Files
Jyotisha/scripts/fonts/build_serif_slices.py
T
Jesse_ChenandClaude Opus 5.5 8112f62b51 perf(fonts): heading serif slices served from public/ under content-hashed names, cached for good (BUG-1128)
Bundled media carry Next's per-deploy ?dpl= and were downloaded again after
every release. Bytes unchanged; /fonts/serif-sc/* is immutable. Overturns the
serif-headings contract line "bundled by Next, not served from public/"
per TASK-home-first-load-20260930 decision 2.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
2026-10-01 00:40:11 +08:00

322 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Build the self-hosted "Jyotisha Serif SC" heading slices.
Input: upstream NotoSerifSC-SemiBold.otf (not committed; see
frontend/src/app/fonts/serif-sc/SOURCE.txt for the URL and sha256).
Output: frontend/public/fonts/serif-sc/jyotisha-serif-sc-NN.<sha256[:10]>.woff2
frontend/src/app/fonts/serif-sc/serif-sc.css (@font-face list)
The slices live in public/ under content-hashed names so their URL never
carries Next's per-deploy `?dpl=` and stays cached across releases
(TASK-home-first-load-20260930 T3-b, BUG-1128); next.config.ts serves
/fonts/serif-sc/* as immutable.
Usage:
python3 scripts/fonts/build_serif_slices.py --source /path/NotoSerifSC-SemiBold.otf
python3 scripts/fonts/build_serif_slices.py --refresh-order --gf-slices /path/simplified-chinese_default.txt
# rewrite char_order.txt only (see scripts/fonts/CHARSET_SOURCE.txt)
Local tooling only: needs `fonttools` and `brotli` (pip). Neither is a repo
dependency and the frontend build never runs this; the woff2 files and the CSS
are committed. Same input -> byte-identical output (timestamps are not
recalculated, slice order and options are fixed).
Coverage (TASK-serif-headings-20260928, 决策 4): 《通用规范汉字表》(2013)
一级 + 二级 = 6500 characters, plus Basic Latin, CJK / full-width punctuation,
digits and common symbols. Anything outside falls back to the sans stack.
Frequency order (char_order.txt, committed; --refresh-order rewrites it):
1. the HOT_COUNT characters most used in this repo's own Chinese corpus
(product copy + the astrology references answers are written from);
2. then Google Fonts' frequency band for Simplified Chinese (the first 20
"FreqRange" subsets of googlefonts/nam-files, Apache-2.0), band by band;
3. ties broken by corpus count, then by table order.
The table itself is ordered by stroke count, so it carries no frequency.
Slicing:
00 Basic Latin, Latin-1 symbols, punctuation, CJK / full-width
punctuation and common symbols (only code points the font has).
01.. Characters with any frequency signal (a Google Fonts band or a
corpus count), in rank order, FREQ_SLICE characters per slice.
..last Characters with no frequency signal, cut into contiguous
code-point chunks; each chunk's unicode-range is its
[first, last] span minus code points owned by other slices. That
keeps serif-sc.css small (ranges instead of single code points).
A code point outside the 6500 inside such a span has no glyph in
the slice, so the browser still falls back to the sans stack.
"""
from __future__ import annotations
import argparse
import collections
import hashlib
import io
import pathlib
import re
import subprocess
import sys
ROOT = pathlib.Path(__file__).resolve().parents[2]
FONT_DIR = pathlib.Path(__file__).resolve().parent
OUT_DIR = ROOT / "frontend" / "src" / "app" / "fonts" / "serif-sc"
SLICE_DIR = ROOT / "frontend" / "public" / "fonts" / "serif-sc"
SLICE_URL = "/fonts/serif-sc"
LEVEL_FILES = (FONT_DIR / "tygfhzb-level-1.txt", FONT_DIR / "tygfhzb-level-2.txt")
ORDER_FILE = FONT_DIR / "char_order.txt"
GF_SLICES_SHA256 = "cd022036048744a478b2d261131a90400bacbb9098b83efd0360f1e44122c2c1"
NO_BAND = 99
SOURCE_SHA256 = "517d9736136b3b4e9aea742a7e1acda1922aea91a4425bd8db79281934055bf5"
FAMILY = "Jyotisha Serif SC"
POSTSCRIPT = "JyotishaSerifSC-SemiBold"
FILE_PREFIX = "jyotisha-serif-sc"
HOT_COUNT = 300
FREQ_BANDS = 20
FREQ_SLICE = 200
UNSEEN_SLICE = 285
# Corpus for --refresh-order: tracked Chinese product copy and the astrology
# references the answers are written from. Headings come from both.
CORPUS_PATHS = ("frontend/src", "frontend/docs", "references", "SKILL.md")
CORPUS_SUFFIXES = {".ts", ".tsx", ".md", ".css", ".json", ".txt"}
LATIN_RANGES = (
(0x0020, 0x007E), # Basic Latin
(0x00A0, 0x00BF), # Latin-1 punctuation and signs (· ° ± « » …)
(0x00D7, 0x00D7), # ×
(0x00F7, 0x00F7), # ÷
(0x2010, 0x2027), # dashes, quotes, ellipsis
(0x2030, 0x203B), # ‰ ′ ″ ※
(0x2103, 0x2103), # ℃
(0x2116, 0x2116), # №
(0x2160, 0x216B), # Roman numerals
(0x2190, 0x2193), # arrows
(0x2460, 0x2473), # circled numbers
(0x3000, 0x303F), # CJK symbols and punctuation
(0x30FB, 0x30FB), # ・
(0xFF01, 0xFF5E), # full-width forms
(0xFFE0, 0xFFE5), # full-width signs
)
def load_table() -> list[str]:
chars: list[str] = []
for path in LEVEL_FILES:
level = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
chars.extend(level)
assert len(chars) == 6500, f"expected 6500 characters, got {len(chars)}"
assert len(set(chars)) == 6500, "duplicate characters in the level files"
assert all(len(ch) == 1 for ch in chars)
return chars
def gf_bands(path: pathlib.Path) -> dict[str, int]:
"""Band index (0 = most frequent) from a nam-files slices file.
The file lists subsets lowest-priority first; the last FREQ_BANDS blocks
are the frequency-ranked "FreqRange" subsets.
"""
text = path.read_text(encoding="utf-8")
blocks = re.split(r"\n(?=# \d+ codepoints )", text)[1:][::-1]
bands: dict[str, int] = {}
for band, block in enumerate(blocks[:FREQ_BANDS]):
assert "FreqRange" in block.splitlines()[0], "unexpected nam-files layout"
for cp in re.findall(r"codepoints: (\d+)", block):
bands.setdefault(chr(int(cp)), band)
return bands
def refresh_order(gf_slices: pathlib.Path) -> None:
table = load_table()
index = {ch: i for i, ch in enumerate(table)}
digest = hashlib.sha256(gf_slices.read_bytes()).hexdigest()
if digest != GF_SLICES_SHA256:
sys.exit(f"nam-files sha256 {digest} != expected {GF_SLICES_SHA256}; see CHARSET_SOURCE.txt")
bands = gf_bands(gf_slices)
files = subprocess.run(
["git", "ls-files", *CORPUS_PATHS], cwd=ROOT, capture_output=True, text=True, check=True
).stdout.split()
counts: collections.Counter[str] = collections.Counter()
for name in files:
path = ROOT / name
if path.suffix not in CORPUS_SUFFIXES or not path.is_file():
continue
counts.update(ch for ch in path.read_text(encoding="utf-8", errors="ignore") if ch in index)
hot = set(sorted(table, key=lambda ch: (-counts[ch], index[ch]))[:HOT_COUNT])
order = sorted(
table,
key=lambda ch: (0 if ch in hot else 1, bands.get(ch, NO_BAND), -counts[ch], index[ch]),
)
head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True).stdout.strip()
lines = [
"# Frequency order of the 6500 《通用规范汉字表》 level-1 + level-2 characters.",
"# Generated by scripts/fonts/build_serif_slices.py --refresh-order; sources in CHARSET_SOURCE.txt.",
f"# Corpus: git-tracked {', '.join(CORPUS_PATHS)} ({', '.join(sorted(CORPUS_SUFFIXES))}) at {head}.",
f"# Sort key: corpus top {HOT_COUNT} first, then Google Fonts SC band (0-{FREQ_BANDS - 1}, {NO_BAND} = none),",
"# then corpus count descending, then table order.",
"# Format: <char>\\t<gf band>\\t<corpus count>.",
]
lines += [f"{ch}\t{bands.get(ch, NO_BAND)}\t{counts[ch]}" for ch in order]
ORDER_FILE.write_text("\n".join(lines) + "\n", encoding="utf-8")
signal = sum(1 for ch in order if ch in bands or counts[ch])
print(f"wrote {ORDER_FILE.relative_to(ROOT)}: {len(order)} characters, {signal} with a frequency signal")
def load_order() -> list[tuple[str, bool]]:
"""(character, has a frequency signal) in rank order."""
rows = []
for line in ORDER_FILE.read_text(encoding="utf-8").splitlines():
if not line or line.startswith("#"):
continue
ch, band, count = line.split("\t")
rows.append((ch, int(band) != NO_BAND or int(count) > 0))
assert sorted(ch for ch, _ in rows) == sorted(load_table()), "char_order.txt is out of step with the level files"
return rows
def runs(codepoints: list[int]) -> list[tuple[int, int]]:
ordered = sorted(set(codepoints))
out: list[tuple[int, int]] = []
start = prev = ordered[0]
for cp in ordered[1:]:
if cp == prev + 1:
prev = cp
continue
out.append((start, prev))
start = prev = cp
out.append((start, prev))
return out
def plan_slices(cmap: dict[int, str]) -> list[dict]:
order = load_order()
missing = [ch for ch, _ in order if ord(ch) not in cmap]
assert not missing, f"font lacks {len(missing)} table characters: {''.join(missing[:20])}"
latin = [cp for lo, hi in LATIN_RANGES for cp in range(lo, hi + 1) if cp in cmap]
slices: list[dict] = [{"chars": latin, "ranges": runs(latin)}]
ranked = [ord(ch) for ch, signal in order if signal]
step = -(-len(ranked) // -(-len(ranked) // FREQ_SLICE)) # even slices of at most FREQ_SLICE
for i in range(0, len(ranked), step):
part = ranked[i : i + step]
slices.append({"chars": part, "ranges": runs(part)})
owned = {cp for s in slices for cp in s["chars"]}
unseen = sorted(ord(ch) for ch, signal in order if not signal)
unseen_slices = max(1, round(len(unseen) / UNSEEN_SLICE))
step = -(-len(unseen) // unseen_slices)
for i in range(0, len(unseen), step):
part = unseen[i : i + step]
span = [cp for cp in range(part[0], part[-1] + 1) if cp not in owned]
slices.append({"chars": part, "ranges": runs(span)})
return slices
def rename(font) -> None:
name = font["name"]
keep = {0, 2, 5, 8, 9, 10, 11, 12, 13, 14}
name.names = [rec for rec in name.names if rec.nameID in keep and rec.platformID == 3 and rec.langID == 0x409]
version = str(name.getName(5, 3, 1, 0x409)).split(";")[0]
name.setName(f"{FAMILY} SemiBold", 1, 3, 1, 0x409)
name.setName(f"{version.replace('Version ', '')};JYOT;{POSTSCRIPT}", 3, 3, 1, 0x409)
name.setName(f"{FAMILY} SemiBold", 4, 3, 1, 0x409)
name.setName(f"{version}; subset of Noto Serif SC SemiBold for Jyotisha headings", 5, 3, 1, 0x409)
name.setName(POSTSCRIPT, 6, 3, 1, 0x409)
name.setName(FAMILY, 16, 3, 1, 0x409)
name.setName("SemiBold", 17, 3, 1, 0x409)
cff = font["CFF "].cff
old = cff.fontNames[0]
cff.fontNames = [POSTSCRIPT]
top = cff.topDictIndex[0]
for attr in ("FullName", "FamilyName"):
if hasattr(top, attr):
setattr(top, attr, f"{FAMILY} SemiBold" if attr == "FullName" else FAMILY)
if hasattr(top, "CIDFontName"):
top.CIDFontName = POSTSCRIPT
assert old != POSTSCRIPT
def build_slice(source_bytes: bytes, codepoints: list[int]) -> bytes:
from fontTools import subset
from fontTools.ttLib import TTFont
options = subset.Options()
options.flavor = "woff2"
options.desubroutinize = True # ~13% smaller woff2 on CJK CFF; hinting is kept for Windows
options.name_IDs = ["*"]
options.name_languages = ["*"]
options.recalc_timestamp = False
font = TTFont(io.BytesIO(source_bytes), recalcTimestamp=False, lazy=False)
subsetter = subset.Subsetter(options)
subsetter.populate(unicodes=codepoints)
subsetter.subset(font)
rename(font)
font.flavor = "woff2"
buf = io.BytesIO()
font.save(buf, reorderTables=False)
return buf.getvalue()
def fmt_ranges(ranges: list[tuple[int, int]]) -> str:
return ", ".join(f"U+{a:X}" if a == b else f"U+{a:X}-{b:X}" for a, b in ranges)
def build(source: pathlib.Path) -> None:
from fontTools.ttLib import TTFont
data = source.read_bytes()
digest = hashlib.sha256(data).hexdigest()
if digest != SOURCE_SHA256:
sys.exit(f"source sha256 {digest} != expected {SOURCE_SHA256}; see SOURCE.txt")
cmap = TTFont(io.BytesIO(data), lazy=True).getBestCmap()
slices = plan_slices(cmap)
OUT_DIR.mkdir(parents=True, exist_ok=True)
SLICE_DIR.mkdir(parents=True, exist_ok=True)
for old in SLICE_DIR.glob(f"{FILE_PREFIX}-*.woff2"):
old.unlink()
css = [
"/* Generated by scripts/fonts/build_serif_slices.py — do not edit by hand.",
f" \"{FAMILY}\" is a renamed subset of Noto Serif SC SemiBold (SIL OFL 1.1, see OFL.txt / SOURCE.txt).",
" One weight is shipped; font-weight 500 700 lets the existing weight-500/600 display rules hit it.",
" No preload and font-display: swap — headings paint in the sans stack first, then swap. */",
]
for i, sl in enumerate(slices):
blob = build_slice(data, sl["chars"])
file_name = f"{FILE_PREFIX}-{i:02d}.{hashlib.sha256(blob).hexdigest()[:10]}.woff2"
(SLICE_DIR / file_name).write_bytes(blob)
css.append(
"@font-face {\n"
f" font-family: \"{FAMILY}\";\n"
" font-style: normal;\n"
" font-weight: 500 700;\n"
" font-display: swap;\n"
f" src: url(\"{SLICE_URL}/{file_name}\") format(\"woff2\");\n"
f" unicode-range: {fmt_ranges(sl['ranges'])};\n"
"}"
)
print(f"{file_name}: {len(sl['chars'])} chars, {len(blob)} bytes")
(OUT_DIR / "serif-sc.css").write_text("\n".join(css) + "\n", encoding="utf-8")
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--source", type=pathlib.Path, help="upstream NotoSerifSC-SemiBold.otf")
parser.add_argument("--refresh-order", action="store_true", help="rewrite char_order.txt")
parser.add_argument("--gf-slices", type=pathlib.Path, help="nam-files slices/simplified-chinese_default.txt")
args = parser.parse_args()
if args.refresh_order:
if not args.gf_slices:
parser.error("--refresh-order needs --gf-slices")
refresh_order(args.gf_slices)
return
if not args.source:
parser.error("--source is required")
build(args.source)
if __name__ == "__main__":
main()