#!/usr/bin/env python3 """Build the self-hosted "Jyotisha Serif SC" heading slices. Input: upstream NotoSerifSC-SemiBold.otf (not committed; see frontend/src/app/fonts/serif-sc/SOURCE.txt for the URL and sha256). Output: frontend/public/fonts/serif-sc/jyotisha-serif-sc-NN..woff2 frontend/src/app/fonts/serif-sc/serif-sc.css (@font-face list) The slices live in public/ under content-hashed names so their URL never carries Next's per-deploy `?dpl=` and stays cached across releases (TASK-home-first-load-20260930 T3-b, BUG-1128); next.config.ts serves /fonts/serif-sc/* as immutable. Usage: python3 scripts/fonts/build_serif_slices.py --source /path/NotoSerifSC-SemiBold.otf python3 scripts/fonts/build_serif_slices.py --refresh-order --gf-slices /path/simplified-chinese_default.txt # rewrite char_order.txt only (see scripts/fonts/CHARSET_SOURCE.txt) Local tooling only: needs `fonttools` and `brotli` (pip). Neither is a repo dependency and the frontend build never runs this; the woff2 files and the CSS are committed. Same input -> byte-identical output (timestamps are not recalculated, slice order and options are fixed). Coverage (TASK-serif-headings-20260928, 决策 4): 《通用规范汉字表》(2013) 一级 + 二级 = 6500 characters, plus Basic Latin, CJK / full-width punctuation, digits and common symbols. Anything outside falls back to the sans stack. Frequency order (char_order.txt, committed; --refresh-order rewrites it): 1. the HOT_COUNT characters most used in this repo's own Chinese corpus (product copy + the astrology references answers are written from); 2. then Google Fonts' frequency band for Simplified Chinese (the first 20 "FreqRange" subsets of googlefonts/nam-files, Apache-2.0), band by band; 3. ties broken by corpus count, then by table order. The table itself is ordered by stroke count, so it carries no frequency. Slicing: 00 Basic Latin, Latin-1 symbols, punctuation, CJK / full-width punctuation and common symbols (only code points the font has). 01.. Characters with any frequency signal (a Google Fonts band or a corpus count), in rank order, FREQ_SLICE characters per slice. ..last Characters with no frequency signal, cut into contiguous code-point chunks; each chunk's unicode-range is its [first, last] span minus code points owned by other slices. That keeps serif-sc.css small (ranges instead of single code points). A code point outside the 6500 inside such a span has no glyph in the slice, so the browser still falls back to the sans stack. """ from __future__ import annotations import argparse import collections import hashlib import io import pathlib import re import subprocess import sys ROOT = pathlib.Path(__file__).resolve().parents[2] FONT_DIR = pathlib.Path(__file__).resolve().parent OUT_DIR = ROOT / "frontend" / "src" / "app" / "fonts" / "serif-sc" SLICE_DIR = ROOT / "frontend" / "public" / "fonts" / "serif-sc" SLICE_URL = "/fonts/serif-sc" LEVEL_FILES = (FONT_DIR / "tygfhzb-level-1.txt", FONT_DIR / "tygfhzb-level-2.txt") ORDER_FILE = FONT_DIR / "char_order.txt" GF_SLICES_SHA256 = "cd022036048744a478b2d261131a90400bacbb9098b83efd0360f1e44122c2c1" NO_BAND = 99 SOURCE_SHA256 = "517d9736136b3b4e9aea742a7e1acda1922aea91a4425bd8db79281934055bf5" FAMILY = "Jyotisha Serif SC" POSTSCRIPT = "JyotishaSerifSC-SemiBold" FILE_PREFIX = "jyotisha-serif-sc" HOT_COUNT = 300 FREQ_BANDS = 20 FREQ_SLICE = 200 UNSEEN_SLICE = 285 # Corpus for --refresh-order: tracked Chinese product copy and the astrology # references the answers are written from. Headings come from both. CORPUS_PATHS = ("frontend/src", "frontend/docs", "references", "SKILL.md") CORPUS_SUFFIXES = {".ts", ".tsx", ".md", ".css", ".json", ".txt"} LATIN_RANGES = ( (0x0020, 0x007E), # Basic Latin (0x00A0, 0x00BF), # Latin-1 punctuation and signs (· ° ± « » …) (0x00D7, 0x00D7), # × (0x00F7, 0x00F7), # ÷ (0x2010, 0x2027), # dashes, quotes, ellipsis (0x2030, 0x203B), # ‰ ′ ″ ※ (0x2103, 0x2103), # ℃ (0x2116, 0x2116), # № (0x2160, 0x216B), # Roman numerals (0x2190, 0x2193), # arrows (0x2460, 0x2473), # circled numbers (0x3000, 0x303F), # CJK symbols and punctuation (0x30FB, 0x30FB), # ・ (0xFF01, 0xFF5E), # full-width forms (0xFFE0, 0xFFE5), # full-width signs ) def load_table() -> list[str]: chars: list[str] = [] for path in LEVEL_FILES: level = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] chars.extend(level) assert len(chars) == 6500, f"expected 6500 characters, got {len(chars)}" assert len(set(chars)) == 6500, "duplicate characters in the level files" assert all(len(ch) == 1 for ch in chars) return chars def gf_bands(path: pathlib.Path) -> dict[str, int]: """Band index (0 = most frequent) from a nam-files slices file. The file lists subsets lowest-priority first; the last FREQ_BANDS blocks are the frequency-ranked "FreqRange" subsets. """ text = path.read_text(encoding="utf-8") blocks = re.split(r"\n(?=# \d+ codepoints )", text)[1:][::-1] bands: dict[str, int] = {} for band, block in enumerate(blocks[:FREQ_BANDS]): assert "FreqRange" in block.splitlines()[0], "unexpected nam-files layout" for cp in re.findall(r"codepoints: (\d+)", block): bands.setdefault(chr(int(cp)), band) return bands def refresh_order(gf_slices: pathlib.Path) -> None: table = load_table() index = {ch: i for i, ch in enumerate(table)} digest = hashlib.sha256(gf_slices.read_bytes()).hexdigest() if digest != GF_SLICES_SHA256: sys.exit(f"nam-files sha256 {digest} != expected {GF_SLICES_SHA256}; see CHARSET_SOURCE.txt") bands = gf_bands(gf_slices) files = subprocess.run( ["git", "ls-files", *CORPUS_PATHS], cwd=ROOT, capture_output=True, text=True, check=True ).stdout.split() counts: collections.Counter[str] = collections.Counter() for name in files: path = ROOT / name if path.suffix not in CORPUS_SUFFIXES or not path.is_file(): continue counts.update(ch for ch in path.read_text(encoding="utf-8", errors="ignore") if ch in index) hot = set(sorted(table, key=lambda ch: (-counts[ch], index[ch]))[:HOT_COUNT]) order = sorted( table, key=lambda ch: (0 if ch in hot else 1, bands.get(ch, NO_BAND), -counts[ch], index[ch]), ) head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True).stdout.strip() lines = [ "# Frequency order of the 6500 《通用规范汉字表》 level-1 + level-2 characters.", "# Generated by scripts/fonts/build_serif_slices.py --refresh-order; sources in CHARSET_SOURCE.txt.", f"# Corpus: git-tracked {', '.join(CORPUS_PATHS)} ({', '.join(sorted(CORPUS_SUFFIXES))}) at {head}.", f"# Sort key: corpus top {HOT_COUNT} first, then Google Fonts SC band (0-{FREQ_BANDS - 1}, {NO_BAND} = none),", "# then corpus count descending, then table order.", "# Format: \\t\\t.", ] lines += [f"{ch}\t{bands.get(ch, NO_BAND)}\t{counts[ch]}" for ch in order] ORDER_FILE.write_text("\n".join(lines) + "\n", encoding="utf-8") signal = sum(1 for ch in order if ch in bands or counts[ch]) print(f"wrote {ORDER_FILE.relative_to(ROOT)}: {len(order)} characters, {signal} with a frequency signal") def load_order() -> list[tuple[str, bool]]: """(character, has a frequency signal) in rank order.""" rows = [] for line in ORDER_FILE.read_text(encoding="utf-8").splitlines(): if not line or line.startswith("#"): continue ch, band, count = line.split("\t") rows.append((ch, int(band) != NO_BAND or int(count) > 0)) assert sorted(ch for ch, _ in rows) == sorted(load_table()), "char_order.txt is out of step with the level files" return rows def runs(codepoints: list[int]) -> list[tuple[int, int]]: ordered = sorted(set(codepoints)) out: list[tuple[int, int]] = [] start = prev = ordered[0] for cp in ordered[1:]: if cp == prev + 1: prev = cp continue out.append((start, prev)) start = prev = cp out.append((start, prev)) return out def plan_slices(cmap: dict[int, str]) -> list[dict]: order = load_order() missing = [ch for ch, _ in order if ord(ch) not in cmap] assert not missing, f"font lacks {len(missing)} table characters: {''.join(missing[:20])}" latin = [cp for lo, hi in LATIN_RANGES for cp in range(lo, hi + 1) if cp in cmap] slices: list[dict] = [{"chars": latin, "ranges": runs(latin)}] ranked = [ord(ch) for ch, signal in order if signal] step = -(-len(ranked) // -(-len(ranked) // FREQ_SLICE)) # even slices of at most FREQ_SLICE for i in range(0, len(ranked), step): part = ranked[i : i + step] slices.append({"chars": part, "ranges": runs(part)}) owned = {cp for s in slices for cp in s["chars"]} unseen = sorted(ord(ch) for ch, signal in order if not signal) unseen_slices = max(1, round(len(unseen) / UNSEEN_SLICE)) step = -(-len(unseen) // unseen_slices) for i in range(0, len(unseen), step): part = unseen[i : i + step] span = [cp for cp in range(part[0], part[-1] + 1) if cp not in owned] slices.append({"chars": part, "ranges": runs(span)}) return slices def rename(font) -> None: name = font["name"] keep = {0, 2, 5, 8, 9, 10, 11, 12, 13, 14} name.names = [rec for rec in name.names if rec.nameID in keep and rec.platformID == 3 and rec.langID == 0x409] version = str(name.getName(5, 3, 1, 0x409)).split(";")[0] name.setName(f"{FAMILY} SemiBold", 1, 3, 1, 0x409) name.setName(f"{version.replace('Version ', '')};JYOT;{POSTSCRIPT}", 3, 3, 1, 0x409) name.setName(f"{FAMILY} SemiBold", 4, 3, 1, 0x409) name.setName(f"{version}; subset of Noto Serif SC SemiBold for Jyotisha headings", 5, 3, 1, 0x409) name.setName(POSTSCRIPT, 6, 3, 1, 0x409) name.setName(FAMILY, 16, 3, 1, 0x409) name.setName("SemiBold", 17, 3, 1, 0x409) cff = font["CFF "].cff old = cff.fontNames[0] cff.fontNames = [POSTSCRIPT] top = cff.topDictIndex[0] for attr in ("FullName", "FamilyName"): if hasattr(top, attr): setattr(top, attr, f"{FAMILY} SemiBold" if attr == "FullName" else FAMILY) if hasattr(top, "CIDFontName"): top.CIDFontName = POSTSCRIPT assert old != POSTSCRIPT def build_slice(source_bytes: bytes, codepoints: list[int]) -> bytes: from fontTools import subset from fontTools.ttLib import TTFont options = subset.Options() options.flavor = "woff2" options.desubroutinize = True # ~13% smaller woff2 on CJK CFF; hinting is kept for Windows options.name_IDs = ["*"] options.name_languages = ["*"] options.recalc_timestamp = False font = TTFont(io.BytesIO(source_bytes), recalcTimestamp=False, lazy=False) subsetter = subset.Subsetter(options) subsetter.populate(unicodes=codepoints) subsetter.subset(font) rename(font) font.flavor = "woff2" buf = io.BytesIO() font.save(buf, reorderTables=False) return buf.getvalue() def fmt_ranges(ranges: list[tuple[int, int]]) -> str: return ", ".join(f"U+{a:X}" if a == b else f"U+{a:X}-{b:X}" for a, b in ranges) def build(source: pathlib.Path) -> None: from fontTools.ttLib import TTFont data = source.read_bytes() digest = hashlib.sha256(data).hexdigest() if digest != SOURCE_SHA256: sys.exit(f"source sha256 {digest} != expected {SOURCE_SHA256}; see SOURCE.txt") cmap = TTFont(io.BytesIO(data), lazy=True).getBestCmap() slices = plan_slices(cmap) OUT_DIR.mkdir(parents=True, exist_ok=True) SLICE_DIR.mkdir(parents=True, exist_ok=True) for old in SLICE_DIR.glob(f"{FILE_PREFIX}-*.woff2"): old.unlink() css = [ "/* Generated by scripts/fonts/build_serif_slices.py — do not edit by hand.", f" \"{FAMILY}\" is a renamed subset of Noto Serif SC SemiBold (SIL OFL 1.1, see OFL.txt / SOURCE.txt).", " One weight is shipped; font-weight 500 700 lets the existing weight-500/600 display rules hit it.", " No preload and font-display: swap — headings paint in the sans stack first, then swap. */", ] for i, sl in enumerate(slices): blob = build_slice(data, sl["chars"]) file_name = f"{FILE_PREFIX}-{i:02d}.{hashlib.sha256(blob).hexdigest()[:10]}.woff2" (SLICE_DIR / file_name).write_bytes(blob) css.append( "@font-face {\n" f" font-family: \"{FAMILY}\";\n" " font-style: normal;\n" " font-weight: 500 700;\n" " font-display: swap;\n" f" src: url(\"{SLICE_URL}/{file_name}\") format(\"woff2\");\n" f" unicode-range: {fmt_ranges(sl['ranges'])};\n" "}" ) print(f"{file_name}: {len(sl['chars'])} chars, {len(blob)} bytes") (OUT_DIR / "serif-sc.css").write_text("\n".join(css) + "\n", encoding="utf-8") def main() -> None: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--source", type=pathlib.Path, help="upstream NotoSerifSC-SemiBold.otf") parser.add_argument("--refresh-order", action="store_true", help="rewrite char_order.txt") parser.add_argument("--gf-slices", type=pathlib.Path, help="nam-files slices/simplified-chinese_default.txt") args = parser.parse_args() if args.refresh_order: if not args.gf_slices: parser.error("--refresh-order needs --gf-slices") refresh_order(args.gf_slices) return if not args.source: parser.error("--source is required") build(args.source) if __name__ == "__main__": main()