Renamed, sliced subset of Noto Serif SC SemiBold (OFL 1.1, noto-cjk Serif2.003) covering the 6500 level-1 + level-2 characters of the 通用规范汉字表 plus Latin, punctuation and common symbols. - scripts/fonts/build_serif_slices.py: fontTools subsetting into 30 woff2 slices + serif-sc.css (@font-face, weight 500 700, font-display swap); byte-identical output for the same input. - Frequency order (char_order.txt): repo corpus top 300, then Google Fonts SC frequency bands (nam-files, Apache-2.0), then corpus count; unranked characters sliced by contiguous code point. - tests/test_serif_font_slices.py: coverage, no overlap, files exist, 120 KB cap; added to the quick quality gate. TASK-serif-headings-20260928 T1. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0199rbQDTsUbCVw84wc8BTFe
314 lines
13 KiB
Python
314 lines
13 KiB
Python
#!/usr/bin/env python3
|
||
"""Build the self-hosted "Jyotisha Serif SC" heading slices.
|
||
|
||
Input: upstream NotoSerifSC-SemiBold.otf (not committed; see
|
||
frontend/src/app/fonts/serif-sc/SOURCE.txt for the URL and sha256).
|
||
Output: frontend/src/app/fonts/serif-sc/jyotisha-serif-sc-NN.woff2
|
||
frontend/src/app/fonts/serif-sc/serif-sc.css (@font-face list)
|
||
|
||
Usage:
|
||
python3 scripts/fonts/build_serif_slices.py --source /path/NotoSerifSC-SemiBold.otf
|
||
python3 scripts/fonts/build_serif_slices.py --refresh-order --gf-slices /path/simplified-chinese_default.txt
|
||
# rewrite char_order.txt only (see scripts/fonts/CHARSET_SOURCE.txt)
|
||
|
||
Local tooling only: needs `fonttools` and `brotli` (pip). Neither is a repo
|
||
dependency and the frontend build never runs this; the woff2 files and the CSS
|
||
are committed. Same input -> byte-identical output (timestamps are not
|
||
recalculated, slice order and options are fixed).
|
||
|
||
Coverage (TASK-serif-headings-20260928, 决策 4): 《通用规范汉字表》(2013)
|
||
一级 + 二级 = 6500 characters, plus Basic Latin, CJK / full-width punctuation,
|
||
digits and common symbols. Anything outside falls back to the sans stack.
|
||
|
||
Frequency order (char_order.txt, committed; --refresh-order rewrites it):
|
||
1. the HOT_COUNT characters most used in this repo's own Chinese corpus
|
||
(product copy + the astrology references answers are written from);
|
||
2. then Google Fonts' frequency band for Simplified Chinese (the first 20
|
||
"FreqRange" subsets of googlefonts/nam-files, Apache-2.0), band by band;
|
||
3. ties broken by corpus count, then by table order.
|
||
The table itself is ordered by stroke count, so it carries no frequency.
|
||
|
||
Slicing:
|
||
00 Basic Latin, Latin-1 symbols, punctuation, CJK / full-width
|
||
punctuation and common symbols (only code points the font has).
|
||
01.. Characters with any frequency signal (a Google Fonts band or a
|
||
corpus count), in rank order, FREQ_SLICE characters per slice.
|
||
..last Characters with no frequency signal, cut into contiguous
|
||
code-point chunks; each chunk's unicode-range is its
|
||
[first, last] span minus code points owned by other slices. That
|
||
keeps serif-sc.css small (ranges instead of single code points).
|
||
A code point outside the 6500 inside such a span has no glyph in
|
||
the slice, so the browser still falls back to the sans stack.
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import collections
|
||
import hashlib
|
||
import io
|
||
import pathlib
|
||
import re
|
||
import subprocess
|
||
import sys
|
||
|
||
ROOT = pathlib.Path(__file__).resolve().parents[2]
|
||
FONT_DIR = pathlib.Path(__file__).resolve().parent
|
||
OUT_DIR = ROOT / "frontend" / "src" / "app" / "fonts" / "serif-sc"
|
||
LEVEL_FILES = (FONT_DIR / "tygfhzb-level-1.txt", FONT_DIR / "tygfhzb-level-2.txt")
|
||
ORDER_FILE = FONT_DIR / "char_order.txt"
|
||
|
||
GF_SLICES_SHA256 = "cd022036048744a478b2d261131a90400bacbb9098b83efd0360f1e44122c2c1"
|
||
NO_BAND = 99
|
||
SOURCE_SHA256 = "517d9736136b3b4e9aea742a7e1acda1922aea91a4425bd8db79281934055bf5"
|
||
FAMILY = "Jyotisha Serif SC"
|
||
POSTSCRIPT = "JyotishaSerifSC-SemiBold"
|
||
FILE_PREFIX = "jyotisha-serif-sc"
|
||
|
||
HOT_COUNT = 300
|
||
FREQ_BANDS = 20
|
||
FREQ_SLICE = 200
|
||
UNSEEN_SLICE = 285
|
||
|
||
# Corpus for --refresh-order: tracked Chinese product copy and the astrology
|
||
# references the answers are written from. Headings come from both.
|
||
CORPUS_PATHS = ("frontend/src", "frontend/docs", "references", "SKILL.md")
|
||
CORPUS_SUFFIXES = {".ts", ".tsx", ".md", ".css", ".json", ".txt"}
|
||
|
||
LATIN_RANGES = (
|
||
(0x0020, 0x007E), # Basic Latin
|
||
(0x00A0, 0x00BF), # Latin-1 punctuation and signs (· ° ± « » …)
|
||
(0x00D7, 0x00D7), # ×
|
||
(0x00F7, 0x00F7), # ÷
|
||
(0x2010, 0x2027), # dashes, quotes, ellipsis
|
||
(0x2030, 0x203B), # ‰ ′ ″ ※
|
||
(0x2103, 0x2103), # ℃
|
||
(0x2116, 0x2116), # №
|
||
(0x2160, 0x216B), # Roman numerals
|
||
(0x2190, 0x2193), # arrows
|
||
(0x2460, 0x2473), # circled numbers
|
||
(0x3000, 0x303F), # CJK symbols and punctuation
|
||
(0x30FB, 0x30FB), # ・
|
||
(0xFF01, 0xFF5E), # full-width forms
|
||
(0xFFE0, 0xFFE5), # full-width signs
|
||
)
|
||
|
||
|
||
def load_table() -> list[str]:
|
||
chars: list[str] = []
|
||
for path in LEVEL_FILES:
|
||
level = [line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
|
||
chars.extend(level)
|
||
assert len(chars) == 6500, f"expected 6500 characters, got {len(chars)}"
|
||
assert len(set(chars)) == 6500, "duplicate characters in the level files"
|
||
assert all(len(ch) == 1 for ch in chars)
|
||
return chars
|
||
|
||
|
||
def gf_bands(path: pathlib.Path) -> dict[str, int]:
|
||
"""Band index (0 = most frequent) from a nam-files slices file.
|
||
|
||
The file lists subsets lowest-priority first; the last FREQ_BANDS blocks
|
||
are the frequency-ranked "FreqRange" subsets.
|
||
"""
|
||
text = path.read_text(encoding="utf-8")
|
||
blocks = re.split(r"\n(?=# \d+ codepoints )", text)[1:][::-1]
|
||
bands: dict[str, int] = {}
|
||
for band, block in enumerate(blocks[:FREQ_BANDS]):
|
||
assert "FreqRange" in block.splitlines()[0], "unexpected nam-files layout"
|
||
for cp in re.findall(r"codepoints: (\d+)", block):
|
||
bands.setdefault(chr(int(cp)), band)
|
||
return bands
|
||
|
||
|
||
def refresh_order(gf_slices: pathlib.Path) -> None:
|
||
table = load_table()
|
||
index = {ch: i for i, ch in enumerate(table)}
|
||
digest = hashlib.sha256(gf_slices.read_bytes()).hexdigest()
|
||
if digest != GF_SLICES_SHA256:
|
||
sys.exit(f"nam-files sha256 {digest} != expected {GF_SLICES_SHA256}; see CHARSET_SOURCE.txt")
|
||
bands = gf_bands(gf_slices)
|
||
files = subprocess.run(
|
||
["git", "ls-files", *CORPUS_PATHS], cwd=ROOT, capture_output=True, text=True, check=True
|
||
).stdout.split()
|
||
counts: collections.Counter[str] = collections.Counter()
|
||
for name in files:
|
||
path = ROOT / name
|
||
if path.suffix not in CORPUS_SUFFIXES or not path.is_file():
|
||
continue
|
||
counts.update(ch for ch in path.read_text(encoding="utf-8", errors="ignore") if ch in index)
|
||
hot = set(sorted(table, key=lambda ch: (-counts[ch], index[ch]))[:HOT_COUNT])
|
||
order = sorted(
|
||
table,
|
||
key=lambda ch: (0 if ch in hot else 1, bands.get(ch, NO_BAND), -counts[ch], index[ch]),
|
||
)
|
||
head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True).stdout.strip()
|
||
lines = [
|
||
"# Frequency order of the 6500 《通用规范汉字表》 level-1 + level-2 characters.",
|
||
"# Generated by scripts/fonts/build_serif_slices.py --refresh-order; sources in CHARSET_SOURCE.txt.",
|
||
f"# Corpus: git-tracked {', '.join(CORPUS_PATHS)} ({', '.join(sorted(CORPUS_SUFFIXES))}) at {head}.",
|
||
f"# Sort key: corpus top {HOT_COUNT} first, then Google Fonts SC band (0-{FREQ_BANDS - 1}, {NO_BAND} = none),",
|
||
"# then corpus count descending, then table order.",
|
||
"# Format: <char>\\t<gf band>\\t<corpus count>.",
|
||
]
|
||
lines += [f"{ch}\t{bands.get(ch, NO_BAND)}\t{counts[ch]}" for ch in order]
|
||
ORDER_FILE.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||
signal = sum(1 for ch in order if ch in bands or counts[ch])
|
||
print(f"wrote {ORDER_FILE.relative_to(ROOT)}: {len(order)} characters, {signal} with a frequency signal")
|
||
|
||
|
||
def load_order() -> list[tuple[str, bool]]:
|
||
"""(character, has a frequency signal) in rank order."""
|
||
rows = []
|
||
for line in ORDER_FILE.read_text(encoding="utf-8").splitlines():
|
||
if not line or line.startswith("#"):
|
||
continue
|
||
ch, band, count = line.split("\t")
|
||
rows.append((ch, int(band) != NO_BAND or int(count) > 0))
|
||
assert sorted(ch for ch, _ in rows) == sorted(load_table()), "char_order.txt is out of step with the level files"
|
||
return rows
|
||
|
||
|
||
def runs(codepoints: list[int]) -> list[tuple[int, int]]:
|
||
ordered = sorted(set(codepoints))
|
||
out: list[tuple[int, int]] = []
|
||
start = prev = ordered[0]
|
||
for cp in ordered[1:]:
|
||
if cp == prev + 1:
|
||
prev = cp
|
||
continue
|
||
out.append((start, prev))
|
||
start = prev = cp
|
||
out.append((start, prev))
|
||
return out
|
||
|
||
|
||
def plan_slices(cmap: dict[int, str]) -> list[dict]:
|
||
order = load_order()
|
||
missing = [ch for ch, _ in order if ord(ch) not in cmap]
|
||
assert not missing, f"font lacks {len(missing)} table characters: {''.join(missing[:20])}"
|
||
|
||
latin = [cp for lo, hi in LATIN_RANGES for cp in range(lo, hi + 1) if cp in cmap]
|
||
slices: list[dict] = [{"chars": latin, "ranges": runs(latin)}]
|
||
|
||
ranked = [ord(ch) for ch, signal in order if signal]
|
||
step = -(-len(ranked) // -(-len(ranked) // FREQ_SLICE)) # even slices of at most FREQ_SLICE
|
||
for i in range(0, len(ranked), step):
|
||
part = ranked[i : i + step]
|
||
slices.append({"chars": part, "ranges": runs(part)})
|
||
|
||
owned = {cp for s in slices for cp in s["chars"]}
|
||
unseen = sorted(ord(ch) for ch, signal in order if not signal)
|
||
unseen_slices = max(1, round(len(unseen) / UNSEEN_SLICE))
|
||
step = -(-len(unseen) // unseen_slices)
|
||
for i in range(0, len(unseen), step):
|
||
part = unseen[i : i + step]
|
||
span = [cp for cp in range(part[0], part[-1] + 1) if cp not in owned]
|
||
slices.append({"chars": part, "ranges": runs(span)})
|
||
return slices
|
||
|
||
|
||
def rename(font) -> None:
|
||
name = font["name"]
|
||
keep = {0, 2, 5, 8, 9, 10, 11, 12, 13, 14}
|
||
name.names = [rec for rec in name.names if rec.nameID in keep and rec.platformID == 3 and rec.langID == 0x409]
|
||
version = str(name.getName(5, 3, 1, 0x409)).split(";")[0]
|
||
name.setName(f"{FAMILY} SemiBold", 1, 3, 1, 0x409)
|
||
name.setName(f"{version.replace('Version ', '')};JYOT;{POSTSCRIPT}", 3, 3, 1, 0x409)
|
||
name.setName(f"{FAMILY} SemiBold", 4, 3, 1, 0x409)
|
||
name.setName(f"{version}; subset of Noto Serif SC SemiBold for Jyotisha headings", 5, 3, 1, 0x409)
|
||
name.setName(POSTSCRIPT, 6, 3, 1, 0x409)
|
||
name.setName(FAMILY, 16, 3, 1, 0x409)
|
||
name.setName("SemiBold", 17, 3, 1, 0x409)
|
||
cff = font["CFF "].cff
|
||
old = cff.fontNames[0]
|
||
cff.fontNames = [POSTSCRIPT]
|
||
top = cff.topDictIndex[0]
|
||
for attr in ("FullName", "FamilyName"):
|
||
if hasattr(top, attr):
|
||
setattr(top, attr, f"{FAMILY} SemiBold" if attr == "FullName" else FAMILY)
|
||
if hasattr(top, "CIDFontName"):
|
||
top.CIDFontName = POSTSCRIPT
|
||
assert old != POSTSCRIPT
|
||
|
||
|
||
def build_slice(source_bytes: bytes, codepoints: list[int]) -> bytes:
|
||
from fontTools import subset
|
||
from fontTools.ttLib import TTFont
|
||
|
||
options = subset.Options()
|
||
options.flavor = "woff2"
|
||
options.desubroutinize = True # ~13% smaller woff2 on CJK CFF; hinting is kept for Windows
|
||
options.name_IDs = ["*"]
|
||
options.name_languages = ["*"]
|
||
options.recalc_timestamp = False
|
||
font = TTFont(io.BytesIO(source_bytes), recalcTimestamp=False, lazy=False)
|
||
subsetter = subset.Subsetter(options)
|
||
subsetter.populate(unicodes=codepoints)
|
||
subsetter.subset(font)
|
||
rename(font)
|
||
font.flavor = "woff2"
|
||
buf = io.BytesIO()
|
||
font.save(buf, reorderTables=False)
|
||
return buf.getvalue()
|
||
|
||
|
||
def fmt_ranges(ranges: list[tuple[int, int]]) -> str:
|
||
return ", ".join(f"U+{a:X}" if a == b else f"U+{a:X}-{b:X}" for a, b in ranges)
|
||
|
||
|
||
def build(source: pathlib.Path) -> None:
|
||
from fontTools.ttLib import TTFont
|
||
|
||
data = source.read_bytes()
|
||
digest = hashlib.sha256(data).hexdigest()
|
||
if digest != SOURCE_SHA256:
|
||
sys.exit(f"source sha256 {digest} != expected {SOURCE_SHA256}; see SOURCE.txt")
|
||
cmap = TTFont(io.BytesIO(data), lazy=True).getBestCmap()
|
||
slices = plan_slices(cmap)
|
||
|
||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
for old in OUT_DIR.glob(f"{FILE_PREFIX}-*.woff2"):
|
||
old.unlink()
|
||
css = [
|
||
"/* Generated by scripts/fonts/build_serif_slices.py — do not edit by hand.",
|
||
f" \"{FAMILY}\" is a renamed subset of Noto Serif SC SemiBold (SIL OFL 1.1, see OFL.txt / SOURCE.txt).",
|
||
" One weight is shipped; font-weight 500 700 lets the existing weight-500/600 display rules hit it.",
|
||
" No preload and font-display: swap — headings paint in the sans stack first, then swap. */",
|
||
]
|
||
for i, sl in enumerate(slices):
|
||
file_name = f"{FILE_PREFIX}-{i:02d}.woff2"
|
||
blob = build_slice(data, sl["chars"])
|
||
(OUT_DIR / file_name).write_bytes(blob)
|
||
css.append(
|
||
"@font-face {\n"
|
||
f" font-family: \"{FAMILY}\";\n"
|
||
" font-style: normal;\n"
|
||
" font-weight: 500 700;\n"
|
||
" font-display: swap;\n"
|
||
f" src: url(\"./{file_name}\") format(\"woff2\");\n"
|
||
f" unicode-range: {fmt_ranges(sl['ranges'])};\n"
|
||
"}"
|
||
)
|
||
print(f"{file_name}: {len(sl['chars'])} chars, {len(blob)} bytes")
|
||
(OUT_DIR / "serif-sc.css").write_text("\n".join(css) + "\n", encoding="utf-8")
|
||
|
||
|
||
def main() -> None:
|
||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||
parser.add_argument("--source", type=pathlib.Path, help="upstream NotoSerifSC-SemiBold.otf")
|
||
parser.add_argument("--refresh-order", action="store_true", help="rewrite char_order.txt")
|
||
parser.add_argument("--gf-slices", type=pathlib.Path, help="nam-files slices/simplified-chinese_default.txt")
|
||
args = parser.parse_args()
|
||
if args.refresh_order:
|
||
if not args.gf_slices:
|
||
parser.error("--refresh-order needs --gf-slices")
|
||
refresh_order(args.gf_slices)
|
||
return
|
||
if not args.source:
|
||
parser.error("--source is required")
|
||
build(args.source)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|