Renamed, sliced subset of Noto Serif SC SemiBold (OFL 1.1, noto-cjk Serif2.003) covering the 6500 level-1 + level-2 characters of the 通用规范汉字表 plus Latin, punctuation and common symbols. - scripts/fonts/build_serif_slices.py: fontTools subsetting into 30 woff2 slices + serif-sc.css (@font-face, weight 500 700, font-display swap); byte-identical output for the same input. - Frequency order (char_order.txt): repo corpus top 300, then Google Fonts SC frequency bands (nam-files, Apache-2.0), then corpus count; unranked characters sliced by contiguous code point. - tests/test_serif_font_slices.py: coverage, no overlap, files exist, 120 KB cap; added to the quick quality gate. TASK-serif-headings-20260928 T1. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_0199rbQDTsUbCVw84wc8BTFe
129 lines
5.2 KiB
Python
129 lines
5.2 KiB
Python
"""Contract for the self-hosted "Jyotisha Serif SC" heading slices.
|
|
|
|
TASK-serif-headings-20260928 T1: every one of the 6500 《通用规范汉字表》
|
|
level-1 + level-2 characters must be reachable through exactly one slice's
|
|
unicode-range, every referenced file must exist, and no slice may exceed
|
|
120 KB. The coverage range is not negotiable (让步顺序 3): if a slice grows,
|
|
cut slices finer instead of dropping characters.
|
|
|
|
Pure file parsing — no fontTools needed. When fontTools happens to be
|
|
installed, the glyph coverage of each slice is checked against its range too.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import pathlib
|
|
import re
|
|
|
|
import pytest
|
|
|
|
ROOT = pathlib.Path(__file__).resolve().parents[1]
|
|
FONT_DIR = ROOT / "frontend" / "src" / "app" / "fonts" / "serif-sc"
|
|
CSS = FONT_DIR / "serif-sc.css"
|
|
LEVEL_FILES = [ROOT / "scripts" / "fonts" / f"tygfhzb-level-{n}.txt" for n in (1, 2)]
|
|
MAX_SLICE_BYTES = 120 * 1024
|
|
|
|
FACE_RE = re.compile(r"@font-face\s*\{(?P<body>[^}]*)\}", re.S)
|
|
|
|
|
|
def _table() -> list[str]:
|
|
chars: list[str] = []
|
|
for path in LEVEL_FILES:
|
|
chars.extend(line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip())
|
|
return chars
|
|
|
|
|
|
def _parse_range(value: str) -> set[int]:
|
|
codepoints: set[int] = set()
|
|
for token in value.split(","):
|
|
token = token.strip().upper()
|
|
assert token.startswith("U+"), token
|
|
span = token[2:]
|
|
if "-" in span:
|
|
lo, hi = (int(part, 16) for part in span.split("-"))
|
|
else:
|
|
lo = hi = int(span, 16)
|
|
assert lo <= hi, token
|
|
codepoints.update(range(lo, hi + 1))
|
|
return codepoints
|
|
|
|
|
|
def _faces() -> list[dict]:
|
|
faces = []
|
|
for match in FACE_RE.finditer(CSS.read_text(encoding="utf-8")):
|
|
body = match.group("body")
|
|
family = re.search(r'font-family:\s*"([^"]+)"', body).group(1)
|
|
src = re.search(r'src:\s*url\("\./([^"]+)"\)\s*format\("woff2"\)', body).group(1)
|
|
unicode_range = re.search(r"unicode-range:\s*([^;]+);", body).group(1)
|
|
faces.append({"family": family, "file": src, "range": _parse_range(unicode_range), "body": body})
|
|
return faces
|
|
|
|
|
|
def test_table_is_the_6500_level_one_and_two_characters():
|
|
table = _table()
|
|
assert len(table) == 6500
|
|
assert len(set(table)) == 6500
|
|
assert all(len(ch) == 1 for ch in table)
|
|
|
|
|
|
def test_every_table_character_is_covered_by_a_slice():
|
|
covered = set().union(*(face["range"] for face in _faces()))
|
|
missing = [ch for ch in _table() if ord(ch) not in covered]
|
|
assert not missing, f"{len(missing)} characters have no slice: {''.join(missing[:40])}"
|
|
|
|
|
|
def test_basic_latin_digits_and_cjk_punctuation_are_covered():
|
|
covered = set().union(*(face["range"] for face in _faces()))
|
|
for text in ("ABCxyz0123456789", ",。、;:?!“”‘’()《》【】…—·", "%+-="):
|
|
for ch in text:
|
|
assert ord(ch) in covered, f"{ch!r} U+{ord(ch):04X} is not covered"
|
|
|
|
|
|
def test_slice_ranges_do_not_overlap():
|
|
seen: dict[int, str] = {}
|
|
for face in _faces():
|
|
for cp in face["range"]:
|
|
assert cp not in seen, f"U+{cp:04X} is in both {seen[cp]} and {face['file']}"
|
|
seen[cp] = face["file"]
|
|
|
|
|
|
def test_every_referenced_file_exists_and_stays_under_the_slice_cap():
|
|
faces = _faces()
|
|
assert len(faces) >= 10
|
|
referenced = set()
|
|
for face in faces:
|
|
path = FONT_DIR / face["file"]
|
|
assert path.is_file(), f"{face['file']} is referenced by serif-sc.css but missing"
|
|
assert path.read_bytes()[:4] == b"wOF2", f"{face['file']} is not a woff2 file"
|
|
size = path.stat().st_size
|
|
assert size <= MAX_SLICE_BYTES, f"{face['file']} is {size} bytes (> {MAX_SLICE_BYTES}); cut slices finer"
|
|
referenced.add(face["file"])
|
|
on_disk = {p.name for p in FONT_DIR.glob("*.woff2")}
|
|
assert on_disk == referenced, f"unreferenced slice files: {sorted(on_disk - referenced)}"
|
|
|
|
|
|
def test_faces_use_the_renamed_family_one_weight_range_and_swap():
|
|
for face in _faces():
|
|
assert face["family"] == "Jyotisha Serif SC"
|
|
assert re.search(r"font-weight:\s*500 700;", face["body"])
|
|
assert re.search(r"font-display:\s*swap;", face["body"])
|
|
|
|
|
|
def test_licence_and_provenance_sit_beside_the_slices():
|
|
assert (FONT_DIR / "OFL.txt").read_text(encoding="utf-8").startswith("This Font Software is licensed under the SIL Open Font License")
|
|
source = (FONT_DIR / "SOURCE.txt").read_text(encoding="utf-8")
|
|
assert "517d9736136b3b4e9aea742a7e1acda1922aea91a4425bd8db79281934055bf5" in source
|
|
assert "Jyotisha Serif SC" in source
|
|
|
|
|
|
def test_slice_glyphs_cover_their_table_characters():
|
|
ttlib = pytest.importorskip("fontTools.ttLib")
|
|
table = {ord(ch) for ch in _table()}
|
|
for face in _faces():
|
|
font = ttlib.TTFont(FONT_DIR / face["file"], lazy=True)
|
|
cmap = set(font.getBestCmap())
|
|
wanted = face["range"] & table
|
|
assert wanted <= cmap, f"{face['file']} lacks glyphs for {len(wanted - cmap)} of its table characters"
|
|
family = font["name"].getDebugName(16) or font["name"].getDebugName(1)
|
|
assert family.startswith("Jyotisha Serif SC"), family
|
|
assert "Noto" not in (font["name"].getDebugName(1) or "")
|