Files
Jyotisha/scripts/pl9_reader_english.py
T
Jesse_ChenandClaude Opus 5.5 419e1f4d37 feat(report): English reader edition from the same packet; yoga combination English twins (E1/E2/E3 wiring)
- /api/professional_report_reference accepts languages [zh,en]; one packet, zh
  byte-identical, markdown_en + reader_dasha_applicability_en, or
  english_unavailable when a Chinese character would leak.
- pl9_reader_english renders the parity volume with report_language=en and a
  fixed term table; yoga_engine carries combination_en / effects_en / strength_en.
- 75 fictional charts: zero Han, same line count, same numbers outside yoga tables.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
2026-09-29 19:26:00 +08:00

170 lines
7.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""English edition of the user-facing PL9 reader (TASK-report-english-edition-20260929).
The Chinese reader is ``pl9_reader_export._render_pl9_user_markdown``: it renders the
parity volume and then localises it with ordered string passes. The English
edition renders the same parity volume from the same packet with
``report_language = 'en'`` (planet and sign tokens stay English, yoga rows use the
engine's ``*_en`` twins), then:
1. drops the same audit-only lines and page references as the Chinese reader;
2. replaces the renderer's hard-coded Chinese labels with fixed English text
(``pl9_reader_english_terms``);
3. replaces engineering vocabulary with English public wording.
Nothing here computes. A Chinese character that survives is a missing entry,
and the caller treats the English edition as unavailable rather than shipping it.
"""
from __future__ import annotations
import copy
import re
try:
from scripts.pl9_reader_english_terms import PHRASES_EN, REGEX_EN
except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution
from pl9_reader_english_terms import PHRASES_EN, REGEX_EN
HAN = re.compile(r"[㐀-鿿豈-﫿]")
def han_leaks(text: str) -> list[str]:
"""Distinct runs of Han characters, for the self-check and for tests."""
return sorted(set(re.findall(r"[㐀-鿿豈-﫿]+", text)))
def _english_packet(packet: dict) -> dict:
"""Deep copy with the language flag set and yoga rows switched to their English twins."""
english = copy.deepcopy(packet)
english['report_language'] = 'en'
advanced = ((english.get('worksheets') or {}).get('advanced_systems') or {})
yoga = advanced.get('yoga') if isinstance(advanced, dict) else None
if isinstance(yoga, dict):
for key in ('detected_yogas', 'yogas'):
rows = yoga.get(key)
if not isinstance(rows, list):
continue
for row in rows:
if not isinstance(row, dict):
continue
if 'combination_en' in row:
row['combination'] = row['combination_en']
if 'effects_en' in row:
row['effects'] = row['effects_en']
if 'strength_en' in row:
row['strength'] = row['strength_en']
row['name_cn'] = row.get('name') or row.get('name_cn')
return english
def _public_wording_en(text: str) -> str:
"""English counterpart of the Chinese reader's engineering-vocabulary pass."""
for raw, public in (
("parameter_sensitive / unverified", "to be checked"),
("pyjhora_behavior_only / not_multiengine_parity", "reference"),
("blocked_and_conflict_fields_cannot_generate_final_predictions", "data only"),
("parameter_sensitive", "reference"),
("partial_verified", "listed"),
("pyjhora_behavior_only", "reference"),
("restricted_or_unclosed", "to be checked"),
("blocked_external_callable", "external replay pending"),
("not_applicable", "not listed"),
("missing_in_local", "not listed separately"),
("functional_role_not_returned", "not listed separately"),
("external_mudda_start_boundary", "reference start"),
("dasha_beginning_dates", "start dates"),
("dasha_ending_dates", "end dates"),
("natural_benefic", "natural benefic"),
("natural_malefic", "natural malefic"),
("functional_benefic", "functional benefic"),
("functional_malefic", "functional malefic"),
("functional_neutral", "functional neutral"),
("observed_relationships", "formed"),
("observed_empty", "not formed"),
("external_replay_pending", "pending"),
("annual_tajika_pack", "annual structure"),
("kp_monthly_report_packet", "monthly focus"),
("source_path", "source"),
("{planets}", "the planets involved"),
("blocked", "not listed"),
):
text = text.replace(raw, public)
text = re.sub(r"\b[a-zA-Z_]+_profile\b", "calculation setting", text)
text = re.sub(r"\b[a-zA-Z_]+_status\b", "status", text)
text = re.sub(r"\b[a-zA-Z_]+_producer\b", "calculation", text)
text = re.sub(r"\b[a-zA-Z_]+_path\b", "source", text)
text = re.sub(r"\bproducer\b", "calculation", text, flags=re.I)
text = re.sub(r"\bschema\b", "structure", text, flags=re.I)
text = re.sub(r"\bprofile\b", "setting", text)
text = re.sub(r"\bProfile\b", "Setting", text)
text = re.sub(r"\baudit\b", "check", text)
return text
def _translate(text: str) -> str:
for pattern, replacement in REGEX_EN:
text = pattern.sub(replacement, text)
for raw, public in PHRASES_EN:
text = text.replace(raw, public)
return text
def render_pl9_user_markdown_en(packet: dict) -> str:
"""Render the English reader volume from the same packet the Chinese one uses."""
try:
from scripts import pl9_reader_export as reader
except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution
import pl9_reader_export as reader
english = _english_packet(packet)
text = reader.render_pl9_parity_markdown(english)
title = reader._pl9_english_report_title(english)
text = re.sub(r"^#\s+.*$", f"# {title}", text, count=1, flags=re.MULTILINE)
text = re.sub(r"(PL9 p\d+(?:-p\d+)?[^)]*)", "", text)
text = re.sub(r"\s*\(PL9 p\d+(?:-p\d+)?[^)]*\)", "", text)
drop_fragments = (
"本册只编排原版",
"研究性证据",
"审计",
"producer",
"PyJHora",
"source_path",
"schema",
"blocked",
"parameter_sensitive",
"native producer",
"Report Layers",
"Worksheet Index",
"Report Pack Manifest",
"发布状态",
"质量检查",
"Chart Identity",
"不自动进入公开发布",
"对标主册",
)
lines: list[str] = []
for raw_line in text.splitlines():
if any(fragment in raw_line for fragment in drop_fragments):
continue
line = re.sub(r"p\d+-p\d+\s*", "", raw_line)
line = re.sub(r"p\d+\s*[·.-]\s*", "", line)
line = re.sub(r"^#+\s*p\d+\s*", lambda match: match.group(0).replace(match.group(0).split()[-1], ""), line)
line = line.replace(" · ", " ")
lines.append(line.rstrip())
text = "\n".join(lines)
text = _translate(text)
text = text.replace("PL9 p", "p. ").replace("PL9", "classical Jyotish")
text = _public_wording_en(text)
text = re.sub(r"(?m)^(#{2,6}\s*)p\d+(?:-p\d+)?\s*[·.-]?\s*", r"\1", text)
text = re.sub(r"(?m)^(#{1,6})\s+/\s+", r"\1 ", text)
text = re.sub(r"(?m)^(#{1,6}) {2,}", r"\1 ", text)
text = re.sub(r"^#\s+.*$", f"# {title}", text, count=1, flags=re.MULTILINE)
# Full-width punctuation from the shared renderer reads badly in English.
for wide, ascii_form in ((";", "; "), (",", ", "), (":", ": "), ("、", ", "), ("(", " ("), (")", ")"), ("—", " - ")):
text = text.replace(wide, ascii_form)
text = re.sub(r"(?<=\S) {2,}(?=\S)", " ", text)
text = re.sub(r"\( ", "(", text)
text = re.sub(r"[ \t]+\n", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text).strip() + "\n"
return text