Files
Jyotisha/scripts/pl9_language_terms.py
T
jesse-uxandJesse_Chen b8eef75871 fix(report): project annual sections and thicken dasha reading
完整数据版第二轮。年运章写出资料包里已有的节,没有数据的不新算。大运解释补回可能段落。瑜伽作用不再断言寿命。术语替换不进标识符。罗睺和计都不写守护第未列出宫。未合入 staging。
2026-10-08 22:30:32 +08:00

912 lines
47 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""PL9 report language normalization and leakage gates.
The term policy is based on this repository's StarTrack language-bridge
boundary and the Jyotish wording guidance in references/modern-language-guide.md.
It only changes reader-facing labels; it does not alter calculation payloads.
"""
from __future__ import annotations
import re
from typing import Any
def replace_natural_language_term(text: str, raw: str, public: str) -> str:
"""Replace one glossary term in prose only.
Identifiers and backtick spans stay as written. A trailing English plural
``s`` is consumed with the English word, so it is not left after Chinese.
"""
if not text or not raw:
return text
plural = "s?" if re.search(r"[A-Za-z]$", raw) and not raw.endswith("s") else ""
pattern = re.compile(rf"(?<![A-Za-z0-9_]){re.escape(raw)}{plural}(?![A-Za-z0-9_])")
parts = re.split(r"(`[^`]*`)", text)
return "".join(
part if index % 2 else pattern.sub(public, part)
for index, part in enumerate(parts)
)
_LIFESPAN_ASSERTION = re.compile(
r"长寿|寿命长|短寿|寿命|longevity|lifespan|long life|short life",
re.IGNORECASE,
)
_LIFESPAN_EFFECT_ZH = "传统上与体质和恢复力有关,仅作参考"
_LIFESPAN_EFFECT_EN = "Traditionally associated with constitution and recovery. Reference only."
def strip_han_plural_s(text: str) -> str:
"""Remove a Latin plural s left on Chinese after a glossary replace."""
return re.sub(r"([一-鿿])s\b", r"\1", text)
def soften_lifespan_wording(text: str, language: str = "zh") -> str:
"""Drop lifespan assertions from an effect cell. Yoga names are not passed here."""
if not text or not _LIFESPAN_ASSERTION.search(str(text)):
return text
if language == "en":
return _LIFESPAN_EFFECT_EN
return _LIFESPAN_EFFECT_ZH
ZH_TERM_REPLACEMENTS: tuple[tuple[str, str], ...] = (
("Natural friends", "天然友星"),
("Natural enemies", "天然敌星"),
("Temporary relation", "临时关系"),
("Planet 1", "行星1"),
("Planet 2", "行星2"),
("Planet", "行星"),
("planet", "行星"),
("sign", "星座"),
("longitude", "黄经"),
("field", "字段"),
("value", "数值"),
("title", "标题"),
("target", "目标"),
("anchor", "依据"),
("yoga", "瑜伽"),
("dt_ut", "UTC 时间"),
("dt_local", "本地时间"),
("sun_lon", "太阳黄经"),
("year_lord", "年主星"),
("field_status", "字段状态"),
("muntha_house", "Muntha 宫位"),
("muntha_sign", "Muntha 星座"),
("muntha_lord", "Muntha 主星"),
("munthesh", "Muntha 宫主"),
("Nakshatra Scheme", "星宿身体对应表"),
("Opinion 1", "说法一"),
("Opinion 2", "说法二"),
("Graha Avastha", "Graha Avastha(行星状态)"),
("Jagradadi", "醒睡状态"),
("Baladi", "年龄状态"),
("Lajjitadi", "羞惭等状态"),
("Deeptadi", "明亮等状态"),
("Shyanadi", "卧姿等状态"),
("Shayanadi", "卧姿等状态"),
("Degree", "落座度数"),
("Declination", "赤纬"),
("Speed", "速度"),
("House", "宫位"),
("Sign", "星座"),
("Lord", "宫主"),
("Score", "分数"),
("Bala", "年龄状态"),
("Jagrat", "醒睡状态"),
("Aspect", "相位"),
("Exact degree", "精确角度"),
("Actual difference", "实际角距"),
("Orb", "容许度"),
("Applying", "入相"),
("From house", "起始宫位"),
("Target house", "目标宫位"),
("Aspect type", "相位类型"),
("Special", "特殊相位"),
("Category", "类别"),
("Strength", "强度"),
("Combination", "组合条件"),
("Effects / notes", "作用说明"),
("friends:", "友星:"),
("enemies:", "敌星:"),
("True", "是"),
("False", "否"),
("conjunction", "合相"),
("opposition", "对冲"),
("special", "特殊组合"),
("solar_yoga", "太阳瑜伽"),
("lunar_yoga", "月亮瑜伽"),
("kalatra", "婚恋"),
("durbhaga", "不利组合"),
("moderate", "中等"),
("common", "常见"),
("strong", "强"),
("weak", "弱"),
("not separately listed", "未单独列出"),
("ownership", "守护宫"),
("dignity", "尊贵状态"),
("house placement", "落宫"),
("natural significations", "自然象征"),
("functional role", "功能角色"),
("Graha Avasthas - Planets and their Moods", "Graha Avastha(行星状态与情绪)"),
("Mangala / Sade Sati / Dosha Results", "火星婚姻煞 / Sade Sati / Dosha 结果"),
("Mangal / Kuja Dosha", "火星婚姻煞(Mangala/Kuja Dosha)"),
("Poorvashadha", "前阿沙陀"),
("Poorva Ashadha", "前阿沙陀"),
("Uttara Phalg.", "后破伽"),
("Uttara Phalguni", "后破伽"),
("Uttarashadha", "后阿沙陀"),
("Uttara Ashadha", "后阿沙陀"),
("Uttarabhadra", "后跋陀罗"),
("Uttara Bhadrapada", "后跋陀罗"),
("Moola", "Mula(根、本源)"),
("Mula", "Mula(根、本源)"),
("Both thighs", "双侧大腿"),
("Both feet", "双脚"),
("Private parts", "私密部位"),
("Sides of body", "身体两侧"),
("Back", "背部"),
("Left side", "左侧"),
("Left hand", "左手"),
("Waist", "腰部"),
("Waiste", "腰部"),
("Shins", "小腿胫部"),
("Swapna", "梦眠"),
("Dreamful", "多梦"),
("Mrita", "死寂"),
("State of death", "死寂状态"),
("Sushupti", "熟睡"),
("State of sleep", "睡眠状态"),
("Jagrad", "觉醒"),
("Wakefulness", "清醒"),
("Kumaravastha", "少年期"),
("Adolescence", "青春期"),
("Balavastha", "童年期"),
("Childhood", "童年"),
("Vriddha", "老年期"),
("Old age", "老年"),
("Mudit Kshobit", "喜悦中带扰动"),
("Trushit Mudit", "渴求中带喜悦"),
("Kshudit Trushit", "饥渴不安"),
("Mudit", "喜悦"),
("Khala", "粗劣"),
("Mudita", "愉悦"),
("Delighted", "愉悦"),
("Shanta", "平静"),
("Quiescent", "安静"),
("Deena", "匮乏"),
("Deficient", "不足"),
("Swastha", "稳定"),
("Stable", "稳定"),
("Nidra", "睡眠"),
("Sleep", "睡眠"),
("Gamenecchha", "欲行"),
("Eager to go", "急于行动"),
("Shayana", "卧躺"),
("Recumbent", "卧躺"),
("Sabhayam Vasti", "集会中"),
("in an assembly", "在集会中"),
("Infant", "婴幼期"),
("Young", "青年期"),
("Youth", "壮年期"),
("Dead", "死寂"),
("Awake", "觉醒"),
("Dreaming", "梦眠"),
("Trushita", "渴求"),
("Kshobhita", "扰动"),
("Vikala", "失衡"),
("Kautuka", "好奇"),
("Agama", "学习/趋近"),
("Panapara", "续宫"),
("Apoklima", "果宫"),
("Kendra", "角宫"),
("Benefic", "自然吉星"),
("Malefic", "自然凶星"),
)
ZH_ALLOWED_LATIN_TOKENS = {
"AI", "AD", "PD", "MD", "PL9", "KP", "BPHS", "PVR", "D1", "D2", "D3", "D4",
"D5", "D6", "D7", "D8", "D9", "D10", "D11", "D12", "D16", "D20", "D24",
"D27", "D30", "D40", "D45", "D60", "Rashi", "Navamsha", "Bhava", "Sripati",
"Sudarshan", "Sudarshana", "Upagraha", "Lagna", "Arudha", "Upapada", "Pada",
"Dasha", "Vimshottari", "Ashtottari", "Yogini", "Kala", "Chakra", "Jaimini",
"Narayana", "Sthira", "Drig", "Shoola", "Tribhagi", "Sade", "Sati", "Dhayya",
"Kantaka", "Varshaphala", "Tajika", "Mudda", "Patyayini", "Saham", "Sahams",
"Shadbala", "Ashtakavarga", "BAV", "SAV", "Pinda", "Avastha", "Vimsopaka",
"Paravatamsa", "Simhasanamsa", "Rahu", "Ketu", "Sun", "Moon", "Mars",
"Mercury", "Jupiter", "Venus", "Saturn", "Ashwini", "Bharani", "Krittika",
"Virupa", "Sthana", "Dig", "Drik", "Chesta",
"Rohini", "Mrigashira", "Ardra", "Punarvasu", "Pushya", "Ashlesha", "Magha",
"Purva", "Uttara", "Phalguni", "Hasta", "Chitra", "Swati", "Vishakha",
"Anuradha", "Jyeshtha", "Mula", "Ashadha", "Shravana", "Dhanishta",
"Shatabhisha", "Bhadrapada", "Revati",
"benefic", "malefic", "debilitated", "Neecha", "Bhanga", "and",
"functional_benefic", "functional_malefic", "functional_neutral",
"natural_benefic", "natural_malefic", "natural_role_not_returned",
"functional_role_not_returned",
}
ZH_FORBIDDEN_MIXED_PHRASES = (
"General effects during",
"General effects which are felt",
"Interpretation of the",
"Effects of",
"Interpretations based on the condition",
"Supported houses gain",
"The immediate focus is",
"brings house",
"is read from house",
"Maha 大运",
"Antar 大运",
"Pratyantar 大运",
"现实场域中阅读",
"这一现实场域",
)
EN_FORBIDDEN_CHARS = (":", ";", ",", "。", "、", "(", ")", "——")
_ZH_REGEX_REPLACEMENTS = tuple(
(raw, public)
for raw, public in ZH_TERM_REPLACEMENTS
if re.fullmatch(r"[A-Za-z0-9_ /:-]+", raw)
)
_ZH_LITERAL_REPLACEMENTS = tuple(
(raw, public)
for raw, public in ZH_TERM_REPLACEMENTS
if not re.fullmatch(r"[A-Za-z0-9_ /:-]+", raw)
)
_ZH_REGEX_REPLACEMENT_MAP = {raw: public for raw, public in _ZH_REGEX_REPLACEMENTS}
def _zh_regex_body(raw: str) -> str:
body = re.escape(raw)
if re.search(r"[A-Za-z]$", raw) and not raw.endswith("s"):
body += "s?"
return body
_ZH_REGEX_REPLACEMENT_PATTERN = re.compile(
r"(?<![A-Za-z0-9_])("
+ "|".join(
_zh_regex_body(raw)
for raw, _public in sorted(_ZH_REGEX_REPLACEMENTS, key=lambda item: len(item[0]), reverse=True)
)
+ r")(?![A-Za-z0-9_])"
)
_ZH_NAKSHATRA_GLOSSES = {
"前阿沙陀": "Purva Ashadha",
"后阿沙陀": "Uttara Ashadha",
"后破伽": "Uttara Phalguni",
"后跋陀罗": "Uttara Bhadrapada",
"阿湿毗尼": "Ashwini",
"巴拉尼": "Bharani",
"基利提卡": "Krittika",
"罗希尼": "Rohini",
"鹿首": "Mrigashira",
"阿尔德拉": "Ardra",
"普那婆苏": "Punarvasu",
"普沙": "Pushya",
"阿什列沙": "Ashlesha",
"摩伽": "Magha",
"前破伽": "Purva Phalguni",
"哈斯塔": "Hasta",
"吉多罗": "Chitra",
"斯瓦蒂": "Swati",
"毗舍佉": "Vishakha",
"阿奴罗陀": "Anuradha",
"杰耶什塔": "Jyeshtha",
"室罗伐那": "Shravana",
"陀尼湿陀": "Dhanishta",
"百药宿": "Shatabhisha",
"前跋陀罗": "Purva Bhadrapada",
"雷瓦蒂": "Revati",
}
_ZH_NAKSHATRA_GLOSS_PATTERN = re.compile(
"("
+ "|".join(re.escape(raw) for raw in sorted(_ZH_NAKSHATRA_GLOSSES, key=len, reverse=True))
+ r")(?![((])"
)
def _zh_regex_public(token: str) -> str:
if token in _ZH_REGEX_REPLACEMENT_MAP:
return _ZH_REGEX_REPLACEMENT_MAP[token]
if token.endswith("s") and token[:-1] in _ZH_REGEX_REPLACEMENT_MAP:
return _ZH_REGEX_REPLACEMENT_MAP[token[:-1]]
return token
def _replace_zh_regex_terms(markdown: str) -> str:
def apply(chunk: str) -> str:
return _ZH_REGEX_REPLACEMENT_PATTERN.sub(
lambda match: _zh_regex_public(match.group(1)),
chunk,
)
parts = re.split(r"(`[^`]*`)", markdown)
return "".join(part if index % 2 else apply(part) for index, part in enumerate(parts))
def normalize_zh_markdown_terms(markdown: str) -> str:
text = _replace_zh_regex_terms(markdown)
for raw, public in _ZH_LITERAL_REPLACEMENTS:
text = text.replace(raw, public)
text = text.replace("e特殊组合ly", "especially")
# Repair field-label substitutions that crossed token boundaries in the
# previous customer pin. These are presentation-only fixes; chart values
# and source evidence remain unchanged.
text = text.replace("强est_第", "最强宫")
text = text.replace("弱est_第", "最弱宫")
text = text.replace("strongest_house", "最强宫")
text = text.replace("weakest_house", "最弱宫")
text = text.replace("7 visible planets occupy exactly 5 distinct signs", "七颗可见行星恰好分布在五个不同星座")
text = text.replace("Many friends, talkative, skilled in various arts", "人际接触较多,善于表达,具多样艺术或技能倾向")
text = text.replace("Father died before birth, ancestral curse", "传统规则提示:涉及父系与家族议题,不能据此判断现实经历")
text = text.replace("Skilled in fine arts, music, dance; cultured and wealthy", "擅长艺术、音乐或舞蹈;重视文化修养与资源积累")
text = text.replace("Ridiculed by others, mocked, subject to derision", "可能面临误解、嘲讽或评价压力;需结合现实处境核验")
text = text.replace("Deception, distrust, household/family complications", "信任、家庭关系或居住事务可能较复杂;需结合现实处境核验")
text = text.replace("rare", "罕见")
text = text.replace("relationship_observation", "关系观察")
text = text.replace("solar 瑜伽", "太阳瑜伽")
text = text.replace("lunar 瑜伽", "月亮瑜伽")
text = re.sub(r"(\d+)\s+第from\s+月亮", r"从月亮起第\1宫", text)
text = re.sub(r"\bin\s+(从月亮起第\d+宫)", r"位于\1", text)
text = text.replace("_aspect_", "宫相位_")
text = _normalize_zh_bhava_factor_text(text)
text = text.replace("Bhava 年龄状态 十二宫力量", "Bhava Bala 十二宫力量")
text = text.replace("Shadbala / Bhava 年龄状态", "Shadbala / Bhava Bala")
text = text.replace("六维力量 / Bhava 年龄状态", "六维力量 / Bhava Bala")
text = text.replace("行星s", "行星")
text = text.replace("星宿 Scheme", "星宿身体对应表")
text = _normalize_mula_gloss(_restore_zh_nakshatra_glosses(_collapse_duplicate_zh_parentheses(text)))
text = text.replace("### 是 Solar Return", "### 真实太阳返照")
text = re.sub(
r"General effects during the 主大运 of (.+?) are read from the 行星's 自然象征, 落宫, owned houses, 尊贵状态, 星宿, and 功能角色\.",
r"\1主大运的总体作用,需要从该行星的自然象征、落宫、守护宫、尊贵状态、星宿与功能角色一起阅读。",
text,
)
text = text.replace(
"Interpretations based on the condition of the 行星 in the birth chart and divisional charts are as follows:",
"结合本命盘与分盘条件后,可按以下方式细读:",
)
text = re.sub(
r"(.+?)落在(.+?),首先把(.+?)放到(.+?)这一现实场域中阅读。",
r"\1落在\2,表示\3会主要通过\4来表现。",
text,
)
text = re.sub(r"第([1-4])足", r"第\1 Pada(星宿四分区)", text)
text = text.replace("Pada | 速度", "Pada(星宿四分区) | 速度")
text = _normalize_zh_dasha_prose(text)
text = _thicken_zh_dasha_density(text)
text = _normalize_mula_gloss(_restore_zh_nakshatra_glosses(_collapse_duplicate_zh_parentheses(text)))
return strip_han_plural_s(text)
def _normalize_zh_bhava_factor_text(markdown: str) -> str:
"""Localize compact Bhava Bala factor formulas in Chinese reports."""
text = markdown
body = r"(?:Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu|太阳|月亮|火星|水星|木星|金星|土星|北交点|南交点)"
text = re.sub(rf"({body})\s+in\s+H(\d+)", r"\1位于第\2宫", text)
text = re.sub(rf"({body})\s+in\s+(角宫|续宫|果宫)\s+\(H(\d+)\)", r"\1位于\2第\3宫", text)
text = re.sub(rf"({body})\s+aspects\s+H(\d+)\s+\((\d+)(?:st|nd|rd|th),\s+(自然吉星|自然凶星)\)", r"\1以\3宫相位照第\2宫(\4)", text)
text = re.sub(rf"({body})\s+aspects\s+H(\d+)\s+\((\d+)(?:st|nd|rd|th)\)", r"\1以\3宫相位照第\2宫", text)
text = re.sub(r"\((自然吉星|自然凶星)\)", r"(\1)", text)
text = text.replace("火星宫相位_4", "火星特殊4宫相位")
text = text.replace("火星宫相位_8", "火星特殊8宫相位")
text = text.replace("木星宫相位_5", "木星特殊5宫相位")
text = text.replace("木星宫相位_9", "木星特殊9宫相位")
text = text.replace("土星宫相位_3", "土星特殊3宫相位")
text = text.replace("土星宫相位_10", "土星特殊10宫相位")
return text
def _normalize_mula_gloss(markdown: str) -> str:
"""Make the Mula gloss idempotent across repeated language passes."""
text = markdown
root_gloss = r"根[,、/]本源"
text = re.sub(rf"Mula(?:[((]{root_gloss}[))])+", "Mula(根、本源)", text)
text = re.sub(rf"根宿[((]Mula[((]{root_gloss}[))][))]", "Mula(根、本源)", text)
text = re.sub(rf"根宿[((](Mula(?:[((]{root_gloss}[))])+)[))]", "Mula(根、本源)", text)
return text
def _collapse_duplicate_zh_parentheses(markdown: str) -> str:
"""Remove duplicated Chinese glosses such as 后破伽(后破伽)."""
return re.sub(r"([\u4e00-\u9fff][\u4e00-\u9fff·/-]{0,12})[((]\1[))]", r"\1", markdown)
def _restore_zh_nakshatra_glosses(markdown: str) -> str:
"""Keep the original Sanskrit/English nakshatra label in Chinese reports."""
return _ZH_NAKSHATRA_GLOSS_PATTERN.sub(
lambda match: f"{match.group(1)}({_ZH_NAKSHATRA_GLOSSES[match.group(1)]})",
markdown,
)
def _thicken_zh_dasha_density(markdown: str) -> str:
"""Restore reader-facing density for translated AD/PD prose.
This adds reading instructions tied to already-present fields. It does not
introduce new predictions or alter dates.
"""
lines = markdown.splitlines()
out: list[str] = []
for index, line in enumerate(lines):
out.append(line)
lookahead = "\n".join(lines[index + 1:index + 6])
if "这一子运不是单独结论" in lookahead or "这个次子运只承担短周期细化" in lookahead:
continue
ad_match = re.search(
r"(.+?)主大运中的(.+?)子运,会把(.+?)的自然主题带入当前阶段。",
line,
)
if ad_match:
md_lord, ad_lord, theme_lord = ad_match.groups()
out.extend([
"",
(
f"这一子运不是单独结论,而是把{theme_lord}的自然象征放进{md_lord}主大运的背景里筛选。"
f"阅读时先看{ad_lord}本身的落宫、守护宫、尊贵状态与星宿,再看它和主大运星之间是否互相支持。"
),
"",
(
"若同一宫位或同一行星在分盘、Ashtakavarga、年度盘和行运中重复出现,"
"该主题的可见度会提高;若这些层彼此冲突,则应把结果视为阶段性倾向,而不是孤立断语。"
),
])
continue
pd_match = re.search(
r"(.+?)子运中的(.+?)次子运,会在(.+?)背景下呈现(.+?)的短周期结果。",
line,
)
if pd_match:
ad_lord, pd_lord, md_lord, theme_lord = pd_match.groups()
out.extend([
"",
(
f"这个次子运只承担短周期细化:{pd_lord}会把{theme_lord}的具体征象带入{ad_lord}子运,"
f"并受{md_lord}主背景限制。"
),
"",
(
"因此这里优先用于判断事情推进的节奏、触发点和轻重缓急;实际领域仍需回到本命落宫、"
"分盘重复、年度盘和当前行运共同核对。"
),
])
continue
return "\n".join(out)
def _normalize_zh_dasha_prose(markdown: str) -> str:
text = markdown
replacements = {
"mind, mother, water, residence, 迁移与出行, public mood, nourishment, fertility, trade, learning, and changeable fortune become active.": "心智、母亲、居住、水象事务、迁移与出行、公众情绪、滋养、生育、贸易、学习与变化中的运势会被启动。",
"There can be interest in mantra, teachers, sacred learning, art, hospitality, garments, ornaments, land, and watery products.": "这一阶段可能增加对咒语、师长、神圣知识、艺术、服务接待、衣物、饰品、土地与水相关事务的兴趣。",
"The mind can become lively, sensitive, restless, affectionate, and responsive to family or social approval.": "心绪可能更活跃、敏感、容易波动,也更重视家庭回应与社会认可。",
"When supported, it brings comfort from home, spouse, children, servants, conveyances, food, clothing, education, fame, and fulfilled desires.": "条件良好时,可带来家庭、伴侣、子女、协助者、交通工具、饮食衣物、教育、名声与愿望满足方面的支持。",
"New places, cultivation, trade, and public-facing work may become profitable.": "新地点、耕作/经营、贸易与面向公众的工作可能带来收益。",
"When afflicted, it can show fear, anger, wavering judgment, family strain, maternal concern, contaminated food, 迁移与出行 pressure, and loss of old security.": "受克时,可能表现为恐惧、怒气、判断摇摆、家庭压力、母亲相关担忧、饮食不洁、迁移压力,以及旧有安全感减弱。",
"Friends may not give full support, and emotional decisions need steadier review.": "朋友支持可能不足,情绪化决定需要更稳妥地复核。",
"absolute Virupas; precise Sthana/Dig/Kala/Drik; bounded BPHS Chesta": "以绝对 Virupa 计分,包含 Sthana、Dig、Kala、Drik 与受限 BPHS Chesta",
"fear, anger, wavering judgment, family strain, maternal concern, contaminated food, 迁移与出行 pressure, and loss of old security": "恐惧、怒气、判断摇摆、家庭压力、母亲相关担忧、饮食不洁、迁移压力,以及旧有安全感减弱",
"unusual openings, foreign benefit, technical or political leverage, and gains through nontraditional channels": "特殊机会、海外或异地收益、技术或权力杠杆,以及非传统渠道带来的收获",
"fear, bondage, deception, sudden reversals, illness, scandals, and trouble from authorities or hidden enemies": "恐惧、束缚、欺骗、突发反转、疾病、名誉风波,以及来自权威或暗中对手的麻烦",
"promotion, respect, wealth, grains, clothes, gold, children, good counsel, and fulfillment of aims": "晋升、尊重、财富、物资、衣物、贵金属、子女、良好建议与目标达成",
"trouble to spouse or children, loss through indulgence, legal worries, fever, or grief from family obligations": "伴侣或子女方面的麻烦、因放纵带来的损耗、法律忧虑、发热或家庭责任带来的忧伤",
"durable property, work stability, public recognition, gain through labor, and capacity to endure responsibility": "稳定资产、工作稳定、公众认可、辛勤劳动带来的收益,以及承担责任的耐力",
"pain, quarrels, obstruction, loss of wealth, separation from parents, danger through low morale, and heavy duties": "痛苦、争执、阻碍、财富损耗、与父母分离、士气低落带来的风险,以及沉重职责",
"knowledge, wealth, garments, jewels, name, ritual merit, useful alliances, and clever problem solving": "知识、财富、衣物、珠宝、名声、仪式功德、有用联盟与灵活的问题解决能力",
"poor judgment, fever, anxiety, argument, loss through paperwork, and instability in business": "判断不稳、发热、焦虑、争论、文书损耗与商业不稳定",
"renunciation, spiritual practice, diagnostic skill, hidden knowledge, and the ability to cut through confusion": "放下执着、灵性修持、诊断能力、隐秘知识,以及切断混乱的能力",
"fear from enemies, loss of reasoning, sudden obstruction, disease, grief, and restless 迁移与出行": "来自对手的恐惧、理性受损、突发阻碍、疾病、忧伤与不安定的迁移出行",
"spouse happiness, property, conveyance, ornaments, fine food, friendship, and pleasant company": "伴侣愉悦、资产、交通工具、饰品、美食、友谊与愉快陪伴",
"indulgence, waste, disputes with women or partners, loss through luxury, and ailments connected with reproductive or urinary balance": "放纵、浪费、与女性或伴侣的争执、奢侈带来的损耗,以及生殖或泌尿平衡相关不适",
"recognition, administrative help, confidence, vehicles, ornaments, and respected duties": "认可、行政助力、自信、交通工具、饰品与受尊重的职责",
"pressure from government, rivals, fire, theft, father-related strain, feverish conditions, and restlessness": "来自政府/权威、竞争者、火灾、盗损、父亲相关压力、发热状态与内在不安的压力",
"courage, stamina, property gains, productive labor, victory over enemies, and technical execution": "勇气、体力、资产收益、有效劳动、战胜对手与技术执行力",
"harsh speech, weapon or fire trouble, stomach ailments, injuries, litigation, and hot-tempered separations": "言语尖锐、武器或火相关麻烦、胃部不适、伤损、诉讼,以及急躁导致的分离",
"stamina, property, gains, productive labor, victory over enemies": "体力、资产、收益、有效劳动和战胜对手",
"foreign places, outsiders, ambition, unusual gains, fear, poisons, snakes, politics, sudden turns, obsession, and unconventional paths become prominent.": "异地、外来者、野心、特殊收益、恐惧、毒性/中毒象征、蛇象、政治、突发转折、执念与非传统路径会变得突出。",
"The period can bring success in ventures, conveyances, garments, distant direction gains, and contact with powerful or foreign circles.": "这一时期可能带来事业尝试、交通工具、衣物、远方收益,以及与有权势或海外/异地圈层接触方面的机会。",
"If afflicted, it brings defamation, business loss, fever, enemies, anxiety, spouse or child distress, and loss of reputation.": "若受克,可能带来名誉受损、事业损失、发热、对手、焦虑、伴侣或子女方面的压力,以及声望下降。",
"It favors strategic risk only when facts and ethics are kept clear.": "只有事实清楚、边界正当时,策略性冒险才更有利。",
"北交点/罗睺-related mantra, restraint, and charity are traditional remedial themes.": "与北交点/罗睺相关的咒语、克制与布施,是传统辅助主题。",
"labor, delay, servants, masses, endurance, old people, land, minerals, iron, chronic pressure, discipline, and separation themes become prominent.": "劳动、延迟、服务者、大众、耐力、长者、土地、矿物、铁器、长期压力、纪律与分离主题会变得突出。",
"The period can give property, recognition, steady gains, service authority, and patient achievement after effort.": "这一时期可能带来资产、认可、稳定收益、服务型权责,以及努力之后的耐心成果。",
"If afflicted, it brings loss of friends, disputes with relatives, fear, confinement, fatigue, rheumatic or digestive trouble, and wandering.": "若受克,可能带来朋友损失、亲属争执、恐惧、受限感、疲惫、风湿或消化问题,以及漂泊。",
"The middle of the period can be more productive than its beginning or end.": "这一时期的中段往往比开始和结束阶段更容易产生成果。",
"Traditional balancing themes include service, discipline, humility, and 土星-related charity.": "传统平衡主题包括服务、纪律、谦逊,以及与土星相关的布施。",
"The period can detach the person from stale ambitions and force a simpler, more inward path.": "这一时期可能使人脱离旧有野心,被迫走向更简单、更内向的路径。",
"If afflicted, business disturbance, loss of wealth, stomach or eye trouble, fear, conflict, and news of death or separation can appear.": "若受克,可能出现事业扰动、财富损耗、胃部或眼部问题、恐惧、冲突,以及死亡或分离相关消息。",
"It can produce intermittent gains after pressure or 迁移与出行.": "它可能在压力或迁移之后带来间歇性收益。",
"Durga worship, protective mantra, and charity are traditional balancing themes.": "杜尔迦崇拜、保护性咒语与布施,是传统平衡主题。",
"authority, government, 状态, father, medicine, land, fire, command, visibility, and public responsibility become more prominent.": "权威、政府、地位、父亲、医疗、土地、火象事务、指挥权、能见度与公共责任会更突出。",
"spiritual discipline, mantra, ritual, leadership, and contact with officials or influential people can increase.": "灵性纪律、咒语、仪式、领导力,以及与官员或有影响力人士的接触可能增加。",
"anxiety, heat, separation from close relatives, disputes with authority, eye, teeth, abdomen, or vitality concerns can also need attention.": "焦虑、热性问题、与近亲分离、和权威争执,以及眼睛、牙齿、腹部或生命力相关议题也需要留意。",
"It can push philanthropic action, construction, public work, or a clearer place in society.": "它可能推动公益行动、建设事务、公共工作,或让个人在社会中取得更明确的位置。",
"The person may feel proud, isolated, or forced to leave a familiar place for work or duty.": "当事人可能感到自尊增强但也更孤立,或因工作与职责被迫离开熟悉环境。",
"energy, courage, land, siblings, weapons, competition, surgery, heat, property disputes, technical action, and decisive breaks become prominent.": "能量、勇气、土地、手足、武器/工具、竞争、手术、热性事务、地产争议、技术行动与果断切割会更突出。",
"The period increases initiative and can bring land, honors, official recognition, or success through bold effort.": "这一时期会增强主动性,也可能通过大胆行动带来土地、荣誉、官方认可或成功。",
"teachers, children, wisdom, religion, wealth, counsel, learning, law, protection, ceremonies, and honorable expansion become prominent.": "师长、子女、智慧、宗教、财富、建议、学习、法律、保护、仪式与体面的扩展会更突出。",
"If afflicted, body pain, family strain, displeasure of authority, missed goals, or loss through poor counsel can appear.": "若受克,可能出现身体疼痛、家庭压力、权威不满、目标落空,或因建议不当导致损失。",
"education, speech, business, writing, analysis, accounts, friends, trade, negotiation, youth, and skills become prominent.": "教育、言语、商业、写作、分析、账务、朋友、贸易、谈判、年轻人事务与技能会更突出。",
"separation, austerity, sharp insight, wandering, spiritual pressure, enemies, sudden loss, animals, cuts, and hidden causes become prominent.": "分离、苦修、锐利洞察、漂泊、灵性压力、对手、突发损失、动物、切割伤与隐秘原因会更突出。",
"comforts, spouse, relationships, ornaments, vehicles, art, luxury, pleasures, agreements, water products, and refined company become prominent.": "舒适、伴侣、关系、饰品、交通工具、艺术、享受、愉悦、协议、水产品与雅致社交会更突出。",
"The period can bring beauty, garments, jewels, enjoyment, hospitality, marriage themes, and social pleasures.": "这一时期可能带来美感、衣物、珠宝、享受、接待、婚姻主题与社交愉悦。",
"If afflicted, comforts are disturbed by excess expense, sensual distraction, household conflict, fever, headaches, or relationship strain.": "若受克,舒适感可能被过度开销、感官分心、家庭冲突、发热、头痛或关系压力扰动。",
}
if replacements:
pattern = re.compile("|".join(re.escape(raw) for raw in sorted(replacements, key=len, reverse=True)))
text = pattern.sub(lambda match: replacements[match.group(0)], text)
text = re.sub(
r"Supported houses gain expression through (.+?); strained houses require steadier handling, especially where the same houses repeat in (?:大运|主大运), divisional charts, Ashtakavarga, or the annual chart\.",
r"受支持的宫位会通过\1获得表达;若同一宫位在大运、分盘、Ashtakavarga 或年度盘中反复承压,则需要更稳妥地处理。",
text,
)
text = re.sub(
r"When supported, (.+?) (?:gives|brings) (.+?)\.",
r"条件良好时,\1会带来\2。",
text,
)
text = re.sub(
r"When strained, (?:it |)(?:can |may |)(?:show|bring) (.+?)\.",
r"受压时,可能表现为\1。",
text,
)
text = re.sub(
r"When afflicted, (?:it |)(?:can |may |)(?:show|bring) (.+?)\.",
r"受克时,可能表现为\1。",
text,
)
text = re.sub(
r"In the (.+?), (.+?)'s significations give the short-period result inside the (.+?) background\.子运中的(.+?)次子运",
r"\1子运中的\4次子运,会在\3背景下呈现\2的短周期结果。",
text,
)
text = re.sub(
r"Consideration: (.+?) is checked from (.+?) against the classical marriage-sensitive houses\. In this chart (.+?) is listed in house (.+?) in (.+?); the detector severity is (.+?)\. Classical reference family: (.+?)\.",
r"判定方式:\1会从\2出发,检查传统婚恋敏感宫位。本盘中,\3位于第\4宫、\5,检测强度为\6。参考文献族:\7。",
text,
)
text = text.replace(
"Consideration: this Dosha row is shown only when the local detector returns a chart-specific condition. Classical reference family is attached in the source map.",
"判定方式:只有本盘检测到对应条件时才列出该 Dosha;对应古典参考族已在资料层登记。",
)
text = text.replace(
"Result: when present, Kuja/Mangala Dosha is read as heat, impatience, conflict, or pressure around partnership handling; when absent or cancelled, the report does not promote it as a dominant relationship obstacle. The result must be read with the seventh house, 金星, Upapada, Navamsha, current 大运, and partner-chart comparison where available.",
"结果:若火星婚姻煞成立,通常表示关系处理中的热度、急躁、冲突或压力;若不存在或有抵消条件,则不应把它提升为主导关系障碍。该项需要与第七宫、金星、Upapada、Navamsha、当前大运以及可用的伴侣盘一起阅读。",
)
text = text.replace(
"Result: this row contributes a supporting condition and is not promoted over the core chart, 大运, and divisional evidence.",
"结果:该项只作为辅助条件,不高于本命盘、大运和分盘证据。",
)
text = text.replace(
"Cancellation: no explicit cancellation reasons were returned by the local detector; partner-chart matching remains a separate comparison step.",
"抵消条件:本地检测器未返回明确抵消原因;伴侣盘匹配仍属于独立比较步骤。",
)
text = text.replace(
"本节按传统大运的总体作用与具体命盘条件两层结构展开,参考:Light on Life An Introduction to the Astrology of India; Predict Effectively through Yogini 大运。",
"本节按传统大运的总体作用与具体命盘条件两层结构展开,参考《印度占星生命之光》和《Yogini Dasha 实战预测》的结构。",
)
text = text.replace(
"When supported in the chart, this period gives 认可、行政助力、自信、交通工具、饰品与受尊重的职责.",
"命盘条件良好时,这一阶段可带来认可、行政助力、自信、交通工具、饰品与受尊重的职责。",
)
text = text.replace(
"When supported in the chart, this period gives 体力、资产、收益、有效劳动和战胜对手.",
"命盘条件良好时,这一阶段可带来体力、资产、收益、有效劳动和战胜对手。",
)
text = re.sub(
r"When supported in the chart, this period gives (.+?)\.",
r"命盘条件良好时,这一阶段可带来\1。",
text,
)
text = re.sub(
r"When afflicted, it brings (.+?)\.",
r"受克时,可能带来\1。",
text,
)
text = re.sub(
r"If afflicted, (.+?)\.",
r"若受克,\1。",
text,
)
text = text.replace("Remedies / Supportive 修持s", "传统辅助建议")
return text
# Fragments left after the longer English glossary. Longer keys are applied first.
_EN_RESIDUAL_HAN: tuple[tuple[str, str], ...] = (
("没有候选以吉相位照年盘", "no candidate aspects the annual Lagna by a benefic aspect"),
("按子运起点星座奇偶", "by the odd or even sign where the sub-period starts"),
("按七级比较所落星座", "compare the occupied signs across seven levels"),
("两主分落别的星座", "the two lords fall in other signs"),
("星座或其主所落星座", "the sign, or the sign its lord occupies"),
("实际位置判断是否", "judge from the actual position whether"),
("以吉相位照年盘", "aspect the annual Lagna by a benefic aspect"),
("不单独增加结论的确定性", "does not by itself make a conclusion more certain"),
("年数数到该", "the year count reaches that sign"),
("一律顺行", "always direct"),
("只倒算计都", "only Ketu is counted backward"),
("未进打分", "not used in the score"),
("时间轴结论", "timeline conclusion"),
("主三角色", "primary triangle role"),
("七轮裁定", "seven-round decision"),
("书内分歧", "the book itself disagrees"),
("待替换为", "still to be replaced by"),
("取为年主", "is taken as the year lord"),
("五分力量", "five-fold strength"),
("五类候选", "five candidate groups"),
("以凶相位照", "aspect by a malefic aspect"),
("阶段框架", "phase frame"),
("起始年龄", "start age"),
("结束年龄", "end age"),
("六重力量", "six-fold strength"),
("矩阵源数据", "matrix source"),
("相位矩阵", "aspect matrix"),
("灵魂使命", "soul aim"),
("核心自我", "core self"),
("人生最高目标", "highest aim"),
("主要谋士", "main advisor"),
("权力代理", "delegated authority"),
("兄弟姐妹", "siblings"),
("冒险精神", "appetite for risk"),
("家庭根基", "home base"),
("情感安全感", "emotional security"),
("祖先业力", "ancestral pattern"),
("传统传承", "inherited tradition"),
("智能成果", "crafted results"),
("竞争对手", "rivals"),
("转化力量", "capacity for change"),
("配偶特质", "spouse traits"),
("伴侣关系", "partnership"),
("书例校准", "book-example calibration"),
("八分法", "eightfold division"),
("要素数量", "factor count"),
("要素总数", "factor total"),
("综合性质", "overall nature"),
("基本年数", "base years"),
("并列参考", "side reference"),
("双主星", "dual lords"),
("子运方向", "sub-period direction"),
("身份按同", "identity follows the same"),
("明显迹象", "a clear indication"),
("严重程度", "severity"),
("轴线之间", "between the axis"),
("颗行星", "planets"),
("高峰期", "peak phase"),
("下降期", "setting phase"),
("最困难", "most difficult"),
("需结合", "read together with"),
("灵魂星", "soul planet"),
("兄弟星", "sibling planet"),
("母亲星", "mother planet"),
("父亲星", "father planet"),
("子女星", "child planet"),
("障碍星", "obstacle planet"),
("配偶星", "spouse planet"),
("旧算法", "previous algorithm"),
("按原书", "per the book"),
("起运", "period start"),
("从主", "from the lord"),
("级取", "the level selects"),
("两主", "two lords"),
("等分", "equal parts"),
("未采用", "not used"),
("星座奇偶", "odd or even sign"),
("的子运", "sub-period"),
("待评估", "not yet assessed"),
("进行medium", "underway"),
("当before", "current "),
("原始值", "raw value"),
("主层", "main level"),
("子层", "sub level"),
("年数", "years"),
("主星", "lord"),
("序号", "index"),
("起始", "start"),
("结束", "end"),
("星座", "sign"),
("相位", "aspect"),
("年度", "annual"),
("细项", "detail"),
("年返", "solar return"),
("摘要", "summary"),
("项目", "item"),
("数值", "value"),
("方法", "method"),
("校验", "check"),
("辅助", "supporting"),
("算法", "algorithm"),
("并列", "listed alongside"),
("顺序", "order"),
("修正", "adjustment"),
("最终", "final"),
("正文", "body text"),
("阶段", "phase"),
("是否", "whether"),
("只有", "only"),
("特殊", "special"),
("方向", "direction"),
("勇气", "courage"),
("母亲", "mother"),
("父亲", "father"),
("子女", "children"),
("创造力", "creativity"),
("学生", "students"),
("障碍", "obstacles"),
("质量", "quality"),
("顺行", "direct"),
("逆行", "retrograde"),
("比较", "compare"),
("另一", "another"),
("部分", "partial"),
("进行", "in progress"),
("占据planet尊贵", "occupied planet dignity"),
("占据planet数", "occupied planet count"),
("partial减weak", "partially reduced"),
("成立条件", "conditions"),
("不in煞", "not in a malefic house"),
("迹象", "indication"),
("相邻", "adjacent"),
("占据", "occupies"),
("尊贵", "dignity"),
("克", "restrains"),
("其", " its "),
("减", "reduced"),
("最", "most "),
("条", "items"),
("的", " of "),
("不", "not "),
("吉", "benefic"),
("数", "count"),
("煞", "malefic"),
("看", "see"),
("起", "from"),
("版", "edition"),
("级", "level"),
("主", "lord"),
("期", "phase"),
("星", "planet"),
("度", "degree"),
("从", "from"),
("取", "take"),
("所", "at"),
("较", "comparatively"),
("当", "when"),
("按", "by"),
("有", "present"),
("无", "absent"),
("表", "table"),
)
def apply_en_residual_han(markdown: str) -> str:
"""Translate Han fragments that survive the longer English glossary."""
text = markdown
for raw, public in sorted(_EN_RESIDUAL_HAN, key=lambda item: len(item[0]), reverse=True):
text = text.replace(raw, public)
return text
def normalize_en_markdown_terms(markdown: str) -> str:
text = markdown.replace("(", "(").replace(")", ")").replace("、", ", ")
text = (
text.replace("dt_ut", "UTC time")
.replace("dt_local", "Local time")
.replace("sun_lon", "Sun longitude")
.replace("year_lord", "Year lord")
.replace("field_status", "Field status")
.replace("muntha_house", "Muntha house")
.replace("muntha_sign", "Muntha sign")
.replace("muntha_lord", "Muntha lord")
)
text = re.sub(
r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)(\d{1,2})\b",
r"\1 \2",
text,
)
return apply_en_residual_han(_thicken_en_dasha_density(text))
def _thicken_en_dasha_density(markdown: str) -> str:
"""Add chapter-level AD context without repeating boilerplate on every PD row."""
lines = markdown.splitlines()
out: list[str] = []
current_md = None
ad_context_added = False
for index, line in enumerate(lines):
md_heading = re.match(r"^#{2,4}\s+(.+?\bMaha Dasha\b.*)$", line)
if md_heading:
next_md = md_heading.group(1)
if next_md != current_md:
current_md = next_md
ad_context_added = False
out.append(line)
lookahead = "\n".join(lines[index + 1:index + 6])
if (
"This Antar Dasha is not read as a standalone verdict" in lookahead
or "This Pratyantar Dasha is the short-cycle refinement" in lookahead
):
continue
ad_match = re.search(
r"The Antar Dasha of (.+?) activates (.+?)'s natural themes inside the broader Maha Dasha of (.+?)\.",
line,
)
if ad_match and not ad_context_added:
ad_lord, theme_lord, md_lord = ad_match.groups()
out.extend([
"",
(
f"This Antar Dasha is not read as a standalone verdict. It places {theme_lord}'s "
f"natural significations inside the background of the {md_lord} Maha Dasha, then "
f"filters them through {ad_lord}'s house placement, house ownership, dignity, "
"nakshatra lord, conjunctions, and divisional-chart repetition."
),
"",
(
"When the same planet or house repeats through divisional charts, Ashtakavarga, "
"annual factors, and transits, the theme becomes more visible. When those layers "
"conflict, the result should be read as a staged tendency rather than an isolated statement."
),
])
ad_context_added = True
continue
pd_match = re.search(
r"In the Pratyantar Dasha of (.+?) in the Antar Dasha of (.+?), (.+?)'s significations give the short-period result inside the (.+?) background\.",
line,
)
if pd_match:
# The row already contains the concrete PD lord and timing context.
# Keep it, but do not append the same methodology paragraph per node.
continue
return "\n".join(out)
def language_quality_receipt(markdown: str, *, language: str) -> dict[str, Any]:
violations: list[str] = []
if language == "en":
chinese_chars = len(re.findall(r"[\u4e00-\u9fff]", markdown))
if chinese_chars:
violations.append(f"chinese_characters:{chinese_chars}")
for char in EN_FORBIDDEN_CHARS:
if char in markdown:
violations.append(f"fullwidth_punctuation:{char}")
else:
for phrase in ZH_FORBIDDEN_MIXED_PHRASES:
if phrase in markdown:
violations.append(f"mixed_template_phrase:{phrase}")
suspicious_lines: list[str] = []
for line_no, line in enumerate(markdown.splitlines(), 1):
if line.lstrip().startswith("|"):
continue
if not re.search(r"[\u4e00-\u9fff]", line):
continue
words = re.findall(r"\b[A-Za-z][A-Za-z/_-]{2,}\b", line)
non_allowed = [word for word in words if word not in ZH_ALLOWED_LATIN_TOKENS]
if len(non_allowed) >= 5:
suspicious_lines.append(f"{line_no}:{','.join(non_allowed[:8])}")
if len(suspicious_lines) >= 20:
break
if suspicious_lines:
violations.append("mixed_english_dense_lines:" + ";".join(suspicious_lines))
return {
"schema": "pl9.language_quality_receipt.v1",
"language": language,
"status": "pass" if not violations else "fail",
"violations": violations,
}