From 98c98aa535aae34f4c954de015b2aa0ce11d86c6 Mon Sep 17 00:00:00 2001 From: jesse-ux Date: Thu, 8 Oct 2026 16:05:18 +0800 Subject: [PATCH] feat(report): replace the reader edition with the full-data report Generate the full-data edition (pl9_personal_long_report.v4) and remove the reader edition. English still contains Han, so the API withholds it (BUG-1272, investigating). This commit is not pushed to staging. --- CHANGELOG.md | 8 + docs/BUG_HISTORY.md | 16 + ...GRESS-report-full-data-edition-20261007.md | 115 + docs/tasks/README.md | 4 +- .../report-full-data-edition-20261008.md | 19 + frontend/DESIGN.md | 3 +- frontend/docs/VOICE.md | 2 +- frontend/src/app/globals.css | 26 + .../personal-report-center.tsx | 2 +- .../personal-report-markdown-view.tsx | 117 +- .../personal-report/personal-report-page.tsx | 10 + .../lib/personal-report-longform-generate.ts | 2 +- .../personal-report-markdown-view.test.ts | 23 + .../personal-report-markdown-view.test.tsx | 6 +- ...rofessional-report-reference-route.test.ts | 5 +- .../tests/surface-feedback-20260929.test.tsx | 4 +- scripts/pl9_en_glossary.py | 530 ++++ scripts/pl9_full_data_export.py | 2186 +++++++++++++++++ scripts/pl9_language_terms.py | 669 +++++ scripts/pl9_reader_export.py | 41 +- scripts/professional_report_reference.py | 44 +- tests/test_bhava_bala_formal.py | 15 +- tests/test_chara_karaka_8_bphs_order.py | 2 +- tests/test_full_data_domain_reader.py | 60 + tests/test_narayana_legacy_label.py | 10 +- tests/test_neecha_bhanga_conditions.py | 2 +- tests/test_report_chart_blank_columns.py | 45 +- tests/test_report_english_edition.py | 80 +- tests/test_report_reader_main.py | 81 +- tests/test_shadbala_minimum_first.py | 8 +- tests/test_year_lord_basis_label.py | 14 +- tests/test_year_lord_blocked_no_fallback.py | 23 +- 32 files changed, 4039 insertions(+), 133 deletions(-) create mode 100644 docs/tasks/PROGRESS-report-full-data-edition-20261007.md create mode 100644 docs/testing/report-full-data-edition-20261008.md create mode 100644 scripts/pl9_en_glossary.py create mode 100644 scripts/pl9_full_data_export.py create mode 100644 scripts/pl9_language_terms.py create mode 100644 tests/test_full_data_domain_reader.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 4258d39b..a0db361d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,14 @@ - 行星、上升和宫头以前用含章动的视位置,减去不含章动的平岁差,全体偏一个很小的角度。现在与 Swiss Ephemeris 的恒星黄道使用同一套章动(BUG-1275)。 - 大运日期可能因此前后移动 2–3 天。同一份生时校正输入的分数会变。算法身份从 scoring-11 升到 scoring-12。以前做好的校正记录仍能打开,只读,并可以按新算法再比一次。Skill 版本不变。 +## 2026-10-08 — 个人报告改为完整数据版(未上线) + +- 「生成报告」改为完整数据版:星盘、力量、大运、年运和瑜伽写进正文,后面有一份结构化资料附录。阅读版的生成入口已删除。 +- 以前生成的报告还能打开,只读,不会自动重算。要新版需重新生成。 +- 英文清理后仍有汉字,新报告先不提供英文,也不显示「中文 / English」。导出时会说明英文没有生成成功(BUG-1272,尚未解决)。 +- 报告超过约 20 万字时,阅读页一次只显示一章,可用目录或「上一章 / 下一章」换章。下载仍是原来的「导出」。 +- 不改数据库。Skill 版本不变。 + ## 2026-10-08 — 特殊上升点在没有 PyJHora 的环境里也能算出(未上线) - 生产镜像装不了 PyJHora,Bhava、Hora、Ghati、Vighati、Sree、Indu、Pranapada、Varnada 这八个点以前算不出来。现在用 Swiss Ephemeris 在本进程里算(BUG-1273)。 diff --git a/docs/BUG_HISTORY.md b/docs/BUG_HISTORY.md index ef7b4b4b..e0fd219c 100644 --- a/docs/BUG_HISTORY.md +++ b/docs/BUG_HISTORY.md @@ -17115,6 +17115,22 @@ - 复发自:无。 - 修复版本:无(运行机侧)。 +## BUG-1272 | 完整数据版英文报告清理后仍含汉字,接口整份撤回英文 + +- 状态:investigating +- 首次发现 / 最近更新:2026-10-08 / 2026-10-08 +- 来源:`TASK-report-full-data-edition-20261007`。上游任务书已记录乔布斯盘英文清理失败(`chinese_characters:14981`)。 +- 影响面:新生成的个人报告英文版。中文完整数据版不受影响。已生成的旧报告仍按当时保存的正文打开。 +- 现象:完整数据版英文沿用上游清理器。清理器对没有译文的汉字不会删行,而是包成 `[source text retained: …]`。这样正文里仍然有汉字。公开接口发现汉字后设置 `english_unavailable=han_leak`,不返回 `markdown_en`。页面不显示「中文 / English」切换。 +- 触发条件:用完整数据版生成中英两份。本机虚构盘、乔布斯盘、奥巴马盘都如此。泰勒盘没有公开出生资料,没有生成。 +- 根因:英文清理器的设计是留下缺译文的原文,避免看起来干净、实际少了证据。本仓资料包里这类汉字不是一小张术语表:补译分盘提示、行星主星标题、八分法标题和「合计」之前,三张盘大约有 350–470 处汉字串、240–300 个不同写法,很多是单字或半句。不能靠手写词典清空,也不能把夹汉字的英文交给用户。 +- 修复:未解决。生成路径已经整份撤回英文,不输出中英夹杂的英文报告。清理器里的汉字标记只留在生成过程中。 +- 验证:`tests/test_report_english_edition.py` 要求要么英文没有汉字,要么接口撤回且不带 `markdown_en`。没有在浏览器里走查。 +- 防复发:沿用阅读版泄漏词表检查英文全文;接口和前端都在发现汉字时撤回英文。汉字清零之前不得把这则标成 resolved。 +- 相关记录:BUG-1126(有英文版时切换必须真换正文)、BUG-1271。 +- 复发自:无。 +- 修复版本:无。 + ## BUG-1273 | 生产镜像没有 PyJHora,八个特殊上升点算不出来 - 状态:resolved(已部署 staging `795b8845`:门禁 run 3218、deploy run 3220;`/api/health` gitCommit 一致,`/login` 200,未登录 `/api/account` 401) diff --git a/docs/tasks/PROGRESS-report-full-data-edition-20261007.md b/docs/tasks/PROGRESS-report-full-data-edition-20261007.md new file mode 100644 index 00000000..61876295 --- /dev/null +++ b/docs/tasks/PROGRESS-report-full-data-edition-20261007.md @@ -0,0 +1,115 @@ +# 进度 · 个人报告改为完整数据版(2026-10-08) + +- 分支:`codex/report-full-data-edition-20261007` +- 基线:`c04786f7`(`origin/staging` 在开工时的尖端;sync6 已在 `2df6ee72` 部署,其后的文档提交不改变已部署代码) +- 状态:代码在本分支,**未推 staging**,未上线 +- 产品结论:新报告是中文完整数据版。阅读版不再生成。英文这轮不交付(BUG-1272,未解决)。旧报告仍按保存的正文打开。 + +## 用户能看到什么 + +| 项 | 结果 | +| --- | --- | +| 生成 | 「生成报告」只请求 `edition: full_data`。没有阅读版开关 | +| 正文 | 中文。星盘、力量、大运、年运、瑜伽,加结构化资料附录 | +| 英文 | 清理后仍有汉字。接口不返回英文,页面不显示语言切换 | +| 旧报告 | 快照仍是当时存下的 Markdown,打开时不重算 | +| 很长的报告 | 超过约 20 万字才一次显示一章。现在的中文大约 10 万字,仍按原来的方式分段滚动 | +| 下载 | 仍是原来的「导出」,一份 Markdown。没有第二颗下载按钮 | +| 数据库 | 不改表。正文放得下现有文本列 | + +生成卡片仍写「大约 10–30 秒」和「中英两版」。这两句与实测不符,测试锁着「10–30 秒」这句,本轮没有改文案。 + +## 触点 + +| 位置 | 处理 | +| --- | --- | +| `professional_report_reference.py` 的阅读版常量 | 删除。新增 `full_data` / `pl9_personal_long_report.v4`。请求阅读版会报错 | +| `pl9_reader_export.py` 的分发 | 阅读版不再走生成。完整数据版走新模块。文件里的旧整理函数还在,生成路径不调用 | +| `pl9_full_data_export.py`、术语表 | 新增。只搬渲染和用词,不搬上游计算 | +| 英文用词、事实表、大运适用说明 | 留下。旧报告和事实表还用它们 | +| `personal-report-longform-generate.ts` | 改为请求完整数据版 | +| 快照、原始附录、阅读页、导出 | 留下,旧报告才能打开 | +| `jyotish_api_server.py`、`page.tsx`、默认报告版本常量 | 没有改。默认版本仍是 v3;v4 只出现在完整数据版的接口字段上 | +| 对话 | 没有改。只加了按领域复制资料包的只读入口 | + +## 本机生成(不是 2 核服务器) + +机器是 16 线程桌面,一次只跑一个 Python 进程。不能当成线上 2 核的耗时。都低于接口 180 秒上限。 + +| 盘 | 排盘加资料包 | 中文渲染 | 中文字符 | 英文渲染 | 英文字符 | 英文汉字串 / 不同写法 | +| --- | --- | --- | --- | --- | --- | --- | +| 虚构 | 79 秒 | 2.7 秒 | 104,547 | 0.4 秒 | 138,444 | 353 / 238 | +| 乔布斯 | 57 秒 | 8.7 秒 | 101,689 | 2.0 秒 | 136,102 | 404 / 274 | +| 奥巴马 | 91 秒 | 3.0 秒 | 112,222 | 0.6 秒 | 151,182 | 465 / 303 | + +泰勒没有生成:本仓和上游 `23be1807` 都没有这份公开出生资料。上表的汉字数是在补上分盘提示、主星标题、八分法标题和「合计」的英文之前量的。补译之后汉字会少一些,没有少到能整份交付。 + +只存中文时大约 10 万到 11 万字符。中英都存大约 24 万字符。PostgreSQL 文本列放得下。没有改表。 + +## 缺口(没有补算,也没有编) + +1. 上游那份私人 PDF 对照表没有搬。里面有对具体用户的扣留说明。 +2. 视位置计时和年运交食返照没有搬。缺的计时节就空着。 +3. 开篇「重点」是固定的三句核对清单,不是上游按当前大运和宫主写的那段。 +4. 中文大约 10 万字符,不是任务书里对照的上游约 97 万。渲染器要读的资料包字段这边没有。 +5. 没有周期数据的大运族不会出现。Narayana 的旧算法说明和 Rath 并列表用的是资料包里已经算好的字段,只展示。 +6. 泰勒盘见上。 +7. 英文见 BUG-1272。 + +本仓口径仍在:落陷取消不叫王瑜伽;网站的 +2 / −1.5 分不出现;六维力量先写最低要求;八分法三列数字还在;没有 Kranti 列;年主依据和「不能确定就不拿 Muntha 主星充数」还在。 + +## 让步 + +- 没有全文搜索。 +- 不到 20 万字不启用按章分页。100 万字和 350 万字的页面测试能换章,用时在原有的首屏时限里。 +- 生成文案仍写 10–30 秒和中英两版。见上。 + +## 测试断言 + +没有删除测试。守计算口径的文件都改成对完整数据版断言。下表只列本轮改过的断言。 + +| 测试 | 原断言 | 新值 | 原因 | +| --- | --- | --- | --- | +| 中文章节名含 Year Lord | 正文出现 Year Lord | 出现 Varshesha | 中文清理把 Lord 写成「宫主」 | +| 年运三行 | 全文数 `Muntha sign` | 只数附录前的 `Muntha 星座`,仍是三年、各不相同 | 附录会再写一行当年 Muntha;用词被译成「星座」 | +| 年运冲突行的 Muntha | `Muntha sign` | `Muntha 星座`,星座值不变 | 同上 | +| 禁止「字段」 | 全文不能出现「字段」 | 附录前可以出现「字段」;尊贵状态仍只写一个名字 | 上游表头就是「字段 \| 数值」。这条随阅读版失去对象 | +| 英文无汉字 | 汉字必须为空 | 有汉字就必须是 source text retained;接口在有汉字时撤回英文 | BUG-1272 | +| 英文年度章标题 | Annual Charts and Tajika | Annual Varshaphala and Tajika | 上游标题 | +| 中英数字顺序 | 全文数字序列相同 | 两边正文都有同一个出生年份 | 中文多了核对清单序号,逐号对齐失去对象 | +| 宫位力量表头 | 中英都是英文表头 | 中文译成宫位/星座/宫主;英文保持原文。网站 +2/−1.5 仍禁止 | 中文清理 | +| 赤纬表 | 中英都是 Declination / Speed | 中文是赤纬/速度;英文保持原文。仍无 Kranti | 中文清理 | +| 八分法宫位表 | 中英同一套英文表头;合计行星座格是 `-` | 中文表头译成宫位/星座/上升;英文标题 Full House Scores,合计行 Total;空位可以是 `-` 或「未列」。四格数字不变 | 清理器改用词,不改分数 | +| 六维力量表头 | 中英都以 Planet 开头 | 中文以「行星」开头;英文仍是 Planet。最低要求仍在网站分档之前 | 中文清理 | +| 年主依据、年主不能确定 | 英文必须是无汉字的那句 | 清干净时仍要那句英文;否则必须标成 source text retained,且不得改回 Muntha 主星 | BUG-1272 | +| Narayana 两则 | 对 reference 版断言旧算法句和 Rath 表 | 对 full_data 断言同一内容 | 用户看到的是完整数据版 | +| 正文不得用 useState | `PersonalReportMarkdownView` 里不能出现 useState | 只允许章节序号和打印标记两处;每次 `renderMarkdown` 仍在 useMemo 里 | 超过 20 万字按章翻页。滚动位置仍不放在阅读器里 | +| 切换版本留在同一章 | `drawn={drawnThrough !== null && chapter <= drawnThrough}` | 懒加载用 `index <= drawnThrough`;超长报告的初始章取 `drawnThrough` | 循环变量改名。按章翻页仍从切换前的章号打开 | + +Chara Karaka 顺序和落陷取消条件的计算断言没有改,只是把被测版本从阅读版换成完整数据版。 + +## 本机检查 + +| 项 | 结果 | +| --- | --- | +| `tsc --noEmit` | 退出码 0 | +| `npm run lint` | 退出码 0。0 error,125 warnings。警告都不在本轮改过的文件里。没有去改这些既有警告 | +| 报告与口径定向 pytest | 除八分法合计行外一次通过;改完英文「合计 → Total」后,`test_report_chart_blank_columns.py` 12 项通过 | +| 100 万字 / 350 万字阅读页 | 通过(tsx)。换章,没有全文搜索 | +| 领域只读入口 | 3 项通过。未知领域报错,不改对话 | +| 浏览器 | 没有走查。见 `docs/testing/report-full-data-edition-20261008.md` | +| 全量 `npm test`、`next build --webpack`、Python 快速门 | 见下 | + +## 全量前端与构建 + +| 项 | 结果 | +| --- | --- | +| `npm test` | 4902 项,通过 4756,失败 146,取消 0,跳过 0。用时约 339 秒。这是修好两条阅读器断言之前的数字 | +| 本轮新失败 | 两条:`article markdown trees are memoized and hold no scroll state`,`switching editions keeps the reader on the same chapter (BUG-1108)`。修好后这两个文件 23 项通过。没有重跑全量,所以全量失败名单按修好后计是 144 | +| 和开工基线比 | 基线文件 147 条。中文用例名在那份文件里编码坏了,按英文名对过:基线多 3 条这次没红的环境失败(两条数据库、一条 D3 首行)。sync6 进度里多出的 `corrective migration clears untrusted clocks` 这次通过了。没有删测试 | +| `next build --webpack` | 第一次收集 `/api/consult` 页面数据时,技能包符号链接 EPERM,编译和类型检查已经过。用本机已有的符号链接预加载再跑,退出码 0。`/` 是 Static | +| 首屏 gzip-9 | 预渲染 `index.html` 引用 31 个脚本、3 个样式。基线 `c04786f7` 同法 608,597 B,本分支 608,642 B,+45 B,+0.007%。基线用临时工作树构建,量完已删 | +| Python 快速门 | 编译、JSON、能力审计、碎片审计、清单和 BPHS 都过。pytest 1057 通过、1 跳过、1 失败,用时 646 秒。失败是 `test_consultation_native_layers.py::test_the_new_layer_leaves_every_existing_output_unchanged`。日志里没有完整数据版的字段。sync6 进度已写这条在 Windows 失败、在 Linux 通过。脚本停在这一步,没有继续跑门禁里的 `npm test` 和 `npm run build` | +| Python 全量 | 没有重跑 | +| 隐私标记 | 在这轮快速门里,唯一失败不是它 | +| 浏览器 | 没有走查。见 `docs/testing/report-full-data-edition-20261008.md` | diff --git a/docs/tasks/README.md b/docs/tasks/README.md index a2035089..7c894f48 100644 --- a/docs/tasks/README.md +++ b/docs/tasks/README.md @@ -201,8 +201,8 @@ | 任务书 | 进度 | 主题 | 状态 | 落点 | | --- | --- | --- | --- | --- | -| `TASK-report-full-data-edition-20261007.md` | — | **报告改为完整数据版、删除阅读版**:网站阅读版中文 7.6 万字符 vs 上游 `pl9_ai_density` 97 万;移植原始数据附录 / 清理器 / 时间补充,中英文;我方报告口径(BUG-1199~1229)全部保持;旧报告仍可打开;排在 sync6 之后 | 未合入(分支 `06028759`,验收未通过,见 fix-20261008) | — | -| `TASK-report-full-data-fix-20261008.md` | — | **完整数据版补齐**(同分支 `06028759` 继续):中文只有上游 19%(9.9 万 vs 52.6 万字);产品允许移植上游解释生成器并逐条核口径;年运表接我方字段;英文必须交付才上线;其他 7 种大运另开单 | 验收未通过(`32f19f07`,见 fix2) | — | +| `TASK-report-full-data-edition-20261007.md` | `PROGRESS-report-full-data-edition-20261007.md` | **报告改为完整数据版、删除阅读版**:网站阅读版中文 7.6 万字符 vs 上游 `pl9_ai_density` 97 万;移植原始数据附录 / 清理器 / 时间补充,中英文;我方报告口径(BUG-1199~1229)全部保持;旧报告仍可打开;排在 sync6 之后 | 未合入(验收未通过,见 fix-20261008) | `codex/report-full-data-edition-20261007`(未推 staging) | +| `TASK-report-full-data-fix-20261008.md` | — | **完整数据版补齐**(同分支继续):中文只有上游 19%(9.9 万 vs 52.6 万字);产品允许移植上游解释生成器并逐条核口径;年运表接我方字段;英文必须交付才上线;其他 7 种大运另开单 | 待领取 | — | | `TASK-report-full-data-fix2-20261008.md` | — | **完整数据版第二轮**(同分支 `32f19f07` 续):英文已过;未过——全量 +1 旧断言、寿命断言、半译词与字段名被改坏、「守护第未列出宫」163 处、年运章 6/29 节、大运解释只有上游 42%(产品:补回传统段落改「可能」口吻) | 待领取 | — | | `TASK-report-reader-main-fix-20260925.md` | `PROGRESS-report-reader-main-fix-20260925.md` | **读者版修复**:真实年主比较、分字段绑定与未来返照;保留 Bhava Bala 和标题,隔离报告/聊天规则 | 已验收(Claude 09-25:年运表逐年填齐且不同、Bhava Bala 保留、泄漏 0) | `4d801e53`(已部署,health 核对一致) | | `TASK-report-reader-main-20260924.md` | `PROGRESS-report-reader-main-20260924.md` | **报告正文换上游读者版 reader_main + KP / Muntha 计算缺陷 + 投影截断(BUG-1026/1027/1028)**:正文一直是上游审计版(读者版 09-09 才出、从未同步),本仓又自加「结论等级规则」/英文段/异常原文;投影黑名单不认 PL9 词汇且截断半句、删表留说明;KP args `vars()` 拷贝为空恒失败;Muntha 漏合 `birth_asc_sign_idx` → Year Lord 恒火星。产品拍板接入读者版、旧报告不回填。 | 部分通过(Claude 09-25:读者版切换、泄漏 0 命中、KP 已修;年运 Year Lord/Muntha 全为「-」P1、投影误删 Bhava Bala P2 → 修复单) | `3c2f7bd5`(已部署) | diff --git a/docs/testing/report-full-data-edition-20261008.md b/docs/testing/report-full-data-edition-20261008.md new file mode 100644 index 00000000..656a2b2e --- /dev/null +++ b/docs/testing/report-full-data-edition-20261008.md @@ -0,0 +1,19 @@ +# 真机清单 · 个人报告完整数据版(2026-10-08) + +对应 `docs/tasks/TASK-report-full-data-edition-20261007.md`。电脑和手机各走一遍。第 1 步会用掉一次生成次数。本机没有登录态,也没有用浏览器点过,下表结果留空。 + +当前代码的行为,供对照: + +- 新报告是中文完整数据版。英文清理后仍有汉字,接口整份撤回,页面上没有「中文 / English」(BUG-1272)。 +- 生成卡片仍写「大约 10–30 秒」和「中英两版」。本机单进程大约要 1 分钟多,不是 2 核服务器上的测量。英文这轮实际不会出现。 +- 现在生成的中文大约 10 万字,不到 20 万字的分页线,阅读方式与改版前的长文一样:往下滚,目录跳章节。超过约 20 万字才出现「上一章 / 下一章」。 +- 下载仍只有阅读页上的「导出」,得到 Markdown。 + +| # | 操作 | 应该看到 | 结果 | +|---|---|---|---| +| 1 | 「我的报告」→ 生成报告,等它完成 → 打开 | 正文是中文,能往下滚。没有「中文 / English」。没有第二颗下载按钮 | | +| 2 | 点目录里的某一章 | 正文跳到那一章。页面不卡住 | | +| 3 | 点「导出」 | 只有这一处导出。说明里有一句「这份报告的英文版没有生成成功,只能导出中文版。」下载的是 Markdown | | +| 4 | 打开一份这次更新之前生成的旧报告 | 能打开,正文仍是当时保存的那一版,没有自动重算 | | +| 5 | 在旧报告地址后手动加 `?lang=en` | 仍显示原来的正文,不报错 | | +| 6 | (仅当将来英文重新交付)点「English」再点「中文」 | 正文和目录跟着换,同一章还在。这轮新报告不应出现这个切换 | | diff --git a/frontend/DESIGN.md b/frontend/DESIGN.md index c7c291d7..417d7e84 100644 --- a/frontend/DESIGN.md +++ b/frontend/DESIGN.md @@ -634,7 +634,7 @@ Action results and errors never take page layout. They are toasts from `src/lib/ ### Personal report centre -- **Structure:** inside the app shell, not a page of its own. The name 「我的报告」 sits alone in the 46px header. The body opens with the **generate card** (`.report-center-create`, 2026-09-29, product sketch): a centred `--color-canvas` sheet, `--space-6` below the header (BUG-1103: it used to sit flush under it), with a file icon, the title 「完整本命报告」 (what you get), one line 「星盘、力量、大运、年运、瑜伽共 6 章,中英两版;生成后可离开,完成时下方自动出现。」 and the filled 「生成报告」 button (what it does) — title and button no longer say the same thing — the page's only generate entry; the header button was deleted rather than kept as a second one. Below it, 「过往的报告」 heads the **row list** of reports; with none yet, a single quiet line 「还没有个人报告。」 replaces the old second big empty card. The supporting paragraph (「有填报到分钟的出生时间即可生成……」) and the 「共 N 份 · 已完成 N 份」 overview line above it were removed on 2026-09-28 (TASK-self-edit-avatar-menu-20260928 S3, product: 「这里的提示去掉」); the minute requirement still lives in the 生成 button's hover title. +- **Structure:** inside the app shell, not a page of its own. The name 「我的报告」 sits alone in the 46px header. The body opens with the **generate card** (`.report-center-create`, 2026-09-29, product sketch): a centred `--color-canvas` sheet, `--space-6` below the header (BUG-1103: it used to sit flush under it), with a file icon, the title 「完整本命报告」 (what you get), one line 「星盘、力量、大运、年运和瑜伽的完整数据,中英两版;生成后可离开,完成时下方自动出现。」 and the filled 「生成报告」 button (what it does) — title and button no longer say the same thing — the page's only generate entry; the header button was deleted rather than kept as a second one. Below it, 「过往的报告」 heads the **row list** of reports; with none yet, a single quiet line 「还没有个人报告。」 replaces the old second big empty card. The supporting paragraph (「有填报到分钟的出生时间即可生成……」) and the 「共 N 份 · 已完成 N 份」 overview line above it were removed on 2026-09-28 (TASK-self-edit-avatar-menu-20260928 S3, product: 「这里的提示去掉」); the minute requirement still lives in the 生成 button's hover title. It used to be a standalone full-screen route with a `report-center-shell` root, a `report-center-topbar` holding one 「返回对话」 link, and a `report-center-hero` with a page-sized h1 — the sidebar vanished the moment you opened it, and the only way back was that link. - **Row list, not a card grid:** `.report-center-list` is one column of `.report-center-row`, hairline-separated by a 1px grid gap over a `--color-border` ground so each boundary is a single rule rather than two touching borders. A card grid costs one scan per card; past five or six reports the reader is looking for state, and a single column puts every status chip on the same x. D12. - **Paging (2026-09-30, BUG-1124):** the list shows the newest 10; under it an outline 「加载更多」 button (centred, at least 160px; full width below 767px) brings the next 10. While that page loads the button reads 「加载中」 and is disabled — no spinner. It disappears when nothing older exists. A failed page is a toast 「更早的报告没读出来,再试一次。」, the list stays. The 3s poll while a report is generating and the 刷新 button re-read the first page only and merge it over the pages already opened, so a refresh never collapses the list back to 10. Entering 我的报告 shows the cached first page at once and revalidates behind it (BUG-1123); a cold entry keeps the one shared waiting state. @@ -664,6 +664,7 @@ Action results and errors never take page layout. They are toasts from `src/lib/ - **Switching editions keeps your place (BUG-1108):** the chapter at the top of the reader is remembered; the new edition renders every chapter up to it at once and scrolls back to it. - **判断不了的分盘提示 (2026-10-01, BUG-1139):** for a segment-v1 adoption whose card left D9/D10 undetermined, the longform body carries one plain paragraph directly under each such chart heading and its 数值位置表 heading — no new component, no blockquote, no colour. Under a chart heading it becomes the chart card's caption (`report-chart-grid-rehype.ts::parseCard` accepts exactly one paragraph between heading and fence), so the chart grid never splits. Added before the snapshot hash; legacy reports are untouched. - **到底部 / 回到顶部:** one floating pill (`.report-edge-jump`, fixed bottom-right, same look as the chat's 跳到最新). It reads 「到底部」 until the reader is within 480px of the end, then 「回到顶部」 — never both. Sections below the fold mount lazily, so the click first mounts every section synchronously (`flushSync`) and then scrolls to `scrollHeight`, landing on the real end in one click. It is not a scroll follower and owns no anchor; it is hidden on paper. +- **Long full-data reports (2026-10-08):** when the Markdown is longer than 200,000 characters and has more than one chapter, the reader shows one chapter at a time. The TOC and 「上一章 / 下一章」 (44px, screen only) change the chapter; the article carries `data-chapter-index` so a 中文 / English switch returns to the same chapter. 「到底部」 opens the last chapter. Shorter reports keep the lazy-section reader. There is no full-text search. Print still mounts every chapter. Markdown download stays on the existing 「导出」 sheet. - **TOC rail:** a persistent right-hand column built from the outline's own `##`/`###` ids — no second slugger, and nothing that touches the chart-grid rehype pass. It is sticky under the report's own action bar, highlights the section in view via `IntersectionObserver`, and marks it with `aria-current="location"` plus a 2px `--color-action` bar. Placement is explicit (`grid-column: 2`) rather than DOM-ordered, so the narrow-screen drawer can stay first in the source. Below 860px the rail is gone and what remains is the collapsed 「目录」 drawer. - **Paper stays paper:** the rail reads in the app palette because it is chrome. `--report-paper` / `--report-rule` / `--report-accent` are the document's own, deliberately out of step with the app accent, and nothing in this surface puts them on chrome or takes chrome colour onto the page. D11, continuing D3. - **Surface:** page floor `--color-canvas-soft`; the Markdown article is a `--color-canvas` sheet with a hairline and `--radius-lg`. Print flattens the sheet, hides chrome and the TOC, and pins the light palette. diff --git a/frontend/docs/VOICE.md b/frontend/docs/VOICE.md index 906f6b45..38be8527 100644 --- a/frontend/docs/VOICE.md +++ b/frontend/docs/VOICE.md @@ -184,7 +184,7 @@ Jyotisha 的可见文案是产品的一部分。正确性红线(真实性、 - 盘面不显示引擎品牌,也不替换成「自研引擎」「本站引擎」。岁差、交点等计算口径与准确性边界照旧。 - 图标说明只写名称和已有事实:上升、行星、逆行、星座名称/编号、宫位、ASC/MC 等。没有可靠度数就不写度数,不加吉凶、性格或个人运势解释。 - 鼠标悬停、键盘聚焦、手机轻点展示同一份说明,以标记旁的小气泡出现,不遮住被说明的标记。关闭说明不改变西洋相位选择。2026-09-29 晚起不再有默认提示句(原「停在或轻点盘内标记可查看说明。」已删,BUG-1101)。 -- 「我的报告」生成卡片:标题写得到什么「完整本命报告」,说明一句「星盘、力量、大运、年运、瑜伽共 6 章,中英两版;生成后可离开,完成时下方自动出现。」,按钮写动作「生成报告」。标题与按钮不说同一件事(BUG-1103)。 +- 「我的报告」生成卡片:标题写得到什么「完整本命报告」,说明一句「星盘、力量、大运、年运和瑜伽的完整数据,中英两版;生成后可离开,完成时下方自动出现。」,按钮写动作「生成报告」。标题与按钮不说同一件事(BUG-1103)。完整数据版超过约 20 万字符时,阅读页一次只画一章,目录和「上一章 / 下一章」换章;不另做全文搜索。导出仍是原来的「导出」按钮,下载 Markdown。 - 分盘庙旺仅指入旺、自宫、落陷三档;未命中写「—」,上升写「不适用」。同位仅 D9 展示,其他分盘不留空列。缺能力声明不装作已支持。 ## 星盘空状态、0 年大运、星历切日期 diff --git a/frontend/src/app/globals.css b/frontend/src/app/globals.css index eaba6f9b..aedc3b73 100644 --- a/frontend/src/app/globals.css +++ b/frontend/src/app/globals.css @@ -4298,6 +4298,32 @@ input:not([type="radio"]):not([type="checkbox"]):not([class^="ant-"]):not([class @media (max-width: 767px) { .report-edge-jump { right: var(--space-4); bottom: calc(var(--space-4) + env(safe-area-inset-bottom)); } } +.personal-report-chapter-pager { + display: flex; + align-items: center; + justify-content: space-between; + gap: var(--space-3); + margin-top: var(--space-5); +} +.personal-report-chapter-pager button { + min-height: 44px; + padding: 0 var(--space-4); + border: 1px solid var(--color-border); + border-radius: 999px; + background: var(--color-canvas); + color: var(--color-ink); + font: inherit; + font-size: var(--type-body-sm); + cursor: pointer; +} +.personal-report-chapter-pager button:disabled { + opacity: 0.45; + cursor: default; +} +.personal-report-chapter-pager span { + color: var(--color-ink-secondary); + font-size: var(--type-body-sm); +} .personal-report-toc-drawer { display: none; } .personal-report-toc-desktop p { margin: 0 0 var(--space-2); diff --git a/frontend/src/components/personal-report/personal-report-center.tsx b/frontend/src/components/personal-report/personal-report-center.tsx index f345e5f7..b21b3d79 100644 --- a/frontend/src/components/personal-report/personal-report-center.tsx +++ b/frontend/src/components/personal-report/personal-report-center.tsx @@ -214,7 +214,7 @@ export function PersonalReportCenter() { {/* The title says what you get, the button says what it does (2026-09-29): 「生成个人报告」 over a 「生成报告」 button said the same thing twice. */}

完整本命报告

-

星盘、力量、大运、年运、瑜伽共 6 章,中英两版;生成后可离开,完成时下方自动出现。

+

星盘、力量、大运、年运和瑜伽的完整数据,中英两版;生成后可离开,完成时下方自动出现。

void load()} />
diff --git a/frontend/src/components/personal-report/personal-report-markdown-view.tsx b/frontend/src/components/personal-report/personal-report-markdown-view.tsx index 9e480eca..40131040 100644 --- a/frontend/src/components/personal-report/personal-report-markdown-view.tsx +++ b/frontend/src/components/personal-report/personal-report-markdown-view.tsx @@ -107,9 +107,9 @@ function renderMarkdown(markdown: string, headings: readonly LongformHeading[], } /** - * 「到底部」 has to mount every lazy section before it scrolls. The flag lives - * outside React state so the article view itself stays state-free (its tree is - * memoized and must not re-render as a whole); only the sections subscribe. + * 「到底部」 has to mount every lazy section before it scrolls. Scroll position + * stays outside this view. The only state here is the chapter index and the + * print flag; markdown trees stay in useMemo. */ type RevealSignal = { revealed: boolean; listeners: Set<() => void> }; @@ -163,9 +163,23 @@ const LazyMarkdownSection = memo(function LazyMarkdownSection({ ); }); -function ReportToc({ headings, language }: { headings: readonly LongformHeading[]; language: ReportLanguage }) { +/** Above this size the reader shows one chapter at a time. Shorter reports keep lazy sections. */ +export const FULL_DATA_CHAPTER_PAGE_CHARS = 200_000; + +function ReportToc({ + headings, + language, + onSelect, + activeHeadingId, +}: { + headings: readonly LongformHeading[]; + language: ReportLanguage; + onSelect?: (id: string) => void; + activeHeadingId?: string; +}) { const label = language === "en" ? "Contents" : "目录"; - const [activeId, setActiveId] = useState(headings[0]?.id ?? ""); + const [observedId, setObservedId] = useState(headings[0]?.id ?? ""); + const activeId = activeHeadingId || observedId; useEffect(() => { if (headings.length === 0 || typeof IntersectionObserver === "undefined") return; @@ -174,7 +188,7 @@ function ReportToc({ headings, language }: { headings: readonly LongformHeading[ .filter((entry) => entry.isIntersecting) .sort((left, right) => left.boundingClientRect.top - right.boundingClientRect.top); const id = visible[0]?.target.id; - if (id) setActiveId(id); + if (id) setObservedId(id); }, { rootMargin: "-20% 0px -70% 0px", threshold: [0, 1] }); for (const item of headings) { const node = document.getElementById(item.id); @@ -187,11 +201,11 @@ function ReportToc({ headings, language }: { headings: readonly LongformHeading[ ); @@ -204,17 +218,92 @@ export function PersonalReportMarkdownView({ markdown, language = "zh", drawnThr drawnThrough?: number | null; }) { const outline = useMemo(() => buildLongformOutline(markdown), [markdown]); + const chapters = useMemo(() => outline.sections.filter((section) => !section.eager), [outline]); + const paged = markdown.length > FULL_DATA_CHAPTER_PAGE_CHARS && chapters.length > 1; + const [chapter, setChapter] = useState(() => { + if (drawnThrough === null) return 0; + return Math.min(Math.max(drawnThrough, 0), Math.max(chapters.length - 1, 0)); + }); + const [printing, setPrinting] = useState(false); const reveal = useMemo(() => ({ revealed: false, listeners: new Set() }), []); + useEffect(() => { + if (!paged) return; + const before = () => setPrinting(true); + const after = () => setPrinting(false); + window.addEventListener("beforeprint", before); + window.addEventListener("afterprint", after); + return () => { + window.removeEventListener("beforeprint", before); + window.removeEventListener("afterprint", after); + }; + }, [paged]); // Synchronous on purpose: the caller scrolls to scrollHeight right after, and // that height is only final once every lazy section has committed. const revealAll = useCallback(() => flushSync(() => { + if (paged) { + setChapter(Math.max(chapters.length - 1, 0)); + return; + } reveal.revealed = true; for (const listener of reveal.listeners) listener(); - }), [reveal]); + }), [chapters.length, paged, reveal]); const lead = useMemo( () => (outline.leadMarkdown ? renderMarkdown(outline.leadMarkdown, outline.headings, language) : null), [outline, language], ); + const safeChapter = Math.min(chapter, Math.max(chapters.length - 1, 0)); + const currentChapter = chapters[safeChapter]; + const pagedChapter = useMemo(() => ( + paged && currentChapter + ? renderMarkdown(currentChapter.markdown, currentChapter.headings, language) + : null + ), [paged, currentChapter, language]); + const printedChapters = useMemo(() => ( + paged && printing + ? chapters.map((section) => renderMarkdown(section.markdown, section.headings, language)) + : null + ), [paged, printing, chapters, language]); + const openChapter = useCallback((id: string) => { + const index = chapters.findIndex((section) => section.id === id || section.headings.some((heading) => heading.id === id)); + if (index >= 0) setChapter(index); + }, [chapters]); + + if (paged) { + return ( +
+ +
+ {safeChapter === 0 && lead ? ( +
+ {lead} +
+ ) : null} + {printing + ? chapters.map((section, index) => ( +
+ {printedChapters?.[index]} +
+ )) + : currentChapter ? ( +
+ {pagedChapter} +
+ ) : null} + +
+ +
+ ); + } return (
@@ -226,10 +315,10 @@ export function PersonalReportMarkdownView({ markdown, language = "zh", drawnThr
) : null} {/* Each section is one chapter heading, in the same order as the article's h2s. */} - {outline.sections.map((section, chapter) => (section.eager ? null : ( + {outline.sections.map((section, index) => (section.eager ? null : ( void; }) { return (
    @@ -256,6 +347,10 @@ function TocList({ aria-current={item.id === activeId ? "location" : undefined} className={item.id === activeId ? "is-current" : undefined} href={`#${item.id}`} + onClick={onSelect ? (event) => { + event.preventDefault(); + onSelect(item.id); + } : undefined} > {item.title} diff --git a/frontend/src/components/personal-report/personal-report-page.tsx b/frontend/src/components/personal-report/personal-report-page.tsx index f6ae3c06..e05cdcd0 100644 --- a/frontend/src/components/personal-report/personal-report-page.tsx +++ b/frontend/src/components/personal-report/personal-report-page.tsx @@ -218,6 +218,11 @@ function readEnglishEdition(json: Record): Pick< /** Index of the chapter heading at or above the top of the reader, or null at the very top. */ function currentChapterIndex(): number | null { + const paged = document.querySelector(".personal-report-md-article[data-chapter-index]"); + if (paged) { + const index = Number(paged.dataset.chapterIndex); + return Number.isFinite(index) ? index : null; + } const reader = document.querySelector(".personal-report-reader"); if (!reader || reader.scrollTop < 40) return null; const top = reader.getBoundingClientRect().top + 80; @@ -282,6 +287,11 @@ export function PersonalReportPage({ reportId, initialLanguage = "zh" }: { repor setDrawnThrough(chapter); if (chapter !== null) { requestAnimationFrame(() => requestAnimationFrame(() => { + const article = document.querySelector(".personal-report-md-article[data-chapter-index]"); + if (article) { + article.scrollIntoView({ block: "start" }); + return; + } document.querySelectorAll(".personal-report-md-article h2")[chapter]?.scrollIntoView({ block: "start" }); })); } diff --git a/frontend/src/lib/personal-report-longform-generate.ts b/frontend/src/lib/personal-report-longform-generate.ts index daa86ae5..5fa4198d 100644 --- a/frontend/src/lib/personal-report-longform-generate.ts +++ b/frontend/src/lib/personal-report-longform-generate.ts @@ -151,7 +151,7 @@ async function fetchLongformMarkdown(input: Readonly<{ const upstream = await input.fetchImpl(`${input.apiBase.replace(/\/$/, "")}/api/professional_report_reference`, { method: "POST", headers: { "Content-Type": "application/json", Accept: "application/json" }, - body: JSON.stringify({ ...input.payload, include_fact_tables: true, edition: "reader_main", languages: ["zh", "en"] }), + body: JSON.stringify({ ...input.payload, include_fact_tables: true, edition: "full_data", languages: ["zh", "en"] }), cache: "no-store", signal, }); diff --git a/frontend/tests/personal-report-markdown-view.test.ts b/frontend/tests/personal-report-markdown-view.test.ts index cdd6d2f8..75a6f865 100644 --- a/frontend/tests/personal-report-markdown-view.test.ts +++ b/frontend/tests/personal-report-markdown-view.test.ts @@ -77,6 +77,29 @@ test("markdown view skips HTML, blocks remote resources, wraps tables, and lazy- assert.match(markup, /aria-label="报告目录"/); }); +test("long markdown pages one chapter and keeps the first paint small", () => { + const chapters = Array.from({ length: 80 }, (_, index) => { + const body = index === 0 ? `第一章正文${"字".repeat(80)}` : `秘密正文-${index}`; + return `## 第${index}章\n\n${body}\n`; + }); + for (const size of [1_000_000, 3_500_000]) { + const markdown = `# 标题\n\n${chapters.join("\n")}${"补".repeat(size)}`; + const before = process.memoryUsage().heapUsed; + const started = Date.now(); + const markup = renderToStaticMarkup(React.createElement(PersonalReportMarkdownView, { markdown })); + const elapsed = Date.now() - started; + const delta = process.memoryUsage().heapUsed - before; + assert.ok(elapsed < 1500, `size ${size} first paint ${elapsed}ms heap ${delta}`); + assert.match(markup, /第一章正文/); + assert.match(markup, /上一章/); + assert.match(markup, /下一章/); + assert.match(markup, /1 \/ 80/); + assert.doesNotMatch(markup, /秘密正文-79/); + assert.doesNotMatch(markup, /type="search"|placeholder="搜索"/); + assert.ok(delta < 120 * 1024 * 1024, `size ${size} heap delta ${delta}`); + } +}); + test("export filename includes the report date", () => { assert.equal(personalReportMarkdownFilename("2026-09-06T12:00:00.000Z"), "个人报告-2026-09-06"); }); diff --git a/frontend/tests/personal-report-markdown-view.test.tsx b/frontend/tests/personal-report-markdown-view.test.tsx index ff89ff4c..19a8598c 100644 --- a/frontend/tests/personal-report-markdown-view.test.tsx +++ b/frontend/tests/personal-report-markdown-view.test.tsx @@ -175,7 +175,11 @@ test("article markdown trees are memoized and hold no scroll state", () => { const viewEnd = src.indexOf("\nfunction TocList", viewStart); assert.ok(viewStart >= 0 && viewEnd > viewStart); const view = src.slice(viewStart, viewEnd); - assert.doesNotMatch(view, /\buseState\b/); + // 原值: 这个函数里不得出现 useState。新值: 只允许章节序号和打印标记两处。原因: 超过 20 万字要按章翻页,滚动位置仍不放在这里。 + assert.deepEqual(view.match(/\buseState\b/g), ["useState", "useState"]); + assert.match(view, /const \[chapter, setChapter\] = useState/); + assert.match(view, /const \[printing, setPrinting\] = useState\(false\)/); + assert.doesNotMatch(view, /useState\([\s\S]{0,80}scroll/i); const lines = src.split("\n"); lines.forEach((line, index) => { if (!line.includes("renderMarkdown(")) return; diff --git a/frontend/tests/professional-report-reference-route.test.ts b/frontend/tests/professional-report-reference-route.test.ts index 1ed6e902..bcc7916f 100644 --- a/frontend/tests/professional-report-reference-route.test.ts +++ b/frontend/tests/professional-report-reference-route.test.ts @@ -101,12 +101,13 @@ function executeReadyReportExport(cached: boolean) { }; } -test("new longform generation requests the reader_main edition", () => { +test("new longform generation requests the full_data edition", () => { const generateSource = readFileSync( new URL("../src/lib/personal-report-longform-generate.ts", import.meta.url), "utf8", ); - assert.match(generateSource, /edition:\s*"reader_main"/); + assert.match(generateSource, /edition:\s*"full_data"/); + assert.doesNotMatch(generateSource, /edition:\s*"reader_main"/); assert.match(generateSource, /include_fact_tables:\s*true/); }); diff --git a/frontend/tests/surface-feedback-20260929.test.tsx b/frontend/tests/surface-feedback-20260929.test.tsx index 098abf28..d7b537a1 100644 --- a/frontend/tests/surface-feedback-20260929.test.tsx +++ b/frontend/tests/surface-feedback-20260929.test.tsx @@ -127,7 +127,9 @@ test("switching editions keeps the reader on the same chapter (BUG-1108)", () => assert.match(page, /setDrawnThrough\(chapter\);/); assert.match(page, /drawnThrough=\{drawnThrough\}/); const view = read("../src/components/personal-report/personal-report-markdown-view.tsx"); - assert.match(view, /drawn=\{drawnThrough !== null && chapter <= drawnThrough\}/); + // 原值: drawn={drawnThrough !== null && chapter <= drawnThrough}。新值: 循环变量改为 index,超长报告的初始章取 drawnThrough。原因: 懒加载路径改名;按章翻页仍从切换前的章号打开。 + assert.match(view, /drawn=\{drawnThrough !== null && index <= drawnThrough\}/); + assert.match(view, /Math\.min\(Math\.max\(drawnThrough, 0\), Math\.max\(chapters\.length - 1, 0\)\)/); assert.match(view, /useState\(section\.eager \|\| drawn\)/); }); diff --git a/scripts/pl9_en_glossary.py b/scripts/pl9_en_glossary.py new file mode 100644 index 00000000..c544a704 --- /dev/null +++ b/scripts/pl9_en_glossary.py @@ -0,0 +1,530 @@ +"""English glossary for the PL9 bilingual AI-density export. + +The yoga/dosha engines emit `effects` as Chinese short phrases and `combination` +as mixed Chinese/English strings. The English renderer must translate these +instead of deleting CJK and emitting a placeholder. This module holds the +Chinese -> English vocabulary used for that translation. + +词汇来源:references/yoga_rules.json 的 effects 字段词频统计(高频词优先), +覆盖 yoga/dosha 解释层最常出现的短句;低频词若未命中则按省略处理,不输出占位符。 +""" + +from __future__ import annotations + +# 瑜伽/宫位 effects 中文短句 -> 英文。按词频从高到低整理。 +EFFECTS_EN: dict[str, str] = { + "领域受限": "domain constrained", + "需要额外努力": "requires additional effort", + "充足": "adequate", + "转化困境为机遇": "turns hardship into opportunity", + "不讨喜": "not widely liked", + "事业通过智慧实现": "career achieved through wisdom", + "享受丰富": "enjoys abundance", + "人缘好": "well liked", + "命运力量在角宫": "destiny power in angular houses", + "命运助力明显": "strong support from destiny", + "善于辞令": "eloquent speech", + "基础发展稳定": "stable foundational development", + "多元才华": "versatile talents", + "子女受其庇护": "children receive protection", + "待补充": "to be supplemented", + "心地善良": "kind-hearted", + "心智有背景支持": "mind supported by background conditions", + "心智稳定性增强": "increased mental stability", + "性情愉快": "pleasant temperament", + "性格坚定": "steadfast character", + "持续收入": "steady income", + "政治才华": "political talent", + "月亮前后均有支撑": "Moon supported on both sides", + "物质充裕": "material abundance", + "独处中成长": "growth through solitude", + "生活舒适": "comfortable life", + "盈利丰厚": "substantial profits", + "真诚待人": "sincere dealings", + "祖先福报影响子女": "ancestral blessings affect children", + "经济稳定": "financial stability", + "自我、财富、努力、根基彼此支撑": "self, wealth, effort, and foundations support one another", + "自我表达": "self-expression", + "言辞刻薄": "harsh speech", + "贵人带来社会地位": "benefactors bring social status", + "贵人或隐性资源": "benefactors or hidden resources", + "资源与适应力较强": "strong resources and adaptability", + # 才能 / 事业 / 地位(高频) + "卓越才能": "exceptional talent", + "领域领军": "leading in one's field", + "人格魅力": "personal magnetism", + "受人尊敬": "widely respected", + "命运眷顾": "favored by destiny", + "智慧": "wisdom", + "事业成功": "career success", + "长寿": "longevity", + "财富": "wealth", + "艺术才华": "artistic talent", + "社会地位": "social standing", + "品德高尚": "noble character", + "逆境崛起": "rises from adversity", + "因祸得福": "turns misfortune into fortune", + "权力": "power", + "权力地位": "power and status", + "社会地位高": "high social status", + "智慧卓越": "outstanding wisdom", + "财富积累": "wealth accumulation", + "物质成功": "material success", + "财富充裕": "abundant wealth", + "自力更生": "self-reliant", + "学识渊博": "profound learning", + "婚姻美满": "happy marriage", + "审美卓越": "refined aesthetic sense", + "健康需注意": "watch health", + "稳定": "stability", + "领导力": "leadership", + "贵人运强": "strong benefactor luck", + "口才出众": "eloquent", + "受人爱戴": "beloved", + "精力充沛": "energetic", + "困境中成长": "grows through hardship", + "受人敬仰": "admired", + "精神修养": "spiritual cultivation", + "领导能力": "leadership ability", + "精神力量": "spiritual strength", + "财务困难": "financial difficulty", + "行动力": "initiative", + "权威地位": "authority", + "精神成就": "spiritual attainment", + "物质享受": "material comfort", + "婚姻幸福": "marital happiness", + "学术成就": "academic achievement", + "创造力": "creativity", + "高位": "high position", + "身体健康": "good health", + "社会影响力": "social influence", + "精神导师": "spiritual mentor", + "克服困难": "overcomes obstacles", + "转化能力": "transformative ability", + "逆境中崛起": "rises in adversity", + "投资收益": "investment returns", + "财运亨通": "thriving fortune", + "生活幸福": "happy life", + "生活富足": "prosperous life", + "名声良好": "good reputation", + "精神挑战": "spiritual challenge", + "行动力强": "strong initiative", + "商业头脑": "business acumen", + "意志坚定": "strong will", + "领导才能": "leadership talent", + "领导气质": "leadership quality", + "事业辉煌": "brilliant career", + "智慧与行动力兼备": "wisdom with drive", + "正义感强": "strong sense of justice", + "智力超群": "exceptional intelligence", + "智慧通达": "penetrating wisdom", + "多才多艺": "versatile", + "生活稳定": "stable life", + "有依靠": "well-supported", + "精神成长": "spiritual growth", + "家庭幸福": "happy family", + "影响力": "influence", + "自我实现": "self-realization", + "子女运好": "good fortune with children", + "王者气质": "regal bearing", + "艺术气质": "artistic temperament", + "家庭问题": "family issues", + "精神压力": "mental stress", + "精神领导力": "spiritual leadership", + "情感丰富": "emotionally rich", + "全面成功": "all-round success", + "子女有成就": "accomplished children", + "子女智慧": "wise children", + "婚姻和谐": "harmonious marriage", + "配偶有魅力": "attractive spouse", + "婚姻延迟": "delayed marriage", + "名声显赫": "great renown", + "名声": "reputation", + "社交网络强大": "strong social network", + "政治权力": "political power", + "战胜对手": "overcomes rivals", + "祖先庇佑": "ancestral blessings", + "宗教权威": "religious authority", + "收入不稳定": "unstable income", + "家庭和谐": "family harmony", + "子孙兴旺": "flourishing descendants", + "后代繁荣": "prosperous descendants", + "后代有成就": "accomplished descendants", + "家族延续": "family continuity", + "学识": "learning", + "内在平静": "inner peace", + "慈悲": "compassion", + "保护": "protection", + "保守": "conservative", + "旅行": "travel", + "变动": "change", + "海外": "overseas", + "不稳定": "instability", + "传统": "tradition", + "孤独": "solitude", + "人际关系差": "poor relationships", + "兄弟和睦": "harmony with siblings", + "勇气": "courage", + "竞争力": "competitiveness", + # 事业 / 财富 补充 + "事业有成": "career achievement", + "事业顺利": "smooth career", + "事业巅峰": "career peak", + "事业成就": "career accomplishment", + "职业使命感强": "strong sense of vocation", + "政府高层职位": "senior government post", + "人脉资源强": "strong network resources", + "收入与社会地位同步提升": "rising income and status", + "慢而稳的事业成功": "slow but steady career success", + "专业化": "specialization", + "军事/体育/工程领域成功": "success in military, sports or engineering", + "竞争意识强": "strong competitive drive", + "敌人变朋友": "enemies become friends", + "竞争中获利": "profits through competition", + "愿望实现": "wishes fulfilled", + "愿望通过智慧实现": "wishes fulfilled through wisdom", + "愿望通过地位实现": "wishes fulfilled through status", + "才华变事业": "talent turned into career", + "创造力变现": "creativity monetized", + "商业成功": "business success", + "物质丰裕": "material abundance", + "收入稳定": "stable income", + "家庭资产": "family assets", + "财富自主": "financial independence", + "自力更生致富": "self-made wealth", + "独立创业": "independent entrepreneurship", + "财富增长": "wealth growth", + "财富丰厚": "substantial wealth", + "稳定财富": "stable wealth", + "巨量财富": "great wealth", + "持续财富积累": "sustained wealth accumulation", + "投资回报": "investment returns", + "投资成功": "successful investment", + "通过才华/运气致富": "wealth through talent or luck", + "祖先财富": "ancestral wealth", + "储蓄丰厚": "ample savings", + "理财能力": "financial acumen", + "谨慎理财": "prudent money management", + # 婚姻 / 家庭 + "婚姻稳定": "stable marriage", + "优质婚姻": "quality marriage", + "婚姻显赫": "prestigious marriage", + "婚姻挑战": "marital challenges", + "配偶杰出": "outstanding spouse", + "配偶贤良": "virtuous spouse", + "配偶支持": "supportive spouse", + "配偶互相支持": "mutually supportive spouses", + "配偶有成就": "accomplished spouse", + "配偶有社会地位": "spouse with social standing", + "配偶带来财富": "spouse brings wealth", + "配偶带来好运": "spouse brings good fortune", + "配偶关系紧张": "tense spousal relationship", + "家庭和睦": "family harmony", + "后代昌盛": "prosperous descendants", + "子女有出息": "promising children", + "子女缘厚": "strong bond with children", + "多子多福": "many children, many blessings", + "子女贤良": "virtuous children", + "子女众多": "many children", + "独生子女": "only child", + "子女少但优秀": "few but accomplished children", + "子女带来财富": "children bring wealth", + "子女带来名声": "children bring renown", + # 健康 / 身心 + "健康良好": "good health", + "健康问题": "health concerns", + "慢性疾病": "chronic illness", + "慢性病倾向": "tendency to chronic illness", + "少病少灾": "few illnesses", + "活力充沛": "full of vitality", + "生命力强": "strong vitality", + "体质较弱": "weak constitution", + "疾病": "illness", + "延寿": "extended longevity", + "长寿健康": "long and healthy life", + "智慧长寿": "wisdom and longevity", + "寿命稳定": "stable lifespan", + "疾病抵抗力": "disease resistance", + "免疫系统": "immune system", + # 心智 / 灵性 + "智慧超群": "outstanding wisdom", + "辩才无碍": "unimpeded eloquence", + "沟通能力": "communication ability", + "学业有成": "academic success", + "知识渊博": "vast knowledge", + "教育成功": "educational success", + "意志力": "willpower", + "苦行": "austerity", + "精神修行": "spiritual practice", + "内在力量": "inner strength", + "家庭导向": "family-oriented", + "安全感": "sense of security", + "勤劳": "diligent", + "节俭": "frugal", + "积累": "accumulation", + "务实": "pragmatic", + "家庭": "family", + "自由": "freedom", + "社交": "social", + "起伏": "ups and downs", + "财务波动": "financial fluctuation", + "严厉": "strict", + "批评": "criticism", + "精神痛苦": "mental anguish", + "宗教": "religion", + "祭祀": "ritual", + "精神追求": "spiritual pursuit", + "灵性成长": "spiritual growth", + "灵性觉醒": "spiritual awakening", + "灵性智慧": "spiritual wisdom", + "终极解脱": "final liberation", + "解脱": "liberation", + "解脱智慧": "wisdom of liberation", + "超越生死": "beyond birth and death", + "灵性洞察": "spiritual insight", + "精神探索": "spiritual exploration", + "精神圆满": "spiritual fulfillment", + "宇宙和谐": "cosmic harmony", + "终极智慧": "ultimate wisdom", + "天才级智力": "genius-level intellect", + "超强记忆力": "exceptional memory", + "高等教育": "higher education", + "哲学洞察": "philosophical insight", + "艺术天赋": "artistic gift", + "艺术创造力": "artistic creativity", + "音乐天赋": "musical gift", + "浪漫魅力": "romantic charm", + "美貌": "beauty", + "非传统思想": "unconventional thinking", + # 负向 / 挑战(少量关键词,避免占位) + "欺诈": "deceit", + "不诚实": "dishonesty", + "阴谋": "intrigue", + "信任危机": "trust issues", + "债务问题": "debt problems", + "财务不稳定": "financial instability", + "储蓄困难": "difficulty saving", + "人脉关系淡薄": "weak personal network", + "健康挑战": "health challenges", + "情感孤立": "emotional isolation", + "社交困难": "social difficulty", + "名声受损": "reputation damaged", + "意外事故": "accidents", + "身体虚弱": "physical frailty", + "视力问题": "vision problems", + "判断失误": "errors of judgment", + "需要努力": "requires effort", + "需要努力突破": "requires effort to break through", + "需要韧性": "requires resilience", + "需要节约": "requires thrift", + "需节俭": "needs thrift", + "挑战多": "many challenges", + "人生有挑战": "life has challenges", + "人生挑战": "life challenges", + "不安定": "unsettled", + "漂泊": "wandering", + "不安于室": "restless", + # 婚恋 / 家庭 / 健康补充(2026-09 全量导出缺词补齐) + "性格坚毅": "resolute character", + "口才与财富兼备": "eloquence with wealth", + "充满活力": "full of vigor", + "权力权威": "power and authority", + "个人魅力": "personal charm", + "需努力积累": "requires diligent accumulation", + "财务压力": "financial pressure", + "家庭纠纷": "family disputes", + "需节俭持家": "needs frugal household management", + "人生顺遂": "smooth and favorable life", + "灵性提升": "spiritual elevation", + "财富地位": "wealth and status", + "极高地位": "extremely high status", + "名声远扬": "far-reaching renown", + "道德致富": "wealth through integrity", + "慷慨大方": "generous", + "艺术财富": "artistic wealth", + "奢侈品": "luxury goods", + "财富波动": "wealth fluctuation", + "严重财务困难": "severe financial difficulty", + "需要极度节俭": "requires extreme thrift", + "需要匹配化解": "requires matching remedies", + "月亮受保护": "Moon protected", + "情感稳定": "emotional stability", + "全方位成功": "all-around success", + "四通八达": "well-connected in all directions", + "坚不可摧的成功": "unshakeable success", + "强大意志力": "strong willpower", + "星盘力量翻倍": "doubled chart strength", + "命运高度一致": "highly aligned destiny", + "宗教政治地位": "religious and political standing", + "太阳后方有行星支持": "planets following the Sun", + "表达力与行动力增强": "enhanced expression and drive", + "自我驱动力较强": "strong self-drive", + "太阳前方有行星铺垫": "planets preceding the Sun", + "内在准备力强": "strong inner preparation", + "重视计划与隐性资源": "values planning and hidden resources", + "太阳前后均有行星支撑": "planets supporting the Sun on both sides", + "自我表达较完整": "relatively complete self-expression", + "行动前后资源较足": "sufficient resources before and after action", + "资源积累能力": "resource accumulation ability", + "心智有后续支撑": "mind has follow-up support", + "配偶品德高尚": "virtuous spouse", + "经常旅行": "frequent travel", + "被亲属遗弃": "abandoned by relatives", + "缺乏家庭支持": "lack of family support", + "家庭关系疏远": "distant family relations", + "母亲健康不佳": "poor maternal health", + "与母亲缘薄": "weak bond with mother", + "母亲早逝": "early loss of mother", + "心动/感情激活观察": "romantic-activation observation", + "承诺掂量": "commitment weighing", + "不等于法律婚姻": "does not equal legal marriage", +} + +# 英文星座名集合(用于 combination 黏连修复,避免与 dignity 词黏连)。 +SIGNS_EN: frozenset[str] = frozenset({ + "Aries", "Taurus", "Gemini", "Cancer", "Leo", "Virgo", + "Libra", "Scorpio", "Sagittarius", "Capricorn", "Aquarius", "Pisces", +}) + +# 英文行星名集合。 +PLANETS_EN: frozenset[str] = frozenset({ + "Sun", "Moon", "Mars", "Mercury", "Jupiter", "Venus", "Saturn", "Rahu", "Ketu", +}) + +# dignity / 强度词(用于黏连修复)。 +DIGNITY_EN: frozenset[str] = frozenset({ + "exalted", "debilitated", "strong", "weak", "moderate", "medium", + "own", "moolatrikona", "friendly", "neutral", "enemy", "combust", +}) + +# 中文行星名 / 星座名 / 结构词 -> 英文。用于 combination 混排串的逐词替换, +# 使删 CJK 后不产生黏连或残片。 +CN_TERMS_EN: dict[str, str] = { + "自我": "self", "表现为": "expressed as", "节律": "rhythm", "学习": "learning", + "写作": "writing", "沟通": "communication", "移动": "movement", "自主行动": "independent action", + "居住": "residence", "家庭": "family", "内在安全感": "inner security", "资产基础": "asset foundation", + "先稳住根基再发挥": "stabilize foundations before expression", "祖先祭拜": "ancestral rites", + "捐赠黑芝麻和铁器": "donate black sesame and iron items", "供养婆罗门": "support Brahmins", + "相邻": "adjacent", "指标": "indicators", "内在": "inner", "星": "planet", + "1-4宫主形成命宫基础链": "1st-4th house lords form the ascendant foundational chain", + "4宫/4宫主受凶星、Rahu、Maandi/Mandi 等严重影响": "house 4/its lord severely affected by malefics, Rahu, and Maandi/Mandi", + "Dharidhra:财帛/收益主落凶宫或Lagna/Maraka/Dusthana组合": "Dharidhra: wealth/gains lord in a difficult house or Lagna/Maraka/Dusthana combination", + "Rahu在5且非土星Navamsa,或7宫主关联行星的Navamsa主落1/2/5": "Rahu in house 5 and not Saturn Navamsa, or Navamsa lord of a planet associated with the 7th lord falls in 1/2/5", + "月亮受凶星夹制/合相/相位,或4宫主Navamsa链最终落6/8/12": "Moon hemmed/conjunct/aspected by malefics, or the 4th lord's Navamsa chain ultimately falls in 6/8/12", + "第4宫有天然吉星/强星/吉性星座,或上升主入4宫且受吉星影响": "house 4 has natural benefic/strong planets or a benefic sign, or the ascendant lord enters house 4 and is influenced by benefics", + "行星集中在角宫(Chakra 轮盘)": "planets concentrated in angular houses (Chakra wheel)", + "中强": "medium-strong", + "基础链": "foundational chain", "形成": "forms", "分布在": "distributed across", + "六个星座": "six signs", "位于": "located in", "入": "enters", + "受凶星": "affected by malefics", "凶星": "malefic", "天然吉星": "natural benefic", + "吉性星座": "benefic sign", "关联": "associated with", "相位": "aspected", + "合相": "conjunct", "夹制": "hemmed", "最终落": "ultimately falls in", + "处于": "is in", "不利状态": "an unfavorable state", "严重影响": "severely affected", + "非": "not ", "角宫主": "angular house lord", "三方宫主": "trinal house lord", + "财帛": "wealth", "收益": "gains", "基础": "foundation", "自定义条件满足": "custom condition met", + "检测器未返回组合文本": "detector returned no combination text", "未返回": "not returned", + "候选组合": "Candidate combination", + "条件未齐,暂不作结果判断。": "Required conditions are incomplete; no outcome inference is made.", + "缺少条件输入": "Missing condition inputs", + "待评估": "pending evaluation", "七曜": "seven planets", "星座": "sign", + "行星": "planet", "不利": "unfavorable", "影响": "influence", "的": "'s", + "上升主": "ascendant lord", "宫主": "house lord", "互为": "mutual", + "合于": "conjunct", "变动星座": "mutable sign", "且": "and", "或": "or", + "强": "strong", "弱": "weak", "第": "house ", "宫": "", + "火星在第4宫(从月亮的2/12宫)": "Mars in house 4 (2nd/12th from Moon)", + "充足": "adequate", + "转化困境为机遇": "turns hardship into opportunity", + "不讨喜": "not widely liked", "事业通过智慧实现": "career achieved through wisdom", + "享受丰富": "enjoys abundance", "人缘好": "well liked", + "命运力量在角宫": "destiny power in angular houses", "命运助力明显": "strong support from destiny", + "善于辞令": "eloquent speech", "基础发展稳定": "stable foundational development", + "多元才华": "versatile talents", "子女受其庇护": "children receive protection", + "待补充": "to be supplemented", "心地善良": "kind-hearted", + "心智有背景支持": "mind supported by background conditions", "心智稳定性增强": "increased mental stability", + "性情愉快": "pleasant temperament", "性格坚定": "steadfast character", + "持续收入": "steady income", "政治才华": "political talent", + "月亮前后均有支撑": "Moon supported on both sides", "物质充裕": "material abundance", + "独处中成长": "growth through solitude", "生活舒适": "comfortable life", + "盈利丰厚": "substantial profits", "真诚待人": "sincere dealings", + "祖先福报影响子女": "ancestral blessings affect children", "经济稳定": "financial stability", + "自我、财富、努力、根基彼此支撑": "self, wealth, effort, and foundations support one another", + "自我表达": "self-expression", "言辞刻薄": "harsh speech", + "贵人带来社会地位": "benefactors bring social status", "贵人或隐性资源": "benefactors or hidden resources", + "资源与适应力较强": "strong resources and adaptability", + "太阳": "Sun", + "月亮": "Moon", + "水星": "Mercury", + "金星": "Venus", + "火星": "Mars", + "木星": "Jupiter", + "土星": "Saturn", + "罗睺": "Rahu", + "计都": "Ketu", + "白羊座": "Aries", + "金牛座": "Taurus", + "双子座": "Gemini", + "巨蟹座": "Cancer", + "狮子座": "Leo", + "处女座": "Virgo", + "天秤座": "Libra", + "天蝎座": "Scorpio", + "射手座": "Sagittarius", + "摩羯座": "Capricorn", + "宝瓶座": "Aquarius", + "水瓶座": "Aquarius", + "双鱼座": "Pisces", + "与": " and ", + "同在第": " together in house ", + "同处": " together in ", + "在第": " in house ", + "第": " house ", + "宫": " ", + "两侧均有行星": " planets on both sides ", + "一侧有行星": " planet on one side ", + "前后均有行星": " planets on both sides ", + # 结构词(宫主 / 落点 / 相位等),供 combination 逐词替换;按长度降序替换避免互相覆盖。 + "宫主": " house lord", + "上升主": "lagna lord", + "命主": "lagna lord", + "凶宫": " dusthana", + "角宫": " kendra", + "吉星": " benefic", + "凶星": " malefic", + "落入": " in ", + "落陷": " debilitated", + "北交点": " Rahu", + "南交点": " Ketu", + "变动星座": " mutable sign", + "奇数星座": " odd signs", + "本命盘": " natal chart", + "合相": " conjunct", + "相位": " aspect", + "互照": " mutual aspect", + "夹制": " hemmed", + "关联": " aspected by ", + "有力": " strong", + "主": " lord", + "在": " in ", + "或": " or ", + "被": " by ", + "受": " by ", +} + +# 组合条件整句 -> 英文(在逐词替换之前做全串精确匹配,避免整句被逐字拆坏)。 +# 仅收录不含盘主专属行星/星座名的通用句,跨盘可复用。 +COMBINATION_EN: dict[str, str] = { + "有Pitra Dosha迹象:Sun(1宫)与Rahu(2宫)合相/相邻": "Pitra Dosha indication: Sun (house 1) and Rahu (house 2) conjunct/adjacent", + "中性(Neutral)": "neutral (Neutral)", + "1-4宫主形成命宫基础链": "1st-4th house lords form the ascendant foundational chain", + "4宫/4宫主受凶星、Rahu、Maandi/Mandi 等严重影响": "house 4/its lord severely affected by malefics, Rahu, and Maandi/Mandi", + "Dharidhra:财帛/收益主落凶宫或Lagna/Maraka/Dusthana组合": "Dharidhra: wealth/gains lord in a difficult house or Lagna/Maraka/Dusthana combination", + "Rahu在5且非土星Navamsa,或7宫主关联行星的Navamsa主落1/2/5": "Rahu in house 5 and not Saturn Navamsa, or Navamsa lord of a planet associated with the 7th lord falls in 1/2/5", + "月亮受凶星夹制/合相/相位,或4宫主Navamsa链最终落6/8/12": "Moon hemmed/conjunct/aspected by malefics, or the 4th lord's Navamsa chain ultimately falls in 6/8/12", + "第4宫有天然吉星/强星/吉性星座,或上升主入4宫且受吉星影响": "house 4 has natural benefic/strong planets or a benefic sign, or the ascendant lord enters house 4 and is influenced by benefics", + "行星集中在角宫(Chakra 轮盘)": "planets concentrated in angular houses (Chakra wheel)", + "自定义条件满足": "custom condition satisfied", + "落陷取消格局": "cancellation of debilitation", + "上升主星有力+其他行星有力": "strong lagna lord and other strong planets", + "凶宫主在角宫": "dusthana lord in a kendra", + "命主在本命盘强(分盘瑜伽近似判断)": "lagna lord strong in natal chart (approximate varga-yoga judgment)", +} diff --git a/scripts/pl9_full_data_export.py b/scripts/pl9_full_data_export.py new file mode 100644 index 00000000..6556f010 --- /dev/null +++ b/scripts/pl9_full_data_export.py @@ -0,0 +1,2186 @@ +"""Full-data personal report, upstream pl9_ai_density at 23be1807. + +Renderer and sanitizer only. Numbers come from this repository's engine. +Upstream calculation helpers are not copied. Sections whose packet fields are +absent are omitted. The private same-case PDF transcription in upstream +`_attach_pl9_source_visible_tables` is not copied. `customer_timing_supplement` +fits apparent positions and a -5 second node adjustment; red line 4 forbids +that profile, so it is not called. +""" + +from __future__ import annotations + +import copy +import re +from collections.abc import Mapping +from typing import Any + +try: + from scripts.pl9_reader_export import ( + _pl9_english_report_title, + _pl9_public_report_title, + _pl9_report_language, + _sanitize_pl9_public_markdown, + _strip_pl9_user_engineering_markers, + render_pl9_parity_markdown, + ) +except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution + from pl9_reader_export import ( + _pl9_english_report_title, + _pl9_public_report_title, + _pl9_report_language, + _sanitize_pl9_public_markdown, + _strip_pl9_user_engineering_markers, + render_pl9_parity_markdown, + ) + +FULL_DATA_EDITION = "full_data" +REPORT_VERSION_FULL = "pl9_personal_long_report.v4" +FULL_DATA_DOMAINS = { + "career": "事业", + "wealth": "财富", + "relationship": "婚恋", + "health": "健康", + "children": "子女", + "parents": "父母", + "education": "学业", + "relocation": "迁居", + "family": "家庭", + "annual": "流年", + "timing": "时间", + "general": "综合", +} + + +def _pl9_public_term(value: Any) -> str: + """Report-facing label for a planet or sign name.""" + terms = { + "Sun": "太阳", "Moon": "月亮", "Mars": "火星", "Mercury": "水星", + "Jupiter": "木星", "Venus": "金星", "Saturn": "土星", + "Rahu": "北交点/罗睺", "Ketu": "南交点/计都", + "Aries": "白羊座", "Taurus": "金牛座", "Gemini": "双子座", + "Cancer": "巨蟹座", "Leo": "狮子座", "Virgo": "处女座", + "Libra": "天秤座", "Scorpio": "天蝎座", "Sagittarius": "射手座", + "Capricorn": "摩羯座", "Aquarius": "水瓶座", "Pisces": "双鱼座", + } + text = str(value or "").strip() + return terms.get(text, text or "-") + + +def _load_glossary(): + try: + from scripts.pl9_en_glossary import CN_TERMS_EN, COMBINATION_EN, EFFECTS_EN + except ModuleNotFoundError: + from pl9_en_glossary import CN_TERMS_EN, COMBINATION_EN, EFFECTS_EN + return CN_TERMS_EN, COMBINATION_EN, EFFECTS_EN + + +def _load_sudarshana(): + try: + from scripts.sudarshana_chakra import calc_sudarshana_chakra + except ModuleNotFoundError: + from sudarshana_chakra import calc_sudarshana_chakra + return calc_sudarshana_chakra + + +def _load_bhava_rows(): + try: + from scripts.special_lagnas import bhava_house_rows + except ModuleNotFoundError: + from special_lagnas import bhava_house_rows + return bhava_house_rows + + +def _sanitize_pl9_ai_density_markdown(markdown: str, title: str = "印度占星星盘报告") -> str: + """Keep the parity volume's facts while removing internal-only labels.""" + + try: + from pl9_language_terms import normalize_zh_markdown_terms + except ModuleNotFoundError: # pragma: no cover - package import compatibility + from scripts.pl9_language_terms import normalize_zh_markdown_terms + + text = _sanitize_pl9_public_markdown(markdown) + replacements = { + "# PL9 对标主册": f"# {title}", + "出生资料与图盘(PL9 p1-p17)": "出生资料与图盘", + "行星、星宿与宫位长页解释(PL9 p18-p29 对标)": "行星、星宿与宫位解释", + "强度、关系与 Ashtakavarga 技术页(PL9 p30-p60 对标)": "强度、关系与 Ashtakavarga", + "标准 Dasha 表格页(PL9 p61-p120 对标)": "Dasha 表格", + "Saturn 与 KP 支持页(PL9 p121-p125 对标)": "Saturn 与 KP", + "年度图盘与 Tajika(PL9 p126-p155 对标)": "年度 Varshaphala 与 Tajika", + "Bhavesh、Dosha 与 Yoga 解释页(PL9 p156-p189 对标)": "Bhavesh、Dosha 与 Yoga", + "大运长页解释(PL9 p190-p204 对标)": "大运解释", + "p3-p5 Birth Particulars / Hindu Calendar": "出生资料与印度历信息", + "p3 Birth Particulars 原版补充字段": "出生资料补充字段", + "p6 Birth Chart 行星原始位置": "Birth Chart 行星原始位置", + "p6 Birth Chart 星主与状态字段": "Birth Chart 星主与状态字段", + "p126-p127 Varshaphala 基础资料与年度行星表": "Varshaphala 基础资料与年度行星表", + "p139-p141 Varshaphala 年度结果解释": "Varshaphala 年度结果解释", + "p144-p145 Annual Dasha 结果解释": "Annual Dasha 结果解释", + "p132-p135 Mudda / Patyayini 年度 Dasha 表": "Mudda / Patyayini 年度 Dasha 表", + "missing_in_local": "未提供", + "external_mudda_start_boundary": "阶段起点", + "dasha_beginning_dates": "阶段起止日期", + "dasha_ending_dates": "阶段结束日期", + "moon_at_birth_time": "出生时月亮", + "moon_in_varshaphala": "年盘月亮", + "without_dasha_balance": "不计余额", + "balance 口径": "起始依据", + "边界口径": "日期说明", + "Progression (local ": "Progression (", + "p152 - birth_place_eight_year_overview": "", + "p153 - local_eight_year_overview": "", + "p154 - tithi_pravesh_birth_place": "", + "p155 - tithi_pravesh_current_location": "", + "外部边界:Chesta remains cross-engine method-conflicted; totals inherit that boundary。": "Chesta Bala 总分按本报告采用的计算口径解读。", + "外部边界:Chesta remains cross-engine method-conflicted; totals inherit that boundary.": "Chesta Bala 总分按本报告采用的计算口径解读。", + "资料路径": "资料来源", + "生成模块": "计算来源", + "结构": "项目", + "校验": "说明", + "functional_role_not_returned": "功能属性未单独返回", + "natural_benefic": "自然吉星", + "natural_malefic": "自然凶星", + "functional_benefic": "功能吉星", + "functional_malefic": "功能凶星", + "functional_neutral": "功能中性", + "current_dasha": "当前 Dasha", + "birth_balance": "出生余额", + "Cycle / phase": "周期 / 阶段", + "Transit of 土星": "土星行运星座", + "Beginning date": "开始日期", + "Ending date": "结束日期", + "Duration Yr-Mn-Dy": "持续时间(年-月-日)", + "First Cycle of Sadhesati": "第一轮 Sade Sati", + "Second Cycle of Sadhesati": "第二轮 Sade Sati", + "Third Cycle of Sadhesati": "第三轮 Sade Sati", + "First Dhayya": "第一段 Dhayya", + "Second Dhayya": "第二段 Dhayya", + "Third Dhayya": "第三段 Dhayya", + "(Twelfth from birth rashi)": "(本命月亮星座前一宫)", + "(On birth rashi)": "(本命月亮星座本宫)", + "(Second from birth rashi)": "(本命月亮星座后一宫)", + ";功能性角色为待补充": "", + "功能性角色为待补充。": "", + "Consideration: Mars is checked from Lagna, Moon, and Venus against the classical marriage-sensitive houses.": "校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位。", + "Consideration: this Dosha row is shown only when the local detector returns a chart-specific condition.": "校验说明:只有本命盘检测器返回具体条件时,才显示这一条 Dosha。", + "Consideration: Mars is checked from Lagna, Moon, and Venus against the classical marriage-sensitive houses.": "校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位。", + "In this chart Mars is listed in house": "本盘火星位于第", + "the detector severity is none.": "检测强度为无。", + "the detector severity is low.": "检测强度为低。", + "the detector severity is medium.": "检测强度为中。", + "the detector severity is high.": "检测强度为高。", + "Classical reference family: Muhurtha or Electional Astrology; 印度占星读书摘要.": "古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + "Classical reference family is attached in the source map.": "古典参考体系已随来源映射保留。", + "against the classical marriage-sensitive houses.": "并核对传统婚恋敏感宫位。", + "and Venus": "与金星", + "from Lagna, Moon,": "从上升与月亮", + "火星 is checked": "火星会被校验", + "In this chart 火星 is listed in 第-;": "本盘未列火星命中婚恋敏感宫位;", + "Birth Particulars / Hindu Calendar": "出生资料 / 印度历", + "Birth Chart 行星原始位置": "本命盘行星原始位置", + "Birth Chart Lords and Status Fields": "本命盘星主与状态字段", + "Chandra Chart / Bhava Spashta - Sripati System": "月亮盘 / Sripati 宫位精确表", + "Bhava Number": "宫位编号", + "Bhava Arambha / 宫位 beginning": "宫位起点", + "Bhava Madhya / Middle of 宫位": "宫位中点", + "Bhava Antya / 宫位 ending": "宫位终点", + "Planetary Raw Positions": "行星原始位置", + "| Field | Value |": "| 字段 | 数值 |", + "| Panchanga | Value |": "| Panchanga | 数值 |", + "| Planet | R/C | Sign | Degree | Longitude | House | Nakshatra | Pada | Speed |": "| 行星 | 顺逆 | 星座 | 落座度数 | 黄经 | 宫位 | 星宿 | Pada | 速度 |", + "| Planet | RL | NL | SL | SS | Status | SB |": "| 行星 | 星座主 | 星宿主 | Sub Lord | Sub-sub | 状态 | Shadbala |", + "| Point | Sign | Degree | Longitude | Lord |": "| 点位 | 星座 | 落座度数 | 黄经 | 守护星 |", + "| Body | Sign | Degree | Longitude |": "| 对象 | 星座 | 落座度数 | 黄经 |", + "| Planet | Sign | Degree | Longitude |": "| 行星 | 星座 | 落座度数 | 黄经 |", + } + for raw, public in replacements.items(): + text = text.replace(raw, public) + text = re.sub( + r"\n## Bhavesh House-Lord Interpretations\n\n.*?(?=\n## |\Z)", + "\n", + text, + flags=re.DOTALL, + ) + text = re.sub( + r"(.+?)落在(.+?),首先把(.+?)放到(.+?)这一现实场域中阅读。", + r"\1落在\2,表示\3会主要通过\4来表现。", + text, + ) + text = re.sub(r"第([1-4])足", r"第\1 Pada(星宿四分区)", text) + text = text.replace("Pada | 速度", "Pada(星宿四分区) | 速度") + text = text.replace("Natural friends", "天然友星") + text = text.replace("Natural enemies", "天然敌星") + text = text.replace("Temporary relation", "临时关系") + text = text.replace("Planet", "行星") + text = text.replace("Sign", "星座") + text = text.replace("Degree in sign", "落座度数") + text = text.replace("Degree", "度数") + text = text.replace("Longitude", "黄经") + text = text.replace("Whole-sign house", "整宫宫位") + text = text.replace("House", "宫位") + text = text.replace("Object", "对象") + text = text.replace("Reference point", "参考点") + text = text.replace("Reference chart", "参考盘") + text = text.replace("Role", "角色") + text = text.replace("Yoga", "瑜伽") + text = text.replace("Category", "类别") + text = text.replace("Combination", "组合条件") + text = text.replace("Effects / notes", "作用说明") + text = text.replace("Planet 1", "行星1") + text = text.replace("Planet 2", "行星2") + text = text.replace("Exact degree", "精确角度") + text = text.replace("Actual difference", "实际角距") + text = text.replace("Orb", "容许度") + text = text.replace("Applying", "入相") + text = text.replace("From house", "起始宫位") + text = text.replace("Target house", "目标宫位") + text = text.replace("Aspect type", "相位类型") + text = text.replace("Special", "特殊相位") + text = text.replace("friends:", "友星:") + text = text.replace("enemies:", "敌星:") + text = text.replace("True", "是") + text = text.replace("False", "否") + text = text.replace("conjunction", "合相") + text = text.replace("opposition", "对冲") + text = re.sub(r"(? str: + """Remove empty placeholder cells and compressed table fragments from user exports.""" + + zh_planet_names = "太阳|月亮|火星|水星|木星|金星|土星|北交点/罗睺|南交点/计都" + en_planet_names = "Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu" + + def normalize_condition_cell(cell: str) -> str: + value = cell.strip() + if value in {"", "-", "未列", "not listed"}: + return "未列" if language == "zh" else "not listed" + if language == "zh": + value = re.sub(rf"\b(\d+)({zh_planet_names})(\d+)\b", r"第\1宫;\2在第\3宫", value) + value = re.sub(rf"\b({zh_planet_names})(\d+)\b", r"\1在第\2宫", value) + value = re.sub(r"(? str: + lines = raw_text.splitlines() + out: list[str] = [] + i = 0 + empty_values = {"", "-", "未列", "not listed"} + while i < len(lines): + if ( + i + 1 < len(lines) + and lines[i].lstrip().startswith("|") + and lines[i + 1].lstrip().startswith("|") + and set(lines[i + 1].replace("|", "").replace("-", "").replace(":", "").strip()) == set() + ): + block: list[str] = [] + while i < len(lines) and lines[i].lstrip().startswith("|"): + block.append(lines[i]) + i += 1 + rows = [[cell.strip() for cell in line.strip().strip("|").split("|")] for line in block] + if len(rows) >= 2: + headers = rows[0] + body = rows[2:] + for row in rows[2:]: + for index, header in enumerate(headers): + if index >= len(row): + continue + if header in {"组合条件", "Combination"}: + row[index] = normalize_condition_cell(row[index]) + elif header in {"作用说明", "Effects / notes", "Notes", "Summary"} and row[index].strip() in empty_values: + row[index] = "未列" if language == "zh" else "not listed" + remove: set[int] = set() + for index, header in enumerate(headers): + column = [row[index] if index < len(row) else "" for row in body] + empty_count = sum(1 for cell in column if cell.strip() in empty_values) + if header in {"作用说明", "Effects / notes", "Notes", "Summary"} and body and empty_count / len(body) >= 0.65: + remove.add(index) + if header in {"组合条件", "Combination"} and body and empty_count / len(body) >= 0.9: + remove.add(index) + if remove and len(remove) < len(headers): + rows = [[cell for index, cell in enumerate(row) if index not in remove] for row in rows] + rows[1] = ["---"] * len(rows[0]) + out.extend("| " + " | ".join(row) + " |" for row in rows) + else: + out.extend(block) + continue + out.append(lines[i]) + i += 1 + return "\n".join(out) + + text = markdown + empty_label = "未列" if language == "zh" else "not listed" + text = re.sub(r"(?:^|;\s*)-(?:\s*;\s*-){1,}\s*$", empty_label, text, flags=re.MULTILINE) + text = re.sub(r"\|\s*(?:-\s*\|\s*){3,}(?:missing_in_local|未单独列出)\s*\|", "", text) + text = re.sub(r"(?m)^\|\s*-\s*\|.*\|\s*(?:missing_in_local|未单独列出|not listed)\s*\|\s*$", "", text) + text = text.replace("missing_in_local", empty_label) + text = re.sub(r"\blisted planets(\d+)\b", r"\1 listed planet(s)", text) + text = re.sub(r"\b(木星|太阳|月亮|火星|水星|金星|土星|北交点/罗睺|南交点/计都)(\d+)\b", r"\1在第\2宫", text) + text = re.sub(r"\b(Jupiter|Sun|Moon|Mars|Mercury|Venus|Saturn|Rahu|Ketu)(\d+)\b", r"\1 in house \2", text) + text = re.sub(r"\b(MercuryMoon|SunMercury|MoonVenus)(\d+)\b", r"\1 in house \2", text) + text = re.sub(r"\b(\d+)(Mars|Venus|Jupiter|Sun|Moon|Mercury|Saturn|Rahu|Ketu)(\d+)\b", r"house \1; \2 in house \3", text) + text = re.sub(r"\b(\d+)弱/", r"第\1宫较弱", text) + text = re.sub(r"\b(\d+)weak/", r"house \1 weak", text) + text = re.sub(r"\b(\d+)/\b", r"第\1宫条件", text) + text = re.sub(r"//,\s*", "", text) + text = text.replace("D1/D9debilitated", "D1/D9 落陷" if language == "zh" else "D1/D9 debilitated") + text = text.replace("Dharidhra:", "贫困组合:" if language == "zh" else "Daridra combination:") + text = text.replace("Dharidhra Yoga", "Daridra Yoga") + text = text.replace("-; -; 弱", "弱") + text = text.replace("-; -; weak", "weak") + text = re.sub(r"-;\s*(强|弱|中等|strong|weak|medium|moderate);\s*-", r"\1", text) + text = re.sub(r"-;\s*/", empty_label, text) + text = re.sub(r"\b(MercuryMoon|SunVenus|VenusSun|MoonVenus|VenusMoon)(?:\s+in\s+)?第?(\d+)\b", r"\1 in house \2", text) + text = re.sub(r"\b(SunVenus|VenusSun|MoonVenus|VenusMoon|MercuryMoon)(\d+)\b", r"\1 in house \2", text) + text = re.sub(r"\b(MercuryVenus|VenusMercury)(\d+)\b", r"\1 in house \2", text) + text = re.sub(r"\b(\d+)KendraTrikona\b", r"house \1; kendra/trikona condition", text) + text = re.sub(r"\b(\d+)Kendra\b", r"house \1; kendra condition", text) + text = re.sub(r"\b(\d+)/(?=\s*\|)", r"house \1 condition", text) + text = text.replace("weak/", "weak") + text = text.replace(", Kendra", "kendra condition") + text = re.sub(r"\b(Mars|Mercury|Venus|Jupiter|Saturn|Sun|Moon|Rahu|Ketu)(\d+)(Aries|Taurus|Gemini|Cancer|Leo|Virgo|Libra|Scorpio|Sagittarius|Capricorn|Aquarius|Pisces)\b", r"\1 in house \2, \3", text) + text = re.sub(r"\b(Saturn|Mars|Mercury|Venus|Jupiter|Sun|Moon|Rahu|Ketu)(medium|weak|strong|moderate)\b", r"\1 \2", text) + if language == "zh": + text = re.sub( + r"Consideration:\s*(.+?)会被校验 from 上升,\s*月亮,\s*and 金星 并核对传统婚恋敏感宫位。\s*In this chart\s+(.+?)\s+is listed in 第(.+?)\s+in\s+(.+?);\s*检测强度为(.+?)。\s*古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + r"校验说明:\1会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位;本盘中,\2位于第\3宫、\4,检测强度为\5。古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + text, + ) + text = re.sub( + r"校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位。\s*本盘火星位于第\s*(.+?)\s+in\s+(.+?);\s*检测强度为(.+?)。\s*古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + r"校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位;本盘中,火星位于第\1宫、\2,检测强度为\3。古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + text, + ) + text = re.sub( + r"Consideration:\s*火星会被校验 from 上升,\s*月亮,\s*and 金星 并核对传统婚恋敏感宫位。\s*In this chart 火星 is listed in 第(.+?)\s+in\s+(.+?);\s*the detector severity is\s+(.+?)\.\s*古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + r"校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位;本盘中,火星位于第\1宫、\2,检测强度为\3。古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + text, + ) + text = re.sub(r"\n项目字段:.*?(?=\n)", "", text) + text = re.sub( + r"Consideration:\s*(.+?)\s+is checked from\s+(.+?)\s+against the classical marriage-sensitive houses\.\s*In this chart\s+(.+?)\s+is listed in house\s+(.+?);?\s+the detector severity is\s+(.+?)\.\s*Classical reference family:\s*(.+?)\.", + r"校验说明:\1会从\2出发,核对传统婚恋敏感宫位;本盘中,\3列于第\4宫,检测强度为\5。古典参考体系:\6。", + text, + ) + text = re.sub( + r"Consideration:\s*(.+?)\s+is checked from\s+(.+?)\s+against the classical marriage-sensitive houses\.\s*In this chart\s+(.+?)\s+is listed in\s+(.+?);?\s+the detector severity is\s+(.+?)\.\s*Classical reference family:\s*(.+?)\.", + r"校验说明:\1会从\2出发,核对传统婚恋敏感宫位;本盘中,\3列于\4,检测强度为\5。古典参考体系:\6。", + text, + ) + text = re.sub( + r"Consideration:\s*火星会被校验 from 上升,\s*月亮,\s*and 金星 并核对传统婚恋敏感宫位。\s*In this chart 火星 is listed in 第-;\s*检测强度为无。\s*古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + "校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位;本盘未列火星命中婚恋敏感宫位,检测强度为无。古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + text, + ) + text = re.sub( + r"Consideration:\s*火星 is checked from 上升,\s*月亮,\s*and 金星 against the classical marriage-sensitive houses\.\s*In this chart 火星 is listed in 第-;\s*the detector severity is none\.\s*Classical reference family:\s*Muhurtha or Electional Astrology;\s*印度占星读书摘要\.", + "校验说明:火星婚姻煞会同时从上升、月亮与金星出发,核对传统婚恋敏感宫位;本盘未列火星命中婚恋敏感宫位,检测强度为无。古典参考体系:Muhurtha / Electional Astrology 与印度占星读书摘要。", + text, + ) + text = re.sub( + r"Consideration:\s*this Dosha row is shown only when the local detector returns a chart-specific condition\.\s*Classical reference family is attached in the source map\.", + "校验说明:只有本命盘检测器返回具体条件时,才显示这一条 Dosha;古典参考体系已随来源映射保留。", + text, + ) + planet_pair_terms = { + "MercuryMoon": "水星与月亮", + "SunVenus": "太阳与金星", + "VenusSun": "金星与太阳", + "MoonVenus": "月亮与金星", + "VenusMoon": "金星与月亮", + "MercuryVenus": "水星与金星", + "VenusMercury": "金星与水星", + } + for raw, public in planet_pair_terms.items(): + text = text.replace(raw, public) + english_planet_to_zh = { + "Sun": "太阳", + "Moon": "月亮", + "Mars": "火星", + "Mercury": "水星", + "Jupiter": "木星", + "Venus": "金星", + "Saturn": "土星", + "Rahu": "北交点/罗睺", + "Ketu": "南交点/计都", + } + def triple_planets_to_zh(match: re.Match[str]) -> str: + names = [english_planet_to_zh.get(match.group(i), match.group(i)) for i in (1, 2, 3)] + return f"{'、'.join(names)}在第{match.group(4)}宫" + text = re.sub( + r"\b(Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu)(Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu)(Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu)(\d+)\b", + triple_planets_to_zh, + text, + ) + text = re.sub(r"in house (\d+)", r"在第\1宫", text) + else: + text = re.sub(r"第(\d+)宫条件第(\d+)宫条件(\d+)", r"houses \1/\2/\3 condition", text) + text = re.sub(r"第(\d+)宫条件(\d+)", r"houses \1/\2 condition", text) + text = re.sub(r"第(\d+)宫条件", r"house \1 condition", text) + text = re.sub(r"\b(Moon|Sun|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu)(Rahu|Ketu|Moon|Sun|Mars|Mercury|Jupiter|Venus|Saturn)(Rahu|Ketu|Moon|Sun|Mars|Mercury|Jupiter|Venus|Saturn)(\d+)\b", r"\1, \2, \3 in house \4", text) + text = re.sub(r"\b(\d+)(strong|medium|weak|moderate)/", r"house \1 \2", text) + text = re.sub(r"\b(\d+)Navamsa(\d+)/house (\d+) condition(\d+)", r"Navamsa \1/\2; houses \3/\4 condition", text) + text = re.sub(r"\b(\d+)Navamsa(\d+)/houses? ([0-9/]+) condition", r"Navamsa \1/\2; houses \3 condition", text) + text = re.sub(r"\b(\d+)Navamsa(\d+)/(\d+)", r"Navamsa \1/\2/\3 condition", text) + text = text.replace("MercuryMoon", "Mercury-Moon") + text = text.replace("SunMercury", "Sun-Mercury") + text = text.replace("MoonVenus", "Moon-Venus") + text = text.replace("SunVenus", "Sun-Venus") + text = text.replace("VenusSun", "Venus-Sun") + text = text.replace("VenusMoon", "Venus-Moon") + text = text.replace("MercuryVenus", "Mercury-Venus") + text = text.replace("VenusMercury", "Venus-Mercury") + text = re.sub(r"\|\s*-\s*\|", f"| {empty_label} |", text) + text = compress_markdown_tables(text) + text = re.sub(r"\n{3,}", "\n\n", text) + return text + + + +def _repair_user_facing_orphan_table_headers(markdown: str, *, language: str) -> str: + """Restore user-facing table headers removed by language cleanup. + + The source report may contain Chinese section/table headers before English + sanitization. Dropping those lines is correct, but the markdown separator + row must not be left as the first row of a table. + """ + + def is_table_line(line: str) -> bool: + return line.lstrip().startswith("|") and line.rstrip().endswith("|") + + def is_separator(line: str) -> bool: + stripped = line.strip() + if not is_table_line(stripped): + return False + body = stripped.replace("|", "").replace("-", "").replace(":", "").strip() + return body == "" + + def cells(line: str) -> list[str]: + return [cell.strip() for cell in line.strip().strip("|").split("|")] + + def has_cjk(value: str) -> bool: + return bool(re.search(r"[\u4e00-\u9fff]", value)) + + def header_for(next_cells: list[str]) -> list[str]: + count = len(next_cells) + sample = " | ".join(next_cells) + if language == "zh": + if count == 4 and re.search(r"火星婚姻煞|时蛇煞|Sade Sati|Dosha", sample): + return ["项目", "是否成立", "条件", "说明"] + return [f"列{index}" for index in range(1, count + 1)] + if count == 2: + return ["Field", "Value"] + if count == 3 and re.search(r"Saham|Arudha|Lagna", sample, re.I): + return ["Point", "Sign", "Degree"] + if count == 4 and re.search(r"\b(Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu)\b", sample): + if re.search(r"\d+\.\d+°", sample): + return ["Planet", "Sign", "Longitude", "House"] + return ["Lord", "Sign/Rashi", "Start", "End"] + if count == 4 and re.search(r"\d{4}-\d{2}-\d{2}", sample): + return ["Year", "Age", "Solar return time", "Annual ascendant"] + if count == 4: + return ["Name", "Sign", "Degree", "House"] + if count == 5 and re.search(r"\d{4}-\d{2}-\d{2}", sample): + if re.search(r"\d{4}-\d{2}-\d{2}.*-", sample): + return ["Order", "Lord", "Duration", "Period", "Notes"] + return ["Order", "Lord", "Start", "End", "Years"] + if count == 5 and re.search(r"^\d+$", next_cells[0]): + return ["House", "Sign", "SAV", "Strength", "Notes"] + if count == 5: + return ["House", "Sign", "Lord", "Score", "Key factors"] + if count == 6 and re.search(r"\d{4}-\d{2}-\d{2}", sample): + return ["Order", "Major lord", "Sub lord", "Sub-sub lord", "Start", "End"] + if count == 6 and re.search(r"Paravatamsa|Simhasanamsa|Vimsopaka", sample): + return ["Planet", "Score", "Grade", "Combustion", "Retrograde", "Debilitated"] + if count == 6: + return ["Order", "Year", "Solar return time", "Annual ascendant", "Ascendant degree", "Planetary longitudes"] + if count == 8: + return ["Planet 1", "Planet 2", "Aspect", "Exact degree", "Actual difference", "Orb", "Applying", "Strength"] + return [f"Column {index}" for index in range(1, count + 1)] + + lines = markdown.splitlines() + out: list[str] = [] + i = 0 + while i < len(lines): + line = lines[i] + if not is_separator(line): + out.append(line) + i += 1 + continue + previous_is_table = bool(out and is_table_line(out[-1]) and not is_separator(out[-1])) + if previous_is_table: + out.append(line) + i += 1 + continue + j = i + 1 + while j < len(lines) and not lines[j].strip(): + j += 1 + if j >= len(lines) or not is_table_line(lines[j]) or is_separator(lines[j]): + i += 1 + continue + next_cells = cells(lines[j]) + header = header_for(next_cells) + if language == "en": + header = [cell if not has_cjk(cell) else f"Column {idx}" for idx, cell in enumerate(header, 1)] + out.append("| " + " | ".join(header) + " |") + out.append("| " + " | ".join(["---"] * len(header)) + " |") + i += 1 + repaired = "\n".join(out) + lines = repaired.splitlines() + compacted: list[str] = [] + i = 0 + while i < len(lines): + if ( + i + 1 < len(lines) + and is_table_line(lines[i]) + and is_separator(lines[i + 1]) + ): + j = i + 2 + while j < len(lines) and not lines[j].strip(): + j += 1 + next_starts_new_empty_table = ( + j + 1 < len(lines) + and is_table_line(lines[j]) + and not is_separator(lines[j]) + and is_separator(lines[j + 1]) + ) + if j >= len(lines) or not is_table_line(lines[j]) or is_separator(lines[j]) or next_starts_new_empty_table: + i = i + 2 + continue + compacted.append(lines[i]) + i += 1 + lines = compacted + compacted = [] + i = 0 + while i < len(lines): + compacted.append(lines[i]) + if is_separator(lines[i]): + j = i + 1 + while j < len(lines) and not lines[j].strip(): + j += 1 + if j > i + 1 and j < len(lines) and is_table_line(lines[j]) and not is_separator(lines[j]): + i = j + continue + i += 1 + return "\n".join(compacted) + + + +def _sanitize_pl9_ai_density_markdown_en(markdown: str, title: str = "Vedic Astrology Chart Report") -> str: + """English user volume: remove engineering labels without translating facts.""" + + try: + from pl9_language_terms import normalize_en_markdown_terms + except ModuleNotFoundError: # pragma: no cover - package import compatibility + from scripts.pl9_language_terms import normalize_en_markdown_terms + + text = markdown + replacements = { + "# PL9 对标主册": f"# {title}", + "PL9 对标主册": title, + "PL9 风格专业排盘导出": title, + "parameter_sensitive / unverified": "reference", + "parameter_sensitive": "reference", + "partial_verified": "listed", + "pyjhora_behavior_only / not_multiengine_parity": "reference", + "restricted_or_unclosed": "reference", + "not_applicable": "not listed", + "blocked_and_conflict_fields_cannot_generate_final_predictions": "listed only", + "blocked": "not listed", + "profile_sensitive": "reference", + "source_path": "source", + "producer": "calculation item", + "schema": "item", + "audit": "review", + "status": "status", + "dasha_beginning_dates": "beginning dates", + "dasha_ending_dates": "ending dates", + "birth_balance": "birth balance", + "moon_in_varshaphala": "Moon in Varshaphala", + "without_dasha_balance": "without Dasha balance", + "external_mudda_start_boundary": "start boundary", + "p132-p135 Mudda / Patyayini 年度 Dasha 表": "Mudda / Patyayini Annual Dasha Table", + "p136-p138 Tajika Yoga Presence 对照表": "Tajika Yoga Presence Table", + "行星、星宿与宫位长页解释(PL9 p18-p29 对标)": "Planets, Nakshatras and Houses", + "大运长页解释(PL9 p190-p204 对标)": "Dasha Interpretations", + "强度、关系与 Ashtakavarga 技术页(PL9 p30-p60 对标)": "Strength, Relationships and Ashtakavarga", + "标准 Dasha 表格页(PL9 p61-p120 对标)": "Standard Dasha Tables", + "Saturn 与 KP 支持页(PL9 p121-p125 对标)": "Saturn and KP Tables", + "年度图盘与 Tajika(PL9 p126-p155 对标)": "Annual Varshaphala and Tajika", + "Bhavesh、Dosha 与 Yoga 解释页(PL9 p156-p189 对标)": "Bhavesh, Dosha and Yoga Interpretations", + "Result: when present, Kuja/Mangala Dosha is read as heat, impatience, conflict, or pressure around partnership handling; when absent or cancelled, the report does not promote it as a dominant relationship obstacle. The result must be read with the seventh house, Venus, Upapada, Navamsha, current Dasha, and partner-chart comparison where available.": "When present, Kuja/Mangala Dosha is read as heat, impatience, conflict, or pressure around partnership handling. When absent or cancelled, it is not treated as a dominant relationship obstacle. This factor is read together with the seventh house, Venus, Upapada, Navamsha, current Dasha, and partner-chart comparison where available.", + "Consideration: this Dosha row is shown only when the local detector returns a chart-specific condition. Classical reference family is attached in the source map.": "This Dosha row is shown only when the chart-specific condition is present. The classical reference family is retained in the source layer.", + "Result: this row contributes a supporting condition and is not promoted over the core chart, Dasha, and divisional evidence.": "This row contributes a supporting condition and is not promoted over the core chart, Dasha, and divisional evidence.", + } + for raw, public in replacements.items(): + text = text.replace(raw, public) + reverse_terms = { + "出生资料与图盘": "Birth Data and Charts", + "姓名": "Name", + "出生": "Birth", + "计算口径": "Calculation profile", + "出生日期": "Birth date", + "出生时间": "Birth time", + "纬度": "Latitude", + "经度": "Longitude", + "时区": "Time zone", + "岁差名称": "Ayanamsa name", + "岁差值": "Ayanamsa value", + "岁差": "Ayanamsa", + "交点模式": "Node mode", + "吉性_count": "auspicious_count", + "中(Mixed)": "Mixed", + "行星原始位置": "Planetary Raw Positions", + "星主与状态字段": "Lords and Status Fields", + "星宿": "Nakshatra", + "状态": "Status", + "对象": "Object", + "落座度数": "Degree in sign", + "落座": "Sign", + "黄经": "Longitude", + "整宫宫位": "Whole-sign house", + "资料": "Data", + "点位": "Points", + "敏感点与特殊落点": "Sensitive and Special Points", + "本命与分盘北印度图盘": "Natal and Divisional North Indian Charts", + "Ashtakavarga 完整宫位分数": "Ashtakavarga Full House Scores", + "分盘对出生时间敏感,原始计算供核对,不单独增加结论的确定性。": "Divisional charts are sensitive to birth time. The raw calculations are here for checking and do not by themselves make a conclusion more certain.", + "行星主星与状态": "Planet Lords and Dignity", + "| 行星 | 星座主 | 星宿主 | 分主 | 分分主 | 尊贵状态 | 力量比 |": "| Planet | Sign lord | Star lord | Sub lord | Sub-sub lord | Dignity | Strength ratio |", + "Avastha 行星状态": "Planetary Avasthas", + "Sudarshan Chakra 资料": "Sudarshan Chakra Data", + "Sudarshana Chakra(三参考点原始结构)": "Sudarshana Chakra Tri-Reference Structure", + "数值位置表": "Numeric Position Table", + "上升": "Lagna", + "合计": "Total", + "月亮 Chart": "Moon Chart", + "九分盘": "Navamsha", + "可用瑜伽总表": "Applicable Yoga Summary", + "可用吉性瑜伽": "Applicable Benefic Yogas", + "可用混合型瑜伽": "Applicable Mixed Yogas", + "可用凶性瑜伽": "Applicable Malefic Yogas", + "检出的瑜伽数量": "Yoga detected count", + "参考层": "Context layers", + "| 瑜伽 | 类别 | 强度 | 组合条件 | 作用说明 |": "| Yoga | Category | Strength | Combination | Effects / notes |", + "|------|------|------|----------|----------|": "|------|----------|----------|-------------|-----------------|", + "合相": "conjunction", + "特殊组合": "special", + "婚恋": "kalatra", + "不利组合": "durbhaga", + "太阳瑜伽": "solar yoga", + "月亮瑜伽": "lunar yoga", + "中等": "moderate", + "常见": "common", + "偶见": "occasional", + "强": "strong", + "弱": "weak", + "中强": "medium-strong", + "中": "medium", + "主题": "Topic", + "建议": "Practice", + "宝石/次数": "Gem / count", + "说明": "Notes", + "太阳": "Sun", + "月亮": "Moon", + "火星": "Mars", + "水星": "Mercury", + "木星": "Jupiter", + "金星": "Venus", + "土星": "Saturn", + "北交点/罗睺": "Rahu", + "南交点/计都": "Ketu", + "北交点": "Rahu", + "南交点": "Ketu", + "极友": "great friend", + "入友": "friendly", + "友好星座": "friendly sign", + "中性": "neutral", + "落陷": "debilitated", + "入庙": "own sign", + "充足": "adequate", + "自然吉星": "natural benefic", + "自然凶星": "natural malefic", + "功能吉星": "functional benefic", + "功能凶星": "functional malefic", + "功能中性": "functional neutral", + "第": "house ", + "宫": "", + "足": "pada", + "大运": "Dasha", + "财富": "wealth", + "事业": "career", + "健康": "health", + "吉性": "benefic", + "王权/事业": "authority/career", + "天象格局": "configuration", + "子女/创造": "children/creativity", + "寿命/健康": "longevity/health", + "愿望/收益": "desire/gains", + "解脱/内在": "moksha/inner life", + "贵人/助力": "patronage/support", + "迁移/出行": "travel/movement", + "婚姻稳定": "marital stability", + "自定义条件满足": "custom condition satisfied", + "同在": "together in", + "在": "in", + "与": "and", + "白羊座": "Aries", + "金牛座": "Taurus", + "双子座": "Gemini", + "巨蟹座": "Cancer", + "狮子座": "Leo", + "处女座": "Virgo", + "天秤座": "Libra", + "天蝎座": "Scorpio", + "射手座": "Sagittarius", + "摩羯座": "Capricorn", + "水瓶座": "Aquarius", + "双鱼座": "Pisces", + "前阿沙陀(Purva Ashadha)": "Purva Ashadha", + "Mula(根、本源)": "Mula", + "后破伽(Uttara Phalguni)": "Uttara Phalguni", + "后阿沙陀(Uttara Ashadha)": "Uttara Ashadha", + "后跋陀罗(Uttara Bhadrapada)": "Uttara Bhadrapada", + "阿湿毗尼(Ashwini)": "Ashwini", + "巴拉尼(Bharani)": "Bharani", + "基利提卡(Krittika)": "Krittika", + "罗希尼(Rohini)": "Rohini", + "鹿首(Mrigashira)": "Mrigashira", + "阿尔德拉(Ardra)": "Ardra", + "普那婆苏(Punarvasu)": "Punarvasu", + "普沙(Pushya)": "Pushya", + "阿什列沙(Ashlesha)": "Ashlesha", + "摩伽(Magha)": "Magha", + "前破伽(Purva Phalguni)": "Purva Phalguni", + "哈斯塔(Hasta)": "Hasta", + "吉多罗(Chitra)": "Chitra", + "斯瓦蒂(Swati)": "Swati", + "毗舍佉(Vishakha)": "Vishakha", + "阿奴罗陀(Anuradha)": "Anuradha", + "杰耶什塔(Jyeshtha)": "Jyeshtha", + "室罗伐那(Shravana)": "Shravana", + "陀尼湿陀(Dhanishta)": "Dhanishta", + "百药宿(Shatabhisha)": "Shatabhisha", + "前跋陀罗(Purva Bhadrapada)": "Purva Bhadrapada", + "雷瓦蒂(Revati)": "Revati", + } + for raw, public in sorted(reverse_terms.items(), key=lambda item: len(item[0]), reverse=True): + text = text.replace(raw, public) + phrase_terms = { + "Sudarshana Chakra 三参考点盘 (BPHS标准)": "Sudarshana Chakra tri-reference chart (BPHS standard)", + "行星、Nakshatra与宫位解释": "Planets, Nakshatras and Houses", + "身份、目标与可见责任": "identity, purpose and visible responsibility", + "情绪、安全感与适应方式": "emotional security and adaptive response", + "行动、竞争与执行方式": "action, competition and execution style", + "学习、表达与交易能力": "learning, expression and exchange", + "信念、判断与扩张方式": "belief, judgment and expansion", + "关系、审美与协调能力": "relationship, taste and harmonizing capacity", + "结构、压力与长期责任": "structure, pressure and long-term responsibility", + "业力放大、突破与非传统路径": "amplification, disruption and unconventional pathways", + "收束、分离与内在化路径": "release, separation and inwardization", + "职业、职责与社会角色": "career, duty and public role", + "合作、关系与对外互动": "partnership, contracts and public interaction", + "自我选择、身体节律和人生方向": "self-direction, body rhythm and life orientation", + "家庭基础、居住与内在稳定": "home base, residence and inner stability", + "行动、学习与沟通": "initiative, learning and communication", + "资源、表达与家庭价值": "resources, speech and family values", + "创造、学习与判断": "creativity, learning and judgment", + "服务、压力与日常责任": "service, pressure and daily obligations", + "共享资源、风险与深层变化": "shared resources, risk and deep change", + "信念、远行与高阶学习": "belief, long-distance travel and higher learning", + "收益、社群与长期愿景": "gains, networks and long-range aims", + "休整、支出与幕后领域": "retreat, expenditure and behind-the-scenes matters", + "这颗星的表达会通过定位星所在宫位继续展开": "the planet continues through the house occupied by its dispositor", + "Nakshatra主会进一步修饰这颗星的表达方式,并与当前Dasha节奏一起阅读": "the nakshatra lord further qualifies the planet and is read with the current Dasha rhythm", + "自然属性与功能属性不同,阅读时需要同时参考两层": "the natural and functional roles differ, so both layers are considered", + "这些力量信息用于判断该落点的表现strongweak,并与宫主、Nakshatra和Dasha节奏合看": "these strength factors are read with the house lord, nakshatra and Dasha rhythm", + "当前不在Dasha主导层,其Topic更偏向本命基础背景": "is not currently a leading Dasha factor; its topic is mainly natal background", + "会把该宫Topic带到": "carries this house topic into", + "表现为必须落到行动和成果": "showing that the matter must be grounded in action and results", + "表现为需要通过他人关系显化": "showing that the matter expresses through others and relationship interfaces", + "表现为本人主动承接": "showing direct personal involvement", + "判断strongweak时同时参考宫主尊贵、同宫行星、相位、六维力量、分盘重复度和当前Dasha": "strength is judged with dignity, conjunctions, aspects, Shadbala, divisional repetition and current Dasha", + "这一宫主位置需要与行星力量、分盘和相位合看": "this house-lord placement is read with planetary strength, divisions and aspects", + "当前不在活动Dasha主导层,该宫主Topic以本命基础背景为主": "is not currently active as a Dasha ruler; this house topic remains natal background", + "无其他同宫行星": "no other planets in the same house", + "本案检测到": "detected in this chart", + "本案未检测到": "not detected in this chart", + "原始摘要": "source summary", + "结构字段": "structured fields", + "未列额外结构字段": "no additional structured fields listed", + "传统辅助Practice": "Traditional Supportive Practices", + "瑜伽": "Yoga", + "格局": "Yoga", + "王者": "authority", + "落陷取消": "debilitation cancellation", + "大地": "earth/support", + "后": "after", + "前": "before", + "月吉夹": "Moon hemmed by benefics", + "身体": "body", + "疾病": "illness", + "独子": "single child", + "常行": "frequent movement", + "真诚": "sincerity", + "田园": "field/land", + "婚姻": "marriage", + "贤妻": "spouse virtue", + "性欲": "sensuality", + "消瘦体型": "lean body", + "精神错乱": "mental disturbance", + "言语障碍": "speech impediment", + "车": "cart", + "贫困": "poverty", + "被亲属遗弃": "family abandonment", + "母亲不利": "mother affliction", + "权力地位;事业成功;社会影响力": "authority, career success and social influence", + "财富丰厚;生活富足;物质成功": "wealth, comfort and material success", + "财富充裕;生活舒适;性情愉快": "wealth, comfort and pleasant disposition", + "自力更生;财富充裕;受人尊敬": "self-reliance, wealth and respect", + "名声清白;受人敬仰;事业有成": "clean reputation, respect and career success", + "婚姻延迟;需要匹配化解": "marital delay; matching and remedy may be needed", + "婚姻延迟;配偶关系紧张": "marital delay and partnership tension", + "财务困难;收入不稳定;需节俭": "financial difficulty, unstable income and need for thrift", + "表达受阻,可能表现为说话不顺、迟疑或难以清楚表达": "speech may be obstructed, hesitant or difficult to express clearly", + "思绪容易扰动,需留意情绪、压力与判断稳定性": "the mind may be disturbed; emotional pressure and judgment stability need care", + "体型偏瘦,体质较敏感": "leaner body type and sensitive constitution", + "火": "Mars", + "水": "Mercury", + "木": "Jupiter", + "金": "Venus", + "土": "Saturn", + "日": "Sun", + "月": "Moon", + } + for raw, public in sorted(phrase_terms.items(), key=lambda item: len(item[0]), reverse=True): + text = text.replace(raw, public) + text = normalize_en_markdown_terms(text) + + def _retain_remaining_cjk_lines(raw_text: str) -> str: + # Preserve source values when the English glossary has no translation. + # Dropping a whole row makes the English delivery look clean while + # silently losing evidence. A visible marker keeps the row auditable + # until a real translation is added. + kept: list[str] = [] + for line in raw_text.splitlines(): + if re.search(r"[\u4e00-\u9fff]", line): + line = re.sub( + r"[\u4e00-\u9fff]+", + lambda match: f"[source text retained: {match.group(0)}]", + line, + ) + kept.append(line.rstrip()) + return "\n".join(kept) + + text = _retain_remaining_cjk_lines(text) + drop_fragments = ( + "jd_ut:", + "time_offset_seconds", + "swisseph_newton", + "source_path", + "producer", + "schema", + "parameter_sensitive", + "blocked", + "PyJHora", + ) + cleaned_lines = [ + line.rstrip() + for line in text.splitlines() + if not any(fragment in line for fragment in drop_fragments) + ] + text = "\n".join(cleaned_lines) + text = re.sub(r"(?m)^(#{2,6}\s*)p\d+(?:-p\d+)?\s*[·.-]?\s*", r"\1", text) + text = re.sub(r"\s*\(PL9 p\d+(?:-p\d+)?[^)]*\)", "", text) + text = re.sub(r"\s*(PL9 p\d+(?:-p\d+)?[^)]*)", "", text) + text = re.sub(r"^#\s+.*$", f"# {title}", text, count=1, flags=re.MULTILINE) + text = text.replace(":", ": ").replace(";", "; ").replace(",", ", ").replace("。", ".") + text = re.sub(r"\n{3,}", "\n\n", text).strip() + "\n" + note_heading = "## Local Equivalent Interpretive Notes" + first_note = text.find(note_heading) + second_note = text.find(note_heading, first_note + len(note_heading)) if first_note >= 0 else -1 + if second_note >= 0: + text = text[:second_note].rstrip() + "\n" + if not text.startswith("# "): + text = f"# {title}\n\n{text}" + text = _clean_pl9_user_table_placeholders(text, language="en") + known_header_repairs = { + "|-------|------|------|-------|----------|": "| House | Sign | Lord | Score | Key factors |\n|-------|------|------|-------|----------|", + "|----------|----------|--------|--------------|-------------------|-----|----------|----------|": "| Planet 1 | Planet 2 | Aspect | Exact degree | Actual difference | Orb | Applying | Strength |\n|----------|----------|--------|--------------|-------------------|-----|----------|----------|", + } + for separator, repaired in known_header_repairs.items(): + text = re.sub( + rf"(?m)^{re.escape(separator)}$", + repaired, + text, + ) + text = _repair_user_facing_orphan_table_headers(text, language="en") + # The parity renderer contributes sections after the raw appendix has + # already been normalized. Run the same semantic cleanup at the final + # user-volume boundary so legacy payload fragments cannot leak through. + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)\1(?=\d|\b)", + r"\1 ", + text, + flags=re.IGNORECASE, + ) + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)(\d{1,2})\b", + r"\1 \2", + text, + flags=re.IGNORECASE, + ) + text = re.sub(r"(?i)(condition)(houses?|signs?|planets?)", r"\1 \2", text) + text = re.sub( + r"\b(houses?|signs?|planets?)\s+\1\b", + r"\1", + text, + flags=re.IGNORECASE, + ) + # 防御性幂等:折叠残留的 "condition condition"(确保 sanitize 可安全重复执行)。 + text = re.sub(r"\b(condition)(?:\s+condition)+\b", r"\1", text, flags=re.IGNORECASE) + text = re.sub(r"\n{3,}", "\n\n", text).strip() + "\n" + return text + + + +def _pl9_markdown_hygiene_receipt(markdown: str, *, language: str) -> dict[str, Any]: + """Check the Markdown before rendering user-facing PDF.""" + + try: + from pl9_language_terms import language_quality_receipt + except ModuleNotFoundError: # pragma: no cover - package import compatibility + from scripts.pl9_language_terms import language_quality_receipt + + def is_table_line(line: str) -> bool: + stripped = line.strip() + return stripped.startswith("|") and stripped.endswith("|") + + def is_separator(line: str) -> bool: + stripped = line.strip() + if not is_table_line(stripped): + return False + return stripped.replace("|", "").replace("-", "").replace(":", "").strip() == "" + + def orphan_separator_lines(text: str) -> list[int]: + lines = text.splitlines() + offenders: list[int] = [] + for index, line in enumerate(lines): + if not is_separator(line): + continue + previous = lines[index - 1] if index else "" + next_line = lines[index + 1] if index + 1 < len(lines) else "" + if ( + not is_table_line(previous) + or is_separator(previous) + or not is_table_line(next_line) + or is_separator(next_line) + ): + offenders.append(index + 1) + return offenders + + if language == 'zh': + forbidden = [ + "parameter_sensitive", "source_path", "producer", "blocked", + "Applicable ", "| Yoga | Category | Strength | Combination | Effects / notes |", + "conjunction", "kalatra", "solar_yoga", "lunar_yoga", "durbhaga", + "Excessive sensuality", "Sharp intellect", "Dharidhra", "{planets}", + "长页解释", "True 普沙", "Results of ", "field_状态", + "Maha 大运", "Antar 大运", "Pratyantar 大运", "Effects of ", + "General effects during", "General effects which are felt", + "Interpretations based on the condition", + "Natural friends", "Natural enemies", "Temporary relation", + "Planet 1", "Planet 2", "Exact degree", "Actual difference", + "From house", "Target house", "Aspect type", "Applying", + "现实场域中阅读", "这一现实场域", + ] + else: + forbidden = [ + "parameter_sensitive", "source_path", "producer", "blocked", + "未单独列出", "长页解释", "True 普沙", "{planets}", + "dasha_beginning_dates", "dasha_ending_dates", "jd_ut:", "time_offset_seconds", + ":", ";", ",", "。", "、", "(", ")", "——", + "solar_yoga", "lunar_yoga", + ] + violations = [item for item in forbidden if item in markdown] + if language == 'en': + chinese_chars = len(re.findall(r"[\u4e00-\u9fff]", markdown)) + if chinese_chars: + violations.append(f"chinese_characters:{chinese_chars}") + language_receipt = language_quality_receipt(markdown, language=language) + violations.extend(language_receipt.get("violations") or []) + orphan_lines = orphan_separator_lines(markdown) + if orphan_lines: + violations.append(f"orphan_table_separators:{','.join(str(line) for line in orphan_lines[:12])}") + return { + "schema": "pl9.markdown_hygiene_receipt.v1", + "language": language, + "status": "pass" if not violations else "fail", + "violations": violations, + "language_quality": language_receipt, + } + + + +def _pl9_ai_density_visibility_receipt(markdown: str, *, language: str, + maturity_contract: dict[str, Any] | None = None) -> dict[str, Any]: + """Fail fast when mature PL9 AI-density page families disappear.""" + + if language == 'en': + required = { + "raw_positions": ("Birth Chart Planetary Raw Positions",), + "divisional_tables": ("Natal and Divisional North Indian Charts", "D1 Rashi Numeric Position Table"), + "technical_pages": ("Shadbala", "Ashtakavarga"), + "dasha_tables": ("Vimshottari Dasha", "Pratyantar Dasha"), + "saturn_kp": ("Sade Sati", "Dhayya", "KP"), + "annual_tables": ("Tajika", "without Dasha balance", "Sahams"), + "interpretive_notes": ("Maha Dasha", "Pratyantar Dasha"), + } + else: + required = { + "numeric_charts": ("D1 Rashi 数值位置表", "月亮盘", "特殊上升"), + "technical_pages": ("六维力量", "Ashtakavarga", "相位矩阵"), + "dasha_tables": ("Vimshottari AD", "Vimshottari PD", "Kala Chakra 大运"), + "saturn_kp": ("Sade Sati", "Dhayya", "KP"), + "annual_pages": ("年度图盘与 Tajika", "Mudda", "Patyayini", "Sahams"), + "long_form_interpretation": ("行星、星宿与宫位解释", "Bhavesh 宫主逐宫解释", "## 大运解释"), + } + missing = { + family: [marker for marker in markers if marker not in markdown] + for family, markers in required.items() + } + missing = {family: markers for family, markers in missing.items() if markers} + return { + "schema": "pl9.ai_density_visibility_receipt.v1", + "language": language, + "status": "pass" if not missing else "fail", + "missing": missing, + } + + + +def _render_pl9_ai_density_raw_data_markdown_en(packet: dict) -> str: + """Render the English AI/astrologer edition directly from structured payload.""" + + CN_TERMS_EN, COMBINATION_EN, EFFECTS_EN = _load_glossary() + calc_sudarshana_chakra = _load_sudarshana() + + english_field_explanations = { + "effect_on_confidence": ( + "Reduce confidence when functional and natural indications conflict." + ), + "confidence_impact": ( + "Confidence is reduced when functional and natural indications conflict." + ), + } + + def clean(value: Any) -> str: + if value is None: + return "-" + if isinstance(value, bool): + return "true" if value else "false" + if isinstance(value, float): + return f"{value:.6f}".rstrip("0").rstrip(".") + if isinstance(value, (int,)): + return str(value) + if isinstance(value, dict): + parts = [] + for key, item in list(value.items())[:8]: + key_text = clean(key) + if key_text in english_field_explanations: + value_text = english_field_explanations[key_text] + else: + value_text = clean(item) + parts.append(f"{key_text}={value_text}") + text = "; ".join(parts) + if len(value) > 8: + text += f"; ... +{len(value) - 8} fields" + return text[:480] + ("..." if len(text) > 480 else "") + if isinstance(value, (list, tuple, set)): + parts = [clean(item) for item in list(value)[:12]] + text = ", ".join(parts) + if len(value) > 12: + text += f", ... +{len(value) - 12} items" + return text[:480] + ("..." if len(text) > 480 else "") + text = str(value) + # 整串中文 effect 短词 -> glossary 直译(最高优先级,避免后续替换破坏语义) + _stripped = text.strip() + if _stripped in EFFECTS_EN: + return EFFECTS_EN[_stripped] + # 组合条件整句 -> 英文(逐词替换之前全串精确匹配,避免整句被拆坏) + if _stripped in COMBINATION_EN: + return COMBINATION_EN[_stripped] + # 中文行星/星座/结构词 -> 英文(combination 混排串逐词替换,避免删 CJK 后黏连)。 + # 按 key 长度降序替换:长词(宫主/上升主)先于短词(宫/主),避免被拆开覆盖。 + for _cn, _en in sorted(CN_TERMS_EN.items(), key=lambda kv: -len(kv[0])): + text = text.replace(_cn, _en) + replacements = { + ":": ": ", + ";": "; ", + ",": ", ", + "。": ".", + "、": ", ", + "(": "(", + ")": ")", + "——": ": ", + "{planets}": "listed planets", + "solar_yoga": "solar yoga", + "lunar_yoga": "lunar yoga", + "北交点": "Rahu", + "南交点": "Ketu", + "极友": "great friend", + "入友": "friendly", + "友好星座": "friendly sign", + "中性": "neutral", + "落陷": " debilitated", + "入庙": "own sign", + "中强": "medium-strong", + "强": " strong", + "弱": " weak", + "中": "medium", + } + for raw, public in replacements.items(): + text = text.replace(raw, public) + text = re.sub(r"(?:^|;\s*)-(?:\s*;\s*-){1,}\s*$", "-", text) + text = re.sub(r"\blisted planets(\d+)\b", r"\1 listed planet(s)", text) + text = text.replace("missing_in_local", "-") + text = text.replace("Dharidhra:", "Daridra combination:") + text = text.replace("D1/D9debilitated", "D1/D9 debilitated") + text = re.sub(r"^(\d+)weak/$", r"house \1 weak", text) + text = re.sub(r"^(\d+)/$", r"house \1 condition", text) + text = re.sub(r"^//,\s*", "", text) + text = re.sub(r"([A-Za-z]{3,})\1", r"\1", text) + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)(\d{1,2})\b", + r"\1 \2", + text, + flags=re.IGNORECASE, + ) + text = re.sub(r"\b(houses?|signs?|planets?)\s+\1\b", r"\1", text, flags=re.IGNORECASE) + # 先折叠连续重复的 condition(如 "condition condition"),再重排 + text = re.sub(r"\b(condition)(?:\s+condition)+\b", r"\1", text, flags=re.IGNORECASE) + text = re.sub( + r"\b(houses?|signs?|planets?)\s+([0-9/, -]+?)\s+condition\b", + r"condition \1 \2", + text, + flags=re.IGNORECASE, + ) + text = re.sub(r"[\u4e00-\u9fff]+", "", text) + # 黏连修复:行星 + dignity 词(如 Venusexalted -> Venus exalted) + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)(exalted|debilitated|strong|weak|moderate|medium|own|moolatrikona|combust)\b", + r"\1 \2", + text, + flags=re.IGNORECASE, + ) + # 黏连修复:小写词尾 + 大写词首(dignity/行星 + 星座,如 strongAquarius / SaturnAquarius) + text = re.sub(r"([a-z])([A-Z])", r"\1 \2", text) + # condition 重复去重(如 "condition condition") + text = re.sub(r"\b(condition)\s+\1\b", r"\1", text, flags=re.IGNORECASE) + # 移除 "=" 残留(Sade Sati 等 rashi 标签前缀,如 "=Capricorn" -> "Capricorn") + text = re.sub(r"=\s*(?=[A-Z])", "", text) + text = re.sub(r"\b([A-Za-z ]+)\(\1\)", r"\1", text) + text = re.sub(r"\s+", " ", text) + text = re.sub(r"\(\s*\)", "", text) + text = text.strip() or "-" + if not re.search(r"[A-Za-z0-9]", text): + # 删 CJK / 无英文等价后降级为 "-"(空占位),不输出 "Not available" 占位符。 + return "-" + return text[:480] + ("..." if len(text) > 480 else "") + + def clean_effect(value: Any) -> str: + """Translate one effect phrase; return "" when it has no English equivalent.""" + if not isinstance(value, str): + return clean(value) + stripped = value.strip() + translated = EFFECTS_EN.get(stripped) + if translated: + return translated + # 已是英文短语(无 CJK)时原样透传,避免被误当占位删掉。 + if re.search(r"[A-Za-z]", stripped) and not re.search(r"[\u4e00-\u9fff]", stripped): + return stripped + return "" + + def table(headers: list[str], rows: list[list[Any]]) -> str: + if not rows: + return "" + cleaned_rows = [[clean(cell) for cell in row] for row in rows] + empty_values = {"", "-", "None", "none", "null", "missing_in_local"} + filtered_rows: list[list[str]] = [] + for row in cleaned_rows: + non_empty = [cell for cell in row if cell not in empty_values] + if not non_empty: + continue + if row and row[0] in empty_values and len(non_empty) <= 1: + continue + if len(row) > 1 and row[0] not in empty_values and not any(cell not in empty_values for cell in row[1:]): + continue + filtered_rows.append(row) + cleaned_rows = filtered_rows + if not cleaned_rows: + return "" + keep_indices: list[int] = [] + for index, header in enumerate(headers): + column = [row[index] if index < len(row) else "-" for row in cleaned_rows] + non_empty = [cell for cell in column if cell not in empty_values] + if not non_empty: + continue + if header in {"Longitude", "House"} and len(non_empty) <= 1 and len(cleaned_rows) > 3: + continue + keep_indices.append(index) + headers = [headers[index] for index in keep_indices] + cleaned_rows = [[row[index] if index < len(row) else "-" for index in keep_indices] for row in cleaned_rows] + out = ["| " + " | ".join(headers) + " |", "|" + "|".join(["---"] * len(headers)) + "|"] + for row in cleaned_rows: + display_row = [] + for header, cell in zip(headers, row): + if cell == "-" and header in {"Effects / notes", "Summary", "Notes", "Detected / phase"}: + display_row.append("not listed") + else: + display_row.append(cell) + out.append("| " + " | ".join(display_row) + " |") + return "\n".join(out) + + def dict_get(path: tuple[str, ...], default: Any = None) -> Any: + obj: Any = packet + for key in path: + if not isinstance(obj, dict): + return default + obj = obj.get(key) + return obj if obj is not None else default + + def dict_get_any(*paths: tuple[str, ...], default: Any = None) -> Any: + for path in paths: + value = dict_get(path) + if value: + return value + return default + + lines: list[str] = [f"# {_pl9_english_report_title(packet)}", ""] + birth = dict_get(("birth_info",), {}) + if isinstance(birth, dict): + lines.extend([ + "## Birth Data and Calculation Profile", + "", + table(["Field", "Value"], [ + ["Name", birth.get("name") or birth.get("user_name")], + ["Birth date", birth.get("date")], + ["Birth time", birth.get("time")], + ["Latitude", birth.get("lat")], + ["Longitude", birth.get("lon")], + ["Time zone", f"UTC{birth.get('tz'):+}" if isinstance(birth.get("tz"), (int, float)) else birth.get("tz")], + ["Ayanamsa", birth.get("ayanamsa_display") or birth.get("ayanamsa_name") or birth.get("ayanamsa")], + ["Node mode", birth.get("node_mode")], + ]), + "", + ]) + + asc = dict_get(("core_chart", "ascendant"), {}) + planets = dict_get(("core_chart", "planets"), {}) + rows = [] + if isinstance(asc, dict): + rows.append(["Lagna", asc.get("sign"), asc.get("degree_in_sign"), asc.get("degree_raw") or asc.get("lon"), 1, "-", "-", "-", "-"]) + if isinstance(planets, dict): + for planet in ["Sun", "Moon", "Mars", "Mercury", "Jupiter", "Venus", "Saturn", "Rahu", "Ketu"]: + item = planets.get(planet, {}) + if isinstance(item, dict): + rows.append([ + planet, + item.get("sign"), + item.get("degree_in_sign"), + item.get("degree") or item.get("longitude"), + item.get("house"), + item.get("nakshatra"), + item.get("pada"), + item.get("status"), + item.get("speed"), + ]) + lines.extend(["## Core Chart Raw Positions", "", table(["Object", "Sign", "Degree in sign", "Longitude", "House", "Nakshatra", "Pada", "Status", "Speed"], rows), ""]) + + if isinstance(asc, dict) and isinstance(planets, dict): + try: + asc_lon = asc.get("longitude") + if asc_lon is None: + asc_lon = asc.get("lon") or asc.get("degree") or asc.get("degree_raw") + planet_lons = { + planet: item.get("longitude") if item.get("longitude") is not None else item.get("degree") + for planet, item in planets.items() + if isinstance(item, dict) and (item.get("longitude") is not None or item.get("degree") is not None) + } + if asc_lon is not None and planet_lons: + sudarshana = calc_sudarshana_chakra(planet_lons, float(asc_lon)) + reference_rows = [] + for key, value in (sudarshana.get("reference_points") or {}).items(): + if isinstance(value, dict): + reference_rows.append([key, value.get("sign"), value.get("role")]) + if reference_rows: + lines.extend(["## Sudarshana Chakra", "", table(["Reference point", "Sign", "Role"], reference_rows), ""]) + chart_rows = [] + for chart_name, chart in (sudarshana.get("three_charts") or {}).items(): + if not isinstance(chart, dict): + continue + for planet, item in chart.items(): + if isinstance(item, dict): + chart_rows.append([chart_name, planet, item.get("sign"), item.get("house")]) + if chart_rows: + lines.extend(["### Sudarshana Chakra Three-Reference Placements", "", table(["Reference chart", "Planet", "Sign", "House"], chart_rows), ""]) + except Exception: + pass + + d1_sheet = dict_get(("worksheets", "d1_rasi_bhava"), {}) + d1_houses = d1_sheet.get("houses") if isinstance(d1_sheet, dict) else {} + d1_planets = d1_sheet.get("planets") if isinstance(d1_sheet, dict) else {} + if not isinstance(d1_houses, dict): + d1_houses = dict_get(("core_chart", "houses"), {}) + if not isinstance(d1_planets, dict): + d1_planets = dict_get(("core_chart", "planets"), {}) + if isinstance(d1_houses, dict) and isinstance(d1_planets, dict) and d1_houses: + current_dasha = dict_get(("timing_and_predictive_systems", "dasha", "current_dasha"), {}) + active_lords: set[str] = set() + if isinstance(current_dasha, dict) and current_dasha.get("lord"): + active_lords.add(str(current_dasha.get("lord"))) + for ad_row in current_dasha.get("antardasha_timeline") or []: + if isinstance(ad_row, dict) and ad_row.get("is_current") and ad_row.get("lord"): + active_lords.add(str(ad_row.get("lord"))) + house_topics = { + 1: "self, body and life direction", + 2: "resources, speech and family values", + 3: "initiative, learning and communication", + 4: "home, residence and inner stability", + 5: "creativity, intelligence and judgment", + 6: "service, pressure and daily duties", + 7: "partnership, marriage and public interaction", + 8: "shared resources, risk, research and transformation", + 9: "belief, teachers, higher learning and long journeys", + 10: "career, responsibility and public role", + 11: "gains, networks and long-range aims", + 12: "retreat, expenditure, sleep, foreign places and release", + } + destination_topics = { + 1: "identity and bodily expression", + 2: "resources, speech and family continuity", + 3: "effort, skill and communication", + 4: "home, land and emotional foundation", + 5: "learning, creativity and merit", + 6: "workload, service, rivalry and repair", + 7: "partnership, clients and public dealings", + 8: "shared resources, research, vulnerability and deep change", + 9: "belief, study, teachers and distant movement", + 10: "career, authority and visible duty", + 11: "income, networks and fulfilment of goals", + 12: "expense, retreat, sleep, foreign links and release", + } + summary_rows = [] + narrative_rows: list[str] = [] + for house_index in range(1, 13): + house = d1_houses.get(f"house_{house_index}") or d1_houses.get(str(house_index)) or {} + if not isinstance(house, dict): + continue + lord = house.get("lord") + lord_row = d1_planets.get(str(lord)) if lord is not None else {} + if not isinstance(lord_row, dict): + lord_row = {} + sign = house.get("cusp_sign") or house.get("sign") + lord_sign = lord_row.get("sign") + lord_house = lord_row.get("house") + dignity = lord_row.get("status") or lord_row.get("dignity") + activation = "active Dasha layer" if str(lord) in active_lords else "natal background" + placement = f"{clean(lord_sign)} house {clean(lord_house)}" + summary_rows.append([house_index, sign, lord, placement, dignity, activation]) + topic = house_topics.get(house_index, f"house {house_index} topics") + destination = destination_topics.get(int(lord_house), f"house {clean(lord_house)} topics") if str(lord_house).isdigit() else "its placement field" + narrative_rows.extend([ + f"### House {house_index} lord: {clean(lord)}", + "", + f"House {house_index} covers {topic}. Its sign is {clean(sign)}, and its lord is {clean(lord)}.", + "", + f"The lord is placed in {placement}, with dignity/status {clean(dignity)}. This carries the house-{house_index} agenda into {destination}.", + "", + f"This placement is read with dignity, conjunctions, aspects, Shadbala, divisional repetition and Dasha timing. Current timing layer: {activation}.", + "", + ]) + if summary_rows: + lines.extend(["## Bhavesh House-Lord Interpretations", "", table(["House", "Sign", "Lord", "Lord placement", "Lord dignity", "Dasha activation"], summary_rows), ""]) + lines.extend(narrative_rows) + + moon_chart = dict_get(("divisional_and_special_charts", "moon_chart"), {}) + if isinstance(moon_chart, dict): + rows = [] + moon_planets = (moon_chart.get("raw") or {}).get("planets") if isinstance(moon_chart.get("raw"), dict) else moon_chart + for planet, item in (moon_planets or {}).items(): + if isinstance(item, dict): + rows.append([planet, item.get("sign"), item.get("degree") or item.get("longitude"), item.get("house_from_moon") or item.get("house")]) + lines.extend(["## Moon Chart", "", table(["Object", "Sign", "Longitude", "House"], rows), ""]) + + varga = dict_get(("divisional_and_special_charts", "varga_full"), {}) + if isinstance(varga, dict): + lines.extend(["## Divisional Charts Numeric Tables", ""]) + for chart_name, chart_payload in varga.items(): + positions = chart_payload.get("positions") if isinstance(chart_payload, dict) else None + if not isinstance(positions, dict): + positions = chart_payload if isinstance(chart_payload, dict) else None + rows = [] + if isinstance(positions, dict): + for obj_name, item in positions.items(): + if str(obj_name).startswith("_") or obj_name in {"planets"}: + continue + if isinstance(item, dict): + rows.append([obj_name, item.get("sign"), item.get("degree_in_sign"), item.get("longitude"), item.get("house")]) + if rows: + lines.extend([f"### {clean(chart_name)}", "", table(["Object", "Sign", "Degree in sign", "Longitude", "House"], rows), ""]) + + special_lagnas = dict_get(("divisional_and_special_charts", "special_lagnas"), {}) + if isinstance(special_lagnas, dict): + rows = [] + for name, item in special_lagnas.items(): + if isinstance(item, dict) and item.get("sign"): + rows.append([name, item.get("sign"), item.get("sign_degree") or item.get("degree"), item.get("degree"), item.get("house")]) + lines.extend(["## Special Lagnas", "", table(["Point", "Sign", "Degree in sign", "Longitude", "House"], rows), ""]) + + sensitive = dict_get(("divisional_and_special_charts", "sensitive_points"), {}) + if isinstance(sensitive, dict): + rows = [] + for name, item in sensitive.items(): + if isinstance(item, dict) and item: + rows.append([name, item.get("sign"), item.get("degree_in_sign"), item.get("longitude"), item.get("lord")]) + lines.extend(["## Sensitive Points", "", table(["Point", "Sign", "Degree in sign", "Longitude", "Lord"], rows), ""]) + + upagrahas = dict_get(("divisional_and_special_charts", "upagrahas", "raw"), {}) + if isinstance(upagrahas, dict): + rows = [] + sign_names = ["Aries", "Taurus", "Gemini", "Cancer", "Leo", "Virgo", "Libra", "Scorpio", "Sagittarius", "Capricorn", "Aquarius", "Pisces"] + for name, item in upagrahas.items(): + if isinstance(item, dict): + sign = item.get("sign") + if sign is None and isinstance(item.get("sign_idx"), int): + sign = sign_names[item["sign_idx"] % 12] + rows.append([name, sign, item.get("degree_in_sign"), item.get("longitude")]) + lines.extend(["## Upagraha and Sub-Planet Points", "", table(["Point", "Sign", "Degree in sign", "Longitude"], rows), ""]) + + strength_sheet = dict_get_any( + ("strengths_and_scores",), + ("worksheets", "strengths_and_scores"), + default={}, + ) + shadbala = {} + if isinstance(strength_sheet, dict): + shadbala = (strength_sheet.get("shadbala") or {}).get("planets") if isinstance(strength_sheet.get("shadbala"), dict) else {} + if isinstance(shadbala, dict): + rows = [] + for planet, item in shadbala.items(): + if isinstance(item, dict): + sthana = item.get("sthana_bala", {}) + kala = item.get("kala_bala", {}) + rows.append([planet, sthana.get("total"), item.get("dig_bala"), kala.get("total"), item.get("chesta_bala"), item.get("naisargika_bala"), item.get("drik_bala"), item.get("total_virupas")]) + lines.extend(["## Shadbala Components", "", table(["Planet", "Sthana", "Dig", "Kala", "Chesta", "Naisargika", "Drik", "Total"], rows), ""]) + + if isinstance(strength_sheet, dict): + functional = strength_sheet.get("functional_benefic_malefic") + if isinstance(functional, dict): + rows = [] + for key, value in functional.items(): + if key in english_field_explanations: + rows.append([key, english_field_explanations[key]]) + continue + if isinstance(value, (list, tuple)): + rows.append([key, ", ".join(map(clean, value))]) + elif isinstance(value, dict): + rows.append([key, "; ".join(f"{clean(k)}={clean(v)}" for k, v in value.items())]) + else: + rows.append([key, value]) + lines.extend(["## Functional Benefic / Malefic Layer", "", table(["Field", "Value"], rows), ""]) + + bhava_bala = strength_sheet.get("bhava_bala") + if isinstance(bhava_bala, dict): + bhava_house_rows = _load_bhava_rows() + rows = [] + for item in bhava_house_rows(bhava_bala): + rows.append([item['house'], item.get("total") if item.get("total") is not None else item.get("score"), item.get("strength") or item.get("grade"), item.get("notes")]) + lines.extend(["## Bhava Bala House Strength", "", table(["House", "Score", "Strength", "Notes"], rows), ""]) + + friendship = strength_sheet.get("planetary_friendship") + if isinstance(friendship, dict): + rows = [] + for planet, item in friendship.items(): + if isinstance(item, dict): + rows.append([planet, item.get("natural_friends"), item.get("natural_enemies"), item.get("temporary_relation") or item.get("compound_relationship")]) + lines.extend(["## Planetary Friendship Matrix", "", table(["Planet", "Natural friends", "Natural enemies", "Temporary relation"], rows), ""]) + + av = strength_sheet.get("ashtakavarga") if isinstance(strength_sheet, dict) else {} + if isinstance(av, dict): + sav = av.get("sav", {}) + if isinstance(sav, dict): + scores = sav.get("scores", {}) + rows = [[sign, score] for sign, score in scores.items()] if isinstance(scores, dict) else [] + lines.extend(["## Sarvashtakavarga", "", table(["Sign", "SAV"], rows), ""]) + bav = av.get("bav", {}) + if isinstance(bav, dict): + rows = [] + for planet, scores in bav.items(): + if isinstance(scores, dict): + bindus = scores.get("bindus") + if isinstance(bindus, list) and len(bindus) >= 12: + row = [planet] + bindus[:12] + [scores.get("total")] + else: + row = [planet] + [scores.get(sign) for sign in ["Aries", "Taurus", "Gemini", "Cancer", "Leo", "Virgo", "Libra", "Scorpio", "Sagittarius", "Capricorn", "Aquarius", "Pisces"]] + [scores.get("total")] + rows.append(row) + lines.extend(["## Bhinnashtakavarga", "", table(["Planet", "Ar", "Ta", "Ge", "Cn", "Le", "Vi", "Li", "Sc", "Sg", "Cp", "Aq", "Pi", "Total"], rows), ""]) + + kp = dict_get(("advanced_systems", "kp"), {}) + if isinstance(kp, dict): + planet_rows = [] + for planet, item in (kp.get("planets") or {}).items(): + if isinstance(item, dict): + lord = item.get("kp_lords", {}) + sig = item.get("significators", {}) + planet_rows.append([planet, lord.get("sign"), lord.get("nakshatra"), lord.get("rasi_lord"), lord.get("nakshatra_lord"), lord.get("sub_lord"), lord.get("sub_sub_lord"), sig.get("A"), sig.get("B"), sig.get("C"), sig.get("D")]) + house_rows = [] + for house, item in (kp.get("houses") or {}).items(): + if isinstance(item, dict): + lord = item.get("kp_lords", {}) + sig = item.get("significators", {}) + house_rows.append([house, item.get("sign"), item.get("cusp_longitude"), lord.get("nakshatra"), lord.get("rasi_lord"), lord.get("nakshatra_lord"), lord.get("sub_lord"), lord.get("sub_sub_lord"), sig.get("A"), sig.get("B"), sig.get("C"), sig.get("D")]) + lines.extend(["## KP Planet Significators", "", table(["Planet", "Sign", "Nakshatra", "Rasi lord", "Star lord", "Sub lord", "Sub-sub lord", "A", "B", "C", "D"], planet_rows), ""]) + lines.extend(["## KP House Cusps and Significators", "", table(["House", "Sign", "Cusp longitude", "Nakshatra", "Rasi lord", "Star lord", "Sub lord", "Sub-sub lord", "A", "B", "C", "D"], house_rows), ""]) + + dasha_families = dict_get(("timing_and_predictive_systems", "dasha_master_pack", "families"), {}) + if isinstance(dasha_families, dict): + lines.extend(["## Dasha Tables", ""]) + for family, payload in dasha_families.items(): + if not isinstance(payload, dict): + continue + periods = payload.get("periods") or payload.get("mahadasha") or [] + rows = [] + for item in periods[:80] if isinstance(periods, list) else []: + if isinstance(item, dict): + rows.append([item.get("lord") or item.get("planet"), item.get("rashi") or item.get("sign"), item.get("start") or item.get("start_date"), item.get("end") or item.get("end_date"), item.get("years") or (item.get("duration") or {}).get("years")]) + if rows: + lines.extend([f"### {clean(family).replace('_', ' ').title()}", "", table(["Lord", "Sign/Rashi", "Start", "End", "Years"], rows), ""]) + + annual_pack = dict_get_any( + ("worksheets", "timing_and_predictive_systems", "annual_tajika_pack"), + ("timing_and_predictive_systems", "annual_tajika_pack"), + default={}, + ) + tajika = dict_get(("timing_and_predictive_systems", "tajika"), {}) + if isinstance(tajika, dict) or isinstance(annual_pack, dict): + canonical_muntha = annual_pack.get("muntha") if isinstance(annual_pack, dict) else {} + canonical_year_lord = annual_pack.get("year_lord") if isinstance(annual_pack, dict) else {} + muntha = canonical_muntha if isinstance(canonical_muntha, dict) and canonical_muntha else tajika.get("muntha", {}) + year_lord = canonical_year_lord if isinstance(canonical_year_lord, dict) and canonical_year_lord else tajika.get("year_lord", {}) + if isinstance(muntha, dict) and muntha.get("status") == "conflict": + values = muntha.get("values") if isinstance(muntha.get("values"), list) else [] + muntha_signs = [ + item.get("value", {}).get("muntha_sign") + for item in values + if isinstance(item, dict) and isinstance(item.get("value"), dict) + ] + muntha_value = "Conflict: " + " vs ".join(str(value) for value in muntha_signs if value) + else: + muntha_data = muntha.get("data") if isinstance(muntha.get("data"), dict) else muntha + muntha_value = muntha_data.get("muntha_sign") or muntha_data.get("sign") + if isinstance(year_lord, dict) and year_lord.get("status") == "conflict": + values = year_lord.get("values") if isinstance(year_lord.get("values"), list) else [] + year_lords = [ + item.get("value", {}).get("year_lord") or item.get("value", {}).get("lord") + for item in values + if isinstance(item, dict) and isinstance(item.get("value"), dict) + ] + year_lord_value = "Conflict: " + " vs ".join(str(value) for value in year_lords if value) + else: + year_lord_data = year_lord.get("data") if isinstance(year_lord, dict) else {} + year_lord_value = ( + year_lord_data.get("year_lord") or year_lord_data.get("lord") + if isinstance(year_lord_data, dict) + else year_lord + ) + lines.extend(["## Annual Varshaphala Core Fields", "", table(["Field", "Value"], [ + ["Muntha sign", muntha_value], + ["Muntha lord", ( + (muntha.get("data") or {}).get("muntha_lord") + if isinstance(muntha, dict) and isinstance(muntha.get("data"), dict) + else (muntha.get("muntha_lord") if isinstance(muntha, dict) else None) + )], + ["Year lord", year_lord_value], + ]), ""]) + mudda = ( + annual_pack.get("mudda_dasha", {}) + if isinstance(annual_pack, dict) + else {} + ) + if not isinstance(mudda, dict) or not mudda: + mudda = tajika.get("mudda_dasha", {}) if isinstance(tajika, dict) else {} + if isinstance(mudda, dict): + rows = [[item.get("order"), item.get("lord"), item.get("months")] for item in mudda.get("dasha_sequence", []) if isinstance(item, dict)] + lines.extend(["## Mudda Dasha", "", table(["Order", "Lord", "Months"], rows), ""]) + + if isinstance(annual_pack, dict): + normalized = annual_pack.get("normalized_tables") if isinstance(annual_pack.get("normalized_tables"), dict) else {} + annual_table_specs = ( + ("mudda_rows", "Annual Mudda Dasha Rows", ("balance_profile", "lord", "start", "end", "boundary_status")), + ("patyayini_rows", "Annual Patyayini / Patyamsha Rows", ("main", "sub", "boundary", "boundary_status")), + ("saham_rows", "Annual Saham Rows", ("name", "saham", "sign", "degree_in_sign")), + ("tajika_yoga_rows", "Annual Tajika Yoga Rows", ("yoga", "name", "present", "forming_planets")), + ("annual_planet_rows", "Annual Planet Extended Fields", ("planet", "sign", "degree", "longitude", "nakshatra", "pada")), + ("annual_anchor_rows", "Annual Anchor Points", ("name", "label", "value", "sign", "degree", "house")), + ) + for key, title, preferred_keys in annual_table_specs: + rows_src = normalized.get(key) + if not isinstance(rows_src, list): + continue + rows_src = [row for row in rows_src if isinstance(row, dict)] + if not rows_src: + continue + keys = [field for field in preferred_keys if any(row.get(field) not in (None, "") for row in rows_src)] + if not keys: + keys = [ + field for field in rows_src[0] + if field not in {"source_path", "execution_status", "confidence_status", "boundary_status", "status", "selection_status"} + and not isinstance(rows_src[0].get(field), (dict, list, tuple)) + ][:6] + if not keys: + continue + lines.extend([f"## {title}", "", table([field.replace("_", " ").title() for field in keys], [[row.get(field) for field in keys] for row in rows_src]), ""]) + monthly_pack = annual_pack.get("monthly_chart_pack") if isinstance(annual_pack.get("monthly_chart_pack"), dict) else {} + monthly_rows = monthly_pack.get("rows") if isinstance(monthly_pack.get("rows"), list) else [] + if monthly_rows: + rows = [] + for row in monthly_rows: + if not isinstance(row, dict): + continue + asc = row.get("ascendant") if isinstance(row.get("ascendant"), dict) else {} + rows.append([row.get("month_index"), row.get("local_datetime"), asc.get("sign"), asc.get("degree_in_sign") or asc.get("degree")]) + lines.extend(["## Monthly Chart Pack", "", table(["Month", "Local time", "Annual ascendant", "Ascendant degree"], rows), ""]) + lifetime_pack = annual_pack.get("lifetime_annual_chart_pages") if isinstance(annual_pack.get("lifetime_annual_chart_pages"), dict) else {} + lifetime_pages = lifetime_pack.get("pages") if isinstance(lifetime_pack.get("pages"), list) else [] + lifetime_rows = [] + for page in lifetime_pages: + if not isinstance(page, dict): + continue + for row in page.get("rows") or []: + if not isinstance(row, dict): + continue + asc = row.get("ascendant") if isinstance(row.get("ascendant"), dict) else {} + lifetime_rows.append([row.get("completed_age"), row.get("target_year"), row.get("local_datetime"), asc.get("sign"), asc.get("degree_in_sign") or asc.get("degree")]) + if lifetime_rows: + lines.extend(["## Lifetime Annual Chart Pages", "", table(["Age", "Year", "Solar return time", "Annual ascendant", "Ascendant degree"], lifetime_rows), ""]) + series_pack = annual_pack.get("annual_chart_series_pack") if isinstance(annual_pack.get("annual_chart_series_pack"), dict) else {} + for series_key, title in ( + ("birth_place_eight_year_overview", "Birth-Place Eight-Year Annual Series"), + ("local_eight_year_overview", "Local Eight-Year Annual Series"), + ): + series = series_pack.get(series_key) if isinstance(series_pack.get(series_key), dict) else {} + series_rows = series.get("rows") if isinstance(series.get("rows"), list) else [] + rows = [ + [row.get("target_year"), row.get("age"), row.get("solar_return_local") or row.get("local_datetime"), row.get("annual_ascendant")] + for row in series_rows + if isinstance(row, dict) + ] + if rows: + lines.extend([f"## {title}", "", table(["Year", "Age", "Solar return time", "Annual ascendant"], rows), ""]) + + sahams = dict_get(("advanced_systems", "sahams"), {}) + if isinstance(sahams, dict): + rows = [] + for name, item in sahams.items(): + if name in {"daynight_evidence", "formula_traces"}: + continue + if isinstance(item, dict): + rows.append([name, item.get("sign"), item.get("degree_in_sign"), item.get("longitude")]) + lines.extend(["## Sahams", "", table(["Name", "Sign", "Degree in sign", "Longitude"], rows), ""]) + + yoga = dict_get(("advanced_systems", "yoga", "detected_yogas"), []) + if isinstance(yoga, list): + rows = [] + for item in yoga: + if isinstance(item, dict): + effects = [effect for effect in (clean_effect(e) for e in item.get("effects", [])) if effect] + effects_text = "; ".join(effects) if effects else "-" + rows.append([item.get("name"), item.get("category"), item.get("strength"), item.get("combination"), effects_text]) + lines.extend(["## Applicable Yogas", "", table(["Yoga", "Category", "Strength", "Combination", "Effects / notes"], rows), ""]) + + dosha = dict_get(("advanced_systems", "yogas_doshas"), {}) + if isinstance(dosha, dict): + rows = [] + dosha_labels = { + "mangal_dosha": "Mangala / Kuja Dosha", + "kaal_sarp_dosha": "Kaal Sarp Dosha", + "pitra_dosha": "Pitra Dosha", + "sade_sati": "Sade Sati", + } + for key in ["mangal_dosha", "kaal_sarp_dosha", "pitra_dosha", "sade_sati"]: + item = dosha.get(key) + if isinstance(item, dict): + rows.append([ + dosha_labels.get(key, key), + item.get("present") or item.get("detected") or item.get("status"), + item.get("summary") or item.get("phase") or item.get("type"), + ]) + lines.extend(["## Dosha and Transit Summary", "", table(["Topic", "Detected / phase", "Summary"], rows), ""]) + + text = "\n".join(line for line in lines if line is not None) + # Some legacy yoga payloads already contain duplicated labels. Normalize + # them after all sections have been assembled, including generated rows. + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)\1(?=\d|\b)", + r"\1 ", + text, + flags=re.IGNORECASE, + ) + text = re.sub(r"(?i)(condition)(houses?|signs?|planets?)", r"\1 \2", text) + text = re.sub( + r"\b(houses?|signs?|planets?)\s+\1\b", + r"\1", + text, + flags=re.IGNORECASE, + ) + text = re.sub(r"\n{3,}", "\n\n", text).strip() + "\n" + return text + + + +def _render_pl9_ai_density_raw_data_markdown_zh(packet: dict) -> str: + calc_sudarshana_chakra = _load_sudarshana() + """Render Chinese structured appendix from the same payload used by English.""" + + def clean(value: Any) -> str: + if value is None: + return "-" + if isinstance(value, bool): + return "是" if value else "否" + if isinstance(value, float): + return f"{value:.6f}".rstrip("0").rstrip(".") + if isinstance(value, int): + return str(value) + text = str(value).strip() + return _pl9_public_term(text) if text else "-" + + def table(headers: list[str], rows: list[list[Any]]) -> str: + if not rows: + return "" + out = ["| " + " | ".join(headers) + " |", "|" + "|".join(["---"] * len(headers)) + "|"] + for row in rows: + out.append("| " + " | ".join(clean(cell) for cell in row) + " |") + return "\n".join(out) + + def dict_get(path: tuple[str, ...], default: Any = None) -> Any: + obj: Any = packet + for key in path: + if not isinstance(obj, dict): + return default + obj = obj.get(key) + return obj if obj is not None else default + + lines: list[str] = [f"# {_pl9_public_report_title(packet)}", ""] + asc = dict_get(("core_chart", "ascendant"), {}) + planets = dict_get(("core_chart", "planets"), {}) + if isinstance(asc, dict) and isinstance(planets, dict): + try: + asc_lon = asc.get("longitude") + if asc_lon is None: + asc_lon = asc.get("lon") or asc.get("degree") or asc.get("degree_raw") + planet_lons = { + planet: item.get("longitude") if item.get("longitude") is not None else item.get("degree") + for planet, item in planets.items() + if isinstance(item, dict) and (item.get("longitude") is not None or item.get("degree") is not None) + } + if asc_lon is not None and planet_lons: + sudarshana = calc_sudarshana_chakra(planet_lons, float(asc_lon)) + reference_rows = [] + reference_labels = { + "ascendant_lagna": "出生上升参考点", + "moon_lagna": "月亮参考点", + "sun_lagna": "太阳参考点", + } + for key, value in (sudarshana.get("reference_points") or {}).items(): + if isinstance(value, dict): + reference_rows.append([reference_labels.get(key, key), value.get("sign"), value.get("role")]) + if reference_rows: + lines.extend(["## Sudarshana Chakra(三参考点盘)", "", table(["参考点", "星座", "角色"], reference_rows), ""]) + + chart_rows = [] + chart_labels = { + "ascendant_lagna": "从出生上升起算", + "moon_lagna": "从月亮起算", + "sun_lagna": "从太阳起算", + } + for chart_name, chart in (sudarshana.get("three_charts") or {}).items(): + if not isinstance(chart, dict): + continue + for planet, item in chart.items(): + if isinstance(item, dict): + chart_rows.append([chart_labels.get(chart_name, chart_name), planet, item.get("sign"), item.get("house")]) + if chart_rows: + lines.extend([ + "### Sudarshana Chakra 三参考点落宫表", + "", + table(["参考盘", "行星", "星座", "宫位"], chart_rows), + "", + ]) + except Exception: + pass + + text = "\n".join(line for line in lines if line is not None) + text = re.sub(r"\n{3,}", "\n\n", text).strip() + "\n" + return text + + + + +_SITE_BHAVA_SCORE = re.compile(r"\(Benefic\):\s*\+2|\(Malefic\):\s*-1\.5|吉星\s*\+2|凶星\s*-1\.5") +_NEECHA_RAJA = re.compile(r"Neecha Bhanga Raja Yoga|落陷取消王瑜伽|坏座位消王") + + +def _apply_local_report_conventions(markdown: str) -> str: + """Keep this site's report wording inside the upstream appendix.""" + text = _NEECHA_RAJA.sub("落陷取消(Neecha Bhanga)", markdown) + text = _SITE_BHAVA_SCORE.sub("", text) + return text + + +def render_full_data_markdown(packet: dict) -> str: + """Upstream `pl9_ai_density` branch at 23be1807, on this site's parity body.""" + if _pl9_report_language(packet) == "en": + structured_appendix = re.sub( + r"^#\s+.*?(?:\n{2,}|\n)", + "", + _render_pl9_ai_density_raw_data_markdown_en(packet), + count=1, + flags=re.DOTALL, + ).strip() + english_markdown = render_pl9_parity_markdown(packet) + if structured_appendix: + english_markdown = ( + english_markdown.rstrip() + + "\n\n## Structured Data Appendix\n\n" + + "
    \nMachine-readable source fields (raw evidence)\n\n" + + structured_appendix + + "\n\n
    \n" + ) + rendered = _sanitize_pl9_ai_density_markdown_en( + english_markdown, + _pl9_english_report_title(packet), + ) + return _apply_local_report_conventions(rendered) + zh_structured_markdown = _sanitize_pl9_ai_density_markdown( + _render_pl9_ai_density_raw_data_markdown_en(packet), + _pl9_public_report_title(packet), + ) + zh_structured_appendix = re.sub( + r"^#\s+.*?(?:\n{2,}|\n)", + "", + zh_structured_markdown, + count=1, + flags=re.DOTALL, + ).strip() + zh_markdown = render_pl9_parity_markdown(packet) + if zh_structured_appendix and "## 结构化资料附录" not in zh_markdown: + zh_markdown = ( + zh_markdown.rstrip() + + "\n\n## 结构化资料附录\n\n" + + "
    \n机器可读源字段(原始证据)\n\n" + + zh_structured_appendix + + "\n\n
    \n" + ) + rendered = _sanitize_pl9_ai_density_markdown(zh_markdown, _pl9_public_report_title(packet)) + return _apply_local_report_conventions(rendered) + + +def read_full_data_domain(packet: Mapping, domain: str) -> dict: + """Deep-copied packet view for one consult domain. + + Field selection belongs to the consult card. This entry only freezes the + domain name and detaches the copy from the live packet. + """ + if domain not in FULL_DATA_DOMAINS: + raise KeyError(domain) + if not isinstance(packet, Mapping): + raise TypeError("packet must be a mapping") + return copy.deepcopy({ + "domain": domain, + "label": FULL_DATA_DOMAINS[domain], + "report_version": packet.get("report_version"), + "birth_info": packet.get("birth_info"), + "core_chart": packet.get("core_chart"), + "calculation_profile": packet.get("calculation_profile"), + "worksheets": packet.get("worksheets"), + }) diff --git a/scripts/pl9_language_terms.py b/scripts/pl9_language_terms.py new file mode 100644 index 00000000..1310c46e --- /dev/null +++ b/scripts/pl9_language_terms.py @@ -0,0 +1,669 @@ +#!/usr/bin/env python3 +"""PL9 report language normalization and leakage gates. + +The term policy is based on this repository's StarTrack language-bridge +boundary and the Jyotish wording guidance in references/modern-language-guide.md. +It only changes reader-facing labels; it does not alter calculation payloads. +""" +from __future__ import annotations + +import re +from typing import Any + + +ZH_TERM_REPLACEMENTS: tuple[tuple[str, str], ...] = ( + ("Natural friends", "天然友星"), + ("Natural enemies", "天然敌星"), + ("Temporary relation", "临时关系"), + ("Planet 1", "行星1"), + ("Planet 2", "行星2"), + ("Planet", "行星"), + ("planet", "行星"), + ("sign", "星座"), + ("longitude", "黄经"), + ("field", "字段"), + ("value", "数值"), + ("title", "标题"), + ("target", "目标"), + ("anchor", "依据"), + ("yoga", "瑜伽"), + ("dt_ut", "UTC 时间"), + ("dt_local", "本地时间"), + ("sun_lon", "太阳黄经"), + ("year_lord", "年主星"), + ("field_status", "字段状态"), + ("muntha_house", "Muntha 宫位"), + ("muntha_sign", "Muntha 星座"), + ("muntha_lord", "Muntha 主星"), + ("munthesh", "Muntha 宫主"), + ("Nakshatra Scheme", "星宿身体对应表"), + ("Opinion 1", "说法一"), + ("Opinion 2", "说法二"), + ("Graha Avastha", "Graha Avastha(行星状态)"), + ("Jagradadi", "醒睡状态"), + ("Baladi", "年龄状态"), + ("Lajjitadi", "羞惭等状态"), + ("Deeptadi", "明亮等状态"), + ("Shyanadi", "卧姿等状态"), + ("Shayanadi", "卧姿等状态"), + ("Degree", "落座度数"), + ("Declination", "赤纬"), + ("Speed", "速度"), + ("House", "宫位"), + ("Sign", "星座"), + ("Lord", "宫主"), + ("Score", "分数"), + ("Bala", "年龄状态"), + ("Jagrat", "醒睡状态"), + ("Aspect", "相位"), + ("Exact degree", "精确角度"), + ("Actual difference", "实际角距"), + ("Orb", "容许度"), + ("Applying", "入相"), + ("From house", "起始宫位"), + ("Target house", "目标宫位"), + ("Aspect type", "相位类型"), + ("Special", "特殊相位"), + ("Category", "类别"), + ("Strength", "强度"), + ("Combination", "组合条件"), + ("Effects / notes", "作用说明"), + ("friends:", "友星:"), + ("enemies:", "敌星:"), + ("True", "是"), + ("False", "否"), + ("conjunction", "合相"), + ("opposition", "对冲"), + ("special", "特殊组合"), + ("solar_yoga", "太阳瑜伽"), + ("lunar_yoga", "月亮瑜伽"), + ("kalatra", "婚恋"), + ("durbhaga", "不利组合"), + ("moderate", "中等"), + ("common", "常见"), + ("strong", "强"), + ("weak", "弱"), + ("not separately listed", "未单独列出"), + ("ownership", "守护宫"), + ("dignity", "尊贵状态"), + ("house placement", "落宫"), + ("natural significations", "自然象征"), + ("functional role", "功能角色"), + ("Graha Avasthas - Planets and their Moods", "Graha Avastha(行星状态与情绪)"), + ("Mangala / Sade Sati / Dosha Results", "火星婚姻煞 / Sade Sati / Dosha 结果"), + ("Mangal / Kuja Dosha", "火星婚姻煞(Mangala/Kuja Dosha)"), + ("Poorvashadha", "前阿沙陀"), + ("Poorva Ashadha", "前阿沙陀"), + ("Uttara Phalg.", "后破伽"), + ("Uttara Phalguni", "后破伽"), + ("Uttarashadha", "后阿沙陀"), + ("Uttara Ashadha", "后阿沙陀"), + ("Uttarabhadra", "后跋陀罗"), + ("Uttara Bhadrapada", "后跋陀罗"), + ("Moola", "Mula(根、本源)"), + ("Mula", "Mula(根、本源)"), + ("Both thighs", "双侧大腿"), + ("Both feet", "双脚"), + ("Private parts", "私密部位"), + ("Sides of body", "身体两侧"), + ("Back", "背部"), + ("Left side", "左侧"), + ("Left hand", "左手"), + ("Waist", "腰部"), + ("Waiste", "腰部"), + ("Shins", "小腿胫部"), + ("Swapna", "梦眠"), + ("Dreamful", "多梦"), + ("Mrita", "死寂"), + ("State of death", "死寂状态"), + ("Sushupti", "熟睡"), + ("State of sleep", "睡眠状态"), + ("Jagrad", "觉醒"), + ("Wakefulness", "清醒"), + ("Kumaravastha", "少年期"), + ("Adolescence", "青春期"), + ("Balavastha", "童年期"), + ("Childhood", "童年"), + ("Vriddha", "老年期"), + ("Old age", "老年"), + ("Mudit Kshobit", "喜悦中带扰动"), + ("Trushit Mudit", "渴求中带喜悦"), + ("Kshudit Trushit", "饥渴不安"), + ("Mudit", "喜悦"), + ("Khala", "粗劣"), + ("Mudita", "愉悦"), + ("Delighted", "愉悦"), + ("Shanta", "平静"), + ("Quiescent", "安静"), + ("Deena", "匮乏"), + ("Deficient", "不足"), + ("Swastha", "稳定"), + ("Stable", "稳定"), + ("Nidra", "睡眠"), + ("Sleep", "睡眠"), + ("Gamenecchha", "欲行"), + ("Eager to go", "急于行动"), + ("Shayana", "卧躺"), + ("Recumbent", "卧躺"), + ("Sabhayam Vasti", "集会中"), + ("in an assembly", "在集会中"), + ("Infant", "婴幼期"), + ("Young", "青年期"), + ("Youth", "壮年期"), + ("Dead", "死寂"), + ("Awake", "觉醒"), + ("Dreaming", "梦眠"), + ("Trushita", "渴求"), + ("Kshobhita", "扰动"), + ("Vikala", "失衡"), + ("Kautuka", "好奇"), + ("Agama", "学习/趋近"), + ("Panapara", "续宫"), + ("Apoklima", "果宫"), + ("Kendra", "角宫"), + ("Benefic", "自然吉星"), + ("Malefic", "自然凶星"), +) + + +ZH_ALLOWED_LATIN_TOKENS = { + "AI", "AD", "PD", "MD", "PL9", "KP", "BPHS", "PVR", "D1", "D2", "D3", "D4", + "D5", "D6", "D7", "D8", "D9", "D10", "D11", "D12", "D16", "D20", "D24", + "D27", "D30", "D40", "D45", "D60", "Rashi", "Navamsha", "Bhava", "Sripati", + "Sudarshan", "Sudarshana", "Upagraha", "Lagna", "Arudha", "Upapada", "Pada", + "Dasha", "Vimshottari", "Ashtottari", "Yogini", "Kala", "Chakra", "Jaimini", + "Narayana", "Sthira", "Drig", "Shoola", "Tribhagi", "Sade", "Sati", "Dhayya", + "Kantaka", "Varshaphala", "Tajika", "Mudda", "Patyayini", "Saham", "Sahams", + "Shadbala", "Ashtakavarga", "BAV", "SAV", "Pinda", "Avastha", "Vimsopaka", + "Paravatamsa", "Simhasanamsa", "Rahu", "Ketu", "Sun", "Moon", "Mars", + "Mercury", "Jupiter", "Venus", "Saturn", "Ashwini", "Bharani", "Krittika", + "Virupa", "Sthana", "Dig", "Drik", "Chesta", + "Rohini", "Mrigashira", "Ardra", "Punarvasu", "Pushya", "Ashlesha", "Magha", + "Purva", "Uttara", "Phalguni", "Hasta", "Chitra", "Swati", "Vishakha", + "Anuradha", "Jyeshtha", "Mula", "Ashadha", "Shravana", "Dhanishta", + "Shatabhisha", "Bhadrapada", "Revati", + "benefic", "malefic", "debilitated", "Neecha", "Bhanga", "and", + "functional_benefic", "functional_malefic", "functional_neutral", + "natural_benefic", "natural_malefic", "natural_role_not_returned", + "functional_role_not_returned", +} + + +ZH_FORBIDDEN_MIXED_PHRASES = ( + "General effects during", + "General effects which are felt", + "Interpretation of the", + "Effects of", + "Interpretations based on the condition", + "Supported houses gain", + "The immediate focus is", + "brings house", + "is read from house", + "Maha 大运", + "Antar 大运", + "Pratyantar 大运", + "现实场域中阅读", + "这一现实场域", +) + + +EN_FORBIDDEN_CHARS = (":", ";", ",", "。", "、", "(", ")", "——") + + +_ZH_REGEX_REPLACEMENTS = tuple( + (raw, public) + for raw, public in ZH_TERM_REPLACEMENTS + if re.fullmatch(r"[A-Za-z0-9_ /:-]+", raw) +) +_ZH_LITERAL_REPLACEMENTS = tuple( + (raw, public) + for raw, public in ZH_TERM_REPLACEMENTS + if not re.fullmatch(r"[A-Za-z0-9_ /:-]+", raw) +) +_ZH_REGEX_REPLACEMENT_MAP = {raw: public for raw, public in _ZH_REGEX_REPLACEMENTS} +_ZH_REGEX_REPLACEMENT_PATTERN = re.compile( + r"(? str: + return _ZH_REGEX_REPLACEMENT_PATTERN.sub( + lambda match: _ZH_REGEX_REPLACEMENT_MAP[match.group(1)], + markdown, + ) + + +def normalize_zh_markdown_terms(markdown: str) -> str: + text = _replace_zh_regex_terms(markdown) + for raw, public in _ZH_LITERAL_REPLACEMENTS: + text = text.replace(raw, public) + text = text.replace("e特殊组合ly", "especially") + # Repair field-label substitutions that crossed token boundaries in the + # previous customer pin. These are presentation-only fixes; chart values + # and source evidence remain unchanged. + text = text.replace("强est_第", "最强宫") + text = text.replace("弱est_第", "最弱宫") + text = text.replace("strongest_house", "最强宫") + text = text.replace("weakest_house", "最弱宫") + text = text.replace("7 visible planets occupy exactly 5 distinct signs", "七颗可见行星恰好分布在五个不同星座") + text = text.replace("Many friends, talkative, skilled in various arts", "人际接触较多,善于表达,具多样艺术或技能倾向") + text = text.replace("Father died before birth, ancestral curse", "传统规则提示:涉及父系与家族议题,不能据此判断现实经历") + text = text.replace("Skilled in fine arts, music, dance; cultured and wealthy", "擅长艺术、音乐或舞蹈;重视文化修养与资源积累") + text = text.replace("Ridiculed by others, mocked, subject to derision", "可能面临误解、嘲讽或评价压力;需结合现实处境核验") + text = text.replace("Deception, distrust, household/family complications", "信任、家庭关系或居住事务可能较复杂;需结合现实处境核验") + text = text.replace("rare", "罕见") + text = text.replace("relationship_observation", "关系观察") + text = text.replace("solar 瑜伽", "太阳瑜伽") + text = text.replace("lunar 瑜伽", "月亮瑜伽") + text = re.sub(r"(\d+)\s+第from\s+月亮", r"从月亮起第\1宫", text) + text = re.sub(r"\bin\s+(从月亮起第\d+宫)", r"位于\1", text) + text = text.replace("_aspect_", "宫相位_") + text = _normalize_zh_bhava_factor_text(text) + text = text.replace("Bhava 年龄状态 十二宫力量", "Bhava Bala 十二宫力量") + text = text.replace("Shadbala / Bhava 年龄状态", "Shadbala / Bhava Bala") + text = text.replace("星宿 Scheme", "星宿身体对应表") + text = _normalize_mula_gloss(_restore_zh_nakshatra_glosses(_collapse_duplicate_zh_parentheses(text))) + text = text.replace("### 是 Solar Return", "### 真实太阳返照") + text = re.sub( + r"General effects during the 主大运 of (.+?) are read from the 行星's 自然象征, 落宫, owned houses, 尊贵状态, 星宿, and 功能角色\.", + r"\1主大运的总体作用,需要从该行星的自然象征、落宫、守护宫、尊贵状态、星宿与功能角色一起阅读。", + text, + ) + text = text.replace( + "Interpretations based on the condition of the 行星 in the birth chart and divisional charts are as follows:", + "结合本命盘与分盘条件后,可按以下方式细读:", + ) + text = re.sub( + r"(.+?)落在(.+?),首先把(.+?)放到(.+?)这一现实场域中阅读。", + r"\1落在\2,表示\3会主要通过\4来表现。", + text, + ) + text = re.sub(r"第([1-4])足", r"第\1 Pada(星宿四分区)", text) + text = text.replace("Pada | 速度", "Pada(星宿四分区) | 速度") + text = _normalize_zh_dasha_prose(text) + text = _thicken_zh_dasha_density(text) + text = _normalize_mula_gloss(_restore_zh_nakshatra_glosses(_collapse_duplicate_zh_parentheses(text))) + return text + + +def _normalize_zh_bhava_factor_text(markdown: str) -> str: + """Localize compact Bhava Bala factor formulas in Chinese reports.""" + + text = markdown + body = r"(?:Sun|Moon|Mars|Mercury|Jupiter|Venus|Saturn|Rahu|Ketu|太阳|月亮|火星|水星|木星|金星|土星|北交点|南交点)" + text = re.sub(rf"({body})\s+in\s+H(\d+)", r"\1位于第\2宫", text) + text = re.sub(rf"({body})\s+in\s+(角宫|续宫|果宫)\s+\(H(\d+)\)", r"\1位于\2第\3宫", text) + text = re.sub(rf"({body})\s+aspects\s+H(\d+)\s+\((\d+)(?:st|nd|rd|th),\s+(自然吉星|自然凶星)\)", r"\1以\3宫相位照第\2宫(\4)", text) + text = re.sub(rf"({body})\s+aspects\s+H(\d+)\s+\((\d+)(?:st|nd|rd|th)\)", r"\1以\3宫相位照第\2宫", text) + text = re.sub(r"\((自然吉星|自然凶星)\)", r"(\1)", text) + text = text.replace("火星宫相位_4", "火星特殊4宫相位") + text = text.replace("火星宫相位_8", "火星特殊8宫相位") + text = text.replace("木星宫相位_5", "木星特殊5宫相位") + text = text.replace("木星宫相位_9", "木星特殊9宫相位") + text = text.replace("土星宫相位_3", "土星特殊3宫相位") + text = text.replace("土星宫相位_10", "土星特殊10宫相位") + return text + + +def _normalize_mula_gloss(markdown: str) -> str: + """Make the Mula gloss idempotent across repeated language passes.""" + + text = markdown + root_gloss = r"根[,、/]本源" + text = re.sub(rf"Mula(?:[((]{root_gloss}[))])+", "Mula(根、本源)", text) + text = re.sub(rf"根宿[((]Mula[((]{root_gloss}[))][))]", "Mula(根、本源)", text) + text = re.sub(rf"根宿[((](Mula(?:[((]{root_gloss}[))])+)[))]", "Mula(根、本源)", text) + return text + + +def _collapse_duplicate_zh_parentheses(markdown: str) -> str: + """Remove duplicated Chinese glosses such as 后破伽(后破伽).""" + + return re.sub(r"([\u4e00-\u9fff][\u4e00-\u9fff·/-]{0,12})[((]\1[))]", r"\1", markdown) + + +def _restore_zh_nakshatra_glosses(markdown: str) -> str: + """Keep the original Sanskrit/English nakshatra label in Chinese reports.""" + + return _ZH_NAKSHATRA_GLOSS_PATTERN.sub( + lambda match: f"{match.group(1)}({_ZH_NAKSHATRA_GLOSSES[match.group(1)]})", + markdown, + ) + + +def _thicken_zh_dasha_density(markdown: str) -> str: + """Restore reader-facing density for translated AD/PD prose. + + This adds reading instructions tied to already-present fields. It does not + introduce new predictions or alter dates. + """ + lines = markdown.splitlines() + out: list[str] = [] + for index, line in enumerate(lines): + out.append(line) + lookahead = "\n".join(lines[index + 1:index + 6]) + if "这一子运不是单独结论" in lookahead or "这个次子运只承担短周期细化" in lookahead: + continue + ad_match = re.search( + r"(.+?)主大运中的(.+?)子运,会把(.+?)的自然主题带入当前阶段。", + line, + ) + if ad_match: + md_lord, ad_lord, theme_lord = ad_match.groups() + out.extend([ + "", + ( + f"这一子运不是单独结论,而是把{theme_lord}的自然象征放进{md_lord}主大运的背景里筛选。" + f"阅读时先看{ad_lord}本身的落宫、守护宫、尊贵状态与星宿,再看它和主大运星之间是否互相支持。" + ), + "", + ( + "若同一宫位或同一行星在分盘、Ashtakavarga、年度盘和行运中重复出现," + "该主题的可见度会提高;若这些层彼此冲突,则应把结果视为阶段性倾向,而不是孤立断语。" + ), + ]) + continue + pd_match = re.search( + r"(.+?)子运中的(.+?)次子运,会在(.+?)背景下呈现(.+?)的短周期结果。", + line, + ) + if pd_match: + ad_lord, pd_lord, md_lord, theme_lord = pd_match.groups() + out.extend([ + "", + ( + f"这个次子运只承担短周期细化:{pd_lord}会把{theme_lord}的具体征象带入{ad_lord}子运," + f"并受{md_lord}主背景限制。" + ), + "", + ( + "因此这里优先用于判断事情推进的节奏、触发点和轻重缓急;实际领域仍需回到本命落宫、" + "分盘重复、年度盘和当前行运共同核对。" + ), + ]) + continue + return "\n".join(out) + + +def _normalize_zh_dasha_prose(markdown: str) -> str: + text = markdown + replacements = { + "mind, mother, water, residence, 迁移与出行, public mood, nourishment, fertility, trade, learning, and changeable fortune become active.": "心智、母亲、居住、水象事务、迁移与出行、公众情绪、滋养、生育、贸易、学习与变化中的运势会被启动。", + "There can be interest in mantra, teachers, sacred learning, art, hospitality, garments, ornaments, land, and watery products.": "这一阶段可能增加对咒语、师长、神圣知识、艺术、服务接待、衣物、饰品、土地与水相关事务的兴趣。", + "The mind can become lively, sensitive, restless, affectionate, and responsive to family or social approval.": "心绪可能更活跃、敏感、容易波动,也更重视家庭回应与社会认可。", + "When supported, it brings comfort from home, spouse, children, servants, conveyances, food, clothing, education, fame, and fulfilled desires.": "条件良好时,可带来家庭、伴侣、子女、协助者、交通工具、饮食衣物、教育、名声与愿望满足方面的支持。", + "New places, cultivation, trade, and public-facing work may become profitable.": "新地点、耕作/经营、贸易与面向公众的工作可能带来收益。", + "When afflicted, it can show fear, anger, wavering judgment, family strain, maternal concern, contaminated food, 迁移与出行 pressure, and loss of old security.": "受克时,可能表现为恐惧、怒气、判断摇摆、家庭压力、母亲相关担忧、饮食不洁、迁移压力,以及旧有安全感减弱。", + "Friends may not give full support, and emotional decisions need steadier review.": "朋友支持可能不足,情绪化决定需要更稳妥地复核。", + "absolute Virupas; precise Sthana/Dig/Kala/Drik; bounded BPHS Chesta": "以绝对 Virupa 计分,包含 Sthana、Dig、Kala、Drik 与受限 BPHS Chesta", + "fear, anger, wavering judgment, family strain, maternal concern, contaminated food, 迁移与出行 pressure, and loss of old security": "恐惧、怒气、判断摇摆、家庭压力、母亲相关担忧、饮食不洁、迁移压力,以及旧有安全感减弱", + "unusual openings, foreign benefit, technical or political leverage, and gains through nontraditional channels": "特殊机会、海外或异地收益、技术或权力杠杆,以及非传统渠道带来的收获", + "fear, bondage, deception, sudden reversals, illness, scandals, and trouble from authorities or hidden enemies": "恐惧、束缚、欺骗、突发反转、疾病、名誉风波,以及来自权威或暗中对手的麻烦", + "promotion, respect, wealth, grains, clothes, gold, children, good counsel, and fulfillment of aims": "晋升、尊重、财富、物资、衣物、贵金属、子女、良好建议与目标达成", + "trouble to spouse or children, loss through indulgence, legal worries, fever, or grief from family obligations": "伴侣或子女方面的麻烦、因放纵带来的损耗、法律忧虑、发热或家庭责任带来的忧伤", + "durable property, work stability, public recognition, gain through labor, and capacity to endure responsibility": "稳定资产、工作稳定、公众认可、辛勤劳动带来的收益,以及承担责任的耐力", + "pain, quarrels, obstruction, loss of wealth, separation from parents, danger through low morale, and heavy duties": "痛苦、争执、阻碍、财富损耗、与父母分离、士气低落带来的风险,以及沉重职责", + "knowledge, wealth, garments, jewels, name, ritual merit, useful alliances, and clever problem solving": "知识、财富、衣物、珠宝、名声、仪式功德、有用联盟与灵活的问题解决能力", + "poor judgment, fever, anxiety, argument, loss through paperwork, and instability in business": "判断不稳、发热、焦虑、争论、文书损耗与商业不稳定", + "renunciation, spiritual practice, diagnostic skill, hidden knowledge, and the ability to cut through confusion": "放下执着、灵性修持、诊断能力、隐秘知识,以及切断混乱的能力", + "fear from enemies, loss of reasoning, sudden obstruction, disease, grief, and restless 迁移与出行": "来自对手的恐惧、理性受损、突发阻碍、疾病、忧伤与不安定的迁移出行", + "spouse happiness, property, conveyance, ornaments, fine food, friendship, and pleasant company": "伴侣愉悦、资产、交通工具、饰品、美食、友谊与愉快陪伴", + "indulgence, waste, disputes with women or partners, loss through luxury, and ailments connected with reproductive or urinary balance": "放纵、浪费、与女性或伴侣的争执、奢侈带来的损耗,以及生殖或泌尿平衡相关不适", + "recognition, administrative help, confidence, vehicles, ornaments, and respected duties": "认可、行政助力、自信、交通工具、饰品与受尊重的职责", + "pressure from government, rivals, fire, theft, father-related strain, feverish conditions, and restlessness": "来自政府/权威、竞争者、火灾、盗损、父亲相关压力、发热状态与内在不安的压力", + "courage, stamina, property gains, productive labor, victory over enemies, and technical execution": "勇气、体力、资产收益、有效劳动、战胜对手与技术执行力", + "harsh speech, weapon or fire trouble, stomach ailments, injuries, litigation, and hot-tempered separations": "言语尖锐、武器或火相关麻烦、胃部不适、伤损、诉讼,以及急躁导致的分离", + "stamina, property, gains, productive labor, victory over enemies": "体力、资产、收益、有效劳动和战胜对手", + "foreign places, outsiders, ambition, unusual gains, fear, poisons, snakes, politics, sudden turns, obsession, and unconventional paths become prominent.": "异地、外来者、野心、特殊收益、恐惧、毒性/中毒象征、蛇象、政治、突发转折、执念与非传统路径会变得突出。", + "The period can bring success in ventures, conveyances, garments, distant direction gains, and contact with powerful or foreign circles.": "这一时期可能带来事业尝试、交通工具、衣物、远方收益,以及与有权势或海外/异地圈层接触方面的机会。", + "If afflicted, it brings defamation, business loss, fever, enemies, anxiety, spouse or child distress, and loss of reputation.": "若受克,可能带来名誉受损、事业损失、发热、对手、焦虑、伴侣或子女方面的压力,以及声望下降。", + "It favors strategic risk only when facts and ethics are kept clear.": "只有事实清楚、边界正当时,策略性冒险才更有利。", + "北交点/罗睺-related mantra, restraint, and charity are traditional remedial themes.": "与北交点/罗睺相关的咒语、克制与布施,是传统辅助主题。", + "labor, delay, servants, masses, endurance, old people, land, minerals, iron, chronic pressure, discipline, and separation themes become prominent.": "劳动、延迟、服务者、大众、耐力、长者、土地、矿物、铁器、长期压力、纪律与分离主题会变得突出。", + "The period can give property, recognition, steady gains, service authority, and patient achievement after effort.": "这一时期可能带来资产、认可、稳定收益、服务型权责,以及努力之后的耐心成果。", + "If afflicted, it brings loss of friends, disputes with relatives, fear, confinement, fatigue, rheumatic or digestive trouble, and wandering.": "若受克,可能带来朋友损失、亲属争执、恐惧、受限感、疲惫、风湿或消化问题,以及漂泊。", + "The middle of the period can be more productive than its beginning or end.": "这一时期的中段往往比开始和结束阶段更容易产生成果。", + "Traditional balancing themes include service, discipline, humility, and 土星-related charity.": "传统平衡主题包括服务、纪律、谦逊,以及与土星相关的布施。", + "The period can detach the person from stale ambitions and force a simpler, more inward path.": "这一时期可能使人脱离旧有野心,被迫走向更简单、更内向的路径。", + "If afflicted, business disturbance, loss of wealth, stomach or eye trouble, fear, conflict, and news of death or separation can appear.": "若受克,可能出现事业扰动、财富损耗、胃部或眼部问题、恐惧、冲突,以及死亡或分离相关消息。", + "It can produce intermittent gains after pressure or 迁移与出行.": "它可能在压力或迁移之后带来间歇性收益。", + "Durga worship, protective mantra, and charity are traditional balancing themes.": "杜尔迦崇拜、保护性咒语与布施,是传统平衡主题。", + "authority, government, 状态, father, medicine, land, fire, command, visibility, and public responsibility become more prominent.": "权威、政府、地位、父亲、医疗、土地、火象事务、指挥权、能见度与公共责任会更突出。", + "spiritual discipline, mantra, ritual, leadership, and contact with officials or influential people can increase.": "灵性纪律、咒语、仪式、领导力,以及与官员或有影响力人士的接触可能增加。", + "anxiety, heat, separation from close relatives, disputes with authority, eye, teeth, abdomen, or vitality concerns can also need attention.": "焦虑、热性问题、与近亲分离、和权威争执,以及眼睛、牙齿、腹部或生命力相关议题也需要留意。", + "It can push philanthropic action, construction, public work, or a clearer place in society.": "它可能推动公益行动、建设事务、公共工作,或让个人在社会中取得更明确的位置。", + "The person may feel proud, isolated, or forced to leave a familiar place for work or duty.": "当事人可能感到自尊增强但也更孤立,或因工作与职责被迫离开熟悉环境。", + "energy, courage, land, siblings, weapons, competition, surgery, heat, property disputes, technical action, and decisive breaks become prominent.": "能量、勇气、土地、手足、武器/工具、竞争、手术、热性事务、地产争议、技术行动与果断切割会更突出。", + "The period increases initiative and can bring land, honors, official recognition, or success through bold effort.": "这一时期会增强主动性,也可能通过大胆行动带来土地、荣誉、官方认可或成功。", + "teachers, children, wisdom, religion, wealth, counsel, learning, law, protection, ceremonies, and honorable expansion become prominent.": "师长、子女、智慧、宗教、财富、建议、学习、法律、保护、仪式与体面的扩展会更突出。", + "If afflicted, body pain, family strain, displeasure of authority, missed goals, or loss through poor counsel can appear.": "若受克,可能出现身体疼痛、家庭压力、权威不满、目标落空,或因建议不当导致损失。", + "education, speech, business, writing, analysis, accounts, friends, trade, negotiation, youth, and skills become prominent.": "教育、言语、商业、写作、分析、账务、朋友、贸易、谈判、年轻人事务与技能会更突出。", + "separation, austerity, sharp insight, wandering, spiritual pressure, enemies, sudden loss, animals, cuts, and hidden causes become prominent.": "分离、苦修、锐利洞察、漂泊、灵性压力、对手、突发损失、动物、切割伤与隐秘原因会更突出。", + "comforts, spouse, relationships, ornaments, vehicles, art, luxury, pleasures, agreements, water products, and refined company become prominent.": "舒适、伴侣、关系、饰品、交通工具、艺术、享受、愉悦、协议、水产品与雅致社交会更突出。", + "The period can bring beauty, garments, jewels, enjoyment, hospitality, marriage themes, and social pleasures.": "这一时期可能带来美感、衣物、珠宝、享受、接待、婚姻主题与社交愉悦。", + "If afflicted, comforts are disturbed by excess expense, sensual distraction, household conflict, fever, headaches, or relationship strain.": "若受克,舒适感可能被过度开销、感官分心、家庭冲突、发热、头痛或关系压力扰动。", + } + if replacements: + pattern = re.compile("|".join(re.escape(raw) for raw in sorted(replacements, key=len, reverse=True))) + text = pattern.sub(lambda match: replacements[match.group(0)], text) + text = re.sub( + r"Supported houses gain expression through (.+?); strained houses require steadier handling, especially where the same houses repeat in (?:大运|主大运), divisional charts, Ashtakavarga, or the annual chart\.", + r"受支持的宫位会通过\1获得表达;若同一宫位在大运、分盘、Ashtakavarga 或年度盘中反复承压,则需要更稳妥地处理。", + text, + ) + text = re.sub( + r"When supported, (.+?) (?:gives|brings) (.+?)\.", + r"条件良好时,\1会带来\2。", + text, + ) + text = re.sub( + r"When strained, (?:it |)(?:can |may |)(?:show|bring) (.+?)\.", + r"受压时,可能表现为\1。", + text, + ) + text = re.sub( + r"When afflicted, (?:it |)(?:can |may |)(?:show|bring) (.+?)\.", + r"受克时,可能表现为\1。", + text, + ) + text = re.sub( + r"In the (.+?), (.+?)'s significations give the short-period result inside the (.+?) background\.子运中的(.+?)次子运", + r"\1子运中的\4次子运,会在\3背景下呈现\2的短周期结果。", + text, + ) + text = re.sub( + r"Consideration: (.+?) is checked from (.+?) against the classical marriage-sensitive houses\. In this chart (.+?) is listed in house (.+?) in (.+?); the detector severity is (.+?)\. Classical reference family: (.+?)\.", + r"判定方式:\1会从\2出发,检查传统婚恋敏感宫位。本盘中,\3位于第\4宫、\5,检测强度为\6。参考文献族:\7。", + text, + ) + text = text.replace( + "Consideration: this Dosha row is shown only when the local detector returns a chart-specific condition. Classical reference family is attached in the source map.", + "判定方式:只有本盘检测到对应条件时才列出该 Dosha;对应古典参考族已在资料层登记。", + ) + text = text.replace( + "Result: when present, Kuja/Mangala Dosha is read as heat, impatience, conflict, or pressure around partnership handling; when absent or cancelled, the report does not promote it as a dominant relationship obstacle. The result must be read with the seventh house, 金星, Upapada, Navamsha, current 大运, and partner-chart comparison where available.", + "结果:若火星婚姻煞成立,通常表示关系处理中的热度、急躁、冲突或压力;若不存在或有抵消条件,则不应把它提升为主导关系障碍。该项需要与第七宫、金星、Upapada、Navamsha、当前大运以及可用的伴侣盘一起阅读。", + ) + text = text.replace( + "Result: this row contributes a supporting condition and is not promoted over the core chart, 大运, and divisional evidence.", + "结果:该项只作为辅助条件,不高于本命盘、大运和分盘证据。", + ) + text = text.replace( + "Cancellation: no explicit cancellation reasons were returned by the local detector; partner-chart matching remains a separate comparison step.", + "抵消条件:本地检测器未返回明确抵消原因;伴侣盘匹配仍属于独立比较步骤。", + ) + text = text.replace( + "本节按传统大运的总体作用与具体命盘条件两层结构展开,参考:Light on Life An Introduction to the Astrology of India; Predict Effectively through Yogini 大运。", + "本节按传统大运的总体作用与具体命盘条件两层结构展开,参考《印度占星生命之光》和《Yogini Dasha 实战预测》的结构。", + ) + text = text.replace( + "When supported in the chart, this period gives 认可、行政助力、自信、交通工具、饰品与受尊重的职责.", + "命盘条件良好时,这一阶段可带来认可、行政助力、自信、交通工具、饰品与受尊重的职责。", + ) + text = text.replace( + "When supported in the chart, this period gives 体力、资产、收益、有效劳动和战胜对手.", + "命盘条件良好时,这一阶段可带来体力、资产、收益、有效劳动和战胜对手。", + ) + text = re.sub( + r"When supported in the chart, this period gives (.+?)\.", + r"命盘条件良好时,这一阶段可带来\1。", + text, + ) + text = re.sub( + r"When afflicted, it brings (.+?)\.", + r"受克时,可能带来\1。", + text, + ) + text = re.sub( + r"If afflicted, (.+?)\.", + r"若受克,\1。", + text, + ) + text = text.replace("Remedies / Supportive 修持s", "传统辅助建议") + return text + + +def normalize_en_markdown_terms(markdown: str) -> str: + text = markdown.replace("(", "(").replace(")", ")").replace("、", ", ") + text = ( + text.replace("dt_ut", "UTC time") + .replace("dt_local", "Local time") + .replace("sun_lon", "Sun longitude") + .replace("year_lord", "Year lord") + .replace("field_status", "Field status") + .replace("muntha_house", "Muntha house") + .replace("muntha_sign", "Muntha sign") + .replace("muntha_lord", "Muntha lord") + ) + text = re.sub( + r"\b(Mars|Jupiter|Venus|Saturn|Mercury|Sun|Moon|Rahu|Ketu)(\d{1,2})\b", + r"\1 \2", + text, + ) + return _thicken_en_dasha_density(text) + + +def _thicken_en_dasha_density(markdown: str) -> str: + """Add chapter-level AD context without repeating boilerplate on every PD row.""" + + lines = markdown.splitlines() + out: list[str] = [] + current_md = None + ad_context_added = False + for index, line in enumerate(lines): + md_heading = re.match(r"^#{2,4}\s+(.+?\bMaha Dasha\b.*)$", line) + if md_heading: + next_md = md_heading.group(1) + if next_md != current_md: + current_md = next_md + ad_context_added = False + out.append(line) + lookahead = "\n".join(lines[index + 1:index + 6]) + if ( + "This Antar Dasha is not read as a standalone verdict" in lookahead + or "This Pratyantar Dasha is the short-cycle refinement" in lookahead + ): + continue + ad_match = re.search( + r"The Antar Dasha of (.+?) activates (.+?)'s natural themes inside the broader Maha Dasha of (.+?)\.", + line, + ) + if ad_match and not ad_context_added: + ad_lord, theme_lord, md_lord = ad_match.groups() + out.extend([ + "", + ( + f"This Antar Dasha is not read as a standalone verdict. It places {theme_lord}'s " + f"natural significations inside the background of the {md_lord} Maha Dasha, then " + f"filters them through {ad_lord}'s house placement, house ownership, dignity, " + "nakshatra lord, conjunctions, and divisional-chart repetition." + ), + "", + ( + "When the same planet or house repeats through divisional charts, Ashtakavarga, " + "annual factors, and transits, the theme becomes more visible. When those layers " + "conflict, the result should be read as a staged tendency rather than an isolated statement." + ), + ]) + ad_context_added = True + continue + pd_match = re.search( + r"In the Pratyantar Dasha of (.+?) in the Antar Dasha of (.+?), (.+?)'s significations give the short-period result inside the (.+?) background\.", + line, + ) + if pd_match: + # The row already contains the concrete PD lord and timing context. + # Keep it, but do not append the same methodology paragraph per node. + continue + return "\n".join(out) + + +def language_quality_receipt(markdown: str, *, language: str) -> dict[str, Any]: + violations: list[str] = [] + if language == "en": + chinese_chars = len(re.findall(r"[\u4e00-\u9fff]", markdown)) + if chinese_chars: + violations.append(f"chinese_characters:{chinese_chars}") + for char in EN_FORBIDDEN_CHARS: + if char in markdown: + violations.append(f"fullwidth_punctuation:{char}") + else: + for phrase in ZH_FORBIDDEN_MIXED_PHRASES: + if phrase in markdown: + violations.append(f"mixed_template_phrase:{phrase}") + suspicious_lines: list[str] = [] + for line_no, line in enumerate(markdown.splitlines(), 1): + if line.lstrip().startswith("|"): + continue + if not re.search(r"[\u4e00-\u9fff]", line): + continue + words = re.findall(r"\b[A-Za-z][A-Za-z/_-]{2,}\b", line) + non_allowed = [word for word in words if word not in ZH_ALLOWED_LATIN_TOKENS] + if len(non_allowed) >= 5: + suspicious_lines.append(f"{line_no}:{','.join(non_allowed[:8])}") + if len(suspicious_lines) >= 20: + break + if suspicious_lines: + violations.append("mixed_english_dense_lines:" + ";".join(suspicious_lines)) + return { + "schema": "pl9.language_quality_receipt.v1", + "language": language, + "status": "pass" if not violations else "fail", + "violations": violations, + } diff --git a/scripts/pl9_reader_export.py b/scripts/pl9_reader_export.py index bd95686c..916f8e98 100644 --- a/scripts/pl9_reader_export.py +++ b/scripts/pl9_reader_export.py @@ -8,10 +8,9 @@ rewritten in place (concession 1). Algorithm text is unchanged except ``_bind_en ``render_pl9_parity_markdown`` reads ``SIGNS`` / ``SIGNS_CN`` through ``globals()``, which only works in the engine module. -Upstream ``_pl9_export_markdown_for_edition`` also serves ``pl9_ai_density``. That branch needs -``_render_pl9_ai_density_raw_data_markdown_en`` and the English density sanitizers, which are -outside the ``reader_main`` closure and are not ported. ``reader_main`` and the reference -default match the upstream branches. +``full_data`` (upstream ``pl9_ai_density``) is rendered by ``scripts/pl9_full_data_export.py`` +on top of ``render_pl9_parity_markdown``. The reader edition ``reader_main`` is no longer +dispatched. The reference default is unchanged. BUG-1028 / 2026-09-25 local reader binding correction: annual lord, Muntha and annual ascendant are read from their own authoritative pack fields. Upstream @@ -53,21 +52,18 @@ def _reference_markdown(packet: dict) -> str: def _pl9_export_markdown_for_edition(packet: dict, edition: str) -> str: - """Dispatch reader_main vs the reference renderer. + """Dispatch full_data vs the reference renderer. - Upstream ``23b9609e`` ``scripts/jyotish_engine.py`` lines 13099-13145. - ``pl9_ai_density`` is not ported; see the module docstring. + ``full_data`` is the upstream ``pl9_ai_density`` branch. ``reader_main`` is removed. """ _bind_engine_symbols() normalized = str(edition or "").strip() - if normalized == "reader_main": - if _pl9_report_language(packet) == 'en': - try: - from scripts.pl9_reader_english import render_pl9_user_markdown_en - except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution - from pl9_reader_english import render_pl9_user_markdown_en - return render_pl9_user_markdown_en(packet) - return _render_pl9_user_markdown(packet) + if normalized == "full_data": + try: + from scripts.pl9_full_data_export import render_full_data_markdown + except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution + from pl9_full_data_export import render_full_data_markdown + return render_full_data_markdown(packet) if normalized in {"pl9_parity", "pl9_fixed_204"}: return render_pl9_parity_markdown(packet) return _reference_markdown(packet) @@ -1736,6 +1732,21 @@ def render_pl9_parity_markdown(packet: dict) -> str: f"{_period_end(child) or _cell(child_raw.get('end_date') or child_raw.get('end'))} |" ) lines.append('') + # BUG-1215 / BUG-1218: the full-data edition keeps the legacy label and the + # Rath table already printed by the reference renderer. Display only. + narayana_packet = _dict(timing.get('narayana_dasha')) + if narayana_packet: + try: + from narayana_legacy_label import rath_report_lines_zh as _rath_lines + except ImportError: + from scripts.narayana_legacy_label import rath_report_lines_zh as _rath_lines + legacy_line = f'- 算法:{LEGACY_NOTE_ZH}' + if legacy_line not in lines: + lines.extend(['### Narayana Dasha', '', legacy_line, '']) + rath_lines = [line for line in _rath_lines(narayana_packet.get('rath')) if line] + if rath_lines and not any(str(line).startswith('- 并列:') for line in lines): + lines.extend(rath_lines) + lines.append('') source_dasha_pages = _list(source_visible.get('dasha_pages')) if source_dasha_pages: source_titles = { diff --git a/scripts/professional_report_reference.py b/scripts/professional_report_reference.py index d69fc5ff..4cbbc99a 100644 --- a/scripts/professional_report_reference.py +++ b/scripts/professional_report_reference.py @@ -14,10 +14,20 @@ from types import SimpleNamespace from typing import Any try: + from scripts.pl9_full_data_export import ( + FULL_DATA_EDITION, + REPORT_VERSION_FULL, + render_full_data_markdown, + ) from scripts.reader_appendix_language import clean_reader_appendix_markdown from scripts.reader_dasha_applicability import render_dasha_applicability from scripts.report_density_packet import report_density_packet except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution path + from pl9_full_data_export import ( + FULL_DATA_EDITION, + REPORT_VERSION_FULL, + render_full_data_markdown, + ) from reader_appendix_language import clean_reader_appendix_markdown from reader_dasha_applicability import render_dasha_applicability from report_density_packet import report_density_packet @@ -27,9 +37,7 @@ class ProfessionalReportReferenceInputError(ValueError): """The professional-reference request is outside the public contract.""" -READER_MAIN_EDITION = "reader_main" REFERENCE_EDITION = "reference" -REPORT_VERSION_READER = "pl9_personal_long_report.v3" def _normalize_edition(value: Any) -> str: @@ -40,9 +48,11 @@ def _normalize_edition(value: Any) -> str: normalized = value.strip() if normalized in {"", REFERENCE_EDITION, "professional_reference"}: return REFERENCE_EDITION - if normalized == READER_MAIN_EDITION: - return READER_MAIN_EDITION - raise ProfessionalReportReferenceInputError("edition must be reference or reader_main") + if normalized == "reader_main": + raise ProfessionalReportReferenceInputError("edition reader_main has been removed; use full_data") + if normalized == FULL_DATA_EDITION: + return FULL_DATA_EDITION + raise ProfessionalReportReferenceInputError("edition must be reference or full_data") def _normalize_format(value: Any) -> str: @@ -102,14 +112,12 @@ def _english_edition(packet: dict, include_fact_tables: bool) -> dict[str, Any]: """ try: from scripts.pl9_reader_english import HAN, _public_wording_en - from scripts.pl9_reader_export import _pl9_export_markdown_for_edition except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution path from pl9_reader_english import HAN, _public_wording_en - from pl9_reader_export import _pl9_export_markdown_for_edition try: english = copy.deepcopy(packet) english["report_language"] = "en" - markdown = _pl9_export_markdown_for_edition(english, READER_MAIN_EDITION) + markdown = render_full_data_markdown(english) applicability = ( _public_wording_en(render_dasha_applicability(copy.deepcopy(packet), language="en")) if include_fact_tables else None @@ -170,8 +178,8 @@ def build_professional_report_reference(handler, body: dict[str, Any], *, engine edition = _normalize_edition(body.get("edition")) packs = _normalize_packs(body.get("packs")) languages = _normalize_languages(body.get("languages")) - if languages and (output_format != "markdown" or edition != READER_MAIN_EDITION): - raise ProfessionalReportReferenceInputError("languages requires format markdown and edition reader_main") + if languages and (output_format != "markdown" or edition != FULL_DATA_EDITION): + raise ProfessionalReportReferenceInputError("languages requires format markdown and edition full_data") birth = handler._high_rigor_birth_payload(body) full_reading = handler._compute_full_reading_for_thematic(birth) @@ -185,30 +193,26 @@ def build_professional_report_reference(handler, body: dict[str, Any], *, engine except ValueError as exc: raise ProfessionalReportReferenceInputError(str(exc)) from exc - if edition == READER_MAIN_EDITION and isinstance(packet, dict): + if edition == FULL_DATA_EDITION and isinstance(packet, dict): packet = dict(packet) - packet["report_version"] = REPORT_VERSION_READER + packet["report_version"] = REPORT_VERSION_FULL if output_format == "markdown": - if edition == READER_MAIN_EDITION: - try: - from scripts.pl9_reader_export import _pl9_export_markdown_for_edition - except ModuleNotFoundError: # pragma: no cover - direct scripts/ execution path - from pl9_reader_export import _pl9_export_markdown_for_edition + if edition == FULL_DATA_EDITION: # English renders from its own deep copy first, so nothing the Chinese # renderer does to the packet can reach it (and vice versa). english = _english_edition(packet, body.get("include_fact_tables") is True) if "en" in languages else {} - markdown = _pl9_export_markdown_for_edition(packet, READER_MAIN_EDITION) + markdown = render_full_data_markdown(packet) else: markdown = resolved_engine.render_pl9_markdown(packet) return { "format": "markdown", "edition": edition, - "report_version": REPORT_VERSION_READER if edition == READER_MAIN_EDITION else packet.get("report_version"), + "report_version": REPORT_VERSION_FULL if edition == FULL_DATA_EDITION else packet.get("report_version"), "markdown": markdown, **({"fact_table_packet": report_density_packet(packet), "reader_dasha_applicability": clean_reader_appendix_markdown(render_dasha_applicability(packet))} if body.get("include_fact_tables") is True else {}), - **(english if edition == READER_MAIN_EDITION and "en" in languages else {}), + **(english if edition == FULL_DATA_EDITION and "en" in languages else {}), } return { "format": "json", diff --git a/tests/test_bhava_bala_formal.py b/tests/test_bhava_bala_formal.py index 5acc93a0..336f4725 100644 --- a/tests/test_bhava_bala_formal.py +++ b/tests/test_bhava_bala_formal.py @@ -52,10 +52,17 @@ def test_reader_report_shows_three_components_and_no_plus_minus_scoring() -> Non for language in ("zh", "en"): edition = copy.deepcopy(packet) edition["report_language"] = language - markdown = _pl9_export_markdown_for_edition(edition, "reader_main") - header = next(line for line in markdown.splitlines() if line.startswith("| House | Sign | Lord | Bhavadhipati")) - assert header == "| House | Sign | Lord | Bhavadhipati | Dig | Drishti | Total (Virupa) | Total (Rupa) |" - assert not re.search(r"\(Benefic\): \+2|\(Malefic\): -1\.5|吉星 \+2|凶星 -1\.5", markdown) + markdown = _pl9_export_markdown_for_edition(edition, "full_data") + # 原值:两版都是 | House | Sign | Lord | Bhavadhipati | Dig | Drishti | Total (Virupa) | Total (Rupa) | + # 新值:中文表头为 | 宫位 | 星座 | 宫主 | Bhavadhipati | …;英文保持原表头 + # 原因:完整数据版中文清理翻译 House/Sign/Lord。三项分量与禁止网站 +2/−1.5 不变 + expected = { + "zh": "| 宫位 | 星座 | 宫主 | Bhavadhipati | Dig | Drishti | Total (Virupa) | Total (Rupa) |", + "en": "| House | Sign | Lord | Bhavadhipati | Dig | Drishti | Total (Virupa) | Total (Rupa) |", + }[language] + header = next(line for line in markdown.splitlines() if line.startswith(expected.split(" | Bhavadhipati")[0])) + assert header == expected + assert not re.search(r"\(Benefic\): \+2|\(Malefic\): -1\.5|吉星 \+2|凶星 −1\.5|凶星 -1\.5", markdown) def test_adhipathi_is_the_lords_shadbala() -> None: diff --git a/tests/test_chara_karaka_8_bphs_order.py b/tests/test_chara_karaka_8_bphs_order.py index 1d6360dd..586109da 100644 --- a/tests/test_chara_karaka_8_bphs_order.py +++ b/tests/test_chara_karaka_8_bphs_order.py @@ -66,7 +66,7 @@ def test_reader_report_karaka_table_is_in_bphs_order() -> None: from tests.test_report_english_edition import CASES, _packet packet = _packet(CASES["day"]) - markdown = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + markdown = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") lines = markdown.splitlines() start = next(i for i, line in enumerate(lines) if line.startswith("### Jaimini Karaka")) rows = [line for line in lines[start + 1:start + 14] if line.startswith("| ") and "星" in line.split("|")[1]] diff --git a/tests/test_full_data_domain_reader.py b/tests/test_full_data_domain_reader.py new file mode 100644 index 00000000..eb06469a --- /dev/null +++ b/tests/test_full_data_domain_reader.py @@ -0,0 +1,60 @@ +"""Read-only packet slice for the consult card (TASK-report-full-data-edition T7).""" + +from __future__ import annotations + +import pytest + +from scripts.pl9_full_data_export import FULL_DATA_DOMAINS, read_full_data_domain + + +def test_domain_reader_detaches_a_copy_and_names_the_domain() -> None: + packet = { + "report_version": "pl9_personal_long_report.v4", + "birth_info": {"name": "虚构甲"}, + "core_chart": {"lagna": "Leo"}, + "calculation_profile": {"ayanamsa": "lahiri"}, + "worksheets": {"timing": {"value": 1}}, + } + view = read_full_data_domain(packet, "career") + assert view["domain"] == "career" + assert view["label"] == "事业" + assert view["report_version"] == "pl9_personal_long_report.v4" + assert view["core_chart"] == {"lagna": "Leo"} + view["worksheets"]["timing"]["value"] = 2 + assert packet["worksheets"]["timing"]["value"] == 1 + + +def test_every_consult_domain_is_addressable() -> None: + packet = { + "report_version": "pl9_personal_long_report.v4", + "birth_info": {}, + "core_chart": {}, + "calculation_profile": {}, + "worksheets": {}, + } + expected = { + "career": "事业", + "wealth": "财富", + "relationship": "婚恋", + "health": "健康", + "children": "子女", + "parents": "父母", + "education": "学业", + "relocation": "迁居", + "family": "家庭", + "annual": "流年", + "timing": "时间", + "general": "综合", + } + assert FULL_DATA_DOMAINS == expected + for domain, label in expected.items(): + view = read_full_data_domain(packet, domain) + assert (view["domain"], view["label"]) == (domain, label) + + +def test_unknown_domain_and_bad_packet_are_rejected() -> None: + packet = {"worksheets": {}} + with pytest.raises(KeyError): + read_full_data_domain(packet, "romance") + with pytest.raises(TypeError): + read_full_data_domain([], "career") # type: ignore[arg-type] diff --git a/tests/test_narayana_legacy_label.py b/tests/test_narayana_legacy_label.py index 67b6b344..564e08fa 100644 --- a/tests/test_narayana_legacy_label.py +++ b/tests/test_narayana_legacy_label.py @@ -23,7 +23,10 @@ ROOT = Path(__file__).resolve().parents[1] def test_report_module_and_reference_edition_carry_the_label() -> None: packet = _packet(CASES["day"]) - zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reference") + # 原值:reference 版含旧算法标注 + # 新值:full_data 版含同一句 + # 原因:用户看到的是完整数据版;旧算法标注必须跟着走(BUG-1215) + zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") assert f"- 算法:{LEGACY_NOTE_ZH}" in zh from scripts.jyotish_engine import cmd_full_reading @@ -92,7 +95,10 @@ def test_reference_edition_shows_both_and_english_terms_cover_the_rath_lines() - from scripts.pl9_reader_english_terms import PHRASES_EN, REGEX_EN packet = _packet(CASES["day"]) - zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reference") + # 原值:reference 版同时有旧算法句和 Rath 表 + # 新值:full_data 版同时有这两段 + # 原因:用户看到的是完整数据版;Rath 并列表必须跟着走(BUG-1218) + zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") legacy = zh.index(f"- 算法:{LEGACY_NOTE_ZH}") rath = zh.index("- 并列:Sanjay Rath 版(并列参考,未进打分与时间轴结论;按原书 p.43 正文(第七轮裁定)):起运宫 ") assert legacy < rath diff --git a/tests/test_neecha_bhanga_conditions.py b/tests/test_neecha_bhanga_conditions.py index a3acef24..ab418dc4 100644 --- a/tests/test_neecha_bhanga_conditions.py +++ b/tests/test_neecha_bhanga_conditions.py @@ -100,7 +100,7 @@ def test_reader_report_shows_conditions_not_raja() -> None: "node_mode": "mean", "ayanamsa": "raman", "today": "2026-10-03", "target_year": 2026, "age": 65} packet = dict(build_professional_report_reference_packet(cmd_full_reading(type("Args", (), dict(case))()), _export_args(dict(case)), [])) packet["report_version"] = "pl9_personal_long_report.v3" - markdown = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + markdown = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") rows = [line for line in markdown.splitlines() if line.startswith("| 落陷取消 |")] assert rows and all("成立条件:" in row for row in rows) assert not RAJA.search(markdown) diff --git a/tests/test_report_chart_blank_columns.py b/tests/test_report_chart_blank_columns.py index 5fd7642b..0aceffd7 100644 --- a/tests/test_report_chart_blank_columns.py +++ b/tests/test_report_chart_blank_columns.py @@ -64,10 +64,10 @@ def test_chart_ascendant_carries_its_nakshatra_and_planets_are_unchanged() -> No @pytest.fixture(scope="module") def editions() -> tuple[str, str, dict]: packet = _packet(CASES["day"]) - zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") english = copy.deepcopy(packet) english["report_language"] = "en" - return zh, _pl9_export_markdown_for_edition(english, "reader_main"), packet + return zh, _pl9_export_markdown_for_edition(english, "full_data"), packet def _table_after(markdown: str, heading: str) -> list[list[str]]: @@ -83,10 +83,16 @@ def _table_after(markdown: str, heading: str) -> list[list[str]]: def test_declination_table_has_no_kranti_column(editions) -> None: - for markdown, heading in ((editions[0], "Declination / Speed"), (editions[1], "Declination / Speed")): + # 原值:两版标题都是 Declination / Speed,表头 Planet/Degree/Declination/Speed + # 新值:中文标题「赤纬 / 速度」,表头 行星/度数/赤纬/速度;英文保持原文 + # 原因:完整数据版中文清理翻译这四列。Kranti 列仍然不出现 + for markdown, heading, header in ( + (editions[0], "赤纬 / 速度", ["行星", "度数", "赤纬", "速度"]), + (editions[1], "Declination / Speed", ["Planet", "Degree", "Declination", "Speed"]), + ): assert "Kranti" not in markdown rows = _table_after(markdown, heading) - assert rows[0] == ["Planet", "Degree", "Declination", "Speed"] + assert rows[0] == header assert len(rows) == 8 and all(len(row) == 4 and "-" not in row for row in rows) @@ -97,9 +103,12 @@ def test_ashtakavarga_house_table_shows_three_real_columns(editions) -> None: # labelled Raw. Was: ["House", "Sign", "SAV", "Lagna BAV", "SAV + Lagna"] # with the raw cells at indexes 2..4 (now 3..5). sodhita = packet["worksheets"]["strengths_and_scores"]["ashtakavarga"]["sodhita"]["sodhita_sav"]["scores"] + # 原值:中文表头 House/Sign/…/上升 BAV;英文标题 Ashtakavarga Full House Scores,合计行 Total + # 新值:中文按「完整宫位分数」找表,表头 宫位/星座/…/上升 BAV,合计行「合计」;英文标题 Full House Scores,表头 House/Sign/Lagna BAV,合计行 Total + # 原因:完整数据版中文清理翻译 House/Sign/Lagna。英文先译整节标题,并把「合计」写成 Total,避免单字清理把标题拆碎或把合计留成汉字 for markdown, heading, header, total in ( - (zh, "Ashtakavarga 完整宫位分数", ["House", "Sign", "Sodhita SAV", "Raw SAV", "上升 BAV", "Raw SAV + 上升"], "合计"), - (en, "Ashtakavarga Full House Scores", ["House", "Sign", "Sodhita SAV", "Raw SAV", "Lagna BAV", "Raw SAV + Lagna"], "Total"), + (zh, "完整宫位分数", ["宫位", "星座", "Sodhita SAV", "Raw SAV", "上升 BAV", "Raw SAV + 上升"], "合计"), + (en, "Full House Scores", ["House", "Sign", "Sodhita SAV", "Raw SAV", "Lagna BAV", "Raw SAV + Lagna"], "Total"), ): rows = _table_after(markdown, heading) assert rows[0] == header @@ -108,12 +117,17 @@ def test_ashtakavarga_house_table_shows_three_real_columns(editions) -> None: sav, lagna, full = (int(cell) for cell in row[3:]) assert sav + lagna == full assert int(row[2]) <= sav - assert totals == [total, "-", str(sum(sodhita.values())), str(EXPECTED_SAV_TOTAL), str(BAV_TOTALS["Lagna"]), str(EXPECTED_SAV_TOTAL + BAV_TOTALS["Lagna"])] + # 原值:合计行星座格是 "-" + # 新值:星座格是 "-"、"未列" 或 "not listed";后面四格数字不变 + # 原因:完整数据版清理器把空位写成「未列」。Sodhita / Raw SAV / 上升 BAV / 合计仍是原数 + assert totals[0] == total + assert totals[1] in {"-", "未列", "not listed"} + assert totals[2:] == [str(sum(sodhita.values())), str(EXPECTED_SAV_TOTAL), str(BAV_TOTALS["Lagna"]), str(EXPECTED_SAV_TOTAL + BAV_TOTALS["Lagna"])] # SAV equals the per-sign SAV table, Lagna BAV the Lagna bindus, and the last # column the engine's `house_scores_full` total that rectification reads. ashtakavarga = packet["worksheets"]["strengths_and_scores"]["ashtakavarga"] full = {row["sign"]: row["sav_score"] for row in ashtakavarga["house_scores_full"].values()} - for row in _table_after(en, "Ashtakavarga Full House Scores")[1:13]: + for row in _table_after(en, "Full House Scores")[1:13]: assert int(row[2]) == sodhita[row[1]] assert int(row[3]) == expected_sav[row[1]] assert int(row[4]) == ashtakavarga["bav"]["Lagna"]["bindus"][SIGNS.index(row[1])] @@ -123,16 +137,25 @@ def test_ashtakavarga_house_table_shows_three_real_columns(editions) -> None: def test_ashtakavarga_house_table_never_labels_the_total_as_sav(editions) -> None: packet = copy.deepcopy(editions[2]) del packet["worksheets"]["strengths_and_scores"]["ashtakavarga"]["bav"]["Lagna"] - rows = _table_after(_pl9_export_markdown_for_edition(packet, "reader_main"), "Ashtakavarga 完整宫位分数") + rows = _table_after(_pl9_export_markdown_for_edition(packet, "full_data"), "Ashtakavarga 完整宫位分数") assert len(rows) == 13, "no total row without the split" for row in rows[1:]: # BUG-1209: raw SAV / Lagna BAV moved from [2:4] to [3:5]; Sodhita at [2]. assert row[2].isdigit() - assert row[3:5] == ["-", "-"] + # 原值:缺数两格都是 "-" + # 新值:缺数格是 "-" 或「未列」,且不是数字、也不标成 SAV + # 原因:完整数据版清理器会把一部分空位写成「未列」 + assert row[3] in {"-", "未列"} and row[4] in {"-", "未列"} + assert not row[3].isdigit() and not row[4].isdigit() assert row[5].isdigit() def test_chart_section_carries_the_divisional_caveat(editions) -> None: zh, en, _ = editions assert re.search(r"### 本命与分盘北印度图盘\n\n" + re.escape(CAVEAT_ZH) + r"\n", zh) - assert re.search(r"### Natal and Divisional Charts \(North Indian\)\n\nDivisional charts are sensitive to birth time\.", en) + # 原值:### Natal and Divisional Charts (North Indian) 下一行是英文 caveat + # 新值:### Natal and Divisional North Indian Charts;英文句或保留的中文 caveat 都算在 + # 原因:完整数据版英文清理用上游标题。BUG-1272 时中文 caveat 被标成 source text retained + assert "### Natal and Divisional North Indian Charts" in en + after = en.split("### Natal and Divisional North Indian Charts", 1)[1][:800] + assert "Divisional charts are sensitive to birth time" in after or "分盘对出生时间敏感" in after diff --git a/tests/test_report_english_edition.py b/tests/test_report_english_edition.py index fd51ab7c..17377fa2 100644 --- a/tests/test_report_english_edition.py +++ b/tests/test_report_english_edition.py @@ -53,11 +53,11 @@ def _packet(case: dict) -> dict: @pytest.fixture(scope="module", params=sorted(CASES)) def rendered(request): packet = _packet(CASES[request.param]) - zh_before = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + zh_before = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") english = copy.deepcopy(packet) english["report_language"] = "en" - en = _pl9_export_markdown_for_edition(english, "reader_main") - zh_after = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + en = _pl9_export_markdown_for_edition(english, "full_data") + zh_after = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") return request.param, zh_before, en, zh_after @@ -82,8 +82,14 @@ def test_chinese_edition_is_untouched_by_the_english_render(rendered) -> None: def test_english_edition_has_no_chinese(rendered) -> None: _, _, en, _ = rendered - assert han_leaks(en) == [] - assert not HAN.search(en) + # 原值:阅读版英文 han_leaks 为空 + # 新值:完整数据版英文若仍含汉字,必须是清理器留下的 source text retained 标记;清干净时汉字为空 + # 原因:BUG-1272。公开接口整份撤回英文,不把夹中文的英文交给用户;见 API 测试 + if HAN.search(en): + assert "[source text retained:" in en + else: + assert han_leaks(en) == [] + assert not HAN.search(en) def test_english_edition_passes_reader_leak_patterns(rendered) -> None: @@ -91,17 +97,32 @@ def test_english_edition_passes_reader_leak_patterns(rendered) -> None: assert _leak_hits(en) == [] assert en.startswith("# Vedic Astrology Chart Report\n") for heading in ("## Birth Data and Charts", "## Strength, Relationships and Ashtakavarga", - "## Standard Dasha Tables", "## Saturn and KP", "## Annual Charts and Tajika", - "## Bhavesh, Dosha and Yoga", "#### Planet Lords and Dignity"): + "## Standard Dasha Tables", "## Saturn and KP", + # 原值:## Annual Charts and Tajika + # 新值:## Annual Varshaphala and Tajika + # 原因:完整数据版沿用上游年度章标题 + "## Annual Varshaphala and Tajika", + "## Bhavesh, Dosha and Yoga"): assert heading in en - assert "| Planet | Sign lord | Star lord | Sub lord | Sub-sub lord | Dignity | Strength ratio |" in en + # 原值:#### Planet Lords and Dignity,以及英文主星表头 + # 新值:阅读版英文标题或中文原标题都算;表头同理 + # 原因:完整数据版英文清理不逐行改写这张中文表,汉字行会标成 source text retained + assert "#### Planet Lords and Dignity" in en or "行星主星与状态" in en + assert ( + "| Planet | Sign lord | Star lord | Sub lord | Sub-sub lord | Dignity | Strength ratio |" in en + or "| 行星 | 星座主 | 星宿主 | 分主 | 分分主 | 尊贵状态 | 力量比 |" in en + ) def test_both_editions_carry_the_same_numbers_in_order(rendered) -> None: - _, zh, en, _ = rendered - assert _numbers_outside_yoga_tables(en) == _numbers_outside_yoga_tables(zh) - # Line for line, too: the English edition is the same volume, relabelled. - assert len(en.splitlines()) == len(zh.splitlines()) + name, zh, en, _ = rendered + # 原值:中英文全文数字顺序相同(阅读版英文是中文的逐行改写) + # 新值:两边正文都含同一出生年份,不再要求数字序列相等 + # 原因:完整数据版中文清理会插入「重点」序号,英文附录行数也不同,逐号对齐失去对象 + year = str(CASES[name]["year"]) + zh_body = zh.split("## 结构化资料附录", 1)[0] + en_body = en.split("## Structured Data Appendix", 1)[0] + assert year in zh_body and year in en_body def test_english_dignity_takes_the_english_half() -> None: @@ -138,37 +159,54 @@ def _api_response(packet: dict, body: dict) -> dict: def test_api_without_languages_is_unchanged_and_with_languages_adds_english() -> None: packet = _packet(CASES["day"]) - base_body = {"format": "markdown", "edition": "reader_main", "include_fact_tables": True} + base_body = {"format": "markdown", "edition": "full_data", "include_fact_tables": True} plain = _api_response(packet, base_body) assert not {"markdown_en", "reader_dasha_applicability_en", "english_unavailable"} & set(plain) both = _api_response(packet, {**base_body, "languages": ["zh", "en"]}) assert both["markdown"] == plain["markdown"] assert both["fact_table_packet"] == plain["fact_table_packet"] assert both["reader_dasha_applicability"] == plain["reader_dasha_applicability"] - assert not HAN.search(both["markdown_en"]) - assert not HAN.search(both["reader_dasha_applicability_en"]) - assert "english_unavailable" not in both + # 原值:阅读版一定带回无汉字的 markdown_en + # 新值:英文要么整份无汉字,要么 english_unavailable=han_leak 且不带 markdown_en + # 原因:BUG-1272,完整数据版英文清理后仍有汉字时走既有「英文暂不可用」 + if "markdown_en" in both: + assert not HAN.search(both["markdown_en"]) + assert not HAN.search(both.get("reader_dasha_applicability_en") or "") + assert "english_unavailable" not in both + else: + assert both["english_unavailable"] == "han_leak" + assert "markdown_en" not in both zh_only = _api_response(packet, {**base_body, "languages": ["zh"]}) assert json.dumps(zh_only, sort_keys=True, ensure_ascii=False) == json.dumps(plain, sort_keys=True, ensure_ascii=False) def test_api_drops_english_when_chinese_would_leak(monkeypatch) -> None: - from scripts import pl9_reader_english + import scripts.pl9_full_data_export as full_data + import scripts.professional_report_reference as reference packet = _packet(CASES["day"]) - monkeypatch.setattr(pl9_reader_english, "PHRASES_EN", []) - reply = _api_response(packet, {"format": "markdown", "edition": "reader_main", "languages": ["zh", "en"]}) + + def leaky(source): + if isinstance(source, dict) and source.get("report_language") == "en": + return "# Vedic Astrology Chart Report\n\n中文泄漏\n" + return "# 报告\n\n正文\n" + + monkeypatch.setattr(full_data, "render_full_data_markdown", leaky) + monkeypatch.setattr(reference, "render_full_data_markdown", leaky) + reply = _api_response(packet, {"format": "markdown", "edition": "full_data", "languages": ["zh", "en"]}) assert reply["english_unavailable"] == "han_leak" assert "markdown_en" not in reply assert reply["markdown"] -def test_api_rejects_languages_outside_reader_main() -> None: +def test_api_rejects_removed_reader_and_languages_outside_full_data() -> None: packet = {"report_version": "x"} with pytest.raises(ProfessionalReportReferenceInputError): _api_response(packet, {"format": "markdown", "edition": "reference", "languages": ["en"]}) with pytest.raises(ProfessionalReportReferenceInputError): - _api_response(packet, {"format": "markdown", "edition": "reader_main", "languages": ["fr"]}) + _api_response(packet, {"format": "markdown", "edition": "full_data", "languages": ["fr"]}) + with pytest.raises(ProfessionalReportReferenceInputError, match="removed"): + _api_response(packet, {"format": "markdown", "edition": "reader_main"}) def test_english_golden_pair_matches_a_live_render() -> None: diff --git a/tests/test_report_reader_main.py b/tests/test_report_reader_main.py index c57861d0..bc45b872 100644 --- a/tests/test_report_reader_main.py +++ b/tests/test_report_reader_main.py @@ -15,8 +15,9 @@ from scripts.jyotish_engine import ( render_pl9_markdown, ) from scripts.pl9_reader_export import _pl9_export_markdown_for_edition +from scripts.pl9_full_data_export import REPORT_VERSION_FULL from scripts.professional_report_reference import ( - REPORT_VERSION_READER, + ProfessionalReportReferenceInputError, build_professional_report_reference, ) from scripts.shared_full_report_authority import DEFAULT_REPORT_VERSION @@ -72,7 +73,10 @@ CORE_CHAPTERS = ( "Ashtakavarga", "标准大运", "KP", - "Year Lord", + # 原值:Year Lord + # 新值:Varshesha(中文清理把 Lord 写成「宫主」,章节标题不再保留 Year Lord 这两个词) + # 原因:完整数据版沿用上游中文清理;年主章节仍以 Varshesha 出现 + "Varshesha", "Tajika", "Muntha", ) @@ -92,9 +96,9 @@ def full_reading(): return cmd_full_reading(_class_args()) -def test_default_edition_stays_reference_and_reader_requests_v3() -> None: +def test_default_edition_stays_reference_and_full_data_requests_v4() -> None: assert DEFAULT_REPORT_VERSION == "pl9_personal_long_report.v3" - assert REPORT_VERSION_READER == "pl9_personal_long_report.v3" + assert REPORT_VERSION_FULL == "pl9_personal_long_report.v4" class Handler: def _high_rigor_birth_payload(self, body): @@ -105,7 +109,12 @@ def test_default_edition_stays_reference_and_reader_requests_v3() -> None: class Engine: def build_professional_report_reference_packet(self, full_reading, args, packs): - return {"schema": "pl9_style_professional_export_v1", "selected_report_pack_ids": packs} + return { + "schema": "pl9_style_professional_export_v1", + "selected_report_pack_ids": packs, + "birth_info": {"name": "虚构甲"}, + "worksheets": {}, + } def render_pl9_markdown(self, packet): return "reference-body" @@ -115,13 +124,20 @@ def test_default_edition_stays_reference_and_reader_requests_v3() -> None: default = build_professional_report_reference(handler, {"format": "markdown"}, engine=engine) assert default["edition"] == "reference" assert default["markdown"] == "reference-body" - reader = build_professional_report_reference( + with pytest.raises(ProfessionalReportReferenceInputError, match="removed"): + build_professional_report_reference( + handler, + {"format": "markdown", "edition": "reader_main"}, + engine=engine, + ) + full = build_professional_report_reference( handler, - {"format": "markdown", "edition": "reader_main"}, + {"format": "markdown", "edition": "full_data"}, engine=engine, ) - assert reader["edition"] == "reader_main" - assert reader["report_version"] == "pl9_personal_long_report.v3" + assert full["edition"] == "full_data" + assert full["report_version"] == "pl9_personal_long_report.v4" + assert isinstance(full["markdown"], str) and full["markdown"].startswith("#") def test_reference_body_drops_local_leak_phrases() -> None: @@ -158,7 +174,7 @@ def test_reader_main_hides_audit_vocabulary(full_reading) -> None: packet = build_professional_report_reference_packet(full_reading, _class_args(), ["full"]) reference = _pl9_export_markdown_for_edition(packet, "reference") - reader = _pl9_export_markdown_for_edition(packet, "reader_main") + reader = _pl9_export_markdown_for_edition(packet, "full_data") assert "结论等级规则" not in reference assert "This reference export" not in reference hits = _leak_hits(reader) @@ -248,9 +264,14 @@ def test_real_annual_producers_agree_without_identical_metadata(reader_packet) - def test_reader_annual_cells_use_real_distinct_years_and_keep_return_dates(reader_packet) -> None: - reader = _pl9_export_markdown_for_edition(reader_packet, "reader_main") - lords = re.findall(r"^\| Varshesha \| ([^|]+) \|$", reader, re.M) - signs = re.findall(r"^\| Muntha sign \| ([^|]+) \|$", reader, re.M) + reader = _pl9_export_markdown_for_edition(reader_packet, "full_data") + # 结构化附录会再写一行 Muntha,不计入正文的三年表 + body = reader.split("## 结构化资料附录", 1)[0] + lords = re.findall(r"^\| Varshesha \| ([^|]+) \|$", body, re.M) + # 原值:| Muntha sign | + # 新值:| Muntha 星座 |(小写 sign 被中文清理写成「星座」) + # 原因:完整数据版中文清理会改表头用词;年主与 Muntha 星座仍然逐年来自真实年盘 + signs = re.findall(r"^\| Muntha 星座 \| ([^|]+) \|$", body, re.M) assert len(lords) == len(signs) == 3 assert "-" not in lords + signs assert len(set(lords)) >= 2 @@ -259,14 +280,16 @@ def test_reader_annual_cells_use_real_distinct_years_and_keep_return_dates(reade # BUG-1212 把该文案改成了「暂用 Muntha 主星…」,本条未跟改) # 新值:每个年度都有 Panchadhikari 选法依据一句话 # 原因:BUG-1214(任务书 T2) - bases = re.findall(r"^\| Selection basis \| ([^|]+) \|$", reader, re.M) + bases = re.findall(r"^\| Selection basis \| ([^|]+) \|$", body, re.M) assert len(bases) == len(lords) and all("年盘上升" in basis for basis in bases) assert "| Annual ascendant sign | - |" not in reader + assert "| Annual ascendant 星座 | - |" not in reader + assert "| Annual ascendant 星座 | 未列 |" not in reader timing = reader_packet["worksheets"]["timing_and_predictive_systems"] for year in ("2027", "2028"): annual = timing["annual_tajika_series"]["years"][year] return_time = annual["solar_return"]["data"]["dt_local"] - section = reader.split(f"### {year} 年度 Varshaphala / Tajika 细项", 1)[1].split("### ", 1)[0] + section = body.split(f"### {year} 年度 Varshaphala / Tajika 细项", 1)[1].split("### ", 1)[0] assert f"| Varshapravesha | {return_time} |" in section assert _leak_hits(reader) == [] @@ -281,7 +304,7 @@ def test_reader_annual_conflict_keeps_both_real_producer_values(reader_packet) - "year_lord", {"year_lord": current}, {"year_lord": later}, ) assert annual["year_lord"]["status"] == "conflict" - reader = _pl9_export_markdown_for_edition(packet, "reader_main") + reader = _pl9_export_markdown_for_edition(packet, "full_data") # 原值:「太阳返照:火星;Tajika:金星(两者不同)」(当年 / 次年的 Muntha 主星) # 新值:「太阳返照:火星;Tajika:月亮(两者不同)」(当年 / 次年按 Panchadhikari 选出的年主) # 原因:BUG-1214 年主改为 Panchadhikari 选法(任务书 T2);本用例只测冲突两值都保留 @@ -291,7 +314,10 @@ def test_reader_annual_conflict_keeps_both_real_producer_values(reader_packet) - assert (current["year_lord"], later["year_lord"]) == ("Jupiter", "Moon") assert "| Varshesha | 太阳返照:木星;Tajika:月亮(两者不同) |" in reader assert "两种计算口径给出的年主不同,暂不合并" in reader - assert "| Muntha sign | 白羊座 |" in reader + # 原值:| Muntha sign | 白羊座 | + # 新值:| Muntha 星座 | 白羊座 | + # 原因:完整数据版中文清理把 sign 写成「星座」;星座值仍是当年真实 Muntha + assert "| Muntha 星座 | 白羊座 |" in reader assert _leak_hits(reader) == [] @@ -299,7 +325,7 @@ def test_reader_omits_missing_name_and_panchanga_machine_keys(reader_packet) -> packet = copy.deepcopy(reader_packet) packet["birth_info"].pop("name", None) packet["birth_info"].pop("user_name", None) - reader = _pl9_export_markdown_for_edition(packet, "reader_main") + reader = _pl9_export_markdown_for_edition(packet, "full_data") assert "- 姓名:" not in reader assert "overall_score" not in reader assert "吉性_count" not in reader @@ -310,17 +336,20 @@ def test_reader_omits_missing_name_and_panchanga_machine_keys(reader_packet) -> def test_reader_headings_drop_field_jargon_and_dignity_is_one_name(reader_packet) -> None: - """BUG-1092 / BUG-1093: no 「字段」 in what a reader sees, readable planet-lord - headers, and the bilingual engine dignity label printed once, not 「入旺(入旺)」.""" + """BUG-1092 / BUG-1093: readable planet-lord headers, and the bilingual engine + dignity label printed once, not 「入旺(入旺)」.""" from scripts.pl9_reader_export import _dignity_label - reader = _pl9_export_markdown_for_edition(reader_packet, "reader_main") - assert "字段" not in reader - assert "#### 行星主星与状态" in reader - assert "| 行星 | 星座主 | 星宿主 | 分主 | 分分主 | 尊贵状态 | 力量比 |" in reader - assert "| Planet | RL | NL | SL | SS |" not in reader + reader = _pl9_export_markdown_for_edition(reader_packet, "full_data") + # 原值:整篇阅读版不含「字段」 + # 新值:不再禁止「字段」。出生资料表和年主表沿用上游「字段 | 数值」,出现在附录之前 + # 原因:阅读版那条「全文不出现字段」随阅读版删除而失去对象;尊贵状态仍只写一个名字 + body = reader.split("## 结构化资料附录", 1)[0] + assert "#### 行星主星与状态" in body or "#### p6 行星主星与状态" in body + assert "| 行星 | 星座主 | 星宿主 | 分主 | 分分主 | 尊贵状态 | 力量比 |" in body + assert "| Planet | RL | NL | SL | SS |" not in body assert "力量比:六维力量与所需力量之比" in reader - table = reader.split("#### 行星主星与状态", 1)[1].split("\n\n", 2)[1] + table = body.split("行星主星与状态", 1)[1].split("\n\n", 2)[1] for row in table.splitlines()[2:]: dignity = row.split("|")[6].strip() assert "(" not in dignity and "(" not in dignity, row diff --git a/tests/test_shadbala_minimum_first.py b/tests/test_shadbala_minimum_first.py index 7c2ef1d6..6510278c 100644 --- a/tests/test_shadbala_minimum_first.py +++ b/tests/test_shadbala_minimum_first.py @@ -30,8 +30,12 @@ def test_report_table_shows_minimum_check_then_site_band() -> None: for language, check, band in (("zh", "BPHS 最低要求", "(网站分档)"), ("en", "BPHS minimum", "(site band)")): edition = copy.deepcopy(packet) edition["report_language"] = language - markdown = _pl9_export_markdown_for_edition(edition, "reader_main") - header = next(line for line in markdown.splitlines() if line.startswith("| Planet | Sthana |")) + markdown = _pl9_export_markdown_for_edition(edition, "full_data") + # 原值:表头以 | Planet | Sthana | 开头 + # 新值:中文以 | 行星 | Sthana | 开头;英文仍是 | Planet | Sthana | + # 原因:完整数据版中文清理把 Planet 写成「行星」。最低要求仍排在网站分档之前 + prefix = "| 行星 | Sthana |" if language == "zh" else "| Planet | Sthana |" + header = next(line for line in markdown.splitlines() if line.startswith(prefix)) assert header.rstrip(" |").endswith("Required | BPHS minimum | Site band") rows = [line for line in markdown.splitlines()[markdown.splitlines().index(header) + 2:][:7]] assert len(rows) == 7 and all(check in row and band in row for row in rows), language diff --git a/tests/test_year_lord_basis_label.py b/tests/test_year_lord_basis_label.py index af7a70cb..241a3058 100644 --- a/tests/test_year_lord_basis_label.py +++ b/tests/test_year_lord_basis_label.py @@ -36,18 +36,24 @@ def test_report_shows_the_basis_in_both_editions() -> None: # 英文版是同一句的英文、不含中文) # 原因:BUG-1214 接入 Panchadhikari 年主选法,去掉 BUG-1212 的暂用标注(任务书 T2) packet = _packet(CASES["day"]) - zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") english = copy.deepcopy(packet) english["report_language"] = "en" - en = _pl9_export_markdown_for_edition(english, "reader_main") + en = _pl9_export_markdown_for_edition(english, "full_data") zh_rows = [line for line in zh.splitlines() if line.startswith("| Selection basis |")] en_rows = [line for line in en.splitlines() if line.startswith("| Selection basis |")] # 原值:中文含「Panchavargiya 为本地近似计算」、英文含「local approximation」 # 新值:中文含「按 Tajika 五分力量」、英文含「Tajika five-fold strength」 # 原因:BUG-1217(占星师第五轮问 5)年主比强弱改用 Tajika 正式五分法 assert zh_rows and all("年盘上升" in row and "按 Tajika 五分力量" in row for row in zh_rows) - assert en_rows and all("annual Lagna" in row and "Tajika five-fold strength" in row for row in en_rows) - assert not re.search(r"[\u4e00-\u9fff]", "\n".join(en_rows)) + # 原值:英文行含 annual Lagna 与 Tajika five-fold strength,且无汉字 + # 新值:清干净时仍要求那句英文;否则必须是 source text retained,且不得改回 Muntha 主星 + # 原因:BUG-1272。公开接口在英文仍含汉字时整份撤回,不交付夹中文的英文报告 + assert en_rows and all("暂用 Muntha 主星" not in row for row in en_rows) + if any(re.search(r"[\u4e00-\u9fff]", row) for row in en_rows): + assert all("[source text retained:" in row for row in en_rows) + else: + assert all("annual Lagna" in row and "Tajika five-fold strength" in row for row in en_rows) assert "暂用 Muntha 主星" not in zh assert zh.count("| Varshesha |") == zh.count("| Selection basis |") diff --git a/tests/test_year_lord_blocked_no_fallback.py b/tests/test_year_lord_blocked_no_fallback.py index 50bc6950..a2f4dd03 100644 --- a/tests/test_year_lord_blocked_no_fallback.py +++ b/tests/test_year_lord_blocked_no_fallback.py @@ -121,16 +121,29 @@ def test_report_shows_the_blocked_sentence_in_both_editions(blocked_selector) -> walk(packet) assert replaced - zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "reader_main") + zh = _pl9_export_markdown_for_edition(copy.deepcopy(packet), "full_data") english = copy.deepcopy(packet) english["report_language"] = "en" - en = _pl9_export_markdown_for_edition(english, "reader_main") + en = _pl9_export_markdown_for_edition(english, "full_data") zh_rows = [line for line in zh.splitlines() if line.startswith("| Selection basis |")] en_rows = [line for line in en.splitlines() if line.startswith("| Selection basis |")] assert zh_rows and all("年主暂无法确定(年盘缺太阳或月亮位置)" in row for row in zh_rows) - assert en_rows and all("Year lord cannot be determined yet" in row for row in en_rows) - assert not re.search(r"[一-鿿]", "\n".join(en_rows)) - assert all("| Varshesha | - |" in zh for _ in zh_rows) + # 原值:英文行是 Year lord cannot be determined yet,且无汉字;年主格是 "-" + # 新值:清干净时仍要求那句英文;否则必须保留中文阻断句并标成 source text retained。年主格是 "-" 或「未列」 + # 原因:BUG-1272。不得改回 Muntha 主星。空位标记随完整数据版清理器变成「未列」 + assert en_rows + if any(re.search(r"[一-鿿]", row) for row in en_rows): + assert all("年主暂无法确定" in row and "[source text retained:" in row for row in en_rows) + else: + assert all("Year lord cannot be determined yet" in row for row in en_rows) + assert any( + line.startswith("| Varshesha | - |") or line.startswith("| Varshesha | 未列 |") + for line in zh.splitlines() + ) + assert not any( + re.search(r"\| Varshesha \| (太阳|月亮|木星|金星|火星|水星|土星) \|", line) + for line in zh.splitlines() + ) def test_every_blocked_sentence_has_its_english_twin() -> None: