feat(research): consult biography backtest harness and pre-change baseline
Independent Staging Quality Gate / validate (push) Successful in 16m20s
Independent Staging Quality Gate / publish (push) Successful in 3m54s

TASK-consult-no-presupposition-and-backtest-20261001 T4 infrastructure only
(T1-T3 untouched). New files only, so it merges cleanly with the parallel
evidence-card brief.

- capture_consult_biography_backtest_golden.py: same handler/body/trim as the
  evidence-card golden; nine public AA charts x parents/marriage/health/career.
- consult-biography-backtest-golden.json: real engine output, byte-reproducible.
- consult_biography_backtest_rubric.json: facts with sources, must_not,
  expected_signals per figure x domain.
- consult-biography-backtest.mts: runs the product's real agent, tools, card,
  methodology, user-turn shape and streamAgentResponse; deterministic checks only.
- Baseline on 9b937c4a: 39/72 severe biography conflicts; all five audit cases
  reproduce in both runs.

BUG-1170 is registered when brief 3 lands.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
This commit is contained in:
Jesse_Chen
2026-10-02 00:10:06 +08:00
co-authored by Claude Opus 5.5
parent e5199106a7
commit 691ee440e3
6 changed files with 2011 additions and 0 deletions
@@ -0,0 +1,179 @@
"""Capture real engine consultation responses for the biography backtest.
Same path as `capture_consult_evidence_card_golden.py` (same handler, same
request body, same research reference date, raman ayanamsa, mean nodes, the
external VedAstro stand-in), extended to nine public Rodden AA charts and to the
four engine routes the backtest asks about: `family` (the engine route the
product's `parents` domain runs as), `marriage`, `health` and `career`. Each
response is trimmed by key only with the same `trim()` as the evidence-card
golden: every kept value is the engine's own value, unchanged.
The chart block (`workflow.chart`) is almost route-independent: the keys all
four routes agree on are stored once per figure as `shared_chart`, and the
keys that differ (today `modules.yogas` and `modules.chara_dasha`) stay with
each route under `chart` / `chart_modules`. The TypeScript reader overlays
them back (`frontend/scripts/research/consult-biography-backtest.mts`), so
every route's chart is exactly what the engine returned.
JYOTISH_API_CHART_CACHE_TTL_SECONDS=0 PYTHONHASHSEED=0 \
python3 scripts/research/capture_consult_biography_backtest_golden.py
Birth data only from the repository's public case libraries (Astro-Databank
AA); no user chart is ever read.
"""
from __future__ import annotations
import json
import os
import sys
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[2]
sys.path[:0] = [str(ROOT), str(ROOT / "scripts"), str(ROOT / "scripts" / "research")]
from consult_evidence_card_lib import AYANAMSA, NODE_MODE, REFERENCE_DATE # noqa: E402
import consult_evidence_card_run as runner # noqa: E402
from capture_consult_evidence_card_golden import trim # noqa: E402
OUT = ROOT / "frontend" / "tests" / "fixtures" / "consult-biography-backtest-golden.json"
CASES = "references/real_case_calibration"
# id, label, case library, case id. The rubric
# (docs/research/consult_biography_backtest_rubric.json) carries why each is here.
FIGURES = (
("steve_jobs", "Steve Jobs", f"{CASES}/minute_rectification_development_v1.json", "steve_jobs_1955_development"),
("barack_obama", "Barack Obama", f"{CASES}/minute_rectification_holdout_v4.json", "barack_obama_1961_aa_v4_holdout"),
("elizabeth_taylor", "Elizabeth Taylor", f"{CASES}/minute_rectification_holdout_v4.json", "elizabeth_taylor_1932_aa_v4_holdout"),
("marilyn_monroe", "Marilyn Monroe", f"{CASES}/minute_rectification_holdout_v5.json", "marilyn_monroe_1926_aa_v5_holdout"),
("judy_garland", "Judy Garland", f"{CASES}/minute_rectification_holdout_v5.json", "judy_garland_1922_aa_v5_holdout"),
("edith_piaf", "Édith Piaf", f"{CASES}/minute_rectification_holdout_v5.json", "edith_piaf_1915_aa_v5_holdout"),
("frida_kahlo", "Frida Kahlo", f"{CASES}/minute_rectification_holdout_v5.json", "frida_kahlo_1907_aa_v5_holdout"),
("george_w_bush", "George W. Bush", f"{CASES}/minute_rectification_holdout_v5.json", "george_w_bush_1946_aa_v5_holdout"),
("zinedine_zidane", "Zinedine Zidane", f"{CASES}/minute_rectification_holdout_v5.json", "zinedine_zidane_1972_aa_v5_holdout"),
)
# Product domain -> engine route and the question the backtest asks. The
# product sends a card-only domain (parents) as its engine route's contract
# (frontend/src/lib/consultation-workflow-request.ts).
ROUTES = (
{"domain": "parents", "engine_route": "family", "question": "我和父母关系如何,他们怎么对待我?"},
{"domain": "marriage", "engine_route": "marriage", "question": "我的婚姻和感情会是什么样?"},
{"domain": "health", "engine_route": "health", "question": "我的身体底子怎么样,健康上要注意什么?"},
{"domain": "career", "engine_route": "career", "question": "我的事业会往什么方向走,能做到什么程度?"},
)
def load_chart(spec: tuple[str, str, str, str]) -> dict[str, Any]:
figure_id, label, path, case_id = spec
payload = json.loads((ROOT / path).read_text(encoding="utf-8"))
case = next(item for item in payload["cases"] if item["case_id"] == case_id)
birth = case["birth"]
year, month, day = (int(part) for part in birth["date"].split("-"))
hour, minute = (int(part) for part in birth["time"].split(":")[:2])
source = birth.get("source") or {}
return {
"id": figure_id,
"label": label,
"source": path,
"case_id": case_id,
"rodden_rating": source.get("rodden_rating"),
"birth_source_url": source.get("url"),
# Same body as consult_evidence_card_lib.load_public_charts.
"body": {
"year": year,
"month": month,
"day": day,
"hour": hour,
"minute": minute,
"second": 0,
"lat": birth["latitude"],
"lon": birth["longitude"],
"tz": birth["timezone_offset"],
"city": birth.get("place") or label,
"ayanamsa": AYANAMSA,
"node_mode": NODE_MODE,
"today": REFERENCE_DATE,
"entry_mode": "direct_chart",
"defer_optional_external_evidence": True,
"declared_accuracy": "minute",
"birth_time_accuracy": "confirmed",
},
}
def split_shared_chart(routes: dict[str, dict[str, Any]]) -> dict[str, Any]:
"""Lift the chart keys every route agrees on into one copy.
A key (or a `modules` key) whose value differs between routes stays with
each route under `chart` (or `chart_modules`); the reader overlays it on
the shared copy, so every route's chart comes back exactly as captured.
"""
charts = [value.pop("chart") for value in routes.values()]
shared: dict[str, Any] = {}
keys = {key for chart in charts for key in chart if key != "modules"}
for key in sorted(keys):
values = [chart.get(key) for chart in charts]
if all(key in chart for chart in charts) and all(item == values[0] for item in values):
shared[key] = values[0]
else:
for value, chart in zip(routes.values(), charts):
if key in chart:
value.setdefault("chart", {})[key] = chart[key]
modules = [chart.get("modules") or {} for chart in charts]
shared["modules"] = {}
for key in sorted({key for item in modules for key in item}):
values = [item.get(key) for item in modules]
if all(key in item for item in modules) and all(entry == values[0] for entry in values):
shared["modules"][key] = values[0]
else:
for value, item in zip(routes.values(), modules):
if key in item:
value.setdefault("chart_modules", {})[key] = item[key]
return shared
def main() -> int:
if os.environ.get("PYTHONHASHSEED") != "0":
raise SystemExit("Set PYTHONHASHSEED=0 before starting this process (ERR-111).")
only = {item for item in os.environ.get("BACKTEST_FIGURES", "").split(",") if item}
runner._block_external_vedastro()
figures = []
for spec in FIGURES:
if only and spec[0] not in only:
continue
chart = load_chart(spec)
routes: dict[str, dict[str, Any]] = {}
for route in ROUTES:
question = {"id": route["domain"], "domain": route["engine_route"], "question": route["question"]}
routes[route["domain"]] = trim(runner._run_workflow(runner._workflow_body(chart, question)))
print(f"{chart['id']} {route['domain']} ok", flush=True)
shared = split_shared_chart(routes)
figures.append({
"id": chart["id"],
"label": chart["label"],
"source": chart["source"],
"case_id": chart["case_id"],
"rodden_rating": chart["rodden_rating"],
"birth_source_url": chart["birth_source_url"],
"shared_chart": shared,
"routes": routes,
})
payload = {
"source": "scripts/research/capture_consult_biography_backtest_golden.py",
"note": "Real engine consultation_workflow responses for nine public AA charts x four routes, trimmed by key only (same trim as consult-evidence-card-golden.json).",
"reference_date": REFERENCE_DATE,
"ayanamsa": AYANAMSA,
"node_mode": NODE_MODE,
"routes": list(ROUTES),
"external_vedastro": "not_called",
"figures": figures,
}
OUT.write_text(json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n", encoding="utf-8")
print(f"wrote {OUT} ({OUT.stat().st_size} bytes)")
return 0
if __name__ == "__main__":
raise SystemExit(main())