feat(research): consult biography backtest harness and pre-change baseline
TASK-consult-no-presupposition-and-backtest-20261001 T4 infrastructure only
(T1-T3 untouched). New files only, so it merges cleanly with the parallel
evidence-card brief.
- capture_consult_biography_backtest_golden.py: same handler/body/trim as the
evidence-card golden; nine public AA charts x parents/marriage/health/career.
- consult-biography-backtest-golden.json: real engine output, byte-reproducible.
- consult_biography_backtest_rubric.json: facts with sources, must_not,
expected_signals per figure x domain.
- consult-biography-backtest.mts: runs the product's real agent, tools, card,
methodology, user-turn shape and streamAgentResponse; deterministic checks only.
- Baseline on 9b937c4a: 39/72 severe biography conflicts; all five audit cases
reproduce in both runs.
BUG-1170 is registered when brief 3 lands.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
e5199106a7
commit
691ee440e3
@@ -0,0 +1,179 @@
|
||||
"""Capture real engine consultation responses for the biography backtest.
|
||||
|
||||
Same path as `capture_consult_evidence_card_golden.py` (same handler, same
|
||||
request body, same research reference date, raman ayanamsa, mean nodes, the
|
||||
external VedAstro stand-in), extended to nine public Rodden AA charts and to the
|
||||
four engine routes the backtest asks about: `family` (the engine route the
|
||||
product's `parents` domain runs as), `marriage`, `health` and `career`. Each
|
||||
response is trimmed by key only with the same `trim()` as the evidence-card
|
||||
golden: every kept value is the engine's own value, unchanged.
|
||||
|
||||
The chart block (`workflow.chart`) is almost route-independent: the keys all
|
||||
four routes agree on are stored once per figure as `shared_chart`, and the
|
||||
keys that differ (today `modules.yogas` and `modules.chara_dasha`) stay with
|
||||
each route under `chart` / `chart_modules`. The TypeScript reader overlays
|
||||
them back (`frontend/scripts/research/consult-biography-backtest.mts`), so
|
||||
every route's chart is exactly what the engine returned.
|
||||
|
||||
JYOTISH_API_CHART_CACHE_TTL_SECONDS=0 PYTHONHASHSEED=0 \
|
||||
python3 scripts/research/capture_consult_biography_backtest_golden.py
|
||||
|
||||
Birth data only from the repository's public case libraries (Astro-Databank
|
||||
AA); no user chart is ever read.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path[:0] = [str(ROOT), str(ROOT / "scripts"), str(ROOT / "scripts" / "research")]
|
||||
|
||||
from consult_evidence_card_lib import AYANAMSA, NODE_MODE, REFERENCE_DATE # noqa: E402
|
||||
import consult_evidence_card_run as runner # noqa: E402
|
||||
from capture_consult_evidence_card_golden import trim # noqa: E402
|
||||
|
||||
OUT = ROOT / "frontend" / "tests" / "fixtures" / "consult-biography-backtest-golden.json"
|
||||
CASES = "references/real_case_calibration"
|
||||
|
||||
# id, label, case library, case id. The rubric
|
||||
# (docs/research/consult_biography_backtest_rubric.json) carries why each is here.
|
||||
FIGURES = (
|
||||
("steve_jobs", "Steve Jobs", f"{CASES}/minute_rectification_development_v1.json", "steve_jobs_1955_development"),
|
||||
("barack_obama", "Barack Obama", f"{CASES}/minute_rectification_holdout_v4.json", "barack_obama_1961_aa_v4_holdout"),
|
||||
("elizabeth_taylor", "Elizabeth Taylor", f"{CASES}/minute_rectification_holdout_v4.json", "elizabeth_taylor_1932_aa_v4_holdout"),
|
||||
("marilyn_monroe", "Marilyn Monroe", f"{CASES}/minute_rectification_holdout_v5.json", "marilyn_monroe_1926_aa_v5_holdout"),
|
||||
("judy_garland", "Judy Garland", f"{CASES}/minute_rectification_holdout_v5.json", "judy_garland_1922_aa_v5_holdout"),
|
||||
("edith_piaf", "Édith Piaf", f"{CASES}/minute_rectification_holdout_v5.json", "edith_piaf_1915_aa_v5_holdout"),
|
||||
("frida_kahlo", "Frida Kahlo", f"{CASES}/minute_rectification_holdout_v5.json", "frida_kahlo_1907_aa_v5_holdout"),
|
||||
("george_w_bush", "George W. Bush", f"{CASES}/minute_rectification_holdout_v5.json", "george_w_bush_1946_aa_v5_holdout"),
|
||||
("zinedine_zidane", "Zinedine Zidane", f"{CASES}/minute_rectification_holdout_v5.json", "zinedine_zidane_1972_aa_v5_holdout"),
|
||||
)
|
||||
|
||||
# Product domain -> engine route and the question the backtest asks. The
|
||||
# product sends a card-only domain (parents) as its engine route's contract
|
||||
# (frontend/src/lib/consultation-workflow-request.ts).
|
||||
ROUTES = (
|
||||
{"domain": "parents", "engine_route": "family", "question": "我和父母关系如何,他们怎么对待我?"},
|
||||
{"domain": "marriage", "engine_route": "marriage", "question": "我的婚姻和感情会是什么样?"},
|
||||
{"domain": "health", "engine_route": "health", "question": "我的身体底子怎么样,健康上要注意什么?"},
|
||||
{"domain": "career", "engine_route": "career", "question": "我的事业会往什么方向走,能做到什么程度?"},
|
||||
)
|
||||
|
||||
|
||||
def load_chart(spec: tuple[str, str, str, str]) -> dict[str, Any]:
|
||||
figure_id, label, path, case_id = spec
|
||||
payload = json.loads((ROOT / path).read_text(encoding="utf-8"))
|
||||
case = next(item for item in payload["cases"] if item["case_id"] == case_id)
|
||||
birth = case["birth"]
|
||||
year, month, day = (int(part) for part in birth["date"].split("-"))
|
||||
hour, minute = (int(part) for part in birth["time"].split(":")[:2])
|
||||
source = birth.get("source") or {}
|
||||
return {
|
||||
"id": figure_id,
|
||||
"label": label,
|
||||
"source": path,
|
||||
"case_id": case_id,
|
||||
"rodden_rating": source.get("rodden_rating"),
|
||||
"birth_source_url": source.get("url"),
|
||||
# Same body as consult_evidence_card_lib.load_public_charts.
|
||||
"body": {
|
||||
"year": year,
|
||||
"month": month,
|
||||
"day": day,
|
||||
"hour": hour,
|
||||
"minute": minute,
|
||||
"second": 0,
|
||||
"lat": birth["latitude"],
|
||||
"lon": birth["longitude"],
|
||||
"tz": birth["timezone_offset"],
|
||||
"city": birth.get("place") or label,
|
||||
"ayanamsa": AYANAMSA,
|
||||
"node_mode": NODE_MODE,
|
||||
"today": REFERENCE_DATE,
|
||||
"entry_mode": "direct_chart",
|
||||
"defer_optional_external_evidence": True,
|
||||
"declared_accuracy": "minute",
|
||||
"birth_time_accuracy": "confirmed",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def split_shared_chart(routes: dict[str, dict[str, Any]]) -> dict[str, Any]:
|
||||
"""Lift the chart keys every route agrees on into one copy.
|
||||
|
||||
A key (or a `modules` key) whose value differs between routes stays with
|
||||
each route under `chart` (or `chart_modules`); the reader overlays it on
|
||||
the shared copy, so every route's chart comes back exactly as captured.
|
||||
"""
|
||||
charts = [value.pop("chart") for value in routes.values()]
|
||||
shared: dict[str, Any] = {}
|
||||
keys = {key for chart in charts for key in chart if key != "modules"}
|
||||
for key in sorted(keys):
|
||||
values = [chart.get(key) for chart in charts]
|
||||
if all(key in chart for chart in charts) and all(item == values[0] for item in values):
|
||||
shared[key] = values[0]
|
||||
else:
|
||||
for value, chart in zip(routes.values(), charts):
|
||||
if key in chart:
|
||||
value.setdefault("chart", {})[key] = chart[key]
|
||||
modules = [chart.get("modules") or {} for chart in charts]
|
||||
shared["modules"] = {}
|
||||
for key in sorted({key for item in modules for key in item}):
|
||||
values = [item.get(key) for item in modules]
|
||||
if all(key in item for item in modules) and all(entry == values[0] for entry in values):
|
||||
shared["modules"][key] = values[0]
|
||||
else:
|
||||
for value, item in zip(routes.values(), modules):
|
||||
if key in item:
|
||||
value.setdefault("chart_modules", {})[key] = item[key]
|
||||
return shared
|
||||
|
||||
|
||||
def main() -> int:
|
||||
if os.environ.get("PYTHONHASHSEED") != "0":
|
||||
raise SystemExit("Set PYTHONHASHSEED=0 before starting this process (ERR-111).")
|
||||
only = {item for item in os.environ.get("BACKTEST_FIGURES", "").split(",") if item}
|
||||
runner._block_external_vedastro()
|
||||
figures = []
|
||||
for spec in FIGURES:
|
||||
if only and spec[0] not in only:
|
||||
continue
|
||||
chart = load_chart(spec)
|
||||
routes: dict[str, dict[str, Any]] = {}
|
||||
for route in ROUTES:
|
||||
question = {"id": route["domain"], "domain": route["engine_route"], "question": route["question"]}
|
||||
routes[route["domain"]] = trim(runner._run_workflow(runner._workflow_body(chart, question)))
|
||||
print(f"{chart['id']} {route['domain']} ok", flush=True)
|
||||
shared = split_shared_chart(routes)
|
||||
figures.append({
|
||||
"id": chart["id"],
|
||||
"label": chart["label"],
|
||||
"source": chart["source"],
|
||||
"case_id": chart["case_id"],
|
||||
"rodden_rating": chart["rodden_rating"],
|
||||
"birth_source_url": chart["birth_source_url"],
|
||||
"shared_chart": shared,
|
||||
"routes": routes,
|
||||
})
|
||||
payload = {
|
||||
"source": "scripts/research/capture_consult_biography_backtest_golden.py",
|
||||
"note": "Real engine consultation_workflow responses for nine public AA charts x four routes, trimmed by key only (same trim as consult-evidence-card-golden.json).",
|
||||
"reference_date": REFERENCE_DATE,
|
||||
"ayanamsa": AYANAMSA,
|
||||
"node_mode": NODE_MODE,
|
||||
"routes": list(ROUTES),
|
||||
"external_vedastro": "not_called",
|
||||
"figures": figures,
|
||||
}
|
||||
OUT.write_text(json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n", encoding="utf-8")
|
||||
print(f"wrote {OUT} ({OUT.stat().st_size} bytes)")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user