Files
Jyotisha/scripts/research/fewer_probes_card_replay.py
T
Jesse_ChenandClaude Opus 5.5 4e6e8d87ef feat(rectification): 少问几道就出卡,交付卡以范围为主(D1–D4)
- D1 引导窗口题每个校正最多 2 道(前端 GUIDED_WINDOW_CASE_LIMIT;引擎
  GUIDED_COLLECT_LIMIT 未动:event_probes.py 属冻结评分身份,改它需重新冻结)
- D2 七条定向线与跳过线重问问完即出卡,没问到的引导窗口不再挡卡,出卡后也不再挂窗口题
- D3 卡头加副标题「最可能 HH:MM」
- D4 前两列相差 ≥5 个百分点才显示相对可能性,否则一句「这几个时刻目前区分不开……」
- 离线回放 scripts/research/fewer_probes_card_replay.py:真值不降、宽度中位 ±1 分钟、提问 11.4→7.8
- Skill 10.0.29 → 10.0.30;DESIGN / VOICE / CHANGELOG / PROGRESS / 真机清单

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-26 14:27:12 +08:00

275 lines
12 KiB
Python

#!/usr/bin/env python3
"""Offline replay for TASK-rectification-fewer-probes-card-20260926 (D1/D2).
Same method as `guided_collect_holdout_replay.py` (the script behind
`docs/research/guided_collect_holdout_2026_09_16.json`): v4 open holdout
(public AA cases), six discriminating probes answered from the true minute,
then guided window events injected on the true candidate's own boundary date
(`truth`) or on the furthest remaining candidate's (`opposite`, control).
What differs is the question being asked. The 09-16 replay stopped at the
first injection that met the precision gate ("how many events to the gate").
This replay models the production delivery rule on both sides and compares
the **final delivered range**:
* before — `GUIDED_COLLECT_LIMIT = 6` and the card waits until the guided pool
is asked out (BUG-751 D6): every window in the receipt is asked.
* after — at most two guided windows per Case (D1, enforced in the frontend
pool as `GUIDED_WINDOW_CASE_LIMIT = 2`; the engine list stays at
`GUIDED_COLLECT_LIMIT = 6` because `event_probes.py` is part of the frozen
scoring identity); once the targeted seven and their one re-ask are asked
the card is delivered whether or not the gate is met (D2). Windows are asked
before the targeted lines, so the first two windows of the receipt are
still asked.
The targeted seven lines are identical on both sides and are not modelled; the
question count below is six probes plus the guided windows asked.
Metrics per radius: truth-in-delivered-range rate, median delivered width,
mean questions asked. Not a merge gate by itself; the numbers go in
`docs/tasks/PROGRESS-rectification-fewer-probes-card-20260926.md`.
"""
from __future__ import annotations
import argparse
import json
import statistics
import sys
import time
import traceback
from datetime import date
from pathlib import Path
from typing import Any, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.active_rectification_event_engine import ( # noqa: E402
AYANAMSA,
NODE_MODE,
compute_candidate_static_contexts,
)
from scripts.rectification.event_probes import ( # noqa: E402
GUIDED_COLLECT_LIMIT,
discriminating_event_probes,
guided_collect_windows,
)
from scripts.rectification.refinement_packet import window_scan # noqa: E402
from scripts.rectification.scoring_service import ( # noqa: E402
build_event_contribution_matrix,
score_from_matrix,
scoreable_request,
)
from scripts.research.guided_collect_holdout_replay import ( # noqa: E402
RADII,
TODAY,
_hhmm,
_minutes,
load_cases,
posterior_state,
precision_gate,
remaining_times,
synthetic_event,
window_boundary_dates,
)
from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_request_for # noqa: E402
from scripts.research.probe_supply_after_six import ASK_COUNT # noqa: E402
REPORT_JSON = ROOT / "docs" / "research" / "fewer_probes_card_replay_2026_09_26.json"
#: Engine receipt cap (unchanged) and the frontend per-Case cap (D1).
BEFORE_LIMIT = GUIDED_COLLECT_LIMIT
AFTER_LIMIT = 2 # frontend `GUIDED_WINDOW_CASE_LIMIT`
def _outcome(state: dict[str, Any], true_time: str) -> dict[str, Any]:
delivery = state["delivery"]
start, end = delivery.get("start"), delivery.get("end")
inside = (
start is not None
and end is not None
and _minutes(start) <= _minutes(true_time) <= _minutes(end)
)
gate = precision_gate(state["valid"], state["scores"])
return {
"start": start,
"end": end,
"width": delivery.get("width"),
"truth_in_range": bool(inside),
"truth_eliminated": true_time in state["eliminated"],
"gate_met": bool(gate["met"]),
"gap": gate["gap"],
"percents": gate["percents"],
}
def evaluate_case(case: dict[str, Any], radius: int, direction: str) -> dict[str, Any]:
true_time = str(case["birth"]["time"])[:5]
request = scoring_request_for(case, radius)
request["ayanamsa"] = AYANAMSA
request["node_mode"] = NODE_MODE
request["minute_step"] = MINUTE_STEP
static_contexts = compute_candidate_static_contexts(request)
built = build_event_contribution_matrix(request, static_contexts=static_contexts)
rows = score_from_matrix(request, built)
times = [stamp for row in rows if (stamp := _hhmm(row.get("time")))]
probes = discriminating_event_probes(
{**request, "refresh_probes": False, "asked_probe_keys": []},
built,
scan=window_scan(built),
candidate_times=times,
representative_time=true_time,
today=TODAY,
)[:ASK_COUNT]
state = posterior_state(rows=rows, contexts=static_contexts, probes=probes, true_time=true_time)
after_six = _outcome(state, true_time)
pool = remaining_times(state) or times
windows_before = guided_collect_windows(request, built, candidate_times=pool, today=TODAY)
# D1: the Case asks the first two windows of the receipt, in receipt order.
windows_after = windows_before[:AFTER_LIMIT]
boundaries = window_boundary_dates(request, built, candidate_times=pool, today=TODAY)
if direction == "truth":
source_time = true_time
else:
source_time = max(pool, key=lambda stamp: abs(_minutes(stamp) - _minutes(true_time))) if pool else true_time
def inject(windows: Sequence[dict[str, Any]]) -> dict[str, Any]:
extras: list[dict[str, Any]] = []
for index, window in enumerate(windows):
key = (int(window["year"]), int(window["month_lo"]), int(window["month_hi"]))
per_time = boundaries.get(key) or {}
when = per_time.get(source_time)
if when is None and per_time:
when = per_time[min(per_time, key=lambda stamp: abs(_minutes(stamp) - _minutes(source_time)))]
if when is None:
when = date(int(window["year"]), int(window["month_lo"]), 1)
extras.append(synthetic_event(window, index, when=when))
if not extras:
return after_six
injected = scoreable_request({**request, "events": list(request["events"]) + extras})
rebuilt = build_event_contribution_matrix(injected, static_contexts=static_contexts)
new_rows = score_from_matrix(injected, rebuilt)
final = posterior_state(rows=new_rows, contexts=static_contexts, probes=probes, true_time=true_time)
return _outcome(final, true_time)
before = inject(windows_before)
after = inject(windows_after)
return {
"case_id": case.get("case_id"),
"radius": radius,
"direction": direction,
"windows_before": len(windows_before),
"windows_after": len(windows_after),
"questions_before": ASK_COUNT + len(windows_before),
"questions_after": ASK_COUNT + len(windows_after),
"after_six": after_six,
"before": before,
"after": after,
"same_range": (before["start"], before["end"]) == (after["start"], after["end"]),
}
def _median(values: Sequence[float]) -> float | None:
return statistics.median(values) if values else None
def summarize(rows: Sequence[dict[str, Any]], radius: int, direction: str) -> dict[str, Any]:
subset = [
row for row in rows
if row.get("radius") == radius and row.get("direction") == direction and not row.get("error")
]
n = len(subset)
def side(name: str) -> dict[str, Any]:
widths = [row[name]["width"] for row in subset if row[name]["width"] is not None]
return {
"truth_in_range": sum(1 for row in subset if row[name]["truth_in_range"]),
"truth_in_range_rate": round(sum(1 for row in subset if row[name]["truth_in_range"]) / n, 4) if n else None,
"median_width": _median(widths),
"gate_met": sum(1 for row in subset if row[name]["gate_met"]),
}
return {
"radius": radius,
"direction": direction,
"n": n,
"after_six": side("after_six"),
"before": {
**side("before"),
"mean_questions": round(statistics.mean(row["questions_before"] for row in subset), 2) if n else None,
"mean_guided": round(statistics.mean(row["windows_before"] for row in subset), 2) if n else None,
},
"after": {
**side("after"),
"mean_questions": round(statistics.mean(row["questions_after"] for row in subset), 2) if n else None,
"mean_guided": round(statistics.mean(row["windows_after"] for row in subset), 2) if n else None,
},
"same_range_cases": sum(1 for row in subset if row["same_range"]),
"errors": sum(
1 for row in rows
if row.get("radius") == radius and row.get("direction") == direction and row.get("error")
),
}
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--radii", default=",".join(str(item) for item in RADII))
parser.add_argument("--directions", default="truth,opposite")
parser.add_argument("--json-out", default=str(REPORT_JSON))
args = parser.parse_args()
radii = tuple(int(item) for item in str(args.radii).split(",") if item.strip())
directions = tuple(item.strip() for item in str(args.directions).split(",") if item.strip())
cases = load_cases()
if args.limit:
cases = cases[: args.limit]
started = time.perf_counter()
rows: list[dict[str, Any]] = []
for case in cases:
for radius in radii:
for direction in directions:
label = f"{case.get('case_id')} ±{radius} {direction}"
try:
result = evaluate_case(case, radius, direction)
rows.append(result)
print(
f"{label} q={result['questions_before']}->{result['questions_after']} "
f"in={result['before']['truth_in_range']}->{result['after']['truth_in_range']} "
f"w={result['before']['width']}->{result['after']['width']} same={result['same_range']}",
flush=True,
)
except Exception as exc: # noqa: BLE001
rows.append({
"case_id": case.get("case_id"),
"radius": radius,
"direction": direction,
"error": f"{type(exc).__name__}: {exc}",
"trace": traceback.format_exc(limit=8),
})
print(f"{label} ERROR {type(exc).__name__}: {exc}", flush=True)
summaries = [summarize(rows, radius, direction) for radius in radii for direction in directions]
payload = {
"generated_at": TODAY.isoformat(),
"method": "guided_collect_holdout_replay.py (2026-09-16), final delivered range compared",
"holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json",
"ask_count": ASK_COUNT,
"guided_collect_limit_before": BEFORE_LIMIT,
"guided_window_case_limit_after": AFTER_LIMIT,
"elapsed_s": round(time.perf_counter() - started, 1),
"summaries": summaries,
"rows": [{key: value for key, value in row.items() if key != "trace"} for row in rows],
"errors": [row for row in rows if row.get("error")],
}
out = Path(args.json_out)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps(summaries, ensure_ascii=False, indent=2), flush=True)
print(f"wrote {out} in {payload['elapsed_s']}s", flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())