#!/usr/bin/env python3 """Offline replay for TASK-rectification-futile-collect-stop-20260929 (D1). Hard red line 1: after the training gate opens, the Case no longer injects spoken / guided events (targeted seven, skip re-ask, guided boundary windows). Same method as `fewer_probes_card_replay.py` (v4 open holdout, six discriminating probes answered from the true minute, guided windows injected on the true candidate's own boundary date (`truth`) or on the furthest remaining candidate's (`opposite`, control)). * baseline — production before D1: the first `GUIDED_WINDOW_CASE_LIMIT = 2` guided windows of the receipt are asked and injected (`after` in `fewer_probes_card_replay.py`). * d1 — nothing is injected after the six probes (`after_six`). The targeted seven lines ask for events whose dates only a real person knows; they cannot be modelled on the open holdout and are left out on both sides (same as the 09-26 replay). This is an open-set replay, not a blind test. Gate: per radius x direction, d1 truth-in-range count >= baseline and d1 median width <= baseline. Writes `docs/research/futile_collect_stop_replay_2026_09_29.json`. """ from __future__ import annotations import argparse import json import statistics import sys import time import traceback from pathlib import Path from typing import Any, Sequence ROOT = Path(__file__).resolve().parents[2] if str(ROOT) not in sys.path: sys.path.insert(0, str(ROOT)) from scripts.research.fewer_probes_card_replay import evaluate_case # noqa: E402 from scripts.research.guided_collect_holdout_replay import RADII, load_cases # noqa: E402 REPORT_JSON = ROOT / "docs" / "research" / "futile_collect_stop_replay_2026_09_29.json" def _median(values: Sequence[float]) -> float | None: return statistics.median(values) if values else None def summarize(rows: Sequence[dict[str, Any]], radius: int, direction: str) -> dict[str, Any]: subset = [ row for row in rows if row.get("radius") == radius and row.get("direction") == direction and not row.get("error") ] n = len(subset) def side(name: str) -> dict[str, Any]: widths = [row[name]["width"] for row in subset if row[name]["width"] is not None] inside = sum(1 for row in subset if row[name]["truth_in_range"]) return { "truth_in_range": inside, "truth_in_range_rate": round(inside / n, 4) if n else None, "median_width": _median(widths), } baseline = side("after") d1 = side("after_six") passed = ( n > 0 and d1["truth_in_range"] >= baseline["truth_in_range"] and d1["median_width"] is not None and baseline["median_width"] is not None and d1["median_width"] <= baseline["median_width"] ) return { "radius": radius, "direction": direction, "n": n, "baseline": { **baseline, "mean_injected": round(statistics.mean(row["windows_after"] for row in subset), 2) if n else None, }, "d1": {**d1, "mean_injected": 0.0}, "width_worse_cases": sum( 1 for row in subset if row["after_six"]["width"] is not None and row["after"]["width"] is not None and row["after_six"]["width"] > row["after"]["width"] ), "truth_lost_cases": sum( 1 for row in subset if row["after"]["truth_in_range"] and not row["after_six"]["truth_in_range"] ), "gate_pass": passed, "errors": sum( 1 for row in rows if row.get("radius") == radius and row.get("direction") == direction and row.get("error") ), } def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--limit", type=int, default=0) parser.add_argument("--radii", default=",".join(str(item) for item in RADII)) parser.add_argument("--directions", default="truth,opposite") parser.add_argument("--json-out", default=str(REPORT_JSON)) args = parser.parse_args() radii = tuple(int(item) for item in str(args.radii).split(",") if item.strip()) directions = tuple(item.strip() for item in str(args.directions).split(",") if item.strip()) cases = load_cases() if args.limit: cases = cases[: args.limit] started = time.perf_counter() rows: list[dict[str, Any]] = [] for case in cases: for radius in radii: for direction in directions: label = f"{case.get('case_id')} ±{radius} {direction}" try: result = evaluate_case(case, radius, direction) rows.append(result) print( f"{label} baseline w={result['after']['width']} in={result['after']['truth_in_range']} " f"d1 w={result['after_six']['width']} in={result['after_six']['truth_in_range']}", flush=True, ) except Exception as exc: # noqa: BLE001 rows.append({ "case_id": case.get("case_id"), "radius": radius, "direction": direction, "error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc(limit=8), }) print(f"{label} ERROR {type(exc).__name__}: {exc}", flush=True) summaries = [summarize(rows, radius, direction) for radius in radii for direction in directions] payload = { "method": "fewer_probes_card_replay.evaluate_case; baseline=after (2 guided windows injected), d1=after_six (none)", "holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json", "elapsed_s": round(time.perf_counter() - started, 1), "all_cells_pass": all(item["gate_pass"] for item in summaries), "summaries": summaries, "rows": [ { "case_id": row.get("case_id"), "radius": row.get("radius"), "direction": row.get("direction"), **({"error": row["error"]} if row.get("error") else { "baseline": {key: row["after"][key] for key in ("start", "end", "width", "truth_in_range")}, "d1": {key: row["after_six"][key] for key in ("start", "end", "width", "truth_in_range")}, "windows_injected": row["windows_after"], }), } for row in rows ], } out = Path(args.json_out) out.parent.mkdir(parents=True, exist_ok=True) out.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") print(json.dumps(summaries, ensure_ascii=False, indent=2), flush=True) print(f"wrote {out} in {payload['elapsed_s']}s all_cells_pass={payload['all_cells_pass']}", flush=True) return 0 if __name__ == "__main__": raise SystemExit(main())