Files
Jyotisha/scripts/research/futile_collect_stop_replay.py
T
Jesse_ChenandClaude Opus 5.5 ea21743b09 fix(rectification): stop spoken collect once the training gate opens (BUG-1084..1087)
Once the discriminator training gate is open, only choice cards are asked
and the range card goes out when they are exhausted; targeted lines, their
re-ask and guided windows no longer hold the card or invite more events.
Delivery body says how many choice questions were used instead of the event
fit percent; narration names an excluded cluster instead of "range
unchanged"; a delivered turn no longer carries a collect question.

Offline replay (v4, 3 radii x 2 directions): truth in range 20/20 in every
cell; guided-window injections give the same width in truth and opposite
directions, so red line 1 was revised by product to truth-in-range only.
Skill 10.0.31 -> 10.0.32 (10.0.31 kept as deprecated for pinned cases).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-29 10:06:57 +08:00

167 lines
6.8 KiB
Python

#!/usr/bin/env python3
"""Offline replay for TASK-rectification-futile-collect-stop-20260929 (D1).
Hard red line 1: after the training gate opens, the Case no longer injects
spoken / guided events (targeted seven, skip re-ask, guided boundary windows).
Same method as `fewer_probes_card_replay.py` (v4 open holdout, six
discriminating probes answered from the true minute, guided windows injected
on the true candidate's own boundary date (`truth`) or on the furthest
remaining candidate's (`opposite`, control)).
* baseline — production before D1: the first `GUIDED_WINDOW_CASE_LIMIT = 2`
guided windows of the receipt are asked and injected (`after` in
`fewer_probes_card_replay.py`).
* d1 — nothing is injected after the six probes (`after_six`).
The targeted seven lines ask for events whose dates only a real person knows;
they cannot be modelled on the open holdout and are left out on both sides
(same as the 09-26 replay). This is an open-set replay, not a blind test.
Gate: per radius x direction, d1 truth-in-range count >= baseline and d1
median width <= baseline. Writes
`docs/research/futile_collect_stop_replay_2026_09_29.json`.
"""
from __future__ import annotations
import argparse
import json
import statistics
import sys
import time
import traceback
from pathlib import Path
from typing import Any, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.research.fewer_probes_card_replay import evaluate_case # noqa: E402
from scripts.research.guided_collect_holdout_replay import RADII, load_cases # noqa: E402
REPORT_JSON = ROOT / "docs" / "research" / "futile_collect_stop_replay_2026_09_29.json"
def _median(values: Sequence[float]) -> float | None:
return statistics.median(values) if values else None
def summarize(rows: Sequence[dict[str, Any]], radius: int, direction: str) -> dict[str, Any]:
subset = [
row for row in rows
if row.get("radius") == radius and row.get("direction") == direction and not row.get("error")
]
n = len(subset)
def side(name: str) -> dict[str, Any]:
widths = [row[name]["width"] for row in subset if row[name]["width"] is not None]
inside = sum(1 for row in subset if row[name]["truth_in_range"])
return {
"truth_in_range": inside,
"truth_in_range_rate": round(inside / n, 4) if n else None,
"median_width": _median(widths),
}
baseline = side("after")
d1 = side("after_six")
passed = (
n > 0
and d1["truth_in_range"] >= baseline["truth_in_range"]
and d1["median_width"] is not None
and baseline["median_width"] is not None
and d1["median_width"] <= baseline["median_width"]
)
return {
"radius": radius,
"direction": direction,
"n": n,
"baseline": {
**baseline,
"mean_injected": round(statistics.mean(row["windows_after"] for row in subset), 2) if n else None,
},
"d1": {**d1, "mean_injected": 0.0},
"width_worse_cases": sum(
1 for row in subset
if row["after_six"]["width"] is not None and row["after"]["width"] is not None
and row["after_six"]["width"] > row["after"]["width"]
),
"truth_lost_cases": sum(
1 for row in subset
if row["after"]["truth_in_range"] and not row["after_six"]["truth_in_range"]
),
"gate_pass": passed,
"errors": sum(
1 for row in rows
if row.get("radius") == radius and row.get("direction") == direction and row.get("error")
),
}
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--radii", default=",".join(str(item) for item in RADII))
parser.add_argument("--directions", default="truth,opposite")
parser.add_argument("--json-out", default=str(REPORT_JSON))
args = parser.parse_args()
radii = tuple(int(item) for item in str(args.radii).split(",") if item.strip())
directions = tuple(item.strip() for item in str(args.directions).split(",") if item.strip())
cases = load_cases()
if args.limit:
cases = cases[: args.limit]
started = time.perf_counter()
rows: list[dict[str, Any]] = []
for case in cases:
for radius in radii:
for direction in directions:
label = f"{case.get('case_id')} ±{radius} {direction}"
try:
result = evaluate_case(case, radius, direction)
rows.append(result)
print(
f"{label} baseline w={result['after']['width']} in={result['after']['truth_in_range']} "
f"d1 w={result['after_six']['width']} in={result['after_six']['truth_in_range']}",
flush=True,
)
except Exception as exc: # noqa: BLE001
rows.append({
"case_id": case.get("case_id"),
"radius": radius,
"direction": direction,
"error": f"{type(exc).__name__}: {exc}",
"trace": traceback.format_exc(limit=8),
})
print(f"{label} ERROR {type(exc).__name__}: {exc}", flush=True)
summaries = [summarize(rows, radius, direction) for radius in radii for direction in directions]
payload = {
"method": "fewer_probes_card_replay.evaluate_case; baseline=after (2 guided windows injected), d1=after_six (none)",
"holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json",
"elapsed_s": round(time.perf_counter() - started, 1),
"all_cells_pass": all(item["gate_pass"] for item in summaries),
"summaries": summaries,
"rows": [
{
"case_id": row.get("case_id"),
"radius": row.get("radius"),
"direction": row.get("direction"),
**({"error": row["error"]} if row.get("error") else {
"baseline": {key: row["after"][key] for key in ("start", "end", "width", "truth_in_range")},
"d1": {key: row["after_six"][key] for key in ("start", "end", "width", "truth_in_range")},
"windows_injected": row["windows_after"],
}),
}
for row in rows
],
}
out = Path(args.json_out)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps(summaries, ensure_ascii=False, indent=2), flush=True)
print(f"wrote {out} in {payload['elapsed_s']}s all_cells_pass={payload['all_cells_pass']}", flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())