Once the discriminator training gate is open, only choice cards are asked and the range card goes out when they are exhausted; targeted lines, their re-ask and guided windows no longer hold the card or invite more events. Delivery body says how many choice questions were used instead of the event fit percent; narration names an excluded cluster instead of "range unchanged"; a delivered turn no longer carries a collect question. Offline replay (v4, 3 radii x 2 directions): truth in range 20/20 in every cell; guided-window injections give the same width in truth and opposite directions, so red line 1 was revised by product to truth-in-range only. Skill 10.0.31 -> 10.0.32 (10.0.31 kept as deprecated for pinned cases). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
167 lines
6.8 KiB
Python
167 lines
6.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Offline replay for TASK-rectification-futile-collect-stop-20260929 (D1).
|
|
|
|
Hard red line 1: after the training gate opens, the Case no longer injects
|
|
spoken / guided events (targeted seven, skip re-ask, guided boundary windows).
|
|
Same method as `fewer_probes_card_replay.py` (v4 open holdout, six
|
|
discriminating probes answered from the true minute, guided windows injected
|
|
on the true candidate's own boundary date (`truth`) or on the furthest
|
|
remaining candidate's (`opposite`, control)).
|
|
|
|
* baseline — production before D1: the first `GUIDED_WINDOW_CASE_LIMIT = 2`
|
|
guided windows of the receipt are asked and injected (`after` in
|
|
`fewer_probes_card_replay.py`).
|
|
* d1 — nothing is injected after the six probes (`after_six`).
|
|
|
|
The targeted seven lines ask for events whose dates only a real person knows;
|
|
they cannot be modelled on the open holdout and are left out on both sides
|
|
(same as the 09-26 replay). This is an open-set replay, not a blind test.
|
|
|
|
Gate: per radius x direction, d1 truth-in-range count >= baseline and d1
|
|
median width <= baseline. Writes
|
|
`docs/research/futile_collect_stop_replay_2026_09_29.json`.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import statistics
|
|
import sys
|
|
import time
|
|
import traceback
|
|
from pathlib import Path
|
|
from typing import Any, Sequence
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
if str(ROOT) not in sys.path:
|
|
sys.path.insert(0, str(ROOT))
|
|
|
|
from scripts.research.fewer_probes_card_replay import evaluate_case # noqa: E402
|
|
from scripts.research.guided_collect_holdout_replay import RADII, load_cases # noqa: E402
|
|
|
|
REPORT_JSON = ROOT / "docs" / "research" / "futile_collect_stop_replay_2026_09_29.json"
|
|
|
|
|
|
def _median(values: Sequence[float]) -> float | None:
|
|
return statistics.median(values) if values else None
|
|
|
|
|
|
def summarize(rows: Sequence[dict[str, Any]], radius: int, direction: str) -> dict[str, Any]:
|
|
subset = [
|
|
row for row in rows
|
|
if row.get("radius") == radius and row.get("direction") == direction and not row.get("error")
|
|
]
|
|
n = len(subset)
|
|
|
|
def side(name: str) -> dict[str, Any]:
|
|
widths = [row[name]["width"] for row in subset if row[name]["width"] is not None]
|
|
inside = sum(1 for row in subset if row[name]["truth_in_range"])
|
|
return {
|
|
"truth_in_range": inside,
|
|
"truth_in_range_rate": round(inside / n, 4) if n else None,
|
|
"median_width": _median(widths),
|
|
}
|
|
|
|
baseline = side("after")
|
|
d1 = side("after_six")
|
|
passed = (
|
|
n > 0
|
|
and d1["truth_in_range"] >= baseline["truth_in_range"]
|
|
and d1["median_width"] is not None
|
|
and baseline["median_width"] is not None
|
|
and d1["median_width"] <= baseline["median_width"]
|
|
)
|
|
return {
|
|
"radius": radius,
|
|
"direction": direction,
|
|
"n": n,
|
|
"baseline": {
|
|
**baseline,
|
|
"mean_injected": round(statistics.mean(row["windows_after"] for row in subset), 2) if n else None,
|
|
},
|
|
"d1": {**d1, "mean_injected": 0.0},
|
|
"width_worse_cases": sum(
|
|
1 for row in subset
|
|
if row["after_six"]["width"] is not None and row["after"]["width"] is not None
|
|
and row["after_six"]["width"] > row["after"]["width"]
|
|
),
|
|
"truth_lost_cases": sum(
|
|
1 for row in subset
|
|
if row["after"]["truth_in_range"] and not row["after_six"]["truth_in_range"]
|
|
),
|
|
"gate_pass": passed,
|
|
"errors": sum(
|
|
1 for row in rows
|
|
if row.get("radius") == radius and row.get("direction") == direction and row.get("error")
|
|
),
|
|
}
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("--limit", type=int, default=0)
|
|
parser.add_argument("--radii", default=",".join(str(item) for item in RADII))
|
|
parser.add_argument("--directions", default="truth,opposite")
|
|
parser.add_argument("--json-out", default=str(REPORT_JSON))
|
|
args = parser.parse_args()
|
|
radii = tuple(int(item) for item in str(args.radii).split(",") if item.strip())
|
|
directions = tuple(item.strip() for item in str(args.directions).split(",") if item.strip())
|
|
cases = load_cases()
|
|
if args.limit:
|
|
cases = cases[: args.limit]
|
|
started = time.perf_counter()
|
|
rows: list[dict[str, Any]] = []
|
|
for case in cases:
|
|
for radius in radii:
|
|
for direction in directions:
|
|
label = f"{case.get('case_id')} ±{radius} {direction}"
|
|
try:
|
|
result = evaluate_case(case, radius, direction)
|
|
rows.append(result)
|
|
print(
|
|
f"{label} baseline w={result['after']['width']} in={result['after']['truth_in_range']} "
|
|
f"d1 w={result['after_six']['width']} in={result['after_six']['truth_in_range']}",
|
|
flush=True,
|
|
)
|
|
except Exception as exc: # noqa: BLE001
|
|
rows.append({
|
|
"case_id": case.get("case_id"),
|
|
"radius": radius,
|
|
"direction": direction,
|
|
"error": f"{type(exc).__name__}: {exc}",
|
|
"trace": traceback.format_exc(limit=8),
|
|
})
|
|
print(f"{label} ERROR {type(exc).__name__}: {exc}", flush=True)
|
|
summaries = [summarize(rows, radius, direction) for radius in radii for direction in directions]
|
|
payload = {
|
|
"method": "fewer_probes_card_replay.evaluate_case; baseline=after (2 guided windows injected), d1=after_six (none)",
|
|
"holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json",
|
|
"elapsed_s": round(time.perf_counter() - started, 1),
|
|
"all_cells_pass": all(item["gate_pass"] for item in summaries),
|
|
"summaries": summaries,
|
|
"rows": [
|
|
{
|
|
"case_id": row.get("case_id"),
|
|
"radius": row.get("radius"),
|
|
"direction": row.get("direction"),
|
|
**({"error": row["error"]} if row.get("error") else {
|
|
"baseline": {key: row["after"][key] for key in ("start", "end", "width", "truth_in_range")},
|
|
"d1": {key: row["after_six"][key] for key in ("start", "end", "width", "truth_in_range")},
|
|
"windows_injected": row["windows_after"],
|
|
}),
|
|
}
|
|
for row in rows
|
|
],
|
|
}
|
|
out = Path(args.json_out)
|
|
out.parent.mkdir(parents=True, exist_ok=True)
|
|
out.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
print(json.dumps(summaries, ensure_ascii=False, indent=2), flush=True)
|
|
print(f"wrote {out} in {payload['elapsed_s']}s all_cells_pass={payload['all_cells_pass']}", flush=True)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|