"""Method-selection + direct reducer diagnostic (not answer persistence). Public holdout-v5 cases only. Truth goes exclusively to the offline answer oracle; the frozen collection, holdout, delivery and stopping gates stay live. The bridge does not exercise option classification, owned contrast registration or persistV9ChoiceAction. Early stop or a missing probe is not six-question validation. This is not the accepted single-chart M2 experiment. """ from __future__ import annotations import argparse import hashlib import json import platform from datetime import timedelta from pathlib import Path from scripts.research import varga_resolution_lib as vr from scripts.research.varga_resolution_production_bridge import ProductionSegmentBridge from scripts.rectification.api_service import score_candidates def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--dataset", default=str(vr.HOLDOUT_V5)) parser.add_argument("--output", required=True) parser.add_argument("--cache-dir", required=True) parser.add_argument("--limit", type=int) parser.add_argument("--radii", default="10,30,60") args = parser.parse_args() cases = vr.load_cases(Path(args.dataset)) if args.limit: cases = cases[:args.limit] cache = Path(args.cache_dir) cache.mkdir(parents=True, exist_ok=True) bridge = ProductionSegmentBridge() records = [] try: for case in cases: for radius in map(int, args.radii.split(",")): replay = vr.CaseReplay(case, radius, cache_dir=cache) key = hashlib.sha256(json.dumps(replay.request, sort_keys=True).encode()).hexdigest() response_file = cache / f"score-api-{key}.json" if response_file.exists(): response = json.loads(response_file.read_text(encoding="utf-8")) else: response = score_candidates(replay.request) response_file.write_text(json.dumps(response, ensure_ascii=False), encoding="utf-8") intervals = response["decision_receipt"]["candidate_intervals"] birth = case["birth"] origin = vr.birth_datetime(case) snapshot = {"birth_date": str(birth["date"]), "reported_birth_time": replay.true_time, "birth_time_source": "approximate", "timezone_offset": replay.request["tz"], "timezone_id": replay.request.get("timezone_id"), "latitude": replay.request["lat"], "longitude": replay.request["lon"]} evidence = [{"id": str(event["id"]), "status": "confirmed", "domain": event["domain"], "datePrecision": event.get("precision", "day"), "occurredFrom": event.get("date_start", event.get("date")), "occurredTo": event.get("date_end", event.get("date")), "eventKind": event.get("event_kind", "dated_event"), "summary": event.get("description", "Public calibration event")} for event in replay.request["events"]] minutes = [{"offset": offset + radius, "date": (origin + timedelta(minutes=offset)).strftime("%Y-%m-%d"), "time": (origin + timedelta(minutes=offset)).strftime("%H:%M"), "signs": {chart: signs[chart] for chart in ("D1", "D9", "D10")}} for offset, signs in sorted(replay.signs.items())] for enabled in (False, True): result = bridge.call({"operation": "method_replay", "request": replay.request, "response": response, "snapshot": snapshot, "range": {"start_time": replay.request["start_time"], "end_time": replay.request["end_time"], "candidate_intervals": intervals}, "evidence": evidence, "minutes": minutes, "targets": ["D1", "D9", "D10"], "enabled": enabled, "true_time": replay.true_time, "ask_count": vr.ASK_COUNT}) state = result.pop("state") raw = bridge.call({"operation": "weights", "rows": [{"time": row["time"], "score": row.get("raw_posterior_score", 0), "cluster_times": row.get("cluster_times")} for row in state["candidates"]], "eliminated": [row["time"] for row in state["candidates"] if row.get("raw_eliminated")]}) metrics = {} for chart in ("D9", "D10"): segments = replay.segments([chart]) summary = bridge.call({"operation": "summary_weights", "weights": raw, "segments": segments, "offsets": {stamp: vr.offset_of(stamp, replay.true_time) for stamp in raw}, "truth_offset": 0}) metrics[chart] = summary records.append({"case_id": replay.case_id, "radius": radius, "strategy": "joint" if enabled else "frozen", "api_response_sha256": hashlib.sha256(response_file.read_bytes()).hexdigest(), **result, "metrics": metrics}) print(f"method replay completed radius={radius}", flush=True) finally: bridge.close() aggregate = {} for radius in map(int, args.radii.split(",")): for strategy in ("frozen", "joint"): rows = [row for row in records if row["radius"] == radius and row["strategy"] == strategy] aggregate[f"{radius}:{strategy}"] = {"cases": len(rows), "asked": sum(len(row["asked"]) for row in rows), "D9": {"truth_kept": sum(row["metrics"]["D9"]["truth_retained"] for row in rows), "top_hit": sum(row["metrics"]["D9"]["top_is_truth"] for row in rows)}, "D10": {"truth_kept": sum(row["metrics"]["D10"]["truth_retained"] for row in rows), "top_hit": sum(row["metrics"]["D10"]["top_is_truth"] for row in rows)}} output = {"scope": "method-selection + direct reducer diagnostic (not answer persistence); real score API/parser/inference/catalog/method/decision, default joint targets; early stop/missing probe is not six-question validation", "python_version": platform.python_version(), "dataset_sha256": hashlib.sha256(Path(args.dataset).read_bytes()).hexdigest(), "aggregate": aggregate, "records": records} Path(args.output).write_text(json.dumps(output, ensure_ascii=False, sort_keys=True, indent=2) + "\n", encoding="utf-8") return 0 if __name__ == "__main__": raise SystemExit(main())