research(rectification): v5 verdict for scoring-method research — LR weights, absence and precision follow-up all fail hard line 1 (BUG-1091 closed_by_design)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-09-30 00:30:34 +08:00
co-authored by Claude Fable 5.1
parent a29fbe0333
commit fa2e7d17f5
9 changed files with 39358 additions and 197 deletions
+128 -42
View File
@@ -29,6 +29,7 @@ Run:
from __future__ import annotations
import argparse
import gc
import json
import pickle
import sys
@@ -49,6 +50,7 @@ from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_reques
from scripts.research.offline_research_20260926_lib import exhaustive_flip_sets, flipped_answers, sample_flip_sets # noqa: E402
from scripts.research.precision_gate_lib import gate_verdict, jitter_day_events # noqa: E402
from scripts.research.probe_supply_after_six import ASK_COUNT, optimal_answer # noqa: E402
from scripts.active_rectification_events import precision_weight # noqa: E402
import scripts.research.scoring_research_lib as lib # noqa: E402
HOLDOUT_V4 = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
@@ -115,12 +117,34 @@ class CaseData:
def examples(self, *, include_extras: bool) -> list[tuple[dict[str, float], bool]]:
out: list[tuple[dict[str, float], bool]] = []
truth = set(self.truth_times)
for event_id in self.training_ids:
matrix = lib.event_feature_matrix(self.store, event_id, self.times, include_extras=include_extras)
for matrix in self.feature_matrices(include_extras=include_extras).values():
for time in self.times:
out.append((matrix.get(time, {}), time in truth))
return out
def feature_matrices(self, *, include_extras: bool) -> dict[str, dict[str, dict[str, float]]]:
cache = self.__dict__.setdefault("_fm_cache", {})
if include_extras not in cache:
cache[include_extras] = {
event_id: lib.event_feature_matrix(self.store, event_id, self.times, include_extras=include_extras)
for event_id in self.training_ids
}
return cache[include_extras]
def proxy_rows(self, table: lib.LRTable, variant: str) -> list[dict[str, Any]]:
"""Linear proxy of `weighted_rows` for fitting scale / temperature (no matrix build)."""
include_extras, allow_negative = lib.VARIANT_SPEC[variant]
universe = [n for n in sorted(table.present) if (include_extras or not lib.is_extra_feature(n))]
matrices = self.feature_matrices(include_extras=include_extras)
rows = []
for time in self.times:
total = 0.0
for event_id, matrix in matrices.items():
weight = precision_weight(str(self.events_by_id[event_id]["precision"]))
total += weight * lib.weighted_points(matrix.get(time, {}), table, allow_negative=allow_negative, universe=universe)
rows.append({"time": time, "score": round(total, 6)})
return rows
# scoring with a weight table ------------------------------------------
def weighted_rows(
self, table: lib.LRTable, variant: str, scale: float,
@@ -149,13 +173,13 @@ class CaseData:
return score_from_matrix(request, built)
def evaluate(data: CaseData, rows: Sequence[dict[str, Any]], *, softmax_prior: bool, pre_answers=(), answers=None, temperature: float = 1.0) -> dict[str, dict[str, Any]]:
def evaluate(data: CaseData, rows: Sequence[dict[str, Any]], *, softmax_prior: bool, pre_answers=(), answers=None, temperature: float = 1.0, probes=None) -> dict[str, dict[str, Any]]:
public = lib.public_for(rows, data.contexts)
out: dict[str, dict[str, Any]] = {}
for mode in PRIORS:
prior = lib.prior_for(public, mode=mode, softmax=softmax_prior, temperature=temperature)
result = lib.replay(
public=public, prior=prior, probes=data.probes, true_time=data.true_time,
public=public, prior=prior, probes=data.probes if probes is None else probes, true_time=data.true_time,
window_times=data.times, pre_answers=pre_answers, answers=answers,
)
result["engine_top1"] = lib.engine_top1(public, data.true_time)
@@ -173,9 +197,12 @@ def fit_prior(training: Sequence[CaseData], table: lib.LRTable, variant: str) ->
scale: research within-case score range matched to the production median
range (so the raw-prior replay's ±2 / 8-lead constants mean the same);
temperature: single Platt temperature for the percent (softmax) prior.
Training-case scores use the linear proxy (`CaseData.proxy_rows`: feature
weights × precision weight, no kind factor / transition proximity) so a
fold costs O(n) dictionary sums instead of n contribution-matrix builds.
"""
prod = [lib.within_case_range(d.prod_rows) for d in training]
rows_by_case = {d.case_id: d.weighted_rows(table, variant, 1.0)[0] for d in training}
rows_by_case = {d.case_id: d.proxy_rows(table, variant) for d in training}
research = [lib.within_case_range(rows) for rows in rows_by_case.values()]
p_med, r_med = lib.median(prod), lib.median(research)
scale = round(p_med / r_med, 6) if p_med and r_med else 1.0
@@ -186,8 +213,8 @@ def fit_prior(training: Sequence[CaseData], table: lib.LRTable, variant: str) ->
return scale, lib.fit_temperature(publics)
def run_ra(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any]:
out: dict[str, Any] = {"variants": list(lib.VARIANTS), "alpha": lib.LAPLACE_ALPHA, "radii": {}}
def run_ra(by_radius: dict[int, list[CaseData]], *, fast: bool, probes_per_variant: bool = False) -> dict[str, Any]:
out: dict[str, Any] = {"variants": list(lib.VARIANTS), "alpha": lib.LAPLACE_ALPHA, "probes_per_variant": probes_per_variant, "radii": {}}
for radius, cases in sorted(by_radius.items()):
block: dict[str, Any] = {"n": len(cases), "baseline": {}, "full_fit": {}, "loo": {}, "robustness": {}, "calibration": {}, "lr_table": {}, "noise_features": {}}
baseline_rows = []
@@ -195,6 +222,7 @@ def run_ra(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
row = evaluate(data, data.prod_rows, softmax_prior=False)
baseline_rows.append({"case_id": data.case_id, **{f"{m}": row[m] for m in PRIORS}})
block["baseline"] = {m: lib.summarize([r[m] for r in baseline_rows]) for m in PRIORS}
block["per_case"] = {"baseline": [{"case_id": r["case_id"], **{m: {k: r[m][k] for k in ("top1", "coverage", "width", "engine_top1")} for m in PRIORS}} for r in baseline_rows]}
per_case_examples = {
include: {d.case_id: d.examples(include_extras=include) for d in cases} for include in (False, True)
}
@@ -241,8 +269,13 @@ def run_ra(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
table = lib.estimate_lr([ex for d in training for ex in examples[d.case_id]])
scale, temperature = fit_prior(training, table, variant)
temperatures.append(temperature)
rows, _ = held.weighted_rows(table, variant, scale)
rows, built_v = held.weighted_rows(table, variant, scale)
result = evaluate(held, rows, softmax_prior=True, temperature=temperature)
if probes_per_variant:
own = lib.production_probes(held.request, built_v, held.times, held.true_time)
result["own_probes"] = evaluate(held, rows, softmax_prior=True, temperature=temperature, probes=own)
result["own_probes_count"] = len(own)
result["case_id"] = held.case_id
loo_rows.append(result)
if variant == "A3":
public = lib.public_for(rows, held.contexts)
@@ -283,9 +316,25 @@ def run_ra(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
))
block["loo"][variant] = {m: lib.summarize([r[m] for r in loo_rows]) for m in PRIORS}
block["loo"][variant]["temperatures"] = sorted(set(temperatures))
block["loo"][variant]["debug_verdict_v4"] = {
block["loo"][variant]["gate_verdict"] = {
m: gate_verdict(block["baseline"][m], block["loo"][variant][m]) for m in PRIORS
}
block["loo"][variant]["hard_line_1"] = {
m: {
"coverage_count": sum(1 for r in loo_rows if r[m]["coverage"]),
"baseline_coverage_count": sum(1 for r in baseline_rows if r[m]["coverage"]),
"top1": block["loo"][variant][m]["top1"],
"baseline_top1": block["baseline"][m]["top1"],
"pass": (sum(1 for r in loo_rows if r[m]["coverage"]) >= sum(1 for r in baseline_rows if r[m]["coverage"])
and float(block["loo"][variant][m]["top1"]) + 1e-9 >= float(block["baseline"][m]["top1"])),
"note": "truth and opposite cells coincide under D1 (nothing is injected after the six probes)",
}
for m in PRIORS
}
if probes_per_variant:
block["loo"][variant]["own_probes"] = {m: lib.summarize([r["own_probes"][m] for r in loo_rows]) for m in PRIORS}
block["loo"][variant]["own_probes"]["mean_count"] = round(sum(r["own_probes_count"] for r in loo_rows) / max(len(loo_rows), 1), 2)
block["per_case"][variant] = [{"case_id": r["case_id"], **{m: {k: r[m][k] for k in ("top1", "coverage", "width", "engine_top1")} for m in PRIORS}} for r in loo_rows]
block["loo"][variant]["verdict"] = "pending_v5"
if robustness:
block["robustness"][variant] = {
@@ -356,7 +405,13 @@ def run_rb(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
for arm in block["arms"]:
if arm == "B0":
continue
block["arms"][arm]["debug_verdict_v4"] = {m: gate_verdict(block["arms"]["B0"][m], block["arms"][arm][m]) for m in PRIORS}
block["arms"][arm]["gate_verdict"] = {m: gate_verdict(block["arms"]["B0"][m], block["arms"][arm][m]) for m in PRIORS}
block["arms"][arm]["hard_line_1"] = {m: {
"coverage_count": sum(1 for r in rows_by_arm[arm] if r[m]["coverage"]),
"baseline_coverage_count": sum(1 for r in rows_by_arm["B0"] if r[m]["coverage"]),
"pass": sum(1 for r in rows_by_arm[arm] if r[m]["coverage"]) >= sum(1 for r in rows_by_arm["B0"] if r[m]["coverage"])
and float(block["arms"][arm][m]["top1"]) + 1e-9 >= float(block["arms"]["B0"][m]["top1"]),
} for m in PRIORS}
block["arms"][arm]["verdict"] = "pending_v5"
out["radii"][str(radius)] = block
return out
@@ -376,8 +431,12 @@ def run_rc(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
raw_events = list(data.case.get("events") or [])
spoken = lib.degrade_to_year(raw_events)
spoken_rows = data.production_rows_for(spoken)
rows_by_arm["B0_true_precision"].append(evaluate(data, data.prod_rows, softmax_prior=False))
rows_by_arm["C0_spoken"].append(evaluate(data, spoken_rows, softmax_prior=False))
def keep(arm: str, result: dict[str, dict[str, Any]]) -> None:
rows_by_arm[arm].append({"case_id": data.case_id, **result})
keep("B0_true_precision", evaluate(data, data.prod_rows, softmax_prior=False))
keep("C0_spoken", evaluate(data, spoken_rows, softmax_prior=False))
# which spoken (year) events would split candidates if the month were known?
training = set(data.training_ids)
id_map = {str(e["id"]): str(rid) for rid, e in zip(
@@ -385,7 +444,7 @@ def run_rc(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
)}
candidates = []
for event in raw_events:
if str(event.get("precision")) != "day":
if str(event.get("precision")) not in {"day", "month"}:
continue
request_id = id_map.get(str(event["id"]))
if request_id not in training:
@@ -406,10 +465,12 @@ def run_rc(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
continue
ids = {item[1] for item in picked}
upgraded = lib.upgrade_to_month(spoken, ids, originals)
rows_by_arm[f"C{k}_prior"].append(evaluate(data, data.production_rows_for(upgraded), softmax_prior=False))
upgraded_rows = data.production_rows_for(upgraded)
pre = [(item[2], "yes", 1.0) for item in picked]
rows_by_arm[f"C{k}_probe"].append(evaluate(data, spoken_rows, softmax_prior=False, pre_answers=pre))
rows_by_arm[f"C{k}_both"].append(evaluate(data, data.production_rows_for(upgraded), softmax_prior=False, pre_answers=pre))
keep(f"C{k}_prior", evaluate(data, upgraded_rows, softmax_prior=False))
keep(f"C{k}_probe", evaluate(data, spoken_rows, softmax_prior=False, pre_answers=pre))
keep(f"C{k}_both", evaluate(data, upgraded_rows, softmax_prior=False, pre_answers=pre))
by_case = {arm: {r["case_id"]: r for r in rows} for arm, rows in rows_by_arm.items()}
block: dict[str, Any] = {
"n": len(cases),
"askable_per_case": {
@@ -419,11 +480,24 @@ def run_rc(by_radius: dict[int, list[CaseData]], *, fast: bool) -> dict[str, Any
"cases_with_at_least_two": sum(1 for c in askable_counts if c >= 2),
},
"arms": {arm: {m: lib.summarize([r[m] for r in rows]) for m in PRIORS} for arm, rows in rows_by_arm.items()},
"per_case": {arm: [{"case_id": r["case_id"], **{m: {k: r[m][k] for k in ("top1", "coverage", "width", "engine_top1")} for m in PRIORS}} for r in rows] for arm, rows in rows_by_arm.items()},
}
for arm in block["arms"]:
for arm, rows in rows_by_arm.items():
if arm.startswith("C0") or arm.startswith("B0"):
continue
block["arms"][arm]["debug_verdict_v4_vs_C0"] = {m: gate_verdict(block["arms"]["C0_spoken"][m], block["arms"][arm][m]) for m in PRIORS}
ids = [r["case_id"] for r in rows]
subset_c0 = [by_case["C0_spoken"][cid] for cid in ids]
subset_b0 = [by_case["B0_true_precision"][cid] for cid in ids]
block["arms"][arm]["subset_C0"] = {m: lib.summarize([r[m] for r in subset_c0]) for m in PRIORS}
block["arms"][arm]["subset_B0"] = {m: lib.summarize([r[m] for r in subset_b0]) for m in PRIORS}
block["arms"][arm]["gate_verdict_vs_subset_C0"] = {m: gate_verdict(block["arms"][arm]["subset_C0"][m], block["arms"][arm][m]) for m in PRIORS}
block["arms"][arm]["hard_line_1_vs_subset_C0"] = {m: {
"coverage_count": sum(1 for r in rows if r[m]["coverage"]),
"subset_C0_coverage_count": sum(1 for r in subset_c0 if r[m]["coverage"]),
"pass": sum(1 for r in rows if r[m]["coverage"]) >= sum(1 for r in subset_c0 if r[m]["coverage"])
and float(block["arms"][arm][m]["top1"]) + 1e-9 >= float(block["arms"][arm]["subset_C0"][m]["top1"]),
} for m in PRIORS}
block["arms"][arm]["gate_verdict_vs_C0"] = {m: gate_verdict(block["arms"]["C0_spoken"][m], block["arms"][arm][m]) for m in PRIORS}
block["arms"][arm]["verdict"] = "pending_v5"
out["radii"][str(radius)] = block
return out
@@ -453,6 +527,7 @@ def main() -> int:
parser.add_argument("--fast", action="store_true", help="fewer repeats (development)")
parser.add_argument("--no-write", action="store_true")
parser.add_argument("--out-suffix", default="2026_09_29")
parser.add_argument("--probes-per-variant", action="store_true", help="also replay LOO variants with probes generated from their own matrix")
args = parser.parse_args()
started = clock_module.time()
holdout_path = Path(args.holdout)
@@ -460,20 +535,41 @@ def main() -> int:
if args.limit:
cases = cases[: args.limit]
cache_dir = Path(args.cache_dir) if args.cache_dir else None
by_radius: dict[int, list[CaseData]] = defaultdict(list)
errors: list[dict[str, Any]] = []
reconciliation: list[dict[str, Any]] = []
for case in cases:
for radius in args.radii:
parts = {"ra": "lr", "rb": "absence", "rc": "precision_followup"}
radii_results: dict[str, dict[str, dict[str, Any]]] = {key: {} for key in parts.values()}
part_meta: dict[str, dict[str, Any]] = {}
# One radius at a time: 77 cases x 3 radii of static contexts do not fit in memory together.
for radius in args.radii:
loaded: list[CaseData] = []
for case in cases:
try:
data = CaseData(case, radius, cache_dir)
except Exception as exc: # noqa: BLE001
errors.append({"case_id": case.get("case_id"), "radius": radius, "error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc()})
continue
reconciliation.append({"case_id": data.case_id, "radius": radius, **data.reconciliation})
by_radius[radius].append(data)
print(f"loaded {case.get('case_id')} ({clock_module.time() - started:.0f}s)", flush=True)
unreconciled = [row for row in reconciliation if row["changed"]]
loaded.append(data)
print(f"loaded radius {radius}: {len(loaded)} cases ({clock_module.time() - started:.0f}s)", flush=True)
if any(row["changed"] for row in reconciliation if row["radius"] == radius):
print(json.dumps({"UNRECONCILED": [row for row in reconciliation if row["changed"]]}, ensure_ascii=False, indent=1))
return 2
by_radius = {radius: loaded}
runners = {
"ra": lambda: run_ra(by_radius, fast=args.fast, probes_per_variant=args.probes_per_variant),
"rb": lambda: run_rb(by_radius, fast=args.fast),
"rc": lambda: run_rc(by_radius, fast=args.fast),
}
for part, key in parts.items():
if part not in args.parts:
continue
block = runners[part]()
radii_results[key].update(block.pop("radii"))
part_meta[key] = block
print(f"{part.upper()} radius {radius} done {clock_module.time() - started:.0f}s", flush=True)
del loaded, by_radius
gc.collect()
header = {
"generated_for": "TASK-rectification-scoring-research-20260929",
"dataset": args.dataset_label,
@@ -485,32 +581,22 @@ def main() -> int:
"node_mode": NODE_MODE,
"ask_count": ASK_COUNT,
"probe_today": lib.TODAY.isoformat(),
"probes": "production G0 probes generated once per case/radius from the production matrix; reused for every arm",
"probes": "production G0 probes generated once per case/radius from the production matrix; reused for every arm (own_probes = regenerated from the variant matrix)",
"open_set_not_blind": True,
"verdict": "pending_v5",
"reconciliation": {"cases": len(reconciliation), "unreconciled": len(unreconciled), "rows": reconciliation},
"verdict": "pending_v5" if args.dataset_label != "v5" else "see_report",
"reconciliation": {"cases": len(reconciliation), "unreconciled": sum(1 for r in reconciliation if r["changed"]), "rows": reconciliation},
"errors": errors,
"seed": SEED,
"fast": bool(args.fast),
}
if unreconciled:
print(json.dumps({"UNRECONCILED": unreconciled}, ensure_ascii=False, indent=1))
return 2
results: dict[str, dict[str, Any]] = {}
if "ra" in args.parts:
results["lr"] = {**header, "part": "R-A", **run_ra(by_radius, fast=args.fast)}
print("R-A done", f"{clock_module.time() - started:.0f}s", flush=True)
if "rb" in args.parts:
results["absence"] = {**header, "part": "R-B", **run_rb(by_radius, fast=args.fast)}
print("R-B done", f"{clock_module.time() - started:.0f}s", flush=True)
if "rc" in args.parts:
results["precision_followup"] = {**header, "part": "R-C", **run_rc(by_radius, fast=args.fast)}
print("R-C done", f"{clock_module.time() - started:.0f}s", flush=True)
for part, key in parts.items():
if part in args.parts:
results[key] = {**header, "part": {"ra": "R-A", "rb": "R-B", "rc": "R-C"}[part], **part_meta.get(key, {}), "radii": radii_results[key]}
for key, payload in results.items():
payload["elapsed_seconds"] = round(clock_module.time() - started, 1)
if not args.no_write:
write_json(OUT_DIR / f"scoring_research_{key}_{args.out_suffix}.json", {k: v for k, v in payload.items() if k != "elapsed_seconds"})
brief: dict[str, Any] = {"reconciliation_unreconciled": len(unreconciled), "errors": len(errors)}
write_json(OUT_DIR / f"scoring_research_{key}_{args.out_suffix}.json", payload)
brief: dict[str, Any] = {"reconciliation_unreconciled": header["reconciliation"]["unreconciled"], "errors": len(errors), "elapsed_seconds": round(clock_module.time() - started, 1)}
for key, payload in results.items():
brief[key] = {}
for radius, block in payload["radii"].items():