research(rectification): typed events scored as answered probes — no benefit (BUG-1089)

v4 open holdout, ±10/±30/±60, raw and percent priors: only 4/9/14 of 57
day-precision training events split candidates; widths unchanged, top-1
drops. Narrowed year blocking (R3) mixed. All arms no_benefit; no
implementation brief recommended. Two runs byte-identical.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
Jesse_Chen
2026-09-29 10:06:13 +08:00
co-authored by Claude Opus 5.5
parent 56b51e2170
commit bcf0c86feb
6 changed files with 8763 additions and 8 deletions
+391
View File
@@ -0,0 +1,391 @@
#!/usr/bin/env python3
"""Offline research for TASK-rectification-typed-event-scoring-research-20260929.
Question: should a dated typed event ("I changed jobs in 2019-10") move the
candidate scores the same way an answered existence probe does?
Today an answered dated probe moves every candidate by ±2 through the
Vimshottari/Narayana boundary-month split (`event_probes._evaluate_contexts`),
while a typed event only reaches the engine prior, and the frontend skips
`classified_from === "evidence"` yes answers (`core/build-state.ts`).
This script replays the v4 open holdout (20 public Rodden-AA cases, 7–8 events
each; 59 day-precision and 84 year-precision events, no month-precision) with
the same six-probe replay used by the 09-14 / 09-16 / 09-26 studies, and adds
research arms:
* ``B0`` baseline: prior, then the first six public probes answered from truth.
* ``R1d`` each training typed event with day precision is scored as an answered
``yes`` probe (split from `_evaluate_contexts` at that year/month),
then the same six probes.
* ``R1y`` as R1d, plus year-precision events scored with the year-level split
(`_evaluate_contexts(month=None)`, the existing year probe path).
* ``R2k`` R1 with k ∈ {1, 2} typed day events moved by ±1–3 months (seeded);
reports how often the truth is pushed out of the delivered range.
* ``R3`` `_existence_blocked_years` narrowed to "same domain, same year"
(probe supply count), alone and combined with R1d.
Two prior scales are reported: ``raw`` (engine row score, the research-lib
convention of the earlier studies) and ``percent`` (proportional percent over
public representatives, the production `relative_support` convention).
Nothing here changes a production default. Module attributes are patched for
one call and restored; the end of `main` asserts they are the originals.
This is an open-set replay, not a blind test: numbers are not accuracy.
"""
from __future__ import annotations
import argparse
import json
import random
import statistics
import sys
from contextlib import contextmanager
from datetime import date
from pathlib import Path
from typing import Any, Iterator, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
import scripts.rectification.event_probes as event_probes # noqa: E402
from scripts.active_rectification_event_engine import compute_candidate_static_contexts # noqa: E402
from scripts.rectification.candidate_contrast import cluster_contexts_by_signature # noqa: E402
from scripts.rectification.case_holdout import holdout_event_ids # noqa: E402
from scripts.rectification.refinement_packet import window_scan # noqa: E402
from scripts.rectification.scoring_service import ( # noqa: E402
build_event_contribution_matrix,
score_from_matrix,
)
from scripts.research.cluster_width_lib import ( # noqa: E402
SEPARATION_LEAD,
delivery_from_public,
merge_adjacent_traced,
public_from_clusters,
raw_signature_clusters,
still_valid_public,
)
from scripts.research.guided_collect_holdout_replay import ( # noqa: E402
TODAY,
_hhmm,
_minutes,
load_cases,
precision_gate,
)
from scripts.research.minute_resolution_sweep import scoring_request_for # noqa: E402
from scripts.research.probe_supply_after_six import ( # noqa: E402
ASK_COUNT,
apply_answer,
optimal_answer,
top1_hit,
)
RADII = (10, 30, 60)
PRIORS = ("raw", "percent")
REPORT_JSON = ROOT / "docs" / "research" / "rectification_typed_event_scoring_2026_09_29.json"
SHIFT_SEED = 20260929
SHIFT_REPEATS = 10
TYPED_SOURCE = "typed_event_research"
ORIGINAL_BLOCKED_YEARS = event_probes._existence_blocked_years
@contextmanager
def narrowed_blocking() -> Iterator[None]:
"""R3: an already-known year blocks only itself in the same domain."""
event_probes._existence_blocked_years = lambda _domain, known_years: set(known_years)
try:
yield
finally:
event_probes._existence_blocked_years = ORIGINAL_BLOCKED_YEARS
def _percent(scores: dict[str, float]) -> dict[str, float]:
total = sum(max(value, 0.0) for value in scores.values())
if total <= 0:
return {key: 0.0 for key in scores}
# Integer percent like `_relative_support_proportional` (largest remainder is not
# needed for ranking; plain rounding keeps the research copy independent).
return {key: float(round(max(value, 0.0) / total * 100)) for key, value in scores.items()}
def _event_month(event: dict[str, Any]) -> int | None:
raw = str(event.get("date_start") or "")
return int(raw[5:7]) if len(raw) >= 7 and raw[4] == "-" else None
def _event_year(event: dict[str, Any]) -> int | None:
raw = str(event.get("date_start") or "")
return int(raw[:4]) if raw[:4].isdigit() else None
def _shift_month(event: dict[str, Any], months: int) -> dict[str, Any]:
year, month = _event_year(event), _event_month(event)
if year is None or month is None:
return event
index = year * 12 + (month - 1) + months
new_year, new_month = divmod(index, 12)
stamp = f"{new_year:04d}-{new_month + 1:02d}-15"
return {**event, "date_start": stamp, "date_end": stamp}
def typed_event_probes(
request: dict[str, Any],
contexts: Sequence[dict[str, Any]],
*,
include_year: bool,
events: Sequence[dict[str, Any]] | None = None,
) -> tuple[list[dict[str, Any]], dict[str, int]]:
"""Score each training typed event through the probe split, answered ``yes``."""
full = [item for item in contexts if event_probes._context_time(item)]
full.sort(key=lambda item: event_probes._clock(str(event_probes._context_time(item))))
clusters = cluster_contexts_by_signature(full)
reps = [cluster["representative"] for cluster in clusters if event_probes._scoreable(cluster["representative"])]
if len(reps) < 2:
reps = [item for item in full if event_probes._scoreable(item)]
set_version = event_probes.candidate_set_version([cluster["times"] for cluster in clusters])
source_events = list(events if events is not None else request["events"])
holdout = set(holdout_event_ids(request["events"]))
stats = {"training_day": 0, "training_year": 0, "split_day": 0, "split_year": 0}
rows: list[dict[str, Any]] = []
for event in source_events:
if str(event.get("id")) in holdout:
continue
domain = event_probes.canonical_domain(event.get("domain"))
if domain not in event_probes.DOMAIN_CATALOG:
continue
precision = str(event.get("precision") or "")
year = _event_year(event)
if year is None:
continue
if precision in {"day", "month"}:
stats["training_day"] += 1
month = _event_month(event)
key = "split_day"
elif precision == "year" and include_year:
stats["training_year"] += 1
month = None
key = "split_year"
else:
continue
row = event_probes._evaluate_contexts(
reps,
birth_date=str(request["birth_date"]),
domain=domain,
year=year,
month=month,
source=TYPED_SOURCE,
clusters=clusters,
set_version=set_version,
)
if row is None:
continue
stats[key] += 1
rows.append(row)
return rows, stats
def replay(
*,
rows: Sequence[dict[str, Any]],
contexts: Sequence[dict[str, Any]],
typed: Sequence[dict[str, Any]],
probes: Sequence[dict[str, Any]],
true_time: str,
prior_mode: str,
) -> dict[str, Any]:
raw = raw_signature_clusters(contexts)
by_time = {stamp: row for row in rows if (stamp := _hhmm(row.get("time")))}
merged, _trace = merge_adjacent_traced(raw, by_time)
public = public_from_clusters(merged, rows)
reps = [str(row["time"])[:5] for row in public]
base = {stamp: float(row.get("score") or 0) for row in public if (stamp := _hhmm(row.get("time")))}
scores = _percent(base) if prior_mode == "percent" else dict(base)
conflicts = {time: 0 for time in reps}
eliminated: set[str] = set()
for probe in typed:
scores, conflicts, eliminated = apply_answer(scores, conflicts, eliminated, probe, "yes", reps)
asked = 0
for probe in list(probes)[:ASK_COUNT]:
answer = optimal_answer(probe, true_time)
if answer is None:
continue
asked += 1
scores, conflicts, eliminated = apply_answer(scores, conflicts, eliminated, probe, answer, reps)
posterior = [{**row, "score": scores.get(str(row["time"])[:5], row.get("score") or 0)} for row in public]
valid = still_valid_public(posterior, scores, eliminated, lead=SEPARATION_LEAD)
delivery = delivery_from_public(valid)
start, end = delivery.get("start"), delivery.get("end")
inside = start is not None and end is not None and _minutes(start) <= _minutes(true_time) <= _minutes(end)
active = [time for time in reps if time not in eliminated]
gate = precision_gate(valid, scores)
return {
"width": delivery.get("width"),
"truth_in_range": bool(inside),
"top1": bool(top1_hit(scores, active, true_time, merged)),
"tied": bool(gate["tied_for_first"]),
"questions": asked,
"typed_scored": len(typed),
"probe_supply": len(probes),
}
def evaluate_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
true_time = str(case["birth"]["time"])[:5]
request = scoring_request_for(case, radius)
contexts = compute_candidate_static_contexts(request)
built = build_event_contribution_matrix(request, static_contexts=contexts)
rows = score_from_matrix(request, built)
times = [stamp for row in rows if (stamp := _hhmm(row.get("time")))]
def probes_now() -> list[dict[str, Any]]:
return event_probes.discriminating_event_probes(
{**request, "refresh_probes": False, "asked_probe_keys": []}, built,
scan=window_scan(built), candidate_times=times, representative_time=true_time, today=TODAY,
)
probes = probes_now()
with narrowed_blocking():
probes_narrow = probes_now()
assert event_probes._existence_blocked_years is ORIGINAL_BLOCKED_YEARS
typed_day, stats = typed_event_probes(request, contexts, include_year=False)
typed_all, stats_all = typed_event_probes(request, contexts, include_year=True)
rng = random.Random(f"{SHIFT_SEED}:{case.get('case_id')}:{radius}")
holdout = set(holdout_event_ids(request["events"]))
day_events = [
index for index, event in enumerate(request["events"])
if str(event.get("precision")) in {"day", "month"} and str(event.get("id")) not in holdout
]
shifted: dict[str, list[dict[str, Any]]] = {"1": [], "2": []}
for k in (1, 2):
if len(day_events) < k:
continue
for _ in range(SHIFT_REPEATS):
picks = rng.sample(day_events, k)
events = [
_shift_month(event, rng.choice((-3, -2, -1, 1, 2, 3))) if index in picks else event
for index, event in enumerate(request["events"])
]
moved, _ = typed_event_probes(request, contexts, include_year=False, events=events)
shifted[str(k)].append({"typed": moved})
out: dict[str, Any] = {"case_id": case.get("case_id"), "radius": radius, "typed_stats": {**stats, **{
"training_year": stats_all["training_year"], "split_year": stats_all["split_year"]}},
"probe_supply": len(probes), "probe_supply_narrow": len(probes_narrow),
"asked_keys_changed_narrow": [p["semantic_key"] for p in probes[:ASK_COUNT]]
!= [p["semantic_key"] for p in probes_narrow[:ASK_COUNT]],
"arms": {}}
for prior in PRIORS:
def run(typed: Sequence[dict[str, Any]], pool: Sequence[dict[str, Any]]) -> dict[str, Any]:
return replay(rows=rows, contexts=contexts, typed=typed, probes=pool, true_time=true_time, prior_mode=prior)
arms = {
"B0": run([], probes),
"R1d": run(typed_day, probes),
"R1y": run(typed_all, probes),
"R3": run([], probes_narrow),
"R3+R1d": run(typed_day, probes_narrow),
}
for k, trials in shifted.items():
results = [run(item["typed"], probes) for item in trials]
if results:
arms[f"R2k{k}"] = {
"trials": len(results),
"truth_out": sum(1 for row in results if not row["truth_in_range"]),
"top1": sum(1 for row in results if row["top1"]),
"median_width": statistics.median(row["width"] for row in results if row["width"] is not None),
}
out["arms"][prior] = arms
return out
def summarize(rows: Sequence[dict[str, Any]], prior: str, radius: int, arm: str) -> dict[str, Any]:
subset = [row["arms"][prior][arm] for row in rows if row.get("radius") == radius and not row.get("error")
and arm in row["arms"][prior]]
n = len(subset)
if not n:
return {"n": 0}
if arm.startswith("R2k"):
trials = sum(item["trials"] for item in subset)
return {
"n": n,
"trials": trials,
"truth_out_rate": round(sum(item["truth_out"] for item in subset) / trials, 4),
"cases_with_truth_out": sum(1 for item in subset if item["truth_out"] > 0),
"top1_rate": round(sum(item["top1"] for item in subset) / trials, 4),
"median_width": statistics.median(item["median_width"] for item in subset),
}
return {
"n": n,
"top1": round(sum(item["top1"] for item in subset) / n, 4),
"truth_in_range": round(sum(item["truth_in_range"] for item in subset) / n, 4),
"median_width": statistics.median(item["width"] for item in subset if item["width"] is not None),
"tie_rate": round(sum(item["tied"] for item in subset) / n, 4),
"mean_questions": round(statistics.mean(item["questions"] for item in subset), 2),
"mean_typed_scored": round(statistics.mean(item["typed_scored"] for item in subset), 2),
"mean_probe_supply": round(statistics.mean(item["probe_supply"] for item in subset), 2),
}
def verdict(summary: dict[str, Any], prior: str, arm: str) -> str:
"""Gate: top1 not lower AND truth-in-range not lower AND median width lower, all radii."""
passes = []
for radius in RADII:
base = summary[prior][str(radius)]["B0"]
cand = summary[prior][str(radius)][arm]
if not cand.get("n"):
return "uncertain"
passes.append(
cand["top1"] >= base["top1"]
and cand["truth_in_range"] >= base["truth_in_range"]
and cand["median_width"] < base["median_width"]
)
return "benefit" if all(passes) else "no_benefit"
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--json-out", default=str(REPORT_JSON))
args = parser.parse_args()
cases = load_cases()
if args.limit:
cases = cases[: args.limit]
rows: list[dict[str, Any]] = []
for case in cases:
for radius in RADII:
try:
rows.append(evaluate_case(case, radius))
except Exception as exc: # noqa: BLE001
rows.append({"case_id": case.get("case_id"), "radius": radius, "error": f"{type(exc).__name__}: {exc}"})
arms = ("B0", "R1d", "R1y", "R3", "R3+R1d", "R2k1", "R2k2")
summary = {prior: {str(radius): {arm: summarize(rows, prior, radius, arm) for arm in arms}
for radius in RADII} for prior in PRIORS}
verdicts = {prior: {arm: verdict(summary, prior, arm) for arm in ("R1d", "R1y", "R3", "R3+R1d")} for prior in PRIORS}
assert event_probes._existence_blocked_years is ORIGINAL_BLOCKED_YEARS
payload = {
"generated_for": "TASK-rectification-typed-event-scoring-research-20260929",
"today": TODAY.isoformat(),
"holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json",
"ask_count": ASK_COUNT,
"separation_lead": SEPARATION_LEAD,
"shift_seed": SHIFT_SEED,
"shift_repeats": SHIFT_REPEATS,
"open_set_not_blind": True,
"summary": summary,
"verdicts": verdicts,
"rows": rows,
"errors": [row for row in rows if row.get("error")],
}
out = Path(args.json_out)
out.write_text(json.dumps(payload, ensure_ascii=False, indent=1, sort_keys=True) + "\n", encoding="utf-8")
print(json.dumps({"summary": summary, "verdicts": verdicts}, ensure_ascii=False, indent=1))
return 0
if __name__ == "__main__":
raise SystemExit(main())