research(rectification): typed events scored as answered probes — no benefit (BUG-1089)
v4 open holdout, ±10/±30/±60, raw and percent priors: only 4/9/14 of 57 day-precision training events split candidates; widths unchanged, top-1 drops. Narrowed year blocking (R3) mixed. All arms no_benefit; no implementation brief recommended. Two runs byte-identical. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
56b51e2170
commit
bcf0c86feb
@@ -0,0 +1,391 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline research for TASK-rectification-typed-event-scoring-research-20260929.
|
||||
|
||||
Question: should a dated typed event ("I changed jobs in 2019-10") move the
|
||||
candidate scores the same way an answered existence probe does?
|
||||
|
||||
Today an answered dated probe moves every candidate by ±2 through the
|
||||
Vimshottari/Narayana boundary-month split (`event_probes._evaluate_contexts`),
|
||||
while a typed event only reaches the engine prior, and the frontend skips
|
||||
`classified_from === "evidence"` yes answers (`core/build-state.ts`).
|
||||
|
||||
This script replays the v4 open holdout (20 public Rodden-AA cases, 7–8 events
|
||||
each; 59 day-precision and 84 year-precision events, no month-precision) with
|
||||
the same six-probe replay used by the 09-14 / 09-16 / 09-26 studies, and adds
|
||||
research arms:
|
||||
|
||||
* ``B0`` baseline: prior, then the first six public probes answered from truth.
|
||||
* ``R1d`` each training typed event with day precision is scored as an answered
|
||||
``yes`` probe (split from `_evaluate_contexts` at that year/month),
|
||||
then the same six probes.
|
||||
* ``R1y`` as R1d, plus year-precision events scored with the year-level split
|
||||
(`_evaluate_contexts(month=None)`, the existing year probe path).
|
||||
* ``R2k`` R1 with k ∈ {1, 2} typed day events moved by ±1–3 months (seeded);
|
||||
reports how often the truth is pushed out of the delivered range.
|
||||
* ``R3`` `_existence_blocked_years` narrowed to "same domain, same year"
|
||||
(probe supply count), alone and combined with R1d.
|
||||
|
||||
Two prior scales are reported: ``raw`` (engine row score, the research-lib
|
||||
convention of the earlier studies) and ``percent`` (proportional percent over
|
||||
public representatives, the production `relative_support` convention).
|
||||
|
||||
Nothing here changes a production default. Module attributes are patched for
|
||||
one call and restored; the end of `main` asserts they are the originals.
|
||||
This is an open-set replay, not a blind test: numbers are not accuracy.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import random
|
||||
import statistics
|
||||
import sys
|
||||
from contextlib import contextmanager
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterator, Sequence
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
import scripts.rectification.event_probes as event_probes # noqa: E402
|
||||
from scripts.active_rectification_event_engine import compute_candidate_static_contexts # noqa: E402
|
||||
from scripts.rectification.candidate_contrast import cluster_contexts_by_signature # noqa: E402
|
||||
from scripts.rectification.case_holdout import holdout_event_ids # noqa: E402
|
||||
from scripts.rectification.refinement_packet import window_scan # noqa: E402
|
||||
from scripts.rectification.scoring_service import ( # noqa: E402
|
||||
build_event_contribution_matrix,
|
||||
score_from_matrix,
|
||||
)
|
||||
from scripts.research.cluster_width_lib import ( # noqa: E402
|
||||
SEPARATION_LEAD,
|
||||
delivery_from_public,
|
||||
merge_adjacent_traced,
|
||||
public_from_clusters,
|
||||
raw_signature_clusters,
|
||||
still_valid_public,
|
||||
)
|
||||
from scripts.research.guided_collect_holdout_replay import ( # noqa: E402
|
||||
TODAY,
|
||||
_hhmm,
|
||||
_minutes,
|
||||
load_cases,
|
||||
precision_gate,
|
||||
)
|
||||
from scripts.research.minute_resolution_sweep import scoring_request_for # noqa: E402
|
||||
from scripts.research.probe_supply_after_six import ( # noqa: E402
|
||||
ASK_COUNT,
|
||||
apply_answer,
|
||||
optimal_answer,
|
||||
top1_hit,
|
||||
)
|
||||
|
||||
RADII = (10, 30, 60)
|
||||
PRIORS = ("raw", "percent")
|
||||
REPORT_JSON = ROOT / "docs" / "research" / "rectification_typed_event_scoring_2026_09_29.json"
|
||||
SHIFT_SEED = 20260929
|
||||
SHIFT_REPEATS = 10
|
||||
TYPED_SOURCE = "typed_event_research"
|
||||
|
||||
ORIGINAL_BLOCKED_YEARS = event_probes._existence_blocked_years
|
||||
|
||||
|
||||
@contextmanager
|
||||
def narrowed_blocking() -> Iterator[None]:
|
||||
"""R3: an already-known year blocks only itself in the same domain."""
|
||||
event_probes._existence_blocked_years = lambda _domain, known_years: set(known_years)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
event_probes._existence_blocked_years = ORIGINAL_BLOCKED_YEARS
|
||||
|
||||
|
||||
def _percent(scores: dict[str, float]) -> dict[str, float]:
|
||||
total = sum(max(value, 0.0) for value in scores.values())
|
||||
if total <= 0:
|
||||
return {key: 0.0 for key in scores}
|
||||
# Integer percent like `_relative_support_proportional` (largest remainder is not
|
||||
# needed for ranking; plain rounding keeps the research copy independent).
|
||||
return {key: float(round(max(value, 0.0) / total * 100)) for key, value in scores.items()}
|
||||
|
||||
|
||||
def _event_month(event: dict[str, Any]) -> int | None:
|
||||
raw = str(event.get("date_start") or "")
|
||||
return int(raw[5:7]) if len(raw) >= 7 and raw[4] == "-" else None
|
||||
|
||||
|
||||
def _event_year(event: dict[str, Any]) -> int | None:
|
||||
raw = str(event.get("date_start") or "")
|
||||
return int(raw[:4]) if raw[:4].isdigit() else None
|
||||
|
||||
|
||||
def _shift_month(event: dict[str, Any], months: int) -> dict[str, Any]:
|
||||
year, month = _event_year(event), _event_month(event)
|
||||
if year is None or month is None:
|
||||
return event
|
||||
index = year * 12 + (month - 1) + months
|
||||
new_year, new_month = divmod(index, 12)
|
||||
stamp = f"{new_year:04d}-{new_month + 1:02d}-15"
|
||||
return {**event, "date_start": stamp, "date_end": stamp}
|
||||
|
||||
|
||||
def typed_event_probes(
|
||||
request: dict[str, Any],
|
||||
contexts: Sequence[dict[str, Any]],
|
||||
*,
|
||||
include_year: bool,
|
||||
events: Sequence[dict[str, Any]] | None = None,
|
||||
) -> tuple[list[dict[str, Any]], dict[str, int]]:
|
||||
"""Score each training typed event through the probe split, answered ``yes``."""
|
||||
full = [item for item in contexts if event_probes._context_time(item)]
|
||||
full.sort(key=lambda item: event_probes._clock(str(event_probes._context_time(item))))
|
||||
clusters = cluster_contexts_by_signature(full)
|
||||
reps = [cluster["representative"] for cluster in clusters if event_probes._scoreable(cluster["representative"])]
|
||||
if len(reps) < 2:
|
||||
reps = [item for item in full if event_probes._scoreable(item)]
|
||||
set_version = event_probes.candidate_set_version([cluster["times"] for cluster in clusters])
|
||||
source_events = list(events if events is not None else request["events"])
|
||||
holdout = set(holdout_event_ids(request["events"]))
|
||||
stats = {"training_day": 0, "training_year": 0, "split_day": 0, "split_year": 0}
|
||||
rows: list[dict[str, Any]] = []
|
||||
for event in source_events:
|
||||
if str(event.get("id")) in holdout:
|
||||
continue
|
||||
domain = event_probes.canonical_domain(event.get("domain"))
|
||||
if domain not in event_probes.DOMAIN_CATALOG:
|
||||
continue
|
||||
precision = str(event.get("precision") or "")
|
||||
year = _event_year(event)
|
||||
if year is None:
|
||||
continue
|
||||
if precision in {"day", "month"}:
|
||||
stats["training_day"] += 1
|
||||
month = _event_month(event)
|
||||
key = "split_day"
|
||||
elif precision == "year" and include_year:
|
||||
stats["training_year"] += 1
|
||||
month = None
|
||||
key = "split_year"
|
||||
else:
|
||||
continue
|
||||
row = event_probes._evaluate_contexts(
|
||||
reps,
|
||||
birth_date=str(request["birth_date"]),
|
||||
domain=domain,
|
||||
year=year,
|
||||
month=month,
|
||||
source=TYPED_SOURCE,
|
||||
clusters=clusters,
|
||||
set_version=set_version,
|
||||
)
|
||||
if row is None:
|
||||
continue
|
||||
stats[key] += 1
|
||||
rows.append(row)
|
||||
return rows, stats
|
||||
|
||||
|
||||
def replay(
|
||||
*,
|
||||
rows: Sequence[dict[str, Any]],
|
||||
contexts: Sequence[dict[str, Any]],
|
||||
typed: Sequence[dict[str, Any]],
|
||||
probes: Sequence[dict[str, Any]],
|
||||
true_time: str,
|
||||
prior_mode: str,
|
||||
) -> dict[str, Any]:
|
||||
raw = raw_signature_clusters(contexts)
|
||||
by_time = {stamp: row for row in rows if (stamp := _hhmm(row.get("time")))}
|
||||
merged, _trace = merge_adjacent_traced(raw, by_time)
|
||||
public = public_from_clusters(merged, rows)
|
||||
reps = [str(row["time"])[:5] for row in public]
|
||||
base = {stamp: float(row.get("score") or 0) for row in public if (stamp := _hhmm(row.get("time")))}
|
||||
scores = _percent(base) if prior_mode == "percent" else dict(base)
|
||||
conflicts = {time: 0 for time in reps}
|
||||
eliminated: set[str] = set()
|
||||
for probe in typed:
|
||||
scores, conflicts, eliminated = apply_answer(scores, conflicts, eliminated, probe, "yes", reps)
|
||||
asked = 0
|
||||
for probe in list(probes)[:ASK_COUNT]:
|
||||
answer = optimal_answer(probe, true_time)
|
||||
if answer is None:
|
||||
continue
|
||||
asked += 1
|
||||
scores, conflicts, eliminated = apply_answer(scores, conflicts, eliminated, probe, answer, reps)
|
||||
posterior = [{**row, "score": scores.get(str(row["time"])[:5], row.get("score") or 0)} for row in public]
|
||||
valid = still_valid_public(posterior, scores, eliminated, lead=SEPARATION_LEAD)
|
||||
delivery = delivery_from_public(valid)
|
||||
start, end = delivery.get("start"), delivery.get("end")
|
||||
inside = start is not None and end is not None and _minutes(start) <= _minutes(true_time) <= _minutes(end)
|
||||
active = [time for time in reps if time not in eliminated]
|
||||
gate = precision_gate(valid, scores)
|
||||
return {
|
||||
"width": delivery.get("width"),
|
||||
"truth_in_range": bool(inside),
|
||||
"top1": bool(top1_hit(scores, active, true_time, merged)),
|
||||
"tied": bool(gate["tied_for_first"]),
|
||||
"questions": asked,
|
||||
"typed_scored": len(typed),
|
||||
"probe_supply": len(probes),
|
||||
}
|
||||
|
||||
|
||||
def evaluate_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
|
||||
true_time = str(case["birth"]["time"])[:5]
|
||||
request = scoring_request_for(case, radius)
|
||||
contexts = compute_candidate_static_contexts(request)
|
||||
built = build_event_contribution_matrix(request, static_contexts=contexts)
|
||||
rows = score_from_matrix(request, built)
|
||||
times = [stamp for row in rows if (stamp := _hhmm(row.get("time")))]
|
||||
|
||||
def probes_now() -> list[dict[str, Any]]:
|
||||
return event_probes.discriminating_event_probes(
|
||||
{**request, "refresh_probes": False, "asked_probe_keys": []}, built,
|
||||
scan=window_scan(built), candidate_times=times, representative_time=true_time, today=TODAY,
|
||||
)
|
||||
|
||||
probes = probes_now()
|
||||
with narrowed_blocking():
|
||||
probes_narrow = probes_now()
|
||||
assert event_probes._existence_blocked_years is ORIGINAL_BLOCKED_YEARS
|
||||
|
||||
typed_day, stats = typed_event_probes(request, contexts, include_year=False)
|
||||
typed_all, stats_all = typed_event_probes(request, contexts, include_year=True)
|
||||
|
||||
rng = random.Random(f"{SHIFT_SEED}:{case.get('case_id')}:{radius}")
|
||||
holdout = set(holdout_event_ids(request["events"]))
|
||||
day_events = [
|
||||
index for index, event in enumerate(request["events"])
|
||||
if str(event.get("precision")) in {"day", "month"} and str(event.get("id")) not in holdout
|
||||
]
|
||||
shifted: dict[str, list[dict[str, Any]]] = {"1": [], "2": []}
|
||||
for k in (1, 2):
|
||||
if len(day_events) < k:
|
||||
continue
|
||||
for _ in range(SHIFT_REPEATS):
|
||||
picks = rng.sample(day_events, k)
|
||||
events = [
|
||||
_shift_month(event, rng.choice((-3, -2, -1, 1, 2, 3))) if index in picks else event
|
||||
for index, event in enumerate(request["events"])
|
||||
]
|
||||
moved, _ = typed_event_probes(request, contexts, include_year=False, events=events)
|
||||
shifted[str(k)].append({"typed": moved})
|
||||
|
||||
out: dict[str, Any] = {"case_id": case.get("case_id"), "radius": radius, "typed_stats": {**stats, **{
|
||||
"training_year": stats_all["training_year"], "split_year": stats_all["split_year"]}},
|
||||
"probe_supply": len(probes), "probe_supply_narrow": len(probes_narrow),
|
||||
"asked_keys_changed_narrow": [p["semantic_key"] for p in probes[:ASK_COUNT]]
|
||||
!= [p["semantic_key"] for p in probes_narrow[:ASK_COUNT]],
|
||||
"arms": {}}
|
||||
for prior in PRIORS:
|
||||
def run(typed: Sequence[dict[str, Any]], pool: Sequence[dict[str, Any]]) -> dict[str, Any]:
|
||||
return replay(rows=rows, contexts=contexts, typed=typed, probes=pool, true_time=true_time, prior_mode=prior)
|
||||
|
||||
arms = {
|
||||
"B0": run([], probes),
|
||||
"R1d": run(typed_day, probes),
|
||||
"R1y": run(typed_all, probes),
|
||||
"R3": run([], probes_narrow),
|
||||
"R3+R1d": run(typed_day, probes_narrow),
|
||||
}
|
||||
for k, trials in shifted.items():
|
||||
results = [run(item["typed"], probes) for item in trials]
|
||||
if results:
|
||||
arms[f"R2k{k}"] = {
|
||||
"trials": len(results),
|
||||
"truth_out": sum(1 for row in results if not row["truth_in_range"]),
|
||||
"top1": sum(1 for row in results if row["top1"]),
|
||||
"median_width": statistics.median(row["width"] for row in results if row["width"] is not None),
|
||||
}
|
||||
out["arms"][prior] = arms
|
||||
return out
|
||||
|
||||
|
||||
def summarize(rows: Sequence[dict[str, Any]], prior: str, radius: int, arm: str) -> dict[str, Any]:
|
||||
subset = [row["arms"][prior][arm] for row in rows if row.get("radius") == radius and not row.get("error")
|
||||
and arm in row["arms"][prior]]
|
||||
n = len(subset)
|
||||
if not n:
|
||||
return {"n": 0}
|
||||
if arm.startswith("R2k"):
|
||||
trials = sum(item["trials"] for item in subset)
|
||||
return {
|
||||
"n": n,
|
||||
"trials": trials,
|
||||
"truth_out_rate": round(sum(item["truth_out"] for item in subset) / trials, 4),
|
||||
"cases_with_truth_out": sum(1 for item in subset if item["truth_out"] > 0),
|
||||
"top1_rate": round(sum(item["top1"] for item in subset) / trials, 4),
|
||||
"median_width": statistics.median(item["median_width"] for item in subset),
|
||||
}
|
||||
return {
|
||||
"n": n,
|
||||
"top1": round(sum(item["top1"] for item in subset) / n, 4),
|
||||
"truth_in_range": round(sum(item["truth_in_range"] for item in subset) / n, 4),
|
||||
"median_width": statistics.median(item["width"] for item in subset if item["width"] is not None),
|
||||
"tie_rate": round(sum(item["tied"] for item in subset) / n, 4),
|
||||
"mean_questions": round(statistics.mean(item["questions"] for item in subset), 2),
|
||||
"mean_typed_scored": round(statistics.mean(item["typed_scored"] for item in subset), 2),
|
||||
"mean_probe_supply": round(statistics.mean(item["probe_supply"] for item in subset), 2),
|
||||
}
|
||||
|
||||
|
||||
def verdict(summary: dict[str, Any], prior: str, arm: str) -> str:
|
||||
"""Gate: top1 not lower AND truth-in-range not lower AND median width lower, all radii."""
|
||||
passes = []
|
||||
for radius in RADII:
|
||||
base = summary[prior][str(radius)]["B0"]
|
||||
cand = summary[prior][str(radius)][arm]
|
||||
if not cand.get("n"):
|
||||
return "uncertain"
|
||||
passes.append(
|
||||
cand["top1"] >= base["top1"]
|
||||
and cand["truth_in_range"] >= base["truth_in_range"]
|
||||
and cand["median_width"] < base["median_width"]
|
||||
)
|
||||
return "benefit" if all(passes) else "no_benefit"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--limit", type=int, default=0)
|
||||
parser.add_argument("--json-out", default=str(REPORT_JSON))
|
||||
args = parser.parse_args()
|
||||
cases = load_cases()
|
||||
if args.limit:
|
||||
cases = cases[: args.limit]
|
||||
rows: list[dict[str, Any]] = []
|
||||
for case in cases:
|
||||
for radius in RADII:
|
||||
try:
|
||||
rows.append(evaluate_case(case, radius))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
rows.append({"case_id": case.get("case_id"), "radius": radius, "error": f"{type(exc).__name__}: {exc}"})
|
||||
arms = ("B0", "R1d", "R1y", "R3", "R3+R1d", "R2k1", "R2k2")
|
||||
summary = {prior: {str(radius): {arm: summarize(rows, prior, radius, arm) for arm in arms}
|
||||
for radius in RADII} for prior in PRIORS}
|
||||
verdicts = {prior: {arm: verdict(summary, prior, arm) for arm in ("R1d", "R1y", "R3", "R3+R1d")} for prior in PRIORS}
|
||||
assert event_probes._existence_blocked_years is ORIGINAL_BLOCKED_YEARS
|
||||
payload = {
|
||||
"generated_for": "TASK-rectification-typed-event-scoring-research-20260929",
|
||||
"today": TODAY.isoformat(),
|
||||
"holdout": "references/real_case_calibration/minute_rectification_holdout_v4.json",
|
||||
"ask_count": ASK_COUNT,
|
||||
"separation_lead": SEPARATION_LEAD,
|
||||
"shift_seed": SHIFT_SEED,
|
||||
"shift_repeats": SHIFT_REPEATS,
|
||||
"open_set_not_blind": True,
|
||||
"summary": summary,
|
||||
"verdicts": verdicts,
|
||||
"rows": rows,
|
||||
"errors": [row for row in rows if row.get("error")],
|
||||
}
|
||||
out = Path(args.json_out)
|
||||
out.write_text(json.dumps(payload, ensure_ascii=False, indent=1, sort_keys=True) + "\n", encoding="utf-8")
|
||||
print(json.dumps({"summary": summary, "verdicts": verdicts}, ensure_ascii=False, indent=1))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user