Files
Jyotisha/scripts/research/answer_flip_tolerance.py
T
Jesse_ChenandClaude Opus 5.5 0f5442cea2 research(rectification): offline R1/R2/R3 — answer-flip tolerance, V1/V2 rerun, dasha shift arithmetic
- R1: flipping 1 answer keeps truth in range 98-100% but cuts head hit by
  a third or more; 2 flips squeeze truth out in 7-10% of ±30/±60 replays
  (two flips = 8 points = SEPARATION_LEAD).
- R2: weights do apply (research scorer == production at V0); V1/V2 are
  identity at ±30/±60 by construction and leave six-question metrics
  unchanged at ±10 -> no_benefit (measured). Supplementary V1n does not
  pass the gate.
- R3: boundary shift is ~3.8 days/minute (1.3-5.9), not 1.1; the 45-day
  gate is ~8-34 minutes. The _representative_pairs hypothesis is refuted
  (all-pairs adds no dated probes); the bottleneck is monthly evaluation.
  New finding recorded as BUG-1048 (investigating): _boundary_windows
  year-straddle exemption and positional zip misalignment bypass the gate.
- Dated errata appended (no deletions) to the 09-14/09-16 briefs and
  research docs; README board row -> 待验收. No production code, scoring,
  thresholds, gates or Skill changed.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-26 14:47:48 +08:00

282 lines
11 KiB
Python

#!/usr/bin/env python3
"""R1 (2026-09-26): yes/no answer-error tolerance of the six-question replay.
Offline only; production scoring, probe gates and Skill text are untouched.
Replays the same six probes the 2026-09-14 precision-gate / cluster-width
research used (production G0 gate, lead-8 union delivery), but with 1 or 2 of
the answered questions flipped (yes<->no). Flip sets are drawn with fixed
string seeds and repeated; an exhaustive all-combinations pass is reported as
a cross-check of the sampled numbers.
This is the public AA open set (v4), not a blind test. Every six-question
number is still an upper bound in one respect: the probe list is not re-drawn
after each answer (the same limitation as the 09-14 replays).
Run:
python3 scripts/research/answer_flip_tolerance.py # full run
python3 scripts/research/answer_flip_tolerance.py --limit 2 # smoke
"""
from __future__ import annotations
import argparse
import json
import sys
import traceback
from pathlib import Path
from typing import Any, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.active_rectification_event_engine import ( # noqa: E402
AYANAMSA,
NODE_MODE,
compute_candidate_static_contexts,
)
import scripts.rectification.event_probes as event_probes # noqa: E402
from scripts.research.cluster_width_lib import ( # noqa: E402
SEPARATION_LEAD,
delivery_from_public,
merge_adjacent_traced,
metrics_bundle,
public_from_clusters,
raw_signature_clusters,
shannon_entropy,
still_valid_public,
)
from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_request_for # noqa: E402
from scripts.research.offline_research_20260926_lib import ( # noqa: E402
FLIP_SEED,
answerable_indices,
exhaustive_flip_sets,
flipped_answers,
median_of,
rate,
sample_flip_sets,
)
from scripts.research.precision_gate_lib import finest_precision # noqa: E402
from scripts.research.precision_gate_sweep import generate_probes, score_bundle # noqa: E402
from scripts.research.probe_supply_after_six import ( # noqa: E402
ASK_COUNT,
apply_answer,
optimal_answer,
)
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
REPORT_JSON = ROOT / "docs" / "research" / "answer_flip_tolerance_2026_09_26.json"
RADII = (10, 30, 60)
FLIP_COUNTS = (0, 1, 2)
REPEATS = 30
def public_for(rows: Sequence[dict[str, Any]], contexts: Sequence[dict[str, Any]]) -> list[dict[str, Any]]:
raw = raw_signature_clusters(contexts)
by_time = {str(row["time"])[:5]: row for row in rows if str(row.get("time"))}
merged, _trace = merge_adjacent_traced(raw, by_time)
public = public_from_clusters(merged, rows)
for row in public:
row["score"] = float(row.get("score") or 0)
return public
def replay_with_answers(
*,
probes: Sequence[dict[str, Any]],
answers: Sequence[str | None],
public: Sequence[dict[str, Any]],
true_time: str,
window_times: Sequence[str],
) -> dict[str, Any]:
"""Same bookkeeping as cluster_width_probe.replay_public + evaluate_delivery,
but the answer to each asked probe comes from `answers`."""
reps = [str(row["time"])[:5] for row in public]
scores = {str(row["time"])[:5]: float(row.get("score") or 0) for row in public}
conflicts = {time: 0 for time in reps}
eliminated: set[str] = set()
for probe, answer in zip(list(probes)[:ASK_COUNT], answers):
if answer is None:
continue
scores, conflicts, eliminated = apply_answer(
scores, conflicts, eliminated, probe, answer, reps,
)
posterior = [
{**row, "score": scores.get(str(row["time"])[:5], row["score"])}
for row in public
]
valid = still_valid_public(posterior, scores, eliminated, lead=SEPARATION_LEAD)
delivery = delivery_from_public(valid)
alive = [row for row in posterior if str(row["time"])[:5] not in eliminated]
metrics = metrics_bundle(
public=alive,
true_time=true_time,
window_times=window_times,
delivery_times=delivery["times"],
delivery_width=delivery["width"],
independent=True,
entropy_scores=[scores.get(str(row["time"])[:5], 0.0) for row in alive],
)
true_rep_eliminated = any(
true_time in (row.get("cluster_times") or [str(row.get("time"))[:5]])
and str(row["time"])[:5] in eliminated
for row in public
)
return {
"top1": bool(metrics["top1_hit"]),
"coverage": bool(metrics["coverage"]),
"width": delivery["width"],
"tie": bool(metrics["tie"]),
"eliminated": len(eliminated),
"truth_eliminated": true_rep_eliminated,
"entropy": round(shannon_entropy(max(s, 0.0) for t, s in scores.items() if t not in eliminated), 4),
}
def run_case(case: dict[str, Any], radius: int, repeats: int) -> dict[str, Any]:
case_id = case["case_id"]
true_time = str(case["birth"]["time"])[:5]
request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
contexts = compute_candidate_static_contexts(request)
events = list(case.get("events") or [])
bundle = score_bundle(case, radius, events, contexts, None)
rows = bundle["rows"]
times = [str(row["time"])[:5] for row in rows]
probes = generate_probes(
bundle["request"], bundle["built"], times, true_time,
gate="G0", precision=finest_precision(events), refresh=False,
)
asked = list(probes)[:ASK_COUNT]
optimal = [optimal_answer(probe, true_time) for probe in asked]
public = public_for(rows, list(bundle["built"].get("static_contexts") or contexts))
out: dict[str, Any] = {
"case_id": case_id,
"radius": radius,
"asked": len(asked),
"answerable": len(answerable_indices(optimal)),
"public_clusters": len(public),
"runs": {},
"exhaustive": {},
}
for k in FLIP_COUNTS:
if k == 0:
sets: list[tuple[int, ...]] = [()]
full: list[tuple[int, ...]] = [()]
else:
sets = sample_flip_sets(optimal, k, repeats, case_id=case_id, radius=radius)
full = exhaustive_flip_sets(optimal, k)
out["runs"][str(k)] = [
{"flipped": list(item), **replay_with_answers(
probes=asked, answers=flipped_answers(optimal, item),
public=public, true_time=true_time, window_times=times,
)}
for item in sets
]
out["exhaustive"][str(k)] = [
{"flipped": list(item), **replay_with_answers(
probes=asked, answers=flipped_answers(optimal, item),
public=public, true_time=true_time, window_times=times,
)}
for item in full
]
return out
def summarize(case_rows: Sequence[dict[str, Any]], key: str) -> dict[str, Any]:
table: dict[str, Any] = {}
for radius in sorted({row["radius"] for row in case_rows}):
table[str(radius)] = {}
for k in FLIP_COUNTS:
runs = [run for row in case_rows if row["radius"] == radius for run in row[key].get(str(k), [])]
eligible = [row for row in case_rows if row["radius"] == radius and row[key].get(str(k))]
per_case_cov = [
all(run["coverage"] for run in row[key][str(k)]) for row in eligible
]
base = [row[key]["0"][0] for row in eligible]
table[str(radius)][str(k)] = {
"baseline_same_cases": {
"truth_in_range": rate(base, "coverage"),
"head_hit": rate(base, "top1"),
"width_median": median_of(base, "width"),
},
"cases": len(eligible),
"runs": len(runs),
"truth_in_range": rate(runs, "coverage"),
"head_hit": rate(runs, "top1"),
"width_median": median_of(runs, "width"),
"tie": rate(runs, "tie"),
"truth_eliminated": rate(runs, "truth_eliminated"),
"cases_ever_squeezed": sum(1 for ok in per_case_cov if not ok),
}
return table
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--repeats", type=int, default=REPEATS)
parser.add_argument("--radii", nargs="+", type=int, default=list(RADII))
parser.add_argument("--no-write", action="store_true")
args = parser.parse_args()
holdout = json.loads(HOLDOUT.read_text(encoding="utf-8"))
cases = list(holdout["cases"])[: args.limit or None]
assert event_probes.MIN_BOUNDARY_DAYS == 45 and event_probes.REFRESH_MIN_BOUNDARY_DAYS == 30
case_rows: list[dict[str, Any]] = []
errors: list[dict[str, Any]] = []
for case in cases:
for radius in args.radii:
try:
case_rows.append(run_case(case, radius, args.repeats))
except Exception as exc: # noqa: BLE001
errors.append({"case_id": case["case_id"], "radius": radius,
"error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc()})
print(f"done {case['case_id']}", flush=True)
assert event_probes.MIN_BOUNDARY_DAYS == 45 and event_probes.REFRESH_MIN_BOUNDARY_DAYS == 30
payload = {
"generated_at": "2026-09-26",
"nature": "offline replay on the public AA open set (v4); not a blind test, not accuracy",
"ayanamsa": AYANAMSA,
"node_mode": NODE_MODE,
"holdout": str(HOLDOUT.relative_to(ROOT)),
"case_count": len(cases),
"radii": list(args.radii),
"minute_step": MINUTE_STEP,
"ask_count": ASK_COUNT,
"flip_seed": FLIP_SEED,
"repeats": args.repeats,
"gate": "G0 production (45 / refresh 30)",
"delivery": f"lead-{SEPARATION_LEAD} union of non-eliminated clusters",
"sampled": summarize(case_rows, "runs"),
"exhaustive": summarize(case_rows, "exhaustive"),
"answerable_per_case": {
str(radius): [row["answerable"] for row in case_rows if row["radius"] == radius]
for radius in args.radii
},
"per_case": [
{
"case_id": row["case_id"],
"radius": row["radius"],
"answerable": row["answerable"],
**{
f"k{k}": {
"truth_in_range": rate(row["exhaustive"].get(str(k), []), "coverage"),
"head_hit": rate(row["exhaustive"].get(str(k), []), "top1"),
"width_median": median_of(row["exhaustive"].get(str(k), []), "width"),
}
for k in FLIP_COUNTS
},
}
for row in case_rows
],
"errors": errors,
"command": "python3 scripts/research/answer_flip_tolerance.py",
}
if not args.no_write:
REPORT_JSON.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"sampled": payload["sampled"], "errors": len(errors)}, ensure_ascii=False, indent=1))
return 0 if not errors else 1
if __name__ == "__main__":
sys.exit(main())