Files
Jyotisha/scripts/research/varga_sensitivity_rerun.py
T
Jesse_ChenandClaude Opus 5.5 0f5442cea2 research(rectification): offline R1/R2/R3 — answer-flip tolerance, V1/V2 rerun, dasha shift arithmetic
- R1: flipping 1 answer keeps truth in range 98-100% but cuts head hit by
  a third or more; 2 flips squeeze truth out in 7-10% of ±30/±60 replays
  (two flips = 8 points = SEPARATION_LEAD).
- R2: weights do apply (research scorer == production at V0); V1/V2 are
  identity at ±30/±60 by construction and leave six-question metrics
  unchanged at ±10 -> no_benefit (measured). Supplementary V1n does not
  pass the gate.
- R3: boundary shift is ~3.8 days/minute (1.3-5.9), not 1.1; the 45-day
  gate is ~8-34 minutes. The _representative_pairs hypothesis is refuted
  (all-pairs adds no dated probes); the bottleneck is monthly evaluation.
  New finding recorded as BUG-1048 (investigating): _boundary_windows
  year-straddle exemption and positional zip misalignment bypass the gate.
- Dated errata appended (no deletions) to the 09-14/09-16 briefs and
  research docs; README board row -> 待验收. No production code, scoring,
  thresholds, gates or Skill changed.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-26 14:47:48 +08:00

304 lines
12 KiB
Python

#!/usr/bin/env python3
"""R2 (2026-09-26): rerun of the M1b divisional-sensitivity weights V1 / V2.
Offline only; production scoring (equal-weight vargas), probe gates and Skill
text are untouched. BUG-692 re-labelled the 09-14 V1/V2 verdict `not_measured`
because V1/V2 printed exactly the V0 numbers. This script first checks, per
case and radius, whether the V1/V2 weights change any candidate score at all
(and whether the research scorer copy equals production at V0), then reruns
the six-question replay with the closure-document metrics.
V1n is a supplementary variant that is not in the 09-14 design: weight per
varga proportional to 1/minutes-per-ascendant-change, normalised so the mean
factor over the 11 production vargas is 1 (total varga mass unchanged, only
redistributed toward the fast vargas). It exists because V1 as specified is
capped at 1 and saturates.
Public AA open set (v4). Not a blind test, not accuracy.
Run:
python3 scripts/research/varga_sensitivity_rerun.py
python3 scripts/research/varga_sensitivity_rerun.py --limit 2 --no-write
"""
from __future__ import annotations
import argparse
import json
import sys
import traceback
from contextlib import contextmanager
from pathlib import Path
from typing import Any, Iterator, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.active_rectification_event_engine import ( # noqa: E402
AYANAMSA,
DOMAIN_CONFIG,
NODE_MODE,
compute_candidate_static_contexts,
)
import scripts.rectification.event_probes as event_probes # noqa: E402
import scripts.research.precision_gate_lib as pgl # noqa: E402
from scripts.research.cluster_width_lib import ( # noqa: E402
merge_adjacent_traced,
public_from_clusters,
raw_signature_clusters,
top1_from_public,
)
from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_request_for # noqa: E402
from scripts.research.offline_research_20260926_lib import median_of, rate # noqa: E402
from scripts.research.precision_gate_lib import ( # noqa: E402
PRODUCTION_VARGA_PREFIXES,
finest_precision,
gate_verdict,
varga_factor,
varga_minutes,
window_minutes_for_radius,
)
from scripts.research.precision_gate_sweep import ( # noqa: E402
run_variant,
score_bundle,
summarize,
varga_policy_for,
)
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
REPORT_JSON = ROOT / "docs" / "research" / "varga_sensitivity_rerun_2026_09_26.json"
RADII = (10, 30, 60)
VARIANTS = ("V0", "V1", "V2", "V1n")
EPS = 1e-9
def v1n_factor_table() -> dict[str, float]:
inverse = {prefix: 1.0 / varga_minutes(prefix) for prefix in PRODUCTION_VARGA_PREFIXES}
mean = sum(inverse.values()) / len(inverse)
return {prefix: round(value / mean, 6) for prefix, value in inverse.items()}
V1N_FACTORS = v1n_factor_table()
@contextmanager
def v1n_factors() -> Iterator[None]:
"""Swap the research module's varga_factor for the normalised table.
Only the research module `precision_gate_lib` is patched, and it is
restored on exit. Production modules are never touched.
"""
previous = pgl.varga_factor
def factor(prefix: str, window_minutes: float, cap: float = pgl.VARGA_CAP) -> float:
return float(V1N_FACTORS.get(prefix, 1.0))
pgl.varga_factor = factor
try:
yield
finally:
pgl.varga_factor = previous
def factor_table() -> dict[str, dict[str, float]]:
return {
str(radius): {
prefix: round(varga_factor(prefix, window_minutes_for_radius(radius)), 4)
for prefix in PRODUCTION_VARGA_PREFIXES
}
for radius in RADII
}
def score_map(rows: Sequence[dict[str, Any]]) -> dict[str, float]:
return {str(row["time"])[:5]: float(row.get("score") or 0) for row in rows}
def diff_stats(base: dict[str, float], other: dict[str, float]) -> dict[str, Any]:
deltas = [abs(other[time] - base[time]) for time in base if time in other]
changed = [item for item in deltas if item > EPS]
base_top = max(base, key=lambda time: (base[time], time)) if base else None
other_top = max(other, key=lambda time: (other[time], time)) if other else None
return {
"candidates": len(deltas),
"changed": len(changed),
"max_abs": round(max(deltas), 6) if deltas else 0.0,
"top_changed": base_top != other_top,
}
def engine_top1(rows: Sequence[dict[str, Any]], contexts: Sequence[dict[str, Any]], true_time: str) -> bool:
raw = raw_signature_clusters(contexts)
by_time = {str(row["time"])[:5]: row for row in rows}
merged, _ = merge_adjacent_traced(raw, by_time)
public = public_from_clusters(merged, rows)
for row in public:
row["score"] = float(row.get("score") or 0)
return top1_from_public(public, true_time)
def domain_prefix_drop(policy: Any, events: Sequence[dict[str, Any]]) -> dict[str, list[str]]:
dropped: dict[str, list[str]] = {}
for domain in sorted({str(event["domain"]) for event in events}):
prefixes = list(DOMAIN_CONFIG[domain][0])
gone = [item for item in prefixes if item not in policy.changing]
if gone:
dropped[domain] = gone
return dropped
def run_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
true_time = str(case["birth"]["time"])[:5]
request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
contexts = compute_candidate_static_contexts(request)
events = list(case.get("events") or [])
precision = finest_precision(events)
production = score_bundle(case, radius, events, contexts, None)
bundles: dict[str, dict[str, Any]] = {}
policies: dict[str, Any] = {}
for name in VARIANTS:
policy_name = "V1" if name == "V1n" else name
policy = varga_policy_for(policy_name, radius, contexts)
if name == "V1n":
with v1n_factors():
bundles[name] = score_bundle(case, radius, events, contexts, policy)
else:
bundles[name] = score_bundle(case, radius, events, contexts, policy)
policies[name] = policy
prod_scores = score_map(production["rows"])
v0_scores = score_map(bundles["V0"]["rows"])
proof = {
"production_vs_V0_policy": diff_stats(prod_scores, v0_scores),
**{
f"{name}_vs_V0": diff_stats(v0_scores, score_map(bundles[name]["rows"]))
for name in VARIANTS if name != "V0"
},
"V0_varga_hits": dict(sorted(policies["V0"].hits.items())),
"V2_changing": sorted(policies["V2"].changing),
"V2_dropped_domain_prefixes": domain_prefix_drop(policies["V2"], events),
}
measured: dict[str, Any] = {}
for name in VARIANTS:
bundle = bundles[name]
row = run_variant(
request=bundle["request"], built=bundle["built"], rows=bundle["rows"],
true_time=true_time, gate="G0", precision=precision,
)
row["engine_top1"] = engine_top1(
bundle["rows"], list(bundle["built"].get("static_contexts") or contexts), true_time,
)
row["case_id"] = case["case_id"]
measured[name] = row
return {"case_id": case["case_id"], "radius": radius, "proof": proof, "measured": measured}
def proof_summary(case_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
out: dict[str, Any] = {}
for radius in sorted({row["radius"] for row in case_rows}):
rows = [row for row in case_rows if row["radius"] == radius]
block: dict[str, Any] = {}
for key in ["production_vs_V0_policy", *[f"{name}_vs_V0" for name in VARIANTS if name != "V0"]]:
stats = [row["proof"][key] for row in rows]
block[key] = {
"cases_with_any_change": sum(1 for item in stats if item["changed"]),
"cases": len(stats),
"candidates_changed": sum(item["changed"] for item in stats),
"candidates_total": sum(item["candidates"] for item in stats),
"max_abs_delta": max((item["max_abs"] for item in stats), default=0.0),
"cases_top_changed": sum(1 for item in stats if item["top_changed"]),
}
block["V2_cases_dropping_any_domain_prefix"] = sum(
1 for row in rows if row["proof"]["V2_dropped_domain_prefixes"]
)
hits: dict[str, int] = {}
for row in rows:
for prefix, count in row["proof"]["V0_varga_hits"].items():
hits[prefix] = hits.get(prefix, 0) + int(count)
block["V0_varga_hits_total"] = dict(sorted(hits.items()))
out[str(radius)] = block
return out
def metric_summary(case_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
out: dict[str, Any] = {}
for radius in sorted({row["radius"] for row in case_rows}):
rows = [row for row in case_rows if row["radius"] == radius]
block: dict[str, Any] = {}
for name in VARIANTS:
measured = [row["measured"][name] for row in rows]
summary = summarize(measured)
summary["engine_top1"] = rate(measured, "engine_top1")
block[name] = summary
for name in VARIANTS:
if name == "V0":
continue
block[name]["verdict_vs_V0"] = gate_verdict(block["V0"], block[name])
out[str(radius)] = block
return out
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--radii", nargs="+", type=int, default=list(RADII))
parser.add_argument("--no-write", action="store_true")
args = parser.parse_args()
holdout = json.loads(HOLDOUT.read_text(encoding="utf-8"))
cases = list(holdout["cases"])[: args.limit or None]
assert event_probes.MIN_BOUNDARY_DAYS == 45 and event_probes.REFRESH_MIN_BOUNDARY_DAYS == 30
case_rows: list[dict[str, Any]] = []
errors: list[dict[str, Any]] = []
for case in cases:
for radius in args.radii:
try:
case_rows.append(run_case(case, radius))
except Exception as exc: # noqa: BLE001
errors.append({"case_id": case["case_id"], "radius": radius,
"error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc()})
print(f"done {case['case_id']}", flush=True)
assert pgl.varga_factor is varga_factor, "research patch leaked"
payload = {
"generated_at": "2026-09-26",
"nature": "offline replay on the public AA open set (v4); not a blind test, not accuracy",
"ayanamsa": AYANAMSA,
"node_mode": NODE_MODE,
"holdout": str(HOLDOUT.relative_to(ROOT)),
"case_count": len(cases),
"radii": list(args.radii),
"minute_step": MINUTE_STEP,
"gate": "G0 production (45 / refresh 30)",
"factor_table_V1": factor_table(),
"factor_table_V1n": V1N_FACTORS,
"varga_minutes": {prefix: round(varga_minutes(prefix), 3) for prefix in PRODUCTION_VARGA_PREFIXES},
"domain_prefixes": {domain: list(config[0]) for domain, config in DOMAIN_CONFIG.items()},
"proof": proof_summary(case_rows),
"metrics": metric_summary(case_rows),
"per_case": [
{
"case_id": row["case_id"], "radius": row["radius"],
"proof": row["proof"],
"measured": {
name: {key: row["measured"][name].get(key) for key in (
"top1", "coverage", "width", "engine_top1", "probes", "refresh")}
for name in VARIANTS
},
}
for row in case_rows
],
"errors": errors,
"command": "python3 scripts/research/varga_sensitivity_rerun.py",
}
if not args.no_write:
REPORT_JSON.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"proof": payload["proof"], "errors": len(errors)}, ensure_ascii=False, indent=1))
for radius, block in payload["metrics"].items():
for name, summary in block.items():
print(radius, name, {key: summary.get(key) for key in (
"top1", "coverage", "width_median", "engine_top1", "verdict_vs_V0")})
return 0 if not errors else 1
if __name__ == "__main__":
sys.exit(main())