- R1: flipping 1 answer keeps truth in range 98-100% but cuts head hit by a third or more; 2 flips squeeze truth out in 7-10% of ±30/±60 replays (two flips = 8 points = SEPARATION_LEAD). - R2: weights do apply (research scorer == production at V0); V1/V2 are identity at ±30/±60 by construction and leave six-question metrics unchanged at ±10 -> no_benefit (measured). Supplementary V1n does not pass the gate. - R3: boundary shift is ~3.8 days/minute (1.3-5.9), not 1.1; the 45-day gate is ~8-34 minutes. The _representative_pairs hypothesis is refuted (all-pairs adds no dated probes); the bottleneck is monthly evaluation. New finding recorded as BUG-1048 (investigating): _boundary_windows year-straddle exemption and positional zip misalignment bypass the gate. - Dated errata appended (no deletions) to the 09-14/09-16 briefs and research docs; README board row -> 待验收. No production code, scoring, thresholds, gates or Skill changed. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
304 lines
12 KiB
Python
304 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""R2 (2026-09-26): rerun of the M1b divisional-sensitivity weights V1 / V2.
|
|
|
|
Offline only; production scoring (equal-weight vargas), probe gates and Skill
|
|
text are untouched. BUG-692 re-labelled the 09-14 V1/V2 verdict `not_measured`
|
|
because V1/V2 printed exactly the V0 numbers. This script first checks, per
|
|
case and radius, whether the V1/V2 weights change any candidate score at all
|
|
(and whether the research scorer copy equals production at V0), then reruns
|
|
the six-question replay with the closure-document metrics.
|
|
|
|
V1n is a supplementary variant that is not in the 09-14 design: weight per
|
|
varga proportional to 1/minutes-per-ascendant-change, normalised so the mean
|
|
factor over the 11 production vargas is 1 (total varga mass unchanged, only
|
|
redistributed toward the fast vargas). It exists because V1 as specified is
|
|
capped at 1 and saturates.
|
|
|
|
Public AA open set (v4). Not a blind test, not accuracy.
|
|
|
|
Run:
|
|
python3 scripts/research/varga_sensitivity_rerun.py
|
|
python3 scripts/research/varga_sensitivity_rerun.py --limit 2 --no-write
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
import traceback
|
|
from contextlib import contextmanager
|
|
from pathlib import Path
|
|
from typing import Any, Iterator, Sequence
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
if str(ROOT) not in sys.path:
|
|
sys.path.insert(0, str(ROOT))
|
|
|
|
from scripts.active_rectification_event_engine import ( # noqa: E402
|
|
AYANAMSA,
|
|
DOMAIN_CONFIG,
|
|
NODE_MODE,
|
|
compute_candidate_static_contexts,
|
|
)
|
|
import scripts.rectification.event_probes as event_probes # noqa: E402
|
|
import scripts.research.precision_gate_lib as pgl # noqa: E402
|
|
from scripts.research.cluster_width_lib import ( # noqa: E402
|
|
merge_adjacent_traced,
|
|
public_from_clusters,
|
|
raw_signature_clusters,
|
|
top1_from_public,
|
|
)
|
|
from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_request_for # noqa: E402
|
|
from scripts.research.offline_research_20260926_lib import median_of, rate # noqa: E402
|
|
from scripts.research.precision_gate_lib import ( # noqa: E402
|
|
PRODUCTION_VARGA_PREFIXES,
|
|
finest_precision,
|
|
gate_verdict,
|
|
varga_factor,
|
|
varga_minutes,
|
|
window_minutes_for_radius,
|
|
)
|
|
from scripts.research.precision_gate_sweep import ( # noqa: E402
|
|
run_variant,
|
|
score_bundle,
|
|
summarize,
|
|
varga_policy_for,
|
|
)
|
|
|
|
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
|
|
REPORT_JSON = ROOT / "docs" / "research" / "varga_sensitivity_rerun_2026_09_26.json"
|
|
RADII = (10, 30, 60)
|
|
VARIANTS = ("V0", "V1", "V2", "V1n")
|
|
EPS = 1e-9
|
|
|
|
|
|
def v1n_factor_table() -> dict[str, float]:
|
|
inverse = {prefix: 1.0 / varga_minutes(prefix) for prefix in PRODUCTION_VARGA_PREFIXES}
|
|
mean = sum(inverse.values()) / len(inverse)
|
|
return {prefix: round(value / mean, 6) for prefix, value in inverse.items()}
|
|
|
|
|
|
V1N_FACTORS = v1n_factor_table()
|
|
|
|
|
|
@contextmanager
|
|
def v1n_factors() -> Iterator[None]:
|
|
"""Swap the research module's varga_factor for the normalised table.
|
|
|
|
Only the research module `precision_gate_lib` is patched, and it is
|
|
restored on exit. Production modules are never touched.
|
|
"""
|
|
previous = pgl.varga_factor
|
|
|
|
def factor(prefix: str, window_minutes: float, cap: float = pgl.VARGA_CAP) -> float:
|
|
return float(V1N_FACTORS.get(prefix, 1.0))
|
|
|
|
pgl.varga_factor = factor
|
|
try:
|
|
yield
|
|
finally:
|
|
pgl.varga_factor = previous
|
|
|
|
|
|
def factor_table() -> dict[str, dict[str, float]]:
|
|
return {
|
|
str(radius): {
|
|
prefix: round(varga_factor(prefix, window_minutes_for_radius(radius)), 4)
|
|
for prefix in PRODUCTION_VARGA_PREFIXES
|
|
}
|
|
for radius in RADII
|
|
}
|
|
|
|
|
|
def score_map(rows: Sequence[dict[str, Any]]) -> dict[str, float]:
|
|
return {str(row["time"])[:5]: float(row.get("score") or 0) for row in rows}
|
|
|
|
|
|
def diff_stats(base: dict[str, float], other: dict[str, float]) -> dict[str, Any]:
|
|
deltas = [abs(other[time] - base[time]) for time in base if time in other]
|
|
changed = [item for item in deltas if item > EPS]
|
|
base_top = max(base, key=lambda time: (base[time], time)) if base else None
|
|
other_top = max(other, key=lambda time: (other[time], time)) if other else None
|
|
return {
|
|
"candidates": len(deltas),
|
|
"changed": len(changed),
|
|
"max_abs": round(max(deltas), 6) if deltas else 0.0,
|
|
"top_changed": base_top != other_top,
|
|
}
|
|
|
|
|
|
def engine_top1(rows: Sequence[dict[str, Any]], contexts: Sequence[dict[str, Any]], true_time: str) -> bool:
|
|
raw = raw_signature_clusters(contexts)
|
|
by_time = {str(row["time"])[:5]: row for row in rows}
|
|
merged, _ = merge_adjacent_traced(raw, by_time)
|
|
public = public_from_clusters(merged, rows)
|
|
for row in public:
|
|
row["score"] = float(row.get("score") or 0)
|
|
return top1_from_public(public, true_time)
|
|
|
|
|
|
def domain_prefix_drop(policy: Any, events: Sequence[dict[str, Any]]) -> dict[str, list[str]]:
|
|
dropped: dict[str, list[str]] = {}
|
|
for domain in sorted({str(event["domain"]) for event in events}):
|
|
prefixes = list(DOMAIN_CONFIG[domain][0])
|
|
gone = [item for item in prefixes if item not in policy.changing]
|
|
if gone:
|
|
dropped[domain] = gone
|
|
return dropped
|
|
|
|
|
|
def run_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
|
|
true_time = str(case["birth"]["time"])[:5]
|
|
request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
|
|
contexts = compute_candidate_static_contexts(request)
|
|
events = list(case.get("events") or [])
|
|
precision = finest_precision(events)
|
|
production = score_bundle(case, radius, events, contexts, None)
|
|
bundles: dict[str, dict[str, Any]] = {}
|
|
policies: dict[str, Any] = {}
|
|
for name in VARIANTS:
|
|
policy_name = "V1" if name == "V1n" else name
|
|
policy = varga_policy_for(policy_name, radius, contexts)
|
|
if name == "V1n":
|
|
with v1n_factors():
|
|
bundles[name] = score_bundle(case, radius, events, contexts, policy)
|
|
else:
|
|
bundles[name] = score_bundle(case, radius, events, contexts, policy)
|
|
policies[name] = policy
|
|
prod_scores = score_map(production["rows"])
|
|
v0_scores = score_map(bundles["V0"]["rows"])
|
|
proof = {
|
|
"production_vs_V0_policy": diff_stats(prod_scores, v0_scores),
|
|
**{
|
|
f"{name}_vs_V0": diff_stats(v0_scores, score_map(bundles[name]["rows"]))
|
|
for name in VARIANTS if name != "V0"
|
|
},
|
|
"V0_varga_hits": dict(sorted(policies["V0"].hits.items())),
|
|
"V2_changing": sorted(policies["V2"].changing),
|
|
"V2_dropped_domain_prefixes": domain_prefix_drop(policies["V2"], events),
|
|
}
|
|
measured: dict[str, Any] = {}
|
|
for name in VARIANTS:
|
|
bundle = bundles[name]
|
|
row = run_variant(
|
|
request=bundle["request"], built=bundle["built"], rows=bundle["rows"],
|
|
true_time=true_time, gate="G0", precision=precision,
|
|
)
|
|
row["engine_top1"] = engine_top1(
|
|
bundle["rows"], list(bundle["built"].get("static_contexts") or contexts), true_time,
|
|
)
|
|
row["case_id"] = case["case_id"]
|
|
measured[name] = row
|
|
return {"case_id": case["case_id"], "radius": radius, "proof": proof, "measured": measured}
|
|
|
|
|
|
def proof_summary(case_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
|
|
out: dict[str, Any] = {}
|
|
for radius in sorted({row["radius"] for row in case_rows}):
|
|
rows = [row for row in case_rows if row["radius"] == radius]
|
|
block: dict[str, Any] = {}
|
|
for key in ["production_vs_V0_policy", *[f"{name}_vs_V0" for name in VARIANTS if name != "V0"]]:
|
|
stats = [row["proof"][key] for row in rows]
|
|
block[key] = {
|
|
"cases_with_any_change": sum(1 for item in stats if item["changed"]),
|
|
"cases": len(stats),
|
|
"candidates_changed": sum(item["changed"] for item in stats),
|
|
"candidates_total": sum(item["candidates"] for item in stats),
|
|
"max_abs_delta": max((item["max_abs"] for item in stats), default=0.0),
|
|
"cases_top_changed": sum(1 for item in stats if item["top_changed"]),
|
|
}
|
|
block["V2_cases_dropping_any_domain_prefix"] = sum(
|
|
1 for row in rows if row["proof"]["V2_dropped_domain_prefixes"]
|
|
)
|
|
hits: dict[str, int] = {}
|
|
for row in rows:
|
|
for prefix, count in row["proof"]["V0_varga_hits"].items():
|
|
hits[prefix] = hits.get(prefix, 0) + int(count)
|
|
block["V0_varga_hits_total"] = dict(sorted(hits.items()))
|
|
out[str(radius)] = block
|
|
return out
|
|
|
|
|
|
def metric_summary(case_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
|
|
out: dict[str, Any] = {}
|
|
for radius in sorted({row["radius"] for row in case_rows}):
|
|
rows = [row for row in case_rows if row["radius"] == radius]
|
|
block: dict[str, Any] = {}
|
|
for name in VARIANTS:
|
|
measured = [row["measured"][name] for row in rows]
|
|
summary = summarize(measured)
|
|
summary["engine_top1"] = rate(measured, "engine_top1")
|
|
block[name] = summary
|
|
for name in VARIANTS:
|
|
if name == "V0":
|
|
continue
|
|
block[name]["verdict_vs_V0"] = gate_verdict(block["V0"], block[name])
|
|
out[str(radius)] = block
|
|
return out
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("--limit", type=int, default=0)
|
|
parser.add_argument("--radii", nargs="+", type=int, default=list(RADII))
|
|
parser.add_argument("--no-write", action="store_true")
|
|
args = parser.parse_args()
|
|
holdout = json.loads(HOLDOUT.read_text(encoding="utf-8"))
|
|
cases = list(holdout["cases"])[: args.limit or None]
|
|
assert event_probes.MIN_BOUNDARY_DAYS == 45 and event_probes.REFRESH_MIN_BOUNDARY_DAYS == 30
|
|
case_rows: list[dict[str, Any]] = []
|
|
errors: list[dict[str, Any]] = []
|
|
for case in cases:
|
|
for radius in args.radii:
|
|
try:
|
|
case_rows.append(run_case(case, radius))
|
|
except Exception as exc: # noqa: BLE001
|
|
errors.append({"case_id": case["case_id"], "radius": radius,
|
|
"error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc()})
|
|
print(f"done {case['case_id']}", flush=True)
|
|
assert pgl.varga_factor is varga_factor, "research patch leaked"
|
|
payload = {
|
|
"generated_at": "2026-09-26",
|
|
"nature": "offline replay on the public AA open set (v4); not a blind test, not accuracy",
|
|
"ayanamsa": AYANAMSA,
|
|
"node_mode": NODE_MODE,
|
|
"holdout": str(HOLDOUT.relative_to(ROOT)),
|
|
"case_count": len(cases),
|
|
"radii": list(args.radii),
|
|
"minute_step": MINUTE_STEP,
|
|
"gate": "G0 production (45 / refresh 30)",
|
|
"factor_table_V1": factor_table(),
|
|
"factor_table_V1n": V1N_FACTORS,
|
|
"varga_minutes": {prefix: round(varga_minutes(prefix), 3) for prefix in PRODUCTION_VARGA_PREFIXES},
|
|
"domain_prefixes": {domain: list(config[0]) for domain, config in DOMAIN_CONFIG.items()},
|
|
"proof": proof_summary(case_rows),
|
|
"metrics": metric_summary(case_rows),
|
|
"per_case": [
|
|
{
|
|
"case_id": row["case_id"], "radius": row["radius"],
|
|
"proof": row["proof"],
|
|
"measured": {
|
|
name: {key: row["measured"][name].get(key) for key in (
|
|
"top1", "coverage", "width", "engine_top1", "probes", "refresh")}
|
|
for name in VARIANTS
|
|
},
|
|
}
|
|
for row in case_rows
|
|
],
|
|
"errors": errors,
|
|
"command": "python3 scripts/research/varga_sensitivity_rerun.py",
|
|
}
|
|
if not args.no_write:
|
|
REPORT_JSON.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
print(json.dumps({"proof": payload["proof"], "errors": len(errors)}, ensure_ascii=False, indent=1))
|
|
for radius, block in payload["metrics"].items():
|
|
for name, summary in block.items():
|
|
print(radius, name, {key: summary.get(key) for key in (
|
|
"top1", "coverage", "width_median", "engine_top1", "verdict_vs_V0")})
|
|
return 0 if not errors else 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|