Files
Jyotisha/scripts/research/representative_pairs_probe.py
T
Jesse_ChenandClaude Opus 5.5 0f5442cea2 research(rectification): offline R1/R2/R3 — answer-flip tolerance, V1/V2 rerun, dasha shift arithmetic
- R1: flipping 1 answer keeps truth in range 98-100% but cuts head hit by
  a third or more; 2 flips squeeze truth out in 7-10% of ±30/±60 replays
  (two flips = 8 points = SEPARATION_LEAD).
- R2: weights do apply (research scorer == production at V0); V1/V2 are
  identity at ±30/±60 by construction and leave six-question metrics
  unchanged at ±10 -> no_benefit (measured). Supplementary V1n does not
  pass the gate.
- R3: boundary shift is ~3.8 days/minute (1.3-5.9), not 1.1; the 45-day
  gate is ~8-34 minutes. The _representative_pairs hypothesis is refuted
  (all-pairs adds no dated probes); the bottleneck is monthly evaluation.
  New finding recorded as BUG-1048 (investigating): _boundary_windows
  year-straddle exemption and positional zip misalignment bypass the gate.
- Dated errata appended (no deletions) to the 09-14/09-16 briefs and
  research docs; README board row -> 待验收. No production code, scoring,
  thresholds, gates or Skill changed.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
2026-09-26 14:47:48 +08:00

475 lines
23 KiB
Python

#!/usr/bin/env python3
"""R3c (2026-09-26): does `_representative_pairs` explain "no dated probes"?
Offline only. Hypothesis under test (brief 2026-09-26): after the window
narrows, the dated (dasha-boundary) probe pool is empty because
`event_probes._representative_pairs` only pairs adjacent cluster
representatives plus first/last, and adjacent representatives are too close
for their boundaries to clear the 45 / 30 day gate.
For every v4 case and radius, at the initial stage (all candidates) and the
refresh stage (after the six-question replay, same as the 09-14 research):
* representative count and minute gaps;
* for each pair (production pairs vs. all pairs): Vimshottari boundary shift,
whether it clears the gate, whether `_boundary_windows`' positional zip is
misaligned;
* boundary dates from `_union_boundary_dates` and probes from the production
generator (before the public cap), with production pairs vs. a research
patch that returns all pairs, and vs. a research patch of
`_boundary_windows` that applies the documented gap to every pair (no
1-January exemption, off-by-k zip re-aligned). Patches are applied to the
module attribute for the duration of one call and restored immediately.
Public AA open set (v4). Not a blind test.
Run:
python3 scripts/research/representative_pairs_probe.py
"""
from __future__ import annotations
import argparse
import json
import statistics
import sys
import traceback
from contextlib import contextmanager
from pathlib import Path
from typing import Any, Iterator, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.active_rectification_event_engine import ( # noqa: E402
AYANAMSA,
NODE_MODE,
compute_candidate_static_contexts,
)
import scripts.rectification.event_probes as ep # noqa: E402
from scripts.rectification.candidate_contrast import cluster_contexts_by_signature # noqa: E402
from scripts.rectification.refinement_packet import window_scan # noqa: E402
from scripts.research.cluster_width_lib import raw_signature_clusters # noqa: E402
from scripts.research.minute_resolution_sweep import MINUTE_STEP, scoring_request_for # noqa: E402
from scripts.research.offline_research_20260926_lib import all_pairs, distribution, zip_misaligned # noqa: E402
from scripts.research.precision_gate_lib import finest_precision # noqa: E402
from scripts.research.precision_gate_sweep import ( # noqa: E402
TODAY,
evaluate_delivery,
generate_probes,
score_bundle,
)
from scripts.research.probe_supply_after_six import asked_key, separates_true # noqa: E402
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
REPORT_JSON = ROOT / "docs" / "research" / "representative_pairs_probe_2026_09_26.json"
RADII = (10, 30, 60)
PRODUCTION_PAIRS = ep._representative_pairs
def clock(context: dict[str, Any]) -> int:
return ep._clock(str(ep._context_time(context)))
def research_all_pairs(reps: Sequence[dict[str, Any]]) -> list[tuple[dict[str, Any], dict[str, Any]]]:
usable = [item for item in reps if ep._scoreable(item) and ep._context_time(item)]
return all_pairs(usable, clock)
PRODUCTION_WINDOWS = ep._boundary_windows
def strict_boundary_windows(left: Sequence[Any], right: Sequence[Any], *, min_days: int | None = None) -> list[Any]:
"""Research-only reading of the gate as documented: the minimum gap applies
to every pair of corresponding boundaries (no 1-January exemption), and an
off-by-k zip is re-aligned when some offset makes the gaps constant."""
threshold = ep.MIN_BOUNDARY_DAYS if min_days is None else min_days
one_list = [item for item in (ep._as_start_date(raw) for raw in left) if item is not None]
two_list = [item for item in (ep._as_start_date(raw) for raw in right) if item is not None]
pairs = list(zip(one_list, two_list))
for offset in (0, 1, -1, 2, -2, 3, -3):
a = one_list[offset:] if offset > 0 else one_list
b = two_list[-offset:] if offset < 0 else two_list
diffs = [(two - one).days for one, two in zip(a, b)]
if diffs and max(diffs) - min(diffs) <= 3:
pairs = list(zip(a, b))
break
windows: list[Any] = []
seen: set[tuple[int, int]] = set()
for one, two in pairs:
if abs((one - two).days) < threshold:
continue
for item in (one, two):
key = (item.year, item.month)
if key not in seen:
seen.add(key)
windows.append(item)
return windows
@contextmanager
def pairs_mode(mode: str) -> Iterator[None]:
if mode == "production":
yield
return
if mode == "strict_gate":
ep._boundary_windows = strict_boundary_windows
try:
yield
finally:
ep._boundary_windows = PRODUCTION_WINDOWS
return
ep._representative_pairs = research_all_pairs
try:
yield
finally:
ep._representative_pairs = PRODUCTION_PAIRS
def aligned_shift(left: Sequence[Any], right: Sequence[Any]) -> int | None:
best: list[int] | None = None
for offset in (0, 1, -1, 2, -2, 3, -3):
a = left[offset:] if offset > 0 else left
b = right[-offset:] if offset < 0 else right
diffs = [(two - one).days for one, two in zip(a, b)]
if diffs and (best is None or max(diffs) - min(diffs) < max(best) - min(best)):
best = diffs
return None if not best else abs(int(statistics.median(best)))
def stage_reps(built: dict[str, Any], times: Sequence[str], refresh: bool) -> list[dict[str, Any]]:
"""Mirror of the representative selection in `_discriminating_event_probe_lists`."""
full = ep._static_contexts(built)
remaining = ep._remaining_contexts(built, times) if times else []
if refresh and remaining:
work = remaining
clusters = cluster_contexts_by_signature(work)
else:
work = full
clusters = cluster_contexts_by_signature(full)
if len(clusters) < 2:
work = remaining or full
clusters = cluster_contexts_by_signature(work)
if len(clusters) < 2:
return []
reps = [cluster["representative"] for cluster in clusters if ep._scoreable(cluster["representative"])]
if len(reps) < 2:
reps = [item for item in work if ep._scoreable(item)]
return reps
def pair_rows(
pairs: Sequence[tuple[dict[str, Any], dict[str, Any]]],
*,
birth_date: str,
lo: int,
hi: int,
threshold: int,
include_pratyantar: bool,
) -> list[dict[str, Any]]:
rows = []
for left, right in pairs:
left_dates = ep._vim_start_dates(
birth_date, float(left["planet_longitudes"]["Moon"]), lo, hi,
include_pratyantar=include_pratyantar)
right_dates = ep._vim_start_dates(
birth_date, float(right["planet_longitudes"]["Moon"]), lo, hi,
include_pratyantar=include_pratyantar)
windows = ep._boundary_windows(left_dates, right_dates, min_days=threshold)
shift = aligned_shift(left_dates, right_dates)
# Replicate `_boundary_windows`' pass rule per zipped pair and say why
# each pair passed: a real gap >= threshold, or the same-year clause
# being skipped because the two dates straddle 1 January.
# A misaligned zip compares non-corresponding boundaries, so its
# "gap" passes are counted separately.
misaligned = zip_misaligned(left_dates, right_dates)
by_gap = straddle = by_misaligned = 0
for one, two in zip(left_dates, right_dates):
gap = abs((one - two).days)
if one.year == two.year and gap < threshold:
continue
if gap < threshold:
straddle += 1
elif misaligned and shift is not None and shift < threshold:
by_misaligned += 1
else:
by_gap += 1
rows.append({
"minutes": abs(clock(right) - clock(left)),
"shift_days": shift,
"clears_gate": shift is not None and shift >= threshold,
"zip_misaligned": misaligned,
"vim_windows": len(windows),
"vim_pass_by_gap": by_gap,
"vim_pass_by_year_straddle": straddle,
"vim_pass_by_misaligned_zip": by_misaligned,
})
return rows
def union_dates(reps: Sequence[dict[str, Any]], mode: str, *, birth_date: str, lo: int, hi: int, refresh: bool) -> int:
kwargs: dict[str, Any] = {}
if refresh:
kwargs = {
"min_boundary_days": ep.REFRESH_MIN_BOUNDARY_DAYS,
"include_pratyantar": True,
"varga_narayana": True,
}
with pairs_mode(mode):
return len(ep._union_boundary_dates(reps, birth_date=birth_date, lo=lo, hi=hi, **kwargs))
def probe_lists(
request: dict[str, Any],
built: dict[str, Any],
times: Sequence[str],
true_time: str,
*,
refresh: bool,
asked: Sequence[str],
mode: str,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, int]]:
payload = {**request, "refresh_probes": refresh, "asked_probe_keys": list(asked)}
original = ep._evaluate_contexts
tally = {"dasha_boundary_evaluated": 0, "dasha_boundary_split": 0}
def counted(*args: Any, **kwargs: Any) -> dict[str, Any] | None:
result = original(*args, **kwargs)
if kwargs.get("source") == "dasha_boundary":
tally["dasha_boundary_evaluated"] += 1
tally["dasha_boundary_split"] += int(result is not None)
return result
ep._evaluate_contexts = counted
try:
with pairs_mode(mode):
public, dropped = ep._discriminating_event_probe_lists(
payload, built, scan=window_scan(built), candidate_times=list(times),
representative_time=true_time, today=TODAY,
)
finally:
ep._evaluate_contexts = original
return public, dropped, tally
def count_probes(
public: Sequence[dict[str, Any]],
dropped: Sequence[dict[str, Any]],
*,
asked: Sequence[str],
true_time: str,
remaining: Sequence[str],
clusters: Sequence[dict[str, Any]],
) -> dict[str, Any]:
asked_set = set(asked)
everything = list(public) + list(dropped)
dated_all = [row for row in everything if str(row.get("source")) == "dasha_boundary"]
new_public = [row for row in public if asked_key(row) and asked_key(row) not in asked_set]
return {
"public": len(public),
"all_before_cap": len(everything),
"dasha_boundary_before_cap": len(dated_all),
"dasha_boundary_public": sum(1 for row in public if str(row.get("source")) == "dasha_boundary"),
"public_new": len(new_public),
"public_new_dasha_boundary": sum(1 for row in new_public if str(row.get("source")) == "dasha_boundary"),
"public_new_separating_truth": sum(
1 for row in new_public if separates_true(row, true_time, remaining, clusters)),
"sources": sorted({str(row.get("source")) for row in everything}),
}
def run_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
true_time = str(case["birth"]["time"])[:5]
request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
contexts = compute_candidate_static_contexts(request)
events = list(case.get("events") or [])
bundle = score_bundle(case, radius, events, contexts, None)
built, rows, req = bundle["built"], bundle["rows"], bundle["request"]
times = [str(row["time"])[:5] for row in rows]
initial = generate_probes(req, built, times, true_time, gate="G0",
precision=finest_precision(events), refresh=False)
delivery = evaluate_delivery(rows=rows, contexts=list(built.get("static_contexts") or contexts),
probes=initial, true_time=true_time)
remaining = delivery["remaining"]
asked = delivery["asked_keys"]
clusters = raw_signature_clusters(list(built.get("static_contexts") or contexts))
birth_date = str(req["birth_date"])
birth_year = int(birth_date[:4])
lo, hi = birth_year + 5, min(TODAY.year, birth_year + 80)
out: dict[str, Any] = {
"case_id": case["case_id"], "radius": radius, "birth_year": birth_year, "stages": {},
}
for stage, refresh, stage_times, stage_asked in (
("initial", False, times, ()),
("refresh", True, remaining, asked),
):
if refresh and len(remaining) < 2:
out["stages"][stage] = {"skipped": "fewer_than_two_remaining", "remaining": len(remaining)}
continue
reps = stage_reps(built, stage_times, refresh)
threshold = ep.REFRESH_MIN_BOUNDARY_DAYS if refresh else ep.MIN_BOUNDARY_DAYS
block: dict[str, Any] = {
"reps": len(reps),
"rep_span_minutes": (clock(max(reps, key=clock)) - clock(min(reps, key=clock))) if reps else 0,
"threshold": threshold,
}
prod_pairs = PRODUCTION_PAIRS(reps)
every_pair = research_all_pairs(reps)
for label, pairs in (("production_pairs", prod_pairs), ("all_pairs", every_pair)):
rows_ = pair_rows(pairs, birth_date=birth_date, lo=lo, hi=hi, threshold=threshold,
include_pratyantar=refresh)
block[label] = {
"count": len(rows_),
"clearing_gate": sum(1 for row in rows_ if row["clears_gate"]),
"zip_misaligned": sum(1 for row in rows_ if row["zip_misaligned"]),
"with_vim_windows": sum(1 for row in rows_ if row["vim_windows"]),
"minutes": [row["minutes"] for row in rows_],
"shift_days": [row["shift_days"] for row in rows_],
}
block[label]["union_dates"] = union_dates(
reps, "production" if label == "production_pairs" else "all",
birth_date=birth_date, lo=lo, hi=hi, refresh=refresh)
public, dropped, tally = probe_lists(
req, built, stage_times, true_time, refresh=refresh, asked=stage_asked,
mode="production" if label == "production_pairs" else "all")
block[label]["probes"] = {**count_probes(
public, dropped, asked=stage_asked, true_time=true_time,
remaining=stage_times, clusters=clusters), **tally}
block[label]["vim_pass_by_gap"] = sum(row["vim_pass_by_gap"] for row in rows_)
block[label]["vim_pass_by_year_straddle"] = sum(row["vim_pass_by_year_straddle"] for row in rows_)
block[label]["vim_pass_by_misaligned_zip"] = sum(row["vim_pass_by_misaligned_zip"] for row in rows_)
strict_public, strict_dropped, strict_tally = probe_lists(
req, built, stage_times, true_time, refresh=refresh, asked=stage_asked, mode="strict_gate")
block["strict_gate"] = {
"union_dates": union_dates(reps, "strict_gate", birth_date=birth_date, lo=lo, hi=hi, refresh=refresh),
"probes": {**count_probes(
strict_public, strict_dropped, asked=stage_asked, true_time=true_time,
remaining=stage_times, clusters=clusters), **strict_tally},
}
out["stages"][stage] = block
return out
def summarize(case_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
out: dict[str, Any] = {}
for radius in sorted({row["radius"] for row in case_rows}):
out[str(radius)] = {}
for stage in ("initial", "refresh"):
blocks = [row["stages"].get(stage) or {} for row in case_rows if row["radius"] == radius]
live = [block for block in blocks if "reps" in block]
summary: dict[str, Any] = {
"cases": len(blocks),
"cases_birth_before_1900": sum(
1 for row in case_rows
if row["radius"] == radius and row.get("birth_year", 2000) < 1900),
"cases_measured": len(live),
"cases_skipped": len(blocks) - len(live),
"reps": distribution([block["reps"] for block in live], 1),
"rep_span_minutes": distribution([block["rep_span_minutes"] for block in live], 1),
}
for label in ("production_pairs", "all_pairs"):
parts = [block[label] for block in live]
probes = [part["probes"] for part in parts]
summary[label] = {
"pairs_mean": round(statistics.fmean([p["count"] for p in parts]), 2) if parts else None,
"pairs_clearing_gate_mean": round(statistics.fmean([p["clearing_gate"] for p in parts]), 2) if parts else None,
"cases_with_zero_pairs_clearing_gate": sum(1 for p in parts if p["clearing_gate"] == 0),
"pairs_zip_misaligned_share": round(
sum(p["zip_misaligned"] for p in parts) / max(sum(p["count"] for p in parts), 1), 3),
"pair_minutes": distribution([m for p in parts for m in p["minutes"]], 1),
"pair_shift_days": distribution([d for p in parts for d in p["shift_days"] if d is not None], 1),
"union_dates_mean": round(statistics.fmean([p["union_dates"] for p in parts]), 2) if parts else None,
"cases_zero_union_dates": sum(1 for p in parts if p["union_dates"] == 0),
"dasha_boundary_before_cap_mean": round(statistics.fmean([q["dasha_boundary_before_cap"] for q in probes]), 2) if probes else None,
"cases_zero_dasha_boundary_before_cap": sum(1 for q in probes if q["dasha_boundary_before_cap"] == 0),
"public_mean": round(statistics.fmean([q["public"] for q in probes]), 2) if probes else None,
"public_new_mean": round(statistics.fmean([q["public_new"] for q in probes]), 2) if probes else None,
"public_new_dasha_boundary_mean": round(statistics.fmean([q["public_new_dasha_boundary"] for q in probes]), 2) if probes else None,
"public_new_separating_truth_mean": round(statistics.fmean([q["public_new_separating_truth"] for q in probes]), 2) if probes else None,
"cases_zero_public_new": sum(1 for q in probes if q["public_new"] == 0),
"dasha_boundary_evaluated_mean": round(statistics.fmean([q["dasha_boundary_evaluated"] for q in probes]), 2) if probes else None,
"dasha_boundary_split_rate": round(
sum(q["dasha_boundary_split"] for q in probes)
/ max(sum(q["dasha_boundary_evaluated"] for q in probes), 1), 3),
"vim_pass_by_gap_total": sum(p["vim_pass_by_gap"] for p in parts),
"vim_pass_by_year_straddle_total": sum(p["vim_pass_by_year_straddle"] for p in parts),
"vim_pass_by_misaligned_zip_total": sum(p["vim_pass_by_misaligned_zip"] for p in parts),
}
strict = [block["strict_gate"] for block in live]
summary["strict_gate"] = {
"union_dates_mean": round(statistics.fmean([p["union_dates"] for p in strict]), 2) if strict else None,
"dasha_boundary_before_cap_mean": round(statistics.fmean([p["probes"]["dasha_boundary_before_cap"] for p in strict]), 2) if strict else None,
"cases_zero_dasha_boundary_before_cap": sum(1 for p in strict if p["probes"]["dasha_boundary_before_cap"] == 0),
"public_new_mean": round(statistics.fmean([p["probes"]["public_new"] for p in strict]), 2) if strict else None,
"public_new_dasha_boundary_mean": round(statistics.fmean([p["probes"]["public_new_dasha_boundary"] for p in strict]), 2) if strict else None,
"cases_zero_public_new": sum(1 for p in strict if p["probes"]["public_new"] == 0),
"cases_zero_public_new_dasha_boundary": sum(1 for p in strict if p["probes"]["public_new_dasha_boundary"] == 0),
}
for label in ("production_pairs", "all_pairs"):
summary[label]["cases_zero_public_new_dasha_boundary"] = sum(
1 for block in live if block[label]["probes"]["public_new_dasha_boundary"] == 0)
out[str(radius)][stage] = summary
return out
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=0)
parser.add_argument("--radii", nargs="+", type=int, default=list(RADII))
parser.add_argument("--no-write", action="store_true")
args = parser.parse_args()
holdout = json.loads(HOLDOUT.read_text(encoding="utf-8"))
cases = list(holdout["cases"])[: args.limit or None]
case_rows: list[dict[str, Any]] = []
errors: list[dict[str, Any]] = []
for case in cases:
for radius in args.radii:
try:
case_rows.append(run_case(case, radius))
except Exception as exc: # noqa: BLE001
errors.append({"case_id": case["case_id"], "radius": radius,
"error": f"{type(exc).__name__}: {exc}", "trace": traceback.format_exc()})
assert ep._representative_pairs is PRODUCTION_PAIRS, "research patch leaked"
assert ep._boundary_windows is PRODUCTION_WINDOWS, "research patch leaked"
assert ep._evaluate_contexts.__name__ == "_evaluate_contexts", "research patch leaked"
print(f"done {case['case_id']}", flush=True)
assert ep.MIN_BOUNDARY_DAYS == 45 and ep.REFRESH_MIN_BOUNDARY_DAYS == 30
payload = {
"generated_at": "2026-09-26",
"nature": "offline replay on the public AA open set (v4); not a blind test",
"ayanamsa": AYANAMSA,
"node_mode": NODE_MODE,
"holdout": str(HOLDOUT.relative_to(ROOT)),
"case_count": len(cases),
"radii": list(args.radii),
"minute_step": MINUTE_STEP,
"today": TODAY.isoformat(),
"summary": summarize(case_rows),
"per_case": [
{
"case_id": row["case_id"], "radius": row["radius"], "birth_year": row["birth_year"],
"stages": {
stage: (block if "reps" not in block else {
"reps": block["reps"], "rep_span_minutes": block["rep_span_minutes"],
**{label: {k: v for k, v in block[label].items() if k not in {"minutes", "shift_days"}}
for label in ("production_pairs", "all_pairs")},
"strict_gate": block["strict_gate"],
})
for stage, block in row["stages"].items()
},
}
for row in case_rows
],
"errors": errors,
"command": "python3 scripts/research/representative_pairs_probe.py",
}
if not args.no_write:
REPORT_JSON.write_text(json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"summary": payload["summary"], "errors": len(errors)}, ensure_ascii=False, indent=1))
return 0 if not errors else 1
if __name__ == "__main__":
sys.exit(main())