research(rectification): archive partial varga-resolution study (BUG-1105)
Archive research scripts, regression tests, M1 results and safe M0 smoke. Keep the incomplete study and failing quick gate explicit. Exclude full M0 JSON, raw logs and unrelated oracle newline changes. Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,302 @@
|
||||
"""Offline segment-oriented rectification research for BUG-1105.
|
||||
|
||||
The production candidate scorer is only used as an observation source. This
|
||||
module never changes production defaults. A segment is a maximal contiguous
|
||||
run of equal divisional ascendant sign; equal signs separated by another sign
|
||||
remain different segments.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from datetime import date
|
||||
from typing import Any, Iterable, Sequence
|
||||
|
||||
from scripts.active_rectification_event_engine import compute_candidate_static_contexts
|
||||
from scripts.rectification.refinement_packet import window_scan
|
||||
from scripts.rectification.scoring_service import build_event_contribution_matrix, score_from_matrix
|
||||
from scripts.research.minute_resolution_sweep import scoring_request_for
|
||||
from scripts.research.probe_supply_after_six import ASK_COUNT, apply_answer, optimal_answer
|
||||
from scripts.research.scoring_research_lib import (
|
||||
public_for,
|
||||
replay,
|
||||
reconcile_rows,
|
||||
score_map,
|
||||
truth_cluster_times,
|
||||
)
|
||||
from scripts.research.cluster_width_lib import SEPARATION_LEAD, delivery_from_public, still_valid_public
|
||||
from scripts.rectification.event_probes import discriminating_event_probes
|
||||
|
||||
VARGA_PREFIXES = ("D1", "D9", "D10")
|
||||
RADII = (10, 15, 30, 60)
|
||||
THRESHOLDS = (0.5, 0.6, 0.7, 0.8, 0.9)
|
||||
TODAY = date(2026, 9, 16)
|
||||
|
||||
|
||||
def clock(stamp: str) -> int:
|
||||
hour, minute = str(stamp)[:5].split(":")
|
||||
return int(hour) * 60 + int(minute)
|
||||
|
||||
|
||||
def sign_value(row: dict[str, Any], prefix: str) -> Any:
|
||||
value = row.get(prefix)
|
||||
if isinstance(value, dict):
|
||||
return value.get("sign_idx", value.get("sign"))
|
||||
return value
|
||||
|
||||
|
||||
def segment_rows(rows: Sequence[dict[str, Any]], prefix: str) -> list[dict[str, Any]]:
|
||||
"""Return maximal sampled runs without merging a non-contiguous sign.
|
||||
|
||||
Clock time is cyclic here: 23:59 followed by 00:00 is one minute apart.
|
||||
A missing sample (for example 23:59 followed by 00:01) still starts a new
|
||||
run. This deliberately uses the observation order rather than a set of
|
||||
signs, so A-B-A produces three separately numbered segments.
|
||||
"""
|
||||
segments: list[dict[str, Any]] = []
|
||||
current: dict[str, Any] | None = None
|
||||
for row in rows:
|
||||
stamp = str(row.get("time") or row.get("stamp") or "")[:5]
|
||||
value = sign_value(row, prefix)
|
||||
if not stamp or value is None:
|
||||
current = None
|
||||
continue
|
||||
minute = clock(stamp)
|
||||
contiguous = (
|
||||
current is not None
|
||||
and (minute - int(current["last_minute"])) % 1440 == 1
|
||||
)
|
||||
if current is None or current["value"] != value or not contiguous:
|
||||
current = {
|
||||
"segment_id": len(segments),
|
||||
"varga": prefix,
|
||||
"value": value,
|
||||
"start": stamp,
|
||||
"end": stamp,
|
||||
"times": [stamp],
|
||||
"last_minute": minute,
|
||||
}
|
||||
segments.append(current)
|
||||
else:
|
||||
current["end"] = stamp
|
||||
current["times"].append(stamp)
|
||||
current["last_minute"] = minute
|
||||
for segment in segments:
|
||||
segment.pop("last_minute", None)
|
||||
return segments
|
||||
|
||||
|
||||
def segment_members(rows: Sequence[dict[str, Any]], prefix: str) -> list[list[str]]:
|
||||
return [list(item["times"]) for item in segment_rows(rows, prefix)]
|
||||
|
||||
|
||||
def row_signature(context: dict[str, Any], prefix: str) -> int | str | None:
|
||||
if prefix == "D1":
|
||||
value = context.get("ascendant_index")
|
||||
else:
|
||||
charts = context.get("varga_charts") or {}
|
||||
value = ((charts.get(prefix) or {}).get("Ascendant") or {}).get("sign_idx")
|
||||
return value if isinstance(value, (int, str)) else None
|
||||
|
||||
|
||||
def contexts_to_rows(contexts: Sequence[dict[str, Any]], prefixes: Iterable[str] = VARGA_PREFIXES) -> list[dict[str, Any]]:
|
||||
rows: list[dict[str, Any]] = []
|
||||
for context in contexts:
|
||||
feature = context.get("feature")
|
||||
stamp = feature.get("time") if isinstance(feature, dict) else None
|
||||
if not stamp:
|
||||
stamp = context["candidate_at"].strftime("%H:%M")
|
||||
rows.append({"time": str(stamp)[:5], **{p: row_signature(context, p) for p in prefixes}})
|
||||
return rows
|
||||
|
||||
|
||||
def truth_segment(rows: Sequence[dict[str, Any]], prefix: str, truth_time: str) -> dict[str, Any] | None:
|
||||
for segment in segment_rows(rows, prefix):
|
||||
if truth_time[:5] in segment["times"]:
|
||||
return segment
|
||||
return None
|
||||
|
||||
|
||||
def unique_sign_count(rows: Sequence[dict[str, Any]], prefix: str) -> int:
|
||||
return len({sign_value(row, prefix) for row in rows if sign_value(row, prefix) is not None})
|
||||
|
||||
|
||||
def unique_segment_count(rows: Sequence[dict[str, Any]], prefix: str) -> int:
|
||||
return len(segment_rows(rows, prefix))
|
||||
|
||||
|
||||
def window_payload(case: dict[str, Any], radius: int) -> tuple[dict[str, Any], list[dict[str, Any]], list[dict[str, Any]]]:
|
||||
"""Build the two-minute scoring grid and an independent one-minute scan grid."""
|
||||
scoring_request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
|
||||
scoring_request["minute_step"] = 2
|
||||
scoring_contexts = compute_candidate_static_contexts(scoring_request)
|
||||
scan_request = scoring_request_for({**case, "candidate_radius_minutes": radius}, radius)
|
||||
scan_request["minute_step"] = 1
|
||||
scan_contexts = compute_candidate_static_contexts(scan_request)
|
||||
return scoring_request, scoring_contexts, contexts_to_rows(scan_contexts)
|
||||
|
||||
|
||||
def _probe_payload(request: dict[str, Any], built: dict[str, Any], times: Sequence[str], true_time: str) -> list[dict[str, Any]]:
|
||||
return discriminating_event_probes(
|
||||
{**request, "refresh_probes": False, "asked_probe_keys": []},
|
||||
built,
|
||||
scan=window_scan(built),
|
||||
candidate_times=list(times),
|
||||
representative_time=true_time,
|
||||
today=TODAY,
|
||||
)
|
||||
|
||||
|
||||
def replay_state(rows: Sequence[dict[str, Any]], contexts: Sequence[dict[str, Any]], request: dict[str, Any], true_time: str, *, probes: Sequence[dict[str, Any]] | None = None, answers: Sequence[str | None] | None = None) -> dict[str, Any]:
|
||||
times = [str(row["time"])[:5] for row in rows]
|
||||
public = public_for(rows, contexts)
|
||||
reps = [str(row["time"])[:5] for row in public]
|
||||
scores = {stamp: float(row.get("score") or 0) for stamp, row in ((str(item["time"])[:5], item) for item in public)}
|
||||
conflicts = {stamp: 0 for stamp in reps}
|
||||
eliminated: set[str] = set()
|
||||
actual_probes = list(probes) if probes is not None else _probe_payload(request, {"static_contexts": list(contexts), "rows": list(rows)}, times, true_time)
|
||||
given = list(answers) if answers is not None else [optimal_answer(p, true_time) for p in actual_probes[:ASK_COUNT]]
|
||||
for probe, answer in zip(actual_probes[:ASK_COUNT], given):
|
||||
if answer in {"yes", "no", "weak_yes"}:
|
||||
scores, conflicts, eliminated = apply_answer(scores, conflicts, eliminated, probe, answer, reps)
|
||||
posterior = [{**row, "score": scores.get(str(row["time"])[:5], row.get("score") or 0)} for row in public]
|
||||
valid = still_valid_public(posterior, scores, eliminated, lead=SEPARATION_LEAD)
|
||||
return {
|
||||
"result": {"questions": sum(answer is not None for answer in given)},
|
||||
"public": public,
|
||||
"posterior": posterior,
|
||||
"valid": valid,
|
||||
"scores": scores,
|
||||
"eliminated": eliminated,
|
||||
"probes": actual_probes,
|
||||
"delivery": delivery_from_public(valid),
|
||||
}
|
||||
|
||||
|
||||
def native_case(case: dict[str, Any], radius: int, *, do_reconcile: bool = False) -> dict[str, Any]:
|
||||
request, contexts, chart_rows = window_payload(case, radius)
|
||||
true_time = str(case["birth"]["time"])[:5]
|
||||
built = build_event_contribution_matrix(request, static_contexts=contexts)
|
||||
rows = score_from_matrix(request, built)
|
||||
times = [str(row["time"])[:5] for row in rows]
|
||||
probes = _probe_payload(request, built, times, true_time)
|
||||
state = replay_state(rows, contexts, request, true_time, probes=probes)
|
||||
reconciliation = {"status": "not_run"}
|
||||
if do_reconcile:
|
||||
from scripts.research.scoring_research_lib import FeatureStore, d60_charts, make_recording_provider
|
||||
store = FeatureStore()
|
||||
recording = build_event_contribution_matrix(request, row_provider=make_recording_provider(contexts, store, d60_charts(contexts)), static_contexts=contexts)
|
||||
recording_rows = score_from_matrix(request, recording)
|
||||
reconciliation = reconcile_rows(rows, recording_rows)
|
||||
return {"case_id": str(case["case_id"]), "radius": radius, "true_time": true_time, "request": request, "contexts": contexts, "chart_rows": chart_rows, "rows": rows, "probes": probes, "state": state, "reconciliation": reconciliation}
|
||||
|
||||
|
||||
def valid_minute_scores(
|
||||
state: dict[str, Any],
|
||||
chart_rows: Sequence[dict[str, Any]],
|
||||
scores: dict[str, float] | None = None,
|
||||
) -> dict[str, float]:
|
||||
"""Project representative posterior scores onto each minute in its cluster."""
|
||||
effective_scores = scores if scores is not None else state["scores"]
|
||||
output: dict[str, float] = {}
|
||||
for row in state["posterior"]:
|
||||
representative = str(row["time"])[:5]
|
||||
value = float(effective_scores.get(representative, row.get("score") or 0))
|
||||
for stamp in row.get("cluster_times") or [representative]:
|
||||
output[str(stamp)[:5]] = value
|
||||
return output
|
||||
|
||||
|
||||
def segment_metrics(state: dict[str, Any], chart_rows: Sequence[dict[str, Any]], prefix: str, true_time: str, mode: str) -> dict[str, Any]:
|
||||
segments = segment_rows(chart_rows, prefix)
|
||||
scores = {str(key)[:5]: float(value) for key, value in state["scores"].items()}
|
||||
if mode == "percent":
|
||||
total = sum(max(value, 0.0) for value in scores.values())
|
||||
scores = (
|
||||
{key: max(value, 0.0) / total * 100.0 for key, value in scores.items()}
|
||||
if total > 0
|
||||
else {key: 0.0 for key in scores}
|
||||
)
|
||||
minute_scores = valid_minute_scores(state, chart_rows, scores)
|
||||
valid_times = {str(t)[:5] for row in state["valid"] for t in (row.get("cluster_times") or [row.get("time")])}
|
||||
truth = truth_segment(chart_rows, prefix, true_time)
|
||||
qualities: list[float] = []
|
||||
for segment in segments:
|
||||
values = [minute_scores.get(t, 0.0) for t in segment["times"] if t in valid_times]
|
||||
if mode == "uniform":
|
||||
quality = float(len(values))
|
||||
elif mode == "percent":
|
||||
quality = sum(max(v, 0.0) for v in values)
|
||||
else:
|
||||
quality = sum(values)
|
||||
qualities.append(quality)
|
||||
total = sum(qualities)
|
||||
share = max(qualities) / total if qualities and total > 0 else None
|
||||
leaders = [i for i, value in enumerate(qualities) if share is not None and abs(value - max(qualities)) <= 1e-9]
|
||||
truth_id = truth["segment_id"] if truth else None
|
||||
retained = truth_id is not None and truth_id in {segment["segment_id"] for segment in segments if any(t in valid_times for t in segment["times"])}
|
||||
correct = truth_id is not None and truth_id in leaders
|
||||
valid_segment_count = sum(1 for segment in segments if any(t in valid_times for t in segment["times"]))
|
||||
return {
|
||||
"prefix": prefix,
|
||||
"mode": mode,
|
||||
"segment_count_window": len(segments),
|
||||
"valid_segment_count": valid_segment_count,
|
||||
"truth_segment_id": truth_id,
|
||||
"truth_retained": bool(retained),
|
||||
"top_segment_correct": bool(correct),
|
||||
"top_segment_tie": len(leaders) > 1,
|
||||
"top_segment_ids": leaders,
|
||||
"top_share": None if share is None else round(share, 8),
|
||||
"segment_qualities": [round(v, 8) for v in qualities],
|
||||
}
|
||||
|
||||
|
||||
def threshold_scan(rows: Sequence[dict[str, Any]], thresholds: Sequence[float] = THRESHOLDS) -> dict[str, Any]:
|
||||
out: dict[str, Any] = {}
|
||||
for threshold in thresholds:
|
||||
eligible = [row for row in rows if row.get("top_share") is not None and float(row["top_share"]) >= threshold]
|
||||
out[str(threshold)] = {"n": len(eligible), "denominator": len(rows), "coverage": round(len(eligible) / len(rows), 8) if rows else None, "accuracy": round(sum(bool(row.get("top_segment_correct")) for row in eligible) / len(eligible), 8) if eligible else None, "truth_retained": round(sum(bool(row.get("truth_retained")) for row in eligible) / len(eligible), 8) if eligible else None}
|
||||
return out
|
||||
|
||||
|
||||
def choose_loo_threshold(training: Sequence[dict[str, Any]], thresholds: Sequence[float] = THRESHOLDS, minimum: int = 5) -> float | None:
|
||||
if not training:
|
||||
return None
|
||||
candidates = []
|
||||
required = min(minimum, len(training))
|
||||
for threshold in thresholds:
|
||||
eligible = [row for row in training if row.get("top_share") is not None and float(row["top_share"]) >= threshold]
|
||||
if len(eligible) < required or not eligible:
|
||||
continue
|
||||
accuracy = sum(bool(row.get("top_segment_correct")) for row in eligible) / len(eligible)
|
||||
retained = sum(bool(row.get("truth_retained")) for row in eligible) / len(eligible)
|
||||
candidates.append((accuracy, retained, len(eligible), -float(threshold), float(threshold)))
|
||||
if not candidates:
|
||||
return None
|
||||
return max(candidates)[-1]
|
||||
|
||||
|
||||
def segment_probe_score(probe: dict[str, Any], chart_rows: Sequence[dict[str, Any]], prefix: str, minute_weights: dict[str, float]) -> float:
|
||||
segments = segment_rows(chart_rows, prefix)
|
||||
segment_by_time = {t: segment["segment_id"] for segment in segments for t in segment["times"]}
|
||||
yes = {str(t)[:5] for item in probe.get("expected_outcomes") or [] if item.get("answer_class") in {"yes", "weak_yes"} for t in item.get("supports") or []}
|
||||
no = {str(t)[:5] for item in probe.get("expected_outcomes") or [] if item.get("answer_class") == "no" for t in item.get("supports") or []}
|
||||
totals: dict[int, float] = defaultdict(float)
|
||||
for stamp in yes | no:
|
||||
if stamp in segment_by_time:
|
||||
totals[segment_by_time[stamp]] += max(minute_weights.get(stamp, 0.0), 0.0)
|
||||
if len(totals) < 2:
|
||||
return 0.0
|
||||
values = sorted(totals.values(), reverse=True)
|
||||
return round(values[0] - values[1], 8)
|
||||
|
||||
|
||||
def reorder_probes_by_segments(probes: Sequence[dict[str, Any]], chart_rows: Sequence[dict[str, Any]], prefix: str, minute_weights: dict[str, float]) -> list[dict[str, Any]]:
|
||||
return sorted(enumerate(probes), key=lambda item: (-segment_probe_score(item[1], chart_rows, prefix, minute_weights), item[0])) and [item[1] for item in sorted(enumerate(probes), key=lambda item: (-segment_probe_score(item[1], chart_rows, prefix, minute_weights), item[0]))]
|
||||
|
||||
|
||||
def strategy_prefixes(domain: str) -> tuple[str, ...]:
|
||||
if domain == "career": return ("D1", "D10")
|
||||
if domain == "relationship": return ("D1", "D9")
|
||||
return VARGA_PREFIXES
|
||||
@@ -0,0 +1,221 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline M1 segment-quality and leave-one-out analysis for BUG-1105.
|
||||
|
||||
This deliberately reuses ``native_case`` and does not modify production
|
||||
rectification code. The output contains both full-sample fixed-threshold
|
||||
figures and case-level leave-one-out validation; no threshold is selected from
|
||||
the validation case itself.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from scripts.research.varga_resolution_lib import ( # noqa: E402
|
||||
RADII,
|
||||
THRESHOLDS,
|
||||
VARGA_PREFIXES,
|
||||
native_case,
|
||||
segment_metrics,
|
||||
choose_loo_threshold,
|
||||
threshold_scan,
|
||||
)
|
||||
|
||||
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v5.json"
|
||||
SCHEMA = "bug-1105-varga-resolution-m1-v1"
|
||||
MODES = ("raw", "percent", "uniform")
|
||||
|
||||
|
||||
def load_cases() -> list[dict[str, Any]]:
|
||||
return list(json.loads(HOLDOUT.read_text(encoding="utf-8")).get("cases") or [])
|
||||
|
||||
|
||||
def is_lmt(case: dict[str, Any]) -> bool:
|
||||
return int(str(case.get("birth", {}).get("date", "9999"))[:4]) < 1900
|
||||
|
||||
|
||||
def metrics_for_case(case: dict[str, Any], radius: int) -> dict[str, Any]:
|
||||
result = native_case(case, radius, do_reconcile=False)
|
||||
state = result["state"]
|
||||
rows = result["chart_rows"]
|
||||
true_time = result["true_time"]
|
||||
return {
|
||||
"case_id": str(case["case_id"]),
|
||||
"radius": radius,
|
||||
"lmt_before_1900": is_lmt(case),
|
||||
"answered_count": int((state.get("result") or {}).get("questions") or 0),
|
||||
"probe_count": len(result.get("probes") or []),
|
||||
"by_varga": {
|
||||
prefix: {
|
||||
mode: segment_metrics(state, rows, prefix, true_time, mode)
|
||||
for mode in MODES
|
||||
}
|
||||
for prefix in VARGA_PREFIXES
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def aggregate_rows(items: Sequence[dict[str, Any]], prefix: str, mode: str) -> dict[str, Any]:
|
||||
rows = [item["by_varga"][prefix][mode] for item in items]
|
||||
denominator = len(rows)
|
||||
return {
|
||||
"denominator": denominator,
|
||||
"top_share_mean": round(sum(float(r["top_share"] or 0) for r in rows) / denominator, 8) if denominator else None,
|
||||
"truth_retained": sum(bool(r["truth_retained"]) for r in rows),
|
||||
"truth_retained_rate": round(sum(bool(r["truth_retained"]) for r in rows) / denominator, 8) if denominator else None,
|
||||
"truth_excluded": sum(not bool(r["truth_retained"]) for r in rows),
|
||||
"top_segment_correct": sum(bool(r["top_segment_correct"]) for r in rows),
|
||||
"top_segment_correct_rate": round(sum(bool(r["top_segment_correct"]) for r in rows) / denominator, 8) if denominator else None,
|
||||
"top_segment_ties": sum(bool(r.get("top_segment_tie")) for r in rows),
|
||||
"top_segment_tie_rate": round(sum(bool(r.get("top_segment_tie")) for r in rows) / denominator, 8) if denominator else None,
|
||||
"valid_segment_count_le_2": sum(int(r["valid_segment_count"]) <= 2 for r in rows),
|
||||
"valid_segment_count_le_2_rate": round(sum(int(r["valid_segment_count"]) <= 2 for r in rows) / denominator, 8) if denominator else None,
|
||||
"thresholds_full_fit": threshold_scan(rows),
|
||||
}
|
||||
|
||||
|
||||
def stratified_rows(items: Sequence[dict[str, Any]], prefix: str, mode: str) -> dict[str, Any]:
|
||||
"""Keep the pre-1900 LMT stratum auditable without changing denominators."""
|
||||
return {
|
||||
"all": aggregate_rows(items, prefix, mode),
|
||||
"lmt_before_1900": aggregate_rows(
|
||||
[item for item in items if item["lmt_before_1900"]], prefix, mode
|
||||
),
|
||||
"post_1900": aggregate_rows(
|
||||
[item for item in items if not item["lmt_before_1900"]], prefix, mode
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def summarize_loo_rows(rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
|
||||
eligible = [row for row in rows if row["eligible"]]
|
||||
return {
|
||||
"eligible": len(eligible),
|
||||
"denominator": len(rows),
|
||||
"coverage": round(len(eligible) / len(rows), 8) if rows else None,
|
||||
"accuracy": round(sum(bool(row["top_segment_correct"]) for row in eligible) / len(eligible), 8) if eligible else None,
|
||||
"truth_retained": round(sum(bool(row["truth_retained"]) for row in eligible) / len(eligible), 8) if eligible else None,
|
||||
}
|
||||
|
||||
|
||||
def loo_fixed(items: Sequence[dict[str, Any]], prefix: str, mode: str) -> dict[str, Any]:
|
||||
out: dict[str, Any] = {}
|
||||
for threshold in THRESHOLDS:
|
||||
selected = [item["by_varga"][prefix][mode] for item in items if float(item["by_varga"][prefix][mode]["top_share"] or 0) >= threshold]
|
||||
out[str(threshold)] = {
|
||||
"eligible": len(selected),
|
||||
"denominator": len(items),
|
||||
"coverage": round(len(selected) / len(items), 8) if items else None,
|
||||
"accuracy": round(sum(bool(row["top_segment_correct"]) for row in selected) / len(selected), 8) if selected else None,
|
||||
"truth_retained": round(sum(bool(row["truth_retained"]) for row in selected) / len(selected), 8) if selected else None,
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
def loo_selected(items: Sequence[dict[str, Any]], prefix: str, mode: str) -> dict[str, Any]:
|
||||
validations: list[dict[str, Any]] = []
|
||||
for index, item in enumerate(items):
|
||||
training = [other["by_varga"][prefix][mode] for j, other in enumerate(items) if j != index]
|
||||
threshold = choose_loo_threshold(training)
|
||||
row = item["by_varga"][prefix][mode]
|
||||
eligible = threshold is not None and float(row["top_share"] or 0) >= threshold
|
||||
validations.append({
|
||||
"case_id": item["case_id"],
|
||||
"threshold": threshold,
|
||||
"eligible": bool(eligible),
|
||||
"top_segment_correct": bool(row["top_segment_correct"]) if eligible else None,
|
||||
"truth_retained": bool(row["truth_retained"]) if eligible else None,
|
||||
"lmt_before_1900": item["lmt_before_1900"],
|
||||
})
|
||||
eligible = [row for row in validations if row["eligible"]]
|
||||
return {
|
||||
"eligible": len(eligible),
|
||||
"denominator": len(validations),
|
||||
"coverage": round(len(eligible) / len(validations), 8) if validations else None,
|
||||
"accuracy": round(sum(bool(row["top_segment_correct"]) for row in eligible) / len(eligible), 8) if eligible else None,
|
||||
"truth_retained": round(sum(bool(row["truth_retained"]) for row in eligible) / len(eligible), 8) if eligible else None,
|
||||
"selected_threshold_counts": {str(t): sum(row["threshold"] == t for row in validations) for t in THRESHOLDS},
|
||||
"validation_rows": validations,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--radii", default=",".join(str(value) for value in RADII))
|
||||
parser.add_argument("--limit", type=int, default=0)
|
||||
parser.add_argument("--json-out", required=True)
|
||||
args = parser.parse_args()
|
||||
radii = tuple(int(value) for value in str(args.radii).split(",") if value.strip())
|
||||
cases = load_cases()
|
||||
if args.limit:
|
||||
cases = cases[: args.limit]
|
||||
items: list[dict[str, Any]] = []
|
||||
errors: list[dict[str, Any]] = []
|
||||
for case in cases:
|
||||
for radius in radii:
|
||||
label = f"{case.get('case_id')} ±{radius}"
|
||||
try:
|
||||
items.append(metrics_for_case(case, radius))
|
||||
print(label, flush=True)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
errors.append({"case_id": str(case.get("case_id")), "radius": radius, "error": f"{type(exc).__name__}: {exc}"})
|
||||
print(label, "ERROR", type(exc).__name__, exc, flush=True)
|
||||
aggregates: list[dict[str, Any]] = []
|
||||
for radius in radii:
|
||||
subset = [item for item in items if item["radius"] == radius]
|
||||
by_varga: dict[str, Any] = {}
|
||||
for prefix in VARGA_PREFIXES:
|
||||
by_varga[prefix] = {}
|
||||
for mode in MODES:
|
||||
loo = loo_selected(subset, prefix, mode)
|
||||
by_varga[prefix][mode] = {
|
||||
"full_fit": stratified_rows(subset, prefix, mode),
|
||||
"loo_fixed_thresholds": loo_fixed(subset, prefix, mode),
|
||||
"loo_selected_threshold": {
|
||||
**summarize_loo_rows(loo["validation_rows"]),
|
||||
"selected_threshold_counts": loo["selected_threshold_counts"],
|
||||
"validation_rows": loo["validation_rows"],
|
||||
"lmt_before_1900": summarize_loo_rows([row for row in loo["validation_rows"] if row["lmt_before_1900"]]),
|
||||
"post_1900": summarize_loo_rows([row for row in loo["validation_rows"] if not row["lmt_before_1900"]]),
|
||||
},
|
||||
}
|
||||
aggregates.append({
|
||||
"radius": radius,
|
||||
"case_count": len(subset),
|
||||
"lmt_case_count": sum(item["lmt_before_1900"] for item in subset),
|
||||
"by_varga": by_varga,
|
||||
})
|
||||
payload = {
|
||||
"schema": SCHEMA,
|
||||
"metadata": {
|
||||
"holdout": str(HOLDOUT.relative_to(ROOT)).replace("\\", "/"),
|
||||
"case_count_requested": len(cases),
|
||||
"case_count_completed": len({item["case_id"] for item in items}),
|
||||
"radii": list(radii),
|
||||
"vargas": list(VARGA_PREFIXES),
|
||||
"modes": list(MODES),
|
||||
"thresholds": list(THRESHOLDS),
|
||||
"loo": "one case held out; threshold selected only from the other cases; validation case never used for selection",
|
||||
"lmt_stratum": "birth year < 1900, reported separately",
|
||||
"production_code_modified": False,
|
||||
"deterministic_json": True,
|
||||
},
|
||||
"aggregates": aggregates,
|
||||
"items": items,
|
||||
"errors": errors,
|
||||
}
|
||||
out = Path(args.json_out)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
out.write_text(json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
||||
return 0 if not errors and len(items) == len(cases) * len(radii) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,349 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the deterministic BUG-1105 varga-resolution M0 replay.
|
||||
|
||||
The production scorer remains untouched. Scoring candidates are sampled at a
|
||||
fixed two-minute step, while an independent one-minute context scan supplies
|
||||
varga rising-sign segments. ``refresh_probes=False`` is intentional and is
|
||||
recorded in the machine result; an empty probe pool is reported rather than
|
||||
silently replaced with refreshed dasha-boundary probes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, Sequence
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from scripts.research.cluster_width_lib import delivery_from_public # noqa: E402
|
||||
from scripts.research.varga_resolution_lib import ( # noqa: E402
|
||||
RADII,
|
||||
VARGA_PREFIXES,
|
||||
native_case,
|
||||
segment_rows,
|
||||
sign_value,
|
||||
)
|
||||
|
||||
HOLDOUT = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v5.json"
|
||||
SCHEMA = "bug-1105-varga-resolution-v1"
|
||||
SCORING_STEP = 2
|
||||
SCAN_STEP = 1
|
||||
REFRESH_PROBES = False
|
||||
|
||||
|
||||
def load_cases() -> list[dict[str, Any]]:
|
||||
payload = json.loads(HOLDOUT.read_text(encoding="utf-8"))
|
||||
return list(payload.get("cases") or [])
|
||||
|
||||
|
||||
def hhmm(value: object) -> str:
|
||||
return str(value or "")[:5]
|
||||
|
||||
|
||||
def clock(stamp: str) -> int:
|
||||
hour, minute = stamp[:5].split(":")
|
||||
return int(hour) * 60 + int(minute)
|
||||
|
||||
|
||||
def in_envelope(stamp: str, start: str | None, end: str | None) -> bool:
|
||||
if not start or not end:
|
||||
return False
|
||||
value, lower, upper = clock(stamp), clock(start), clock(end)
|
||||
return lower <= value <= upper if lower <= upper else value >= lower or value <= upper
|
||||
|
||||
|
||||
def stable_unique(values: Iterable[str]) -> list[str]:
|
||||
return list(dict.fromkeys(str(value)[:5] for value in values if value))
|
||||
|
||||
|
||||
def interval_times(chart_rows: Sequence[dict[str, Any]], delivery: dict[str, Any]) -> list[str]:
|
||||
return [
|
||||
hhmm(row.get("time"))
|
||||
for row in chart_rows
|
||||
if in_envelope(hhmm(row.get("time")), delivery.get("start"), delivery.get("end"))
|
||||
]
|
||||
|
||||
|
||||
def valid_candidate_times(state: dict[str, Any]) -> list[str]:
|
||||
values: list[str] = []
|
||||
for row in state.get("valid") or []:
|
||||
values.extend(hhmm(value) for value in row.get("cluster_times") or [row.get("time")])
|
||||
return stable_unique(values)
|
||||
|
||||
|
||||
def represented_segments(
|
||||
chart_rows: Sequence[dict[str, Any]],
|
||||
prefix: str,
|
||||
times: Iterable[str],
|
||||
) -> list[dict[str, Any]]:
|
||||
wanted = set(stable_unique(times))
|
||||
return [
|
||||
segment for segment in segment_rows(chart_rows, prefix)
|
||||
if wanted.intersection(segment.get("times") or [])
|
||||
]
|
||||
|
||||
|
||||
def represented_values(rows: Sequence[dict[str, Any]], prefix: str, times: Iterable[str]) -> list[Any]:
|
||||
wanted = set(stable_unique(times))
|
||||
return stable_values(sign_value(row, prefix) for row in rows if hhmm(row.get("time")) in wanted)
|
||||
|
||||
|
||||
def stable_values(values: Iterable[Any]) -> list[Any]:
|
||||
result: list[Any] = []
|
||||
for value in values:
|
||||
if value is None or value in result:
|
||||
continue
|
||||
result.append(value)
|
||||
return result
|
||||
|
||||
|
||||
def majority(values: Sequence[Any], truth: Any) -> tuple[bool | None, bool]:
|
||||
if not values:
|
||||
return None, False
|
||||
counts = Counter(values)
|
||||
peak = max(counts.values())
|
||||
leaders = {value for value, count in counts.items() if count == peak}
|
||||
return (truth in leaders if len(leaders) == 1 else None), len(leaders) > 1
|
||||
|
||||
|
||||
def window_summary(chart_rows: Sequence[dict[str, Any]]) -> dict[str, Any]:
|
||||
counts: dict[str, int] = {}
|
||||
segments: dict[str, int] = {}
|
||||
for prefix in VARGA_PREFIXES:
|
||||
counts[prefix] = len(stable_values(sign_value(row, prefix) for row in chart_rows))
|
||||
segments[prefix] = len(segment_rows(chart_rows, prefix))
|
||||
combinations = len({tuple(sign_value(row, prefix) for prefix in VARGA_PREFIXES) for row in chart_rows})
|
||||
return {
|
||||
"scan_point_count": len(chart_rows),
|
||||
"sign_counts": counts,
|
||||
"segment_counts": segments,
|
||||
"combination_count": combinations,
|
||||
"d1_single_sign": counts["D1"] == 1,
|
||||
}
|
||||
|
||||
|
||||
def interval_summary(
|
||||
chart_rows: Sequence[dict[str, Any]],
|
||||
state: dict[str, Any],
|
||||
true_time: str,
|
||||
) -> dict[str, Any]:
|
||||
delivery = state.get("delivery") or {}
|
||||
envelope = interval_times(chart_rows, delivery)
|
||||
real = valid_candidate_times(state)
|
||||
values: dict[str, dict[str, Any]] = {}
|
||||
for prefix in VARGA_PREFIXES:
|
||||
truth_row = next((row for row in chart_rows if hhmm(row.get("time")) == true_time), None)
|
||||
truth = sign_value(truth_row or {}, prefix)
|
||||
envelope_segments = represented_segments(chart_rows, prefix, envelope)
|
||||
real_segments = represented_segments(chart_rows, prefix, real)
|
||||
envelope_values = [sign_value(row, prefix) for row in chart_rows if hhmm(row.get("time")) in set(envelope)]
|
||||
exact_values = represented_values(chart_rows, prefix, real)
|
||||
envelope_majority, envelope_tie = majority(envelope_values, truth)
|
||||
exact_majority, exact_tie = majority(exact_values, truth)
|
||||
truth_segment = next(
|
||||
(segment for segment in segment_rows(chart_rows, prefix) if true_time in (segment.get("times") or [])),
|
||||
None,
|
||||
)
|
||||
truth_segment_id = truth_segment.get("segment_id") if truth_segment else None
|
||||
values[prefix] = {
|
||||
"truth_sign": truth,
|
||||
"truth_segment_id": truth_segment_id,
|
||||
"envelope_start": delivery.get("start"),
|
||||
"envelope_end": delivery.get("end"),
|
||||
"envelope_scan_point_count": len(envelope),
|
||||
"envelope_sign_count": len(stable_values(envelope_values)),
|
||||
"envelope_segment_count": len(envelope_segments),
|
||||
"envelope_majority_truth": envelope_majority,
|
||||
"envelope_majority_tie": envelope_tie,
|
||||
"real_valid_candidate_count": len(real),
|
||||
"real_valid_sign_count": len(exact_values),
|
||||
"real_valid_segment_count": len(real_segments),
|
||||
"real_valid_majority_truth": exact_majority,
|
||||
"real_valid_majority_tie": exact_tie,
|
||||
"truth_segment_retained_in_real_set": truth_segment_id is not None and any(
|
||||
segment.get("segment_id") == truth_segment_id for segment in real_segments
|
||||
),
|
||||
}
|
||||
combo_envelope = {
|
||||
tuple(sign_value(row, prefix) for prefix in VARGA_PREFIXES)
|
||||
for row in chart_rows
|
||||
if hhmm(row.get("time")) in set(envelope)
|
||||
}
|
||||
combo_real = {
|
||||
tuple(sign_value(row, prefix) for prefix in VARGA_PREFIXES)
|
||||
for row in chart_rows
|
||||
if hhmm(row.get("time")) in set(real)
|
||||
}
|
||||
truth_row = next((row for row in chart_rows if hhmm(row.get("time")) == true_time), None)
|
||||
truth_combo = tuple(sign_value(truth_row or {}, prefix) for prefix in VARGA_PREFIXES)
|
||||
envelope_combo_majority, envelope_combo_tie = majority(
|
||||
[tuple(sign_value(row, prefix) for prefix in VARGA_PREFIXES) for row in chart_rows if hhmm(row.get("time")) in set(envelope)],
|
||||
truth_combo,
|
||||
)
|
||||
return {
|
||||
"scoring_candidate_times": real,
|
||||
"interval_envelope": {
|
||||
"start": delivery.get("start"),
|
||||
"end": delivery.get("end"),
|
||||
"scan_point_count": len(envelope),
|
||||
"times": envelope,
|
||||
},
|
||||
"by_varga": values,
|
||||
"combination": {
|
||||
"envelope_count": len(combo_envelope),
|
||||
"real_valid_count": len(combo_real),
|
||||
"envelope_majority_truth": envelope_combo_majority,
|
||||
"envelope_majority_tie": envelope_combo_tie,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def case_result(case: dict[str, Any], radius: int) -> dict[str, Any]:
|
||||
result = native_case(case, radius, do_reconcile=True)
|
||||
state = result["state"]
|
||||
chart_rows = result["chart_rows"]
|
||||
return {
|
||||
"case_id": str(case.get("case_id") or ""),
|
||||
"radius": radius,
|
||||
"true_time": result["true_time"],
|
||||
"scoring_candidate_count": len(result["rows"]),
|
||||
"scan_point_count": len(chart_rows),
|
||||
"probe_count": len(result["probes"]),
|
||||
"answered_count": int((state.get("result") or {}).get("questions") or 0),
|
||||
"reconciliation": result["reconciliation"],
|
||||
"window": window_summary(chart_rows),
|
||||
"six_question_delivery": interval_summary(chart_rows, state, result["true_time"]),
|
||||
}
|
||||
|
||||
|
||||
def ratio(numerator: int, denominator: int) -> float | None:
|
||||
return round(numerator / denominator, 8) if denominator else None
|
||||
|
||||
|
||||
def aggregate(results: Sequence[dict[str, Any]], radius: int) -> dict[str, Any]:
|
||||
rows = [row for row in results if row["radius"] == radius]
|
||||
n = len(rows)
|
||||
window = {
|
||||
"case_count": n,
|
||||
"d1_single_sign": sum(row["window"]["d1_single_sign"] for row in rows),
|
||||
"mean_sign_counts": {
|
||||
prefix: round(sum(row["window"]["sign_counts"][prefix] for row in rows) / n, 8) if n else None
|
||||
for prefix in VARGA_PREFIXES
|
||||
},
|
||||
"mean_segment_counts": {
|
||||
prefix: round(sum(row["window"]["segment_counts"][prefix] for row in rows) / n, 8) if n else None
|
||||
for prefix in VARGA_PREFIXES
|
||||
},
|
||||
"mean_combination_count": round(sum(row["window"]["combination_count"] for row in rows) / n, 8) if n else None,
|
||||
}
|
||||
by_varga: dict[str, Any] = {}
|
||||
for prefix in VARGA_PREFIXES:
|
||||
items = [row["six_question_delivery"]["by_varga"][prefix] for row in rows]
|
||||
by_varga[prefix] = {
|
||||
"envelope_only_one_sign": sum(item["envelope_sign_count"] == 1 for item in items),
|
||||
"envelope_at_most_two_signs": sum(item["envelope_sign_count"] <= 2 for item in items),
|
||||
"envelope_majority_truth": sum(item["envelope_majority_truth"] is True for item in items),
|
||||
"envelope_majority_ties": sum(item["envelope_majority_tie"] for item in items),
|
||||
"real_truth_segment_retained": sum(item["truth_segment_retained_in_real_set"] for item in items),
|
||||
"real_at_most_two_segments": sum(item["real_valid_segment_count"] <= 2 for item in items),
|
||||
"denominator": n,
|
||||
}
|
||||
for key in (
|
||||
"envelope_only_one_sign",
|
||||
"envelope_at_most_two_signs",
|
||||
"envelope_majority_truth",
|
||||
"real_truth_segment_retained",
|
||||
"real_at_most_two_segments",
|
||||
):
|
||||
by_varga[prefix][f"{key}_rate"] = ratio(by_varga[prefix][key], n)
|
||||
combo = [row["six_question_delivery"]["combination"] for row in rows]
|
||||
return {
|
||||
"radius": radius,
|
||||
"case_count": n,
|
||||
"reconciliation": {
|
||||
"all_zero_diff": all(not row["reconciliation"].get("changed") for row in rows),
|
||||
"case_count": n,
|
||||
"changed_case_count": sum(bool(row["reconciliation"].get("changed")) for row in rows),
|
||||
"denominator": n,
|
||||
},
|
||||
"window": window,
|
||||
"delivery_envelope": {
|
||||
"by_varga": by_varga,
|
||||
"combination_only_one": sum(item["envelope_count"] == 1 for item in combo),
|
||||
"combination_at_most_two": sum(item["envelope_count"] <= 2 for item in combo),
|
||||
"combination_majority_truth": sum(item["envelope_majority_truth"] is True for item in combo),
|
||||
"combination_majority_ties": sum(item["envelope_majority_tie"] for item in combo),
|
||||
"denominator": n,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--radii", default=",".join(str(value) for value in RADII))
|
||||
parser.add_argument("--vargas", default=",".join(VARGA_PREFIXES))
|
||||
parser.add_argument("--limit", type=int, default=0)
|
||||
parser.add_argument("--json-out", required=True)
|
||||
args = parser.parse_args()
|
||||
radii = tuple(int(value) for value in str(args.radii).split(",") if value.strip())
|
||||
vargas = tuple(value.strip() for value in str(args.vargas).split(",") if value.strip())
|
||||
cases = load_cases()
|
||||
if args.limit:
|
||||
cases = cases[: args.limit]
|
||||
results: list[dict[str, Any]] = []
|
||||
errors: list[dict[str, str | int]] = []
|
||||
for case in cases:
|
||||
for radius in radii:
|
||||
label = f"{case.get('case_id')} ±{radius}"
|
||||
try:
|
||||
row = case_result(case, radius)
|
||||
results.append(row)
|
||||
print(
|
||||
f"{label} score={row['scoring_candidate_count']} scan={row['scan_point_count']} "
|
||||
f"probes={row['probe_count']} reconcile_changed={bool(row['reconciliation'].get('changed'))}",
|
||||
flush=True,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
errors.append({"case_id": str(case.get("case_id") or ""), "radius": radius, "error": f"{type(exc).__name__}: {exc}"})
|
||||
print(f"{label} ERROR {type(exc).__name__}: {exc}", flush=True)
|
||||
payload = {
|
||||
"schema": SCHEMA,
|
||||
"metadata": {
|
||||
"holdout": str(HOLDOUT.relative_to(ROOT)).replace("\\", "/"),
|
||||
"case_count_requested": len(cases),
|
||||
"case_count_completed": len({row["case_id"] for row in results}),
|
||||
"radii": list(radii),
|
||||
"vargas": list(vargas),
|
||||
"ayanamsa": "raman",
|
||||
"node_mode": "mean",
|
||||
"scoring_candidate_step_minutes": SCORING_STEP,
|
||||
"segment_scan_step_minutes": SCAN_STEP,
|
||||
"refresh_probes": REFRESH_PROBES,
|
||||
"replay_probe_source": "existing event_probes.discriminating_event_probes; no refresh",
|
||||
"questions_requested": 6,
|
||||
"questions_answered_is_recorded_per_case": True,
|
||||
"real_valid_candidate_set": "union of cluster_times from still_valid_public after replay; scoring-grid candidates only",
|
||||
"interval_envelope": "unionStillValidRange equivalent: min/max clock edges over the real valid candidate clusters; envelope is not the real set",
|
||||
"segment_definition": "maximal contiguous one-minute scan run of equal D1/D9/D10 sign; repeated non-contiguous signs retain separate IDs; 23:59->00:00 is contiguous",
|
||||
"aggregation_denominator": "77 cases per radius when full run completes; empty coverage and ties are separate counts",
|
||||
"reconciliation_scope": "every case and radius, scoring grid only; single-case zero-diff is smoke evidence, not full-set evidence",
|
||||
"deterministic_json": True,
|
||||
},
|
||||
"aggregates": [aggregate(results, radius) for radius in radii],
|
||||
"results": results,
|
||||
"errors": errors,
|
||||
}
|
||||
out = Path(args.json_out)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
out.write_text(json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
||||
print(json.dumps(payload["aggregates"], ensure_ascii=False, indent=2, sort_keys=True), flush=True)
|
||||
return 0 if not errors and all(item["reconciliation"]["all_zero_diff"] for item in payload["aggregates"]) else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user