fix(rectification): use candidate dates for cross-midnight dasha scoring

Add date-isolated caches and regression coverage, align scoring identity, and freeze full research reruns while preserving historical artifacts. Record unresolved cache/receipt identity and end-to-end acceptance gaps for branch review only.

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
jesse-ux
2026-09-20 13:56:11 +08:00
co-authored by Claude Code
parent 03cba4780a
commit aa46da1016
49 changed files with 62080 additions and 109 deletions
@@ -42,7 +42,7 @@
"confirmation_allowed": false,
"accept_allowed": false,
"confirm_allowed": false,
"representative_candidate_id": "85ca5490-07ab-54b7-acf0-59385f705015",
"representative_candidate_id": "e8c11277-022c-519c-95e9-6a0a04e8bf23",
"representative_time": "12:00",
"overall_confidence": "low",
"margin_percent": 5.3006,
@@ -1150,7 +1150,7 @@
},
"candidate_feature_snapshot": {
"calculation_spec_hash": "e03dfc555a247e78549768cc0b487cb427414633982212debd9cd6e60bb67292",
"algorithm_version": "rectification-v5-matrix-scoring-7",
"algorithm_version": "rectification-v5-matrix-scoring-8",
"candidate_count": 3,
"feature_hash": "36190982db75d3c9fe7c584607f25342531049c3d751060075519dbbd68033db",
"features": [
@@ -0,0 +1,156 @@
"""Candidate-local dasha dates; public AA replay and synthetic cache boundaries."""
from __future__ import annotations
import hashlib
import json
from datetime import date, datetime, timedelta
import pytest
from scripts.active_rectification_event_engine import compute_candidate_static_contexts
from scripts.rectification import dasha_transition_proximity as proximity
from scripts.rectification.scoring_service import (
build_event_contribution_matrix,
public_technique_layers,
score_from_matrix,
)
from scripts.research.reported_offset_sweep import shifted_window
from scripts.research.sealed_holdout_rerun import DATASET
def _canonical(value):
return json.dumps(value, sort_keys=True, separators=(",", ":")).encode()
def _cases():
return json.loads(DATASET.read_text(encoding="utf-8"))["cases"]
def test_fixed_scoring_identity_is_exposed_without_changing_input_contract():
from scripts.rectification.api_service import engine_scoring_versions
from scripts.rectification.scoring_service import ALGORITHM_VERSION, INPUT_CONTRACT_VERSION
assert engine_scoring_versions()["algorithm_version"] == ALGORITHM_VERSION == "rectification-v5-matrix-scoring-8"
assert INPUT_CONTRACT_VERSION == "rectification-calculation-spec-v4"
@pytest.mark.parametrize("start", ["2000-01-01", "2000-02-29", "2000-12-31"])
def test_every_candidate_uses_own_date_and_caches_do_not_cross_dates(monkeypatch, start):
# Intentionally identical synthetic chart values: only date distinguishes caches.
anchor = date.fromisoformat(start)
moments = [datetime.combine(anchor, datetime.min.time()) + timedelta(hours=23, minutes=50+i)
for i in range(21)]
contexts = [{"candidate_at": at, "feature": {"time": at.strftime("%H:%M")},
"planet_longitudes": {"Moon": 42.0}, "ascendant_index": 1}
for at in moments]
event_date = date(2020, 1, 10)
calls = {"ad": [], "pd": [], "narayana": []}
def vim(birth_date, moon, lo, hi, *, include_pratyantar=False):
calls["pd" if include_pratyantar else "ad"].append(birth_date)
delta = (date.fromisoformat(birth_date) - anchor).days
return [event_date + timedelta(days=delta + (3 if include_pratyantar else 2))]
def narayana(asc, planets, birth_date, lo, hi):
calls["narayana"].append(birth_date)
delta = (date.fromisoformat(birth_date) - anchor).days
return [event_date + timedelta(days=delta + 4)]
monkeypatch.setattr(proximity, "_vim_start_dates", vim)
monkeypatch.setattr(proximity, "_narayana_start_dates", narayana)
# Two events sharing the year band exercise cache reuse, not merely a new call.
events = [{"id": precision, "domain": "career", "precision": precision,
"date_start": event_date.isoformat(), "date_end": event_date.isoformat()}
for precision in ("day", "month")]
matrix = {event["id"]: {at.strftime("%H:%M"): {"points": 2.0, "rule_ids": []}
for at in moments} for event in events}
proximity.merge_transition_proximity(matrix, events, contexts, start,
public_technique_layers=public_technique_layers)
for at in moments:
delta = (at.date() - anchor).days
for precision in ("day", "month"):
expected = proximity.score_transition_proximity(
event_date=event_date, precision=precision,
vim_starts=[event_date + timedelta(days=delta + 2)],
vim_pd_starts=[event_date + timedelta(days=delta + 3)],
narayana_starts=[event_date + timedelta(days=delta + 4)],
)
actual = matrix[precision][at.strftime("%H:%M")]
assert actual["points"] == round(2.0 + expected["points"], 4)
assert actual["rule_ids"] == sorted(expected["rule_ids"])
expected_dates = [anchor.isoformat(), (anchor + timedelta(days=1)).isoformat()]
assert calls == {kind: expected_dates for kind in calls}
def test_legacy_context_without_candidate_at_retains_request_date(monkeypatch):
calls = []
def vim(birth_date, moon, lo, hi, **kwargs):
calls.append(birth_date)
return []
monkeypatch.setattr(proximity, "_vim_start_dates", vim)
monkeypatch.setattr(proximity, "_narayana_start_dates", lambda *args: [])
proximity.merge_transition_proximity(
{"event": {"12:00": {"points": 2.0}}},
[{"id": "event", "precision": "day", "date": "2020-01-10"}],
[{"feature": {"time": "12:00"}, "planet_longitudes": {"Moon": 42.0}}],
"2000-01-01", public_technique_layers=public_technique_layers,
)
assert calls == ["2000-01-01", "2000-01-01"]
@pytest.mark.parametrize("ordinal,score_sha256", [
(1, "2aabdda6bb56baf6a9d0b119964ee1ab8023fede9ea8703f6d265207c490bec2"),
(2, "6585d895aeb79e26257b1002696c16b5e252f9a702f2a5702f929fd80486abb9"),
(3, "246296903d915e4886229530fa554428d7f65fb686f70cc369711347ebaf2610"),
])
def test_same_day_public_aa_scores_keep_pre_fix_bytes(ordinal, score_sha256):
# Golden hashes captured from the unmodified production path, 121 minutes each.
# Only ordinal and score bytes are retained; no birth data or coordinates.
request, moments = shifted_window(_cases()[ordinal - 1], 0, 60)
assert len({moment.date() for moment in moments}) == 1
built = build_event_contribution_matrix(request)
scores = [row["score"] for row in score_from_matrix(request, built)]
assert len(scores) == 121
assert hashlib.sha256(_canonical(scores)).hexdigest() == score_sha256
def test_real_cross_midnight_all_candidates_match_independent_dated_calculation(monkeypatch):
# Existing public AA case naturally crosses midnight at radius 60; no birth mutation.
request, moments = shifted_window(_cases()[5], 0, 60)
contexts = compute_candidate_static_contexts(request)
assert [context["candidate_at"] for context in contexts] == moments
assert len({moment.date() for moment in moments}) == 2
original_vim, original_narayana = proximity._vim_start_dates, proximity._narayana_start_dates
seen_vim, seen_narayana = set(), set()
def vim(birth_date, moon, lo, hi, *, include_pratyantar=False):
seen_vim.add((birth_date, round(moon, 6), include_pratyantar))
return original_vim(birth_date, moon, lo, hi, include_pratyantar=include_pratyantar)
def narayana(asc, planets, birth_date, lo, hi):
seen_narayana.add((birth_date, asc, round(planets["Moon"], 6)))
return original_narayana(asc, planets, birth_date, lo, hi)
with monkeypatch.context() as capture:
capture.setattr(proximity, "_vim_start_dates", vim)
capture.setattr(proximity, "_narayana_start_dates", narayana)
actual = build_event_contribution_matrix(request, static_contexts=contexts)
expected_rows, expected_matrix = [], {}
for context in contexts:
candidate_date = context["candidate_at"].date().isoformat()
dated_request = {**request, "birth_date": candidate_date}
# One correctly dated candidate per independent matrix, with fresh local caches.
built = build_event_contribution_matrix(dated_request, static_contexts=[context])
expected_rows.extend(score_from_matrix(dated_request, built))
for event_id, cells in built["matrix"].items():
expected_matrix.setdefault(event_id, {}).update(cells)
assert _canonical(actual["matrix"]) == _canonical(expected_matrix)
assert _canonical(score_from_matrix(request, actual)) == _canonical(expected_rows)
for context in contexts:
candidate_date = context["candidate_at"].date().isoformat()
moon = round(context["planet_longitudes"]["Moon"], 6)
assert (candidate_date, moon, False) in seen_vim
assert (candidate_date, moon, True) in seen_vim
assert (candidate_date, context["ascendant_index"], moon) in seen_narayana
@@ -117,6 +117,9 @@ def test_sealed_holdout_contract_matches_v3_report_and_stays_closed() -> None:
assert produced_by["implementation_hash_matches_at_replay"] is True
assert produced_by["source_report"] == "references/real_case_calibration/minute_rectification_holdout_v3_report.json"
assert current_tree["implementation_sha256"] == rerun["frozen_record"]["implementation_sha256"]
assert current_tree["extended_identity"] == rerun["frozen_record"]["extended_identity"]
assert "scripts/rectification/dasha_transition_proximity.py" in current_tree["extended_identity"]["production_scoring_files"]
assert product["current_tree_fixed_protocol_rerun"]["extended_identity"] == current_tree["extended_identity"]
assert current_tree["fixed_protocol_rerun_hash_matches"] is rerun["implementation_hash_matches_at_replay"] is True
assert current_tree["fixed_protocol_rerun_trial_count"] == rerun["trial_count"] == 20
assert rerun["official_valid_independent_blind"] is False
@@ -0,0 +1,9 @@
"""Collect candidate-date regressions through quick's test_rectification_*.py glob."""
from tests.test_dasha_transition_proximity_cross_midnight import ( # noqa: F401
test_every_candidate_uses_own_date_and_caches_do_not_cross_dates,
test_fixed_scoring_identity_is_exposed_without_changing_input_contract,
test_legacy_context_without_candidate_at_retains_request_date,
test_real_cross_midnight_all_candidates_match_independent_dated_calculation,
test_same_day_public_aa_scores_keep_pre_fix_bytes,
)
@@ -2,6 +2,12 @@
Golden payload in tests/golden/rectification_engine_memoization_v1.json was
produced from origin/staging @ a8d29d1b before any memoization landed.
On 2026-09-20 only two identity leaves were refreshed from the real engine:
representative_candidate_id 85ca5490-07ab-54b7-acf0-59385f705015 ->
e8c11277-022c-519c-95e9-6a0a04e8bf23, and snapshot algorithm_version
rectification-v5-matrix-scoring-7 -> rectification-v5-matrix-scoring-8.
The authorized cross-midnight version bump changes UUID identity, not this
same-day fixture's scores. Historical numeric values and feature hashes remain.
Do not compare that payload with a whole-structure ``==``. Cross-machine
libm / pyswisseph rounding already drifted ``margin_percent`` by 1.1e-3
+2 -1
View File
@@ -19,7 +19,8 @@ def _row(time: str, score: float) -> dict:
class RelativeSupportScaleTest(unittest.TestCase):
def test_holdout_gate_keeps_proportional_default(self) -> None:
self.assertEqual(POLICY_VERSION, "rectification-candidate-policy-v3")
self.assertEqual(ALGORITHM_VERSION, "rectification-v5-matrix-scoring-7")
# Cross-midnight date fix: scoring identity -7 -> -8, not a prior/policy change.
self.assertEqual(ALGORITHM_VERSION, "rectification-v5-matrix-scoring-8")
self.assertEqual(RELATIVE_SUPPORT_MODE, "proportional")
def test_offset_top_two_lead_is_at_least_proportional(self) -> None:
@@ -7,10 +7,13 @@ from tests.test_reported_offset_research import ( # noqa: F401
test_recorded_specification_and_all_prespecified_cells,
test_shifted_window_preserves_dates_across_midnight,
test_zero_offset_centres_on_truth_and_has_complete_grid,
test_sweep_changed_frozen_identity_fails_before_scoring,
)
from tests.test_sealed_holdout_contract_freshness import ( # noqa: F401
test_changed_frozen_identity_fails_before_any_replay,
test_contract_tracks_actual_current_scorer_and_dataset_audit,
test_fixed_protocol_rerun_is_auditable_but_never_independent_blind,
test_frozen_record_matches_dataset_scorer_and_evaluator_bytes,
test_extended_identity_drift_rejected_even_when_legacy_hash_unchanged,
test_historical_artifacts_are_byte_preserved_not_refreshed,
)
+42 -8
View File
@@ -68,25 +68,33 @@ def test_shifted_window_preserves_dates_across_midnight(clock, offset):
assert metrics["delivery_width_minutes"] == 31
def test_cross_midnight_real_engine_scores_match_candidate_date_replay():
def test_cross_midnight_real_engine_scores_match_candidate_date_replay(monkeypatch):
case = json.loads(DATASET.read_text(encoding="utf-8"))["cases"][5]
request, candidates = sweep.shifted_window(case, 0, 60)
assert len({candidate.date() for candidate in candidates}) == 2
contexts = sweep.compute_candidate_static_contexts(request, candidates=candidates)
grouped = sweep.score_window(request, contexts)
calls = []
build = sweep.build_event_contribution_matrix
def traced(request, **kwargs):
calls.append(len(kwargs["static_contexts"]))
return build(request, **kwargs)
with monkeypatch.context() as patch:
patch.setattr(sweep, "build_event_contribution_matrix", traced)
native = sweep.score_window(request, contexts)
assert calls == [len(contexts)]
expected = []
for context in contexts:
dated = {**request, "birth_date": context["candidate_at"].date().isoformat()}
built = sweep.build_event_contribution_matrix(dated, static_contexts=[context])
expected.extend(sweep.score_from_matrix(dated, built))
assert grouped == expected
old_matrix = sweep.build_event_contribution_matrix(request, static_contexts=contexts)
old_rows = sweep.score_from_matrix(request, old_matrix)
assert old_rows != expected
assert native == expected
native_matrix = sweep.build_event_contribution_matrix(request, static_contexts=contexts)
native_rows = sweep.score_from_matrix(request, native_matrix)
assert native_rows == expected
def test_recorded_specification_and_all_prespecified_cells():
report = json.loads((sweep.ROOT / "docs/research/reported_offset_2026_09_20.json").read_text(encoding="utf-8"))
report = json.loads(sweep.REPORT.read_text(encoding="utf-8"))
spec = report["specification"]
assert spec["ayanamsa"] == "raman"
assert spec["node_mode"] == "mean"
@@ -98,7 +106,17 @@ def test_recorded_specification_and_all_prespecified_cells():
assert spec["evaluator_sha256"] == sweep.file_sha256(sweep.ROOT / "scripts/research/reported_offset_sweep.py")
assert spec["production_scoring_sha256"] == sweep.implementation_sha256(spec["production_scoring_files"])
assert spec["research_implementation_sha256"] == sweep.implementation_sha256(spec["research_files"])
assert spec["replay_revision"] == "candidate_date_grouped_v2"
assert spec["replay_revision"] == "native_candidate_date_v3"
frozen = json.loads(sweep.FREEZE.read_text(encoding="utf-8"))
assert report["frozen_record"] == spec == frozen
sweep.verify_frozen_record(frozen, sweep.freeze_record())
assert report["implementation_hash_matches_at_replay"] is True
assert report["dataset_hash_matches_at_replay"] is True
assert frozen["frozen_at_utc"] <= report["replay_started_at_utc"] <= report["replay_finished_at_utc"]
assert spec["official_valid_independent_blind"] is False
assert spec["official_blind_trial_count"] == 0
assert spec["results_previously_seen"] is True
assert spec["must_not_use_for_tuning"] is True
assert spec["truth_hidden_from_ranker"] is True
assert spec["is_blind_evaluation"] is False
assert report["trial_count"] == 20 * len(sweep.RADII) * len(sweep.OFFSETS)
@@ -115,3 +133,19 @@ def test_recorded_specification_and_all_prespecified_cells():
assert row["truth_in_window_rate"] == expected
assert row["delivery_coverage_rate"] <= expected
assert row["top_1_rate"] <= expected
assert report["historical_comparison"] == sweep.historical_comparison(
sweep.LEGACY_REPORT, report["trials"], ("case_ordinal", "radius_minutes", "offset_minutes"),
)
@pytest.mark.parametrize("key", ["dataset_sha256", "production_scoring_sha256", "research_implementation_sha256", "evaluator_sha256"])
def test_sweep_changed_frozen_identity_fails_before_scoring(tmp_path, monkeypatch, key):
frozen = sweep.freeze_record()
frozen[key] = "0" * 64
path = tmp_path / "bad-freeze.json"
path.write_text(json.dumps(frozen), encoding="utf-8")
def unexpected(*args, **kwargs):
raise AssertionError("must reject identity before scoring")
monkeypatch.setattr(sweep, "compute_candidate_static_contexts", unexpected)
with pytest.raises(ValueError, match=f"frozen_record_mismatch:{key}"):
sweep.run(freeze_path=path)
@@ -7,7 +7,10 @@ import pytest
from scripts.minute_rectification_blind_eval import implementation_sha256, summarize_trials
from scripts.rectification.sealed_holdout import holdout_passed, load_sealed_minute_holdout
from scripts.research.sealed_holdout_rerun import DATASET, FREEZE, REPORT, file_sha256, freeze_record, run
from scripts.research.sealed_holdout_rerun import (
ARCHIVE, DATASET, FREEZE, REPORT, LEGACY_REPORT, PRODUCTION_FILES, file_sha256,
freeze_record, historical_comparison, implementation_identity, run,
)
ROOT = Path(__file__).resolve().parents[1]
@@ -21,6 +24,12 @@ def test_contract_tracks_actual_current_scorer_and_dataset_audit():
contract = read(ROOT / "references/rectification_sealed_holdout.v1.json")
actual_hash = implementation_sha256(dataset["frozen_scoring"]["files"])
assert contract["current_tree_scorer"]["implementation_sha256"] == actual_hash
assert contract["current_tree_scorer"]["extended_identity"] == implementation_identity()
historical_contract = read(ARCHIVE / "references/rectification_sealed_holdout.v1.json")
runtime_keys = ("status", "valid_public_aa_cases", "required_cases", "top_1_rate", "confirmation_coverage_rate", "sealed_benchmark_id")
for key in runtime_keys:
assert type(contract[key]) is type(historical_contract[key])
assert contract[key] == historical_contract[key]
assert contract["source_audit_status"] == dataset["source_audit_status"]
assert contract["evaluated_on"] == read(REPORT)["evaluated_on"]
assert contract["status"] == "not_ready"
@@ -37,6 +46,11 @@ def test_frozen_record_matches_dataset_scorer_and_evaluator_bytes():
assert frozen["dataset_sha256"] == file_sha256(DATASET)
assert len(frozen["files"]) == 12
assert read(DATASET)["frozen_scoring"]["implementation_sha256"] == frozen["historical_frozen_sha256"]
assert frozen["extended_identity"] == implementation_identity()
assert set(PRODUCTION_FILES) <= set(frozen["extended_identity"]["production_scoring_files"])
assert {"scripts/rectification/scoring_service.py", "scripts/rectification/dasha_transition_proximity.py"} <= set(PRODUCTION_FILES)
for path, digest in frozen["extended_identity"]["file_sha256"].items():
assert digest == file_sha256(ROOT / path)
def test_fixed_protocol_rerun_is_auditable_but_never_independent_blind():
@@ -45,6 +59,11 @@ def test_fixed_protocol_rerun_is_auditable_but_never_independent_blind():
scorer = contract["current_tree_scorer"]
rerun = contract["current_tree_fixed_protocol_rerun"]
assert report["frozen_record"] == read(FREEZE)
assert report["frozen_record"]["frozen_at_utc"] <= report["replay_started_at_utc"] <= report["replay_finished_at_utc"]
assert scorer["extended_identity"] == report["frozen_record"]["extended_identity"]
assert rerun["extended_identity"] == scorer["extended_identity"]
assert rerun["freeze_record_path"] == FREEZE.relative_to(ROOT).as_posix()
assert report["historical_comparison"] == historical_comparison(LEGACY_REPORT, report["trials"], ("case_ordinal",))
assert report["trial_count"] == len(report["trials"]) == 20
assert report["excluded_cases"] == []
aggregate = summarize_trials(report["trials"], read(DATASET)["release_metrics"])
@@ -79,3 +98,42 @@ def test_changed_frozen_identity_fails_before_any_replay(tmp_path, monkeypatch):
monkeypatch.setattr("scripts.research.sealed_holdout_rerun.build_feature_fact_rows", unexpected)
with pytest.raises(ValueError, match="frozen_record_mismatch:implementation_sha256"):
run(path)
@pytest.mark.parametrize("path", ["scripts/rectification/scoring_service.py", "scripts/rectification/dasha_transition_proximity.py"])
def test_extended_identity_drift_rejected_even_when_legacy_hash_unchanged(tmp_path, monkeypatch, path):
from scripts.research import sealed_holdout_rerun as replay
frozen = replay.freeze_record()
freeze_path = tmp_path / "extended-freeze.json"
freeze_path.write_text(json.dumps(frozen), encoding="utf-8")
original = replay.implementation_identity
def changed(dataset=DATASET):
actual = original(dataset)
actual["file_sha256"][path] = "0" * 64
return actual
monkeypatch.setattr(replay, "implementation_identity", changed)
def unexpected(*args, **kwargs):
raise AssertionError("extended drift must reject before shadow scoring")
monkeypatch.setattr(replay, "build_feature_fact_rows", unexpected)
assert frozen["implementation_sha256"] == replay.freeze_record()["implementation_sha256"]
with pytest.raises(ValueError, match="frozen_record_mismatch:extended_identity"):
replay.run(freeze_path)
def test_historical_artifacts_are_byte_preserved_not_refreshed():
manifest = read(ARCHIVE / "manifest.json")
assert file_sha256(ARCHIVE / "manifest.json") == "6102a26a840be207b5858b3a4c0509274468ae9071e86d308cbbac7371d6864e"
assert len(manifest["files"]) == 8
for record in manifest["files"]:
archived = ROOT / record["archive_path"]
assert archived.stat().st_size == record["size_bytes"]
assert file_sha256(archived) == record["sha256"]
old_freeze = ROOT / "docs/research/sealed_holdout_rerun_2026_09_20.freeze.json"
assert old_freeze.read_bytes() == (ARCHIVE / old_freeze.relative_to(ROOT)).read_bytes()
assert file_sha256(old_freeze) == "d17651cb50224acdd1af7a4692c777ed169b216a962a0675371254ce561d0921"
assert FREEZE != old_freeze
for name, digest in (
("reported_offset_2026_09_20.json", "9878f2b50c957a470fafcb2ed0a9eb16405c3f6fa422ea50f467e84b0189ace1"),
("sealed_holdout_rerun_2026_09_20.json", "40df220bfcb51a683b31fbd626896e47a5aed7ee4f30ff669543691f7ce53c1e"),
):
assert file_sha256(ARCHIVE / "docs/research" / name) == digest