"""Regression tests for the 2026-09-29 scoring-method research scaffold. Pure helpers use fictional inputs. The one integration test reconciles the research recording scorer against the production scorer on a public v4 case (task hard line 4: zero difference before any experiment). """ from __future__ import annotations import json from pathlib import Path import pytest import scripts.research.scoring_research_lib as lib ROOT = Path(__file__).resolve().parents[1] HOLDOUT_V4 = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json" def test_estimate_lr_rewards_discriminating_feature_and_leaves_noise_near_zero() -> None: examples = [] for _ in range(20): examples.append(({"good": 1.0, "noise": 1.0}, True)) examples.append(({"good": 1.0, "noise": 0.0}, True)) for _ in range(80): examples.append(({"noise": 1.0}, False)) examples.append(({"noise": 0.0}, False)) table = lib.estimate_lr(examples, alpha=1.0) assert table.positives == 40 and table.negatives == 160 assert table.present["good"] > 1.5 assert table.absent["good"] < -1.5 assert abs(table.present["noise"]) < 0.05 # reward-only policy clips the negative absent term; naive-Bayes keeps it assert table.weight("good", allow_negative=False) == (table.present["good"], 0.0) assert table.weight("good", allow_negative=True) == (table.present["good"], table.absent["good"]) def test_loo_folds_never_contain_the_held_out_case() -> None: folds = lib.loo_folds(["a", "b", "c"]) assert [held for held, _ in folds] == ["a", "b", "c"] for held, training in folds: assert held not in training and len(training) == 2 def test_event_feature_matrix_is_the_fraction_of_sample_dates() -> None: store = lib.FeatureStore() store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house", "event_kind:career"], ["pranapada_in_target_house"]) store.put("e1", "1990-02-15", "10:00", ["vim_ad_domain_lord"], []) store.put("e1", "1990-01-15", "10:02", ["no_domain_activation"], []) store.put("e1", "1990-02-15", "10:02", ["no_domain_activation"], []) matrix = lib.event_feature_matrix(store, "e1", ["10:00", "10:02"], include_extras=True) assert matrix["10:00"] == {"vim_md_domain_house": 0.5, "vim_ad_domain_lord": 0.5, "pranapada_in_target_house": 0.5} assert matrix["10:02"] == {} without = lib.event_feature_matrix(store, "e1", ["10:00"], include_extras=False) assert "pranapada_in_target_house" not in without["10:00"] def test_scoring_rule_filter_drops_event_kind_and_constant_rules() -> None: assert lib.is_scoring_rule("vim_md_domain_varga") assert lib.is_scoring_rule("controlled_transit_jupiter_domain_house") assert not lib.is_scoring_rule("event_kind:career") assert not lib.is_scoring_rule("event_kind_profile:career:change") assert not lib.is_scoring_rule("no_domain_activation") assert not lib.is_scoring_rule("occupation_auxiliary_not_primary") assert not lib.is_scoring_rule("kp_asc_sublord_is_vim_md") # extras are a separate namespace assert lib.is_extra_feature("vim_md_domain_varga_d60") def test_weighted_provider_scores_each_sample_from_the_store() -> None: store = lib.FeatureStore() store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house"], ["kp_asc_sublord_is_vim_md"]) store.put("e1", "1990-01-15", "10:02", [], []) table = lib.LRTable( alpha=1.0, positives=1, negatives=1, present={"vim_md_domain_house": 1.0, "kp_asc_sublord_is_vim_md": 0.5}, absent={"vim_md_domain_house": -0.25, "kp_asc_sublord_is_vim_md": -0.1}, support={}, ) request = {"events": [{"id": "e1", "date": "1990-01-15", "domain": "career"}]} a1 = lib.make_weighted_provider(store, table, variant="A1")(request) a2 = lib.make_weighted_provider(store, table, variant="A2")(request) a3 = lib.make_weighted_provider(store, table, variant="A3", scale=2.0)(request) assert [(r["time"], r["score"]) for r in a1] == [("10:00", 1.0), ("10:02", 0.0)] assert [(r["time"], r["score"]) for r in a2] == [("10:00", 1.5), ("10:02", 0.0)] assert [(r["time"], r["score"]) for r in a3] == [("10:00", 3.0), ("10:02", -0.7)] assert a1[0]["evidence"][0]["rule_ids"] == ["vim_md_domain_house"] def _probe() -> dict: return {"expected_outcomes": [ {"answer_class": "yes", "supports": ["10:00"], "conflicts": ["10:04"]}, {"answer_class": "no", "supports": ["10:04"], "conflicts": ["10:00"]}, ]} def test_scaled_answer_halves_the_delta_and_never_counts_a_conflict() -> None: scores = {"10:00": 10.0, "10:02": 10.0, "10:04": 10.0} conflicts = {time: 0 for time in scores} weak, weak_conflicts, weak_elim = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 0.5) assert weak == {"10:00": 9.0, "10:02": 10.0, "10:04": 11.0} assert weak_conflicts == conflicts and weak_elim == set() full, full_conflicts, _ = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 1.0) assert full == {"10:00": 8.0, "10:02": 10.0, "10:04": 12.0} assert full_conflicts["10:00"] == 1 def test_absent_years_are_gaps_strictly_inside_each_domain_span() -> None: events = [ {"domain": "career", "date": "2017-05-01", "precision": "day"}, {"domain": "career", "date": "2019", "precision": "year"}, {"domain": "career", "date": "2025-02", "precision": "month"}, {"domain": "relocation", "date": "2000", "precision": "year"}, ] assert lib.absent_years(events) == {"career": [2018, 2020, 2021, 2022, 2023, 2024]} def test_degrade_and_upgrade_precision_helpers() -> None: events = [{"id": "a", "domain": "career", "date": "2017-05-20", "precision": "day"}, {"id": "b", "domain": "career", "date": "2019", "precision": "year"}] spoken = lib.degrade_to_year(events) assert spoken[0]["precision"] == "year" and spoken[0]["date"] == "2017" assert spoken[1] == events[1] upgraded = lib.upgrade_to_month(spoken, {"a"}, {e["id"]: e for e in events}) assert upgraded[0]["precision"] == "month" and upgraded[0]["date"] == "2017-05" shifted = lib.shift_day_events_by_months(events, seed="fictional") assert shifted[0]["shift_months"] in {-3, -2, -1, 1, 2, 3} assert shifted[1] == events[1] assert lib.drop_events(events, 1, seed="x") != events and len(lib.drop_events(events, 1, seed="x")) == 1 def test_softmax_and_proportional_percent_sum_to_about_one_hundred() -> None: scores = {"10:00": 12.0, "10:02": 11.0, "10:04": 9.0} assert abs(sum(lib.percent_proportional(scores).values()) - 100) <= 2 soft = lib.percent_softmax(scores) assert abs(sum(soft.values()) - 100) <= 2 and soft["10:00"] > soft["10:02"] > soft["10:04"] assert lib.calibration_bins([(0.9, True), (0.02, False)])[-1]["observed_rate"] == 1.0 @pytest.mark.skipif(not HOLDOUT_V4.exists(), reason="v4 open set not present") def test_recording_scorer_reconciles_with_production_on_a_public_case() -> None: from scripts.active_rectification_event_engine import compute_candidate_static_contexts from scripts.rectification.scoring_service import build_event_contribution_matrix, score_from_matrix from scripts.research.minute_resolution_sweep import scoring_request_for case = json.loads(HOLDOUT_V4.read_text(encoding="utf-8"))["cases"][0] request = scoring_request_for({**case, "candidate_radius_minutes": 10}, 10) contexts = compute_candidate_static_contexts(request) production = score_from_matrix(request, build_event_contribution_matrix(request, static_contexts=contexts)) store = lib.FeatureStore() built = build_event_contribution_matrix( request, row_provider=lib.make_recording_provider(contexts, store, lib.d60_charts(contexts)), static_contexts=contexts, ) research = score_from_matrix(request, built) outcome = lib.reconcile_rows(production, research) assert outcome["changed"] == 0, outcome assert outcome["candidates"] == len(contexts) == 11 # extras were recorded alongside the production rules recorded = {name for cells in store.rows.values() for cell in cells.values() for name in cell["extras"]} assert any(name.startswith("kp_") for name in recorded) or any(name.endswith("_in_target_house") for name in recorded)