Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
160 lines
8.2 KiB
Python
160 lines
8.2 KiB
Python
"""Regression tests for the 2026-09-29 scoring-method research scaffold.
|
|
|
|
Pure helpers use fictional inputs. The one integration test reconciles the
|
|
research recording scorer against the production scorer on a public v4 case
|
|
(task hard line 4: zero difference before any experiment).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
import scripts.research.scoring_research_lib as lib
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
HOLDOUT_V4 = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
|
|
|
|
|
|
def test_estimate_lr_rewards_discriminating_feature_and_leaves_noise_near_zero() -> None:
|
|
examples = []
|
|
for _ in range(20):
|
|
examples.append(({"good": 1.0, "noise": 1.0}, True))
|
|
examples.append(({"good": 1.0, "noise": 0.0}, True))
|
|
for _ in range(80):
|
|
examples.append(({"noise": 1.0}, False))
|
|
examples.append(({"noise": 0.0}, False))
|
|
table = lib.estimate_lr(examples, alpha=1.0)
|
|
assert table.positives == 40 and table.negatives == 160
|
|
assert table.present["good"] > 1.5
|
|
assert table.absent["good"] < -1.5
|
|
assert abs(table.present["noise"]) < 0.05
|
|
# reward-only policy clips the negative absent term; naive-Bayes keeps it
|
|
assert table.weight("good", allow_negative=False) == (table.present["good"], 0.0)
|
|
assert table.weight("good", allow_negative=True) == (table.present["good"], table.absent["good"])
|
|
|
|
|
|
def test_loo_folds_never_contain_the_held_out_case() -> None:
|
|
folds = lib.loo_folds(["a", "b", "c"])
|
|
assert [held for held, _ in folds] == ["a", "b", "c"]
|
|
for held, training in folds:
|
|
assert held not in training and len(training) == 2
|
|
|
|
|
|
def test_event_feature_matrix_is_the_fraction_of_sample_dates() -> None:
|
|
store = lib.FeatureStore()
|
|
store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house", "event_kind:career"], ["pranapada_in_target_house"])
|
|
store.put("e1", "1990-02-15", "10:00", ["vim_ad_domain_lord"], [])
|
|
store.put("e1", "1990-01-15", "10:02", ["no_domain_activation"], [])
|
|
store.put("e1", "1990-02-15", "10:02", ["no_domain_activation"], [])
|
|
matrix = lib.event_feature_matrix(store, "e1", ["10:00", "10:02"], include_extras=True)
|
|
assert matrix["10:00"] == {"vim_md_domain_house": 0.5, "vim_ad_domain_lord": 0.5, "pranapada_in_target_house": 0.5}
|
|
assert matrix["10:02"] == {}
|
|
without = lib.event_feature_matrix(store, "e1", ["10:00"], include_extras=False)
|
|
assert "pranapada_in_target_house" not in without["10:00"]
|
|
|
|
|
|
def test_scoring_rule_filter_drops_event_kind_and_constant_rules() -> None:
|
|
assert lib.is_scoring_rule("vim_md_domain_varga")
|
|
assert lib.is_scoring_rule("controlled_transit_jupiter_domain_house")
|
|
assert not lib.is_scoring_rule("event_kind:career")
|
|
assert not lib.is_scoring_rule("event_kind_profile:career:change")
|
|
assert not lib.is_scoring_rule("no_domain_activation")
|
|
assert not lib.is_scoring_rule("occupation_auxiliary_not_primary")
|
|
assert not lib.is_scoring_rule("kp_asc_sublord_is_vim_md") # extras are a separate namespace
|
|
assert lib.is_extra_feature("vim_md_domain_varga_d60")
|
|
|
|
|
|
def test_weighted_provider_scores_each_sample_from_the_store() -> None:
|
|
store = lib.FeatureStore()
|
|
store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house"], ["kp_asc_sublord_is_vim_md"])
|
|
store.put("e1", "1990-01-15", "10:02", [], [])
|
|
table = lib.LRTable(
|
|
alpha=1.0, positives=1, negatives=1,
|
|
present={"vim_md_domain_house": 1.0, "kp_asc_sublord_is_vim_md": 0.5},
|
|
absent={"vim_md_domain_house": -0.25, "kp_asc_sublord_is_vim_md": -0.1},
|
|
support={},
|
|
)
|
|
request = {"events": [{"id": "e1", "date": "1990-01-15", "domain": "career"}]}
|
|
a1 = lib.make_weighted_provider(store, table, variant="A1")(request)
|
|
a2 = lib.make_weighted_provider(store, table, variant="A2")(request)
|
|
a3 = lib.make_weighted_provider(store, table, variant="A3", scale=2.0)(request)
|
|
assert [(r["time"], r["score"]) for r in a1] == [("10:00", 1.0), ("10:02", 0.0)]
|
|
assert [(r["time"], r["score"]) for r in a2] == [("10:00", 1.5), ("10:02", 0.0)]
|
|
assert [(r["time"], r["score"]) for r in a3] == [("10:00", 3.0), ("10:02", -0.7)]
|
|
assert a1[0]["evidence"][0]["rule_ids"] == ["vim_md_domain_house"]
|
|
|
|
|
|
def _probe() -> dict:
|
|
return {"expected_outcomes": [
|
|
{"answer_class": "yes", "supports": ["10:00"], "conflicts": ["10:04"]},
|
|
{"answer_class": "no", "supports": ["10:04"], "conflicts": ["10:00"]},
|
|
]}
|
|
|
|
|
|
def test_scaled_answer_halves_the_delta_and_never_counts_a_conflict() -> None:
|
|
scores = {"10:00": 10.0, "10:02": 10.0, "10:04": 10.0}
|
|
conflicts = {time: 0 for time in scores}
|
|
weak, weak_conflicts, weak_elim = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 0.5)
|
|
assert weak == {"10:00": 9.0, "10:02": 10.0, "10:04": 11.0}
|
|
assert weak_conflicts == conflicts and weak_elim == set()
|
|
full, full_conflicts, _ = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 1.0)
|
|
assert full == {"10:00": 8.0, "10:02": 10.0, "10:04": 12.0}
|
|
assert full_conflicts["10:00"] == 1
|
|
|
|
|
|
def test_absent_years_are_gaps_strictly_inside_each_domain_span() -> None:
|
|
events = [
|
|
{"domain": "career", "date": "2017-05-01", "precision": "day"},
|
|
{"domain": "career", "date": "2019", "precision": "year"},
|
|
{"domain": "career", "date": "2025-02", "precision": "month"},
|
|
{"domain": "relocation", "date": "2000", "precision": "year"},
|
|
]
|
|
assert lib.absent_years(events) == {"career": [2018, 2020, 2021, 2022, 2023, 2024]}
|
|
|
|
|
|
def test_degrade_and_upgrade_precision_helpers() -> None:
|
|
events = [{"id": "a", "domain": "career", "date": "2017-05-20", "precision": "day"},
|
|
{"id": "b", "domain": "career", "date": "2019", "precision": "year"}]
|
|
spoken = lib.degrade_to_year(events)
|
|
assert spoken[0]["precision"] == "year" and spoken[0]["date"] == "2017"
|
|
assert spoken[1] == events[1]
|
|
upgraded = lib.upgrade_to_month(spoken, {"a"}, {e["id"]: e for e in events})
|
|
assert upgraded[0]["precision"] == "month" and upgraded[0]["date"] == "2017-05"
|
|
shifted = lib.shift_day_events_by_months(events, seed="fictional")
|
|
assert shifted[0]["shift_months"] in {-3, -2, -1, 1, 2, 3}
|
|
assert shifted[1] == events[1]
|
|
assert lib.drop_events(events, 1, seed="x") != events and len(lib.drop_events(events, 1, seed="x")) == 1
|
|
|
|
|
|
def test_softmax_and_proportional_percent_sum_to_about_one_hundred() -> None:
|
|
scores = {"10:00": 12.0, "10:02": 11.0, "10:04": 9.0}
|
|
assert abs(sum(lib.percent_proportional(scores).values()) - 100) <= 2
|
|
soft = lib.percent_softmax(scores)
|
|
assert abs(sum(soft.values()) - 100) <= 2 and soft["10:00"] > soft["10:02"] > soft["10:04"]
|
|
assert lib.calibration_bins([(0.9, True), (0.02, False)])[-1]["observed_rate"] == 1.0
|
|
|
|
|
|
@pytest.mark.skipif(not HOLDOUT_V4.exists(), reason="v4 open set not present")
|
|
def test_recording_scorer_reconciles_with_production_on_a_public_case() -> None:
|
|
from scripts.active_rectification_event_engine import compute_candidate_static_contexts
|
|
from scripts.rectification.scoring_service import build_event_contribution_matrix, score_from_matrix
|
|
from scripts.research.minute_resolution_sweep import scoring_request_for
|
|
|
|
case = json.loads(HOLDOUT_V4.read_text(encoding="utf-8"))["cases"][0]
|
|
request = scoring_request_for({**case, "candidate_radius_minutes": 10}, 10)
|
|
contexts = compute_candidate_static_contexts(request)
|
|
production = score_from_matrix(request, build_event_contribution_matrix(request, static_contexts=contexts))
|
|
store = lib.FeatureStore()
|
|
built = build_event_contribution_matrix(
|
|
request, row_provider=lib.make_recording_provider(contexts, store, lib.d60_charts(contexts)), static_contexts=contexts,
|
|
)
|
|
research = score_from_matrix(request, built)
|
|
outcome = lib.reconcile_rows(production, research)
|
|
assert outcome["changed"] == 0, outcome
|
|
assert outcome["candidates"] == len(contexts) == 11
|
|
# extras were recorded alongside the production rules
|
|
recorded = {name for cells in store.rows.values() for cell in cells.values() for name in cell["extras"]}
|
|
assert any(name.startswith("kp_") for name in recorded) or any(name.endswith("_in_target_house") for name in recorded)
|