Files
Jyotisha/tests/test_scoring_research.py
T

160 lines
8.2 KiB
Python

"""Regression tests for the 2026-09-29 scoring-method research scaffold.
Pure helpers use fictional inputs. The one integration test reconciles the
research recording scorer against the production scorer on a public v4 case
(task hard line 4: zero difference before any experiment).
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
import scripts.research.scoring_research_lib as lib
ROOT = Path(__file__).resolve().parents[1]
HOLDOUT_V4 = ROOT / "references" / "real_case_calibration" / "minute_rectification_holdout_v4.json"
def test_estimate_lr_rewards_discriminating_feature_and_leaves_noise_near_zero() -> None:
examples = []
for _ in range(20):
examples.append(({"good": 1.0, "noise": 1.0}, True))
examples.append(({"good": 1.0, "noise": 0.0}, True))
for _ in range(80):
examples.append(({"noise": 1.0}, False))
examples.append(({"noise": 0.0}, False))
table = lib.estimate_lr(examples, alpha=1.0)
assert table.positives == 40 and table.negatives == 160
assert table.present["good"] > 1.5
assert table.absent["good"] < -1.5
assert abs(table.present["noise"]) < 0.05
# reward-only policy clips the negative absent term; naive-Bayes keeps it
assert table.weight("good", allow_negative=False) == (table.present["good"], 0.0)
assert table.weight("good", allow_negative=True) == (table.present["good"], table.absent["good"])
def test_loo_folds_never_contain_the_held_out_case() -> None:
folds = lib.loo_folds(["a", "b", "c"])
assert [held for held, _ in folds] == ["a", "b", "c"]
for held, training in folds:
assert held not in training and len(training) == 2
def test_event_feature_matrix_is_the_fraction_of_sample_dates() -> None:
store = lib.FeatureStore()
store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house", "event_kind:career"], ["pranapada_in_target_house"])
store.put("e1", "1990-02-15", "10:00", ["vim_ad_domain_lord"], [])
store.put("e1", "1990-01-15", "10:02", ["no_domain_activation"], [])
store.put("e1", "1990-02-15", "10:02", ["no_domain_activation"], [])
matrix = lib.event_feature_matrix(store, "e1", ["10:00", "10:02"], include_extras=True)
assert matrix["10:00"] == {"vim_md_domain_house": 0.5, "vim_ad_domain_lord": 0.5, "pranapada_in_target_house": 0.5}
assert matrix["10:02"] == {}
without = lib.event_feature_matrix(store, "e1", ["10:00"], include_extras=False)
assert "pranapada_in_target_house" not in without["10:00"]
def test_scoring_rule_filter_drops_event_kind_and_constant_rules() -> None:
assert lib.is_scoring_rule("vim_md_domain_varga")
assert lib.is_scoring_rule("controlled_transit_jupiter_domain_house")
assert not lib.is_scoring_rule("event_kind:career")
assert not lib.is_scoring_rule("event_kind_profile:career:change")
assert not lib.is_scoring_rule("no_domain_activation")
assert not lib.is_scoring_rule("occupation_auxiliary_not_primary")
assert not lib.is_scoring_rule("kp_asc_sublord_is_vim_md") # extras are a separate namespace
assert lib.is_extra_feature("vim_md_domain_varga_d60")
def test_weighted_provider_scores_each_sample_from_the_store() -> None:
store = lib.FeatureStore()
store.put("e1", "1990-01-15", "10:00", ["vim_md_domain_house"], ["kp_asc_sublord_is_vim_md"])
store.put("e1", "1990-01-15", "10:02", [], [])
table = lib.LRTable(
alpha=1.0, positives=1, negatives=1,
present={"vim_md_domain_house": 1.0, "kp_asc_sublord_is_vim_md": 0.5},
absent={"vim_md_domain_house": -0.25, "kp_asc_sublord_is_vim_md": -0.1},
support={},
)
request = {"events": [{"id": "e1", "date": "1990-01-15", "domain": "career"}]}
a1 = lib.make_weighted_provider(store, table, variant="A1")(request)
a2 = lib.make_weighted_provider(store, table, variant="A2")(request)
a3 = lib.make_weighted_provider(store, table, variant="A3", scale=2.0)(request)
assert [(r["time"], r["score"]) for r in a1] == [("10:00", 1.0), ("10:02", 0.0)]
assert [(r["time"], r["score"]) for r in a2] == [("10:00", 1.5), ("10:02", 0.0)]
assert [(r["time"], r["score"]) for r in a3] == [("10:00", 3.0), ("10:02", -0.7)]
assert a1[0]["evidence"][0]["rule_ids"] == ["vim_md_domain_house"]
def _probe() -> dict:
return {"expected_outcomes": [
{"answer_class": "yes", "supports": ["10:00"], "conflicts": ["10:04"]},
{"answer_class": "no", "supports": ["10:04"], "conflicts": ["10:00"]},
]}
def test_scaled_answer_halves_the_delta_and_never_counts_a_conflict() -> None:
scores = {"10:00": 10.0, "10:02": 10.0, "10:04": 10.0}
conflicts = {time: 0 for time in scores}
weak, weak_conflicts, weak_elim = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 0.5)
assert weak == {"10:00": 9.0, "10:02": 10.0, "10:04": 11.0}
assert weak_conflicts == conflicts and weak_elim == set()
full, full_conflicts, _ = lib.apply_scaled_answer(scores, conflicts, set(), _probe(), "no", list(scores), 1.0)
assert full == {"10:00": 8.0, "10:02": 10.0, "10:04": 12.0}
assert full_conflicts["10:00"] == 1
def test_absent_years_are_gaps_strictly_inside_each_domain_span() -> None:
events = [
{"domain": "career", "date": "2017-05-01", "precision": "day"},
{"domain": "career", "date": "2019", "precision": "year"},
{"domain": "career", "date": "2025-02", "precision": "month"},
{"domain": "relocation", "date": "2000", "precision": "year"},
]
assert lib.absent_years(events) == {"career": [2018, 2020, 2021, 2022, 2023, 2024]}
def test_degrade_and_upgrade_precision_helpers() -> None:
events = [{"id": "a", "domain": "career", "date": "2017-05-20", "precision": "day"},
{"id": "b", "domain": "career", "date": "2019", "precision": "year"}]
spoken = lib.degrade_to_year(events)
assert spoken[0]["precision"] == "year" and spoken[0]["date"] == "2017"
assert spoken[1] == events[1]
upgraded = lib.upgrade_to_month(spoken, {"a"}, {e["id"]: e for e in events})
assert upgraded[0]["precision"] == "month" and upgraded[0]["date"] == "2017-05"
shifted = lib.shift_day_events_by_months(events, seed="fictional")
assert shifted[0]["shift_months"] in {-3, -2, -1, 1, 2, 3}
assert shifted[1] == events[1]
assert lib.drop_events(events, 1, seed="x") != events and len(lib.drop_events(events, 1, seed="x")) == 1
def test_softmax_and_proportional_percent_sum_to_about_one_hundred() -> None:
scores = {"10:00": 12.0, "10:02": 11.0, "10:04": 9.0}
assert abs(sum(lib.percent_proportional(scores).values()) - 100) <= 2
soft = lib.percent_softmax(scores)
assert abs(sum(soft.values()) - 100) <= 2 and soft["10:00"] > soft["10:02"] > soft["10:04"]
assert lib.calibration_bins([(0.9, True), (0.02, False)])[-1]["observed_rate"] == 1.0
@pytest.mark.skipif(not HOLDOUT_V4.exists(), reason="v4 open set not present")
def test_recording_scorer_reconciles_with_production_on_a_public_case() -> None:
from scripts.active_rectification_event_engine import compute_candidate_static_contexts
from scripts.rectification.scoring_service import build_event_contribution_matrix, score_from_matrix
from scripts.research.minute_resolution_sweep import scoring_request_for
case = json.loads(HOLDOUT_V4.read_text(encoding="utf-8"))["cases"][0]
request = scoring_request_for({**case, "candidate_radius_minutes": 10}, 10)
contexts = compute_candidate_static_contexts(request)
production = score_from_matrix(request, build_event_contribution_matrix(request, static_contexts=contexts))
store = lib.FeatureStore()
built = build_event_contribution_matrix(
request, row_provider=lib.make_recording_provider(contexts, store, lib.d60_charts(contexts)), static_contexts=contexts,
)
research = score_from_matrix(request, built)
outcome = lib.reconcile_rows(production, research)
assert outcome["changed"] == 0, outcome
assert outcome["candidates"] == len(contexts) == 11
# extras were recorded alongside the production rules
recorded = {name for cells in store.rows.values() for cell in cells.values() for name in cell["extras"]}
assert any(name.startswith("kp_") for name in recorded) or any(name.endswith("_in_target_house") for name in recorded)