Product decision 2026-10-02 (TASK-upstream-sync5 R1): a time range is offered
only with at least 4 dated, primary-scoreable events covering 3 domains,
counted on all of them (training + reserved holdout). Was 3 training events /
2 domains in three TS copies and the Python acceptance gate while the policy
file already said 4/3.
- One definition: references/rectification_policy.v1.json
(minConfirmationEvents / minConfirmationDomains). TS core/types MIN_DATED_*,
rectification-decision MIN_STANDALONE_*, evidence-model MIN_ACCEPTANCE_*,
the convergence evaluator and the post-inference trainingGateOpen all read
it; Python decision_policy MIN_ACCEPTANCE_* alias MIN_CONFIRMATION_*.
- Python receipt counts all scoreable events / domains for event_quality and
domain_diversity; decision policy identity v3 -> v4 (candidate UUIDs carry
it). Candidate scores unchanged (77 v5 cases A/B identical), so the
algorithm stays rectification-v5-matrix-scoring-10.
- Memoization golden v3 written by write_golden; v2 frozen by sha256 with a
test that its scores equal v3 and only the receipt policy moved.
- Collect gap copy names the exact gap ("再来两件……其中至少一件不是……")
instead of always "再来一件"; VOICE.md updated. Legacy life-events form copy
4/3 as well.
- 30 frontend test files, 4 Python tests: fixtures extended to the same
scenario at 4/3, or assertions changed with 原值/新值/原因 notes.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N4f2nya58RoRu4yEmJgRGE
69 lines
3.3 KiB
Python
69 lines
3.3 KiB
Python
from __future__ import annotations
|
|
|
|
import unittest
|
|
from decimal import Decimal
|
|
|
|
from scripts.rectification.decision_policy import (
|
|
POLICY_VERSION,
|
|
RELATIVE_SUPPORT_MODE,
|
|
_relative_support,
|
|
build_candidate_decisions,
|
|
)
|
|
from scripts.rectification.scoring_service import ALGORITHM_VERSION
|
|
|
|
|
|
def _row(time: str, score: float) -> dict:
|
|
return {"time": time, "score": score, "evidence": [], "missing_layers": []}
|
|
|
|
|
|
class RelativeSupportScaleTest(unittest.TestCase):
|
|
def test_holdout_gate_keeps_proportional_default(self) -> None:
|
|
# 原值: policy-v3;新值: policy-v4;原因: R1 区间门槛 4 件 3 域(BUG-1193),相对支持口径不变。
|
|
self.assertEqual(POLICY_VERSION, "rectification-candidate-policy-v4")
|
|
# Explicit date windows: scoring identity -8 -> -9, not a prior/policy change.
|
|
# 原值: -9;新值: -10;原因: 功能吉凶 v2 改变打分(BUG-1181),策略 v3 与相对支持口径不变。
|
|
self.assertEqual(ALGORITHM_VERSION, "rectification-v5-matrix-scoring-10")
|
|
self.assertEqual(RELATIVE_SUPPORT_MODE, "proportional")
|
|
|
|
def test_offset_top_two_lead_is_at_least_proportional(self) -> None:
|
|
scores = [
|
|
Decimal("15.52"), Decimal("15.40"), Decimal("14.10"), Decimal("13.00"),
|
|
Decimal("12.50"), Decimal("12.00"), Decimal("11.80"), Decimal("11.50"),
|
|
Decimal("11.40"), Decimal("11.20"), Decimal("11.10"), Decimal("10.96"),
|
|
]
|
|
floor = min(scores)
|
|
proportional = _relative_support(scores, floor=floor, mode="proportional")
|
|
offset = _relative_support(scores, floor=floor, mode="offset")
|
|
prop_lead = sorted(proportional, reverse=True)
|
|
offset_lead = sorted(offset, reverse=True)
|
|
self.assertGreaterEqual(offset_lead[0] - offset_lead[1], prop_lead[0] - prop_lead[1])
|
|
self.assertEqual(sum(offset), 100)
|
|
self.assertEqual(sum(proportional), 100)
|
|
|
|
def test_equal_scores_split_evenly_for_offset_and_proportional(self) -> None:
|
|
scores = [Decimal("12")] * 12
|
|
expected = _relative_support(scores, floor=Decimal("12"), mode="proportional")
|
|
offset = _relative_support(scores, floor=Decimal("12"), mode="offset")
|
|
self.assertEqual(offset, expected)
|
|
self.assertEqual(sum(offset), 100)
|
|
self.assertTrue(all(value in {8, 9} for value in offset))
|
|
|
|
def test_build_candidate_decisions_offset_lead_beats_proportional_on_spaced_grid(self) -> None:
|
|
scores = [
|
|
15.52, 15.40, 14.10, 13.00, 12.50, 12.00,
|
|
11.80, 11.50, 11.40, 11.20, 11.10, 10.96,
|
|
]
|
|
rows = [_row(f"05:{index * 5:02d}", score) for index, score in enumerate(scores)]
|
|
proportional = build_candidate_decisions(rows, result_id="00000000-0000-4000-8000-000000000001", support_mode="proportional")
|
|
offset = build_candidate_decisions(rows, result_id="00000000-0000-4000-8000-000000000001", support_mode="offset")
|
|
self.assertEqual(len(proportional), 12)
|
|
self.assertEqual(len(offset), 12)
|
|
prop_lead = proportional[0]["relative_support"] - proportional[1]["relative_support"]
|
|
offset_lead = offset[0]["relative_support"] - offset[1]["relative_support"]
|
|
self.assertGreaterEqual(offset_lead, prop_lead)
|
|
self.assertEqual(sum(item["relative_support"] for item in offset), 100)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|