Wording only: _quality_user_meaning names start/change/interruption as 升学/学业变动/学业中断; _display_date_label drops the stored month for year-precision events. Split hash and month field unchanged; same-machine A/B (PYTHONHASHSEED=0) differs only in user_meaning/display_date_label. Re-frozen per ERR-110 under new quality_wording_2026_09_29 records; sealed rerun (20) and reported-offset sweep (900) identical to 09-21. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017eEAG8HD3mm8gsKXgk8uU8
115 lines
5.0 KiB
Python
115 lines
5.0 KiB
Python
#!/usr/bin/env python3
|
|
"""A/B dump for BUG-1088 (quality-probe wording, year-precision label).
|
|
|
|
Dumps engine rows, public candidate decisions and every discriminating probe
|
|
for a few v4 open-holdout cases plus fictional cases that carry education
|
|
quality events. Run on the base and on the patched tree with the same
|
|
PYTHONHASHSEED and diff: only `user_meaning` / `display_date_label` text may
|
|
change.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from uuid import NAMESPACE_URL, uuid5
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
if str(ROOT) not in sys.path:
|
|
sys.path.insert(0, str(ROOT))
|
|
|
|
from scripts.active_rectification_event_engine import compute_candidate_static_contexts # noqa: E402
|
|
from scripts.rectification.decision_policy import build_candidate_decisions # noqa: E402
|
|
from scripts.rectification.event_probes import _discriminating_event_probe_lists # noqa: E402
|
|
from scripts.rectification.refinement_packet import window_scan # noqa: E402
|
|
from scripts.rectification.scoring_service import ( # noqa: E402
|
|
build_event_contribution_matrix,
|
|
score_from_matrix,
|
|
scoreable_request,
|
|
)
|
|
from scripts.research.guided_collect_holdout_replay import TODAY, load_cases # noqa: E402
|
|
from scripts.research.minute_resolution_sweep import scoring_request_for # noqa: E402
|
|
from scripts.rectification.contracts import normalize_rectification_request # noqa: E402
|
|
|
|
#: Fictional births only (no real person). Education events carry quality kinds.
|
|
FICTIONAL = [
|
|
("1994-06-21", "14:50", [
|
|
("education", "education_start", "2011-01-01", "2011-12-31", "year"),
|
|
("education", "education_change", "2012-09-01", "2012-09-30", "month"),
|
|
("relocation", "relocation", "2000-01-01", "2000-12-31", "year"),
|
|
("career", "career_entry", "2017-05-01", "2017-05-31", "month"),
|
|
]),
|
|
("1993-11-17", "09:20", [
|
|
("education", "education_interruption", "2010-01-01", "2010-12-31", "year"),
|
|
("education", "education_start", "2011-09-03", "2011-09-03", "day"),
|
|
("career", "career_entry", "2016-07-01", "2016-07-31", "month"),
|
|
("relocation", "relocation", "2018-01-01", "2018-12-31", "year"),
|
|
]),
|
|
]
|
|
|
|
|
|
def fictional_request(birth_date: str, time: str, events: list[tuple[str, str, str, str, str]]) -> dict[str, Any]:
|
|
hh, mm = int(time[:2]), int(time[3:])
|
|
start = f"{(hh * 60 + mm - 15) // 60:02d}:{(hh * 60 + mm - 15) % 60:02d}"
|
|
end = f"{(hh * 60 + mm + 15) // 60:02d}:{(hh * 60 + mm + 15) % 60:02d}"
|
|
return scoreable_request(normalize_rectification_request({
|
|
"birth_date": birth_date, "start_time": start, "end_time": end,
|
|
"lat": 39.9042, "lon": 116.4074, "tz": 8.0,
|
|
"events": [
|
|
{"id": str(uuid5(NAMESPACE_URL, f"fictional-{birth_date}-{index}")), "domain": domain, "event_kind": kind, "date_start": lo,
|
|
"date_end": hi, "precision": precision, "summary": f"fictional {kind}"}
|
|
for index, (domain, kind, lo, hi, precision) in enumerate(events)
|
|
],
|
|
}, today=TODAY))
|
|
|
|
|
|
def dump(request: dict[str, Any]) -> dict[str, Any]:
|
|
import scripts.rectification.event_probes as event_probes
|
|
|
|
request = {**request, "minute_step": 1}
|
|
captured: list[dict[str, Any]] = []
|
|
original = event_probes._quality_distinguish_probes
|
|
|
|
def capture(*args: Any, **kwargs: Any) -> list[dict[str, Any]]:
|
|
rows = original(*args, **kwargs)
|
|
captured.extend(rows)
|
|
return rows
|
|
|
|
# Quality probes are ranked after the boundary probes and often fall past MAX_PROBES;
|
|
# capture the builder output so the A/B sees them. Restored below.
|
|
event_probes._quality_distinguish_probes = capture
|
|
contexts = compute_candidate_static_contexts(request)
|
|
built = build_event_contribution_matrix(request, static_contexts=contexts)
|
|
rows = score_from_matrix(request, built)
|
|
times = [str(row["time"])[:5] for row in rows]
|
|
probes, dropped = _discriminating_event_probe_lists(
|
|
{**request, "refresh_probes": False, "asked_probe_keys": []}, built,
|
|
scan=window_scan(built), candidate_times=times, representative_time=times[len(times) // 2],
|
|
today=TODAY,
|
|
)
|
|
event_probes._quality_distinguish_probes = original
|
|
assert event_probes._quality_distinguish_probes is original
|
|
decisions = build_candidate_decisions(rows, result_id="ab", static_contexts=contexts)
|
|
return {
|
|
"rows": [{"time": str(row["time"])[:5], "score": str(row.get("score"))} for row in rows],
|
|
"decisions": decisions,
|
|
"probes": probes,
|
|
"dropped": dropped,
|
|
"quality_built": captured,
|
|
}
|
|
|
|
|
|
def main() -> int:
|
|
out: dict[str, Any] = {}
|
|
for case in load_cases()[:4]:
|
|
out[str(case["case_id"])] = dump(scoring_request_for(case, 10))
|
|
for birth_date, time, events in FICTIONAL:
|
|
out[f"fictional-{birth_date}"] = dump(fictional_request(birth_date, time, events))
|
|
json.dump(out, sys.stdout, ensure_ascii=False, indent=1, sort_keys=True, default=str)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|