diff --git a/docs/research/ACTIVE_FRONTS.md b/docs/research/ACTIVE_FRONTS.md index 7ed728ac..245211c3 100644 --- a/docs/research/ACTIVE_FRONTS.md +++ b/docs/research/ACTIVE_FRONTS.md @@ -12,9 +12,11 @@ This file is the small index for the current engineering fronts that still drive ## Relationship Adjudication - `/Users/wuyongnaren/Documents/印度占星/docs/research/marriage_adjudicator_first_pass_audit_2026_06_27.md` +- `/Users/wuyongnaren/Documents/印度占星/docs/research/marriage_benchmark_summary_bridge_audit_2026_06_28.md` - `/Users/wuyongnaren/Documents/印度占星/docs/research/isolated_asset_bridge_audit_2026_06_28.md` - `/Users/wuyongnaren/Documents/印度占星/references/event_judgment_marriage.md` - `/Users/wuyongnaren/Documents/印度占星/docs/superpowers/specs/2026-06-28-jaimini-marriage-bridge-v1-design.md` +- `/Users/wuyongnaren/Documents/印度占星/scripts/marriage_benchmark_summary.py` ## Wealth Adjudication diff --git a/docs/research/marriage_benchmark_summary_bridge_audit_2026_06_28.md b/docs/research/marriage_benchmark_summary_bridge_audit_2026_06_28.md new file mode 100644 index 00000000..c15adb5f --- /dev/null +++ b/docs/research/marriage_benchmark_summary_bridge_audit_2026_06_28.md @@ -0,0 +1,59 @@ +# Marriage Benchmark Summary Bridge Audit - 2026-06-28 + +## Scope + +This pass turns the v6.1 marriage verification dataset from a large historical JSON asset into a small adjudicator-ready summary. + +## Entrypoint + +- `/Users/wuyongnaren/Documents/印度占星/scripts/marriage_benchmark_summary.py` +- Source dataset: `/Users/wuyongnaren/Documents/印度占星/tests/test-data/verify-results-v6.1.json` + +## Why This Was Needed + +The v6.1 benchmark already contains: + +- 18 public cases +- 26 marriage events +- Rao P1-P8 hit data +- UL / Argala / D7 / D60 / Karakamsha context +- divorce markers + +But the useful event statistics were buried inside a large JSON file and a generated report. The new summary helper makes the benchmark reusable by strict adjudicators without requiring every agent to manually parse the full file. + +## Current Summary + +- `case_count = 18` +- `ascendant_match_count = 18` +- `marriage_event_count = 26` +- `divorce_event_count = 15` +- Rao hit distribution: + - `2/8`: 2 events + - `3/8`: 2 events + - `4/8`: 9 events + - `5/8`: 8 events + - `6/8`: 2 events + - `7/8`: 3 events + +## Label-Lift Seed Cases + +The helper exposes the strongest `label_lift_failure_seed` candidates: + +- `Britney Spears|Kevin Federline|2004-10-06` +- `Elon Musk|Justine Wilson|2000-01-01` +- `Tom Cruise|Katie Holmes|2006-11-18` +- `Albert Einstein|Mileva Maric|1903-01-06` +- `Nelson Mandela|Winnie Madikizela|1958-06-14` + +These are high-signal calibration targets for future relationship adjudicator work. + +## Boundary + +- The helper does not recompute astrology. +- It does not alter the source dataset. +- It does not promote `legal_marriage` or any other label by itself. +- It only exposes benchmark evidence in a stable shape. + +## Regression Coverage + +- `/Users/wuyongnaren/Documents/印度占星/tests/test_marriage_benchmark_summary.py` diff --git a/scripts/marriage_benchmark_summary.py b/scripts/marriage_benchmark_summary.py new file mode 100644 index 00000000..84ec0903 --- /dev/null +++ b/scripts/marriage_benchmark_summary.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +"""Summarize the v6.1 marriage timing benchmark into adjudicator-ready evidence.""" + +from __future__ import annotations + +import argparse +import json +from collections import Counter +from pathlib import Path +from typing import Any + + +ROOT = Path(__file__).resolve().parents[1] +DEFAULT_BENCHMARK = ROOT / "tests" / "test-data" / "verify-results-v6.1.json" +RAO_PARAMETERS = [f"P{index}" for index in range(1, 9)] + + +def _resolve(path: str | Path) -> Path: + resolved = Path(path) + if not resolved.is_absolute(): + resolved = ROOT / resolved + return resolved + + +def _load_rows(path: str | Path) -> list[dict[str, Any]]: + loaded = json.loads(_resolve(path).read_text(encoding="utf-8")) + if not isinstance(loaded, list): + raise ValueError("Marriage benchmark must be a JSON list") + return [row for row in loaded if isinstance(row, dict)] + + +def _event_id(case_name: str, marriage: dict[str, Any]) -> str: + return f"{case_name}|{marriage.get('spouse')}|{marriage.get('date')}" + + +def _hit_count(marriage: dict[str, Any]) -> int | None: + summary = ((marriage.get("rao_8_params") or {}).get("summary") or {}) + value = summary.get("hit_count") + return int(value) if isinstance(value, int) else None + + +def summarize_benchmark(path: str | Path = DEFAULT_BENCHMARK) -> dict[str, Any]: + rows = _load_rows(path) + parameter_hits = {key: 0 for key in RAO_PARAMETERS} + hit_distribution: Counter[str] = Counter() + label_lift_seed_cases: list[dict[str, Any]] = [] + event_count = 0 + divorce_count = 0 + + for row in rows: + case_name = row.get("name") or "unknown" + for marriage in row.get("marriages") or []: + if not isinstance(marriage, dict): + continue + event_count += 1 + if marriage.get("divorce"): + divorce_count += 1 + rao = marriage.get("rao_8_params") or {} + for parameter in RAO_PARAMETERS: + if (rao.get(parameter) or {}).get("hit") is True: + parameter_hits[parameter] += 1 + score = _hit_count(marriage) + if score is not None: + hit_distribution[str(score)] += 1 + if score >= 6: + label_lift_seed_cases.append( + { + "event_id": _event_id(case_name, marriage), + "case": case_name, + "spouse": marriage.get("spouse"), + "date": marriage.get("date"), + "rao_hit_count": score, + "rao_hit_rate_pct": round(score / len(RAO_PARAMETERS) * 100, 2), + "use": "label_lift_failure_seed", + } + ) + + parameter_summary = { + parameter: { + "hit_count": hits, + "event_count": event_count, + "hit_rate_pct": round(hits / event_count * 100, 2) if event_count else 0.0, + } + for parameter, hits in parameter_hits.items() + } + return { + "scope": "marriage_timing_benchmark_summary", + "schema_version": 1, + "source_file": str(_resolve(path)), + "case_count": len(rows), + "ascendant_match_count": sum(1 for row in rows if row.get("asc_match") is True), + "marriage_event_count": event_count, + "divorce_event_count": divorce_count, + "rao_hit_distribution": dict(sorted(hit_distribution.items(), key=lambda item: int(item[0]))), + "rao_parameter_hits": parameter_summary, + "label_lift_seed_cases": sorted( + label_lift_seed_cases, + key=lambda item: (-item["rao_hit_count"], item["case"], item["spouse"] or ""), + ), + "boundary": ( + "This summary preserves the v6.1 benchmark as adjudicator evidence. " + "It does not recompute astrology, alter source data, or promote labels by itself." + ), + } + + +def render_markdown(report: dict[str, Any]) -> str: + lines = [ + "# Marriage Timing Benchmark Summary", + "", + f"- source_file: `{report['source_file']}`", + f"- case_count: `{report['case_count']}`", + f"- ascendant_match_count: `{report['ascendant_match_count']}`", + f"- marriage_event_count: `{report['marriage_event_count']}`", + f"- divorce_event_count: `{report['divorce_event_count']}`", + "", + "## Rao Hit Distribution", + "", + "| Rao hits | Event count |", + "| ---: | ---: |", + ] + for score, count in report["rao_hit_distribution"].items(): + lines.append(f"| {score} | {count} |") + lines.extend( + [ + "", + "## Rao Parameter Hits", + "", + "| Parameter | Hits | Hit rate |", + "| --- | ---: | ---: |", + ] + ) + for parameter, row in report["rao_parameter_hits"].items(): + lines.append(f"| {parameter} | {row['hit_count']} | {row['hit_rate_pct']}% |") + lines.extend( + [ + "", + "## Label Lift Seed Cases", + "", + "| Event | Rao hits |", + "| --- | ---: |", + ] + ) + for row in report["label_lift_seed_cases"]: + lines.append(f"| `{row['event_id']}` | {row['rao_hit_count']} |") + lines.extend(["", "## Boundary", "", report["boundary"]]) + return "\n".join(lines) + + +def main() -> int: + parser = argparse.ArgumentParser(description="Summarize the v6.1 marriage timing benchmark") + parser.add_argument("--benchmark-file", default=str(DEFAULT_BENCHMARK)) + parser.add_argument("--format", choices=("json", "markdown"), default="json") + args = parser.parse_args() + + report = summarize_benchmark(args.benchmark_file) + if args.format == "markdown": + print(render_markdown(report)) + else: + print(json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_marriage_benchmark_summary.py b/tests/test_marriage_benchmark_summary.py new file mode 100644 index 00000000..27dedac2 --- /dev/null +++ b/tests/test_marriage_benchmark_summary.py @@ -0,0 +1,41 @@ +#!/usr/bin/env python3 +"""Regression tests for the marriage timing benchmark summary helper.""" + +from __future__ import annotations + +from pathlib import Path + +from scripts.marriage_benchmark_summary import summarize_benchmark + + +ROOT = Path(__file__).resolve().parents[1] +BENCHMARK_FILE = ROOT / "tests" / "test-data" / "verify-results-v6.1.json" + + +def test_marriage_benchmark_summary_preserves_rao_v61_event_statistics() -> None: + report = summarize_benchmark(str(BENCHMARK_FILE)) + + assert report["scope"] == "marriage_timing_benchmark_summary" + assert report["case_count"] == 18 + assert report["ascendant_match_count"] == 18 + assert report["marriage_event_count"] == 26 + assert report["divorce_event_count"] == 15 + assert report["rao_hit_distribution"] == { + "2": 2, + "3": 2, + "4": 9, + "5": 8, + "6": 2, + "7": 3, + } + assert report["rao_parameter_hits"]["P1"]["hit_count"] == 25 + assert report["rao_parameter_hits"]["P6"]["hit_rate_pct"] == 69.23 + + +def test_marriage_benchmark_summary_exposes_label_lift_seed_cases() -> None: + report = summarize_benchmark(str(BENCHMARK_FILE)) + seed_ids = {row["event_id"] for row in report["label_lift_seed_cases"]} + + assert "Britney Spears|Kevin Federline|2004-10-06" in seed_ids + assert "Tom Cruise|Katie Holmes|2006-11-18" in seed_ids + assert all(row["rao_hit_count"] >= 6 for row in report["label_lift_seed_cases"])