From 3cdd6bfdf0184dc6155bafaa31ace1fe28c5d433 Mon Sep 17 00:00:00 2001 From: 732642856 <732642856@qq.com> Date: Sun, 19 Jul 2026 16:36:41 +0800 Subject: [PATCH] feat: sync skill truth audit safeguards --- SKILL.md | 10 +++ ...s_repo_strategy_diff_audit_2026_07_19.json | 74 +++++++++++++++ ...on_audit_vimsopaka_avastha_2026_07_19.json | 51 +++++++++++ ...level_holdout_pilot_source_queue_report.py | 90 +++++++++++++++++++ ...nique_promotion_audit_vimsopaka_avastha.py | 85 ++++++++++++++++++ tests/test_cross_repo_strategy_diff_audit.py | 30 +++++++ ...st_day_level_holdout_pilot_source_queue.py | 33 +++++++ ...level_holdout_pilot_source_queue_report.py | 40 +++++++++ tests/test_skill_truth_boundary_text.py | 19 ++++ ...nique_promotion_audit_vimsopaka_avastha.py | 42 +++++++++ 10 files changed, 474 insertions(+) create mode 100644 references/cross_project_contract/cross_repo_strategy_diff_audit_2026_07_19.json create mode 100644 references/oracle/technique_promotion_audit_vimsopaka_avastha_2026_07_19.json create mode 100644 scripts/day_level_holdout_pilot_source_queue_report.py create mode 100644 scripts/technique_promotion_audit_vimsopaka_avastha.py create mode 100644 tests/test_cross_repo_strategy_diff_audit.py create mode 100644 tests/test_day_level_holdout_pilot_source_queue.py create mode 100644 tests/test_day_level_holdout_pilot_source_queue_report.py create mode 100644 tests/test_skill_truth_boundary_text.py create mode 100644 tests/test_technique_promotion_audit_vimsopaka_avastha.py diff --git a/SKILL.md b/SKILL.md index aa206710..9908c3d1 100644 --- a/SKILL.md +++ b/SKILL.md @@ -13,8 +13,18 @@ description: "印度占星(Jyotish)专业解盘与推运系统。核心能 > **执行总控**:`references/quick-reference-guide.md` > **严格路由**:`references/strict-workflow-router.md`(涉及事业/婚恋/财务/应期/技法验证时必须优先读取) > **机器注册表**:`references/technique_registry.json` + `scripts/audit_capabilities.py` +> **能力真相边界**:回答前必须参考 `references/oracle/skill_truth_overlay_2026_07_19.json` 与 `references/oracle/effective_skill_capability_view_2026_07_19.json`;不得直接把 `references/technique_registry.json` 的旧 `covered` 当作完整闭环。 > **文章级细节模板**:`references/interpretation_template_registry.json` + `scripts/validate_interpretation_templates.py` +### Skill truth overlay 硬边界 + +KP/Muhurta/Gochara/Sahams/Sphuta/Tajika等高阶分支必须按 skill truth overlay 降级使用;商业回答必须区分 `ready / partial / reference-only / blocked`。 + +| 技法 | 商业回答边界 | +|---|---| +| KP系统 | reference-only / partial | +| Muhurta/Gochara/Sahams/Sphuta/Tajika | 只作参考或探索性证据,除非 evidence packet 明确升级 | + ## v6.9.14 核心能力 | 维度 | 数据 | diff --git a/references/cross_project_contract/cross_repo_strategy_diff_audit_2026_07_19.json b/references/cross_project_contract/cross_repo_strategy_diff_audit_2026_07_19.json new file mode 100644 index 00000000..109f0e44 --- /dev/null +++ b/references/cross_project_contract/cross_repo_strategy_diff_audit_2026_07_19.json @@ -0,0 +1,74 @@ +{ + "created_at": "2026-07-19", + "items": [ + { + "commercial_sha256": "548d177786b76d846e5694e74dc32841bb2618989c1a1ec7dd4d0b7f2b298fbc", + "notes": [ + "commercial_retains_annotation_protocol_pointer" + ], + "path": "references/real_case_calibration/day_level_holdout_v3_preregistration.json", + "research_sha256": "d7f794630d3def6f9f7b5d94c4365bb508c60a8b44228f35a2d2d4548ea9f360", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "f5812c7ce35e67d1760ba0d0ada0157bd28ece2cb9be817e0d47321e592a0116", + "notes": [ + "commercial_retains_observational_validation_mode_for_product_intake" + ], + "path": "scripts/day_level_holdout_validator.py", + "research_sha256": "a37f4fb63af2fdf42591c6a4fe3870c11663f3126309628001ed365f18817b90", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "6e3063eeacfc6730a877a9780b9a82892defe71f74cd4561182ffc7fbfa9fac5", + "notes": [ + "research_has_stricter_self_host_truth_upgrade_gate" + ], + "path": "scripts/vedastro_identity_archive.py", + "research_sha256": "333e490f5a2086dc046819d24285962d8660a6fd7cd022eeca2b586e8887b82e", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "d4a2957ebbe336091544f34da20983d3152dcdca8e11b9b5f5da4eed21c69beb", + "notes": [ + "diff_requires_manual_review_before_copy" + ], + "path": "tests/test_day_level_holdout_validator.py", + "research_sha256": "e130327895f848980da79fa4dc2d5b6802038473c8b457390f1d96aae60715d5", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "6ab6a5b5052334eaad04b2c8f2260e5966ba789e862c69a8ebfd4adcf0d96f40", + "notes": [ + "test_policy_diff_only_no_runtime_effect" + ], + "path": "tests/test_shadbala_av_normative_benchmark_plan.py", + "research_sha256": "73d80cd00b21b48993ba31a2018205ffe6d2c75474062ac19ed156a3e9dce163", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "58024551bd9bb620e2cbfe1c9a28ccdce2d2c5f1588ffae19ca3fbe0082bdd64", + "notes": [ + "commercial_answer_boundary_assertions_may_include_product_context" + ], + "path": "tests/test_skill_truth_overlay_application.py", + "research_sha256": "845aa750f740690c880b79d44960799a56ef379cc2f96c99919a12eb52e16077", + "status": "intentional_diff_review_required" + }, + { + "commercial_sha256": "1300c089410d914936f44abb478dbeb4b7c17ad502b587867df2fce31487aacd", + "notes": [ + "paired_with_vedastro_identity_archive_policy" + ], + "path": "tests/test_vedastro_identity_archive.py", + "research_sha256": "de9c183b39dd2a11b9b8d4f39877e794e2d4e5fb37236629fdfd768ecbe0f6e0", + "status": "intentional_diff_review_required" + } + ], + "scope": "cross_repo_strategy_diff_audit", + "status": "do_not_blind_overwrite", + "summary": { + "diff_count": 7, + "policy": "Commercial may keep product-safe wrappers while importing research truth gates only when compatible." + } +} diff --git a/references/oracle/technique_promotion_audit_vimsopaka_avastha_2026_07_19.json b/references/oracle/technique_promotion_audit_vimsopaka_avastha_2026_07_19.json new file mode 100644 index 00000000..e8d97206 --- /dev/null +++ b/references/oracle/technique_promotion_audit_vimsopaka_avastha_2026_07_19.json @@ -0,0 +1,51 @@ +{ + "created_at": "2026-07-19", + "items": [ + { + "claim_boundary": "Runtime presence is real, but Vimsopaka weight table/source/oracle closure is still separate.", + "current_call_status": "formally_called_in_full_reading", + "entrypoints": [ + "full_reading prompt pack", + "vimsopaka_semantic_summary" + ], + "historical_artifacts": [ + "skills/jyotish-engine-modules/scripts/vimsopaka_calculator.py" + ], + "main_artifacts": [ + "scripts/full_reading.py", + "tests/test_cli_smoke.py" + ], + "next_action": "add_formula_source_and_oracle_packet", + "reuse_decision": "do_not_duplicate_runtime", + "technique_id": "vimsopaka_bala" + }, + { + "claim_boundary": "Avastha is endpoint-visible, but formula variants and interpretive claim level still need source/oracle packet.", + "current_call_status": "formally_called_via_deep_varga_avastha_endpoint", + "entrypoints": [ + "/api/deep_varga_avastha", + "jyotish-app skill-map deepVargaAvastha" + ], + "historical_artifacts": [ + "skills/jyotish-engine-modules/scripts/avastha_calculator.py" + ], + "main_artifacts": [ + "scripts/deep_varga_avastha.py", + "scripts/jyotish_api_server.py", + "jyotish-app/skill-map.js", + "tests/test_deep_varga_avastha.py" + ], + "next_action": "add_display_contract_and_source_oracle_packet", + "reuse_decision": "do_not_duplicate_runtime", + "technique_id": "avastha_states" + } + ], + "production_tuning_allowed": false, + "scope": "technique_promotion_audit_vimsopaka_avastha", + "summary": { + "duplicate_runtime_needed": 0, + "formally_called_count": 2, + "items_checked": 2 + }, + "truth_policy": "runtime_presence_not_oracle_closure" +} diff --git a/scripts/day_level_holdout_pilot_source_queue_report.py b/scripts/day_level_holdout_pilot_source_queue_report.py new file mode 100644 index 00000000..3d059077 --- /dev/null +++ b/scripts/day_level_holdout_pilot_source_queue_report.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python3 +"""Expand pilot day-level holdout source candidates into unscored windows.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +def build_report(queue_path: Path) -> dict: + queue = json.loads(queue_path.read_text(encoding="utf-8")) + windows = [] + for subject in queue["subjects"]: + positive = subject["candidate_positive_event"] + windows.append( + { + "subject_id": subject["subject_id"], + "name": subject["name"], + "domain": subject["domain"], + "label_candidate": positive["label"], + "start": positive["start"], + "end": positive["end"], + "event_description": positive["event_description"], + "event_absent_assertion": "", + "source_urls": positive.get("source_urls", []), + "claim_status": "candidate_not_label", + "required_next_step": "independent_human_adjudication", + "scoring_status": "blocked_not_frozen", + } + ) + for item in subject["candidate_negative_windows"]: + windows.append( + { + "subject_id": subject["subject_id"], + "name": subject["name"], + "domain": subject["domain"], + "label_candidate": "no_target_event", + "start": item["start"], + "end": item["end"], + "event_description": "", + "event_absent_assertion": item["event_absent_assertion"], + "source_urls": positive.get("source_urls", []), + "claim_status": "candidate_not_label", + "required_next_step": "independent_human_adjudication", + "scoring_status": "blocked_not_frozen", + } + ) + + positive_count = sum(1 for row in windows if row["label_candidate"] == "target_event") + negative_count = sum(1 for row in windows if row["label_candidate"] == "no_target_event") + return { + "scope": "day_level_holdout_pilot_source_queue_report", + "source_queue": str(queue_path), + "status": "awaiting_independent_human_labels", + "production_tuning_allowed": False, + "blind_scoring_allowed": False, + "truth_boundary": queue["truth_boundary"], + "summary": { + "subject_count": len(queue["subjects"]), + "window_count": len(windows), + "positive_candidate_count": positive_count, + "negative_candidate_count": negative_count, + "ready_annotation_count": 0, + "blocked_annotation_count": len(windows), + }, + "windows": windows, + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--queue", + type=Path, + default=Path("references/real_case_calibration/day_level_holdout_v3_pilot_source_queue_2026_07_19.json"), + ) + parser.add_argument("--output", type=Path) + args = parser.parse_args() + report = build_report(args.queue) + text = json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(text, encoding="utf-8") + print(text, end="") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/technique_promotion_audit_vimsopaka_avastha.py b/scripts/technique_promotion_audit_vimsopaka_avastha.py new file mode 100644 index 00000000..d67d71f2 --- /dev/null +++ b/scripts/technique_promotion_audit_vimsopaka_avastha.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python3 +"""Audit Vimsopaka/Avastha fragments against current runtime entrypoints.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +def _contains(path: Path, tokens: list[str]) -> bool: + text = path.read_text(encoding="utf-8") + return all(token in text for token in tokens) + + +def build_audit(root: Path) -> dict: + full_reading = root / "scripts/full_reading.py" + api_server = root / "scripts/jyotish_api_server.py" + deep = root / "scripts/deep_varga_avastha.py" + skill_map = root / "jyotish-app/skill-map.js" + tests = root / "tests/test_cli_smoke.py" + deep_tests = root / "tests/test_deep_varga_avastha.py" + + vimsopaka_called = _contains(tests, ["vimsopaka_semantic_summary", "modules", "vimsopaka"]) + avastha_called = _contains(api_server, ["/api/deep_varga_avastha", "_compute_deep_varga_avastha"]) and _contains( + deep, ["AvasthaCalculator", "DivisionalChartsCalculator"] + ) + + items = [ + { + "technique_id": "vimsopaka_bala", + "current_call_status": "formally_called_in_full_reading" if vimsopaka_called else "partial", + "main_artifacts": ["scripts/full_reading.py", "tests/test_cli_smoke.py"], + "historical_artifacts": ["skills/jyotish-engine-modules/scripts/vimsopaka_calculator.py"], + "entrypoints": ["full_reading prompt pack", "vimsopaka_semantic_summary"], + "reuse_decision": "do_not_duplicate_runtime", + "next_action": "add_formula_source_and_oracle_packet", + "claim_boundary": "Runtime presence is real, but Vimsopaka weight table/source/oracle closure is still separate.", + }, + { + "technique_id": "avastha_states", + "current_call_status": "formally_called_via_deep_varga_avastha_endpoint" if avastha_called else "partial", + "main_artifacts": [ + "scripts/deep_varga_avastha.py", + "scripts/jyotish_api_server.py", + "jyotish-app/skill-map.js", + "tests/test_deep_varga_avastha.py", + ], + "historical_artifacts": ["skills/jyotish-engine-modules/scripts/avastha_calculator.py"], + "entrypoints": ["/api/deep_varga_avastha", "jyotish-app skill-map deepVargaAvastha"], + "reuse_decision": "do_not_duplicate_runtime", + "next_action": "add_display_contract_and_source_oracle_packet", + "claim_boundary": "Avastha is endpoint-visible, but formula variants and interpretive claim level still need source/oracle packet.", + }, + ] + return { + "scope": "technique_promotion_audit_vimsopaka_avastha", + "created_at": "2026-07-19", + "truth_policy": "runtime_presence_not_oracle_closure", + "production_tuning_allowed": False, + "summary": { + "items_checked": len(items), + "formally_called_count": sum("formally_called" in item["current_call_status"] for item in items), + "duplicate_runtime_needed": 0, + }, + "items": items, + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--root", type=Path, default=Path(".")) + parser.add_argument("--output", type=Path) + args = parser.parse_args() + audit = build_audit(args.root) + text = json.dumps(audit, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(text, encoding="utf-8") + print(text, end="") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_cross_repo_strategy_diff_audit.py b/tests/test_cross_repo_strategy_diff_audit.py new file mode 100644 index 00000000..13899153 --- /dev/null +++ b/tests/test_cross_repo_strategy_diff_audit.py @@ -0,0 +1,30 @@ +from __future__ import annotations + +import json +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +AUDIT = ROOT / "references" / "cross_project_contract" / "cross_repo_strategy_diff_audit_2026_07_19.json" + + +def test_strategy_diff_audit_blocks_blind_overwrite() -> None: + data = json.loads(AUDIT.read_text(encoding="utf-8")) + + assert data["scope"] == "cross_repo_strategy_diff_audit" + assert data["status"] == "do_not_blind_overwrite" + assert data["summary"]["diff_count"] == 7 + + +def test_strategy_diff_audit_tracks_vedastro_and_holdout_policy_diffs() -> None: + data = json.loads(AUDIT.read_text(encoding="utf-8")) + items = {item["path"]: item for item in data["items"]} + + assert "scripts/day_level_holdout_validator.py" in items + assert "commercial_retains_observational_validation_mode_for_product_intake" in items[ + "scripts/day_level_holdout_validator.py" + ]["notes"] + assert "scripts/vedastro_identity_archive.py" in items + assert "research_has_stricter_self_host_truth_upgrade_gate" in items[ + "scripts/vedastro_identity_archive.py" + ]["notes"] diff --git a/tests/test_day_level_holdout_pilot_source_queue.py b/tests/test_day_level_holdout_pilot_source_queue.py new file mode 100644 index 00000000..6f1d930f --- /dev/null +++ b/tests/test_day_level_holdout_pilot_source_queue.py @@ -0,0 +1,33 @@ +from __future__ import annotations + +import json +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +QUEUE = ROOT / "references" / "real_case_calibration" / "day_level_holdout_v3_pilot_source_queue_2026_07_19.json" + + +def test_pilot_source_queue_has_three_public_subjects_but_no_truth_upgrade() -> None: + data = json.loads(QUEUE.read_text(encoding="utf-8")) + + assert data["status"] == "candidate_requires_adjudication" + assert data["production_tuning_allowed"] is False + assert data["candidate_count"] == 3 + assert "not holdout annotations" in data["truth_boundary"] + assert "independent adjudicator" in data["required_next_step"] + + subjects = {subject["subject_id"]: subject for subject in data["subjects"]} + assert set(subjects) == {"steve_jobs", "barack_obama", "albert_einstein"} + for subject in subjects.values(): + assert subject["candidate_positive_event"]["status"] == "candidate_requires_adjudication" + assert len(subject["candidate_negative_windows"]) == 2 + assert all(window["status"] == "candidate_requires_adjudication" for window in subject["candidate_negative_windows"]) + assert all(url.startswith("https://") for url in subject["candidate_positive_event"]["source_urls"]) + + +def test_pilot_source_queue_is_not_misrepresented_as_day_level_holdout_manifest() -> None: + data = json.loads(QUEUE.read_text(encoding="utf-8")) + + assert "annotations" not in data + assert data["scope"] == "day_level_holdout_v3_pilot_source_queue" + assert data["status"] != "ready_for_blind_replay" diff --git a/tests/test_day_level_holdout_pilot_source_queue_report.py b/tests/test_day_level_holdout_pilot_source_queue_report.py new file mode 100644 index 00000000..85966b90 --- /dev/null +++ b/tests/test_day_level_holdout_pilot_source_queue_report.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +from pathlib import Path + +from scripts.day_level_holdout_pilot_source_queue_report import build_report + +ROOT = Path(__file__).resolve().parents[1] +QUEUE = ROOT / "references/real_case_calibration/day_level_holdout_v3_pilot_source_queue_2026_07_19.json" + + +def test_pilot_source_queue_expands_to_nine_unscored_windows() -> None: + report = build_report(QUEUE) + + assert report["scope"] == "day_level_holdout_pilot_source_queue_report" + assert report["status"] == "awaiting_independent_human_labels" + assert report["production_tuning_allowed"] is False + assert report["blind_scoring_allowed"] is False + assert report["summary"] == { + "subject_count": 3, + "window_count": 9, + "positive_candidate_count": 3, + "negative_candidate_count": 6, + "ready_annotation_count": 0, + "blocked_annotation_count": 9, + } + + +def test_pilot_source_queue_windows_keep_public_source_and_boundary() -> None: + report = build_report(QUEUE) + + for window in report["windows"]: + assert window["claim_status"] == "candidate_not_label" + assert window["required_next_step"] == "independent_human_adjudication" + assert window["scoring_status"] == "blocked_not_frozen" + if window["label_candidate"] == "target_event": + assert window["source_urls"] + assert all(url.startswith("https://") for url in window["source_urls"]) + else: + assert window["event_absent_assertion"] + diff --git a/tests/test_skill_truth_boundary_text.py b/tests/test_skill_truth_boundary_text.py new file mode 100644 index 00000000..9c845186 --- /dev/null +++ b/tests/test_skill_truth_boundary_text.py @@ -0,0 +1,19 @@ +from __future__ import annotations + +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +SKILL = ROOT / "SKILL.md" + + +def test_skill_text_requires_effective_capability_view_before_claims() -> None: + text = SKILL.read_text(encoding="utf-8") + assert "references/oracle/effective_skill_capability_view_2026_07_19.json" in text + assert "references/oracle/skill_truth_overlay_2026_07_19.json" in text + assert "不得直接把 `references/technique_registry.json` 的旧 `covered` 当作完整闭环" in text + + +def test_skill_text_does_not_overclaim_kp_muhurta_sahams() -> None: + text = SKILL.read_text(encoding="utf-8") + assert "KP/Muhurta/Gochara/Sahams/Sphuta/Tajika等高阶分支必须按 skill truth overlay 降级使用" in text + assert "KP系统 | reference-only / partial" in text diff --git a/tests/test_technique_promotion_audit_vimsopaka_avastha.py b/tests/test_technique_promotion_audit_vimsopaka_avastha.py new file mode 100644 index 00000000..61541d5d --- /dev/null +++ b/tests/test_technique_promotion_audit_vimsopaka_avastha.py @@ -0,0 +1,42 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from scripts.technique_promotion_audit_vimsopaka_avastha import build_audit + +ROOT = Path(__file__).resolve().parents[1] + + +def test_vimsopaka_avastha_audit_corrects_call_status() -> None: + audit = build_audit(ROOT) + assert audit["scope"] == "technique_promotion_audit_vimsopaka_avastha" + assert audit["truth_policy"] == "runtime_presence_not_oracle_closure" + assert audit["production_tuning_allowed"] is False + statuses = {item["technique_id"]: item["current_call_status"] for item in audit["items"]} + assert statuses["vimsopaka_bala"] == "formally_called_in_full_reading" + assert statuses["avastha_states"] == "formally_called_via_deep_varga_avastha_endpoint" + + +def test_vimsopaka_avastha_audit_records_entrypoints_and_boundaries() -> None: + audit = build_audit(ROOT) + for item in audit["items"]: + assert item["main_artifacts"] + assert item["entrypoints"] + assert item["next_action"] in { + "add_formula_source_and_oracle_packet", + "add_display_contract_and_source_oracle_packet", + } + assert item["claim_boundary"] + assert item["reuse_decision"] == "do_not_duplicate_runtime" + + +def test_vimsopaka_avastha_audit_artifact_exists() -> None: + artifact = ROOT / "references/oracle/technique_promotion_audit_vimsopaka_avastha_2026_07_19.json" + data = json.loads(artifact.read_text(encoding="utf-8")) + assert data["summary"] == { + "items_checked": 2, + "formally_called_count": 2, + "duplicate_runtime_needed": 0, + } +