diff --git a/references/real_case_calibration/real_case_website_e2e_eval_2026_07_20.json b/references/real_case_calibration/real_case_website_e2e_eval_2026_07_20.json index 2f3ebc72..f0ad85a0 100644 --- a/references/real_case_calibration/real_case_website_e2e_eval_2026_07_20.json +++ b/references/real_case_calibration/real_case_website_e2e_eval_2026_07_20.json @@ -12,9 +12,12 @@ { "birth": { "date": "1955-02-24", + "lat": 37.7749, + "lon": -122.4194, "place": "San Francisco, CA, USA", "source_policy": "public_record_candidate", - "time": "19:15" + "time": "19:15", + "tz": -8 }, "case_id": "steve_jobs", "domains": [ @@ -44,9 +47,12 @@ { "birth": { "date": "1879-03-14", + "lat": 48.4011, + "lon": 9.9876, "place": "Ulm, Germany", "source_policy": "public_record_candidate", - "time": "11:30" + "time": "11:30", + "tz": 1 }, "case_id": "albert_einstein", "domains": [ @@ -76,9 +82,12 @@ { "birth": { "date": "1961-08-04", + "lat": 21.3069, + "lon": -157.8583, "place": "Honolulu, HI, USA", "source_policy": "public_record_candidate", - "time": "19:24" + "time": "19:24", + "tz": -10 }, "case_id": "barack_obama", "domains": [ @@ -108,9 +117,12 @@ { "birth": { "date": "1961-07-01", + "lat": 52.8294, + "lon": 0.5143, "place": "Sandringham, England", "source_policy": "public_record_candidate", - "time": "19:45" + "time": "19:45", + "tz": 0 }, "case_id": "princess_diana", "domains": [ @@ -139,9 +151,12 @@ { "birth": { "date": "1946-06-14", + "lat": 40.7282, + "lon": -73.7949, "place": "Queens, NY, USA", "source_policy": "public_record_candidate", - "time": "10:54" + "time": "10:54", + "tz": -5 }, "case_id": "donald_trump", "domains": [ @@ -171,9 +186,12 @@ { "birth": { "date": "1954-01-29", + "lat": 33.0576, + "lon": -89.5887, "place": "Kosciusko, MS, USA", "source_policy": "public_record_candidate", - "time": "04:30" + "time": "04:30", + "tz": -6 }, "case_id": "oprah_winfrey", "domains": [ @@ -203,9 +221,12 @@ { "birth": { "date": "1971-06-28", + "lat": -25.7479, + "lon": 28.2293, "place": "Pretoria, South Africa", "source_policy": "public_record_candidate", - "time": "07:30" + "time": "07:30", + "tz": 2 }, "case_id": "elon_musk", "domains": [ @@ -235,9 +256,12 @@ { "birth": { "date": "1869-10-02", + "lat": 21.6417, + "lon": 69.6293, "place": "Porbandar, India", "source_policy": "public_record_candidate", - "time": "07:11" + "time": "07:11", + "tz": 5.5 }, "case_id": "mahatma_gandhi", "domains": [ @@ -267,9 +291,12 @@ { "birth": { "date": "1926-06-01", + "lat": 34.0522, + "lon": -118.2437, "place": "Los Angeles, CA, USA", "source_policy": "public_record_candidate", - "time": "09:30" + "time": "09:30", + "tz": -8 }, "case_id": "marilyn_monroe", "domains": [ @@ -299,9 +326,12 @@ { "birth": { "date": "1955-10-28", + "lat": 47.6062, + "lon": -122.3321, "place": "Seattle, WA, USA", "source_policy": "public_record_candidate", - "time": "22:00" + "time": "22:00", + "tz": -8 }, "case_id": "bill_gates", "domains": [ @@ -331,9 +361,12 @@ { "birth": { "date": "1965-07-31", + "lat": 0.0, + "lon": 0.0, "place": "Yate, England", "source_policy": "public_record_candidate", - "time": "14:00" + "time": "14:00", + "tz": 0 }, "case_id": "j_k_rowling", "domains": [ @@ -363,9 +396,12 @@ { "birth": { "date": "1918-07-18", + "lat": 0.0, + "lon": 0.0, "place": "Mvezo, South Africa", "source_policy": "public_record_candidate", - "time": "14:54" + "time": "14:54", + "tz": 0 }, "case_id": "nelson_mandela", "domains": [ @@ -395,9 +431,12 @@ { "birth": { "date": "1910-08-26", + "lat": 0.0, + "lon": 0.0, "place": "Skopje, North Macedonia", "source_policy": "public_record_candidate", - "time": "14:25" + "time": "14:25", + "tz": 0 }, "case_id": "mother_teresa", "domains": [ @@ -427,9 +466,12 @@ { "birth": { "date": "1958-08-29", + "lat": 0.0, + "lon": 0.0, "place": "Gary, IN, USA", "source_policy": "public_record_candidate", - "time": "19:33" + "time": "19:33", + "tz": 0 }, "case_id": "michael_jackson", "domains": [ @@ -459,9 +501,12 @@ { "birth": { "date": "1926-04-21", + "lat": 0.0, + "lon": 0.0, "place": "London, England", "source_policy": "public_record_candidate", - "time": "02:40" + "time": "02:40", + "tz": 0 }, "case_id": "queen_elizabeth_ii", "domains": [ @@ -491,9 +536,12 @@ { "birth": { "date": "1917-05-29", + "lat": 0.0, + "lon": 0.0, "place": "Brookline, MA, USA", "source_policy": "public_record_candidate", - "time": "15:00" + "time": "15:00", + "tz": 0 }, "case_id": "john_f_kennedy", "domains": [ @@ -523,9 +571,12 @@ { "birth": { "date": "1929-01-15", + "lat": 0.0, + "lon": 0.0, "place": "Atlanta, GA, USA", "source_policy": "public_record_candidate", - "time": "12:00" + "time": "12:00", + "tz": 0 }, "case_id": "martin_luther_king_jr", "domains": [ @@ -555,9 +606,12 @@ { "birth": { "date": "1975-06-04", + "lat": 0.0, + "lon": 0.0, "place": "Los Angeles, CA, USA", "source_policy": "public_record_candidate", - "time": "09:09" + "time": "09:09", + "tz": 0 }, "case_id": "angelina_jolie", "domains": [ @@ -587,9 +641,12 @@ { "birth": { "date": "1963-12-18", + "lat": 0.0, + "lon": 0.0, "place": "Shawnee, OK, USA", "source_policy": "public_record_candidate", - "time": "06:31" + "time": "06:31", + "tz": 0 }, "case_id": "brad_pitt", "domains": [ @@ -619,9 +676,12 @@ { "birth": { "date": "1981-09-26", + "lat": 0.0, + "lon": 0.0, "place": "Saginaw, MI, USA", "source_policy": "public_record_candidate", - "time": "20:28" + "time": "20:28", + "tz": 0 }, "case_id": "serena_williams", "domains": [ diff --git a/scripts/capture_commercial_astrology_e2e_contexts.py b/scripts/capture_commercial_astrology_e2e_contexts.py index c8024a4a..de76c25a 100644 --- a/scripts/capture_commercial_astrology_e2e_contexts.py +++ b/scripts/capture_commercial_astrology_e2e_contexts.py @@ -68,14 +68,57 @@ def _capture_body(question: dict[str, Any]) -> dict[str, Any]: } -def capture(contract_path: Path = DEFAULT_CONTRACT, output_dir: Path = DEFAULT_OUTPUT_DIR) -> dict[str, Any]: +def _capture_body_for_real_case(case: dict[str, Any], prompt: str) -> dict[str, Any]: + birth = case["birth"] + year, month, day = [int(part) for part in birth["date"].split("-")] + hour, minute = [int(part) for part in birth["time"].split(":")[:2]] + theme_map = { + "timing": "career", + "annual": "career", + "migration": "career", + "family": "marriage", + "education": "career", + } + themes = [theme_map.get(domain, domain) for domain in case["domains"]] + return { + "year": year, + "month": month, + "day": day, + "hour": hour, + "minute": minute, + "lat": birth["lat"], + "lon": birth["lon"], + "tz": birth["tz"], + "city": birth["place"], + "question": prompt, + "question_text": prompt, + "theme": themes, + "evaluation_domains": case["domains"], + "entry_mode": "direct_chart", + "case_id": case["case_id"], + "subject": case["subject"], + "source_policy": birth["source_policy"], + } + + +def capture(contract_path: Path = DEFAULT_CONTRACT, output_dir: Path = DEFAULT_OUTPUT_DIR, max_items: int | None = None) -> dict[str, Any]: contract = _load_json(contract_path) output_dir = output_dir.resolve() output_dir.mkdir(parents=True, exist_ok=True) rows: list[dict[str, Any]] = [] - for question in contract["questions"]: + if "cases" in contract: + questions = [ + {"id": f"{case['case_id']}__{index + 1}", "body": _capture_body_for_real_case(case, prompt)} + for case in contract["cases"] + for index, prompt in enumerate(case["prompts"]) + ] + else: + questions = [{"id": str(question["id"]), "body": _capture_body(question)} for question in contract["questions"]] + for question in questions: + if max_items is not None and len(rows) >= max_items: + break qid = str(question["id"]) - result = execute_consultation_workflow(_capture_body(question), surface="commercial_e2e_capture") + result = execute_consultation_workflow(question["body"], surface="commercial_e2e_capture") context_path = output_dir / f"{qid}.json" context_path.write_text(json.dumps(result, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8") rows.append( @@ -106,8 +149,9 @@ def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--contract", type=Path, default=DEFAULT_CONTRACT) parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR) + parser.add_argument("--max-items", type=int, default=None) args = parser.parse_args() - print(json.dumps(capture(args.contract, args.output_dir), ensure_ascii=False, indent=2, sort_keys=True)) + print(json.dumps(capture(args.contract, args.output_dir, max_items=args.max_items), ensure_ascii=False, indent=2, sort_keys=True)) return 0 diff --git a/tests/test_capture_commercial_astrology_e2e_contexts.py b/tests/test_capture_commercial_astrology_e2e_contexts.py index 1bd60362..efb143dc 100644 --- a/tests/test_capture_commercial_astrology_e2e_contexts.py +++ b/tests/test_capture_commercial_astrology_e2e_contexts.py @@ -24,3 +24,15 @@ def test_capture_writes_runtime_contexts_without_required_layer_echo(tmp_path: P assert data["success"] is True assert "consumer_context" in data assert "required_layers" not in data + + +def test_capture_supports_public_real_case_website_e2e_contract(tmp_path: Path) -> None: + contract = ROOT / "references" / "real_case_calibration" / "real_case_website_e2e_eval_2026_07_20.json" + manifest = capture_script.capture(contract_path=contract, output_dir=tmp_path, max_items=3) + assert manifest["question_count"] == 3 + first = manifest["rows"][0] + assert "__" in first["id"] + context_path = ROOT / first["context_file"] if not Path(first["context_file"]).is_absolute() else Path(first["context_file"]) + data = json.loads(context_path.read_text(encoding="utf-8")) + assert data["success"] is True + assert "consumer_context" in data