research: score Jev intent variants with the previous turn

The earlier report had no previous-turn rows. This run measures V0, V1, V2, and Flash on the re-extracted corpus and records that the real sample is not representative of the simulated set.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Jesse_Chen
2026-09-27 13:10:35 +08:00
co-authored by Cursor
parent fdb7087b19
commit 0b68fa97c4
10 changed files with 338682 additions and 99563 deletions
+139 -3
View File
@@ -11,6 +11,7 @@ from __future__ import annotations
import argparse
import json
import sys
from datetime import datetime
from pathlib import Path
from typing import Any, Mapping, Sequence
@@ -18,9 +19,13 @@ ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
CACHE_DIR = Path(r"G:\Ferti\Jyotisha\.cache\jev_intent")
from scripts.research.jev_intent_cache import cache_dir # noqa: E402
CACHE_DIR = cache_dir()
LEGACY_PATH = CACHE_DIR / "source_b.jsonl"
OUTPUT_PATH = CACHE_DIR / "source_b_v2.jsonl"
ANSWER_CLASSES = {"yes", "weak_yes", "no", "unsure"}
CLOSED_FOCUS = {"resolved", "declined", "skipped", "superseded"}
EXTRACT_SQL = """
select
@@ -32,8 +37,12 @@ select
t.created_at,
c.status as case_status,
f.question_id,
f.intent as focus_intent,
f.expected_answer_schema,
f.status as focus_status
f.status as focus_status,
f.asked_at,
f.resolved_at,
t.message_origin
from public.agentic_rectification_turns t
join public.agentic_rectification_cases c on c.id = t.case_id
left join lateral (
@@ -50,6 +59,124 @@ order by t.case_id, t.created_at
"""
def _timestamp(value: Any) -> datetime | None:
if value is None or value == "":
return None
text = str(value).strip().replace("Z", "+00:00")
try:
return datetime.fromisoformat(text)
except ValueError:
return None
def _choice_copy(schema: Mapping[str, Any]) -> tuple[str, list[dict[str, str]]] | None:
choice = schema.get("choice") if isinstance(schema.get("choice"), dict) else schema
if not isinstance(choice, dict):
return None
prompt = choice.get("prompt")
options = choice.get("options")
if not isinstance(prompt, str) or not prompt.strip():
return None
if not isinstance(options, list) or len(options) != 4:
return None
cleaned: list[dict[str, str]] = []
for item in options:
if not isinstance(item, dict):
return None
key = item.get("key")
label = item.get("label")
answer = item.get("answer_class")
if key not in {"A", "B", "C", "D"} or not isinstance(label, str) or not label.strip():
return None
if answer not in ANSWER_CLASSES:
return None
cleaned.append({"key": str(key), "label": label.strip(), "answer_class": str(answer)})
if len({item["key"] for item in cleaned}) != 4:
return None
if len({item["label"] for item in cleaned}) != 4:
return None
return prompt.strip(), cleaned
def focus_is_open(row: Mapping[str, Any]) -> bool:
"""Production classifies the focus that is still open when the user speaks.
A focus asked earlier and already resolved, declined, skipped, or superseded
is not the current question. The row stays in the extract; its layer is none.
"""
schema = row.get("schema")
if schema is None:
schema = row.get("expected_answer_schema")
if not isinstance(schema, dict) or not schema:
return False
created = _timestamp(row.get("created_at"))
asked = _timestamp(row.get("asked_at"))
resolved = _timestamp(row.get("resolved_at"))
if asked and created and asked > created:
return False
if resolved and created and resolved < created:
return False
status = str(row.get("focus_status") or "")
if status in CLOSED_FOCUS and resolved and created and resolved < created:
return False
return True
def focus_payload(row: Mapping[str, Any]) -> tuple[dict[str, Any] | None, str, bool]:
schema = row.get("schema")
if schema is None:
schema = row.get("expected_answer_schema")
stale = isinstance(schema, dict) and bool(schema) and not focus_is_open(row)
if not focus_is_open(row) or not isinstance(schema, dict):
return None, "none", stale
case_status = str(row.get("case_status") or "collecting_evidence")
choice = _choice_copy(schema)
if choice:
prompt, options = choice
return {
"current_question": prompt,
"options": options,
"case_status": case_status,
"question_id": row.get("question_id"),
}, "choice", False
prompt = schema.get("prompt") if schema.get("collect") is True else None
if isinstance(prompt, str) and prompt.strip():
return {
"current_question": prompt.strip(),
"options": [],
"case_status": case_status,
"question_id": row.get("question_id"),
}, "collect", False
return None, "none", False
def normalize_extract_row(raw: Mapping[str, Any]) -> dict[str, Any]:
focus, layer, stale = focus_payload(raw)
turn_id = str(raw.get("turn_id") or raw.get("id") or "")
return {
"id": turn_id,
"turn_id": turn_id,
"source": "B",
"layer": layer,
"case_id": str(raw.get("case_id") or "") or None,
"user_message": raw.get("user_message"),
"assistant_message": raw.get("assistant_message"),
"created_at": raw.get("created_at"),
"case_status": raw.get("case_status"),
"turn_status": raw.get("turn_status"),
"message_origin": raw.get("message_origin"),
"question_id": raw.get("question_id"),
"focus_intent": raw.get("focus_intent"),
"focus_status": raw.get("focus_status"),
"focus": focus,
"focus_stale": stale,
}
def prepare_extract(raw_rows: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
return attach_previous_turns([normalize_extract_row(row) for row in raw_rows])
def load_jsonl(path: Path) -> list[dict[str, Any]]:
if not path.is_file():
return []
@@ -185,13 +312,22 @@ def main(argv: Sequence[str] | None = None) -> int:
return 0
if args.raw:
raw_rows = load_jsonl(args.raw)
rows = attach_previous_turns(raw_rows)
if raw_rows and any(key in raw_rows[0] for key in ("schema", "expected_answer_schema")):
rows = prepare_extract(raw_rows)
else:
rows = attach_previous_turns([dict(row) for row in raw_rows])
counts = match_gold(rows, legacy)
write_jsonl(args.out, rows)
layers: dict[str, int] = {}
for row in rows:
layer = str(row.get("layer") or "none")
layers[layer] = layers.get(layer, 0) + 1
print(json.dumps({
"out": str(args.out),
"n": len(rows),
"n_with_previous": sum(1 for row in rows if row.get("previous_turn")),
"n_stale_focus": sum(1 for row in rows if row.get("focus_stale")),
"layers": layers,
"gold": counts,
}, ensure_ascii=False))
return 0