Files
Jyotisha/scripts/research/jev_intent_probe_v2.py
T
jesse-ux e3bd3930e3
Independent Staging Quality Gate / validate (push) Successful in 12m0s
Independent Staging Quality Gate / publish (push) Successful in 3m35s
research: compare Jev intent state with the previous turn
V0 on the existing 157 real rows matches the 09-19 cache. The staging extract has no case linkage, so V1/V2 are unmeasured and the verdict stays 缺数据.
2026-09-27 11:41:53 +08:00

460 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Jev intent research v2: previous-turn state variants.
Offline by default for the report. Source B text stays in the gitignored cache.
Committed rows keep gold and model output only.
"""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from typing import Any, Mapping, Sequence
ROOT = Path(__file__).resolve().parents[2]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from scripts.research.jev_intent_probe import ( # noqa: E402
CACHE_DIR,
cache_key,
confusion_counts,
layer_metrics,
load_cache,
load_jsonl,
self_consistency,
strip_confidence,
)
from scripts.research.jev_intent_questions import ( # noqa: E402
JEV_MODEL,
previous_turn_payload,
)
from scripts.research.jev_intent_source_b_v2 import OUTPUT_PATH, from_legacy # noqa: E402
REPORT_JSON = ROOT / "docs" / "research" / "jev_intent_2026_09_27.json"
REPORT_MD = ROOT / "docs" / "research" / "jev_intent_2026_09_27.md"
SAMPLES_DIR = ROOT / "scripts" / "research" / "jev_intent_samples"
MIN_WITH_PREVIOUS = 100
PUBLISHED_C_INTENT = {"choice": 0.990, "collect": 0.922, "none": 0.960}
PRED_KEEP = (
"ok",
"unavailable",
"model",
"intent",
"answer_class",
"has_new_dated_event",
"confidence",
"raw",
"input_tokens",
"output_tokens",
"elapsed_ms",
"continued",
"continues_previous_turn",
"variant",
"error",
)
def pct(value: float | None) -> str:
if value is None:
return "—"
return f"{value:.1%}"
def public_pred(pred: Mapping[str, Any] | None) -> dict[str, Any] | None:
if not pred:
return None
return {key: pred.get(key) for key in PRED_KEEP if key in pred}
def has_previous(row: Mapping[str, Any]) -> bool:
return previous_turn_payload(row) is not None
def unclear_to_focus(rows: Sequence[Mapping[str, Any]], pred_key: str) -> int:
count = 0
for row in rows:
if row.get("layer") != "none":
continue
if (row.get("gold") or {}).get("intent") != "unclear":
continue
if (row.get(pred_key) or {}).get("intent") == "answer_current_focus":
count += 1
return count
def strip_block(block: dict[str, Any] | None) -> dict[str, Any] | None:
"""Flash has no confidence. Leave those rates empty instead of printing 0%."""
if not block:
return None
out = dict(block)
out["all"] = strip_confidence(block.get("all"))
out["by_layer"] = {
layer: strip_confidence(metrics) if metrics else None
for layer, metrics in (block.get("by_layer") or {}).items()
}
out["with_previous"] = strip_confidence(block.get("with_previous"))
out["without_previous"] = strip_confidence(block.get("without_previous"))
return out
def pack_metrics(rows: Sequence[Mapping[str, Any]], pred_key: str) -> dict[str, Any] | None:
usable = [row for row in rows if row.get(pred_key)]
if not usable:
return None
by_layer = {}
for layer in ("choice", "collect", "none"):
subset = [row for row in usable if row.get("layer") == layer]
by_layer[layer] = layer_metrics(subset, pred_key=pred_key) if subset else None
with_prev = [row for row in usable if row.get("has_previous_turn")]
without_prev = [row for row in usable if not row.get("has_previous_turn")]
return {
"all": layer_metrics(usable, pred_key=pred_key),
"by_layer": by_layer,
"with_previous": layer_metrics(with_prev, pred_key=pred_key) if with_prev else None,
"without_previous": layer_metrics(without_prev, pred_key=pred_key) if without_prev else None,
"none_unclear_to_focus": unclear_to_focus(usable, pred_key),
"self_consistency": None,
}
def attach_legacy_preds(samples: Sequence[dict[str, Any]], cache: Mapping[str, Any], mapping: Mapping[str, str]) -> dict[str, int]:
missing = {dest: 0 for dest in mapping}
for sample in samples:
for dest, legacy in mapping.items():
key = cache_key(legacy, sample)
if key in cache:
pred = dict(cache[key])
pred["variant"] = "v0"
sample[dest] = pred
else:
missing[dest] += 1
return missing
def public_rows(samples: Sequence[Mapping[str, Any]], pred_keys: Sequence[str]) -> list[dict[str, Any]]:
rows = []
for sample in samples:
row = {
"id": sample.get("id"),
"source": sample.get("source"),
"layer": sample.get("layer"),
"gold": sample.get("gold"),
"gold_source": sample.get("gold_source"),
"has_previous_turn": bool(sample.get("has_previous_turn")),
"previous_turn_source": sample.get("previous_turn_source"),
}
for key in pred_keys:
pred = public_pred(sample.get(key) if isinstance(sample.get(key), dict) else None)
if pred:
row[key] = pred
rows.append(row)
return rows
def decide(meta: Mapping[str, Any], metrics: Mapping[str, Any]) -> dict[str, str]:
blocked: list[str] = []
if meta.get("n_with_previous", 0) < MIN_WITH_PREVIOUS:
blocked.append(f"n_有上一轮 {meta.get('n_with_previous', 0)} < {MIN_WITH_PREVIOUS}")
if not metrics.get("jev_v1"):
blocked.append("V1 没有预测")
if not metrics.get("jev_v2_chain"):
blocked.append("V2 链式没有预测")
if not metrics.get("jev_v2_gold"):
blocked.append("V2 gold 上界没有预测")
if not metrics.get("flash_v1"):
blocked.append("Flash+V1 没有预测")
if not metrics.get("flash_v0"):
blocked.append("Flash V0 没有预测")
represent = metrics.get("representativeness") or {}
if represent.get("fail"):
blocked.append("来源 B 与来源 C 同层 intent 差 > 10pp")
if blocked:
return {
"verdict": "缺数据",
"blocked": ";".join(blocked),
"reason": "上一轮变量没有测全,不能判过门。已有的 V0 数字只作对照,不代替 V1/V2。",
}
v0_none = ((metrics.get("jev_v0") or {}).get("by_layer") or {}).get("none") or {}
v2_none = ((metrics.get("jev_v2_chain") or {}).get("by_layer") or {}).get("none") or {}
v0_high = v0_none.get("high_conf_error_rate")
v2_high = v2_none.get("high_conf_error_rate")
if v0_high is None or v2_high is None:
return {"verdict": "blocked", "blocked": "无焦点层高置信错误写不出", "reason": "V2 相对 V0 的无焦点层比较缺数字。"}
jev = (metrics.get("jev_v2_chain") or {}).get("all") or {}
flash = (metrics.get("flash_v0") or {}).get("all") or {}
high = jev.get("high_conf_error_rate")
recall = jev.get("low_conf_recall")
jev_intent = jev.get("intent_acc")
flash_intent = flash.get("intent_acc")
missing = [
name for name, value in (
("高置信错误", high),
("低置信召回", recall),
("V2 intent", jev_intent),
("Flash V0 intent", flash_intent),
) if value is None
]
if missing:
return {"verdict": "blocked", "blocked": "、".join(missing), "reason": "门槛里有写不出的数字。"}
gates = []
if high > 0.03:
gates.append(f"高置信错误 {high:.1%} > 3%")
if recall < 0.60:
gates.append(f"低置信召回 {recall:.1%} < 60%")
if jev_intent < flash_intent - 0.03:
gates.append(f"intent {jev_intent:.1%} < 现行 {flash_intent:.1%} − 3pp")
if v2_high >= v0_high:
gates.append(f"无焦点层高置信错误 V2 {v2_high:.1%} 没有低于 V0 {v0_high:.1%}")
if gates:
return {"verdict": "未过门", "blocked": "", "reason": ";".join(gates)}
return {
"verdict": "过门",
"blocked": "",
"reason": "三项门槛都过,且 V2 无焦点层高置信错误低于 V0。",
}
def representativeness(source_b: Sequence[Mapping[str, Any]], source_c: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
layers = {}
fail = False
for layer in ("choice", "collect", "none"):
b_rows = [row for row in source_b if row.get("layer") == layer and row.get("jev_v0_1")]
c_rows = [row for row in source_c if row.get("layer") == layer and row.get("jev_v0_1")]
if not b_rows or not c_rows:
layers[layer] = {"n_b": len(b_rows), "n_c": len(c_rows), "delta": None}
continue
b_acc = layer_metrics(b_rows, pred_key="jev_v0_1")["intent_acc"]
c_acc = layer_metrics(c_rows, pred_key="jev_v0_1")["intent_acc"]
delta = abs(b_acc - c_acc)
fail = fail or delta > 0.10
published = PUBLISHED_C_INTENT[layer]
layers[layer] = {
"n_b": len(b_rows),
"n_c": len(c_rows),
"source_b_intent": b_acc,
"source_c_intent": c_acc,
"delta": delta,
"published_c_intent": published,
"published_delta_pp": (c_acc - published) * 100,
}
return {"fail": fail, "by_layer": layers}
def build_from_samples(source_b: Sequence[dict[str, Any]], source_c: Sequence[dict[str, Any]]) -> dict[str, Any]:
for row in list(source_b) + list(source_c):
if "previous_turn" in row:
row["has_previous_turn"] = has_previous(row)
else:
row["has_previous_turn"] = bool(row.get("has_previous_turn"))
pred_keys = ("jev_v0_1", "jev_v0_2", "flash_v0")
b_metrics = {
"jev_v0": pack_metrics(source_b, "jev_v0_1"),
"flash_v0": strip_block(pack_metrics(source_b, "flash_v0")),
"jev_v1": pack_metrics(source_b, "jev_v1_1"),
"jev_v2_chain": pack_metrics(source_b, "jev_v2_chain_1"),
"jev_v2_gold": pack_metrics(source_b, "jev_v2_gold_1"),
"flash_v1": strip_block(pack_metrics(source_b, "flash_v1")),
}
if b_metrics["jev_v0"]:
b_metrics["jev_v0"]["self_consistency"] = self_consistency(source_b, "jev_v0_1", "jev_v0_2")
none_b = [row for row in source_b if row.get("layer") == "none"]
meta = {
"model": JEV_MODEL,
"sdk": "typesafe-sdk 0.7.0",
"source_b_n": len(source_b),
"n_with_previous": sum(1 for row in source_b if row.get("has_previous_turn")),
"n_without_previous": sum(1 for row in source_b if not row.get("has_previous_turn")),
"layers": {
layer: sum(1 for row in source_b if row.get("layer") == layer)
for layer in ("choice", "collect", "none")
},
"gold_source": {},
"source_c_n": len(source_c),
"previous_decision_note": "链式与 gold 上界都未跑。09-19 文件没有 turn_id / case_id,本机没有 staging 库。",
}
gold_counts: dict[str, int] = {}
for row in source_b:
source = str(row.get("gold_source") or "unknown")
gold_counts[source] = gold_counts.get(source, 0) + 1
meta["gold_source"] = gold_counts
metrics = {
**b_metrics,
"source_c_v0": pack_metrics(source_c, "jev_v0_1"),
"representativeness": representativeness(source_b, source_c),
"source_b_none_confusion": {
"jev_v0": confusion_counts(none_b, "jev_v0_1"),
"flash_v0": confusion_counts(none_b, "flash_v0"),
} if none_b else {},
}
if metrics["source_c_v0"]:
metrics["source_c_v0"]["self_consistency"] = self_consistency(source_c, "jev_v0_1", "jev_v0_2")
conclusion = decide(meta, metrics)
return {
"meta": meta,
"metrics": metrics,
"conclusion": conclusion,
"rows": public_rows(list(source_b) + list(source_c), pred_keys + (
"jev_v1_1", "jev_v1_2", "jev_v2_chain_1", "jev_v2_chain_2",
"jev_v2_gold_1", "jev_v2_gold_2", "flash_v1",
)),
}
def load_samples() -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, int]]:
if OUTPUT_PATH.is_file():
source_b = load_jsonl(OUTPUT_PATH)
else:
source_b = from_legacy(load_jsonl(CACHE_DIR / "source_b.jsonl"))
source_b = [row for row in source_b if isinstance(row.get("gold"), dict) and row["gold"].get("intent")]
source_c = load_jsonl(SAMPLES_DIR / "simulated.jsonl")
cache = load_cache()
missing = {}
missing.update(attach_legacy_preds(source_b, cache, {
"jev_v0_1": "jev_1",
"jev_v0_2": "jev_2",
"flash_v0": "current_1",
}))
missing.update(attach_legacy_preds(source_c, cache, {
"jev_v0_1": "jev_1",
"jev_v0_2": "jev_2",
}))
return source_b, source_c, missing
def recompute_from_rows(rows: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
source_b = [dict(row) for row in rows if row.get("source") == "B"]
source_c = [dict(row) for row in rows if row.get("source") == "C"]
return build_from_samples(source_b, source_c)
def metric_line(title: str, block: Mapping[str, Any] | None) -> str:
if not block or not block.get("all"):
return f"| {title} | — | — | — | — | — | — | — |"
all_m = block["all"]
none_high = ((block.get("by_layer") or {}).get("none") or {}).get("high_conf_error_rate")
return (
f"| {title} | {all_m['n']} | {pct(all_m['intent_acc'])} | {pct(all_m['answer_class_acc'])} | "
f"{pct(all_m['dated_acc'])} | {pct(all_m['high_conf_error_rate'])} | {pct(all_m['low_conf_recall'])} | "
f"{block.get('none_unclear_to_focus')} / 无焦点高置信错误 {pct(none_high)} |"
)
def write_markdown(report: Mapping[str, Any]) -> None:
meta = report["meta"]
metrics = report["metrics"]
conclusion = report["conclusion"]
layers = meta.get("layers") or {}
gold = meta.get("gold_source") or {}
represent = metrics.get("representativeness") or {}
lines = [
"# TypeSafe Jev 意图分类 · 上一轮 state 对照(2026-09-27)",
"",
f"- 任务:`docs/tasks/TASK-rectification-jev-intent-classifier-research-v2-20260927.md`",
f"- 基线:`origin/staging` @ `710c848b`",
f"- 模型:`{meta.get('model')}`;SDK `{meta.get('sdk')}`",
f"- 结论:**{conclusion.get('verdict')}**。{conclusion.get('reason')}",
"",
"## 样本",
"",
f"- 来源 B:{meta.get('source_b_n')} 条。有上一轮 {meta.get('n_with_previous')},无上一轮 {meta.get('n_without_previous')}。",
f"- 层:点选 {layers.get('choice', 0)} / 采集 {layers.get('collect', 0)} / 无焦点 {layers.get('none', 0)}。",
f"- gold 来源:{json.dumps(gold, ensure_ascii=False)}。",
f"- {meta.get('previous_decision_note')}",
"",
"## 来源 B",
"",
"| 变体 | n | intent | answer_class | dated | 高置信错误 | 低置信召回 | 无焦点 unclear→answer |",
"| --- | ---: | ---: | ---: | ---: | ---: | ---: | --- |",
metric_line("Jev V0(09-19 缓存)", metrics.get("jev_v0")),
metric_line("Flash V0(09-19 缓存,生产提示)", metrics.get("flash_v0")),
metric_line("Jev V1", metrics.get("jev_v1")),
metric_line("Jev V2 链式", metrics.get("jev_v2_chain")),
metric_line("Jev V2 gold 上界", metrics.get("jev_v2_gold")),
metric_line("Flash + V1", metrics.get("flash_v1")),
"",
f"Jev V0 自洽率:{pct((metrics.get('jev_v0') or {}).get('self_consistency'))}。",
"",
"## 来源 C 回归锚(只 V0)",
"",
]
c_block = metrics.get("source_c_v0") or {}
for layer in ("choice", "collect", "none"):
cell = (represent.get("by_layer") or {}).get(layer) or {}
published = cell.get("published_c_intent")
got = cell.get("source_c_intent")
delta = cell.get("published_delta_pp")
lines.append(
f"- {layer}:重算 {pct(got)},09-19 公布 {pct(published)},差 {delta if delta is None else round(delta, 2)} pp(n={cell.get('n_c')})。"
)
lines += [
f"- 来源 C 自洽率:{pct(c_block.get('self_consistency'))}。",
"",
"## 无焦点层混淆(来源 B,V0)",
"",
"计数来自报告 JSON,不含原文。",
"",
]
confusion = (metrics.get("source_b_none_confusion") or {}).get("jev_v0") or {}
labels = confusion.get("labels") or []
counts = confusion.get("counts") or {}
if labels:
lines.append("| gold \\ pred | " + " | ".join(labels) + " |")
lines.append("| --- | " + " | ".join("---:" for _ in labels) + " |")
for gold in labels:
cells = [str((counts.get(gold) or {}).get(pred, 0)) for pred in labels]
lines.append(f"| {gold} | " + " | ".join(cells) + " |")
lines += [
"",
"## 写不出的项",
"",
conclusion.get("blocked") or "无",
"",
"V1、V2、Flash+V1 要等 staging 库抽出带 `case_id` 的上一轮,并且本机有 `DEEPSEEK_API_KEY` 之后才能补。门槛不放宽。",
"",
]
REPORT_MD.write_text("\n".join(lines) + "\n", encoding="utf-8")
def main(argv: Sequence[str] | None = None) -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--offline", action="store_true")
parser.add_argument("--from-report", action="store_true", help="recompute tables from the committed JSON rows")
args = parser.parse_args(argv)
if args.from_report:
if not REPORT_JSON.is_file():
print("report json missing", file=sys.stderr)
return 2
payload = json.loads(REPORT_JSON.read_text(encoding="utf-8"))
report = recompute_from_rows(payload.get("rows") or [])
report["meta"]["recomputed_from"] = "report_rows"
write_markdown(report)
print(json.dumps({
"verdict": report["conclusion"]["verdict"],
"source_b_intent": ((report["metrics"].get("jev_v0") or {}).get("all") or {}).get("intent_acc"),
"n": report["meta"]["source_b_n"],
"n_with_previous": report["meta"]["n_with_previous"],
}, ensure_ascii=False))
return 0
source_b, source_c, missing = load_samples()
report = build_from_samples(source_b, source_c)
report["meta"]["legacy_cache_missing"] = missing
report["meta"]["offline"] = True
REPORT_JSON.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
write_markdown(report)
print(json.dumps({
"verdict": report["conclusion"]["verdict"],
"source_b_n": report["meta"]["source_b_n"],
"n_with_previous": report["meta"]["n_with_previous"],
"missing": missing,
"json": str(REPORT_JSON),
}, ensure_ascii=False))
return 0
if __name__ == "__main__":
raise SystemExit(main())