Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
507ef959f5 | ||
|
|
33367fb081 | ||
|
|
eb8f5147f4 |
@@ -1,5 +1,9 @@
|
||||
# 印度占星 Skill 更新日志
|
||||
|
||||
## 2026-09-19 — 设置弹窗布局与资料入口整理
|
||||
|
||||
设置弹窗的四个分区继续共用固定外框,桌面导航加宽并与内容区分隔,右侧内容增加稳定内边距,表单阅读宽度更舒适;导航去掉会误导成页面跳转的右箭头。个人资料的账户头像预览放大到 56px,星盘资料的“添加其他人”移到分组标题旁,列表较长时也容易找到。Skill 版本不变(纯界面改动)。
|
||||
|
||||
## 2026-09-18 — 账户弹窗重新能点;校正读不到下一题时会说出来
|
||||
|
||||
侧栏打开个人资料、星盘资料、设置、点数或退出登录时,弹窗里的按钮和表单又能点了,确认退出也恢复了。生时校正如果没读到下一题,会写「这一问还没读到。」并给出「重新读取」,不再停在空白里装没事。Skill 版本不变。
|
||||
|
||||
@@ -12781,3 +12781,19 @@
|
||||
- 关联记录:BUG-525(题干经 asked_turn_id 挂上)、BUG-930(回答钉顶:只见最后一轮是设计)、BUG-965
|
||||
- 复发自:待确认
|
||||
- 修复版本:`824ecff0`
|
||||
|
||||
## BUG-970 | 设置弹窗内容区留白过大、导航入口语义像页面跳转
|
||||
|
||||
- 状态:investigating
|
||||
- 首次发现:2026-09-19
|
||||
- 最近更新:2026-09-19
|
||||
- 影响面:设置弹窗的桌面布局、个人资料头像区、星盘资料列表入口。
|
||||
- 用户现象:设置弹窗左侧导航过窄,右侧内容贴边且留下较大空白;分区按钮尾部箭头暗示会进入下一页;个人资料头像锚点偏小;星盘资料的添加入口位于列表末尾,资料较多时不稳定。
|
||||
- 触发条件:登录后打开设置弹窗,在桌面宽度切换四个设置分区,查看个人资料或星盘资料列表。
|
||||
- 根因:旧布局使用 176px 导航、三列导航按钮(为同弹窗内切换保留无实际用途的箭头列)、内容区没有明确内边距,表单子内容上限为 440px;头像编辑仍以 48px 预览布局;添加入口与列表内容耦合在列表底部。
|
||||
- 修复:导航调整为 200px 并增加 hairline 边界,移除误导性尾部箭头;内容区增加 32px 桌面内边距,表单阅读上限调整为 560px;个人资料头像预览调整为 56px;“添加其他人”移动到“其他人”分组标题操作区。保留四分区共享 `.settings-modal` 和 `vh` + `@supports (height: 1dvh)` 回退契约。
|
||||
- 验证:`./node_modules/.bin/tsc --noEmit` 安装依赖后退出码 0;`npm run lint` 退出码 0(0 errors、120 warnings);设置相关定向测试 21/21 通过。广泛测试筛选命令为 3481 tests / 3402 pass / 79 fail,失败集中在重复迁移、Windows 路径、Docker/数据库和技能 symlink 等环境或基线问题,不能作为本轮 UI 回归归因。`npm run build` 已完成编译和 TypeScript,但在收集 `/api/daily-starlanguage` 页面数据时因 Windows 禁止创建 Skill runtime symlink(`EPERM`)失败,因此尚未确认 `/` Static 或首屏 gzip;登录态浏览器走查仍待受控环境。
|
||||
- 防复发:设置导航继续使用 icon + label 的两列结构且不添加页面跳转箭头;右侧滚动容器负责内边距,表单 cap 只作用于其子内容;列表视图默认不渲染表单,添加入口固定在“其他人”标题操作区;`frontend/DESIGN.md` 与合同测试同步锁定这些关系。
|
||||
- 相关记录:BUG-554、BUG-698
|
||||
- 复发自:无
|
||||
- 修复版本:待发布
|
||||
|
||||
@@ -244,6 +244,180 @@
|
||||
"cost_usd": 0.001293768
|
||||
}
|
||||
},
|
||||
"current_source_b": {
|
||||
"n": 157,
|
||||
"unavailable": 0,
|
||||
"intent_acc": 0.89171974522293,
|
||||
"answer_class_acc": 0.9702970297029703,
|
||||
"answer_class_n": 101,
|
||||
"dated_acc": 0.9363057324840764,
|
||||
"all_acc": 0.8407643312101911,
|
||||
"high_conf_error_rate": null,
|
||||
"low_conf_coverage": null,
|
||||
"low_conf_recall": null,
|
||||
"no_vs_unsure": 0,
|
||||
"median_ms": 685.0,
|
||||
"p95_ms": 938.0,
|
||||
"mean_input_tokens": 516.6496815286624,
|
||||
"cost_usd": 0.0034067880000000004
|
||||
},
|
||||
"current_source_b_by_layer": {
|
||||
"choice": {
|
||||
"n": 17,
|
||||
"unavailable": 0,
|
||||
"intent_acc": 0.8823529411764706,
|
||||
"answer_class_acc": 1.0,
|
||||
"answer_class_n": 1,
|
||||
"dated_acc": 0.8823529411764706,
|
||||
"all_acc": 0.8823529411764706,
|
||||
"high_conf_error_rate": null,
|
||||
"low_conf_coverage": null,
|
||||
"low_conf_recall": null,
|
||||
"no_vs_unsure": 0,
|
||||
"median_ms": 657.0,
|
||||
"p95_ms": 763.0,
|
||||
"mean_input_tokens": 450.5882352941176,
|
||||
"cost_usd": 0.00032172
|
||||
},
|
||||
"collect": {
|
||||
"n": 107,
|
||||
"unavailable": 0,
|
||||
"intent_acc": 0.9532710280373832,
|
||||
"answer_class_acc": 0.97,
|
||||
"answer_class_n": 100,
|
||||
"dated_acc": 0.9439252336448598,
|
||||
"all_acc": 0.8785046728971962,
|
||||
"high_conf_error_rate": null,
|
||||
"low_conf_coverage": null,
|
||||
"low_conf_recall": null,
|
||||
"no_vs_unsure": 0,
|
||||
"median_ms": 689.0,
|
||||
"p95_ms": 938.0,
|
||||
"mean_input_tokens": 528.9345794392524,
|
||||
"cost_usd": 0.0023770320000000003
|
||||
},
|
||||
"none": {
|
||||
"n": 33,
|
||||
"unavailable": 0,
|
||||
"intent_acc": 0.696969696969697,
|
||||
"answer_class_acc": null,
|
||||
"answer_class_n": 0,
|
||||
"dated_acc": 0.9393939393939394,
|
||||
"all_acc": 0.696969696969697,
|
||||
"high_conf_error_rate": null,
|
||||
"low_conf_coverage": null,
|
||||
"low_conf_recall": null,
|
||||
"no_vs_unsure": 0,
|
||||
"median_ms": 715.0,
|
||||
"p95_ms": 978.0,
|
||||
"mean_input_tokens": 510.8484848484849,
|
||||
"cost_usd": 0.0007080360000000001
|
||||
}
|
||||
},
|
||||
"source_b_jev_self_consistency": {
|
||||
"all": 0.9617834394904459,
|
||||
"choice": 0.9411764705882353,
|
||||
"collect": 0.9906542056074766,
|
||||
"none": 0.8787878787878788
|
||||
},
|
||||
"source_b_none_confusion": {
|
||||
"jev": {
|
||||
"labels": [
|
||||
"provide_new_evidence",
|
||||
"stop_rectification",
|
||||
"ask_about_result",
|
||||
"unclear",
|
||||
"answer_current_focus"
|
||||
],
|
||||
"counts": {
|
||||
"provide_new_evidence": {
|
||||
"provide_new_evidence": 24,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
},
|
||||
"stop_rectification": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
},
|
||||
"ask_about_result": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 1,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
},
|
||||
"unclear": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 1,
|
||||
"answer_current_focus": 7
|
||||
},
|
||||
"answer_current_focus": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
}
|
||||
},
|
||||
"other": 0,
|
||||
"n": 33
|
||||
},
|
||||
"current": {
|
||||
"labels": [
|
||||
"provide_new_evidence",
|
||||
"stop_rectification",
|
||||
"ask_about_result",
|
||||
"unclear",
|
||||
"answer_current_focus"
|
||||
],
|
||||
"counts": {
|
||||
"provide_new_evidence": {
|
||||
"provide_new_evidence": 21,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 3
|
||||
},
|
||||
"stop_rectification": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
},
|
||||
"ask_about_result": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 1,
|
||||
"answer_current_focus": 0
|
||||
},
|
||||
"unclear": {
|
||||
"provide_new_evidence": 1,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 2,
|
||||
"answer_current_focus": 5
|
||||
},
|
||||
"answer_current_focus": {
|
||||
"provide_new_evidence": 0,
|
||||
"stop_rectification": 0,
|
||||
"ask_about_result": 0,
|
||||
"unclear": 0,
|
||||
"answer_current_focus": 0
|
||||
}
|
||||
},
|
||||
"other": 0,
|
||||
"n": 33
|
||||
}
|
||||
},
|
||||
"representativeness": {
|
||||
"note": "来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。",
|
||||
"fail": true,
|
||||
@@ -101740,7 +101914,7 @@
|
||||
},
|
||||
"conclusion": {
|
||||
"verdict": "缺数据",
|
||||
"reason": "来源 B 与来源 C 同层 intent 准确率差 > 10pp,模拟语料不代表真人,来源 C 门槛结论降为缺数据。来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。 同时来源 C 绝对门槛未过:低置信召回 23.1% / 45.5% / 25.0% < 60%(错了却仍高置信)。 相对 −3pp(同一样本):choice Jev 99.0% vs 现行 98.8%(门槛 95.8%,过);collect Jev 92.2% vs 现行 97.5%(门槛 94.5%,未过);none Jev 96.0% vs 现行 94.0%(门槛 91.0%,过)。",
|
||||
"reason": "来源 B 与来源 C 同层 intent 准确率差 > 10pp,模拟语料不代表真人,来源 C 门槛结论降为缺数据。来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。 同时来源 C 绝对门槛未过:低置信召回 23.1% / 45.5% / 25.0% < 60%(错了却仍高置信)。 相对 −3pp(同一样本):choice Jev 99.0% vs 现行 98.8%(门槛 95.8%,过);collect Jev 92.2% vs 现行 97.5%(门槛 94.5%,不可判(复核与对照同源));none Jev 96.0% vs 现行 94.0%(门槛 91.0%,过)。",
|
||||
"if_connect": "不得上线。先补真机样本或重造更像真人的来源 C,再测。"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
|
||||
## 结论
|
||||
|
||||
**缺数据**。来源 B 与来源 C 同层 intent 准确率差 > 10pp,模拟语料不代表真人,来源 C 门槛结论降为缺数据。来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。 同时来源 C 绝对门槛未过:低置信召回 23.1% / 45.5% / 25.0% < 60%(错了却仍高置信)。 相对 −3pp(同一样本):choice Jev 99.0% vs 现行 98.8%(门槛 95.8%,过);collect Jev 92.2% vs 现行 97.5%(门槛 94.5%,未过);none Jev 96.0% vs 现行 94.0%(门槛 91.0%,过)。
|
||||
**缺数据**。来源 B 与来源 C 同层 intent 准确率差 > 10pp,模拟语料不代表真人,来源 C 门槛结论降为缺数据。来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。 同时来源 C 绝对门槛未过:低置信召回 23.1% / 45.5% / 25.0% < 60%(错了却仍高置信)。 相对 −3pp(同一样本):choice Jev 99.0% vs 现行 98.8%(门槛 95.8%,过);collect Jev 92.2% vs 现行 97.5%(门槛 94.5%,不可判(复核与对照同源));none Jev 96.0% vs 现行 94.0%(门槛 91.0%,过)。
|
||||
|
||||
若接:不得上线。先补真机样本或重造更像真人的来源 C,再测。
|
||||
|
||||
@@ -41,18 +41,49 @@
|
||||
|
||||
同一样本上 Jev vs 现行(相对门槛用这一表):
|
||||
|
||||
| 层 | n | Jev intent | 现行 intent | 差(Jev−现行) | 门槛现行−3pp |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: |
|
||||
| choice | 400 | 99.0% | 98.8% | 0.2% | 95.8% |
|
||||
| collect | 400 | 92.2% | 97.5% | -5.2% | 94.5% |
|
||||
| none | 100 | 96.0% | 94.0% | 2.0% | 91.0% |
|
||||
| 层 | n | Jev intent | 现行 intent | 差(Jev−现行) | 门槛现行−3pp | 判定 |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: | --- |
|
||||
| choice | 400 | 99.0% | 98.8% | 0.2% | 95.8% | 过 |
|
||||
| collect | 400 | 92.2% | 97.5% | -5.2% | 94.5% | 不可判(复核与对照同源) |
|
||||
| none | 100 | 96.0% | 94.0% | 2.0% | 91.0% | 过 |
|
||||
|
||||
## 代表性检验(来源 B vs 来源 C)
|
||||
|
||||
来源 B 已标注 157 条(点选 17 / 采集 107 / 无焦点 33;未标注 0),长度 P25/中位/P75/最长 = 12/21/28/172。人工标注,未用模型代标。Jev 在来源 B 上 intent:点选 94.1%、采集 94.4%、无焦点 78.8%;全集 91.1%。无焦点层差 17.2pp > 10pp。
|
||||
|
||||
来源 B 与来源 C 同层 intent 准确率:choice B 94.1% vs C 99.0%(差 4.9%,n_B=17);collect B 94.4% vs C 92.2%(差 2.1%,n_B=107);none B 78.8% vs C 96.0%(差 17.2%,n_B=33)。 有层差 > 10pp,结论降为缺数据。
|
||||
|
||||
来源 B 是真人 + 人工标注 + 线上模型三者齐备的唯一一组。现行无置信度,置信度三列为空。
|
||||
|
||||
| 范围 | n | Jev intent | 现行 intent | Jev 高置信错误 | Jev 低置信召回 | Jev 自洽 |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
||||
| 全集 | 157 | 91.1% | 89.2% | 6.4% | 24.1% | 96.2% |
|
||||
| choice | 17 | 94.1% | 88.2% | 0.0% | 66.7% | 94.1% |
|
||||
| collect | 107 | 94.4% | 95.3% | 5.6% | 18.2% | 99.1% |
|
||||
| none | 33 | 78.8% | 69.7% | 12.1% | 20.0% | 87.9% |
|
||||
|
||||
无焦点层现行 intent 69.7%、Jev 78.8%。现行更低,说明 78.8% 主要是这 33 条本身难,不是单 Jev 不行。
|
||||
|
||||
无焦点层 gold × 预测混淆计数(只有计数,无原文)。预测出现 `answer_current_focus` 是因为模型把无焦点句当成在回答采集题。
|
||||
|
||||
### gold × Jev(n=33)
|
||||
|
||||
| gold \ pred | provide_new_evidence | stop_rectification | ask_about_result | unclear | answer_current_focus |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: |
|
||||
| provide_new_evidence | 24 | 0 | 0 | 0 | 0 |
|
||||
| stop_rectification | 0 | 0 | 0 | 0 | 0 |
|
||||
| ask_about_result | 0 | 0 | 1 | 0 | 0 |
|
||||
| unclear | 0 | 0 | 0 | 1 | 7 |
|
||||
| answer_current_focus | 0 | 0 | 0 | 0 | 0 |
|
||||
|
||||
### gold × 现行(n=33)
|
||||
|
||||
| gold \ pred | provide_new_evidence | stop_rectification | ask_about_result | unclear | answer_current_focus |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: |
|
||||
| provide_new_evidence | 21 | 0 | 0 | 0 | 3 |
|
||||
| stop_rectification | 0 | 0 | 0 | 0 | 0 |
|
||||
| ask_about_result | 0 | 0 | 0 | 1 | 0 |
|
||||
| unclear | 1 | 0 | 0 | 2 | 5 |
|
||||
| answer_current_focus | 0 | 0 | 0 | 0 | 0 |
|
||||
|
||||
## 置信度–准确率曲线与 θ
|
||||
|
||||
推荐 θ = 0.9。点:
|
||||
@@ -154,6 +185,25 @@
|
||||
| 若同一句话既明确否定当前采集题又补充了新的带时间经历,intent 仍为 answer_current_focus 且 answer_class 为 no,不要改成 provide_new_evidence。 | Code keeps no + Noul true. Jev intent/Noul are independent. | Jev may split this pair; code does not re-vote intent from the Noul. |
|
||||
| 不要按关键词表或正则猜测,只根据当前问题与用户这句话的语义分类。 | Not sent. Jev has no keyword table in the question. | The meta-instruction is dropped (Jev answers the written question, not the intended one). |
|
||||
|
||||
## 限制
|
||||
|
||||
1. **复核 ≈ 生产提示,且复核模型 = 对照模型。** `REVIEW_RUBRIC` 与生产 `COLLECT_INSTRUCTIONS` 逐句对应;生成 / 复核 / 对照都是 `deepseek-flash`。进入测试集的 900 条是「Flash 用近生产提示能答对目标标签」的那 900 条,被剔的 36 条恰是 Flash 不同意的。因此来源 C 上现行 97.5% / 98.8% / 94.0% 是构造出来的上界。采集层「Jev 92.2% 未过相对门槛 94.5%」**不可当作 Jev 输给现行的证据**。
|
||||
2. **采集层 intent 错例的 gold 有争议。** 来源 C 采集层 Jev intent 错例 31 条:`provide_new_evidence → answer_current_focus` 17、`unclear → answer_current_focus` 9、`stop → unclear` 4、`ask → unclear` 1。17 条 pne 几乎全是「另外 2019 年我换工作搬了家」句式,若干尾句落在当前题域,按生产提示可读成 `answer_current_focus + unsure`。9 条 unclear(「一时半会儿真捋不明白」)按「记不清 → unsure」也读得通。两类合计 ≥ 20 条,占该层 intent 错例约 2/3。gold 来自「生成目标 + Flash 复核同意」,不等于人工真值。改写示例:
|
||||
- 「另外 2019 年换过工作,感情那会儿真没细想。」gold=provide_new_evidence;可读成在回答感情采集题。
|
||||
- 「另外 2019 年搬了家,工作那摊子反而没顾上细想。」gold=provide_new_evidence;可读成在回答工作采集题。
|
||||
- 「这事我一时真说不上来。」gold=unclear;按生产提示是 unsure。
|
||||
3. **语料仍不像真人。** 修复单 1 只把长度和人设写成硬红线(已过)。原单还要求按来源 B 的标点 / 语气词比例约束,未进红线:
|
||||
|
||||
| 指标 | 来源 B(真人) | 来源 C(模拟) |
|
||||
| --- | ---: | ---: |
|
||||
| 含标点 | 13% | 98.9% |
|
||||
| 含语气词(吧/呢/啊/嗯/哦/额/emm) | 4% | 25.6% |
|
||||
| 含年份或月份 | 81% | 47.8% |
|
||||
| 带年份句里写「2019」 | — | 236 / 429 = 55% |
|
||||
| `provide_new_evidence` 里是搬家/换工作 | — | 152 / 157 |
|
||||
|
||||
根因:生成脚本 `ALT_EVENT_HINTS` 给七个领域的「另一件事」全是搬家/换工作,年份未约束,模型收敛到「另外 2019 年搬过家」。这解释了模拟语料不代表真人的一部分,也解释了采集层错例为何长得一样。本单不修,留给产品决定是否再造一轮。
|
||||
|
||||
## 回退
|
||||
|
||||
任何上线方案必须保留回退到现行会话模型的路径。官方限流会动态调整。
|
||||
|
||||
@@ -119,3 +119,20 @@ cache 样本 id 与 `simulated.jsonl` 一致(900/900)。unavailable = 0。
|
||||
- `python -m pytest tests/test_jev_intent_research.py -q`:11 passed
|
||||
- 构建脚本无 `bank = {`;复核函数无 `re.search` 判标签
|
||||
- 线上分类器文件未改
|
||||
|
||||
## 修复轮 2(2026-09-19,`codex/rectification-jev-intent-research-fix2-20260919`)
|
||||
|
||||
任务书:`TASK-rectification-jev-intent-classifier-research-fix2-20260919.md`。**0 token**:从 `.cache/jev_intent/jev_runs_v2.json` 聚合已跑结果,未重造语料、未重跑模型。
|
||||
|
||||
补进报告的三组数(来源 B 157 条):
|
||||
|
||||
| 范围 | n | Jev intent | 现行 Flash intent | Jev 高置信错误 | Jev 低置信召回 | Jev 自洽 |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
||||
| 全集 | 157 | 91.1% | 89.2% | **6.4%** | 24.1% | 96.2% |
|
||||
| 点选 | 17 | 94.1% | 88.2% | 0.0% | 66.7% | 94.1% |
|
||||
| 采集 | 107 | 94.4% | 95.3% | 5.6% | 18.2% | 99.1% |
|
||||
| 无焦点 | 33 | 78.8% | **69.7%** | **12.1%** | 20.0% | 87.9% |
|
||||
|
||||
无焦点层 7 条 Jev 错全是 `unclear → answer_current_focus`。现行在无焦点上更低(69.7%),10 条错里 5 条同样是 `unclear → answer_current_focus`。说明 78.8% 主要是这 33 条难,不是单 Jev 不行。
|
||||
|
||||
采集层相对门槛改为「不可判(复核与对照同源)」。限制三节已写入报告。结论仍 **缺数据**,不上线。未立 BUG。
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
# 设置面板 UI 改造进度
|
||||
|
||||
- 日期:2026-09-19
|
||||
- 基线:`origin/staging` / `4f4cd684`
|
||||
- 分支:`codex/settings-ui-20260919`
|
||||
- 范围:设置弹窗布局、个人资料、星盘资料列表入口;不修改账户 API、星盘 API、数据库或工作流。
|
||||
|
||||
## 已完成
|
||||
|
||||
- 设置弹窗保留四个分区共用 `.settings-modal`,桌面左侧导航从 176px 调整为 200px,并增加导航与内容之间的 hairline 分隔。
|
||||
- 设置导航移除误导性的尾部右箭头,保留图标 + 分区名称;标题增加“设置”眉标题并保留当前分区标题。
|
||||
- 右侧内容区增加桌面 32px 内边距,表单阅读宽度上限从 440px 调整为 560px;移动端使用 16px 内容内边距。
|
||||
- 保留 `vh` 基线与 `@supports (height: 1dvh)` 升级,不回退到会被压缩器吃掉的重复声明写法。
|
||||
- 个人资料头像文案改为“账户头像”,Beam 说明收紧,预览从 48px 改为 56px;同步桌面与移动端 CSS。
|
||||
- “添加其他人”移动到“其他人”分组标题右侧,列表继续默认不展开表单。
|
||||
- 更新 `frontend/DESIGN.md`、设置布局合同测试、资料面板测试和星盘资料合同测试。
|
||||
|
||||
## 断言变更记录
|
||||
|
||||
| 文件 | 原值 | 新值 | 原因 |
|
||||
| --- | --- | --- | --- |
|
||||
| `frontend/tests/profile-panel.test.ts` | `size={48}` | `size={56}` | 头像预览需要更清晰的视觉锚点,并与账户身份区留出稳定间距。 |
|
||||
| `frontend/tests/account-dialog-overlay.test.ts` | 表单 cap 420–460px | `max-width: 560px` | 内容区已有明确内边距,资料表单需要舒适阅读宽度,避免右侧留下过大的空白。 |
|
||||
|
||||
## 测试结果
|
||||
|
||||
已执行结果:
|
||||
|
||||
- `npm run lint`:退出码 0,120 warnings、0 errors;警告与本轮改动无关,未顺手修改。
|
||||
- `./node_modules/.bin/tsc --noEmit`:安装当前 worktree 依赖后无输出并以退出码 0 结束。
|
||||
- 设置相关定向测试:21 tests / 21 pass / 0 fail,覆盖设置导航、固定外框、两列导航、内容内边距、560px 表单 cap、头像 56px、星盘资料列表入口与详情动作。
|
||||
- 全量 `npm test -- --test-name-pattern=...`:退出码 1,3481 tests / 3402 pass / 79 fail;失败集中包含重复迁移文件、Windows 路径拼接、Docker/数据库和技能 symlink 权限等环境或基线问题,不能宣称全量通过,需与基线逐条比对。
|
||||
- `npm run build`:编译与 TypeScript 阶段通过,但在收集 `/api/daily-starlanguage` 页面数据时因 Windows 禁止创建 Skill runtime symlink 失败;这是环境权限缺口,不是本轮设置 UI 编译错误。构建未形成可用于确认 `/` Static 或首屏 gzip 的完整产物。
|
||||
|
||||
## 浏览器验收环境缺口
|
||||
|
||||
尚未进行登录态下的浏览器截图级走查。需要受控账号、Chrome/真实设备和真实设置内容,按 `docs/testing/settings-dialog-20260916.md` 复核四分区切换、深色主题、移动端四栏、个人资料头像和星盘资料列表。没有受控登录态或 Chrome 时,不把该项写成通过。
|
||||
|
||||
## 未完成事项
|
||||
|
||||
- 浏览器/真实设备验收仍待具备受控登录态与 Chrome 的环境执行;BUG-970 继续保持 `investigating`。
|
||||
- 构建因 Windows Skill runtime symlink 权限缺口未完成页面数据收集,需在允许 symlink 的环境复跑,才能确认 `/` Static 与首屏 gzip。
|
||||
- 完整测试的 79 个失败需在可用环境与基线逐条比对;本记录不把它们归因于本轮 UI 改动。
|
||||
@@ -234,7 +234,7 @@
|
||||
| `TASK-rectification-open-collect-invite-20260914.md` | `PROGRESS-rectification-open-collect-invite-20260914.md` | **P0**:固定七条采集线问完后只说「能问的都问完了」,用户不知道还能补经历、也不知道补了有用;而两轮研究证明补带年月经历是唯一有效手段。产品拍板:交付卡照出 + 卡上给不限领域的补充邀请(先要确切日期,再退年月;举七条线之外的例子),补完必须可见生效(BUG-689) | 待验收 | `codex/rectification-open-collect-invite-20260914` |
|
||||
|
||||
| `TASK-rectification-cluster-width-research-20260914.md` | `PROGRESS-rectification-cluster-width-research-20260914.md` | **研究单**:上一轮证明调权重改不动交付区间宽度——所有方案宽度中位数都等于整个搜索窗。先确认 sweep 的宽度口径是否含淘汰(M0),再画簇结构像(M1),最后量三个改法:放宽簇上限、按分差决定是否合并、交付区间改分位覆盖(M2)。真值覆盖率不得下降 | 待验收 | `codex/rectification-cluster-width-research-20260914` |
|
||||
| `TASK-rectification-jev-intent-classifier-research-20260919.md` | `PROGRESS-rectification-jev-intent-classifier-research-20260919.md` | **研究单**:TypeSafe Jev(只做 Choice/Score/Noul 的校准判断模型,$0.042/Mtok)能否接管校正流的意图分类。产品 09-19 授权评估(推翻 09-15「分类只用贵模型」需重新拍板)。Agent 模拟校正流造 ≥900 条中文语料(标签先定、独立复核)+ 真机样本做代表性锚,量准确率 / 高置信错误率 / 低置信召回 / 延迟;只离线测,不改线上 | **待验收**(修复轮已交,结论 **缺数据**) | 修复单 `TASK-rectification-jev-intent-classifier-research-fix-20260919.md` 已做完。来源 C 由 DeepSeek Flash 生成+复核(900 条,争议率 2.85%);来源 B 157 条已人工标注。Jev 无焦点层 vs 真人差 17.2pp > 10pp。落点分支 `codex/rectification-jev-intent-research-fix-20260919` |
|
||||
| `TASK-rectification-jev-intent-classifier-research-20260919.md` | `PROGRESS-rectification-jev-intent-classifier-research-20260919.md` | **研究单**:TypeSafe Jev(只做 Choice/Score/Noul 的校准判断模型,$0.042/Mtok)能否接管校正流的意图分类。产品 09-19 授权评估(推翻 09-15「分类只用贵模型」需重新拍板)。Agent 模拟校正流造 ≥900 条中文语料(标签先定、独立复核)+ 真机样本做代表性锚,量准确率 / 高置信错误率 / 低置信召回 / 延迟;只离线测,不改线上 | **待验收**(fix2 已交,结论仍 **缺数据**) | `codex/rectification-jev-intent-research-fix2-20260919`。来源 B 现行 Flash 已入表(全集 89.2%、无焦点 69.7%);Jev 高置信错误全集 6.4% / 无焦点 12.1%;无焦点混淆全是 unclear→answer_current_focus。采集层相对门槛标不可判。 |
|
||||
|
||||
| `TASK-rectification-minute-resolution-research-20260914.md` | `PROGRESS-rectification-minute-resolution-research-20260914.md` | **研究单**:候选分不开的根因是打分尺度——窗口内恒定项 11.5 分 vs 随分钟变化项 2.125 分(≈5:1)。先修封存基准(v3 每例仅 3 件事且被标 invalidated)出 v4,再离线量五个改法:分盘除数、去底座、**KP 宫头子主计分(产品 09-14 拍板,推翻 BUG-325 一条红线)**、年精度事件改边际似然、聚类签名层对齐。有收益才立实现单 | 待验收 | `codex/rectification-minute-resolution-research-20260914` |
|
||||
|
||||
@@ -283,6 +283,7 @@
|
||||
| `TASK-sidebar-unify-20260916.md` | `PROGRESS-sidebar-unify-20260916.md` | **侧栏统一成一个组件 + 次级页共享外壳 + 跳转不整页刷新(纯前端)**:`/` 用 `AppSidebar`、四个次级页用另写的 `AppNavRail`,两套标记结构共用一份 CSS,会话行少 `.session-main` 一层(无内边距 / 44px 高 / 当前项标记、留 44px 空列)、页脚 56px 菜单 vs 44px 文字;首页进次级页是 `window.location.assign` 整页刷新,次级页各自在组件里挂 `SecondaryShell`、无共享 layout 无缓存,每次跳转重拉 `/api/sessions` + `/api/account`;折叠状态不持久。产品拍板 D1 只留一个 `AppSidebar`(操作回调可选 = 只读模式,D9「不带写操作」维持)、D2 路由组 `app/(secondary)/` layout 承载外壳、D3 改 `<Link>`、D4 模块级 60 s 缓存 + 首页写操作失效、D5 折叠状态存 cookie、D6 回首页慢的启动链另开一单。实测:接口 TTFB 0.63~0.90 s,回 `/` 启动链串行 ≥4 次往返 + 揭幕闸门 4 s。BUG-744~746 | 已验收(2026-09-16,Opus 子代理实现 `22edabbc`,门禁全绿、失败清单与基线逐条相同;折叠状态取 localStorage 触发让步 1;真机清单待走) | `codex/sidebar-unify-20260916` |
|
||||
| `TASK-rectification-precision-gate-guided-collect-fix-20260916.md` | `PROGRESS-rectification-precision-gate-guided-collect-fix-20260916.md` | **验收修复单**:F1 `decision_receipt` 新增 `guided_collect_windows` 让 `test_rectification_engine_memoization` golden 红(在 CORE_PYTEST_TARGETS,staging 未部署 cfb41daf);F2 16 条既有校正断言红未按三栏改;F3 `mayDeliverOnPrecision` 缺省即放行、短路 BUG-654 且 `guidedCollectExhausted` 不含 refresh;F4 无领域轨道窗口轮询贴领域、一窗只问一领域;F5 离线回放注入 month_lo 非真值方向、0/20 不作数;F6 录入卡合成「或」列表句走模型轮;F7 page.tsx +1、手造 fixture。D6 产品已拍板:门槛只是必要条件,仍问完线再出。BUG 段 744 起 | **已验收通过,已合入 staging**;合入后 run `2697` 仍红 capability audit(`(secondary)/chart`),跟进 BUG-753 | `dc732825`(Opus 子代理实现,Claude 独立验收):tsc 0 / lint 0 error / npm test 3405 条 31 红与基线 a396bbe0 逐条一致(16 条校正红转绿、0 新红)/ pytest 193 通过 / `○ /` Static / gzip 582,552 → 582,800(+0.04%)/ page.tsx 1836 → 1835 / golden diff 仅加 `guided_collect_windows` 一键 / Skill 10.0.28。F5 真值方向回放三档仍 0 例靠引导件达标,作研究结论记录。D2+D6 合并后 10 分钟门槛只上报不改出卡时机 |
|
||||
| — (产品口头拍板,无任务书) | `PROGRESS-agent-voice-20260917.md` | **对话口气改形状(纯提示词与文案,无 BUG 号)**:产品判定现有人设出来的是顾问报告。人设改成「把人当一个人认真对待 / 行动力很强、嘴有点毒但靠谱的同事」;开场从「一句结论 + 2–3 条短要点 + 一句下一步」换成固定形状——**反差**(表面 A 底下 B,命名成一个格局)→ **谁在推谁在修**(大运主星推、行运修体面)→ **别去应 X 的象,去扮演 Y 的象** → **最多三条短行动**(各 ≤ 20 字),仍无标题、≤ 400 字。新增希望纪律:有转机且允许精确应期就说到月份,没有就说这段时间拿来干什么、能扮演哪个象,禁「一切都会好 / 相信自己 / 加油 / 你值得更好的 / 宇宙自有安排」。申报时段与无出生分钟两条降级路线形状照给、不编月份。改 `product-voice.ts` / `consult/route.ts` 用户回合文案 / 三处复述旧形状的系统指令 / `VOICE.md`(五条原则→七条)/ `CHANGELOG.md`。零业务逻辑改动,Skill 版本不变 | 已实现,待验收 | `codex/agent-voice-20260917`:tsc 0 / lint 0 error(118 warning 不变)/ npm test 3468→3471 条、36 红与基线 `ff0427cf` 逐条相同、0 新红 / `○ /` Static / 首屏 gzip 1,416,965 两侧**字节相同**(改动全在服务端模块,客户端 chunk 无该文本)。真机口气走查欠 |
|
||||
| — (产品口头拍板,无任务书) | `PROGRESS-settings-ui-20260919.md` | **设置面板布局与资料入口整理**:基线 `4f4cd684`;四分区继续共用固定 `.settings-modal`,桌面导航 176px→200px 并加分隔,内容区增加内边距,表单 cap 440px→560px,导航移除误导性右箭头;账户头像 48px→56px;“添加其他人”移到分组标题操作区。已同步 `frontend/DESIGN.md` 与合同测试;tsc 0、lint 0 error、定向测试 21/21;build 被 Windows Skill runtime symlink 权限阻塞,浏览器走查待受控环境;BUG-970 保持 `investigating` | 待验收 | `codex/settings-ui-20260919` |
|
||||
|
||||
## 命名与归档
|
||||
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
# TASK · Jev 意图分类研究修复单 2:把已跑未报的数字补进报告(2026-09-19)
|
||||
|
||||
- 基线:`origin/staging` @ `eb8f5147`(含 `c2ecbc74` 修复轮)
|
||||
- 原单:`TASK-rectification-jev-intent-classifier-research-20260919.md`;修复单 1:`TASK-rectification-jev-intent-classifier-research-fix-20260919.md`
|
||||
- 分支:`codex/rectification-jev-intent-research-fix2-20260919`(从 `origin/staging` 新建)
|
||||
- 性质:报告补漏;**不重造语料、不重跑 Jev、不改线上代码、不立 BUG**。除 F1 第 3 条外全部可离线从 `.cache/jev_intent/` 出,不消耗 token。
|
||||
|
||||
## 1. 修复轮验收结论(对照修复单 1 逐条)
|
||||
|
||||
| 项 | 结论 | 证据(验收方在 `origin/staging` @ `eb8f5147` 独立重算) |
|
||||
| --- | --- | --- |
|
||||
| 红线 1 · 不改线上 | 通过 | `git diff --stat d6fc4fb8..c2ecbc74` 只有 `scripts/research/**`、`tests/test_jev_intent_research.py`、`docs/**`、`BLOCKED.md` |
|
||||
| 红线 2 · key 不进仓库 | 通过 | 全仓 grep 无 token;脚本只读环境变量 |
|
||||
| 红线 3 · 真实消息不提交 | 通过 | 报告 JSON 里 `"source": "B"` 0 行,只有聚合数;`.cache/jev_intent/` 不在仓库 |
|
||||
| 红线 4 · sha256 固化 | 通过 | 三份 sha256 与提交文件实算一致;报告 `meta.sha256` 同值;`id_check_ok = true`,`rows` 的 C 类 id 集合与 `simulated.jsonl` 完全相等 |
|
||||
| 红线 5 · 快速门 | 通过(Python 定向)/ 环境缺口 | `python3 -m unittest tests.test_jev_intent_research`:11 passed;本机无 `.venv`,`run_quality_gate.py --profile quick` 未跑 |
|
||||
| 红线 6 · 真实模型生成与复核 | 通过 | `complete_chat` 走 DeepSeek Chat Completions;`GENERATOR_NAME` / `REVIEWER_NAME` = `deepseek-flash`;README 有生成 / 复核提示原文;`review_sample` 无正则;源码无 `bank = {` |
|
||||
| 红线 7 · 语料三项统计 | 通过 | 去重 825 / 900、单条最多重复 5、八种人设最少 48(点选+采集)、惜字如金+方言 = 20.0%(≤ 20% 的边界值)、长度 P25 / 中位 / P75 = 16 / 20 / 24(P25 落在 8–16 的上沿) |
|
||||
| 红线 8 · 现行对照全量 | 通过 | 来源 C 现行 400 / 400 / 100 全量一次,第二次 133 / 133 / 33 |
|
||||
| F1 语料重造 | 通过 | 争议率 2.85%(36 / 1261),三层 400 / 400 / 100,配额偏差 ≤ 4.5% |
|
||||
| F2 重跑 T2 | 通过 | Jev×2、现行×1 + 1/3;`unavailable = 0`;报告表内每个数字与我从 `rows` 重算一致 |
|
||||
| F3 报告 | **部分未通过** | 结论「缺数据」成立;但 §2 列出的已跑未报数字缺席 |
|
||||
| F4 来源 B | **部分未通过** | 157 条人工标注、Jev×2、现行×1 都跑了;现行在来源 B 上的准确率**没有出现在任何文件里**;T3 要求「写明差在哪一类标签」只写到层,没到标签 |
|
||||
| F5 记录 | 通过 | PROGRESS 修复轮段、BLOCKED 三条划掉、README 状态行、token 分列齐全 |
|
||||
|
||||
**结论「缺数据」采纳,产品不据此上线。** 未通过项只影响「下一步该补什么」的判断,不影响本轮结论。
|
||||
|
||||
## 2. 事故实证
|
||||
|
||||
### 2.1 现行模型在来源 B 上跑了、没报(P1)
|
||||
|
||||
- `scripts/research/jev_intent_probe.py` `main()`:`if source_b: run_batch(source_b, …, run_id="current_1", call_fn=call_current_retry)` —— 现行模型确实在 157 条真人样本上跑了一次;PROGRESS token 表也计了「B 一次」。
|
||||
- 报告 JSON `metrics` 只有 `source_b` / `source_b_by_layer` 两项,且全是 `jev_1` 的数;没有 `current` 在来源 B 上的任何聚合。MD、PROGRESS 同样没有。
|
||||
- 后果:无焦点层 Jev 78.8%(26 / 33)是「Jev 不行」还是「这 33 条本身难 / 标注有争议」,现在无法判断;而这是真人 + 人工标注 + 线上模型三者齐备的**唯一**一组数,也是产品下一步决策最需要的数。
|
||||
|
||||
### 2.2 来源 B 上 Jev 的高置信错误没进报告(P1)
|
||||
|
||||
报告 JSON 里已有、MD 没写的数:
|
||||
|
||||
| 来源 B(Jev) | 全集 n=157 | 点选 n=17 | 采集 n=107 | 无焦点 n=33 |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| intent 准确率 | 91.1% | 94.1% | 94.4% | 78.8% |
|
||||
| 三题全对 | 81.5% | 82.4% | 89.7% | 54.5% |
|
||||
| 高置信错误率(≥0.8 且错) | **6.4%** | 0.0% | 5.6% | **12.1%** |
|
||||
| 低置信召回 | 24.1% | 66.7% | 18.2% | 20.0% |
|
||||
| dated 准确率 | 91.7% | 88.2% | 97.2% | 75.8% |
|
||||
|
||||
高置信错误的绝对门槛是 ≤ 3%;真人样本上全集 6.4%、无焦点 12.1%,比来源 C 的 0–2.8% 差得多。这是比「层差 17.2pp」更直接的不可接证据,报告必须写。
|
||||
|
||||
### 2.3 无焦点层差在哪一类标签没写(P2,原单 T3 明文要求)
|
||||
|
||||
原单 T3:「报告必须写明差在哪一类标签」。现报告只到层。来源 B 无焦点层 7 条错,gold 与 Jev 各是什么、是否集中在 `provide_new_evidence` ↔ `unclear`,要按标签给混淆表(只给计数,不给原文)。
|
||||
|
||||
### 2.4 复核模型 = 对照模型 + 复核提示 ≈ 生产提示(P2,方法学限制,写进报告即可)
|
||||
|
||||
- `REVIEW_RUBRIC`(`jev_intent_corpus_build.py`)与生产 `COLLECT_INSTRUCTIONS` 逐句对应:「没有/没发生/这方面没什么 → no」「记不清/…/以后再说 → unsure」「不要把「没有」或「记不清」标成 yes」「不要按 A/B/C/D 位置猜」「既明确否定又补充新经历 → answer_current_focus + no」。
|
||||
- 复核与对照都是 `deepseek-flash`。于是进入 `simulated.jsonl` 的 900 条,是「Flash 用近生产提示能答对目标标签」的那 900 条;被剔的 36 条恰是 Flash 不同意的。
|
||||
- 后果:现行模型在来源 C 上的 97.5% / 98.8% / 94.0% 是构造出来的上界,「采集层 Jev 92.2% 未过相对门槛 94.5%」不能当作 Jev 输给现行的证据。修复单 1 允许同一模型生成 + 复核,但没预见复核提示与对照提示同源;本单不要求换模型重复核(产品未拍板再花 token),只要求报告写明这条限制并把相对门槛那一行标「不可判」。
|
||||
|
||||
### 2.5 Jev 采集层 31 条 intent 错例里有标签争议(P2,写进错例画像即可)
|
||||
|
||||
验收方逐条看了 Jev 在采集层的 intent 错例(gold → Jev):`provide_new_evidence → answer_current_focus` 17 条、`unclear → answer_current_focus` 9 条、`stop → unclear` 4 条、`ask → unclear` 1 条。
|
||||
|
||||
- 17 条 pne 错例几乎全是「另外 2019 年我换工作搬了家」句式;其中若干条尾句直接落在当前题域(例:题问感情,答「…感情那会儿真没细想」;题问工作,答「工作那摊子反而没顾上细想」),按生产提示读是 `answer_current_focus + unsure`,Jev 的判法有依据。
|
||||
- 9 条 unclear 错例的句子是「一时半会儿真捋不明白」「这事我一时真说不上来」,按生产提示「记不清/想不起来 → unsure」读也是 `answer_current_focus + unsure`;现行模型自己也在其中 3 条上判了 `answer_current_focus`。
|
||||
- 这两类合计 ≥ 20 条,占 Jev 采集层 intent 错例的 2/3。它们的 gold 由「生成目标 + Flash 复核同意」得来,不等于人工真值。
|
||||
|
||||
### 2.6 语料像真人的程度(观察项,不判失败)
|
||||
|
||||
修复单 1 只把长度和人设写成硬红线,三项都过;原单 T0 还要求按来源 B 的标点 / 语气词比例约束生成,没进红线,实测差距如下:
|
||||
|
||||
| 指标 | 来源 B(真人) | 来源 C(模拟) |
|
||||
| --- | ---: | ---: |
|
||||
| 含标点 | 13% | 98.9% |
|
||||
| 含语气词 | 4% | 25.6% |
|
||||
| 含年份或月份 | 81% | 47.8% |
|
||||
| 带年份句里写「2019」 | — | 236 / 429 = 55% |
|
||||
| `provide_new_evidence` 里是搬家/换工作 | — | 152 / 157 |
|
||||
|
||||
根因在 `ALT_EVENT_HINTS`:七个领域的「另一件事」提示全是「搬家 / 换工作」两种,年份没约束,模型就收敛到「另外 2019 年搬过家」。这解释了「模拟语料不代表真人」的一部分,也解释了 2.5 的错例为何长得一样。不在本单修,写进报告「限制」一节,留给产品决定是否再造一轮。
|
||||
|
||||
## 3. 决策记录
|
||||
|
||||
1. 本轮结论「缺数据」采纳。**不上线、不影子双跑**,直到 2.1 的数字出来后产品再拍板下一步(补真机样本 / 再造语料 / 关单)。
|
||||
2. 本单不新增任何模型调用(F1 第 3 条例外见下),全部从 `.cache/jev_intent/` 已有结果聚合;执行方若发现 cache 里没有来源 B 的 `current_1` 结果,按让步 1 处理。
|
||||
3. 不换复核模型、不重造语料;2.4 / 2.6 只写限制。
|
||||
|
||||
## 4. 硬红线
|
||||
|
||||
原单 §4 与修复单 1 §5 全部继续生效。另加:
|
||||
|
||||
9. 来源 B 仍只写聚合与混淆计数,**不得**出现任何一条真人原文、id、时间戳。
|
||||
10. 不得为让来源 B 的数字好看而改 gold;标注争议只能写成「争议 n 条」并列出争议类型。
|
||||
|
||||
## 5. 任务分解
|
||||
|
||||
### F1 · 报告补齐来源 B 三组数(P1)
|
||||
|
||||
1. `jev_intent_probe.py` 的指标段增加 `current_source_b` / `current_source_b_by_layer`(口径与 `source_b` 相同;现行无置信度,置信度三列记 `null`),MD「代表性检验」一节改成一张表:行 = 三层 + 全集,列 = n / Jev intent / 现行 intent / Jev 高置信错误 / Jev 低置信召回 / Jev 自洽(jev_1 vs jev_2 在来源 B 上)。
|
||||
2. 无焦点层按标签给 gold × Jev 与 gold × 现行两张混淆计数表(4 × 4,只有计数)。
|
||||
3. 若 cache 里没有来源 B 的现行结果,允许只为这 157 条跑一次现行(约 16 万 input token),PROGRESS 记明。
|
||||
4. 验收:MD 表里每个数能从报告 JSON 重算;`tests/test_jev_intent_research.py` 新增一条断言:报告 JSON 的 `metrics` 含 `current_source_b` 且 `n == meta.source_b_n`。
|
||||
|
||||
### F2 · 报告增加「限制」一节(P2)
|
||||
|
||||
写三条:复核 ≈ 生产提示且同模型(2.4)、采集层错例的标签争议(2.5,给计数与 3 条改写示例)、语料风格差距表(2.6)。「现行模型对照」一节的采集层「未过」改成「不可判(复核与对照同源)」。
|
||||
|
||||
### F3 · 记录
|
||||
|
||||
- PROGRESS 追加「修复轮 2」段。
|
||||
- `docs/tasks/README.md` 状态行改为本单结论。
|
||||
- 不立 BUG。
|
||||
|
||||
## 6. 让步顺序
|
||||
|
||||
1. cache 里没有来源 B 现行结果且产品不再给 key:F1 第 1 条只填 Jev 列,现行列标 `unavailable`,PROGRESS 写明;F1 第 2 条照做。
|
||||
2. 无焦点层 7 条错例里有执行方自己都拿不准的标注:写「标注争议 n 条」,不改 gold。
|
||||
|
||||
## 7. 开工前置命令
|
||||
|
||||
```bash
|
||||
git fetch origin --prune
|
||||
git worktree add -b codex/rectification-jev-intent-research-fix2-20260919 .worktrees/rectification-jev-intent-research-fix2-20260919 origin/staging
|
||||
cd .worktrees/rectification-jev-intent-research-fix2-20260919
|
||||
ls .cache/jev_intent/ # 确认 source_b.jsonl 与 runs cache 在
|
||||
python3 -m unittest tests.test_jev_intent_research
|
||||
```
|
||||
|
||||
## 8. BUG 编号
|
||||
|
||||
不立 BUG。`docs/BUG_HISTORY.md` 当前最大号 BUG-970。
|
||||
+6
-6
@@ -533,12 +533,12 @@ page. Three parts now, in reading order:
|
||||
### Settings dialog
|
||||
|
||||
- **Placement:** the overlay and the onboarding paywall portal to `document.body`. `SidebarInset` stays `inert` while a dialog is open (the BUG-744~746 focus contract); the dialog itself must not sit inside that subtree (BUG-968). Closing still uses the existing overlay click, the header button, and the window-level Escape listener.
|
||||
- **Structure:** one fixed chrome for four panes — 个人资料, 星盘资料, 账户与点数, 通用设置. Left nav is 176px and does not scroll; the title bar stays put; only the right-hand content pane scrolls. Logout stays a separate 400px confirmation.
|
||||
- **Width / height:** desktop `width: min(100vw - 32px, 880px)`, with `height: min(84vh, 640px)` as the base and `min(84dvh, 640px)` applied inside `@supports (height: 1dvh)`. All four panes share one class (`.settings-modal`), so switching panes cannot change the frame. At ≤767px the dialog is full-screen with four equal tabs along the top.
|
||||
- **Pane menu states:** default is transparent with secondary ink; hover is a 55% wash of `--color-canvas-muted` keeping secondary ink; current is the solid muted surface with primary ink. No accent bar, and no weight change — hierarchy here comes from ink rank and surface, matching “Hierarchy inside the nav comes from ink rank, not hue”. The sidebar's 2px `--sidebar-ring` on the active session is deliberately **not** changed to match; the two surfaces read differently on purpose until that is revisited.
|
||||
- **Content width:** the 880px frame leaves roughly 690px of content, which is too wide for a single column of fields. Form panes (个人资料, 通用设置) cap their children at 440px and stay left aligned (`.settings-dialog-content--form > *`); list panes (星盘资料, 账户与点数) stay full-bleed so tables and card grids keep their columns. The cap sits on the children of the scroll container only — never on `.settings-modal` or `.account-settings-shell` — so it cannot make the frame resize between panes (BUG-554 / BUG-698).
|
||||
- **Personal profile:** one row of avatar editing (48px preview, eight palettes, 换一个形象) plus nickname and login email. No duplicate 管理星盘资料 button.
|
||||
- **Chart library:** list first (self row, other rows, 添加其他人). A row opens a detail with ← 星盘资料. Other details own 设为默认 / 删除 / 用于合盘 and that person's synastry history. The add form is a view, not an always-on stack.
|
||||
- **Structure:** one fixed chrome for four panes — 个人资料, 星盘资料, 账户与点数, 通用设置. Left nav is 200px, has a hairline boundary, and does not scroll; the title bar stays put; only the right-hand content pane scrolls. The pane menu uses icon + label rows without a trailing chevron because these controls switch content inside the same dialog. Logout stays a separate 400px confirmation.
|
||||
- **Width / height:** desktop `width: min(100vw - 32px, 880px)`, with `height: min(84vh, 640px)` as the base and `min(84dvh, 640px)` applied inside `@supports (height: 1dvh)`. The right content pane uses 32px horizontal and bottom inset on desktop. All four panes share one class (`.settings-modal`), so switching panes cannot change the frame. At ≤767px the dialog is full-screen with four equal tabs along the top and 16px content inset.
|
||||
- **Pane menu states:** default is transparent with secondary ink; hover is a 55% wash of `--color-canvas-muted` keeping secondary ink; current is the solid muted surface with primary ink. No accent bar, no trailing navigation chevron, and no weight change — hierarchy here comes from ink rank and surface, matching “Hierarchy inside the nav comes from ink rank, not hue”. The sidebar's 2px `--sidebar-ring` on the active session is deliberately **not** changed to match; the two surfaces read differently on purpose until that is revisited.
|
||||
- **Content width:** the 880px frame now gives the content pane a deliberate inset instead of leaving controls against the divider. Form panes (个人资料, 通用设置) cap their children at 560px and stay left aligned (`.settings-dialog-content--form > *`); list panes (星盘资料, 账户与点数) stay full-bleed within the inset so tables and card grids keep their columns. The cap sits on the children of the scroll container only — never on `.settings-modal` or `.account-settings-shell` — so it cannot make the frame resize between panes (BUG-554 / BUG-698).
|
||||
- **Personal profile:** one row of avatar editing (56px preview, eight palettes, 换一个形象) plus nickname and login email. The header uses the product-level “设置” eyebrow above the current pane title; there is no duplicate 管理星盘资料 button.
|
||||
- **Chart library:** list first (self row, other rows, 添加其他人). 添加其他人 sits in the 其他人 group heading action area instead of below the list, so the entry remains discoverable when the list grows. A row opens a detail with ← 星盘资料. Other details own 设为默认 / 删除 / 用于合盘 and that person's synastry history. The add form is a view, not an always-on stack.
|
||||
- **States:** open, pane switch, list / self / other / add, saving, success, and error.
|
||||
|
||||
### Billing pane
|
||||
|
||||
@@ -1831,23 +1831,21 @@ button:disabled { cursor: default; opacity: .45; }
|
||||
.account-modal-overlay { position: fixed; z-index: 40; inset: 0; display: grid; place-items: center; padding: var(--space-4); background: var(--color-scrim); animation: account-overlay-enter 180ms ease-out both; }
|
||||
.account-modal { width: min(100%, 560px); max-height: min(84vh, 760px); overflow-y: auto; padding: var(--space-8); border: 1px solid var(--color-border); border-radius: var(--radius-xl); background: var(--color-canvas); box-shadow: var(--shadow-elevated); animation: account-dialog-enter 180ms var(--ease-out) both; }
|
||||
.settings-modal { width: min(100vw - 32px, 880px); height: min(84vh, 640px); max-height: min(84vh, 640px); overflow: hidden; display: grid; grid-template-rows: auto 1fr; }
|
||||
.account-settings-shell { display: grid; grid-template-columns: 176px minmax(0, 1fr); min-height: 0; overflow: hidden; align-items: stretch; }
|
||||
.settings-dialog-nav { display: grid; align-content: start; gap: var(--space-1); overflow: hidden; }
|
||||
.settings-dialog-nav-item { min-height: 44px; display: grid; grid-template-columns: 18px minmax(0, 1fr) 16px; align-items: center; gap: var(--space-2); padding: 0 var(--space-3); border: 0; border-radius: var(--radius-md); background: transparent; color: var(--color-ink-secondary); cursor: pointer; text-align: left; }
|
||||
.account-settings-shell { display: grid; grid-template-columns: 200px minmax(0, 1fr); min-height: 0; overflow: hidden; align-items: stretch; }
|
||||
.settings-dialog-nav { display: grid; align-content: start; gap: var(--space-1); overflow: hidden; padding-right: var(--space-4); border-right: 1px solid var(--color-border); }
|
||||
.settings-dialog-nav-item { min-height: 44px; display: grid; grid-template-columns: 18px minmax(0, 1fr); align-items: center; gap: var(--space-2); padding: 0 var(--space-3); border: 0; border-radius: var(--radius-md); background: transparent; color: var(--color-ink-secondary); cursor: pointer; text-align: left; }
|
||||
.settings-dialog-nav-item:hover { background: color-mix(in srgb, var(--color-canvas-muted) 55%, transparent); color: var(--color-ink-secondary); }
|
||||
.settings-dialog-nav-item[aria-current="page"] { background: var(--color-canvas-muted); color: var(--color-ink); }
|
||||
.settings-dialog-nav-item > svg { width: 18px; height: 18px; color: var(--color-ink-tertiary); }
|
||||
.settings-dialog-nav-item > svg:last-child { width: 15px; height: 15px; margin-left: auto; }
|
||||
.settings-dialog-content { min-width: 0; min-height: 0; overflow-y: auto; }
|
||||
/* Form panes (个人资料 / 通用设置) read as one column, so they are capped and left
|
||||
aligned instead of stretching across the ~690px content area. The cap lives on the
|
||||
children, never on the scroll container or the dialog box: .settings-modal keeps its
|
||||
fixed width/height, so the four panes still cannot change the dialog size (BUG-554 /
|
||||
BUG-698). List panes (星盘资料 / 账户与点数) stay full-bleed. */
|
||||
.settings-dialog-content--form > * { max-width: 440px; margin-right: auto; }
|
||||
.settings-dialog-copy { margin: 0 0 var(--space-5); color: var(--color-ink-secondary); font-size: var(--type-body-sm); line-height: 1.6; }
|
||||
.settings-dialog-content { min-width: 0; min-height: 0; overflow-y: auto; padding: 0 var(--space-8) var(--space-8); }
|
||||
/* Form panes (个人资料 / 通用设置) use a comfortable reading width instead of
|
||||
leaving the content stranded in a narrow 440px column. The cap stays on pane
|
||||
children, never on the scroll container or dialog box, so pane switches keep the
|
||||
same frame (BUG-554 / BUG-698). List panes remain full-bleed within this inset. */
|
||||
.settings-dialog-content--form > * { max-width: 560px; margin-right: auto; }
|
||||
.settings-dialog-copy { max-width: 760px; margin: 0 0 var(--space-5); color: var(--color-ink-secondary); font-size: var(--type-body-sm); line-height: 1.6; }
|
||||
.avatar-section { padding-top: 0; }
|
||||
.avatar-editor { display: grid; grid-template-columns: 48px minmax(0, 1fr); align-items: center; gap: var(--space-5); margin-top: var(--space-5); }
|
||||
.avatar-editor { display: grid; grid-template-columns: 56px minmax(0, 1fr); align-items: center; gap: var(--space-5); margin-top: var(--space-5); }
|
||||
.avatar-editor-controls { min-width: 0; display: grid; gap: var(--space-4); }
|
||||
.avatar-palette-list { display: grid; grid-template-columns: repeat(4, minmax(48px, 1fr)); gap: var(--space-2); }
|
||||
.avatar-palette { min-width: 0; height: 34px; overflow: hidden; display: flex; padding: 3px; border: 1px solid var(--color-border); border-radius: var(--radius-md); background: var(--color-canvas); cursor: pointer; transition: border-color 120ms ease-out, box-shadow 120ms ease-out, transform 120ms ease-out; }
|
||||
@@ -1864,6 +1862,8 @@ button:disabled { cursor: default; opacity: .45; }
|
||||
.logout-modal { width: min(100%, 400px); }
|
||||
.account-modal h2, .auth-panel h1 { font-family: var(--font-display); font-weight: 500; letter-spacing: -.5px; text-wrap: balance; }
|
||||
.account-modal h2 { margin: 0; font-size: var(--type-display-sm); letter-spacing: -.025em; }
|
||||
.account-modal-heading { min-width: 0; display: grid; gap: var(--space-1); }
|
||||
.account-modal-eyebrow { color: var(--color-ink-tertiary); font-size: var(--type-overline); font-weight: 600; letter-spacing: .08em; text-transform: uppercase; }
|
||||
.account-modal-header { display: flex; align-items: flex-start; justify-content: space-between; gap: var(--space-4); padding-bottom: var(--space-5); }
|
||||
.dialog-close { width: 44px; height: 44px; display: grid; flex: 0 0 auto; place-items: center; border: 0; cursor: pointer; transition: background-color 120ms ease-out, transform 120ms ease-out; border-radius: var(--radius-md); background: var(--color-canvas-muted); }
|
||||
.redeem-balance { display: flex; align-items: baseline; justify-content: space-between; gap: var(--space-4); padding: var(--space-4) 0; border-top: 1px solid var(--color-border); border-bottom: 1px solid var(--color-border); }
|
||||
@@ -1890,6 +1890,8 @@ button:disabled { cursor: default; opacity: .45; }
|
||||
.chart-library-panel { display: grid; gap: var(--space-5); margin-top: var(--space-5); }
|
||||
.chart-library-group { display: grid; gap: var(--space-3); }
|
||||
.chart-library-group-heading { display: flex; align-items: center; justify-content: space-between; gap: var(--space-3); }
|
||||
.chart-library-group-heading > div:first-child { min-width: 0; }
|
||||
.chart-library-group-heading-actions { display: flex; align-items: center; flex: 0 0 auto; gap: var(--space-2); }
|
||||
.chart-library-group-heading b, .chart-library-group-heading small { display: block; }
|
||||
.chart-library-group-heading b { color: var(--color-ink); font-size: var(--type-caption); font-weight: 600; }
|
||||
.chart-library-group-heading small { margin-top: 3px; color: var(--color-ink-secondary); font-size: var(--type-caption); font-weight: 400; }
|
||||
@@ -1899,6 +1901,7 @@ button.chart-library-item { cursor: pointer; text-align: left; }
|
||||
.chart-library-item-chevron { width: 16px; height: 16px; flex: 0 0 auto; color: var(--color-ink-tertiary); }
|
||||
.chart-library-back { min-height: 44px; display: inline-flex; align-items: center; gap: var(--space-2); margin: 0 0 var(--space-4); padding: 0; border: 0; background: transparent; color: var(--color-ink-secondary); cursor: pointer; font-size: var(--type-body-sm); }
|
||||
.chart-library-add { margin-top: var(--space-4); }
|
||||
.chart-library-add--heading { margin-top: 0; white-space: nowrap; }
|
||||
.chart-library-item-main { min-width: 0; }
|
||||
.chart-library-item strong, .chart-library-item small { display: block; }
|
||||
.chart-library-item strong { color: var(--color-ink); font-size: var(--type-body-md); font-weight: 500; }
|
||||
@@ -2025,13 +2028,16 @@ input:not([type="radio"]):not([type="checkbox"]):not([class^="ant-"]):not([class
|
||||
.account-modal { width: 100%; max-height: 100vh; min-height: 100vh; border-radius: 0; padding: var(--space-6); }
|
||||
.settings-modal { width: 100%; height: 100vh; max-height: 100vh; }
|
||||
.account-settings-shell { grid-template-columns: 1fr; grid-template-rows: auto 1fr; gap: var(--space-4); }
|
||||
.settings-dialog-nav { position: static; grid-template-columns: repeat(4, minmax(0, 1fr)); gap: var(--space-1); padding-bottom: var(--space-3); border-bottom: 1px solid var(--color-border); }
|
||||
.settings-dialog-nav-item { min-height: 40px; grid-template-columns: 1fr; justify-items: center; gap: 2px; padding: var(--space-2); text-align: center; font-size: var(--type-caption); }
|
||||
.settings-dialog-nav-item > svg:last-child { display: none; }
|
||||
.settings-dialog-content { min-height: 0; overflow-y: auto; }
|
||||
.avatar-editor { grid-template-columns: 48px minmax(0, 1fr); gap: var(--space-4); }
|
||||
.avatar-editor > .user-avatar { width: 48px !important; height: 48px !important; }
|
||||
.settings-dialog-nav { position: static; grid-template-columns: repeat(4, minmax(0, 1fr)); gap: var(--space-1); padding-bottom: var(--space-3); border-right: 0; border-bottom: 1px solid var(--color-border); }
|
||||
.settings-dialog-nav-item { min-height: 44px; grid-template-columns: 1fr; justify-items: center; gap: 2px; padding: var(--space-2); text-align: center; font-size: var(--type-caption); }
|
||||
.settings-dialog-content { min-height: 0; overflow-y: auto; padding: 0 var(--space-4) var(--space-6); }
|
||||
.settings-dialog-content--form > * { max-width: none; }
|
||||
.avatar-editor { grid-template-columns: 56px minmax(0, 1fr); gap: var(--space-4); }
|
||||
.avatar-editor > .user-avatar { width: 56px !important; height: 56px !important; }
|
||||
.avatar-palette-list { grid-template-columns: repeat(2, minmax(64px, 1fr)); }
|
||||
.chart-library-group-heading { align-items: flex-start; }
|
||||
.chart-library-group-heading-actions { align-items: flex-end; flex-direction: column; gap: var(--space-1); }
|
||||
.chart-library-add--heading { min-height: 40px; padding-inline: var(--space-3); }
|
||||
.auth-page { height: 100vh; overflow-x: hidden; overflow-y: auto; -webkit-overflow-scrolling: touch; padding: 0; }
|
||||
.auth-shell { min-height: 100%; overflow: visible; align-content: start; grid-template-columns: 1fr; grid-template-rows: auto auto; border-radius: 0; box-shadow: none; }
|
||||
.auth-story { min-height: 0; justify-content: flex-start; gap: var(--space-3); padding: var(--space-6) var(--space-6) var(--space-5); }
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"use client";
|
||||
|
||||
import { ChevronRight, Settings, UserRound, Users, WalletCards, X } from "lucide-react";
|
||||
import { Settings, UserRound, Users, WalletCards, X } from "lucide-react";
|
||||
import { memo, type MutableRefObject, type ReactNode } from "react";
|
||||
import { createPortal } from "react-dom";
|
||||
|
||||
@@ -72,7 +72,10 @@ export const AccountDialogOverlay = memo(function AccountDialogOverlay({
|
||||
onMouseDown={(event) => event.stopPropagation()}
|
||||
>
|
||||
<header className="account-modal-header">
|
||||
<h2 id="account-dialog-title">{model.title}</h2>
|
||||
<div className="account-modal-heading">
|
||||
{isSettingsDialog ? <span className="account-modal-eyebrow">设置</span> : null}
|
||||
<h2 id="account-dialog-title">{model.title}</h2>
|
||||
</div>
|
||||
<button
|
||||
className="dialog-close"
|
||||
ref={model.closeButtonRef}
|
||||
@@ -97,7 +100,6 @@ export const AccountDialogOverlay = memo(function AccountDialogOverlay({
|
||||
>
|
||||
<Icon aria-hidden="true" />
|
||||
<span>{label}</span>
|
||||
<ChevronRight aria-hidden="true" />
|
||||
</button>
|
||||
))}
|
||||
</nav>
|
||||
|
||||
@@ -323,24 +323,26 @@ export function ChartLibraryPanel({
|
||||
<div className="chart-library-group">
|
||||
<div className="chart-library-group-heading">
|
||||
<div><b>其他人</b><small>亲友、伴侣或客户资料</small></div>
|
||||
<span className="chart-library-count">{otherCharts.length}</span>
|
||||
<div className="chart-library-group-heading-actions">
|
||||
<span className="chart-library-count">{otherCharts.length}</span>
|
||||
<button
|
||||
className="button-secondary chart-library-add chart-library-add--heading"
|
||||
type="button"
|
||||
onClick={() => {
|
||||
setOtherProfileDraft(emptyProfile);
|
||||
setOtherChartRelationship("other");
|
||||
setEditingChartId(null);
|
||||
setAccountError("");
|
||||
setProfileNotice("");
|
||||
setView({ kind: "add" });
|
||||
}}
|
||||
>
|
||||
添加其他人
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
{otherCharts.length === 0 && <p className="empty-library-copy">还没有其他星盘,先添加一份资料。</p>}
|
||||
{otherCharts.map(listRow)}
|
||||
<button
|
||||
className="button-secondary chart-library-add"
|
||||
type="button"
|
||||
onClick={() => {
|
||||
setOtherProfileDraft(emptyProfile);
|
||||
setOtherChartRelationship("other");
|
||||
setEditingChartId(null);
|
||||
setAccountError("");
|
||||
setProfileNotice("");
|
||||
setView({ kind: "add" });
|
||||
}}
|
||||
>
|
||||
添加其他人
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
@@ -27,11 +27,11 @@ export function ProfilePanel({
|
||||
{account.avatar && (
|
||||
<section className="sheet-section avatar-section" aria-labelledby="avatar-section-title">
|
||||
<div className="section-heading">
|
||||
<b id="avatar-section-title">头像</b>
|
||||
<small>Beam 形象由随机种子生成,刷新和换设备后保持一致</small>
|
||||
<b id="avatar-section-title">账户头像</b>
|
||||
<small>Beam 形象会在刷新和换设备后保持一致</small>
|
||||
</div>
|
||||
<div className="avatar-editor">
|
||||
<UserAvatar avatar={account.avatar} size={48} label="当前头像预览" />
|
||||
<UserAvatar avatar={account.avatar} size={56} label="当前头像预览" />
|
||||
<div className="avatar-editor-controls">
|
||||
<div className="avatar-palette-list" role="radiogroup" aria-label="头像配色">
|
||||
{beamAvatarPalettes.map((palette, index) => (
|
||||
|
||||
@@ -127,6 +127,15 @@ test("the settings pane menu separates hover from current without an accent bar"
|
||||
for (const rule of hoverRules) assert.ok(!rule.includes('[aria-current="page"]'), `current and hover must be separate rules: ${rule}`);
|
||||
});
|
||||
|
||||
test("settings navigation uses two columns without a misleading chevron", () => {
|
||||
const overlay = readFileSync(new URL("../src/components/account-dialog-overlay.tsx", import.meta.url), "utf8");
|
||||
const styles = readFileSync(new URL("../src/app/globals.css", import.meta.url), "utf8");
|
||||
assert.doesNotMatch(overlay, /<ChevronRight/);
|
||||
assert.match(cssDeclarations(".settings-dialog-nav-item", styles), /grid-template-columns:\s*18px\s+minmax\(0,\s*1fr\)/);
|
||||
assert.match(cssDeclarations(".settings-dialog-nav", styles), /border-right:\s*1px\s+solid\s+var\(--color-border\)/);
|
||||
assert.match(cssDeclarations(".settings-dialog-content", styles), /padding:\s*0\s+var\(--space-8\)\s+var\(--space-8\)/);
|
||||
});
|
||||
|
||||
test("form panes get a reading-width cap, list panes stay full-bleed", () => {
|
||||
// T8.1 / E13: the 880px dialog leaves ~690px of content, which pulls a one-column
|
||||
// form apart. The cap is applied per pane, inside the content box.
|
||||
@@ -149,7 +158,8 @@ test("form panes get a reading-width cap, list panes stay full-bleed", () => {
|
||||
}
|
||||
|
||||
const cap = cssDeclarations(".settings-dialog-content--form > *", styles);
|
||||
assert.match(cap, /max-width:\s*4[2-6]\dpx/, "cap belongs in the 420-460px reading-width band");
|
||||
// 原值:420–460px;新值:560px;原因:右侧内容区增加内边距后,资料表单仍需使用舒适的阅读宽度,避免右侧留下过大的空白。
|
||||
assert.match(cap, /max-width:\s*560px/, "cap uses the wider 560px reading width");
|
||||
assert.match(cap, /margin-right:\s*auto/, "capped content is left aligned");
|
||||
});
|
||||
|
||||
|
||||
@@ -110,6 +110,8 @@ test("the chart library starts as a list and only renders a form after a view ch
|
||||
assert.match(charts, /useState<ChartLibraryView>\(CHART_LIBRARY_LIST_VIEW\)/);
|
||||
assert.doesNotMatch(list, /ChartProfileForm|<form/);
|
||||
assert.match(list, /添加其他人/);
|
||||
// 原值:仅断言列表中存在“添加其他人”;新值:同步锁定标题操作区;原因:入口从列表末尾移到“其他人”分组标题,长列表中位置仍稳定。
|
||||
assert.match(charts, /chart-library-group-heading-actions[\s\S]{0,900}添加其他人/);
|
||||
assert.match(charts, /setView\(\{ kind: "add" \}\)/);
|
||||
assert.match(charts, /setView\(CHART_LIBRARY_LIST_VIEW\)/);
|
||||
});
|
||||
|
||||
@@ -8,7 +8,8 @@ test("the profile pane keeps avatar palettes and drops the duplicate chart-libra
|
||||
assert.doesNotMatch(panel, /管理星盘资料/);
|
||||
assert.match(panel, /role="radiogroup"/);
|
||||
assert.match(panel, /换一个形象/);
|
||||
assert.match(panel, /size=\{48\}/);
|
||||
// 原值:48px;新值:56px;原因:设置面板的头像预览需要更清晰的视觉锚点,并与账户身份区留出稳定间距。
|
||||
assert.match(panel, /size=\{56\}/);
|
||||
assert.match(panel, /账户信息/);
|
||||
assert.match(panel, /登录邮箱/);
|
||||
});
|
||||
|
||||
@@ -327,6 +327,58 @@ def cache_key(run_id: str, sample: Mapping[str, Any]) -> str:
|
||||
return f"{run_id}:{sample['id']}:{digest}"
|
||||
|
||||
|
||||
NONE_CONFUSION_LABELS = (
|
||||
"provide_new_evidence",
|
||||
"stop_rectification",
|
||||
"ask_about_result",
|
||||
"unclear",
|
||||
"answer_current_focus",
|
||||
)
|
||||
|
||||
|
||||
def attach_from_cache(
|
||||
samples: Sequence[dict[str, Any]],
|
||||
cache: Mapping[str, Any],
|
||||
run_ids: Sequence[str],
|
||||
) -> dict[str, int]:
|
||||
missing = {rid: 0 for rid in run_ids}
|
||||
for sample in samples:
|
||||
for rid in run_ids:
|
||||
key = cache_key(rid, sample)
|
||||
if key in cache:
|
||||
sample[rid] = cache[key]
|
||||
else:
|
||||
missing[rid] += 1
|
||||
return missing
|
||||
|
||||
|
||||
def strip_confidence(metrics: Mapping[str, Any] | None) -> dict[str, Any] | None:
|
||||
if metrics is None:
|
||||
return None
|
||||
out = dict(metrics)
|
||||
out["high_conf_error_rate"] = None
|
||||
out["low_conf_coverage"] = None
|
||||
out["low_conf_recall"] = None
|
||||
return out
|
||||
|
||||
|
||||
def confusion_counts(
|
||||
rows: Sequence[Mapping[str, Any]],
|
||||
pred_key: str,
|
||||
labels: Sequence[str] = NONE_CONFUSION_LABELS,
|
||||
) -> dict[str, Any]:
|
||||
counts = {gold: {pred: 0 for pred in labels} for gold in labels}
|
||||
other = 0
|
||||
for row in rows:
|
||||
gold = (row.get("gold") or {}).get("intent")
|
||||
pred = (row.get(pred_key) or {}).get("intent")
|
||||
if gold in counts and pred in counts[gold]:
|
||||
counts[gold][pred] += 1
|
||||
else:
|
||||
other += 1
|
||||
return {"labels": list(labels), "counts": counts, "other": other, "n": len(rows)}
|
||||
|
||||
|
||||
def run_batch(
|
||||
samples: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
@@ -471,8 +523,8 @@ def write_markdown(report: Mapping[str, Any]) -> None:
|
||||
lines.append("")
|
||||
lines.append("同一样本上 Jev vs 现行(相对门槛用这一表):")
|
||||
lines.append("")
|
||||
lines.append("| 层 | n | Jev intent | 现行 intent | 差(Jev−现行) | 门槛现行−3pp |")
|
||||
lines.append("| --- | ---: | ---: | ---: | ---: | ---: |")
|
||||
lines.append("| 层 | n | Jev intent | 现行 intent | 差(Jev−现行) | 门槛现行−3pp | 判定 |")
|
||||
lines.append("| --- | ---: | ---: | ---: | ---: | ---: | --- |")
|
||||
for layer in ("choice", "collect", "none"):
|
||||
cell = paired.get(layer) or {}
|
||||
def pct(value: float | None) -> str:
|
||||
@@ -481,8 +533,16 @@ def write_markdown(report: Mapping[str, Any]) -> None:
|
||||
cur = cell.get("current_intent")
|
||||
delta = None if jev is None or cur is None else jev - cur
|
||||
gate = None if cur is None else cur - 0.03
|
||||
if layer == "collect":
|
||||
verdict = "不可判(复核与对照同源)"
|
||||
elif jev is None or cur is None:
|
||||
verdict = "—"
|
||||
elif round(jev * 100, 1) >= round((cur - 0.03) * 100, 1):
|
||||
verdict = "过"
|
||||
else:
|
||||
verdict = "未过"
|
||||
lines.append(
|
||||
f"| {layer} | {cell.get('n', 0)} | {pct(jev)} | {pct(cur)} | {pct(delta)} | {pct(gate)} |"
|
||||
f"| {layer} | {cell.get('n', 0)} | {pct(jev)} | {pct(cur)} | {pct(delta)} | {pct(gate)} | {verdict} |"
|
||||
)
|
||||
else:
|
||||
lines.append(report["meta"].get("current_model_note") or "未跑现行模型。")
|
||||
@@ -492,6 +552,65 @@ def write_markdown(report: Mapping[str, Any]) -> None:
|
||||
"",
|
||||
report["metrics"]["representativeness"]["note"],
|
||||
"",
|
||||
]
|
||||
b_jev = report["metrics"].get("source_b")
|
||||
b_jev_layer = report["metrics"].get("source_b_by_layer") or {}
|
||||
b_cur = report["metrics"].get("current_source_b")
|
||||
b_cur_layer = report["metrics"].get("current_source_b_by_layer") or {}
|
||||
b_cons = report["metrics"].get("source_b_jev_self_consistency") or {}
|
||||
if b_jev:
|
||||
def pct(value: float | None) -> str:
|
||||
return "—" if value is None else f"{value:.1%}"
|
||||
lines += [
|
||||
"来源 B 是真人 + 人工标注 + 线上模型三者齐备的唯一一组。现行无置信度,置信度三列为空。",
|
||||
"",
|
||||
"| 范围 | n | Jev intent | 现行 intent | Jev 高置信错误 | Jev 低置信召回 | Jev 自洽 |",
|
||||
"| --- | ---: | ---: | ---: | ---: | ---: | ---: |",
|
||||
]
|
||||
rows_spec = [("全集", b_jev, b_cur, b_cons.get("all"))]
|
||||
for layer in ("choice", "collect", "none"):
|
||||
rows_spec.append((
|
||||
layer,
|
||||
b_jev_layer.get(layer) or {},
|
||||
b_cur_layer.get(layer) or {},
|
||||
b_cons.get(layer),
|
||||
))
|
||||
for name, jev_m, cur_m, cons in rows_spec:
|
||||
lines.append(
|
||||
f"| {name} | {jev_m.get('n', 0)} | {pct(jev_m.get('intent_acc'))} | {pct((cur_m or {}).get('intent_acc'))} | "
|
||||
f"{pct(jev_m.get('high_conf_error_rate'))} | {pct(jev_m.get('low_conf_recall'))} | {pct(cons)} |"
|
||||
)
|
||||
lines.append("")
|
||||
none_cur = (b_cur_layer.get("none") or {}).get("intent_acc")
|
||||
none_jev = (b_jev_layer.get("none") or {}).get("intent_acc")
|
||||
if none_cur is not None and none_jev is not None:
|
||||
lines.append(
|
||||
f"无焦点层现行 intent {none_cur:.1%}、Jev {none_jev:.1%}。"
|
||||
"现行更低,说明 78.8% 主要是这 33 条本身难,不是单 Jev 不行。"
|
||||
)
|
||||
lines.append("")
|
||||
confusion = report["metrics"].get("source_b_none_confusion") or {}
|
||||
if confusion:
|
||||
lines += [
|
||||
"无焦点层 gold × 预测混淆计数(只有计数,无原文)。预测出现 `answer_current_focus` 是因为模型把无焦点句当成在回答采集题。",
|
||||
"",
|
||||
]
|
||||
for title, key in (("gold × Jev", "jev"), ("gold × 现行", "current")):
|
||||
table = confusion.get(key) or {}
|
||||
labels = table.get("labels") or list(NONE_CONFUSION_LABELS)
|
||||
counts = table.get("counts") or {}
|
||||
lines.append(f"### {title}(n={table.get('n', 0)})")
|
||||
lines.append("")
|
||||
header = "| gold \\ pred | " + " | ".join(labels) + " |"
|
||||
sep = "| --- | " + " | ".join("---:" for _ in labels) + " |"
|
||||
lines.append(header)
|
||||
lines.append(sep)
|
||||
for gold in labels:
|
||||
row_counts = counts.get(gold) or {}
|
||||
cells = " | ".join(str(row_counts.get(pred, 0)) for pred in labels)
|
||||
lines.append(f"| {gold} | {cells} |")
|
||||
lines.append("")
|
||||
lines += [
|
||||
"## 置信度–准确率曲线与 θ",
|
||||
"",
|
||||
f"推荐 θ = {report['metrics']['theta']['recommended_theta']}。点:",
|
||||
@@ -530,6 +649,25 @@ def write_markdown(report: Mapping[str, Any]) -> None:
|
||||
lost = row["lost"].replace("|", "\\|") if row["lost"] else "—"
|
||||
lines.append(f"| {row['production']} | {row['jev']} | {lost} |")
|
||||
lines += [
|
||||
"",
|
||||
"## 限制",
|
||||
"",
|
||||
"1. **复核 ≈ 生产提示,且复核模型 = 对照模型。** `REVIEW_RUBRIC` 与生产 `COLLECT_INSTRUCTIONS` 逐句对应;生成 / 复核 / 对照都是 `deepseek-flash`。进入测试集的 900 条是「Flash 用近生产提示能答对目标标签」的那 900 条,被剔的 36 条恰是 Flash 不同意的。因此来源 C 上现行 97.5% / 98.8% / 94.0% 是构造出来的上界。采集层「Jev 92.2% 未过相对门槛 94.5%」**不可当作 Jev 输给现行的证据**。",
|
||||
"2. **采集层 intent 错例的 gold 有争议。** 来源 C 采集层 Jev intent 错例 31 条:`provide_new_evidence → answer_current_focus` 17、`unclear → answer_current_focus` 9、`stop → unclear` 4、`ask → unclear` 1。17 条 pne 几乎全是「另外 2019 年我换工作搬了家」句式,若干尾句落在当前题域,按生产提示可读成 `answer_current_focus + unsure`。9 条 unclear(「一时半会儿真捋不明白」)按「记不清 → unsure」也读得通。两类合计 ≥ 20 条,占该层 intent 错例约 2/3。gold 来自「生成目标 + Flash 复核同意」,不等于人工真值。改写示例:",
|
||||
" - 「另外 2019 年换过工作,感情那会儿真没细想。」gold=provide_new_evidence;可读成在回答感情采集题。",
|
||||
" - 「另外 2019 年搬了家,工作那摊子反而没顾上细想。」gold=provide_new_evidence;可读成在回答工作采集题。",
|
||||
" - 「这事我一时真说不上来。」gold=unclear;按生产提示是 unsure。",
|
||||
"3. **语料仍不像真人。** 修复单 1 只把长度和人设写成硬红线(已过)。原单还要求按来源 B 的标点 / 语气词比例约束,未进红线:",
|
||||
"",
|
||||
"| 指标 | 来源 B(真人) | 来源 C(模拟) |",
|
||||
"| --- | ---: | ---: |",
|
||||
"| 含标点 | 13% | 98.9% |",
|
||||
"| 含语气词(吧/呢/啊/嗯/哦/额/emm) | 4% | 25.6% |",
|
||||
"| 含年份或月份 | 81% | 47.8% |",
|
||||
"| 带年份句里写「2019」 | — | 236 / 429 = 55% |",
|
||||
"| `provide_new_evidence` 里是搬家/换工作 | — | 152 / 157 |",
|
||||
"",
|
||||
"根因:生成脚本 `ALT_EVENT_HINTS` 给七个领域的「另一件事」全是搬家/换工作,年份未约束,模型收敛到「另外 2019 年搬过家」。这解释了模拟语料不代表真人的一部分,也解释了采集层错例为何长得一样。本单不修,留给产品决定是否再造一轮。",
|
||||
"",
|
||||
"## 回退",
|
||||
"",
|
||||
@@ -570,8 +708,9 @@ def decide_verdict(report: dict[str, Any]) -> dict[str, str]:
|
||||
if jev is None or cur is None:
|
||||
continue
|
||||
ok = round(jev * 100, 1) >= round((cur - 0.03) * 100, 1)
|
||||
mark = "不可判(复核与对照同源)" if layer == "collect" else ("过" if ok else "未过")
|
||||
bits.append(
|
||||
f"{layer} Jev {jev:.1%} vs 现行 {cur:.1%}(门槛 {cur-0.03:.1%},{'过' if ok else '未过'})"
|
||||
f"{layer} Jev {jev:.1%} vs 现行 {cur:.1%}(门槛 {cur-0.03:.1%},{mark})"
|
||||
)
|
||||
if bits:
|
||||
relative_note = " 相对 −3pp(同一样本):" + ";".join(bits) + "。"
|
||||
@@ -693,11 +832,12 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
parser.add_argument("--sample-fraction", type=float, default=1.0)
|
||||
parser.add_argument("--current-second-fraction", type=float, default=1.0 / 3)
|
||||
parser.add_argument("--sample-seed", type=int, default=20260919)
|
||||
parser.add_argument("--offline", action="store_true", help="aggregate from cache only, no API calls")
|
||||
args = parser.parse_args(argv)
|
||||
if not args.current_only and not os.environ.get("TYPESAFE_API_KEY"):
|
||||
if not args.offline and not args.current_only and not os.environ.get("TYPESAFE_API_KEY"):
|
||||
print("TYPESAFE_API_KEY missing", file=sys.stderr)
|
||||
return 2
|
||||
if args.current_only and not os.environ.get("DEEPSEEK_API_KEY"):
|
||||
if not args.offline and args.current_only and not os.environ.get("DEEPSEEK_API_KEY"):
|
||||
print("DEEPSEEK_API_KEY missing", file=sys.stderr)
|
||||
return 2
|
||||
synthetic = load_jsonl(SAMPLES_DIR / "synthetic.jsonl")
|
||||
@@ -715,8 +855,13 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
_merge_report_preds(samples)
|
||||
cache = load_cache()
|
||||
current_sample: list[dict[str, Any]] = []
|
||||
second_current: list[dict[str, Any]] = []
|
||||
if args.offline:
|
||||
miss_c = attach_from_cache(samples, cache, ("jev_1", "jev_2", "current_1", "current_2"))
|
||||
miss_b = attach_from_cache(source_b, cache, ("jev_1", "jev_2", "current_1"))
|
||||
print(json.dumps({"offline_missing": {"samples": miss_c, "source_b": miss_b}}, ensure_ascii=False), flush=True)
|
||||
try:
|
||||
if not args.current_only:
|
||||
if not args.offline and not args.current_only:
|
||||
run_batch(samples, workers=args.workers, run_id="jev_1", cache=cache)
|
||||
save_cache(cache)
|
||||
second = samples
|
||||
@@ -734,7 +879,7 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
run_batch(source_b, workers=args.workers, run_id="jev_2", cache=cache)
|
||||
save_cache(cache)
|
||||
source_c_rows = [row for row in samples if row.get("source") == "C"]
|
||||
if args.current_only or os.environ.get("DEEPSEEK_API_KEY"):
|
||||
if not args.offline and (args.current_only or os.environ.get("DEEPSEEK_API_KEY")):
|
||||
if args.sample_fraction < 1:
|
||||
current_sample = stratified_sample(
|
||||
source_c_rows, fraction=args.sample_fraction, seed=args.sample_seed,
|
||||
@@ -802,6 +947,38 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
layer: layer_metrics([row for row in source_b if row.get("layer") == layer], pred_key="jev_1")
|
||||
for layer in ("choice", "collect", "none")
|
||||
} if source_b else {}
|
||||
current_source_b = strip_confidence(
|
||||
layer_metrics(source_b, pred_key="current_1") if source_b and any(row.get("current_1") for row in source_b) else None
|
||||
)
|
||||
current_source_b_by_layer = {
|
||||
layer: strip_confidence(layer_metrics(
|
||||
[row for row in source_b if row.get("layer") == layer and row.get("current_1")],
|
||||
pred_key="current_1",
|
||||
))
|
||||
for layer in ("choice", "collect", "none")
|
||||
} if source_b and any(row.get("current_1") for row in source_b) else {}
|
||||
source_b_jev_self_consistency = (
|
||||
{
|
||||
"all": self_consistency(source_b, "jev_1", "jev_2"),
|
||||
**{
|
||||
layer: self_consistency(
|
||||
[row for row in source_b if row.get("layer") == layer],
|
||||
"jev_1",
|
||||
"jev_2",
|
||||
)
|
||||
for layer in ("choice", "collect", "none")
|
||||
},
|
||||
}
|
||||
if source_b else {}
|
||||
)
|
||||
none_b = [row for row in source_b if row.get("layer") == "none"]
|
||||
source_b_none_confusion = (
|
||||
{
|
||||
"jev": confusion_counts(none_b, "jev_1"),
|
||||
"current": confusion_counts(none_b, "current_1"),
|
||||
}
|
||||
if none_b else {}
|
||||
)
|
||||
represent_fail = False
|
||||
represent_layers: dict[str, Any] = {}
|
||||
if not source_b or len(source_b) < 30:
|
||||
@@ -904,6 +1081,10 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
"source_a": source_a_metrics,
|
||||
"source_b": source_b_metrics,
|
||||
"source_b_by_layer": source_b_by_layer,
|
||||
"current_source_b": current_source_b,
|
||||
"current_source_b_by_layer": current_source_b_by_layer,
|
||||
"source_b_jev_self_consistency": source_b_jev_self_consistency,
|
||||
"source_b_none_confusion": source_b_none_confusion,
|
||||
"representativeness": {
|
||||
"note": represent_note,
|
||||
"fail": represent_fail,
|
||||
@@ -933,6 +1114,11 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
} for k, v in jev_metrics.items()},
|
||||
"self_consistency": jev_cons,
|
||||
"source_b_n": len(source_b),
|
||||
"current_source_b": {
|
||||
"n": (current_source_b or {}).get("n"),
|
||||
"intent": (current_source_b or {}).get("intent_acc"),
|
||||
},
|
||||
"source_b_none_confusion_n": (source_b_none_confusion.get("jev") or {}).get("n"),
|
||||
}, ensure_ascii=False, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
@@ -132,6 +132,16 @@ class JevIntentCorpusTests(unittest.TestCase):
|
||||
for name in names:
|
||||
self.assertNotIn(name, message)
|
||||
|
||||
def test_report_json_includes_current_source_b(self) -> None:
|
||||
path = ROOT / "docs" / "research" / "jev_intent_2026_09_19.json"
|
||||
if not path.is_file():
|
||||
self.skipTest("report json not generated yet")
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
current = (payload.get("metrics") or {}).get("current_source_b")
|
||||
self.assertIsNotNone(current)
|
||||
self.assertEqual(current["n"], payload["meta"]["source_b_n"])
|
||||
self.assertIn("source_b_none_confusion", payload["metrics"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user