/** * Biography backtest for the ordinary (natal) consultation answer * (TASK-consult-no-presupposition-and-backtest-20261001 T4, D5). * * Runs the product's real consultation agent against public Rodden AA charts * and writes each answer for a human / Claude to score against * docs/research/consult_biography_backtest_rubric.json. Nothing here copies * prompt text: the system prompt, the skill method block, the tools, the * evidence card, the methodology checklist, the user-turn shape and the answer * stream handling all come from the product modules, so a run always tests the * prompts on the current checkout. * * What is real Where it comes from * system prompt + bound skill method getJyotishAgent (src/mastra/index.ts) * tool loop, prepareStep, settings Mastra Agent.stream + consultationNatalPrepareStep * + consultationGenerationSettings (route.ts natalStreamOptions) * tool result (contract, card, createConsultationTools -> toModelEvidenceView * methodology checklist) (engine call replaced by the golden workflow) * evidence lookup tool the real read-consultation-evidence * user turn consultationUserTurnContent + natalUserTurnShape * answer extraction, pass 4, retries, streamAgentResponse (stepScopedAnswer, pass4Mode, * length continuation retry / retryForAnswer / continueAfterLength as route.ts) * * Gaps against app/api/consult/route.ts are listed in GAPS below and in * docs/testing/consult-affliction-backtest-20261001.md. * * Usage (from frontend/, Node 22): * DS_KEY=... ./node_modules/.bin/tsx scripts/research/consult-biography-backtest.mts \ * [--out ] [--repeats 2] [--figures a,b] [--domains parents,marriage,health,career] \ * [--model deepseek-flash] [--base-url https://api.deepseek.com] [--concurrency 4] [--dump-cards] * [--correction barack_obama,marilyn_monroe] [--correction-text 我爸其实没怎么管过我] * [--correction-domain career] [--correction-texts "zinedine_zidane=我是足球运动员|barack_obama=我是政治家"] * Several follow-up domains in one run: --correction-domain parents,career with * domain-qualified entries, e.g. --correction "parents:barack_obama,career:zinedine_zidane" * --correction-texts "career:zinedine_zidane=我是足球运动员" (an unqualified entry or text * applies to every listed domain). * * --dump-cards writes the model-visible tool result per figure x domain and * calls no model. --correction adds, for each listed figure's parents answer, * one follow-up turn in the same session (history = that question and answer) * with --correction-text (T3 / D4 of the task brief); those answers are written * as parents-correction-run.md and listed in the summary for reading. * --recheck re-runs the deterministic checks of an * earlier run against the current rubric and rewrites its summary. The key is read from DS_KEY (or BACKTEST_MODEL_KEY) only and * is never written anywhere. Output defaults to /scratch/ (gitignored); * only the summary may be quoted into docs. */ import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; import { dirname, join, resolve } from "node:path"; import { fileURLToPath } from "node:url"; import { getJyotishAgent } from "../../src/mastra/index.ts"; import { AGENT_MAX_STEPS, consultationContinueGenerationSettings, consultationGenerationSettings, consultationNatalPrepareStep, consultationStepBudgetReceipt, createConsultationAgentContext, createConsultationRuntimeState, publicConsultationRuntimeSteps, type ConsultationRuntimeState, } from "../../src/mastra/consultation-tools.ts"; import { consultationWorkflowResponseSchema, minuteSensitiveThemesFromBirthTimeSensitivity, } from "../../src/mastra/consultation-workflow.ts"; import type { ResolvedLanguageModel } from "../../src/mastra/model.ts"; import { createConsultationPlan } from "../../src/lib/consultation-plan.ts"; import { consultationUserTurnContent } from "../../src/lib/consultation-session-history.ts"; import { consultationContinueMessages, natalUserTurnShape } from "../../src/lib/consultation-thinking-plan.ts"; import { streamAgentResponse } from "../../src/lib/stream-agent-response.ts"; import type { ConsultationDomain } from "../../src/lib/consultation-domain-registry.ts"; import type { ServerChartConsultation } from "../../src/lib/consultation-route-service.ts"; const HERE = dirname(fileURLToPath(import.meta.url)); const FRONTEND = resolve(HERE, "../.."); const ROOT = resolve(FRONTEND, ".."); const GOLDEN = join(FRONTEND, "tests/fixtures/consult-biography-backtest-golden.json"); const RUBRIC = join(ROOT, "docs/research/consult_biography_backtest_rubric.json"); /** Where this harness differs from the live route; repeated in the summary. */ export const GAPS = [ "引擎调用换成 golden(同 capture 路径、raman 岁差、mean 交点、参考日 2026-09-27);产品用用户档案的岁差设置。", "用户回合不带「用户称呼」:真实路由会带档案名,这里故意不给名人名字,免得模型凭记忆答生平。", "natalToolInstruction 一句与 currentTimeContext 是 route.ts 内部函数,未导出,脚本里按原文复刻(见 routeToolInstruction / routeCurrentTime)。", "没有内容审核(moderateOutput)、计费、标题旁路事件、历史摘要;主回测单轮、无历史;--correction 纠正追问轮的历史只有上一问一答。", "性别一律按档案未填(性别未知)处理。", "模型经 Mastra 的 OpenAI 兼容通道直连 DeepSeek(model id 由参数给),不经产品数据库的模型目录。", ] as const; // route.ts (natalToolInstruction, entrypoint undefined = ordinary chat). Not exported there. const routeToolInstruction = "如需新的个人星盘结论,必须调用服务器绑定的排盘工具。"; // route.ts currentTimeContext. Not exported there. function routeCurrentTime(now: Date) { const chinaTime = new Date(now.getTime() + 8 * 60 * 60 * 1000).toISOString().replace("T", " ").slice(0, 19); return `服务端当前时间(权威):${now.toISOString()};中国标准时间(UTC+8):${chinaTime}。涉及“现在、今天、今年、未来几个月”等相对时间时,以此为准。`; } // The golden is computed for this reference date; the request clock matches it. const REQUEST_TIME = new Date("2026-09-27T04:00:00.000Z"); type Json = Record; type GoldenRoute = { domain: string; engine_route: string; question: string }; type GoldenFigure = { id: string; label: string; rodden_rating: string; shared_chart: Json; routes: Record; }; type Golden = { routes: GoldenRoute[]; figures: GoldenFigure[] }; type RubricFigure = { id: string; domains: Record; }; function parseArgs(argv: readonly string[]) { const value = (name: string) => { const index = argv.indexOf(`--${name}`); return index >= 0 ? argv[index + 1] : undefined; }; const stamp = new Date().toISOString().replace(/[:.]/g, "-"); return { out: resolve(value("out") ?? join(ROOT, "scratch/consult-biography-backtest", stamp)), repeats: Math.max(1, Number(value("repeats") ?? 2)), figures: value("figures")?.split(",").filter(Boolean) ?? null, domains: value("domains")?.split(",").filter(Boolean) ?? ["parents", "marriage", "health", "career"], model: value("model") ?? "deepseek-flash", baseUrl: value("base-url") ?? "https://api.deepseek.com", concurrency: Math.max(1, Number(value("concurrency") ?? 4)), dumpCards: argv.includes("--dump-cards"), recheck: value("recheck"), correction: value("correction")?.split(",").filter(Boolean) ?? [], correctionText: value("correction-text") ?? "我爸其实没怎么管过我", // Follow-up domain (default parents) and optional per-figure texts // "figure=text|figure=text" (career follow-up: the user states the real job). correctionDomains: (value("correction-domain") ?? "parents").split(",").filter(Boolean), correctionTexts: Object.fromEntries((value("correction-texts") ?? "").split("|").filter(Boolean).map((pair) => { const at = pair.indexOf("="); return [pair.slice(0, at), pair.slice(at + 1)]; })) as Record, }; } /** One route's engine response, with the figure's shared chart put back exactly as captured. */ function routeWorkflow(figure: GoldenFigure, domain: string): Json { const route = figure.routes[domain]; if (!route) throw new Error(`golden has no ${domain} route for ${figure.id}`); const { chart: chartOverrides, chart_modules: moduleOverrides, ...rest } = structuredClone(route); const shared = structuredClone(figure.shared_chart); const chart = { ...shared, ...(chartOverrides ?? {}), modules: { ...(shared.modules as Json), ...(moduleOverrides ?? {}) }, }; return { ...rest, chart }; } // The engine route each product domain runs as (consultation-domain-registry): // parents / children / family all run the engine's family route. const GOLDEN_ROUTE_FOR_DOMAIN: Record = { parents: "parents", family: "parents", children: "parents", marriage: "marriage", health: "health", career: "career", }; function stubRunWorkflow(figure: GoldenFigure, executed: string[]) { return async (input: { theme: string }) => { const key = GOLDEN_ROUTE_FOR_DOMAIN[input.theme]; if (!key) throw new Error(`backtest golden has no engine route for domain ${input.theme}`); executed.push(input.theme); // runConsultationWorkflow: schema parse, then minute-sensitive themes on the policy. const data = consultationWorkflowResponseSchema.parse(routeWorkflow(figure, key)); const consumer = data.consumer_context as Json; return { ...data, consumer_context: { ...consumer, answer_policy: { ...(consumer.answer_policy as Json), minute_sensitive_themes: minuteSensitiveThemesFromBirthTimeSensitivity(data.birth_time_sensitivity), }, }, } as unknown as typeof data; }; } function serverChart(figure: GoldenFigure): ServerChartConsultation { const birth = (figure.shared_chart.birth ?? {}) as Json; const route = routeWorkflow(figure, "parents"); const body = ((route.chart as Json).birth ?? birth) as Json; const num = (key: string, fallback = 0) => (typeof body[key] === "number" ? body[key] as number : fallback); // Birth fields only shape the tool input schema; the stub ignores them. const toolInput = { year: num("year", 1950), month: num("month", 1), day: num("day", 1), hour: num("hour"), minute: num("minute"), city: "public chart", lat: num("lat", num("latitude")), lon: num("lon", num("longitude")), tz: num("tz", num("timezone")), ayanamsa: "raman", declared_accuracy: "minute", time_source: "birth_record", } as unknown as ServerChartConsultation["toolInput"]; return { name: "backtest", toolInput, truth: { birthDate: "1950-01-01", reportedBirthTime: null, activeBirthTime: null, selectedTimeKind: "reported", birthTimeSource: "birth_record", birthTimeStatus: "reported", placeLabel: "public chart", placeCodes: { countryCode: null, provinceCode: null, cityCode: null, districtCode: null }, placeId: null, placeType: null, placeProvider: null, timezoneId: null, timezoneSource: null, latitude: toolInput.lat, longitude: toolInput.lon, timezoneOffset: toolInput.tz, }, gender: null, } as ServerChartConsultation; } function model(args: ReturnType, apiKey: string): ResolvedLanguageModel { return { id: `backtest-${args.model}`, label: args.model, description: "biography backtest", creditCost: 0, isDefault: false, mode: "compatible", model: { providerId: "deepseek", modelId: args.model, url: args.baseUrl, apiKey } as ResolvedLanguageModel["model"], }; } /** A follow-up turn in the same session: the earlier question and answer are the history. */ type FollowUp = { priorQuestion: string; priorAnswer: string }; type Job = { figure: GoldenFigure; domain: string; question: string; run: number; followUp?: FollowUp }; type Usage = { inputTokens?: number; outputTokens?: number; reasoningTokens?: number; cachedInputTokens?: number }; function addUsage(total: Usage, usage: unknown) { const row = (usage ?? {}) as Record; for (const key of ["inputTokens", "outputTokens", "reasoningTokens", "cachedInputTokens"] as const) { const value = row[key]; if (typeof value === "number") total[key] = (total[key] ?? 0) + value; } } function agentContext(job: Job, state: ConsultationRuntimeState, executed: string[]) { return createConsultationAgentContext({ userId: "biography-backtest", sessionId: `backtest-${job.figure.id}`, requestId: `backtest-${job.figure.id}-${job.domain}-${job.run}`, consultationMode: "verified_chart", plan: createConsultationPlan({ userIntent: job.question, theme: job.domain as ConsultationDomain }), theme: job.domain as ConsultationDomain, followUpTurn: Boolean(job.followUp), serverChart: serverChart(job.figure), state, runWorkflow: stubRunWorkflow(job.figure, executed) as never, }); } function historyOf(followUp: FollowUp | undefined) { return followUp ? [ { role: "user" as const, text: followUp.priorQuestion }, { role: "assistant" as const, text: followUp.priorAnswer }, ] : []; } function userTurn(question: string, followUp?: FollowUp) { return consultationUserTurnContent({ currentTime: routeCurrentTime(REQUEST_TIME), // natalUserTurnShape reads the stored history, as route.ts does: a prior answer -> follow-up shape. instruction: `${routeToolInstruction}${natalUserTurnShape({ history: historyOf(followUp) })}`, question, }); } async function runJob(job: Job, resolved: ResolvedLanguageModel) { const state = createConsultationRuntimeState(); const executed: string[] = []; const ctx = agentContext(job, state, executed); const agent = getJyotishAgent(resolved, ctx); const runId = ctx.requestId; // route.ts consultationBaseMessages: stored history as plain messages, then this turn. const baseMessages = [ ...historyOf(job.followUp).map((message) => (message.role === "user" ? { role: "user" as const, content: message.text } : { role: "assistant" as const, content: message.text })), { role: "user" as const, content: userTurn(job.question, job.followUp) }, ]; const streamOptions = { runId, maxSteps: AGENT_MAX_STEPS, ...consultationGenerationSettings(resolved.model), }; const natalStreamOptions = { ...streamOptions, prepareStep: consultationNatalPrepareStep }; const usages: Promise[] = []; const startedAt = Date.now(); let output: string | null = null; let failure: string | null = null; const first = await agent.stream(baseMessages, natalStreamOptions); usages.push(first.totalUsage); const response = streamAgentResponse({ runId, requestId: runId, state, stream: first.fullStream as never, requireTool: true, retry: async () => { const retried = await agent.stream([ ...baseMessages, { role: "user" as const, content: "运行合同不完整:本次尚未取得服务器计算结果。请调用 run-jyotish-consultation 完成计算,再据此回答;不要在工具参数中添加出生资料。" }, ], natalStreamOptions); usages.push(retried.totalUsage); return retried.fullStream as never; }, retryForAnswer: async (retryHint?: string) => { const retried = await agent.stream([ ...baseMessages, { role: "user" as const, content: `服务器计算已经完成,但上一轮没有输出任何回答文本。请重新取回本次计算结果,然后直接给出回答;不要只描述过程或工具调用。${retryHint ? `\n${retryHint}` : ""}` }, ], natalStreamOptions); usages.push(retried.totalUsage); return retried.fullStream as never; }, continueAfterLength: async (text: string, evidence?: unknown) => { const continued = await agent.stream(consultationContinueMessages(baseMessages, text, evidence), { ...streamOptions, ...consultationContinueGenerationSettings(resolved.model), }); usages.push(continued.totalUsage); return continued.fullStream as never; }, stepScopedAnswer: true, pass4Mode: "verified_chart", toolStatus: () => { const status = state.workflowReceipt?.status; return status === "ready" || status === "degraded" ? status : "blocked"; }, receipt: () => ({ runId, runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: state.jyotishSkillBound, referenceReads: state.skillReferenceReadCount, methodologySections: state.methodologySectionCount, }, steps: publicConsultationRuntimeSteps(state), stepBudget: consultationStepBudgetReceipt(state), workflow: state.workflowReceipt ?? { route: job.domain, status: "blocked", preciseTiming: "blocked", missingLayers: [], domains: [job.domain as ConsultationDomain] }, techniqueTruth: state.techniqueTruth ?? "unknown", }) as never, onComplete: (text: string) => { output = text; }, onError: (error: unknown) => { failure = error instanceof Error ? error.message : String(error); }, }); // Drain the public event stream; that is what drives the run. await response.text(); const usage: Usage = {}; for (const item of await Promise.allSettled(usages)) { if (item.status === "fulfilled") addUsage(usage, item.value); } // Assigned inside the stream callbacks, which TypeScript cannot see. const settledOutput = output as string | null; const settledFailure = failure as string | null; return { output: settledOutput ?? "", failure: settledFailure, executed, seconds: (Date.now() - startedAt) / 1000, usage, cardChars: state.evidenceCardChars ?? null, visibleChars: state.modelVisibleChars ?? null, methodologySections: state.methodologySectionCount, lookups: state.evidenceLookupCallCount, finish: state.composeFinishReason ?? state.modelFinishReason ?? null, }; } // ---- deterministic checks (judgment scoring is done by reading the answers) ---- const BANNED = [ { id: "retrograde_repeat", pattern: /逆行[^。!?\n]{0,30}(反复|打回来|来回|拉回|折返|回头)/gu }, { id: "repeat_retrograde", pattern: /(反复|打回来|来回)[^。!?\n]{0,12}逆行/gu }, ]; // 他 / 她 when the card says 性别未知: only for the partner (marriage) or children. const PRONOUN = /(? answer.slice(Math.max(0, index - 14), index + 16).replace(/\s+/g, " "); const banned = BANNED.flatMap((rule) => [...answer.matchAll(rule.pattern)].map((hit) => ({ rule: rule.id, text: hit[0] }))); const pronouns = domain === "marriage" || domain === "children" ? [...answer.matchAll(PRONOUN)].map((hit) => context(hit.index ?? 0)) : []; const literalHits = mustNot.flatMap((item) => (item.literals ?? []) .filter((literal) => answer.includes(literal)) .map((literal) => ({ claim: item.claim, literal }))); return { banned, pronouns, literalHits }; } async function pool(items: readonly T[], size: number, fn: (item: T) => Promise) { const results: R[] = new Array(items.length); let next = 0; await Promise.all(Array.from({ length: Math.min(size, items.length) }, async () => { while (next < items.length) { const index = next++; results[index] = await fn(items[index]!); } })); return results; } async function dumpCards(jobs: readonly Job[], out: string) { for (const job of jobs.filter((item) => item.run === 1)) { const state = createConsultationRuntimeState(); const ctx = agentContext(job, state, []); const fake = model({ ...parseArgs([]), model: "card-dump" }, "unused"); const agent = getJyotishAgent(fake, ctx); const tools = await agent.listTools(); const tool = tools["run-jyotish-consultation"] as unknown as { execute: (input: unknown, context: unknown) => Promise }; const view = await tool.execute({ question: job.question, domains: [job.domain] }, {}); const path = join(out, "cards", `${job.figure.id}-${job.domain}.json`); mkdirSync(dirname(path), { recursive: true }); writeFileSync(path, `${JSON.stringify(view, null, 1)}\n`); console.log(`card ${path} (${JSON.stringify(view).length} chars)`); } } // The product logs every reasoning delta (consultation-budget.ts); a batch run // would drown its own progress lines in them. const productInfo = console.info.bind(console); console.info = (...items: unknown[]) => { if (typeof items[0] === "string" && items[0].startsWith("[consult-reasoning]")) return; productInfo(...items); }; type Row = { figure: string; domain: string; run: number; chars: number; seconds: number; checks: ReturnType }; function writeSummary(out: string, summary: Json & { rows: Row[] }, modelId: string) { const rows = summary.rows; summary.banned_hits = rows.reduce((sum, row) => sum + row.checks.banned.length, 0); summary.pronoun_hits = rows.reduce((sum, row) => sum + row.checks.pronouns.length, 0); summary.must_not_literal_hits = rows.reduce((sum, row) => sum + row.checks.literalHits.length, 0); writeFileSync(join(out, "summary.json"), `${JSON.stringify(summary, null, 1)}\n`); writeFileSync(join(out, "summary.md"), [ `# 名人生平回测 · ${modelId}`, "", `开始 ${summary.started_at},${rows.length} 份,墙钟 ${Number(summary.wall_seconds).toFixed(0)} s,失败 ${summary.failures} 份。`, `用量合计:${JSON.stringify(summary.usage)}`, `确定性检查:禁句 ${summary.banned_hits},性别代词(婚恋/子女)${summary.pronoun_hits},must_not 原文 ${summary.must_not_literal_hits}。判断类评分需逐份阅读。`, "", "| 名人 | 领域 | run | 字数 | 秒 | 禁句 | 代词 | must_not |", "| --- | --- | --- | --- | --- | --- | --- | --- |", ...rows.map((row) => `| ${row.figure} | ${row.domain} | ${row.run} | ${row.chars} | ${row.seconds.toFixed(0)} | ${row.checks.banned.length} | ${row.checks.pronouns.length} | ${row.checks.literalHits.length} |`), "", "## must_not 原文命中", "", ...rows.flatMap((row) => row.checks.literalHits.map((hit) => `- ${row.figure} · ${row.domain} · run${row.run}:「${hit.literal}」(${hit.claim})`)), "", "## 与真实路由的差异", "", ...GAPS.map((gap) => `- ${gap}`), "", ].join("\n")); } function recheck(out: string, rubric: { figures: RubricFigure[] }) { const summary = JSON.parse(readFileSync(join(out, "summary.json"), "utf8")) as Json & { rows: Row[]; model: string }; for (const row of summary.rows) { const text = readFileSync(join(out, row.figure, `${row.domain}-run${row.run}.md`), "utf8"); const answer = text.slice(text.indexOf("\n---\n\n") + 6).trim(); const mustNot = rubric.figures.find((item) => item.id === row.figure)?.domains[row.domain]?.must_not ?? []; row.checks = deterministicChecks(answer === "(无回答)" ? "" : answer, row.domain, mustNot); } writeSummary(out, summary, summary.model); console.log(`rechecked: ${join(out, "summary.md")}`); } async function main() { const args = parseArgs(process.argv.slice(2)); const golden = JSON.parse(readFileSync(GOLDEN, "utf8")) as Golden; const rubric = JSON.parse(readFileSync(RUBRIC, "utf8")) as { figures: RubricFigure[] }; if (args.recheck) { recheck(resolve(args.recheck), rubric); return; } const questions = new Map(golden.routes.map((route) => [route.domain, route.question])); const figures = golden.figures.filter((figure) => !args.figures || args.figures.includes(figure.id)); const jobs: Job[] = figures.flatMap((figure) => args.domains.flatMap((domain) => { const question = questions.get(domain); if (!question) throw new Error(`no golden question for domain ${domain}`); return Array.from({ length: args.repeats }, (_, index) => ({ figure, domain, question, run: index + 1 })); })); mkdirSync(args.out, { recursive: true }); if (args.dumpCards) { await dumpCards(jobs, args.out); return; } const apiKey = process.env.DS_KEY ?? process.env.BACKTEST_MODEL_KEY; if (!apiKey) throw new Error("Set DS_KEY (or BACKTEST_MODEL_KEY); without a key the backtest is an environment gap."); const resolved = model(args, apiKey); const started = Date.now(); const rows = await pool(jobs, args.concurrency, async (job) => { let result: Awaited>; try { result = await runJob(job, resolved); } catch (error) { result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null }; } const mustNot = rubric.figures.find((item) => item.id === job.figure.id)?.domains[job.domain]?.must_not ?? []; const checks = deterministicChecks(result.output, job.domain, mustNot); const file = join(args.out, job.figure.id, `${job.domain}-run${job.run}.md`); mkdirSync(dirname(file), { recursive: true }); writeFileSync(file, [ `# ${job.figure.label} · ${job.domain} · run ${job.run}`, "", `- 问题:${job.question}`, `- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次;清单段 ${result.methodologySections}`, `- 执行领域:${result.executed.join(", ") || "无"};卡 ${result.cardChars ?? "?"} 字符,工具结果 ${result.visibleChars ?? "?"} 字符`, `- 用量:${JSON.stringify(result.usage)}`, `- 确定性检查:禁句 ${checks.banned.length},性别代词 ${checks.pronouns.length},must_not 原文 ${checks.literalHits.length}`, ...(result.failure ? [`- 失败:${result.failure}`] : []), ...checks.banned.map((hit) => ` - 禁句 ${hit.rule}:「${hit.text}」`), ...checks.pronouns.map((hit) => ` - 代词:「${hit}」`), ...checks.literalHits.map((hit) => ` - must_not:「${hit.literal}」(${hit.claim})`), "", "---", "", result.output || "(无回答)", "", ].join("\n")); console.log(`${job.figure.id} ${job.domain} run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`); return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, ...result, output: undefined, checks }; }); const corrections = await pool( jobs.filter((job) => args.correctionDomains.includes(job.domain) && (args.correction.includes(job.figure.id) || args.correction.includes(`${job.domain}:${job.figure.id}`))), args.concurrency, async (prior) => { const priorFile = join(args.out, prior.figure.id, `${prior.domain}-run${prior.run}.md`); const priorText = readFileSync(priorFile, "utf8"); const priorAnswer = priorText.slice(priorText.indexOf("\n---\n\n") + 6).trim(); const question = args.correctionTexts[`${prior.domain}:${prior.figure.id}`] ?? args.correctionTexts[prior.figure.id] ?? args.correctionText; const job: Job = { ...prior, question, followUp: { priorQuestion: prior.question, priorAnswer } }; let result: Awaited>; try { result = await runJob(job, resolved); } catch (error) { result = { output: "", failure: error instanceof Error ? error.message : String(error), executed: [], seconds: 0, usage: {}, cardChars: null, visibleChars: null, methodologySections: 0, lookups: 0, finish: null }; } const file = join(args.out, job.figure.id, `${job.domain}-correction-run${job.run}.md`); writeFileSync(file, [ `# ${job.figure.label} · ${job.domain} 纠正追问 · run ${job.run}`, "", `- 上一轮:${priorFile}`, `- 追问:${job.question}`, `- 模型:${args.model};用时 ${result.seconds.toFixed(1)} s;结束:${result.finish ?? "?"};查卡 ${result.lookups} 次`, `- 用量:${JSON.stringify(result.usage)}`, ...(result.failure ? [`- 失败:${result.failure}`] : []), "", "---", "", result.output || "(无回答)", "", ].join("\n")); console.log(`${job.figure.id} ${job.domain} correction run${job.run}: ${result.output.length} chars, ${result.seconds.toFixed(0)} s${result.failure ? `, FAILED ${result.failure}` : ""}`); return { figure: job.figure.id, domain: job.domain, run: job.run, chars: result.output.length, seconds: result.seconds, failure: result.failure, usage: result.usage }; }, ); const total: Usage = {}; for (const row of [...rows, ...corrections]) addUsage(total, row.usage); const summary = { model: args.model, started_at: new Date(started).toISOString(), wall_seconds: (Date.now() - started) / 1000, jobs: rows.length, failures: rows.filter((row) => row.failure || row.chars === 0).length, usage: total, gaps: GAPS, rows, corrections, }; writeSummary(args.out, summary as unknown as Json & { rows: Row[] }, args.model); console.log(`summary: ${join(args.out, "summary.md")}`); } await main();