fix(consult): stop the consult-gate VedAstro wait and record step timing
Consult turns reuse a same-day official snapshot inside the rectification gate, or record deferred_in_consultation instead of starting another 4-second snapshot. Classification, step 0, step 1, and the time to the first answer character go into the existing usage metadata and a new admin usage column. BUG-1231, BUG-1232
This commit is contained in:
@@ -0,0 +1,199 @@
|
||||
import assert from "node:assert/strict";
|
||||
import { readFileSync } from "node:fs";
|
||||
import test from "node:test";
|
||||
|
||||
import { agentObservabilityEventSchema } from "../src/lib/agent-observability.ts";
|
||||
import { streamAgentResponse } from "../src/lib/stream-agent-response.ts";
|
||||
import {
|
||||
consultationModelStepTelemetry,
|
||||
consultationStepBudgetReceipt,
|
||||
createConsultationRuntimeState,
|
||||
} from "../src/mastra/consultation-tools.ts";
|
||||
import {
|
||||
consultationModelSteps,
|
||||
consultationUsageMetadata,
|
||||
createConsultationLatencyRecord,
|
||||
formatUsageTiming,
|
||||
noteAnswerBodyCharacter,
|
||||
observeConsultationModelChunk,
|
||||
usageTimingFromLedger,
|
||||
} from "../src/lib/consultation-step-timing.ts";
|
||||
|
||||
const SECRET = "这段正文不该出现在计时里";
|
||||
|
||||
test("simulated step results fill the timing fields and leave the answer text out", () => {
|
||||
const record = createConsultationLatencyRecord();
|
||||
observeConsultationModelChunk(record, { type: "step-start" }, 1_000);
|
||||
observeConsultationModelChunk(record, {
|
||||
type: "step-finish",
|
||||
payload: {
|
||||
output: {
|
||||
text: SECRET,
|
||||
usage: { inputTokens: 11, outputTokens: 4, reasoningTokens: 9, cachedInputTokens: 2 },
|
||||
},
|
||||
},
|
||||
}, 1_600);
|
||||
observeConsultationModelChunk(record, { type: "step-start" }, 5_000);
|
||||
noteAnswerBodyCharacter(record, 5_450);
|
||||
observeConsultationModelChunk(record, {
|
||||
type: "step-finish",
|
||||
payload: {
|
||||
output: {
|
||||
text: SECRET,
|
||||
usage: { inputTokens: 20, outputTokens: 30, reasoningTokens: 100 },
|
||||
},
|
||||
},
|
||||
}, 9_000);
|
||||
|
||||
const modelSteps = consultationModelSteps(record);
|
||||
const usageMetadata = {
|
||||
classification: {
|
||||
outcome: "consult",
|
||||
usageKnown: true,
|
||||
inputTokens: 3,
|
||||
outputTokens: 1,
|
||||
durationMs: 180,
|
||||
},
|
||||
...consultationUsageMetadata({
|
||||
modelSteps,
|
||||
answerReasoningMs: record.answerReasoningMs,
|
||||
}),
|
||||
};
|
||||
const event = agentObservabilityEventSchema.parse({
|
||||
requestId: "req-timing",
|
||||
classification: { durationMs: 180 },
|
||||
modelSteps,
|
||||
answer: { reasoning_ms: record.answerReasoningMs },
|
||||
toolCalls: [{ name: "run-jyotish-consultation", durationMs: 700, status: "completed" }],
|
||||
});
|
||||
|
||||
assert.equal(event.classification?.durationMs, 180);
|
||||
assert.deepEqual(event.modelSteps?.[0], {
|
||||
index: 0,
|
||||
durationMs: 600,
|
||||
reasoningTokens: 9,
|
||||
outputTokens: 4,
|
||||
inputTokens: 11,
|
||||
cachedInputTokens: 2,
|
||||
});
|
||||
assert.deepEqual(event.modelSteps?.[1], {
|
||||
index: 1,
|
||||
durationMs: 4_000,
|
||||
reasoningTokens: 100,
|
||||
outputTokens: 30,
|
||||
inputTokens: 20,
|
||||
cachedInputTokens: null,
|
||||
});
|
||||
assert.equal(event.answer?.reasoning_ms, 450);
|
||||
assert.equal(event.toolCalls?.[0]?.durationMs, 700);
|
||||
assert.equal(usageMetadata.classification.durationMs, 180);
|
||||
assert.equal(usageMetadata.answer.reasoning_ms, 450);
|
||||
assert.equal(JSON.stringify({ event, usageMetadata }).includes(SECRET), false);
|
||||
|
||||
const ledger = usageTimingFromLedger(
|
||||
JSON.stringify(usageMetadata.modelSteps),
|
||||
"180",
|
||||
"450",
|
||||
);
|
||||
assert.equal(ledger.modelSteps[1]?.reasoningTokens, 100);
|
||||
const shown = formatUsageTiming(ledger);
|
||||
assert.match(shown, /分类 180 ms/);
|
||||
assert.match(shown, /第 1 步 4000 ms/);
|
||||
assert.match(shown, /写到正文 450 ms/);
|
||||
assert.equal(shown.includes(SECRET), false);
|
||||
});
|
||||
|
||||
test("a missing provider usage field stays null and step 0 text is not the answer clock", () => {
|
||||
const record = createConsultationLatencyRecord();
|
||||
observeConsultationModelChunk(record, { type: "step-finish" }, 2_000);
|
||||
noteAnswerBodyCharacter(record, 2_100);
|
||||
observeConsultationModelChunk(record, {
|
||||
type: "step-finish",
|
||||
payload: { output: { usage: { outputTokens: 6 } } },
|
||||
}, 3_000);
|
||||
const steps = consultationModelSteps(record);
|
||||
assert.equal(steps[0]?.durationMs, null);
|
||||
assert.equal(steps[0]?.inputTokens, null);
|
||||
assert.equal(steps[1]?.durationMs, null);
|
||||
assert.equal(steps[1]?.outputTokens, 6);
|
||||
assert.equal(steps[1]?.reasoningTokens, null);
|
||||
assert.equal(steps[1]?.cachedInputTokens, null);
|
||||
assert.equal(record.answerReasoningMs, null);
|
||||
});
|
||||
|
||||
test("the answer stream stores step timings on the run and not in the public receipt", async () => {
|
||||
const state = createConsultationRuntimeState();
|
||||
state.consultationToolCallCount = 1;
|
||||
state.consultationToolSuccessCount = 1;
|
||||
state.consultationToolCompleted = true;
|
||||
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
|
||||
const sentence = "事业方向的判断如下。";
|
||||
async function* chunks() {
|
||||
yield { type: "step-start" };
|
||||
yield {
|
||||
type: "step-finish",
|
||||
payload: {
|
||||
stepResult: { reason: "tool-calls" },
|
||||
output: { usage: { inputTokens: 8, outputTokens: 1, reasoningTokens: 4, cachedInputTokens: 0 }, text: sentence },
|
||||
},
|
||||
};
|
||||
yield { type: "step-start" };
|
||||
yield { type: "text-delta", payload: { text: sentence } };
|
||||
yield {
|
||||
type: "step-finish",
|
||||
payload: {
|
||||
stepResult: { reason: "stop" },
|
||||
output: { usage: { reasoningTokens: 12 }, text: sentence },
|
||||
},
|
||||
};
|
||||
yield { type: "finish", payload: { stepResult: { reason: "stop" }, output: { usage: {}, steps: [{}, {}] } } };
|
||||
}
|
||||
const response = streamAgentResponse({
|
||||
runId: "run",
|
||||
requestId: "req",
|
||||
state,
|
||||
stream: chunks(),
|
||||
requireTool: true,
|
||||
toolStatus: () => "ready",
|
||||
receipt: () => ({
|
||||
runId: "run",
|
||||
runtime: "mastra-agentic" as const,
|
||||
skill: {
|
||||
name: "jyotish-vedic-astrology" as const,
|
||||
loaded: state.jyotishSkillBound,
|
||||
referenceReads: state.skillReferenceReadCount,
|
||||
methodologySections: state.methodologySectionCount,
|
||||
},
|
||||
steps: state.steps,
|
||||
stepBudget: consultationStepBudgetReceipt(state),
|
||||
workflow: state.workflowReceipt!,
|
||||
}),
|
||||
});
|
||||
const body = await response.text();
|
||||
const telemetry = consultationModelStepTelemetry(state);
|
||||
const event = agentObservabilityEventSchema.parse({ requestId: "req-stream", ...telemetry });
|
||||
assert.equal(event.modelSteps?.[0]?.inputTokens, 8);
|
||||
assert.equal(event.modelSteps?.[0]?.cachedInputTokens, 0);
|
||||
assert.equal(event.modelSteps?.[0]?.reasoningTokens, 4);
|
||||
assert.equal(event.modelSteps?.[1]?.reasoningTokens, 12);
|
||||
assert.equal(event.modelSteps?.[1]?.inputTokens, null);
|
||||
assert.equal(typeof event.answer?.reasoning_ms, "number");
|
||||
assert.equal(JSON.stringify(event).includes(sentence), false);
|
||||
assert.equal(body.includes("modelSteps"), false);
|
||||
assert.equal(body.includes("reasoning_ms"), false);
|
||||
});
|
||||
|
||||
test("the consult route and the usage list keep the same timing fields", () => {
|
||||
const route = readFileSync(new URL("../src/app/api/consult/route.ts", import.meta.url), "utf8");
|
||||
const list = readFileSync(new URL("../src/app/api/admin/usage/route.ts", import.meta.url), "utf8");
|
||||
const page = readFileSync(new URL("../src/components/admin/billing-operations-resources.tsx", import.meta.url), "utf8");
|
||||
assert.match(route, /classification: \{ durationMs: observation\.durationMs \}/);
|
||||
assert.match(route, /durationMs: classificationDurationMs/);
|
||||
assert.match(route, /consultationUsageMetadata/);
|
||||
assert.match(list, /metadata->'cache'->>'readTokens'/);
|
||||
assert.match(list, /metadata->'modelSteps'/);
|
||||
assert.match(list, /metadata->'classification'->>'durationMs'/);
|
||||
assert.match(list, /metadata->'answer'->>'reasoning_ms'/);
|
||||
assert.match(page, /分段/);
|
||||
assert.doesNotMatch(list, /alter table|create table/i);
|
||||
});
|
||||
Reference in New Issue
Block a user