Files
Jyotisha/frontend/tests/consultation-agentic-runtime.test.ts
T
Jesse_Chen 4e247c112e fix(web): keep thinking off the spoken consult and rectification answer
Enumerate evidence kinds so education cannot be proposed as a kind, and stream Chinese thinking on a separate channel that collapses when the reply arrives.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-21 21:30:55 +08:00

1481 lines
75 KiB
TypeScript

import assert from "node:assert/strict";
import test from "node:test";
import {
AGENT_MAX_STEPS,
AGENT_TIMEOUT_MS,
CONSULTATION_MAX_OUTPUT_TOKENS,
mergeConsultationAnswerPolicies,
CONSULTATION_DOMAIN_WALL_CLOCK_MS,
MAX_CONSULTATION_DOMAINS,
appendConsultationRuntimeStep,
canonicalDomainPlan,
consultationGenerationSettings,
consultationModelStepTelemetry,
consultationStepBudgetReceipt,
consultationToolFailureCode,
createConsultationRuntimeHooks,
createConsultationTools,
createConsultationRuntimeState,
domainFitsRunBudget,
executableDomainPlan,
publicConsultationRuntimeSteps,
} from "../src/mastra/consultation-tools.ts";
import {
ConsultationWorkflowError,
consultationWorkflowFailureCode,
} from "../src/mastra/consultation-workflow.ts";
import { agentExecutionReceiptSchema } from "../src/lib/consultation-agent-events.ts";
import { consultationDomainIds, consultationDomainPlanValues } from "../src/lib/consultation-domain-registry.ts";
import { getJyotishAgent } from "../src/mastra/index.ts";
import { consultationAgentPublicEventSchema, createNdjsonParser } from "../src/lib/consultation-agent-events.ts";
import { createConsultationPlan } from "../src/lib/consultation-plan.ts";
import {
collectAgentPublicEvents,
streamAgentResponse,
} from "../src/lib/stream-agent-response.ts";
const serverChart = {
name: "测试",
toolInput: { year: 1990, month: 1, day: 2, hour: 3, minute: 4, city: "台北", lat: 25.03, lon: 121.56, tz: 8 },
truth: {
birthDate: "1990-01-02", reportedBirthTime: "03:04", activeBirthTime: null,
selectedTimeKind: "reported" as const, birthTimeSource: "reported", birthTimeStatus: "reported",
placeLabel: "台北", placeCodes: { countryCode: "TW", provinceCode: null, cityCode: null, districtCode: null },
placeId: null, placeType: "city", placeProvider: "profile", latitude: 25.03, longitude: 121.56,
timezoneId: "Asia/Taipei", timezoneSource: "profile", timezoneOffset: 8,
},
};
// Mastra validates against the model-facing schema before execute() runs, so a
// test can only send what the model can send. The cast keeps the argument
// checked against that shape without depending on Mastra's inferred type.
type ModelConsultationToolInput = { question: string; domains?: string[] };
const modelInput = (input: ModelConsultationToolInput) => input as never;
const rejectedByInputSchema = (input: { question: string; theme?: string; domains?: string[] }) => input as never;
type WorkflowOptions = {
status?: "ready" | "degraded" | "blocked";
missingLayers?: string[];
preciseTiming?: boolean;
availableLayers?: string[];
hardBlockers?: string[];
leadWithLimitations?: boolean;
limitation?: string;
chart?: Record<string, unknown>;
};
function workflow(theme = "career", options: WorkflowOptions = {}) {
return {
success: true,
question: "综合看看",
chart: options.chart ?? {},
routing: { primary_theme: theme },
consumer_context: {
route: theme,
core_status: options.status ?? "ready",
available_layers: options.availableLayers ?? [],
missing_route_layers: options.missingLayers ?? [],
hard_blockers: options.hardBlockers ?? [],
technique_truth: { status: "verified" },
answer_policy: {
can_answer_direction: true,
can_answer_precise_timing: options.preciseTiming ?? true,
...(options.leadWithLimitations === undefined ? {} : { should_lead_with_limitations: options.leadWithLimitations }),
},
...(options.limitation === undefined ? {} : { user_facing_limitation: options.limitation }),
},
};
}
// The natal projection only survives the evidence allowlist when the chart
// actually carries allowlisted placements, and the hoisting test needs it to.
const natalChart = {
ascendant: { sign: "Leo", degree: 12.5 },
planets: [{ name: "Sun", sign: "Leo", degree: 1.25 }, { name: "Moon", sign: "Pisces", degree: 20.5 }],
houses: [{ number: 1, sign: "Leo" }],
};
const toolContext = { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never;
type PlanResult = Record<string, unknown> & {
domains: string[];
omitted_domains: string[];
consultations: Array<Record<string, unknown> & { domain: string; claim_cards: Array<{ category: string }> }>;
evidence_contract: {
available_layers: string[];
missing_route_layers: string[];
hard_blockers: string[];
answer_policy: Record<string, unknown>;
user_facing_limitation?: string;
};
rectification: { boundary: string };
claim_cards: Array<{ category: string }>;
};
async function runDomainPlan(
domains: string[],
runWorkflow: (theme: string) => ReturnType<typeof workflow>,
options: { now?: () => number; requestId?: string } = {},
) {
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: options.requestId ?? `r-${domains.join("-")}`,
consultationMode: "verified_chart", serverChart, state,
...(options.now ? { now: options.now } : {}),
runWorkflow: async (input) => runWorkflow(input.theme),
})["run-jyotish-consultation"];
const result = await tool.execute!(modelInput({ question: "综合看看", domains }), toolContext) as PlanResult;
return { result, state };
}
test("the server-selected domain stays authoritative when the model omits domains", async () => {
let calls = 0;
let captured: unknown;
let capturedPlan: unknown;
const state = createConsultationRuntimeState();
const plan = createConsultationPlan({
userIntent: "事业如何", theme: "career", consultationMode: "unverified_birth_time", modelCreditCost: 1,
});
assert.throws(
() => (plan.requestedDomains as unknown as string[]).push("timing"),
TypeError,
);
const tools = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r", consultationMode: "unverified_birth_time",
plan,
theme: "career", serverChart, state,
runWorkflow: async (input, options) => {
calls += 1; captured = input; capturedPlan = options?.plan; return workflow();
},
});
const tool = tools["run-jyotish-consultation"];
// A single-value theme is deliberately absent: two mutually exclusive ways to
// name a domain cost the model a step per call to discover the rule.
assert.deepEqual(Object.keys((tool.inputSchema as unknown as { shape: object }).shape), ["question", "domains"]);
const execute = tool.execute!;
const context = { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never;
const [first, second] = await Promise.all([
execute(modelInput({ question: "尝试改成精确应期" }), context),
execute(modelInput({ question: "尝试改成婚恋" }), context),
]);
assert.equal(calls, 1);
assert.deepEqual(first, second);
assert.deepEqual(captured, { ...serverChart.toolInput, entryMode: "direct_chart", question: "事业如何", theme: "career" });
assert.strictEqual(capturedPlan, plan);
assert.equal(state.consultationToolCallCount, 1);
assert.equal(state.consultationToolSuccessCount, 1);
assert.equal(state.workflowReceipt?.preciseTiming, "allowed");
assert.deepEqual(state.workflowReceipt?.domains, ["career"]);
assert.deepEqual((first as { domains?: string[] }).domains, ["career"]);
});
test("multi-domain plan canonicalizes aliases, de-duplicates, preserves order, and runs every domain", async () => {
const calls: Array<{ theme: string; question: string }> = [];
const state = createConsultationRuntimeState();
const tools = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async (input) => {
calls.push({ theme: input.theme, question: input.question });
if (input.theme === "wealth") return workflow(input.theme, { status: "degraded", missingLayers: ["D11"] });
return workflow(input.theme);
},
});
// Aliases, not repetitions: the array bound is now the executable domain cap,
// so a duplicate spends one of the slots the clock can actually pay for.
// canonicalDomainPlan keeps the de-duplication coverage for longer raw lists.
const result = await tools["run-jyotish-consultation"].execute!(
modelInput({ question: "事业、财富和迁居怎么一起规划", domains: ["career", "finance", "home"] }),
{ observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never,
) as { domains: string[]; consultations: Array<{ domain: string }> };
assert.deepEqual(calls, [
{ theme: "career", question: "事业、财富和迁居怎么一起规划" },
{ theme: "wealth", question: "事业、财富和迁居怎么一起规划" },
{ theme: "migration", question: "事业、财富和迁居怎么一起规划" },
]);
assert.deepEqual(result.domains, ["career", "wealth", "migration"]);
assert.deepEqual(result.consultations.map((item) => item.domain), ["career", "wealth", "migration"]);
assert.deepEqual(state.workflowReceipt, {
route: "multi-domain",
status: "degraded",
preciseTiming: "allowed",
missingLayers: ["D11"],
domains: ["career", "wealth", "migration"],
});
});
test("a multi-domain result exposes the same top-level answer contract as a single domain", async () => {
const single = await runDomainPlan(["career"], (theme) => workflow(theme, {
availableLayers: ["D10"], chart: natalChart,
}));
const multi = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, {
availableLayers: theme === "career" ? ["D10"] : ["D11"], chart: natalChart,
}));
// Every path jyotishInstructions states an output rule against has to resolve
// in both shapes. With none of them present the model has no contract that
// authorizes it to speak, which is how a successful calculation produced no
// answer at all.
for (const key of ["packet_version", "question", "route", "status", "evidence_contract", "claim_cards", "rectification"]) {
assert.equal(key in single.result, true, `single-domain result is missing ${key}`);
assert.equal(key in multi.result, true, `multi-domain result is missing ${key}`);
}
for (const contract of [single.result.evidence_contract, multi.result.evidence_contract]) {
assert.equal(Array.isArray(contract.available_layers), true);
assert.equal(Array.isArray(contract.missing_route_layers), true);
assert.equal(Array.isArray(contract.hard_blockers), true);
assert.equal(typeof contract.answer_policy.can_answer_direction, "boolean");
assert.equal(typeof contract.answer_policy.can_answer_precise_timing, "boolean");
}
assert.equal(single.result.rectification.boundary, "not_auto_rectified");
assert.equal(multi.result.rectification.boundary, "not_auto_rectified");
assert.equal(multi.result.packet_version, single.result.packet_version);
assert.equal(multi.result.question, single.result.question);
assert.equal(multi.result.route, "multi-domain");
assert.equal(multi.result.status, "ready");
assert.equal(multi.result.success, true);
assert.equal(single.result.success, true);
// An available layer stays available: it really was computed for one domain.
assert.deepEqual(multi.result.evidence_contract.available_layers, ["D10", "D11"]);
assert.deepEqual(multi.result.consultations.map((item) => item.domain), ["career", "wealth"]);
});
test("the merged answer policy is the most restrictive of the executed domains", async () => {
const { result, state } = await runDomainPlan(["career", "timing", "wealth"], (theme) => workflow(theme, {
// One domain forbidding precise timing must forbid it for the whole answer.
preciseTiming: theme !== "timing",
status: theme === "wealth" ? "degraded" : "ready",
missingLayers: theme === "wealth" ? ["D11"] : [],
hardBlockers: theme === "timing" ? ["negative_holdout_gate"] : [],
leadWithLimitations: theme === "timing",
limitation: theme === "wealth" ? "财富层证据不完整。" : undefined,
chart: natalChart,
}));
const policy = result.evidence_contract.answer_policy;
assert.equal(policy.can_answer_precise_timing, false);
assert.equal(policy.can_answer_direction, true);
assert.equal(policy.should_lead_with_limitations, true);
assert.deepEqual(result.evidence_contract.hard_blockers, ["negative_holdout_gate"]);
assert.deepEqual(result.evidence_contract.missing_route_layers, ["D11"]);
assert.equal(result.status, "degraded");
assert.equal(result.evidence_contract.user_facing_limitation, "财富层证据不完整。");
assert.equal(state.workflowReceipt?.preciseTiming, "blocked");
assert.equal(state.workflowReceipt?.status, "degraded");
const blocked = await runDomainPlan(["career", "health"], (theme) => workflow(theme, {
status: theme === "health" ? "blocked" : "ready", chart: natalChart,
}));
assert.equal(blocked.result.status, "blocked");
});
test("merging answer policies can only ever restrict", () => {
// Merged directly, because the projection currently emits only three policy
// fields and the rules have to hold for any field it may emit later.
assert.deepEqual(
mergeConsultationAnswerPolicies([
{ can_answer_direction: true, can_answer_precise_timing: true },
{ can_answer_direction: true, can_answer_precise_timing: false },
]),
{ can_answer_direction: true, can_answer_precise_timing: false },
);
assert.deepEqual(
mergeConsultationAnswerPolicies([
{ can_answer_direction: true, can_answer_precise_timing: true, should_lead_with_limitations: false },
{ can_answer_direction: false, can_answer_precise_timing: true, should_lead_with_limitations: true },
]),
{ can_answer_direction: false, can_answer_precise_timing: true, should_lead_with_limitations: true },
);
// A prohibition list unions: a technique one domain forbids stays forbidden.
const prohibitions = mergeConsultationAnswerPolicies([
{ can_answer_direction: true, can_answer_precise_timing: true, deterministic_claims_forbidden_for: ["narayana"] },
{ can_answer_direction: true, can_answer_precise_timing: true, deterministic_claims_forbidden_for: ["transit", "narayana"] },
]);
assert.deepEqual(prohibitions.deterministic_claims_forbidden_for, ["narayana", "transit"]);
// A boolean that is absent for one domain is not consent from that domain.
assert.equal(
mergeConsultationAnswerPolicies([{ can_answer_chart_interpretation: true }, {}]).can_answer_chart_interpretation,
false,
);
// A field the domains disagree on in a way that cannot be merged is reported
// as unresolved and forces the answer to lead with its limits, rather than
// being dropped, which would remove whatever it was restricting.
const conflicted = mergeConsultationAnswerPolicies([
{ can_answer_direction: true, can_answer_precise_timing: true, claim_ceiling: "direction_only" },
{ can_answer_direction: true, can_answer_precise_timing: true, claim_ceiling: "structure_only" },
]);
assert.deepEqual(conflicted.unresolved_policy_fields, ["claim_ceiling"]);
assert.equal(conflicted.should_lead_with_limitations, true);
assert.equal("claim_ceiling" in conflicted, false);
});
test("the merged contract exposes every policy field a single domain exposes", async () => {
const options: WorkflowOptions = {
leadWithLimitations: false, limitation: "边界说明。", chart: natalChart,
};
const single = await runDomainPlan(["career"], (theme) => workflow(theme, options));
const multi = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, options));
// Guards drift: a field added to the per-domain projection without being
// merged would silently vanish from the multi-domain contract.
for (const key of Object.keys(single.result.evidence_contract.answer_policy)) {
assert.equal(key in multi.result.evidence_contract.answer_policy, true, `merged policy is missing ${key}`);
}
for (const key of Object.keys(single.result.evidence_contract)) {
assert.equal(key in multi.result.evidence_contract, true, `merged contract is missing ${key}`);
}
assert.equal(multi.result.evidence_contract.answer_policy.should_lead_with_limitations, false);
});
test("the identical natal projection is carried once instead of per domain", async () => {
const { result } = await runDomainPlan(["career", "wealth", "timing"], (theme) => workflow(theme, { chart: natalChart }));
assert.deepEqual(result.claim_cards.map((card) => card.category), ["natal_foundation"]);
assert.equal(
result.consultations.every((item) => item.claim_cards.every((card) => card.category !== "natal_foundation")),
true,
);
assert.equal(result.consultations.some((item) => item.claim_cards.length > 0), true);
// When the domains genuinely disagree, nothing is presented as shared.
const differing = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, {
chart: theme === "career" ? natalChart : { ...natalChart, ascendant: { sign: "Virgo", degree: 1 } },
}));
assert.deepEqual(differing.result.claim_cards, []);
assert.equal(
differing.result.consultations.every((item) => item.claim_cards.some((card) => card.category === "natal_foundation")),
true,
);
});
test("the domain cap is what the run budget can actually pay for", () => {
// 21s per sequential domain against the 110s run budget, minus the reserve a
// three-domain staging run actually left for composing the answer.
assert.equal(MAX_CONSULTATION_DOMAINS, 3);
assert.equal(AGENT_TIMEOUT_MS, 110_000);
assert.equal(CONSULTATION_DOMAIN_WALL_CLOCK_MS, 65_000);
assert.ok(MAX_CONSULTATION_DOMAINS * 21_000 <= CONSULTATION_DOMAIN_WALL_CLOCK_MS);
// Six domains, the previous cap, could never finish inside the deadline.
assert.ok(6 * 21_000 > AGENT_TIMEOUT_MS);
assert.deepEqual(
executableDomainPlan(["career", "wealth", "timing", "marriage", "health"]),
{ domains: ["career", "wealth", "timing"], omittedDomains: ["marriage", "health"] },
);
assert.deepEqual(executableDomainPlan(["career"]), { domains: ["career"], omittedDomains: [] });
// The first domain always runs; after that the next one has to be projected
// to finish, judged by how long the executed ones really took.
assert.equal(domainFitsRunBudget(0, 0), true);
assert.equal(domainFitsRunBudget(21_000, 1), true);
assert.equal(domainFitsRunBudget(42_000, 2), true);
assert.equal(domainFitsRunBudget(60_000, 2), false);
assert.equal(domainFitsRunBudget(40_000, 1), false);
});
test("a plan larger than the cap cannot be expressed and never starts a calculation", async () => {
let calls = 0;
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-cap", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async (input) => { calls += 1; return workflow(input.theme); },
})["run-jyotish-consultation"];
const inputSchema = tool.inputSchema as unknown as { safeParse: (value: unknown) => { success: boolean } };
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career", "wealth", "timing"] }).success, true);
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career", "wealth", "timing", "marriage"] }).success, false);
// Mastra rejects the over-budget plan before the tool body runs, so it costs
// one correctable step and nothing about the run advances.
const refused = await tool.execute!(
{ question: "全都看看", domains: ["career", "wealth", "timing", "marriage", "health"] } as never,
toolContext,
) as Record<string, unknown>;
assert.equal(calls, 0);
assert.equal("domains" in refused, false);
assert.equal(state.consultationToolStarted, false);
assert.deepEqual(runSteps(state), []);
});
test("a plan that runs long stops early and discloses the domains it dropped", async () => {
let clock = 0;
const executed: string[] = [];
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-slow", consultationMode: "verified_chart",
serverChart, state,
now: () => clock,
runWorkflow: async (input) => {
executed.push(input.theme);
clock += 40_000;
return workflow(input.theme, { chart: natalChart });
},
})["run-jyotish-consultation"];
const result = await tool.execute!(
modelInput({ question: "三个领域", domains: ["career", "wealth", "timing"] }),
toolContext,
) as PlanResult;
// 40s each cannot fit a second domain inside the loop's share of the budget,
// so the run answers what it has instead of aborting mid-loop and losing it.
assert.deepEqual(executed, ["career"]);
assert.deepEqual(result.domains, ["career"]);
assert.deepEqual(result.omitted_domains, ["wealth", "timing"]);
assert.equal(result.status, "degraded");
assert.equal(state.consultationToolCompleted, true);
assert.equal(state.consultationToolSuccessCount, 1);
assert.equal(state.consultationToolDurationMs, 40_000);
// The single executed domain still has to carry the full top-level contract.
for (const key of ["packet_version", "route", "status", "evidence_contract", "claim_cards", "rectification"]) {
assert.equal(key in result, true, `truncated result is missing ${key}`);
}
});
test("the advertised domain limit matches the enforced one", () => {
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-description", consultationMode: "verified_chart",
serverChart, state: createConsultationRuntimeState(),
runWorkflow: async (input) => workflow(input.theme),
})["run-jyotish-consultation"];
const description = tool.description ?? "";
assert.match(description, new RegExp(`at most ${MAX_CONSULTATION_DOMAINS} allowlisted`));
assert.doesNotMatch(description, /up to six|six allowlisted/);
assert.match(description, /omitted_domains/);
assert.match(description, /top-level answer contract/);
});
test("domain plan rejects unknown and product domains before any workflow runs", async () => {
for (const domain of ["unknown", "prashna", "muhurta", "rectification", "compatibility"]) {
let calls = 0;
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: `r-${domain}`, consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async () => { calls += 1; return workflow(); },
})["run-jyotish-consultation"];
// Refused by the enumerated schema before execute, so Mastra resolves with a
// validation envelope instead of the tool throwing from the registry check.
const rejected = await tool.execute!(
modelInput({ question: "测试", domains: ["career", domain] }),
{ observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never,
) as { error?: unknown };
assert.equal(rejected.error, true, domain);
assert.equal(calls, 0);
assert.equal(state.consultationToolCompleted, false);
}
});
test("the model-facing schema names the domain vocabulary it accepts", async () => {
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-vocabulary", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async () => workflow(),
})["run-jyotish-consultation"];
const inputSchema = tool.inputSchema as unknown as {
safeParse: (value: unknown) => { success: boolean };
shape: { domains: { unwrap: () => { element: { options?: readonly string[] } } } };
};
// The skill's methodology names strict-workflow checklists, and while this was
// a free-form string those labels passed validation and died inside the call.
// Enumerating the values is what puts the vocabulary in front of the model.
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["event-timing-strict"] }).success, false);
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["wealth-timing-strict"] }).success, false);
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"] }).success, true);
// Aliases stay accepted: enumerating states the vocabulary, it does not narrow it.
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["finance"] }).success, true);
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["感情"] }).success, true);
// Enumerable, so the JSON schema handed to the model carries the values rather
// than an opaque string. A wrapper that hid them would pass the checks above.
const options = inputSchema.shape.domains.unwrap().element.options;
assert.deepEqual(consultationDomainPlanValues, options);
for (const id of consultationDomainIds) assert.ok(options?.includes(id), id);
});
test("domain plan enforces the raw plan upper bound and one input mode", async () => {
let calls = 0;
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-limit", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async () => { calls += 1; return workflow(); },
})["run-jyotish-consultation"];
const context = { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never;
const inputSchema = tool.inputSchema as unknown as { safeParse: (value: unknown) => { success: boolean } };
assert.equal(inputSchema.safeParse({
question: "测试",
domains: ["career", "career", "career", "career", "career", "career", "career"],
}).success, false);
assert.equal(calls, 0);
// The model can only express a domain plan one way. The pair that used to be
// representable, and cost a model step to be told was invalid, is now refused
// by the schema before the tool body runs.
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"] }).success, true);
assert.equal(inputSchema.safeParse({ question: "测试" }).success, true);
assert.equal(inputSchema.safeParse({ question: "测试", theme: "career" }).success, false);
assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"], theme: "career" }).success, false);
const rejectedState = createConsultationRuntimeState();
const secondTool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-modes", consultationMode: "verified_chart",
serverChart, state: rejectedState,
runWorkflow: async () => { calls += 1; return workflow(); },
})["run-jyotish-consultation"];
const refused = await secondTool.execute!(
rejectedByInputSchema({ question: "测试", domains: ["career"], theme: "career" }),
context,
) as Record<string, unknown>;
// Nothing about the run advances: no calculation, no counted attempt, and no
// consultation payload the model could mistake for a result.
assert.equal(calls, 0);
assert.equal("domains" in refused, false);
assert.equal(rejectedState.consultationToolStarted, false);
assert.equal(rejectedState.consultationToolCallCount, 0);
assert.deepEqual(runSteps(rejectedState), []);
});
test("the single-value domain form stays available to callers without the model schema", () => {
const plan = createConsultationPlan({
userIntent: "事业如何", theme: "career", consultationMode: "verified_chart", modelCreditCost: 1,
});
// A single-value theme must never override the route-selected server domain.
assert.deepEqual(canonicalDomainPlan({ theme: "timing" }, { plan, theme: "career" }), ["career"]);
assert.deepEqual(canonicalDomainPlan({}, { plan, theme: "career" }), ["career"]);
assert.deepEqual(canonicalDomainPlan({ theme: "marriage" }, {}), ["marriage"]);
assert.deepEqual(canonicalDomainPlan({ domains: ["career", "finance"] }, {}), ["career", "wealth"]);
assert.deepEqual(canonicalDomainPlan({ domains: ["career", "finance", "career", "home"] }, {}), ["career", "wealth", "migration"]);
assert.deepEqual(canonicalDomainPlan({ domains: ["timing"] }, { plan, theme: "career" }), ["timing"]);
assert.throws(
() => canonicalDomainPlan({ domains: ["career"], theme: "career" }, {}),
/invalid_consultation_domain_plan/,
);
assert.throws(() => canonicalDomainPlan({}, {}), /invalid_consultation_domain_plan/);
assert.throws(() => canonicalDomainPlan({ theme: "prashna" }, {}), /unsupported_consultation_domain/);
});
test("invalid model input does not poison a later valid contract retry", async () => {
let calls = 0;
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-retry", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async (input) => { calls += 1; return workflow(input.theme); },
})["run-jyotish-consultation"];
const context = { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never;
// BUG-205: a bad domain must be refused before the request-scoped calculation
// cache is written. The refusal now happens at the schema, one layer earlier
// than the registry check it used to reach, because the domain ids are
// enumerated in the schema the model is handed. Mastra reports that refusal by
// resolving with a validation envelope instead of throwing, so this asserts the
// envelope rather than a rejection.
const rejected = await tool.execute!(
modelInput({ question: "先给出错误参数", domains: ["career", "unknown"] }),
context,
) as { error?: unknown; message?: unknown };
assert.equal(rejected.error, true);
// The envelope has to name the legal ids: it is the only correction the model
// gets, and an unnamed vocabulary is what produced the invalid call.
assert.match(String(rejected.message), /'career'/);
assert.match(String(rejected.message), /'timing'/);
assert.equal(calls, 0);
assert.equal(state.consultationToolCallCount, 0);
assert.equal(state.consultationToolSuccessCount, 0);
const result = await tool.execute!(modelInput({ question: "改用合法参数", domains: ["timing"] }), context) as { domains: string[] };
assert.equal(calls, 1);
assert.deepEqual(result.domains, ["timing"]);
assert.equal(state.consultationToolCallCount, 1);
assert.equal(state.consultationToolSuccessCount, 1);
assert.equal(state.consultationToolCompleted, true);
});
test("a rejected workflow promise is cleared before a later tool call", async () => {
let calls = 0;
const state = createConsultationRuntimeState();
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "r-rejected-promise", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async (input) => {
calls += 1;
if (calls === 1) throw new Error("workflow_temporarily_failed");
return workflow(input.theme);
},
})["run-jyotish-consultation"];
const context = { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never;
await assert.rejects(
tool.execute!(modelInput({ question: "第一次计算", domains: ["career"] }), context),
/workflow_temporarily_failed/,
);
const result = await tool.execute!(modelInput({ question: "重新计算", domains: ["timing"] }), context) as { domains: string[] };
assert.equal(calls, 2);
assert.deepEqual(result.domains, ["timing"]);
assert.equal(state.consultationToolCallCount, 2);
assert.equal(state.consultationToolSuccessCount, 1);
});
test("a failed calculation records why it failed and forwards the request id", async () => {
const state = createConsultationRuntimeState();
const seen: Array<string | undefined> = [];
const tool = createConsultationTools({
userId: "u", sessionId: "s", requestId: "req-correlation", consultationMode: "verified_chart",
serverChart, state,
runWorkflow: async (_input, options) => {
seen.push(options?.requestId);
throw new ConsultationWorkflowError("workflow_queue_full", "Async job queue is full");
},
})["run-jyotish-consultation"];
await assert.rejects(
tool.execute!(modelInput({ question: "队列满时的表现", domains: ["career"] }), { observe: { span: async (_n: string, fn: () => Promise<unknown>) => fn(), log() {} } } as never),
/Async job queue is full/,
);
assert.deepEqual(seen, ["req-correlation"]);
const failedStep = state.steps.find((step) => step.status === "failed");
assert.equal(failedStep?.name, "run-jyotish-consultation");
assert.equal(failedStep?.failureCode, "workflow_queue_full");
});
test("workflow failure codes classify transport and contract faults", () => {
assert.equal(consultationWorkflowFailureCode(new ConsultationWorkflowError("workflow_rate_limited", "x")), "workflow_rate_limited");
assert.equal(consultationWorkflowFailureCode(new DOMException("slow", "TimeoutError")), "workflow_timeout");
assert.equal(consultationWorkflowFailureCode(new DOMException("stop", "AbortError")), "workflow_aborted");
assert.equal(consultationWorkflowFailureCode(new Error("anything else")), undefined);
});
test("every tool failure resolves to a code, not an absent field", () => {
// The workflow classifier returns undefined for anything it does not own, and
// the append site omitted the field when it was undefined, so the failures
// raised inside the tool reached the log as the one record with no reason.
assert.equal(consultationToolFailureCode(new ConsultationWorkflowError("workflow_rate_limited", "x")), "workflow_rate_limited");
assert.equal(consultationToolFailureCode(new DOMException("stop", "AbortError")), "workflow_aborted");
assert.equal(consultationToolFailureCode(new Error("invalid_consultation_domain_plan")), "invalid_domain_plan");
assert.equal(consultationToolFailureCode(new Error("unsupported_consultation_domain")), "invalid_domain_plan");
assert.equal(consultationToolFailureCode(new Error("anything else")), "unexpected_error");
assert.equal(consultationToolFailureCode("not an error"), "unexpected_error");
});
test("a call rejected before the tool body runs still appears in the receipt", async () => {
// Observed on staging: the model's arguments were refused against the tool's
// strict input schema, so `execute` never ran. The client saw tool.failed
// while the receipt showed no failed step and the budget counted no call.
const state = createConsultationRuntimeState();
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation" } };
yield { type: "tool-error", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation", error: new Error("bad arguments") } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "blocked",
receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }),
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
const failedStep = state.steps.find((step) => step.kind === "tool" && step.status === "failed");
assert.equal(failedStep?.name, "run-jyotish-consultation");
assert.equal(failedStep?.failureCode, "tool_call_rejected");
assert.equal(consultationStepBudgetReceipt(state).used, 2);
// The client learns a step failed; the classification stays server-side.
const failed = events.find((event) => (event as { type?: string }).type === "run.failed") as {
receipt?: { steps: Array<{ status: string; name: string }> };
};
assert.deepEqual(failed.receipt?.steps.map((step) => step.status), ["completed", "failed"]);
assert.doesNotMatch(JSON.stringify(failed), /tool_call_rejected/);
});
test("a failure the tool already recorded is not recorded twice", async () => {
// The tool records what it can see, with the duration and cause it alone
// knows. The stream must only fill the gap, never double-count.
const state = createConsultationRuntimeState();
appendConsultationRuntimeStep(state, {
kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 20936, failureCode: "workflow_queue_full",
});
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation" } };
yield { type: "tool-error", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation", error: new Error("boom") } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "blocked",
receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }),
onError: () => {},
});
await response.text();
const failed = state.steps.filter((step) => step.status === "failed");
assert.equal(failed.length, 1);
assert.equal(failed[0]?.failureCode, "workflow_queue_full");
});
test("the public receipt never carries the internal failure classification", () => {
const state = createConsultationRuntimeState();
appendConsultationRuntimeStep(state, {
kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 12, failureCode: "workflow_server_error",
});
const steps = publicConsultationRuntimeSteps(state);
assert.equal(steps.every((step) => !("failureCode" in step)), true);
assert.equal(state.steps.find((step) => step.kind === "tool")?.failureCode, "workflow_server_error");
// A strict receipt schema would reject the internal field, so this also
// guards the run from failing while building a successful response.
const receipt = agentExecutionReceiptSchema.parse({
runId: "run", runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps,
workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] },
});
assert.equal(receipt.steps.length, 2);
assert.throws(() => agentExecutionReceiptSchema.parse({
runId: "run", runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: state.steps,
workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] },
}));
});
test("the public receipt never carries the model step budget diagnostics", () => {
const state = createConsultationRuntimeState();
state.modelStepCount = 8;
state.modelFinishReason = "tool-calls";
appendConsultationRuntimeStep(state, { kind: "skill", name: "jyotish-vedic-astrology", status: "completed" });
assert.deepEqual(
consultationModelStepTelemetry(state),
{ modelStepCount: 8, skillReferenceReads: 0, methodologySections: 0, modelFinishReason: "tool-calls" },
);
assert.deepEqual(
consultationModelStepTelemetry(createConsultationRuntimeState()),
{ modelStepCount: 0, skillReferenceReads: 0, methodologySections: 0 },
);
// The client receipt schema is strict, so leaking either field would make a
// successful run fail while serializing its own answer.
const receipt = agentExecutionReceiptSchema.parse({
runId: "run", runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: publicConsultationRuntimeSteps(state),
stepBudget: consultationStepBudgetReceipt(state),
workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] },
});
assert.doesNotMatch(JSON.stringify(receipt), /modelStepCount|modelFinishReason|tool-calls/);
assert.throws(() => agentExecutionReceiptSchema.parse({
runId: "run", runtime: "mastra-agentic",
skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: publicConsultationRuntimeSteps(state),
workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] },
...consultationModelStepTelemetry(state),
}));
});
test("the receipt reports how many reference documents the model opened", () => {
const state = createConsultationRuntimeState();
const hooks = createConsultationRuntimeHooks(state);
// The method is bound before the model runs, so a run starts with it in hand and
// with nothing read. Having the method says nothing about the model having gone
// past it to a reference of its own, which is the number this counts.
assert.equal(state.jyotishSkillBound, true);
assert.equal(state.skillReferenceReadCount, 0);
hooks.afterToolCall({ toolName: "skill_read" });
hooks.afterToolCall({ toolName: "read_file" });
hooks.afterToolCall({ toolName: "skill_read", error: new Error("denied") });
assert.equal(state.skillReferenceReadCount, 2);
// Reference reads are not runtime steps, so the step list cannot answer this on its own.
assert.equal(state.steps.filter((step) => step.kind === "skill").length, 1);
assert.equal(agentExecutionReceiptSchema.parse(receipt(state)).skill.referenceReads, 2);
});
test("the receipt separates method the server delivered from method the model went looking for", () => {
const state = createConsultationRuntimeState();
// A run where the model opened nothing is no longer a run composed without method: the strict
// checklist for the route travels with the evidence, so the two counts have to be readable apart.
state.methodologySectionCount = 3;
const parsed = agentExecutionReceiptSchema.parse(receipt(state));
assert.equal(parsed.skill.referenceReads, 0);
assert.equal(parsed.skill.methodologySections, 3);
assert.equal(consultationModelStepTelemetry(state).methodologySections, 3);
});
test("personal Agent exposes the Jyotish Skill and named server tool", async () => {
const state = createConsultationRuntimeState();
const agent = getJyotishAgent({
id: "personal-agent-probe", label: "Probe", description: "", creditCost: 1, isDefault: false,
mode: "openai", model: "openai/gpt-5-mini",
} as never, {
userId: "u", sessionId: "s", requestId: "r", consultationMode: "verified_chart", serverChart, state,
} as never);
const skills = await agent.listSkills();
const toolNames = Object.keys(await agent.getToolsForExecution({ runId: "r" }));
assert.equal(skills.some((skill) => skill.name === "jyotish-vedic-astrology"), true);
// Activation is withdrawn: answering an activation meant resending the whole
// package listing every turn, and the method is bound into the instructions
// instead. Reading a named reference is still the model's own to do.
assert.equal(toolNames.includes("skill"), false);
assert.equal(toolNames.includes("skill_search"), false);
assert.equal(toolNames.includes("skill_read"), true);
assert.equal(toolNames.includes("run-jyotish-consultation"), true);
assert.equal(toolNames.includes("consultationTool"), false);
});
test("public stream filters private chunks and completes once", async () => {
const chunks = [
{ type: "reasoning-delta", payload: { text: "secret" } },
{ type: "tool-call", payload: { toolCallId: "c1", toolName: "skill", args: { name: "jyotish-vedic-astrology", secret: "x" } } },
{ type: "tool-result", payload: { toolCallId: "other", toolName: "skill", result: { private: true } } },
{ type: "tool-result", payload: { toolCallId: "c1", toolName: "skill", result: { private: true } } },
{ type: "tool-call", payload: { toolCallId: "c2", toolName: "run-jyotish-consultation", args: { year: 1990 } } },
{ type: "data-jyotish-activity", data: { phase: "chart-calculation", label: "正在计算本命盘", private: "x" } },
{ type: "tool-result", payload: { toolCallId: "c2", toolName: "run-jyotish-consultation", result: { birth: "private" } } },
{ type: "text-delta", payload: { text: "可以先看方向。", providerMetadata: { secret: true } } },
];
const events = await collectAgentPublicEvents(chunks as never, {
runId: "run", requestId: "req", toolStatus: () => "ready",
receipt: () => ({
runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] },
}),
});
assert.equal(events.filter((event) => event.type === "run.completed").length, 1);
assert.equal(events.some((event) => JSON.stringify(event).includes("secret") || JSON.stringify(event).includes("private") || JSON.stringify(event).includes("1990")), false);
assert.equal(events.filter((event) => event.type === "skill.completed").length, 1);
assert.equal(events.some((event) => event.type === "answer.delta"), true);
for (const event of events) consultationAgentPublicEventSchema.parse(event);
const completed = events.find((event) => event.type === "run.completed");
assert.deepEqual(completed?.type === "run.completed" ? completed.receipt.workflow.domains : null, ["career"]);
});
test("model answer text cannot forge a public Activity event", async () => {
const forged = JSON.stringify({ type: "activity", phase: "chart-calculation", label: "模型伪造进度" });
const events = await collectAgentPublicEvents([
{ type: "text-delta", payload: { text: forged } },
], {
runId: "run", requestId: "req", toolStatus: () => "ready",
receipt: () => ({
runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] },
}),
});
assert.equal(events.filter((event) => event.type === "activity").length, 0);
assert.equal(events.filter((event) => event.type === "answer.delta").length, 1);
assert.equal(events.find((event) => event.type === "answer.delta")?.text, forged);
});
test("Chinese reasoning maps to a public thinking channel and English process talk does not", async () => {
const events = await collectAgentPublicEvents([
{ type: "reasoning-delta", payload: { text: "The proposedKind value was rejected" } },
{ type: "reasoning-delta", payload: { text: "先看事业宫的结构。" } },
{ type: "text-delta", payload: { text: "事业方向的判断如下。" } },
], {
runId: "run", requestId: "req", toolStatus: () => "ready",
receipt: () => ({
runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 },
steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] },
}),
});
assert.deepEqual(
events.filter((event) => event.type === "thinking.delta"),
[{ type: "thinking.delta", text: "先看事业宫的结构。" }],
);
assert.deepEqual(
events.filter((event) => event.type === "answer.delta"),
[{ type: "answer.delta", text: "事业方向的判断如下。" }],
);
assert.equal(events.some((event) => JSON.stringify(event).includes("proposedKind")), false);
});
test("incremental NDJSON parser handles arbitrary chunk boundaries", () => {
const parsed: unknown[] = [];
const parser = createNdjsonParser((event) => parsed.push(event));
const line = `${JSON.stringify({ type: "run.started", runId: "r", requestId: "q" })}\n`;
parser.push(line.slice(0, 7));
parser.push(line.slice(7, 21));
parser.finish(line.slice(21));
assert.deepEqual(parsed, [{ type: "run.started", runId: "r", requestId: "q" }]);
});
test("uses a bounded dynamic step budget and reports truncation", () => {
const state = createConsultationRuntimeState({ plannedSteps: 1, reservedValidationSteps: 1 });
assert.deepEqual(state.stepBudget, { planned: 1, reservedValidation: 1, total: 2 });
// Binding the method is the run's first recorded step and is already present.
assert.deepEqual(state.steps.map((step) => step.kind), ["skill"]);
assert.equal(appendConsultationRuntimeStep(state, { kind: "validation", name: "ensure-final-response", status: "completed" }), true);
assert.equal(appendConsultationRuntimeStep(state, { kind: "tool", name: "unexpected-extra-step", status: "completed" }), false);
assert.equal(state.steps.length, 2);
assert.equal(state.stepsTruncated, true);
assert.deepEqual(consultationStepBudgetReceipt(state), { planned: 2, used: 2, remaining: 0, truncated: true });
});
/**
* The steps a run took, without the method binding every run starts with. The
* binding is recorded when the state is created, so a test asking whether a run
* advanced has to say which steps it means.
*/
function runSteps(state: ReturnType<typeof createConsultationRuntimeState>) {
return state.steps.filter((step) => step.kind !== "skill");
}
function receipt(state: ReturnType<typeof createConsultationRuntimeState>) {
return {
runId: "run",
runtime: "mastra-agentic" as const,
skill: {
name: "jyotish-vedic-astrology" as const,
loaded: state.jyotishSkillBound,
referenceReads: state.skillReferenceReadCount,
methodologySections: state.methodologySectionCount,
},
steps: state.steps,
stepBudget: consultationStepBudgetReceipt(state),
workflow: state.workflowReceipt ?? { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] },
};
}
test("text written before the contract completes is dropped, not released later", async () => {
const state = createConsultationRuntimeState();
let completed = 0;
async function* chunks() {
// Production shape (run a5f4409e): between rejected calls the model narrates
// its own tool errors. Holding that text meant the eventual success released
// it as the visible answer, so a recovered run read as the model explaining
// itself and never answering the question.
yield { type: "text-delta", payload: { text: "域名单有误,我改为不指定域。" } };
yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } };
state.jyotishSkillBound = true;
yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } };
yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: {} } };
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
yield { type: "text-delta", payload: { text: "这是真正的回答。" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 1);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1);
const answer = events
.filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta")
.map((event) => event.text)
.join("");
assert.equal(answer, "这是真正的回答。");
assert.doesNotMatch(JSON.stringify(events), /域名单有误/);
});
test("a run that only narrated its failures is not delivered or billed as an answer", async () => {
const state = createConsultationRuntimeState();
let completed = 0;
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } };
state.jyotishSkillBound = true;
yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } };
yield { type: "text-delta", payload: { text: "两次域名单都不被服务端接受,我改为不指定域。" } };
yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: {} } };
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "general", status: "ready", preciseTiming: "allowed", missingLayers: [] };
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
// Presenting the narration as the reading is dishonest, and so is charging for
// a fixed apology that says there is nothing to say.
assert.equal(completed, 0);
assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false);
const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string };
assert.equal(failure.code, "empty_answer");
assert.doesNotMatch(JSON.stringify(events), /不被服务端接受/);
});
test("a call Mastra rejected against the input schema is not reported as completed", async () => {
const state = createConsultationRuntimeState({ plannedSteps: 8 });
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: { domains: ["event-timing-strict"] } } };
// Mastra resolves rather than throws when arguments fail inputSchema, so the
// tool body never runs and cannot record anything. Reported as completed this
// would claim a calculation that never happened.
yield {
type: "tool-result",
payload: {
toolCallId: "tool-1",
toolName: "run-jyotish-consultation",
result: { error: true, message: "Tool input validation failed", validationErrors: { errors: [], fields: {} } },
},
};
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }),
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(events.some((event) => (event as { type?: string }).type === "tool.completed"), false);
const failed = events.find((event) => (event as { type?: string }).type === "tool.failed") as { code: string };
assert.equal(failed.code, "calculation_failed");
assert.deepEqual(
state.steps.map((step) => `${step.kind}:${step.status}`),
["skill:completed", "tool:failed"],
);
assert.equal(runSteps(state)[0]?.failureCode, "tool_call_rejected");
// The rejection reason is a server-side diagnostic; the client sees only that a step failed.
assert.doesNotMatch(JSON.stringify(events), /tool_call_rejected|validationErrors/);
});
test("a calculation that succeeds only after failed attempts still satisfies the contract", async () => {
const state = createConsultationRuntimeState();
let completed = 0;
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } };
state.jyotishSkillBound = true;
yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } };
// Two transient workflow failures, then one success, as observed in production.
state.consultationToolCallCount = 3;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
yield { type: "tool-result", payload: { toolCallId: "tool-3", toolName: "run-jyotish-consultation", result: {} } };
yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 1);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 0);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1);
assert.equal((events.find((event) => (event as { type?: string }).type === "answer.delta") as { text?: string }).text, "事业方向的判断如下。");
});
test("a second successful calculation still fails the single-calculation boundary", async () => {
const state = createConsultationRuntimeState();
let failed = 0;
async function* chunks() {
yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } };
state.jyotishSkillBound = true;
yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } };
state.consultationToolCallCount = 2;
state.consultationToolSuccessCount = 2;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
yield { type: "text-delta", payload: { text: "不应显示" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onError: () => { failed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(failed, 1);
assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1);
});
test("incomplete runtime contract fails without saving a successful answer", async () => {
const state = createConsultationRuntimeState();
let completed = 0;
let failed = 0;
async function* chunks() {
yield { type: "text-delta", payload: { text: "不能保存" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "blocked", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
onError: () => { failed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 0);
assert.equal(failed, 1);
assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1);
});
function toolOnlyRunState() {
const state = createConsultationRuntimeState();
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
return state;
}
test("a calculation the model never wrote up is asked again instead of apologised for", async () => {
// Production shape: the tool succeeded in 63.5s and the model then produced no
// text at all. That used to be answered with a fixed apology and billed as a
// completed consultation, so the user paid for a sentence saying nothing.
const state = toolOnlyRunState();
let completedOutput = "";
let retries = 0;
async function* chunks() {
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
}
async function* answerChunks() {
yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
retryForAnswer: async () => { retries += 1; return answerChunks(); },
onComplete: (output) => { completedOutput = output; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(retries, 1);
assert.equal(completedOutput, "事业方向的判断如下。");
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1);
assert.equal(state.steps.at(-1)?.name, "answer-retry");
});
test("a calculation still unanswered after the retry fails the run rather than billing it", async () => {
const state = toolOnlyRunState();
let completed = 0;
async function* chunks() {
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
}
async function* silentChunks() {
yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
retryForAnswer: async () => silentChunks(),
onComplete: () => { completed += 1; },
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 0);
assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false);
const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string; message: string };
assert.equal(failure.code, "empty_answer");
assert.match(failure.message, /不会扣点/);
});
test("a length-limited answer is not billed or delivered as a completed consultation", async () => {
// Staging persisted a 253-character pinch that ended mid-heading, then treated
// the run as completed. The model had finished with reason `length`; the public
// stream still emitted `run.completed`, so the composer unlocked as if the
// reading were done.
const state = toolOnlyRunState();
const pinchedHeading = "**先看命盘结构(Lahiri岁差、均交点口径";
let completed = 0;
let failed = 0;
async function* chunks() {
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
yield { type: "text-delta", payload: { text: pinchedHeading } };
yield { type: "finish", payload: { stepResult: { reason: "length" }, output: { usage: {}, steps: [{}, {}] } } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
onError: () => { failed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 0);
assert.equal(failed, 1);
assert.equal(state.modelFinishReason, "length");
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 0);
const answer = events
.filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta")
.map((event) => event.text)
.join("");
assert.equal(answer, pinchedHeading);
const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as {
code: string;
message: string;
};
assert.equal(failure.code, "answer_truncated");
assert.match(failure.message, /回答未完成/);
assert.match(failure.message, /不会扣点/);
assert.doesNotMatch(JSON.stringify(failure), /modelFinishReason|"length"/);
});
test("a timeout after partial visible text is the same truncation, not a successful answer", async () => {
const state = toolOnlyRunState();
const pinchedHeading = "**先看命盘结构(Lahiri岁差、均交点口径";
let completed = 0;
async function* chunks() {
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
yield { type: "text-delta", payload: { text: pinchedHeading } };
throw new DOMException("The operation was aborted due to timeout", "TimeoutError");
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(completed, 0);
const answer = events
.filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta")
.map((event) => event.text)
.join("");
assert.equal(answer, pinchedHeading);
const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string };
assert.equal(failure.code, "answer_truncated");
});
test("consult generation reserves visible output tokens and enables a separate thinking channel", () => {
const settings = consultationGenerationSettings("deepseek");
assert.equal(CONSULTATION_MAX_OUTPUT_TOKENS, 8192);
assert.equal(settings.modelSettings.maxOutputTokens, CONSULTATION_MAX_OUTPUT_TOKENS);
assert.deepEqual(settings.providerOptions.openai, { thinking: { type: "enabled" } });
assert.deepEqual(settings.providerOptions.deepseek, { thinking: { type: "enabled" } });
assert.equal(AGENT_MAX_STEPS, 8);
});
test("Chinese thinking stays off the spoken answer and does not bill a thought-only run", async () => {
const state = createConsultationRuntimeState();
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
async function* chunks() {
yield { type: "reasoning-delta", payload: { text: "The proposedKind value was rejected" } };
yield { type: "reasoning-delta", payload: { text: "先看事业宫的结构。" } };
yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.deepEqual(
events.filter((event) => (event as { type?: string }).type === "thinking.delta"),
[{ type: "thinking.delta", text: "先看事业宫的结构。" }],
);
const answer = events
.filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta")
.map((event) => event.text)
.join("");
assert.equal(answer, "事业方向的判断如下。");
assert.doesNotMatch(JSON.stringify(events), /proposedKind/);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1);
});
test("a completed run records the finish reason and the authoritative step count", async () => {
const state = createConsultationRuntimeState();
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
async function* chunks() {
yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } };
yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } };
yield { type: "step-finish", payload: { stepResult: { reason: "stop" } } };
// The terminal chunk carries the runtime's own step list, which wins over
// the chunks we counted.
yield { type: "finish", payload: { stepResult: { reason: "stop" }, output: { usage: {}, steps: [{}, {}, {}] } } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
});
await response.text();
assert.equal(state.modelFinishReason, "stop");
assert.equal(state.modelStepCount, 3);
});
test("a run that stops while still wanting tools records the exhausted step budget", async () => {
const state = createConsultationRuntimeState({ plannedSteps: 8 });
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
// The production shape: the calculation succeeded, the model never wrote an
// answer, and only the fallback text reached the user. Nothing in the public
// event stream said the step budget ran out. A model that stopped while it
// still wanted tool calls is the reading of `tool-calls` here, and it is what
// the answer retry exists to recover from.
async function* chunks() {
yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } };
yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } };
yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } };
yield { type: "finish", payload: { stepResult: { reason: "tool-calls" }, output: { usage: {} } } };
}
let completed = 0;
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { completed += 1; },
onError: () => {},
});
await response.text();
assert.equal(completed, 0);
assert.equal(state.modelFinishReason, "tool-calls");
assert.equal(state.modelStepCount, 2);
});
test("a retry accumulates model steps and reports the latest finish reason", async () => {
const state = createConsultationRuntimeState({ plannedSteps: 8 });
async function* firstAttempt() {
yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } };
yield { type: "finish", payload: { stepResult: { reason: "tool-calls" }, output: { usage: {} } } };
}
async function* retriedAttempt() {
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
yield { type: "text-delta", payload: { text: "补齐后的回答。" } };
yield { type: "step-finish", payload: { stepResult: { reason: "stop" } } };
yield { type: "finish", payload: { stepResult: { reason: "unrecognized-provider-reason" }, output: { usage: {} } } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: firstAttempt(), requireTool: true,
retry: async () => retriedAttempt(),
toolStatus: () => "ready", receipt: () => receipt(state),
});
await response.text();
assert.equal(state.modelStepCount, 2);
assert.equal(state.modelFinishReason, "unknown");
});
test("a failed run reports the same allowlisted receipt a completed run does", async () => {
const state = createConsultationRuntimeState({ plannedSteps: 8 });
state.modelFinishReason = "tool-calls";
state.modelStepCount = 8;
appendConsultationRuntimeStep(state, {
kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 20936, failureCode: "workflow_queue_full",
});
async function* chunks() {
yield { type: "tool-error", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", error: new Error("boom") } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "blocked",
receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }),
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
const failed = events.find((event) => (event as { type?: string }).type === "run.failed") as {
code: string;
receipt?: { steps: Array<{ durationMs?: number; name: string }>; stepBudget?: { used: number } };
};
assert.equal(failed.code, "runtime_contract_incomplete");
// Without this the caller learned only the code: no step durations, no budget,
// exactly when the run needed explaining most.
// Binding costs no time, so it reports no duration where the old activation
// reported a round trip.
assert.deepEqual(failed.receipt?.steps.map((step) => step.durationMs), [undefined, 20936]);
assert.equal(failed.receipt?.stepBudget?.used, 2);
// The internal classification and the model loop diagnostics stay server-side.
assert.doesNotMatch(JSON.stringify(failed), /workflow_queue_full|modelFinishReason|modelStepCount|tool-calls/);
});
test("a receipt that cannot be built still leaves a failure event", async () => {
const state = createConsultationRuntimeState();
async function* chunks() {
yield { type: "text-delta", payload: { text: "不能保存" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "blocked",
receipt: () => { throw new Error("receipt_unavailable"); },
onError: () => {},
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
const failed = events.filter((event) => (event as { type?: string }).type === "run.failed");
assert.equal(failed.length, 1);
assert.equal("receipt" in (failed[0] as object), false);
});
test("persistence failure emits run.failed instead of run.completed", async () => {
const state = createConsultationRuntimeState();
state.jyotishSkillBound = true;
state.consultationToolCallCount = 1;
state.consultationToolSuccessCount = 1;
state.consultationToolCompleted = true;
state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] };
let failed = 0;
async function* chunks() {
yield { type: "text-delta", payload: { text: "不能在持久化失败后标记完成。" } };
}
const response = streamAgentResponse({
runId: "run", requestId: "req", state, stream: chunks(), requireTool: true,
toolStatus: () => "ready", receipt: () => receipt(state),
onComplete: () => { throw new Error("persistence failed"); },
onError: () => { failed += 1; },
});
const events: unknown[] = [];
const parser = createNdjsonParser((event) => events.push(event));
parser.finish(await response.text());
assert.equal(failed, 1);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 0);
assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1);
});