import assert from "node:assert/strict"; import test from "node:test"; import { AGENT_MAX_STEPS, AGENT_TIMEOUT_MS, CONSULTATION_MAX_OUTPUT_TOKENS, mergeConsultationAnswerPolicies, CONSULTATION_DOMAIN_WALL_CLOCK_MS, MAX_CONSULTATION_DOMAINS, appendConsultationRuntimeStep, canonicalDomainPlan, consultationGenerationSettings, consultationModelStepTelemetry, consultationStepBudgetReceipt, consultationToolFailureCode, createConsultationRuntimeHooks, createConsultationTools, createConsultationRuntimeState, domainFitsRunBudget, executableDomainPlan, publicConsultationRuntimeSteps, } from "../src/mastra/consultation-tools.ts"; import { ConsultationWorkflowError, consultationWorkflowFailureCode, } from "../src/mastra/consultation-workflow.ts"; import { agentExecutionReceiptSchema } from "../src/lib/consultation-agent-events.ts"; import { consultationDomainIds, consultationDomainPlanValues } from "../src/lib/consultation-domain-registry.ts"; import { getJyotishAgent } from "../src/mastra/index.ts"; import { consultationAgentPublicEventSchema, createNdjsonParser } from "../src/lib/consultation-agent-events.ts"; import { createConsultationPlan } from "../src/lib/consultation-plan.ts"; import { collectAgentPublicEvents, streamAgentResponse, } from "../src/lib/stream-agent-response.ts"; const serverChart = { name: "测试", toolInput: { year: 1990, month: 1, day: 2, hour: 3, minute: 4, city: "台北", lat: 25.03, lon: 121.56, tz: 8 }, truth: { birthDate: "1990-01-02", reportedBirthTime: "03:04", activeBirthTime: null, selectedTimeKind: "reported" as const, birthTimeSource: "reported", birthTimeStatus: "reported", placeLabel: "台北", placeCodes: { countryCode: "TW", provinceCode: null, cityCode: null, districtCode: null }, placeId: null, placeType: "city", placeProvider: "profile", latitude: 25.03, longitude: 121.56, timezoneId: "Asia/Taipei", timezoneSource: "profile", timezoneOffset: 8, }, }; // Mastra validates against the model-facing schema before execute() runs, so a // test can only send what the model can send. The cast keeps the argument // checked against that shape without depending on Mastra's inferred type. type ModelConsultationToolInput = { question: string; domains?: string[] }; const modelInput = (input: ModelConsultationToolInput) => input as never; const rejectedByInputSchema = (input: { question: string; theme?: string; domains?: string[] }) => input as never; type WorkflowOptions = { status?: "ready" | "degraded" | "blocked"; missingLayers?: string[]; preciseTiming?: boolean; availableLayers?: string[]; hardBlockers?: string[]; leadWithLimitations?: boolean; limitation?: string; chart?: Record; }; function workflow(theme = "career", options: WorkflowOptions = {}) { return { success: true, question: "综合看看", chart: options.chart ?? {}, routing: { primary_theme: theme }, consumer_context: { route: theme, core_status: options.status ?? "ready", available_layers: options.availableLayers ?? [], missing_route_layers: options.missingLayers ?? [], hard_blockers: options.hardBlockers ?? [], technique_truth: { status: "verified" }, answer_policy: { can_answer_direction: true, can_answer_precise_timing: options.preciseTiming ?? true, ...(options.leadWithLimitations === undefined ? {} : { should_lead_with_limitations: options.leadWithLimitations }), }, ...(options.limitation === undefined ? {} : { user_facing_limitation: options.limitation }), }, }; } // The natal projection only survives the evidence allowlist when the chart // actually carries allowlisted placements, and the hoisting test needs it to. const natalChart = { ascendant: { sign: "Leo", degree: 12.5 }, planets: [{ name: "Sun", sign: "Leo", degree: 1.25 }, { name: "Moon", sign: "Pisces", degree: 20.5 }], houses: [{ number: 1, sign: "Leo" }], }; const toolContext = { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never; type PlanResult = Record & { domains: string[]; omitted_domains: string[]; consultations: Array & { domain: string; claim_cards: Array<{ category: string }> }>; evidence_contract: { available_layers: string[]; missing_route_layers: string[]; hard_blockers: string[]; answer_policy: Record; user_facing_limitation?: string; }; rectification: { boundary: string }; claim_cards: Array<{ category: string }>; }; async function runDomainPlan( domains: string[], runWorkflow: (theme: string) => ReturnType, options: { now?: () => number; requestId?: string } = {}, ) { const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: options.requestId ?? `r-${domains.join("-")}`, consultationMode: "verified_chart", serverChart, state, ...(options.now ? { now: options.now } : {}), runWorkflow: async (input) => runWorkflow(input.theme), })["run-jyotish-consultation"]; const result = await tool.execute!(modelInput({ question: "综合看看", domains }), toolContext) as PlanResult; return { result, state }; } test("the server-selected domain stays authoritative when the model omits domains", async () => { let calls = 0; let captured: unknown; let capturedPlan: unknown; const state = createConsultationRuntimeState(); const plan = createConsultationPlan({ userIntent: "事业如何", theme: "career", consultationMode: "unverified_birth_time", modelCreditCost: 1, }); assert.throws( () => (plan.requestedDomains as unknown as string[]).push("timing"), TypeError, ); const tools = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r", consultationMode: "unverified_birth_time", plan, theme: "career", serverChart, state, runWorkflow: async (input, options) => { calls += 1; captured = input; capturedPlan = options?.plan; return workflow(); }, }); const tool = tools["run-jyotish-consultation"]; // A single-value theme is deliberately absent: two mutually exclusive ways to // name a domain cost the model a step per call to discover the rule. assert.deepEqual(Object.keys((tool.inputSchema as unknown as { shape: object }).shape), ["question", "domains"]); const execute = tool.execute!; const context = { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never; const [first, second] = await Promise.all([ execute(modelInput({ question: "尝试改成精确应期" }), context), execute(modelInput({ question: "尝试改成婚恋" }), context), ]); assert.equal(calls, 1); assert.deepEqual(first, second); assert.deepEqual(captured, { ...serverChart.toolInput, entryMode: "direct_chart", question: "事业如何", theme: "career" }); assert.strictEqual(capturedPlan, plan); assert.equal(state.consultationToolCallCount, 1); assert.equal(state.consultationToolSuccessCount, 1); assert.equal(state.workflowReceipt?.preciseTiming, "allowed"); assert.deepEqual(state.workflowReceipt?.domains, ["career"]); assert.deepEqual((first as { domains?: string[] }).domains, ["career"]); }); test("multi-domain plan canonicalizes aliases, de-duplicates, preserves order, and runs every domain", async () => { const calls: Array<{ theme: string; question: string }> = []; const state = createConsultationRuntimeState(); const tools = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r", consultationMode: "verified_chart", serverChart, state, runWorkflow: async (input) => { calls.push({ theme: input.theme, question: input.question }); if (input.theme === "wealth") return workflow(input.theme, { status: "degraded", missingLayers: ["D11"] }); return workflow(input.theme); }, }); // Aliases, not repetitions: the array bound is now the executable domain cap, // so a duplicate spends one of the slots the clock can actually pay for. // canonicalDomainPlan keeps the de-duplication coverage for longer raw lists. const result = await tools["run-jyotish-consultation"].execute!( modelInput({ question: "事业、财富和迁居怎么一起规划", domains: ["career", "finance", "home"] }), { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never, ) as { domains: string[]; consultations: Array<{ domain: string }> }; assert.deepEqual(calls, [ { theme: "career", question: "事业、财富和迁居怎么一起规划" }, { theme: "wealth", question: "事业、财富和迁居怎么一起规划" }, { theme: "migration", question: "事业、财富和迁居怎么一起规划" }, ]); assert.deepEqual(result.domains, ["career", "wealth", "migration"]); assert.deepEqual(result.consultations.map((item) => item.domain), ["career", "wealth", "migration"]); assert.deepEqual(state.workflowReceipt, { route: "multi-domain", status: "degraded", preciseTiming: "allowed", missingLayers: ["D11"], domains: ["career", "wealth", "migration"], }); }); test("a multi-domain result exposes the same top-level answer contract as a single domain", async () => { const single = await runDomainPlan(["career"], (theme) => workflow(theme, { availableLayers: ["D10"], chart: natalChart, })); const multi = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, { availableLayers: theme === "career" ? ["D10"] : ["D11"], chart: natalChart, })); // Every path jyotishInstructions states an output rule against has to resolve // in both shapes. With none of them present the model has no contract that // authorizes it to speak, which is how a successful calculation produced no // answer at all. for (const key of ["packet_version", "question", "route", "status", "evidence_contract", "claim_cards", "rectification"]) { assert.equal(key in single.result, true, `single-domain result is missing ${key}`); assert.equal(key in multi.result, true, `multi-domain result is missing ${key}`); } for (const contract of [single.result.evidence_contract, multi.result.evidence_contract]) { assert.equal(Array.isArray(contract.available_layers), true); assert.equal(Array.isArray(contract.missing_route_layers), true); assert.equal(Array.isArray(contract.hard_blockers), true); assert.equal(typeof contract.answer_policy.can_answer_direction, "boolean"); assert.equal(typeof contract.answer_policy.can_answer_precise_timing, "boolean"); } assert.equal(single.result.rectification.boundary, "not_auto_rectified"); assert.equal(multi.result.rectification.boundary, "not_auto_rectified"); assert.equal(multi.result.packet_version, single.result.packet_version); assert.equal(multi.result.question, single.result.question); assert.equal(multi.result.route, "multi-domain"); assert.equal(multi.result.status, "ready"); assert.equal(multi.result.success, true); assert.equal(single.result.success, true); // An available layer stays available: it really was computed for one domain. assert.deepEqual(multi.result.evidence_contract.available_layers, ["D10", "D11"]); assert.deepEqual(multi.result.consultations.map((item) => item.domain), ["career", "wealth"]); }); test("the merged answer policy is the most restrictive of the executed domains", async () => { const { result, state } = await runDomainPlan(["career", "timing", "wealth"], (theme) => workflow(theme, { // One domain forbidding precise timing must forbid it for the whole answer. preciseTiming: theme !== "timing", status: theme === "wealth" ? "degraded" : "ready", missingLayers: theme === "wealth" ? ["D11"] : [], hardBlockers: theme === "timing" ? ["negative_holdout_gate"] : [], leadWithLimitations: theme === "timing", limitation: theme === "wealth" ? "财富层证据不完整。" : undefined, chart: natalChart, })); const policy = result.evidence_contract.answer_policy; assert.equal(policy.can_answer_precise_timing, false); assert.equal(policy.can_answer_direction, true); assert.equal(policy.should_lead_with_limitations, true); assert.deepEqual(result.evidence_contract.hard_blockers, ["negative_holdout_gate"]); assert.deepEqual(result.evidence_contract.missing_route_layers, ["D11"]); assert.equal(result.status, "degraded"); assert.equal(result.evidence_contract.user_facing_limitation, "财富层证据不完整。"); assert.equal(state.workflowReceipt?.preciseTiming, "blocked"); assert.equal(state.workflowReceipt?.status, "degraded"); const blocked = await runDomainPlan(["career", "health"], (theme) => workflow(theme, { status: theme === "health" ? "blocked" : "ready", chart: natalChart, })); assert.equal(blocked.result.status, "blocked"); }); test("merging answer policies can only ever restrict", () => { // Merged directly, because the projection currently emits only three policy // fields and the rules have to hold for any field it may emit later. assert.deepEqual( mergeConsultationAnswerPolicies([ { can_answer_direction: true, can_answer_precise_timing: true }, { can_answer_direction: true, can_answer_precise_timing: false }, ]), { can_answer_direction: true, can_answer_precise_timing: false }, ); assert.deepEqual( mergeConsultationAnswerPolicies([ { can_answer_direction: true, can_answer_precise_timing: true, should_lead_with_limitations: false }, { can_answer_direction: false, can_answer_precise_timing: true, should_lead_with_limitations: true }, ]), { can_answer_direction: false, can_answer_precise_timing: true, should_lead_with_limitations: true }, ); // A prohibition list unions: a technique one domain forbids stays forbidden. const prohibitions = mergeConsultationAnswerPolicies([ { can_answer_direction: true, can_answer_precise_timing: true, deterministic_claims_forbidden_for: ["narayana"] }, { can_answer_direction: true, can_answer_precise_timing: true, deterministic_claims_forbidden_for: ["transit", "narayana"] }, ]); assert.deepEqual(prohibitions.deterministic_claims_forbidden_for, ["narayana", "transit"]); // A boolean that is absent for one domain is not consent from that domain. assert.equal( mergeConsultationAnswerPolicies([{ can_answer_chart_interpretation: true }, {}]).can_answer_chart_interpretation, false, ); // A field the domains disagree on in a way that cannot be merged is reported // as unresolved and forces the answer to lead with its limits, rather than // being dropped, which would remove whatever it was restricting. const conflicted = mergeConsultationAnswerPolicies([ { can_answer_direction: true, can_answer_precise_timing: true, claim_ceiling: "direction_only" }, { can_answer_direction: true, can_answer_precise_timing: true, claim_ceiling: "structure_only" }, ]); assert.deepEqual(conflicted.unresolved_policy_fields, ["claim_ceiling"]); assert.equal(conflicted.should_lead_with_limitations, true); assert.equal("claim_ceiling" in conflicted, false); }); test("the merged contract exposes every policy field a single domain exposes", async () => { const options: WorkflowOptions = { leadWithLimitations: false, limitation: "边界说明。", chart: natalChart, }; const single = await runDomainPlan(["career"], (theme) => workflow(theme, options)); const multi = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, options)); // Guards drift: a field added to the per-domain projection without being // merged would silently vanish from the multi-domain contract. for (const key of Object.keys(single.result.evidence_contract.answer_policy)) { assert.equal(key in multi.result.evidence_contract.answer_policy, true, `merged policy is missing ${key}`); } for (const key of Object.keys(single.result.evidence_contract)) { assert.equal(key in multi.result.evidence_contract, true, `merged contract is missing ${key}`); } assert.equal(multi.result.evidence_contract.answer_policy.should_lead_with_limitations, false); }); test("the identical natal projection is carried once instead of per domain", async () => { const { result } = await runDomainPlan(["career", "wealth", "timing"], (theme) => workflow(theme, { chart: natalChart })); assert.deepEqual(result.claim_cards.map((card) => card.category), ["natal_foundation"]); assert.equal( result.consultations.every((item) => item.claim_cards.every((card) => card.category !== "natal_foundation")), true, ); assert.equal(result.consultations.some((item) => item.claim_cards.length > 0), true); // When the domains genuinely disagree, nothing is presented as shared. const differing = await runDomainPlan(["career", "wealth"], (theme) => workflow(theme, { chart: theme === "career" ? natalChart : { ...natalChart, ascendant: { sign: "Virgo", degree: 1 } }, })); assert.deepEqual(differing.result.claim_cards, []); assert.equal( differing.result.consultations.every((item) => item.claim_cards.some((card) => card.category === "natal_foundation")), true, ); }); test("the domain cap is what the run budget can actually pay for", () => { // 21s per sequential domain against the 110s run budget, minus the reserve a // three-domain staging run actually left for composing the answer. assert.equal(MAX_CONSULTATION_DOMAINS, 3); assert.equal(AGENT_TIMEOUT_MS, 110_000); assert.equal(CONSULTATION_DOMAIN_WALL_CLOCK_MS, 65_000); assert.ok(MAX_CONSULTATION_DOMAINS * 21_000 <= CONSULTATION_DOMAIN_WALL_CLOCK_MS); // Six domains, the previous cap, could never finish inside the deadline. assert.ok(6 * 21_000 > AGENT_TIMEOUT_MS); assert.deepEqual( executableDomainPlan(["career", "wealth", "timing", "marriage", "health"]), { domains: ["career", "wealth", "timing"], omittedDomains: ["marriage", "health"] }, ); assert.deepEqual(executableDomainPlan(["career"]), { domains: ["career"], omittedDomains: [] }); // The first domain always runs; after that the next one has to be projected // to finish, judged by how long the executed ones really took. assert.equal(domainFitsRunBudget(0, 0), true); assert.equal(domainFitsRunBudget(21_000, 1), true); assert.equal(domainFitsRunBudget(42_000, 2), true); assert.equal(domainFitsRunBudget(60_000, 2), false); assert.equal(domainFitsRunBudget(40_000, 1), false); }); test("a plan larger than the cap cannot be expressed and never starts a calculation", async () => { let calls = 0; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-cap", consultationMode: "verified_chart", serverChart, state, runWorkflow: async (input) => { calls += 1; return workflow(input.theme); }, })["run-jyotish-consultation"]; const inputSchema = tool.inputSchema as unknown as { safeParse: (value: unknown) => { success: boolean } }; assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career", "wealth", "timing"] }).success, true); assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career", "wealth", "timing", "marriage"] }).success, false); // Mastra rejects the over-budget plan before the tool body runs, so it costs // one correctable step and nothing about the run advances. const refused = await tool.execute!( { question: "全都看看", domains: ["career", "wealth", "timing", "marriage", "health"] } as never, toolContext, ) as Record; assert.equal(calls, 0); assert.equal("domains" in refused, false); assert.equal(state.consultationToolStarted, false); assert.deepEqual(runSteps(state), []); }); test("a plan that runs long stops early and discloses the domains it dropped", async () => { let clock = 0; const executed: string[] = []; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-slow", consultationMode: "verified_chart", serverChart, state, now: () => clock, runWorkflow: async (input) => { executed.push(input.theme); clock += 40_000; return workflow(input.theme, { chart: natalChart }); }, })["run-jyotish-consultation"]; const result = await tool.execute!( modelInput({ question: "三个领域", domains: ["career", "wealth", "timing"] }), toolContext, ) as PlanResult; // 40s each cannot fit a second domain inside the loop's share of the budget, // so the run answers what it has instead of aborting mid-loop and losing it. assert.deepEqual(executed, ["career"]); assert.deepEqual(result.domains, ["career"]); assert.deepEqual(result.omitted_domains, ["wealth", "timing"]); assert.equal(result.status, "degraded"); assert.equal(state.consultationToolCompleted, true); assert.equal(state.consultationToolSuccessCount, 1); assert.equal(state.consultationToolDurationMs, 40_000); // The single executed domain still has to carry the full top-level contract. for (const key of ["packet_version", "route", "status", "evidence_contract", "claim_cards", "rectification"]) { assert.equal(key in result, true, `truncated result is missing ${key}`); } }); test("the advertised domain limit matches the enforced one", () => { const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-description", consultationMode: "verified_chart", serverChart, state: createConsultationRuntimeState(), runWorkflow: async (input) => workflow(input.theme), })["run-jyotish-consultation"]; const description = tool.description ?? ""; assert.match(description, new RegExp(`at most ${MAX_CONSULTATION_DOMAINS} allowlisted`)); assert.doesNotMatch(description, /up to six|six allowlisted/); assert.match(description, /omitted_domains/); assert.match(description, /top-level answer contract/); }); test("domain plan rejects unknown and product domains before any workflow runs", async () => { for (const domain of ["unknown", "prashna", "muhurta", "rectification", "compatibility"]) { let calls = 0; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: `r-${domain}`, consultationMode: "verified_chart", serverChart, state, runWorkflow: async () => { calls += 1; return workflow(); }, })["run-jyotish-consultation"]; // Refused by the enumerated schema before execute, so Mastra resolves with a // validation envelope instead of the tool throwing from the registry check. const rejected = await tool.execute!( modelInput({ question: "测试", domains: ["career", domain] }), { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never, ) as { error?: unknown }; assert.equal(rejected.error, true, domain); assert.equal(calls, 0); assert.equal(state.consultationToolCompleted, false); } }); test("the model-facing schema names the domain vocabulary it accepts", async () => { const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-vocabulary", consultationMode: "verified_chart", serverChart, state, runWorkflow: async () => workflow(), })["run-jyotish-consultation"]; const inputSchema = tool.inputSchema as unknown as { safeParse: (value: unknown) => { success: boolean }; shape: { domains: { unwrap: () => { element: { options?: readonly string[] } } } }; }; // The skill's methodology names strict-workflow checklists, and while this was // a free-form string those labels passed validation and died inside the call. // Enumerating the values is what puts the vocabulary in front of the model. assert.equal(inputSchema.safeParse({ question: "测试", domains: ["event-timing-strict"] }).success, false); assert.equal(inputSchema.safeParse({ question: "测试", domains: ["wealth-timing-strict"] }).success, false); assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"] }).success, true); // Aliases stay accepted: enumerating states the vocabulary, it does not narrow it. assert.equal(inputSchema.safeParse({ question: "测试", domains: ["finance"] }).success, true); assert.equal(inputSchema.safeParse({ question: "测试", domains: ["感情"] }).success, true); // Enumerable, so the JSON schema handed to the model carries the values rather // than an opaque string. A wrapper that hid them would pass the checks above. const options = inputSchema.shape.domains.unwrap().element.options; assert.deepEqual(consultationDomainPlanValues, options); for (const id of consultationDomainIds) assert.ok(options?.includes(id), id); }); test("domain plan enforces the raw plan upper bound and one input mode", async () => { let calls = 0; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-limit", consultationMode: "verified_chart", serverChart, state, runWorkflow: async () => { calls += 1; return workflow(); }, })["run-jyotish-consultation"]; const context = { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never; const inputSchema = tool.inputSchema as unknown as { safeParse: (value: unknown) => { success: boolean } }; assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career", "career", "career", "career", "career", "career", "career"], }).success, false); assert.equal(calls, 0); // The model can only express a domain plan one way. The pair that used to be // representable, and cost a model step to be told was invalid, is now refused // by the schema before the tool body runs. assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"] }).success, true); assert.equal(inputSchema.safeParse({ question: "测试" }).success, true); assert.equal(inputSchema.safeParse({ question: "测试", theme: "career" }).success, false); assert.equal(inputSchema.safeParse({ question: "测试", domains: ["career"], theme: "career" }).success, false); const rejectedState = createConsultationRuntimeState(); const secondTool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-modes", consultationMode: "verified_chart", serverChart, state: rejectedState, runWorkflow: async () => { calls += 1; return workflow(); }, })["run-jyotish-consultation"]; const refused = await secondTool.execute!( rejectedByInputSchema({ question: "测试", domains: ["career"], theme: "career" }), context, ) as Record; // Nothing about the run advances: no calculation, no counted attempt, and no // consultation payload the model could mistake for a result. assert.equal(calls, 0); assert.equal("domains" in refused, false); assert.equal(rejectedState.consultationToolStarted, false); assert.equal(rejectedState.consultationToolCallCount, 0); assert.deepEqual(runSteps(rejectedState), []); }); test("the single-value domain form stays available to callers without the model schema", () => { const plan = createConsultationPlan({ userIntent: "事业如何", theme: "career", consultationMode: "verified_chart", modelCreditCost: 1, }); // A single-value theme must never override the route-selected server domain. assert.deepEqual(canonicalDomainPlan({ theme: "timing" }, { plan, theme: "career" }), ["career"]); assert.deepEqual(canonicalDomainPlan({}, { plan, theme: "career" }), ["career"]); assert.deepEqual(canonicalDomainPlan({ theme: "marriage" }, {}), ["marriage"]); assert.deepEqual(canonicalDomainPlan({ domains: ["career", "finance"] }, {}), ["career", "wealth"]); assert.deepEqual(canonicalDomainPlan({ domains: ["career", "finance", "career", "home"] }, {}), ["career", "wealth", "migration"]); assert.deepEqual(canonicalDomainPlan({ domains: ["timing"] }, { plan, theme: "career" }), ["timing"]); assert.throws( () => canonicalDomainPlan({ domains: ["career"], theme: "career" }, {}), /invalid_consultation_domain_plan/, ); assert.throws(() => canonicalDomainPlan({}, {}), /invalid_consultation_domain_plan/); assert.throws(() => canonicalDomainPlan({ theme: "prashna" }, {}), /unsupported_consultation_domain/); }); test("invalid model input does not poison a later valid contract retry", async () => { let calls = 0; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-retry", consultationMode: "verified_chart", serverChart, state, runWorkflow: async (input) => { calls += 1; return workflow(input.theme); }, })["run-jyotish-consultation"]; const context = { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never; // BUG-205: a bad domain must be refused before the request-scoped calculation // cache is written. The refusal now happens at the schema, one layer earlier // than the registry check it used to reach, because the domain ids are // enumerated in the schema the model is handed. Mastra reports that refusal by // resolving with a validation envelope instead of throwing, so this asserts the // envelope rather than a rejection. const rejected = await tool.execute!( modelInput({ question: "先给出错误参数", domains: ["career", "unknown"] }), context, ) as { error?: unknown; message?: unknown }; assert.equal(rejected.error, true); // The envelope has to name the legal ids: it is the only correction the model // gets, and an unnamed vocabulary is what produced the invalid call. assert.match(String(rejected.message), /'career'/); assert.match(String(rejected.message), /'timing'/); assert.equal(calls, 0); assert.equal(state.consultationToolCallCount, 0); assert.equal(state.consultationToolSuccessCount, 0); const result = await tool.execute!(modelInput({ question: "改用合法参数", domains: ["timing"] }), context) as { domains: string[] }; assert.equal(calls, 1); assert.deepEqual(result.domains, ["timing"]); assert.equal(state.consultationToolCallCount, 1); assert.equal(state.consultationToolSuccessCount, 1); assert.equal(state.consultationToolCompleted, true); }); test("a rejected workflow promise is cleared before a later tool call", async () => { let calls = 0; const state = createConsultationRuntimeState(); const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "r-rejected-promise", consultationMode: "verified_chart", serverChart, state, runWorkflow: async (input) => { calls += 1; if (calls === 1) throw new Error("workflow_temporarily_failed"); return workflow(input.theme); }, })["run-jyotish-consultation"]; const context = { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never; await assert.rejects( tool.execute!(modelInput({ question: "第一次计算", domains: ["career"] }), context), /workflow_temporarily_failed/, ); const result = await tool.execute!(modelInput({ question: "重新计算", domains: ["timing"] }), context) as { domains: string[] }; assert.equal(calls, 2); assert.deepEqual(result.domains, ["timing"]); assert.equal(state.consultationToolCallCount, 2); assert.equal(state.consultationToolSuccessCount, 1); }); test("a failed calculation records why it failed and forwards the request id", async () => { const state = createConsultationRuntimeState(); const seen: Array = []; const tool = createConsultationTools({ userId: "u", sessionId: "s", requestId: "req-correlation", consultationMode: "verified_chart", serverChart, state, runWorkflow: async (_input, options) => { seen.push(options?.requestId); throw new ConsultationWorkflowError("workflow_queue_full", "Async job queue is full"); }, })["run-jyotish-consultation"]; await assert.rejects( tool.execute!(modelInput({ question: "队列满时的表现", domains: ["career"] }), { observe: { span: async (_n: string, fn: () => Promise) => fn(), log() {} } } as never), /Async job queue is full/, ); assert.deepEqual(seen, ["req-correlation"]); const failedStep = state.steps.find((step) => step.status === "failed"); assert.equal(failedStep?.name, "run-jyotish-consultation"); assert.equal(failedStep?.failureCode, "workflow_queue_full"); }); test("workflow failure codes classify transport and contract faults", () => { assert.equal(consultationWorkflowFailureCode(new ConsultationWorkflowError("workflow_rate_limited", "x")), "workflow_rate_limited"); assert.equal(consultationWorkflowFailureCode(new DOMException("slow", "TimeoutError")), "workflow_timeout"); assert.equal(consultationWorkflowFailureCode(new DOMException("stop", "AbortError")), "workflow_aborted"); assert.equal(consultationWorkflowFailureCode(new Error("anything else")), undefined); }); test("every tool failure resolves to a code, not an absent field", () => { // The workflow classifier returns undefined for anything it does not own, and // the append site omitted the field when it was undefined, so the failures // raised inside the tool reached the log as the one record with no reason. assert.equal(consultationToolFailureCode(new ConsultationWorkflowError("workflow_rate_limited", "x")), "workflow_rate_limited"); assert.equal(consultationToolFailureCode(new DOMException("stop", "AbortError")), "workflow_aborted"); assert.equal(consultationToolFailureCode(new Error("invalid_consultation_domain_plan")), "invalid_domain_plan"); assert.equal(consultationToolFailureCode(new Error("unsupported_consultation_domain")), "invalid_domain_plan"); assert.equal(consultationToolFailureCode(new Error("anything else")), "unexpected_error"); assert.equal(consultationToolFailureCode("not an error"), "unexpected_error"); }); test("a call rejected before the tool body runs still appears in the receipt", async () => { // Observed on staging: the model's arguments were refused against the tool's // strict input schema, so `execute` never ran. The client saw tool.failed // while the receipt showed no failed step and the budget counted no call. const state = createConsultationRuntimeState(); async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation" } }; yield { type: "tool-error", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation", error: new Error("bad arguments") } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "blocked", receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }), onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); const failedStep = state.steps.find((step) => step.kind === "tool" && step.status === "failed"); assert.equal(failedStep?.name, "run-jyotish-consultation"); assert.equal(failedStep?.failureCode, "tool_call_rejected"); assert.equal(consultationStepBudgetReceipt(state).used, 2); // The client learns a step failed; the classification stays server-side. const failed = events.find((event) => (event as { type?: string }).type === "run.failed") as { receipt?: { steps: Array<{ status: string; name: string }> }; }; assert.deepEqual(failed.receipt?.steps.map((step) => step.status), ["completed", "failed"]); assert.doesNotMatch(JSON.stringify(failed), /tool_call_rejected/); }); test("a failure the tool already recorded is not recorded twice", async () => { // The tool records what it can see, with the duration and cause it alone // knows. The stream must only fill the gap, never double-count. const state = createConsultationRuntimeState(); appendConsultationRuntimeStep(state, { kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 20936, failureCode: "workflow_queue_full", }); async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation" } }; yield { type: "tool-error", payload: { toolCallId: "call-1", toolName: "run-jyotish-consultation", error: new Error("boom") } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "blocked", receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }), onError: () => {}, }); await response.text(); const failed = state.steps.filter((step) => step.status === "failed"); assert.equal(failed.length, 1); assert.equal(failed[0]?.failureCode, "workflow_queue_full"); }); test("the public receipt never carries the internal failure classification", () => { const state = createConsultationRuntimeState(); appendConsultationRuntimeStep(state, { kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 12, failureCode: "workflow_server_error", }); const steps = publicConsultationRuntimeSteps(state); assert.equal(steps.every((step) => !("failureCode" in step)), true); assert.equal(state.steps.find((step) => step.kind === "tool")?.failureCode, "workflow_server_error"); // A strict receipt schema would reject the internal field, so this also // guards the run from failing while building a successful response. const receipt = agentExecutionReceiptSchema.parse({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps, workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }, }); assert.equal(receipt.steps.length, 2); assert.throws(() => agentExecutionReceiptSchema.parse({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: state.steps, workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }, })); }); test("the public receipt never carries the model step budget diagnostics", () => { const state = createConsultationRuntimeState(); state.modelStepCount = 8; state.modelFinishReason = "tool-calls"; appendConsultationRuntimeStep(state, { kind: "skill", name: "jyotish-vedic-astrology", status: "completed" }); assert.deepEqual( consultationModelStepTelemetry(state), { modelStepCount: 8, skillReferenceReads: 0, methodologySections: 0, modelFinishReason: "tool-calls" }, ); assert.deepEqual( consultationModelStepTelemetry(createConsultationRuntimeState()), { modelStepCount: 0, skillReferenceReads: 0, methodologySections: 0 }, ); // The client receipt schema is strict, so leaking either field would make a // successful run fail while serializing its own answer. const receipt = agentExecutionReceiptSchema.parse({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: publicConsultationRuntimeSteps(state), stepBudget: consultationStepBudgetReceipt(state), workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }, }); assert.doesNotMatch(JSON.stringify(receipt), /modelStepCount|modelFinishReason|tool-calls/); assert.throws(() => agentExecutionReceiptSchema.parse({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: publicConsultationRuntimeSteps(state), workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }, ...consultationModelStepTelemetry(state), })); }); test("the receipt reports how many reference documents the model opened", () => { const state = createConsultationRuntimeState(); const hooks = createConsultationRuntimeHooks(state); // The method is bound before the model runs, so a run starts with it in hand and // with nothing read. Having the method says nothing about the model having gone // past it to a reference of its own, which is the number this counts. assert.equal(state.jyotishSkillBound, true); assert.equal(state.skillReferenceReadCount, 0); hooks.afterToolCall({ toolName: "skill_read" }); hooks.afterToolCall({ toolName: "read_file" }); hooks.afterToolCall({ toolName: "skill_read", error: new Error("denied") }); assert.equal(state.skillReferenceReadCount, 2); // Reference reads are not runtime steps, so the step list cannot answer this on its own. assert.equal(state.steps.filter((step) => step.kind === "skill").length, 1); assert.equal(agentExecutionReceiptSchema.parse(receipt(state)).skill.referenceReads, 2); }); test("the receipt separates method the server delivered from method the model went looking for", () => { const state = createConsultationRuntimeState(); // A run where the model opened nothing is no longer a run composed without method: the strict // checklist for the route travels with the evidence, so the two counts have to be readable apart. state.methodologySectionCount = 3; const parsed = agentExecutionReceiptSchema.parse(receipt(state)); assert.equal(parsed.skill.referenceReads, 0); assert.equal(parsed.skill.methodologySections, 3); assert.equal(consultationModelStepTelemetry(state).methodologySections, 3); }); test("personal Agent exposes the Jyotish Skill and named server tool", async () => { const state = createConsultationRuntimeState(); const agent = getJyotishAgent({ id: "personal-agent-probe", label: "Probe", description: "", creditCost: 1, isDefault: false, mode: "openai", model: "openai/gpt-5-mini", } as never, { userId: "u", sessionId: "s", requestId: "r", consultationMode: "verified_chart", serverChart, state, } as never); const skills = await agent.listSkills(); const toolNames = Object.keys(await agent.getToolsForExecution({ runId: "r" })); assert.equal(skills.some((skill) => skill.name === "jyotish-vedic-astrology"), true); // Activation is withdrawn: answering an activation meant resending the whole // package listing every turn, and the method is bound into the instructions // instead. Reading a named reference is still the model's own to do. assert.equal(toolNames.includes("skill"), false); assert.equal(toolNames.includes("skill_search"), false); assert.equal(toolNames.includes("skill_read"), true); assert.equal(toolNames.includes("run-jyotish-consultation"), true); assert.equal(toolNames.includes("consultationTool"), false); }); test("public stream filters private chunks and completes once", async () => { const chunks = [ { type: "reasoning-delta", payload: { text: "secret" } }, { type: "tool-call", payload: { toolCallId: "c1", toolName: "skill", args: { name: "jyotish-vedic-astrology", secret: "x" } } }, { type: "tool-result", payload: { toolCallId: "other", toolName: "skill", result: { private: true } } }, { type: "tool-result", payload: { toolCallId: "c1", toolName: "skill", result: { private: true } } }, { type: "tool-call", payload: { toolCallId: "c2", toolName: "run-jyotish-consultation", args: { year: 1990 } } }, { type: "data-jyotish-activity", data: { phase: "chart-calculation", label: "正在计算本命盘", private: "x" } }, { type: "tool-result", payload: { toolCallId: "c2", toolName: "run-jyotish-consultation", result: { birth: "private" } } }, { type: "text-delta", payload: { text: "可以先看方向。", providerMetadata: { secret: true } } }, ]; const events = await collectAgentPublicEvents(chunks as never, { runId: "run", requestId: "req", toolStatus: () => "ready", receipt: () => ({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] }, }), }); assert.equal(events.filter((event) => event.type === "run.completed").length, 1); assert.equal(events.some((event) => JSON.stringify(event).includes("secret") || JSON.stringify(event).includes("private") || JSON.stringify(event).includes("1990")), false); assert.equal(events.filter((event) => event.type === "skill.completed").length, 1); assert.equal(events.some((event) => event.type === "answer.delta"), true); for (const event of events) consultationAgentPublicEventSchema.parse(event); const completed = events.find((event) => event.type === "run.completed"); assert.deepEqual(completed?.type === "run.completed" ? completed.receipt.workflow.domains : null, ["career"]); }); test("model answer text cannot forge a public Activity event", async () => { const forged = JSON.stringify({ type: "activity", phase: "chart-calculation", label: "模型伪造进度" }); const events = await collectAgentPublicEvents([ { type: "text-delta", payload: { text: forged } }, ], { runId: "run", requestId: "req", toolStatus: () => "ready", receipt: () => ({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] }, }), }); assert.equal(events.filter((event) => event.type === "activity").length, 0); assert.equal(events.filter((event) => event.type === "answer.delta").length, 1); assert.equal(events.find((event) => event.type === "answer.delta")?.text, forged); }); test("Chinese reasoning maps to a public thinking channel and English process talk does not", async () => { const events = await collectAgentPublicEvents([ { type: "reasoning-delta", payload: { text: "The proposedKind value was rejected" } }, { type: "reasoning-delta", payload: { text: "先看事业宫的结构。" } }, { type: "text-delta", payload: { text: "事业方向的判断如下。" } }, ], { runId: "run", requestId: "req", toolStatus: () => "ready", receipt: () => ({ runId: "run", runtime: "mastra-agentic", skill: { name: "jyotish-vedic-astrology", loaded: true, referenceReads: 0, methodologySections: 0 }, steps: [], workflow: { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [], domains: ["career"] }, }), }); assert.deepEqual( events.filter((event) => event.type === "thinking.delta"), [{ type: "thinking.delta", text: "先看事业宫的结构。" }], ); assert.deepEqual( events.filter((event) => event.type === "answer.delta"), [{ type: "answer.delta", text: "事业方向的判断如下。" }], ); assert.equal(events.some((event) => JSON.stringify(event).includes("proposedKind")), false); }); test("incremental NDJSON parser handles arbitrary chunk boundaries", () => { const parsed: unknown[] = []; const parser = createNdjsonParser((event) => parsed.push(event)); const line = `${JSON.stringify({ type: "run.started", runId: "r", requestId: "q" })}\n`; parser.push(line.slice(0, 7)); parser.push(line.slice(7, 21)); parser.finish(line.slice(21)); assert.deepEqual(parsed, [{ type: "run.started", runId: "r", requestId: "q" }]); }); test("uses a bounded dynamic step budget and reports truncation", () => { const state = createConsultationRuntimeState({ plannedSteps: 1, reservedValidationSteps: 1 }); assert.deepEqual(state.stepBudget, { planned: 1, reservedValidation: 1, total: 2 }); // Binding the method is the run's first recorded step and is already present. assert.deepEqual(state.steps.map((step) => step.kind), ["skill"]); assert.equal(appendConsultationRuntimeStep(state, { kind: "validation", name: "ensure-final-response", status: "completed" }), true); assert.equal(appendConsultationRuntimeStep(state, { kind: "tool", name: "unexpected-extra-step", status: "completed" }), false); assert.equal(state.steps.length, 2); assert.equal(state.stepsTruncated, true); assert.deepEqual(consultationStepBudgetReceipt(state), { planned: 2, used: 2, remaining: 0, truncated: true }); }); /** * The steps a run took, without the method binding every run starts with. The * binding is recorded when the state is created, so a test asking whether a run * advanced has to say which steps it means. */ function runSteps(state: ReturnType) { return state.steps.filter((step) => step.kind !== "skill"); } function receipt(state: ReturnType) { return { runId: "run", runtime: "mastra-agentic" as const, skill: { name: "jyotish-vedic-astrology" as const, loaded: state.jyotishSkillBound, referenceReads: state.skillReferenceReadCount, methodologySections: state.methodologySectionCount, }, steps: state.steps, stepBudget: consultationStepBudgetReceipt(state), workflow: state.workflowReceipt ?? { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }, }; } test("text written before the contract completes is dropped, not released later", async () => { const state = createConsultationRuntimeState(); let completed = 0; async function* chunks() { // Production shape (run a5f4409e): between rejected calls the model narrates // its own tool errors. Holding that text meant the eventual success released // it as the visible answer, so a recovered run read as the model explaining // itself and never answering the question. yield { type: "text-delta", payload: { text: "域名单有误,我改为不指定域。" } }; yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } }; state.jyotishSkillBound = true; yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } }; yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: {} } }; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; yield { type: "text-delta", payload: { text: "这是真正的回答。" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 1); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1); const answer = events .filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta") .map((event) => event.text) .join(""); assert.equal(answer, "这是真正的回答。"); assert.doesNotMatch(JSON.stringify(events), /域名单有误/); }); test("a run that only narrated its failures is not delivered or billed as an answer", async () => { const state = createConsultationRuntimeState(); let completed = 0; async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } }; state.jyotishSkillBound = true; yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } }; yield { type: "text-delta", payload: { text: "两次域名单都不被服务端接受,我改为不指定域。" } }; yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: {} } }; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "general", status: "ready", preciseTiming: "allowed", missingLayers: [] }; yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); // Presenting the narration as the reading is dishonest, and so is charging for // a fixed apology that says there is nothing to say. assert.equal(completed, 0); assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false); const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string }; assert.equal(failure.code, "empty_answer"); assert.doesNotMatch(JSON.stringify(events), /不被服务端接受/); }); test("a call Mastra rejected against the input schema is not reported as completed", async () => { const state = createConsultationRuntimeState({ plannedSteps: 8 }); async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", args: { domains: ["event-timing-strict"] } } }; // Mastra resolves rather than throws when arguments fail inputSchema, so the // tool body never runs and cannot record anything. Reported as completed this // would claim a calculation that never happened. yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: { error: true, message: "Tool input validation failed", validationErrors: { errors: [], fields: {} } }, }, }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }), onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(events.some((event) => (event as { type?: string }).type === "tool.completed"), false); const failed = events.find((event) => (event as { type?: string }).type === "tool.failed") as { code: string }; assert.equal(failed.code, "calculation_failed"); assert.deepEqual( state.steps.map((step) => `${step.kind}:${step.status}`), ["skill:completed", "tool:failed"], ); assert.equal(runSteps(state)[0]?.failureCode, "tool_call_rejected"); // The rejection reason is a server-side diagnostic; the client sees only that a step failed. assert.doesNotMatch(JSON.stringify(events), /tool_call_rejected|validationErrors/); }); test("a calculation that succeeds only after failed attempts still satisfies the contract", async () => { const state = createConsultationRuntimeState(); let completed = 0; async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } }; state.jyotishSkillBound = true; yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } }; // Two transient workflow failures, then one success, as observed in production. state.consultationToolCallCount = 3; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; yield { type: "tool-result", payload: { toolCallId: "tool-3", toolName: "run-jyotish-consultation", result: {} } }; yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 1); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 0); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1); assert.equal((events.find((event) => (event as { type?: string }).type === "answer.delta") as { text?: string }).text, "事业方向的判断如下。"); }); test("a second successful calculation still fails the single-calculation boundary", async () => { const state = createConsultationRuntimeState(); let failed = 0; async function* chunks() { yield { type: "tool-call", payload: { toolCallId: "skill-1", toolName: "skill", args: { name: "jyotish-vedic-astrology" } } }; state.jyotishSkillBound = true; yield { type: "tool-result", payload: { toolCallId: "skill-1", toolName: "skill", result: {} } }; state.consultationToolCallCount = 2; state.consultationToolSuccessCount = 2; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; yield { type: "text-delta", payload: { text: "不应显示" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onError: () => { failed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(failed, 1); assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1); }); test("incomplete runtime contract fails without saving a successful answer", async () => { const state = createConsultationRuntimeState(); let completed = 0; let failed = 0; async function* chunks() { yield { type: "text-delta", payload: { text: "不能保存" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "blocked", receipt: () => receipt(state), onComplete: () => { completed += 1; }, onError: () => { failed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 0); assert.equal(failed, 1); assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1); }); function toolOnlyRunState() { const state = createConsultationRuntimeState(); state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; return state; } test("a calculation the model never wrote up is asked again instead of apologised for", async () => { // Production shape: the tool succeeded in 63.5s and the model then produced no // text at all. That used to be answered with a fixed apology and billed as a // completed consultation, so the user paid for a sentence saying nothing. const state = toolOnlyRunState(); let completedOutput = ""; let retries = 0; async function* chunks() { yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; } async function* answerChunks() { yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), retryForAnswer: async () => { retries += 1; return answerChunks(); }, onComplete: (output) => { completedOutput = output; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(retries, 1); assert.equal(completedOutput, "事业方向的判断如下。"); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1); assert.equal(state.steps.at(-1)?.name, "answer-retry"); }); test("a calculation still unanswered after the retry fails the run rather than billing it", async () => { const state = toolOnlyRunState(); let completed = 0; async function* chunks() { yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; } async function* silentChunks() { yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), retryForAnswer: async () => silentChunks(), onComplete: () => { completed += 1; }, onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 0); assert.equal(events.some((event) => (event as { type?: string }).type === "answer.delta"), false); const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string; message: string }; assert.equal(failure.code, "empty_answer"); assert.match(failure.message, /不会扣点/); }); test("a length-limited answer is not billed or delivered as a completed consultation", async () => { // Staging persisted a 253-character pinch that ended mid-heading, then treated // the run as completed. The model had finished with reason `length`; the public // stream still emitted `run.completed`, so the composer unlocked as if the // reading were done. const state = toolOnlyRunState(); const pinchedHeading = "**先看命盘结构(Lahiri岁差、均交点口径"; let completed = 0; let failed = 0; async function* chunks() { yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; yield { type: "text-delta", payload: { text: pinchedHeading } }; yield { type: "finish", payload: { stepResult: { reason: "length" }, output: { usage: {}, steps: [{}, {}] } } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, onError: () => { failed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 0); assert.equal(failed, 1); assert.equal(state.modelFinishReason, "length"); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 0); const answer = events .filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta") .map((event) => event.text) .join(""); assert.equal(answer, pinchedHeading); const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string; message: string; }; assert.equal(failure.code, "answer_truncated"); assert.match(failure.message, /回答未完成/); assert.match(failure.message, /不会扣点/); assert.doesNotMatch(JSON.stringify(failure), /modelFinishReason|"length"/); }); test("a timeout after partial visible text is the same truncation, not a successful answer", async () => { const state = toolOnlyRunState(); const pinchedHeading = "**先看命盘结构(Lahiri岁差、均交点口径"; let completed = 0; async function* chunks() { yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; yield { type: "text-delta", payload: { text: pinchedHeading } }; throw new DOMException("The operation was aborted due to timeout", "TimeoutError"); } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(completed, 0); const answer = events .filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta") .map((event) => event.text) .join(""); assert.equal(answer, pinchedHeading); const failure = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string }; assert.equal(failure.code, "answer_truncated"); }); test("consult generation reserves visible output tokens and enables a separate thinking channel", () => { const settings = consultationGenerationSettings("deepseek"); assert.equal(CONSULTATION_MAX_OUTPUT_TOKENS, 8192); assert.equal(settings.modelSettings.maxOutputTokens, CONSULTATION_MAX_OUTPUT_TOKENS); assert.deepEqual(settings.providerOptions.openai, { thinking: { type: "enabled" } }); assert.deepEqual(settings.providerOptions.deepseek, { thinking: { type: "enabled" } }); assert.equal(AGENT_MAX_STEPS, 8); }); test("Chinese thinking stays off the spoken answer and does not bill a thought-only run", async () => { const state = createConsultationRuntimeState(); state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; async function* chunks() { yield { type: "reasoning-delta", payload: { text: "The proposedKind value was rejected" } }; yield { type: "reasoning-delta", payload: { text: "先看事业宫的结构。" } }; yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.deepEqual( events.filter((event) => (event as { type?: string }).type === "thinking.delta"), [{ type: "thinking.delta", text: "先看事业宫的结构。" }], ); const answer = events .filter((event): event is { type: string; text: string } => (event as { type?: string }).type === "answer.delta") .map((event) => event.text) .join(""); assert.equal(answer, "事业方向的判断如下。"); assert.doesNotMatch(JSON.stringify(events), /proposedKind/); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 1); }); test("a completed run records the finish reason and the authoritative step count", async () => { const state = createConsultationRuntimeState(); state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; async function* chunks() { yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } }; yield { type: "text-delta", payload: { text: "事业方向的判断如下。" } }; yield { type: "step-finish", payload: { stepResult: { reason: "stop" } } }; // The terminal chunk carries the runtime's own step list, which wins over // the chunks we counted. yield { type: "finish", payload: { stepResult: { reason: "stop" }, output: { usage: {}, steps: [{}, {}, {}] } } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), }); await response.text(); assert.equal(state.modelFinishReason, "stop"); assert.equal(state.modelStepCount, 3); }); test("a run that stops while still wanting tools records the exhausted step budget", async () => { const state = createConsultationRuntimeState({ plannedSteps: 8 }); state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; // The production shape: the calculation succeeded, the model never wrote an // answer, and only the fallback text reached the user. Nothing in the public // event stream said the step budget ran out. A model that stopped while it // still wanted tool calls is the reading of `tool-calls` here, and it is what // the answer retry exists to recover from. async function* chunks() { yield { type: "tool-result", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", result: {} } }; yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } }; yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } }; yield { type: "finish", payload: { stepResult: { reason: "tool-calls" }, output: { usage: {} } } }; } let completed = 0; const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { completed += 1; }, onError: () => {}, }); await response.text(); assert.equal(completed, 0); assert.equal(state.modelFinishReason, "tool-calls"); assert.equal(state.modelStepCount, 2); }); test("a retry accumulates model steps and reports the latest finish reason", async () => { const state = createConsultationRuntimeState({ plannedSteps: 8 }); async function* firstAttempt() { yield { type: "step-finish", payload: { stepResult: { reason: "tool-calls" } } }; yield { type: "finish", payload: { stepResult: { reason: "tool-calls" }, output: { usage: {} } } }; } async function* retriedAttempt() { state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; yield { type: "text-delta", payload: { text: "补齐后的回答。" } }; yield { type: "step-finish", payload: { stepResult: { reason: "stop" } } }; yield { type: "finish", payload: { stepResult: { reason: "unrecognized-provider-reason" }, output: { usage: {} } } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: firstAttempt(), requireTool: true, retry: async () => retriedAttempt(), toolStatus: () => "ready", receipt: () => receipt(state), }); await response.text(); assert.equal(state.modelStepCount, 2); assert.equal(state.modelFinishReason, "unknown"); }); test("a failed run reports the same allowlisted receipt a completed run does", async () => { const state = createConsultationRuntimeState({ plannedSteps: 8 }); state.modelFinishReason = "tool-calls"; state.modelStepCount = 8; appendConsultationRuntimeStep(state, { kind: "tool", name: "run-jyotish-consultation", status: "failed", durationMs: 20936, failureCode: "workflow_queue_full", }); async function* chunks() { yield { type: "tool-error", payload: { toolCallId: "tool-1", toolName: "run-jyotish-consultation", error: new Error("boom") } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "blocked", receipt: () => ({ ...receipt(state), steps: publicConsultationRuntimeSteps(state) }), onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); const failed = events.find((event) => (event as { type?: string }).type === "run.failed") as { code: string; receipt?: { steps: Array<{ durationMs?: number; name: string }>; stepBudget?: { used: number } }; }; assert.equal(failed.code, "runtime_contract_incomplete"); // Without this the caller learned only the code: no step durations, no budget, // exactly when the run needed explaining most. // Binding costs no time, so it reports no duration where the old activation // reported a round trip. assert.deepEqual(failed.receipt?.steps.map((step) => step.durationMs), [undefined, 20936]); assert.equal(failed.receipt?.stepBudget?.used, 2); // The internal classification and the model loop diagnostics stay server-side. assert.doesNotMatch(JSON.stringify(failed), /workflow_queue_full|modelFinishReason|modelStepCount|tool-calls/); }); test("a receipt that cannot be built still leaves a failure event", async () => { const state = createConsultationRuntimeState(); async function* chunks() { yield { type: "text-delta", payload: { text: "不能保存" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "blocked", receipt: () => { throw new Error("receipt_unavailable"); }, onError: () => {}, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); const failed = events.filter((event) => (event as { type?: string }).type === "run.failed"); assert.equal(failed.length, 1); assert.equal("receipt" in (failed[0] as object), false); }); test("persistence failure emits run.failed instead of run.completed", async () => { const state = createConsultationRuntimeState(); state.jyotishSkillBound = true; state.consultationToolCallCount = 1; state.consultationToolSuccessCount = 1; state.consultationToolCompleted = true; state.workflowReceipt = { route: "career", status: "ready", preciseTiming: "blocked", missingLayers: [] }; let failed = 0; async function* chunks() { yield { type: "text-delta", payload: { text: "不能在持久化失败后标记完成。" } }; } const response = streamAgentResponse({ runId: "run", requestId: "req", state, stream: chunks(), requireTool: true, toolStatus: () => "ready", receipt: () => receipt(state), onComplete: () => { throw new Error("persistence failed"); }, onError: () => { failed += 1; }, }); const events: unknown[] = []; const parser = createNdjsonParser((event) => events.push(event)); parser.finish(await response.text()); assert.equal(failed, 1); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.completed").length, 0); assert.equal(events.filter((event) => (event as { type?: string }).type === "run.failed").length, 1); });