Files
Jyotisha/frontend/src/lib/report-public-projection.ts
T
jesse-ux 3c2f7bd559 fix(report): use the upstream reader edition for new longform bodies
New reports request reader_main from upstream origin/main 23b9609e.
KP and transit no longer copy an empty vars() dict, solar returns keep
birth_asc_sign_idx, and ordinary projection deletes internal lines whole.
2026-09-25 02:21:46 +08:00

965 lines
32 KiB
TypeScript

/**
* Ordinary-user report projection.
*
* Document kind is an explicit argument. A route name never decides visibility.
* Public text is assembled from an allowlist. Internal keys are not copied;
* a limitation that must survive is rewritten in natural language.
* Professional reference has no separate permission in this release, so that
* kind fail-safes to the same ordinary projection instead of leaking.
*/
export const REPORT_DOCUMENT_KINDS = [
"chat_export",
"personal_report_detail",
"ordinary_markdown_download",
"professional_reference",
] as const;
export type ReportDocumentKind = (typeof REPORT_DOCUMENT_KINDS)[number];
export const ORDINARY_REPORT_DOCUMENT_KINDS = [
"chat_export",
"personal_report_detail",
"ordinary_markdown_download",
] as const;
export type OrdinaryReportDocumentKind = (typeof ORDINARY_REPORT_DOCUMENT_KINDS)[number];
/** Fields an ordinary report may contain. Anything else is not copied. */
export const ORDINARY_PUBLIC_FIELDS = [
"title",
"prose",
"conclusion",
"action",
"limitation",
"chart_fence",
"engine_svg",
] as const;
/** Classified internal. Never serialized into ordinary output. */
export const INTERNAL_REPORT_FIELDS = [
"technique_truth",
"workflow_route",
"workflow_status",
"precise_timing",
"missing_layers",
"score",
"weight",
"execution_ledger",
"internal_url",
"secret",
"model_debug",
"raw_tool_response",
"job",
"attempt",
"provider",
] as const;
export const ORDINARY_LIMITATION_COPY = {
preciseTiming: "这次说不到具体哪一天。",
missingEvidence: "有些依据还没补上,相关说法不能当成确定预测。",
techniqueOpen: "有些判断还没闭合,不能写成确定结论。",
statusLimited: "这次能读的,就是已经写明的这些。",
} as const;
const LIMITATION_ORDER = [
ORDINARY_LIMITATION_COPY.preciseTiming,
ORDINARY_LIMITATION_COPY.missingEvidence,
ORDINARY_LIMITATION_COPY.techniqueOpen,
ORDINARY_LIMITATION_COPY.statusLimited,
] as const;
const LIMITATION_HEADING = "需要知道的限制";
const INTERNAL_KEYS = new Set([
"technique_truth",
"workflow_route",
"workflow_status",
"precise_timing",
"missing_layers",
"execution_ledger",
"execution_receipt",
"claim_boundary",
"model_debug",
"model_id",
"provider",
"provider_payload",
"job_id",
"attempt_count",
"attempt_id",
"tool_call_id",
"tool_result",
"raw_tool_response",
"finish_reason",
"prompt_tokens",
"completion_tokens",
"system_prompt",
"score",
"weight",
"confidence_score",
"rubric_score",
]);
const INTERNAL_HEADINGS = new Set([
"claim boundary",
"boundary",
"technique audit",
"technique audit table",
"workflow receipt",
"workflow status",
"execution ledger",
"execution receipt",
"model debug",
"raw tool response",
"tool response",
"provider metadata",
"job metadata",
"attempt metadata",
"质量验收矩阵",
"技法审计",
"执行账本",
"模型调试",
]);
const INTERNAL_KEY_TOKEN = /\b(technique_truth|workflow_route|workflow_status|precise_timing|missing_layers|execution_ledger|execution_receipt|claim_boundary|model_debug|model_id|provider_payload|job_id|attempt_count|attempt_id|tool_call_id|tool_result|raw_tool_response|finish_reason|prompt_tokens|completion_tokens|system_prompt|confidence_score|rubric_score)\b|参数敏感|parameter_sensitive|\bMEVG\b/i;
const FIELD_LINE = /^\s*(?:>\s*)?(?:[-*+]\s+|\d+\.\s+)?(?:\*\*|__)?["'`]?([A-Za-z][A-Za-z0-9_-]*)["'`]?(?:\*\*|__)?\s*[:=]\s*(.*?)\s*$/;
const SECRET_PATTERNS = [
/\bsk-[A-Za-z0-9]{8,}\b/g,
/\bsk_(?:live|test)_[A-Za-z0-9]+\b/g,
/\bBearer\s+[A-Za-z0-9._-]{8,}\b/gi,
/\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b/g,
/\b(?:api[_-]?key|secret|password|access[_-]?token)\s*[:=]\s*\S+/gi,
/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g,
/\bSUPABASE_SERVICE_ROLE_KEY\b/g,
/\bAUTH_SECRET\b/g,
];
const BARE_URL = /\b(?:https?|javascript|vbscript|data|file|blob):[^\s)<>\]]+/gi;
const WINDOWS_PATH = /\b[A-Za-z]:\\[^\s)]+/g;
const UNIX_PRIVATE_PATH = /(?:^|[\s(])\/(?:opt|var|home|Users|root|tmp|srv)\/[^\s)]*/g;
const SVG_TAGS = new Set([
"svg", "g", "path", "line", "circle", "rect", "polygon", "polyline",
"text", "tspan", "title", "desc", "defs", "use", "ellipse",
]);
const DROP_WITH_CONTENT = new Set(["script", "style", "iframe", "object"]);
const DROP_TAG = new Set(["img", "embed", "script", "style", "iframe", "object"]);
const PUBLIC_PROTOCOLS = new Set(["http:", "https:", "mailto:"]);
export const ORDINARY_OUTPUT_LEAK_PATTERNS: readonly RegExp[] = [
/\btechnique_truth\b/i,
/\bworkflow_route\b/i,
/\bworkflow_status\b/i,
/\bprecise_timing\b/i,
/\bmissing_layers\b/i,
/\bexecution_ledger\b/i,
/\bexecution_receipt\b/i,
/\bprovider_payload\b/i,
/\btool_call_id\b/i,
/\btool_result\b/i,
/\bmodel_debug\b/i,
/\bjob_id\b/i,
/\battempt_count\b/i,
/\bprompt_tokens\b/i,
/\bfinish_reason\b/i,
/\bsystem_prompt\b/i,
/\bparameter_sensitive\b/i,
/\bMEVG\b/,
/javascript:/i,
/vbscript:/i,
/data:text\/html/i,
/\bfile:/i,
/127\.0\.0\.1/,
/\blocalhost\b/i,
/metadata\.google\.internal/i,
/\bsk-[A-Za-z0-9]{8,}/,
/\bBearer\s+/i,
/\bSUPABASE_SERVICE_ROLE_KEY\b/,
/<\s*script\b/i,
/<\s*iframe\b/i,
/<\s*object\b/i,
/<\s*embed\b/i,
/<\s*img\b/i,
/\bblocked\b/i,
/\bexecuted\b/i,
/\bpartial_verified\b/i,
/\bmissing_in_local\b/i,
/\bproducer\b/i,
/\bparity\b/i,
/PL9\s*第\s*\d+\s*页/,
/full-report:\/\//,
/\bpl9_[A-Za-z0-9_]+\b/i,
/\b[A-Za-z][A-Za-z0-9]*(?:_[A-Za-z0-9]+)*_unavailable\b/,
/\b[A-Za-z][A-Za-z0-9]*(?:_[A-Za-z0-9]+)*_(?:pack|packet)\b/i,
/object has no attribute/,
/\bTraceback\b/,
/(?:\b[A-Za-z]{2,}(?:'[A-Za-z]+)?\b\s+){11}\b[A-Za-z]{2,}(?:'[A-Za-z]+)?\b/,
];
export function ordinaryOutputLeaks(text: string): string[] {
return ORDINARY_OUTPUT_LEAK_PATTERNS.filter((pattern) => pattern.test(text)).map((pattern) => pattern.source);
}
export function isOrdinaryReportDocumentKind(kind: ReportDocumentKind): kind is OrdinaryReportDocumentKind {
return kind !== "professional_reference";
}
export type ReportMarkdownEnvelope = Readonly<{
documentKind: ReportDocumentKind;
format: "markdown";
markdown: string;
}>;
export type ChatExportInput = Readonly<{
documentKind?: "chat_export";
title: string;
prose: string;
techniqueTruth?: string | null;
workflowStatus?: string | null;
preciseTiming?: string | null;
missingLayers?: readonly string[] | null;
}>;
type Analysis = {
prose: string;
limitations: string[];
changed: boolean;
};
type LineSlice = { text: string; raw: string };
type Block = {
type: "fence" | "heading" | "html" | "table" | "list" | "paragraph" | "blank";
raw: string;
text: string;
level?: number;
lang?: string;
tag?: string;
};
export function projectChatExportMarkdown(input: ChatExportInput): string {
const analyzed = analyzeOrdinaryReportMarkdown(input.prose.trim());
const prose = analyzed.prose.trim() || "暂无回答。";
const limitations = orderedLimitations([
...analyzed.limitations,
...limitationsFromSignals(input),
]);
const lines = [`# ${publicTitle(input.title)}`, "", "## 最新回答", prose];
if (limitations.length > 0) {
lines.push("", `## ${LIMITATION_HEADING}`, "", ...limitations);
}
return lines.join("\n").trim();
}
export function projectOrdinaryReportMarkdown(markdown: string): string {
const analyzed = analyzeOrdinaryReportMarkdown(markdown);
if (!analyzed.changed && analyzed.limitations.length === 0) return markdown;
return appendLimitations(analyzed.prose, analyzed.limitations);
}
export function projectOrdinarySnippet(text: string): string {
return analyzeOrdinaryReportMarkdown(text).prose.trim();
}
export function projectReportEnvelope(input: {
documentKind: ReportDocumentKind;
markdown: string;
}): ReportMarkdownEnvelope {
return {
documentKind: input.documentKind,
format: "markdown",
markdown: projectOrdinaryReportMarkdown(input.markdown),
};
}
/** Explicit professional-reference envelope. Content still uses the ordinary projection. */
export function releaseProfessionalReference(markdown: string): ReportMarkdownEnvelope {
return projectReportEnvelope({
documentKind: "professional_reference",
markdown,
});
}
export function projectOrdinaryReportDocument(document: unknown): unknown {
if (!isReportShaped(document)) return document;
const limitations = new Set<string>();
const summary = readRecord(document.executiveSummary);
const sections = [
...narrativeSections(document.thematicNarrative, limitations),
...foundationSection(document.natalFoundation, limitations),
...phaseSection(document.currentPhase, limitations),
];
const actions = actionNotes(document.actionNotes, limitations);
for (const disclosure of recordList(document.blockedConflictDisclosure)) {
const reason = takeText(disclosure.reason, limitations);
if (reason) limitations.add(ORDINARY_LIMITATION_COPY.techniqueOpen);
if (Array.isArray(disclosure.missingEvidence) && disclosure.missingEvidence.length > 0) {
limitations.add(ORDINARY_LIMITATION_COPY.missingEvidence);
}
}
return {
documentKind: "personal_report_detail" as const,
headline: takeText(summary?.headline, limitations),
summary: takeText(summary?.summary, limitations),
priorities: takeTextList(summary?.priorities, limitations),
sections,
actions,
limitations: orderedLimitations(limitations),
disclaimer: takeText(document.disclaimer, limitations),
};
}
function limitationsFromSignals(input: ChatExportInput): string[] {
const notes: string[] = [];
const timing = normalizeToken(input.preciseTiming);
if (timing && /blocked|denied|false|unavailable|forbidden/.test(timing)) {
notes.push(ORDINARY_LIMITATION_COPY.preciseTiming);
}
if (input.missingLayers && input.missingLayers.some((layer) => layer.trim().length > 0)) {
notes.push(ORDINARY_LIMITATION_COPY.missingEvidence);
}
const truth = normalizeToken(input.techniqueTruth);
if (truth && /partial|blocked|degraded|not_applicable|not-applicable|unverified|unknown/.test(truth)) {
notes.push(ORDINARY_LIMITATION_COPY.techniqueOpen);
}
const status = normalizeToken(input.workflowStatus);
if (status && /blocked|degraded|failed|partial|error|incomplete/.test(status)) {
notes.push(ORDINARY_LIMITATION_COPY.statusLimited);
}
return notes;
}
function adjacentInternalTableIndexes(blocks: Block[]): Set<number> {
const drop = new Set<number>();
blocks.forEach((block, index) => {
if (block.type !== "table" || !tableIsInternal(block.text)) return;
drop.add(index);
let cursor = index - 1;
while (cursor >= 0 && blocks[cursor]?.type === "blank") cursor -= 1;
if (cursor >= 0 && (blocks[cursor]?.type === "paragraph" || blocks[cursor]?.type === "list")) {
drop.add(cursor);
cursor -= 1;
while (cursor >= 0 && blocks[cursor]?.type === "blank") cursor -= 1;
}
if (cursor >= 0 && blocks[cursor]?.type === "heading") drop.add(cursor);
});
return drop;
}
function analyzeOrdinaryReportMarkdown(markdown: string): Analysis {
if (!markdown) return { prose: markdown, limitations: [], changed: false };
const blocks = parseBlocks(markdown);
const dropWithTable = adjacentInternalTableIndexes(blocks);
const limitations = new Set<string>();
const kept: string[] = [];
let changed = false;
let skipUntilLevel: number | null = null;
for (let index = 0; index < blocks.length; index += 1) {
const block = blocks[index];
if (!block) continue;
if (dropWithTable.has(index)) {
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
if (block.type === "heading") {
const level = block.level ?? 1;
if (skipUntilLevel !== null && level <= skipUntilLevel) skipUntilLevel = null;
if (skipUntilLevel !== null) {
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
if (isInternalHeading(block.text) || containsInternalToken(block.text)) {
skipUntilLevel = level;
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
} else if (skipUntilLevel !== null) {
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
if (block.type === "fence") {
if (block.lang === "jyotish-chart" && fenceIsPublic(block.raw)) {
kept.push(block.raw);
continue;
}
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
if (block.type === "html") {
if (block.tag === "svg" && isAllowlistedSvg(block.raw)) {
kept.push(block.raw);
continue;
}
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
if (block.type === "blank") {
kept.push(block.raw);
continue;
}
if (block.type === "table" && tableIsInternal(block.text)) {
changed = true;
collectLimitations(block.raw, limitations);
continue;
}
const projected = projectPreservedBlock(block.raw);
if (projected.changed) changed = true;
for (const note of projected.limitations) limitations.add(note);
if (projected.text) kept.push(projected.text);
}
if (skipUntilLevel !== null) changed = true;
const prose = changed ? collapseBlankLines(kept.join("")) : markdown;
return { prose, limitations: orderedLimitations(limitations), changed };
}
function projectPreservedBlock(raw: string): { text: string; changed: boolean; limitations: string[] } {
const limitations = new Set<string>();
const lines = splitLines(raw);
const kept: string[] = [];
let changed = false;
for (const line of lines) {
const projected = projectProseLine(line.text);
for (const note of projected.limitations) limitations.add(note);
if (!projected.keep) {
changed = true;
continue;
}
if (projected.text !== line.text) {
changed = true;
const ending = line.raw.endsWith("\r\n") ? "\r\n" : line.raw.endsWith("\n") ? "\n" : "";
kept.push(`${projected.text}${ending}`);
continue;
}
kept.push(line.raw);
}
return { text: kept.join(""), changed, limitations: [...limitations] };
}
function projectProseLine(line: string): { keep: boolean; text: string; limitations: string[] } {
if (line.trim() === "") return { keep: true, text: line, limitations: [] };
const field = FIELD_LINE.exec(line);
if (field && isInternalKey(field[1])) {
return { keep: false, text: "", limitations: notesForField(field[1], field[2] ?? "") };
}
const cleaned = projectInline(line);
if (!cleaned.trim() || containsInternalToken(cleaned) || ordinaryOutputLeaks(cleaned).length > 0) {
return { keep: false, text: "", limitations: notesFromText(line) };
}
return { keep: true, text: cleaned, limitations: [] };
}
/** Safety-only boundary for the separately authorized raw attachment, not prose projection. */
export function sanitizeRawAppendixMarkdown(markdown: string): string {
return parseBlocks(markdown).map((block) => {
if (block.type === "fence" && block.lang === "jyotish-chart") {
return fenceIsPublic(block.raw) ? block.raw : "";
}
return stripSecrets(stripHtml(block.raw)).split(/(\r?\n)/).map((line) => (
/^\r?\n$/.test(line) ? line : projectInline(line)
)).join("");
}).join("");
}
function projectInline(text: string): string {
let next = text.replace(/<!--[\s\S]*?-->/g, "");
next = next.replace(/!\[([^\]]*)\]\([^)]*\)/g, (_match, alt: string) => stripSecrets(stripHtml(String(alt))).trim());
next = next.replace(/\[([^\]]*)\]\(([^)]+)\)/g, (_match, label: string, url: string) => {
const publicLabel = stripSecrets(stripHtml(String(label))).trim();
return isPublicUrl(String(url)) ? `[${publicLabel}](${String(url).trim()})` : publicLabel;
});
next = stripHtml(next);
next = stripSecrets(next);
next = next.replace(BARE_URL, (url) => (isPublicUrl(url) ? url : ""));
next = next.replace(WINDOWS_PATH, "");
next = next.replace(UNIX_PRIVATE_PATH, " ");
return next.replace(/[ \t]{2,}/g, " ").replace(/[ \t]+$/g, "").replace(/^[ \t]+/g, "");
}
function notesForField(key: string, value: string): string[] {
const normalizedKey = normalizeKey(key);
const normalized = normalizeToken(value);
if (normalizedKey === "precise_timing" && normalized && /blocked|denied|false|unavailable|forbidden/.test(normalized)) {
return [ORDINARY_LIMITATION_COPY.preciseTiming];
}
if (normalizedKey === "missing_layers" && normalized && !/^(none|unknown|n\/a|null|\[\]|\{\})$/.test(normalized)) {
return [ORDINARY_LIMITATION_COPY.missingEvidence];
}
if (normalizedKey === "technique_truth" && normalized && /partial|blocked|degraded|not_applicable|unverified|unknown/.test(normalized)) {
return [ORDINARY_LIMITATION_COPY.techniqueOpen];
}
if (normalizedKey === "workflow_status" && normalized && /blocked|degraded|failed|partial|error|incomplete/.test(normalized)) {
return [ORDINARY_LIMITATION_COPY.statusLimited];
}
return [];
}
function notesFromText(text: string): string[] {
const notes: string[] = [];
const field = FIELD_LINE.exec(text.trim());
if (field) notes.push(...notesForField(field[1], field[2] ?? ""));
if (/\bmissing_layers\b|\bMEVG\b/i.test(text)) notes.push(ORDINARY_LIMITATION_COPY.missingEvidence);
if (/\bprecise_timing\b/i.test(text) && /blocked|denied|false/i.test(text)) {
notes.push(ORDINARY_LIMITATION_COPY.preciseTiming);
}
if (/\btechnique_truth\b|\bparameter_sensitive\b/i.test(text)) notes.push(ORDINARY_LIMITATION_COPY.techniqueOpen);
if (/\bworkflow_status\b/i.test(text) && /blocked|degraded|failed|partial/i.test(text)) {
notes.push(ORDINARY_LIMITATION_COPY.statusLimited);
}
return notes;
}
function collectLimitations(text: string, into: Set<string>) {
for (const line of text.split(/\r?\n/)) {
for (const note of projectProseLine(line).limitations) into.add(note);
for (const note of notesFromText(line)) into.add(note);
}
}
function appendLimitations(prose: string, limitations: readonly string[]): string {
const notes = orderedLimitations(limitations).filter((note) => !prose.includes(note));
const body = prose.replace(/\s+$/u, "");
if (notes.length === 0) return body;
const section = [`## ${LIMITATION_HEADING}`, "", ...notes].join("\n");
return body ? `${body}\n\n${section}` : section;
}
function orderedLimitations(notes: Iterable<string>): string[] {
const present = new Set(notes);
return LIMITATION_ORDER.filter((note) => present.has(note));
}
function publicTitle(title: string): string {
const line = projectInline(title.replace(/[\r\n]+/g, " ")).replace(/^#+\s*/, "").trim();
return line || "咨询报告";
}
function isInternalKey(key: string): boolean {
return INTERNAL_KEYS.has(normalizeKey(key));
}
function normalizeKey(key: string): string {
return key.trim().toLowerCase().replace(/-/g, "_");
}
function normalizeToken(value: string | null | undefined): string {
return (value ?? "").trim().toLowerCase().replace(/^["'`[\]]+|["'`[\]]+$/g, "").replace(/-/g, "_");
}
function isInternalHeading(text: string): boolean {
const normalized = text.trim().toLowerCase().replace(/[`*_]/g, "").replace(/\s+/g, " ");
return INTERNAL_HEADINGS.has(normalized);
}
function containsInternalToken(text: string): boolean {
return INTERNAL_KEY_TOKEN.test(text);
}
function fenceIsPublic(raw: string): boolean {
if (containsInternalToken(raw)) return false;
if (ordinaryOutputLeaks(raw).length > 0) return false;
if (/<\s*(?:script|iframe|object|embed|img)\b/i.test(raw)) return false;
return true;
}
function tableIsInternal(text: string): boolean {
if (containsInternalToken(text) || ordinaryOutputLeaks(text).length > 0) return true;
const cells = text.split("|").map((cell) => cell.trim()).filter(Boolean);
return cells.some((cell) => isInternalKey(cell) || /^(?:score|weight|provider|attempt|job|model)$/i.test(cell));
}
function isPublicUrl(raw: string): boolean {
const value = raw.trim();
if (value.startsWith("#") && !value.includes(":")) return true;
let url: URL;
try {
url = new URL(value);
} catch {
return false;
}
if (!PUBLIC_PROTOCOLS.has(url.protocol)) return false;
if (url.username || url.password) return false;
if (url.port === "5200") return false;
if (/[?&](?:api[_-]?key|token|secret|password)=/i.test(url.search)) return false;
return !isInternalHost(url.hostname);
}
function isInternalHost(hostname: string): boolean {
const host = hostname.replace(/^\[|\]$/g, "").toLowerCase();
if (
host === "localhost"
|| host.endsWith(".localhost")
|| host.endsWith(".local")
|| host.endsWith(".internal")
|| host === "metadata.google.internal"
|| host === "0.0.0.0"
|| host === "::1"
) {
return true;
}
const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
if (!ipv4) return false;
const parts = ipv4.slice(1).map((part) => Number(part));
if (parts.some((part) => part > 255)) return false;
const [a, b] = parts;
if (a === 10 || a === 127 || a === 0) return true;
if (a === 169 && b === 254) return true;
if (a === 172 && b >= 16 && b <= 31) return true;
if (a === 192 && b === 168) return true;
return false;
}
function stripSecrets(text: string): string {
let next = text;
for (const pattern of SECRET_PATTERNS) {
pattern.lastIndex = 0;
next = next.replace(pattern, "");
}
return next;
}
function stripHtml(input: string): string {
let output = "";
let index = 0;
while (index < input.length) {
const start = input.indexOf("<", index);
if (start === -1) {
output += input.slice(index);
break;
}
output += input.slice(index, start);
if (input.startsWith("<!--", start)) {
const end = input.indexOf("-->", start + 4);
index = end === -1 ? input.length : end + 3;
continue;
}
if (input.startsWith("<?", start) || input.startsWith("<!", start)) {
const end = input.indexOf(">", start + 2);
index = end === -1 ? input.length : end + 1;
continue;
}
const tag = readTag(input, start);
if (!tag) {
output += "<";
index = start + 1;
continue;
}
const name = tag.name.toLowerCase();
if (name === "svg") {
const end = tag.selfClosing ? tag.end : findClose(input, tag.end, "svg");
const raw = input.slice(start, end);
if (isAllowlistedSvg(raw)) output += raw;
index = end;
continue;
}
if (DROP_WITH_CONTENT.has(name) && !tag.selfClosing) {
index = findClose(input, tag.end, name);
continue;
}
if (DROP_TAG.has(name) || tag.selfClosing) {
index = tag.end;
continue;
}
index = tag.end;
}
return output;
}
function isAllowlistedSvg(raw: string): boolean {
const tags = [...raw.matchAll(/<\/?\s*([a-zA-Z0-9]+)/g)].map((match) => match[1].toLowerCase());
if (tags[0] !== "svg") return false;
if (!tags.every((tag) => SVG_TAGS.has(tag))) return false;
if (/\bon[a-z]+\s*=/i.test(raw)) return false;
if (/\b(?:javascript|vbscript|data|file):/i.test(raw)) return false;
return ordinaryOutputLeaks(raw).length === 0;
}
function readTag(input: string, start: number): { name: string; end: number; selfClosing: boolean } | null {
if (input[start] !== "<") return null;
let index = start + 1;
if (input[index] === "/") index += 1;
const nameStart = index;
while (index < input.length && /[A-Za-z0-9]/.test(input[index] ?? "")) index += 1;
if (index === nameStart) return null;
const name = input.slice(nameStart, index);
let quote: string | null = null;
while (index < input.length) {
const char = input[index];
if (quote) {
if (char === quote) quote = null;
index += 1;
continue;
}
if (char === "\"" || char === "'") {
quote = char;
index += 1;
continue;
}
if (char === ">") {
const selfClosing = input[index - 1] === "/";
return { name, end: index + 1, selfClosing };
}
index += 1;
}
return null;
}
function findClose(input: string, from: number, name: string): number {
const open = new RegExp(`<\\s*${name}\\b`, "gi");
const close = new RegExp(`</\\s*${name}\\s*>`, "gi");
open.lastIndex = from;
close.lastIndex = from;
let depth = 1;
while (depth > 0) {
const nextClose = close.exec(input);
if (!nextClose) return input.length;
open.lastIndex = from;
let nested = 0;
let nextOpen = open.exec(input);
while (nextOpen && nextOpen.index < nextClose.index) {
nested += 1;
nextOpen = open.exec(input);
}
depth += nested - 1;
from = nextClose.index + nextClose[0].length;
if (depth === 0) return from;
open.lastIndex = from;
close.lastIndex = from;
}
return input.length;
}
function parseBlocks(markdown: string): Block[] {
const lines = splitLines(markdown);
const blocks: Block[] = [];
let index = 0;
while (index < lines.length) {
const line = lines[index];
if (!line) break;
const fence = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line.text);
if (fence) {
const marker = fence[1][0];
const width = fence[1].length;
const collected = [line];
index += 1;
while (index < lines.length) {
const next = lines[index];
collected.push(next);
index += 1;
if (new RegExp(`^ {0,3}${marker}{${width},}\\s*$`).test(next.text)) break;
}
blocks.push({
type: "fence",
raw: collected.map((item) => item.raw).join(""),
text: collected.map((item) => item.text).join("\n"),
lang: fence[2].trim().split(/\s+/)[0] ?? "",
});
continue;
}
const html = /^ {0,3}<([a-zA-Z][a-zA-Z0-9]*)\b/.exec(line.text);
if (html && /^(script|style|iframe|object|embed|img|svg)$/i.test(html[1])) {
const tag = html[1].toLowerCase();
const collected = [line];
index += 1;
if (!/\/\s*>$/.test(line.text) && tag !== "img" && tag !== "embed") {
const close = new RegExp(`</\\s*${tag}\\s*>`, "i");
while (index < lines.length && !close.test(collected[collected.length - 1]?.text ?? "")) {
collected.push(lines[index]);
index += 1;
if (close.test(lines[index - 1]?.text ?? "")) break;
}
}
blocks.push({
type: "html",
raw: collected.map((item) => item.raw).join(""),
text: collected.map((item) => item.text).join("\n"),
tag,
});
continue;
}
if (/^ {0,3}#{1,6}\s+\S/.test(line.text)) {
const level = /^( {0,3})(#+)/.exec(line.text)?.[2].length ?? 1;
blocks.push({
type: "heading",
raw: line.raw,
text: line.text.replace(/^ {0,3}#{1,6}\s+/, "").trim(),
level,
});
index += 1;
continue;
}
if (isTableStart(lines, index)) {
const collected = [line];
index += 1;
while (index < lines.length && lines[index].text.includes("|") && lines[index].text.trim()) {
collected.push(lines[index]);
index += 1;
}
blocks.push({
type: "table",
raw: collected.map((item) => item.raw).join(""),
text: collected.map((item) => item.text).join("\n"),
});
continue;
}
if (/^\s*$/.test(line.text)) {
const collected = [line];
index += 1;
while (index < lines.length && /^\s*$/.test(lines[index].text)) {
collected.push(lines[index]);
index += 1;
}
blocks.push({
type: "blank",
raw: collected.map((item) => item.raw).join(""),
text: "",
});
continue;
}
const collected = [line];
index += 1;
while (index < lines.length && !isBlockStart(lines, index)) {
collected.push(lines[index]);
index += 1;
}
const list = collected.every((item) => /^(\s*)([-*+]|\d+\.)\s+/.test(item.text) || /^\s+/.test(item.text) || item.text.trim() === "");
blocks.push({
type: list ? "list" : "paragraph",
raw: collected.map((item) => item.raw).join(""),
text: collected.map((item) => item.text).join("\n"),
});
}
return blocks;
}
function isBlockStart(lines: LineSlice[], index: number): boolean {
const text = lines[index]?.text ?? "";
if (/^\s*$/.test(text)) return true;
if (/^ {0,3}(`{3,}|~{3,})/.test(text)) return true;
if (/^ {0,3}#{1,6}\s+\S/.test(text)) return true;
if (/^ {0,3}<(script|style|iframe|object|embed|img|svg)\b/i.test(text)) return true;
return isTableStart(lines, index);
}
function isTableStart(lines: LineSlice[], index: number): boolean {
const current = lines[index]?.text ?? "";
const next = lines[index + 1]?.text ?? "";
return current.includes("|") && /^\s*\|?\s*:?-{3,}/.test(next);
}
function splitLines(markdown: string): LineSlice[] {
const lines: LineSlice[] = [];
let index = 0;
while (index < markdown.length) {
const next = markdown.indexOf("\n", index);
if (next === -1) {
lines.push({ text: markdown.slice(index), raw: markdown.slice(index) });
break;
}
const raw = markdown.slice(index, next + 1);
lines.push({ text: raw.replace(/\r?\n$/, ""), raw });
index = next + 1;
}
return lines;
}
function collapseBlankLines(text: string): string {
return text.replace(/[ \t]+\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
}
function isReportShaped(value: unknown): value is Record<string, unknown> {
if (!value || typeof value !== "object" || Array.isArray(value)) return false;
const record = value as Record<string, unknown>;
if (record.schemaVersion === "report_document.v1" || record.schemaVersion === "report_document.v2") return true;
return isRecord(record.evidenceAppendix) || isRecord(record.executiveSummary);
}
function takeText(value: unknown, limitations: Set<string>): string {
if (typeof value !== "string") return "";
const analyzed = analyzeOrdinaryReportMarkdown(value);
for (const note of analyzed.limitations) limitations.add(note);
return analyzed.prose.trim();
}
function takeTextList(value: unknown, limitations: Set<string>): string[] {
if (!Array.isArray(value)) return [];
return value.flatMap((item) => {
const text = takeText(item, limitations);
return text ? [text] : [];
});
}
function narrativeSections(value: unknown, limitations: Set<string>) {
return recordList(value).flatMap((section) => {
const narrative = takeText(section.narrative, limitations);
const actions = takeTextList(section.actions, limitations);
const caveats = takeTextList(section.caveats, limitations);
const title = takeText(section.title, limitations);
if (!title && !narrative && actions.length === 0 && caveats.length === 0) return [];
return [{ title, narrative, actions, caveats }];
});
}
function foundationSection(value: unknown, limitations: Set<string>) {
const section = readRecord(value);
if (!section) return [];
return [{
title: takeText(section.title, limitations),
narrative: takeText(section.narrative, limitations),
actions: takeTextList(section.keyFactors, limitations),
caveats: takeTextList(section.caveats, limitations),
}];
}
function phaseSection(value: unknown, limitations: Set<string>) {
const section = readRecord(value);
if (!section) return [];
return [{
title: takeText(section.title, limitations) || takeText(section.phaseLabel, limitations),
narrative: takeText(section.narrative, limitations),
actions: takeTextList(section.timingNotes, limitations),
caveats: takeTextList(section.caveats, limitations),
}];
}
function actionNotes(value: unknown, limitations: Set<string>) {
return recordList(value).flatMap((note) => {
const title = takeText(note.title, limitations);
const body = takeText(note.note, limitations);
const priority = note.priority === "now" || note.priority === "next" || note.priority === "watch"
? note.priority
: "";
if (!title && !body) return [];
return [{ title, note: body, ...(priority ? { priority } : {}) }];
});
}
function recordList(value: unknown): Record<string, unknown>[] {
if (!Array.isArray(value)) return [];
return value.filter(isRecord);
}
function readRecord(value: unknown): Record<string, unknown> | null {
return isRecord(value) ? value : null;
}
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === "object" && value !== null && !Array.isArray(value);
}