New reports request reader_main from upstream origin/main 23b9609e. KP and transit no longer copy an empty vars() dict, solar returns keep birth_asc_sign_idx, and ordinary projection deletes internal lines whole.
965 lines
32 KiB
TypeScript
965 lines
32 KiB
TypeScript
/**
|
|
* Ordinary-user report projection.
|
|
*
|
|
* Document kind is an explicit argument. A route name never decides visibility.
|
|
* Public text is assembled from an allowlist. Internal keys are not copied;
|
|
* a limitation that must survive is rewritten in natural language.
|
|
* Professional reference has no separate permission in this release, so that
|
|
* kind fail-safes to the same ordinary projection instead of leaking.
|
|
*/
|
|
|
|
export const REPORT_DOCUMENT_KINDS = [
|
|
"chat_export",
|
|
"personal_report_detail",
|
|
"ordinary_markdown_download",
|
|
"professional_reference",
|
|
] as const;
|
|
|
|
export type ReportDocumentKind = (typeof REPORT_DOCUMENT_KINDS)[number];
|
|
|
|
export const ORDINARY_REPORT_DOCUMENT_KINDS = [
|
|
"chat_export",
|
|
"personal_report_detail",
|
|
"ordinary_markdown_download",
|
|
] as const;
|
|
|
|
export type OrdinaryReportDocumentKind = (typeof ORDINARY_REPORT_DOCUMENT_KINDS)[number];
|
|
|
|
/** Fields an ordinary report may contain. Anything else is not copied. */
|
|
export const ORDINARY_PUBLIC_FIELDS = [
|
|
"title",
|
|
"prose",
|
|
"conclusion",
|
|
"action",
|
|
"limitation",
|
|
"chart_fence",
|
|
"engine_svg",
|
|
] as const;
|
|
|
|
/** Classified internal. Never serialized into ordinary output. */
|
|
export const INTERNAL_REPORT_FIELDS = [
|
|
"technique_truth",
|
|
"workflow_route",
|
|
"workflow_status",
|
|
"precise_timing",
|
|
"missing_layers",
|
|
"score",
|
|
"weight",
|
|
"execution_ledger",
|
|
"internal_url",
|
|
"secret",
|
|
"model_debug",
|
|
"raw_tool_response",
|
|
"job",
|
|
"attempt",
|
|
"provider",
|
|
] as const;
|
|
|
|
export const ORDINARY_LIMITATION_COPY = {
|
|
preciseTiming: "这次说不到具体哪一天。",
|
|
missingEvidence: "有些依据还没补上,相关说法不能当成确定预测。",
|
|
techniqueOpen: "有些判断还没闭合,不能写成确定结论。",
|
|
statusLimited: "这次能读的,就是已经写明的这些。",
|
|
} as const;
|
|
|
|
const LIMITATION_ORDER = [
|
|
ORDINARY_LIMITATION_COPY.preciseTiming,
|
|
ORDINARY_LIMITATION_COPY.missingEvidence,
|
|
ORDINARY_LIMITATION_COPY.techniqueOpen,
|
|
ORDINARY_LIMITATION_COPY.statusLimited,
|
|
] as const;
|
|
|
|
const LIMITATION_HEADING = "需要知道的限制";
|
|
|
|
const INTERNAL_KEYS = new Set([
|
|
"technique_truth",
|
|
"workflow_route",
|
|
"workflow_status",
|
|
"precise_timing",
|
|
"missing_layers",
|
|
"execution_ledger",
|
|
"execution_receipt",
|
|
"claim_boundary",
|
|
"model_debug",
|
|
"model_id",
|
|
"provider",
|
|
"provider_payload",
|
|
"job_id",
|
|
"attempt_count",
|
|
"attempt_id",
|
|
"tool_call_id",
|
|
"tool_result",
|
|
"raw_tool_response",
|
|
"finish_reason",
|
|
"prompt_tokens",
|
|
"completion_tokens",
|
|
"system_prompt",
|
|
"score",
|
|
"weight",
|
|
"confidence_score",
|
|
"rubric_score",
|
|
]);
|
|
|
|
const INTERNAL_HEADINGS = new Set([
|
|
"claim boundary",
|
|
"boundary",
|
|
"technique audit",
|
|
"technique audit table",
|
|
"workflow receipt",
|
|
"workflow status",
|
|
"execution ledger",
|
|
"execution receipt",
|
|
"model debug",
|
|
"raw tool response",
|
|
"tool response",
|
|
"provider metadata",
|
|
"job metadata",
|
|
"attempt metadata",
|
|
"质量验收矩阵",
|
|
"技法审计",
|
|
"执行账本",
|
|
"模型调试",
|
|
]);
|
|
|
|
const INTERNAL_KEY_TOKEN = /\b(technique_truth|workflow_route|workflow_status|precise_timing|missing_layers|execution_ledger|execution_receipt|claim_boundary|model_debug|model_id|provider_payload|job_id|attempt_count|attempt_id|tool_call_id|tool_result|raw_tool_response|finish_reason|prompt_tokens|completion_tokens|system_prompt|confidence_score|rubric_score)\b|参数敏感|parameter_sensitive|\bMEVG\b/i;
|
|
|
|
const FIELD_LINE = /^\s*(?:>\s*)?(?:[-*+]\s+|\d+\.\s+)?(?:\*\*|__)?["'`]?([A-Za-z][A-Za-z0-9_-]*)["'`]?(?:\*\*|__)?\s*[:=]\s*(.*?)\s*$/;
|
|
|
|
const SECRET_PATTERNS = [
|
|
/\bsk-[A-Za-z0-9]{8,}\b/g,
|
|
/\bsk_(?:live|test)_[A-Za-z0-9]+\b/g,
|
|
/\bBearer\s+[A-Za-z0-9._-]{8,}\b/gi,
|
|
/\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b/g,
|
|
/\b(?:api[_-]?key|secret|password|access[_-]?token)\s*[:=]\s*\S+/gi,
|
|
/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g,
|
|
/\bSUPABASE_SERVICE_ROLE_KEY\b/g,
|
|
/\bAUTH_SECRET\b/g,
|
|
];
|
|
|
|
const BARE_URL = /\b(?:https?|javascript|vbscript|data|file|blob):[^\s)<>\]]+/gi;
|
|
const WINDOWS_PATH = /\b[A-Za-z]:\\[^\s)]+/g;
|
|
const UNIX_PRIVATE_PATH = /(?:^|[\s(])\/(?:opt|var|home|Users|root|tmp|srv)\/[^\s)]*/g;
|
|
|
|
const SVG_TAGS = new Set([
|
|
"svg", "g", "path", "line", "circle", "rect", "polygon", "polyline",
|
|
"text", "tspan", "title", "desc", "defs", "use", "ellipse",
|
|
]);
|
|
|
|
const DROP_WITH_CONTENT = new Set(["script", "style", "iframe", "object"]);
|
|
const DROP_TAG = new Set(["img", "embed", "script", "style", "iframe", "object"]);
|
|
|
|
const PUBLIC_PROTOCOLS = new Set(["http:", "https:", "mailto:"]);
|
|
|
|
export const ORDINARY_OUTPUT_LEAK_PATTERNS: readonly RegExp[] = [
|
|
/\btechnique_truth\b/i,
|
|
/\bworkflow_route\b/i,
|
|
/\bworkflow_status\b/i,
|
|
/\bprecise_timing\b/i,
|
|
/\bmissing_layers\b/i,
|
|
/\bexecution_ledger\b/i,
|
|
/\bexecution_receipt\b/i,
|
|
/\bprovider_payload\b/i,
|
|
/\btool_call_id\b/i,
|
|
/\btool_result\b/i,
|
|
/\bmodel_debug\b/i,
|
|
/\bjob_id\b/i,
|
|
/\battempt_count\b/i,
|
|
/\bprompt_tokens\b/i,
|
|
/\bfinish_reason\b/i,
|
|
/\bsystem_prompt\b/i,
|
|
/\bparameter_sensitive\b/i,
|
|
/\bMEVG\b/,
|
|
/javascript:/i,
|
|
/vbscript:/i,
|
|
/data:text\/html/i,
|
|
/\bfile:/i,
|
|
/127\.0\.0\.1/,
|
|
/\blocalhost\b/i,
|
|
/metadata\.google\.internal/i,
|
|
/\bsk-[A-Za-z0-9]{8,}/,
|
|
/\bBearer\s+/i,
|
|
/\bSUPABASE_SERVICE_ROLE_KEY\b/,
|
|
/<\s*script\b/i,
|
|
/<\s*iframe\b/i,
|
|
/<\s*object\b/i,
|
|
/<\s*embed\b/i,
|
|
/<\s*img\b/i,
|
|
/\bblocked\b/i,
|
|
/\bexecuted\b/i,
|
|
/\bpartial_verified\b/i,
|
|
/\bmissing_in_local\b/i,
|
|
/\bproducer\b/i,
|
|
/\bparity\b/i,
|
|
/PL9\s*第\s*\d+\s*页/,
|
|
/full-report:\/\//,
|
|
/\bpl9_[A-Za-z0-9_]+\b/i,
|
|
/\b[A-Za-z][A-Za-z0-9]*(?:_[A-Za-z0-9]+)*_unavailable\b/,
|
|
/\b[A-Za-z][A-Za-z0-9]*(?:_[A-Za-z0-9]+)*_(?:pack|packet)\b/i,
|
|
/object has no attribute/,
|
|
/\bTraceback\b/,
|
|
/(?:\b[A-Za-z]{2,}(?:'[A-Za-z]+)?\b\s+){11}\b[A-Za-z]{2,}(?:'[A-Za-z]+)?\b/,
|
|
];
|
|
|
|
export function ordinaryOutputLeaks(text: string): string[] {
|
|
return ORDINARY_OUTPUT_LEAK_PATTERNS.filter((pattern) => pattern.test(text)).map((pattern) => pattern.source);
|
|
}
|
|
|
|
export function isOrdinaryReportDocumentKind(kind: ReportDocumentKind): kind is OrdinaryReportDocumentKind {
|
|
return kind !== "professional_reference";
|
|
}
|
|
|
|
export type ReportMarkdownEnvelope = Readonly<{
|
|
documentKind: ReportDocumentKind;
|
|
format: "markdown";
|
|
markdown: string;
|
|
}>;
|
|
|
|
export type ChatExportInput = Readonly<{
|
|
documentKind?: "chat_export";
|
|
title: string;
|
|
prose: string;
|
|
techniqueTruth?: string | null;
|
|
workflowStatus?: string | null;
|
|
preciseTiming?: string | null;
|
|
missingLayers?: readonly string[] | null;
|
|
}>;
|
|
|
|
type Analysis = {
|
|
prose: string;
|
|
limitations: string[];
|
|
changed: boolean;
|
|
};
|
|
|
|
type LineSlice = { text: string; raw: string };
|
|
|
|
type Block = {
|
|
type: "fence" | "heading" | "html" | "table" | "list" | "paragraph" | "blank";
|
|
raw: string;
|
|
text: string;
|
|
level?: number;
|
|
lang?: string;
|
|
tag?: string;
|
|
};
|
|
|
|
export function projectChatExportMarkdown(input: ChatExportInput): string {
|
|
const analyzed = analyzeOrdinaryReportMarkdown(input.prose.trim());
|
|
const prose = analyzed.prose.trim() || "暂无回答。";
|
|
const limitations = orderedLimitations([
|
|
...analyzed.limitations,
|
|
...limitationsFromSignals(input),
|
|
]);
|
|
const lines = [`# ${publicTitle(input.title)}`, "", "## 最新回答", prose];
|
|
if (limitations.length > 0) {
|
|
lines.push("", `## ${LIMITATION_HEADING}`, "", ...limitations);
|
|
}
|
|
return lines.join("\n").trim();
|
|
}
|
|
|
|
export function projectOrdinaryReportMarkdown(markdown: string): string {
|
|
const analyzed = analyzeOrdinaryReportMarkdown(markdown);
|
|
if (!analyzed.changed && analyzed.limitations.length === 0) return markdown;
|
|
return appendLimitations(analyzed.prose, analyzed.limitations);
|
|
}
|
|
|
|
export function projectOrdinarySnippet(text: string): string {
|
|
return analyzeOrdinaryReportMarkdown(text).prose.trim();
|
|
}
|
|
|
|
export function projectReportEnvelope(input: {
|
|
documentKind: ReportDocumentKind;
|
|
markdown: string;
|
|
}): ReportMarkdownEnvelope {
|
|
return {
|
|
documentKind: input.documentKind,
|
|
format: "markdown",
|
|
markdown: projectOrdinaryReportMarkdown(input.markdown),
|
|
};
|
|
}
|
|
|
|
/** Explicit professional-reference envelope. Content still uses the ordinary projection. */
|
|
export function releaseProfessionalReference(markdown: string): ReportMarkdownEnvelope {
|
|
return projectReportEnvelope({
|
|
documentKind: "professional_reference",
|
|
markdown,
|
|
});
|
|
}
|
|
|
|
export function projectOrdinaryReportDocument(document: unknown): unknown {
|
|
if (!isReportShaped(document)) return document;
|
|
const limitations = new Set<string>();
|
|
const summary = readRecord(document.executiveSummary);
|
|
const sections = [
|
|
...narrativeSections(document.thematicNarrative, limitations),
|
|
...foundationSection(document.natalFoundation, limitations),
|
|
...phaseSection(document.currentPhase, limitations),
|
|
];
|
|
const actions = actionNotes(document.actionNotes, limitations);
|
|
for (const disclosure of recordList(document.blockedConflictDisclosure)) {
|
|
const reason = takeText(disclosure.reason, limitations);
|
|
if (reason) limitations.add(ORDINARY_LIMITATION_COPY.techniqueOpen);
|
|
if (Array.isArray(disclosure.missingEvidence) && disclosure.missingEvidence.length > 0) {
|
|
limitations.add(ORDINARY_LIMITATION_COPY.missingEvidence);
|
|
}
|
|
}
|
|
return {
|
|
documentKind: "personal_report_detail" as const,
|
|
headline: takeText(summary?.headline, limitations),
|
|
summary: takeText(summary?.summary, limitations),
|
|
priorities: takeTextList(summary?.priorities, limitations),
|
|
sections,
|
|
actions,
|
|
limitations: orderedLimitations(limitations),
|
|
disclaimer: takeText(document.disclaimer, limitations),
|
|
};
|
|
}
|
|
|
|
function limitationsFromSignals(input: ChatExportInput): string[] {
|
|
const notes: string[] = [];
|
|
const timing = normalizeToken(input.preciseTiming);
|
|
if (timing && /blocked|denied|false|unavailable|forbidden/.test(timing)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.preciseTiming);
|
|
}
|
|
if (input.missingLayers && input.missingLayers.some((layer) => layer.trim().length > 0)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.missingEvidence);
|
|
}
|
|
const truth = normalizeToken(input.techniqueTruth);
|
|
if (truth && /partial|blocked|degraded|not_applicable|not-applicable|unverified|unknown/.test(truth)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.techniqueOpen);
|
|
}
|
|
const status = normalizeToken(input.workflowStatus);
|
|
if (status && /blocked|degraded|failed|partial|error|incomplete/.test(status)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.statusLimited);
|
|
}
|
|
return notes;
|
|
}
|
|
|
|
function adjacentInternalTableIndexes(blocks: Block[]): Set<number> {
|
|
const drop = new Set<number>();
|
|
blocks.forEach((block, index) => {
|
|
if (block.type !== "table" || !tableIsInternal(block.text)) return;
|
|
drop.add(index);
|
|
let cursor = index - 1;
|
|
while (cursor >= 0 && blocks[cursor]?.type === "blank") cursor -= 1;
|
|
if (cursor >= 0 && (blocks[cursor]?.type === "paragraph" || blocks[cursor]?.type === "list")) {
|
|
drop.add(cursor);
|
|
cursor -= 1;
|
|
while (cursor >= 0 && blocks[cursor]?.type === "blank") cursor -= 1;
|
|
}
|
|
if (cursor >= 0 && blocks[cursor]?.type === "heading") drop.add(cursor);
|
|
});
|
|
return drop;
|
|
}
|
|
|
|
function analyzeOrdinaryReportMarkdown(markdown: string): Analysis {
|
|
if (!markdown) return { prose: markdown, limitations: [], changed: false };
|
|
const blocks = parseBlocks(markdown);
|
|
const dropWithTable = adjacentInternalTableIndexes(blocks);
|
|
const limitations = new Set<string>();
|
|
const kept: string[] = [];
|
|
let changed = false;
|
|
let skipUntilLevel: number | null = null;
|
|
|
|
for (let index = 0; index < blocks.length; index += 1) {
|
|
const block = blocks[index];
|
|
if (!block) continue;
|
|
if (dropWithTable.has(index)) {
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
if (block.type === "heading") {
|
|
const level = block.level ?? 1;
|
|
if (skipUntilLevel !== null && level <= skipUntilLevel) skipUntilLevel = null;
|
|
if (skipUntilLevel !== null) {
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
if (isInternalHeading(block.text) || containsInternalToken(block.text)) {
|
|
skipUntilLevel = level;
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
} else if (skipUntilLevel !== null) {
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
|
|
if (block.type === "fence") {
|
|
if (block.lang === "jyotish-chart" && fenceIsPublic(block.raw)) {
|
|
kept.push(block.raw);
|
|
continue;
|
|
}
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
|
|
if (block.type === "html") {
|
|
if (block.tag === "svg" && isAllowlistedSvg(block.raw)) {
|
|
kept.push(block.raw);
|
|
continue;
|
|
}
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
|
|
if (block.type === "blank") {
|
|
kept.push(block.raw);
|
|
continue;
|
|
}
|
|
|
|
if (block.type === "table" && tableIsInternal(block.text)) {
|
|
changed = true;
|
|
collectLimitations(block.raw, limitations);
|
|
continue;
|
|
}
|
|
|
|
const projected = projectPreservedBlock(block.raw);
|
|
if (projected.changed) changed = true;
|
|
for (const note of projected.limitations) limitations.add(note);
|
|
if (projected.text) kept.push(projected.text);
|
|
}
|
|
|
|
if (skipUntilLevel !== null) changed = true;
|
|
const prose = changed ? collapseBlankLines(kept.join("")) : markdown;
|
|
return { prose, limitations: orderedLimitations(limitations), changed };
|
|
}
|
|
|
|
function projectPreservedBlock(raw: string): { text: string; changed: boolean; limitations: string[] } {
|
|
const limitations = new Set<string>();
|
|
const lines = splitLines(raw);
|
|
const kept: string[] = [];
|
|
let changed = false;
|
|
for (const line of lines) {
|
|
const projected = projectProseLine(line.text);
|
|
for (const note of projected.limitations) limitations.add(note);
|
|
if (!projected.keep) {
|
|
changed = true;
|
|
continue;
|
|
}
|
|
if (projected.text !== line.text) {
|
|
changed = true;
|
|
const ending = line.raw.endsWith("\r\n") ? "\r\n" : line.raw.endsWith("\n") ? "\n" : "";
|
|
kept.push(`${projected.text}${ending}`);
|
|
continue;
|
|
}
|
|
kept.push(line.raw);
|
|
}
|
|
return { text: kept.join(""), changed, limitations: [...limitations] };
|
|
}
|
|
|
|
function projectProseLine(line: string): { keep: boolean; text: string; limitations: string[] } {
|
|
if (line.trim() === "") return { keep: true, text: line, limitations: [] };
|
|
const field = FIELD_LINE.exec(line);
|
|
if (field && isInternalKey(field[1])) {
|
|
return { keep: false, text: "", limitations: notesForField(field[1], field[2] ?? "") };
|
|
}
|
|
const cleaned = projectInline(line);
|
|
if (!cleaned.trim() || containsInternalToken(cleaned) || ordinaryOutputLeaks(cleaned).length > 0) {
|
|
return { keep: false, text: "", limitations: notesFromText(line) };
|
|
}
|
|
return { keep: true, text: cleaned, limitations: [] };
|
|
}
|
|
|
|
/** Safety-only boundary for the separately authorized raw attachment, not prose projection. */
|
|
export function sanitizeRawAppendixMarkdown(markdown: string): string {
|
|
return parseBlocks(markdown).map((block) => {
|
|
if (block.type === "fence" && block.lang === "jyotish-chart") {
|
|
return fenceIsPublic(block.raw) ? block.raw : "";
|
|
}
|
|
return stripSecrets(stripHtml(block.raw)).split(/(\r?\n)/).map((line) => (
|
|
/^\r?\n$/.test(line) ? line : projectInline(line)
|
|
)).join("");
|
|
}).join("");
|
|
}
|
|
|
|
function projectInline(text: string): string {
|
|
let next = text.replace(/<!--[\s\S]*?-->/g, "");
|
|
next = next.replace(/!\[([^\]]*)\]\([^)]*\)/g, (_match, alt: string) => stripSecrets(stripHtml(String(alt))).trim());
|
|
next = next.replace(/\[([^\]]*)\]\(([^)]+)\)/g, (_match, label: string, url: string) => {
|
|
const publicLabel = stripSecrets(stripHtml(String(label))).trim();
|
|
return isPublicUrl(String(url)) ? `[${publicLabel}](${String(url).trim()})` : publicLabel;
|
|
});
|
|
next = stripHtml(next);
|
|
next = stripSecrets(next);
|
|
next = next.replace(BARE_URL, (url) => (isPublicUrl(url) ? url : ""));
|
|
next = next.replace(WINDOWS_PATH, "");
|
|
next = next.replace(UNIX_PRIVATE_PATH, " ");
|
|
return next.replace(/[ \t]{2,}/g, " ").replace(/[ \t]+$/g, "").replace(/^[ \t]+/g, "");
|
|
}
|
|
|
|
function notesForField(key: string, value: string): string[] {
|
|
const normalizedKey = normalizeKey(key);
|
|
const normalized = normalizeToken(value);
|
|
if (normalizedKey === "precise_timing" && normalized && /blocked|denied|false|unavailable|forbidden/.test(normalized)) {
|
|
return [ORDINARY_LIMITATION_COPY.preciseTiming];
|
|
}
|
|
if (normalizedKey === "missing_layers" && normalized && !/^(none|unknown|n\/a|null|\[\]|\{\})$/.test(normalized)) {
|
|
return [ORDINARY_LIMITATION_COPY.missingEvidence];
|
|
}
|
|
if (normalizedKey === "technique_truth" && normalized && /partial|blocked|degraded|not_applicable|unverified|unknown/.test(normalized)) {
|
|
return [ORDINARY_LIMITATION_COPY.techniqueOpen];
|
|
}
|
|
if (normalizedKey === "workflow_status" && normalized && /blocked|degraded|failed|partial|error|incomplete/.test(normalized)) {
|
|
return [ORDINARY_LIMITATION_COPY.statusLimited];
|
|
}
|
|
return [];
|
|
}
|
|
|
|
function notesFromText(text: string): string[] {
|
|
const notes: string[] = [];
|
|
const field = FIELD_LINE.exec(text.trim());
|
|
if (field) notes.push(...notesForField(field[1], field[2] ?? ""));
|
|
if (/\bmissing_layers\b|\bMEVG\b/i.test(text)) notes.push(ORDINARY_LIMITATION_COPY.missingEvidence);
|
|
if (/\bprecise_timing\b/i.test(text) && /blocked|denied|false/i.test(text)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.preciseTiming);
|
|
}
|
|
if (/\btechnique_truth\b|\bparameter_sensitive\b/i.test(text)) notes.push(ORDINARY_LIMITATION_COPY.techniqueOpen);
|
|
if (/\bworkflow_status\b/i.test(text) && /blocked|degraded|failed|partial/i.test(text)) {
|
|
notes.push(ORDINARY_LIMITATION_COPY.statusLimited);
|
|
}
|
|
return notes;
|
|
}
|
|
|
|
function collectLimitations(text: string, into: Set<string>) {
|
|
for (const line of text.split(/\r?\n/)) {
|
|
for (const note of projectProseLine(line).limitations) into.add(note);
|
|
for (const note of notesFromText(line)) into.add(note);
|
|
}
|
|
}
|
|
|
|
function appendLimitations(prose: string, limitations: readonly string[]): string {
|
|
const notes = orderedLimitations(limitations).filter((note) => !prose.includes(note));
|
|
const body = prose.replace(/\s+$/u, "");
|
|
if (notes.length === 0) return body;
|
|
const section = [`## ${LIMITATION_HEADING}`, "", ...notes].join("\n");
|
|
return body ? `${body}\n\n${section}` : section;
|
|
}
|
|
|
|
function orderedLimitations(notes: Iterable<string>): string[] {
|
|
const present = new Set(notes);
|
|
return LIMITATION_ORDER.filter((note) => present.has(note));
|
|
}
|
|
|
|
function publicTitle(title: string): string {
|
|
const line = projectInline(title.replace(/[\r\n]+/g, " ")).replace(/^#+\s*/, "").trim();
|
|
return line || "咨询报告";
|
|
}
|
|
|
|
function isInternalKey(key: string): boolean {
|
|
return INTERNAL_KEYS.has(normalizeKey(key));
|
|
}
|
|
|
|
function normalizeKey(key: string): string {
|
|
return key.trim().toLowerCase().replace(/-/g, "_");
|
|
}
|
|
|
|
function normalizeToken(value: string | null | undefined): string {
|
|
return (value ?? "").trim().toLowerCase().replace(/^["'`[\]]+|["'`[\]]+$/g, "").replace(/-/g, "_");
|
|
}
|
|
|
|
function isInternalHeading(text: string): boolean {
|
|
const normalized = text.trim().toLowerCase().replace(/[`*_]/g, "").replace(/\s+/g, " ");
|
|
return INTERNAL_HEADINGS.has(normalized);
|
|
}
|
|
|
|
function containsInternalToken(text: string): boolean {
|
|
return INTERNAL_KEY_TOKEN.test(text);
|
|
}
|
|
|
|
function fenceIsPublic(raw: string): boolean {
|
|
if (containsInternalToken(raw)) return false;
|
|
if (ordinaryOutputLeaks(raw).length > 0) return false;
|
|
if (/<\s*(?:script|iframe|object|embed|img)\b/i.test(raw)) return false;
|
|
return true;
|
|
}
|
|
|
|
function tableIsInternal(text: string): boolean {
|
|
if (containsInternalToken(text) || ordinaryOutputLeaks(text).length > 0) return true;
|
|
const cells = text.split("|").map((cell) => cell.trim()).filter(Boolean);
|
|
return cells.some((cell) => isInternalKey(cell) || /^(?:score|weight|provider|attempt|job|model)$/i.test(cell));
|
|
}
|
|
|
|
function isPublicUrl(raw: string): boolean {
|
|
const value = raw.trim();
|
|
if (value.startsWith("#") && !value.includes(":")) return true;
|
|
let url: URL;
|
|
try {
|
|
url = new URL(value);
|
|
} catch {
|
|
return false;
|
|
}
|
|
if (!PUBLIC_PROTOCOLS.has(url.protocol)) return false;
|
|
if (url.username || url.password) return false;
|
|
if (url.port === "5200") return false;
|
|
if (/[?&](?:api[_-]?key|token|secret|password)=/i.test(url.search)) return false;
|
|
return !isInternalHost(url.hostname);
|
|
}
|
|
|
|
function isInternalHost(hostname: string): boolean {
|
|
const host = hostname.replace(/^\[|\]$/g, "").toLowerCase();
|
|
if (
|
|
host === "localhost"
|
|
|| host.endsWith(".localhost")
|
|
|| host.endsWith(".local")
|
|
|| host.endsWith(".internal")
|
|
|| host === "metadata.google.internal"
|
|
|| host === "0.0.0.0"
|
|
|| host === "::1"
|
|
) {
|
|
return true;
|
|
}
|
|
const ipv4 = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/.exec(host);
|
|
if (!ipv4) return false;
|
|
const parts = ipv4.slice(1).map((part) => Number(part));
|
|
if (parts.some((part) => part > 255)) return false;
|
|
const [a, b] = parts;
|
|
if (a === 10 || a === 127 || a === 0) return true;
|
|
if (a === 169 && b === 254) return true;
|
|
if (a === 172 && b >= 16 && b <= 31) return true;
|
|
if (a === 192 && b === 168) return true;
|
|
return false;
|
|
}
|
|
|
|
function stripSecrets(text: string): string {
|
|
let next = text;
|
|
for (const pattern of SECRET_PATTERNS) {
|
|
pattern.lastIndex = 0;
|
|
next = next.replace(pattern, "");
|
|
}
|
|
return next;
|
|
}
|
|
|
|
function stripHtml(input: string): string {
|
|
let output = "";
|
|
let index = 0;
|
|
while (index < input.length) {
|
|
const start = input.indexOf("<", index);
|
|
if (start === -1) {
|
|
output += input.slice(index);
|
|
break;
|
|
}
|
|
output += input.slice(index, start);
|
|
if (input.startsWith("<!--", start)) {
|
|
const end = input.indexOf("-->", start + 4);
|
|
index = end === -1 ? input.length : end + 3;
|
|
continue;
|
|
}
|
|
if (input.startsWith("<?", start) || input.startsWith("<!", start)) {
|
|
const end = input.indexOf(">", start + 2);
|
|
index = end === -1 ? input.length : end + 1;
|
|
continue;
|
|
}
|
|
const tag = readTag(input, start);
|
|
if (!tag) {
|
|
output += "<";
|
|
index = start + 1;
|
|
continue;
|
|
}
|
|
const name = tag.name.toLowerCase();
|
|
if (name === "svg") {
|
|
const end = tag.selfClosing ? tag.end : findClose(input, tag.end, "svg");
|
|
const raw = input.slice(start, end);
|
|
if (isAllowlistedSvg(raw)) output += raw;
|
|
index = end;
|
|
continue;
|
|
}
|
|
if (DROP_WITH_CONTENT.has(name) && !tag.selfClosing) {
|
|
index = findClose(input, tag.end, name);
|
|
continue;
|
|
}
|
|
if (DROP_TAG.has(name) || tag.selfClosing) {
|
|
index = tag.end;
|
|
continue;
|
|
}
|
|
index = tag.end;
|
|
}
|
|
return output;
|
|
}
|
|
|
|
function isAllowlistedSvg(raw: string): boolean {
|
|
const tags = [...raw.matchAll(/<\/?\s*([a-zA-Z0-9]+)/g)].map((match) => match[1].toLowerCase());
|
|
if (tags[0] !== "svg") return false;
|
|
if (!tags.every((tag) => SVG_TAGS.has(tag))) return false;
|
|
if (/\bon[a-z]+\s*=/i.test(raw)) return false;
|
|
if (/\b(?:javascript|vbscript|data|file):/i.test(raw)) return false;
|
|
return ordinaryOutputLeaks(raw).length === 0;
|
|
}
|
|
|
|
function readTag(input: string, start: number): { name: string; end: number; selfClosing: boolean } | null {
|
|
if (input[start] !== "<") return null;
|
|
let index = start + 1;
|
|
if (input[index] === "/") index += 1;
|
|
const nameStart = index;
|
|
while (index < input.length && /[A-Za-z0-9]/.test(input[index] ?? "")) index += 1;
|
|
if (index === nameStart) return null;
|
|
const name = input.slice(nameStart, index);
|
|
let quote: string | null = null;
|
|
while (index < input.length) {
|
|
const char = input[index];
|
|
if (quote) {
|
|
if (char === quote) quote = null;
|
|
index += 1;
|
|
continue;
|
|
}
|
|
if (char === "\"" || char === "'") {
|
|
quote = char;
|
|
index += 1;
|
|
continue;
|
|
}
|
|
if (char === ">") {
|
|
const selfClosing = input[index - 1] === "/";
|
|
return { name, end: index + 1, selfClosing };
|
|
}
|
|
index += 1;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
function findClose(input: string, from: number, name: string): number {
|
|
const open = new RegExp(`<\\s*${name}\\b`, "gi");
|
|
const close = new RegExp(`</\\s*${name}\\s*>`, "gi");
|
|
open.lastIndex = from;
|
|
close.lastIndex = from;
|
|
let depth = 1;
|
|
while (depth > 0) {
|
|
const nextClose = close.exec(input);
|
|
if (!nextClose) return input.length;
|
|
open.lastIndex = from;
|
|
let nested = 0;
|
|
let nextOpen = open.exec(input);
|
|
while (nextOpen && nextOpen.index < nextClose.index) {
|
|
nested += 1;
|
|
nextOpen = open.exec(input);
|
|
}
|
|
depth += nested - 1;
|
|
from = nextClose.index + nextClose[0].length;
|
|
if (depth === 0) return from;
|
|
open.lastIndex = from;
|
|
close.lastIndex = from;
|
|
}
|
|
return input.length;
|
|
}
|
|
|
|
function parseBlocks(markdown: string): Block[] {
|
|
const lines = splitLines(markdown);
|
|
const blocks: Block[] = [];
|
|
let index = 0;
|
|
while (index < lines.length) {
|
|
const line = lines[index];
|
|
if (!line) break;
|
|
const fence = /^ {0,3}(`{3,}|~{3,})(.*)$/.exec(line.text);
|
|
if (fence) {
|
|
const marker = fence[1][0];
|
|
const width = fence[1].length;
|
|
const collected = [line];
|
|
index += 1;
|
|
while (index < lines.length) {
|
|
const next = lines[index];
|
|
collected.push(next);
|
|
index += 1;
|
|
if (new RegExp(`^ {0,3}${marker}{${width},}\\s*$`).test(next.text)) break;
|
|
}
|
|
blocks.push({
|
|
type: "fence",
|
|
raw: collected.map((item) => item.raw).join(""),
|
|
text: collected.map((item) => item.text).join("\n"),
|
|
lang: fence[2].trim().split(/\s+/)[0] ?? "",
|
|
});
|
|
continue;
|
|
}
|
|
const html = /^ {0,3}<([a-zA-Z][a-zA-Z0-9]*)\b/.exec(line.text);
|
|
if (html && /^(script|style|iframe|object|embed|img|svg)$/i.test(html[1])) {
|
|
const tag = html[1].toLowerCase();
|
|
const collected = [line];
|
|
index += 1;
|
|
if (!/\/\s*>$/.test(line.text) && tag !== "img" && tag !== "embed") {
|
|
const close = new RegExp(`</\\s*${tag}\\s*>`, "i");
|
|
while (index < lines.length && !close.test(collected[collected.length - 1]?.text ?? "")) {
|
|
collected.push(lines[index]);
|
|
index += 1;
|
|
if (close.test(lines[index - 1]?.text ?? "")) break;
|
|
}
|
|
}
|
|
blocks.push({
|
|
type: "html",
|
|
raw: collected.map((item) => item.raw).join(""),
|
|
text: collected.map((item) => item.text).join("\n"),
|
|
tag,
|
|
});
|
|
continue;
|
|
}
|
|
if (/^ {0,3}#{1,6}\s+\S/.test(line.text)) {
|
|
const level = /^( {0,3})(#+)/.exec(line.text)?.[2].length ?? 1;
|
|
blocks.push({
|
|
type: "heading",
|
|
raw: line.raw,
|
|
text: line.text.replace(/^ {0,3}#{1,6}\s+/, "").trim(),
|
|
level,
|
|
});
|
|
index += 1;
|
|
continue;
|
|
}
|
|
if (isTableStart(lines, index)) {
|
|
const collected = [line];
|
|
index += 1;
|
|
while (index < lines.length && lines[index].text.includes("|") && lines[index].text.trim()) {
|
|
collected.push(lines[index]);
|
|
index += 1;
|
|
}
|
|
blocks.push({
|
|
type: "table",
|
|
raw: collected.map((item) => item.raw).join(""),
|
|
text: collected.map((item) => item.text).join("\n"),
|
|
});
|
|
continue;
|
|
}
|
|
if (/^\s*$/.test(line.text)) {
|
|
const collected = [line];
|
|
index += 1;
|
|
while (index < lines.length && /^\s*$/.test(lines[index].text)) {
|
|
collected.push(lines[index]);
|
|
index += 1;
|
|
}
|
|
blocks.push({
|
|
type: "blank",
|
|
raw: collected.map((item) => item.raw).join(""),
|
|
text: "",
|
|
});
|
|
continue;
|
|
}
|
|
const collected = [line];
|
|
index += 1;
|
|
while (index < lines.length && !isBlockStart(lines, index)) {
|
|
collected.push(lines[index]);
|
|
index += 1;
|
|
}
|
|
const list = collected.every((item) => /^(\s*)([-*+]|\d+\.)\s+/.test(item.text) || /^\s+/.test(item.text) || item.text.trim() === "");
|
|
blocks.push({
|
|
type: list ? "list" : "paragraph",
|
|
raw: collected.map((item) => item.raw).join(""),
|
|
text: collected.map((item) => item.text).join("\n"),
|
|
});
|
|
}
|
|
return blocks;
|
|
}
|
|
|
|
function isBlockStart(lines: LineSlice[], index: number): boolean {
|
|
const text = lines[index]?.text ?? "";
|
|
if (/^\s*$/.test(text)) return true;
|
|
if (/^ {0,3}(`{3,}|~{3,})/.test(text)) return true;
|
|
if (/^ {0,3}#{1,6}\s+\S/.test(text)) return true;
|
|
if (/^ {0,3}<(script|style|iframe|object|embed|img|svg)\b/i.test(text)) return true;
|
|
return isTableStart(lines, index);
|
|
}
|
|
|
|
function isTableStart(lines: LineSlice[], index: number): boolean {
|
|
const current = lines[index]?.text ?? "";
|
|
const next = lines[index + 1]?.text ?? "";
|
|
return current.includes("|") && /^\s*\|?\s*:?-{3,}/.test(next);
|
|
}
|
|
|
|
function splitLines(markdown: string): LineSlice[] {
|
|
const lines: LineSlice[] = [];
|
|
let index = 0;
|
|
while (index < markdown.length) {
|
|
const next = markdown.indexOf("\n", index);
|
|
if (next === -1) {
|
|
lines.push({ text: markdown.slice(index), raw: markdown.slice(index) });
|
|
break;
|
|
}
|
|
const raw = markdown.slice(index, next + 1);
|
|
lines.push({ text: raw.replace(/\r?\n$/, ""), raw });
|
|
index = next + 1;
|
|
}
|
|
return lines;
|
|
}
|
|
|
|
function collapseBlankLines(text: string): string {
|
|
return text.replace(/[ \t]+\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
}
|
|
|
|
function isReportShaped(value: unknown): value is Record<string, unknown> {
|
|
if (!value || typeof value !== "object" || Array.isArray(value)) return false;
|
|
const record = value as Record<string, unknown>;
|
|
if (record.schemaVersion === "report_document.v1" || record.schemaVersion === "report_document.v2") return true;
|
|
return isRecord(record.evidenceAppendix) || isRecord(record.executiveSummary);
|
|
}
|
|
|
|
function takeText(value: unknown, limitations: Set<string>): string {
|
|
if (typeof value !== "string") return "";
|
|
const analyzed = analyzeOrdinaryReportMarkdown(value);
|
|
for (const note of analyzed.limitations) limitations.add(note);
|
|
return analyzed.prose.trim();
|
|
}
|
|
|
|
function takeTextList(value: unknown, limitations: Set<string>): string[] {
|
|
if (!Array.isArray(value)) return [];
|
|
return value.flatMap((item) => {
|
|
const text = takeText(item, limitations);
|
|
return text ? [text] : [];
|
|
});
|
|
}
|
|
|
|
function narrativeSections(value: unknown, limitations: Set<string>) {
|
|
return recordList(value).flatMap((section) => {
|
|
const narrative = takeText(section.narrative, limitations);
|
|
const actions = takeTextList(section.actions, limitations);
|
|
const caveats = takeTextList(section.caveats, limitations);
|
|
const title = takeText(section.title, limitations);
|
|
if (!title && !narrative && actions.length === 0 && caveats.length === 0) return [];
|
|
return [{ title, narrative, actions, caveats }];
|
|
});
|
|
}
|
|
|
|
function foundationSection(value: unknown, limitations: Set<string>) {
|
|
const section = readRecord(value);
|
|
if (!section) return [];
|
|
return [{
|
|
title: takeText(section.title, limitations),
|
|
narrative: takeText(section.narrative, limitations),
|
|
actions: takeTextList(section.keyFactors, limitations),
|
|
caveats: takeTextList(section.caveats, limitations),
|
|
}];
|
|
}
|
|
|
|
function phaseSection(value: unknown, limitations: Set<string>) {
|
|
const section = readRecord(value);
|
|
if (!section) return [];
|
|
return [{
|
|
title: takeText(section.title, limitations) || takeText(section.phaseLabel, limitations),
|
|
narrative: takeText(section.narrative, limitations),
|
|
actions: takeTextList(section.timingNotes, limitations),
|
|
caveats: takeTextList(section.caveats, limitations),
|
|
}];
|
|
}
|
|
|
|
function actionNotes(value: unknown, limitations: Set<string>) {
|
|
return recordList(value).flatMap((note) => {
|
|
const title = takeText(note.title, limitations);
|
|
const body = takeText(note.note, limitations);
|
|
const priority = note.priority === "now" || note.priority === "next" || note.priority === "watch"
|
|
? note.priority
|
|
: "";
|
|
if (!title && !body) return [];
|
|
return [{ title, note: body, ...(priority ? { priority } : {}) }];
|
|
});
|
|
}
|
|
|
|
function recordList(value: unknown): Record<string, unknown>[] {
|
|
if (!Array.isArray(value)) return [];
|
|
return value.filter(isRecord);
|
|
}
|
|
|
|
function readRecord(value: unknown): Record<string, unknown> | null {
|
|
return isRecord(value) ? value : null;
|
|
}
|
|
|
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
}
|