From 4d3fc21ec941a3fed08e7283e87f3de6955d9812 Mon Sep 17 00:00:00 2001
From: alda-art
Date: Thu, 1 Oct 2026 12:15:26 +0000
Subject: [PATCH 1/3] feat(diagnostics): unify failure evidence and verifiable
exports
---
apps/web/locales/en.json | 8 +-
apps/web/locales/zh-CN.json | 8 +-
apps/web/src/dashboard-ui.ts | 4 +-
apps/web/src/index.ts | 22 ++-
apps/web/src/workspace.ts | 7 +-
apps/web/tests/web.test.ts | 34 +++-
docs/guides/failure-diagnostics.md | 24 +++
packages/cli/README.md | 8 +
packages/cli/src/index.ts | 37 ++++-
packages/cli/src/project-run.ts | 1 +
.../cli/tests/failure-diagnostics.test.ts | 39 +++++
packages/core/src/checks.ts | 2 +
packages/trace/src/diagnostics.ts | 157 ++++++++++++++++++
packages/trace/src/index.ts | 2 +
packages/trace/tests/diagnostics.test.ts | 55 ++++++
15 files changed, 399 insertions(+), 9 deletions(-)
create mode 100644 docs/guides/failure-diagnostics.md
create mode 100644 packages/cli/tests/failure-diagnostics.test.ts
create mode 100644 packages/trace/src/diagnostics.ts
create mode 100644 packages/trace/tests/diagnostics.test.ts
diff --git a/apps/web/locales/en.json b/apps/web/locales/en.json
index 72a211a..a992463 100644
--- a/apps/web/locales/en.json
+++ b/apps/web/locales/en.json
@@ -786,5 +786,11 @@
"进程检查": "Process check",
"HTTP 检查": "HTTP check",
"文件检查": "File check",
- "Docker 检查": "Docker check"
+ "Docker 检查": "Docker check",
+ "统一失败证据": "Unified failure evidence",
+ "缺失源码映射": "Missing source mapping",
+ "源码不可用": "Source unavailable",
+ "原始脱敏证据": "Original redacted evidence",
+ "未记录输出": "No recorded output",
+ "导出诊断包": "Export diagnostics"
}
diff --git a/apps/web/locales/zh-CN.json b/apps/web/locales/zh-CN.json
index ab56523..57fe1e1 100644
--- a/apps/web/locales/zh-CN.json
+++ b/apps/web/locales/zh-CN.json
@@ -786,5 +786,11 @@
"进程检查": "进程检查",
"HTTP 检查": "HTTP 检查",
"文件检查": "文件检查",
- "Docker 检查": "Docker 检查"
+ "Docker 检查": "Docker 检查",
+ "统一失败证据": "统一失败证据",
+ "缺失源码映射": "缺失源码映射",
+ "源码不可用": "源码不可用",
+ "原始脱敏证据": "原始脱敏证据",
+ "未记录输出": "未记录输出",
+ "导出诊断包": "导出诊断包"
}
diff --git a/apps/web/src/dashboard-ui.ts b/apps/web/src/dashboard-ui.ts
index 2eddd0c..1b3aed5 100644
--- a/apps/web/src/dashboard-ui.ts
+++ b/apps/web/src/dashboard-ui.ts
@@ -28,7 +28,9 @@ function renderTrend(items){const box=$('trend-chart'),duration=trendMetric==='d
function diagnosticAdvice(category){return ({configuration:t('检查配置项、命令参数和工作目录,再执行该检查。'),timeout:t('查看最后一条输出及耗时,确认阻塞步骤后再评估超时设置。'),environment:t('检查运行环境、工具安装和资源观测值。'),platform:t('核对检查支持的平台,在对应真实平台补齐验证。'),dependency:t('检查依赖服务及前置步骤是否可用。'),artifact:t('检查关联 artifact、manifest 与运行谱系,重新生成缺失证据。'),policy:t('查看违规断言及轨迹,核对策略要求。'),assertion:t('从失败断言或错误堆栈定位实现与测试预期。'),budget:t('核对执行预算与实际消耗,确认提前终止原因。'),cancelled:t('确认中断原因,补跑尚未完成的检查。')})[category]||t('结合下方原始证据定位原因;当前记录不足以直接确定根因。')}
function diagnosticExcerpt(item){const assertion=(item.assertions||[]).find(a=>!a.passed);if(assertion)return assertion.message||assertion.id||t('断言未通过');const lines=checkLog(item).split('\n').map(s=>s.trim()).filter(s=>s&&!/\[output omitted|\[redacted\]/i.test(s));if(!lines.length&&item.outputTruncated)return t('日志已截断,当前证据不足以判断具体原因。');return lines.find(s=>/error|fail|exception|timeout|错误|失败|阻塞/i.test(s))||lines[0]||t('未记录具体错误输出,请查看执行状态与关联证据。')}
function renderDiagnosis(run){const box=$('diagnosis-list');box.replaceChildren();const bad=run.kind==='project'?(run.checks||[]).filter(c=>['failed','blocked'].includes(c.status)):(run.cases||[]).filter(c=>!c.passed);const gate=run.gate?.passed===false;const missing=(run.coverageSources||[]).filter(s=>s.error);box.closest('.diagnosis-panel').hidden=!bad.length&&!gate&&!missing.length;$('diagnosis-summary').textContent=bad.length?bad.length+t(' 项执行异常。先查看证据,再定位并处理对应检查。'):gate?t('执行结果之外仍有门禁未通过,当前不能判定验证通过。'):run.status==='running'?t('检查正在进行,当前结果尚未完整。'):run.status==='completed'?t('本次已记录的检查通过;结论仅适用于当前检查范围。'):t('运行未完整结束,请核查中断原因与未完成项。');for(const item of bad){const id=item.id||item.caseId,row=text('article',undefined,'diagnosis-row'),content=text('div');content.append(text('span',item.status==='blocked'?t('阻塞 · 尚未完成验证'):t('失败 · 需处理'),'diagnosis-label'),text('h3',id),text('p',diagnosticExcerpt(item),'diagnosis-excerpt'),text('p',t('排查建议:')+diagnosticAdvice(item.category||item.failureCategory),'scope'));row.append(content,button(t('查看证据与定位'),()=>openCheckDrawer(run,id),'secondary'));box.append(row)}if(gate){for(const f of run.gate.failures||[]){const row=text('article',undefined,'diagnosis-row'),content=text('div');content.append(text('span',t('门禁未通过'),'diagnosis-label'),text('h3',f.target||f.code),text('p',f.message||t('查看门禁详情')),text('p',(f.actual!==undefined?t('实际值 ')+f.actual+' · ':'')+(f.required!==undefined?t('要求 ')+f.required:''),'scope'));row.append(content,button(t('查看门禁证据'),()=>showTab('evidence'),'secondary'));box.append(row)}}for(const source of missing){const row=text('article',undefined,'diagnosis-row');row.append(text('p',(source.checkId||source.runId)+':'+source.error),button(t('查看证据来源'),()=>showTab('coverage'),'secondary'));box.append(row)}if(!bad.length&&!gate&&!missing.length){const note=text('div',undefined,'diagnosis-context');note.append(text('p',run.retryOf?t('这是重试子集,通过不代表来源运行的全部检查通过。'):!(run.coverageSources||[]).some(s=>s.coverage)?t('覆盖率未采集,无法据此判断源码覆盖程度。'):t('覆盖率仅适用于声明的文件;查看未覆盖位置与测量精度。')));note.append(button(run.retryOf?t('查看来源运行'):t('查看覆盖范围'),()=>run.retryOf?select(run.retryOf):showTab('coverage'),'text-button'));box.append(note)}}
-function openCheckDrawer(run,id){const project=run.kind==='project',item=project?run.checks.find(c=>c.id===id):run.cases.find(c=>c.caseId===id);if(!item)return;const dialog=$('check-drawer'),body=$('drawer-content');$('drawer-title').textContent=id;body.replaceChildren(text('p',run.runId+' · '+date(run.startedAt),'scope'));const stats=text('div',undefined,'drawer-stats');stats.append(text('strong',project?(labels[item.status]||item.status):item.passed?t('通过'):t('失败')),text('strong',time(project?item.durationMs:item.metrics?.latencyMs)));body.append(stats);const failed=project?['failed','blocked'].includes(item.status):!item.passed;if(failed)body.append(text('h3',t('已记录的异常')),text('p',diagnosticExcerpt(item),'notice error'),text('h3',t('建议排查路径')),text('p',diagnosticAdvice(item.category||item.failureCategory),'scope'));if(project){body.append(text('h3',t('执行上下文')),text('pre',t('类别:')+(item.category||t('未记录'))+t('\n检查退出码:')+item.exitCode+t('\n进程退出码:')+(item.processExit??t('未记录'))+t('\n工作目录:')+(item.cwd||t('未记录'))+(item.httpStatus?'\nHTTP:'+item.httpStatus:'')));if(item.command)body.append(text('h3',t('执行命令与参数')),text('pre',item.command+t('\n参数:')+JSON.stringify(item.args||[])));body.append(text('h3',t('标准错误 · 已脱敏')),text('pre',checkLogStream(item,'stderr')||t('无标准错误输出')),text('h3',t('标准输出 · 已脱敏')),text('pre',checkLogStream(item,'stdout')||t('无标准输出')));if(item.outputTruncated)body.append(text('p',t('输出采用有界保留;展示错误附近与末尾上下文,未保留全部日志。'),'notice'));if(item.childRun)body.append(button(t('进入关联 Agent 用例'),async()=>{dialog.close();await select(item.childRun.runId);showTab('cases')},'secondary'))}else{for(const a of item.assertions||[])body.append(text('p',(a.passed?'✓ ':'! ')+(a.message||a.id),'assertion'));if(item.trajectoryId)body.append(text('p',t('关联轨迹:')+item.trajectoryId,'scope'))}const logs=checkLog(item);const locations=[...new Set(logs.match(/(?:[A-Za-z]:[\\/]|(?:\.{0,2}[\\/])?)[^\s<>"()]+\.(?:[cm]?[jt]sx?|py|go|rs|java|vue|svelte):\d+(?::\d+)?/g)||[])].slice(0,12);body.append(text('h3',t('日志中的文件位置')));body.append(text('p',locations.length?t('以下位置从日志提取,需结合堆栈判断是否为根因。'):t('当前证据未提供可识别的文件行号。可先从执行命令、工作目录或关联轨迹排查。'),'scope'));if(locations.length)body.append(text('pre',locations.join('\n')));if(failed)body.append(button(t('进入问题处理流程'),()=>{dialog.close();showTab('improve');const card=[...document.querySelectorAll('.issue-card')].find(node=>node.querySelector('h3')?.textContent===id);card?.scrollIntoView({block:'center',behavior:'smooth'})},'secondary'));$('drawer-full').onclick=async()=>{dialog.close();await select(run.runId);showTab(project?'checks':'cases');const container=$(project?'checks':'cases-content');const row=[...container.children].find(el=>el.dataset.checkId===id);if(row){const details=row.matches('details')?row:row.querySelector('details');if(details)details.open=true;row.tabIndex=-1;row.focus();row.scrollIntoView({block:'center',behavior:'smooth'})}};if(!dialog.open)dialog.showModal()}
+function openCheckDrawer(run,id){const project=run.kind==='project',item=project?run.checks.find(c=>c.id===id):run.cases.find(c=>c.caseId===id);if(!item)return;const dialog=$('check-drawer'),body=$('drawer-content');$('drawer-title').textContent=id;body.replaceChildren(text('p',run.runId+' · '+date(run.startedAt),'scope'));const stats=text('div',undefined,'drawer-stats');stats.append(text('strong',project?(labels[item.status]||item.status):item.passed?t('通过'):t('失败')),text('strong',time(project?item.durationMs:item.metrics?.latencyMs)));body.append(stats);const failed=project?['failed','blocked'].includes(item.status):!item.passed;if(failed)body.append(text('h3',t('已记录的异常')),text('p',diagnosticExcerpt(item),'notice error'),text('h3',t('建议排查路径')),text('p',diagnosticAdvice(item.category||item.failureCategory),'scope'));if(project){body.append(text('h3',t('执行上下文')),text('pre',t('类别:')+(item.category||t('未记录'))+t('\n检查退出码:')+item.exitCode+t('\n进程退出码:')+(item.processExit??t('未记录'))+t('\n工作目录:')+(item.cwd||t('未记录'))+(item.httpStatus?'\nHTTP:'+item.httpStatus:'')));if(item.command)body.append(text('h3',t('执行命令与参数')),text('pre',item.command+t('\n参数:')+JSON.stringify(item.args||[])));body.append(text('h3',t('标准错误 · 已脱敏')),text('pre',checkLogStream(item,'stderr')||t('无标准错误输出')),text('h3',t('标准输出 · 已脱敏')),text('pre',checkLogStream(item,'stdout')||t('无标准输出')));if(item.outputTruncated)body.append(text('p',t('输出采用有界保留;展示错误附近与末尾上下文,未保留全部日志。'),'notice'));if(item.childRun)body.append(button(t('进入关联 Agent 用例'),async()=>{dialog.close();await select(item.childRun.runId);showTab('cases')},'secondary'))}else{for(const a of item.assertions||[])body.append(text('p',(a.passed?'✓ ':'! ')+(a.message||a.id),'assertion'));if(item.trajectoryId)body.append(text('p',t('关联轨迹:')+item.trajectoryId,'scope'))}renderFailureDiagnostic(run,id,body);const logs=checkLog(item);const locations=[...new Set(logs.match(/(?:[A-Za-z]:[\\/]|(?:\.{0,2}[\\/])?)[^\s<>"()]+\.(?:[cm]?[jt]sx?|py|go|rs|java|vue|svelte):\d+(?::\d+)?/g)||[])].slice(0,12);body.append(text('h3',t('日志中的文件位置')));body.append(text('p',locations.length?t('以下位置从日志提取,需结合堆栈判断是否为根因。'):t('当前证据未提供可识别的文件行号。可先从执行命令、工作目录或关联轨迹排查。'),'scope'));if(locations.length)body.append(text('pre',locations.join('\n')));if(failed)body.append(button(t('进入问题处理流程'),()=>{dialog.close();showTab('improve');const card=[...document.querySelectorAll('.issue-card')].find(node=>node.querySelector('h3')?.textContent===id);card?.scrollIntoView({block:'center',behavior:'smooth'})},'secondary'));$('drawer-full').onclick=async()=>{dialog.close();await select(run.runId);showTab(project?'checks':'cases');const container=$(project?'checks':'cases-content');const row=[...container.children].find(el=>el.dataset.checkId===id);if(row){const details=row.matches('details')?row:row.querySelector('details');if(details)details.open=true;row.tabIndex=-1;row.focus();row.scrollIntoView({block:'center',behavior:'smooth'})}};if(!dialog.open)dialog.showModal()}
+
+function renderFailureDiagnostic(run,id,body){const failure=run.diagnostics?.failures.find(item=>item.checkId===id||item.caseId===id);if(!failure)return;body.append(text('h3',t('统一失败证据')));body.append(text('pre',JSON.stringify({testNames:failure.testNames,assertions:failure.assertions,relatedFailures:failure.relatedFailures,rootCause:failure.rootCause,sourceMapping:failure.sourceMapping},null,2)));for(const location of failure.locations){body.append(button(location.path+':'+location.line+' (path-line)',async()=>{try{const data=await api('/api/structure?runId='+encodeURIComponent(run.runId));const node=data.structure.nodes.find(node=>node.kind==='file'&&node.path===location.path);if(!node)throw new Error(t('缺失源码映射'));const source=await api('/api/structure/source?runId='+encodeURIComponent(run.runId)+'&nodeId='+encodeURIComponent(node.id)+'&line='+location.line);body.append(text('pre',source.lines?source.lines.map((line,index)=>(source.startLine+index)+': '+line).join('\n'):t('源码不可用')));}catch(error){body.append(text('p',error.message,'notice'))}},'secondary'))}body.append(text('h3',t('原始脱敏证据')),text('pre',[failure.originalError.stderr,failure.originalError.stdout].filter(Boolean).join('\n')||t('未记录输出')),text('p',failure.nextSteps.join('\n'),'scope'));const download=text('a',t('导出诊断包'),'secondary');download.href='/api/runs/'+encodeURIComponent(run.runId)+'/diagnostics/bundle';download.download='diagnostics.json';body.append(download)}
function renderOutcome(run){const box=$('outcome-chart');box.replaceChildren();const total=Math.max(0,Number(run.total)||0),passed=Math.min(total,Math.max(0,Number(run.passed)||0)),failed=run.kind==='project'?(run.checks||[]).filter(c=>c.status==='failed'||c.status==='blocked').length:run.cases.filter(c=>!c.passed).length,other=Math.max(0,total-passed-failed);if(!total){box.append(text('div',t('本次未计划检查项'),'empty'));return}const a=passed/total*100,b=Math.min(100,(passed+failed)/total*100),layout=text('div',undefined,'outcome-layout'),donut=text('div',undefined,'donut');donut.style.background='conic-gradient(#668878 0% '+a+'%,#df795f '+a+'% '+b+'%,#e6e3dc '+b+'% 100%)';donut.setAttribute('role','img');donut.setAttribute('aria-label',t('通过 ')+passed+t(',失败或阻塞 ')+failed+t(',其余 ')+other);const center=text('div');center.append(text('strong',Math.round(a)+'%'),text('small',t('计划项通过率')));donut.append(center);const key=text('div',undefined,'chart-key');for(const [label,value,color] of [[t('通过'),passed,'#668878'],[t('失败 / 阻塞'),failed,'#df795f'],[t('待执行 / 其他'),other,'#b6b2a9']]){const row=text('div'),dot=text('i');dot.style.background=color;row.append(dot,text('span',label),text('strong',value));key.append(row)}layout.append(donut,key);box.append(layout)}
function barRow(label,value,max,display){const row=text('div',undefined,'bar-row'),name=text('span',label,'bar-label'),track=text('div',undefined,'bar-track'),fill=text('i');name.title=label;fill.style.width=(max?Math.max(0,Math.min(100,value/max*100)):0)+'%';track.append(fill);row.append(name,track,text('strong',display));return row}
diff --git a/apps/web/src/index.ts b/apps/web/src/index.ts
index 856b4d8..ccb573d 100644
--- a/apps/web/src/index.ts
+++ b/apps/web/src/index.ts
@@ -26,6 +26,8 @@ import {
} from "@canary/improvement";
import {
ArtifactIntegrityError,
+ buildRunDiagnostics,
+ createDiagnosticBundle,
FileArtifactRepository,
redactRunSnapshot,
redactValue,
@@ -205,7 +207,8 @@ async function handleRequest(
return;
}
const value = parts[3] ? workspace?.read(parts[3]) : workspace?.list();
- writeJson(response, value ? 200 : 404, value ?? { error: "Run not found" });
+ response.writeHead(value ? 200 : 404, { "content-type": "application/json; charset=utf-8" });
+ response.end(JSON.stringify(value ?? { error: "Run not found" }));
return;
}
if (parts[0] === "api" && parts[1] === "structure" && (parts.length === 2 || (parts.length === 3 && parts[2] === "source"))) {
@@ -398,6 +401,23 @@ async function handleRequest(
response.end("Not found");
return;
}
+ if (parts[3] === "diagnostics") {
+ if (request.method !== "GET") { writeJson(response, 405, { error: "GET required" }); return; }
+ const root = artifacts ? resolveProjectRootFromArtifacts(artifacts.rootDir) : process.cwd();
+ if (parts[4] === "bundle") {
+ if (!artifacts || artifacts.verify(runId).status !== "verified") { writeJson(response, 409, { error: "Export requires sealed, verified evidence" }); return; }
+ response.setHeader("content-disposition", 'attachment; filename="diagnostics.json"');
+ const source = artifacts.readRun(runId);
+ if (!source) { writeJson(response, 404, { error: "Run not found" }); return; }
+ const bundle = createDiagnosticBundle(source, root);
+ response.writeHead(200, { "content-type": "application/json; charset=utf-8" });
+ response.end(JSON.stringify(bundle));
+ } else {
+ response.writeHead(200, { "content-type": "application/json; charset=utf-8" });
+ response.end(JSON.stringify(buildRunDiagnostics(store.get(runId)!, root)));
+ }
+ return;
+ }
if (parts[3] === "retry") {
if (request.method !== "POST") {
writeJson(response, 405, { error: "POST required" });
diff --git a/apps/web/src/workspace.ts b/apps/web/src/workspace.ts
index 1eb7b65..3990501 100644
--- a/apps/web/src/workspace.ts
+++ b/apps/web/src/workspace.ts
@@ -1,8 +1,9 @@
import { existsSync, readdirSync, readFileSync, statSync } from "node:fs";
-import { join } from "node:path";
+import { join, resolve } from "node:path";
import type { CoverageSummary, RunSnapshot } from "@canary/core";
import {
ArtifactIntegrityError,
+ buildRunDiagnostics,
FileArtifactRepository,
RunStore,
readArtifactManifest,
@@ -221,7 +222,7 @@ export class WorkspaceReader {
});
}
}
- return this.store.sanitize({
+ return { ...this.store.sanitize({
...summary(run),
issues,
activeCheck: run.activeCheck,
@@ -243,6 +244,6 @@ export class WorkspaceReader {
})),
timeline: run.events.slice(-60).map((event) => ({ type: event.type, at: "at" in event ? event.at : undefined })),
improvements: run.improvements ?? this.repository?.readJson(id, "improvement.json") ?? [],
- });
+ }), diagnostics: buildRunDiagnostics(run, this.repository ? resolve(this.repository.rootDir, "../..") : process.cwd()) };
}
}
diff --git a/apps/web/tests/web.test.ts b/apps/web/tests/web.test.ts
index 162974f..7ffd53e 100644
--- a/apps/web/tests/web.test.ts
+++ b/apps/web/tests/web.test.ts
@@ -5,7 +5,7 @@ import { join } from "node:path";
import { request } from "node:http";
import { execFileSync } from "node:child_process";
import { createWebServer, FileArtifactRepository, RunStore } from "../src/index.js";
-import { beginArtifacts, sealArtifacts, writePrivateJson } from "@canary/trace";
+import { buildRunDiagnostics, verifyDiagnosticBundle, beginArtifacts, sealArtifacts, writePrivateJson } from "@canary/trace";
import { buildStructure } from "@canary/structure";
function get(url: string): Promise<{ status: number; body: string }> {
@@ -32,6 +32,38 @@ function coverage(runId: string, status: "provisional" | "final" = "final", cove
}
describe("web run store and HTTP/SSE", () => {
+ it("serves the shared diagnosis and a verifiable bundle without truncating retained evidence", async () => {
+ const root = mkdtempSync(join(tmpdir(), "canary-web-diagnostics-"));
+ const artifactRoot = join(root, ".canary", "artifacts");
+ const store = new RunStore();
+ const run = store.create(1, "run_diagnostics");
+ const check = { id: "test", type: "command" as const, version: 1 as const, required: true, status: "failed" as const, evidence: "verified" as const, exitCode: 1, category: "assertion" as const, retryable: false, durationMs: 1, cwd: root, envAllowlist: ["API_KEY"], outputEvidence: { policy: "bounded-redacted-lines-v1" as const, stdout: [], stderr: ["x".repeat(3000), "AssertionError: wrong value", "password=superprivate"] } };
+ const snapshot = { ...run, status: "failed" as const, checks: [check] };
+ const dir = join(artifactRoot, run.runId);
+ beginArtifacts(dir);
+ writePrivateJson(join(dir, "run.json"), snapshot, { maxStringLength: Infinity });
+ sealArtifacts(dir);
+ const repo = new FileArtifactRepository(artifactRoot);
+ const expected = buildRunDiagnostics(repo.readRun(run.runId)!, root);
+ const web = createWebServer(new RunStore(), "127.0.0.1", 0, artifactRoot);
+ const listening = await web.listen();
+ try {
+ const diagnostics = await get(`${listening.url}/api/runs/${run.runId}/diagnostics`);
+ expect(diagnostics.status).toBe(200);
+ expect(JSON.parse(diagnostics.body)).toEqual(expected);
+ const workspace = JSON.parse((await get(`${listening.url}/api/workspace/runs/${run.runId}`)).body);
+ expect(workspace.diagnostics).toEqual(expected);
+ const exported = await get(`${listening.url}/api/runs/${run.runId}/diagnostics/bundle`);
+ expect(exported.status).toBe(200);
+ expect(exported.body).not.toContain("superprivate");
+ expect(verifyDiagnosticBundle(JSON.parse(exported.body))).toBe(true);
+ expect(JSON.parse(exported.body).files["diagnostics.json"]).toEqual(expected);
+ expect((await post(`${listening.url}/api/runs/${run.runId}/diagnostics/bundle`, {})).status).toBe(405);
+ writeFileSync(join(dir, "run.json"), "{}");
+ expect((await get(`${listening.url}/api/runs/${run.runId}/diagnostics/bundle`)).status).toBe(409);
+ } finally { await close(web.server); }
+ });
+
it("serves the sealed historical structure and rejects a damaged graph", async () => {
const root = mkdtempSync(join(tmpdir(), "canary-web-structure-"));
writeFileSync(join(root, "index.ts"), "export function value() { return 1; }\n");
diff --git a/docs/guides/failure-diagnostics.md b/docs/guides/failure-diagnostics.md
new file mode 100644
index 0000000..f49208b
--- /dev/null
+++ b/docs/guides/failure-diagnostics.md
@@ -0,0 +1,24 @@
+# Unified failure evidence / 统一失败证据
+
+Inspect a failed check in the workspace drawer or run `canary diagnostics `. Both use the same structured diagnosis. Existing JSON, Markdown and JUnit reports remain compatible.
+
+在页面检查详情中查看统一失败证据,或执行 `canary diagnostics `。页面和 CLI 使用同一诊断模型,既有 JSON、Markdown、JUnit 报告保持兼容。
+
+Export / 导出:
+
+```bash
+canary diagnostics --out diagnostics.json
+canary diagnostics verify diagnostics.json
+```
+
+Export requires a sealed, verified run and refuses to overwrite files. The bundle includes redacted original evidence, failed assertions, reported stack positions, run ID/start time, recorded commit/runtime, commands, working directories and required environment names. It omits environment values and credentials. SHA-256 and byte lengths check completeness and corruption, not who produced the package. Verify returns exit code 0 on success and 1 on failure.
+
+导出要求运行已封存且通过完整性检查,不覆盖已有文件。包内保留脱敏原始错误、失败断言、堆栈位置、运行身份、已记录的 commit 与运行环境、命令、工作目录和所需环境变量名称,不包含环境变量值和凭据。SHA-256 与字节长度用于校验完整性,不能证明来源身份。校验成功退出码为 0,失败为 1。
+
+Source positions are reported `path-line` evidence, not verified source maps. The page loads source through hash verification; absent snapshots or mismatching source remain unavailable. Missing positions are labeled `missing`; root causes are `unknown`. Only explicitly recorded dependencies associate failures; a prerequisite explanation is a `hypothesis` requiring confirmation. Similar messages alone never establish a relationship. Older reports without runtime or dependency metadata remain readable with explicit limitations.
+
+日志位置标记为 `path-line`,不视为已验证的源码映射。页面通过源码哈希检查读取片段;缺失快照或哈希不匹配时显示不可用。缺失位置标记为 `missing`,未确认根因标记为 `unknown`。仅通过已记录的显式依赖关联失败;前置失败解释标记为 `hypothesis`,需要确认。不会仅凭类似错误文本关联失败。旧报告缺失运行环境或依赖元数据时仍可读取,并明确显示限制。
+
+Follow the package's next steps: inspect original evidence, restore the recorded context, confirm the location, repair the failing behavior and rerun the same check. Truncated logs are labeled; the package cannot recover discarded output.
+
+按包内下一步说明处理:阅读原始证据,恢复记录的运行上下文,确认源码位置,修复并重跑同一检查。已截断日志会明确标记,诊断包不能恢复已丢弃的输出。
diff --git a/packages/cli/README.md b/packages/cli/README.md
index aa956f7..e76316e 100644
--- a/packages/cli/README.md
+++ b/packages/cli/README.md
@@ -14,3 +14,11 @@
## 当前使用入口
命令、项目定位和产物位置以[运行指南](../../docs/guides/running-and-ui.md)及[安装指南](../../docs/guides/getting-started.md)为准。全局启动器保留调用目录,用 `CANARY_HOME` 指向安装仓库;本地 `canary.config.ts` 优先于安装 Demo。所有读写命令共用 `ProjectContext`。
+
+## Failure diagnostics
+
+`canary diagnostics [--json] [--config ]` reads the same structured failure evidence shown in the check drawer. Export a sealed run with `canary diagnostics --out diagnostics.json`, then validate it with `canary diagnostics verify diagnostics.json` (exit 0 for valid, 1 for invalid). Existing report formats and exit codes are unchanged. Export refuses to overwrite an existing file.
+
+The JSON bundle contains `diagnostics.json`, `NEXT-STEPS.txt` and a manifest with byte lengths and SHA-256 hashes. Hashes detect corruption, not authenticity. Only retained, redacted errors are included; environment requirements contain names, never values. Commit/runtime information is explicitly unavailable for historical runs that did not record it.
+
+Each failure keeps its check or case identity, test names where recognizable, failed assertions, stack frames and reported source positions. Positions have `path-line` precision; they do not establish a verified source map or a root cause. The page opens source only through the existing hash-checked source viewer. Missing positions stay `missing`, causes stay `unknown`, and declared prerequisite failures are labeled as `hypothesis`. Similar text never groups unrelated failures.
diff --git a/packages/cli/src/index.ts b/packages/cli/src/index.ts
index 4d13d9b..8144a58 100644
--- a/packages/cli/src/index.ts
+++ b/packages/cli/src/index.ts
@@ -7,7 +7,7 @@ import { resolve, relative } from "node:path";
import { pathToFileURL, fileURLToPath } from "node:url";
import { createRequire } from "node:module";
import type { RunSnapshot } from "@canary/core";
-import { FileArtifactRepository, RunStore, applyRetention, planRetention, verifyArtifacts, redactValue, readArtifactManifest, safeArtifactPath, writePrivateJson, writePrivateText, sha256 } from "@canary/trace";
+import { FileArtifactRepository, RunStore, applyRetention, planRetention, verifyArtifacts, redactValue, readArtifactManifest, safeArtifactPath, writePrivateJson, writePrivateText, sha256, buildRunDiagnostics, createDiagnosticBundle, verifyDiagnosticBundle } from "@canary/trace";
import { projectChecksConfigSchema, ciResultSchema } from "@canary/core";
import { runProjectSession } from "./project-session.js";
import { discoverProject } from "./discovery.js";
@@ -84,6 +84,8 @@ const USAGE = `Usage: canary run [--ci] [--project ] [--affected --ba
canary soft-trial prepare --experience --regression --holdout
canary soft-trial validate|approve|run|rollback [--actor ] [--reason ]
canary replay [--headless] [--no-open]
+ canary diagnostics [--out ] [--config ]
+ canary diagnostics verify
canary verify [--json] [--config ]
canary prune [--apply] [--json] [--config ]
canary host discover [--config ]
@@ -1336,6 +1338,39 @@ export async function main(argv = process.argv.slice(2)): Promise {
return 0;
} catch (error) { console.error(error instanceof Error ? error.message : String(error)); return 1; }
}
+ if (command === "diagnostics") {
+ try {
+ if (rest[0] === "verify") {
+ if (rest.length !== 2) throw new Error("Use canary diagnostics verify ");
+ const valid = verifyDiagnosticBundle(JSON.parse(readFileSync(resolve(rest[1]!), "utf8")));
+ console.log(JSON.stringify({ kind: "canary.diagnostic-verification", valid }));
+ return valid ? 0 : 1;
+ }
+ const runId = rest[0];
+ if (!runId || runId.startsWith("--")) throw new Error("A run ID is required");
+ for (let i = 1; i < rest.length; i++) {
+ if (rest[i] === "--json") continue;
+ if (!["--out", "--config"].includes(rest[i]!) || !rest[i + 1] || rest[i + 1]!.startsWith("--")) throw new Error("Invalid diagnostics arguments");
+ i++;
+ }
+ const context = resolveProjectContext({ configPath });
+ const repository = new FileArtifactRepository(context.artifactRoot);
+ const integrity = repository.verify(runId);
+ if (integrity.status === "invalid") throw new Error("Run evidence failed integrity verification");
+ const run = repository.readRun(runId);
+ if (!run) throw new Error("Run not found");
+ const output = flagValue(rest, "--out");
+ if (output) {
+ if (integrity.status !== "verified") throw new Error("Export requires a sealed, verified source run");
+ const bundle = createDiagnosticBundle(run, context.projectRoot);
+ const destination = resolve(output);
+ mkdirSync(resolve(destination, ".."), { recursive: true });
+ writeFileSync(destination, JSON.stringify(bundle, null, 2) + "\n", { flag: "wx", mode: 0o600 });
+ console.log(JSON.stringify({ kind: "canary.diagnostic-export", path: destination, manifest: bundle.manifest }));
+ } else console.log(JSON.stringify(buildRunDiagnostics(run, context.projectRoot), null, 2));
+ return 0;
+ } catch (error) { console.error(error instanceof Error ? error.message : String(error)); return 1; }
+ }
if (command === "export") {
const output = flagValue(rest, "--out");
if (!output) {
diff --git a/packages/cli/src/project-run.ts b/packages/cli/src/project-run.ts
index c3d3d25..9295aa9 100644
--- a/packages/cli/src/project-run.ts
+++ b/packages/cli/src/project-run.ts
@@ -237,6 +237,7 @@ export async function runProjectChecks(
outputTruncated:
outcome.outputTruncated || (outcome.stdout?.length ?? 0) > 2048 || (outcome.stderr?.length ?? 0) > 2048,
id: check.id,
+ dependsOn: check.dependsOn,
type: check.type,
version: check.version,
required: check.required || Boolean(session?.lineage?.retryOf),
diff --git a/packages/cli/tests/failure-diagnostics.test.ts b/packages/cli/tests/failure-diagnostics.test.ts
new file mode 100644
index 0000000..60f0bc9
--- /dev/null
+++ b/packages/cli/tests/failure-diagnostics.test.ts
@@ -0,0 +1,39 @@
+import { describe, expect, it, vi } from "vitest";
+import { mkdtempSync, readFileSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+import { beginArtifacts, sealArtifacts, writePrivateJson, verifyDiagnosticBundle } from "@canary/trace";
+import { main } from "../src/index.js";
+
+describe("diagnostic CLI", () => {
+ it("exports and verifies sealed evidence and refuses overwrite, corruption and invalid arguments", async () => {
+ const root = mkdtempSync(join(tmpdir(), "canary-diagnostic-cli-"));
+ const config = join(root, "canary.config.ts");
+ writeFileSync(config, "export default {};\n");
+ const dir = join(root, ".canary", "artifacts", "run_diag");
+ beginArtifacts(dir);
+ writePrivateJson(join(dir, "run.json"), { runId: "run_diag", status: "failed", startedAt: "2026-10-01T00:00:00Z", totalCases: 1, completedCases: 1, passedCases: 0, results: [], events: [], checks: [{ id: "test", type: "command", status: "failed", category: "assertion", cwd: root, envAllowlist: [], stderr: "AssertionError: password=superprivate" }] });
+ sealArtifacts(dir);
+ const output = join(root, "diagnostics.json");
+ const log = vi.spyOn(console, "log").mockImplementation(() => {});
+ const error = vi.spyOn(console, "error").mockImplementation(() => {});
+ try {
+ expect(await main(["diagnostics", "run_diag", "--config", config])).toBe(0);
+ const diagnostic = JSON.parse(String(log.mock.calls.at(-1)![0]));
+ expect(diagnostic.kind).toBe("canary.diagnostics");
+ expect(await main(["diagnostics", "run_diag", "--config", config, "--out", output])).toBe(0);
+ const bundle = JSON.parse(readFileSync(output, "utf8"));
+ expect(bundle.files["diagnostics.json"]).toEqual(diagnostic);
+ expect(verifyDiagnosticBundle(bundle)).toBe(true);
+ expect(readFileSync(output, "utf8")).not.toContain("superprivate");
+ expect(await main(["diagnostics", "verify", output])).toBe(0);
+ expect(await main(["diagnostics", "run_diag", "--config", config, "--out", output])).toBe(1);
+ expect(await main(["diagnostics", "run_diag", "--config", config, "--bad"])).toBe(1);
+ bundle.files["NEXT-STEPS.txt"] += "changed";
+ writeFileSync(output, JSON.stringify(bundle));
+ expect(await main(["diagnostics", "verify", output])).toBe(1);
+ writeFileSync(join(dir, "run.json"), "{}");
+ expect(await main(["diagnostics", "run_diag", "--config", config])).toBe(1);
+ } finally { log.mockRestore(); error.mockRestore(); }
+ });
+});
diff --git a/packages/core/src/checks.ts b/packages/core/src/checks.ts
index 55063ef..b720d10 100644
--- a/packages/core/src/checks.ts
+++ b/packages/core/src/checks.ts
@@ -91,6 +91,8 @@ export function defineProjectConfig(input: z.input;
+ originalError: { stdout: string; stderr: string; truncated: boolean };
+ stack: string[];
+ locations: Array<{ path: string; line: number; column?: number; mapping: "path-line" }>;
+ sourceMapping: "path-line" | "missing";
+ relatedFailures: Array<{ id: string; evidence: "declared-dependency" }>;
+ rootCause: { status: "unknown" | "hypothesis"; reason: string };
+ nextSteps: string[];
+}
+export interface RunDiagnostics {
+ v: 1;
+ kind: "canary.diagnostics";
+ runId: string;
+ reproduction: {
+ gitCommit?: string;
+ identity: { runId: string; startedAt: string };
+ runtime?: { node: string; platform: string; arch: string };
+ lockfiles: Record;
+ environmentNames: string[];
+ sourceHash?: string;
+ configHash?: string;
+ commands: Array<{ checkId: string; command?: string; args?: string[]; cwd: string; environmentNames: string[] }>;
+ };
+ failures: FailureDiagnostic[];
+ limitations: string[];
+}
+
+const clean = (value: T): T => redactValue(value, { maxStringLength: Infinity }) as T;
+const stripAnsi = (value: string) => value.replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, "");
+function locations(text: string, root: string, cwd: string): FailureDiagnostic["locations"] {
+ const found: FailureDiagnostic["locations"] = [];
+ for (const row of stripAnsi(text).split(/\r?\n/)) {
+ const python = row.match(/\bFile ["']([^"']+)["'], line (\d+)/);
+ const frame = row.match(/(?:\(|\s|^)([^\s()]+\.(?:[cm]?[jt]sx?|py|go|rs|java|vue|svelte)):(\d+)(?::(\d+))?/);
+ const match = python ?? frame;
+ if (!match) continue;
+ let raw = match[1]!;
+ if (raw.startsWith("file://")) { try { raw = fileURLToPath(raw); } catch { continue; } }
+ const path = relative(resolve(root), resolve(root, cwd, raw)).replaceAll("\\", "/");
+ const line = Number(match[2]);
+ if (!path || isAbsolute(path) || path === ".." || path.startsWith("../") || !Number.isSafeInteger(line) || line < 1) continue;
+ const column = !python && match[3] ? Number(match[3]) : undefined;
+ if (!found.some((item) => item.path === path && item.line === line && item.column === column))
+ found.push({ path, line, ...(column ? { column } : {}), mapping: "path-line" });
+ }
+ return found.slice(0, 40);
+}
+function failure(id: string, category: string, stdout: string, stderr: string, root: string, cwd: string): FailureDiagnostic {
+ const text = [stderr, stdout].join("\n");
+ const mapped = locations(text, root, cwd);
+ return {
+ id, category, testNames: [], assertions: [], originalError: { stdout, stderr, truncated: false },
+ stack: stripAnsi(text).split(/\r?\n/).filter((row) => /^\s*at\s|\bFile ["']/.test(row)).slice(0, 80),
+ locations: mapped, sourceMapping: mapped.length ? "path-line" : "missing", relatedFailures: [],
+ rootCause: { status: "unknown", reason: "No evidence establishes a root cause; stack locations are reported positions, not verified source maps." },
+ nextSteps: ["Inspect the original redacted error and failed assertions.", "Confirm the recorded commit, runtime, working directory and required environment names.", "Inspect the reported source location; repair and rerun the same check or case."],
+ };
+}
+/** Derive one shared view without changing the original report or evaluation result. */
+export function buildRunDiagnostics(input: RunSnapshot, projectRoot: string): RunDiagnostics {
+ const commandSecrets: string[] = [];
+ for (const check of input.checks ?? []) for (const [index, arg] of (check.args ?? []).entries()) {
+ if (!arg.startsWith("-")) continue;
+ const [name, ...value] = arg.split("=");
+ if (!SECRET_KEY.test(name!.replace(/^-+/, ""))) continue;
+ const secret = value.length ? value.join("=") : check.args?.[index + 1];
+ if (secret && !secret.startsWith("--")) commandSecrets.push(secret);
+ }
+ const run = redactValue(input, { maxStringLength: Infinity, secretValues: commandSecrets }) as RunSnapshot;
+ const failures: FailureDiagnostic[] = [];
+ for (const check of run.checks ?? []) {
+ if (check.status !== "failed" && check.status !== "blocked") continue;
+ const stdout = check.outputEvidence?.stdout.join("\n") || check.stdout || "";
+ const stderr = check.outputEvidence?.stderr.join("\n") || check.stderr || "";
+ const item = failure(`${run.runId}:check:${check.id}`, check.category, stdout, stderr, projectRoot, check.cwd);
+ item.checkId = check.id;
+ item.originalError.truncated = Boolean(check.outputTruncated);
+ item.testNames = stripAnsi([stderr, stdout].join("\n")).split(/\r?\n/).flatMap((row) => {
+ const match = row.match(/^\s*(?:FAIL\s+|[×✕]\s+|not ok \d+\s*-?\s*)(.+)/);
+ return match ? [match[1]!.trim()] : [];
+ }).slice(0, 80);
+ failures.push(item);
+ }
+ for (const result of run.results.filter((item) => !item.passed)) {
+ const assertions = result.assertions.filter((item) => !item.passed).map(({ id, message, details }) => ({ id, message, details }));
+ const errors = run.events.flatMap((event) => event.type === "execution.failed" && event.executionId === result.executionId ? [event.error] : []);
+ const assertionEvidence = assertions.map((a) => [a.message, a.details ? JSON.stringify(a.details) : ""].filter(Boolean).join("\n"));
+ const item = failure(`${run.runId}:case:${result.executionId}`, result.failureCategory ?? "assertion", "", [...errors, ...assertionEvidence].join("\n"), projectRoot, ".");
+ Object.assign(item, { caseId: result.caseId, executionId: result.executionId, testNames: [result.caseId], assertions });
+ failures.push(item);
+ }
+ for (const [index, gate] of (run.gate?.passed === false ? run.gate.failures : []).entries()) {
+ failures.push(failure(`${run.runId}:gate:${index}`, "coverage_below_threshold", "", gate.message, projectRoot, "."));
+ }
+ for (const item of failures) {
+ const check = run.checks?.find((check) => check.id === item.checkId);
+ if (check?.category !== "dependency") continue;
+ for (const dependency of check.dependsOn ?? []) {
+ const parent = failures.find((other) => other.checkId === dependency);
+ if (parent) item.relatedFailures.push({ id: parent.id, evidence: "declared-dependency" });
+ }
+ if (item.relatedFailures.length) item.rootCause = { status: "hypothesis", reason: "A declared prerequisite failed. This explains blocking, but does not establish the prerequisite's root cause." };
+ }
+ const recorded = run.evidence?.reproduction;
+ return clean({
+ v: 1, kind: "canary.diagnostics", runId: run.runId,
+ reproduction: {
+ gitCommit: recorded?.gitCommit,
+ identity: { runId: run.runId, startedAt: run.startedAt },
+ runtime: recorded ? { node: recorded.node, platform: recorded.platform, arch: recorded.arch } : undefined,
+ lockfiles: recorded?.lockfiles ?? {},
+ environmentNames: recorded?.environmentNames ?? [],
+ sourceHash: recorded?.sourceHash,
+ configHash: recorded?.configHash,
+ commands: (run.checks ?? []).map((check) => ({ checkId: check.id, command: check.command, args: check.args, cwd: check.cwd, environmentNames: check.envAllowlist })),
+ }, failures,
+ limitations: ["Only retained redacted evidence is included; credentials and environment values are omitted.", "Missing source maps and unknown causes remain explicit. Similar messages never establish a failure relationship.", ...(!recorded ? ["This historical run has no recorded runtime or commit context."] : [])],
+ } as RunDiagnostics);
+}
+
+export interface DiagnosticBundle {
+ v: 1;
+ kind: "canary.diagnostic-bundle";
+ files: { "diagnostics.json": RunDiagnostics; "NEXT-STEPS.txt": string };
+ manifest: Array<{ path: string; bytes: number; sha256: string }>;
+}
+const fileContent = (value: unknown) => typeof value === "string" ? value : JSON.stringify(value);
+export function createDiagnosticBundle(run: RunSnapshot, projectRoot: string): DiagnosticBundle {
+ const files = clean({
+ "diagnostics.json": buildRunDiagnostics(run, projectRoot),
+ "NEXT-STEPS.txt": "Verify with canary diagnostics verify . Inspect originalError, assertions and locations for each failure. Restore the recorded commit and runtime, supply required environment variables locally, fix the failing check, then rerun it. Hypotheses need confirmation; missing mappings need manual inspection. Hashes detect corruption, not authorship.\n",
+ });
+ if (containsSensitiveValue(files, { maxStringLength: Infinity })) throw new Error("Diagnostic privacy scan failed");
+ return { v: 1, kind: "canary.diagnostic-bundle", files, manifest: Object.entries(files).map(([path, value]) => ({ path, bytes: Buffer.byteLength(fileContent(value)), sha256: sha256(fileContent(value)) })) };
+}
+export function verifyDiagnosticBundle(input: unknown): boolean {
+ try {
+ const bundle = input as DiagnosticBundle;
+ if (bundle.v !== 1 || bundle.kind !== "canary.diagnostic-bundle" || Object.keys(bundle).sort().join() !== "files,kind,manifest,v") return false;
+ if (Object.keys(bundle.files).sort().join() !== "NEXT-STEPS.txt,diagnostics.json" || bundle.files["diagnostics.json"].kind !== "canary.diagnostics" || bundle.files["diagnostics.json"].v !== 1 || typeof bundle.files["NEXT-STEPS.txt"] !== "string") return false;
+ if (!Array.isArray(bundle.manifest) || bundle.manifest.length !== 2 || new Set(bundle.manifest.map((item) => item.path)).size !== 2 || containsSensitiveValue(bundle.files, { maxStringLength: Infinity })) return false;
+ return bundle.manifest.every((entry) => Object.hasOwn(bundle.files, entry.path) && entry.bytes === Buffer.byteLength(fileContent(bundle.files[entry.path as keyof typeof bundle.files])) && entry.sha256 === sha256(fileContent(bundle.files[entry.path as keyof typeof bundle.files])));
+ } catch { return false; }
+}
diff --git a/packages/trace/src/index.ts b/packages/trace/src/index.ts
index f7a6d10..1f1fe0a 100644
--- a/packages/trace/src/index.ts
+++ b/packages/trace/src/index.ts
@@ -19,3 +19,5 @@ export * from "./artifacts.js";
export * from "./recovery.js";
export * from "./retention.js";
export { redactText, containsSensitiveText, containsSensitiveValue, collectSecretValues, SECRET_KEY } from "./privacy.js";
+
+export * from "./diagnostics.js";
diff --git a/packages/trace/tests/diagnostics.test.ts b/packages/trace/tests/diagnostics.test.ts
new file mode 100644
index 0000000..9e024e4
--- /dev/null
+++ b/packages/trace/tests/diagnostics.test.ts
@@ -0,0 +1,55 @@
+import { describe, expect, it } from "vitest";
+import type { ProjectCheckResult, RunSnapshot } from "@canary/core";
+import { buildRunDiagnostics, createDiagnosticBundle, verifyDiagnosticBundle } from "../src/diagnostics.js";
+
+const check = (id: string, overrides: Partial = {}): ProjectCheckResult => ({ id, type: "command", version: 1, required: true, status: "failed", evidence: "verified", exitCode: 1, category: "assertion", retryable: false, durationMs: 1, cwd: "/project", envAllowlist: ["API_KEY"], command: "node", args: ["test.js"], ...overrides });
+const fixture = (): RunSnapshot => ({ runId: "run_1", status: "failed", startedAt: "2026-10-01T00:00:00Z", totalCases: 3, completedCases: 3, passedCases: 0, results: [], events: [], checks: [check("test", { stderr: "FAIL adds numbers\nAssertionError: wrong result\n at test (/project/src/test.ts:12:3)", outputTruncated: true }), check("similar", { stderr: "AssertionError: wrong result" }), check("dependent", { status: "blocked", category: "dependency", dependsOn: ["test"] })] });
+
+describe("unified failure diagnostics", () => {
+ it("retains errors, test names and stack positions without merging similar failures", () => {
+ const run = fixture(), original = JSON.stringify(run);
+ const report = buildRunDiagnostics(run, "/project");
+ expect(report.failures).toHaveLength(3);
+ expect(report.failures[0]).toMatchObject({ testNames: ["adds numbers"], locations: [{ path: "src/test.ts", line: 12, column: 3, mapping: "path-line" }], originalError: { truncated: true }, rootCause: { status: "unknown" } });
+ expect(report.failures[1]).toMatchObject({ sourceMapping: "missing", relatedFailures: [], rootCause: { status: "unknown" } });
+ expect(report.failures[2]).toMatchObject({ relatedFailures: [{ id: "run_1:check:test", evidence: "declared-dependency" }], rootCause: { status: "hypothesis" } });
+ expect(JSON.stringify(run)).toBe(original);
+ });
+ it("links case assertions and execution identity and rejects external stack positions", () => {
+ const run = fixture();
+ run.results.push({ runId: run.runId, caseId: "case-1", executionId: "exec-1", passed: false, assertions: [{ id: "equals", passed: false, message: 'File "src/test.py", line 9', details: { expected: 2, actual: 3 } }], coverage: {} as never });
+ run.events.push({ type: "execution.failed", executionId: "exec-1", error: "Error: failed\n at task (/project/src/task.ts:5:2)" });
+ run.events.push({ type: "execution.failed", executionId: "unrelated", error: "Error: unrelated" });
+ run.checks![0]!.stderr = "at x (/outside/file.ts:3:1)\nat x (../../outside.ts:2:1)";
+ const report = buildRunDiagnostics(run, "/project");
+ expect(report.failures[0]!.locations).toEqual([]);
+ expect(report.failures[3]!.originalError.stderr).not.toContain("unrelated");
+ expect(report.failures[3]).toMatchObject({ caseId: "case-1", executionId: "exec-1", assertions: [{ id: "equals", details: { expected: 2, actual: 3 } }], locations: [{ path: "src/task.ts", line: 5, column: 2 }, { path: "src/test.py", line: 9 }] });
+ });
+ it("redacts credentials everywhere before hashing and verifies a JSON roundtrip", () => {
+ const run = fixture();
+ run.checks![0]!.args = ["--token=ghp_12345678901234567890", "--password", "opaque-value-123"];
+ run.checks![0]!.stderr += "\npassword=superprivate\nhttps://user:pass@example.com";
+ const bundle = createDiagnosticBundle(run, "/project");
+ const serialized = JSON.stringify(bundle);
+ expect(serialized).not.toContain("superprivate");
+ expect(serialized).not.toContain("opaque-value-123");
+ expect(serialized).not.toContain("ghp_12345678901234567890");
+ expect(serialized).not.toContain("user:pass");
+ expect(verifyDiagnosticBundle(JSON.parse(serialized))).toBe(true);
+ bundle.files["NEXT-STEPS.txt"] += "tampered";
+ expect(verifyDiagnosticBundle(bundle)).toBe(false);
+ });
+ it("rejects altered diagnostics, duplicate entries, extra files and malformed bundles", () => {
+ for (const mutate of [
+ (b: ReturnType) => { b.files["diagnostics.json"].runId = "other"; },
+ (b: ReturnType) => { b.manifest[1] = b.manifest[0]!; },
+ (b: ReturnType) => { Object.assign(b.files, { "extra.txt": "extra" }); },
+ ]) {
+ const bundle = createDiagnosticBundle(fixture(), "/project"); mutate(bundle);
+ expect(verifyDiagnosticBundle(bundle)).toBe(false);
+ }
+ expect(verifyDiagnosticBundle(null)).toBe(false);
+ expect(verifyDiagnosticBundle({})).toBe(false);
+ });
+});
From e2c60a21c3bc2d1ffe2656891245b431c8403004 Mon Sep 17 00:00:00 2001
From: nat-xu <19707027371@163.com>
Date: Thu, 1 Oct 2026 20:31:00 +0800
Subject: [PATCH 2/3] fix(fixtures): pin R6 hashes to canonical LF source bytes
---
integrations/fixtures/README.md | 2 +-
integrations/fixtures/deterministic-agent/fixture.json | 8 ++++----
integrations/fixtures/http-agent/fixture.json | 8 ++++----
integrations/fixtures/mcp-agent/fixture.json | 8 ++++----
integrations/fixtures/tool-calling-agent/fixture.json | 10 +++++-----
5 files changed, 18 insertions(+), 18 deletions(-)
diff --git a/integrations/fixtures/README.md b/integrations/fixtures/README.md
index 494c7d4..6222a56 100644
--- a/integrations/fixtures/README.md
+++ b/integrations/fixtures/README.md
@@ -8,7 +8,7 @@
- `mcp-agent`:独立 stdio JSON-RPC 进程,验证现有 MCP run 适配器。不是完整 MCP 协议一致性认证。
- `pi-agent`:固定真实上游来源和版本;当前模型执行 blocked,不用本地 echo 冒充 Pi 成功。
-每个目录的 `fixture.json` 定义来源、版本、启动方式、凭证、offline 行为、期望输出及副作用。四个本地 fixture 固定为 `r6-v1`,其文件 SHA-256 在执行前验证;修改 fixture 必须同时更新版本说明和哈希,并重新验收。
+每个目录的 `fixture.json` 定义来源、版本、启动方式、凭证、offline 行为、期望输出及副作用。四个本地 fixture 固定为 `r6-v1-lf`,哈希对应 Git 中使用 LF 换行的原始文件字节,与仓库 `.gitattributes` 一致。该版本仅修正此前按 CRLF 生成的哈希,不改变 fixture 行为。文件 SHA-256 在执行前严格验证;修改 fixture 必须同时更新版本说明和哈希,并重新验收。
脚本不扫描凭证、不连接模型、不更改被测源码。只有 HTTP fixture 使用临时回环监听。临时项目及 artifact 保留供审计,位置见验收 JSON 的 `artifactBase`。Windows 权限测试仅对本次创建的 `.canary` 目录加入当前用户写入拒绝 ACL,并在 finally 中恢复;POSIX 仅对自建目录临时 chmod,root 运行会将该项标 blocked。
diff --git a/integrations/fixtures/deterministic-agent/fixture.json b/integrations/fixtures/deterministic-agent/fixture.json
index feda5f1..1fe2930 100644
--- a/integrations/fixtures/deterministic-agent/fixture.json
+++ b/integrations/fixtures/deterministic-agent/fixture.json
@@ -2,7 +2,7 @@
"version": 1,
"id": "deterministic-agent",
"source": "Canary repository / integrations/fixtures/deterministic-agent",
- "revision": "r6-v1",
+ "revision": "r6-v1-lf",
"credentials": [],
"offline": true,
"start": "node scripts/verify-r6.mjs",
@@ -15,8 +15,8 @@
],
"limitations": "Deterministic local contract fixture; no model provider.",
"files": {
- "agent.mjs": "49e893efe3f2b12cf52b67e61278b7d9e23647741d37c58dd84c386915ab3c61",
- "canary.config.ts": "2fda6d0b10e33a4ed6641da522d807baee3e462588171561de5e92e58bb43fb0",
- "cases.ts": "a8f280efaedead6a745297a2a6e1ac7fb661f0c67cfefa8c2fa78eb473fbfb5b"
+ "agent.mjs": "5e9f75cf6c99927386cc4461ce9cb0944e858e2683e39afa8b3b947709b7f900",
+ "canary.config.ts": "ce6c63a8ed1e2b71b0ff5d4941847182e0f50283d0a2a6421ef09c515084530e",
+ "cases.ts": "8e4b34c22ecfa94c362275696bcc27e366a55de640c0a722ade1e2d2fd1546bd"
}
}
diff --git a/integrations/fixtures/http-agent/fixture.json b/integrations/fixtures/http-agent/fixture.json
index 0ad0f20..3f15a52 100644
--- a/integrations/fixtures/http-agent/fixture.json
+++ b/integrations/fixtures/http-agent/fixture.json
@@ -2,7 +2,7 @@
"version": 1,
"id": "http-agent",
"source": "Canary repository / integrations/fixtures/http-agent",
- "revision": "r6-v1",
+ "revision": "r6-v1-lf",
"credentials": [],
"offline": true,
"start": "node scripts/verify-r6.mjs",
@@ -16,8 +16,8 @@
],
"limitations": "Deterministic local contract fixture; no model provider.",
"files": {
- "canary.config.ts": "75190168f9bd5d9d35c517099edfd65b00227377d744cd575a422e45d2ad7c2a",
- "cases.ts": "8d16a4c98bafb26bf73b459a12a8a872b788bcdc7c4206adaa8d9f7f8d918ef4",
- "server.mjs": "b1576d5e7c88d26d944d5cad7ab731a26abb8a1c38a081ad8cb0d066254a8df3"
+ "canary.config.ts": "aac5adb0169ff2b2a6bef596e8f39cce6e79d4198a41260a025ab8ea672a85da",
+ "cases.ts": "66be98f81cfce34a966c5f913cdfba209071f3d1778dfdcf45ddf56e52447f1a",
+ "server.mjs": "7f85f382126a675c6325bda4275e909f8e4405365f4f57edb1596df050be6b58"
}
}
diff --git a/integrations/fixtures/mcp-agent/fixture.json b/integrations/fixtures/mcp-agent/fixture.json
index f4148f7..5c18391 100644
--- a/integrations/fixtures/mcp-agent/fixture.json
+++ b/integrations/fixtures/mcp-agent/fixture.json
@@ -2,7 +2,7 @@
"version": 1,
"id": "mcp-agent",
"source": "Canary repository / integrations/fixtures/mcp-agent",
- "revision": "r6-v1",
+ "revision": "r6-v1-lf",
"credentials": [],
"offline": true,
"start": "node scripts/verify-r6.mjs",
@@ -16,8 +16,8 @@
],
"limitations": "MCP fixture covers existing newline JSON-RPC run adapter, not full MCP protocol conformance.",
"files": {
- "canary.config.ts": "d8ff573cc4e789ec7c81e311437ef5e00c3acba0fdd32345335380f212aab74c",
- "cases.ts": "82b2db43299adeb4259a6f7f57a12df93da49579e66770a13d57f961d146c36b",
- "server.mjs": "2c9788818702cfaec43b5aa3f3cc7a198c6973ea11f18a909fa65b28b834e990"
+ "canary.config.ts": "40acb0240f2bcff9581c217bc4eca0a55bb5fb3811cc9bac899492fe7cb46f43",
+ "cases.ts": "741189119830ba91bf5a48fafe54803898431223e3a4775b2cb8387056bc3f0c",
+ "server.mjs": "70bb99e0b253cc2e8bf723893863abd22c13dbe761fbffc858dc408bb31e8388"
}
}
diff --git a/integrations/fixtures/tool-calling-agent/fixture.json b/integrations/fixtures/tool-calling-agent/fixture.json
index 7b6dd56..e80b9ff 100644
--- a/integrations/fixtures/tool-calling-agent/fixture.json
+++ b/integrations/fixtures/tool-calling-agent/fixture.json
@@ -2,7 +2,7 @@
"version": 1,
"id": "tool-calling-agent",
"source": "Canary repository / integrations/fixtures/tool-calling-agent",
- "revision": "r6-v1",
+ "revision": "r6-v1-lf",
"credentials": [],
"offline": true,
"start": "node scripts/verify-r6.mjs",
@@ -15,9 +15,9 @@
],
"limitations": "Deterministic local contract fixture; no model provider.",
"files": {
- "agent.mjs": "fb05e5c499cc65f498e9cf8970c5cad4ba158f7ba83da8aefb2d0d43b0b6a49d",
- "canary.config.ts": "8eafb337fb8bcf81ea0e518e5d3bcfae350ff0a568884ba52ebc4bb1143be1aa",
- "cases.ts": "04f15c10ebef8e521858d6a2abd9485153fbbfa124ae9721c3247ae1cb86f583",
- "tools.mjs": "fbeaf6a0c5443d16035a153d86960b9b29bbd1093727a089fd5fe22ceaab676b"
+ "agent.mjs": "6ca40411deab96e1f473674b9e7d41fad4a747a88e3cdbef48006417cf263719",
+ "canary.config.ts": "6acb406874edfde47adcfcf176033d1fa0294a0beb781ab9601a4ac245e8ea7d",
+ "cases.ts": "bb113aa57ca4889aa4078305605d10c8eeddd2b825e40759b4f1cc18713d0b2d",
+ "tools.mjs": "6d0c8a7519d51a84660f12de53cab45d6b6f8befa13499334b8d3d6faea7f51f"
}
}
From 1d2f0c8fe129942b92a059fcd2ca5e675867fdd4 Mon Sep 17 00:00:00 2001
From: nat-xu <19707027371@163.com>
Date: Thu, 1 Oct 2026 20:44:09 +0800
Subject: [PATCH 3/3] ci: consolidate platform status on the Node 24 baseline
---
.github/workflows/r6.yml | 34 ++++++++++++++++++----
README.cn.md | 4 +--
README.md | 4 +--
apps/site/index.html | 2 +-
apps/site/package.json | 2 +-
apps/site/site.js | 4 +--
docs/guides/architecture-ci.md | 2 +-
docs/guides/getting-started.md | 2 +-
docs/guides/l03-distribution-and-export.md | 2 +-
docs/guides/r6-platform-fixtures.md | 2 +-
docs/guides/support-matrix.md | 2 +-
package.json | 2 +-
scripts/install-global.mjs | 2 +-
scripts/verify-distribution.mjs | 2 +-
14 files changed, 45 insertions(+), 21 deletions(-)
diff --git a/.github/workflows/r6.yml b/.github/workflows/r6.yml
index 6113bfc..ff5b141 100644
--- a/.github/workflows/r6.yml
+++ b/.github/workflows/r6.yml
@@ -1,4 +1,4 @@
-name: R6 platform evidence
+name: Platform checks
on:
workflow_dispatch:
pull_request:
@@ -17,11 +17,17 @@ permissions:
contents: read
jobs:
verify:
+ name: ${{ matrix.platform }} / Node 24
strategy:
fail-fast: false
matrix:
- os: [windows-latest, ubuntu-latest, macos-latest]
- node: [24, 22]
+ include:
+ - os: windows-latest
+ platform: Windows
+ - os: ubuntu-latest
+ platform: Ubuntu
+ - os: macos-latest
+ platform: macOS
runs-on: ${{ matrix.os }}
timeout-minutes: 20
steps:
@@ -31,7 +37,7 @@ jobs:
version: 10.15.0
- uses: actions/setup-node@v4
with:
- node-version: ${{ matrix.node }}
+ node-version: 24
cache: pnpm
- run: pnpm install --frozen-lockfile
- run: pnpm build
@@ -43,9 +49,27 @@ jobs:
- uses: actions/upload-artifact@v4
if: always()
with:
- name: r6-${{ matrix.os }}-node${{ matrix.node }}
+ name: r6-${{ matrix.os }}-node24
path: |
r6-evidence/
.canary/logs/verification/r6/
include-hidden-files: true
if-no-files-found: error
+
+ summary:
+ name: All platforms / Node 24
+ needs: verify
+ if: ${{ always() }}
+ runs-on: ubuntu-latest
+ steps:
+ - name: Aggregate platform results
+ env:
+ PLATFORM_RESULT: ${{ needs.verify.result }}
+ shell: bash
+ run: |
+ echo "## Node 24 platform compatibility" >> "$GITHUB_STEP_SUMMARY"
+ echo "Windows, Ubuntu and macOS: **$PLATFORM_RESULT**" >> "$GITHUB_STEP_SUMMARY"
+ if [[ "$PLATFORM_RESULT" != "success" ]]; then
+ echo "Platform validation did not complete successfully. Inspect the individual platform jobs." >&2
+ exit 1
+ fi
diff --git a/README.cn.md b/README.cn.md
index 028ba8e..60d8bba 100644
--- a/README.cn.md
+++ b/README.cn.md
@@ -9,7 +9,7 @@
-
+
@@ -31,7 +31,7 @@
## 快速开始
-准备 **Node.js 22+(推荐 24)和 Git**,按系统执行一条安装命令:
+准备 **Node.js 24和 Git**,按系统执行一条安装命令:
**Windows / PowerShell**
diff --git a/README.md b/README.md
index b67c218..080d7cb 100644
--- a/README.md
+++ b/README.md
@@ -9,7 +9,7 @@
-
+
@@ -31,7 +31,7 @@ See the [support scope](docs/guides/support-matrix.md).
## Quick start
-Use **Node.js 22+ (24 recommended) and Git**, then run one installation command for your platform:
+Use **Node.js 24 and Git**, then run one installation command for your platform:
**Windows / PowerShell**
diff --git a/apps/site/index.html b/apps/site/index.html
index 7660060..cfa53bd 100644
--- a/apps/site/index.html
+++ b/apps/site/index.html
@@ -379,7 +379,7 @@ Your next signal