From f5d2b3786ab23b326960312f11d091d19b9f0536 Mon Sep 17 00:00:00 2001 From: AIGameCreator App Date: Wed, 15 Jul 2026 04:03:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=AE=8C=E5=96=84=E5=8D=95Agent=E6=8C=81?= =?UTF-8?q?=E4=B9=85=E8=AE=A1=E5=88=92=E4=B8=8E=E6=81=A2=E5=A4=8D=E9=AA=8C?= =?UTF-8?q?=E6=94=B6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 新增结构化计划更新、单调修订、完成门禁和失败进度保留 升级上下文与最终回复恢复协议并绑定规划仓库指纹 收紧模型正文公共审计和CLI本地存储路径输出 补齐开发界面计划展示、Supervisor紧凑摘要与乱序竞态合并 强化真实Provider验收器并记录TLS外部门禁未通过 补充Rust前端回归及长期技术文档 --- .../scripts/agent-runtime-real-e2e.mjs | 1948 +++++++++++++++- .../src-tauri/src/agent.rs | 1300 +++++++++-- .../src-tauri/src/cli.rs | 70 +- .../src-tauri/src/main.rs | 4 + .../src-tauri/src/swarm_cli.rs | 208 +- .../src-tauri/src/tests.rs | 2006 ++++++++++++++++- apps/ai-game-creator-shell/src/App.tsx | 327 ++- .../tests/appSurface.test.ts | 509 ++++- .../shared-memory/decision-log.md | 14 + docs/project-memory/shared-memory/pitfalls.md | 38 + ...案】AI游戏创作Agent Runtime V1.1-2026-07-12.md | 73 + ...¹案】AI游戏创作智能体App实施计划-2026-06-24.md | 27 +- 12 files changed, 6208 insertions(+), 316 deletions(-) diff --git a/apps/ai-game-creator-shell/scripts/agent-runtime-real-e2e.mjs b/apps/ai-game-creator-shell/scripts/agent-runtime-real-e2e.mjs index 71298a4d5..21a8e6fcb 100644 --- a/apps/ai-game-creator-shell/scripts/agent-runtime-real-e2e.mjs +++ b/apps/ai-game-creator-shell/scripts/agent-runtime-real-e2e.mjs @@ -5,6 +5,7 @@ import fs from 'node:fs/promises'; import os from 'node:os'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; + import { buildProcessSessionFixtureSource } from './process-session-real-e2e-fixture.mjs'; const appRoot = path.resolve(fileURLToPath(new URL('..', import.meta.url))); @@ -29,6 +30,8 @@ const commandPassedMarker = 'real-e2e-command=passed'; const commandRootErrorMarker = `real-e2e-root-${randomUUID().replaceAll('-', '')}`; const commandRootErrorLine = 170; const commandDiagnosticLineCount = 240; +const steerInstruction = + '继续完成原任务,并依据恢复后的真实进展重审、重排尚未完成的安排,确保最终交付完整。'; const processFixtureScriptPath = 'fixtures/process-session-service.mjs'; const processReadyPrefix = 'GENARRATIVE_PROCESS_READY'; const processEchoPrefix = 'GENARRATIVE_PROCESS_ECHO'; @@ -115,7 +118,11 @@ const state = { commandOutputMarkerSeenInContext: false, commandOutputContextPages: new Set(), commandMarkerReportLeakCount: 0, + steerInstructionReportLeakCount: 0, + projectPathTranscriptLeakCount: 0, + projectPathReportLeakCount: 0, transcriptScanner: null, + projectPathTranscriptScanner: null, projectRoot: null, sentinelToken: null, cliBinary: null, @@ -124,6 +131,9 @@ const state = { identityStable: false, initialRunId: null, initialSessionId: null, + initialTask: null, + planRecovery: null, + steer: null, confirmedActionIds: new Set(), cleanupPerformed: false, process: { @@ -194,10 +204,16 @@ try { } } state.transcriptLeakCount = state.transcriptScanner?.count ?? 0; + state.projectPathTranscriptLeakCount = + state.projectPathTranscriptScanner?.count ?? 0; if (state.transcriptLeakCount + state.projectLeakCount > 0) { state.status = 'FAIL'; recordError('loaded-key-leak-detected'); } + if (state.projectPathTranscriptLeakCount > 0) { + state.status = 'FAIL'; + recordError('disposable-project-path-transcript-leak-detected'); + } if (state.projectRoot && !state.options?.keepProject) { try { state.cleanupPerformed = await removeDisposableProject(); @@ -240,6 +256,29 @@ try { summary = buildSummary(); report = JSON.stringify(summary, null, 2); } + state.steerInstructionReportLeakCount = countExactSecrets( + Buffer.from(report), + [steerInstruction], + ); + state.evidence.steerInstructionReportLeakCount = + state.steerInstructionReportLeakCount; + if (state.steerInstructionReportLeakCount > 0) { + state.status = 'FAIL'; + recordError('steer-instruction-report-leak-detected'); + summary = buildSummary(); + report = JSON.stringify(summary, null, 2); + } + state.projectPathReportLeakCount = countExactSecrets( + Buffer.from(report), + disposableProjectPathVariants(), + ); + state.evidence.projectPathReportLeakCount = state.projectPathReportLeakCount; + if (state.projectPathReportLeakCount > 0) { + state.status = 'FAIL'; + recordError('disposable-project-path-report-leak-detected'); + summary = buildSummary(); + report = JSON.stringify(summary, null, 2); + } state.reportLeakCount = countExactSecrets(Buffer.from(report), state.secrets); if (state.reportLeakCount > 0) { state.status = 'FAIL'; @@ -247,6 +286,33 @@ try { summary = buildSummary(); report = JSON.stringify(summary, null, 2); } + const remainingProjectPathReportLeakCount = countExactSecrets( + Buffer.from(report), + disposableProjectPathVariants(), + ); + if (remainingProjectPathReportLeakCount > 0) { + state.status = 'FAIL'; + recordError('disposable-project-path-report-redaction-required'); + const safeSummary = { + status: state.status, + suite: state.suite, + blocked: state.blocked, + cleanup: { + performed: state.cleanupPerformed, + kept: Boolean(state.options?.keepProject), + }, + evidence: { + projectPathReportLeakCount: remainingProjectPathReportLeakCount, + }, + errorCount: state.errors.length, + errorHashes: state.errors.map((error) => ({ + code: error.code, + detailHash: error.detailHash, + })), + }; + safeSummary.summaryHash = hashValue(JSON.stringify(safeSummary)); + report = JSON.stringify(safeSummary, null, 2); + } process.stdout.write(`${report}\n`); process.exitCode = state.status === 'PASS' ? 0 : state.status === 'BLOCKED' ? 2 : 1; @@ -258,6 +324,10 @@ async function runRealE2e() { const task = buildTaskPrompt(state.suite); assertUnscriptedTaskPrompt(task); + state.initialTask = { + chars: [...task].length, + sha256: hashValue(task), + }; await runCli( [ '--agent-enqueue', @@ -273,17 +343,34 @@ async function runRealE2e() { const beforeKill = await waitForCanonicalRuntime(); state.initialRunId = beforeKill.runId; state.initialSessionId = beforeKill.sessionId; + const preKillPlan = await waitForPartiallyCompletedStructuredPlan(); + state.planRecovery = { + preKillRevision: preKillPlan.revision, + preKillCompletedStepHashes: preKillPlan.completedStepHashes, + preKillIncompleteStepCount: preKillPlan.incompleteStepCount, + preKillTerminalStepHash: preKillPlan.terminalStepHash, + recoveredRevision: 0, + recoveredCompletedStepHashes: [], + recoveredTerminalStepHash: null, + }; await killRunnerOnce(); await runCli(['--agent-resume', state.projectRoot], { timeoutMs: 120_000 }); state.resumed = true; - const afterResume = await waitForRuntimeIdentity(); + const recoveredPlan = await waitForRecoveredStructuredPlan(preKillPlan); + const afterResume = recoveredPlan.runtime; assert( afterResume.runId === state.initialRunId && afterResume.sessionId === state.initialSessionId, 'run-session-changed-after-resume', ); state.identityStable = true; + state.planRecovery.recoveredRevision = recoveredPlan.revision; + state.planRecovery.recoveredCompletedStepHashes = + recoveredPlan.completedStepHashes; + state.planRecovery.recoveredTerminalStepHash = recoveredPlan.terminalStepHash; + + await injectSameRunSteer(); await driveRuntimeToQuiescence(); state.evidence = await validateLandedEvidence(); @@ -484,6 +571,9 @@ function supportedBrowserCandidates(platform, environment) { async function seedDisposableProject() { const prefix = path.join(os.tmpdir(), 'genarrative-agent-runtime-real-e2e-'); state.projectRoot = await fs.mkdtemp(prefix); + state.projectPathTranscriptScanner = new StreamingSecretScanner( + disposableProjectPathVariants(), + ); state.sentinelToken = randomUUID(); await fs.writeFile( path.join(state.projectRoot, sentinelFileName), @@ -583,6 +673,9 @@ async function seedProcessSessionDisposableProject() { 'genarrative-agent-runtime-process-real-e2e-', ); state.projectRoot = await fs.mkdtemp(prefix); + state.projectPathTranscriptScanner = new StreamingSecretScanner( + disposableProjectPathVariants(), + ); state.sentinelToken = randomUUID(); await fs.writeFile( path.join(state.projectRoot, sentinelFileName), @@ -725,31 +818,34 @@ function seededGameHtml() { } function buildTaskPrompt(suite) { - const canvasStep = + const editorAssetOutcome = suite === 'full' - ? `在最终验证前必须调用一次 canvas.asset_generate,prompt 为“${editorAssetPrompt}”,并使用真实 editor API 结果。` - : '本套件禁止调用 canvas.asset_generate。'; - return `修复当前项目唯一的真实验证失败,并完成可执行、可审阅、可恢复的交付;所有具体路径、脚本、旧值、新值、产物内容和 actionId 都必须从仓库与真实工具 observation 自行发现,禁止猜测截断输出。 + ? `- 通过项目已配置的外部编辑器生成服务回流一项真实的透明背景琥珀街机代币素材,生成意图为“${editorAssetPrompt}”,并保留可核验的资源身份与本地产物。` + : '- 本次交付不产生外部编辑器生成素材。'; + return `修复当前 disposable 项目唯一的真实验收失败,交付一份可执行、可审阅、可恢复的完整结果。具体路径、脚本、变更内容和持久身份均应从仓库事实与真实运行证据中自行发现,不得猜测被截断的诊断。 -验收要求(不规定固定调用顺序): -- 读取适用仓库规范、项目索引、脚本和游戏源码;修改前后各形成一次安全 Git 工作树审阅。 -- 修改前真实运行项目验证。短 observation 不足以定位根因,必须按源 actionId 分页读取完整命令输出,确认唯一根因后才能修改;不得通过直接阅读验证脚本绕过这份失败反馈。 -- 只用一次原子多文件 patchset 完成根因输出要求的既有文件修改和新证据文件创建;不得再用其他写工具补修。用自动 checkpointId 审阅包含正文的两项 diff。 -- 使用三个隔离 reviewer 并行只读检查已有的三个独立 evidence 区域;前两个使用程序模板,第三个使用质量模板,writeScopes 互不重叠,并形成唯一 all-join。 -- 修复后再次真实运行验证并通过,再按项目清单中的原始脚本完成 project.verify。 -- 对桌面和移动视口完成真实 preview.validate,并把两张实际截图交给 image.inspect 做一次视觉检查。 -- 收束前回查当前 run 中唯一成功 patchset 的持久动作记录;不得猜测 actionId,不得重复有副作用动作。 -- 全部验证与审阅证据完成后,只创建一个受控本地 Git 提交,精确包含本次原子变更的两个安全路径;提交前使用最新可提交快照,提交后再次安全审阅并确认 HEAD 已前移、本次两项变更不再处于工作树变更。不得提交诱饵、Runtime 控制面或其它预存文件。 -- ${canvasStep} +交付结果: +- 保留现有可见内容、非空动画画布与仓库安全边界;根因要求的既有文件修复和新证据文件由一次原子多文件变更完整落地,不留下半完成文件或额外写入。 +- 项目清单声明的原始验收全部通过;桌面与移动视口均有真实渲染、非空画布、无致命控制台错误的截图证据,并完成两张实际截图的视觉审查。 +- 三个既有独立 evidence 区域各由隔离 reviewer 并行只读审查,其中两名采用程序实现视角、一名采用质量视角;边界互不重叠,结果只汇总一次。 +- 本次原子变更形成一个受控本地 Git 提交,提交只包含两项安全产物;提交身份、父提交、树内容和提交后工作树状态均可核验,诱饵、Runtime 控制面及预存文件不进入提交。 +${editorAssetOutcome} -保留现有可见内容、非空动画画布和仓库安全边界;不得读取或转述敏感诱饵、配置密钥或绝对路径。只有上述证据全部落盘后才能最终回复。`; +可验证行为: +- 同一 run 维护有界计划,真实里程碑变化会更新进度,所有必要步骤完成前不形成最终回复。 +- 证据同时覆盖原始失败、未截断的唯一根因、原子变更正文差异、修复后的项目验收、项目清单验收、双视口渲染、隔离审查汇总和提交后安全审阅。 +- 原始失败的定位来自真实运行反馈;不得直接读取验收实现来规避失败,也不得重复任何会产生副作用的动作。 +- 唯一原子变更的持久动作身份可从当前 run 的历史事实精确回查,所有恢复与重试保持幂等。 + +安全约束:不得读取、提交或转述敏感诱饵、配置密钥、私有 Runtime 正文或项目绝对路径;不得用额外写入补修原子变更;全部交付证据落盘且相互一致时才可收束。`; } function buildProcessSessionTaskPrompt() { - return `完成当前 disposable 项目的真实交互服务验收:先从项目清单确认唯一服务,整个验收最多启动一个进程;启动后只沿同一会话等待 readiness、按服务给出的一次性 challenge 完成一次交互并确认精确回显,challenge 必须原样作为单独一行输入,并且只有观察到精确回显后才可终止服务,不得为探测、试错、重试或停止另起进程。只有服务形成可信终态后才能简短报告完成,不得修改项目文件,也不要在最终回复中复述 challenge、回显或其他私有进程输出。`; + return `交付当前 disposable 项目的真实交互服务验收结果。有效证据必须同时证明:项目清单中的唯一交互服务只有一个实例;同一会话观察到 readiness;服务给出的一次性 challenge 被原样作为单独一行交互一次并得到精确回显;服务形成可信终态;项目文件保持不变。不得为探测、试错、重试或停止另起实例,不得在最终回复中复述 challenge、回显或其他私有进程输出。`; } function assertProcessSessionTaskPrompt(task) { + assertResultOrientedDisposableTask(task, 'process-session-task'); for (const forbidden of [ 'command.start', 'command.poll', @@ -769,6 +865,7 @@ function assertProcessSessionTaskPrompt(task) { } function assertUnscriptedTaskPrompt(task) { + assertResultOrientedDisposableTask(task, 'real-e2e-task'); for (const forbidden of [ 'AGENTS.md', 'package.json', @@ -786,6 +883,65 @@ function assertUnscriptedTaskPrompt(task) { } } +function assertResultOrientedDisposableTask(task, codePrefix) { + for (const forbidden of [ + 'project.index', + 'project.search', + 'project.diff', + 'project.patchset', + 'project.verify', + 'project.git_commit', + 'file.read', + 'file.write', + 'file.patch', + 'file.delete', + 'git.inspect', + 'command.exec', + 'command.output_read', + 'agent.spawn_isolated', + 'agent.action_history', + 'agent.run_status', + 'preview.validate', + 'image.inspect', + 'canvas.asset_generate', + 'actionId', + 'checkpointId', + 'writeScopes', + ]) { + assert(!task.includes(forbidden), `${codePrefix}-tool-recipe-leak`); + } + for (const pattern of [ + /首先/u, + /随后/u, + /依次/u, + /固定(?:调用)?顺序/u, + /第[一二三四五六七八九十0-9]+步/u, + /先[^。;\n]{0,120}(?:再|然后)/u, + ]) { + assert(!pattern.test(task), `${codePrefix}-ordered-recipe-leak`); + } +} + +function assertUnscriptedSteerInstruction(instruction) { + for (const forbidden of [ + '/', + '\\', + '--', + '.agent', + 'AGENTS.md', + 'package.json', + 'command.', + 'project.', + 'file.', + 'agent.', + 'preview.', + 'git.', + 'canvas.', + ]) { + assert(!instruction.includes(forbidden), 'steer-instruction-recipe-leak'); + } +} + async function prepareCliBinary() { const cargo = process.platform === 'win32' ? 'cargo.exe' : 'cargo'; await runProcess( @@ -824,16 +980,20 @@ async function runCli(args, options = {}) { return runProcess( state.cliBinary, [...args, '--config-dir', state.options.configDir], - { cwd: appRoot, timeoutMs: options.timeoutMs ?? 60_000 }, + { + cwd: appRoot, + timeoutMs: options.timeoutMs ?? 60_000, + stdin: options.stdin, + }, ); } -async function runProcess(program, args, { cwd, timeoutMs }) { +async function runProcess(program, args, { cwd, timeoutMs, stdin }) { return new Promise((resolve, reject) => { const child = spawn(program, args, { cwd, env: { ...process.env, NO_COLOR: '1', RUST_BACKTRACE: '0' }, - stdio: ['ignore', 'pipe', 'pipe'], + stdio: [stdin === undefined ? 'ignore' : 'pipe', 'pipe', 'pipe'], }); let stdout = Buffer.alloc(0); let stderr = Buffer.alloc(0); @@ -844,10 +1004,12 @@ async function runProcess(program, args, { cwd, timeoutMs }) { }, timeoutMs); child.stdout.on('data', (chunk) => { state.transcriptScanner?.scan('stdout', chunk); + state.projectPathTranscriptScanner?.scan('stdout', chunk); stdout = appendBounded(stdout, chunk, commandOutputLimit); }); child.stderr.on('data', (chunk) => { state.transcriptScanner?.scan('stderr', chunk); + state.projectPathTranscriptScanner?.scan('stderr', chunk); stderr = appendBounded(stderr, chunk, commandOutputLimit); }); child.on('error', (error) => { @@ -870,6 +1032,7 @@ async function runProcess(program, args, { cwd, timeoutMs }) { resolve(result); } }); + if (stdin !== undefined) child.stdin.end(stdin); }); } @@ -884,6 +1047,39 @@ async function readRuntime(agentId) { return runtime; } +function mainRuntimeStatePath() { + return path.join( + state.projectRoot, + '.agent/runtime/agents', + `${mainAgentId}.json`, + ); +} + +function mainContextBundlePath() { + return path.join( + state.projectRoot, + '.agent/runtime/context-bundles', + mainAgentId, + `${state.initialRunId}.json`, + ); +} + +function agentConversationPath(agentId, sessionId) { + return sessionId === `agent-session-${agentId}` + ? path.join( + state.projectRoot, + '.agent/conversations/agents', + `${agentId}.jsonl`, + ) + : path.join( + state.projectRoot, + '.agent/conversations/agents', + agentId, + 'sessions', + `${sessionId}.jsonl`, + ); +} + async function readRunnerStatus() { const result = await runCli(['--runner-status'], { timeoutMs: 60_000 }); return parseAssignedJson(result.stdout, [ @@ -969,6 +1165,581 @@ async function waitForRuntimeIdentity() { throw codedError('runtime-not-readable-after-resume'); } +async function waitForPartiallyCompletedStructuredPlan() { + const deadline = Date.now() + runTimeoutMs; + let lastError = null; + while (Date.now() < deadline) { + const taskSnapshot = await readTaskSnapshot(); + const initial = taskSnapshot.latest.find( + (task) => + task.agentId === mainAgentId && task.runId === state.initialRunId, + ); + if (initial && isFailedTask(initial)) { + throw codedError('main-runtime-failed-before-plan-checkpoint'); + } + if ( + initial?.status === 'completed' || + initial?.phase === 'completed' || + initial?.status === 'cancelled' + ) { + throw codedError('main-runtime-terminal-before-plan-checkpoint'); + } + + try { + const plan = await readVerifiedDurablePlanSnapshot('pre-kill-plan'); + if (plan.completedStepHashes.length > 0 && plan.incompleteStepCount > 0) { + const messages = await readOptionalJsonl( + agentConversationPath(mainAgentId, state.initialSessionId), + ); + assert( + messages.filter( + (message) => + message.role === 'assistant' && message.agentId === mainAgentId, + ).length === 0, + 'assistant-persisted-before-plan-checkpoint', + ); + return plan; + } + } catch (error) { + lastError = error; + } + + await captureCommandOutputContextEvidence(); + await confirmPendingActions(); + await sleep(pollIntervalMs); + } + throw codedError('partial-structured-plan-before-kill-timeout', lastError); +} + +async function waitForRecoveredStructuredPlan(preKillPlan) { + const deadline = Date.now() + 120_000; + let lastError = null; + while (Date.now() < deadline) { + let recovered; + try { + recovered = await readVerifiedDurablePlanSnapshot('recovered-plan'); + } catch (error) { + lastError = error; + await sleep(pollIntervalMs); + continue; + } + assert( + recovered.revision >= preKillPlan.revision, + 'recovered-plan-revision-regressed', + ); + const recoveredCompleted = new Set(recovered.completedStepHashes); + assert( + preKillPlan.completedStepHashes.every((stepHash) => + recoveredCompleted.has(stepHash), + ), + 'recovered-plan-completed-step-lost', + ); + if (!isSteerableRuntime(recovered.runtime)) { + if (isTerminalRuntime(recovered.runtime)) { + throw codedError('recovered-runtime-terminal-before-steer'); + } + lastError = codedError('recovered-runtime-not-yet-steerable'); + await sleep(pollIntervalMs); + continue; + } + const taskSnapshot = await readTaskSnapshot(); + const initial = taskSnapshot.latest.find( + (task) => + task.agentId === mainAgentId && task.runId === state.initialRunId, + ); + if (initial?.sessionId !== state.initialSessionId || !isLiveTask(initial)) { + lastError = codedError('recovered-plan-task-not-yet-live'); + await sleep(pollIntervalMs); + continue; + } + return recovered; + } + throw codedError('recovered-structured-plan-timeout', lastError); +} + +async function readVerifiedDurablePlanSnapshot(codePrefix) { + const [runtime, contextBundle, records] = await Promise.all([ + readJson(mainRuntimeStatePath()), + readJson(mainContextBundlePath()), + readOptionalJsonl(path.join(state.projectRoot, '.agent/agent.db')), + ]); + const snapshot = inspectStructuredPlanSnapshot(runtime, codePrefix); + assertStructuredPlanContextSnapshot( + runtime, + contextBundle, + `${codePrefix}-context`, + ); + assertStructuredPlanAuditSnapshot(runtime, records, `${codePrefix}-audit`); + return { ...snapshot, runtime }; +} + +function inspectStructuredPlanSnapshot(runtime, codePrefix) { + assert( + runtime.agentId === mainAgentId && + runtime.runId === state.initialRunId && + runtime.sessionId === state.initialSessionId, + `${codePrefix}-identity-invalid`, + ); + assert( + Number.isSafeInteger(runtime.planRevision) && runtime.planRevision > 0, + `${codePrefix}-revision-invalid`, + ); + assert( + isNonEmptyString(runtime.planExplanation) && + Array.isArray(runtime.planSteps) && + runtime.planSteps.length >= 3 && + runtime.planSteps.length <= 8, + `${codePrefix}-snapshot-invalid`, + ); + const stepHashes = []; + const completedStepHashes = []; + const incompleteSteps = []; + let activePlanStepIndex = null; + for (const [index, step] of runtime.planSteps.entries()) { + assert( + step.index === index && + isNonEmptyString(step.title) && + ['pending', 'in_progress', 'completed'].includes(step.status), + `${codePrefix}-step-invalid`, + ); + const stepHash = hashValue(step.title); + assert(!stepHashes.includes(stepHash), `${codePrefix}-step-duplicate`); + stepHashes.push(stepHash); + if (step.status === 'completed') completedStepHashes.push(stepHash); + else incompleteSteps.push({ index, status: step.status, stepHash }); + if (step.status === 'in_progress') { + assert( + activePlanStepIndex === null, + `${codePrefix}-multiple-in-progress`, + ); + activePlanStepIndex = index; + } + } + assert( + runtime.activePlanStepIndex === activePlanStepIndex, + `${codePrefix}-active-step-invalid`, + ); + completedStepHashes.sort(); + return { + revision: runtime.planRevision, + completedStepHashes, + incompleteStepCount: runtime.planSteps.length - completedStepHashes.length, + incompleteSteps, + terminalStepHash: hashValue(JSON.stringify(completedStepHashes)), + }; +} + +function assertStructuredPlanContextSnapshot(runtime, contextBundle, code) { + assert( + contextBundle.schemaVersion === 'game-creator-runtime-context-bundle.v3' && + contextBundle.agentId === runtime.agentId && + contextBundle.taskId === runtime.taskId && + contextBundle.sessionId === runtime.sessionId && + contextBundle.runId === runtime.runId && + contextBundle.planRevision === runtime.planRevision && + contextBundle.planExplanation === runtime.planExplanation && + JSON.stringify(contextBundle.planSteps) === + JSON.stringify(runtime.planSteps) && + contextBundle.activePlanStepIndex === runtime.activePlanStepIndex, + `${code}-snapshot-mismatch`, + ); +} + +function assertStructuredPlanAuditSnapshot(runtime, records, code) { + const audits = records.filter( + (record) => + record.recordType === 'agent.runtime.plan_update' && + record.agentId === runtime.agentId && + record.taskId === runtime.taskId && + record.sessionId === runtime.sessionId && + record.runId === runtime.runId && + record.planRevision === runtime.planRevision, + ); + assert(audits.length === 1, `${code}-record-count-invalid`); + const audit = audits[0]; + assert( + audit.explanationSha256 === hashValue(runtime.planExplanation) && + audit.explanationChars === [...runtime.planExplanation].length && + Array.isArray(audit.steps) && + audit.steps.length === runtime.planSteps.length && + audit.steps.every( + (step, index) => + step.stepSha256 === hashValue(runtime.planSteps[index].title) && + step.status === runtime.planSteps[index].status, + ), + `${code}-snapshot-mismatch`, + ); +} + +async function injectSameRunSteer() { + assertUnscriptedSteerInstruction(steerInstruction); + const steerTargetPlan = await waitForProviderPlanningSteerTarget(); + const runtime = steerTargetPlan.runtime; + assert( + runtime.runId === state.initialRunId && + runtime.sessionId === state.initialSessionId && + isProviderPlanningWait(runtime) && + steerTargetPlan.incompleteStepCount > 0, + 'steer-target-runtime-invalid', + ); + const before = steerTargetPlan.targetRunIds; + assert( + JSON.stringify(before) === JSON.stringify([state.initialRunId]), + 'steer-target-task-missing', + ); + + const steerId = `real-e2e-steer-${randomUUID().slice(0, 12)}`; + const result = await runCli( + [ + '--agent-steer', + state.projectRoot, + mainAgentId, + state.initialSessionId, + state.initialRunId, + steerId, + '--stdin', + ], + { timeoutMs: 120_000, stdin: steerInstruction }, + ); + assert( + !result.stdout.includes(steerInstruction) && + !result.stderr.includes(steerInstruction), + 'steer-instruction-cli-leak', + ); + const steer = parseAssignedJson(result.stdout, ['steerJson']); + assert( + steer?.steerId === steerId && + steer.sequence === 1 && + ['queued', 'applied'].includes(steer.status) && + steer.providerInterrupted === true, + 'steer-cli-result-invalid', + ); + + const afterSnapshot = await readTaskSnapshot(); + const after = targetMainTaskRunIds( + afterSnapshot, + steerTargetPlan.taskIdentity, + ); + assert( + JSON.stringify(after) === JSON.stringify(before), + 'steer-created-new-task-run', + ); + const afterRuntime = await readRuntime(mainAgentId); + assert( + afterRuntime.runId === state.initialRunId && + afterRuntime.sessionId === state.initialSessionId && + runtime.taskQueue && + afterRuntime.taskQueue && + afterRuntime.taskQueue.total === runtime.taskQueue.total && + afterRuntime.taskQueue.latestRunId === runtime.taskQueue.latestRunId, + 'steer-changed-runtime-or-task-queue', + ); + state.steer = { + steerId, + steerIdHash: hashValue(steerId), + instructionSha256: hashValue(steerInstruction), + sequence: steer.sequence, + providerInterrupted: steer.providerInterrupted, + planRevisionAtAcceptance: runtime.planRevision, + completedStepHashesAtAcceptance: steerTargetPlan.completedStepHashes, + incompleteStepsAtAcceptance: steerTargetPlan.incompleteSteps, + incompletePlanSignatureAtAcceptance: hashValue( + JSON.stringify(steerTargetPlan.incompleteSteps), + ), + providerWaitStatus: runtime.status, + providerWaitPhase: runtime.phase, + providerWaitUpdatedAt: runtime.updatedAt, + agentDbSequenceBefore: steerTargetPlan.agentDbSequenceBefore, + taskIdentity: steerTargetPlan.taskIdentity, + initialMessageId: steerTargetPlan.initialMessageId, + activeActionsAtAcceptance: steerTargetPlan.activeActions, + durableActionsAtAcceptance: steerTargetPlan.durableActions, + sideEffectReceiptsAtAcceptance: steerTargetPlan.sideEffectReceipts, + projectRevisionAtAcceptance: steerTargetPlan.projectRevision, + projectSideEffectFingerprintAtAcceptance: + steerTargetPlan.projectSideEffectFingerprint, + taskRunIdsBefore: before, + taskRunIdsAfter: after, + taskRunSetHash: hashValue(JSON.stringify(before)), + }; +} + +async function waitForProviderPlanningSteerTarget() { + const deadline = Date.now() + runTimeoutMs; + let lastError = null; + while (Date.now() < deadline) { + const taskSnapshot = await readTaskSnapshot(); + const initial = taskSnapshot.latest.find( + (task) => + task.agentId === mainAgentId && task.runId === state.initialRunId, + ); + if (initial && isFailedTask(initial)) { + throw codedError('main-runtime-failed-before-provider-wait'); + } + if (!initial || !isLiveTask(initial)) { + lastError = codedError('main-runtime-not-live-before-provider-wait'); + await sleep(pollIntervalMs); + continue; + } + try { + const plan = await readVerifiedDurablePlanSnapshot( + 'steer-provider-wait-plan', + ); + if ( + plan.completedStepHashes.length > 0 && + plan.incompleteStepCount > 0 && + isProviderPlanningWait(plan.runtime) + ) { + const snapshot = await capturePreSteerSnapshot( + plan, + initial, + taskSnapshot, + ); + const current = await readRuntime(mainAgentId); + if ( + isProviderPlanningWait(current) && + current.runId === plan.runtime.runId && + current.sessionId === plan.runtime.sessionId && + current.planRevision === plan.runtime.planRevision && + current.updatedAt === plan.runtime.updatedAt + ) { + return { ...plan, ...snapshot, runtime: current }; + } + } + } catch (error) { + lastError = error; + } + await captureCommandOutputContextEvidence(); + await confirmPendingActions(); + await sleep(pollIntervalMs); + } + throw codedError('provider-planning-wait-before-steer-timeout', lastError); +} + +async function capturePreSteerSnapshot(plan, initial, taskSnapshot) { + const taskIdentity = { + agentId: mainAgentId, + taskId: initial.taskId, + sessionId: state.initialSessionId, + source: initial.source, + }; + const targetRunIds = targetMainTaskRunIds(taskSnapshot, taskIdentity); + assert( + JSON.stringify(targetRunIds) === JSON.stringify([state.initialRunId]), + 'pre-steer-target-run-set-invalid', + ); + + const conversationPath = agentConversationPath( + mainAgentId, + state.initialSessionId, + ); + const messages = await readOptionalJsonl(conversationPath); + const initialMessageId = backgroundTaskMessageId( + mainAgentId, + state.initialSessionId, + state.initialRunId, + initial.source, + ); + assert( + messages.length === 1 && + messages[0].role === 'user' && + messages[0].agentId === mainAgentId && + messages[0].messageId === initialMessageId && + hashValue(messages[0].content) === state.initialTask?.sha256 && + [...messages[0].content].length === state.initialTask?.chars, + 'pre-steer-initial-conversation-invalid', + ); + + const durableActions = await readTargetDurableActions(); + const activeActions = durableActions.filter((action) => + ['pending-confirmation', 'approved', 'executing'].includes(action.status), + ); + assert( + activeActions.length === 0 && + plan.runtime.pendingToolAction == null && + plan.runtime.pendingAction == null, + 'provider-planning-wait-has-active-action', + ); + + const records = await readOptionalJsonl( + path.join(state.projectRoot, '.agent/agent.db'), + ); + const sideEffectReceipts = records + .map((record, index) => ({ record, sequence: index + 1 })) + .filter( + ({ record }) => + record.recordType === 'agent.runtime.action_receipt' && + record.agentId === mainAgentId && + record.runId === state.initialRunId && + !idempotentObservationTools.has(record.tool), + ) + .map(({ record, sequence }) => ({ + actionFingerprint: record.actionFingerprint, + actionId: record.actionId, + identityHash: hashValue(actionReceiptIdentity(record)), + sequence, + status: record.status, + tool: record.tool, + updatedAt: record.updatedAt, + })); + const revision = await readJson( + path.join(state.projectRoot, '.agent/runtime/project-revision.json'), + ); + assert( + Number.isSafeInteger(revision.revision) && revision.revision >= 0, + 'pre-steer-project-revision-invalid', + ); + const [head, worktree] = await Promise.all([ + runProcess('git', ['rev-parse', '--verify', 'HEAD'], { + cwd: state.projectRoot, + timeoutMs: 30_000, + }), + runProcess( + 'git', + ['status', '--porcelain=v2', '-z', '--untracked-files=all'], + { + cwd: state.projectRoot, + timeoutMs: 30_000, + }, + ), + ]); + return { + activeActions, + agentDbSequenceBefore: records.length, + durableActions, + initialMessageId, + projectRevision: revision.revision, + projectSideEffectFingerprint: hashValue( + `${head.stdout.trim()}\n${worktree.stdout}`, + ), + sideEffectReceipts, + targetRunIds, + taskIdentity, + }; +} + +async function readTargetDurableActions() { + const root = path.join(state.projectRoot, '.agent/runtime/pending-actions'); + const actions = []; + for (const file of (await listFiles(root)).filter((entry) => + entry.endsWith('.json'), + )) { + const value = await readJson(file).catch(() => null); + if ( + !value || + value.agentId !== mainAgentId || + value.runId !== state.initialRunId + ) { + continue; + } + assert( + isNonEmptyString(value.actionId) && + isNonEmptyString(value.actionFingerprint) && + isNonEmptyString(value.action?.tool) && + isNonEmptyString(value.status) && + Number.isSafeInteger(value.plannedSteerCursor), + 'durable-action-identity-invalid', + ); + actions.push({ + actionFingerprint: value.actionFingerprint, + actionId: value.actionId, + plannedSteerCursor: value.plannedSteerCursor, + status: value.status, + tool: value.action.tool, + updatedAt: value.updatedAt, + }); + } + return actions.sort((left, right) => + left.actionId.localeCompare(right.actionId), + ); +} + +function isProviderPlanningWait(runtime) { + return runtime.status === 'running' && runtime.phase === 'planning'; +} + +function isSteerableRuntime(runtime) { + return ( + ['running', 'waiting-for-confirmation'].includes(runtime.status) && + !['cancelling', 'finalizing', 'needs-reconciliation'].includes( + runtime.phase, + ) + ); +} + +function targetMainTaskRunIds(taskSnapshot, identity) { + return [ + ...new Set( + taskSnapshot.all + .filter( + (task) => + task.agentId === identity.agentId && + task.taskId === identity.taskId && + task.sessionId === identity.sessionId && + task.source === identity.source, + ) + .map((task) => task.runId) + .filter(isNonEmptyString), + ), + ].sort(); +} + +function validateFinalMainRunSet(taskSnapshot, initial, spawnRecord) { + const targetRunIds = targetMainTaskRunIds( + taskSnapshot, + state.steer.taskIdentity, + ); + assert( + JSON.stringify(targetRunIds) === + JSON.stringify(state.steer.taskRunIdsBefore) && + JSON.stringify(targetRunIds) === JSON.stringify([state.initialRunId]), + 'final-target-main-run-set-changed', + ); + const legalLineageRunIds = new Set([spawnRecord.joinRunId]); + for (const child of spawnRecord.children ?? []) { + if (isNonEmptyString(child.runId)) legalLineageRunIds.add(child.runId); + } + const nonTargetMainRecords = taskSnapshot.all.filter( + (task) => + task.agentId === mainAgentId && + !( + task.taskId === initial.taskId && + task.sessionId === state.initialSessionId && + task.source === initial.source && + task.runId === state.initialRunId + ), + ); + const legalLineageRecords = nonTargetMainRecords.filter( + (task) => + task.source === 'agent-isolated-join' && + task.runId === spawnRecord.joinRunId && + task.sessionId === state.initialSessionId && + task.parentRunId === state.initialRunId && + task.delegationId === spawnRecord.delegationGroupId, + ); + const unexpectedRecords = nonTargetMainRecords.filter( + (task) => !legalLineageRecords.includes(task), + ); + assert( + unexpectedRecords.length === 0 && + legalLineageRecords.every((task) => legalLineageRunIds.has(task.runId)), + 'final-unexpected-main-run-detected', + ); + const legalLineageRuns = new Set( + legalLineageRecords.map((task) => task.runId), + ); + assert( + legalLineageRuns.size <= 1, + 'final-legal-main-lineage-run-count-invalid', + ); + return { + legalLineageRunCount: legalLineageRuns.size, + targetRunCount: targetRunIds.length, + targetRunSetHash: hashValue(JSON.stringify(targetRunIds)), + unexpectedRunCount: 0, + }; +} + async function driveRuntimeToQuiescence() { const deadline = Date.now() + runTimeoutMs; let quietPolls = 0; @@ -1197,6 +1968,7 @@ async function confirmPendingActions(allowedTools = null) { 'project.patchset', 'project.git_commit', 'command.exec', + ...(state.suite === 'llm-runtime' ? ['command.run_limited'] : []), 'project.verify', 'preview.start', 'preview.validate', @@ -1391,6 +2163,8 @@ async function validateProcessSessionEvidence() { processActivityLeakCount: publicLeaks.activity, processOutputLeakCount: publicLeaks.output, processRuntimeStateLeakCount: publicLeaks.runtimeState, + projectPathPublicLeakCount: publicLeaks.projectPathLeakCount, + projectPathPublicSurfaceCount: publicLeaks.projectPathSurfaceCount, processReportLeakCount: state.process.reportLeakCount, secretLeakCount: state.transcriptLeakCount + state.projectLeakCount, lureLeakCount: state.lureLeakCount, @@ -1591,6 +2365,8 @@ async function validateProcessRunnerKillEvidence() { processActivityLeakCount: publicLeaks.activity, processOutputLeakCount: publicLeaks.output, processRuntimeStateLeakCount: publicLeaks.runtimeState, + projectPathPublicLeakCount: publicLeaks.projectPathLeakCount, + projectPathPublicSurfaceCount: publicLeaks.projectPathSurfaceCount, processReportLeakCount: state.process.reportLeakCount, secretLeakCount: state.transcriptLeakCount + state.projectLeakCount, lureLeakCount: state.lureLeakCount, @@ -2039,6 +2815,29 @@ function validateProcessPublicLeakBoundary(persistence) { ); assert(counts[surface] === 0, `process-private-output-${surface}-leak`); } + const projectPathCounts = validateProjectRootPublicLeakBoundary( + surfaces, + 'process-public', + ); + return { + ...counts, + projectPathLeakCount: sumObjectValues(projectPathCounts), + projectPathSurfaceCount: Object.keys(projectPathCounts).length, + }; +} + +function validateProjectRootPublicLeakBoundary(surfaces, codePrefix) { + const variants = disposableProjectPathVariants(); + assert(variants.length >= 2, `${codePrefix}-path-variants-missing`); + const counts = {}; + for (const [surface, records] of Object.entries(surfaces)) { + const values = Array.isArray(records) ? records : [records]; + counts[surface] = countExactSecrets( + Buffer.from(values.map((record) => JSON.stringify(record)).join('\n')), + variants, + ); + assert(counts[surface] === 0, `${codePrefix}-${surface}-project-path-leak`); + } return counts; } @@ -2445,21 +3244,34 @@ async function validateLandedEvidence() { const agentDb = await readJsonl( path.join(state.projectRoot, '.agent/agent.db'), ); + const conversationFiles = await listFiles( + path.join(state.projectRoot, '.agent/conversations'), + ); + const conversationEntries = []; + for (const file of conversationFiles.filter((entry) => + entry.endsWith('.jsonl'), + )) { + for (const message of await readJsonl(file)) { + conversationEntries.push({ file: path.resolve(file), message }); + } + } + const conversations = conversationEntries.map(({ message }) => message); + const [activity, output] = await Promise.all([ + readOptionalJsonl(path.join(state.projectRoot, '.agent/activity.jsonl')), + readOptionalJsonl(path.join(state.projectRoot, '.agent/output.jsonl')), + ]); assert(taskSnapshot.all.length > 0, 'task-evidence-missing'); assert(events.length > 0, 'event-evidence-missing'); assert(agentDb.length > 0, 'agent-db-evidence-missing'); assertNoPersistedImagePayload('task', taskSnapshot.all); assertNoPersistedImagePayload('event', events); assertNoPersistedImagePayload('agent-db', agentDb); - const contextBundlePath = path.join( - state.projectRoot, - '.agent/runtime/context-bundles', - mainAgentId, - `${state.initialRunId}.json`, - ); + const contextBundlePath = mainContextBundlePath(); const contextBundle = await readJson(contextBundlePath); + const runtimeStatePath = mainRuntimeStatePath(); + const runtimeState = await readJson(runtimeStatePath); assert( - contextBundle.schemaVersion === 'game-creator-runtime-context-bundle.v2' && + contextBundle.schemaVersion === 'game-creator-runtime-context-bundle.v3' && contextBundle.agentId === mainAgentId && contextBundle.runId === state.initialRunId && typeof contextBundle.repositoryContextFingerprint === 'string' && @@ -2471,6 +3283,11 @@ async function validateLandedEvidence() { ); const toolPlanProtocolCount = validateMainRunToolPlanProtocols(agentDb); + const structuredPlanEvidence = validateStructuredPlanEvidence( + agentDb, + runtimeState, + contextBundle, + ); const confirmedActionLifecycleCount = validateConfirmedActionLifecycles(agentDb); const replayEvidence = validateToolActionReplays(agentDb); @@ -2801,11 +3618,28 @@ async function validateLandedEvidence() { agentDb: countExactSecrets(Buffer.from(JSON.stringify(agentDb)), [ commandRootErrorMarker, ]), + conversation: countExactSecrets( + Buffer.from(JSON.stringify(conversations)), + [commandRootErrorMarker], + ), + activity: countExactSecrets(Buffer.from(JSON.stringify(activity)), [ + commandRootErrorMarker, + ]), + output: countExactSecrets(Buffer.from(JSON.stringify(output)), [ + commandRootErrorMarker, + ]), + runtimeState: countExactSecrets(Buffer.from(JSON.stringify(runtimeState)), [ + commandRootErrorMarker, + ]), }; assert( markerLeakCounts.task === 0 && markerLeakCounts.event === 0 && markerLeakCounts.agentDb === 0 && + markerLeakCounts.conversation === 0 && + markerLeakCounts.activity === 0 && + markerLeakCounts.output === 0 && + markerLeakCounts.runtimeState === 0 && state.commandOutputMarkerSeenInContext && state.commandOutputContextPages.size >= 1, 'command-output-transcript-persistence-boundary-invalid', @@ -3075,6 +3909,17 @@ async function validateLandedEvidence() { initial?.status === 'completed' || initial?.phase === 'completed', 'main-run-not-completed', ); + const steerEvidence = await validateSameRunSteerEvidence({ + agentDb, + activity, + contextBundle, + conversationEntries, + events, + initial, + output, + runtimeState, + taskSnapshot, + }); const actionReceiptEvidence = validateMainRunActionReceipts( agentDb, initial, @@ -3572,6 +4417,11 @@ async function validateLandedEvidence() { joinCompletionRecordIndex < actionHistoryExecution.startIndex, 'action-history-not-after-isolated-join', ); + const finalRunSetEvidence = validateFinalMainRunSet( + taskSnapshot, + initial, + spawnRecord, + ); const completedProjections = agentDb.filter( (record) => @@ -3637,37 +4487,93 @@ async function validateLandedEvidence() { 'completed-run-finalization-journal-present', ); - const conversationFiles = await listFiles( - path.join(state.projectRoot, '.agent/conversations'), - ); - const conversations = []; - for (const file of conversationFiles.filter((entry) => - entry.endsWith('.jsonl'), - )) { - conversations.push(...(await readJsonl(file))); - } const expectedFinalMessageId = finalMessageId( completedProjection.agentId, completedProjection.sessionId, completedProjection.runId, ); - const finalAssistant = conversations.filter( - (message) => - message.role === 'assistant' && - message.agentId === completedProjection.agentId && - message.messageId === expectedFinalMessageId, + const expectedConversationPath = path.resolve( + agentConversationPath( + completedProjection.agentId, + completedProjection.sessionId, + ), ); - assert(finalAssistant.length === 1, 'final-assistant-count-invalid'); - const finalAssistantAudits = agentDb.filter( - (record) => - record.recordType === 'conversation.message' && - record.role === 'assistant' && - record.agentId === completedProjection.agentId && - record.sessionId === completedProjection.sessionId && - record.messageId === expectedFinalMessageId, + const expectedInitialMessageId = backgroundTaskMessageId( + completedProjection.agentId, + completedProjection.sessionId, + completedProjection.runId, + completedProjection.source, + ); + const expectedSteerMessageId = steerEvidence.messageId; + const targetSessionMessages = conversationEntries.filter( + ({ file }) => file === expectedConversationPath, ); assert( - finalAssistantAudits.length === 1, + expectedInitialMessageId === state.steer.initialMessageId && + isNonEmptyString(expectedSteerMessageId) && + targetSessionMessages.length === 3 && + JSON.stringify( + targetSessionMessages.map(({ message }) => message.role), + ) === JSON.stringify(['user', 'user', 'assistant']) && + JSON.stringify( + targetSessionMessages.map(({ message }) => message.messageId), + ) === + JSON.stringify([ + expectedInitialMessageId, + expectedSteerMessageId, + expectedFinalMessageId, + ]) && + targetSessionMessages.every( + ({ message }) => + message.agentId === completedProjection.agentId && + Number.isSafeInteger(message.updatedAt), + ) && + targetSessionMessages[0].message.updatedAt <= + targetSessionMessages[1].message.updatedAt && + targetSessionMessages[1].message.updatedAt <= + targetSessionMessages[2].message.updatedAt && + hashValue(targetSessionMessages[0].message.content) === + state.initialTask?.sha256 && + [...targetSessionMessages[0].message.content].length === + state.initialTask?.chars && + hashValue(targetSessionMessages[1].message.content) === + state.steer.instructionSha256 && + isNonEmptyString(targetSessionMessages[2].message.content), + 'target-session-message-contract-invalid', + ); + const finalAssistant = [targetSessionMessages[2].message]; + const targetConversationAudits = agentDb.filter( + (record) => + record.recordType === 'conversation.message' && + record.agentId === completedProjection.agentId && + record.sessionId === completedProjection.sessionId, + ); + assert( + targetConversationAudits.length === 3 && + JSON.stringify(targetConversationAudits.map((record) => record.role)) === + JSON.stringify(['user', 'user', 'assistant']) && + JSON.stringify( + targetConversationAudits.map((record) => record.messageId), + ) === + JSON.stringify([ + expectedInitialMessageId, + expectedSteerMessageId, + expectedFinalMessageId, + ]) && + targetConversationAudits.every( + (record) => + isNonEmptyString(record.path) && + path.resolve(resolveProjectRelative(record.path)) === + expectedConversationPath, + ), + 'target-session-conversation-audit-contract-invalid', + ); + const finalAssistantAudits = targetConversationAudits.filter( + (record) => record.role === 'assistant', + ); + assert( + finalAssistantAudits.length === 1 && + finalAssistantAudits[0].messageId === expectedFinalMessageId, 'final-assistant-audit-count-invalid', ); assert( @@ -3678,9 +4584,10 @@ async function validateLandedEvidence() { finalAssistantAudits[0].path, ); assert( - conversationFiles.some( - (file) => path.resolve(file) === auditedConversationPath, - ), + auditedConversationPath === expectedConversationPath && + conversationFiles.some( + (file) => path.resolve(file) === auditedConversationPath, + ), 'final-assistant-audit-path-invalid', ); assert( @@ -3712,6 +4619,24 @@ async function validateLandedEvidence() { assert(duplicateActionCount === 0, 'duplicate-action-detected'); assert(duplicateMessageCount === 0, 'duplicate-message-detected'); assert(duplicateReceiptCount === 0, 'duplicate-receipt-detected'); + const projectPathPublicLeakCounts = validateProjectRootPublicLeakBoundary( + { + task: taskSnapshot.all, + event: events, + agentDb, + receipt: agentDb.filter( + (record) => record.recordType === 'agent.runtime.action_receipt', + ), + conversation: conversations, + activity, + output, + runtimeState: [runtimeState], + }, + 'real-e2e-public', + ); + const projectPathPublicLeakCount = sumObjectValues( + projectPathPublicLeakCounts, + ); const [html, createdFile, gameEntries] = await Promise.all([ fs.readFile(path.join(state.projectRoot, 'game/index.html'), 'utf8'), fs.readFile(path.join(state.projectRoot, patchsetCreatedPath), 'utf8'), @@ -3773,6 +4698,72 @@ async function validateLandedEvidence() { agentDbRecordCount: agentDb.length, successfulToolExecutionCount: successfulToolExecutions.length, toolPlanProtocolCount, + structuredPlanUpdateCount: structuredPlanEvidence.updateCount, + structuredPlanRevision: structuredPlanEvidence.planRevision, + structuredPlanCompletedStepCount: structuredPlanEvidence.completedStepCount, + structuredPlanRegressionCount: structuredPlanEvidence.regressionCount, + structuredPlanPreKillRevision: structuredPlanEvidence.preKillRevision, + structuredPlanPreKillCompletedStepCount: + structuredPlanEvidence.preKillCompletedStepCount, + structuredPlanPreKillIncompleteStepCount: + structuredPlanEvidence.preKillIncompleteStepCount, + structuredPlanPreKillTerminalStepHash: + structuredPlanEvidence.preKillTerminalStepHash, + structuredPlanRecoveredRevision: structuredPlanEvidence.recoveredRevision, + structuredPlanRecoveredCompletedStepCount: + structuredPlanEvidence.recoveredCompletedStepCount, + structuredPlanRecoveredTerminalStepHash: + structuredPlanEvidence.recoveredTerminalStepHash, + structuredPlanTimelineUpdateCount: + structuredPlanEvidence.timelineUpdateCount, + structuredPlanTimelineAnchoredUpdateCount: + structuredPlanEvidence.timelineAnchoredUpdateCount, + structuredPlanCompletionTransitionCount: + structuredPlanEvidence.completionTransitionCount, + structuredPlanCompletionObservationCount: + structuredPlanEvidence.completionObservationCount, + structuredPlanPrematureCompletionCount: + structuredPlanEvidence.prematureCompletionCount, + structuredPlanTimelineHash: structuredPlanEvidence.timelineHash, + steerAcceptedPlanRevision: steerEvidence.planRevisionAtAcceptance, + steerSequence: steerEvidence.sequence, + steerIdHash: steerEvidence.steerIdHash, + steerMessageIdHash: steerEvidence.messageIdHash, + steerInstructionSha256: steerEvidence.instructionSha256, + steerProviderInterrupted: steerEvidence.providerInterrupted, + steerProviderPlanningWaitMatched: steerEvidence.providerPlanningWaitMatched, + steerCompletedStepCountAtAcceptance: + steerEvidence.completedStepCountAtAcceptance, + steerIncompleteStepCountAtAcceptance: + steerEvidence.incompleteStepCountAtAcceptance, + steerFirstPostPlanRevision: steerEvidence.firstPostSteerPlanRevision, + steerFirstPostIncompleteStepCount: + steerEvidence.firstPostSteerIncompleteStepCount, + steerIncompletePlanReordered: steerEvidence.incompletePlanReordered, + steerOldPendingActionCount: steerEvidence.oldPendingActionCount, + steerOldPendingActionSetHash: steerEvidence.oldPendingActionSetHash, + steerOldPendingExecutionCount: steerEvidence.oldPendingExecutionCount, + steerOldPlanMaterializedActionCount: + steerEvidence.oldPlanMaterializedActionCount, + steerAcceptanceWindowExecutionCount: + steerEvidence.acceptanceWindowExecutionCount, + steerSideEffectReceiptCountAtAcceptance: + steerEvidence.sideEffectReceiptCountAtAcceptance, + steerSideEffectSnapshotHash: steerEvidence.sideEffectSnapshotHash, + steerPreSideEffectReplayCount: steerEvidence.preSteerSideEffectReplayCount, + steerLedgerRecordCount: steerEvidence.ledgerRecordCount, + steerAppliedCount: steerEvidence.appliedCount, + steerClosedCount: steerEvidence.closedCount, + steerAuditCount: steerEvidence.auditCount, + steerTaskRunCountBefore: steerEvidence.taskRunCountBefore, + steerTaskRunCountAfter: steerEvidence.taskRunCountAfter, + steerTaskRunSetHash: steerEvidence.taskRunSetHash, + finalTargetMainRunCount: finalRunSetEvidence.targetRunCount, + finalLegalMainLineageRunCount: finalRunSetEvidence.legalLineageRunCount, + finalUnexpectedMainRunCount: finalRunSetEvidence.unexpectedRunCount, + finalTargetMainRunSetHash: finalRunSetEvidence.targetRunSetHash, + steerPublicInstructionLeakCount: steerEvidence.publicInstructionLeakCount, + steerInstructionReportLeakCount: state.steerInstructionReportLeakCount, confirmedActionLifecycleCount, sideEffectActionCount: replayEvidence.sideEffectActionCount, sideEffectReplayCount: replayEvidence.sideEffectReplayCount, @@ -3821,6 +4812,10 @@ async function validateLandedEvidence() { commandOutputMarkerTaskLeakCount: markerLeakCounts.task, commandOutputMarkerEventLeakCount: markerLeakCounts.event, commandOutputMarkerAgentDbLeakCount: markerLeakCounts.agentDb, + commandOutputMarkerConversationLeakCount: markerLeakCounts.conversation, + commandOutputMarkerActivityLeakCount: markerLeakCounts.activity, + commandOutputMarkerOutputLeakCount: markerLeakCounts.output, + commandOutputMarkerRuntimeStateLeakCount: markerLeakCounts.runtimeState, commandOutputMarkerReceiptLeakCount: actionReceiptEvidence.commandOutputMarkerLeakCount, commandOutputReadReceiptCount: @@ -3852,11 +4847,20 @@ async function validateLandedEvidence() { actionReceiptSecretLeakCount: actionReceiptEvidence.secretLeakCount, actionReceiptLureLeakCount: actionReceiptEvidence.lureLeakCount, conversationMessageCount: conversations.length, + targetSessionMessageCount: targetSessionMessages.length, + targetSessionUserMessageCount: targetSessionMessages.filter( + ({ message }) => message.role === 'user', + ).length, + targetSessionAssistantMessageCount: finalAssistant.length, + targetSessionConversationAuditCount: targetConversationAudits.length, finalAssistantCount: finalAssistant.length, duplicateActionCount, duplicateMessageCount, duplicateReceiptCount, confirmedActionCount: state.confirmedActionIds.size, + projectPathPublicLeakCount, + projectPathPublicSurfaceCount: Object.keys(projectPathPublicLeakCounts) + .length, secretLeakCount: state.transcriptLeakCount + state.projectLeakCount, lureLeakCount: state.lureLeakCount, paths: [ @@ -3868,6 +4872,7 @@ async function validateLandedEvidence() { relativeProjectPath(successfulCommandSidecarPath), patchsetCreatedPath, relativeProjectPath(contextBundlePath), + relativeProjectPath(runtimeStatePath), relativeProjectPath(checkpointManifestPath), relativeProjectPath(mainGate.file), relativeBrowserReport, @@ -3960,6 +4965,8 @@ function buildSummary() { ...state.evidence, secretLeakCount, lureLeakCount: state.lureLeakCount, + projectPathTranscriptLeakCount: state.projectPathTranscriptLeakCount, + projectPathReportLeakCount: state.projectPathReportLeakCount, }, cleanup: { performed: state.cleanupPerformed, @@ -3971,9 +4978,6 @@ function buildSummary() { detailHash: error.detailHash, })), }; - if (state.options?.keepProject && state.projectRoot) { - base.projectPath = state.projectRoot; - } base.summaryHash = hashValue(JSON.stringify(base)); return base; } @@ -3985,6 +4989,56 @@ function emptyEvidence() { agentDbRecordCount: 0, successfulToolExecutionCount: 0, toolPlanProtocolCount: 0, + structuredPlanUpdateCount: 0, + structuredPlanRevision: 0, + structuredPlanCompletedStepCount: 0, + structuredPlanRegressionCount: 0, + structuredPlanPreKillRevision: 0, + structuredPlanPreKillCompletedStepCount: 0, + structuredPlanPreKillIncompleteStepCount: 0, + structuredPlanPreKillTerminalStepHash: null, + structuredPlanRecoveredRevision: 0, + structuredPlanRecoveredCompletedStepCount: 0, + structuredPlanRecoveredTerminalStepHash: null, + structuredPlanTimelineUpdateCount: 0, + structuredPlanTimelineAnchoredUpdateCount: 0, + structuredPlanCompletionTransitionCount: 0, + structuredPlanCompletionObservationCount: 0, + structuredPlanPrematureCompletionCount: 0, + structuredPlanTimelineHash: null, + steerAcceptedPlanRevision: 0, + steerSequence: 0, + steerIdHash: null, + steerMessageIdHash: null, + steerInstructionSha256: null, + steerProviderInterrupted: false, + steerProviderPlanningWaitMatched: false, + steerCompletedStepCountAtAcceptance: 0, + steerIncompleteStepCountAtAcceptance: 0, + steerFirstPostPlanRevision: 0, + steerFirstPostIncompleteStepCount: 0, + steerIncompletePlanReordered: false, + steerOldPendingActionCount: 0, + steerOldPendingActionSetHash: null, + steerOldPendingExecutionCount: 0, + steerOldPlanMaterializedActionCount: 0, + steerAcceptanceWindowExecutionCount: 0, + steerSideEffectReceiptCountAtAcceptance: 0, + steerSideEffectSnapshotHash: null, + steerPreSideEffectReplayCount: 0, + steerLedgerRecordCount: 0, + steerAppliedCount: 0, + steerClosedCount: 0, + steerAuditCount: 0, + steerTaskRunCountBefore: 0, + steerTaskRunCountAfter: 0, + steerTaskRunSetHash: null, + finalTargetMainRunCount: 0, + finalLegalMainLineageRunCount: 0, + finalUnexpectedMainRunCount: 0, + finalTargetMainRunSetHash: null, + steerPublicInstructionLeakCount: 0, + steerInstructionReportLeakCount: 0, confirmedActionLifecycleCount: 0, sideEffectActionCount: 0, sideEffectReplayCount: 0, @@ -4027,6 +5081,10 @@ function emptyEvidence() { commandOutputMarkerTaskLeakCount: 0, commandOutputMarkerEventLeakCount: 0, commandOutputMarkerAgentDbLeakCount: 0, + commandOutputMarkerConversationLeakCount: 0, + commandOutputMarkerActivityLeakCount: 0, + commandOutputMarkerOutputLeakCount: 0, + commandOutputMarkerRuntimeStateLeakCount: 0, commandOutputMarkerReceiptLeakCount: 0, commandOutputReadReceiptCount: 0, commandOutputMarkerReportLeakCount: 0, @@ -4054,11 +5112,19 @@ function emptyEvidence() { actionReceiptSecretLeakCount: 0, actionReceiptLureLeakCount: 0, conversationMessageCount: 0, + targetSessionMessageCount: 0, + targetSessionUserMessageCount: 0, + targetSessionAssistantMessageCount: 0, + targetSessionConversationAuditCount: 0, finalAssistantCount: 0, duplicateActionCount: 0, duplicateMessageCount: 0, duplicateReceiptCount: 0, confirmedActionCount: 0, + projectPathPublicLeakCount: 0, + projectPathPublicSurfaceCount: 0, + projectPathTranscriptLeakCount: 0, + projectPathReportLeakCount: 0, secretLeakCount: 0, lureLeakCount: 0, paths: [], @@ -4104,6 +5170,10 @@ function emptyProcessEvidence() { processOutputLeakCount: 0, processRuntimeStateLeakCount: 0, processReportLeakCount: 0, + projectPathPublicLeakCount: 0, + projectPathPublicSurfaceCount: 0, + projectPathTranscriptLeakCount: 0, + projectPathReportLeakCount: 0, secretLeakCount: 0, lureLeakCount: 0, paths: [], @@ -4289,6 +5359,716 @@ function validateMainRunToolPlanProtocols(records) { return protocols.length; } +async function validateSameRunSteerEvidence({ + agentDb, + activity, + contextBundle, + conversationEntries, + events, + initial, + output, + runtimeState, + taskSnapshot, +}) { + assert(state.steer, 'same-run-steer-state-missing'); + assert( + state.steer.providerInterrupted === true && + state.steer.providerWaitStatus === 'running' && + state.steer.providerWaitPhase === 'planning' && + Number.isSafeInteger(state.steer.providerWaitUpdatedAt) && + JSON.stringify(state.steer.taskRunIdsBefore) === + JSON.stringify(state.steer.taskRunIdsAfter) && + state.steer.taskRunIdsBefore.includes(state.initialRunId) && + state.steer.taskRunSetHash === + hashValue(JSON.stringify(state.steer.taskRunIdsBefore)) && + Number.isSafeInteger(state.steer.planRevisionAtAcceptance) && + runtimeState.planRevision > state.steer.planRevisionAtAcceptance, + 'same-run-steer-task-queue-invalid', + ); + const finalMainRunIds = targetMainTaskRunIds( + taskSnapshot, + state.steer.taskIdentity, + ); + assert( + JSON.stringify(finalMainRunIds) === + JSON.stringify(state.steer.taskRunIdsBefore), + 'same-run-steer-final-target-run-set-invalid', + ); + + const ledgerPath = path.join( + state.projectRoot, + '.agent/runtime/steers', + mainAgentId, + `${state.initialRunId}.jsonl`, + ); + const ledger = await readJsonl(ledgerPath); + const entryRecords = ledger.filter( + (record) => record.steerId === state.steer.steerId, + ); + const closedRecords = ledger.filter( + (record) => record.steerId == null && record.status === 'closed', + ); + assert( + ledger.length === 5 && + entryRecords.length === 4 && + JSON.stringify(entryRecords.map((record) => record.status)) === + JSON.stringify([ + 'prepared', + 'conversation-persisted', + 'queued', + 'applied', + ]) && + closedRecords.length === 1 && + ledger.at(-1) === closedRecords[0], + 'same-run-steer-ledger-lifecycle-invalid', + ); + const prepared = entryRecords[0]; + const expectedMessageId = prepared.messageId; + assert( + isNonEmptyString(expectedMessageId) && + entryRecords.every( + (record) => + record.schemaVersion === 'game-creator-runtime-steer.v1' && + isNonEmptyString(record.projectId) && + record.agentId === mainAgentId && + record.taskId === initial.taskId && + record.sessionId === state.initialSessionId && + record.runId === state.initialRunId && + record.source === initial.source && + record.sequence === state.steer.sequence && + record.messageId === expectedMessageId && + record.instructionSha256 === state.steer.instructionSha256 && + record.contentChars === [...steerInstruction].length && + record.contentBytes === Buffer.byteLength(steerInstruction) && + record.acceptedVia === 'cli', + ) && + prepared.instruction === steerInstruction && + entryRecords.slice(1).every((record) => record.instruction == null) && + Number.isSafeInteger(entryRecords.at(-1).appliedAt), + 'same-run-steer-ledger-entry-invalid', + ); + const closed = closedRecords[0]; + assert( + closed.schemaVersion === prepared.schemaVersion && + closed.projectId === prepared.projectId && + closed.agentId === mainAgentId && + closed.taskId === initial.taskId && + closed.sessionId === state.initialSessionId && + closed.runId === state.initialRunId && + closed.source === initial.source && + closed.sequence === state.steer.sequence && + closed.steerId == null && + closed.messageId == null && + closed.instructionSha256 == null && + closed.instruction == null && + closed.contentChars === 0 && + closed.contentBytes === 0 && + closed.acceptedVia === 'runtime-finalization', + 'same-run-steer-ledger-closed-invalid', + ); + + const runtimeRefs = runtimeState.appliedSteerRefs ?? []; + const contextRefs = contextBundle.appliedSteerRefs ?? []; + assert( + runtimeState.agentId === mainAgentId && + runtimeState.sessionId === state.initialSessionId && + runtimeState.runId === state.initialRunId && + runtimeState.appliedSteerCursor === state.steer.sequence && + runtimeState.queuedSteerCount === 0 && + runtimeRefs.length === 1 && + runtimeRefs[0].steerId === state.steer.steerId && + runtimeRefs[0].sequence === state.steer.sequence && + runtimeRefs[0].messageId === expectedMessageId && + runtimeRefs[0].instructionSha256 === state.steer.instructionSha256 && + runtimeRefs[0].contentChars === [...steerInstruction].length && + contextBundle.appliedSteerCursor === runtimeState.appliedSteerCursor && + JSON.stringify(contextRefs) === JSON.stringify(runtimeRefs), + 'same-run-steer-runtime-state-invalid', + ); + + const targetConversationPath = path.resolve( + agentConversationPath(mainAgentId, state.initialSessionId), + ); + const steerMessages = conversationEntries.filter( + ({ file, message }) => + file === targetConversationPath && + message.messageId === expectedMessageId, + ); + assert( + steerMessages.length === 1 && + steerMessages[0].message.role === 'user' && + steerMessages[0].message.agentId === mainAgentId && + steerMessages[0].message.content === steerInstruction && + hashValue(steerMessages[0].message.content) === + state.steer.instructionSha256, + 'same-run-steer-conversation-invalid', + ); + const steerConversationAudits = agentDb.filter( + (record) => + record.recordType === 'conversation.message' && + record.role === 'user' && + record.agentId === mainAgentId && + record.sessionId === state.initialSessionId && + record.messageId === expectedMessageId, + ); + assert( + steerConversationAudits.length === 1 && + isNonEmptyString(steerConversationAudits[0].path) && + path.resolve(resolveProjectRelative(steerConversationAudits[0].path)) === + targetConversationPath, + 'same-run-steer-conversation-audit-invalid', + ); + + const indexedSteerAudits = agentDb + .map((record, index) => ({ index, record })) + .filter( + ({ record }) => + record.recordType === 'agent.runtime.steer' && + record.agentId === mainAgentId && + record.taskId === initial.taskId && + record.sessionId === state.initialSessionId && + record.runId === state.initialRunId && + record.steerId === state.steer.steerId, + ); + const steerAudits = indexedSteerAudits.map(({ record }) => record); + assert( + steerAudits.length === 2 && + JSON.stringify(steerAudits.map((record) => record.status)) === + JSON.stringify(['queued', 'applied']) && + steerAudits.every( + (record) => + record.sequence === state.steer.sequence && + record.messageId === expectedMessageId && + record.instructionSha256 === state.steer.instructionSha256 && + record.contentChars === [...steerInstruction].length, + ), + 'same-run-steer-audit-invalid', + ); + + const appliedAuditIndex = indexedSteerAudits.find( + ({ record }) => record.status === 'applied', + )?.index; + assert( + Number.isSafeInteger(appliedAuditIndex) && + appliedAuditIndex >= state.steer.agentDbSequenceBefore, + 'same-run-steer-applied-sequence-invalid', + ); + const postSteerPlanUpdates = agentDb + .map((record, index) => ({ index, record })) + .filter( + ({ index, record }) => + index > appliedAuditIndex && + record.recordType === 'agent.runtime.plan_update' && + record.agentId === mainAgentId && + record.taskId === initial.taskId && + record.sessionId === state.initialSessionId && + record.runId === state.initialRunId, + ); + assert(postSteerPlanUpdates.length > 0, 'same-run-steer-replan-missing'); + const firstPostSteerPlan = postSteerPlanUpdates[0].record; + const completedBeforeSteer = new Set( + state.steer.completedStepHashesAtAcceptance, + ); + const completedAfterSteer = new Set( + firstPostSteerPlan.steps + .filter((step) => step.status === 'completed') + .map((step) => step.stepSha256), + ); + const incompleteAfterSteer = firstPostSteerPlan.steps + .map((step, index) => ({ + index, + status: step.status, + stepHash: step.stepSha256, + })) + .filter((step) => step.status !== 'completed'); + assert( + firstPostSteerPlan.planRevision > state.steer.planRevisionAtAcceptance && + completedBeforeSteer.size > 0 && + [...completedBeforeSteer].every((stepHash) => + completedAfterSteer.has(stepHash), + ) && + incompleteAfterSteer.length > 0 && + hashValue(JSON.stringify(incompleteAfterSteer)) !== + state.steer.incompletePlanSignatureAtAcceptance, + 'same-run-steer-incomplete-plan-not-reordered', + ); + + const oldActionIds = new Set( + state.steer.activeActionsAtAcceptance.map((action) => action.actionId), + ); + const oldPendingExecutionCount = agentDb + .slice(state.steer.agentDbSequenceBefore) + .filter( + (record) => + oldActionIds.has(record.actionId) && + [ + 'agent.runtime.tool_action.executing', + 'agent.runtime.action_receipt', + ].includes(record.recordType), + ).length; + const acceptanceWindowExecutionCount = agentDb + .slice(state.steer.agentDbSequenceBefore, appliedAuditIndex + 1) + .filter( + (record) => + record.recordType === 'agent.runtime.tool_action.executing' && + record.agentId === mainAgentId && + record.runId === state.initialRunId, + ).length; + const durableActionIdsAtAcceptance = new Set( + state.steer.durableActionsAtAcceptance.map((action) => action.actionId), + ); + const oldPlanMaterializedActions = (await readTargetDurableActions()).filter( + (action) => + action.plannedSteerCursor < state.steer.sequence && + !durableActionIdsAtAcceptance.has(action.actionId), + ); + assert( + oldPendingExecutionCount === 0 && + acceptanceWindowExecutionCount === 0 && + oldPlanMaterializedActions.length === 0, + 'same-run-steer-old-pending-action-executed', + ); + + let preSteerSideEffectReplayCount = 0; + for (const latched of state.steer.sideEffectReceiptsAtAcceptance) { + const matches = agentDb.filter( + (record) => + record.recordType === 'agent.runtime.action_receipt' && + record.agentId === mainAgentId && + record.runId === state.initialRunId && + record.actionId === latched.actionId && + record.actionFingerprint === latched.actionFingerprint && + record.tool === latched.tool && + record.status === latched.status && + hashValue(actionReceiptIdentity(record)) === latched.identityHash, + ); + if (matches.length > 1) preSteerSideEffectReplayCount += matches.length - 1; + assert(matches.length === 1, 'same-run-steer-side-effect-receipt-changed'); + } + const finalRevision = await readJson( + path.join(state.projectRoot, '.agent/runtime/project-revision.json'), + ); + assert( + Number.isSafeInteger(state.steer.projectRevisionAtAcceptance) && + finalRevision.revision >= state.steer.projectRevisionAtAcceptance && + /^[0-9a-f]{64}$/u.test( + state.steer.projectSideEffectFingerprintAtAcceptance ?? '', + ) && + preSteerSideEffectReplayCount === 0, + 'same-run-steer-side-effect-snapshot-invalid', + ); + + const publicInstructionLeakCount = countExactSecrets( + Buffer.from( + JSON.stringify([ + taskSnapshot.all, + events, + agentDb, + activity, + output, + runtimeState, + ]), + ), + [steerInstruction], + ); + assert( + publicInstructionLeakCount === 0, + 'same-run-steer-public-instruction-leak', + ); + return { + planRevisionAtAcceptance: state.steer.planRevisionAtAcceptance, + sequence: state.steer.sequence, + steerIdHash: state.steer.steerIdHash, + messageId: expectedMessageId, + messageIdHash: hashValue(expectedMessageId), + instructionSha256: state.steer.instructionSha256, + providerInterrupted: state.steer.providerInterrupted, + providerPlanningWaitMatched: true, + completedStepCountAtAcceptance: completedBeforeSteer.size, + incompleteStepCountAtAcceptance: + state.steer.incompleteStepsAtAcceptance.length, + firstPostSteerPlanRevision: firstPostSteerPlan.planRevision, + firstPostSteerIncompleteStepCount: incompleteAfterSteer.length, + incompletePlanReordered: true, + oldPendingActionCount: oldActionIds.size, + oldPendingActionSetHash: hashValue( + JSON.stringify( + state.steer.activeActionsAtAcceptance.map((action) => [ + action.actionId, + action.actionFingerprint, + action.tool, + action.status, + action.plannedSteerCursor, + ]), + ), + ), + oldPendingExecutionCount, + oldPlanMaterializedActionCount: oldPlanMaterializedActions.length, + acceptanceWindowExecutionCount, + sideEffectReceiptCountAtAcceptance: + state.steer.sideEffectReceiptsAtAcceptance.length, + sideEffectSnapshotHash: hashValue( + JSON.stringify( + state.steer.sideEffectReceiptsAtAcceptance.map( + (receipt) => receipt.identityHash, + ), + ), + ), + preSteerSideEffectReplayCount, + ledgerRecordCount: ledger.length, + appliedCount: entryRecords.filter((record) => record.status === 'applied') + .length, + closedCount: closedRecords.length, + auditCount: steerAudits.length, + taskRunCountBefore: state.steer.taskRunIdsBefore.length, + taskRunCountAfter: state.steer.taskRunIdsAfter.length, + taskRunSetHash: state.steer.taskRunSetHash, + publicInstructionLeakCount, + }; +} + +function validateStructuredPlanEvidence(records, runtimeState, contextBundle) { + const indexedUpdates = records + .map((record, index) => ({ index, record })) + .filter( + ({ record }) => + record.recordType === 'agent.runtime.plan_update' && + record.agentId === mainAgentId && + record.sessionId === state.initialSessionId && + record.runId === state.initialRunId, + ); + const updates = indexedUpdates.map(({ record }) => record); + assert(updates.length >= 3, 'structured-plan-update-count-invalid'); + const completed = new Set(); + let regressionCount = 0; + let previousPlanSignature = null; + for (const [index, update] of updates.entries()) { + assert( + update.planRevision === index + 1 && + Number.isSafeInteger(update.updatedAt) && + /^[0-9a-f]{64}$/u.test(update.explanationSha256 ?? '') && + Number.isInteger(update.explanationChars) && + update.explanationChars > 0 && + Array.isArray(update.steps) && + update.steps.length >= 3 && + update.steps.length <= 8, + 'structured-plan-update-invalid', + ); + const planSignature = hashValue( + JSON.stringify({ + explanationSha256: update.explanationSha256, + steps: update.steps, + }), + ); + assert( + planSignature !== previousPlanSignature, + 'structured-plan-idempotent-revision-incremented', + ); + previousPlanSignature = planSignature; + const stepNames = new Set(); + let inProgressCount = 0; + const byName = new Map(); + for (const step of update.steps) { + assert( + /^[0-9a-f]{64}$/u.test(step.stepSha256 ?? '') && + ['pending', 'in_progress', 'completed'].includes(step.status) && + !stepNames.has(step.stepSha256), + 'structured-plan-step-invalid', + ); + stepNames.add(step.stepSha256); + byName.set(step.stepSha256, step.status); + if (step.status === 'in_progress') inProgressCount += 1; + } + assert(inProgressCount <= 1, 'structured-plan-multiple-in-progress'); + for (const step of completed) { + if (byName.get(step) !== 'completed') regressionCount += 1; + } + for (const step of update.steps) { + if (step.status === 'completed') completed.add(step.stepSha256); + } + } + assert(regressionCount === 0, 'structured-plan-terminal-regression'); + const timeline = validatePlanUpdateActionTimeline(records, indexedUpdates); + const finalUpdate = updates.at(-1); + assert( + finalUpdate.steps.every((step) => step.status === 'completed'), + 'structured-plan-final-not-completed', + ); + const finalSnapshot = inspectStructuredPlanSnapshot( + runtimeState, + 'final-structured-plan', + ); + assert( + runtimeState.planRevision === finalUpdate.planRevision && + createHash('sha256') + .update(runtimeState.planExplanation) + .digest('hex') === finalUpdate.explanationSha256 && + Array.isArray(runtimeState.planSteps) && + runtimeState.planSteps.length === finalUpdate.steps.length && + runtimeState.planSteps.every( + (step, index) => + step.index === index && + createHash('sha256').update(step.title).digest('hex') === + finalUpdate.steps[index].stepSha256 && + step.status === 'completed', + ) && + runtimeState.activePlanStepIndex === null, + 'structured-plan-runtime-state-invalid', + ); + assertStructuredPlanContextSnapshot( + runtimeState, + contextBundle, + 'structured-plan-final-context', + ); + + const recovery = state.planRecovery; + assert( + recovery && + Number.isSafeInteger(recovery.preKillRevision) && + recovery.preKillRevision > 0 && + Number.isSafeInteger(recovery.recoveredRevision) && + recovery.recoveredRevision >= recovery.preKillRevision && + Array.isArray(recovery.preKillCompletedStepHashes) && + recovery.preKillCompletedStepHashes.length > 0 && + recovery.preKillIncompleteStepCount > 0 && + /^[0-9a-f]{64}$/u.test(recovery.preKillTerminalStepHash ?? '') && + Array.isArray(recovery.recoveredCompletedStepHashes) && + /^[0-9a-f]{64}$/u.test(recovery.recoveredTerminalStepHash ?? ''), + 'structured-plan-recovery-state-invalid', + ); + const preKillUpdate = updates.find( + (update) => update.planRevision === recovery.preKillRevision, + ); + const recoveredUpdate = updates.find( + (update) => update.planRevision === recovery.recoveredRevision, + ); + assert( + preKillUpdate && + recoveredUpdate && + preKillUpdate.steps.length >= 3 && + preKillUpdate.steps.some((step) => step.status !== 'completed'), + 'structured-plan-recovery-revision-missing', + ); + const preKillCompleted = preKillUpdate.steps + .filter((step) => step.status === 'completed') + .map((step) => step.stepSha256) + .sort(); + const recoveredCompleted = recoveredUpdate.steps + .filter((step) => step.status === 'completed') + .map((step) => step.stepSha256) + .sort(); + assert( + JSON.stringify(preKillCompleted) === + JSON.stringify(recovery.preKillCompletedStepHashes) && + recovery.preKillIncompleteStepCount === + preKillUpdate.steps.length - preKillCompleted.length && + recovery.preKillTerminalStepHash === + hashValue(JSON.stringify(preKillCompleted)) && + JSON.stringify(recoveredCompleted) === + JSON.stringify(recovery.recoveredCompletedStepHashes) && + recovery.recoveredTerminalStepHash === + hashValue(JSON.stringify(recoveredCompleted)) && + preKillCompleted.every((stepHash) => + recoveredCompleted.includes(stepHash), + ) && + recoveredCompleted.every((stepHash) => + finalSnapshot.completedStepHashes.includes(stepHash), + ), + 'structured-plan-recovery-terminal-step-invalid', + ); + return { + updateCount: updates.length, + planRevision: runtimeState.planRevision, + completedStepCount: runtimeState.planSteps.length, + regressionCount, + preKillRevision: recovery.preKillRevision, + preKillCompletedStepCount: preKillCompleted.length, + preKillIncompleteStepCount: recovery.preKillIncompleteStepCount, + preKillTerminalStepHash: recovery.preKillTerminalStepHash, + recoveredRevision: recovery.recoveredRevision, + recoveredCompletedStepCount: recoveredCompleted.length, + recoveredTerminalStepHash: recovery.recoveredTerminalStepHash, + timelineUpdateCount: timeline.updateCount, + timelineAnchoredUpdateCount: timeline.anchoredUpdateCount, + completionTransitionCount: timeline.completionTransitionCount, + completionObservationCount: timeline.completionObservationCount, + prematureCompletionCount: timeline.prematureCompletionCount, + timelineHash: timeline.timelineHash, + }; +} + +function validatePlanUpdateActionTimeline(records, indexedUpdates) { + const terminalActions = records + .map((record, index) => ({ index, record })) + .filter( + ({ record }) => + record.recordType === 'agent.runtime.tool_observation' && + record.agentId === mainAgentId && + record.runId === state.initialRunId && + record.status !== 'waiting-for-confirmation' && + isNonEmptyString(record.actionId) && + isNonEmptyString(record.actionFingerprint) && + isNonEmptyString(record.tool) && + Number.isSafeInteger(record.updatedAt), + ) + .map(({ index: observationIndex, record: observation }) => { + const receiptIndex = records.findIndex( + (record) => + record.recordType === 'agent.runtime.action_receipt' && + record.agentId === mainAgentId && + record.runId === state.initialRunId && + record.actionId === observation.actionId && + record.actionFingerprint === observation.actionFingerprint && + record.tool === observation.tool && + record.status === observation.status, + ); + const actionIndex = records.findIndex( + (record) => + record.agentId === mainAgentId && + record.runId === state.initialRunId && + record.actionId === observation.actionId && + record.actionFingerprint === observation.actionFingerprint && + record.tool === observation.tool && + [ + 'agent.runtime.tool_action.executing', + 'agent.runtime.tool_confirmation_required', + 'agent.runtime.tool_confirmation.approved', + 'agent.runtime.tool_confirmation.rejected', + 'agent.runtime.tool_action.observed', + ].includes(record.recordType), + ); + const receipt = records[receiptIndex]; + const action = records[actionIndex]; + assert( + actionIndex >= 0 && + actionIndex < observationIndex && + receiptIndex >= 0 && + receiptIndex < observationIndex && + Number.isSafeInteger(action.updatedAt) && + Number.isSafeInteger(receipt.updatedAt) && + action.updatedAt <= receipt.updatedAt && + receipt.updatedAt <= observation.updatedAt, + 'structured-plan-terminal-action-lifecycle-invalid', + ); + return { + actionIdentityHash: hashValue( + `${observation.actionId}\0${observation.actionFingerprint}\0${observation.tool}`, + ), + actionIndex, + observation, + observationIndex, + receiptIndex, + }; + }); + assert( + duplicateCount( + terminalActions.map((action) => action.observation.actionId), + ) === 0, + 'structured-plan-terminal-observation-duplicate', + ); + + const appliedSteerIndexes = records + .map((record, index) => ({ index, record })) + .filter( + ({ record }) => + record.recordType === 'agent.runtime.steer' && + record.agentId === mainAgentId && + record.runId === state.initialRunId && + record.status === 'applied', + ) + .map(({ index }) => index); + const timeline = []; + let anchoredUpdateCount = 0; + let completionTransitionCount = 0; + let completionObservationCount = 0; + let prematureCompletionCount = 0; + let previousCompleted = new Set(); + let previousUpdateIndex = -1; + + for (const [updateIndex, indexedUpdate] of indexedUpdates.entries()) { + const update = indexedUpdate.record; + const completedNow = new Set( + update.steps + .filter((step) => step.status === 'completed') + .map((step) => step.stepSha256), + ); + const newlyCompleted = [...completedNow].filter( + (stepHash) => !previousCompleted.has(stepHash), + ); + const intervalActions = terminalActions.filter( + (action) => + action.observationIndex > previousUpdateIndex && + action.observationIndex < indexedUpdate.index, + ); + const intervalSteerIndexes = appliedSteerIndexes.filter( + (index) => index > previousUpdateIndex && index < indexedUpdate.index, + ); + if (updateIndex === 0) { + assert( + newlyCompleted.length === 0, + 'structured-plan-initial-update-precompleted-step', + ); + } else { + assert( + intervalActions.length > 0 || intervalSteerIndexes.length > 0, + 'structured-plan-update-without-action-or-steer-anchor', + ); + anchoredUpdateCount += 1; + } + if (newlyCompleted.length > intervalActions.length) { + prematureCompletionCount += + newlyCompleted.length - intervalActions.length; + } + assert( + newlyCompleted.length <= intervalActions.length, + 'structured-plan-completed-before-terminal-observation', + ); + const completionActions = + newlyCompleted.length === 0 + ? [] + : intervalActions.slice(-newlyCompleted.length); + for (const action of completionActions) { + assert( + action.observation.updatedAt <= update.updatedAt && + action.receiptIndex < action.observationIndex && + action.observationIndex < indexedUpdate.index, + 'structured-plan-completion-timestamp-invalid', + ); + } + completionTransitionCount += newlyCompleted.length; + completionObservationCount += completionActions.length; + timeline.push({ + actionIdentityHashes: completionActions.map( + (action) => action.actionIdentityHash, + ), + completedStepHashes: newlyCompleted.sort(), + planRevision: update.planRevision, + planSequence: indexedUpdate.index + 1, + steerSequences: intervalSteerIndexes.map((index) => index + 1), + terminalObservationSequences: completionActions.map( + (action) => action.observationIndex + 1, + ), + updatedAt: update.updatedAt, + }); + previousCompleted = completedNow; + previousUpdateIndex = indexedUpdate.index; + } + assert( + completionTransitionCount > 0 && + completionTransitionCount === completionObservationCount && + prematureCompletionCount === 0, + 'structured-plan-completion-observation-coverage-invalid', + ); + return { + updateCount: indexedUpdates.length, + anchoredUpdateCount, + completionTransitionCount, + completionObservationCount, + prematureCompletionCount, + timelineHash: hashValue(JSON.stringify(timeline)), + }; +} + function validateConfirmedActionLifecycles(records) { const indexed = records.map((record, index) => ({ record, index })); const approvals = indexed.filter( @@ -5659,6 +7439,57 @@ function finalMessageId(agentId, sessionId, runId) { return `agent-finalization-${fingerprint.slice(0, 32)}`; } +function backgroundTaskMessageId(agentId, sessionId, runId, source) { + const fingerprint = createHash('sha256') + .update(`${agentId}\n${sessionId}\n${runId}\n${source}`) + .digest('hex'); + return `runtime-task-${fingerprint.slice(0, 32)}`; +} + +function actionReceiptIdentity(record) { + return JSON.stringify([ + record.agentId, + record.taskId, + record.sessionId, + record.runId, + record.actionId, + record.actionFingerprint, + record.tool, + record.executionMode, + record.status, + record.inputSummary, + record.summary, + record.safeDetail, + record.detailUnavailable, + record.updatedAt, + ]); +} + +function disposableProjectPathVariants() { + if (!isNonEmptyString(state.projectRoot)) return []; + const absolute = path.resolve(state.projectRoot); + const forward = path.posix.normalize(absolute.replaceAll('\\', '/')); + const backward = path.win32.normalize(absolute.replaceAll('/', '\\')); + return [ + ...new Set([ + state.projectRoot, + absolute, + path.normalize(absolute), + forward, + backward, + forward.replaceAll('/', '\\'), + backward.replaceAll('\\', '/'), + ]), + ].filter((value) => isNonEmptyString(value) && value.length > 1); +} + +function sumObjectValues(value) { + return Object.values(value).reduce((total, count) => { + assert(Number.isSafeInteger(count) && count >= 0, 'leak-count-invalid'); + return total + count; + }, 0); +} + function countBy(values) { const counts = new Map(); for (const value of values) counts.set(value, (counts.get(value) ?? 0) + 1); @@ -5732,8 +7563,9 @@ function redactSecrets(value) { result = result.split(secret).join('[REDACTED]'); if (state.options?.configDir) result = result.split(state.options.configDir).join('[CONFIG_DIR]'); - if (state.projectRoot) - result = result.split(state.projectRoot).join('[PROJECT]'); + for (const projectPath of disposableProjectPathVariants()) { + result = result.split(projectPath).join('[PROJECT]'); + } return result; } diff --git a/apps/ai-game-creator-shell/src-tauri/src/agent.rs b/apps/ai-game-creator-shell/src-tauri/src/agent.rs index e3455ff0c..a0fa396d6 100644 --- a/apps/ai-game-creator-shell/src-tauri/src/agent.rs +++ b/apps/ai-game-creator-shell/src-tauri/src/agent.rs @@ -13,7 +13,7 @@ fn external_agent_runner_owns_background_execution() -> bool { } pub(crate) const AGENT_RUNTIME_PENDING_ACTION_SCHEMA_VERSION: &str = - "game-creator-pending-action.v3"; + "game-creator-pending-action.v4"; pub(crate) const AGENT_RUNTIME_PENDING_ACTION_STATUS_PENDING: &str = "pending-confirmation"; pub(crate) const AGENT_RUNTIME_PENDING_ACTION_STATUS_APPROVED: &str = "approved"; pub(crate) const AGENT_RUNTIME_PENDING_ACTION_STATUS_EXECUTING: &str = "executing"; @@ -25,6 +25,8 @@ pub(crate) const AGENT_RUNTIME_ACTION_EXECUTION_MODE_AUTO: &str = "auto"; pub(crate) const AGENT_RUNTIME_ACTION_EXECUTION_MODE_CONFIRMATION: &str = "confirmation"; pub(crate) const AGENT_RUNTIME_ACTION_FINGERPRINT_VERSION: &str = "sha256-serde-json-v2"; pub(crate) const AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION: &str = + "game-creator-runtime-context-bundle.v3"; +const AGENT_RUNTIME_CONTEXT_BUNDLE_LEGACY_SCHEMA_VERSION: &str = "game-creator-runtime-context-bundle.v2"; pub(crate) const AGENT_RUNTIME_PROJECT_REVISION_SCHEMA_VERSION: &str = "game-creator-project-revision.v1"; @@ -838,15 +840,28 @@ fn game_creator_agent_runtime_finalization_matches_state( && journal.response_steer_cursor == state.applied_steer_cursor } +fn game_creator_agent_runtime_finalization_plan_matches_state( + journal: &AgentRuntimeFinalizationJournal, + state: &AgentRuntimeState, +) -> bool { + journal.schema_version == AGENT_RUNTIME_FINALIZATION_LEGACY_SCHEMA_VERSION + || journal.plan_revision == 0 + || (journal.plan_revision == state.plan_revision + && journal.plan_explanation == state.plan_explanation + && journal.plan == state.plan + && journal.plan_steps == state.plan_steps + && journal.active_plan_step_index == state.active_plan_step_index) +} + fn read_game_creator_agent_runtime_state_for_finalization_resume( root: &Path, agent_id: &str, -) -> Result { +) -> Result<(AgentRuntimeState, bool), String> { let state_result = read_game_creator_agent_runtime_at(root, agent_id).map(|result| result.state); if let Ok(state) = &state_result { if !state.run_id.trim().is_empty() { - return Ok(state.clone()); + return Ok((state.clone(), false)); } } @@ -881,9 +896,10 @@ fn read_game_creator_agent_runtime_state_for_finalization_resume( } } match finalization_tasks.len() { - 0 => state_result, - 1 => Ok(agent_runtime_state_from_task_record( - &finalization_tasks.remove(0), + 0 => state_result.map(|state| (state, false)), + 1 => Ok(( + agent_runtime_state_from_task_record(&finalization_tasks.remove(0)), + true, )), count => Err(format!( "Agent Runtime 同一 Agent 存在 {count} 个未完成 finalization,已阻断自动恢复" @@ -896,7 +912,8 @@ fn resume_game_creator_agent_finalization_at( agent_id: &str, runtime_lock: AgentRuntimeTaskLock, ) -> Result { - let mut state = read_game_creator_agent_runtime_state_for_finalization_resume(root, agent_id)?; + let (mut state, state_reconstructed_from_task) = + read_game_creator_agent_runtime_state_for_finalization_resume(root, agent_id)?; if state.run_id.trim().is_empty() { return Ok(AgentRuntimeFinalizationResume::NotFound(runtime_lock)); } @@ -911,6 +928,15 @@ fn resume_game_creator_agent_finalization_at( .map(AgentRuntimeFinalizationResume::Blocked); } }; + if state_reconstructed_from_task + && journal.schema_version == AGENT_RUNTIME_FINALIZATION_SCHEMA_VERSION + { + state.plan_revision = journal.plan_revision; + state.plan_explanation = journal.plan_explanation.clone(); + state.plan = journal.plan.clone(); + state.plan_steps = journal.plan_steps.clone(); + state.active_plan_step_index = journal.active_plan_step_index; + } if !game_creator_agent_runtime_finalization_matches_state(&journal, &state) { let error = "Agent Runtime finalization 恢复已阻断:与当前 Runtime 身份不匹配"; record_game_creator_agent_runtime_finalization_pending(root, &state, error); @@ -924,6 +950,53 @@ fn resume_game_creator_agent_finalization_at( )?; let assistant_exists = game_creator_agent_runtime_finalization_assistant_exists(root, &journal)?; + if assistant_exists + && !state_reconstructed_from_task + && !game_creator_agent_runtime_finalization_plan_matches_state(&journal, &state) + { + let error = "Agent Runtime finalization 恢复已阻断:assistant 已提交但结构化计划快照与当前 Runtime 不匹配"; + record_game_creator_agent_runtime_finalization_pending(root, &state, error); + return read_game_creator_agent_runtime_at(root, agent_id) + .map(AgentRuntimeFinalizationResume::Blocked); + } + if state_reconstructed_from_task && !assistant_exists { + let error = "Agent Runtime finalization 恢复已阻断:Runtime state 缺失,不能从 task record 猜测结构化计划快照"; + state.status = "failed".to_string(); + state.phase = "needs-reconciliation".to_string(); + state.current_action = "最终回复恢复需要人工核对".to_string(); + state.waiting_on = + "开发者核对 Runtime state、context bundle 与 finalization journal".to_string(); + state.next_step = "恢复完整计划快照后显式恢复当前 run".to_string(); + state.error = Some(error.to_string()); + state.updated_at = unix_timestamp(); + append_game_creator_agent_runtime_task(root, &state)?; + refresh_game_creator_agent_runtime_task_queue(root, &mut state)?; + write_game_creator_agent_runtime_state(root, &state)?; + let _ = append_game_creator_agent_runtime_event( + root, + &state, + "finalization.needs_reconciliation", + "failed", + "needs-reconciliation", + "最终回复恢复缺少可信 Runtime state,已停止自动完成。", + None, + ); + let _ = append_agent_db_record( + root, + serde_json::json!({ + "recordType": "agent.runtime.background_task.finalization_needs_reconciliation", + "agentId": state.agent_id, + "taskId": state.task_id, + "sessionId": state.session_id, + "runId": state.run_id, + "source": state.source, + "reason": "runtime-state-missing", + }), + ); + emit_game_creator_agent_runtime_update(root, agent_id); + return read_game_creator_agent_runtime_at(root, agent_id) + .map(AgentRuntimeFinalizationResume::Blocked); + } let task_cancelled = read_latest_game_creator_agent_runtime_task_by_run_id(root, agent_id, &state.run_id)? .is_some_and(|task| task.status == "cancelled"); @@ -967,12 +1040,13 @@ fn resume_game_creator_agent_finalization_at( } if journal.status == AGENT_RUNTIME_FINALIZATION_STATUS_PREPARED && !assistant_exists { let current_revision = read_game_creator_agent_runtime_project_revision(root)?; - let blocker = if let Some(blocker) = - agent_runtime_non_verification_completion_blocker_at_locked( - root, - &journal.agent_id, - &journal.run_id, - ) { + let blocker = if let Some(blocker) = structured_plan_completion_blocker(&state) { + Some(blocker) + } else if let Some(blocker) = agent_runtime_non_verification_completion_blocker_at_locked( + root, + &journal.agent_id, + &journal.run_id, + ) { Some(blocker) } else if current_revision.revision != journal.response_revision { Some(agent_runtime_verification_blocker( @@ -2548,7 +2622,6 @@ pub(crate) fn agent_runtime_tool_requires_repository_context_fingerprint_gate(to fn pending_repository_context_drift_observation( root: &Path, - runtime: &AgentRuntimeState, pending: &AgentRuntimePendingToolAction, ) -> Result, String> { if !agent_runtime_tool_requires_repository_context_fingerprint_gate( @@ -2556,25 +2629,18 @@ fn pending_repository_context_drift_observation( ) { return Ok(None); } - let Some(bundle) = read_game_creator_agent_runtime_context_bundle(root, runtime)? else { - return Ok(None); - }; - if bundle.repository_context_fingerprint.is_empty() { - return Ok(None); - } let current = build_repository_startup_context_at(root)?; - if current.fingerprint == bundle.repository_context_fingerprint { + if current.fingerprint == pending.planned_repository_context_fingerprint { return Ok(None); } Ok(Some(AgentRuntimeToolObservation { tool: pending.action.tool.clone(), status: "blocked".to_string(), summary: "仓库规范或启动上下文已漂移,旧动作未执行".to_string(), - detail: Some(format!( - "repositoryContextDrift=true · plannedFingerprint={} · currentFingerprint={} · 请在同一 run 下一轮 planning 重新确认适用规范", - bundle.repository_context_fingerprint, - current.fingerprint - )), + detail: Some( + "repositoryContextDrift=true · 请在同一 run 下一轮 planning 重新确认适用规范" + .to_string(), + ), })) } @@ -2770,7 +2836,7 @@ pub(crate) async fn continue_game_creator_agent_pending_tool_action( let observation = match pending.status.as_str() { AGENT_RUNTIME_PENDING_ACTION_STATUS_APPROVED => { let repository_context_drift = - match pending_repository_context_drift_observation(&root, &runtime, &pending) { + match pending_repository_context_drift_observation(&root, &pending) { Ok(observation) => observation, Err(error) => { let _ = mark_game_creator_agent_runtime_needs_reconciliation_at( @@ -4166,7 +4232,8 @@ async fn run_game_creator_agent_background_task_pass_with_context( } } }; - plan = requested_plan; + let planning_repository_context_fingerprint = requested_plan.repository_context_fingerprint; + plan = requested_plan.plan; match consume_game_creator_agent_runtime_steers( &root, @@ -4212,22 +4279,149 @@ async fn run_game_creator_agent_background_task_pass_with_context( runtime.status.as_str(), runtime.phase.as_str(), &format!("Agent 已形成第 {} 轮任务理解摘要。", loop_index + 1), - Some(&plan.thinking_summary), + Some(&format!( + "thinkingSummarySha256={:x} · chars={}", + Sha256::digest(plan.thinking_summary.as_bytes()), + plan.thinking_summary.chars().count() + )), ); } - if !plan.plan.is_empty() { + if let Some(plan_update) = plan.plan_update.as_ref() { + match apply_agent_runtime_plan_update(&mut runtime, plan_update) { + Ok(true) => { + runtime.updated_at = unix_timestamp(); + if let Err(error) = write_game_creator_agent_runtime_state(&root, &runtime) { + return fail_game_creator_agent_background_context_at( + &root, + &agent_id, + &session_id, + runtime, + &format!("持久化结构化计划 Runtime state 失败:{error}"), + ); + } + let plan_audit_steps = runtime + .plan_steps + .iter() + .take(AGENT_RUNTIME_PLAN_STEP_LIMIT) + .map(|step| { + serde_json::json!({ + "stepSha256": format!( + "{:x}", + Sha256::digest(step.title.as_bytes()) + ), + "status": step.status, + }) + }) + .collect::>(); + let _ = append_game_creator_agent_runtime_event( + &root, + &runtime, + "plan_update", + runtime.status.as_str(), + runtime.phase.as_str(), + &format!( + "Agent 已提交第 {} 版结构化计划,进度 {}/{}。", + runtime.plan_revision, + runtime + .plan_steps + .iter() + .filter(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_COMPLETED) + .count(), + runtime.plan_steps.len() + ), + Some(&format!( + "planRevision={} · stepCount={}", + runtime.plan_revision, + runtime.plan_steps.len() + )), + ); + let _ = append_agent_db_record( + &root, + serde_json::json!({ + "recordType": "agent.runtime.plan_update", + "agentId": runtime.agent_id, + "taskId": runtime.task_id, + "sessionId": runtime.session_id, + "runId": runtime.run_id, + "planRevision": runtime.plan_revision, + "explanationSha256": format!( + "{:x}", + Sha256::digest(runtime.plan_explanation.as_bytes()) + ), + "explanationChars": runtime.plan_explanation.chars().count(), + "steps": plan_audit_steps, + }), + ); + } + Ok(false) => {} + Err(error) => { + let observation = AgentRuntimeToolObservation { + tool: "runtime.plan_update".to_string(), + status: "rejected".to_string(), + summary: "结构化计划更新被 Runtime 拒绝".to_string(), + detail: Some(sanitize_agent_runtime_text(&error, 500)), + }; + runtime.current_action = "修正结构化计划更新".to_string(); + runtime.waiting_on = "Agent 提交满足单调约束的 planUpdate".to_string(); + runtime.next_step = "保留已完成步骤并修正计划状态后重新提交".to_string(); + runtime.observations.push(observation.summary()); + runtime.updated_at = unix_timestamp(); + let _ = write_game_creator_agent_runtime_state(&root, &runtime); + let _ = append_game_creator_agent_runtime_event( + &root, + &runtime, + "plan_update.rejected", + runtime.status.as_str(), + runtime.phase.as_str(), + &observation.summary, + Some(&format!( + "planUpdateErrorSha256={:x}", + Sha256::digest(error.as_bytes()) + )), + ); + context_tracker.record(&observation); + observations.push(observation); + match checkpoint_game_creator_agent_runtime_context( + &root, + &mut runtime, + &task, + &plan, + &mut observations, + loop_index + 1, + &mut context_tracker, + ) { + Ok(AgentRuntimeContextCheckpoint::Stalled) => { + context_stalled = true; + break 'agent_loop; + } + Ok(_) => continue 'agent_loop, + Err(error) => { + return fail_game_creator_agent_background_context_at( + &root, + &agent_id, + &session_id, + runtime, + &error, + ); + } + } + } + } + } else if !plan.plan.is_empty() { update_agent_runtime_plan_steps(&mut runtime, plan.plan.clone()); - runtime.updated_at = unix_timestamp(); - let _ = write_game_creator_agent_runtime_state(&root, &runtime); - let _ = append_game_creator_agent_runtime_event( - &root, - &runtime, - "plan", - runtime.status.as_str(), - runtime.phase.as_str(), - &format!("Agent 已生成第 {} 轮行动计划。", loop_index + 1), - Some(&runtime.plan.join(" / ")), - ); + if !agent_runtime_has_structured_plan(&runtime) { + runtime.updated_at = unix_timestamp(); + let _ = write_game_creator_agent_runtime_state(&root, &runtime); + let _ = append_game_creator_agent_runtime_event( + &root, + &runtime, + "plan", + runtime.status.as_str(), + runtime.phase.as_str(), + &format!("Agent 已生成第 {} 轮行动计划。", loop_index + 1), + Some(&format!("planStepCount={}", runtime.plan.len())), + ); + } } if let Err(error) = persist_game_creator_agent_runtime_context( &root, @@ -4248,25 +4442,32 @@ async fn run_game_creator_agent_background_task_pass_with_context( } if plan.actions.is_empty() { - let completion_blocker = - process_session_completion_blocker_at(&root, &agent_id, &runtime.run_id) - .or_else(|| { - isolated_join_completion_blocker_at(&root, &agent_id, &runtime.run_id) - }) - .or_else(|| { - static_delegate_completion_blocker_at(&root, &agent_id, &runtime.run_id) - }) - .or_else(|| { - project_verification_completion_blocker_at( - &root, - &agent_id, - &runtime.run_id, - &observations, - ) - }); + let completion_blocker = structured_plan_completion_blocker(&runtime) + .or_else(|| { + process_session_completion_blocker_at(&root, &agent_id, &runtime.run_id) + }) + .or_else(|| isolated_join_completion_blocker_at(&root, &agent_id, &runtime.run_id)) + .or_else(|| { + static_delegate_completion_blocker_at(&root, &agent_id, &runtime.run_id) + }) + .or_else(|| { + project_verification_completion_blocker_at( + &root, + &agent_id, + &runtime.run_id, + &observations, + ) + }); if let Some(blocker) = completion_blocker { let blocker_summary = blocker.summary(); - if blocker.tool == "runtime.process_session" { + if blocker.tool == "runtime.plan_update" { + runtime.status = "running".to_string(); + runtime.phase = "planning".to_string(); + runtime.current_action = "等待结构化计划进度更新".to_string(); + runtime.waiting_on = "当前计划的必要步骤全部 completed".to_string(); + runtime.next_step = + "根据真实工具观察提交新的 planUpdate,再决定下一步动作".to_string(); + } else if blocker.tool == "runtime.process_session" { runtime.status = "running".to_string(); runtime.phase = "waiting-for-process-session".to_string(); runtime.current_action = "等待进程会话收束".to_string(); @@ -4503,6 +4704,23 @@ async fn run_game_creator_agent_background_task_pass_with_context( runtime.current_action.as_str(), action.reason.as_deref(), ); + if let Err(error) = persist_game_creator_agent_runtime_context( + &root, + &runtime, + &task, + &plan, + &observations, + loop_index, + &context_tracker, + ) { + return fail_game_creator_agent_background_context_at( + &root, + &agent_id, + &session_id, + runtime, + &error, + ); + } if stop_game_creator_agent_runtime_if_cancel_requested(&root, &mut runtime) { return AgentBackgroundTaskOutcome::Finished; } @@ -4513,6 +4731,7 @@ async fn run_game_creator_agent_background_task_pass_with_context( &plan, &observations, &planning_request_revision, + &planning_repository_context_fingerprint, action, action_index, AGENT_RUNTIME_ACTION_EXECUTION_MODE_CONFIRMATION, @@ -4582,22 +4801,19 @@ async fn run_game_creator_agent_background_task_pass_with_context( ); return AgentBackgroundTaskOutcome::Finished; } - let repository_context_drift = match pending_repository_context_drift_observation( - &root, - &runtime, - &pending_action, - ) { - Ok(observation) => observation, - Err(error) => { - let _ = mark_game_creator_agent_runtime_needs_reconciliation_at( - &root, - &mut runtime, - &pending_action, - &error, - ); - return AgentBackgroundTaskOutcome::NeedsReconciliation; - } - }; + let repository_context_drift = + match pending_repository_context_drift_observation(&root, &pending_action) { + Ok(observation) => observation, + Err(error) => { + let _ = mark_game_creator_agent_runtime_needs_reconciliation_at( + &root, + &mut runtime, + &pending_action, + &error, + ); + return AgentBackgroundTaskOutcome::NeedsReconciliation; + } + }; if let Some(observation) = repository_context_drift { pending_action.status = AGENT_RUNTIME_PENDING_ACTION_STATUS_OBSERVED_REJECTED.to_string(); @@ -4962,6 +5178,34 @@ async fn run_game_creator_agent_background_task_pass_with_context( let _ = fail_game_creator_agent_runtime_turn_at(&root, runtime, &error); return AgentBackgroundTaskOutcome::Finished; } + context_tracker.record(&observation); + observations.push(observation.clone()); + if let Err(error) = persist_game_creator_agent_runtime_context( + &root, + &runtime, + &task, + &plan, + &observations, + loop_index, + &context_tracker, + ) { + if let Some(pending_action) = durable_action.as_ref() { + let _ = mark_game_creator_agent_runtime_needs_reconciliation_at( + &root, + &mut runtime, + pending_action, + &format!("工具 observation 已持久化,但 context bundle 落盘失败:{error}"), + ); + return AgentBackgroundTaskOutcome::NeedsReconciliation; + } + return fail_game_creator_agent_background_context_at( + &root, + &agent_id, + &session_id, + runtime, + &error, + ); + } let (action_id, action_fingerprint, _, execution_mode) = observation_action_identity .as_ref() .expect("durable observation has action identity"); @@ -5034,8 +5278,6 @@ async fn run_game_creator_agent_background_task_pass_with_context( } return AgentBackgroundTaskOutcome::WaitingForConfirmation; } - context_tracker.record(&observation); - observations.push(observation); if stop_game_creator_agent_runtime_if_cancel_requested(&root, &mut runtime) { return AgentBackgroundTaskOutcome::Finished; } @@ -5373,7 +5615,9 @@ async fn run_game_creator_agent_background_task_pass_with_context( pub(crate) const AGENT_RUNTIME_BACKGROUND_LOOP_LIMIT: usize = 6; const AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT: usize = 3; -const AGENT_RUNTIME_FINALIZATION_SCHEMA_VERSION: &str = "game-creator-runtime-finalization.v1"; +const AGENT_RUNTIME_FINALIZATION_SCHEMA_VERSION: &str = "game-creator-runtime-finalization.v2"; +const AGENT_RUNTIME_FINALIZATION_LEGACY_SCHEMA_VERSION: &str = + "game-creator-runtime-finalization.v1"; const AGENT_RUNTIME_FINALIZATION_STATUS_PREPARED: &str = "prepared"; const AGENT_RUNTIME_FINALIZATION_STATUS_ASSISTANT_PERSISTED: &str = "assistant-persisted"; const AGENT_RUNTIME_FINALIZATION_STATUS_RUNTIME_COMPLETED: &str = "runtime-completed"; @@ -5387,6 +5631,10 @@ const AGENT_RUNTIME_ACTION_HISTORY_MAX_DB_BYTES: u64 = 32 * 1024 * 1024; pub(crate) const AGENT_RUNTIME_ACTION_HISTORY_MAX_OUTPUT_CHARS: usize = 7_200; const AGENT_RUNTIME_ACTION_RECEIPT_SAFE_DETAIL_MAX_CHARS: usize = 500; pub(crate) const AGENT_RUNTIME_PLAN_STEP_LIMIT: usize = 8; +const AGENT_RUNTIME_PLAN_STATUS_PENDING: &str = "pending"; +const AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS: &str = "in_progress"; +const AGENT_RUNTIME_PLAN_STATUS_COMPLETED: &str = "completed"; +const AGENT_RUNTIME_PLAN_STATUS_FAILED: &str = "failed"; const AGENT_RUNTIME_TOOL_OBSERVATION_MAX_CHARS: usize = 900; const AGENT_RUNTIME_COMMAND_OUTPUT_CONTEXT_MAX_CHARS: usize = 64_000; const AGENT_RUNTIME_PROCESS_POLL_CONTEXT_MAX_CHARS: usize = 32_000; @@ -5443,11 +5691,27 @@ pub(crate) enum AgentRuntimeFinalizationCheckpoint { #[serde(deny_unknown_fields, rename_all = "camelCase")] pub(crate) struct AgentRuntimeToolPlan { pub(crate) thinking_summary: String, + #[serde(default)] + pub(crate) plan_update: Option, pub(crate) plan: Vec, pub(crate) actions: Vec, pub(crate) response: String, } +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields, rename_all = "camelCase")] +pub(crate) struct AgentRuntimePlanUpdate { + pub(crate) explanation: String, + pub(crate) steps: Vec, +} + +#[derive(Clone, Debug, Deserialize, Eq, PartialEq, Serialize)] +#[serde(deny_unknown_fields, rename_all = "camelCase")] +pub(crate) struct AgentRuntimePlanUpdateStep { + pub(crate) step: String, + pub(crate) status: String, +} + #[derive(Clone, Debug, Eq, PartialEq)] pub(crate) struct ParsedAgentRuntimeToolPlan { pub(crate) plan: AgentRuntimeToolPlan, @@ -5456,6 +5720,11 @@ pub(crate) struct ParsedAgentRuntimeToolPlan { pub(crate) function_name: Option, } +struct RequestedAgentRuntimeToolPlan { + plan: AgentRuntimeToolPlan, + repository_context_fingerprint: String, +} + #[derive(Clone, Debug, Default, Deserialize, Eq, PartialEq, Serialize)] #[serde(rename_all = "camelCase")] pub(crate) struct AgentRuntimeToolAction { @@ -5568,6 +5837,18 @@ pub(crate) struct AgentRuntimeFinalizationJournal { pub(crate) response_revision: u64, #[serde(default)] pub(crate) response_steer_cursor: u64, + #[serde(default)] + pub(crate) plan_revision: u64, + #[serde(default)] + pub(crate) plan_explanation: String, + #[serde(default)] + pub(crate) plan: Vec, + #[serde(default)] + pub(crate) plan_steps: Vec, + #[serde(default)] + pub(crate) active_plan_step_index: Option, + #[serde(default)] + pub(crate) plan_snapshot_fingerprint: String, pub(crate) verification_gate: AgentRuntimeVerificationGate, pub(crate) finalization_id: String, pub(crate) message_id: String, @@ -5597,7 +5878,15 @@ pub(crate) struct AgentRuntimeContextBundle { pub(crate) next_loop_index: u32, pub(crate) context_window: u32, pub(crate) thinking_summary: String, + #[serde(default)] + pub(crate) plan_revision: u64, + #[serde(default)] + pub(crate) plan_explanation: String, pub(crate) plan: Vec, + #[serde(default)] + pub(crate) plan_steps: Vec, + #[serde(default)] + pub(crate) active_plan_step_index: Option, pub(crate) fallback_response: String, pub(crate) observations: Vec, pub(crate) verification_gate: AgentRuntimeVerificationGate, @@ -6556,13 +6845,19 @@ pub(crate) fn consume_game_creator_agent_runtime_steers( runtime.applied_steer_refs = refs; runtime.queued_steer_count = 0; validate_agent_runtime_steer_refs(runtime.applied_steer_cursor, &runtime.applied_steer_refs)?; + let plan_recheck = if agent_runtime_has_structured_plan(runtime) { + "需基于原结构化计划重审未完成步骤,已完成步骤继续保留" + } else { + "旧短计划需重新生成" + }; let observation = AgentRuntimeToolObservation { tool: "runtime.steer".to_string(), status: "ok".to_string(), summary: format!( - "已应用 {} 条运行中追加指令,steerCursor={},旧计划需重新生成", + "已应用 {} 条运行中追加指令,steerCursor={},{}", pending.len(), - runtime.applied_steer_cursor + runtime.applied_steer_cursor, + plan_recheck ), detail: Some(format!( "previousCursor={previous_cursor} · appliedSequences={}", @@ -6598,7 +6893,11 @@ pub(crate) fn consume_game_creator_agent_runtime_steers( "steer.applied", runtime.status.as_str(), runtime.phase.as_str(), - "运行中追加指令已进入当前 run 上下文,旧计划已作废。", + if agent_runtime_has_structured_plan(runtime) { + "运行中追加指令已进入当前 run 上下文;结构化计划等待重审,已完成步骤不会回退。" + } else { + "运行中追加指令已进入当前 run 上下文,旧短计划已作废。" + }, Some(&format!( "sequence={} · chars={} · sha256={}", entry.identity.sequence, @@ -7781,10 +8080,17 @@ fn game_creator_agent_runtime_finalization_id( response_fingerprint: &str, response_revision: u64, response_steer_cursor: u64, + plan_snapshot_fingerprint: &str, ) -> String { - let payload = format!( - "{project_id}\n{agent_id}\n{session_id}\n{run_id}\n{response_fingerprint}\n{response_revision}\n{response_steer_cursor}" - ); + let payload = if plan_snapshot_fingerprint.is_empty() { + format!( + "{project_id}\n{agent_id}\n{session_id}\n{run_id}\n{response_fingerprint}\n{response_revision}\n{response_steer_cursor}" + ) + } else { + format!( + "{project_id}\n{agent_id}\n{session_id}\n{run_id}\n{response_fingerprint}\n{response_revision}\n{response_steer_cursor}\n{plan_snapshot_fingerprint}" + ) + }; let fingerprint = format!("{:x}", Sha256::digest(payload.as_bytes())); format!( "agent-finalization-{}", @@ -7792,6 +8098,24 @@ fn game_creator_agent_runtime_finalization_id( ) } +fn game_creator_agent_runtime_plan_snapshot_fingerprint( + plan_revision: u64, + plan_explanation: &str, + plan: &[String], + plan_steps: &[AgentRuntimePlanStep], + active_plan_step_index: Option, +) -> String { + let payload = serde_json::json!({ + "planRevision": plan_revision, + "planExplanation": plan_explanation, + "plan": plan, + "planSteps": plan_steps, + "activePlanStepIndex": active_plan_step_index, + }); + let encoded = serde_json::to_vec(&payload).unwrap_or_default(); + format!("{:x}", Sha256::digest(encoded)) +} + fn build_game_creator_agent_runtime_finalization_journal( root: &Path, state: &AgentRuntimeState, @@ -7810,6 +8134,13 @@ fn build_game_creator_agent_runtime_finalization_journal( } let project_id = game_creator_agent_runtime_context_project_id(root)?; let response_fingerprint = format!("{:x}", Sha256::digest(response.as_bytes())); + let plan_snapshot_fingerprint = game_creator_agent_runtime_plan_snapshot_fingerprint( + state.plan_revision, + &state.plan_explanation, + &state.plan, + &state.plan_steps, + state.active_plan_step_index, + ); let finalization_id = game_creator_agent_runtime_finalization_id( &project_id, &state.agent_id, @@ -7818,6 +8149,7 @@ fn build_game_creator_agent_runtime_finalization_journal( &response_fingerprint, response_revision, state.applied_steer_cursor, + &plan_snapshot_fingerprint, ); let message_id = game_creator_agent_runtime_finalization_message_id( &state.agent_id, @@ -7841,6 +8173,12 @@ fn build_game_creator_agent_runtime_finalization_journal( response_fingerprint, response_revision, response_steer_cursor: state.applied_steer_cursor, + plan_revision: state.plan_revision, + plan_explanation: state.plan_explanation.clone(), + plan: state.plan.clone(), + plan_steps: state.plan_steps.clone(), + active_plan_step_index: state.active_plan_step_index, + plan_snapshot_fingerprint, verification_gate: read_game_creator_agent_runtime_verification_gate( root, &state.agent_id, @@ -7869,7 +8207,8 @@ fn validate_game_creator_agent_runtime_finalization_journal( expected_agent_id: &str, expected_run_id: &str, ) -> Result<(), String> { - if journal.schema_version != AGENT_RUNTIME_FINALIZATION_SCHEMA_VERSION { + let legacy_schema = journal.schema_version == AGENT_RUNTIME_FINALIZATION_LEGACY_SCHEMA_VERSION; + if journal.schema_version != AGENT_RUNTIME_FINALIZATION_SCHEMA_VERSION && !legacy_schema { return Err(format!( "不支持的 Agent Runtime finalization 版本:{}", journal.schema_version @@ -7894,6 +8233,47 @@ fn validate_game_creator_agent_runtime_finalization_journal( if journal.response_fingerprint != response_fingerprint { return Err("Agent Runtime finalization 回复指纹不匹配".to_string()); } + if legacy_schema { + if journal.plan_revision != 0 + || !journal.plan_explanation.is_empty() + || !journal.plan.is_empty() + || !journal.plan_steps.is_empty() + || journal.active_plan_step_index.is_some() + || !journal.plan_snapshot_fingerprint.is_empty() + { + return Err("旧版 Agent Runtime finalization 不能携带计划快照".to_string()); + } + } else { + validate_agent_runtime_structured_plan_snapshot( + journal.plan_revision, + &journal.plan_explanation, + &journal.plan, + &journal.plan_steps, + journal.active_plan_step_index, + )?; + if journal.plan_revision > 0 + && (journal.active_plan_step_index.is_some() + || journal + .plan_steps + .iter() + .any(|step| step.status != AGENT_RUNTIME_PLAN_STATUS_COMPLETED)) + { + return Err( + "Agent Runtime finalization 必须绑定已全部完成的结构化计划快照".to_string(), + ); + } + let expected_plan_snapshot_fingerprint = + game_creator_agent_runtime_plan_snapshot_fingerprint( + journal.plan_revision, + &journal.plan_explanation, + &journal.plan, + &journal.plan_steps, + journal.active_plan_step_index, + ); + if journal.plan_snapshot_fingerprint != expected_plan_snapshot_fingerprint { + return Err("Agent Runtime finalization 计划快照指纹不匹配".to_string()); + } + } let expected_finalization_id = game_creator_agent_runtime_finalization_id( &journal.project_id, &journal.agent_id, @@ -7902,6 +8282,7 @@ fn validate_game_creator_agent_runtime_finalization_journal( &journal.response_fingerprint, journal.response_revision, journal.response_steer_cursor, + &journal.plan_snapshot_fingerprint, ); let expected_message_id = game_creator_agent_runtime_finalization_message_id( &journal.agent_id, @@ -8065,12 +8446,31 @@ fn sanitize_game_creator_agent_runtime_context_bundle( next_loop_index: bundle.next_loop_index, context_window: bundle.context_window, thinking_summary: redact_agent_runtime_project_paths(root, &bundle.thinking_summary, 240), + plan_revision: bundle.plan_revision, + plan_explanation: redact_agent_runtime_project_paths(root, &bundle.plan_explanation, 240), plan: bundle .plan .iter() .take(AGENT_RUNTIME_PLAN_STEP_LIMIT) .map(|item| redact_agent_runtime_project_paths(root, item, 180)) .collect(), + plan_steps: bundle + .plan_steps + .iter() + .take(AGENT_RUNTIME_PLAN_STEP_LIMIT) + .map(|step| AgentRuntimePlanStep { + index: step.index, + title: redact_agent_runtime_project_paths(root, &step.title, 180), + status: sanitize_agent_runtime_text(&step.status, 80), + detail: step + .detail + .as_deref() + .map(|detail| redact_agent_runtime_project_paths(root, detail, 220)) + .filter(|detail| !detail.trim().is_empty()), + updated_at: step.updated_at, + }) + .collect(), + active_plan_step_index: bundle.active_plan_step_index, fallback_response: redact_agent_runtime_project_paths( root, &bundle.fallback_response, @@ -8167,12 +8567,16 @@ pub(crate) fn build_game_creator_agent_runtime_context_bundle( context_window: u32::try_from(next_loop_index / AGENT_RUNTIME_BACKGROUND_LOOP_LIMIT + 1) .unwrap_or(u32::MAX), thinking_summary: redact_agent_runtime_project_paths(root, &plan.thinking_summary, 240), - plan: plan + plan_revision: runtime.plan_revision, + plan_explanation: runtime.plan_explanation.clone(), + plan: runtime .plan .iter() .take(AGENT_RUNTIME_PLAN_STEP_LIMIT) .map(|item| redact_agent_runtime_project_paths(root, item, 180)) .collect(), + plan_steps: runtime.plan_steps.clone(), + active_plan_step_index: runtime.active_plan_step_index, fallback_response: redact_agent_runtime_project_paths(root, &plan.response, 1_200), observations: compact_agent_runtime_context_observations(root, observations), verification_gate: read_game_creator_agent_runtime_verification_gate( @@ -8307,7 +8711,8 @@ pub(crate) fn read_game_creator_agent_runtime_context_bundle( .get("schemaVersion") .and_then(serde_json::Value::as_str) .unwrap_or_default(); - if schema_version != AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION { + let legacy_schema = schema_version == AGENT_RUNTIME_CONTEXT_BUNDLE_LEGACY_SCHEMA_VERSION; + if schema_version != AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION && !legacy_schema { return Err(format!( "不支持的 Agent Runtime context bundle 版本:{}", if schema_version.is_empty() { @@ -8317,7 +8722,7 @@ pub(crate) fn read_game_creator_agent_runtime_context_bundle( } )); } - let bundle = serde_json::from_value::(raw).map_err(|error| { + let mut bundle = serde_json::from_value::(raw).map_err(|error| { format!( "解析 Agent Runtime context bundle 失败:{}: {error}", path.display() @@ -8325,6 +8730,17 @@ pub(crate) fn read_game_creator_agent_runtime_context_bundle( })?; validate_agent_runtime_pending_serialized_content(root, &content) .map_err(|error| format!("Agent Runtime context bundle 不安全:{error}"))?; + if legacy_schema { + bundle.schema_version = AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION.to_string(); + bundle.plan_revision = runtime.plan_revision; + bundle.plan_explanation = runtime.plan_explanation.clone(); + if runtime.plan_revision > 0 { + bundle.plan = runtime.plan.clone(); + } + bundle.plan_steps = runtime.plan_steps.clone(); + bundle.active_plan_step_index = runtime.active_plan_step_index; + bundle = sanitize_game_creator_agent_runtime_context_bundle(root, &bundle); + } if bundle.project_id != game_creator_agent_runtime_context_project_id(root)? || bundle.agent_id != redact_agent_runtime_project_paths(root, &runtime.agent_id, 160) || bundle.task_id != redact_agent_runtime_project_paths(root, &runtime.task_id, 160) @@ -8407,6 +8823,40 @@ pub(crate) fn read_game_creator_agent_runtime_context_bundle( if bundle.plan.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { return Err("Agent Runtime context bundle 计划步骤超过上限".to_string()); } + if bundle.plan_steps.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { + return Err("Agent Runtime context bundle 结构化计划步骤超过上限".to_string()); + } + if bundle.plan_revision != runtime.plan_revision { + return Err(format!( + "Agent Runtime context bundle 计划 revision 与当前状态不匹配:bundle={} state={}", + bundle.plan_revision, runtime.plan_revision + )); + } + validate_agent_runtime_structured_plan_snapshot( + bundle.plan_revision, + &bundle.plan_explanation, + &bundle.plan, + &bundle.plan_steps, + bundle.active_plan_step_index, + )?; + let expected_plan = sanitize_game_creator_agent_runtime_context_bundle( + root, + &AgentRuntimeContextBundle { + plan_revision: runtime.plan_revision, + plan_explanation: runtime.plan_explanation.clone(), + plan: runtime.plan.clone(), + plan_steps: runtime.plan_steps.clone(), + active_plan_step_index: runtime.active_plan_step_index, + ..bundle.clone() + }, + ); + if bundle.plan_explanation != expected_plan.plan_explanation + || bundle.plan != expected_plan.plan + || bundle.plan_steps != expected_plan.plan_steps + || bundle.active_plan_step_index != expected_plan.active_plan_step_index + { + return Err("Agent Runtime context bundle 结构化计划快照与当前状态不匹配".to_string()); + } if bundle.observations.len() > AGENT_RUNTIME_CONTEXT_OBSERVATION_LIMIT { return Err("Agent Runtime context bundle 观察数量超过上限".to_string()); } @@ -8420,6 +8870,7 @@ pub(crate) fn continuation_from_game_creator_agent_runtime_context_bundle( AgentRuntimeContinuationContext { plan: AgentRuntimeToolPlan { thinking_summary: bundle.thinking_summary, + plan_update: None, plan: bundle.plan, actions: Vec::new(), response: bundle.fallback_response, @@ -8552,6 +9003,7 @@ pub(crate) struct AgentRuntimePendingToolAction { pub(crate) observations: Vec, pub(crate) project_revision_before: AgentRuntimeProjectRevision, pub(crate) verification_gate_before: AgentRuntimeVerificationGate, + pub(crate) planned_repository_context_fingerprint: String, #[serde(default)] pub(crate) planned_steer_cursor: u64, pub(crate) action: AgentRuntimeToolAction, @@ -8586,6 +9038,7 @@ impl AgentRuntimePendingToolAction { fn tool_plan(&self) -> AgentRuntimeToolPlan { AgentRuntimeToolPlan { thinking_summary: self.thinking_summary.clone(), + plan_update: None, plan: self.plan.clone(), actions: Vec::new(), response: self.fallback_response.clone(), @@ -8613,6 +9066,7 @@ fn build_game_creator_agent_runtime_pending_tool_action( plan: &AgentRuntimeToolPlan, observations: &[AgentRuntimeToolObservation], project_revision_before: &AgentRuntimeProjectRevision, + planned_repository_context_fingerprint: &str, action: &AgentRuntimeToolAction, action_index: usize, execution_mode: &str, @@ -8658,6 +9112,7 @@ fn build_game_creator_agent_runtime_pending_tool_action( &runtime.agent_id, &runtime.run_id, )?, + planned_repository_context_fingerprint: planned_repository_context_fingerprint.to_string(), planned_steer_cursor: runtime.applied_steer_cursor, action: action.clone(), action_id, @@ -9253,6 +9708,86 @@ fn agent_runtime_non_verification_completion_blocker_at_locked( .or_else(|| static_delegate_completion_blocker_at_locked(root, agent_id, run_id)) } +pub(crate) fn structured_plan_completion_blocker( + runtime: &AgentRuntimeState, +) -> Option { + if !agent_runtime_has_structured_plan(runtime) { + return None; + } + if let Err(error) = validate_agent_runtime_structured_plan_snapshot( + runtime.plan_revision, + &runtime.plan_explanation, + &runtime.plan, + &runtime.plan_steps, + runtime.active_plan_step_index, + ) { + return Some(AgentRuntimeToolObservation { + tool: "runtime.plan_update".to_string(), + status: "needs-reconciliation".to_string(), + summary: "结构化计划快照无效,不能收束当前任务".to_string(), + detail: Some(format!( + "planRevision={} · snapshotErrorSha256={:x}", + runtime.plan_revision, + Sha256::digest(error.as_bytes()) + )), + }); + } + let incomplete = runtime + .plan_steps + .iter() + .filter(|step| step.status != AGENT_RUNTIME_PLAN_STATUS_COMPLETED) + .collect::>(); + if incomplete.is_empty() { + return None; + } + let completed = runtime + .plan_steps + .iter() + .filter(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_COMPLETED) + .count(); + let pending = incomplete + .iter() + .filter(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_PENDING) + .count(); + let in_progress = incomplete + .iter() + .filter(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS) + .count(); + let failed = incomplete + .iter() + .filter(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_FAILED) + .count(); + let step_status_hashes = incomplete + .iter() + .take(AGENT_RUNTIME_PLAN_STEP_LIMIT) + .map(|step| { + format!( + "{}:{:x}", + step.status, + Sha256::digest(step.title.as_bytes()) + ) + }) + .collect::>() + .join(","); + Some(AgentRuntimeToolObservation { + tool: "runtime.plan_update".to_string(), + status: "blocked".to_string(), + summary: format!( + "结构化计划尚未完成:{completed}/{}", + runtime.plan_steps.len() + ), + detail: Some(format!( + "planRevision={} · completed={} · pending={} · inProgress={} · failed={} · stepStatusSha256={};请按真实进度提交 planUpdate,Runtime 不会按工具数组下标自动完成步骤。", + runtime.plan_revision, + completed, + pending, + in_progress, + failed, + step_status_hashes + )), + }) +} + pub(crate) fn process_session_completion_blocker_at( root: &Path, agent_id: &str, @@ -10549,7 +11084,239 @@ pub(crate) fn agent_runtime_tool_action_input_summary( (!summary.trim().is_empty()).then_some(summary) } +fn agent_runtime_has_structured_plan(runtime: &AgentRuntimeState) -> bool { + runtime.plan_revision > 0 +} + +fn validate_agent_runtime_structured_plan_snapshot( + plan_revision: u64, + plan_explanation: &str, + plan: &[String], + plan_steps: &[AgentRuntimePlanStep], + active_plan_step_index: Option, +) -> Result<(), String> { + if plan_revision == 0 { + if !plan_explanation.trim().is_empty() { + return Err("Agent 旧计划不能携带结构化计划说明".to_string()); + } + return Ok(()); + } + if plan_explanation.trim().is_empty() { + return Err("Agent 结构化计划缺少 explanation".to_string()); + } + if plan_steps.is_empty() || plan_steps.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { + return Err(format!( + "Agent 结构化计划步骤数量必须在 1..={AGENT_RUNTIME_PLAN_STEP_LIMIT}" + )); + } + if plan.len() != plan_steps.len() { + return Err("Agent 结构化计划文本与步骤数量不匹配".to_string()); + } + + let mut seen_steps = std::collections::BTreeSet::new(); + let mut in_progress_index = None; + for (index, step) in plan_steps.iter().enumerate() { + if step.index != index as u32 + || step.title.trim().is_empty() + || plan.get(index) != Some(&step.title) + { + return Err("Agent 结构化计划步骤索引或标题不匹配".to_string()); + } + if !seen_steps.insert(step.title.as_str()) { + return Err("Agent 结构化计划包含重复步骤".to_string()); + } + if !matches!( + step.status.as_str(), + AGENT_RUNTIME_PLAN_STATUS_PENDING + | AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS + | AGENT_RUNTIME_PLAN_STATUS_COMPLETED + | AGENT_RUNTIME_PLAN_STATUS_FAILED + ) { + return Err("Agent 结构化计划持久状态无效".to_string()); + } + if step.status == AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS { + if in_progress_index.replace(step.index).is_some() { + return Err("Agent 结构化计划同时存在多个 in_progress 步骤".to_string()); + } + } + } + if active_plan_step_index != in_progress_index { + return Err("Agent 结构化计划 activePlanStepIndex 与 in_progress 步骤不匹配".to_string()); + } + Ok(()) +} + +pub(crate) fn sanitize_agent_runtime_plan_update( + update: &AgentRuntimePlanUpdate, +) -> Result { + let explanation = sanitize_agent_runtime_text(update.explanation.trim(), 240); + if explanation.is_empty() { + return Err("Agent 结构化计划更新的 explanation 不能为空".to_string()); + } + if update.steps.is_empty() { + return Err("Agent 结构化计划更新必须包含至少一个步骤".to_string()); + } + if update.steps.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { + return Err(format!( + "Agent 结构化计划更新最多包含 {AGENT_RUNTIME_PLAN_STEP_LIMIT} 个步骤" + )); + } + + let mut seen_steps = std::collections::BTreeSet::new(); + let mut in_progress_count = 0usize; + let mut steps = Vec::with_capacity(update.steps.len()); + for raw_step in &update.steps { + let step = sanitize_agent_runtime_text(raw_step.step.trim(), 180); + if step.is_empty() { + return Err("Agent 结构化计划更新不能包含空步骤".to_string()); + } + if !seen_steps.insert(step.clone()) { + return Err("Agent 结构化计划更新包含重复步骤".to_string()); + } + let status = raw_step.status.trim(); + if !matches!( + status, + AGENT_RUNTIME_PLAN_STATUS_PENDING + | AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS + | AGENT_RUNTIME_PLAN_STATUS_COMPLETED + ) { + return Err( + "Agent 结构化计划步骤状态无效,只允许 pending、in_progress、completed".to_string(), + ); + } + if status == AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS { + in_progress_count += 1; + if in_progress_count > 1 { + return Err("Agent 结构化计划同时最多只能有一个 in_progress 步骤".to_string()); + } + } + steps.push(AgentRuntimePlanUpdateStep { + step, + status: status.to_string(), + }); + } + + Ok(AgentRuntimePlanUpdate { explanation, steps }) +} + +pub(crate) fn apply_agent_runtime_plan_update( + runtime: &mut AgentRuntimeState, + update: &AgentRuntimePlanUpdate, +) -> Result { + let update = sanitize_agent_runtime_plan_update(update)?; + let mut terminal_steps = std::collections::BTreeMap::new(); + if agent_runtime_has_structured_plan(runtime) { + for step in &runtime.plan_steps { + if matches!( + step.status.as_str(), + AGENT_RUNTIME_PLAN_STATUS_COMPLETED | AGENT_RUNTIME_PLAN_STATUS_FAILED + ) { + terminal_steps.insert(step.title.clone(), step.clone()); + } + } + } + + for step in &update.steps { + let Some(existing) = terminal_steps.get(&step.step) else { + continue; + }; + if existing.status == AGENT_RUNTIME_PLAN_STATUS_COMPLETED + && step.status != AGENT_RUNTIME_PLAN_STATUS_COMPLETED + { + return Err("Agent 结构化计划不能让已完成步骤回退".to_string()); + } + if existing.status == AGENT_RUNTIME_PLAN_STATUS_FAILED { + return Err("Agent 结构化计划不能改写已失败步骤".to_string()); + } + } + + let incoming_titles = update + .steps + .iter() + .map(|step| step.step.as_str()) + .collect::>(); + let mut merged = runtime + .plan_steps + .iter() + .filter(|step| { + matches!( + step.status.as_str(), + AGENT_RUNTIME_PLAN_STATUS_COMPLETED | AGENT_RUNTIME_PLAN_STATUS_FAILED + ) && !incoming_titles.contains(step.title.as_str()) + }) + .map(|step| (step.title.clone(), step.status.clone())) + .collect::>(); + merged.extend(update.steps.iter().map(|step| { + let status = terminal_steps + .get(&step.step) + .map(|existing| existing.status.clone()) + .unwrap_or_else(|| step.status.clone()); + (step.step.clone(), status) + })); + if merged.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { + return Err(format!( + "Agent 结构化计划保留终态步骤后超过 {AGENT_RUNTIME_PLAN_STEP_LIMIT} 步上限" + )); + } + + let unchanged = agent_runtime_has_structured_plan(runtime) + && runtime.plan_explanation == update.explanation + && runtime.plan_steps.len() == merged.len() + && runtime + .plan_steps + .iter() + .zip(merged.iter()) + .all(|(existing, (title, status))| { + existing.title == *title && existing.status == *status + }); + if unchanged { + return Ok(false); + } + + let now = unix_timestamp(); + let previous_steps = runtime + .plan_steps + .iter() + .map(|step| (step.title.clone(), step.clone())) + .collect::>(); + runtime.plan_steps = merged + .into_iter() + .enumerate() + .map(|(index, (title, status))| { + let previous = previous_steps.get(&title); + let unchanged_status = previous.is_some_and(|step| step.status == status); + AgentRuntimePlanStep { + index: index as u32, + title, + status, + detail: previous.and_then(|step| step.detail.clone()), + updated_at: if unchanged_status { + previous.map(|step| step.updated_at).unwrap_or(now) + } else { + now + }, + } + }) + .collect(); + runtime.plan = runtime + .plan_steps + .iter() + .map(|step| step.title.clone()) + .collect(); + runtime.active_plan_step_index = runtime + .plan_steps + .iter() + .find(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS) + .map(|step| step.index); + runtime.plan_explanation = update.explanation; + runtime.plan_revision = runtime.plan_revision.saturating_add(1).max(1); + Ok(true) +} + fn update_agent_runtime_plan_steps(runtime: &mut AgentRuntimeState, plan: Vec) { + if agent_runtime_has_structured_plan(runtime) { + return; + } runtime.plan = plan .into_iter() .filter(|item| !item.trim().is_empty()) @@ -10575,12 +11342,12 @@ fn update_agent_runtime_plan_steps(runtime: &mut AgentRuntimeState, plan: Vec Result, String> { - let (llm, config_path, mut request) = build_game_creator_agent_background_tool_plan_request( - root, - agent_id, - session_id, - run_id, - task, - observations, - loop_index, - )?; +) -> Result, String> { + let (llm, config_path, mut request, repository_context_fingerprint) = + build_game_creator_agent_background_tool_plan_request( + root, + agent_id, + session_id, + run_id, + task, + observations, + loop_index, + )?; let client = build_game_creator_llm_client_from_llm_config(&llm, &config_path)?; for repair_attempt in 0..=AGENT_RUNTIME_TOOL_PLAN_FORMAT_REPAIR_ATTEMPTS { if game_creator_agent_runtime_cancel_requested_for(root, agent_id, run_id) { @@ -10787,7 +11579,10 @@ async fn request_game_creator_agent_background_tool_plan_at( "responseId": response.response_id, }), )?; - return Ok(Some(parsed.plan)); + return Ok(Some(RequestedAgentRuntimeToolPlan { + plan: parsed.plan, + repository_context_fingerprint, + })); } Err(error) if repair_attempt < AGENT_RUNTIME_TOOL_PLAN_FORMAT_REPAIR_ATTEMPTS => { if game_creator_agent_runtime_cancel_requested_for(root, agent_id, run_id) { @@ -10814,11 +11609,25 @@ async fn request_game_creator_agent_background_tool_plan_at( "loopIteration": loop_index, "attempt": next_attempt, "maxAttempts": AGENT_RUNTIME_TOOL_PLAN_FORMAT_REPAIR_ATTEMPTS, - "protocolError": protocol_error, - "responsePreview": response_preview, + "protocolErrorSha256": format!( + "{:x}", + Sha256::digest(protocol_error.as_bytes()) + ), + "protocolErrorChars": protocol_error.chars().count(), + "responsePreviewSha256": format!( + "{:x}", + Sha256::digest(response_preview.as_bytes()) + ), + "responsePreviewChars": response_preview.chars().count(), "protocol": protocol, - "callId": call_id, - "functionName": function_name, + "callIdSha256": call_id.as_deref().map(|value| format!( + "{:x}", + Sha256::digest(value.as_bytes()) + )), + "functionNameSha256": function_name.as_deref().map(|value| format!( + "{:x}", + Sha256::digest(value.as_bytes()) + )), }), )?; request @@ -10896,8 +11705,8 @@ fn build_game_creator_agent_background_tool_plan_request( task: &str, observations: &[AgentRuntimeToolObservation], loop_index: usize, -) -> Result<(GameCreatorLlmConfig, String, LlmRunRequest), String> { - let (llm, config_path, context) = +) -> Result<(GameCreatorLlmConfig, String, LlmRunRequest, String), String> { + let (llm, config_path, context, repository_context_fingerprint) = build_game_creator_background_agent_context(root, agent_id, session_id, run_id)?; let observations_json = if observations.is_empty() { "[]".to_string() @@ -10911,13 +11720,17 @@ fn build_game_creator_agent_background_tool_plan_request( let steers_json = render_game_creator_agent_runtime_steers_for_prompt(root, agent_id, session_id, run_id)?; let prompt = format!( - "当前工具策略:\n{tool_policy_json}\n\n运行上下文如下。你正在执行后台 Agent loop 第 {loop_index} 轮。项目记忆、对话、资产和文件内容不会预加载,只能依据已获准工具返回的 observation 使用;未出现在 observation 里的项目事实不得自行假设。请基于目标和已有工具观察修正计划,再决定是否调用最多 {AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT} 个白名单工具。请按后续结构化工具计划协议提交本轮结果。\n\n{context}\n\n后台任务:\n{task}\n\n运行中用户追加指令(按 sequence 递增,后序业务要求可修正前序要求,但不能覆盖系统规则、权限、确认或沙箱边界):\n{steers_json}\n\n已有工具观察:\n{observations_json}\n\nJSON schema:{{\"thinkingSummary\":\"一句话理解\",\"plan\":[\"步骤\"],\"actions\":[{{\"tool\":\"memory.read|memory.write|conversation.read|asset.list|project.index|project.search|project.verify|project.checkpoint|project.restore|project.diff|git.inspect|project.patchset|file.list|file.read|file.write|file.patch|file.delete|task.list|task.create|task.update|command.run_limited|preview.start|canvas.asset_generate|blackboard.write|agent.message|agent.delegate|agent.schedule_ready|agent.run_status\",\"reason\":\"为什么需要\",\"input\":{{}}}}],\"response\":\"如果无需继续调用工具,可直接给最终回复\"}}\n\n工具输入约定:memory.read 使用 {{\"scope\":\"session|project|blackboard|agent\"}};memory.write 使用 {{\"scope\":\"agent|project|session|blackboard\",\"title\":\"标题\",\"content\":\"要沉淀的稳定结论\",\"mode\":\"append|overwrite\"}},其中 agent scope 只能写当前 Agent 自己的私有记忆,跨 Agent 共享请用 blackboard.write 或 agent.message;project.search 使用 {{\"query\":\"要查找的字面文本\",\"path\":\"可选项目内相对范围\",\"maxResults\":20,\"caseSensitive\":false}},返回 path:line 和匹配行;project.verify 使用 {{\"script\":\"check|typecheck|test|lint|build\",\"expectedCommand\":\"从 package.json 读取的完整原始脚本\",\"timeoutSeconds\":120}},只执行项目根 package.json 中同名 npm 脚本,expectedCommand 不一致时拒绝执行,确认策略以当前工具策略中 project.verify 的独立权限为准;project.checkpoint input 可为空,用于在写文件或批量修改前创建本地 checkpoint;project.restore 使用 {{\"checkpointId\":\"checkpoint id\"}},用于在确认后把当前项目恢复到指定 checkpoint;project.diff 使用 {{\"checkpointId\":\"checkpoint id\",\"includeContent\":true,\"maxFiles\":20,\"maxChars\":24000}},用于读取路径摘要或有界统一 diff hunks;git.inspect 使用 {{\"includeDiff\":true,\"maxFiles\":20,\"maxChars\":24000}},只读当前项目根的 Git staged / unstaged / untracked 安全路径和有界 staged / unstaged diff,不推进 revision;不得用它提交、暂存、切分支、合并、重置、stash、worktree 或访问 remote;project.patchset 使用 {{\"changes\":[{{\"operation\":\"create|update|delete\",\"path\":\"项目内相对文件\",\"content\":\"create 内容\",\"expectedSha256\":\"update/delete 必填\",\"oldText\":\"update 必填\",\"newText\":\"update 必填\",\"expectedReplacements\":1}}]}},会自动 checkpoint 并在一把锁内应用多文件变更,成功后必须用返回的 checkpointId 调用 project.diff includeContent=true 审查整体变更;file.list 使用 {{\"path\":\"可选项目内相对目录或文件\"}},path 为空时列出项目摘要;file.read 使用 {{\"path\":\"项目内相对路径\",\"startLine\":1,\"maxLines\":120}},按行读取并返回行号和完整内容 SHA-256;file.write 使用 {{\"path\":\"项目内相对路径\",\"content\":\"完整文件内容\"}};file.patch 使用 {{\"path\":\"项目内相对路径\",\"oldText\":\"必须精确匹配的原文\",\"newText\":\"替换后的文本\",\"expectedReplacements\":1}},匹配数不符时不写入;file.delete 使用 {{\"path\":\"项目内相对路径\"}},只删除项目内普通文件,不删除目录或任何 .agent 控制面文件;task.list input 可为空,用于读取 manifest 任务图、状态和 readyTaskIds;task.create 使用 {{\"taskId\":\"可选自定义 taskId\",\"title\":\"任务标题\",\"group\":\"design|art|code|balance|audio|publishing\",\"role\":\"角色名\",\"dependencies\":[\"已有 taskId\"],\"artifacts\":[\"预期产物\"],\"acceptanceCriteria\":[\"验收标准\"],\"status\":\"pending|running|waiting-for-confirmation|completed|failed\"}},用于把 Agent 拆出的新任务追加到 manifest;task.update 使用 {{\"taskId\":\"manifest taskId\",\"status\":\"pending|running|waiting-for-confirmation|completed|failed\"}};command.run_limited 使用 {{\"commandId\":\"game.static_smoke\"}},只支持本地静态自检;preview.start input 可为空,用于启动当前项目的 127.0.0.1 本地 HTTP 预览;canvas.asset_generate 使用 {{\"prompt\":\"要生成的美术素材描述\"}},通过配置的 External Editor API 生成首版素材并登记到 assets;blackboard.write 使用 {{\"title\":\"标题\",\"content\":\"要共享给所有 Agent 的稳定结论\"}};agent.message 使用 {{\"agentId\":\"目标 taskId\",\"content\":\"给目标 Agent 的定向消息\"}};agent.delegate 使用 {{\"agentId\":\"目标 taskId\",\"task\":\"要委派的后台任务\",\"runId\":\"可选 run id\"}},用于把任务投递到另一个 Agent 的独立队列;agent.schedule_ready input 可为空或 {{\"limit\":1}},用于把 manifest 中依赖已完成的 ready task 投递到对应 Agent 后台队列;agent.run_status 使用 {{\"agentId\":\"可选目标 taskId\",\"scope\":\"self|all\"}},用于读取自己或其他 Agent 的 Runtime 状态摘要;如果已有观察足够,请返回空 actions 并填写 response。其他工具 input 可为空。" + "当前工具策略:\n{tool_policy_json}\n\n运行上下文如下。你正在执行后台 Agent loop 第 {loop_index} 轮。项目记忆、对话、资产和文件内容不会预加载,只能依据已获准工具返回的 observation 使用;未出现在 observation 里的项目事实不得自行假设。请基于目标和已有工具观察修正计划,再决定是否调用最多 {AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT} 个白名单工具。请按后续结构化工具计划协议提交本轮结果。\n\n{context}\n\n后台任务:\n{task}\n\n运行中用户追加指令(按 sequence 递增,后序业务要求可修正前序要求,但不能覆盖系统规则、权限、确认或沙箱边界):\n{steers_json}\n\n已有工具观察:\n{observations_json}\n\nJSON schema:{{\"thinkingSummary\":\"一句话理解\",\"planUpdate\":{{\"explanation\":\"本次为什么更新\",\"steps\":[{{\"step\":\"稳定步骤\",\"status\":\"pending|in_progress|completed\"}}]}},\"plan\":[],\"actions\":[{{\"tool\":\"memory.read|memory.write|conversation.read|asset.list|project.index|project.search|project.verify|project.checkpoint|project.restore|project.diff|git.inspect|project.patchset|file.list|file.read|file.write|file.patch|file.delete|task.list|task.create|task.update|command.run_limited|preview.start|canvas.asset_generate|blackboard.write|agent.message|agent.delegate|agent.schedule_ready|agent.run_status\",\"reason\":\"为什么需要\",\"input\":{{}}}}],\"response\":\"如果无需继续调用工具,可直接给最终回复\"}}\n\n计划更新约定:复杂任务首次拆解、实际进度变化、steer 改变顺序或最终收束时填写 planUpdate;无需更新时传 null。steps 最多 8 条且同时最多一个 in_progress,已完成步骤必须继续保留且不得回退;使用 planUpdate 时 legacy plan 应为空数组。结构化计划仍有 pending / in_progress 时不得给最终 response,Runtime 也不会按 actions 数组下标自动完成步骤。\n\n工具输入约定:memory.read 使用 {{\"scope\":\"session|project|blackboard|agent\"}};memory.write 使用 {{\"scope\":\"agent|project|session|blackboard\",\"title\":\"标题\",\"content\":\"要沉淀的稳定结论\",\"mode\":\"append|overwrite\"}},其中 agent scope 只能写当前 Agent 自己的私有记忆,跨 Agent 共享请用 blackboard.write 或 agent.message;project.search 使用 {{\"query\":\"要查找的字面文本\",\"path\":\"可选项目内相对范围\",\"maxResults\":20,\"caseSensitive\":false}},返回 path:line 和匹配行;project.verify 使用 {{\"script\":\"check|typecheck|test|lint|build\",\"expectedCommand\":\"从 package.json 读取的完整原始脚本\",\"timeoutSeconds\":120}},只执行项目根 package.json 中同名 npm 脚本,expectedCommand 不一致时拒绝执行,确认策略以当前工具策略中 project.verify 的独立权限为准;project.checkpoint input 可为空,用于在写文件或批量修改前创建本地 checkpoint;project.restore 使用 {{\"checkpointId\":\"checkpoint id\"}},用于在确认后把当前项目恢复到指定 checkpoint;project.diff 使用 {{\"checkpointId\":\"checkpoint id\",\"includeContent\":true,\"maxFiles\":20,\"maxChars\":24000}},用于读取路径摘要或有界统一 diff hunks;git.inspect 使用 {{\"includeDiff\":true,\"maxFiles\":20,\"maxChars\":24000}},只读当前项目根的 Git staged / unstaged / untracked 安全路径和有界 staged / unstaged diff,不推进 revision;不得用它提交、暂存、切分支、合并、重置、stash、worktree 或访问 remote;project.patchset 使用 {{\"changes\":[{{\"operation\":\"create|update|delete\",\"path\":\"项目内相对文件\",\"content\":\"create 内容\",\"expectedSha256\":\"update/delete 必填\",\"oldText\":\"update 必填\",\"newText\":\"update 必填\",\"expectedReplacements\":1}}]}},会自动 checkpoint 并在一把锁内应用多文件变更,成功后必须用返回的 checkpointId 调用 project.diff includeContent=true 审查整体变更;file.list 使用 {{\"path\":\"可选项目内相对目录或文件\"}},path 为空时列出项目摘要;file.read 使用 {{\"path\":\"项目内相对路径\",\"startLine\":1,\"maxLines\":120}},按行读取并返回行号和完整内容 SHA-256;file.write 使用 {{\"path\":\"项目内相对路径\",\"content\":\"完整文件内容\"}};file.patch 使用 {{\"path\":\"项目内相对路径\",\"oldText\":\"必须精确匹配的原文\",\"newText\":\"替换后的文本\",\"expectedReplacements\":1}},匹配数不符时不写入;file.delete 使用 {{\"path\":\"项目内相对路径\"}},只删除项目内普通文件,不删除目录或任何 .agent 控制面文件;task.list input 可为空,用于读取 manifest 任务图、状态和 readyTaskIds;task.create 使用 {{\"taskId\":\"可选自定义 taskId\",\"title\":\"任务标题\",\"group\":\"design|art|code|balance|audio|publishing\",\"role\":\"角色名\",\"dependencies\":[\"已有 taskId\"],\"artifacts\":[\"预期产物\"],\"acceptanceCriteria\":[\"验收标准\"],\"status\":\"pending|running|waiting-for-confirmation|completed|failed\"}},用于把 Agent 拆出的新任务追加到 manifest;task.update 使用 {{\"taskId\":\"manifest taskId\",\"status\":\"pending|running|waiting-for-confirmation|completed|failed\"}};command.run_limited 使用 {{\"commandId\":\"game.static_smoke\"}},只支持本地静态自检;preview.start input 可为空,用于启动当前项目的 127.0.0.1 本地 HTTP 预览;canvas.asset_generate 使用 {{\"prompt\":\"要生成的美术素材描述\"}},通过配置的 External Editor API 生成首版素材并登记到 assets;blackboard.write 使用 {{\"title\":\"标题\",\"content\":\"要共享给所有 Agent 的稳定结论\"}};agent.message 使用 {{\"agentId\":\"目标 taskId\",\"content\":\"给目标 Agent 的定向消息\"}};agent.delegate 使用 {{\"agentId\":\"目标 taskId\",\"task\":\"要委派的后台任务\",\"runId\":\"可选 run id\"}},用于把任务投递到另一个 Agent 的独立队列;agent.schedule_ready input 可为空或 {{\"limit\":1}},用于把 manifest 中依赖已完成的 ready task 投递到对应 Agent 后台队列;agent.run_status 使用 {{\"agentId\":\"可选目标 taskId\",\"scope\":\"self|all\"}},用于读取自己或其他 Agent 的 Runtime 状态摘要;如果已有观察足够,请返回空 actions 并填写 response。其他工具 input 可为空。" ); let prompt = prompt .replace( "项目记忆、对话、资产和文件内容不会预加载", "除下方有界仓库启动上下文外,项目记忆、对话、资产和源码正文不会预加载", ) + .replace( + "project.checkpoint input 可为空,用于在写文件或批量修改前创建本地 checkpoint", + "project.checkpoint input 可为空,只用于多个 file.* 写动作前或需要独立回退点时创建本地 checkpoint;project.patchset 会自动创建 checkpoint,不要为同一批变更额外调用 project.checkpoint", + ) .replace( "preview.start|canvas.asset_generate", "preview.start|preview.validate|canvas.asset_generate", @@ -11002,16 +11815,16 @@ fn build_game_creator_agent_background_tool_plan_request( .with_tool_choice(platform_llm::LlmToolChoice::Required); } request = apply_game_creator_llm_reasoning_effort(request, &llm)?; - Ok((llm, config_path, request)) + Ok((llm, config_path, request, repository_context_fingerprint)) } fn game_creator_agent_tool_plan_function_tool() -> platform_llm::LlmFunctionTool { platform_llm::LlmFunctionTool::new( AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME, - "提交本轮 Agent 的任务理解、短计划、白名单工具动作或最终回复。Runtime 只执行 arguments 中经过本地策略校验的动作。", + "提交本轮 Agent 的任务理解、可选持久计划更新、白名单工具动作或最终回复。Runtime 只执行 arguments 中经过本地策略校验的动作。", serde_json::json!({ "type": "object", - "required": ["thinkingSummary", "plan", "actions", "response"], + "required": ["thinkingSummary", "planUpdate", "plan", "actions", "response"], "additionalProperties": false, "properties": { "thinkingSummary": { @@ -11019,6 +11832,35 @@ fn game_creator_agent_tool_plan_function_tool() -> platform_llm::LlmFunctionTool "minLength": 1, "description": "一句话概括当前任务理解和决策依据" }, + "planUpdate": { + "type": ["object", "null"], + "description": "复杂任务的持久计划更新;没有真实进度变化时传 null", + "required": ["explanation", "steps"], + "additionalProperties": false, + "properties": { + "explanation": { + "type": "string", + "minLength": 1 + }, + "steps": { + "type": "array", + "minItems": 1, + "maxItems": AGENT_RUNTIME_PLAN_STEP_LIMIT, + "items": { + "type": "object", + "required": ["step", "status"], + "additionalProperties": false, + "properties": { + "step": { "type": "string", "minLength": 1 }, + "status": { + "type": "string", + "enum": ["pending", "in_progress", "completed"] + } + } + } + } + } + }, "plan": { "type": "array", "maxItems": AGENT_RUNTIME_PLAN_STEP_LIMIT, @@ -11065,7 +11907,7 @@ fn build_game_creator_agent_background_final_reply_request( plan: &AgentRuntimeToolPlan, observations: &[AgentRuntimeToolObservation], ) -> Result<(GameCreatorLlmConfig, String, LlmRunRequest), String> { - let (llm, config_path, context) = + let (llm, config_path, context, _repository_context_fingerprint) = build_game_creator_background_agent_context(root, agent_id, session_id, run_id)?; let observations_json = serde_json::to_string_pretty(observations) .map_err(|error| format!("序列化 Agent 工具观察失败:{error}"))?; @@ -11104,7 +11946,7 @@ fn build_game_creator_background_agent_context( agent_id: &str, session_id: &str, run_id: &str, -) -> Result<(GameCreatorLlmConfig, String, String), String> { +) -> Result<(GameCreatorLlmConfig, String, String, String), String> { let agent_id = normalize_game_creator_runtime_agent_id(agent_id)?; validate_project_root(root)?; let session_id = @@ -11176,10 +12018,22 @@ fn build_game_creator_background_agent_context( role_definition.id, sanitize_agent_runtime_text(run_id, 160) ); - let context = format!("{context}{supervisor_context}"); + let runtime_context = + render_agent_runtime_prompt_context_for_session(root, &agent_id, Some(&session_id), false)?; + let runtime_context = if runtime_context.trim().is_empty() { + String::new() + } else { + format!("\n\n# 持久 Runtime 状态\n\n{runtime_context}") + }; + let context = format!("{context}{runtime_context}{supervisor_context}"); let app_config = load_game_creator_app_config()?; let llm = resolve_game_creator_llm_config_for_agent(&app_config, &template_agent_id); - Ok((llm, format!("agentLlm.{template_agent_id}"), context)) + Ok(( + llm, + format!("agentLlm.{template_agent_id}"), + context, + repository_context.fingerprint, + )) } pub(crate) fn parse_game_creator_agent_tool_plan_response( @@ -11188,7 +12042,7 @@ pub(crate) fn parse_game_creator_agent_tool_plan_response( let stripped = strip_llm_thinking_blocks(content); let payload = extract_json_payload(stripped.as_str()) .ok_or_else(|| "Agent 工具计划协议错误:未返回完整 JSON 对象".to_string())?; - parse_game_creator_agent_tool_plan_payload(payload) + parse_game_creator_agent_tool_plan_payload(payload, false) } pub(crate) fn parse_game_creator_agent_tool_plan_llm_response( @@ -11213,11 +12067,10 @@ pub(crate) fn parse_game_creator_agent_tool_plan_llm_response( let call = &response.tool_calls[0]; if call.name != AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME { return Err(format!( - "Agent 工具计划协议错误:期望调用 {AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME},实际调用 {}", - call.name + "Agent 工具计划协议错误:必须调用 {AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME},实际调用了非预期函数" )); } - let plan = parse_game_creator_agent_tool_plan_payload(call.arguments.as_str()) + let plan = parse_game_creator_agent_tool_plan_payload(call.arguments.as_str(), true) .map_err(|error| format!("{error};function arguments 解析失败"))?; Ok(ParsedAgentRuntimeToolPlan { plan, @@ -11241,7 +12094,21 @@ fn game_creator_agent_tool_plan_response_preview( fn parse_game_creator_agent_tool_plan_payload( payload: &str, + require_plan_update_field: bool, ) -> Result { + if require_plan_update_field { + let value = serde_json::from_str::(payload) + .map_err(|error| format!("解析 Agent 工具计划失败:{error}"))?; + if !value + .as_object() + .is_some_and(|object| object.contains_key("planUpdate")) + { + return Err( + "Agent 工具计划协议错误:native function arguments 必须显式包含 planUpdate" + .to_string(), + ); + } + } let mut plan = serde_json::from_str::(payload) .map_err(|error| format!("解析 Agent 工具计划失败:{error}"))?; if plan.thinking_summary.trim().is_empty() { @@ -11255,6 +12122,11 @@ fn parse_game_creator_agent_tool_plan_payload( return Err("Agent 工具计划协议错误:action.tool 不能为空".to_string()); } plan.thinking_summary = truncate_agent_runtime_text(&plan.thinking_summary, 240); + plan.plan_update = plan + .plan_update + .as_ref() + .map(sanitize_agent_runtime_plan_update) + .transpose()?; plan.plan = plan .plan .into_iter() @@ -11756,7 +12628,7 @@ fn validate_agent_runtime_project_snapshot_action_after_lock( ) { return Err(agent_runtime_tool_policy_block_observation(tool, blocked)); } - match pending_repository_context_drift_observation(root, &runtime, pending) { + match pending_repository_context_drift_observation(root, pending) { Ok(Some(observation)) => return Err(observation), Ok(None) => {} Err(error) => { @@ -12084,6 +12956,16 @@ fn validate_agent_runtime_pending_tool_action_record( &pending.agent_id, &pending.run_id, )?; + if pending.planned_repository_context_fingerprint.len() != 64 + || !pending + .planned_repository_context_fingerprint + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err( + "Agent Runtime 待确认动作缺少有效的 planning repository fingerprint".to_string(), + ); + } if !matches!( pending.status.as_str(), AGENT_RUNTIME_PENDING_ACTION_STATUS_PENDING @@ -13215,8 +14097,8 @@ where { return agent_runtime_mutation_gate_failure_observation(root, tool, &error); } - let runtime = match read_game_creator_agent_runtime_at(root, agent_id) { - Ok(runtime) if runtime.state.run_id == run_id => runtime.state, + match read_game_creator_agent_runtime_at(root, agent_id) { + Ok(runtime) if runtime.state.run_id == run_id => {} Ok(_) => { return agent_runtime_mutation_gate_failure_observation( root, @@ -13227,8 +14109,8 @@ where Err(error) => { return agent_runtime_mutation_gate_failure_observation(root, tool, &error); } - }; - match pending_repository_context_drift_observation(root, &runtime, pending_action) { + } + match pending_repository_context_drift_observation(root, pending_action) { Ok(Some(observation)) => return observation, Ok(None) => {} Err(error) => { @@ -20274,6 +21156,18 @@ fn start_game_creator_agent_runtime_task_for_session_in_session_lane_at( state.loop_iteration = previous_state.loop_iteration; state.max_loop_iterations = previous_state.max_loop_iterations; state.tool_action_budget = previous_state.tool_action_budget; + validate_agent_runtime_structured_plan_snapshot( + previous_state.plan_revision, + &previous_state.plan_explanation, + &previous_state.plan, + &previous_state.plan_steps, + previous_state.active_plan_step_index, + )?; + state.plan_revision = previous_state.plan_revision; + state.plan_explanation = previous_state.plan_explanation; + state.plan = previous_state.plan; + state.plan_steps = previous_state.plan_steps; + state.active_plan_step_index = previous_state.active_plan_step_index; } state.recent_tool_calls = previous_state.recent_tool_calls; state.last_response = previous_state.last_response; @@ -20901,7 +21795,9 @@ where )); } let current_revision = read_game_creator_agent_runtime_project_revision(root)?; - let blocker = if let Some(blocker) = agent_runtime_non_verification_completion_blocker_at_locked( + let blocker = if let Some(blocker) = structured_plan_completion_blocker(&state) { + Some(blocker) + } else if let Some(blocker) = agent_runtime_non_verification_completion_blocker_at_locked( root, &state.agent_id, &state.run_id, @@ -21184,7 +22080,10 @@ fn unique_game_creator_agent_runtime_run_id( Err(format!("无法生成唯一 Agent Runtime runId:{base_run_id}")) } -fn default_game_creator_agent_runtime_state(agent_id: &str, run_id: &str) -> AgentRuntimeState { +pub(crate) fn default_game_creator_agent_runtime_state( + agent_id: &str, + run_id: &str, +) -> AgentRuntimeState { AgentRuntimeState { schema_version: AGENT_RUNTIME_SCHEMA_VERSION.to_string(), agent_id: agent_id.to_string(), @@ -21205,6 +22104,8 @@ fn default_game_creator_agent_runtime_state(agent_id: &str, run_id: &str) -> Age loop_iteration: 0, max_loop_iterations: AGENT_RUNTIME_BACKGROUND_LOOP_LIMIT as u32, tool_action_budget: AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT as u32, + plan_revision: 0, + plan_explanation: String::new(), plan: vec![ "读取项目上下文".to_string(), "按角色职责推理".to_string(), @@ -21294,14 +22195,22 @@ fn normalize_game_creator_agent_runtime_state(state: &mut AgentRuntimeState, age if state.tool_action_budget == 0 { state.tool_action_budget = AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT as u32; } - if state.plan.is_empty() { + state.plan_explanation = if agent_runtime_has_structured_plan(state) { + sanitize_agent_runtime_text(&state.plan_explanation, 240) + } else { + String::new() + }; + if state.plan.is_empty() && !agent_runtime_has_structured_plan(state) { state.plan = vec![ "读取项目上下文".to_string(), "按角色职责推理".to_string(), "回复并记录 runtime 事件".to_string(), ]; } - if state.plan_steps.is_empty() && !state.plan.is_empty() { + if state.plan_steps.is_empty() + && !state.plan.is_empty() + && !agent_runtime_has_structured_plan(state) + { state.plan_steps = state .plan .iter() @@ -21324,17 +22233,34 @@ fn normalize_game_creator_agent_runtime_state(state: &mut AgentRuntimeState, age }) .collect(); } - if state.plan_steps.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT { + if !agent_runtime_has_structured_plan(state) + && state.plan_steps.len() > AGENT_RUNTIME_PLAN_STEP_LIMIT + { state.plan_steps.truncate(AGENT_RUNTIME_PLAN_STEP_LIMIT); } - if let Some(active_index) = state.active_plan_step_index { - if !state - .plan_steps - .iter() - .any(|step| step.index == active_index) - { - state.active_plan_step_index = None; + if !agent_runtime_has_structured_plan(state) { + if let Some(active_index) = state.active_plan_step_index { + if !state + .plan_steps + .iter() + .any(|step| step.index == active_index) + { + state.active_plan_step_index = None; + } } + } else if let Err(error) = validate_agent_runtime_structured_plan_snapshot( + state.plan_revision, + &state.plan_explanation, + &state.plan, + &state.plan_steps, + state.active_plan_step_index, + ) { + state.status = "needs-reconciliation".to_string(); + state.phase = "needs-reconciliation".to_string(); + state.current_action = "结构化计划状态损坏".to_string(); + state.waiting_on = "人工核对当前 run 的计划快照".to_string(); + state.next_step = "修复计划快照后恢复当前 run".to_string(); + state.error = Some(sanitize_agent_runtime_text(&error, 500)); } if state.allowed_tools.is_empty() { state.allowed_tools = default_game_creator_agent_runtime_allowed_tools(); @@ -23064,13 +23990,14 @@ fn render_agent_runtime_tool_names(values: &[String], limit: usize) -> String { } fn render_agent_runtime_prompt_context(root: &Path, agent_id: &str) -> Result { - render_agent_runtime_prompt_context_for_session(root, agent_id, None) + render_agent_runtime_prompt_context_for_session(root, agent_id, None, true) } fn render_agent_runtime_prompt_context_for_session( root: &Path, agent_id: &str, session_id: Option<&str>, + include_cross_run_continuity: bool, ) -> Result { let runtime_result = match session_id { Some(session_id) => { @@ -23149,15 +24076,15 @@ fn render_agent_runtime_prompt_context_for_session( state.loop_iteration, state.max_loop_iterations, state.tool_action_budget )); } - if runtime.task_queue.total > 0 { + if include_cross_run_continuity && runtime.task_queue.total > 0 { lines.push(format!( "任务队列:{}", format_agent_runtime_task_queue_observation(&runtime.task_queue) )); } - if let Some(last_response) = state - .last_response - .as_deref() + if let Some(last_response) = include_cross_run_continuity + .then_some(state.last_response.as_deref()) + .flatten() .filter(|value| !value.trim().is_empty()) { lines.push(format!( @@ -23190,6 +24117,13 @@ fn render_agent_runtime_prompt_context_for_session( } } if !state.plan_steps.is_empty() { + if state.plan_revision > 0 { + lines.push(format!( + "结构化计划:revision={};说明={}", + state.plan_revision, + redact_agent_runtime_project_paths(root, &state.plan_explanation, 240) + )); + } lines.push("计划进度:".to_string()); for step in state.plan_steps.iter().take(AGENT_RUNTIME_PLAN_STEP_LIMIT) { let mut line = format!( @@ -23212,13 +24146,17 @@ fn render_agent_runtime_prompt_context_for_session( } } - let recent_observations = state - .observations - .iter() - .filter(|item| !item.trim().is_empty()) - .rev() - .take(3) - .collect::>(); + let recent_observations = if include_cross_run_continuity { + state + .observations + .iter() + .filter(|item| !item.trim().is_empty()) + .rev() + .take(3) + .collect::>() + } else { + Vec::new() + }; if !recent_observations.is_empty() { lines.push("最近观察:".to_string()); for item in recent_observations.iter().rev() { @@ -23229,12 +24167,16 @@ fn render_agent_runtime_prompt_context_for_session( } } - let recent_tool_calls = state - .recent_tool_calls - .iter() - .rev() - .take(3) - .collect::>(); + let recent_tool_calls = if include_cross_run_continuity { + state + .recent_tool_calls + .iter() + .rev() + .take(3) + .collect::>() + } else { + Vec::new() + }; if !recent_tool_calls.is_empty() { lines.push("最近工具动作:".to_string()); for call in recent_tool_calls.iter().rev() { @@ -23278,13 +24220,17 @@ fn render_agent_runtime_prompt_context_for_session( } } - let recent_events = runtime - .recent_events - .iter() - .filter(|event| !event.summary.trim().is_empty()) - .rev() - .take(4) - .collect::>(); + let recent_events = if include_cross_run_continuity { + runtime + .recent_events + .iter() + .filter(|event| !event.summary.trim().is_empty()) + .rev() + .take(4) + .collect::>() + } else { + Vec::new() + }; if !recent_events.is_empty() { lines.push("最近事件:".to_string()); for event in recent_events.iter().rev() { @@ -23309,12 +24255,16 @@ fn render_agent_runtime_prompt_context_for_session( } } - let recent_tasks = runtime - .recent_tasks - .iter() - .rev() - .take(3) - .collect::>(); + let recent_tasks = if include_cross_run_continuity { + runtime + .recent_tasks + .iter() + .rev() + .take(3) + .collect::>() + } else { + Vec::new() + }; if !recent_tasks.is_empty() { lines.push("最近任务:".to_string()); for task in recent_tasks.iter().rev() { @@ -23422,7 +24372,7 @@ fn build_game_creator_role_agent_context_for_session( let project_blackboard = read_optional_text(&root.join(PROJECT_BLACKBOARD_MEMORY_PATH))?; let agent_memory = read_local_agent_memory_at(root, &agent_id)?.content; let runtime_context = - render_agent_runtime_prompt_context_for_session(root, &agent_id, Some(&session_id))?; + render_agent_runtime_prompt_context_for_session(root, &agent_id, Some(&session_id), true)?; let asset_context = render_local_asset_prompt_context(root)?; let conversation_context = render_local_conversation_prompt_context_for_session( root, @@ -23494,6 +24444,10 @@ pub(crate) fn game_creator_agent_runtime_tool_plan_system_prompt_for_agent( pub(crate) fn game_creator_agent_runtime_tool_plan_system_prompt() -> String { let prompt = "你是 Genarrative AI 游戏创作多智能体 Runtime 中的专业 Agent。你必须在白名单工具内规划行动:先给一句 thinkingSummary,再给短计划,再决定是否请求工具。只能请求 memory.read、memory.write、conversation.read、asset.list、project.index、project.search、project.verify、project.checkpoint、project.restore、project.diff、file.list、file.read、file.write、file.patch、file.delete、task.list、task.create、task.update、command.run_limited、preview.start、canvas.asset_generate、blackboard.write、agent.message、agent.delegate、agent.schedule_ready、agent.run_status。处理代码任务时先用 project.search 定位,再用带行号的 file.read 获取足够上下文;优先使用 file.patch 做精确局部修改,只有确认文件已废弃时才请求 file.delete,批量修改前创建 project.checkpoint,修改后再次读取验证。每次成功执行 file.write、file.patch、file.delete 或 project.restore 都会产生新的项目 revision;最后一次修改后必须成功执行 project.verify,或成功执行 command.run_limited 的 game.static_smoke,才能返回空 actions 收束。文件回读不能替代可执行验证,验证后再次修改必须重新验证。需要执行 package.json 中的验证脚本时,先读取 package.json,再把真实脚本名和读到的完整命令原样提交给 project.verify;script 可以是 check、typecheck、test、lint、build,或使用 check:、test:(例如 test:unit)、lint:、typecheck:、build:、verify:、validate: 形式的命名脚本,其中冒号后的每个非空段必须以字母或数字开头且只能包含字母、数字、连字符、下划线或点;不得猜测或改写 expectedCommand。每 6 轮只是一个上下文压缩窗口,不是 run 的终止上限;只要 observation 出现新的独立进展,就在同一 run 继续下一窗口,只有窗口没有新进展时才按停滞处理。Agent 私有记忆只能由本人写入,跨 Agent 共享稳定结论用 blackboard.write,给单个 Agent 留上下文用 agent.message。不要假装工具已执行;工具结果会由 Runtime 作为 observation 返回。优先调用 submit_agent_tool_plan function tool 提交结构化计划;只有上游不支持 function tool 时才返回同结构的单个 JSON 对象。不要 markdown,不要泄露密钥。" + .replace( + "先给一句 thinkingSummary,再给短计划,再决定是否请求工具", + "先给一句 thinkingSummary;复杂任务首次拆解、实际进度变化、steer 调整顺序或最终收束时提交 planUpdate,再决定是否请求工具。planUpdate 只允许 pending、in_progress、completed 且同时最多一个 in_progress;无需更新时传 null,使用时 legacy plan 传空数组;已完成步骤必须保留且不得回退,所有必要步骤 completed 前不得给最终回复,Runtime 不会按工具动作下标代替你更新进度", + ) .replace( "project.diff、file.list", "project.diff、git.inspect、project.git_commit、project.patchset、file.list", @@ -23528,7 +24482,7 @@ pub(crate) fn game_creator_agent_runtime_tool_plan_system_prompt() -> String { ) .replace( "优先使用 file.patch 做精确局部修改,只有确认文件已废弃时才请求 file.delete,批量修改前创建 project.checkpoint,修改后再次读取验证", - "单文件小改优先使用 file.patch;涉及多个文件时优先使用 project.patchset,并在成功后用返回的 checkpointId 调用 project.diff(includeContent=true) 审查整体变更;只有确认文件已废弃时才删除", + "单文件小改优先使用 file.patch;涉及多个文件时优先使用 project.patchset,它会自动创建 checkpoint,无需额外调用 project.checkpoint,并在成功后用返回的 checkpointId 调用 project.diff(includeContent=true) 审查整体变更;只有确认文件已废弃时才删除", ) .replace( "每次成功执行 file.write、file.patch、file.delete 或 project.restore 都会产生新的项目 revision;最后一次修改后必须成功执行 project.verify,或成功执行 command.run_limited 的 game.static_smoke", diff --git a/apps/ai-game-creator-shell/src-tauri/src/cli.rs b/apps/ai-game-creator-shell/src-tauri/src/cli.rs index 8d9a1f0a2..4cf90aad0 100644 --- a/apps/ai-game-creator-shell/src-tauri/src/cli.rs +++ b/apps/ai-game-creator-shell/src-tauri/src/cli.rs @@ -446,6 +446,32 @@ pub(crate) fn parse_cli_command(args: &[String]) -> Result, S })) } +fn strip_agent_runtime_cli_private_paths(value: &mut serde_json::Value) { + match value { + serde_json::Value::Array(values) => { + for value in values { + strip_agent_runtime_cli_private_paths(value); + } + } + serde_json::Value::Object(object) => { + object.remove("sessionPath"); + object.remove("eventPath"); + object.remove("taskPath"); + for value in object.values_mut() { + strip_agent_runtime_cli_private_paths(value); + } + } + _ => {} + } +} + +fn serialize_agent_runtime_cli_payload(payload: &T) -> Result { + let mut value = serde_json::to_value(payload) + .map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}"))?; + strip_agent_runtime_cli_private_paths(&mut value); + serde_json::to_string(&value).map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}")) +} + pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { match command { CliCommand::LlmStatus => { @@ -595,8 +621,7 @@ pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { println!("requestedRunId={run_id}"); println!( "runtimeJson={}", - serde_json::to_string(&runtime) - .map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}"))? + serialize_agent_runtime_cli_payload(&runtime)? ); Ok(()) } @@ -607,8 +632,7 @@ pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { let runtime = read_game_creator_agent_runtime_at(&project_path, &agent_id)?; println!( "runtimeJson={}", - serde_json::to_string(&runtime) - .map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}"))? + serialize_agent_runtime_cli_payload(&runtime)? ); Ok(()) } @@ -630,8 +654,7 @@ pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { println!("agent.confirm.accepted"); println!( "runtimeJson={}", - serde_json::to_string(&runtime) - .map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}"))? + serialize_agent_runtime_cli_payload(&runtime)? ); Ok(()) } @@ -665,8 +688,7 @@ pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { println!("steerId={steer_id}"); println!( "steerJson={}", - serde_json::to_string(&result) - .map_err(|error| format!("序列化 Agent steer 结果失败:{error}"))? + serialize_agent_runtime_cli_payload(&result)? ); Ok(()) } @@ -677,8 +699,7 @@ pub(crate) fn run_cli_command(command: CliCommand) -> Result<(), String> { println!("agent.resume.accepted"); println!( "runtimesJson={}", - serde_json::to_string(&runtimes) - .map_err(|error| format!("序列化 Agent Runtime 状态失败:{error}"))? + serialize_agent_runtime_cli_payload(&runtimes)? ); Ok(()) } @@ -763,6 +784,35 @@ mod tests { use super::*; use std::io::Cursor; + #[test] + fn runtime_cli_payload_omits_private_storage_paths_recursively() { + let payload = serde_json::json!({ + "sessionPath": "/tmp/private/session.json", + "runtime": { + "eventPath": "/tmp/private/events.jsonl", + "state": { + "sessionId": "session-7", + "runId": "run-9" + } + }, + "runtimes": [{ + "taskPath": "/tmp/private/tasks.jsonl", + "state": { "status": "running" } + }] + }); + + let encoded = serialize_agent_runtime_cli_payload(&payload) + .expect("serialize redacted Runtime CLI payload"); + assert!(!encoded.contains("/tmp/private")); + let parsed = serde_json::from_str::(&encoded) + .expect("parse redacted Runtime CLI payload"); + assert!(parsed.get("sessionPath").is_none()); + assert!(parsed["runtime"].get("eventPath").is_none()); + assert!(parsed["runtimes"][0].get("taskPath").is_none()); + assert_eq!(parsed["runtime"]["state"]["sessionId"], "session-7"); + assert_eq!(parsed["runtimes"][0]["state"]["status"], "running"); + } + #[test] fn parses_agent_steer_with_stdin_only_contract() { let command = parse_cli_command(&[ diff --git a/apps/ai-game-creator-shell/src-tauri/src/main.rs b/apps/ai-game-creator-shell/src-tauri/src/main.rs index 052ad11fd..3fc8a710f 100644 --- a/apps/ai-game-creator-shell/src-tauri/src/main.rs +++ b/apps/ai-game-creator-shell/src-tauri/src/main.rs @@ -202,6 +202,10 @@ struct AgentRuntimeState { #[serde(default)] tool_action_budget: u32, #[serde(default)] + plan_revision: u64, + #[serde(default)] + plan_explanation: String, + #[serde(default)] plan: Vec, #[serde(default)] plan_steps: Vec, diff --git a/apps/ai-game-creator-shell/src-tauri/src/swarm_cli.rs b/apps/ai-game-creator-shell/src-tauri/src/swarm_cli.rs index c95f96260..fdbb3657a 100644 --- a/apps/ai-game-creator-shell/src-tauri/src/swarm_cli.rs +++ b/apps/ai-game-creator-shell/src-tauri/src/swarm_cli.rs @@ -7,6 +7,7 @@ use std::time::{Duration, Instant}; const SWARM_CHAT_POLL_INTERVAL: Duration = Duration::from_millis(250); const SWARM_CHAT_SETTLE_WINDOW: Duration = Duration::from_millis(1_500); const SWARM_CHAT_HISTORY_LIMIT: usize = 50; +const SWARM_CHAT_PLAN_STEP_LIMIT: usize = 8; #[derive(Debug, Eq, PartialEq)] enum SwarmChatInput { @@ -694,15 +695,113 @@ fn print_runtime_state( relation, state.current_action ) - .map_err(|error| format!("写入终端失败:{error}")) + .map_err(|error| format!("写入终端失败:{error}"))?; + + let completed = state + .plan_steps + .iter() + .filter(|step| step.status == "completed") + .count(); + let current_step = runtime_current_plan_step(state) + .map(|step| { + format!( + "#{} [{}] {}", + step.index.saturating_add(1), + runtime_cli_value(&step.status), + runtime_cli_value(&step.title) + ) + }) + .unwrap_or_else(|| "-".to_string()); + writeln!( + output, + "[计划] revision={} completed={}/{} current={} | waiting={} | next={}", + state.plan_revision, + completed, + state.plan_steps.len(), + current_step, + runtime_cli_value(&state.waiting_on), + runtime_cli_value(&state.next_step) + ) + .map_err(|error| format!("写入终端失败:{error}"))?; + if !state.plan_explanation.trim().is_empty() { + writeln!( + output, + " [计划说明] {}", + runtime_cli_value(&state.plan_explanation) + ) + .map_err(|error| format!("写入终端失败:{error}"))?; + } + + for step in state.plan_steps.iter().take(SWARM_CHAT_PLAN_STEP_LIMIT) { + writeln!( + output, + " [计划步骤] #{} [{}] {}", + step.index.saturating_add(1), + runtime_cli_value(&step.status), + runtime_cli_value(&step.title) + ) + .map_err(|error| format!("写入终端失败:{error}"))?; + } + if state.plan_steps.len() > SWARM_CHAT_PLAN_STEP_LIMIT { + writeln!( + output, + " [计划] 另有 {} 条步骤未显示", + state.plan_steps.len() - SWARM_CHAT_PLAN_STEP_LIMIT + ) + .map_err(|error| format!("写入终端失败:{error}"))?; + } + Ok(()) +} + +fn runtime_current_plan_step(state: &AgentRuntimeState) -> Option<&AgentRuntimePlanStep> { + state + .active_plan_step_index + .and_then(|active_index| { + state + .plan_steps + .iter() + .find(|step| step.index == active_index) + .or_else(|| state.plan_steps.get(active_index as usize)) + }) + .or_else(|| { + state.plan_steps.iter().find(|step| { + matches!( + step.status.as_str(), + "active" | "in_progress" | "running" | "waiting-for-confirmation" + ) + }) + }) + .or_else(|| { + state + .plan_steps + .iter() + .find(|step| step.status == "pending") + }) +} + +fn runtime_cli_value(value: &str) -> &str { + let value = value.trim(); + if value.is_empty() { + "-" + } else { + value + } } fn runtime_state_signature( state: &AgentRuntimeState, queue: &AgentRuntimeTaskQueueSummary, ) -> String { - format!( - "{}:{}:{}:{}:{}:{}:{}:{}:{}", + let completed_plan_steps = state + .plan_steps + .iter() + .filter(|step| step.status == "completed") + .count(); + let current_plan_step = runtime_current_plan_step(state) + .map(|step| format!("{}:{}:{}", step.index, step.status, step.title)) + .unwrap_or_default(); + let mut signature = format!( + "{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}:{}", state.run_id, state.status, state.phase, @@ -711,8 +810,21 @@ fn runtime_state_signature( queue.pending, queue.running, queue.waiting_for_confirmation, - queue.updated_at - ) + queue.updated_at, + state.plan_revision, + state + .active_plan_step_index + .map(|index| index.to_string()) + .unwrap_or_default(), + completed_plan_steps, + state.plan_steps.len(), + current_plan_step, + state.waiting_on, + state.next_step + ); + signature.push(':'); + signature.push_str(&state.plan_explanation); + signature } fn runtime_event_key(event: &AgentRuntimeEvent) -> String { @@ -812,6 +924,92 @@ mod tests { assert!(output.is_empty()); } + #[test] + fn runtime_plan_revision_and_current_step_change_state_signature() { + let mut snapshot = runtime("running", "planning", 0); + snapshot.state.updated_at = 100; + snapshot.task_queue.updated_at = 100; + snapshot.state.plan_revision = 1; + snapshot.state.plan_steps = vec![ + AgentRuntimePlanStep { + index: 0, + title: "读取现有 CLI".to_string(), + status: "in_progress".to_string(), + detail: None, + updated_at: 100, + }, + AgentRuntimePlanStep { + index: 1, + title: "补充计划展示".to_string(), + status: "pending".to_string(), + detail: None, + updated_at: 100, + }, + ]; + snapshot.state.active_plan_step_index = Some(0); + + let initial = runtime_state_signature(&snapshot.state, &snapshot.task_queue); + snapshot.state.plan_revision = 2; + let revised = runtime_state_signature(&snapshot.state, &snapshot.task_queue); + assert_ne!(initial, revised); + + snapshot.state.plan_steps[0].title = "核对现有 CLI".to_string(); + let current_step_changed = runtime_state_signature(&snapshot.state, &snapshot.task_queue); + assert_ne!(revised, current_step_changed); + + snapshot.state.plan_steps[0].status = "completed".to_string(); + snapshot.state.plan_steps[1].status = "in_progress".to_string(); + snapshot.state.active_plan_step_index = Some(1); + let advanced = runtime_state_signature(&snapshot.state, &snapshot.task_queue); + assert_ne!(current_step_changed, advanced); + } + + #[test] + fn runtime_plan_output_is_bounded_and_omits_private_observations() { + let mut snapshot = runtime("running", "planning", 0); + snapshot.state.plan_revision = 7; + snapshot.state.plan_explanation = "已完成读取,进入验证".to_string(); + snapshot.state.current_action = "展示持久计划".to_string(); + snapshot.state.waiting_on = "开发者确认".to_string(); + snapshot.state.next_step = "运行 focused cargo test".to_string(); + snapshot.state.observations = vec![ + "PRIVATE_OBSERVATION_SENTINEL".to_string(), + "PRIVATE_DETAIL_SENTINEL".to_string(), + ]; + snapshot.state.plan_steps = (0..10) + .map(|index| AgentRuntimePlanStep { + index, + title: format!("计划步骤 {}", index + 1), + status: match index { + 0 | 1 => "completed", + 2 => "in_progress", + _ => "pending", + } + .to_string(), + detail: Some(format!("PRIVATE_STEP_DETAIL_{index}")), + updated_at: 100, + }) + .collect(); + snapshot.state.active_plan_step_index = Some(2); + + let mut output = Vec::new(); + print_runtime_state(&snapshot.state, &snapshot.task_queue, &mut output) + .expect("print runtime plan progress"); + let output = String::from_utf8(output).expect("runtime output is utf-8"); + + assert!(output.contains( + "[计划] revision=7 completed=2/10 current=#3 [in_progress] 计划步骤 3 | waiting=开发者确认 | next=运行 focused cargo test" + )); + assert!(output.contains("[计划说明] 已完成读取,进入验证")); + assert_eq!(output.matches("[计划步骤]").count(), 8); + assert!(output.contains("[计划步骤] #8 [pending] 计划步骤 8")); + assert!(output.contains("另有 2 条步骤未显示")); + assert!(!output.contains("计划步骤 9")); + assert!(!output.contains("PRIVATE_OBSERVATION_SENTINEL")); + assert!(!output.contains("PRIVATE_DETAIL_SENTINEL")); + assert!(!output.contains("PRIVATE_STEP_DETAIL")); + } + #[test] fn input_channel_preserves_lines_and_eof() { let (tx, rx) = mpsc::channel(); diff --git a/apps/ai-game-creator-shell/src-tauri/src/tests.rs b/apps/ai-game-creator-shell/src-tauri/src/tests.rs index 0749ab96b..935992cbf 100644 --- a/apps/ai-game-creator-shell/src-tauri/src/tests.rs +++ b/apps/ai-game-creator-shell/src-tauri/src/tests.rs @@ -180,6 +180,9 @@ fn pending_tool_action_for_test( &state.run_id, ) .expect("read verification gate before pending action"), + planned_repository_context_fingerprint: build_repository_startup_context_at(root) + .expect("build pending action repository context") + .fingerprint, planned_steer_cursor: state.applied_steer_cursor, action, action_id: agent_runtime_tool_action_id( @@ -1351,6 +1354,7 @@ fn spawn_mock_llm_server_responses(response_contents: Vec) -> String { fn final_tool_plan_response(response: impl Into) -> String { serde_json::json!({ "thinkingSummary": "已有工具观察足够,可以收束后台任务", + "planUpdate": null, "plan": [], "actions": [], "response": response.into(), @@ -3078,6 +3082,948 @@ async fn role_agent_runtime_turn_persists_session_events_and_index() { fs::remove_dir_all(root).ok(); } +#[test] +fn structured_plan_update_validates_and_advances_monotonically() { + let parsed = parse_game_creator_agent_tool_plan_response( + &serde_json::json!({ + "thinkingSummary": "先建立可恢复计划", + "planUpdate": { + "explanation": "首次拆解", + "steps": [ + { "step": "读取项目", "status": "in_progress" }, + { "step": "验证结果", "status": "pending" } + ] + }, + "plan": ["旧计划应被结构化更新覆盖"], + "actions": [], + "response": "" + }) + .to_string(), + ) + .expect("parse structured plan"); + assert!(parsed.plan_update.is_some()); + + let legacy = parse_game_creator_agent_tool_plan_response( + &serde_json::json!({ + "thinkingSummary": "旧 Provider fallback", + "plan": ["旧步骤"], + "actions": [], + "response": "" + }) + .to_string(), + ) + .expect("parse legacy plan"); + assert!(legacy.plan_update.is_none()); + + let mut runtime = default_game_creator_agent_runtime_state("design-director", "plan-run"); + assert!(apply_agent_runtime_plan_update( + &mut runtime, + parsed.plan_update.as_ref().expect("plan update"), + ) + .expect("apply first plan")); + assert_eq!(runtime.plan_revision, 1); + assert_eq!(runtime.active_plan_step_index, Some(0)); + assert_eq!(runtime.plan_steps[0].status, "in_progress"); + assert_eq!(runtime.plan, vec!["读取项目", "验证结果"]); + + assert!(!apply_agent_runtime_plan_update( + &mut runtime, + parsed.plan_update.as_ref().expect("same plan update"), + ) + .expect("idempotent plan update")); + assert_eq!(runtime.plan_revision, 1); + + let progressed = AgentRuntimePlanUpdate { + explanation: "读取完成,开始验证".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取项目".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "验证结果".to_string(), + status: "in_progress".to_string(), + }, + ], + }; + assert!(apply_agent_runtime_plan_update(&mut runtime, &progressed).expect("advance plan")); + assert_eq!(runtime.plan_revision, 2); + assert_eq!(runtime.plan_steps[0].status, "completed"); + assert_eq!(runtime.active_plan_step_index, Some(1)); + + let regression = AgentRuntimePlanUpdate { + explanation: "错误回退".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取项目".to_string(), + status: "pending".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "验证结果".to_string(), + status: "in_progress".to_string(), + }, + ], + }; + assert!(apply_agent_runtime_plan_update(&mut runtime, ®ression) + .expect_err("completed step cannot regress") + .contains("不能让已完成步骤回退")); + assert_eq!(runtime.plan_revision, 2); + + let completed = AgentRuntimePlanUpdate { + explanation: "全部完成".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "验证结果".to_string(), + status: "completed".to_string(), + }], + }; + assert!(apply_agent_runtime_plan_update(&mut runtime, &completed) + .expect("terminal step is retained")); + assert_eq!(runtime.plan_revision, 3); + assert_eq!(runtime.plan, vec!["读取项目", "验证结果"]); + assert!(runtime + .plan_steps + .iter() + .all(|step| step.status == "completed")); + assert!(structured_plan_completion_blocker(&runtime).is_none()); +} + +#[test] +fn structured_plan_update_rejects_invalid_shapes_and_ignores_action_indexes() { + let invalid_updates = [ + AgentRuntimePlanUpdate { + explanation: " ".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "步骤".to_string(), + status: "pending".to_string(), + }], + }, + AgentRuntimePlanUpdate { + explanation: "重复".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "步骤".to_string(), + status: "pending".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "步骤".to_string(), + status: "completed".to_string(), + }, + ], + }, + AgentRuntimePlanUpdate { + explanation: "多活动步骤".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "步骤一".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "步骤二".to_string(), + status: "in_progress".to_string(), + }, + ], + }, + AgentRuntimePlanUpdate { + explanation: "未知状态".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "步骤".to_string(), + status: "blocked".to_string(), + }], + }, + ]; + for update in invalid_updates { + assert!(sanitize_agent_runtime_plan_update(&update).is_err()); + } + + let mut runtime = default_game_creator_agent_runtime_state("design-director", "guard-run"); + let update = AgentRuntimePlanUpdate { + explanation: "等待工具真实结果".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取项目".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "验证结果".to_string(), + status: "pending".to_string(), + }, + ], + }; + apply_agent_runtime_plan_update(&mut runtime, &update).expect("apply guarded plan"); + let snapshot = runtime.plan_steps.clone(); + activate_agent_runtime_plan_step(&mut runtime, 1, "工具数组第二项"); + complete_agent_runtime_active_plan_step(&mut runtime, "completed", "工具成功"); + retry_agent_runtime_active_plan_step(&mut runtime, "等待重试"); + activate_agent_runtime_response_plan_step(&mut runtime, "准备回复"); + complete_agent_runtime_remaining_plan_steps(&mut runtime, "自动收束"); + assert_eq!(runtime.plan_steps, snapshot); + let blocker = structured_plan_completion_blocker(&runtime).expect("incomplete plan blocks"); + assert_eq!(blocker.tool, "runtime.plan_update"); + assert!(blocker.summary.contains("0/2")); + let blocker_detail = blocker.detail.expect("structured blocker detail"); + assert!(blocker_detail.contains("planRevision=1")); + assert!(blocker_detail.contains("completed=0")); + assert!(blocker_detail.contains("pending=1")); + assert!(blocker_detail.contains("inProgress=1")); + assert!(blocker_detail.contains("failed=0")); + assert!(blocker_detail.contains(&format!( + "in_progress:{:x}", + Sha256::digest("读取项目".as_bytes()) + ))); + assert!(blocker_detail.contains(&format!( + "pending:{:x}", + Sha256::digest("验证结果".as_bytes()) + ))); + assert!(!blocker.summary.contains("读取项目")); + assert!(!blocker.summary.contains("验证结果")); + assert!(!blocker_detail.contains("读取项目")); + assert!(!blocker_detail.contains("验证结果")); +} + +#[test] +fn structured_plan_state_normalization_fails_closed_on_corruption() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "损坏计划状态项目").expect("project init"); + let mut runtime = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + "验证损坏计划失败关闭", + "structured-plan-corrupt-run", + "agent-background-task", + "损坏计划测试", + vec!["旧计划".to_string()], + ) + .expect("start corrupt plan state"); + apply_agent_runtime_plan_update( + &mut runtime, + &AgentRuntimePlanUpdate { + explanation: "当前步骤执行中".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "执行验证".to_string(), + status: "in_progress".to_string(), + }], + }, + ) + .expect("apply plan before corruption"); + runtime.active_plan_step_index = None; + write_game_creator_agent_runtime_state(&root, &runtime).expect("persist corrupt plan state"); + + let loaded = read_game_creator_agent_runtime_at(&root, "design-director") + .expect("read corrupt plan state") + .state; + assert_eq!(loaded.status, "needs-reconciliation"); + assert_eq!(loaded.phase, "needs-reconciliation"); + assert!(loaded.error.as_deref().is_some_and(|error| { + error.contains("activePlanStepIndex") && error.contains("in_progress") + })); + fs::remove_dir_all(root).ok(); +} + +#[test] +fn structured_plan_failure_after_all_steps_completed_preserves_terminal_snapshot() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "结构化计划失败收束项目").expect("project init"); + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + "计划完成后模拟 Runtime 失败", + "structured-plan-completed-before-failure-run", + "agent-background-task", + "准备失败收束回归", + vec!["旧计划占位".to_string()], + ) + .expect("start completed structured plan fixture"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "所有必要步骤已完成".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "完成实现".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "完成验证".to_string(), + status: "completed".to_string(), + }, + ], + }, + ) + .expect("apply fully completed structured plan"); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist completed structured plan before failure"); + let expected_revision = state.plan_revision; + let expected_explanation = state.plan_explanation.clone(); + let expected_plan = state.plan.clone(); + let expected_steps = state.plan_steps.clone(); + let expected_active_index = state.active_plan_step_index; + assert_eq!(expected_revision, 1); + assert!(expected_steps.iter().all(|step| step.status == "completed")); + assert_eq!(expected_active_index, None); + + let failed = fail_game_creator_agent_runtime_turn_at( + &root, + state, + "结构化计划完成后的模拟 Runtime 失败", + ) + .expect("fail completed structured plan through real terminal path"); + assert_eq!(failed.status, "failed"); + assert_eq!(failed.phase, "failed"); + assert_eq!(failed.plan_revision, expected_revision); + assert_eq!(failed.plan_explanation, expected_explanation); + assert_eq!(failed.plan, expected_plan); + assert_eq!(failed.plan_steps, expected_steps); + assert_eq!(failed.active_plan_step_index, expected_active_index); + assert!(failed + .plan_steps + .iter() + .all(|step| step.status == "completed")); + assert!(structured_plan_completion_blocker(&failed).is_none()); + + let persisted = read_game_creator_agent_runtime_at(&root, "design-director") + .expect("read failed structured plan state") + .state; + assert_eq!(persisted.status, "failed"); + assert_eq!(persisted.phase, "failed"); + assert_eq!(persisted.plan_revision, expected_revision); + assert_eq!(persisted.plan_explanation, expected_explanation); + assert_eq!(persisted.plan, expected_plan); + assert_eq!(persisted.plan_steps, expected_steps); + assert_eq!(persisted.active_plan_step_index, expected_active_index); + + fs::remove_dir_all(root).ok(); +} + +#[test] +fn structured_plan_failure_and_budget_preserve_last_trusted_progress() { + for budget_exhausted in [false, true] { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "结构化计划失败进度项目") + .expect("project init"); + let run_id = if budget_exhausted { + "structured-plan-budget-preserves-progress-run" + } else { + "structured-plan-failure-preserves-progress-run" + }; + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + "保留失败前最后一次可信计划进度", + run_id, + "agent-background-task", + "准备失败进度回归", + vec!["旧计划占位".to_string()], + ) + .expect("start structured failure fixture"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "实现已完成,验证仍在执行".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "完成实现".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "运行验证".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "整理结果".to_string(), + status: "pending".to_string(), + }, + ], + }, + ) + .expect("apply structured failure plan"); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist structured state before failure"); + persist_game_creator_agent_runtime_context( + &root, + &state, + &state.current_task, + &AgentRuntimeToolPlan::default(), + &[], + 0, + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("persist structured context before failure"); + let expected_revision = state.plan_revision; + let expected_explanation = state.plan_explanation.clone(); + let expected_plan = state.plan.clone(); + let expected_steps = state.plan_steps.clone(); + let expected_active_index = state.active_plan_step_index; + + let failed = if budget_exhausted { + fail_game_creator_agent_runtime_budget_at(&root, state, "loop-budget-exhausted:test") + } else { + fail_game_creator_agent_runtime_turn_at(&root, state, "structured failure test") + } + .expect("persist structured terminal runtime"); + assert_eq!(failed.plan_revision, expected_revision); + assert_eq!(failed.plan_explanation, expected_explanation); + assert_eq!(failed.plan, expected_plan); + assert_eq!(failed.plan_steps, expected_steps); + assert_eq!(failed.active_plan_step_index, expected_active_index); + assert_eq!( + failed.phase, + if budget_exhausted { + "budget-exhausted" + } else { + "failed" + } + ); + let context = read_game_creator_agent_runtime_context_bundle(&root, &failed) + .expect("read terminal structured context") + .expect("terminal structured context exists"); + assert_eq!(context.plan_revision, expected_revision); + assert_eq!(context.plan_steps, expected_steps); + assert_eq!(context.active_plan_step_index, expected_active_index); + + fs::remove_dir_all(root).ok(); + } +} + +#[test] +fn structured_plan_state_write_failure_stops_before_context_and_audit() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "结构化计划状态写失败项目") + .expect("project init"); + let (request_sender, request_receiver) = mpsc::channel(); + let (release_sender, release_receiver) = mpsc::channel(); + let response = serde_json::json!({ + "thinkingSummary": "准备提交结构化计划", + "planUpdate": { + "explanation": "状态写入失败时不得推进其它投影", + "steps": [ + { "step": "完成检查", "status": "completed" } + ] + }, + "plan": [], + "actions": [], + "response": "该回复不得落盘" + }) + .to_string(); + let base_url = spawn_releasable_mock_llm_server_responses_with_capture( + vec![response], + request_sender, + release_receiver, + ); + let _config_guard = write_test_local_config(format!( + r#"{{ + "agentLlm": {{ + "design-director": {{ + "apiKey": "design-key", + "baseUrl": {base_url:?}, + "model": "design-runtime-model", + "apiKind": "openai_responses" + }} + }} +}}"# + )); + let run_id = "structured-plan-state-write-failure-run"; + start_game_creator_agent_background_task_at( + &root, + "design-director", + "验证结构化计划状态写失败边界", + run_id, + ) + .expect("start structured state write failure task"); + request_receiver + .recv_timeout(Duration::from_secs(2)) + .expect("planning request reached provider"); + + let state_path = root.join(".agent/runtime/agents/design-director.json"); + fs::remove_file(&state_path).expect("remove runtime state before response"); + fs::create_dir(&state_path).expect("replace runtime state with directory"); + release_sender + .send(()) + .expect("release structured plan response"); + drop(wait_to_acquire_agent_runtime_lock(&root, "design-director")); + + let context_path = + game_creator_agent_runtime_context_bundle_path(&root, "design-director", run_id); + let context: Value = serde_json::from_str( + &fs::read_to_string(&context_path).expect("read unchanged context bundle"), + ) + .expect("parse unchanged context bundle"); + assert_eq!(context["planRevision"], 0); + let records = read_agent_db_records_for_test(&root); + assert!(!records.iter().any(|record| { + record["recordType"] == "agent.runtime.plan_update" && record["runId"] == run_id + })); + let conversation = read_local_conversation_for_session_at( + &root, + Some("design-director"), + Some("agent-session-design-director"), + ) + .expect("read state write failure conversation"); + assert!(!conversation.messages.iter().any(|message| { + message.role == "assistant" && message.content.contains("该回复不得落盘") + })); + assert!(conversation.messages.iter().any(|message| { + message.role == "assistant" && message.content.contains("后台任务失败") + })); + + fs::remove_dir(&state_path).expect("remove sabotaged runtime state directory"); + fs::remove_dir_all(root).ok(); +} + +#[tokio::test] +async fn structured_plan_background_runtime_requires_explicit_completion() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "持久计划项目").expect("project init"); + fs::write(root.join("game/notes.txt"), "结构化计划证据").expect("write notes"); + let (sender, receiver) = mpsc::channel(); + let responses = vec![ + serde_json::json!({ + "thinkingSummary": "先读取项目证据", + "planUpdate": { + "explanation": "首次拆解", + "steps": [ + { "step": "读取项目证据", "status": "in_progress" }, + { "step": "确认并回复", "status": "pending" } + ] + }, + "plan": [], + "actions": [{ + "tool": "file.read", + "reason": "读取真实证据", + "input": { "path": "game/notes.txt" } + }], + "response": "" + }) + .to_string(), + serde_json::json!({ + "thinkingSummary": "证据已读,仍需完成确认", + "planUpdate": { + "explanation": "读取已完成", + "steps": [ + { "step": "读取项目证据", "status": "completed" }, + { "step": "确认并回复", "status": "in_progress" } + ] + }, + "plan": [], + "actions": [], + "response": "这条回复必须被未完成计划阻断" + }) + .to_string(), + serde_json::json!({ + "thinkingSummary": "全部必要步骤已经完成", + "planUpdate": { + "explanation": "确认完成并收束", + "steps": [ + { "step": "读取项目证据", "status": "completed" }, + { "step": "确认并回复", "status": "completed" } + ] + }, + "plan": [], + "actions": [], + "response": "结构化计划完成后才允许返回。" + }) + .to_string(), + ]; + let base_url = spawn_mock_llm_server_responses_with_capture(responses, Some(sender)); + let _config_guard = write_test_local_config(format!( + r#"{{ + "agentLlm": {{ + "design-director": {{ + "apiKey": "design-key", + "baseUrl": {base_url:?}, + "model": "design-runtime-model", + "apiKind": "openai_responses" + }} + }} +}}"# + )); + + start_game_creator_agent_background_task_at( + &root, + "design-director", + "读取证据并按持久计划收束", + "structured-plan-loop-run", + ) + .expect("start structured plan task"); + let requests = (0..3) + .map(|_| { + receiver + .recv_timeout(Duration::from_secs(3)) + .expect("structured plan request") + }) + .collect::>(); + assert!(requests[1].contains("结构化计划证据")); + assert!(requests[1].contains("结构化计划:revision=1")); + assert!(requests[1].contains("[in_progress] 读取项目证据")); + assert!(requests[2].contains("runtime.plan_update")); + assert!(requests[2].contains("结构化计划:revision=2")); + assert!(!requests[2].contains("这条回复必须被未完成计划阻断")); + + let runtime = wait_for_agent_runtime_idle(&root, "design-director"); + assert_eq!(runtime.status, "idle"); + assert_eq!(runtime.phase, "completed"); + assert_eq!(runtime.plan_revision, 3); + assert_eq!(runtime.plan_explanation, "确认完成并收束"); + assert_eq!(runtime.active_plan_step_index, None); + assert!(runtime + .plan_steps + .iter() + .all(|step| step.status == "completed")); + assert_eq!( + runtime.last_response.as_deref(), + Some("结构化计划完成后才允许返回。") + ); + assert_eq!( + read_game_creator_agent_runtime_project_revision(&root) + .expect("read project revision after plan updates") + .revision, + 0, + "plan metadata must not advance the project revision" + ); + let plan_audits = read_agent_db_records_for_test(&root) + .into_iter() + .filter(|record| { + record["recordType"] == "agent.runtime.plan_update" + && record["runId"] == "structured-plan-loop-run" + }) + .collect::>(); + assert_eq!(plan_audits.len(), 3); + assert_eq!( + plan_audits + .iter() + .filter_map(|record| record["planRevision"].as_u64()) + .collect::>(), + vec![1, 2, 3] + ); + assert!(plan_audits.iter().all(|record| { + record["steps"].as_array().is_some_and(|steps| { + steps + .iter() + .filter(|step| step["status"] == "in_progress") + .count() + <= 1 + }) + })); + let plan_audit_json = serde_json::to_string(&plan_audits).expect("serialize plan audits"); + assert!(!plan_audit_json.contains("读取项目证据")); + assert!(!plan_audit_json.contains("确认并回复")); + assert!(plan_audits.iter().all(|record| { + record["steps"].as_array().is_some_and(|steps| { + steps.iter().all(|step| { + step["stepSha256"] + .as_str() + .is_some_and(|value| value.len() == 64) + && step.get("step").is_none() + }) + }) + })); + fs::remove_dir_all(root).ok(); +} + +#[tokio::test] +async fn structured_plan_same_run_stale_running_resume_preserves_v3_snapshot() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "结构化计划 Runner 恢复项目") + .expect("project init"); + let task = "恢复同一 run 的结构化计划"; + let run_id = "structured-plan-stale-running-resume-run"; + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + task, + run_id, + "agent-background-task", + "Runner 中断前正在规划", + vec!["旧计划占位".to_string()], + ) + .expect("start stale structured runtime"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "恢复时必须保留完整计划快照".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取持久上下文".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "继续 Runner 恢复".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "完成最终核对".to_string(), + status: "pending".to_string(), + }, + ], + }, + ) + .expect("apply pre-restart structured plan"); + state.updated_at = unix_timestamp().saturating_sub(600); + append_game_creator_agent_runtime_task(&root, &state).expect("append stale running task"); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist stale structured runtime"); + persist_game_creator_agent_runtime_context( + &root, + &state, + task, + &AgentRuntimeToolPlan::default(), + &[], + 0, + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("persist pre-restart v3 context"); + let persisted_before = read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect("read pre-restart context") + .expect("pre-restart context exists"); + assert_eq!(persisted_before.plan_revision, 1); + assert_eq!(persisted_before.plan_steps, state.plan_steps); + assert_eq!(persisted_before.active_plan_step_index, Some(1)); + + let lock_path = root.join(".agent/runtime/locks/design-director.lock"); + fs::create_dir_all(lock_path.parent().expect("lock parent")).expect("create lock parent"); + fs::write( + &lock_path, + serde_json::json!({ + "agentId": "design-director", + "pid": 0, + "createdAt": 1, + }) + .to_string(), + ) + .expect("write stale Runner lock projection"); + + let (request_sender, request_receiver) = mpsc::channel(); + let (response_sender, response_receiver) = mpsc::channel(); + let base_url = + spawn_interactive_mock_llm_server_with_capture(1, request_sender, response_receiver); + let _config_guard = write_test_local_config(format!( + r#"{{ + "agentLlm": {{ + "design-director": {{ + "apiKey": "design-key", + "baseUrl": {base_url:?}, + "model": "design-runtime-model", + "apiKind": "openai_responses" + }} + }} +}}"# + )); + + let resumed = resume_game_creator_agent_background_tasks_at(&root) + .expect("resume stale structured running task"); + assert_eq!(resumed.len(), 1); + assert_eq!(resumed[0].state.task_id, state.task_id); + assert_eq!(resumed[0].state.session_id, state.session_id); + assert_eq!(resumed[0].state.run_id, run_id); + assert_eq!(resumed[0].state.plan_revision, 1); + assert_eq!(resumed[0].state.plan_explanation, state.plan_explanation); + assert_eq!(resumed[0].state.plan_steps, state.plan_steps); + assert_eq!(resumed[0].state.active_plan_step_index, Some(1)); + + let replanning_request = request_receiver + .recv_timeout(Duration::from_secs(3)) + .expect("same-run resumed planning request"); + assert!(replanning_request.contains("结构化计划:revision=1")); + assert!(replanning_request.contains("恢复时必须保留完整计划快照")); + assert!(replanning_request.contains("[completed] 读取持久上下文")); + assert!(replanning_request.contains("[in_progress] 继续 Runner 恢复")); + assert!(replanning_request.contains("[pending] 完成最终核对")); + + let running = read_game_creator_agent_runtime_at(&root, "design-director") + .expect("read resumed running state") + .state; + assert_eq!(running.task_id, state.task_id); + assert_eq!(running.session_id, state.session_id); + assert_eq!(running.run_id, run_id); + assert_eq!(running.plan_revision, 1); + assert_eq!(running.plan_explanation, state.plan_explanation); + assert_eq!(running.plan_steps, state.plan_steps); + assert_eq!(running.active_plan_step_index, Some(1)); + let resumed_context = read_game_creator_agent_runtime_context_bundle(&root, &running) + .expect("read resumed v3 context") + .expect("resumed v3 context exists"); + assert_eq!(resumed_context.plan_revision, 1); + assert_eq!(resumed_context.plan_explanation, running.plan_explanation); + assert_eq!(resumed_context.plan, running.plan); + assert_eq!(resumed_context.plan_steps, running.plan_steps); + assert_eq!(resumed_context.active_plan_step_index, Some(1)); + + response_sender + .send( + serde_json::json!({ + "thinkingSummary": "恢复快照可信,可以完成剩余核对", + "planUpdate": { + "explanation": "Runner 恢复后完成剩余步骤", + "steps": [ + { "step": "读取持久上下文", "status": "completed" }, + { "step": "继续 Runner 恢复", "status": "completed" }, + { "step": "完成最终核对", "status": "completed" } + ] + }, + "plan": [], + "actions": [], + "response": "同一 run 已保留结构化计划并完成恢复。" + }) + .to_string(), + ) + .expect("release resumed planning response"); + let completed = wait_for_agent_runtime_idle(&root, "design-director"); + assert_eq!(completed.task_id, state.task_id); + assert_eq!(completed.session_id, state.session_id); + assert_eq!(completed.run_id, run_id); + assert_eq!(completed.phase, "completed"); + assert_eq!(completed.plan_revision, 2); + assert_eq!(completed.active_plan_step_index, None); + assert!(completed + .plan_steps + .iter() + .all(|step| step.status == "completed")); + assert_eq!( + completed.last_response.as_deref(), + Some("同一 run 已保留结构化计划并完成恢复。") + ); + + fs::remove_dir_all(root).ok(); +} + +#[test] +fn structured_plan_finalization_without_readable_runtime_state_needs_reconciliation() { + for (state_mode, run_id) in [ + ("missing", "structured-plan-finalization-missing-state-run"), + ( + "unreadable", + "structured-plan-finalization-unreadable-state-run", + ), + ] { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "finalization state 失败关闭项目") + .expect("project init"); + let task = format!("验证 {state_mode} Runtime state 不得降级完成"); + let response = format!("{state_mode} state 时不得写入的 assistant"); + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + &task, + run_id, + "agent-background-task", + "准备最终回复", + vec!["旧计划占位".to_string()], + ) + .expect("start finalization state fixture"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "finalization 前计划已可信完成".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "完成最终核对".to_string(), + status: "completed".to_string(), + }], + }, + ) + .expect("apply completed finalization plan"); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist completed structured state"); + persist_game_creator_agent_runtime_context( + &root, + &state, + &task, + &AgentRuntimeToolPlan::default(), + &[], + 0, + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("persist completed finalization context"); + let outcome = finish_game_creator_agent_background_runtime_turn_with_checkpoint_at( + &root, + state.clone(), + &response, + 0, + &[], + |checkpoint| { + if checkpoint == AgentRuntimeFinalizationCheckpoint::Prepared { + Err(format!("injected-{state_mode}-state-checkpoint")) + } else { + Ok(()) + } + }, + ) + .expect("prepare finalization journal without assistant"); + assert!(matches!( + outcome, + AgentBackgroundFinalizationOutcome::Pending(_) + )); + let before = read_local_conversation_for_session_at( + &root, + Some("design-director"), + Some(&state.session_id), + ) + .expect("read conversation before state loss"); + assert_eq!( + before + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 0 + ); + + let state_path = root.join(".agent/runtime/agents/design-director.json"); + if state_mode == "missing" { + fs::remove_file(&state_path).expect("remove Runtime state"); + } else { + fs::write(&state_path, b"{not-readable-runtime-state").expect("corrupt Runtime state"); + } + + let resumed = resume_game_creator_agent_background_tasks_at(&root) + .expect("resume finalization without readable Runtime state"); + assert_eq!(resumed.len(), 1); + let reconciled = read_game_creator_agent_runtime_at(&root, "design-director") + .expect("read reconciled Runtime state") + .state; + assert_eq!(reconciled.run_id, run_id); + assert_eq!(reconciled.status, "failed"); + assert_eq!(reconciled.phase, "needs-reconciliation"); + assert!(reconciled.error.as_deref().is_some_and(|error| { + error.contains("Runtime state 缺失") && error.contains("结构化计划快照") + })); + assert!( + read_game_creator_agent_runtime_finalization_journal(&root, "design-director", run_id,) + .expect("read retained finalization journal") + .is_some(), + "unresolved journal must remain for {state_mode} state" + ); + let after = read_local_conversation_for_session_at( + &root, + Some("design-director"), + Some(&state.session_id), + ) + .expect("read conversation after failed-closed recovery"); + assert_eq!( + after + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 0, + "{state_mode} state must not complete assistant" + ); + let records = read_agent_db_records_for_test(&root); + assert!(records.iter().any(|record| { + record["recordType"] + == "agent.runtime.background_task.finalization_needs_reconciliation" + && record["runId"] == run_id + && record["reason"] == "runtime-state-missing" + })); + assert!(!records.iter().any(|record| { + record["runId"] == run_id + && (record["recordType"] == "agent.runtime.background_task.completed" + || (record["recordType"] == "conversation.message" + && record["role"] == "assistant")) + })); + + fs::remove_dir_all(root).ok(); + } +} + #[tokio::test] async fn background_agent_runtime_task_executes_plan_tool_observation_loop() { let root = unique_project_path(); @@ -3267,6 +4213,30 @@ async fn background_agent_runtime_task_executes_plan_tool_observation_loop() { assert!(event_types.contains(&"observation")); assert!(event_types.contains(&"response")); assert!(!event_types.contains(&"error")); + let thinking_event = runtime_result + .recent_events + .iter() + .find(|event| event.event_type == "thinking_summary") + .expect("thinking summary event"); + assert!(thinking_event.detail.as_deref().is_some_and(|detail| detail + .contains("thinkingSummarySha256=") + && detail.contains("chars="))); + assert!(!thinking_event + .detail + .as_deref() + .unwrap_or_default() + .contains("需要先读项目笔记再给策划建议")); + let plan_event = runtime_result + .recent_events + .iter() + .find(|event| event.event_type == "plan") + .expect("legacy plan event"); + assert_eq!(plan_event.detail.as_deref(), Some("planStepCount=3")); + assert!(!plan_event + .detail + .as_deref() + .unwrap_or_default() + .contains("读取项目笔记")); let agent_db = fs::read_to_string(root.join(".agent/agent.db")).expect("agent db"); assert!(agent_db.contains("\"recordType\":\"agent.runtime.tool_observation\"")); assert!(agent_db.contains("\"tool\":\"file.read\"")); @@ -3282,6 +4252,7 @@ async fn background_agent_runtime_executes_native_function_tool_plan() { let (sender, receiver) = mpsc::channel(); let first_arguments = serde_json::json!({ "thinkingSummary": "先读取项目索引确认结构", + "planUpdate": null, "plan": ["读取项目索引", "根据观察回复"], "actions": [{ "tool": "project.index", @@ -3879,7 +4850,11 @@ async fn background_agent_runtime_recovers_bound_context_for_the_same_session_an "nextLoopIndex": AGENT_RUNTIME_BACKGROUND_LOOP_LIMIT, "contextWindow": 2, "thinkingSummary": "前六轮已经定位关键约束", - "plan": ["使用恢复证据收束"], + "planRevision": state.plan_revision, + "planExplanation": state.plan_explanation, + "plan": state.plan, + "planSteps": state.plan_steps, + "activePlanStepIndex": state.active_plan_step_index, "fallbackResponse": "", "observations": [{ "tool": "project.search", @@ -4003,6 +4978,211 @@ fn background_agent_runtime_rejects_cross_session_context_bundle_on_resume() { fs::remove_dir_all(root).ok(); } +#[test] +fn structured_plan_context_bundle_migrates_v2_and_rejects_v3_plan_mismatch() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "计划快照恢复项目").expect("project init"); + let task = "验证结构化计划快照恢复"; + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + task, + "structured-plan-context-run", + "agent-background-task", + "结构化计划恢复测试", + vec!["旧短计划".to_string()], + ) + .expect("start structured context state"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "恢复时保持计划进度".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取上下文".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "继续验证".to_string(), + status: "in_progress".to_string(), + }, + ], + }, + ) + .expect("apply structured context plan"); + write_game_creator_agent_runtime_state(&root, &state).expect("persist structured state"); + let bundle = build_game_creator_agent_runtime_context_bundle( + &root, + &state, + task, + &AgentRuntimeToolPlan::default(), + &[], + 0, + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("build v3 context bundle"); + let bundle_path = game_creator_agent_runtime_context_bundle_path( + &root, + "design-director", + "structured-plan-context-run", + ); + fs::create_dir_all(bundle_path.parent().expect("bundle parent")).expect("create bundle parent"); + + let mut legacy = serde_json::to_value(&bundle).expect("serialize v3 bundle"); + let legacy_object = legacy.as_object_mut().expect("bundle object"); + legacy_object.insert( + "schemaVersion".to_string(), + Value::String("game-creator-runtime-context-bundle.v2".to_string()), + ); + legacy_object.remove("planRevision"); + legacy_object.remove("planExplanation"); + legacy_object.remove("planSteps"); + legacy_object.remove("activePlanStepIndex"); + fs::write( + &bundle_path, + serde_json::to_string_pretty(&legacy).expect("serialize v2 bundle"), + ) + .expect("write v2 bundle"); + let migrated = read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect("read v2 bundle") + .expect("v2 bundle exists"); + assert_eq!( + migrated.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!(migrated.plan_revision, state.plan_revision); + assert_eq!(migrated.plan_steps, state.plan_steps); + assert_eq!( + migrated.active_plan_step_index, + state.active_plan_step_index + ); + + let mut revision_mismatch = bundle.clone(); + revision_mismatch.plan_revision += 1; + write_game_creator_agent_runtime_context_bundle(&root, &revision_mismatch) + .expect("write revision-mismatched v3 bundle"); + assert!( + read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect_err("v3 plan revision mismatch must fail") + .contains("计划 revision 与当前状态不匹配") + ); + + let mut snapshot_mismatch = bundle; + snapshot_mismatch.plan_steps[1].title = "被篡改的步骤".to_string(); + snapshot_mismatch.plan[1] = "被篡改的步骤".to_string(); + write_game_creator_agent_runtime_context_bundle(&root, &snapshot_mismatch) + .expect("write snapshot-mismatched v3 bundle"); + assert!( + read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect_err("v3 plan snapshot mismatch must fail") + .contains("结构化计划快照与当前状态不匹配") + ); + + fs::remove_dir_all(root).ok(); +} + +#[test] +fn structured_plan_context_v3_revision_zero_rejects_each_legacy_snapshot_mismatch() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "revision zero context 项目") + .expect("project init"); + let task = "验证 revision zero 的 v3 完整快照"; + let state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + task, + "structured-plan-context-revision-zero-run", + "agent-background-task", + "准备 legacy context 快照", + vec![ + "读取 legacy 项目".to_string(), + "完成 legacy 核对".to_string(), + ], + ) + .expect("start revision zero context state"); + assert_eq!(state.plan_revision, 0); + assert_eq!(state.active_plan_step_index, Some(0)); + let bundle = build_game_creator_agent_runtime_context_bundle( + &root, + &state, + task, + &AgentRuntimeToolPlan::default(), + &[], + 0, + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("build revision zero v3 context"); + assert_eq!( + bundle.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!(bundle.plan_revision, 0); + + let mut plan_mismatch = bundle.clone(); + plan_mismatch.plan[0] = "被篡改的 legacy plan".to_string(); + write_game_creator_agent_runtime_context_bundle(&root, &plan_mismatch) + .expect("write revision zero plan mismatch"); + assert!( + read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect_err("revision zero v3 legacy plan mismatch must fail") + .contains("计划快照与当前状态不匹配") + ); + + let mut steps_mismatch = bundle.clone(); + steps_mismatch.plan_steps[0].title = "被篡改的 legacy planSteps".to_string(); + write_game_creator_agent_runtime_context_bundle(&root, &steps_mismatch) + .expect("write revision zero planSteps mismatch"); + assert!( + read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect_err("revision zero v3 legacy planSteps mismatch must fail") + .contains("计划快照与当前状态不匹配") + ); + + let mut active_index_mismatch = bundle.clone(); + active_index_mismatch.active_plan_step_index = None; + write_game_creator_agent_runtime_context_bundle(&root, &active_index_mismatch) + .expect("write revision zero active index mismatch"); + assert!( + read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect_err("revision zero v3 active index mismatch must fail") + .contains("计划快照与当前状态不匹配") + ); + + let bundle_path = game_creator_agent_runtime_context_bundle_path( + &root, + "design-director", + "structured-plan-context-revision-zero-run", + ); + let mut legacy_v2 = serde_json::to_value(&bundle).expect("serialize revision zero v3 bundle"); + let legacy_object = legacy_v2.as_object_mut().expect("legacy v2 bundle object"); + legacy_object.insert( + "schemaVersion".to_string(), + Value::String("game-creator-runtime-context-bundle.v2".to_string()), + ); + legacy_object.remove("planRevision"); + legacy_object.remove("planExplanation"); + legacy_object.remove("planSteps"); + legacy_object.remove("activePlanStepIndex"); + fs::write( + &bundle_path, + serde_json::to_string_pretty(&legacy_v2).expect("serialize revision zero v2 bundle"), + ) + .expect("write compatible revision zero v2 bundle"); + let migrated = read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect("read compatible revision zero v2 bundle") + .expect("revision zero v2 bundle exists"); + assert_eq!( + migrated.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!(migrated.plan_revision, 0); + assert_eq!(migrated.plan, state.plan); + assert_eq!(migrated.plan_steps, state.plan_steps); + assert_eq!(migrated.active_plan_step_index, Some(0)); + + fs::remove_dir_all(root).ok(); +} + #[test] fn legacy_context_and_pending_records_fail_closed() { let root = unique_project_path(); @@ -4046,6 +5226,7 @@ fn legacy_context_and_pending_records_fail_closed() { for version in [ "game-creator-pending-action.v1", "game-creator-pending-action.v2", + "game-creator-pending-action.v3", ] { fs::write( &pending_path, @@ -4072,7 +5253,7 @@ fn legacy_context_and_pending_records_fail_closed() { runtime.state.run_id == "legacy-v1-run" && runtime.state.status == "failed" && runtime.state.error.as_deref().is_some_and(|error| { - error.contains("待确认动作恢复失败") && error.contains("pending-action.v2") + error.contains("待确认动作恢复失败") && error.contains("pending-action.v3") }) })); @@ -4528,7 +5709,7 @@ fn agent_runtime_context_bundle_preserves_project_verification_gate_evidence() { fn agent_runtime_context_bundle_redacts_paths_and_secrets_and_limits_plan() { let root = unique_project_path(); init_local_game_project_at(&root, "project-1", "上下文安全项目").expect("project init"); - let state = start_game_creator_agent_runtime_task_at( + let mut state = start_game_creator_agent_runtime_task_at( &root, "design-director", "验证上下文安全持久化", @@ -4542,12 +5723,14 @@ fn agent_runtime_context_bundle_redacts_paths_and_secrets_and_limits_plan() { let secret = ["s", "k-1234567890abcdef"].concat(); let plan = AgentRuntimeToolPlan { thinking_summary: format!("检查 {root_display},凭据为 {secret}"), + plan_update: None, plan: (0..AGENT_RUNTIME_PLAN_STEP_LIMIT + 3) .map(|index| format!("步骤 {index} 读取 {root_display}")) .collect(), actions: Vec::new(), response: format!(r#"结果位于 {root_display},{{"credential":"{secret}"}}"#), }; + state.plan = plan.plan.clone(); let observations = vec![AgentRuntimeToolObservation { tool: "project.search".to_string(), status: "ok".to_string(), @@ -4818,7 +6001,7 @@ fn agent_runtime_context_compaction_preserves_completed_milestones_across_window fn agent_runtime_context_bundle_size_limit_includes_trailing_newline() { let root = unique_project_path(); init_local_game_project_at(&root, "project-1", "上下文大小项目").expect("project init"); - let state = start_game_creator_agent_runtime_task_at( + let mut state = start_game_creator_agent_runtime_task_at( &root, "design-director", "验证上下文大小边界", @@ -4830,12 +6013,14 @@ fn agent_runtime_context_bundle_size_limit_includes_trailing_newline() { .expect("start runtime state"); let plan = AgentRuntimeToolPlan { thinking_summary: "上下文大小边界".to_string(), + plan_update: None, plan: (0..AGENT_RUNTIME_PLAN_STEP_LIMIT) .map(|_| "p".repeat(180)) .collect(), actions: Vec::new(), response: String::new(), }; + state.plan = plan.plan.clone(); let context_tracker = AgentRuntimeContextWindowTracker::default(); let build_candidate = |content_diff_chars: usize| { let mut observations = (0..AGENT_RUNTIME_CONTEXT_OBSERVATION_LIMIT - 1) @@ -13019,7 +14204,7 @@ fn finalization_resume_recovers_persisted_assistant_without_runtime_state() { init_local_game_project_at(&root, "project-1", "最终回复状态投影丢失恢复项目") .expect("project init"); let run_id = "design-finalization-missing-runtime-state-run"; - let state = start_game_creator_agent_runtime_task_at( + let mut state = start_game_creator_agent_runtime_task_at( &root, "design-director", "在 Runtime state 丢失后恢复 finalization", @@ -13029,6 +14214,27 @@ fn finalization_resume_recovers_persisted_assistant_without_runtime_state() { vec!["保存最终回复".to_string()], ) .expect("start missing state finalization runtime"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "最终回复前结构化计划已完成".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "完成实现".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "完成验证".to_string(), + status: "completed".to_string(), + }, + ], + }, + ) + .expect("apply structured plan before missing state finalization"); + let expected_plan_revision = state.plan_revision; + let expected_plan_explanation = state.plan_explanation.clone(); + let expected_plan = state.plan.clone(); + let expected_plan_steps = state.plan_steps.clone(); let response = "state 丢失后仍不能重放的最终回复"; let outcome = finish_game_creator_agent_background_runtime_turn_with_checkpoint_at( &root, @@ -13081,6 +14287,14 @@ fn finalization_resume_recovers_persisted_assistant_without_runtime_state() { assert_eq!(runtime.phase, "completed"); assert_eq!(runtime.run_id, run_id); assert_eq!(runtime.last_response.as_deref(), Some(response)); + assert_eq!(runtime.plan_revision, expected_plan_revision); + assert_eq!(runtime.plan_explanation, expected_plan_explanation); + assert_eq!(runtime.plan, expected_plan); + assert_eq!(runtime.plan_steps, expected_plan_steps); + assert!(runtime + .plan_steps + .iter() + .all(|step| step.status == "completed")); let conversation = read_local_conversation_for_session_at( &root, Some("design-director"), @@ -13711,6 +14925,112 @@ fn finalization_resume_blocks_tampered_journal_identity() { fs::remove_dir_all(root).ok(); } +#[test] +fn finalization_resume_blocks_internally_consistent_incomplete_plan_snapshot() { + let root = unique_project_path(); + init_local_game_project_at(&root, "project-1", "最终回复未完成计划篡改项目") + .expect("project init"); + let run_id = "design-finalization-incomplete-plan-journal-run"; + let mut state = start_game_creator_agent_runtime_task_at( + &root, + "design-director", + "阻断绑定未完成计划的 finalization", + run_id, + "agent-background-task", + "准备最终回复", + vec!["旧计划占位".to_string()], + ) + .expect("start incomplete plan finalization runtime"); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: "最终回复前计划可信完成".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: "完成最终核对".to_string(), + status: "completed".to_string(), + }], + }, + ) + .expect("apply completed structured plan"); + let outcome = finish_game_creator_agent_background_runtime_turn_with_checkpoint_at( + &root, + state, + "未完成计划篡改前的 finalization 回复", + 0, + &[], + |checkpoint| { + if checkpoint == AgentRuntimeFinalizationCheckpoint::Prepared { + Err("injected-incomplete-plan-journal-crash".to_string()) + } else { + Ok(()) + } + }, + ) + .expect("prepared checkpoint injection is recoverable"); + assert!(matches!( + outcome, + AgentBackgroundFinalizationOutcome::Pending(_) + )); + + let journal_path = + game_creator_agent_runtime_finalization_path(&root, "design-director", run_id); + let mut journal = serde_json::from_str::( + &fs::read_to_string(&journal_path).expect("read structured finalization journal"), + ) + .expect("parse structured finalization journal"); + journal.plan_steps[0].status = "pending".to_string(); + journal.active_plan_step_index = None; + let plan_snapshot = serde_json::json!({ + "planRevision": journal.plan_revision, + "planExplanation": journal.plan_explanation, + "plan": journal.plan, + "planSteps": journal.plan_steps, + "activePlanStepIndex": journal.active_plan_step_index, + }); + journal.plan_snapshot_fingerprint = format!( + "{:x}", + Sha256::digest(serde_json::to_vec(&plan_snapshot).expect("serialize plan snapshot")) + ); + let finalization_identity = format!( + "{}\n{}\n{}\n{}\n{}\n{}\n{}\n{}", + journal.project_id, + journal.agent_id, + journal.session_id, + journal.run_id, + journal.response_fingerprint, + journal.response_revision, + journal.response_steer_cursor, + journal.plan_snapshot_fingerprint, + ); + journal.finalization_id = format!( + "agent-finalization-{}", + format!("{:x}", Sha256::digest(finalization_identity.as_bytes())) + .chars() + .take(32) + .collect::() + ); + fs::write( + &journal_path, + serde_json::to_string_pretty(&journal) + .expect("serialize internally consistent incomplete plan journal"), + ) + .expect("write internally consistent incomplete plan journal"); + + let resumed = resume_game_creator_agent_background_tasks_at(&root) + .expect("incomplete plan finalization must fail closed per agent"); + assert_eq!(resumed.len(), 1); + let runtime = read_game_creator_agent_runtime_at(&root, "design-director") + .expect("read blocked incomplete plan finalization runtime") + .state; + assert_eq!(runtime.phase, "finalizing"); + assert!(runtime.error.as_deref().is_some_and(|error| { + error.contains("finalization") && error.contains("已全部完成") + })); + assert!(journal_path.exists()); + + fs::remove_dir_all(root).ok(); +} + #[cfg(unix)] #[test] fn finalization_resume_rejects_symlinked_journal() { @@ -14229,6 +15549,7 @@ async fn stale_finalization_context_survives_restart_before_same_run_replanning( state.loop_iteration = 1; let mut plan = AgentRuntimeToolPlan { thinking_summary: "旧 revision 已经收束".to_string(), + plan_update: None, plan: Vec::new(), actions: Vec::new(), response: "重启后不得复用的旧回复".to_string(), @@ -14991,6 +16312,11 @@ fn agent_runtime_tool_plan_prompt_explains_named_verification_scripts_and_contex assert!(prompt.contains("只读任务填写被检查的现有文件")); assert!(prompt.contains("writeScopes 必须是互不重叠的项目内非私有相对目录")); assert!(prompt.contains("agent.action_history")); + assert!(prompt.contains("planUpdate")); + assert!(prompt.contains("in_progress")); + assert!(prompt.contains("所有必要步骤 completed 前不得给最终回复")); + assert!(prompt.contains("无需额外调用 project.checkpoint")); + assert!(!prompt.contains("批量修改前创建 project.checkpoint")); } #[tokio::test] @@ -16763,6 +18089,12 @@ async fn background_agent_runtime_can_confirm_and_continue_waiting_tool_actions( ); assert_eq!(persisted_delete.action.tool, "file.delete"); assert_eq!(persisted_delete.action_id, delete_pending.action_id); + assert_eq!( + persisted_delete.planned_repository_context_fingerprint, + build_repository_startup_context_at(&root) + .expect("build waiting delete repository context") + .fingerprint + ); assert_eq!( persisted_delete.project_revision_before, read_game_creator_agent_runtime_project_revision(&root) @@ -19483,12 +20815,25 @@ async fn background_agent_runtime_repairs_malformed_native_function_arguments() assert_eq!(repair_records.len(), 2); assert_eq!(repair_records[0]["attempt"], 1); assert_eq!(repair_records[0]["protocol"], "native_function"); - assert_eq!(repair_records[0]["callId"], "call-native-malformed-1"); assert_eq!(repair_records[1]["attempt"], 2); - assert_eq!(repair_records[1]["callId"], "call-native-malformed-2"); - assert!(repair_records + for (record, call_id) in repair_records .iter() - .all(|record| record["functionName"] == AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME)); + .zip(["call-native-malformed-1", "call-native-malformed-2"]) + { + assert_eq!( + record["callIdSha256"], + format!("{:x}", Sha256::digest(call_id.as_bytes())) + ); + assert_eq!( + record["functionNameSha256"], + format!( + "{:x}", + Sha256::digest(AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME.as_bytes()) + ) + ); + assert!(record.get("callId").is_none()); + assert!(record.get("functionName").is_none()); + } assert!(records.iter().any(|record| { record["recordType"] == "agent.runtime.tool_plan.protocol" && record["runId"] == run_id @@ -19580,10 +20925,21 @@ async fn background_agent_runtime_repairs_malformed_tool_plan_in_same_run() { repair_record["maxAttempts"], AGENT_RUNTIME_TOOL_PLAN_FORMAT_REPAIR_ATTEMPTS ); - assert!(repair_record["protocolError"] - .as_str() - .is_some_and(|error| error.contains("解析 Agent 工具计划失败"))); - assert_eq!(repair_record["responsePreview"], malformed_plan); + assert_eq!( + repair_record["responsePreviewSha256"], + format!("{:x}", Sha256::digest(malformed_plan.as_bytes())) + ); + assert_eq!( + repair_record["responsePreviewChars"], + malformed_plan.chars().count() + ); + assert_eq!( + repair_record["protocolErrorSha256"].as_str().map(str::len), + Some(64) + ); + assert!(repair_record["protocolErrorChars"].as_u64().is_some()); + assert!(repair_record.get("protocolError").is_none()); + assert!(repair_record.get("responsePreview").is_none()); assert!(!records.iter().any(|record| { record["recordType"] == "agent.runtime.background_task.failed" && record["runId"] == run_id })); @@ -25998,6 +27354,7 @@ fn agent_tool_plan_llm_response( fn agent_tool_plan_parser_accepts_native_function_arguments() { let arguments = serde_json::json!({ "thinkingSummary": "先读取项目索引", + "planUpdate": null, "plan": ["读取索引", "根据观察继续"], "actions": [{ "tool": "project.index", @@ -26029,6 +27386,29 @@ fn agent_tool_plan_parser_accepts_native_function_arguments() { assert_eq!(parsed.plan.actions[0].tool, "project.index"); } +#[test] +fn agent_tool_plan_parser_requires_plan_update_in_native_arguments() { + let arguments = serde_json::json!({ + "thinkingSummary": "错误省略结构化计划字段", + "plan": ["旧计划不应被 native function 接受"], + "actions": [], + "response": "" + }) + .to_string(); + let response = agent_tool_plan_llm_response( + "", + vec![platform_llm::LlmToolCall { + id: "call-missing-plan-update".to_string(), + name: AGENT_RUNTIME_TOOL_PLAN_FUNCTION_NAME.to_string(), + arguments, + }], + ); + + let error = parse_game_creator_agent_tool_plan_llm_response(&response) + .expect_err("native function must include planUpdate even when null"); + assert!(error.contains("必须显式包含 planUpdate")); +} + #[test] fn agent_tool_plan_parser_rejects_wrong_or_multiple_native_calls() { let valid_arguments = final_tool_plan_response("不应执行"); @@ -26042,7 +27422,8 @@ fn agent_tool_plan_parser_rejects_wrong_or_multiple_native_calls() { ); let wrong_error = parse_game_creator_agent_tool_plan_llm_response(&wrong_function) .expect_err("wrong function must be rejected"); - assert!(wrong_error.contains("实际调用 wrong_function")); + assert!(wrong_error.contains("实际调用了非预期函数")); + assert!(!wrong_error.contains("wrong_function")); let multiple_calls = agent_tool_plan_llm_response( "", @@ -32812,6 +34193,100 @@ fn start_agent_runtime_steer_fixture(root: &Path, run_id: &str) -> AgentRuntimeS .expect("start steer runtime") } +fn start_structured_plan_with_applied_steer_for_test( + root: &Path, + run_id: &str, + explanation: &str, + steps: Vec, + steer_id: &str, + steer_body: &str, +) -> AgentRuntimeState { + let mut state = start_agent_runtime_steer_fixture(root, run_id); + apply_agent_runtime_plan_update( + &mut state, + &AgentRuntimePlanUpdate { + explanation: explanation.to_string(), + steps, + }, + ) + .expect("apply structured plan before steer"); + write_game_creator_agent_runtime_state(root, &state) + .expect("persist structured plan before steer"); + let task = state.current_task.clone(); + let plan = AgentRuntimeToolPlan::default(); + let mut observations = Vec::new(); + let tracker = AgentRuntimeContextWindowTracker::default(); + persist_game_creator_agent_runtime_context( + root, + &state, + &task, + &plan, + &observations, + 0, + &tracker, + ) + .expect("persist structured plan context before steer"); + steer_game_creator_agent_runtime_task_at( + root, + &state.agent_id, + &state.session_id, + &state.run_id, + steer_id, + steer_body, + "test", + ) + .expect("queue structured plan steer"); + assert!(consume_game_creator_agent_runtime_steers( + root, + &mut state, + &task, + &plan, + &mut observations, + 0, + &tracker, + ) + .expect("consume structured plan steer")); + state +} + +fn assert_structured_plan_blocker_audit_for_test( + record: &Value, + plan_revision: u64, + completed: usize, + pending: usize, + in_progress: usize, + failed: usize, + total: usize, + incomplete_steps: &[(&str, &str)], + forbidden_text: &[&str], +) { + let summary = record["summary"] + .as_str() + .expect("structured plan blocker audit summary"); + let detail = record["detail"] + .as_str() + .expect("structured plan blocker audit detail"); + assert!(summary.contains(&format!("{completed}/{total}"))); + assert!(detail.contains(&format!("planRevision={plan_revision}"))); + assert!(detail.contains(&format!("completed={completed}"))); + assert!(detail.contains(&format!("pending={pending}"))); + assert!(detail.contains(&format!("inProgress={in_progress}"))); + assert!(detail.contains(&format!("failed={failed}"))); + assert!(detail.contains("stepStatusSha256=")); + for (status, title) in incomplete_steps { + let title_hash = format!("{:x}", Sha256::digest(title.as_bytes())); + assert_eq!(title_hash.len(), 64); + assert!(detail.contains(&format!("{status}:{title_hash}"))); + } + let encoded = serde_json::to_string(record).expect("serialize structured blocker audit"); + for forbidden in forbidden_text { + assert!( + !encoded.contains(forbidden), + "structured blocker audit leaked forbidden text: {forbidden}" + ); + } +} + fn read_agent_runtime_steer_jsonl( root: &Path, agent_id: &str, @@ -32827,6 +34302,509 @@ fn read_agent_runtime_steer_jsonl( .collect() } +#[test] +fn structured_plan_same_run_steer_preserves_monotonic_runtime_and_v3_context() { + let root = unique_project_path(); + let mut state = start_agent_runtime_steer_fixture(&root, "structured-plan-steer-run"); + let task = state.current_task.clone(); + let session_id = state.session_id.clone(); + let task_id = state.task_id.clone(); + let plan = AgentRuntimeToolPlan::default(); + let tracker = AgentRuntimeContextWindowTracker::default(); + let mut observations = Vec::new(); + + let initial_update = AgentRuntimePlanUpdate { + explanation: "已读取项目,开始实现键盘控制".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取项目".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "实现键盘控制".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "运行定向验证".to_string(), + status: "pending".to_string(), + }, + ], + }; + assert!(apply_agent_runtime_plan_update(&mut state, &initial_update) + .expect("apply pre-steer structured plan")); + let pre_steer_revision = state.plan_revision; + assert_eq!(pre_steer_revision, 1); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist pre-steer structured runtime"); + persist_game_creator_agent_runtime_context( + &root, + &state, + &task, + &plan, + &observations, + 0, + &tracker, + ) + .expect("persist pre-steer v3 context"); + let pre_steer_context = read_game_creator_agent_runtime_context_bundle(&root, &state) + .expect("read pre-steer context") + .expect("pre-steer context exists"); + assert_eq!( + pre_steer_context.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!(pre_steer_context.plan_revision, pre_steer_revision); + assert_eq!(pre_steer_context.plan_steps, state.plan_steps); + + steer_game_creator_agent_runtime_task_at( + &root, + &state.agent_id, + &state.session_id, + &state.run_id, + "structured-plan-steer-1", + "保留已完成的读取步骤,先运行定向验证,再继续实现键盘控制。", + "test", + ) + .expect("queue same-run steer"); + assert!(consume_game_creator_agent_runtime_steers( + &root, + &mut state, + &task, + &plan, + &mut observations, + 0, + &tracker, + ) + .expect("consume same-run steer")); + assert_eq!(state.plan_revision, pre_steer_revision); + assert!(state + .plan_steps + .iter() + .any(|step| { step.title == "读取项目" && step.status == "completed" })); + + let consumed_state = read_game_creator_agent_runtime_at(&root, &state.agent_id) + .expect("read consumed runtime") + .state; + assert_eq!(consumed_state.task_id, task_id); + assert_eq!(consumed_state.session_id, session_id); + assert_eq!(consumed_state.run_id, state.run_id); + assert_eq!(consumed_state.plan_revision, pre_steer_revision); + assert_eq!(consumed_state.plan_steps, state.plan_steps); + let consumed_context = read_game_creator_agent_runtime_context_bundle(&root, &consumed_state) + .expect("read consumed context") + .expect("consumed context exists"); + assert_eq!( + consumed_context.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!(consumed_context.plan_revision, consumed_state.plan_revision); + assert_eq!( + consumed_context.plan_explanation, + consumed_state.plan_explanation + ); + assert_eq!(consumed_context.plan, consumed_state.plan); + assert_eq!(consumed_context.plan_steps, consumed_state.plan_steps); + assert_eq!( + consumed_context.active_plan_step_index, + consumed_state.active_plan_step_index + ); + assert!(consumed_context.plan_revision >= pre_steer_context.plan_revision); + + state = consumed_state; + let rejected_revision = state.plan_revision; + let completed_regression = AgentRuntimePlanUpdate { + explanation: "错误地回退已完成步骤".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "读取项目".to_string(), + status: "pending".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "运行定向验证".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "实现键盘控制".to_string(), + status: "pending".to_string(), + }, + ], + }; + assert!( + apply_agent_runtime_plan_update(&mut state, &completed_regression) + .expect_err("same-run steer replan cannot regress completed steps") + .contains("不能让已完成步骤回退") + ); + assert_eq!(state.plan_revision, rejected_revision); + + let steered_update = AgentRuntimePlanUpdate { + explanation: "按追加要求重排未完成步骤".to_string(), + steps: vec![ + AgentRuntimePlanUpdateStep { + step: "运行定向验证".to_string(), + status: "in_progress".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: "实现键盘控制".to_string(), + status: "pending".to_string(), + }, + ], + }; + assert!(apply_agent_runtime_plan_update(&mut state, &steered_update) + .expect("apply post-steer structured plan")); + assert!(state.plan_revision > rejected_revision); + assert!(state + .plan_steps + .iter() + .any(|step| { step.title == "读取项目" && step.status == "completed" })); + write_game_creator_agent_runtime_state(&root, &state) + .expect("persist post-steer structured runtime"); + persist_game_creator_agent_runtime_context( + &root, + &state, + &task, + &plan, + &observations, + 0, + &tracker, + ) + .expect("persist post-steer v3 context"); + + let persisted_state = read_game_creator_agent_runtime_at(&root, &state.agent_id) + .expect("read post-steer runtime") + .state; + let persisted_context = read_game_creator_agent_runtime_context_bundle(&root, &persisted_state) + .expect("read post-steer context") + .expect("post-steer context exists"); + assert_eq!( + persisted_context.schema_version, + AGENT_RUNTIME_CONTEXT_BUNDLE_SCHEMA_VERSION + ); + assert_eq!( + persisted_context.plan_revision, + persisted_state.plan_revision + ); + assert_eq!( + persisted_context.plan_explanation, + persisted_state.plan_explanation + ); + assert_eq!(persisted_context.plan, persisted_state.plan); + assert_eq!(persisted_context.plan_steps, persisted_state.plan_steps); + assert_eq!( + persisted_context.active_plan_step_index, + persisted_state.active_plan_step_index + ); + assert!(persisted_context.plan_revision > consumed_context.plan_revision); + assert!(persisted_context + .plan_steps + .iter() + .any(|step| { step.title == "读取项目" && step.status == "completed" })); + + fs::remove_dir_all(root).ok(); +} + +#[test] +fn structured_plan_incomplete_normal_and_resumed_finalization_audits_are_redacted() { + let ordinary_root = unique_project_path(); + let ordinary_root_text = ordinary_root.to_string_lossy().into_owned(); + let ordinary_step_title = format!("STRUCTURED_PLAN_ORDINARY_STEP_TITLE {ordinary_root_text}"); + let ordinary_steer_body = + format!("STRUCTURED_PLAN_ORDINARY_STEER_BODY 只审计哈希 {ordinary_root_text}"); + let ordinary_run_id = "structured-plan-ordinary-finalization-audit-run"; + let ordinary_state = start_structured_plan_with_applied_steer_for_test( + &ordinary_root, + ordinary_run_id, + "普通 finalization 的未完成计划", + vec![ + AgentRuntimePlanUpdateStep { + step: "已完成普通准备".to_string(), + status: "completed".to_string(), + }, + AgentRuntimePlanUpdateStep { + step: ordinary_step_title.clone(), + status: "in_progress".to_string(), + }, + ], + "structured-plan-ordinary-steer", + &ordinary_steer_body, + ); + let ordinary_outcome = finish_game_creator_agent_background_runtime_turn_at( + &ordinary_root, + ordinary_state.clone(), + "未完成计划不得落盘的普通 finalization 回复", + 0, + &[], + ) + .expect("ordinary incomplete finalization is recoverable"); + let ordinary_blocker = match ordinary_outcome { + AgentBackgroundFinalizationOutcome::Stale(blocker) => blocker, + AgentBackgroundFinalizationOutcome::Completed(_) => { + panic!("ordinary incomplete plan must not complete finalization") + } + AgentBackgroundFinalizationOutcome::Pending(error) => { + panic!("ordinary incomplete plan must block before journal: {error}") + } + }; + assert_eq!(ordinary_blocker.tool, "runtime.plan_update"); + assert!(!ordinary_blocker.summary.contains(&ordinary_step_title)); + assert!(!ordinary_blocker + .detail + .as_deref() + .unwrap_or_default() + .contains(&ordinary_step_title)); + let ordinary_records = read_agent_db_records_for_test(&ordinary_root); + let ordinary_audit = ordinary_records + .iter() + .find(|record| { + record["recordType"] == "agent.runtime.background_task.completion_blocked" + && record["runId"] == ordinary_run_id + }) + .expect("ordinary structured plan completion-blocked audit"); + assert_structured_plan_blocker_audit_for_test( + ordinary_audit, + 1, + 1, + 0, + 1, + 0, + 2, + &[("in_progress", ordinary_step_title.as_str())], + &[ + ordinary_step_title.as_str(), + ordinary_steer_body.as_str(), + ordinary_root_text.as_str(), + ], + ); + let ordinary_agent_db = + fs::read_to_string(ordinary_root.join(".agent/agent.db")).expect("ordinary Agent DB"); + assert!(!ordinary_agent_db.contains(&ordinary_step_title)); + assert!(!ordinary_agent_db.contains(&ordinary_steer_body)); + assert!(!ordinary_agent_db.contains(&ordinary_root_text)); + let ordinary_conversation = read_local_conversation_for_session_at( + &ordinary_root, + Some(&ordinary_state.agent_id), + Some(&ordinary_state.session_id), + ) + .expect("read ordinary finalization conversation"); + assert_eq!( + ordinary_conversation + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 0 + ); + fs::remove_dir_all(&ordinary_root).ok(); + + let resumed_root = unique_project_path(); + let resumed_root_text = resumed_root.to_string_lossy().into_owned(); + let resumed_step_title = format!("STRUCTURED_PLAN_RESUMED_STEP_TITLE {resumed_root_text}"); + let resumed_steer_body = + format!("STRUCTURED_PLAN_RESUMED_STEER_BODY 恢复审计只留哈希 {resumed_root_text}"); + let resumed_run_id = "structured-plan-resumed-finalization-audit-run"; + let completed_state = start_structured_plan_with_applied_steer_for_test( + &resumed_root, + resumed_run_id, + "prepared journal 前计划已完成", + vec![AgentRuntimePlanUpdateStep { + step: "已完成恢复准备".to_string(), + status: "completed".to_string(), + }], + "structured-plan-resumed-steer", + &resumed_steer_body, + ); + let response = "恢复时计划变为未完成后不得写入的 assistant"; + let prepared = finish_game_creator_agent_background_runtime_turn_with_checkpoint_at( + &resumed_root, + completed_state.clone(), + response, + 0, + &[], + |checkpoint| { + if checkpoint == AgentRuntimeFinalizationCheckpoint::Prepared { + Err("injected-structured-plan-resumed-finalization".to_string()) + } else { + Ok(()) + } + }, + ) + .expect("prepare resumable finalization journal"); + assert!(matches!( + prepared, + AgentBackgroundFinalizationOutcome::Pending(_) + )); + let before_resume = read_local_conversation_for_session_at( + &resumed_root, + Some(&completed_state.agent_id), + Some(&completed_state.session_id), + ) + .expect("read conversation before resumed finalization"); + assert_eq!( + before_resume + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 0 + ); + + let mut incomplete_state = + read_game_creator_agent_runtime_at(&resumed_root, &completed_state.agent_id) + .expect("read prepared finalization state") + .state; + assert_eq!(incomplete_state.plan_revision, 1); + apply_agent_runtime_plan_update( + &mut incomplete_state, + &AgentRuntimePlanUpdate { + explanation: "恢复前新增必要核对步骤".to_string(), + steps: vec![AgentRuntimePlanUpdateStep { + step: resumed_step_title.clone(), + status: "pending".to_string(), + }], + }, + ) + .expect("make prepared finalization plan incomplete"); + assert_eq!(incomplete_state.plan_revision, 2); + append_game_creator_agent_runtime_task(&resumed_root, &incomplete_state) + .expect("append incomplete resumed finalization task"); + write_game_creator_agent_runtime_state(&resumed_root, &incomplete_state) + .expect("persist incomplete resumed finalization state"); + persist_game_creator_agent_runtime_context( + &resumed_root, + &incomplete_state, + &incomplete_state.current_task, + &AgentRuntimeToolPlan::default(), + &[], + usize::try_from(incomplete_state.loop_iteration).unwrap_or_default(), + &AgentRuntimeContextWindowTracker::default(), + ) + .expect("persist incomplete resumed finalization context"); + + let (request_sender, request_receiver) = mpsc::channel(); + let (response_sender, response_receiver) = mpsc::channel(); + let base_url = + spawn_interactive_mock_llm_server_with_capture(1, request_sender, response_receiver); + let _config_guard = write_test_local_config(format!( + r#"{{ + "agentLlm": {{ + "code-prototype": {{ + "apiKey": "code-key", + "baseUrl": {base_url:?}, + "model": "code-runtime-model", + "apiKind": "openai_responses" + }} + }} +}}"# + )); + let resumed = resume_game_creator_agent_background_tasks_at(&resumed_root) + .expect("resume incomplete structured finalization through public entry"); + assert_eq!(resumed.len(), 1); + assert_eq!(resumed[0].state.run_id, resumed_run_id); + assert_eq!(resumed[0].state.plan_revision, 2); + assert_eq!(resumed[0].state.plan_steps, incomplete_state.plan_steps); + let replanning_request = request_receiver + .recv_timeout(Duration::from_secs(3)) + .expect("resumed finalization same-run replanning request"); + assert!(replanning_request.contains("结构化计划:revision=2")); + assert!(replanning_request.contains("STRUCTURED_PLAN_RESUMED_STEP_TITLE")); + assert!(read_game_creator_agent_runtime_finalization_journal( + &resumed_root, + &completed_state.agent_id, + resumed_run_id, + ) + .expect("read discarded stale finalization journal") + .is_none()); + let after_resume = read_local_conversation_for_session_at( + &resumed_root, + Some(&completed_state.agent_id), + Some(&completed_state.session_id), + ) + .expect("read conversation after resumed finalization blocker"); + assert_eq!( + after_resume + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 0 + ); + let resumed_records = read_agent_db_records_for_test(&resumed_root); + let resumed_audit = resumed_records + .iter() + .find(|record| { + record["recordType"] == "agent.runtime.background_task.finalization_stale_recovered" + && record["runId"] == resumed_run_id + }) + .expect("resumed structured plan finalization audit"); + assert_structured_plan_blocker_audit_for_test( + resumed_audit, + 2, + 1, + 1, + 0, + 0, + 2, + &[("pending", resumed_step_title.as_str())], + &[ + resumed_step_title.as_str(), + resumed_steer_body.as_str(), + resumed_root_text.as_str(), + ], + ); + let resumed_agent_db = + fs::read_to_string(resumed_root.join(".agent/agent.db")).expect("resumed Agent DB"); + assert!(!resumed_agent_db.contains(&resumed_step_title)); + assert!(!resumed_agent_db.contains(&resumed_steer_body)); + assert!(!resumed_agent_db.contains(&resumed_root_text)); + + response_sender + .send( + serde_json::json!({ + "thinkingSummary": "恢复审计完成后显式完成剩余步骤", + "planUpdate": { + "explanation": "恢复后的必要核对已经完成", + "steps": [ + { "step": "已完成恢复准备", "status": "completed" }, + { "step": resumed_step_title.clone(), "status": "completed" } + ] + }, + "plan": [], + "actions": [], + "response": "恢复后的结构化计划已显式完成。" + }) + .to_string(), + ) + .expect("release resumed finalization replanning response"); + let completed = wait_for_agent_runtime_idle(&resumed_root, &completed_state.agent_id); + assert_eq!(completed.run_id, resumed_run_id); + assert_eq!(completed.phase, "completed"); + assert_eq!(completed.plan_revision, 3); + assert!(completed + .plan_steps + .iter() + .all(|step| step.status == "completed")); + let completed_conversation = read_local_conversation_for_session_at( + &resumed_root, + Some(&completed_state.agent_id), + Some(&completed_state.session_id), + ) + .expect("read completed resumed finalization conversation"); + assert_eq!( + completed_conversation + .messages + .iter() + .filter(|message| message.role == "assistant") + .count(), + 1 + ); + let completed_agent_db = + fs::read_to_string(resumed_root.join(".agent/agent.db")).expect("completed Agent DB"); + assert!(!completed_agent_db.contains(&resumed_step_title)); + assert!(!completed_agent_db.contains(&resumed_steer_body)); + assert!(!completed_agent_db.contains(&resumed_root_text)); + + fs::remove_dir_all(resumed_root).ok(); +} + #[test] fn agent_runtime_steer_is_idempotent_rejects_conflicts_and_enforces_capacity() { let root = unique_project_path(); diff --git a/apps/ai-game-creator-shell/src/App.tsx b/apps/ai-game-creator-shell/src/App.tsx index c109b7f0a..00a988f54 100644 --- a/apps/ai-game-creator-shell/src/App.tsx +++ b/apps/ai-game-creator-shell/src/App.tsx @@ -270,6 +270,8 @@ interface AgentRuntimeState { loopIteration?: number; maxLoopIterations?: number; toolActionBudget?: number; + planRevision?: number; + planExplanation?: string; plan: string[]; planSteps?: AgentRuntimePlanStep[]; activePlanStepIndex?: number | null; @@ -315,11 +317,12 @@ interface AgentRuntimePendingToolActionSummary { } interface AgentRuntimePlanStep { - index: number; - title: string; + step?: string; status: string; - detail: string | null; - updatedAt: number; + index?: number; + title?: string; + detail?: string | null; + updatedAt?: number; } interface AgentRuntimeTaskQueueSummary { @@ -664,6 +667,7 @@ function agentRuntimePlanStepsFromPlan(plan: string[]): AgentRuntimePlanStep[] { .filter((item) => item.trim().length > 0) .slice(0, 8) .map((title, index) => ({ + step: title, index, title, status: index === 0 ? 'active' : 'pending', @@ -672,39 +676,147 @@ function agentRuntimePlanStepsFromPlan(plan: string[]): AgentRuntimePlanStep[] { })); } +function agentRuntimePlanStepText(step: AgentRuntimePlanStep) { + return (step.step ?? step.title ?? '').trim(); +} + +function normalizeAgentRuntimePlanStep( + step: AgentRuntimePlanStep, + fallbackIndex: number, +): AgentRuntimePlanStep | null { + const text = agentRuntimePlanStepText(step); + if (!text) { + return null; + } + const index = + typeof step.index === 'number' && + Number.isFinite(step.index) && + step.index >= 0 + ? Math.trunc(step.index) + : fallbackIndex; + return { + ...step, + step: text, + index, + title: step.title?.trim() || text, + status: step.status?.trim() || 'pending', + detail: step.detail ?? null, + updatedAt: + typeof step.updatedAt === 'number' && Number.isFinite(step.updatedAt) + ? step.updatedAt + : 0, + }; +} + +function normalizeAgentRuntimePlanStepList(steps: AgentRuntimePlanStep[]) { + return steps + .map((step, index) => normalizeAgentRuntimePlanStep(step, index)) + .filter((step): step is AgentRuntimePlanStep => step !== null); +} + function normalizeAgentRuntimePlanSteps( state: AgentRuntimeState, previous?: AgentRuntimeState | null, ) { if (state.planSteps && state.planSteps.length > 0) { - return state.planSteps; + return normalizeAgentRuntimePlanStepList(state.planSteps); + } + if (state.planSteps !== undefined && state.planRevision !== undefined) { + return []; } if (previous?.planSteps && previous.planSteps.length > 0) { - return previous.planSteps; + return normalizeAgentRuntimePlanStepList(previous.planSteps); } return agentRuntimePlanStepsFromPlan(state.plan ?? []); } function normalizeAgentRuntimeActivePlanStepIndex( state: AgentRuntimeState, + planSteps: AgentRuntimePlanStep[], previous?: AgentRuntimeState | null, ) { if (state.activePlanStepIndex !== undefined) { - return state.activePlanStepIndex; + if (state.activePlanStepIndex === null) { + return null; + } + return Number.isFinite(state.activePlanStepIndex) && + state.activePlanStepIndex >= 0 + ? Math.trunc(state.activePlanStepIndex) + : null; } if (previous?.activePlanStepIndex !== undefined) { return previous.activePlanStepIndex; } - const activeStep = (state.planSteps ?? []).find( - (step) => step.status === 'active', + const activeStepIndex = planSteps.findIndex((step) => + ['active', 'in_progress', 'running'].includes(step.status), ); - return activeStep?.index ?? null; + if (activeStepIndex < 0) { + return null; + } + return planSteps[activeStepIndex]?.index ?? activeStepIndex; +} + +function sameAgentRuntimeRun( + state: AgentRuntimeState, + previous?: AgentRuntimeState | null, +) { + return Boolean( + previous && + previous.agentId === state.agentId && + previous.sessionId === state.sessionId && + previous.runId === state.runId, + ); +} + +function normalizeAgentRuntimePlanRevision( + revision: number | undefined, + previousRevision: number | undefined, +) { + if ( + typeof revision === 'number' && + Number.isFinite(revision) && + revision >= 0 + ) { + return Math.trunc(revision); + } + return previousRevision; +} + +function shouldReplaceAgentRuntimePlan( + state: AgentRuntimeState, + previous?: AgentRuntimeState | null, +) { + if (!previous) { + return true; + } + const previousRevision = normalizeAgentRuntimePlanRevision( + previous.planRevision, + undefined, + ); + if (previousRevision === undefined) { + return true; + } + const nextRevision = normalizeAgentRuntimePlanRevision( + state.planRevision, + undefined, + ); + return nextRevision !== undefined && nextRevision >= previousRevision; } function normalizeAgentRuntimeState( state: AgentRuntimeState, previous?: AgentRuntimeState | null, ): AgentRuntimeState { + const previousPlanState = sameAgentRuntimeRun(state, previous) + ? previous + : null; + const replacePlan = shouldReplaceAgentRuntimePlan(state, previousPlanState); + const planState = replacePlan ? state : previousPlanState!; + const planFallbackState = replacePlan ? previousPlanState : null; + const planSteps = normalizeAgentRuntimePlanSteps( + planState, + planFallbackState, + ); return { ...state, currentGoal: state.currentGoal ?? state.currentTask ?? '', @@ -714,10 +826,18 @@ function normalizeAgentRuntimeState( maxLoopIterations: state.maxLoopIterations ?? previous?.maxLoopIterations ?? 3, toolActionBudget: state.toolActionBudget ?? previous?.toolActionBudget ?? 3, - planSteps: normalizeAgentRuntimePlanSteps(state, previous), + plan: planState.plan ?? planFallbackState?.plan ?? [], + planRevision: normalizeAgentRuntimePlanRevision( + planState.planRevision, + planFallbackState?.planRevision, + ), + planExplanation: + planState.planExplanation ?? planFallbackState?.planExplanation, + planSteps, activePlanStepIndex: normalizeAgentRuntimeActivePlanStepIndex( - state, - previous, + planState, + planSteps, + planFallbackState, ), recentToolCalls: state.recentToolCalls ?? previous?.recentToolCalls ?? [], toolPolicy: state.toolPolicy ?? @@ -745,6 +865,32 @@ function normalizeAgentRuntimeState( }; } +function mergeAgentRuntimeStateIntoMap( + current: Record, + incoming: AgentRuntimeState, + preserveNewerRun: boolean, +) { + const previous = Object.values(current).find( + (runtime) => runtime?.agentId === incoming.agentId, + ); + const mergedRuntime = + preserveNewerRun && + previous && + !sameAgentRuntimeRun(incoming, previous) && + previous.updatedAt >= incoming.updatedAt + ? previous + : normalizeAgentRuntimeState(incoming, previous); + const next = { ...current }; + for (const [key, runtime] of Object.entries(next)) { + if (runtime?.agentId === incoming.agentId) { + delete next[key]; + } + } + next[mergedRuntime.agentId] = mergedRuntime; + next[mergedRuntime.taskId] = mergedRuntime; + return next; +} + function agentRuntimeStateFromResult( result: AgentRuntimeResult, previous?: AgentRuntimeState | null, @@ -966,8 +1112,12 @@ function formatAgentRuntimeLoopProgress( }`; } -function formatAgentRuntimePlanStep(step: AgentRuntimePlanStep) { - return `#${step.index + 1} ${step.status} · ${step.title}${ +function formatAgentRuntimePlanStep( + step: AgentRuntimePlanStep, + fallbackIndex = 0, +) { + const index = step.index ?? fallbackIndex; + return `#${index + 1} ${step.status} · ${agentRuntimePlanStepText(step)}${ step.detail ? ` · ${step.detail}` : '' }`; } @@ -976,9 +1126,20 @@ function agentRuntimeActivePlanStep( runtime: Pick, ) { const steps = runtime.planSteps ?? []; + const activePlanStepIndex = runtime.activePlanStepIndex; + if (activePlanStepIndex !== null && activePlanStepIndex !== undefined) { + const indexedStep = + steps.find((step) => step.index === activePlanStepIndex) ?? + steps[activePlanStepIndex]; + if (indexedStep) { + return indexedStep; + } + } return ( - steps.find((step) => step.index === runtime.activePlanStepIndex) ?? - steps.find((step) => step.status === 'active') ?? + steps.find((step) => + ['active', 'in_progress', 'running'].includes(step.status), + ) ?? + steps.find((step) => step.status === 'pending') ?? null ); } @@ -1089,7 +1250,11 @@ function AgentRuntimeStatusPanel({ ).reverse(); const recentToolCalls = (runtime.recentToolCalls ?? []).slice(-3).reverse(); const recentTasks = (runtime.recentTasks ?? []).slice(-3).reverse(); - const planSteps = (runtime.planSteps ?? []).slice(0, 5); + const planSteps = (runtime.planSteps ?? []).slice(0, 8); + const hasPersistentPlan = + runtime.planRevision !== undefined || + Boolean(runtime.planExplanation) || + planSteps.length > 0; const toolPolicy = runtime.toolPolicy; const taskQueueSummary = formatAgentRuntimeTaskQueue(runtime.taskQueue); const loopProgress = formatAgentRuntimeLoopProgress(runtime); @@ -1233,14 +1398,20 @@ function AgentRuntimeStatusPanel({ : ''} ) : null} - {planSteps.length > 0 ? ( + {hasPersistentPlan ? (
计划进度 - {planSteps.map((step) => ( + {(runtime.planRevision ?? 0) > 0 ? ( + {`计划修订号:${runtime.planRevision}`} + ) : null} + {runtime.planExplanation ? ( + {`计划说明:${runtime.planExplanation}`} + ) : null} + {planSteps.map((step, index) => ( - {formatAgentRuntimePlanStep(step)} + {formatAgentRuntimePlanStep(step, index)} ))}
@@ -1796,6 +1967,58 @@ function projectSupervisorRuntimeStatusLabel( return '分析'; } +function projectSupervisorCollaboratingAgentCount( + supervisorRuntime: AgentRuntimeState | null, + runtimeByAgentId: Record, +) { + if (!supervisorRuntime?.runId) { + return 0; + } + const agentIds = new Set(); + for (const runtime of Object.values(runtimeByAgentId)) { + if ( + !runtime || + runtime.agentId === PROJECT_SUPERVISOR_AGENT_ID || + !['agent-delegate', 'agent-delegate-retry'].includes(runtime.source) || + runtime.parentAgentId !== PROJECT_SUPERVISOR_AGENT_ID || + runtime.parentRunId !== supervisorRuntime.runId + ) { + continue; + } + agentIds.add(runtime.agentId); + } + return agentIds.size; +} + +function formatProjectSupervisorCompactProgress( + runtime: AgentRuntimeState, + collaboratingAgentCount: number, +) { + const planSteps = (runtime.planSteps ?? []).filter( + (step) => agentRuntimePlanStepText(step).length > 0, + ); + const completedPlanStepCount = planSteps.filter( + (step) => step.status === 'completed', + ).length; + const activePlanStep = agentRuntimeActivePlanStep(runtime); + const currentPlanStep = activePlanStep + ? agentRuntimePlanStepText(activePlanStep) + : planSteps.length > 0 && completedPlanStepCount === planSteps.length + ? '已完成' + : '暂无'; + const waitingOn = + runtime.waitingOn?.trim() || agentRuntimeWaitingOnFromPhase(runtime.phase); + const nextStep = + runtime.nextStep?.trim() || agentRuntimeNextStepFromPhase(runtime.phase); + return [ + `计划完成:${completedPlanStepCount}/${planSteps.length}`, + `当前步骤:${currentPlanStep}`, + `等待:${waitingOn}`, + `下一步:${nextStep}`, + `专业 Agent 协作:${collaboratingAgentCount}`, + ].join(' · '); +} + function isTransientProjectOpenMessage( message: ChatMessage, projectPath: string, @@ -12629,7 +12852,10 @@ export function deriveAgentStatusCards( runtimeMaxLoopIterations: runtime?.maxLoopIterations ?? null, runtimeToolActionBudget: runtime?.toolActionBudget ?? null, runtimeActivePlanStep: activePlanStep - ? formatAgentRuntimePlanStep(activePlanStep) + ? formatAgentRuntimePlanStep( + activePlanStep, + runtime?.planSteps?.indexOf(activePlanStep) ?? 0, + ) : null, runtimeTaskQueue: runtime?.taskQueue ?? null, runtimeRecentTasks: runtime?.recentTasks ?? [], @@ -14454,9 +14680,14 @@ export function App() { const nextRuntime = agentRuntimeStateFromResult(payload.runtime); if (payload.agentId === PROJECT_SUPERVISOR_AGENT_ID) { const expectedSessionId = projectSupervisorSessionIdRef.current; + const currentRuntime = projectSupervisorRuntimeRef.current; if ( - expectedSessionId !== null && - nextRuntime.sessionId !== expectedSessionId + nextRuntime.agentId !== PROJECT_SUPERVISOR_AGENT_ID || + nextRuntime.runId !== payload.runId || + (expectedSessionId !== null && + nextRuntime.sessionId !== expectedSessionId) || + (currentRuntime !== null && + !sameAgentRuntimeRun(nextRuntime, currentRuntime)) ) { return; } @@ -23072,15 +23303,9 @@ export function App() { if (!runtime) { return; } - setAgentRuntimeById((current) => { - const previous = current[runtime.agentId] ?? current[runtime.taskId]; - const mergedRuntime = normalizeAgentRuntimeState(runtime, previous); - return { - ...current, - [mergedRuntime.agentId]: mergedRuntime, - [mergedRuntime.taskId]: mergedRuntime, - }; - }); + setAgentRuntimeById((current) => + mergeAgentRuntimeStateIntoMap(current, runtime, false), + ); } async function refreshAgentRuntimes( @@ -23173,13 +23398,16 @@ export function App() { if (localProjectPathRef.current !== nextProjectPath) { return; } - const nextRuntimeById: Record = {}; + const nextRuntimes: AgentRuntimeState[] = []; for (const runtimeResult of runtimes) { - const runtime = agentRuntimeStateFromResult(runtimeResult); - nextRuntimeById[runtime.agentId] = runtime; - nextRuntimeById[runtime.taskId] = runtime; + nextRuntimes.push(agentRuntimeStateFromResult(runtimeResult)); } - setAgentRuntimeById(nextRuntimeById); + setAgentRuntimeById((current) => + nextRuntimes.reduce( + (next, runtime) => mergeAgentRuntimeStateIntoMap(next, runtime, true), + current, + ), + ); } catch { if (localProjectPathRef.current === nextProjectPath) { setAgentRuntimeById({}); @@ -23389,12 +23617,16 @@ export function App() { projectSupervisorRuntimeError, ); const projectSupervisorStatusDetail = - projectSupervisorRuntimeError || - projectSupervisorRuntime?.error || - (projectSupervisorStatus && projectSupervisorStatus !== '已完成' - ? projectSupervisorRuntime?.currentAction - : '') || - ''; + projectSupervisorRuntimeError || projectSupervisorRuntime?.error || ''; + const projectSupervisorCompactProgress = projectSupervisorRuntime + ? formatProjectSupervisorCompactProgress( + projectSupervisorRuntime, + projectSupervisorCollaboratingAgentCount( + projectSupervisorRuntime, + agentRuntimeById, + ), + ) + : ''; const projectSupervisorPendingToolAction = projectSupervisorRuntime?.pendingToolAction ?? null; const visibleAgentConversationMessages = latestVisibleItems( @@ -24026,6 +24258,11 @@ export function App() { {projectSupervisorStatusDetail ? ( {projectSupervisorStatusDetail} ) : null} + {projectSupervisorCompactProgress ? ( + + {projectSupervisorCompactProgress} + + ) : null} {projectSupervisorPendingToolAction ? (
>; supervisorMessages?: Array>; initialRuntime?: Record; + runtimeMapLoader?: () => Promise>>; } = {}) { const manifest = createGameCreationAppManifest( 'local-project-draft', @@ -177,9 +179,9 @@ function createProjectSupervisorRuntimeHarness({ }); const runtimeResult = (state = currentRuntime) => ({ state, - sessionPath: `${projectPath}/.agent/runtime/agents/project-supervisor.json`, - eventPath: `${projectPath}/.agent/runtime/events/project-supervisor.jsonl`, - taskPath: `${projectPath}/.agent/runtime/tasks/project-supervisor.jsonl`, + sessionPath: `${projectPath}/.agent/runtime/agents/${String(state.agentId)}.json`, + eventPath: `${projectPath}/.agent/runtime/events/${String(state.agentId)}.jsonl`, + taskPath: `${projectPath}/.agent/runtime/tasks/${String(state.agentId)}.jsonl`, taskQueue: { total: state.status === 'idle' ? 0 : 1, pending: state.status === 'pending' ? 1 : 0, @@ -282,6 +284,10 @@ function createProjectSupervisorRuntimeHarness({ } return runtimeResult(); } + if (command === 'read_game_creator_agent_runtimes') { + const states = runtimeMapLoader ? await runtimeMapLoader() : []; + return states.map((state) => runtimeResult(state)); + } if (command === 'start_game_creator_agent_runtime_task') { const runId = String(args?.runId ?? ''); currentSupervisorMessages.push( @@ -397,6 +403,18 @@ function createProjectSupervisorRuntimeHarness({ }, }); }, + emitAgentRuntime(state: Record) { + runtimeUpdateHandler?.({ + payload: { + projectPath, + agentId: String(state.agentId ?? ''), + runId: String(state.runId ?? ''), + status: String(state.status ?? ''), + phase: String(state.phase ?? ''), + runtime: runtimeResult(state), + }, + }); + }, }; } @@ -3463,6 +3481,163 @@ describe('AI 游戏创作 App 界面边界', () => { ]); }); + it('normalizes and restores the V1.17 persistent plan snapshot after refresh', async () => { + const sessionId = 'agent-session-plan-v117'; + const planSteps = [ + { step: '读取项目上下文', status: 'completed' }, + { title: '确认目标与约束', status: 'completed' }, + { step: '补齐核心玩法', status: 'in_progress' }, + { step: '实现交互反馈', status: 'pending' }, + { step: '运行定向验证', status: 'pending' }, + { step: '检查移动端布局', status: 'pending' }, + { step: '整理变更摘要', status: 'pending' }, + { step: '准备最终回复', status: 'pending' }, + { step: '不应展示的第九步', status: 'pending' }, + ]; + const runtimeState = { + schemaVersion: 'game-creator-agent-runtime.v1', + agentId: 'design-director', + taskId: 'design-director', + sessionId, + runId: 'runtime-plan-v117', + source: 'agent-background-task', + status: 'running', + phase: 'action', + currentTask: '持续更新本轮执行计划', + currentGoal: '完成首版玩法实现', + currentAction: '补齐核心玩法', + waitingOn: '核心玩法实现结果', + nextStep: '运行定向验证', + loopIteration: 3, + maxLoopIterations: 6, + toolActionBudget: 3, + planRevision: 7, + planExplanation: '根据最新项目观察调整实现与验证顺序', + plan: planSteps.map((step) => step.step ?? step.title ?? ''), + planSteps, + activePlanStepIndex: 2, + observations: [], + allowedTools: ['file.read', 'file.patch', 'project.verify'], + lastResponse: null, + error: null, + updatedAt: 7000, + }; + const runtimeResult = () => ({ + state: { + ...runtimeState, + planSteps: runtimeState.planSteps.map((step) => ({ ...step })), + }, + sessionPath: + '/tmp/authorized-game/.agent/runtime/agents/design-director.json', + eventPath: + '/tmp/authorized-game/.agent/runtime/events/design-director.jsonl', + taskPath: + '/tmp/authorized-game/.agent/runtime/tasks/design-director.jsonl', + taskQueue: { + total: 1, + pending: 0, + running: 1, + completed: 0, + failed: 0, + latestRunId: runtimeState.runId, + updatedAt: runtimeState.updatedAt, + }, + recentEvents: [], + recentTasks: [], + }); + let runtimeReadCount = 0; + const invoke = vi.fn(async (command: string) => { + if (command === 'check_game_creator_llm_config') { + return { + configured: true, + apiKeyPresent: true, + baseUrl: 'https://llm.example.test/v1', + model: 'gpt-5.5', + apiKind: 'openai_responses', + reasoningEffort: 'high', + stream: true, + error: null, + agents: [], + }; + } + if (command === 'resume_game_creator_agent_runtime_tasks') { + return []; + } + if (command === 'list_game_creator_agent_sessions') { + return { + path: '/tmp/authorized-game/.agent/conversations/agents/design-director/sessions.json', + agentId: 'design-director', + activeSessionId: sessionId, + sessions: [ + { + sessionId, + title: '持久计划验证', + createdAt: 6000, + updatedAt: 7000, + archivedAt: null, + messageCount: 0, + legacy: false, + }, + ], + }; + } + if (command === 'read_local_conversation') { + return { + path: `/tmp/authorized-game/.agent/conversations/agents/design-director/sessions/${sessionId}.jsonl`, + agentId: 'design-director', + sessionId, + messages: [], + }; + } + if (command === 'read_game_creator_agent_runtime') { + runtimeReadCount += 1; + return runtimeResult(); + } + throw new Error(`unexpected invoke ${command}`); + }); + window.__TAURI__ = { core: { invoke } }; + renderLauncherAgentChatAt('/?agent-chat'); + + fireEvent.change(screen.getByLabelText('Agent 聊天项目目录'), { + target: { value: '/tmp/authorized-game' }, + }); + fireEvent.click(screen.getByRole('button', { name: '读取历史' })); + + const planPanel = await screen.findByLabelText('Agent 计划进度'); + expect(within(planPanel).getByText('计划修订号:7')).not.toBeNull(); + expect( + within(planPanel).getByText( + '计划说明:根据最新项目观察调整实现与验证顺序', + ), + ).not.toBeNull(); + expect(within(planPanel).getAllByText(/^#\d+ /)).toHaveLength(8); + expect( + within(planPanel).getByText('#2 completed · 确认目标与约束'), + ).not.toBeNull(); + expect( + within(planPanel).getByText('#3 in_progress · 补齐核心玩法'), + ).not.toBeNull(); + expect(within(planPanel).queryByText(/不应展示的第九步/)).toBeNull(); + const firstSnapshot = planPanel.textContent; + const runtimeReadsBeforeRefresh = runtimeReadCount; + + fireEvent.click( + within(screen.getByLabelText('Agent Runtime 操作')).getByRole('button', { + name: '刷新状态', + }), + ); + + await waitFor(() => { + expect(runtimeReadCount).toBeGreaterThan(runtimeReadsBeforeRefresh); + }); + const refreshedPlanPanel = screen.getByLabelText('Agent 计划进度'); + expect(refreshedPlanPanel.textContent).toBe(firstSnapshot); + expect(within(refreshedPlanPanel).getAllByText(/^#\d+ /)).toHaveLength(8); + expect( + within(refreshedPlanPanel).queryByText(/不应展示的第九步/), + ).toBeNull(); + }); + it('does not sync an old runtime reply into another Agent session', async () => { const activeSessionId = 'agent-session-design-active'; const archivedSessionId = 'agent-session-design-archived'; @@ -23819,6 +23994,334 @@ describe('AI 游戏创作 App 界面边界', () => { expect(screen.getAllByText('做一个反弹弹幕厨房游戏')).toHaveLength(1); }); + it('keeps Project Supervisor plan progress compact on the ordinary chat surface', async () => { + const supervisorRunId = 'supervisor-plan-run'; + const harness = createProjectSupervisorRuntimeHarness({ + initialRuntime: { + runId: supervisorRunId, + status: 'running', + phase: 'waiting-for-delegate-receipts', + currentTask: '协调专业 Agent 完成首版方案', + currentAction: '内部控制台:轮询 durable delivery claim', + waitingOn: '专业 Agent 回执', + nextStep: '汇总专业 Agent 结果', + planRevision: 4, + planExplanation: '内部计划说明不应出现在普通聊天', + plan: ['确认目标', '并行委派专业 Agent', '汇总并回复'], + planSteps: [ + { step: '确认目标', status: 'completed' }, + { step: '并行委派专业 Agent', status: 'active' }, + { step: '汇总并回复', status: 'pending' }, + ], + activePlanStepIndex: 1, + }, + }); + window.__TAURI__ = { + core: { invoke: harness.invoke }, + event: { listen: harness.listen }, + }; + renderAppAt('/'); + await openMainProject(harness.projectPath); + expect( + await screen.findByText('项目总控 Agent · 等待专业 Agent'), + ).not.toBeNull(); + await waitFor(() => { + expect(harness.listen).toHaveBeenCalledWith( + 'game-creator-agent-runtime-update', + expect.any(Function), + ); + }); + + const delegatedRuntime = (agentId: string, delegationId: string) => + harness.runtimeState({ + agentId, + taskId: agentId, + sessionId: `session-${agentId}`, + runId: `run-${agentId}`, + source: 'agent-delegate', + parentAgentId: 'project-supervisor', + parentRunId: supervisorRunId, + delegationId, + status: 'running', + phase: 'action', + currentTask: `执行 ${agentId} 专业任务`, + currentAction: '执行专业工具动作', + waitingOn: '工具结果', + nextStep: '返回专业结论', + updatedAt: 5000, + }); + await act(async () => { + const designRuntime = delegatedRuntime( + 'design-director', + 'delegation-design', + ); + harness.emitAgentRuntime(designRuntime); + harness.emitAgentRuntime(designRuntime); + harness.emitAgentRuntime( + delegatedRuntime('art-asset-plan', 'delegation-art'), + ); + harness.emitAgentRuntime({ + ...delegatedRuntime('dynamic-review-child', 'isolated-review'), + source: 'agent-isolated-child', + }); + }); + + const compactProgress = await screen.findByLabelText('项目总控 Agent 进度'); + expect(compactProgress.textContent).toBe( + '计划完成:1/3 · 当前步骤:并行委派专业 Agent · 等待:专业 Agent 回执 · 下一步:汇总专业 Agent 结果 · 专业 Agent 协作:2', + ); + const supervisorPanel = screen.getByLabelText('项目总控 Agent 状态'); + expect(supervisorPanel.textContent).not.toContain('计划修订号:4'); + expect(supervisorPanel.textContent).not.toContain( + '内部计划说明不应出现在普通聊天', + ); + expect(supervisorPanel.textContent).not.toContain( + '内部控制台:轮询 durable delivery claim', + ); + expect(within(supervisorPanel).queryByText(/#2 active/)).toBeNull(); + expect( + within(supervisorPanel).queryByLabelText('Agent 计划进度'), + ).toBeNull(); + }); + + it('keeps completed Supervisor steps when a lower planRevision arrives out of order', async () => { + const runId = 'supervisor-revision-run'; + const harness = createProjectSupervisorRuntimeHarness({ + initialRuntime: { + runId, + status: 'running', + phase: 'action', + planRevision: 5, + planExplanation: '初始计划', + plan: ['确认目标', '完成实现', '收束验证'], + planSteps: [ + { step: '确认目标', status: 'completed' }, + { step: '完成实现', status: 'in_progress' }, + { step: '收束验证', status: 'pending' }, + ], + activePlanStepIndex: 1, + waitingOn: '实现结果', + nextStep: '收束验证', + updatedAt: 5000, + }, + }); + window.__TAURI__ = { + core: { invoke: harness.invoke }, + event: { listen: harness.listen }, + }; + renderAppAt('/'); + await openMainProject(harness.projectPath); + await waitFor(() => { + expect(harness.listen).toHaveBeenCalledWith( + 'game-creator-agent-runtime-update', + expect.any(Function), + ); + }); + + await act(async () => { + harness.emitRuntime( + harness.runtimeState({ + runId, + status: 'running', + phase: 'action', + planRevision: 6, + planExplanation: '实现已完成,进入验证', + plan: ['确认目标', '完成实现', '收束验证'], + planSteps: [ + { step: '确认目标', status: 'completed' }, + { step: '完成实现', status: 'completed' }, + { step: '收束验证', status: 'in_progress' }, + ], + activePlanStepIndex: 2, + waitingOn: '验证结果', + nextStep: '整理最终回复', + updatedAt: 6000, + }), + ); + }); + expect( + (await screen.findByLabelText('项目总控 Agent 进度')).textContent, + ).toContain('计划完成:2/3 · 当前步骤:收束验证'); + + await act(async () => { + harness.emitRuntime( + harness.runtimeState({ + runId, + status: 'running', + phase: 'waiting-for-delegate-receipts', + planRevision: 5, + planExplanation: '迟到的旧计划', + plan: ['确认目标', '完成实现', '收束验证'], + planSteps: [ + { step: '确认目标', status: 'completed' }, + { step: '完成实现', status: 'in_progress' }, + { step: '收束验证', status: 'pending' }, + ], + activePlanStepIndex: 1, + waitingOn: '迟到读取仍报告等待回执', + nextStep: '迟到读取的运行摘要', + updatedAt: 7000, + }), + ); + }); + + expect( + await screen.findByText('项目总控 Agent · 等待专业 Agent'), + ).not.toBeNull(); + expect(screen.getByLabelText('项目总控 Agent 进度').textContent).toBe( + '计划完成:2/3 · 当前步骤:收束验证 · 等待:迟到读取仍报告等待回执 · 下一步:迟到读取的运行摘要 · 专业 Agent 协作:0', + ); + }); + + it('clears the old plan for a new Supervisor run and ignores a late old-run event', async () => { + const oldRunId = 'supervisor-completed-old-run'; + const harness = createProjectSupervisorRuntimeHarness({ + initialRuntime: { + runId: oldRunId, + status: 'completed', + phase: 'completed', + currentTask: '旧 run 任务', + planRevision: 8, + planExplanation: '旧 run 已完成计划', + plan: ['旧 run 设计', '旧 run 验证'], + planSteps: [ + { step: '旧 run 设计', status: 'completed' }, + { step: '旧 run 验证', status: 'completed' }, + ], + activePlanStepIndex: null, + updatedAt: 5000, + }, + }); + window.__TAURI__ = { + core: { invoke: harness.invoke }, + event: { listen: harness.listen }, + }; + renderAppAt('/'); + await openMainProject(harness.projectPath); + expect(await screen.findByText('项目总控 Agent · 已完成')).not.toBeNull(); + + submitChat('开始真正的新一轮实现'); + await waitFor(() => { + expect( + harness.invoke.mock.calls.filter( + ([command]) => command === 'start_game_creator_agent_runtime_task', + ), + ).toHaveLength(1); + }); + const startCall = harness.invoke.mock.calls.find( + ([command]) => command === 'start_game_creator_agent_runtime_task', + ); + const newRunId = String(startCall?.[1]?.runId ?? ''); + expect(newRunId).not.toBe(''); + expect(newRunId).not.toBe(oldRunId); + expect(await screen.findByText('项目总控 Agent · 分析')).not.toBeNull(); + expect(screen.getByLabelText('项目总控 Agent 进度').textContent).toContain( + '计划完成:0/0 · 当前步骤:暂无', + ); + + await act(async () => { + harness.emitRuntime( + harness.runtimeState({ + runId: oldRunId, + status: 'running', + phase: 'waiting-for-delegate-receipts', + currentTask: '迟到的旧 run', + planRevision: 9, + planExplanation: '迟到旧 run 不得覆盖新 run', + plan: ['迟到旧步骤'], + planSteps: [{ step: '迟到旧步骤', status: 'in_progress' }], + activePlanStepIndex: 0, + waitingOn: '旧 run 专业 Agent', + nextStep: '旧 run 摘要', + updatedAt: 9999, + }), + ); + }); + + expect(screen.getByText('项目总控 Agent · 分析')).not.toBeNull(); + const progress = screen.getByLabelText('项目总控 Agent 进度'); + expect(progress.textContent).toContain('计划完成:0/0 · 当前步骤:暂无'); + expect(progress.textContent).not.toContain('迟到旧步骤'); + expect(progress.textContent).not.toContain('旧 run 专业 Agent'); + }); + + it('preserves a newer runtime event when an older full runtime map refresh resolves', async () => { + const supervisorRunId = 'supervisor-runtime-map-run'; + let resolveRuntimeMap!: (states: Array>) => void; + const runtimeMap = new Promise>>( + (resolve) => { + resolveRuntimeMap = resolve; + }, + ); + const harness = createProjectSupervisorRuntimeHarness({ + initialRuntime: { + runId: supervisorRunId, + status: 'running', + phase: 'action', + updatedAt: 5000, + }, + runtimeMapLoader: () => runtimeMap, + }); + window.__TAURI__ = { + core: { invoke: harness.invoke }, + event: { listen: harness.listen }, + }; + renderAppAt('/'); + await openMainProject(harness.projectPath); + await waitFor(() => { + expect(harness.invoke).toHaveBeenCalledWith( + 'read_game_creator_agent_runtimes', + { projectPath: harness.projectPath }, + ); + expect(harness.listen).toHaveBeenCalledWith( + 'game-creator-agent-runtime-update', + expect.any(Function), + ); + }); + + const currentDelegatedRuntime = harness.runtimeState({ + agentId: 'design-director', + taskId: 'design-director', + sessionId: 'design-current-session', + runId: 'design-current-run', + source: 'agent-delegate', + parentAgentId: 'project-supervisor', + parentRunId: supervisorRunId, + delegationId: 'delegation-current', + status: 'running', + phase: 'action', + updatedAt: 9000, + }); + await act(async () => { + harness.emitAgentRuntime(currentDelegatedRuntime); + }); + expect( + (await screen.findByLabelText('项目总控 Agent 进度')).textContent, + ).toContain('专业 Agent 协作:1'); + + const staleDelegatedRuntime = harness.runtimeState({ + agentId: 'design-director', + taskId: 'design-director', + sessionId: 'design-old-session', + runId: 'design-old-run', + source: 'agent-delegate', + parentAgentId: 'project-supervisor', + parentRunId: 'supervisor-old-run', + delegationId: 'delegation-old', + status: 'completed', + phase: 'completed', + updatedAt: 8000, + }); + await act(async () => { + resolveRuntimeMap([staleDelegatedRuntime]); + await runtimeMap; + }); + + expect(screen.getByLabelText('项目总控 Agent 进度').textContent).toContain( + '专业 Agent 协作:1', + ); + }); + it('steers the running Project Supervisor run without changing run or Session', async () => { const harness = createProjectSupervisorRuntimeHarness(); window.__TAURI__ = { diff --git a/docs/project-memory/shared-memory/decision-log.md b/docs/project-memory/shared-memory/decision-log.md index 57dbb4ba1..ff7a5d673 100644 --- a/docs/project-memory/shared-memory/decision-log.md +++ b/docs/project-memory/shared-memory/decision-log.md @@ -4563,6 +4563,20 @@ - UI:普通用户只看到总控 Agent 的紧凑状态、等待对象、协作数量、安全确认和唯一最终回复;不展示内部工具计划、原始 observation、动态 child 或开发控制台。Runner 未提供 token delta 时只显示真实状态,不做伪流式。 - 验收:Rust `project_supervisor_` 定向回归覆盖 ID/prompt/config、delivery/claim 幂等、排序锁零部分认领、Agent DB 旁路、未 Observed 门禁、Provider planning 前 durable 等待、parent-wake coalescing/结构性错误投影、重启损坏 barrier、完整身份与迟到 child suppression、delivery `.previous` 恢复、旧 receipt runId 冲突、executing `run_status` 续接与委派 policy 重验;Runner 内部回归覆盖定向 wake 只有目标推进后成功、可重试结果不缓存;Session lane 回归覆盖入队后才通知 Runner。客户端定向回归覆盖 active Supervisor Session、same-run steer、legacy 历史合并、确认/拒绝和唯一终态 assistant。修复后真实 Provider 已证明 design/art 两个专业 Agent 同秒进入 running 并重叠 20 秒,父 run 只写 1 条 waiting、同一 `Observed` claim 认领 2 份回执、首轮恰好 1 条 user / 1 条 assistant、无 reconciliation;同一 Session 第二轮引用上文完成且未新增委派。项目范围精确密钥扫描为 0。V1.15 首轮跑偏和本轮修复前 revision 误伤仍只保留为负向历史,不作为通过证据。 +## 2026-07-15 AI 游戏创作 Agent Runtime V1.17 单 Agent 持久计划 + +- 决策:`submit_agent_tool_plan` 顶层新增 nullable `planUpdate={explanation,steps[{step,status}]}`。strict function arguments 必须出现该字段,无真实变化时传 `null`;结构化更新最多 8 个唯一步骤,状态只允许 `pending / in_progress / completed` 且至多一个 `in_progress`。旧文本协议可缺字段,legacy `plan` 只作 fallback;当前 run 一旦有 `planRevision > 0`,legacy `plan` 不得再覆盖结构化计划。 +- 决策:Runtime state 持久化 `planRevision / planExplanation / planSteps / activePlanStepIndex`。有效变化使 revision 单调递增,完全相同的更新幂等不增号,非法更新不改快照;已完成或历史快照中已有的失败终态步骤必须保留,completed 不得回退。外层 run 进入 `failed / budget-exhausted` 时保留最后一次可信计划的 revision、说明、步骤状态和 active index,不把未完成步骤机械改写为失败。结构化计划建立后,工具 action 下标和旧自动步骤 helper 全部失去进度写权限,Agent 必须依据真实 observation 显式更新计划。 +- 决策:任一结构化步骤未完成时,空 actions、Provider response 和恢复中的 finalization 都由 `runtime.plan_update` blocker 拦截,不能写 assistant 或 completed。计划更新只属于 Runtime 私有元数据,不是工具 action,不读取或改写项目 policy,不触发 confirm/deny,不推进 project revision 或 verification gate,也不改变待确认动作 fingerprint。 +- 审计边界:`thinking_summary` 公共 event 只留正文 SHA-256 与字符数,legacy `plan` event 只留步骤数;`agent.runtime.plan_update` 只留 explanation 哈希与字符数,以及 step 标题哈希、状态和数量。`agent.runtime.tool_plan.repair` 只留尝试计数、协议以及模型输出/调用体预览、解析错误、callId / functionName 的哈希与长度,不落原始正文、错误或 function arguments;仅当前 planning 的私有有界 repair 请求可保留经过过滤的必要上下文。 +- 恢复与 steer:context bundle 升级为 `game-creator-runtime-context-bundle.v3` 并保存完整计划快照;v3 revision 或快照与 Runtime state 不一致时失败关闭,损坏 state 进入 `needs-reconciliation`。v2 继续可读,但只能在原身份、task、revision 与 verification gate 校验通过后从当前 state 补齐计划字段,后续 checkpoint 写 v3;v1 仍拒绝。Runner 重启、确认续跑和 stale finalization 不得重建或自动完成计划。same-run steer 丢弃旧 actions / 旧回复但保留终态步骤和 revision,下一版只重审未完成部分。 +- Finalization:journal 升级为 `game-creator-runtime-finalization.v2`,在 `prepared` 时绑定最终完整计划快照与 `planSnapshotFingerprint`,并把计划指纹纳入幂等 `finalizationId`。assistant 已落盘而 Runtime state 丢失时,从唯一 task record 与 v2 journal 恢复原 structured plan 后补齐 completed,不请求 Provider、不重放工具;assistant 尚未落盘而 state 丢失时保留 journal 并进入 `needs-reconciliation`,不得只凭 task 或 prepared journal 猜计划并写回复。 +- 展示:开发 Agent UI、项目内开发面板、CLI / `agent.run_status` 有界展示 revision、说明和最多 8 个完整步骤;刷新合并只沿用同一 Agent/Session/run。正式用户 Project Supervisor 只显示完成数、当前步骤、等待对象、下一步和专业 Agent 协作数量,不暴露 revision、内部说明、完整步骤、原始 observation、内部动作或动态 child。 +- 验收:确定性回归覆盖 schema、native function 显式字段与文本兼容、限制、单调性、外层失败进度保留、终态保留、动作下标零推进、未完成 final 门禁、损坏状态、v3/v2 恢复、finalization v2 state 丢失恢复、公共审计零正文、计划元数据 revision/policy 中立和两类 UI。恢复/steer 专项必须证明同一 run/session、revision 不回退、终态不丢、旧动作零执行和副作用零重放。真实 Provider 必须在无计划/工具配方的 disposable 项目中自行建立并多次更新计划,经历一次 same-run steer 与一次 Runner 重启,最终只在全部步骤 completed 后写唯一 assistant,并由 Runtime state、v3 bundle、task/event/Agent DB/conversation 和副作用计数交叉取证;截至 2026-07-15 尚未记录该专项 PASS。 +- 全量回归修正:context bundle v3 为保持计划快照一致,会在每个 action / observation 后同步;repository startup fingerprint 因此不能继续从最新 bundle 读取,否则同一 planning 批次的前置验证改变规范文件后,后续旧写动作会错误放行。pending action 升级为 `game-creator-pending-action.v4`,绑定 Provider planning 实际渲染的 repository fingerprint;同批 actions、确认和恢复统一复核该快照,旧 v1-v3 失败关闭。五类写动作 drift 回归证明旧动作零执行。 +- 终审补充:finalization v2 读取边界必须再次要求所有结构化步骤 `completed` 且 active index 为空;仅重算合法 `planSnapshotFingerprint / finalizationId` 的未完成快照也失败关闭。开发 CLI 的 Runtime JSON 只输出状态和安全身份,递归移除 `sessionPath / eventPath / taskPath`,不把项目绝对存储位置写入命令 transcript;Tauri/App 内部结果结构保持不变。 +- 验收现状:确定性回归已通过,真实 `gpt-5.5 llm-runtime` 连续三轮均在首个 Provider planning POST 返回前因同一 TLS record-layer failure 失败,未产生 plan/tool/kill/steer 证据;第三轮绝对路径、密钥和诱饵泄漏为 0。V1.17 继续保持未 PASS,Provider 恢复后必须完整重跑。 + ## 2026-07-13 普通微信支付 V3 退款使用统一观察事务闭环 - 背景:普通微信支付 V3 的退款申请响应、退款结果回调、主动查单和商户平台手工退款发现可能重复、乱序或只出现其中一种;原充值订单只有单一终态,无法表达多次部分退款、权益回收欠款和会员人工处理。 diff --git a/docs/project-memory/shared-memory/pitfalls.md b/docs/project-memory/shared-memory/pitfalls.md index bd4f0d2e4..b18212c3f 100644 --- a/docs/project-memory/shared-memory/pitfalls.md +++ b/docs/project-memory/shared-memory/pitfalls.md @@ -3171,6 +3171,44 @@ - 验证:Rust 定向回归使用 `project_supervisor_` 前缀,覆盖 delivery/claim 状态机、同 action 幂等、第 4 个新委派拒绝与已预留委派复用/拒绝 suppression、后续 delivery 锁忙时零部分认领、Agent DB 故障后回执仍可重放、未 Observed 阻断 final、Provider planning 前 durable 等待、parent-wake coalescing/结构性错误、重启损坏 barrier、错配和迟到 child、executing `run_status` 续接与 delegate policy 重验;`agent_background_enqueue_notifies_only_after_session_lane_release` 覆盖入队锁序,Runner 内部测试覆盖定向 wake 与不缓存重试。真实 Provider 必须同时证明专业 Agent 时间区间重叠、父 run 仅一次 waiting、同一 Observed claim 认领全部回执、唯一 assistant、第二轮历史引用不新增委派和项目范围密钥扫描为 0。 - 关联:`docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md`、`apps/ai-game-creator-shell/src-tauri/src/delegation.rs`、`agent.rs`、`runner.rs`、`tests.rs`。 +## 单 Agent 持久计划不能靠工具下标或恢复猜进度 + +- 现象:工具 action 1 成功后第二个计划步骤被自动标成完成,模型仍有 pending / in_progress 步骤却写出最终回复;或 Runner 重启、刷新 UI、same-run steer 后 `planRevision` 回退、已完成步骤消失,legacy `plan` 又覆盖新计划。另一类错误是仅更新计划就触发项目 revision 漂移、verification 失效或权限确认。 +- 原因:旧 `planSteps` 由短 `plan` 派生,并按 actions 数组下标驱动 `active / completed`,它无法表达跨窗口、恢复和 steer 后的真实任务进度。把 context bundle 当唯一计划事实源、把 v2 缺字段当空计划,或把 `planUpdate` 伪装成受策略工具,也会让 Runtime state、恢复快照和项目副作用门禁互相污染。 +- 处理:V1.17 的 `planUpdate` 只接受 1 到 8 个唯一 `pending / in_progress / completed` 步骤,至多一个 `in_progress`;native function 即使无变化也必须显式传 `planUpdate: null`,只有旧文本 JSON 可省略。当前 run 建立结构化计划后,legacy `plan` 和所有按 action 下标推进的 helper 都只能读不能写。`planRevision` 只在有效变化时单调增加;completed 与历史快照中已有的 failed 终态即使被下一版省略也必须合并保留,completed 回退或合并后超过 8 步时整次拒绝。外层 run 进入 `failed / budget-exhausted` 时必须原样保留最后可信进度,不把 pending / in_progress 机械标成 failed。任何非 completed 步骤都阻止成功 final,不能用 response 文本绕过。 +- 恢复与 steer:context bundle v3 必须保存并复核 revision、说明、步骤和 active index;与 Runtime state 不一致时失败关闭,不能选“看起来更新”的一份。v2 只能在原身份、task、revision 和 verification gate 校验通过后从当前 state 补齐计划,v1 继续拒绝。same-run steer 只作废旧 Provider actions / 回复并要求重审未完成部分,不能清空终态步骤或重置 revision;Runner 重启同样不得自动勾选。finalization 另以 v2 journal 绑定最终完整计划快照:只有 assistant 已落盘时,state 丢失才可从该快照恢复终态;assistant 未落盘且 state 不可读时必须进入 reconciliation。 +- 边界:计划更新是 `.agent/runtime` 私有元数据,不经过项目工具 policy,不推进 project revision 或 verification gate,不改变 pending action fingerprint。开发 UI/CLI 可展示最多 8 步完整计划;普通用户 Supervisor 只能显示完成数、当前步骤、等待、下一步和协作数量,不能把内部 explanation、完整步骤或 currentAction 搬到主聊天。 +- 验证:运行 `structured_plan_` 与 `agent_runtime_context_bundle_migrates_v2_and_rejects_v3_plan_mismatch` Rust 定向用例,并用 `appSurface.test.ts` 覆盖刷新恢复和 Supervisor 紧凑摘要。真实 Provider 必须在无计划配方下多次更新计划,完成一步后接受 same-run steer,再经历 Runner 强杀恢复;最终证明 run/session 不变、revision 不回退、终态不丢、旧动作与副作用不重放、未完成时零 assistant、完成后唯一 assistant,计划更新前后 project revision / policy 不变。未运行该门禁时不得写 V1.17 PASS。 + +### Finalization 只绑定回复会在 Runtime state 丢失后丢计划 + +- 症状:assistant 已经按稳定 messageId 写入 conversation,进程却在 Runtime completed 投影前退出;重启后 state 文件缺失,系统从 task record 重建出默认或 legacy 计划,最终回复虽然没有重复,结构化计划 revision 和步骤却丢失。反向地,assistant 尚未写入时若也用 journal 单独猜计划,会把过期回复错误提交给用户。 +- 原因:task record 不携带完整结构化计划,context bundle 也可能对应 finalization 前的其它 checkpoint;只给 journal 绑定回复和 verification gate,无法证明准备提交时的最终计划快照。 +- 处理:`game-creator-runtime-finalization.v2` 在 prepared 时保存完整 `planRevision / planExplanation / plan / planSteps / activePlanStepIndex` 与 `planSnapshotFingerprint`,并将指纹纳入 finalizationId。读取时除校验格式和指纹外,还要再次要求全部结构化步骤 completed 且 active index 为空。assistant 已存在且 task 唯一时,允许从 v2 快照恢复原计划并补齐 Runtime completed;assistant 不存在而 state 缺失或不可读时保留 journal、进入 `needs-reconciliation`,不能自动写回复。已有 state 与 journal 快照冲突时同样失败关闭或让 prepared 回复失效后在同一 run 重规划。 +- 验证:`finalization_resume_recovers_persisted_assistant_without_runtime_state` 必须证明无 Provider 重放、assistant 唯一且恢复后的 revision/说明/步骤与 prepared 快照完全一致;`structured_plan_finalization_without_readable_runtime_state_needs_reconciliation` 必须证明 assistant 未落盘时 missing/corrupt state 都零回复、journal 保留且无 completed 审计;`finalization_resume_blocks_internally_consistent_incomplete_plan_snapshot` 必须证明重算合法指纹和 finalizationId 也不能提交未完成计划。 + +### CLI Runtime JSON 不能暴露项目绝对存储路径 + +- 症状:真实 E2E 的 task/event/Agent DB/report 均无项目绝对路径,子进程 transcript 扫描却稳定命中 6 次;入队和首次状态读取各返回 3 个路径。 +- 原因:开发 CLI 直接序列化 `AgentRuntimeResult`,把仅供 Tauri/App 定位本地 sidecar 的 `sessionPath / eventPath / taskPath` 一并写进 `runtimeJson`。后续 confirm、steer 和 resume 复用同一结果结构,也会重复暴露。 +- 处理:保持 Tauri 内部契约不变,只在 CLI JSON 输出视图递归删除三个存储路径;`state`、task queue、events、tasks、run/session/action 身份和 steer 状态继续保留,验收器仍能解析必要证据。不要靠 E2E 忽略 CLI stdout,也不要笼统删除所有 `path` 字段破坏安全相对产物证据。 +- 验证:CLI serializer 单测覆盖顶层、嵌套和数组结果;真实 `--agent-runtime-status` 输出对 disposable 项目路径命中为 0,后续完整 `llm-runtime` 报告的 `projectPathTranscriptLeakCount / projectPathReportLeakCount` 必须同时为 0。 + +### 把模型修复上下文写入公共审计会泄露正文 + +- 症状:Runtime event 或 `.agent/agent.db` 为了排障直接记录 thinking summary、legacy plan 标题、解析错误、malformed JSON 或 native function arguments;私有任务内容、项目路径或模型调用体因此进入公共审计和 UI 最近事件。 +- 原因:格式修复确实需要把上一条输出与错误反馈给同一次 Provider 请求,但“Provider 私有修复上下文”和“持久公共诊断投影”被误当成同一份数据。 +- 处理:`thinking_summary` event 只留正文 SHA-256 与字符数,legacy `plan` event 只留步骤数;结构化计划审计只留 explanation 哈希与字符数,以及 step 标题哈希、状态和计数。`agent.runtime.tool_plan.repair` 只留 attempt/maxAttempts、protocol,以及错误、输出或调用体预览、callId/functionName 的哈希与长度。经过过滤和限长的上一条输出与协议错误只可进入当前 planning 的私有 repair 请求,不得落到 event、task 或 Agent DB 正文字段。 +- 验证:后台 loop 回归必须断言 thinking event 不含摘要正文、legacy plan event 不含标题;文本与 native repair 回归必须同时证明私有请求仍含足够修复上下文,而 Agent DB 不存在 `protocolError / responsePreview / function arguments / callId / functionName` 原文字段。 + +### 每次同步 context bundle 时刷新 planning fingerprint,导致同批旧动作越过仓库规范漂移 + +- 症状:同一 Provider planning 返回多个 actions;前一个验证动作修改了 `AGENTS.md` 或其它启动上下文来源,后一个写动作仍执行成功,下一轮只看到普通 `ok` observation,没有 `repositoryContextDrift=true`。 +- 原因:context bundle v3 在 action 激活和 observation 落盘后都会同步完整计划,同时重新扫描 repository startup context。若 drift gate 从最新 bundle 读取 fingerprint,前一个动作造成的漂移会被同步成新基线,后续动作不再与 Provider planning 真正看到的旧规范比较。 +- 处理:Provider request builder 必须把实际渲染的 repository fingerprint 和工具计划一起返回,并写入 `game-creator-pending-action.v4` 的 `plannedRepositoryContextFingerprint`;同批自动动作、待确认动作和恢复动作都只复核该持久快照。旧 v1-v3 缺少身份,失败关闭,不能从最新 bundle 猜回。 +- 验证:`runtime_v11_closure_repository_context_drift_replans_before_auto_mutations` 必须覆盖 file.write / file.patch / file.delete / project.patchset / project.restore 五种动作,证明前置验证导致规范漂移后旧动作零执行、同 run 收到稳定 drift observation;`legacy_context_and_pending_records_fail_closed` 覆盖 v3 拒绝。 +- 关联:`docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md`、`apps/ai-game-creator-shell/src-tauri/src/agent.rs`、`main.rs`、`tests.rs`、`apps/ai-game-creator-shell/src/App.tsx`、`tests/appSurface.test.ts`。 + ## iOS 退款问询的 result_code 不是 debug 状态 - 现象:为了先观察真实 iOS 退款通知,回调返回 `ErrCode=0 + IosRefundQueryResponse.result_code=1`,并把 evidence 写成“调试阶段不执行自动退款决策”,看起来像安全 ACK,实际已经向微信建议拒绝退款。 diff --git a/docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md b/docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md index 87f443719..b6cc8202a 100644 --- a/docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md +++ b/docs/technical/【技术方案】AI游戏创作Agent Runtime V1.1-2026-07-12.md @@ -686,8 +686,81 @@ V1.16 把正式用户主聊天从一次性自然语言问答升级为现有 Exte 2026-07-14 修复后真实 Provider 验收已通过。一次性项目 `/tmp/gameagent-supervisor-parallel-e2e-pass-*` 中,父 run `swarm-project-supervisor-1784044475952` 同时创建 `design-director` 与 `art-director` 两个静态委派;两者均在时间戳 `1784044486` 进入 running,分别于 `1784044506`、`1784044536` completed,存在 20 秒真实重叠。父 run 只写入 1 条 `waiting-for-delegate-receipts` task 记录,期间没有继续 Provider 轮询;随后同一 actionId 认领两份 delivery,两个 delivery 均为 `claimed-by-parent`,唯一 claim 为 `Observed` 且 receiptCount=2,父 run 无 reconciliation 并 completed。首轮 Supervisor Session 恰好写入 1 条 user 与 1 条 assistant;同一 Session 的第二轮“基于上文、不要重新委派”请求直接使用历史完成三句回复,delivery 总数仍为 2。项目范围精确 secret 扫描无 API Key、Bearer token 或 `sk-*` 命中。 +## V1.17 单 Agent 持久计划 + +V1.17 把后台单 Agent 每轮临时生成的短 `plan` 升级为同一 run 内可恢复、可单调更新并参与完成门禁的结构化计划。它不新增模型工具、任务系统或项目权限;计划更新仍通过既有 `submit_agent_tool_plan` function arguments 提交,Runtime state 与 context bundle 是持久事实源。 + +### 工具计划契约 + +OpenAI Chat / Responses 的 strict function schema 顶层固定为 `thinkingSummary / planUpdate / plan / actions / response`,其中 `planUpdate` 是必填但可为 `null` 的字段: + +```json +{ + "thinkingSummary": "已完成项目读取,开始实现", + "planUpdate": { + "explanation": "根据真实项目观察推进实现步骤", + "steps": [ + { "step": "读取项目上下文", "status": "completed" }, + { "step": "实现核心玩法", "status": "in_progress" }, + { "step": "运行定向验证", "status": "pending" } + ] + }, + "plan": [], + "actions": [], + "response": "" +} +``` + +- `planUpdate.explanation` 必须是非空有界文本;`steps` 必须包含 1 到 8 个标题互不重复的步骤,每个步骤标题也是有界文本。 +- Provider 只允许提交 `pending / in_progress / completed`,同一更新至多一个 `in_progress`。持久快照仍可读取历史已有的内部 `failed` 终态以兼容恢复,但它不是 Provider 可提交状态,也不能由外层 run 失败临时生成。 +- 复杂任务在首次拆解、真实进度变化、same-run steer 后重审顺序和最终收束时提交 `planUpdate`;没有变化时传 `null`。提交结构化更新时 `plan` 传空数组。 +- native function 的 arguments 即使本轮没有计划变化也必须显式包含 `planUpdate: null`;省略字段属于协议错误并进入既有格式修复或失败路径。只有兼容旧实现的文本 JSON 可以省略该字段并继续走 legacy `plan` fallback。 +- `plan` 只保留给文本 JSON 兼容协议或旧 Provider 作为 fallback。只要当前 run 已有 `planRevision > 0`,后续 legacy `plan` 不得覆盖结构化计划;同一响应同时带有两者时以 `planUpdate` 为准。 + +### 单调状态与完成门禁 + +- `AgentRuntimeState` 持久化 `planRevision / planExplanation / plan / planSteps / activePlanStepIndex`。第一次有效结构化更新把 revision 推到 1;内容或状态真实变化时单调加一,完全相同的幂等更新不增加 revision,拒绝的更新也不改变现有快照。 +- 已进入 `completed` 或历史快照中已有 `failed` 的终态步骤必须保留。后续更新即使省略它们,Runtime 也会把终态步骤并回快照;`completed` 不得回退,已有 `failed` 不得由 Provider 改写。合并后仍受 8 步上限约束,超限整次拒绝。 +- `activePlanStepIndex` 只对应唯一 `in_progress` 步骤。结构化计划建立后,旧的“按 actions 数组下标激活、完成或重试步骤”辅助逻辑全部失效;工具成功、失败或 action 序号都不能替模型改写结构化进度,Agent 必须根据真实 observation 显式提交下一版 `planUpdate`。 +- 只要结构化计划中仍有非 `completed` 步骤,空 actions、非空 response 或恢复中的 prepared finalization 都不能写 assistant、completed 或成功 final。Runtime 返回 `runtime.plan_update` blocker 并在同一 run 继续 planning;普通失败、取消和 reconciliation 仍可按既有失败路径收束,不能伪装成计划成功完成。 +- 外层 run 进入 `failed` 或 `budget-exhausted` 时,就结构化计划字段而言,必须原样保留最后一次可信的 `planRevision / planExplanation / plan / planSteps / activePlanStepIndex`。不得把当时的 `pending / in_progress` 机械改写为 `failed`,也不得把 run 失败反推成步骤进度事实。 +- `planUpdate` 是 Runtime 控制面元数据,不是白名单工具 action。更新计划不读取或改写 `.agent/policy.json`,不触发 confirm/deny,不推进 project revision,不改变 verification gate,也不改变 pending action fingerprint;真正的文件、命令、验证、委派和提交仍独立经过原有门禁。 + +### 恢复与 steer + +- `.agent/runtime/context-bundles//.json` schema 升级为 `game-creator-runtime-context-bundle.v3`,新增结构化计划 revision、说明、步骤和 active index 快照。读取 v3 时必须与当前 Runtime state 的完整计划快照一致;revision、步骤、索引或状态不匹配时失败关闭,损坏 Runtime state 投影为 `needs-reconciliation`,不得猜测进度或自动完成。 +- `game-creator-runtime-context-bundle.v2` 保持读取兼容:先完成原有身份、task、revision 和 verification gate 校验,再从当前 Runtime state 补入结构化计划字段并按 v3 继续;后续 checkpoint 写 v3。v1 以及缺失既有安全关联的旧记录仍按原规则失败关闭。 +- 最终回复 journal 升级为 `game-creator-runtime-finalization.v2`。它在 `prepared` 时绑定最终可完成的完整计划快照与 `planSnapshotFingerprint`,并把该指纹纳入 `finalizationId`;读取 v2 时还必须重新确认结构化步骤全部为 `completed` 且不存在 active index,不能让内部一致但未完成的篡改快照越过完成门禁。恢复不能只凭回复、task record 或 context bundle 猜测最终计划。 +- assistant 已按稳定 `messageId` 落入 conversation、但 Runtime state 随后缺失或不可读时,恢复可从唯一 task record 重建同一 run,再用可信的 v2 journal 恢复原 `planRevision / planExplanation / plan / planSteps / activePlanStepIndex` 并补齐 completed 投影,不重新请求 Provider 或重放工具。若 assistant 尚未落盘而 Runtime state 已丢失,则保留 journal 并进入 `needs-reconciliation`,不得仅凭 prepared journal 新写 assistant;已有 state 与 journal 计划快照冲突时同样失败关闭或丢弃过期 prepared 回复回到同 run 重规划。 +- Runner 重启、确认续跑和 stale finalization 重规划必须保持同一 Agent/task/Session/run、原 `planRevision`、全部终态步骤和未完成步骤;恢复不能重新从 legacy `plan` 派生进度,也不能因为工具回执已经存在而自动勾选步骤。 +- 同一 Provider planning 返回的所有 actions 必须绑定该请求实际渲染的 repository startup fingerprint。pending action schema 升级为 `game-creator-pending-action.v4` 并持久化 `plannedRepositoryContextFingerprint`;后续 action、用户确认和恢复都只用这份 planning 快照做 drift gate,不能从每个 action / observation 后持续刷新的 context bundle 回读。旧 v1-v3 记录缺少该身份,统一失败关闭进入核对。 +- same-run steer 继续由 V1.13 丢弃过期 Provider 计划、剩余 actions 或旧最终回复;结构化计划本身不清空。steer observation 明确要求重审未完成步骤,下一版可以调整未完成步骤的标题和顺序,但已完成或失败步骤继续保留,`planRevision` 继续单调递增。 + +### 展示边界 + +- 开发 Agent UI、项目内开发面板和 CLI Runtime 状态输出展示有界的完整计划:revision、说明、最多 8 个步骤、每步状态、当前步骤及已有安全 detail。刷新或事件合并只可沿用同一 Agent/Session/run 的上一个快照,切换 run 时不得把旧计划带入新 run。 +- 正式用户的 Project Supervisor 聊天只展示紧凑摘要:`已完成数/总数`、当前步骤、等待对象、下一步和当前专业 Agent 协作数量。不得展示 `planRevision`、内部 explanation、完整步骤列表、原始 observation、内部 currentAction 或动态 child 身份。 + +### 公共审计与私有修复上下文 + +- `thinking_summary` 公共 Runtime event 只写固定语义摘要以及 `thinkingSummarySha256 / chars`,不写模型摘要正文;legacy `plan` event 只写 `planStepCount`,不写步骤标题。结构化 `plan_update` 的 Agent DB 投影同样只保存 explanation 的 SHA-256 与字符数,以及步骤标题的 SHA-256、状态和数量。 +- `agent.runtime.tool_plan.repair` 只保存尝试次数、协议类型,以及协议错误、经过过滤的模型输出或 function call 预览的 SHA-256 与字符数;repair 中的 callId / functionName 也只写哈希。原始模型正文、解析错误、function arguments 或调用体不得进入 event、task 或 Agent DB 公共审计。 +- 为了让 Provider 修正格式,同一次 planning 的私有、瞬时 repair 请求可以携带经过统一敏感信息过滤和长度限制的上一条输出预览与协议错误。该上下文只服务当前 Provider 请求,不得反向复制到公共审计或用户可见状态。 + +### 验收口径 + +确定性验收至少覆盖:native function 显式 `planUpdate` 与文本 JSON omission 兼容;空说明、重复步骤、未知状态、超过 8 步和多个 `in_progress` 拒绝;幂等更新不增 revision、真实更新单调递增、终态步骤保留和 completed 回退拒绝;工具 action 下标零推进;外层 failed / budget-exhausted 保留最后可信计划;未完成计划阻止 response 与恢复 finalization;损坏 state 进入 reconciliation;context bundle v3 快照一致性、v2 兼容和 v3 mismatch 失败关闭;finalization v2 计划指纹及 assistant 已落盘后的 state 丢失恢复;thinking / legacy plan / repair 公共审计零正文;计划元数据不推进 project revision、verification gate 或权限确认;开发 UI 刷新后仍显示完整 8 步,以及 Supervisor 只显示紧凑摘要。 + +恢复与 steer 专项必须在同一 run 中先完成至少一个步骤,再分别覆盖 context checkpoint 后强杀 Runner、恢复继续、Provider planning 中接受 steer、旧 actions 零执行和重排未完成步骤。恢复前后 `taskId / sessionId / runId` 必须不变,`planRevision` 不回退,终态步骤不丢失,已有副作用不重放;最后一版所有必要步骤均为 `completed` 后才允许唯一 assistant。 + +真实 Provider 验收必须使用不提供计划内容、工具顺序或状态迁移配方的 disposable 项目,让模型自行建立至少三步计划、依据真实 observation 更新至少两次、经历一次 same-run steer 和一次 Runner 重启后完成。验收器从 function call arguments、Runtime state、v3 context bundle、task/event/Agent DB、conversation 和副作用计数交叉证明 revision 单调、终态保留、未完成时零 final、最终唯一 assistant、零动作重放、计划元数据零 project revision / policy 变化以及密钥和项目绝对路径零泄漏;开发 UI/CLI 与 Supervisor 摘要另做展示断言。未实际完成这套真实 Provider 门禁前,只能记录“未验收”或外部阻塞,不能把确定性测试外推为 V1.17 PASS。 + +截至 2026-07-15,确定性门禁已通过:Tauri 单线程全量 690 项中 686 passed / 4 ignored,结构化计划、finalization 恢复和前端竞态定向回归全部通过。真实 `gpt-5.5` `llm-runtime` 连续三轮均在首个 Provider planning POST 返回前因 TLS record-layer failure 进入 failed,尚未产生 function plan、工具动作、Runner kill 或 steer,因此仍未记录 V1.17 真实 Provider PASS。首轮验收器同时发现 CLI `runtimeJson` 暴露 `sessionPath / eventPath / taskPath`;CLI 输出视图移除这三个绝对存储路径后,第三轮项目绝对路径 transcript/report 泄漏计数均为 0。外部请求失败不能替代完整真实门禁,后续 Provider 恢复后必须重跑本节命令。 + ## 验收命令 +- `cargo test --manifest-path apps/ai-game-creator-shell/src-tauri/Cargo.toml structured_plan_ -- --nocapture` +- `cargo test --manifest-path apps/ai-game-creator-shell/src-tauri/Cargo.toml agent_runtime_context_bundle_migrates_v2_and_rejects_v3_plan_mismatch -- --nocapture` - `npm run ai-game-creator-shell:typecheck` - `npm run test -- apps/ai-game-creator-shell/tests` - `cargo test --manifest-path apps/ai-game-creator-shell/src-tauri/Cargo.toml` diff --git a/docs/technical/【技术方案】AI游戏创作智能体App实施计划-2026-06-24.md b/docs/technical/【技术方案】AI游戏创作智能体App实施计划-2026-06-24.md index 43bcfc335..17c7a801e 100644 --- a/docs/technical/【技术方案】AI游戏创作智能体App实施计划-2026-06-24.md +++ b/docs/technical/【技术方案】AI游戏创作智能体App实施计划-2026-06-24.md @@ -54,6 +54,14 @@ V1.11 的受保护仓库控制目录同时包含 `.git / .agent / .agents / .cod 当前委派协议已把 delivery 与 `Prepared -> Committed -> Observed` claim journal 作为事实源,认领前按 delegationId 排序并取得全部 delivery 锁,`.agent/agent.db` 只作 best-effort 诊断投影。恢复中的 executing 动作只允许 Supervisor `agent.delegate / agent.run_status` 经项目锁、pending 全身份和 policy 重验后补交;parent-wake 使用 singleflight、有界错误分类和稳定 Runner requestId。子终态只有在 parent/child/delivery 完整身份一致后才能 ready 或 suppression;错配不得改写 delivery。最终回复继续由原父 run 的 finalization journal 幂等写入。 +2026-07-15 起,Runtime V1.1 文档的“V1.17 单 Agent 持久计划”作为后台工具规划进度的新事实源。`submit_agent_tool_plan` 新增 nullable `planUpdate={explanation,steps[{step,status}]}`;步骤只接受 `pending / in_progress / completed`,最多 8 步且至多一个 `in_progress`。结构化计划一旦建立,legacy `plan` 只作旧协议 fallback;终态步骤必须保留,`planRevision` 只在真实变化时单调递增,工具 action 下标不得自动完成结构化步骤,存在未完成步骤时不得写最终回复或 completed。 + +V1.17 计划快照随 `game-creator-runtime-context-bundle.v3` 持久化,v2 在通过原身份、revision 和 verification gate 校验后从当前 Runtime state 补齐计划字段继续恢复;计划元数据本身不推进项目 revision、不改变 verification gate,也不触发项目权限确认。开发 UI 和 CLI 有界展示 revision、说明与完整 8 步;正式用户的 Supervisor 只展示完成数、当前步骤、等待对象、下一步和协作数量的紧凑摘要。恢复、same-run steer 和真实 Provider 的完整验收矩阵以 Runtime V1.17 章节为准;截至 2026-07-15 尚未记录 V1.17 真实 Provider PASS。 + +V1.17 同时把 finalization journal 升级为 v2 并绑定最终完整计划快照:assistant 已落盘而 Runtime state 丢失时,从 v2 journal 恢复原结构化计划后补齐终态;assistant 尚未落盘且 state 丢失时失败关闭。外层 `failed / budget-exhausted` 只保留最后可信计划,不把未完成步骤机械改成失败。thinking summary、legacy plan event 和 tool-plan repair 公共审计只保留哈希、字符数或计数,必要的模型输出与错误上下文仅留在有界私有 repair 请求中。 + +2026-07-15 V1.17 收口时,确定性回归已通过,但真实 `gpt-5.5` `llm-runtime` 连续三轮都在首个 planning POST 返回前遇到相同 TLS record-layer failure,未形成结构化计划或工具动作,因此继续保持“真实 Provider 未 PASS”。首轮严格泄漏扫描另发现开发 CLI 的 `runtimeJson` 直接带出 `sessionPath / eventPath / taskPath`;CLI 输出现已移除这三个绝对存储路径,第三轮 transcript/report 路径泄漏计数归零。Provider 恢复后仍需完整重跑 Runner kill + same-run steer 套件,不能以短 `/models` 鉴权成功或确定性测试代替。 + 2026-07-12 真实验收:发布 AppData 中的真实 `gpt-5.5` 已通过最终安全收紧后的 `llm-runtime` 套件,覆盖 Runner 强杀恢复且 run/session 身份稳定、仓库上下文、checkpoint/精确修改、失败命令诊断与修复复验、6 套确认生命周期、项目验证、桌面与移动非空画布证据、3 个隔离实例并行和唯一 all-join;95 条 task、161 条 event、137 条 Agent DB、13 条合法工具协议、副作用判重、终态投影、assistant audit、消息、回执和密钥泄露均以结构化落盘事实验收。`full` 套件仍要求 External Editor API 配置,缺失时必须返回 `BLOCKED(editorApi)`,不得记为通过。 2026-07-13 V1.3 真实验收:同一真实 Provider 套件已改为先读取 SHA-256,再用唯一一次 `project.patchset` 同时更新和创建文件,并使用自动 checkpointId 读取 2 项内容 hunks;prepared / completed 审计各 1 条、patchset revision 增量为 1,Runner 强杀恢复、命令和项目验证、双视口浏览器验证、隔离 Agent join、重复副作用与密钥扫描继续全部通过。 @@ -101,7 +109,7 @@ Agent Runtime 负责: - 2026-07-10 补充:后台 Agent 任务的 `memory.write scope=agent` 只允许写当前 Agent 自己的私有记忆。若 action 指定其他 `agentId / targetAgentId`,Runtime 返回 `blocked` observation,不写目标 Agent 私有记忆,也不写 `agent.runtime.memory.write` 审计;跨 Agent 共享稳定结论必须使用 `blackboard.write`,给单个 Agent 留上下文必须使用 `agent.message`。 - 历史边界说明:下一条“只做进程内 JSONL 串行化、跨进程不支持”已由 V1.1 的进程内互斥 + OS 文件锁替代,仅保留为演进记录。 - 2026-07-10 补充:本地 append-only JSONL 追加写入按目标文件路径做进程内串行化。`.agent/agent.db`、项目 / Agent 对话、Runtime events、Runtime tasks、Agent activity 和 output 都通过共享 helper 写入完整 JSON 行,防止多个后台 Agent 并行运行时 record 内容与换行交错;该约束服务于当前单客户端进程内并行,不把跨进程同项目写入作为 v1 支持目标。 -- 2026-07-10 补充,2026-07-12 冻结并更新:后台 Agent 待确认工具动作改用 durable `AgentRuntimePendingToolAction`。完整记录包含精确工具 action、当前 task/run、loop 轮次、action 序号、计划、已有 observations、续跑上下文、创建时的全局 project revision 和 per-run verification gate;schema 升级为 `game-creator-pending-action.v3`。写入前拒绝密钥、Token、Cookie、App 配置痕迹和项目绝对路径,再通过临时文件替换原子写入 `.agent/runtime/pending-actions//.json`。公共 runtime state 的 `pendingToolAction` 只暴露 `actionId / actionFingerprint / tool / inputSummary / reason / requestedAt` 安全摘要。`actionFingerprint` 绑定工具名、完整输入 JSON 与实际执行 task context;`actionId` 再绑定 run、loop、action 序号和 occurrence nonce,防止同一 run 内相同输入的旧 UI 点击批准后一次动作。开发窗口和项目内 Agent 面板的“确认继续 / 拒绝并继续”都提交当前 `runId + actionId`,Runtime 与私有动作、公共摘要交叉校验后才迁移账本状态。批准或自动执行前必须重读并匹配创建时的全局 revision 与 gate;任一漂移都进入 `needs-reconciliation`,不得执行旧动作。确认后在同一 run 直接执行原 action 并把 observation 接回后续 loop,不创建新 run、不要求模型重复动作;拒绝不执行工具,写 `blocked` observation 后在同一 run 继续规划。账本状态使用 `pending-confirmation / approved / executing / observed-approved / observed-rejected`:重启可恢复 waiting、未执行的 approved action 或已落盘 observation;若进程中断在 `executing`,Runtime 进入 `failed / needs-reconciliation`,禁止自动重放外部副作用,开发者核对项目状态后先取消原任务。旧版 pending action 或缺失 revision / gate 关联的记录恢复时必须失败关闭,不得按默认值补齐、自动重放外部副作用或写 completed。等待期间同 Agent 新任务保持 `pending`,重启不会越过 waiting run,确认、拒绝或取消后再串行排空。`.agent/runtime/` 作为私有控制面,不允许通用文件工具列出、读取或写入;`file.delete` 进一步禁止整个 `.agent/**`。每 Agent 锁使用唯一 token,旧持有者不会删除替换后的新锁,Linux 上仍存活的其他进程锁不会按超时强占。 +- 2026-07-10 补充,2026-07-12 冻结,2026-07-15 更新:后台 Agent 待确认工具动作改用 durable `AgentRuntimePendingToolAction`。完整记录包含精确工具 action、当前 task/run、loop 轮次、action 序号、计划、已有 observations、续跑上下文、创建时的全局 project revision、per-run verification gate,以及本次 Provider planning 实际使用的 repository startup fingerprint;schema 当前为 `game-creator-pending-action.v4`。写入前拒绝密钥、Token、Cookie、App 配置痕迹和项目绝对路径,再通过临时文件替换原子写入 `.agent/runtime/pending-actions//.json`。公共 runtime state 的 `pendingToolAction` 只暴露 `actionId / actionFingerprint / tool / inputSummary / reason / requestedAt` 安全摘要。`actionFingerprint` 绑定工具名、完整输入 JSON 与实际执行 task context;`actionId` 再绑定 run、loop、action 序号和 occurrence nonce,防止同一 run 内相同输入的旧 UI 点击批准后一次动作。开发窗口和项目内 Agent 面板的“确认继续 / 拒绝并继续”都提交当前 `runId + actionId`,Runtime 与私有动作、公共摘要交叉校验后才迁移账本状态。批准或自动执行前必须重读并匹配创建时的全局 revision、gate 与 planning repository fingerprint;任一漂移都拒绝旧动作并回到同一 run 重新规划,结果未知才进入 `needs-reconciliation`。确认后在同一 run 直接执行原 action 并把 observation 接回后续 loop,不创建新 run、不要求模型重复动作;拒绝不执行工具,写 `blocked` observation 后在同一 run 继续规划。账本状态使用 `pending-confirmation / approved / executing / observed-approved / observed-rejected`:重启可恢复 waiting、未执行的 approved action 或已落盘 observation;若进程中断在 `executing`,Runtime 进入 `failed / needs-reconciliation`,禁止自动重放外部副作用,开发者核对项目状态后先取消原任务。旧版 v1-v3 pending action 或缺失 revision / gate / planning fingerprint 关联的记录恢复时必须失败关闭,不得按默认值补齐、自动重放外部副作用或写 completed。等待期间同 Agent 新任务保持 `pending`,重启不会越过 waiting run,确认、拒绝或取消后再串行排空。`.agent/runtime/` 作为私有控制面,不允许通用文件工具列出、读取或写入;`file.delete` 进一步禁止整个 `.agent/**`。每 Agent 锁使用唯一 token,旧持有者不会删除替换后的新锁,Linux 上仍存活的其他进程锁不会按超时强占。 - 2026-07-10 补充:per-agent 锁最终采用 OS 级文件锁,取代上一条末尾的 token/PID/超时抢占方案。Unix 使用非阻塞独占 `flock`,Windows 使用禁止共享的文件句柄;`.agent/runtime/locks/.lock` 只保存诊断元数据并可长期存在,真正所有权随文件句柄和进程生命周期释放。任何确认、拒绝、恢复、取消和队列 drain 都必须使用同一系统锁;确认、拒绝和取消只能在拿锁后重新读取当前 runtime、task 与待确认动作再迁移状态,恢复也必须先拿锁再读取 durable pending action 或 recoverable task,不能用拿锁前的旧快照覆盖并发结果。waiting 状态只允许短暂等待原 worker 正常释放,不得按状态删除并重建锁文件;running 取消在拿不到锁时只保留取消 tombstone,由原 worker 在 LLM / 工具成功或失败返回后的检查点收束。 - 2026-07-10 补充:白名单自动工具也必须使用 durable `AgentRuntimePendingToolAction`,并以 `executionMode = auto` 区别待开发者确认的动作。Runtime 在副作用前依次落盘 `approved`、`executing`,返回后落盘 `observed-approved` 和 observation;该账本继续覆盖下一轮 LLM planning,直到下一条精确动作接管或 completed / failed / cancelled 终态可靠落盘,不能在 observation 刚落盘时提前删除。恢复 `approved + auto` 时执行同一 action 一次,恢复 `observed-approved + auto` 时只把 observation 交回 Agent,恢复 `executing + auto` 时停止在 `failed / needs-reconciliation`;该核对阶段是硬屏障,即使磁盘仍是 `approved` 也禁止继续。恢复时策略由 auto 收紧为 confirm,则保留原 actionId / fingerprint 并转换成 `pending-confirmation + confirmation`。自动动作在 `.agent/agent.db` 写 `agent.runtime.tool_action.executing`、`agent.runtime.tool_action.observed` 和 `agent.runtime.tool_action.needs_reconciliation` 审计。 - 2026-07-10 补充:`needs-reconciliation` 按 Agent 队列级屏障处理。该 Agent 的新聊天后台任务、delegate 和 ready-task 调度仍可入队,但只能保持 `pending`;恢复、正常 drain 和取消其他排队 run 后触发的 drain 都不得越过当前核对 run。新 run 只能在取得 per-agent OS 锁后重新读取 waiting / cancelling / reconciliation 状态并通过准入检查,不能在锁外检查后直接启动;锁内通过检查后统一从 task JSONL 选择最早 pending run,保证并发投递时仍按 FIFO 启动。屏障判定同时读取当前 Runtime state 与 task JSONL 最新记录,因此 pending ledger 缺失时也不放行、不允许 retry;开发者核对外部副作用后必须显式取消该 run,后续队列才继续。 @@ -109,13 +117,13 @@ Agent Runtime 负责: - 2026-07-10 补充:Agent Runtime state 新增 `recentToolCalls`,每次后台工具执行后记录最近 20 条结构化工具动作,包含 tool、status、actionFingerprint、inputSummary、reason、summary、detail 和 updatedAt;开发窗口、项目内 Agent 对话弹窗和主窗口 Agent 状态列表可直接展示“最近动作”和安全目标摘要,不再只能从 observation 字符串里猜测 action / observation 对应关系。`inputSummary` 只保留相对路径、checkpoint id、目标 Agent、内容字符数等确认所需信息,不保存原始 API Key、待写正文、消息正文、素材 prompt 或任意未过滤输入。 - 2026-07-10 补充:Agent Runtime state 新增 `currentGoal` 和 `waitingOn`,把本轮目标与当前等待对象从 `currentTask / currentAction / nextStep` 中显式拆出来;后台任务启动、工具 observation、完成和失败都会刷新该状态,开发窗口、项目内 Agent 对话弹窗、主窗口 Agent 状态列表、`agent.run_status` observation 和下一轮 planning prompt 都展示同一份目标 / 等待信息,避免开发者只能从动作文本里猜 Agent 卡在 LLM、工具、同伴还是人工输入。 - 2026-07-10 补充,2026-07-12 更新:Agent Runtime state 新增 `loopIteration / maxLoopIterations / toolActionBudget`,结构化记录后台 Agent loop 当前轮次、当前 6 轮上下文压缩窗口的结束轮次和每轮工具动作预算;开发窗口 Runtime 面板、主窗口 Agent 状态列表、`agent.run_status` observation 和下一轮 planning prompt 都展示该进度,帮助判断 Agent 是刚开始规划、正在 replan,还是接近当前窗口边界。`maxLoopIterations` 会随窗口推进显示 6、12 等结束轮次,只做运行观测,不构成单个 run 的固定轮数上限,也不改变工具权限。 -- 2026-07-10 补充:Agent Runtime state 新增 `planSteps / activePlanStepIndex`,从 Agent 输出的 `plan` 派生结构化计划步骤,并在工具 action / observation / response / error 生命周期中更新 `pending / active / completed / failed` 和 detail;开发窗口 Runtime 面板、主窗口 Agent 状态列表、`agent.run_status` observation 和下一轮 planning prompt 都展示当前计划步骤与步骤进度,避免只能展示一串不可定位的 plan 文本。 +- 2026-07-10 首版、2026-07-15 由 V1.17 更新:Agent Runtime state 使用 `planRevision / planExplanation / planSteps / activePlanStepIndex` 保存同一 run 的持久结构化计划。`planUpdate` 最多提交 8 个 `pending / in_progress / completed` 步骤且至多一个 `in_progress`;已完成和历史快照中已有的失败终态步骤保留,revision 只在真实变化时单调递增。外层 run 进入 `failed / budget-exhausted` 时原样保留最后可信 revision、说明、步骤状态与 active index,不把 `pending / in_progress` 机械改成 `failed`。旧 `plan` 派生 `active / pending` 并按工具 action 下标自动推进的逻辑只保留给尚未建立结构化计划的 legacy fallback;`planRevision > 0` 后工具动作、observation 和 response 不得自动改写步骤状态,未完成步骤阻止 final。开发 UI、项目内开发面板、`agent.run_status` 与 CLI 有界展示完整计划,普通用户 Supervisor 只显示紧凑摘要。 - 2026-07-10 补充:`recentEvents` 接入前端归一态和 Runtime 状态面板,事件事实源仍是 `.agent/runtime/events/.jsonl`;面板按时间展示最近 `thinking_summary / plan / action / observation / response / error` 事件,现在能同时看到 Agent 的计划、最近观察、最近事件、最近工具动作和任务队列。 - 2026-07-10 补充:后台 Runtime 每次追加 `.agent/runtime/events/.jsonl` 后会通过 Tauri `game-creator-agent-runtime-update` 事件广播当前 `AgentRuntimeResult`,开发单 Agent 聊天页、项目内 Agent 对话弹窗和主窗口 Agent 状态卡用同一套前端归一化逻辑合并状态;该事件只做实时 UI 通知,`.agent/runtime/agents`、`events` 和 `tasks` 仍是重开项目后的事实源。 - 2026-07-10 补充:后台 Agent loop 的统一语义事件类型为 `thinking_summary / plan / action / observation / response / error`。普通失败和 loop 预算耗尽都会追加 `error` 事件,并继续保留 `turn.failed / turn.budget_exhausted` 生命周期事件兼容既有读取方;开发窗口、项目内 Agent 对话弹窗和主窗口状态卡通过现有最近事件列表直接展示统一错误事件及其安全详情。状态面板默认保持最新 4 条的紧凑视图,当前后端返回的最近事件超过 4 条时可展开查看全部返回记录,确保同一 run 的六类语义事件不会因 UI 硬截断而无法检查。 - 2026-07-11 补充:开发单 Agent 聊天页继续使用整页纵向滚动,不把 Runtime 锁进固定视口;聊天消息区使用固定响应式高度并在内部滚动,避免历史消息持续撑高聊天面板。可选的 Runtime 恢复确认区始终占据独立布局行,不能与 Runtime 详情或聊天消息重叠。Runtime 状态面板支持折叠详情,折叠时只卸载目标、计划、事件、动作和任务等详情 DOM,仍保留状态标题与取消、重试、确认、拒绝、刷新操作;等待 LLM 时在消息区持续显示连接 / 等待首包 / 接收中的动态状态和“请求仍在进行中”提示。流式聊天的连续 delta 通过 `requestAnimationFrame` 合并为每帧最多一次消息更新,delta 不重复提交未变化的 Runtime state;OpenAI Chat SSE 的空数组或 `null` `choices` 心跳 / 元数据事件会跳过,usage-only 尾包会回填最终 token usage,finish-only 事件会把结束原因送入状态流,上游 error 保留真实消息,`[DONE]` 立即结束读取;正文与 finish reason 已接收后即使尾包异常也保存完整正文,不再改判整轮失败。持久事件订阅失败时显示非致命 Runtime 错误;聊天事件监听不可用时必须调用真正的非流式入口,不得因 Agent 配置仍为 `stream=true` 在 Rust 内部再发 SSE 请求;流式请求在首个文本片段前遇到允许的协议兼容错误时只降级一次普通回复。 - 2026-07-12 调整:开发单 Agent 对话框新增 `执行 / 聊天` 分段模式,默认 `执行`。默认发送直接调用 `start_game_creator_agent_runtime_task`,复用工具规划、权限确认、取消、队列和 Runtime 实时状态;`聊天` 作为显式模式继续走不执行工具的流式回复。消息区在 Runtime 启动、排队、等待 LLM、执行工具、等待确认和同步终态回复期间持续显示当前状态,不再要求开发者从页头文案猜测请求是否仍在运行;原独立“后台运行”按钮移除。Runtime 必须先取得项目写锁并重读项目 revision 与当前 run 的 verification gate;只有 run 从未要求验证,或成功验证绑定的 revision 与锁内当前 revision 完全一致,才允许在同一把锁内先把最终 assistant 回复可靠写入当前 Agent Session,再写 completed 终态并广播事件。完成预检、finalization、恢复和取消收束遇到同项目另一个 Agent 的短暂写锁时,最多等待约 1 秒后重试;锁持续占用才返回 blocker,不能把毫秒级竞争误判为验证缺失并重新请求 LLM。每次可形成最终回复的 planning 请求或独立 final reply 请求开始前都记录 `responseRevision`;回复完成后锁内当前 revision 与它不同即返回可恢复 `Stale`,该规则同样覆盖 `requiresVerification=false` 的只读 run。旧回复不得进入会话或 completed,Runtime 记录 completion blocker、`response.stale` 事件与审计,并保持原 Agent、Task、Session、Run、loop 计数和 per-Agent 锁回到 planning,重新读取或验证当前项目状态后再生成回复。revision / gate 读取失败或 stale continuation 持久化失败仍进入 failed;一旦 finalization journal 已进入 `prepared`,后续对话或终态落盘失败改为保持可恢复 `finalizing`,不得把当前 run 误记为 failed。前端只对当前项目、Agent、Session 和 runId 匹配的终态事件自动重读对话,直到看到新 assistant 消息或重试结束,切换 Session 后旧 run 不得污染当前聊天记录。 -- 2026-07-12 补充并冻结:后台最终回复通过 `.agent/runtime/finalizations//.json` 的 `game-creator-runtime-finalization.v1` journal 跨越多文件落盘,状态严格按 `prepared -> assistant-persisted -> runtime-completed` 推进,终态可靠投影后删除 journal。journal 绑定项目、Agent、Task、Session、Run、source、父委派身份、任务正文、回复指纹、`responseRevision`、verification gate、`finalizationId` 和稳定 `messageId`;finalization 单独使用 512 KiB 上限,必须容纳 32,000 字符的最大合法回复及元数据。跨平台替换先写临时文件;目标平台不能原子覆盖旧文件时,先把旧 journal 原子移动为同目录 `.previous` 恢复副本,再安装新文件,读取时主文件缺失必须回退恢复副本,成功推进或终态清理时同时删除副本,禁止先删除唯一旧 journal。`resume` 在 pending action、普通 running/pending task 和 delegate receipt 修复之前优先恢复 finalization,只按 journal 补齐 assistant 与终态,不重新请求 LLM、不重放工具或 receipt 任务;Runtime state 只是可重建投影,状态文件缺失或损坏但 task ledger 仍能唯一定位主 journal 或恢复副本时,必须从 task ledger 重建同一 run 后继续恢复。`prepared` 且 assistant 尚未落盘时若任务已取消,或当前 revision / verification gate 已使回复过期,则丢弃 journal,分别保持取消终态或回到同 run planning;assistant JSONL 是用户可见提交点,取消 command 必须在持有 per-Agent 锁后检查 journal,assistant 尚未存在时立即删除 prepared journal 再取消,assistant 已存在时则完成原 finalization 并忽略迟到取消,不能留下孤儿 journal 或把可见回复改判为 cancelled。journal 损坏、版本不支持、身份/回复指纹/幂等 ID 冲突必须 fail closed,并把同一 live run 投影为 `status=running / phase=finalizing` 供 UI 明确显示,不能降级为普通任务恢复。 +- 2026-07-12 补充并冻结,2026-07-15 由 V1.17 升级:后台最终回复通过 `.agent/runtime/finalizations//.json` 的 `game-creator-runtime-finalization.v2` journal 跨越多文件落盘,状态严格按 `prepared -> assistant-persisted -> runtime-completed` 推进,终态可靠投影后删除 journal。journal 除绑定项目、Agent、Task、Session、Run、source、父委派身份、任务正文、回复指纹、`responseRevision`、verification gate、稳定 `messageId` 外,还绑定最终 `planRevision / planExplanation / plan / planSteps / activePlanStepIndex` 完整快照及 `planSnapshotFingerprint`,并把计划指纹纳入 `finalizationId`;v1 只作不含计划快照的 legacy 读取。finalization 单独使用 512 KiB 上限,必须容纳 32,000 字符的最大合法回复及元数据。跨平台替换先写临时文件;目标平台不能原子覆盖旧文件时,先把旧 journal 原子移动为同目录 `.previous` 恢复副本,再安装新文件,读取时主文件缺失必须回退恢复副本,成功推进或终态清理时同时删除副本,禁止先删除唯一旧 journal。`resume` 在 pending action、普通 running/pending task 和 delegate receipt 修复之前优先恢复 finalization,不重新请求 LLM、不重放工具或 receipt 任务。assistant 已落盘而 Runtime state 缺失或损坏时,可从唯一 task record 重建同一 run,并从 v2 journal 恢复原结构化计划后补齐 completed;assistant 尚未落盘而 state 不可读时必须保留 journal、进入 `needs-reconciliation`,不得只凭 task 或 prepared journal 猜进度并写 assistant。`prepared` 且 assistant 尚未落盘时若任务已取消,或当前 revision、verification gate、结构化计划已使回复过期,则丢弃 journal,分别保持取消终态或回到同 run planning;assistant JSONL 是用户可见提交点,取消 command 必须在持有 per-Agent 锁后检查 journal,assistant 尚未存在时立即删除 prepared journal 再取消,assistant 已存在时则完成原 finalization 并忽略迟到取消,不能留下孤儿 journal 或把可见回复改判为 cancelled。journal 损坏、版本不支持、身份/回复/计划指纹/幂等 ID 冲突必须 fail closed,并把同一 live run 投影为 `status=running / phase=finalizing` 或 `needs-reconciliation` 供 UI 明确显示,不能降级为普通任务恢复。 - 2026-07-12 补充并冻结:本地 conversation JSONL 的 `messageId` 是可选向后兼容字段,旧记录无需迁移仍可读取。finalization 使用稳定 `messageId` 幂等追加 assistant;同一 Agent / Session 下相同 ID 且 role/content 一致时不得重复写 JSONL,若消息已存在但 `.agent/agent.db` 缺少对应 `conversation.message` audit,重试必须在 audit 追加锁内补写一次,已有 audit 不重复;同一 Agent / Session 作用域下相同 ID 对应不同 role 或 content 时继续按冲突失败关闭。completed task JSONL、Runtime state、`turn.completed / response` 事件、`agent.runtime.completed / background_task.completed` audit、pending/confirmation 清理和 delegate result 发布均按既有身份幂等补齐,重启不得制造第二份终态投影。 - 2026-07-11 补充:后台单 Agent 新增 Codex 风格的代码导航与局部编辑闭环。`project.search` 接受 `query / path / maxResults / caseSensitive`,在项目边界内做字面量搜索并返回 `path:line`,最多扫描 500 个、单个不超过 512 KiB 的文本文件,跳过 `.agent`、`.git`、`node_modules`、`dist`、`build`、`target`、`.next`、`coverage` 和 `.env*`;该工具映射到 `file.read` 权限。`file.read` 接受 `startLine / maxLines`,返回带行号的指定片段、总行数和下一页提示,单次最多 240 行、8,000 字符。`file.patch` 接受 `path / oldText / newText / expectedReplacements`,只在实际匹配数与预期一致时持锁写入,目标文件和修改后文件最大 2 MiB,成功后写 `agent.runtime.file.patch` 审计;该工具映射到 `file.write` 权限。Agent planning prompt 明确要求批量修改前创建 checkpoint,并可在修改后再次 `file.read` 验证;本轮不开放任意 shell 命令。 - 2026-07-12 补充:后台单 Agent 文件生命周期加入 `file.delete`。输入只接受项目内相对 `path`,工具使用独立且默认需确认的 `file.delete` 权限,不继承 `file.write`;Runtime 复用通用文件工具的绝对路径、父目录、反斜杠、有效或悬空符号链接和目录防护,并额外禁止删除整个 `.agent/**` 控制面。pending action v3 绑定本轮 planning 请求发出前的全局 project revision;`file.delete` 取得项目写锁后会在实际副作用前再次校验该 revision 和 verification gate,再保守推进新 revision 与 gate 并删除文件。项目/session memory、blackboard 和 canvas asset 的 Runtime 写入同样推进全局 revision,防止它们改写待删除目标却不触发漂移。删除成功后写入不含文件正文的 `agent.runtime.file.delete` 审计;目标已不存在时返回幂等 observation,但仍保留已经推进的验证门禁。等待期间发生进程内 Agent 修改时,旧动作进入 `needs-reconciliation`,不能删除漂移后的路径;进程中断在 `executing` 时同样不得自动重放删除。v1 威胁模型只承诺客户端自身遵守项目锁的并发写入;外部进程在路径校验后把父目录替换为符号链接的 TOCTOU 攻击不在本轮承诺内,如需防御必须升级为目录句柄与 no-follow `unlinkat` 级别的平台实现。该工具只进入开发单 Agent Runtime,普通用户聊天和正式用户窗口不增加文件删除入口。 @@ -124,12 +132,13 @@ Agent Runtime 负责: - 2026-07-11 补充:开发验证可用 `npm run ai-game-creator-shell:agent-task -- [--init] ` 无 UI 启动单 Agent 后台任务。CLI 只负责可选初始化、调用现有 Runtime、按 runId 轮询终态并打印 `status / phase / replyText / pendingActionId`,不复制 planning 或工具执行逻辑;默认 10 分钟轮询上限。`waiting-for-confirmation` 会以非零状态退出并要求转到开发窗口确认,CLI 不提供跳过项目权限的自动确认参数。该入口用于真实 provider 的可重复端到端验收,不进入普通用户界面。 - 2026-07-11 补充,2026-07-13 调整:后台工具规划与最终回复的 LLM 请求新增可恢复错误重试:`LlmError::EmptyResponse` 原样自动重试最多 3 次;`Timeout / Connectivity / Transport` 及上游 `408 / 429 / 5xx` 最多额外重试 5 次并按 `500ms / 1000ms / 1500ms / 2000ms / 2500ms` 退避。配置、请求、流能力、反序列化错误及其他 `4xx` 不重试。重试发生在工具计划被解析和执行前,或最终回复尚未落盘时,因此不会重复执行已经落盘的工具副作用;重试耗尽后仍写入原有 `error / turn.failed` 事件并把失败消息追加到当前 Agent 会话。 - 2026-07-11 调整,2026-07-12 更新:后台单 Agent planning loop 每 6 轮形成一个上下文压缩窗口,每轮最多 3 个工具动作;6 轮是窗口大小,不是单个 run 的固定上限。`loopIteration` 在同一 run 内连续递增,`maxLoopIterations` 指向当前窗口的结束轮次;待确认或重启恢复后按 context bundle 的 `nextLoopIndex` 在同一 run 继续。每个窗口结束时压缩已有 observation;窗口产生新的独立观察时继续下一窗口,最近 6 轮没有独立进展或相邻窗口指纹重复时才写入 `failed / budget-exhausted` 和 `loop-budget-exhausted`,不生成总结伪装完成。这只调整后台单 Agent Runtime;游戏草案 Generator/Evaluator 仍保持独立的 3 轮修复预算。旧摘要中“后台最多 3 轮”或“整个 run 最多 6 轮”的描述不再有效。 -- 2026-07-12 补充并冻结,2026-07-13 调整容量:后台 Agent 每个 run 的可恢复 planning 上下文通过临时文件替换原子写入 `.agent/runtime/context-bundles//.json`,绑定 Agent、Task、Session、Run、任务正文和 revision / verification gate 关联,schema 固定升级为 `game-creator-runtime-context-bundle.v2`;保存 `nextLoopIndex`、当前窗口、计划、fallback response、压缩后的 observation、上一窗口指纹和 `contextStalled`。stale continuation 必须清空旧 actions 与 fallback response,保留 blocker、loop 位置和窗口进度;`contextStalled` 一旦在窗口边界成立,同 run 重规划和进程重启都不得清除。`runtime.verification` 的上下文指纹忽略动态 revision 数值前缀,仅保留稳定处置指引;成功 `project.verify / game.static_smoke` 的动态命令输出不进入窗口指纹。revision 数字或时间戳持续变化本身不算独立进展,重复 stale 最迟在相邻窗口指纹重复时以 `loop-budget-exhausted` 终止。单文件最多 128 KiB、最多 12 条 observation;普通 observation detail 仍压到 1,600 字符,最新内容 diff 可保留到 24,256 字符并在窗口压缩时优先保留。写入前统一截断并过滤敏感内容和项目绝对路径,安全校验失败时拒绝落盘。读取时要求普通文件并校验 schema、Agent、Session、Run、任务正文、observation 数量和 revision / gate 关联,身份不一致时拒绝续跑。v1 context bundle 恢复必须失败关闭,不自动迁移,也不能把缺失 gate 当成 `requiresVerification=false`;revision 与验证资格仍以锁内重读的独立持久化文件为准,bundle 只保存恢复上下文。该文件属于 Runtime 私有控制面,不等同于根级 `.agent/context.bundle.json`,不得由通用文件工具暴露。 +- 2026-07-12 补充并冻结,2026-07-13 调整容量,2026-07-15 由 V1.17 升级计划快照:后台 Agent 每个 run 的可恢复 planning 上下文通过临时文件替换原子写入 `.agent/runtime/context-bundles//.json`,绑定 Agent、Task、Session、Run、任务正文和 revision / verification gate 关联;当前 schema 为 `game-creator-runtime-context-bundle.v3`,除 `nextLoopIndex`、当前窗口、fallback response、压缩 observation、上一窗口指纹和 `contextStalled` 外,还持久化 `planRevision / planExplanation / planSteps / activePlanStepIndex`。读取 v3 必须让结构化计划 revision 与完整快照匹配当前 Runtime state;任一不一致失败关闭。v2 保持读取兼容:通过原有身份、任务、revision 和 gate 校验后,从当前 Runtime state 补入计划字段并按 v3 继续,后续 checkpoint 写 v3;v1 以及缺失既有 gate 关联的记录仍失败关闭,不自动迁移。stale continuation 必须清空旧 actions 与 fallback response,保留 blocker、loop 位置、窗口进度和结构化计划;`contextStalled` 一旦在窗口边界成立,同 run 重规划和进程重启都不得清除。`runtime.verification` 的上下文指纹忽略动态 revision 数值前缀,仅保留稳定处置指引;成功 `project.verify / game.static_smoke` 的动态命令输出不进入窗口指纹。revision 数字或时间戳持续变化本身不算独立进展,重复 stale 最迟在相邻窗口指纹重复时以 `loop-budget-exhausted` 终止。单文件最多 128 KiB、最多 12 条 observation;普通 observation detail 仍压到 1,600 字符,最新内容 diff 可保留到 24,256 字符并在窗口压缩时优先保留。写入前统一截断并过滤敏感内容和项目绝对路径,安全校验失败时拒绝落盘。revision 与验证资格仍以锁内重读的独立持久化文件为准,bundle 只保存恢复上下文;该文件属于 Runtime 私有控制面,不等同于根级 `.agent/context.bundle.json`,不得由通用文件工具暴露。 - 2026-07-11 调整:后台任务的可执行正文上限统一为 4,000 字符。入队 JSONL、启动后的 `currentTask/currentGoal`、planning prompt、待确认动作 task context、确认续跑和重启恢复都保留同一份正文;对话仍保存用户原始消息。状态事件、列表卡片和 `agent.db` 摘要可继续使用较短安全预览,但不能再反向作为后续 LLM 执行输入。这样长任务末尾的验收标记和输出格式要求不会在队列边界被 180 字符截断。 - 2026-07-11 调整,2026-07-12 由 Runtime V1.2 更新:后台 planning 使用 4,000 输出 token,最终回复使用 2,400,并继续叠加最多 3 次 EmptyResponse 重试。推理档位不再硬编码为 `low`:planning、普通单 Agent 聊天和最终回复统一使用解析后的 `llm.reasoningEffort`,`agentLlm..reasoningEffort` 有值时覆盖全局、缺省时继承全局;取值只允许 `default / low / medium / high`,发布默认 `high`,`default` 表示不向 Provider 发送推理档位。 -- 2026-07-11 补充:后台单 Agent 的工具 planning 响应必须提供可反序列化为 `thinkingSummary / plan / actions / response` schema 的 JSON object。Runtime 从模型输出中解析首个完整对象,因此对象后的尾随说明可以忽略;只有普通文本、没有完整对象,或对象无法反序列化时都不构成有效工具计划。对于这两类无效输出,Runtime 最多追加 2 次自动格式修复请求,每次只把限长且经过统一敏感信息过滤的上一次输出作为修复上下文,并把修复尝试写入 `.agent/agent.db` 的 `agent.runtime.tool_plan.repair` 审计。修复预算耗尽后进入既有工具规划失败路径,不得把普通文本折算为空 actions + response,也不得因此进入 completed;最终回复阶段仍按其独立的普通文本契约处理。 -- 2026-07-12 补充:OpenAI Chat / Responses 的后台工具 planning 优先注册唯一的 `submit_agent_tool_plan` function tool,并使用字符串形式 `tool_choice=required` 和 strict schema;Runtime 只接受恰好一次同名 function call,并把 arguments 复用现有 `AgentRuntimeToolPlan` 校验与两次格式修复循环。错误函数名、多次调用和非法 arguments 都不得执行工具。Anthropic 保留文本 JSON 回退,planning 强制非流式,最终普通回复继续按 Agent 配置决定是否流式。`platform-llm` 会在本地拒绝无 function tools 的 tool choice 和 Anthropic function tools,并把协议类型写入 `agent.runtime.tool_plan.protocol` 审计。 -- 2026-07-11 调整:工具计划四个顶层字段均为必填并拒绝未知顶层字段;thinkingSummary 与 action.tool 必须非空。这样 `{}`、前置无关 JSON 或结构不完整对象会触发格式修复,不会成为假完成信号。空 actions 表示 planning 收束;response 非空时直接采用,response 为空时进入独立最终回复生成。`agent.runtime.project.verify` 记录补充 `runId / actionId / actionFingerprint`,用于在多 Agent 并行验证时把命令终态与具体 Runtime 动作关联。 +- 2026-07-11 补充,2026-07-15 由 V1.17 更新:后台单 Agent 的工具 planning 响应必须提供可反序列化为 `thinkingSummary / planUpdate / plan / actions / response` schema 的 JSON object。Runtime 从模型输出中解析首个完整对象,因此对象后的尾随说明可以忽略;只有普通文本、没有完整对象,或对象无法反序列化时都不构成有效工具计划。对于这两类无效输出,Runtime 最多追加 2 次自动格式修复请求;同一次 planning 的私有 repair 请求可携带限长且经过统一敏感信息过滤的上一条模型输出或 function call 预览与协议错误,以便 Provider 真正修正格式。`.agent/agent.db` 的 `agent.runtime.tool_plan.repair` 公共审计只写 attempt/maxAttempts、protocol,以及错误、输出/调用体预览、callId 和 functionName 的 SHA-256、字符数或计数,不保存原始模型正文、错误或 function arguments。修复预算耗尽后进入既有工具规划失败路径,不得把普通文本折算为空 actions + response,也不得因此进入 completed;最终回复阶段仍按其独立的普通文本契约处理。旧文本协议可省略 `planUpdate`,但只能继续走 legacy `plan` fallback。 +- 2026-07-12 补充,2026-07-15 由 V1.17 更新:OpenAI Chat / Responses 的后台工具 planning 优先注册唯一的 `submit_agent_tool_plan` function tool,并使用字符串形式 `tool_choice=required` 和 strict schema;Runtime 只接受恰好一次同名 function call,并把 arguments 复用现有 `AgentRuntimeToolPlan` 校验与两次格式修复循环。strict arguments 中 `planUpdate` 必须出现但可为 `null`,使用结构化更新时 legacy `plan` 必须为空。错误函数名、多次调用和非法 arguments 都不得执行工具。Anthropic 保留文本 JSON 回退,planning 强制非流式,最终普通回复继续按 Agent 配置决定是否流式。`platform-llm` 会在本地拒绝无 function tools 的 tool choice 和 Anthropic function tools,并把协议类型写入 `agent.runtime.tool_plan.protocol` 审计。 +- 2026-07-11 调整,2026-07-15 由 V1.17 更新:工具计划五个顶层字段均为必填并拒绝未知顶层字段;`thinkingSummary`、结构化计划的 `explanation / step` 与 `action.tool` 必须非空。`planUpdate` 只接受 `null` 或最多 8 个唯一步骤,状态限于 `pending / in_progress / completed` 且至多一个 `in_progress`。这样 `{}`、前置无关 JSON 或结构不完整对象会触发格式修复,不会成为假完成信号。空 actions 只有在 verification、process/join/delivery 和结构化计划完成门禁都通过后才表示 planning 收束;response 非空时直接采用,response 为空时进入独立最终回复生成。`agent.runtime.project.verify` 记录补充 `runId / actionId / actionFingerprint`,用于在多 Agent 并行验证时把命令终态与具体 Runtime 动作关联。 +- 2026-07-15 V1.17 公共审计收紧:`thinking_summary` event 只保存固定摘要、正文 SHA-256 与字符数,legacy `plan` event 只保存步骤数;结构化计划审计只保存 explanation 的哈希与字符数,以及 step 标题哈希、状态和数量。模型 thinking、legacy plan 标题、repair 错误和调用体只允许出现在对应私有 Runtime 上下文或有界 repair 请求中,不得复制到公共 event、task 或 Agent DB 正文字段。 - 2026-07-10 补充:Agent Runtime state / result 新增 `taskQueue`,从 `.agent/runtime/tasks/.jsonl` 中每个 `runId` 的最新记录汇总 `total / pending / running / completed / failed / latestRunId`;开发窗口 Runtime 面板、主窗口 Agent 状态列表、`agent.run_status` observation 和下一轮 planning prompt 都读取该摘要,用于判断同一 Agent 是否仍有排队任务。该字段是运行观测摘要,不新增调度器、SQLite 或独立 worker。 - 2026-07-10 补充:Runtime 新增 `agent.schedule_ready` 调度入口。开发构建可在权限确认后扫描 manifest ready task,把依赖已完成且仍为 `pending` 的任务标成 `running`,并按 taskId 投递到对应 Agent 的既有后台队列;source 固定为 `agent-ready-task-scheduler`,审计记录写 `agent.runtime.ready_task.scheduled`。该入口只把 manifest ready task 接入现有 per-agent 队列、锁、JSONL、LLM loop、工具策略和事件流,不新增独立 worker,也不会在默认确认策略下静默启动。 - 2026-07-10 补充:主窗口 Agent 状态栏的“调度 Ready”只在开发模式显示。点击后复用项目策略确认弹窗,确认通过才调用 `schedule_game_creator_agent_ready_tasks`,并把返回的 Runtime 合并回 Agent 状态卡;普通用户窗口继续只展示状态和单 Agent 对话入口,不直接暴露 ready-task 调度按钮。 @@ -150,7 +159,7 @@ Agent Runtime 负责: - 2026-07-10 补充:后台 planning 与预算内 final reply 使用专用最小上下文,只预置 Agent 身份、sessionId、runId、执行模式和工具策略;Agent 私有记忆、项目记忆、黑板、对话、资产、项目索引与文件正文只能经对应工具通过权限 gate 后作为 observation 进入下一轮。普通前台聊天仍可使用角色上下文。长黑板、记忆和对话按尾部截断,确保最新结论与最新定向消息优先保留。 - 2026-07-10 补充:同一 Agent 的前台直接聊天、流式聊天和后台任务统一使用 `.agent/runtime/locks/.lock` OS 文件锁。前台聊天不再在整个 LLM 请求期间占用项目级写锁;同 Agent 后台任务在前台运行时只入队,前台成功或失败后把当前 Agent 锁直接移交给 drain,不重新抢锁,也不允许 drain 启动异常把已经完成的聊天结果改判为失败。不同 Agent 继续并行,真实项目写工具只在副作用执行期间短暂申请项目写锁。 - 2026-07-10 补充:默认 `agent.resume=confirm` 时,客户端自动恢复命令只做 auto gate 并返回待确认错误;主工作区和独立开发 Agent 聊天窗口在首次读取项目 Runtime 时都必须显示 `agent.resume` 确认条,确认对象绑定发起时的项目路径,切换项目会取消旧确认,异步返回后也不得把旧项目 Runtime 合并到新项目 UI。开发者确认后调用独立 `confirm_resume_game_creator_agent_runtime_tasks`,该命令仍执行 deny-only 权限检查后才接回 durable queue。临时调用失败不锁死项目路径,允许后续刷新重试;明确 deny 或取消都不恢复任务。 -- 2026-07-10 补充,2026-07-12 更新:后台 Agent 只有返回空 `actions` 且不存在 `project.verify` blocker 才视为 loop 已收束。每 6 轮只是上下文压缩窗口;窗口产生新的独立 observation 时压缩上下文并继续同一 run,最近 6 轮没有独立进展或相邻窗口重复时终态才写为 `status=failed / phase=budget-exhausted`,error 使用 `loop-budget-exhausted` 机器可读前缀,不再调用 final reply 后写 completed 审计。解析阶段保留过滤后的 action 总数,每轮超过 3 个 action 时写入 `runtime.tool_budget` observation 并只执行前三个,要求下一轮重新排序。Runtime 默认 `allowedTools` 直接由实际可执行工具白名单派生,避免 UI 观测与执行边界漂移。 +- 2026-07-10 补充,2026-07-12 更新,2026-07-15 增加 V1.17 完成门禁:后台 Agent 返回空 `actions` 后,只有不存在 `project.verify` 等既有 blocker,且当前结构化计划的全部必要步骤均为 `completed`,才视为 loop 已收束。工具 action 序号和成功 observation 不会自动推进结构化计划;未完成时 Runtime 返回 `runtime.plan_update` blocker,在同一 run 要求 Agent 按真实进度更新。每 6 轮只是上下文压缩窗口;窗口产生新的独立 observation 时压缩上下文并继续同一 run,最近 6 轮没有独立进展或相邻窗口重复时终态才写为 `status=failed / phase=budget-exhausted`,error 使用 `loop-budget-exhausted` 机器可读前缀,不再调用 final reply 后写 completed 审计。解析阶段保留过滤后的 action 总数,每轮超过 3 个 action 时写入 `runtime.tool_budget` observation 并只执行前三个,要求下一轮重新排序。Runtime 默认 `allowedTools` 直接由实际可执行工具白名单派生,避免 UI 观测与执行边界漂移。 - 任务图能力:每轮 Orchestrator agenda、ready / active task 选择、Evaluator 结构化返工路由、返工轮 carry-over。 - 记忆能力:短期记忆 `memory/session.md`、长期记忆 `memory/project.md`、项目级黑板 `memory/blackboard.md` 和角色私有记忆 `memory/agents//.md`;黑板用于共享重要跨 agent 记忆,角色私有记忆只给对应角色 brief 读取和追加。最近 project / agent conversation 会作为短期 prompt 上下文读取,不替代正式 memory 文件。 - 对话能力:结构化对话记录统一落在 `.agent/conversations/` 的 append-only JSONL;普通聊天写 `.agent/conversations/project.jsonl`,进入单个 agent 后只写对应 `.agent/conversations/agents/.jsonl`,不把原始对话混进项目黑板或角色私有记忆。 @@ -305,6 +314,7 @@ game-project/ ## v1 验收证据矩阵 - `npm run ai-game-creator-shell:check`:覆盖壳 typecheck、聊天命令单测、用户 / 开发窗口 UI 边界 smoke、主窗口命令按钮复用 `/help` 命令列表、主窗口项目摘要从 manifest / trace 派生任务完成数、ready 数、资产来源分布和最近命令且未选项目时不显示、灵感草稿只填充输入框不提交、能力按钮复用 `/capabilities`、LLM状态按钮复用 `/llm-status` 且结果回填 Agent 状态列表、聊天侧 `/agents` 汇总和单 Agent 对话的 provider / 模型 / 流式 / API Key 读取状态、开发日志面板只读读取 `.agent/logs/command.log` / `preview.log` / `agent.log`、项目状态按钮复用 `/status`、权限按钮复用 `/policy` 且策略草稿按钮只填入 `/policy-confirm project.index` / `/policy-confirm asset.register` / `/policy-confirm memory.write` / `/policy-confirm preview.start` / `/policy-confirm preview.open` / `/policy-confirm preview.stop` / `/policy-confirm agent.run_status` / `/policy-confirm conversation.read` / `/policy-confirm conversation.write`、审计按钮复用 `/audit`、资产按钮复用 `/assets` 且资产结果可一键复用 `/read`、任务按钮复用 `/tasks`、聊天侧 `/agents` 汇总每个 Agent 的当前状态、聊天侧 `/agent-conversations` 列出 Agent 对话读取命令、聊天侧 `/agent-memories` 列出 Agent 私有记忆读取命令、Trace 按钮复用 `/trace`、文件按钮复用 `/files` 且文件结果可一键复用 `/read`、索引按钮复用 `/index`、记忆 / 短期记忆 / 黑板按钮复用 `/memory long|short|blackboard`、快照按钮复用 `/checkpoint`、快照列表按钮复用 `/checkpoints` 且 checkpoint 结果可一键复用 `/diff` / `/restore`、历史按钮复用 `/history`、受限命令白名单按钮复用 `/commands` 且无需项目初始化、静态自检快捷按钮复用 `/smoke`、预览状态快捷按钮复用 `/preview-status`、主窗口运行时配置面板读写 Tauri 配置目录中的 `game-creator.config.json`、支持全局与每个 agent 单独选择 LLM Provider 且不把 API Key 写入聊天、单 Agent 对话面板可手动追加私有记忆且走 `memory.write` 策略、主窗口提供音效登记、画板音频导入和常用生成产物读取草稿入口,聊天侧 `/art` 可盘点美术素材且不直接触发平台生成或画板同步,聊天侧 `/context` 可盘点生成上下文来源且不直接读取上下文文件,聊天侧 `/timeline` 可汇总项目活动时间线且不直接读取日志或 trace 文件,聊天侧 `/artifacts` 可列出常用生成产物读取命令,聊天侧 `/run-artifacts` 可列出最近 run 产物读取命令,聊天侧 `/run-files` 可列出 Agent 运行辅助文件读取命令,聊天侧 `/logs` 可列出固定日志读取命令且不直接读取日志,聊天侧 `/brief` 只基于当前已加载的 manifest / 最近 run trace / 预览状态 / 资产数量 / 最近命令生成项目简报,聊天侧 `/goal` 只基于当前 manifest.goal / 最近 run goal / taskGraph.goal 汇总创作目标来源,提供 `/next` 或 `/agent-resume 细化目标:` 后续草稿且不直接触发 Tauri 读写、文件读取、预览启动或新增面板,聊天侧 `/mvp` 只基于当前 manifest / 最近 run trace / preview / 任务 / 资产 / 最近命令汇总本轮最小可玩范围,提供 `/run` 等后续草稿且不直接触发 Tauri 读写、文件读取、预览启动、导出或新增面板,聊天侧 `/audience` 只基于当前 manifest / 最近 run trace / preview 准备首批试玩对象和观察重点且不直接触发 Tauri 读写、预览启动、导出或继续 run,聊天侧 `/feedback` 只基于当前 manifest / 最近 run trace / preview 准备试玩反馈模板和修改说明草稿且不直接触发 Tauri 读写、预览启动或继续 run,聊天侧 `/next` 基于当前已加载的 manifest / 最近 run trace 输出下一步建议和 `/goal` / `/mvp` / `/accessibility` / `/performance` / `/tasks` / `/criteria` / `/groups` / `/budget` / `/qa` / `/changes` / `/trace` / `/run` / `/open-preview` / `/test-plan` / `/audience` / `/feedback` / `/assets` / `/art` / `/context` / `/timeline` / `/artifacts` / `/run-artifacts` / `/run-files` / `/logs` / `/agent-resume ` 等安全命令草稿方向,提供一个首选草稿且不直接触发 Tauri 读写、预览启动或文件读取,聊天侧 `/capabilities` 展示标准 Agent 能力清单且不打开开发面板、聊天侧 `/audit` 从 manifest / 本地文件 / `.agent/run.latest.json` 分别汇总用户面、6 组任务配置、6 组协作证据、任务编排、loop、记忆、本地产物、HTTP 预览、画板回流和权限日志证据且不打开开发面板;未生成 `.agent/run.latest.json` 前,`/audit` 只能标记任务配置通过,不能把 6 组协作证据误判为通过;trace 已存在但状态为 `failed`、`needs-revision`、`running`、`max-passes-exhausted` 或缺少 `Evaluator passed` 步骤时,`/audit` 不能把 loop 误判为通过。聊天侧 `/llm-status` 只显示 base_url / model / API Key 已读取状态且不泄露密钥本体、聊天侧长期记忆查看 / 追加 / 覆盖 / 删除的授权本地项目路径、上传资产写入后的 manifest 刷新和 `/assets` 聊天可见性、`/smoke` 聊天侧确认后只通过授权本地项目路径执行白名单 `game.static_smoke`、`/run` 聊天侧确认后通过授权本地项目路径执行 `game.static_smoke`、启动 `127.0.0.1` 本地预览并交给外部浏览器、`/preview` 聊天侧确认后通过授权本地项目路径启动 `127.0.0.1` 本地预览并交给外部浏览器、`/status` 聊天侧项目 / 任务 / 资产 / 预览 / 最近命令汇总、`/files` 聊天侧本地项目文件列表、`/read` 聊天侧文件读取的授权本地项目路径、`/tasks` 聊天侧任务拆分与下一步专业组展示的授权本地项目路径、聊天确认生成后实时展示 Planner / Orchestrator / 角色 brief / Generator / Evaluator / 写盘 / 自检进度,并自动读取 `.agent/run.latest.json` 在普通聊天消息里展示 Run、LLM 对话、loop 轮次、工具调用、active / carry-over 任务、返工焦点、编排轮次、最近步骤、画板同步建议命令和本地产物快照、`/trace` 聊天侧读取 `.agent/run.latest.json` 并展示 loop 轮次 / active 任务 / 返工路线 / dependency waves 的授权本地项目路径、`platform-agent` 编排测试、共享契约测试、Tauri 本地能力测试和无密钥本地 provider 端到端 smoke;用于证明独立 App、真实 LLM-compatible loop、本地落盘、自检和 HTTP 预览闭环,并覆盖 loop 跑满 3 轮失败时不会写入最终游戏产物。 +- V1.17 单 Agent 持久计划验收:Rust 定向用例覆盖 native function 显式 `planUpdate`、文本 JSON omission 兼容、输入上限、单调 revision、外层 failed / budget-exhausted 保留最后可信进度、终态保留、工具 action 下标零推进、未完成步骤阻止 final、损坏状态失败关闭、context bundle v3/v2 恢复、finalization v2 计划快照与 assistant 已落盘后的 state 丢失恢复,以及 thinking / legacy plan / repair 公共审计零正文;`appSurface.test.ts` 覆盖开发 UI 刷新后完整 8 步仍在,以及普通用户 Supervisor 只显示完成数、当前步骤、等待、下一步和协作数量。恢复/steer 专项还必须证明 Runner 重启和 same-run steer 后身份不变、旧动作零执行、终态步骤不丢、revision 不回退;真实 Provider 必须按 Runtime V1.17 章节完成无配方验收后才能记 PASS,当前状态为未验收。 - `file.delete` 的 Runtime 验收必须覆盖:删除普通文件与缺失文件的幂等结果、缺少路径、目录、绝对路径、父目录、反斜杠、有效与悬空符号链接和整个 `.agent/**` 控制面拒绝、独立 `confirm / deny` 策略、确认前无副作用、确认期间全局 revision 漂移失败关闭、durable action ledger 的 approved / executing / observed 恢复边界、`agent.runtime.file.delete` 审计,以及删除前 revision 推进、删除后必须通过当前 revision 的 `project.verify` 或 `game.static_smoke` 才能收束。另用完整后台 loop 和开发 CLI 真实任务证明 Agent 能自主选择删除并完成验证;普通用户窗口继续没有文件写入或删除入口。 - `/risks` 聊天入口由 `appSurface.test.ts` 的主窗口 smoke 覆盖:只基于当前已加载的 manifest / trace / 预览 / 任务 / 资产 / 最近命令生成风险摘要,提供首个风险处理草稿,不触发 Tauri 读写、文件读取、预览启动或新增普通用户面板。 - `/goal` 聊天入口由 `appSurface.test.ts` 主窗口 smoke 覆盖:只基于当前已加载 manifest.goal、最近 run goal 和 taskGraph.goal 汇总创作目标来源,提供 `/agent-resume 细化目标:` 或 `/next` 草稿,不触发 Tauri 读写、不读取 spec / 上下文 / trace 文件、不新增普通用户目标面板。 @@ -384,6 +394,7 @@ game-project/ - 本地项目初始化会创建 `game/`、`assets/`、`memory/`、`memory/agents/`、`exports/`、`.agent/logs/`,写入 `.agent/manifest.json`,生成 append-only JSONL 本地项目索引 `.agent/agent.db`,并生成默认 `game/index.html`。 - v1 conversation 记录使用 append-only JSONL,每行带 `schemaVersion`、`role`、`content`、`agentId` 和 `updatedAt`,作为聊天历史和单 agent 对话历史的事实源;目录在首次写入时创建。 - 开发窗口和项目内 Agent 对话弹窗的“后台运行”只启动或排队单 Agent 后台任务,不阻塞等待回复;用户可刷新同一 Agent 对话或 runtime 状态查看进度和结果,也可对当前 run 执行取消 / 重试,待确认 run 还可执行“确认继续”或“拒绝并继续”。后台任务会向 `.agent/runtime/tasks/.jsonl` 追加任务视角记录,任务状态使用 `pending / running / waiting-for-confirmation / cancelled / completed / failed`,读取时按 `runId` 去重返回最近任务;`runId` 在同一 Agent 内是单个 run 的身份,后台入队会自动把重复 runId 改写为唯一实际 runId,防止不同任务互相覆盖;runtime state 自身仍可在完成后显示 `idle / completed`,二者语义分开。同一 Agent 的 pending 任务由持有 `.agent/runtime/locks/.lock` 的后台 drain 串行执行,避免同一 Agent 并发抢上下文;不同 Agent 仍可并行;若某个工具动作命中确认策略,该 Agent 会停在 `waiting-for-confirmation` 并暂停继续消费队列,等待后续确认或策略调整;Runtime 会把完整 `AgentRuntimePendingToolAction` 经过敏感内容和项目绝对路径校验后原子写入 `.agent/runtime/pending-actions//.json`,公共 `pendingToolAction` 只公开安全摘要;确认或拒绝必须匹配 `runId + actionId` 并通过工具名与完整输入 JSON 的 SHA-256 校验。确认在同一 run 直接执行原 action 并把 observation 接回后续 loop;拒绝不执行工具,而是写入 `blocked` observation 后在同一 run 继续规划。待确认状态可跨 App 重启读取并回收上一进程锁;等待期间同 Agent 新任务保持 pending,确认/拒绝续跑结束后由同一 drain 串行排空;若用户取消 pending 任务,drain 不再消费该 run,若取消 running 任务,则在当前 LLM 或工具调用返回后的检查点停止,不继续执行工具或保存最终 assistant 回复。客户端重开项目时会对当前项目路径自动尝试一次 Runtime 恢复;恢复命令必须通过 `agent.resume` 自动权限,默认确认策略下不会静默启动;同一 Agent 同时存在上一进程遗留 `running` 和 `pending` 时,先重接 `running`,再由 drain 继续 `pending`。开发构建和后台 Agent 工具箱都可通过 `agent.schedule_ready` 权限确认入口把 manifest ready task 投递进同一后台队列,命令会先把 ready task 标成 `running`,再用 taskId 作为 Agent id 入队,source 为 `agent-ready-task-scheduler`;该入口不新增独立 worker。后台任务的核心 loop 每 6 轮形成一个上下文压缩窗口:每轮把已有 observation 带回 LLM 让 Agent 重新规划;只有合法工具计划返回空 actions,且不存在未通过或项目修改后未重跑的 `project.verify` 时才收束,response 为空时进入独立最终回复生成。窗口边界会压缩 observation;有新的独立观察时在同一 run 继续下一窗口,最近窗口重复无进展时才以 `budget-exhausted / loop-budget-exhausted` 失败。后台任务完成后会把 assistant 回复追加到对应 `.agent/conversations/agents/.jsonl`,并向 `.agent/agent.db` 写入 `agent.runtime.background_task.queued` / `agent.runtime.background_task` / `agent.runtime.background_task.recovered` / `agent.runtime.ready_task.scheduled` / `agent.runtime.tool_observation` / `agent.runtime.tool_confirmation_required` / `agent.runtime.tool_confirmation.approved` / `agent.runtime.tool_confirmation.rejected` / `agent.runtime.memory.write` / `agent.runtime.project.verify` / `agent.runtime.file.write` / `agent.runtime.file.patch` / `agent.runtime.file.delete` / `agent.runtime.tool_plan.repair` / `agent.runtime.task.create` / `agent.runtime.task.update` / `agent.runtime.command.run_limited` / `agent.runtime.blackboard.write` / `agent.runtime.agent.message` / `agent.runtime.agent.delegate` / `agent.runtime.background_task.cancelled` / `agent.runtime.background_task.retry` / `agent.runtime.background_task.completed` / `agent.runtime.background_task.failed` 审计记录。当前工具箱开放只读工具 `memory.read`、`conversation.read`、`asset.list`、`project.index`、`project.search`、`project.diff`、`file.list`、`file.read`、`task.list`、`agent.run_status`,以及受策略保护的写/运行工具 `memory.write`、`project.checkpoint`、`project.restore`、`project.verify`、`file.write`、`file.patch`、`file.delete`、`task.create`、`task.update`、`command.run_limited`、`preview.start`、`canvas.asset_generate`、`blackboard.write`、`agent.message`、`agent.delegate` 和 `agent.schedule_ready`;`memory.write scope=agent` 只允许写当前 Agent 自己的私有记忆,跨 Agent 共享必须改用 `blackboard.write` 或 `agent.message`;`project.checkpoint` 只创建本地 checkpoint,不返回本机绝对路径;`project.restore` 只按 checkpoint id 恢复当前项目,不返回本机绝对路径,默认确认策略下不会静默回滚;`file.delete` 只删除项目内普通文件,默认确认且不能访问 `.agent/**`;`task.create` 只追加新 manifest 任务,`task.update` 只更新已有任务状态;`agent.schedule_ready` 只调度 manifest ready task,不创建平行 runtime;若项目策略拒绝,对应工具不会执行,Runtime 会把策略结果作为 observation 回给 Agent 修正计划;若项目策略要求确认,Runtime 会持久化精确待确认动作并保留 waiting 状态,不执行该工具;只有确认入口通过 `runId + actionId + SHA-256` 校验后才直接执行原 action,拒绝入口则生成 `blocked` observation。`toolPolicy` 保存当前工具级权限快照,供 planning prompt 和状态面板展示;`recentToolCalls` 保存最近 20 条结构化工具动作及安全 `inputSummary`,供状态面板展示最近动作和确认目标;append-only JSONL 写入按目标文件路径在当前进程内串行追加完整行,覆盖 `.agent/agent.db`、对话、Runtime events/tasks、activity 和 output,减少多个 Agent 同时完成时的行交错风险。 +- 2026-07-15 V1.17 收束补充:上一条“空 actions 且 verification 通过即可收束”只适用于没有结构化计划的 legacy run。`planRevision > 0` 后,Runtime 还必须确认全部计划步骤均为 `completed`;未完成时在同一 run 返回 `runtime.plan_update` blocker,不写 assistant 或 completed。计划更新只写 Runtime 控制面,不经过工具权限策略,也不推进 project revision 或 verification gate。 - 2026-07-10 补充:当前工具箱还开放 `preview.start`,审计记录类型为 `agent.runtime.preview.start`;该工具不会打开任意 URL,只启动当前授权项目的 `127.0.0.1` 本地预览,并和 Tauri 用户命令共用同一个 `PreviewRegistry`。 - 2026-07-10 补充:当前工具箱还开放 `canvas.asset_generate`,审计记录类型为 `agent.runtime.canvas.asset_generate`;该工具只通过配置好的 External Editor API 生成并回流素材,不暴露任意上传 / 任意网络请求能力。 - 2026-07-10 补充:当前工具箱还开放 `task.list`;该工具只读取 manifest 任务图、状态、依赖、产物和 `readyTaskIds`,并受 `task.list` 项目权限策略保护。