掐断结构化计划空转活锁
apply_agent_runtime_plan_update 把计划解释算进「有变化」判据里,模型每轮重写 一遍解释就返回「计划有进展」,Runtime 还发一个新 planRevision 替它背书。观察 detail 里那句「只有步骤或状态真实变化时才调用 update_agent_plan」是 prompt, 不是门禁,于是总控可以只改解释、不落地任何动作地无限循环,每轮烧一次 Provider 请求。现场实测四分钟走了八轮,planRevision 到 4 而步骤状态一格没动。 判据拆成三态:解释照旧落盘,但只改解释既不算进展也不再 bump planRevision, 事件改发 plan_update.explanation_only。计数落在持久 Runtime state 上(照 plan_submit_gdd_rejection_count 的先例),进程重启不能把一次活锁洗成新的无限 开销;只在「本轮没有任何动作、且未完成原因就是计划自己」这一支累加——等委派 回执、等 provider 批次、等用户问询走的是别的 blocker 类型,不会被误判成空转。 连续 2 轮把 update_agent_plan 从工具目录里摘掉逼模型自愈,只摘这一个工具,其 余工具面原样保留;连续 4 轮以 plan-update-idle-rounds-exhausted 收束。摘工具 的阈值严格小于收束限额,由单测守住,否则 run 会在从没被逼过一次真动作的情况 下直接失败。 Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -135,8 +135,9 @@ pub(crate) use response_stream::filter_agent_runtime_response_stream_for_test;
|
||||
pub(crate) use structured_plan::{
|
||||
activate_agent_runtime_plan_step, activate_agent_runtime_response_plan_step,
|
||||
apply_agent_runtime_plan_update, complete_agent_runtime_active_plan_step,
|
||||
complete_agent_runtime_remaining_plan_steps, retry_agent_runtime_active_plan_step,
|
||||
sanitize_agent_runtime_plan_update,
|
||||
complete_agent_runtime_remaining_plan_steps, plan_update_idle_rounds_require_repair,
|
||||
retry_agent_runtime_active_plan_step, sanitize_agent_runtime_plan_update,
|
||||
AgentRuntimePlanUpdateOutcome,
|
||||
};
|
||||
// 不带 agentId 的三个解析入口走 `"__all_agents__"` 哨兵、跳过按身份的工具面
|
||||
// 复核,只对测试开放;生产代码必须用 `_for_agent`。
|
||||
|
||||
+117
@@ -619,6 +619,24 @@ pub(in crate::agent) fn build_game_creator_agent_background_tool_plan_request(
|
||||
"上一轮 runtime.plan_update 被拒绝。本轮必须立即提交当前 in_progress 步骤对应的实际项目 mutation,或在只读合同已满足时 respond_to_user;禁止再次规划、读取、搜索、验证、委派或普通文本解释。"
|
||||
}));
|
||||
}
|
||||
// 计划空转和 plan_update 被拒绝一样,是 request-scoped 的 liveness 信号:
|
||||
// 连续若干轮只改计划解释、没有任何动作也没有步骤推进时,本轮直接把
|
||||
// update_agent_plan 从工具目录里摘掉。判据是 Runtime 拥有的持久计数,不是
|
||||
// prompt 提醒——观察 detail 里那句「只有步骤或状态真实变化时才调用」拦不住
|
||||
// 任何东西。只摘这一个工具,其余工具面原样保留。
|
||||
let plan_update_idle_rounds = read_game_creator_agent_runtime_at(root, agent_id)
|
||||
.ok()
|
||||
.map(|result| result.state)
|
||||
.filter(|state| state.run_id == run_id)
|
||||
.map_or(0, |state| state.plan_update_idle_rounds);
|
||||
if plan_update_idle_rounds_require_repair(plan_update_idle_rounds) {
|
||||
request
|
||||
.function_tools
|
||||
.retain(|tool| tool.name != AGENT_RUNTIME_UPDATE_PLAN_FUNCTION_NAME);
|
||||
request.messages.push(LlmMessage::user(format!(
|
||||
"已连续 {plan_update_idle_rounds} 轮只改结构化计划解释,没有任何动作,步骤状态也没有变化。本轮 update_agent_plan 已从工具目录中移除:必须直接调用当前 in_progress 步骤对应的实际动作函数,或在证据已足够时 respond_to_user。"
|
||||
)));
|
||||
}
|
||||
if autonomous_game_build && !editor_api_key_is_configured() {
|
||||
let canvas_function = native_runtime_function_name("canvas.asset_generate")
|
||||
.ok_or_else(|| "无法生成画布素材工具函数名".to_string())?;
|
||||
@@ -994,6 +1012,105 @@ mod tests {
|
||||
.any(|message| message.content.contains("runtime.plan_update 被拒绝")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn idle_plan_update_rounds_drop_the_plan_tool_from_the_request_catalog() {
|
||||
let directory = crate::tests::canonical_test_tempdir("provider-plan-idle-repair-");
|
||||
let root = directory.path().join("project");
|
||||
init_local_game_project_at(&root, "plan-idle-repair", "修复现有游戏")
|
||||
.expect("project init");
|
||||
let binding = bind_game_creator_agent_runtime_run_profile_at(
|
||||
&root,
|
||||
GAME_CREATOR_PROJECT_SUPERVISOR_AGENT_ID,
|
||||
"plan-idle-repair-root",
|
||||
AGENT_RUNTIME_SUPERVISOR_GAME_CHAT_SOURCE,
|
||||
Some(AGENT_RUNTIME_RUN_PROFILE_AUTONOMOUS_GAME_BUILD),
|
||||
None,
|
||||
)
|
||||
.expect("bind root");
|
||||
let state = start_game_creator_agent_runtime_task_at(
|
||||
&root,
|
||||
&binding.agent_id,
|
||||
"修复现有游戏",
|
||||
&binding.run_id,
|
||||
&binding.source,
|
||||
"执行当前计划中的项目修改",
|
||||
vec!["立即修改 game/index.html".to_string()],
|
||||
)
|
||||
.expect("start task");
|
||||
crate::agent::create_game_creator_agent_runtime_goal_contract_at(
|
||||
&root,
|
||||
&binding.agent_id,
|
||||
&binding.run_id,
|
||||
&state.current_task,
|
||||
&AgentRuntimeGoalContractDraft {
|
||||
outcome: "修复现有游戏".to_string(),
|
||||
non_negotiables: Vec::new(),
|
||||
preferences: Vec::new(),
|
||||
forbidden_assumptions: Vec::new(),
|
||||
open_questions: Vec::new(),
|
||||
acceptance_nodes: vec![AgentRuntimeGoalContractAcceptanceNodeDraft {
|
||||
criterion_id: "repair-game".to_string(),
|
||||
criterion: "完成项目修改".to_string(),
|
||||
required: true,
|
||||
required_evidence: vec!["file.patch".to_string()],
|
||||
dependencies: Vec::new(),
|
||||
}],
|
||||
},
|
||||
)
|
||||
.expect("create goal contract");
|
||||
let catalog = GameCreatorMcpCatalog {
|
||||
fingerprint: String::new(),
|
||||
servers: Vec::new(),
|
||||
tools: Vec::new(),
|
||||
};
|
||||
|
||||
// 没有空转计数时 update_agent_plan 必须还在,否则这条判据就等于永远生效。
|
||||
let (_, _, baseline, _, _) = build_game_creator_agent_background_tool_plan_request(
|
||||
&root,
|
||||
&state.agent_id,
|
||||
&state.session_id,
|
||||
&state.run_id,
|
||||
&state.current_task,
|
||||
&[],
|
||||
1,
|
||||
&catalog,
|
||||
)
|
||||
.expect("build baseline request");
|
||||
assert!(baseline
|
||||
.function_tools
|
||||
.iter()
|
||||
.any(|tool| tool.name == AGENT_RUNTIME_UPDATE_PLAN_FUNCTION_NAME));
|
||||
|
||||
let mut idle_state = state.clone();
|
||||
idle_state.plan_update_idle_rounds = 2;
|
||||
crate::agent::write_game_creator_agent_runtime_state(&root, &idle_state)
|
||||
.expect("persist idle rounds");
|
||||
|
||||
let (_, _, request, _, _) = build_game_creator_agent_background_tool_plan_request(
|
||||
&root,
|
||||
&state.agent_id,
|
||||
&state.session_id,
|
||||
&state.run_id,
|
||||
&state.current_task,
|
||||
&[],
|
||||
2,
|
||||
&catalog,
|
||||
)
|
||||
.expect("build idle-repair request");
|
||||
assert!(!request
|
||||
.function_tools
|
||||
.iter()
|
||||
.any(|tool| tool.name == AGENT_RUNTIME_UPDATE_PLAN_FUNCTION_NAME));
|
||||
// 只摘这一个工具:真动作和收束都必须还在,否则模型无路可走。
|
||||
assert!(request
|
||||
.function_tools
|
||||
.iter()
|
||||
.any(|tool| tool.name == AGENT_RUNTIME_RESPOND_FUNCTION_NAME));
|
||||
assert!(request.messages.iter().any(|message| message
|
||||
.content
|
||||
.contains("update_agent_plan 已从工具目录中移除")));
|
||||
}
|
||||
|
||||
fn build_request_system_prompt_for_root_source(
|
||||
agent_id: &str,
|
||||
root_source: &str,
|
||||
|
||||
@@ -115,10 +115,41 @@ pub(crate) fn sanitize_agent_runtime_plan_update(
|
||||
Ok(AgentRuntimePlanUpdate { explanation, steps })
|
||||
}
|
||||
|
||||
/// 只改计划解释、不落地任何动作的一轮是「计划空转」。第一轮可能只是模型把
|
||||
/// 思考和动作拆成了两步,第二轮就是模式了:从这一轮起把 update_agent_plan 从
|
||||
/// 工具目录里摘掉,逼它要么调真动作要么收束。
|
||||
const AGENT_RUNTIME_PLAN_UPDATE_IDLE_REPAIR_ROUNDS: u32 = 2;
|
||||
|
||||
pub(crate) fn plan_update_idle_rounds_require_repair(idle_rounds: u32) -> bool {
|
||||
idle_rounds >= AGENT_RUNTIME_PLAN_UPDATE_IDLE_REPAIR_ROUNDS
|
||||
}
|
||||
|
||||
/// 一次 `update_agent_plan` 究竟改动了什么。
|
||||
///
|
||||
/// 拆开的理由:计划解释是给人看的自由文本,只改解释不构成计划进展。把它算进
|
||||
/// 「有变化」里,模型每轮重写一遍解释就能无限续命,Runtime 还会给它发一个新
|
||||
/// planRevision 背书,看上去像在推进。
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
pub(crate) enum AgentRuntimePlanUpdateOutcome {
|
||||
/// 步骤集合、状态和解释都没变,Runtime 什么都没写。
|
||||
Unchanged,
|
||||
/// 只有计划解释变了:新解释照旧落盘,但不算进展,也不 bump planRevision。
|
||||
ExplanationOnly,
|
||||
/// 步骤集合或步骤状态真实变化。
|
||||
StepsChanged,
|
||||
}
|
||||
|
||||
impl AgentRuntimePlanUpdateOutcome {
|
||||
/// 计划是否真的往前走了一格。空转守卫只认这一个判据。
|
||||
pub(crate) fn advanced_steps(self) -> bool {
|
||||
matches!(self, Self::StepsChanged)
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn apply_agent_runtime_plan_update(
|
||||
runtime: &mut AgentRuntimeState,
|
||||
update: &AgentRuntimePlanUpdate,
|
||||
) -> Result<bool, String> {
|
||||
) -> Result<AgentRuntimePlanUpdateOutcome, String> {
|
||||
let update = sanitize_agent_runtime_plan_update(update)?;
|
||||
let had_structured_plan = agent_runtime_has_structured_plan(runtime);
|
||||
let mut terminal_steps = std::collections::BTreeMap::new();
|
||||
@@ -178,18 +209,19 @@ pub(crate) fn apply_agent_runtime_plan_update(
|
||||
));
|
||||
}
|
||||
|
||||
let unchanged = had_structured_plan
|
||||
&& runtime.plan_explanation == update.explanation
|
||||
&& runtime.plan_steps.len() == merged.len()
|
||||
&& runtime
|
||||
let steps_changed = !had_structured_plan
|
||||
|| runtime.plan_steps.len() != merged.len()
|
||||
|| !runtime
|
||||
.plan_steps
|
||||
.iter()
|
||||
.zip(merged.iter())
|
||||
.all(|(existing, (title, status))| {
|
||||
existing.title == *title && existing.status == *status
|
||||
});
|
||||
if unchanged {
|
||||
return Ok(false);
|
||||
let explanation_changed =
|
||||
!had_structured_plan || runtime.plan_explanation != update.explanation;
|
||||
if !steps_changed && !explanation_changed {
|
||||
return Ok(AgentRuntimePlanUpdateOutcome::Unchanged);
|
||||
}
|
||||
|
||||
let now = unix_timestamp();
|
||||
@@ -232,8 +264,11 @@ pub(crate) fn apply_agent_runtime_plan_update(
|
||||
.find(|step| step.status == AGENT_RUNTIME_PLAN_STATUS_IN_PROGRESS)
|
||||
.map(|step| step.index);
|
||||
runtime.plan_explanation = update.explanation;
|
||||
if !steps_changed {
|
||||
return Ok(AgentRuntimePlanUpdateOutcome::ExplanationOnly);
|
||||
}
|
||||
runtime.plan_revision = runtime.plan_revision.saturating_add(1).max(1);
|
||||
Ok(true)
|
||||
Ok(AgentRuntimePlanUpdateOutcome::StepsChanged)
|
||||
}
|
||||
|
||||
pub(in crate::agent) fn update_agent_runtime_plan_steps(
|
||||
|
||||
@@ -196,6 +196,65 @@ fn finish_plan_submit_business_rejection_limit_at(
|
||||
Ok(AgentBackgroundTaskOutcome::Finished)
|
||||
}
|
||||
|
||||
/// 摘掉工具还继续空转,说明自愈失败。和 Fast GDD 拒绝限额同理:把它记在持久
|
||||
/// Runtime state 上,进程重启不能把一次活锁洗成新的无限 Provider 开销。
|
||||
const AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT: u32 = 4;
|
||||
|
||||
fn plan_update_idle_limit_reached(idle_rounds: u32) -> bool {
|
||||
idle_rounds >= AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod plan_update_idle_guard_tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn idle_guard_always_tries_self_repair_before_killing_the_run() {
|
||||
// 摘工具的阈值必须严格小于收束限额,否则 run 会在从没被逼过一次真动作
|
||||
// 的情况下直接失败,自愈这一级就等于不存在。
|
||||
assert!(plan_update_idle_rounds_require_repair(
|
||||
AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT
|
||||
));
|
||||
assert!(!plan_update_idle_limit_reached(
|
||||
AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT - 1
|
||||
));
|
||||
let first_repair_round = (0..=AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT)
|
||||
.find(|rounds| plan_update_idle_rounds_require_repair(*rounds))
|
||||
.expect("repair threshold within limit");
|
||||
assert!(first_repair_round < AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn idle_limit_is_reached_only_at_the_configured_round() {
|
||||
assert!(!plan_update_idle_limit_reached(0));
|
||||
assert!(!plan_update_idle_limit_reached(
|
||||
AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT - 1
|
||||
));
|
||||
assert!(plan_update_idle_limit_reached(
|
||||
AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT
|
||||
));
|
||||
assert!(plan_update_idle_limit_reached(
|
||||
AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT + 1
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
fn finish_plan_update_idle_limit_at(
|
||||
root: &Path,
|
||||
runtime: &AgentRuntimeState,
|
||||
) -> Result<AgentBackgroundTaskOutcome, String> {
|
||||
let error = format!(
|
||||
"结构化计划连续 {AGENT_RUNTIME_PLAN_UPDATE_IDLE_LIMIT} 轮只改解释、没有任何动作也没有步骤推进,已停止自动续跑;请检查最后一次 runtime.plan_update observation 后重新发起任务。"
|
||||
);
|
||||
let failed = fail_game_creator_agent_runtime_turn_at(root, runtime.clone(), &error)?;
|
||||
let _ = append_game_creator_agent_background_task_failed_audit(
|
||||
root,
|
||||
&failed,
|
||||
AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_PLAN_UPDATE_IDLE_LIMIT,
|
||||
);
|
||||
Ok(AgentBackgroundTaskOutcome::Finished)
|
||||
}
|
||||
|
||||
/// A strict submit payload rejection is a normal planning observation, not a
|
||||
/// Provider/lifecycle reconciliation failure. Close the exact sole-action
|
||||
/// batch, persist the rejected observation, and return a same-run continuation
|
||||
@@ -491,6 +550,8 @@ pub(in crate::agent) fn prepare_game_chat_single_round_convergence_at(
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_TOOL_PLAN: &str = "tool-plan-failed";
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_PLAN_SUBMIT_REJECTION_LIMIT: &str =
|
||||
"plan-submit-validation-retries-exhausted";
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_PLAN_UPDATE_IDLE_LIMIT: &str =
|
||||
"plan-update-idle-rounds-exhausted";
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_BUDGET: &str = "loop-budget-exhausted";
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_FINAL_REPLY: &str = "final-reply-failed";
|
||||
const AGENT_RUNTIME_BACKGROUND_FAILURE_KIND_FINALIZATION: &str = "finalization-failed";
|
||||
@@ -967,6 +1028,21 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
};
|
||||
}
|
||||
|
||||
// 和上面同理:计数已经随上一轮的 blocker 一起落盘,恢复后不能把第 N 次空转
|
||||
// 变成第 N+1 次 Provider 请求。
|
||||
if plan_update_idle_limit_reached(runtime.plan_update_idle_rounds) {
|
||||
return match finish_plan_update_idle_limit_at(&root, &runtime) {
|
||||
Ok(outcome) => outcome,
|
||||
Err(error) => fail_game_creator_agent_background_context_at(
|
||||
&root,
|
||||
&agent_id,
|
||||
&session_id,
|
||||
runtime,
|
||||
&format!("收束已耗尽的结构化计划空转失败:{error}"),
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
if continuation.applied_steer_cursor < runtime.applied_steer_cursor {
|
||||
return fail_game_creator_agent_background_context_at(
|
||||
&root,
|
||||
@@ -1374,6 +1450,10 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
let mut planning_repository_context_fingerprint: String;
|
||||
let mut planning_session_binding: Option<PlanProviderSessionBindingV1>;
|
||||
let action_start_index: usize;
|
||||
// 本轮 update_agent_plan 的分型结果,供下面的计划空转守卫判据使用。
|
||||
// 恢复既有批次的那一支不会新提交计划,保持 None 即可:那一支本来就带着
|
||||
// 真实动作,走不到空转分支。
|
||||
let mut plan_update_outcome = None;
|
||||
if let Some(batch) = resumed_provider_batch.as_ref() {
|
||||
let Some(first_pending) = batch.actions.first() else {
|
||||
return fail_game_creator_agent_background_context_at(
|
||||
@@ -1911,7 +1991,8 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
}
|
||||
if let Some(plan_update) = plan.plan_update.as_ref() {
|
||||
match apply_agent_runtime_plan_update(&mut runtime, plan_update) {
|
||||
Ok(true) => {
|
||||
Ok(outcome) if outcome.advanced_steps() => {
|
||||
plan_update_outcome = Some(outcome);
|
||||
runtime.updated_at = unix_timestamp();
|
||||
if let Err(error) = write_game_creator_agent_runtime_state(&root, &runtime)
|
||||
{
|
||||
@@ -1979,7 +2060,38 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
}),
|
||||
);
|
||||
}
|
||||
Ok(false) => {}
|
||||
Ok(outcome) => {
|
||||
plan_update_outcome = Some(outcome);
|
||||
if outcome == AgentRuntimePlanUpdateOutcome::ExplanationOnly {
|
||||
// 解释照旧落盘,但不发 plan_update 事件、不 bump revision:
|
||||
// 这一轮没有任何计划进展,事件流不该替它背书。
|
||||
runtime.updated_at = unix_timestamp();
|
||||
if let Err(error) =
|
||||
write_game_creator_agent_runtime_state(&root, &runtime)
|
||||
{
|
||||
return fail_game_creator_agent_background_context_at(
|
||||
&root,
|
||||
&agent_id,
|
||||
&session_id,
|
||||
runtime,
|
||||
&format!("持久化结构化计划解释 Runtime state 失败:{error}"),
|
||||
);
|
||||
}
|
||||
let _ = append_game_creator_agent_runtime_event(
|
||||
&root,
|
||||
&runtime,
|
||||
"plan_update.explanation_only",
|
||||
runtime.status.as_str(),
|
||||
runtime.phase.as_str(),
|
||||
"Agent 只改写了结构化计划解释,步骤与状态没有变化。",
|
||||
Some(&format!(
|
||||
"planRevision={} · explanationSha256={:x}",
|
||||
runtime.plan_revision,
|
||||
Sha256::digest(runtime.plan_explanation.as_bytes())
|
||||
)),
|
||||
);
|
||||
}
|
||||
}
|
||||
Err(error) => {
|
||||
let observation = AgentRuntimeToolObservation {
|
||||
tool: "runtime.plan_update".to_string(),
|
||||
@@ -2070,6 +2182,10 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
action_start_index = 0;
|
||||
}
|
||||
|
||||
if !plan.actions.is_empty() {
|
||||
// 本轮有真实动作,计划没有空转。
|
||||
runtime.plan_update_idle_rounds = 0;
|
||||
}
|
||||
if plan.actions.is_empty() {
|
||||
// blocked 的 plan_gdd blocker 有三种截然不同的继续推进态,phase 与
|
||||
// next_step 必须按类型化子状态选,不能回去猜 detail 字符串。
|
||||
@@ -2124,6 +2240,15 @@ async fn run_game_creator_agent_background_task_pass_without_deadline(
|
||||
if let Some(blocker) = completion_blocker {
|
||||
let blocker_summary = blocker.summary();
|
||||
if blocker.tool == "runtime.plan_update" {
|
||||
// 走到这里说明本轮没有任何动作,而且未完成的原因就是计划自己。
|
||||
// 等委派回执、等 provider 批次、等用户问询都是别的 blocker 类型,
|
||||
// 不会落到这一支,所以合法等待不会被算成空转。
|
||||
if plan_update_outcome.is_some_and(|outcome| outcome.advanced_steps()) {
|
||||
runtime.plan_update_idle_rounds = 0;
|
||||
} else {
|
||||
runtime.plan_update_idle_rounds =
|
||||
runtime.plan_update_idle_rounds.saturating_add(1);
|
||||
}
|
||||
runtime.status = "running".to_string();
|
||||
runtime.phase = "planning".to_string();
|
||||
runtime.current_action = "等待结构化计划进度更新".to_string();
|
||||
|
||||
@@ -1601,6 +1601,7 @@ pub(crate) fn default_game_creator_agent_runtime_state(
|
||||
next_step: "等待输入".to_string(),
|
||||
loop_iteration: 0,
|
||||
plan_submit_gdd_rejection_count: 0,
|
||||
plan_update_idle_rounds: 0,
|
||||
max_loop_iterations: AGENT_RUNTIME_BACKGROUND_LOOP_LIMIT as u32,
|
||||
tool_action_budget: AGENT_RUNTIME_BACKGROUND_TOOL_ACTION_LIMIT as u32,
|
||||
plan_revision: 0,
|
||||
|
||||
@@ -252,6 +252,12 @@ struct AgentRuntimeState {
|
||||
/// an invalid-provider-output loop back into an unbounded retry.
|
||||
#[serde(default)]
|
||||
plan_submit_gdd_rejection_count: u32,
|
||||
/// Consecutive rounds where this run produced no action and no structured
|
||||
/// plan step advance. Runtime-owned durable state so a runner restart
|
||||
/// cannot launder an explanation-only planning loop back into an unbounded
|
||||
/// Provider spend.
|
||||
#[serde(default)]
|
||||
plan_update_idle_rounds: u32,
|
||||
#[serde(default)]
|
||||
max_loop_iterations: u32,
|
||||
#[serde(default)]
|
||||
|
||||
@@ -516,17 +516,21 @@ fn structured_plan_update_validates_and_advances_monotonically() {
|
||||
&mut runtime,
|
||||
parsed.plan_update.as_ref().expect("plan update"),
|
||||
)
|
||||
.expect("apply first plan"));
|
||||
.expect("apply first plan")
|
||||
.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 1);
|
||||
assert_eq!(runtime.active_plan_step_index, Some(0));
|
||||
assert_eq!(runtime.plan_steps[0].status, "in_progress");
|
||||
assert_eq!(runtime.plan, vec!["读取项目", "验证结果"]);
|
||||
|
||||
assert!(!apply_agent_runtime_plan_update(
|
||||
&mut runtime,
|
||||
parsed.plan_update.as_ref().expect("same plan update"),
|
||||
)
|
||||
.expect("idempotent plan update"));
|
||||
assert_eq!(
|
||||
apply_agent_runtime_plan_update(
|
||||
&mut runtime,
|
||||
parsed.plan_update.as_ref().expect("same plan update"),
|
||||
)
|
||||
.expect("idempotent plan update"),
|
||||
AgentRuntimePlanUpdateOutcome::Unchanged
|
||||
);
|
||||
assert_eq!(runtime.plan_revision, 1);
|
||||
|
||||
let progressed = AgentRuntimePlanUpdate {
|
||||
@@ -542,7 +546,9 @@ fn structured_plan_update_validates_and_advances_monotonically() {
|
||||
},
|
||||
],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut runtime, &progressed).expect("advance plan"));
|
||||
assert!(apply_agent_runtime_plan_update(&mut runtime, &progressed)
|
||||
.expect("advance plan")
|
||||
.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 2);
|
||||
assert_eq!(runtime.plan_steps[0].status, "completed");
|
||||
assert_eq!(runtime.active_plan_step_index, Some(1));
|
||||
@@ -573,7 +579,8 @@ fn structured_plan_update_validates_and_advances_monotonically() {
|
||||
}],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut runtime, &completed)
|
||||
.expect("terminal step is retained"));
|
||||
.expect("terminal step is retained")
|
||||
.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 3);
|
||||
assert_eq!(runtime.plan, vec!["读取项目", "验证结果"]);
|
||||
assert!(runtime
|
||||
@@ -629,6 +636,7 @@ fn first_structured_plan_replaces_terminal_legacy_scaffolding_steps() {
|
||||
assert!(
|
||||
apply_agent_runtime_plan_update(&mut runtime, &first_structured_update)
|
||||
.expect("first structured plan replaces legacy scaffolding")
|
||||
.advanced_steps()
|
||||
);
|
||||
assert_eq!(runtime.plan_revision, 1);
|
||||
assert_eq!(runtime.plan_steps.len(), AGENT_RUNTIME_PLAN_STEP_LIMIT);
|
||||
@@ -650,6 +658,71 @@ fn first_structured_plan_replaces_terminal_legacy_scaffolding_steps() {
|
||||
assert_ne!(same_title_step.updated_at, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explanation_only_plan_update_is_not_plan_progress() {
|
||||
let mut runtime = default_game_creator_agent_runtime_state("design-director", "plan-idle-run");
|
||||
let initial = AgentRuntimePlanUpdate {
|
||||
explanation: "先读项目再验证".to_string(),
|
||||
steps: vec![
|
||||
AgentRuntimePlanUpdateStep {
|
||||
step: "读取项目".to_string(),
|
||||
status: "in_progress".to_string(),
|
||||
},
|
||||
AgentRuntimePlanUpdateStep {
|
||||
step: "验证结果".to_string(),
|
||||
status: "pending".to_string(),
|
||||
},
|
||||
],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut runtime, &initial)
|
||||
.expect("apply first plan")
|
||||
.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 1);
|
||||
|
||||
// 步骤集合与状态一字未动,只换了解释:这是活锁的形状,不能算计划进展,
|
||||
// 也不能发一个新 planRevision 替它背书。
|
||||
let explanation_only = AgentRuntimePlanUpdate {
|
||||
explanation: "换个说法解释同一个计划".to_string(),
|
||||
steps: initial.steps.clone(),
|
||||
};
|
||||
let outcome = apply_agent_runtime_plan_update(&mut runtime, &explanation_only)
|
||||
.expect("apply explanation-only plan");
|
||||
assert_eq!(outcome, AgentRuntimePlanUpdateOutcome::ExplanationOnly);
|
||||
assert!(!outcome.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 1);
|
||||
assert_eq!(runtime.plan_explanation, "换个说法解释同一个计划");
|
||||
assert_eq!(runtime.plan_steps[0].status, "in_progress");
|
||||
assert_eq!(runtime.plan_steps[1].status, "pending");
|
||||
|
||||
// 真把步骤推到下一格才重新计数。
|
||||
let progressed = AgentRuntimePlanUpdate {
|
||||
explanation: "换个说法解释同一个计划".to_string(),
|
||||
steps: vec![
|
||||
AgentRuntimePlanUpdateStep {
|
||||
step: "读取项目".to_string(),
|
||||
status: "completed".to_string(),
|
||||
},
|
||||
AgentRuntimePlanUpdateStep {
|
||||
step: "验证结果".to_string(),
|
||||
status: "in_progress".to_string(),
|
||||
},
|
||||
],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut runtime, &progressed)
|
||||
.expect("advance plan")
|
||||
.advanced_steps());
|
||||
assert_eq!(runtime.plan_revision, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plan_update_idle_repair_threshold_tolerates_exactly_one_split_round() {
|
||||
// 第一轮只更新计划、下一轮才动手是正常的两步走,不该被当成活锁。
|
||||
assert!(!plan_update_idle_rounds_require_repair(0));
|
||||
assert!(!plan_update_idle_rounds_require_repair(1));
|
||||
assert!(plan_update_idle_rounds_require_repair(2));
|
||||
assert!(plan_update_idle_rounds_require_repair(3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_plan_failed_step_remains_immutable_after_migration() {
|
||||
let mut runtime = default_game_creator_agent_runtime_state("code-prototype", "failed-plan-run");
|
||||
@@ -4787,7 +4860,8 @@ fn structured_plan_same_run_steer_preserves_monotonic_runtime_and_v3_context() {
|
||||
],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut state, &initial_update)
|
||||
.expect("apply pre-steer structured plan"));
|
||||
.expect("apply pre-steer structured plan")
|
||||
.advanced_steps());
|
||||
let pre_steer_revision = state.plan_revision;
|
||||
assert_eq!(pre_steer_revision, 1);
|
||||
write_game_creator_agent_runtime_state(&root, &state)
|
||||
@@ -4906,7 +4980,8 @@ fn structured_plan_same_run_steer_preserves_monotonic_runtime_and_v3_context() {
|
||||
],
|
||||
};
|
||||
assert!(apply_agent_runtime_plan_update(&mut state, &steered_update)
|
||||
.expect("apply post-steer structured plan"));
|
||||
.expect("apply post-steer structured plan")
|
||||
.advanced_steps());
|
||||
assert!(state.plan_revision > rejected_revision);
|
||||
assert!(state
|
||||
.plan_steps
|
||||
|
||||
Reference in New Issue
Block a user