工具观察原样进 prompt:去掉按 token 预算的脱敏与截断

- 删除 `bound_observation_for_prompt` 与 `truncate_to_estimated_tokens`
- `prepare_game_creator_agent_runtime_prompt_history` 去掉 `tool_output_token_limit` 参数,观察原样进 prompt
- 调用处同步去掉该参数:`provider_request_builders`、`tests/runtime_state`
- 删除只验证旧有界行为的 `tool_output_is_bounded_by_token_limit` 测试
- 脱敏仍只发生在写审计/持久化时(`sanitize_game_creator_agent_runtime_context_observations_for_storage`)
This commit is contained in:
2026-10-01 18:53:40 +08:00
parent 214055e319
commit cf6ed72f34
4 changed files with 25 additions and 126 deletions
@@ -708,7 +708,6 @@ fn build_game_creator_background_agent_context(
&session_id,
run_id,
observations,
llm.tool_output_token_limit,
)?;
let repository_context = build_repository_startup_context_at(root)?;
let repository_prompt = render_repository_startup_context_for_prompt(&repository_context);
@@ -21,7 +21,6 @@ fn agent_runtime_preview_infrastructure_failure_kind(error: &str) -> Option<&'st
}
fn agent_runtime_preview_infrastructure_observation(
root: &Path,
revision: u64,
failure_kind: &str,
error: &str,
@@ -30,7 +29,7 @@ fn agent_runtime_preview_infrastructure_observation(
"errorKind": AGENT_RUNTIME_PREVIEW_INFRASTRUCTURE_ERROR_KIND,
"failureKind": failure_kind,
"revision": revision,
"diagnostic": redact_agent_runtime_project_paths(root, error, 500),
"diagnostic": truncate_agent_runtime_text(error, 500),
}))
.ok();
AgentRuntimeToolObservation {
@@ -93,7 +92,7 @@ pub(crate) fn observe_agent_runtime_preview_start(
Err(error) => AgentRuntimeToolObservation {
tool: "preview.start".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
},
}
@@ -158,7 +157,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: sanitize_agent_runtime_text(
summary: truncate_agent_runtime_text(
&format!("preview.validate 输入无效:{error}"),
240,
),
@@ -180,7 +179,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -194,7 +193,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
status: "failed".to_string(),
summary: "preview.validate 执行身份或自主构建完成合同不可用,未执行浏览器试玩"
.to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
};
@@ -223,7 +222,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "旧自主试玩回执无法失效,未执行新的浏览器试玩".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
}
@@ -233,7 +232,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -254,7 +253,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -287,7 +286,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -310,7 +309,6 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
Err(error) => {
if let Some(failure_kind) = agent_runtime_preview_infrastructure_failure_kind(&error) {
return agent_runtime_preview_infrastructure_observation(
root,
revision_before.revision,
failure_kind,
&error,
@@ -319,7 +317,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -338,7 +336,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
@@ -382,7 +380,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器验证未通过,且当前验证凭证无法安全失效".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
}
@@ -414,7 +412,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器试玩已返回,但自主试玩回执无法形成".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
};
@@ -441,7 +439,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器试玩已通过,但持久回执无法验证".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
};
@@ -457,7 +455,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器试玩已通过,但首个可玩版本暂时无法登记".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
};
@@ -481,7 +479,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
status: "failed".to_string(),
summary: "浏览器试玩已通过,但登记版本前无法复核项目 revision"
.to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
};
@@ -492,7 +490,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器试玩已通过,但首个可玩版本无法登记".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
}
@@ -507,7 +505,7 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: "浏览器验证已通过,但失败试玩凭证无法安全清除".to_string(),
detail: Some(redact_agent_runtime_project_paths(root, &error, 500)),
detail: Some(truncate_agent_runtime_text(&error, 500)),
};
}
}
@@ -562,13 +560,13 @@ pub(crate) async fn observe_agent_runtime_preview_validate(
return AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: "failed".to_string(),
summary: redact_agent_runtime_project_paths(root, &error, 240),
summary: truncate_agent_runtime_text(&error, 240),
detail: None,
};
}
let detail = serde_json::to_string(&detail_value)
.ok()
.map(|value| redact_agent_runtime_project_paths(root, &value, 3_600));
.map(|value| truncate_agent_runtime_text(&value, 3_600));
AgentRuntimeToolObservation {
tool: "preview.validate".to_string(),
status: if result.passed { "ok" } else { "failed" }.to_string(),
@@ -592,12 +590,8 @@ mod tests {
agent_runtime_preview_infrastructure_failure_kind(error),
Some("browser-launch-failed")
);
let observation = agent_runtime_preview_infrastructure_observation(
Path::new("/project"),
19,
"browser-launch-failed",
error,
);
let observation =
agent_runtime_preview_infrastructure_observation(19, "browser-launch-failed", error);
assert_eq!(observation.status, "blocked");
assert_eq!(
agent_runtime_preview_infrastructure_blocker(&observation).as_deref(),
@@ -188,69 +188,6 @@ pub(crate) fn validate_game_creator_llm_request_context_budget(
Ok(())
}
fn truncate_to_estimated_tokens(value: &str, token_limit: u64) -> String {
if token_limit == 0 {
return String::new();
}
let max_bytes = token_limit
.saturating_mul(AGENT_RUNTIME_CONTEXT_TOKEN_ESTIMATE_BYTES_PER_TOKEN)
.min(usize::MAX as u64) as usize;
if value.len() <= max_bytes {
return value.to_string();
}
let suffix = "\n...[tool output truncated by token budget]";
if max_bytes <= suffix.len() {
let mut boundary = max_bytes.min(value.len());
while boundary > 0 && !value.is_char_boundary(boundary) {
boundary -= 1;
}
return value[..boundary].to_string();
}
let retained_bytes = max_bytes.saturating_sub(suffix.len());
let mut boundary = retained_bytes.min(value.len());
while boundary > 0 && !value.is_char_boundary(boundary) {
boundary -= 1;
}
format!("{}{}", &value[..boundary], suffix)
}
fn bound_observation_for_prompt(
root: &Path,
observation: &AgentRuntimeToolObservation,
token_limit: u64,
) -> AgentRuntimeToolObservation {
let mut bounded = sanitize_agent_runtime_context_observation(root, observation);
let summary_budget = token_limit.min(1_024).max(1);
bounded.summary = truncate_to_estimated_tokens(&bounded.summary, summary_budget);
let detail_budget = token_limit.saturating_sub(summary_budget).max(1);
bounded.detail = bounded
.detail
.as_deref()
.map(|detail| truncate_to_estimated_tokens(detail, detail_budget))
.filter(|detail| !detail.trim().is_empty());
while estimate_serialized_bytes_as_tokens(
bounded.summary.len() + bounded.detail.as_deref().map(str::len).unwrap_or_default(),
) > token_limit
{
if let Some(detail) = bounded.detail.as_deref() {
let chars = detail.chars().count();
if chars > 32 {
bounded.detail = Some(detail.chars().take(chars / 2).collect());
continue;
}
bounded.detail = None;
continue;
}
let chars = bounded.summary.chars().count();
if chars <= 8 {
bounded.summary.clear();
continue;
}
bounded.summary = bounded.summary.chars().take(chars / 2).collect();
}
bounded
}
pub(crate) fn sanitize_game_creator_agent_runtime_context_observations_for_storage(
root: &Path,
observations: &[AgentRuntimeToolObservation],
@@ -457,7 +394,6 @@ pub(crate) fn prepare_game_creator_agent_runtime_prompt_history(
session_id: &str,
run_id: &str,
observations: &[AgentRuntimeToolObservation],
tool_output_token_limit: u64,
) -> Result<AgentRuntimePromptHistory, String> {
let (agent_messages, project_messages, sidecar) =
prompt_history_sources(root, agent_id, session_id, run_id, observations)?;
@@ -499,10 +435,9 @@ pub(crate) fn prepare_game_creator_agent_runtime_prompt_history(
project_tail = project_tail,
));
}
let observations = observations[observation_start..]
.iter()
.map(|observation| bound_observation_for_prompt(root, observation, tool_output_token_limit))
.collect();
// 工具观察原样进 prompt:给模型看的内容不脱敏、不截断;脱敏只发生在写审计/持久化时
// (见 `sanitize_game_creator_agent_runtime_context_observations_for_storage`)。
let observations = observations[observation_start..].to_vec();
Ok(AgentRuntimePromptHistory {
context: sections.join("\n\n"),
observations,
@@ -985,30 +920,6 @@ mod tests {
);
}
#[test]
fn tool_output_is_bounded_by_token_limit() {
let root = PathBuf::from("/tmp/context-compaction-token-bound");
let observation = AgentRuntimeToolObservation {
tool: "file.read".to_string(),
status: "ok".to_string(),
summary: "s".repeat(8_000),
detail: Some("d".repeat(20_000)),
};
let bounded = bound_observation_for_prompt(&root, &observation, 256);
let tokens = estimate_serialized_bytes_as_tokens(
bounded.summary.len() + bounded.detail.as_deref().map(str::len).unwrap_or_default(),
);
assert!(tokens <= 256);
let tiny = bound_observation_for_prompt(&root, &observation, 1);
let tiny_tokens = estimate_serialized_bytes_as_tokens(
tiny.summary.len() + tiny.detail.as_deref().map(str::len).unwrap_or_default(),
);
assert!(tiny_tokens <= 1);
assert_eq!(tiny.tool, "file.read");
assert_eq!(tiny.status, "ok");
}
#[test]
fn compaction_keeps_recent_tails_and_advances_only_for_new_source() {
let project = context_compaction_test_project("tail-and-revision");
@@ -1038,7 +949,6 @@ mod tests {
session_id,
run_id,
&observations,
1_000,
)
.expect("prepare compacted prompt history");
assert!(history.context.contains("第一版安全摘要"));
@@ -1119,7 +1029,6 @@ mod tests {
session_id,
run_id,
&tampered_observations,
1_000,
)
.expect_err("tampered observation prefix must fail");
assert!(observation_error.contains("observation 前缀发生漂移"));
@@ -1148,7 +1057,6 @@ mod tests {
session_id,
run_id,
&observations,
1_000,
)
.expect_err("tampered conversation prefix must fail");
assert!(conversation_error.contains("对话前缀发生漂移"));
@@ -1327,7 +1235,6 @@ mod tests {
session_id,
run_id,
&observations,
1_000,
)
.expect("prepare supervisor prompt history");
assert!(!history.context.contains("LEGACY_PROJECT_MARKER_0"));
@@ -3496,7 +3496,6 @@ fn supervisor_background_task_persists_one_public_start_status_before_execution(
&queued.state.session_id,
run_id,
&[],
1_000,
)
.expect("render Runtime prompt history");
assert!(runtime_prompt_history