Merge branch 'master' of ssh://192.168.35.82:2222/GenarrativeAI/Genarrative into feat/scene_v1
Project CI / Repository checks (pull_request) Successful in 53s
Project CI / Frontend tests (pull_request) Successful in 3m47s
Project CI / Backend tests (pull_request) Successful in 4m3s
Project CI / Native shell tests (pull_request) Successful in 15m23s

This commit is contained in:
2026-08-07 16:42:03 +08:00
11 changed files with 217 additions and 55 deletions
@@ -887,6 +887,11 @@ mod tests {
requests[0].body["model"],
BACKGROUND_MUSIC_PROMPT_ASSIST_MODEL
);
assert_eq!(
requests[0].body["max_completion_tokens"],
BACKGROUND_MUSIC_PROMPT_ASSIST_MAX_OUTPUT_TOKENS
);
assert!(requests[0].body.get("max_tokens").is_none());
assert_eq!(requests[0].body["reasoning_effort"], "medium");
assert!(requests[0].body.get("tools").is_none());
assert!(requests[0].body.get("tool_choice").is_none());
+7 -2
View File
@@ -23,7 +23,7 @@ use platform_auth::{
RefreshCookieConfig, RefreshCookieError, RefreshCookieSameSite, SmsAuthConfig, SmsAuthProvider,
SmsAuthProviderKind, SmsProviderError, WechatProvider, sign_access_token, verify_access_token,
};
use platform_llm::{LlmClient, LlmConfig, LlmError, LlmProvider};
use platform_llm::{LlmClient, LlmConfig, LlmError, LlmProvider, OpenAiChatTokenBudgetField};
use platform_matting::{MattingClient, MattingConfig};
use platform_oss::{OssClient, OssConfig, OssError};
use platform_wechat::{WechatClient, WechatConfig, pay::WechatPayClient};
@@ -2217,7 +2217,8 @@ fn build_editor_agent_llm_client(
config
.llm_retry_backoff_ms
.min(EDITOR_AGENT_LLM_MAX_RETRY_BACKOFF_MS),
)?;
)?
.with_openai_chat_token_budget_field(OpenAiChatTokenBudgetField::MaxCompletionTokens);
Ok(Some(LlmClient::new(llm_config)?))
}
@@ -2538,6 +2539,10 @@ mod tests {
"https://api.vectorengine.test/v1/chat/completions"
);
assert!(!client.config().official_fallback());
assert_eq!(
client.config().openai_chat_token_budget_field(),
OpenAiChatTokenBudgetField::MaxCompletionTokens
);
assert_eq!(client.config().max_retries(), 1);
assert_eq!(client.config().retry_backoff_ms(), 60_000);
}
+19 -17
View File
@@ -1,6 +1,6 @@
# platform-llm 平台适配 crate
日期:`2026-07-27`
更新:`2026-08-06`
## 1. crate 职责
@@ -36,10 +36,11 @@ Responses 如果只发送 `response.completed` 或 `response.incomplete`,解
## 4. 流式与参数契约
1. `LlmStreamDelta` 只包含 `accumulated_text`、`delta_text` 和 `finish_reason`,工具调用不会进入 `on_delta`;纯工具响应允许 `text` 为空。
2. 工具片段按协议索引聚合:Chat 使用 `delta.tool_calls[].index`,Responses 使用 `output_index`,Anthropic 使用 content block `index`。Responses 的 `.done`、`response.completed` 和 `response.incomplete` 中的完整 arguments 是权威值,可以覆盖之前的分片拼接。
3. 流结束固化工具调用时,缺少 id 或函数名返回 `Deserialize`;空参数默认保存为 `{}`;非空参数必须能反序列化为完整 JSON,截断或半截 JSON 不会交给业务层。这里是 JSON 语法完整性检查,不是针对 `parameters` 的 JSON Schema 业务校验。
4. 非流式工具调用采用不同的参数边界:缺失或空白 `arguments` 统一归一为 `{}`;Chat / Responses 的非空畸形 `arguments` 不在平台层做 JSON 校验、修复或静默丢弃,而是保留参数内容(仅按统一归一策略去除首尾空白),连同 call id 和函数名交给调用方的 repair 循环。Anthropic `tool_use.input` 缺失时同样按 `{}` 归一;调用方不能把非流式参数自动假定为统一 schema 校验通过。
1. `LlmRunRequest.max_output_tokens` 是协议中立的生成预算,包含可见输出与 Provider 可能使用的隐藏 reasoning token,不包含输入 token,也不保证可见正文长度。Responses 映射为 `max_output_tokens`,Anthropic 映射为 `max_tokens`;Chat 由 `LlmConfig.openai_chat_token_budget_field` 显式映射为当前 `max_completion_tokens` 或兼容网关旧字段 `max_tokens`,每次只发送一个。默认保留 legacy,已验证支持新字段的 endpoint 必须显式 opt-in;禁止按模型名猜测或收到 `400` 后自动重放。
2. `LlmStreamDelta` 只包含 `accumulated_text`、`delta_text` 和 `finish_reason`,工具调用不会进入 `on_delta`;纯工具响应允许 `text` 为空。
3. 工具片段按协议索引聚合:Chat 使用 `delta.tool_calls[].index`,Responses 使用 `output_index`,Anthropic 使用 content block `index`。Responses 的 `.done`、`response.completed` 和 `response.incomplete` 中的完整 arguments 是权威值,可以覆盖之前的分片拼接。
4. 流结束固化工具调用时,缺少 id 或函数名返回 `Deserialize`;空参数默认保存为 `{}`;非空参数必须能反序列化为完整 JSON,截断或半截 JSON 不会交给业务层。这里是 JSON 语法完整性检查,不是针对 `parameters` 的 JSON Schema 业务校验。
5. 非流式工具调用采用不同的参数边界:缺失或空白 `arguments` 统一归一为 `{}`;Chat / Responses 的非空畸形 `arguments` 不在平台层做 JSON 校验、修复或静默丢弃,而是保留参数内容(仅按统一归一策略去除首尾空白),连同 call id 和函数名交给调用方的 repair 循环。Anthropic `tool_use.input` 缺失时同样按 `{}` 归一;调用方不能把非流式参数自动假定为统一 schema 校验通过。
## 5. 错误边界
@@ -54,18 +55,19 @@ Responses 如果只发送 `response.completed` 或 `response.incomplete`,解
1. `LlmProvider`
2. `LlmConfig`
3. `LlmMessageRole`
4. `LlmMessage`
5. `LlmRunRequest`
6. `LlmApiKind`
7. `LlmStreamDelta`
8. `LlmFunctionTool`
9. `LlmToolChoice`
10. `LlmToolCall`
11. `LlmRunResponse`
12. `LlmTokenUsage`
13. `LlmClient`
14. `LlmError`
3. `OpenAiChatTokenBudgetField`
4. `LlmMessageRole`
5. `LlmMessage`
6. `LlmRunRequest`
7. `LlmApiKind`
8. `LlmStreamDelta`
9. `LlmFunctionTool`
10. `LlmToolChoice`
11. `LlmToolCall`
12. `LlmRunResponse`
13. `LlmTokenUsage`
14. `LlmClient`
15. `LlmError`
## 7. 设计文档
+169 -27
View File
@@ -47,6 +47,18 @@ pub enum LlmProvider {
OpenAiCompatible,
}
/// OpenAI Chat Completions 的生成预算字段方言。
///
/// `max_completion_tokens` 是当前 OpenAI 契约,包含可见输出与隐藏 reasoning token;
/// `max_tokens` 仅用于尚未支持新字段的兼容网关。能力必须由调用方按 endpoint 显式声明,
/// 不能根据模型名或请求级 model override 猜测。
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum OpenAiChatTokenBudgetField {
MaxCompletionTokens,
LegacyMaxTokens,
}
// 统一收口文本模型网关配置,避免 api-server 和业务模块各自重复解析环境变量。
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct LlmConfig {
@@ -60,6 +72,7 @@ pub struct LlmConfig {
retry_backoff_ms: u64,
official_fallback: bool,
anthropic_strict_tool_support: bool,
openai_chat_token_budget_field: OpenAiChatTokenBudgetField,
}
// 首版只冻结当前项目已稳定使用的 system/user/assistant 三种消息角色。
@@ -154,6 +167,8 @@ pub struct LlmToolCall {
pub struct LlmRunRequest {
pub model: Option<String>,
pub messages: Vec<LlmMessage>,
/// 生成侧 token 预算,包含可见输出与 Provider 可能使用的隐藏 reasoning token;
/// 不包含输入 token,也不保证可见正文长度。
pub max_output_tokens: Option<u32>,
pub enable_web_search: bool,
pub api_kind: LlmApiKind,
@@ -294,8 +309,9 @@ struct ChatCompletionsRequestBody {
#[serde(skip_serializing_if = "Option::is_none")]
official_fallback: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "max_tokens")]
max_output_tokens: Option<u32>,
max_completion_tokens: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
max_tokens: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
reasoning_effort: Option<&'static str>,
#[serde(skip_serializing_if = "Option::is_none")]
@@ -1023,6 +1039,7 @@ impl LlmConfig {
retry_backoff_ms,
official_fallback: false,
anthropic_strict_tool_support: false,
openai_chat_token_budget_field: OpenAiChatTokenBudgetField::LegacyMaxTokens,
})
}
@@ -1040,6 +1057,15 @@ impl LlmConfig {
self
}
/// 显式选择当前 Chat Completions endpoint 接受的生成预算字段。
pub fn with_openai_chat_token_budget_field(
mut self,
field: OpenAiChatTokenBudgetField,
) -> Self {
self.openai_chat_token_budget_field = field;
self
}
pub fn with_raw_log_dir(mut self, raw_log_dir: impl Into<PathBuf>) -> Self {
self.raw_log_dir = raw_log_dir.into();
self
@@ -1097,6 +1123,10 @@ impl LlmConfig {
self.anthropic_strict_tool_support
}
pub fn openai_chat_token_budget_field(&self) -> OpenAiChatTokenBudgetField {
self.openai_chat_token_budget_field
}
pub fn chat_completions_url(&self) -> String {
format!(
"{}/{}",
@@ -2252,31 +2282,41 @@ fn build_request_body(request: &LlmRunRequest, config: &LlmConfig, stream: bool)
let fallback_model = config.model();
let official_fallback = config.official_fallback().then_some(true);
match request.api_kind {
LlmApiKind::OpenAiChat => LlmRequestBody::ChatCompletions(ChatCompletionsRequestBody {
model: request.resolved_model(fallback_model).to_string(),
messages: map_chat_completions_input_messages(request.messages.as_slice()),
stream,
official_fallback,
max_output_tokens: request.max_output_tokens,
reasoning_effort: request
.response_reasoning_effort
.map(LlmResponseReasoningEffort::as_str),
web_search_options: request
.enable_web_search
.then_some(ChatCompletionsWebSearchOptions {}),
tools: (!request.function_tools.is_empty()).then(|| {
request
.function_tools
.iter()
.cloned()
.map(|function| ChatCompletionsFunctionTool {
tool_type: "function",
function,
})
.collect()
}),
tool_choice: request.tool_choice.map(LlmToolChoice::as_str),
}),
LlmApiKind::OpenAiChat => {
let (max_completion_tokens, max_tokens) = match config.openai_chat_token_budget_field()
{
OpenAiChatTokenBudgetField::MaxCompletionTokens => {
(request.max_output_tokens, None)
}
OpenAiChatTokenBudgetField::LegacyMaxTokens => (None, request.max_output_tokens),
};
LlmRequestBody::ChatCompletions(ChatCompletionsRequestBody {
model: request.resolved_model(fallback_model).to_string(),
messages: map_chat_completions_input_messages(request.messages.as_slice()),
stream,
official_fallback,
max_completion_tokens,
max_tokens,
reasoning_effort: request
.response_reasoning_effort
.map(LlmResponseReasoningEffort::as_str),
web_search_options: request
.enable_web_search
.then_some(ChatCompletionsWebSearchOptions {}),
tools: (!request.function_tools.is_empty()).then(|| {
request
.function_tools
.iter()
.cloned()
.map(|function| ChatCompletionsFunctionTool {
tool_type: "function",
function,
})
.collect()
}),
tool_choice: request.tool_choice.map(LlmToolChoice::as_str),
})
}
LlmApiKind::OpenAiResponses => LlmRequestBody::Responses(ResponsesRequestBody {
model: request.resolved_model(fallback_model).to_string(),
stream,
@@ -4017,6 +4057,33 @@ mod tests {
);
}
#[test]
fn llm_config_chat_token_budget_field_defaults_to_legacy_and_is_explicitly_selectable() {
let config = LlmConfig::new(
LlmProvider::OpenAiCompatible,
"https://example.com/v1".to_string(),
"secret".to_string(),
"model-a".to_string(),
DEFAULT_REQUEST_TIMEOUT_MS,
DEFAULT_MAX_RETRIES,
DEFAULT_RETRY_BACKOFF_MS,
)
.expect("config should be valid");
assert_eq!(
config.openai_chat_token_budget_field(),
OpenAiChatTokenBudgetField::LegacyMaxTokens
);
assert_eq!(
config
.with_openai_chat_token_budget_field(
OpenAiChatTokenBudgetField::MaxCompletionTokens,
)
.openai_chat_token_budget_field(),
OpenAiChatTokenBudgetField::MaxCompletionTokens
);
}
#[test]
fn run_request_defaults_to_openai_responses_api_kind() {
let request = LlmRunRequest::single_turn("系统", "用户");
@@ -4025,6 +4092,75 @@ mod tests {
assert_eq!(request.with_openai_chat().api_kind, LlmApiKind::OpenAiChat);
}
#[test]
fn chat_request_body_uses_configured_token_budget_field_without_model_guessing() {
let legacy_config = LlmConfig::new(
LlmProvider::OpenAiCompatible,
"https://legacy-gateway.example/v1".to_string(),
"secret".to_string(),
"legacy-chat-model".to_string(),
DEFAULT_REQUEST_TIMEOUT_MS,
DEFAULT_MAX_RETRIES,
DEFAULT_RETRY_BACKOFF_MS,
)
.expect("config should be valid");
let modern_config = legacy_config
.clone()
.with_openai_chat_token_budget_field(OpenAiChatTokenBudgetField::MaxCompletionTokens);
let request = LlmRunRequest::single_turn("系统", "用户")
.with_openai_chat()
.with_model("gpt-5.4-mini")
.with_max_output_tokens(256);
let legacy_json = serde_json::to_value(build_request_body(&request, &legacy_config, false))
.expect("legacy body should serialize");
assert_eq!(legacy_json["model"], "gpt-5.4-mini");
assert_eq!(legacy_json["max_tokens"], 256);
assert!(legacy_json.get("max_completion_tokens").is_none());
let modern_json = serde_json::to_value(build_request_body(&request, &modern_config, false))
.expect("modern body should serialize");
assert_eq!(modern_json["model"], "gpt-5.4-mini");
assert_eq!(modern_json["max_completion_tokens"], 256);
assert!(modern_json.get("max_tokens").is_none());
}
#[test]
fn responses_and_anthropic_token_budget_wire_fields_remain_unchanged() {
let config = LlmConfig::new(
LlmProvider::OpenAiCompatible,
"https://example.com/v1".to_string(),
"secret".to_string(),
"model-a".to_string(),
DEFAULT_REQUEST_TIMEOUT_MS,
DEFAULT_MAX_RETRIES,
DEFAULT_RETRY_BACKOFF_MS,
)
.expect("config should be valid")
.with_openai_chat_token_budget_field(OpenAiChatTokenBudgetField::MaxCompletionTokens);
let base_request = LlmRunRequest::single_turn("系统", "用户").with_max_output_tokens(384);
let responses_json = serde_json::to_value(build_request_body(
&base_request.clone().with_openai_responses(),
&config,
false,
))
.expect("Responses body should serialize");
assert_eq!(responses_json["max_output_tokens"], 384);
assert!(responses_json.get("max_completion_tokens").is_none());
assert!(responses_json.get("max_tokens").is_none());
let anthropic_json = serde_json::to_value(build_request_body(
&base_request.with_anthropic(),
&config,
false,
))
.expect("Anthropic body should serialize");
assert_eq!(anthropic_json["max_tokens"], 384);
assert!(anthropic_json.get("max_completion_tokens").is_none());
assert!(anthropic_json.get("max_output_tokens").is_none());
}
#[test]
fn run_request_rejects_tool_choice_without_function_tools() {
let error = LlmRunRequest::single_turn("系统", "用户")
@@ -4865,6 +5001,8 @@ mod tests {
assert_eq!(response.text, "搜索成功");
assert_eq!(request_json["web_search_options"], serde_json::json!({}));
assert_eq!(request_json["max_tokens"], 128);
assert!(request_json.get("max_completion_tokens").is_none());
assert!(request_json.get("official_fallback").is_none());
}
@@ -4973,6 +5111,7 @@ mod tests {
1,
)
.expect("config should be valid")
.with_openai_chat_token_budget_field(OpenAiChatTokenBudgetField::MaxCompletionTokens)
.with_official_fallback(true);
let client = LlmClient::new(config).expect("client should be created");
let response = client
@@ -4989,6 +5128,7 @@ mod tests {
]),
])
.with_openai_chat()
.with_max_output_tokens(256)
.with_response_reasoning_effort(LlmResponseReasoningEffort::Low),
)
.await
@@ -5007,6 +5147,8 @@ mod tests {
assert_eq!(response.model, "gpt-4o-mini");
assert_eq!(response.text, r#"{"levelName":"雨夜猫街"}"#);
assert_eq!(request_json["official_fallback"], serde_json::json!(true));
assert_eq!(request_json["max_completion_tokens"], 256);
assert!(request_json.get("max_tokens").is_none());
assert_eq!(request_json["reasoning_effort"], "low");
assert_eq!(
request_json["messages"][1]["content"],