Add GLM 5.2 model support

This commit is contained in:
Bryan Helmkamp 2026-07-22 17:19:16 -04:00
parent 03b1b790d1
commit c67cfef141
No known key found for this signature in database
9 changed files with 335 additions and 6 deletions

View file

@ -301,6 +301,7 @@ mod tests {
("gemini-3.1-pro-preview-customtools", "gemini-3.1-pro-preview-customtools", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini),
("gemini-3.5-flash", "gemini-3.5-flash", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini),
("glm-4.7", "glm-4.7", T::OpenAiCompatible, C::OpenAiCompatible, B::OpenAi, P::OpenAi),
("glm-5.2", "glm-5.2", T::OpenAiCompatible, C::OpenAiCompatible, B::OpenAi, P::OpenAi),
("gpt-5.4", "gpt-5.4", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),
("gpt-5.4-mini", "gpt-5.4-mini", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),
("gpt-5.4-pro", "gpt-5.4-pro", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),

View file

@ -25,10 +25,10 @@ pub(super) fn decode_response(
})?;
let mut content_parts = Vec::new();
if let Some(reasoning) = &choice.message.reasoning_content {
if let Some(reasoning) = choice.message.reasoning() {
if !reasoning.is_empty() {
content_parts.push(ContentPart::Thinking(ThinkingData {
text: reasoning.clone(),
text: reasoning.to_string(),
signature: None,
redacted: false,
}));

View file

@ -87,7 +87,7 @@ impl StreamState {
let delta = choice.delta.as_ref()?;
// Accumulate reasoning/thinking content (Kimi, etc.).
if let Some(reasoning) = &delta.reasoning_content {
if let Some(reasoning) = delta.reasoning() {
if !reasoning.is_empty() {
self.accumulated_reasoning.push_str(reasoning);
}

View file

@ -75,9 +75,19 @@ pub(super) struct ApiChoice {
pub(super) struct ApiChoiceMessage {
pub content: Option<String>,
pub reasoning_content: Option<String>,
/// OpenRouter's normalized spelling for reasoning text.
pub reasoning: Option<String>,
pub tool_calls: Option<Vec<ApiToolCall>>,
}
impl ApiChoiceMessage {
pub(super) fn reasoning(&self) -> Option<&str> {
self.reasoning_content
.as_deref()
.or(self.reasoning.as_deref())
}
}
#[derive(serde::Deserialize)]
pub(super) struct ApiToolCall {
pub id: String,
@ -178,9 +188,19 @@ pub(super) struct StreamDelta {
pub content: Option<String>,
/// Reasoning/thinking content (used by Kimi and other reasoning models).
pub reasoning_content: Option<String>,
/// OpenRouter's normalized spelling for reasoning text.
pub reasoning: Option<String>,
pub tool_calls: Option<Vec<StreamToolCall>>,
}
impl StreamDelta {
pub(super) fn reasoning(&self) -> Option<&str> {
self.reasoning_content
.as_deref()
.or(self.reasoning.as_deref())
}
}
#[derive(serde::Deserialize)]
pub(super) struct StreamToolCall {
pub index: usize,
@ -202,3 +222,46 @@ pub(super) struct AccumulatedToolCall {
pub arguments: String,
pub started: bool,
}
#[cfg(test)]
mod tests {
use super::{ApiResponse, StreamChunk};
#[test]
fn reasoning_accepts_provider_and_openrouter_spellings() {
let provider_response: ApiResponse = serde_json::from_value(serde_json::json!({
"id": "response-1",
"model": "reasoning-model",
"choices": [{
"message": {
"content": null,
"reasoning_content": "provider reasoning"
},
"finish_reason": "stop"
}]
}))
.unwrap();
assert_eq!(
provider_response.choices[0].message.reasoning(),
Some("provider reasoning")
);
let openrouter_chunk: StreamChunk = serde_json::from_value(serde_json::json!({
"id": "response-2",
"model": "reasoning-model",
"choices": [{
"delta": {"reasoning": "OpenRouter reasoning"},
"finish_reason": null
}]
}))
.unwrap();
assert_eq!(
openrouter_chunk.choices.unwrap()[0]
.delta
.as_ref()
.unwrap()
.reasoning(),
Some("OpenRouter reasoning")
);
}
}

View file

@ -14,6 +14,7 @@ use fabro_llm::types::{
CostSource, FinishReason, Message, ReasoningEffort, Request, ToolChoice, ToolDefinition,
};
use fabro_model::Catalog;
use fabro_model::catalog::LlmCatalogSettings;
use fabro_static::EnvVars;
fn make_request(model: &str) -> Request {
@ -332,6 +333,94 @@ async fn openrouter_complete() {
assert_eq!(response.cost_source, Some(CostSource::Authoritative));
}
#[fabro_macros::e2e_test(live("OPENROUTER_API_KEY"))]
async fn openrouter_glm_5_2_reasoning_tool_round_trip() {
let api_key =
std::env::var(EnvVars::OPENROUTER_API_KEY).expect("OPENROUTER_API_KEY must be set");
let overrides: LlmCatalogSettings = toml::from_str(
r"
[providers.openrouter]
enabled = true
",
)
.expect("OpenRouter catalog override should parse");
let catalog = Catalog::from_builtin_with_overrides(&overrides)
.expect("enabled OpenRouter catalog should build");
let adapter = OpenAiCompatibleAdapter::new(api_key, "https://openrouter.ai/api/v1")
.with_name("openrouter")
.with_catalog(Arc::new(catalog));
let tool = ToolDefinition::function(
"multiply",
"Multiply two integers",
serde_json::json!({
"type": "object",
"properties": {
"a": {"type": "integer"},
"b": {"type": "integer"}
},
"required": ["a", "b"]
}),
);
let request = Request {
model: "z-ai/glm-5.2".to_string(),
messages: vec![Message::user(
"Use the multiply tool to calculate 19 times 23. Do not calculate it yourself.",
)],
tools: Some(vec![tool]),
tool_choice: Some(ToolChoice::Required),
temperature: Some(0.0),
max_tokens: Some(4096),
reasoning_effort: Some(ReasoningEffort::High),
..make_request("z-ai/glm-5.2")
};
let tool_response = adapter.complete(&request).await.unwrap();
assert_eq!(tool_response.finish_reason, FinishReason::ToolCalls);
let raw_message_keys = tool_response
.raw
.as_ref()
.and_then(|raw| raw.pointer("/choices/0/message"))
.and_then(serde_json::Value::as_object)
.map(|message| message.keys().cloned().collect::<Vec<_>>())
.unwrap_or_default();
assert!(
tool_response.reasoning().is_some(),
"GLM 5.2 should return reasoning content before its tool call; raw message keys: \
{raw_message_keys:?}"
);
assert_eq!(tool_response.cost_source, Some(CostSource::Authoritative));
let tool_call = tool_response
.tool_calls()
.into_iter()
.next()
.expect("GLM 5.2 should call the required tool");
assert_eq!(tool_call.name, "multiply");
let mut messages = request.messages.clone();
messages.push(tool_response.message);
messages.push(Message::tool_result(
tool_call.id,
serde_json::json!({"product": 437}),
false,
));
let final_request = Request {
model: "z-ai/glm-5.2".to_string(),
messages,
temperature: Some(0.0),
max_tokens: Some(2048),
reasoning_effort: Some(ReasoningEffort::High),
..make_request("z-ai/glm-5.2")
};
let final_response = adapter.complete(&final_request).await.unwrap();
assert_eq!(final_response.finish_reason, FinishReason::Stop);
assert!(
final_response.text().contains("437"),
"GLM 5.2 should incorporate the replayed tool result"
);
assert_eq!(final_response.cost_source, Some(CostSource::Authoritative));
}
async fn run_multi_turn_cache_test(
adapter: &dyn ProviderAdapter,
model: &str,

View file

@ -476,7 +476,7 @@ async fn stream_tool_call_deltas() {
#[tokio::test]
async fn stream_reasoning_content_deltas() {
let sse = support::sse_data_transcript(&[
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","reasoning_content":"Let me "},"finish_reason":null}]}"#,
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","reasoning":"Let me "},"finish_reason":null}]}"#,
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"reasoning_content":"think"},"finish_reason":null}]}"#,
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"content":"4."},"finish_reason":null}]}"#,
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}"#,

View file

@ -2056,6 +2056,70 @@ enabled = true
);
}
#[test]
fn builtin_openrouter_includes_glm_5_2_when_enabled() {
let catalog = Catalog::from_builtin_with_overrides(&minimal_settings(
r"
[providers.openrouter]
enabled = true
",
))
.expect("enabled OpenRouter override should build from the built-in provider settings");
let model = catalog
.get("z-ai/glm-5.2")
.expect("OpenRouter GLM 5.2 should be present");
insta::assert_debug_snapshot!(model, @r#"
Model {
id: "z-ai/glm-5.2",
provider: openrouter,
family: "glm-5",
display_name: "GLM 5.2 (via OpenRouter)",
limits: ModelLimits {
context_window: 1048576,
max_output: Some(
131072,
),
},
training: None,
knowledge_cutoff: None,
features: ModelFeatures {
tools: true,
vision: false,
reasoning: true,
reasoning_effort: Levels,
prompt_cache: true,
sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
0.784,
),
output_cost_per_mtok: Some(
2.464,
),
cache_input_cost_per_mtok: Some(
0.1456,
),
},
estimated_output_tps: None,
aliases: [],
default: false,
small_default: false,
configured: false,
}
"#);
let settings = catalog
.model_settings("z-ai/glm-5.2")
.expect("OpenRouter GLM 5.2 settings should be present");
assert_eq!(settings.api_id, "z-ai/glm-5.2");
assert_eq!(settings.controls.reasoning_effort, vec![
ReasoningEffort::High,
ReasoningEffort::XHigh
]);
}
#[test]
fn builtin_ollama_provider_is_opt_in() {
let ollama = ProviderId::new("ollama");
@ -4352,6 +4416,67 @@ sampling_params = false
fn glm_4_7_in_catalog() {
let m = Catalog::builtin().get("glm-4.7").unwrap();
assert_eq!(m.provider, ProviderId::new("zai"));
assert_eq!(Catalog::builtin().get("glm4").unwrap().id, "glm-4.7");
}
#[test]
fn glm_5_2_in_catalog() {
let catalog = Catalog::builtin();
let model = catalog.get("glm-5.2").expect("GLM 5.2 should be present");
insta::assert_debug_snapshot!(model, @r#"
Model {
id: "glm-5.2",
provider: zai,
family: "glm-5",
display_name: "GLM 5.2",
limits: ModelLimits {
context_window: 1048576,
max_output: Some(
131072,
),
},
training: None,
knowledge_cutoff: None,
features: ModelFeatures {
tools: true,
vision: false,
reasoning: true,
reasoning_effort: Levels,
prompt_cache: true,
sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
1.4,
),
output_cost_per_mtok: Some(
4.4,
),
cache_input_cost_per_mtok: Some(
0.26,
),
},
estimated_output_tps: None,
aliases: [
"glm",
"glm5",
],
default: true,
small_default: false,
configured: false,
}
"#);
let settings = catalog
.model_settings("glm-5.2")
.expect("GLM 5.2 settings should be present");
assert_eq!(settings.api_id, "glm-5.2");
assert_eq!(settings.controls.reasoning_effort, vec![
ReasoningEffort::High,
ReasoningEffort::Max
]);
assert_eq!(catalog.get("glm").unwrap().id, "glm-5.2");
assert_eq!(catalog.get("glm5").unwrap().id, "glm-5.2");
}
#[test]

View file

@ -316,6 +316,31 @@ reasoning = false
input_cost_per_mtok = 0.1875
output_cost_per_mtok = 1.125
[models."z-ai/glm-5.2"]
provider = "openrouter"
api_id = "z-ai/glm-5.2"
display_name = "GLM 5.2 (via OpenRouter)"
family = "glm-5"
[models."z-ai/glm-5.2".limits]
context_window = 1048576
max_output = 131072
[models."z-ai/glm-5.2".features]
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
prompt_cache = true
[models."z-ai/glm-5.2".controls]
reasoning_effort = ["high", "xhigh"]
[models."z-ai/glm-5.2".costs]
input_cost_per_mtok = 0.784
output_cost_per_mtok = 2.464
cache_input_cost_per_mtok = 0.1456
[models."z-ai/glm-4.6"]
provider = "openrouter"
api_id = "z-ai/glm-4.6"

View file

@ -8,14 +8,40 @@ priority = 60
[providers.zai.auth]
credentials = ["env:ZAI_API_KEY", "vault:ZAI_API_KEY"]
[models."glm-5.2"]
provider = "zai"
api_id = "glm-5.2"
display_name = "GLM 5.2"
family = "glm-5"
default = true
aliases = ["glm", "glm5"]
[models."glm-5.2".limits]
context_window = 1048576
max_output = 131072
[models."glm-5.2".features]
tools = true
vision = false
reasoning = true
reasoning_effort = "levels"
prompt_cache = true
[models."glm-5.2".controls]
reasoning_effort = ["high", "max"]
[models."glm-5.2".costs]
input_cost_per_mtok = 1.4
output_cost_per_mtok = 4.4
cache_input_cost_per_mtok = 0.26
[models."glm-4.7"]
provider = "zai"
api_id = "glm-4.7"
display_name = "GLM 4.7"
family = "glm-4"
default = true
estimated_output_tps = 100
aliases = ["glm", "glm4"]
aliases = ["glm4"]
[models."glm-4.7".limits]
context_window = 202752