mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-09 03:20:56 +00:00
Add GLM 5.2 model support
This commit is contained in:
parent
03b1b790d1
commit
c67cfef141
9 changed files with 335 additions and 6 deletions
|
|
@ -301,6 +301,7 @@ mod tests {
|
|||
("gemini-3.1-pro-preview-customtools", "gemini-3.1-pro-preview-customtools", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini),
|
||||
("gemini-3.5-flash", "gemini-3.5-flash", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini),
|
||||
("glm-4.7", "glm-4.7", T::OpenAiCompatible, C::OpenAiCompatible, B::OpenAi, P::OpenAi),
|
||||
("glm-5.2", "glm-5.2", T::OpenAiCompatible, C::OpenAiCompatible, B::OpenAi, P::OpenAi),
|
||||
("gpt-5.4", "gpt-5.4", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),
|
||||
("gpt-5.4-mini", "gpt-5.4-mini", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),
|
||||
("gpt-5.4-pro", "gpt-5.4-pro", T::OpenAi, C::OpenAiResponses, B::OpenAi, P::OpenAi),
|
||||
|
|
|
|||
|
|
@ -25,10 +25,10 @@ pub(super) fn decode_response(
|
|||
})?;
|
||||
|
||||
let mut content_parts = Vec::new();
|
||||
if let Some(reasoning) = &choice.message.reasoning_content {
|
||||
if let Some(reasoning) = choice.message.reasoning() {
|
||||
if !reasoning.is_empty() {
|
||||
content_parts.push(ContentPart::Thinking(ThinkingData {
|
||||
text: reasoning.clone(),
|
||||
text: reasoning.to_string(),
|
||||
signature: None,
|
||||
redacted: false,
|
||||
}));
|
||||
|
|
|
|||
|
|
@ -87,7 +87,7 @@ impl StreamState {
|
|||
let delta = choice.delta.as_ref()?;
|
||||
|
||||
// Accumulate reasoning/thinking content (Kimi, etc.).
|
||||
if let Some(reasoning) = &delta.reasoning_content {
|
||||
if let Some(reasoning) = delta.reasoning() {
|
||||
if !reasoning.is_empty() {
|
||||
self.accumulated_reasoning.push_str(reasoning);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -75,9 +75,19 @@ pub(super) struct ApiChoice {
|
|||
pub(super) struct ApiChoiceMessage {
|
||||
pub content: Option<String>,
|
||||
pub reasoning_content: Option<String>,
|
||||
/// OpenRouter's normalized spelling for reasoning text.
|
||||
pub reasoning: Option<String>,
|
||||
pub tool_calls: Option<Vec<ApiToolCall>>,
|
||||
}
|
||||
|
||||
impl ApiChoiceMessage {
|
||||
pub(super) fn reasoning(&self) -> Option<&str> {
|
||||
self.reasoning_content
|
||||
.as_deref()
|
||||
.or(self.reasoning.as_deref())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
pub(super) struct ApiToolCall {
|
||||
pub id: String,
|
||||
|
|
@ -178,9 +188,19 @@ pub(super) struct StreamDelta {
|
|||
pub content: Option<String>,
|
||||
/// Reasoning/thinking content (used by Kimi and other reasoning models).
|
||||
pub reasoning_content: Option<String>,
|
||||
/// OpenRouter's normalized spelling for reasoning text.
|
||||
pub reasoning: Option<String>,
|
||||
pub tool_calls: Option<Vec<StreamToolCall>>,
|
||||
}
|
||||
|
||||
impl StreamDelta {
|
||||
pub(super) fn reasoning(&self) -> Option<&str> {
|
||||
self.reasoning_content
|
||||
.as_deref()
|
||||
.or(self.reasoning.as_deref())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
pub(super) struct StreamToolCall {
|
||||
pub index: usize,
|
||||
|
|
@ -202,3 +222,46 @@ pub(super) struct AccumulatedToolCall {
|
|||
pub arguments: String,
|
||||
pub started: bool,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{ApiResponse, StreamChunk};
|
||||
|
||||
#[test]
|
||||
fn reasoning_accepts_provider_and_openrouter_spellings() {
|
||||
let provider_response: ApiResponse = serde_json::from_value(serde_json::json!({
|
||||
"id": "response-1",
|
||||
"model": "reasoning-model",
|
||||
"choices": [{
|
||||
"message": {
|
||||
"content": null,
|
||||
"reasoning_content": "provider reasoning"
|
||||
},
|
||||
"finish_reason": "stop"
|
||||
}]
|
||||
}))
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
provider_response.choices[0].message.reasoning(),
|
||||
Some("provider reasoning")
|
||||
);
|
||||
|
||||
let openrouter_chunk: StreamChunk = serde_json::from_value(serde_json::json!({
|
||||
"id": "response-2",
|
||||
"model": "reasoning-model",
|
||||
"choices": [{
|
||||
"delta": {"reasoning": "OpenRouter reasoning"},
|
||||
"finish_reason": null
|
||||
}]
|
||||
}))
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
openrouter_chunk.choices.unwrap()[0]
|
||||
.delta
|
||||
.as_ref()
|
||||
.unwrap()
|
||||
.reasoning(),
|
||||
Some("OpenRouter reasoning")
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ use fabro_llm::types::{
|
|||
CostSource, FinishReason, Message, ReasoningEffort, Request, ToolChoice, ToolDefinition,
|
||||
};
|
||||
use fabro_model::Catalog;
|
||||
use fabro_model::catalog::LlmCatalogSettings;
|
||||
use fabro_static::EnvVars;
|
||||
|
||||
fn make_request(model: &str) -> Request {
|
||||
|
|
@ -332,6 +333,94 @@ async fn openrouter_complete() {
|
|||
assert_eq!(response.cost_source, Some(CostSource::Authoritative));
|
||||
}
|
||||
|
||||
#[fabro_macros::e2e_test(live("OPENROUTER_API_KEY"))]
|
||||
async fn openrouter_glm_5_2_reasoning_tool_round_trip() {
|
||||
let api_key =
|
||||
std::env::var(EnvVars::OPENROUTER_API_KEY).expect("OPENROUTER_API_KEY must be set");
|
||||
let overrides: LlmCatalogSettings = toml::from_str(
|
||||
r"
|
||||
[providers.openrouter]
|
||||
enabled = true
|
||||
",
|
||||
)
|
||||
.expect("OpenRouter catalog override should parse");
|
||||
let catalog = Catalog::from_builtin_with_overrides(&overrides)
|
||||
.expect("enabled OpenRouter catalog should build");
|
||||
let adapter = OpenAiCompatibleAdapter::new(api_key, "https://openrouter.ai/api/v1")
|
||||
.with_name("openrouter")
|
||||
.with_catalog(Arc::new(catalog));
|
||||
let tool = ToolDefinition::function(
|
||||
"multiply",
|
||||
"Multiply two integers",
|
||||
serde_json::json!({
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"a": {"type": "integer"},
|
||||
"b": {"type": "integer"}
|
||||
},
|
||||
"required": ["a", "b"]
|
||||
}),
|
||||
);
|
||||
let request = Request {
|
||||
model: "z-ai/glm-5.2".to_string(),
|
||||
messages: vec![Message::user(
|
||||
"Use the multiply tool to calculate 19 times 23. Do not calculate it yourself.",
|
||||
)],
|
||||
tools: Some(vec![tool]),
|
||||
tool_choice: Some(ToolChoice::Required),
|
||||
temperature: Some(0.0),
|
||||
max_tokens: Some(4096),
|
||||
reasoning_effort: Some(ReasoningEffort::High),
|
||||
..make_request("z-ai/glm-5.2")
|
||||
};
|
||||
|
||||
let tool_response = adapter.complete(&request).await.unwrap();
|
||||
assert_eq!(tool_response.finish_reason, FinishReason::ToolCalls);
|
||||
let raw_message_keys = tool_response
|
||||
.raw
|
||||
.as_ref()
|
||||
.and_then(|raw| raw.pointer("/choices/0/message"))
|
||||
.and_then(serde_json::Value::as_object)
|
||||
.map(|message| message.keys().cloned().collect::<Vec<_>>())
|
||||
.unwrap_or_default();
|
||||
assert!(
|
||||
tool_response.reasoning().is_some(),
|
||||
"GLM 5.2 should return reasoning content before its tool call; raw message keys: \
|
||||
{raw_message_keys:?}"
|
||||
);
|
||||
assert_eq!(tool_response.cost_source, Some(CostSource::Authoritative));
|
||||
let tool_call = tool_response
|
||||
.tool_calls()
|
||||
.into_iter()
|
||||
.next()
|
||||
.expect("GLM 5.2 should call the required tool");
|
||||
assert_eq!(tool_call.name, "multiply");
|
||||
|
||||
let mut messages = request.messages.clone();
|
||||
messages.push(tool_response.message);
|
||||
messages.push(Message::tool_result(
|
||||
tool_call.id,
|
||||
serde_json::json!({"product": 437}),
|
||||
false,
|
||||
));
|
||||
let final_request = Request {
|
||||
model: "z-ai/glm-5.2".to_string(),
|
||||
messages,
|
||||
temperature: Some(0.0),
|
||||
max_tokens: Some(2048),
|
||||
reasoning_effort: Some(ReasoningEffort::High),
|
||||
..make_request("z-ai/glm-5.2")
|
||||
};
|
||||
|
||||
let final_response = adapter.complete(&final_request).await.unwrap();
|
||||
assert_eq!(final_response.finish_reason, FinishReason::Stop);
|
||||
assert!(
|
||||
final_response.text().contains("437"),
|
||||
"GLM 5.2 should incorporate the replayed tool result"
|
||||
);
|
||||
assert_eq!(final_response.cost_source, Some(CostSource::Authoritative));
|
||||
}
|
||||
|
||||
async fn run_multi_turn_cache_test(
|
||||
adapter: &dyn ProviderAdapter,
|
||||
model: &str,
|
||||
|
|
|
|||
|
|
@ -476,7 +476,7 @@ async fn stream_tool_call_deltas() {
|
|||
#[tokio::test]
|
||||
async fn stream_reasoning_content_deltas() {
|
||||
let sse = support::sse_data_transcript(&[
|
||||
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","reasoning_content":"Let me "},"finish_reason":null}]}"#,
|
||||
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","reasoning":"Let me "},"finish_reason":null}]}"#,
|
||||
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"reasoning_content":"think"},"finish_reason":null}]}"#,
|
||||
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"content":"4."},"finish_reason":null}]}"#,
|
||||
r#"{"id":"chatcmpl_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}"#,
|
||||
|
|
|
|||
|
|
@ -2056,6 +2056,70 @@ enabled = true
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtin_openrouter_includes_glm_5_2_when_enabled() {
|
||||
let catalog = Catalog::from_builtin_with_overrides(&minimal_settings(
|
||||
r"
|
||||
[providers.openrouter]
|
||||
enabled = true
|
||||
",
|
||||
))
|
||||
.expect("enabled OpenRouter override should build from the built-in provider settings");
|
||||
|
||||
let model = catalog
|
||||
.get("z-ai/glm-5.2")
|
||||
.expect("OpenRouter GLM 5.2 should be present");
|
||||
insta::assert_debug_snapshot!(model, @r#"
|
||||
Model {
|
||||
id: "z-ai/glm-5.2",
|
||||
provider: openrouter,
|
||||
family: "glm-5",
|
||||
display_name: "GLM 5.2 (via OpenRouter)",
|
||||
limits: ModelLimits {
|
||||
context_window: 1048576,
|
||||
max_output: Some(
|
||||
131072,
|
||||
),
|
||||
},
|
||||
training: None,
|
||||
knowledge_cutoff: None,
|
||||
features: ModelFeatures {
|
||||
tools: true,
|
||||
vision: false,
|
||||
reasoning: true,
|
||||
reasoning_effort: Levels,
|
||||
prompt_cache: true,
|
||||
sampling_params: true,
|
||||
},
|
||||
costs: ModelCosts {
|
||||
input_cost_per_mtok: Some(
|
||||
0.784,
|
||||
),
|
||||
output_cost_per_mtok: Some(
|
||||
2.464,
|
||||
),
|
||||
cache_input_cost_per_mtok: Some(
|
||||
0.1456,
|
||||
),
|
||||
},
|
||||
estimated_output_tps: None,
|
||||
aliases: [],
|
||||
default: false,
|
||||
small_default: false,
|
||||
configured: false,
|
||||
}
|
||||
"#);
|
||||
|
||||
let settings = catalog
|
||||
.model_settings("z-ai/glm-5.2")
|
||||
.expect("OpenRouter GLM 5.2 settings should be present");
|
||||
assert_eq!(settings.api_id, "z-ai/glm-5.2");
|
||||
assert_eq!(settings.controls.reasoning_effort, vec![
|
||||
ReasoningEffort::High,
|
||||
ReasoningEffort::XHigh
|
||||
]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtin_ollama_provider_is_opt_in() {
|
||||
let ollama = ProviderId::new("ollama");
|
||||
|
|
@ -4352,6 +4416,67 @@ sampling_params = false
|
|||
fn glm_4_7_in_catalog() {
|
||||
let m = Catalog::builtin().get("glm-4.7").unwrap();
|
||||
assert_eq!(m.provider, ProviderId::new("zai"));
|
||||
assert_eq!(Catalog::builtin().get("glm4").unwrap().id, "glm-4.7");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn glm_5_2_in_catalog() {
|
||||
let catalog = Catalog::builtin();
|
||||
let model = catalog.get("glm-5.2").expect("GLM 5.2 should be present");
|
||||
insta::assert_debug_snapshot!(model, @r#"
|
||||
Model {
|
||||
id: "glm-5.2",
|
||||
provider: zai,
|
||||
family: "glm-5",
|
||||
display_name: "GLM 5.2",
|
||||
limits: ModelLimits {
|
||||
context_window: 1048576,
|
||||
max_output: Some(
|
||||
131072,
|
||||
),
|
||||
},
|
||||
training: None,
|
||||
knowledge_cutoff: None,
|
||||
features: ModelFeatures {
|
||||
tools: true,
|
||||
vision: false,
|
||||
reasoning: true,
|
||||
reasoning_effort: Levels,
|
||||
prompt_cache: true,
|
||||
sampling_params: true,
|
||||
},
|
||||
costs: ModelCosts {
|
||||
input_cost_per_mtok: Some(
|
||||
1.4,
|
||||
),
|
||||
output_cost_per_mtok: Some(
|
||||
4.4,
|
||||
),
|
||||
cache_input_cost_per_mtok: Some(
|
||||
0.26,
|
||||
),
|
||||
},
|
||||
estimated_output_tps: None,
|
||||
aliases: [
|
||||
"glm",
|
||||
"glm5",
|
||||
],
|
||||
default: true,
|
||||
small_default: false,
|
||||
configured: false,
|
||||
}
|
||||
"#);
|
||||
|
||||
let settings = catalog
|
||||
.model_settings("glm-5.2")
|
||||
.expect("GLM 5.2 settings should be present");
|
||||
assert_eq!(settings.api_id, "glm-5.2");
|
||||
assert_eq!(settings.controls.reasoning_effort, vec![
|
||||
ReasoningEffort::High,
|
||||
ReasoningEffort::Max
|
||||
]);
|
||||
assert_eq!(catalog.get("glm").unwrap().id, "glm-5.2");
|
||||
assert_eq!(catalog.get("glm5").unwrap().id, "glm-5.2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
|
|||
|
|
@ -316,6 +316,31 @@ reasoning = false
|
|||
input_cost_per_mtok = 0.1875
|
||||
output_cost_per_mtok = 1.125
|
||||
|
||||
[models."z-ai/glm-5.2"]
|
||||
provider = "openrouter"
|
||||
api_id = "z-ai/glm-5.2"
|
||||
display_name = "GLM 5.2 (via OpenRouter)"
|
||||
family = "glm-5"
|
||||
|
||||
[models."z-ai/glm-5.2".limits]
|
||||
context_window = 1048576
|
||||
max_output = 131072
|
||||
|
||||
[models."z-ai/glm-5.2".features]
|
||||
tools = true
|
||||
vision = false
|
||||
reasoning = true
|
||||
reasoning_effort = "levels"
|
||||
prompt_cache = true
|
||||
|
||||
[models."z-ai/glm-5.2".controls]
|
||||
reasoning_effort = ["high", "xhigh"]
|
||||
|
||||
[models."z-ai/glm-5.2".costs]
|
||||
input_cost_per_mtok = 0.784
|
||||
output_cost_per_mtok = 2.464
|
||||
cache_input_cost_per_mtok = 0.1456
|
||||
|
||||
[models."z-ai/glm-4.6"]
|
||||
provider = "openrouter"
|
||||
api_id = "z-ai/glm-4.6"
|
||||
|
|
|
|||
|
|
@ -8,14 +8,40 @@ priority = 60
|
|||
[providers.zai.auth]
|
||||
credentials = ["env:ZAI_API_KEY", "vault:ZAI_API_KEY"]
|
||||
|
||||
[models."glm-5.2"]
|
||||
provider = "zai"
|
||||
api_id = "glm-5.2"
|
||||
display_name = "GLM 5.2"
|
||||
family = "glm-5"
|
||||
default = true
|
||||
aliases = ["glm", "glm5"]
|
||||
|
||||
[models."glm-5.2".limits]
|
||||
context_window = 1048576
|
||||
max_output = 131072
|
||||
|
||||
[models."glm-5.2".features]
|
||||
tools = true
|
||||
vision = false
|
||||
reasoning = true
|
||||
reasoning_effort = "levels"
|
||||
prompt_cache = true
|
||||
|
||||
[models."glm-5.2".controls]
|
||||
reasoning_effort = ["high", "max"]
|
||||
|
||||
[models."glm-5.2".costs]
|
||||
input_cost_per_mtok = 1.4
|
||||
output_cost_per_mtok = 4.4
|
||||
cache_input_cost_per_mtok = 0.26
|
||||
|
||||
[models."glm-4.7"]
|
||||
provider = "zai"
|
||||
api_id = "glm-4.7"
|
||||
display_name = "GLM 4.7"
|
||||
family = "glm-4"
|
||||
default = true
|
||||
estimated_output_tps = 100
|
||||
aliases = ["glm", "glm4"]
|
||||
aliases = ["glm4"]
|
||||
|
||||
[models."glm-4.7".limits]
|
||||
context_window = 202752
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue