From 21f94f48de9e047f8a3c3a9982d4ebbb7d858dac Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 7 Mar 2026 18:01:28 -0500 Subject: [PATCH] Add POST /completions endpoint with Anthropic-style SSE streaming Adds a completions API endpoint that supports both streaming (SSE) and non-streaming (JSON) modes, with structured output via JSON Schema. Wires up the CLI `arc llm prompt` command to use the server when `--mode server` is specified. Co-Authored-By: Claude Opus 4.6 --- Cargo.lock | 1 + crates/arc-api/Cargo.toml | 1 + crates/arc-api/src/server.rs | 326 ++++++++++++++++++ crates/arc-cli/src/main.rs | 20 +- crates/arc-llm/src/cli.rs | 264 ++++++++++++++ docs/api-reference/arc-api.yaml | 95 +++++ .../arc-api-client/src/api/completions-api.ts | 136 ++++++++ ...etting-group.ts => completion-response.ts} | 25 +- ...ting-field-type.ts => completion-usage.ts} | 17 +- .../src/models/create-completion-request.ts | 39 +++ .../src/models/evaluation-result.ts | 30 -- .../src/models/setting-field.ts | 48 --- .../src/models/verification-status.ts | 30 -- 13 files changed, 894 insertions(+), 138 deletions(-) create mode 100644 packages/arc-api-client/src/api/completions-api.ts rename packages/arc-api-client/src/models/{setting-group.ts => completion-response.ts} (52%) rename packages/arc-api-client/src/models/{setting-field-type.ts => completion-usage.ts} (58%) create mode 100644 packages/arc-api-client/src/models/create-completion-request.ts delete mode 100644 packages/arc-api-client/src/models/evaluation-result.ts delete mode 100644 packages/arc-api-client/src/models/setting-field.ts delete mode 100644 packages/arc-api-client/src/models/verification-status.ts diff --git a/Cargo.lock b/Cargo.lock index 18baa3bac..279213138 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -145,6 +145,7 @@ dependencies = [ "chrono", "clap", "dirs", + "futures-util", "http-body-util", "hyper", "hyper-util", diff --git a/crates/arc-api/Cargo.toml b/crates/arc-api/Cargo.toml index 835dc34b9..f54ad9178 100644 --- a/crates/arc-api/Cargo.toml +++ b/crates/arc-api/Cargo.toml @@ -16,6 +16,7 @@ arc-util = { path = "../arc-util" } arc-db = { path = "../arc-db" } arc-types = { path = "../arc-types" } chrono.workspace = true +futures-util.workspace = true axum = "0.8" dirs.workspace = true sqlx.workspace = true diff --git a/crates/arc-api/src/server.rs b/crates/arc-api/src/server.rs index b32fade4f..8e443198a 100644 --- a/crates/arc-api/src/server.rs +++ b/crates/arc-api/src/server.rs @@ -239,6 +239,7 @@ fn demo_routes() -> Router> { .route("/insights/history", get(crate::demo::list_query_history)) .route("/models", get(crate::demo::list_models)) .route("/models/{id}/test", post(test_model)) + .route("/completions", post(create_completion)) .route("/settings", get(crate::demo::get_server_configuration)) .route("/usage", get(crate::demo::get_aggregate_usage)) } @@ -290,6 +291,7 @@ fn real_routes() -> Router> { .route("/insights/history", get(not_implemented)) .route("/models", get(crate::demo::list_models)) .route("/models/{id}/test", post(test_model)) + .route("/completions", post(create_completion)) .route("/settings", get(not_implemented)) .route("/usage", get(get_aggregate_usage)) } @@ -980,6 +982,243 @@ async fn test_model( } } +fn finish_reason_to_stop_reason(reason: &arc_llm::types::FinishReason) -> &'static str { + match reason { + arc_llm::types::FinishReason::Stop => "end_turn", + arc_llm::types::FinishReason::Length => "max_tokens", + _ => "end_turn", + } +} + +async fn create_completion( + _auth: AuthenticatedService, + State(state): State>, + Json(req): Json, +) -> Response { + // Resolve model + let model_id = req.model.unwrap_or_else(|| { + arc_llm::catalog::list_models(None) + .first() + .map_or_else(|| "claude-sonnet-4-5".to_string(), |m| m.id.clone()) + }); + + let info = arc_llm::catalog::get_model_info(&model_id); + + // Build GenerateParams + let mut params = arc_llm::generate::GenerateParams::new(&model_id).prompt(&req.prompt); + if let Some(ref info) = info { + params = params.provider(&info.provider); + } + if let Some(system) = req.system { + params = params.system(&system); + } + if let Some(temp) = req.temperature { + params = params.temperature(temp); + } + if let Some(max_tokens) = req.max_tokens { + params = params.max_tokens(max_tokens); + } + if let Some(top_p) = req.top_p { + params = params.top_p(top_p); + } + + // Force non-streaming for structured output + let use_stream = req.stream && req.schema.is_none(); + + // Dry-run mode returns a stub response + if state.dry_run { + let msg_id = ulid::Ulid::new().to_string(); + if use_stream { + let sse_stream = futures_util::stream::iter(vec![ + Ok::<_, std::convert::Infallible>( + Event::default() + .event("message_start") + .data(serde_json::json!({ + "type": "message_start", + "message": { + "id": msg_id, + "type": "message", + "role": "assistant", + "content": [], + "model": model_id, + "stop_reason": null, + "usage": {"input_tokens": 0} + } + }).to_string()), + ), + Ok(Event::default() + .event("message_stop") + .data(serde_json::json!({"type": "message_stop"}).to_string())), + ]); + return Sse::new(sse_stream).into_response(); + } + return Json(arc_types::CompletionResponse { + id: msg_id, + model: model_id, + content: String::new(), + stop_reason: "end_turn".to_string(), + usage: arc_types::CompletionUsage { + input_tokens: 0, + output_tokens: 0, + }, + output: None, + }) + .into_response(); + } + + if use_stream { + // Streaming path + let stream_result = match arc_llm::generate::stream(params).await { + Ok(s) => s, + Err(e) => { + return ApiError::new(StatusCode::BAD_GATEWAY, format!("LLM error: {e}")) + .into_response() + } + }; + + let msg_id = ulid::Ulid::new().to_string(); + let model_for_stream = model_id.clone(); + + let sse_stream = futures_util::stream::once(futures_util::future::ready(Ok::<_, std::convert::Infallible>( + Event::default() + .event("message_start") + .data(serde_json::json!({ + "type": "message_start", + "message": { + "id": msg_id, + "type": "message", + "role": "assistant", + "content": [], + "model": model_for_stream, + "stop_reason": null, + "usage": {"input_tokens": 0} + } + }).to_string()), + ))) + .chain(futures_util::stream::once(futures_util::future::ready(Ok( + Event::default() + .event("content_block_start") + .data(serde_json::json!({ + "type": "content_block_start", + "index": 0, + "content_block": {"type": "text", "text": ""} + }).to_string()), + )))) + .chain(futures_util::stream::once(futures_util::future::ready(Ok( + Event::default() + .event("ping") + .data(serde_json::json!({"type": "ping"}).to_string()), + )))) + .chain( + tokio_stream::StreamExt::map(stream_result, |event| { + match event { + Ok(arc_llm::types::StreamEvent::TextDelta { delta, .. }) => Ok( + Event::default() + .event("content_block_delta") + .data(serde_json::json!({ + "type": "content_block_delta", + "index": 0, + "delta": {"type": "text_delta", "text": delta} + }).to_string()), + ), + Ok(arc_llm::types::StreamEvent::TextEnd { .. }) => Ok( + Event::default() + .event("content_block_stop") + .data(serde_json::json!({"type": "content_block_stop", "index": 0}).to_string()), + ), + Ok(arc_llm::types::StreamEvent::Finish { finish_reason, usage, .. }) => Ok( + Event::default() + .event("message_delta") + .data(serde_json::json!({ + "type": "message_delta", + "delta": {"stop_reason": finish_reason_to_stop_reason(&finish_reason)}, + "usage": {"output_tokens": usage.output_tokens} + }).to_string()), + ), + Ok(arc_llm::types::StreamEvent::Error { error, .. }) => Ok( + Event::default() + .event("error") + .data(serde_json::json!({ + "type": "error", + "error": {"type": "server_error", "message": error.to_string()} + }).to_string()), + ), + Err(e) => Ok( + Event::default() + .event("error") + .data(serde_json::json!({ + "type": "error", + "error": {"type": "server_error", "message": e.to_string()} + }).to_string()), + ), + // Skip events we don't map (StreamStart, TextStart, reasoning, tool calls, etc.) + _ => Ok(Event::default().comment("ignored")), + } + }), + ) + .chain(futures_util::stream::once(futures_util::future::ready(Ok( + Event::default() + .event("message_stop") + .data(serde_json::json!({"type": "message_stop"}).to_string()), + )))); + + Sse::new(sse_stream) + .keep_alive( + axum::response::sse::KeepAlive::new() + .interval(Duration::from_secs(15)) + .event( + Event::default() + .event("ping") + .data(serde_json::json!({"type": "ping"}).to_string()), + ), + ) + .into_response() + } else { + // Non-streaming path + let msg_id = ulid::Ulid::new().to_string(); + + if let Some(schema) = req.schema { + match arc_llm::generate::generate_object(params, schema).await { + Ok(result) => Json(arc_types::CompletionResponse { + id: msg_id, + model: model_id, + content: result.text(), + stop_reason: finish_reason_to_stop_reason(&result.finish_reason).to_string(), + usage: arc_types::CompletionUsage { + input_tokens: result.usage.input_tokens, + output_tokens: result.usage.output_tokens, + }, + output: result.output, + }) + .into_response(), + Err(e) => { + ApiError::new(StatusCode::BAD_GATEWAY, format!("LLM error: {e}")) + .into_response() + } + } + } else { + match arc_llm::generate::generate(params).await { + Ok(result) => Json(arc_types::CompletionResponse { + id: msg_id, + model: model_id, + content: result.text(), + stop_reason: finish_reason_to_stop_reason(&result.finish_reason).to_string(), + usage: arc_types::CompletionUsage { + input_tokens: result.usage.input_tokens, + output_tokens: result.usage.output_tokens, + }, + output: None, + }) + .into_response(), + Err(e) => { + ApiError::new(StatusCode::BAD_GATEWAY, format!("LLM error: {e}")) + .into_response() + } + } + } + } +} + async fn get_retro( _auth: AuthenticatedService, State(state): State>, @@ -1914,4 +2153,91 @@ mod tests { let response = app.oneshot(req).await.unwrap(); assert_eq!(response.status(), StatusCode::CONFLICT); } + + #[tokio::test] + async fn create_completion_non_streaming_returns_json() { + let state = create_app_state_with_options( + test_db().await, + test_registry, + true, + 5, + arc_workflows::git::GitAuthor::default(), + ); + let app = build_router(state, AuthMode::Disabled); + + let req = Request::builder() + .method("POST") + .uri("/completions") + .header("content-type", "application/json") + .body(Body::from( + serde_json::to_string(&serde_json::json!({ + "prompt": "Hello", + "stream": false + })) + .unwrap(), + )) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + assert_eq!(response.status(), StatusCode::OK); + + let body = body_json(response.into_body()).await; + assert!(body["id"].is_string()); + assert!(body["model"].is_string()); + assert_eq!(body["stop_reason"], "end_turn"); + assert!(body["usage"]["input_tokens"].is_number()); + assert!(body["usage"]["output_tokens"].is_number()); + } + + #[tokio::test] + async fn create_completion_streaming_returns_sse() { + let state = create_app_state_with_options( + test_db().await, + test_registry, + true, + 5, + arc_workflows::git::GitAuthor::default(), + ); + let app = build_router(state, AuthMode::Disabled); + + let req = Request::builder() + .method("POST") + .uri("/completions") + .header("content-type", "application/json") + .body(Body::from( + serde_json::to_string(&serde_json::json!({ + "prompt": "Hello", + "stream": true + })) + .unwrap(), + )) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + assert_eq!(response.status(), StatusCode::OK); + assert_eq!( + response + .headers() + .get("content-type") + .unwrap() + .to_str() + .unwrap(), + "text/event-stream" + ); + } + + #[tokio::test] + async fn create_completion_missing_prompt_returns_422() { + let app = test_app_with(test_db().await); + + let req = Request::builder() + .method("POST") + .uri("/completions") + .header("content-type", "application/json") + .body(Body::from("{}")) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + assert_eq!(response.status(), StatusCode::UNPROCESSABLE_ENTITY); + } } diff --git a/crates/arc-cli/src/main.rs b/crates/arc-cli/src/main.rs index 7ff9f783a..54efabc07 100644 --- a/crates/arc-cli/src/main.rs +++ b/crates/arc-cli/src/main.rs @@ -153,7 +153,25 @@ async fn main() -> Result<()> { if args.model.is_none() { args.model = llm_defaults.and_then(|l| l.model.clone()); } - arc_llm::cli::run_prompt(args).await? + let resolved = cli_config::resolve_mode( + cli.mode, + cli.server_url.as_deref(), + &cli_config, + ); + match resolved.mode { + cli_config::ExecutionMode::Server => { + let client = + cli_config::build_server_client(resolved.tls.as_ref())?; + let server = arc_llm::cli::ServerConnection { + client, + base_url: resolved.server_base_url, + }; + arc_llm::cli::run_prompt_via_server(args, &server).await? + } + cli_config::ExecutionMode::Standalone => { + arc_llm::cli::run_prompt(args).await? + } + } } LlmCommand::Chat(mut args) => { if args.model.is_none() { diff --git a/crates/arc-llm/src/cli.rs b/crates/arc-llm/src/cli.rs index 4e675b22f..5a50ac6a0 100644 --- a/crates/arc-llm/src/cli.rs +++ b/crates/arc-llm/src/cli.rs @@ -354,6 +354,191 @@ pub async fn run_prompt(args: PromptArgs) -> Result<()> { Ok(()) } +pub async fn run_prompt_via_server(args: PromptArgs, server: &ServerConnection) -> Result<()> { + let stdin_prompt = read_stdin_prompt(); + let prompt_text = resolve_prompt(args.prompt, stdin_prompt)?; + + // Extract known options + let mut temperature: Option = None; + let mut max_tokens: Option = None; + let mut top_p: Option = None; + for (key, value) in &args.option { + match key.as_str() { + "temperature" => { + temperature = Some( + value + .parse() + .with_context(|| format!("invalid temperature value: {value}"))?, + ); + } + "max_tokens" => { + max_tokens = Some( + value + .parse() + .with_context(|| format!("invalid max_tokens value: {value}"))?, + ); + } + "top_p" => { + top_p = Some( + value + .parse() + .with_context(|| format!("invalid top_p value: {value}"))?, + ); + } + _ => {} + } + } + + let schema: Option = match &args.schema { + Some(s) => Some(serde_json::from_str(s).context("--schema must be valid JSON")?), + None => None, + }; + + // Force non-streaming for structured output + let use_stream = !args.no_stream && schema.is_none(); + + let mut body = serde_json::json!({ + "prompt": prompt_text, + "stream": use_stream, + }); + if let Some(ref model) = args.model { + body["model"] = serde_json::Value::String(model.clone()); + } + if let Some(ref system) = args.system { + body["system"] = serde_json::Value::String(system.clone()); + } + if let Some(ref schema) = schema { + body["schema"] = schema.clone(); + } + if let Some(t) = temperature { + body["temperature"] = serde_json::json!(t); + } + if let Some(m) = max_tokens { + body["max_tokens"] = serde_json::json!(m); + } + if let Some(t) = top_p { + body["top_p"] = serde_json::json!(t); + } + + let url = format!("{}/completions", server.base_url); + + if use_stream { + let response = server + .client + .post(&url) + .json(&body) + .send() + .await + .with_context(|| format!("Failed to connect to server at {}", server.base_url))?; + + let status = response.status(); + if !status.is_success() { + let text = response.text().await.unwrap_or_default(); + bail!("Server returned {status}: {text}"); + } + + // Parse SSE stream + let mut stream = response.bytes_stream(); + let mut buffer = String::new(); + let mut output_usage: Option = None; + + use futures::StreamExt; + while let Some(chunk) = stream.next().await { + let chunk = chunk.context("Error reading stream")?; + buffer.push_str(&String::from_utf8_lossy(&chunk)); + + // Process complete SSE frames (separated by blank lines) + while let Some(pos) = buffer.find("\n\n") { + let frame = buffer[..pos].to_string(); + buffer = buffer[pos + 2..].to_string(); + + let mut event_type = String::new(); + let mut data = String::new(); + for line in frame.lines() { + if let Some(val) = line.strip_prefix("event: ") { + event_type = val.to_string(); + } else if let Some(val) = line.strip_prefix("data: ") { + data = val.to_string(); + } + } + + match event_type.as_str() { + "content_block_delta" => { + if let Ok(parsed) = serde_json::from_str::(&data) { + if let Some(text) = parsed["delta"]["text"].as_str() { + print!("{text}"); + let _ = io::stdout().flush(); + } + } + } + "message_delta" => { + if args.usage { + if let Ok(parsed) = serde_json::from_str::(&data) { + output_usage = Some(parsed); + } + } + } + "error" => { + if let Ok(parsed) = serde_json::from_str::(&data) { + let msg = parsed["error"]["message"] + .as_str() + .unwrap_or("Unknown error"); + bail!("Server error: {msg}"); + } + } + _ => {} + } + } + } + println!(); + + if args.usage { + if let Some(usage) = output_usage { + let output_tokens = usage["usage"]["output_tokens"].as_i64().unwrap_or(0); + eprintln!("Tokens: output {output_tokens}"); + } + } + } else { + // Non-streaming + let response = server + .client + .post(&url) + .json(&body) + .send() + .await + .with_context(|| format!("Failed to connect to server at {}", server.base_url))?; + + let status = response.status(); + if !status.is_success() { + let text = response.text().await.unwrap_or_default(); + bail!("Server returned {status}: {text}"); + } + + let result: serde_json::Value = response + .json() + .await + .context("Failed to parse completion response")?; + + if schema.is_some() { + if let Some(output) = result.get("output") { + println!("{}", serde_json::to_string_pretty(output)?); + } else if let Some(content) = result["content"].as_str() { + print!("{content}"); + } + } else if let Some(content) = result["content"].as_str() { + print!("{content}"); + } + + if args.usage { + let input = result["usage"]["input_tokens"].as_i64().unwrap_or(0); + let output = result["usage"]["output_tokens"].as_i64().unwrap_or(0); + eprintln!("Tokens: {} input, {} output, {} total", input, output, input + output); + } + } + + Ok(()) +} + #[derive(Deserialize)] struct PaginatedModelsResponse { data: Vec, @@ -951,4 +1136,83 @@ mod tests { let result = fetch_models_from_server(&client, &server.url(""), None).await; assert!(result.is_err()); } + + // --- run_prompt_via_server --- + + #[tokio::test] + async fn run_prompt_via_server_non_streaming() { + let mock_server = httpmock::MockServer::start_async().await; + let mock = mock_server.mock_async(|when, then| { + when.method("POST").path("/completions"); + then.status(200) + .header("Content-Type", "application/json") + .body(serde_json::json!({ + "id": "msg_123", + "model": "test-model", + "content": "Hello world", + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 5} + }).to_string()); + }).await; + + let server = ServerConnection { + client: reqwest::Client::new(), + base_url: mock_server.url(""), + }; + + let args = PromptArgs { + prompt: Some("Hello".into()), + model: Some("test-model".into()), + system: None, + no_stream: true, + usage: false, + schema: None, + option: vec![], + }; + + let result = run_prompt_via_server(args, &server).await; + assert!(result.is_ok()); + mock.assert_async().await; + } + + #[tokio::test] + async fn run_prompt_via_server_streaming() { + let mock_server = httpmock::MockServer::start_async().await; + let sse_body = "\ +event: message_start\n\ +data: {\"type\":\"message_start\",\"message\":{\"id\":\"msg_1\",\"type\":\"message\",\"role\":\"assistant\",\"content\":[],\"model\":\"test\",\"stop_reason\":null,\"usage\":{\"input_tokens\":5}}}\n\ +\n\ +event: content_block_delta\n\ +data: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"text_delta\",\"text\":\"Hi\"}}\n\ +\n\ +event: message_stop\n\ +data: {\"type\":\"message_stop\"}\n\ +\n"; + + let mock = mock_server.mock_async(|when, then| { + when.method("POST").path("/completions"); + then.status(200) + .header("Content-Type", "text/event-stream") + .body(sse_body); + }).await; + + let server = ServerConnection { + client: reqwest::Client::new(), + base_url: mock_server.url(""), + }; + + let args = PromptArgs { + prompt: Some("Hello".into()), + model: Some("test-model".into()), + system: None, + no_stream: false, + usage: false, + schema: None, + option: vec![], + }; + + let result = run_prompt_via_server(args, &server).await; + assert!(result.is_ok()); + mock.assert_async().await; + } } diff --git a/docs/api-reference/arc-api.yaml b/docs/api-reference/arc-api.yaml index ddc5f1219..29b348653 100644 --- a/docs/api-reference/arc-api.yaml +++ b/docs/api-reference/arc-api.yaml @@ -29,6 +29,8 @@ tags: description: Run retrospectives - name: Models description: Available LLM models + - name: Completions + description: Single-turn LLM completions - name: Settings description: Platform configuration @@ -1093,6 +1095,38 @@ paths: schema: $ref: "#/components/schemas/ErrorResponse" + # ── Completions ─────────────────────────────────────────────────────── + + /completions: + post: + operationId: createCompletion + tags: [Completions] + summary: Create Completion + description: | + Generate a text completion. Set `stream: true` for Anthropic-style SSE streaming. + + SSE event types: message_start, content_block_start, content_block_delta, + content_block_stop, message_delta, message_stop, ping, error. + requestBody: + required: true + content: + application/json: + schema: + $ref: "#/components/schemas/CreateCompletionRequest" + responses: + "200": + description: Completion result (JSON when stream=false, SSE when stream=true) + content: + application/json: + schema: + $ref: "#/components/schemas/CompletionResponse" + "400": + description: Invalid request + content: + application/json: + schema: + $ref: "#/components/schemas/ErrorResponse" + # ── Settings ────────────────────────────────────────────────────────── /settings: @@ -1474,6 +1508,67 @@ components: nullable: true description: Error details when status is "error". + # ── Completion Schemas ───────────────────────────────────────────── + + CreateCompletionRequest: + type: object + required: [prompt] + properties: + prompt: + type: string + description: The user prompt text. + model: + type: string + description: Model ID or alias. Server picks default if omitted. + system: + type: string + description: System prompt. + stream: + type: boolean + default: true + description: Stream response via SSE. + schema: + description: JSON Schema for structured output. + temperature: + type: number + format: double + max_tokens: + type: integer + format: int64 + top_p: + type: number + format: double + + CompletionUsage: + type: object + required: [input_tokens, output_tokens] + properties: + input_tokens: + type: integer + format: int64 + output_tokens: + type: integer + format: int64 + + CompletionResponse: + type: object + required: [id, model, content, stop_reason, usage] + properties: + id: + type: string + model: + type: string + content: + type: string + description: Generated text. + stop_reason: + type: string + description: Why generation stopped (end_turn, max_tokens). + usage: + $ref: "#/components/schemas/CompletionUsage" + output: + description: Parsed structured output when schema was provided. + PaginatedSavedQueryList: description: Paginated list of saved queries. type: object diff --git a/packages/arc-api-client/src/api/completions-api.ts b/packages/arc-api-client/src/api/completions-api.ts new file mode 100644 index 000000000..3c0993b77 --- /dev/null +++ b/packages/arc-api-client/src/api/completions-api.ts @@ -0,0 +1,136 @@ +/* tslint:disable */ +/* eslint-disable */ +/** + * Arc Run API + * HTTP API for managing Arc workflow run executions. + * + * The version of the OpenAPI document: 0.1.0 + * + * + * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). + * https://openapi-generator.tech + * Do not edit the class manually. + */ + + +import type { Configuration } from '../configuration'; +import type { AxiosPromise, AxiosInstance, RawAxiosRequestConfig } from 'axios'; +import globalAxios from 'axios'; +// Some imports not used depending on template conditions +// @ts-ignore +import { DUMMY_BASE_URL, assertParamExists, setApiKeyToObject, setBasicAuthToObject, setBearerAuthToObject, setOAuthToObject, setSearchParams, serializeDataIfNeeded, toPathString, createRequestFunction, replaceWithSerializableTypeIfNeeded } from '../common'; +// @ts-ignore +import { BASE_PATH, COLLECTION_FORMATS, type RequestArgs, BaseAPI, RequiredError, operationServerMap } from '../base'; +// @ts-ignore +import type { CompletionResponse } from '../models'; +// @ts-ignore +import type { CreateCompletionRequest } from '../models'; +// @ts-ignore +import type { ErrorResponse } from '../models'; +/** + * CompletionsApi - axios parameter creator + */ +export const CompletionsApiAxiosParamCreator = function (configuration?: Configuration) { + return { + /** + * Generate a text completion. Set `stream: true` for Anthropic-style SSE streaming. SSE event types: message_start, content_block_start, content_block_delta, content_block_stop, message_delta, message_stop, ping, error. + * @summary Create Completion + * @param {CreateCompletionRequest} createCompletionRequest + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + createCompletion: async (createCompletionRequest: CreateCompletionRequest, options: RawAxiosRequestConfig = {}): Promise => { + // verify required parameter 'createCompletionRequest' is not null or undefined + assertParamExists('createCompletion', 'createCompletionRequest', createCompletionRequest) + const localVarPath = `/completions`; + // use dummy base URL string because the URL constructor only accepts absolute URLs. + const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL); + let baseOptions; + if (configuration) { + baseOptions = configuration.baseOptions; + } + + const localVarRequestOptions = { method: 'POST', ...baseOptions, ...options}; + const localVarHeaderParameter = {} as any; + const localVarQueryParameter = {} as any; + + // authentication mTLS required + await setApiKeyToObject(localVarHeaderParameter, "X-mTLS-Client-CN", configuration) + + // authentication BearerAuth required + // http bearer authentication required + await setBearerAuthToObject(localVarHeaderParameter, configuration) + + localVarHeaderParameter['Content-Type'] = 'application/json'; + localVarHeaderParameter['Accept'] = 'application/json'; + + setSearchParams(localVarUrlObj, localVarQueryParameter); + let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {}; + localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers}; + localVarRequestOptions.data = serializeDataIfNeeded(createCompletionRequest, localVarRequestOptions, configuration) + + return { + url: toPathString(localVarUrlObj), + options: localVarRequestOptions, + }; + }, + } +}; + +/** + * CompletionsApi - functional programming interface + */ +export const CompletionsApiFp = function(configuration?: Configuration) { + const localVarAxiosParamCreator = CompletionsApiAxiosParamCreator(configuration) + return { + /** + * Generate a text completion. Set `stream: true` for Anthropic-style SSE streaming. SSE event types: message_start, content_block_start, content_block_delta, content_block_stop, message_delta, message_stop, ping, error. + * @summary Create Completion + * @param {CreateCompletionRequest} createCompletionRequest + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + async createCompletion(createCompletionRequest: CreateCompletionRequest, options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise> { + const localVarAxiosArgs = await localVarAxiosParamCreator.createCompletion(createCompletionRequest, options); + const localVarOperationServerIndex = configuration?.serverIndex ?? 0; + const localVarOperationServerBasePath = operationServerMap['CompletionsApi.createCompletion']?.[localVarOperationServerIndex]?.url; + return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath); + }, + } +}; + +/** + * CompletionsApi - factory interface + */ +export const CompletionsApiFactory = function (configuration?: Configuration, basePath?: string, axios?: AxiosInstance) { + const localVarFp = CompletionsApiFp(configuration) + return { + /** + * Generate a text completion. Set `stream: true` for Anthropic-style SSE streaming. SSE event types: message_start, content_block_start, content_block_delta, content_block_stop, message_delta, message_stop, ping, error. + * @summary Create Completion + * @param {CreateCompletionRequest} createCompletionRequest + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + createCompletion(createCompletionRequest: CreateCompletionRequest, options?: RawAxiosRequestConfig): AxiosPromise { + return localVarFp.createCompletion(createCompletionRequest, options).then((request) => request(axios, basePath)); + }, + }; +}; + +/** + * CompletionsApi - object-oriented interface + */ +export class CompletionsApi extends BaseAPI { + /** + * Generate a text completion. Set `stream: true` for Anthropic-style SSE streaming. SSE event types: message_start, content_block_start, content_block_delta, content_block_stop, message_delta, message_stop, ping, error. + * @summary Create Completion + * @param {CreateCompletionRequest} createCompletionRequest + * @param {*} [options] Override http request option. + * @throws {RequiredError} + */ + public createCompletion(createCompletionRequest: CreateCompletionRequest, options?: RawAxiosRequestConfig) { + return CompletionsApiFp(this.configuration).createCompletion(createCompletionRequest, options).then((request) => request(this.axios, this.basePath)); + } +} + diff --git a/packages/arc-api-client/src/models/setting-group.ts b/packages/arc-api-client/src/models/completion-response.ts similarity index 52% rename from packages/arc-api-client/src/models/setting-group.ts rename to packages/arc-api-client/src/models/completion-response.ts index f4a3d254a..9a0dca0fa 100644 --- a/packages/arc-api-client/src/models/setting-group.ts +++ b/packages/arc-api-client/src/models/completion-response.ts @@ -15,27 +15,20 @@ // May contain unused imports in some cases // @ts-ignore -import type { SettingField } from './setting-field'; +import type { CompletionUsage } from './completion-usage'; -/** - * A logical group of related settings. - */ -export interface SettingGroup { - /** - * Machine-readable group identifier. - */ +export interface CompletionResponse { 'id': string; + 'model': string; /** - * Human-readable group name. + * Generated text. */ - 'name': string; + 'content': string; /** - * Prose description of the settings group. + * Why generation stopped (end_turn, max_tokens). */ - 'description': string; - /** - * Settings within this group. - */ - 'fields': Array; + 'stop_reason': string; + 'usage': CompletionUsage; + 'output'?: any; } diff --git a/packages/arc-api-client/src/models/setting-field-type.ts b/packages/arc-api-client/src/models/completion-usage.ts similarity index 58% rename from packages/arc-api-client/src/models/setting-field-type.ts rename to packages/arc-api-client/src/models/completion-usage.ts index d830df867..e3e6ddab1 100644 --- a/packages/arc-api-client/src/models/setting-field-type.ts +++ b/packages/arc-api-client/src/models/completion-usage.ts @@ -14,17 +14,8 @@ -/** - * Input type for a setting field. - */ - -export const SettingFieldType = { - TEXT: 'text', - SELECT: 'select', - TOGGLE: 'toggle' -} as const; - -export type SettingFieldType = typeof SettingFieldType[keyof typeof SettingFieldType]; - - +export interface CompletionUsage { + 'input_tokens': number; + 'output_tokens': number; +} diff --git a/packages/arc-api-client/src/models/create-completion-request.ts b/packages/arc-api-client/src/models/create-completion-request.ts new file mode 100644 index 000000000..b0bd6c76b --- /dev/null +++ b/packages/arc-api-client/src/models/create-completion-request.ts @@ -0,0 +1,39 @@ +/* tslint:disable */ +/* eslint-disable */ +/** + * Arc Run API + * HTTP API for managing Arc workflow run executions. + * + * The version of the OpenAPI document: 0.1.0 + * + * + * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). + * https://openapi-generator.tech + * Do not edit the class manually. + */ + + + +export interface CreateCompletionRequest { + /** + * The user prompt text. + */ + 'prompt': string; + /** + * Model ID or alias. Server picks default if omitted. + */ + 'model'?: string; + /** + * System prompt. + */ + 'system'?: string; + /** + * Stream response via SSE. + */ + 'stream'?: boolean; + 'schema'?: any; + 'temperature'?: number; + 'max_tokens'?: number; + 'top_p'?: number; +} + diff --git a/packages/arc-api-client/src/models/evaluation-result.ts b/packages/arc-api-client/src/models/evaluation-result.ts deleted file mode 100644 index 495361638..000000000 --- a/packages/arc-api-client/src/models/evaluation-result.ts +++ /dev/null @@ -1,30 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Arc Run API - * HTTP API for managing Arc workflow run executions. - * - * The version of the OpenAPI document: 0.1.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * Outcome of a single verification evaluation. - */ - -export const EvaluationResult = { - PASS: 'pass', - FAIL: 'fail', - SKIP: 'skip' -} as const; - -export type EvaluationResult = typeof EvaluationResult[keyof typeof EvaluationResult]; - - - diff --git a/packages/arc-api-client/src/models/setting-field.ts b/packages/arc-api-client/src/models/setting-field.ts deleted file mode 100644 index 3f33ffb67..000000000 --- a/packages/arc-api-client/src/models/setting-field.ts +++ /dev/null @@ -1,48 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Arc Run API - * HTTP API for managing Arc workflow run executions. - * - * The version of the OpenAPI document: 0.1.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { SettingFieldType } from './setting-field-type'; - -/** - * A single configurable setting within a group. - */ -export interface SettingField { - /** - * Machine-readable setting key. - */ - 'key': string; - /** - * Human-readable label displayed in the UI. - */ - 'label': string; - /** - * Current value of the setting. - */ - 'value': string; - 'type': SettingFieldType; - /** - * Available options for select-type fields. - */ - 'options'?: Array; - /** - * Additional help text for the setting. - */ - 'description'?: string; -} - - - diff --git a/packages/arc-api-client/src/models/verification-status.ts b/packages/arc-api-client/src/models/verification-status.ts deleted file mode 100644 index 7debfc41d..000000000 --- a/packages/arc-api-client/src/models/verification-status.ts +++ /dev/null @@ -1,30 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Arc Run API - * HTTP API for managing Arc workflow run executions. - * - * The version of the OpenAPI document: 0.1.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * Result status of a verification control evaluation. - */ - -export const VerificationStatus = { - PASS: 'pass', - FAIL: 'fail', - NA: 'na' -} as const; - -export type VerificationStatus = typeof VerificationStatus[keyof typeof VerificationStatus]; - - -