diff --git a/crates/llm-client/src/openai.rs b/crates/llm-client/src/openai.rs index c9a655e..6b8ee09 100644 --- a/crates/llm-client/src/openai.rs +++ b/crates/llm-client/src/openai.rs @@ -223,6 +223,26 @@ impl OpenAiClient { warn!(model = %options.model, ?output_tokens, "openai: response truncated (max_tokens reached)"); } + // Reassemble the streamed message for the payload log, so a streamed call + // leaves the same debugging trail as a buffered one — including + // reasoning_content and tool_calls, which previously existed only as + // transient deltas and never appeared in the logged body. Built here, + // before `turn` consumes the accumulators (clones are cheap vs. the round-trip). + let logged_tool_calls: Vec = tool_calls.iter() + .map(|(_idx, (id, name, args))| json!({ + "id": id, + "type": "function", + "function": { "name": name, "arguments": args }, + })) + .collect(); + let mut logged_message = json!({ "role": "assistant", "content": content.clone() }); + if let Some(rc) = &reasoning_content { + logged_message["reasoning_content"] = rc.clone().into(); + } + if !logged_tool_calls.is_empty() { + logged_message["tool_calls"] = Value::Array(logged_tool_calls); + } + let turn = if !tool_calls.is_empty() { let calls = tool_calls .into_values() @@ -242,7 +262,7 @@ impl OpenAiClient { // streamed call leaves the same debugging trail as a buffered one. let response_body = json!({ "streamed": true, - "choices": [{ "finish_reason": finish }], + "choices": [{ "finish_reason": finish, "message": logged_message }], "usage": usage, }); let raw_meta = LlmRawMeta { diff --git a/crates/skald-core/src/llm/providers/mod.rs b/crates/skald-core/src/llm/providers/mod.rs index c9d0829..d6b6b3a 100644 --- a/crates/skald-core/src/llm/providers/mod.rs +++ b/crates/skald-core/src/llm/providers/mod.rs @@ -3,6 +3,7 @@ pub mod declared; pub mod ollama; pub mod openai; pub mod openrouter; +pub mod requesty; // Re-export so existing code that uses `providers::ServiceType` / `providers::RemoteLlmModelInfo` keeps working. pub use crate::provider::ServiceType; diff --git a/crates/skald-core/src/llm/providers/requesty.rs b/crates/skald-core/src/llm/providers/requesty.rs new file mode 100644 index 0000000..601d06f --- /dev/null +++ b/crates/skald-core/src/llm/providers/requesty.rs @@ -0,0 +1,197 @@ +use anyhow::{Result, anyhow}; + +use crate::llm::providers::{RemoteLlmModelInfo, build_openai_llm, fetch_openai_models}; +use crate::llm::{LlmModelRecord, LlmProviderRecord}; +use crate::provider::{ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ReasoningMode, ServiceType}; + +/// Requesty router base URL. +const BASE_URL: &str = "https://router.requesty.ai/v1"; + +/// Reasoning effort values Requesty passes through to supporting models +/// (OpenAI o-series, Anthropic extended thinking, DeepSeek-R, etc.). +const REASONING_EFFORTS: &[&str] = &["minimal", "low", "medium", "high"]; + +pub struct RequestyProvider { + /// Lazy: a `reqwest::Client` can only be built after the process installs + /// a crypto provider (done by the shell at startup), so we defer until + /// first use — same pattern as `DeclaredProvider`. + http: std::sync::OnceLock, +} + +impl RequestyProvider { + pub fn new() -> Self { + Self { http: std::sync::OnceLock::new() } + } + + fn http(&self) -> &reqwest::Client { + self.http.get_or_init(reqwest::Client::new) + } + + async fn fetch_catalog(&self, api_key: &str) -> Result> { + let raw = fetch_openai_models(self.http(), BASE_URL, Some(api_key), "Requesty").await?; + Ok(raw.iter().filter_map(map_model).collect()) + } +} + +/// Maps one raw JSON model object from Requesty's `GET /v1/models` response +/// to `RemoteLlmModelInfo`. The endpoint returns flat fields (not nested +/// objects): `context_window`, `max_output_tokens`, `input_price`, +/// `output_price` (all per-token USD), and `supports_*` booleans. +fn map_model(m: &serde_json::Value) -> Option { + let id = m["id"].as_str()?.to_string(); + + let context_length = m["context_window"].as_u64(); + let max_completion_tokens = m["max_output_tokens"].as_u64(); + + // Prices are per-token USD → convert to per-million. + let price_input_per_million = m["input_price"].as_f64().map(|v| v * 1_000_000.0); + let price_output_per_million = m["output_price"].as_f64().map(|v| v * 1_000_000.0); + + let vision = m["supports_vision"].as_bool().unwrap_or(false); + let reasoning = m["supports_reasoning"].as_bool().unwrap_or(false); + + let mut capabilities = vec!["function_calling".to_string()]; + if vision { capabilities.push("vision".to_string()); } + if reasoning { capabilities.push("reasoning".to_string()); } + capabilities.sort(); + capabilities.dedup(); + + Some(RemoteLlmModelInfo { + name: id.clone(), + id, + context_length, + max_completion_tokens, + knowledge_cutoff: None, + capabilities, + vision: Some(vision), + price_input_per_million, + price_output_per_million, + reasoning: None, + }) +} + +#[async_trait::async_trait] +impl ApiProvider for RequestyProvider { + fn type_id(&self) -> &'static str { "requesty" } + fn display_name(&self) -> &'static str { "Requesty" } + fn supported_types(&self) -> &'static [ServiceType] { + &[ServiceType::Llm] + } + + async fn list_llm_models(&self, record: &LlmProviderRecord) -> Result>> { + let api_key = record.api_key.as_deref() + .ok_or_else(|| anyhow!("provider '{}': api_key required for requesty model listing", record.name))?; + Ok(Some(self.fetch_catalog(api_key).await?)) + } + + fn reasoning_mode(&self, _model_id: &str, capabilities: &[String]) -> Option { + if capabilities.iter().any(|c| c == "reasoning") { + Some(ReasoningMode::ValueSet { + values: REASONING_EFFORTS.iter().map(|s| s.to_string()).collect(), + default: Some("medium".to_string()), + }) + } else { + None + } + } + + fn reasoning_request(&self, value: &serde_json::Value) -> Option { + value.as_str().map(|s| serde_json::json!({ "reasoning_effort": s })) + } + + fn build_llm(&self, record: &LlmProviderRecord, model: &LlmModelRecord) -> Option> { + Some(build_openai_llm(self, BASE_URL, record, model, false)) + } + + fn ui_meta(&self) -> ProviderUiMeta { + ProviderUiMeta { + type_id: "requesty", + display_name: "Requesty", + description: Some("Requesty AI gateway — 300+ models from OpenAI, Anthropic, Google and more"), + color: "#10b981", + icon: "bi-shuffle", + lists_models: true, + fields: &[ + ProviderField { key: "api_key", label: "API Key", required: true, secret: true }, + ], + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn map_full_model() { + let m = serde_json::json!({ + "id": "anthropic/claude-opus-4-8", + "object": "model", + "created": 1715367049, + "owned_by": "anthropic", + "context_window": 1048576, + "max_output_tokens": 128000, + "input_price": 0.000015, + "output_price": 0.000075, + "supports_vision": true, + "supports_reasoning": true + }); + let info = map_model(&m).unwrap(); + assert_eq!(info.id, "anthropic/claude-opus-4-8"); + assert_eq!(info.context_length, Some(1_048_576)); + assert_eq!(info.max_completion_tokens, Some(128_000)); + assert_eq!(info.price_input_per_million, Some(15.0)); + assert_eq!(info.price_output_per_million, Some(75.0)); + assert_eq!(info.vision, Some(true)); + assert!(info.capabilities.contains(&"vision".to_string())); + assert!(info.capabilities.contains(&"reasoning".to_string())); + assert!(info.capabilities.contains(&"function_calling".to_string())); + } + + #[test] + fn map_bare_model() { + // Some models may omit the enriched fields entirely. + let m = serde_json::json!({ + "id": "experimental/model-x", + "object": "model", + "owned_by": "test" + }); + let info = map_model(&m).unwrap(); + assert_eq!(info.id, "experimental/model-x"); + assert_eq!(info.context_length, None); + assert_eq!(info.max_completion_tokens, None); + assert_eq!(info.price_input_per_million, None); + assert_eq!(info.vision, Some(false)); + assert!(!info.capabilities.contains(&"vision".to_string())); + } + + #[test] + fn map_non_vision_non_reasoning() { + let m = serde_json::json!({ + "id": "deepseek/deepseek-chat", + "context_window": 64000, + "supports_vision": false, + "supports_reasoning": false + }); + let info = map_model(&m).unwrap(); + assert_eq!(info.context_length, Some(64_000)); + assert_eq!(info.vision, Some(false)); + assert!(!info.capabilities.contains(&"reasoning".to_string())); + } + + #[test] + fn reasoning_request_maps_effort() { + let p = RequestyProvider::new(); + let req = |v: &str| p.reasoning_request(&serde_json::json!(v)).unwrap(); + assert_eq!(req("high"), serde_json::json!({ "reasoning_effort": "high" })); + assert_eq!(req("minimal"), serde_json::json!({ "reasoning_effort": "minimal" })); + assert!(p.reasoning_request(&serde_json::json!(42)).is_none()); + } + + #[test] + fn reasoning_mode_from_capabilities() { + let p = RequestyProvider::new(); + assert!(p.reasoning_mode("any", &["reasoning".to_string()]).is_some()); + assert!(p.reasoning_mode("any", &[]).is_none()); + } +} diff --git a/crates/skald-core/src/session/handler/message_builder.rs b/crates/skald-core/src/session/handler/message_builder.rs index 1952a28..6e4c78c 100644 --- a/crates/skald-core/src/session/handler/message_builder.rs +++ b/crates/skald-core/src/session/handler/message_builder.rs @@ -17,6 +17,14 @@ use crate::tools::tool_names as tn; /// that have `inject_skills` enabled (the default). const SKILLS_INDEX_PATH: &str = "skills/index.md"; +/// Stand-in for a tool-call turn's `reasoning_content` when none was recorded. +/// DeepSeek's thinking mode 400s if an assistant turn that made tool calls is +/// replayed with an absent or empty `reasoning_content` (it "must be passed back"), +/// and a bare tool call sometimes arrives with no reasoning at all — so the field +/// must always be present and non-empty. Neutral text: it stands in for the model's +/// own prior chain-of-thought. +const REASONING_ROUNDTRIP_PLACEHOLDER: &str = "(no reasoning recorded for this step)"; + /// OS description (type + version), computed once — it does not change at runtime. fn os_description() -> &'static str { static OS: std::sync::OnceLock = std::sync::OnceLock::new(); @@ -298,11 +306,14 @@ impl MessageBuilder { if tool_calls.is_empty() { let mut msg = json!({ "role": "assistant", "content": entry.content }); - if let Some(rc) = &entry.reasoning_content { + // A plain (non-tool) assistant turn does not need its reasoning + // round-tripped; echo it only when we actually have some, and never + // as "" (DeepSeek rejects an empty reasoning_content). + if let Some(rc) = entry.reasoning_content.as_deref().filter(|s| !s.is_empty()) { // Echo under both names: DeepSeek expects "reasoning_content", // MiniMax M3 and others expect "reasoning". - msg["reasoning_content"] = rc.clone().into(); - msg["reasoning"] = rc.clone().into(); + msg["reasoning_content"] = rc.into(); + msg["reasoning"] = rc.into(); } out.push(msg); } else { @@ -323,12 +334,19 @@ impl MessageBuilder { "content": entry.content, "tool_calls": tc_array, }); - if let Some(rc) = &entry.reasoning_content { - // Echo under both names: DeepSeek expects "reasoning_content", - // MiniMax M3 and others expect "reasoning". - msg["reasoning_content"] = rc.clone().into(); - msg["reasoning"] = rc.clone().into(); - } + // DeepSeek thinking mode: an assistant turn that made tool calls + // must carry a NON-EMPTY reasoning_content back on the request that + // continues from its tool result, or the API 400s ("reasoning_content + // in the thinking mode must be passed back"). DeepSeek sometimes + // streams a bare tool call with no reasoning, so the stored value can + // be absent/empty — backfill a placeholder, since both an absent and + // an empty field are rejected on replay. Echoed under both names + // (MiniMax M3 uses "reasoning"); harmless for providers that ignore it. + let rc = entry.reasoning_content.as_deref() + .filter(|s| !s.is_empty()) + .unwrap_or(REASONING_ROUNDTRIP_PLACEHOLDER); + msg["reasoning_content"] = rc.into(); + msg["reasoning"] = rc.into(); out.push(msg); for tc in &tool_calls { diff --git a/crates/skald-core/src/skald/bundles.rs b/crates/skald-core/src/skald/bundles.rs index 47cc11e..7cd13c3 100644 --- a/crates/skald-core/src/skald/bundles.rs +++ b/crates/skald-core/src/skald/bundles.rs @@ -64,6 +64,7 @@ impl Models { provider_registry.register_builtin(crate::llm::providers::openai::OpenAiProvider); provider_registry.register_builtin(crate::llm::providers::anthropic::AnthropicProvider::new()); provider_registry.register_builtin(crate::llm::providers::openrouter::OpenRouterProvider::new()); + provider_registry.register_builtin(crate::llm::providers::requesty::RequestyProvider::new()); provider_registry.register_builtin(crate::llm::providers::ollama::OllamaProvider::new()); // OpenAI-compatible providers are runtime data (providers.yaml), not code. for p in crate::llm::providers::declared::load(std::path::Path::new(