From 4fea04c57f26a765e2f50ee60740f1998232a371 Mon Sep 17 00:00:00 2001 From: xavix-yo Date: Sat, 18 Jul 2026 16:19:05 +0100 Subject: [PATCH] feat(providers): OpenAI-compatible provider types as runtime data (providers.yaml) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace the five copy-paste OpenAI-compatible provider structs (moonshot, moonshot_code, deepseek, zai, lm_studio) with one DeclaredProvider engine driven by a providers.yaml catalog loaded at boot from the cwd — edit and restart, no rebuild. The YAML carries identity, endpoints, per-model JSON field mapping, id-glob enrichment rules and the reasoning knob (effort / thinking request kinds); capability_flags keep vision and future input modalities declarative. Anthropic, Ollama, OpenAI and OpenRouter stay native (different wire protocols or bespoke parsing) and register alongside; colliding declared ids are skipped. The shipped catalog is validated by a unit test. --- CLAUDE.md | 6 +- Cargo.lock | 1 + crates/skald-core/Cargo.toml | 1 + .../skald-core/src/llm/providers/declared.rs | 819 ++++++++++++++++++ .../skald-core/src/llm/providers/deepseek.rs | 123 --- .../skald-core/src/llm/providers/lm_studio.rs | 77 -- crates/skald-core/src/llm/providers/mod.rs | 5 +- .../skald-core/src/llm/providers/moonshot.rs | 197 ----- crates/skald-core/src/llm/providers/zai.rs | 135 --- crates/skald-core/src/skald/bundles.rs | 16 +- providers.yaml | 168 ++++ 11 files changed, 1005 insertions(+), 543 deletions(-) create mode 100644 crates/skald-core/src/llm/providers/declared.rs delete mode 100644 crates/skald-core/src/llm/providers/deepseek.rs delete mode 100644 crates/skald-core/src/llm/providers/lm_studio.rs delete mode 100644 crates/skald-core/src/llm/providers/moonshot.rs delete mode 100644 crates/skald-core/src/llm/providers/zai.rs create mode 100644 providers.yaml diff --git a/CLAUDE.md b/CLAUDE.md index 0aa6062..90655d6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -85,7 +85,7 @@ Two rules keep the boundary real, and both are enforced by the compiler: | `crates/skald-core/src/clarification/` | `ClarificationManager`: background-session question/answer | | `crates/skald-core/src/elicitation/` | `ElicitationManager` + bridge: MCP server-initiated input (`elicitation/create`), surfaced in the Inbox; secrets never logged/persisted | | `crates/skald-core/src/inbox.rs` | `Inbox`: unified façade for pending approvals + clarifications + elicitations (wraps ApprovalManager, ClarificationManager, ElicitationManager) | -| `crates/skald-core/src/llm/` | LLM client abstraction (OpenAI-compat, Anthropic, Ollama…) | +| `crates/skald-core/src/llm/` | LLM client abstraction (OpenAI-compat, Anthropic, Ollama…). OpenAI-compatible provider *types* are runtime data, not code: `providers/declared.rs` loads `providers.yaml` at boot (see Config); only non-OpenAI-compatible or bespoke providers (anthropic, ollama, openai, openrouter) stay native | | `crates/skald-core/src/transcribe/` | Transcription providers | | `crates/skald-core/src/image_generate/` | Image generation providers | | `crates/skald-core/src/memory/` | Agent memory tools | @@ -194,7 +194,7 @@ Resolution is **source-agnostic**: the WS + Inbox paths resolve by `request_id`; - **Headless** (default): no handler installed, so `restart` calls `libc::_exit(-1)` (= exit code 255); `run.sh` re-executes the same binary *by path*. - **Desktop** (`--features desktop`): the Tauri shell installs a handler via `tools::restart::set_restart_handler` — cleanup + respawn of the bundled binary + `exit(0)`. The core does not know Tauri exists. -Use it to pick up `config.yml` / database changes, which are only read at startup. To load new **code**: `./build.sh`, then restart — the supervisor picks up the new binary on the next loop, since `build.sh` installs it with an atomic rename. +Use it to pick up `config.yml` / `providers.yaml` / database changes, which are only read at startup. To load new **code**: `./build.sh`, then restart — the supervisor picks up the new binary on the next loop, since `build.sh` installs it with an atomic rename. > `run.bat` is still stale (`cargo run`) and must be fixed. @@ -235,6 +235,8 @@ The `docs/` directory is **ignored** for now — do not read it, reference it, o Copy `default.config.yaml` → `config.yml`. Never commit `config.yml` (contains API keys). +`providers.yaml` (repo root, cwd-relative like `config.yml`) declares the **OpenAI-compatible LLM provider types** — endpoints, UI metadata, per-model JSON field mapping, id-glob enrichment rules, reasoning knobs. Loaded at boot by `llm::providers::declared`; edit + `restart`, no rebuild. An invalid entry is logged and skipped, never fatal; an `id` colliding with a native provider is skipped. Adding a new OpenAI-compatible provider is a YAML edit, not a Rust file. The shipped file is validated by a unit test (`declared::tests::shipped_providers_yaml_is_valid`). + ## Python environment All Python scripts (MCP servers, setup scripts) use a local virtualenv at `.venv/` in the project root. diff --git a/Cargo.lock b/Cargo.lock index 4267cc3..6e8512c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5530,6 +5530,7 @@ dependencies = [ "reqwest 0.13.4", "serde", "serde_json", + "serde_yaml", "sha2 0.10.9", "sqlx", "subtle", diff --git a/crates/skald-core/Cargo.toml b/crates/skald-core/Cargo.toml index ab00853..0777bba 100644 --- a/crates/skald-core/Cargo.toml +++ b/crates/skald-core/Cargo.toml @@ -42,6 +42,7 @@ uuid = { version = "1", features = ["v4"] } reqwest = { version = "0.13.4", default-features = false, features = ["rustls-no-provider", "charset", "http2", "system-proxy", "json", "multipart"] } async-trait = "0.1" serde_json = "1" +serde_yaml = "0.9" indexmap = { version = "2", features = ["serde"] } tracing = "0.1" chrono = { version = "0.4", default-features = false, features = ["clock", "std"] } diff --git a/crates/skald-core/src/llm/providers/declared.rs b/crates/skald-core/src/llm/providers/declared.rs new file mode 100644 index 0000000..c11388f --- /dev/null +++ b/crates/skald-core/src/llm/providers/declared.rs @@ -0,0 +1,819 @@ +//! Declarative OpenAI-compatible LLM providers, loaded at boot from a runtime +//! YAML file ([`PROVIDERS_FILE`], resolved against the process working +//! directory — the same convention as `config.yml`). Editing the file and +//! restarting picks up changes without a rebuild; a provider entry that fails +//! validation is logged and skipped, never fatal. +//! +//! One engine, [`DeclaredProvider`], implements [`ApiProvider`] for every +//! entry: the YAML carries identity, endpoints, per-model JSON field mapping, +//! id-prefix enrichment rules and the reasoning knob, while providers whose +//! behavior is not OpenAI-compatible (Anthropic, Ollama) or too bespoke +//! (OpenRouter's pricing/reasoning parsing, OpenAI's extra TTS/transcribe +//! services) stay native Rust and register alongside. + +use std::collections::HashMap; +use std::path::Path; +use std::sync::Arc; + +use anyhow::{anyhow, Context, Result}; +use tracing::{info, warn}; + +use crate::chatbot::openai::OpenAiClient; +use crate::llm::providers::{extra_with_reasoning, RemoteLlmModelInfo}; +use crate::llm::{LlmModelRecord, LlmProviderRecord}; +use crate::provider::{ + ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ReasoningMode, ServiceType, +}; + +/// Runtime catalog of declarative providers, relative to the process cwd. +pub const PROVIDERS_FILE: &str = "providers.yaml"; + +const LLM_ONLY: &[ServiceType] = &[ServiceType::Llm]; + +// ── YAML spec ──────────────────────────────────────────────────────────────── + +#[derive(Debug, serde::Deserialize)] +struct ProvidersFile { + #[serde(default)] + providers: Vec, +} + +#[derive(Debug, serde::Deserialize)] +struct ProviderSpec { + id: String, + name: String, + /// Only `openai_compatible` exists today; the key is accepted (and + /// validated) so the file format can grow other kinds later. + kind: Option, + base_url: String, + /// When true, the instance-level `base_url` stored in the DB overrides + /// `base_url` (local providers like LM Studio). + #[serde(default)] + base_url_overridable: bool, + #[serde(default)] + api_key: ApiKeySpec, + #[serde(default)] + prompt_cache: bool, + ui: UiSpec, + #[serde(default)] + fields: Vec, + models: Option, + reasoning: Option, +} + +#[derive(Debug, Default, serde::Deserialize)] +#[serde(rename_all = "snake_case")] +enum ApiKeySpec { + #[default] + Required, + Optional, + None, +} + +#[derive(Debug, serde::Deserialize)] +struct UiSpec { + color: String, + icon: String, + description: Option, +} + +#[derive(Debug, serde::Deserialize)] +struct FieldSpec { + key: String, + label: String, + #[serde(default)] + required: bool, + #[serde(default)] + secret: bool, +} + +#[derive(Debug, Default, serde::Deserialize)] +struct ModelsSpec { + /// GET path joined to `models_url`; the response must be the OpenAI + /// `{ "data": [...] }` envelope. Mutually exclusive with `static`. + endpoint: Option, + /// Base for the listing endpoint; defaults to the resolved `base_url`. + models_url: Option, + /// Defaults to `bearer` unless `api_key: none`. + auth: Option, + /// Static model-id catalog (provider exposes no listing endpoint). + #[serde(rename = "static")] + static_models: Option>, + #[serde(default)] + map: MapSpec, + #[serde(default)] + base_capabilities: Vec, + #[serde(default)] + defaults: DefaultsSpec, + /// First-matching rule wins. + #[serde(default)] + enrich: Vec, +} + +#[derive(Debug, serde::Deserialize)] +#[serde(rename_all = "snake_case")] +enum AuthSpec { + Bearer, + None, +} + +/// Per-model JSON field names → `RemoteLlmModelInfo` fields. Absent mappings +/// leave the corresponding field `None` (id defaults to `"id"`, name to id). +#[derive(Debug, Default, serde::Deserialize)] +struct MapSpec { + id: Option, + name: Option, + context_length: Option, + max_completion_tokens: Option, + knowledge_cutoff: Option, + /// Boolean JSON field; when true it also adds the `vision` capability. + vision: Option, + price_input_per_million: Option, + price_output_per_million: Option, + /// capability name → boolean JSON field that enables it. + #[serde(default)] + capability_flags: HashMap, +} + +#[derive(Debug, Default, serde::Deserialize)] +struct DefaultsSpec { + vision: Option, +} + +#[derive(Debug, serde::Deserialize)] +struct EnrichRule { + #[serde(rename = "match")] + glob: String, + #[serde(default)] + mode: EnrichMode, + context_length: Option, + max_completion_tokens: Option, + vision: Option, + #[serde(default)] + add_capabilities: Vec, +} + +#[derive(Debug, Default, serde::Deserialize)] +#[serde(rename_all = "snake_case")] +enum EnrichMode { + /// Only fill fields the endpoint/static list left unset (endpoint wins). + #[default] + Fill, + /// Overwrite whatever the endpoint returned (rule wins). + Override, +} + +#[derive(Debug, serde::Deserialize)] +struct ReasoningSpec { + request: ReasoningRequestSpec, + /// First-matching rule wins. + #[serde(default)] + modes: Vec, +} + +#[derive(Debug, serde::Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case")] +enum ReasoningRequestSpec { + /// `{"reasoning_effort": value}`, with optional value remapping + /// (e.g. `disabled` → `none`). + Effort { + #[serde(default)] + remap: HashMap, + }, + /// `disabled`/`enabled` toggle via `{"thinking": {...}}`; any other value + /// also carries `reasoning_effort`. + Thinking, +} + +#[derive(Debug, serde::Deserialize)] +struct ReasoningModeRule { + #[serde(default)] + when: WhenSpec, + values: Vec, + default: Option, +} + +/// Selector for a reasoning mode: with both keys present the model matches on +/// either (OR); with neither, it matches every model. +#[derive(Debug, Default, serde::Deserialize)] +struct WhenSpec { + models: Option>, + capability: Option, +} + +impl WhenSpec { + fn matches(&self, model_id: &str, capabilities: &[String]) -> bool { + let id_ok = self + .models + .as_ref() + .is_some_and(|gs| gs.iter().any(|g| glob_match(g, model_id))); + let cap_ok = self + .capability + .as_ref() + .is_some_and(|c| capabilities.iter().any(|x| x == c)); + match (self.models.is_some(), self.capability.is_some()) { + (false, false) => true, + _ => id_ok || cap_ok, + } + } +} + +// ── DeclaredProvider ───────────────────────────────────────────────────────── + +/// [`ApiProvider`] strings are `&'static str`, but the spec is runtime data: +/// the handful of strings a provider exposes are leaked once at boot. The set +/// of declared providers is fixed for the process lifetime and tiny, so the +/// leak is bounded and intentional. +struct LeakedMeta { + id: &'static str, + name: &'static str, + description: Option<&'static str>, + color: &'static str, + icon: &'static str, + fields: &'static [ProviderField], +} + +fn leak_str(s: &str) -> &'static str { + Box::leak(s.to_string().into_boxed_str()) +} + +pub struct DeclaredProvider { + spec: ProviderSpec, + meta: LeakedMeta, + /// Built on first use: with the `rustls-no-provider` reqwest feature a + /// `Client` can only be constructed after the process installs a crypto + /// provider (done by the shell at startup), so building one per provider + /// at registration time would panic outside the server binary (tests). + http: std::sync::OnceLock, +} + +impl DeclaredProvider { + fn new(spec: ProviderSpec) -> Self { + let fields: Vec = spec + .fields + .iter() + .map(|f| ProviderField { + key: leak_str(&f.key), + label: leak_str(&f.label), + required: f.required, + secret: f.secret, + }) + .collect(); + let meta = LeakedMeta { + id: leak_str(&spec.id), + name: leak_str(&spec.name), + description: spec.ui.description.as_deref().map(leak_str), + color: leak_str(&spec.ui.color), + icon: leak_str(&spec.ui.icon), + fields: Box::leak(fields.into_boxed_slice()), + }; + Self { spec, meta, http: std::sync::OnceLock::new() } + } + + fn http(&self) -> &reqwest::Client { + self.http.get_or_init(reqwest::Client::new) + } + + fn base_url(&self, record: &LlmProviderRecord) -> String { + if self.spec.base_url_overridable { + record + .base_url + .clone() + .unwrap_or_else(|| self.spec.base_url.clone()) + } else { + self.spec.base_url.clone() + } + } + + fn models_url(&self, models: &ModelsSpec, record: &LlmProviderRecord) -> String { + models + .models_url + .clone() + .unwrap_or_else(|| self.base_url(record)) + } + + /// Bearer key for the listing endpoint, or `None` when the entry is + /// unauthenticated. Errors when auth is required but the instance has no key. + fn auth_key(&self, models: &ModelsSpec, record: &LlmProviderRecord) -> Result> { + let bearer = match models.auth { + Some(AuthSpec::Bearer) => true, + Some(AuthSpec::None) => false, + None => !matches!(self.spec.api_key, ApiKeySpec::None), + }; + if !bearer { + return Ok(None); + } + let key = record.api_key.clone().ok_or_else(|| { + anyhow!( + "provider '{}': api_key required for {} model listing", + record.name, + self.meta.name + ) + })?; + Ok(Some(key)) + } + + fn blank_info(&self, id: &str, models: &ModelsSpec) -> RemoteLlmModelInfo { + RemoteLlmModelInfo { + id: id.to_string(), + name: id.to_string(), + context_length: None, + max_completion_tokens: None, + knowledge_cutoff: None, + capabilities: models.base_capabilities.clone(), + vision: models.defaults.vision, + price_input_per_million: None, + price_output_per_million: None, + reasoning: None, + } + } + + fn map_model(&self, m: &serde_json::Value, models: &ModelsSpec) -> Option { + let map = &models.map; + let get = |f: &Option| f.as_deref().map(|k| &m[k]); + let id = get(&map.id) + .or_else(|| Some(&m["id"])) + .and_then(|v| v.as_str())? + .to_string(); + let name = get(&map.name) + .and_then(|v| v.as_str()) + .unwrap_or(&id) + .to_string(); + let mut vision = get(&map.vision).and_then(|v| v.as_bool()); + if vision.is_none() { + vision = models.defaults.vision; + } + let mut capabilities = models.base_capabilities.clone(); + let mut add_cap = |cap: &str| { + if !capabilities.iter().any(|c| c == cap) { + capabilities.push(cap.to_string()); + } + }; + if vision == Some(true) { + add_cap("vision"); + } + for (cap, field) in &map.capability_flags { + if m[field].as_bool().unwrap_or(false) { + add_cap(cap); + } + } + Some(RemoteLlmModelInfo { + id, + name, + context_length: get(&map.context_length).and_then(|v| v.as_u64()), + max_completion_tokens: get(&map.max_completion_tokens).and_then(|v| v.as_u64()), + knowledge_cutoff: get(&map.knowledge_cutoff) + .and_then(|v| v.as_str()) + .map(String::from), + capabilities, + vision, + price_input_per_million: get(&map.price_input_per_million).and_then(|v| v.as_f64()), + price_output_per_million: get(&map.price_output_per_million).and_then(|v| v.as_f64()), + reasoning: None, + }) + } + + async fn list_models(&self, record: &LlmProviderRecord) -> Result> { + let models = self.spec.models.as_ref().expect("checked by caller"); + let mut list = if let Some(ids) = &models.static_models { + ids.iter().map(|id| self.blank_info(id, models)).collect() + } else { + let endpoint = models.endpoint.as_deref().expect("validated"); + let base = self.models_url(models, record); + let url = format!("{}{}", base.trim_end_matches('/'), endpoint); + let key = self.auth_key(models, record)?; + let mut req = self.http().get(&url); + if let Some(k) = key { + req = req.bearer_auth(k); + } + let who = self.meta.name; + let resp: serde_json::Value = req + .send() + .await + .map_err(|e| anyhow!("{who} request failed: {e}"))? + .error_for_status() + .map_err(|e| anyhow!("{who} error response: {e}"))? + .json() + .await + .map_err(|e| anyhow!("{who} response parse failed: {e}"))?; + let raw = resp["data"] + .as_array() + .cloned() + .ok_or_else(|| anyhow!("unexpected {who} response shape"))?; + raw.iter().filter_map(|m| self.map_model(m, models)).collect() + }; + for info in &mut list { + apply_enrich(&models.enrich, info); + } + Ok(list) + } +} + +/// Applies the first matching enrich rule (later rules are not consulted). +fn apply_enrich(rules: &[EnrichRule], info: &mut RemoteLlmModelInfo) { + let Some(rule) = rules.iter().find(|r| glob_match(&r.glob, &info.id)) else { + return; + }; + let set = |dst: &mut Option, v: Option| { + if let Some(v) = v { + match rule.mode { + EnrichMode::Fill => { + if dst.is_none() { + *dst = Some(v); + } + } + EnrichMode::Override => *dst = Some(v), + } + } + }; + set(&mut info.context_length, rule.context_length); + set(&mut info.max_completion_tokens, rule.max_completion_tokens); + if let Some(v) = rule.vision { + match rule.mode { + EnrichMode::Fill => { + if info.vision.is_none() { + info.vision = Some(v); + } + } + EnrichMode::Override => info.vision = Some(v), + } + } + for cap in &rule.add_capabilities { + if !info.capabilities.iter().any(|c| c == cap) { + info.capabilities.push(cap.clone()); + } + } +} + +/// Case-insensitive glob where `*` is the only wildcard (any run, empty +/// included): `"k3*"` is a prefix match, `"*reasoner*"` a contains match, and +/// a pattern without `*` is an exact match. +fn glob_match(pattern: &str, text: &str) -> bool { + let p = pattern.to_lowercase(); + let t = text.to_lowercase(); + if !p.contains('*') { + return p == t; + } + let anchored_start = !p.starts_with('*'); + let anchored_end = !p.ends_with('*'); + let parts: Vec<&str> = p.split('*').filter(|s| !s.is_empty()).collect(); + if parts.is_empty() { + return true; + } + let mut rest = t.as_str(); + for (i, part) in parts.iter().enumerate() { + if i == 0 && anchored_start { + if !rest.starts_with(part) { + return false; + } + rest = &rest[part.len()..]; + continue; + } + match rest.find(part) { + Some(pos) => rest = &rest[pos + part.len()..], + None => return false, + } + } + if anchored_end { + return t.ends_with(parts[parts.len() - 1]); + } + true +} + +#[async_trait::async_trait] +impl ApiProvider for DeclaredProvider { + fn type_id(&self) -> &'static str { + self.meta.id + } + + fn display_name(&self) -> &'static str { + self.meta.name + } + + fn supported_types(&self) -> &'static [ServiceType] { + LLM_ONLY + } + + async fn list_llm_models( + &self, + record: &LlmProviderRecord, + ) -> Result>> { + if self.spec.models.is_none() { + return Ok(None); + } + Ok(Some(self.list_models(record).await?)) + } + + fn reasoning_mode(&self, model_id: &str, capabilities: &[String]) -> Option { + let spec = self.spec.reasoning.as_ref()?; + let rule = spec + .modes + .iter() + .find(|r| r.when.matches(model_id, capabilities))?; + Some(ReasoningMode::ValueSet { + values: rule.values.clone(), + default: rule.default.clone(), + }) + } + + fn reasoning_request(&self, value: &serde_json::Value) -> Option { + let spec = self.spec.reasoning.as_ref()?; + let s = value.as_str()?; + match &spec.request { + ReasoningRequestSpec::Effort { remap } => { + let v = remap.get(s).map(String::as_str).unwrap_or(s); + Some(serde_json::json!({ "reasoning_effort": v })) + } + ReasoningRequestSpec::Thinking => match s { + "disabled" => Some(serde_json::json!({ "thinking": { "type": "disabled" } })), + "enabled" => Some(serde_json::json!({ "thinking": { "type": "enabled" } })), + effort => Some(serde_json::json!({ + "thinking": { "type": "enabled" }, + "reasoning_effort": effort, + })), + }, + } + } + + fn build_llm( + &self, + record: &LlmProviderRecord, + model: &LlmModelRecord, + ) -> Option> { + Some((|| { + let key = match self.spec.api_key { + ApiKeySpec::Required => record.api_key.clone().with_context(|| { + format!( + "provider '{}': api_key required for {}", + record.name, self.meta.id + ) + })?, + ApiKeySpec::Optional => record.api_key.clone().unwrap_or_default(), + ApiKeySpec::None => String::new(), + }; + let extra = extra_with_reasoning(self, model); + let prompt_cache = self.spec.prompt_cache; + Ok(BuiltLlmClient { + client: Arc::new(OpenAiClient::new(self.base_url(record), key, extra, prompt_cache)), + prompt_cache, + }) + })()) + } + + fn ui_meta(&self) -> ProviderUiMeta { + ProviderUiMeta { + type_id: self.meta.id, + display_name: self.meta.name, + description: self.meta.description, + color: self.meta.color, + icon: self.meta.icon, + lists_models: self.spec.models.is_some(), + fields: self.meta.fields, + } + } +} + +// ── Loader ─────────────────────────────────────────────────────────────────── + +/// Loads every valid entry from `path`. A missing file, an unreadable file or +/// a malformed entry is logged and skipped — the native providers always +/// register regardless. +pub fn load(path: &Path) -> Vec { + let text = match std::fs::read_to_string(path) { + Ok(t) => t, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + info!(path = %path.display(), "no declarative providers file — skipped"); + return Vec::new(); + } + Err(e) => { + warn!(path = %path.display(), error = %e, "cannot read declarative providers file — skipped"); + return Vec::new(); + } + }; + let file: ProvidersFile = match serde_yaml::from_str(&text) { + Ok(f) => f, + Err(e) => { + warn!(path = %path.display(), error = %e, "invalid declarative providers file — skipped"); + return Vec::new(); + } + }; + let mut seen = std::collections::HashSet::new(); + let mut out = Vec::new(); + for (i, entry) in file.providers.into_iter().enumerate() { + match parse_entry(entry) { + Ok(p) => { + if !seen.insert(p.meta.id) { + warn!(type_id = p.meta.id, "duplicate provider id in declarative file — skipped"); + continue; + } + out.push(p); + } + Err(e) => { + warn!(path = %path.display(), entry = i, error = %e, "invalid provider entry — skipped"); + } + } + } + info!(path = %path.display(), count = out.len(), "declarative providers loaded"); + out +} + +fn parse_entry(v: serde_yaml::Value) -> Result { + let spec: ProviderSpec = serde_yaml::from_value(v)?; + validate(&spec)?; + Ok(DeclaredProvider::new(spec)) +} + +fn validate(spec: &ProviderSpec) -> Result<()> { + if spec.id.trim().is_empty() { + return Err(anyhow!("provider id must not be empty")); + } + if let Some(kind) = &spec.kind + && kind != "openai_compatible" + { + return Err(anyhow!( + "provider '{}': unsupported kind '{kind}' (only 'openai_compatible')", + spec.id + )); + } + if spec.base_url.trim().is_empty() { + return Err(anyhow!("provider '{}': base_url must not be empty", spec.id)); + } + if let Some(m) = &spec.models { + match (m.endpoint.is_some(), m.static_models.is_some()) { + (true, true) => { + return Err(anyhow!( + "provider '{}': models.endpoint and models.static are mutually exclusive", + spec.id + )); + } + (false, false) => { + return Err(anyhow!( + "provider '{}': models needs either endpoint or static", + spec.id + )); + } + _ => {} + } + } + Ok(()) +} + +// ── Tests ──────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn glob_prefix_contains_exact() { + assert!(glob_match("k3*", "K3-0718")); + assert!(glob_match("*coder*", "deepseek-Coder-v2")); + assert!(glob_match("*128k*", "glm-4-32b-0414-128k")); + assert!(glob_match("glm-5", "glm-5")); + assert!(!glob_match("glm-5", "glm-5-turbo")); + assert!(glob_match("a*b", "axbxb")); + assert!(!glob_match("a*b", "abx")); + assert!(glob_match("*b", "ab")); + assert!(!glob_match("k3*", "kimi-for-coding")); + } + + fn provider(yaml: &str) -> DeclaredProvider { + parse_entry(serde_yaml::from_str(yaml).unwrap()).unwrap() + } + + #[test] + fn thinking_request_mapping() { + let p = provider( + r#" + id: t + name: T + base_url: http://x + ui: { color: c, icon: i } + reasoning: + request: { kind: thinking } + modes: + - { values: [disabled, enabled], default: enabled } + "#, + ); + let req = |v: &str| p.reasoning_request(&serde_json::json!(v)).unwrap(); + assert_eq!(req("disabled"), serde_json::json!({ "thinking": { "type": "disabled" } })); + assert_eq!(req("enabled"), serde_json::json!({ "thinking": { "type": "enabled" } })); + assert_eq!( + req("high"), + serde_json::json!({ "thinking": { "type": "enabled" }, "reasoning_effort": "high" }) + ); + } + + #[test] + fn effort_request_remap() { + let p = provider( + r#" + id: t + name: T + base_url: http://x + ui: { color: c, icon: i } + reasoning: + request: { kind: effort, remap: { disabled: none } } + modes: + - { values: [disabled, max], default: max } + "#, + ); + let req = |v: &str| p.reasoning_request(&serde_json::json!(v)).unwrap(); + assert_eq!(req("disabled"), serde_json::json!({ "reasoning_effort": "none" })); + assert_eq!(req("max"), serde_json::json!({ "reasoning_effort": "max" })); + } + + #[test] + fn reasoning_mode_first_match_wins() { + let p = provider( + r#" + id: t + name: T + base_url: http://x + ui: { color: c, icon: i } + reasoning: + request: { kind: thinking } + modes: + - when: { models: ["glm-5.2*"] } + values: [disabled, max] + - when: { models: ["glm-5*"] } + values: [disabled, enabled] + "#, + ); + let mode = |id: &str| p.reasoning_mode(id, &[]); + assert!(matches!( + mode("glm-5.2"), + Some(ReasoningMode::ValueSet { values, .. }) if values == ["disabled", "max"] + )); + assert!(matches!( + mode("glm-5-turbo"), + Some(ReasoningMode::ValueSet { values, .. }) if values == ["disabled", "enabled"] + )); + assert!(mode("glm-4").is_none()); + } + + #[test] + fn reasoning_mode_capability_or_models() { + let p = provider( + r#" + id: t + name: T + base_url: http://x + ui: { color: c, icon: i } + reasoning: + request: { kind: thinking } + modes: + - when: { capability: reasoning, models: ["*reasoner*"] } + values: [disabled, high] + "#, + ); + assert!(p.reasoning_mode("anything", &["reasoning".to_string()]).is_some()); + assert!(p.reasoning_mode("deepseek-reasoner", &[]).is_some()); + assert!(p.reasoning_mode("deepseek-chat", &[]).is_none()); + } + + #[test] + fn enrich_first_match_fill_vs_override() { + let rules: Vec = serde_yaml::from_str( + r#" + - { match: "*coder*", context_length: 16384, mode: override } + - { match: "k3*", context_length: 1048576, vision: true } + "#, + ) + .unwrap(); + let mut info = RemoteLlmModelInfo { + id: "x-coder".into(), + name: "x".into(), + context_length: Some(999), + max_completion_tokens: None, + knowledge_cutoff: None, + capabilities: vec![], + vision: None, + price_input_per_million: None, + price_output_per_million: None, + reasoning: None, + }; + apply_enrich(&rules, &mut info); + assert_eq!(info.context_length, Some(16384)); // override wins over endpoint + + info.id = "k3-pro".into(); + info.context_length = Some(999); + apply_enrich(&rules, &mut info); + assert_eq!(info.context_length, Some(999)); // fill keeps the endpoint value + assert_eq!(info.vision, Some(true)); + } + + /// The catalog shipped at the repository root must always parse: the file + /// is runtime data, but this test keeps a typo from reaching users. + #[test] + fn shipped_providers_yaml_is_valid() { + let text = include_str!("../../../../../providers.yaml"); + let file: ProvidersFile = serde_yaml::from_str(text).unwrap(); + assert!(!file.providers.is_empty()); + let mut ids = std::collections::HashSet::new(); + for entry in file.providers { + let p = parse_entry(entry).unwrap(); + assert!(ids.insert(p.meta.id), "duplicate id {}", p.meta.id); + } + } +} diff --git a/crates/skald-core/src/llm/providers/deepseek.rs b/crates/skald-core/src/llm/providers/deepseek.rs deleted file mode 100644 index 41439d0..0000000 --- a/crates/skald-core/src/llm/providers/deepseek.rs +++ /dev/null @@ -1,123 +0,0 @@ -use anyhow::{Result, anyhow}; - -use crate::llm::{LlmModelRecord, LlmProviderRecord}; -use crate::llm::providers::{RemoteLlmModelInfo, build_openai_llm, fetch_openai_models}; -use crate::provider::{ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ReasoningMode, ServiceType}; - -pub struct DeepSeekProvider { - http: reqwest::Client, -} - -impl DeepSeekProvider { - pub fn new() -> Self { - Self { http: reqwest::Client::new() } - } - - fn known_context_length(model_id: &str) -> Option { - let id = model_id.to_lowercase(); - if id.contains("coder") { Some(16384) } - else if id.contains("reasoner") { Some(65536) } - else if id.starts_with("deepseek-v4") { Some(1_048_576) } - else if id.starts_with("deepseek-chat") || id.starts_with("deepseek-v3") { Some(65536) } - else { None } - } - - fn known_max_output(model_id: &str) -> Option { - if model_id.to_lowercase().starts_with("deepseek-v4") { Some(393_216) } else { None } - } - - fn known_capabilities(model_id: &str) -> Vec { - let mut caps = vec!["function_calling".to_string()]; - if model_id.to_lowercase().contains("reasoner") { - caps.push("reasoning".to_string()); - } - caps - } -} - -#[async_trait::async_trait] -impl ApiProvider for DeepSeekProvider { - fn type_id(&self) -> &'static str { "deepseek" } - fn display_name(&self) -> &'static str { "DeepSeek" } - fn supported_types(&self) -> &'static [ServiceType] { - &[ServiceType::Llm] - } - - async fn list_llm_models(&self, record: &LlmProviderRecord) -> Result>> { - let api_key = record.api_key.as_deref() - .ok_or_else(|| anyhow!("provider '{}': api_key required for deepseek model listing", record.name))?; - - let raw = fetch_openai_models(&self.http, "https://api.deepseek.com", Some(api_key), "DeepSeek").await?; - let models = raw - .iter() - .filter_map(|m| { - let id = m["id"].as_str()?.to_string(); - let name = id.clone(); - let context_length = Self::known_context_length(&id).or_else(|| m["context_length"].as_u64()); - let capabilities = Self::known_capabilities(&id); - let max_output = Self::known_max_output(&id); - Some(RemoteLlmModelInfo { - id, name, context_length, - max_completion_tokens: max_output, - knowledge_cutoff: None, - capabilities, - vision: None, - price_input_per_million: None, - price_output_per_million: None, - reasoning: None, - }) - }) - .collect(); - - Ok(Some(models)) - } - - fn reasoning_mode(&self, model_id: &str, capabilities: &[String]) -> Option { - // Thinking mode (thinking.type) + graded reasoning_effort. "disabled" - // turns thinking off; effort levels low/medium map to high, xhigh to max. - let id = model_id.to_lowercase(); - if capabilities.iter().any(|c| c == "reasoning") - || id.contains("reasoner") - || id.starts_with("deepseek-v4") - { - Some(ReasoningMode::ValueSet { - values: ["disabled", "low", "medium", "high", "xhigh", "max"] - .iter().map(|s| s.to_string()).collect(), - default: Some("high".to_string()), - }) - } else { - None - } - } - - fn reasoning_request(&self, value: &serde_json::Value) -> Option { - // "disabled" → thinking off; "enabled" → thinking on (no effort); - // any effort level → thinking on + `reasoning_effort`. - match value.as_str()? { - "disabled" => Some(serde_json::json!({ "thinking": { "type": "disabled" } })), - "enabled" => Some(serde_json::json!({ "thinking": { "type": "enabled" } })), - effort => Some(serde_json::json!({ - "thinking": { "type": "enabled" }, - "reasoning_effort": effort, - })), - } - } - - fn build_llm(&self, record: &LlmProviderRecord, model: &LlmModelRecord) -> Option> { - Some(build_openai_llm(self, "https://api.deepseek.com/v1", record, model, false)) - } - - fn ui_meta(&self) -> ProviderUiMeta { - ProviderUiMeta { - type_id: "deepseek", - display_name: "DeepSeek", - description: None, - color: "#0ea5e9", - icon: "bi-search", - lists_models: true, - fields: &[ - ProviderField { key: "api_key", label: "API Key", required: true, secret: true }, - ], - } - } -} diff --git a/crates/skald-core/src/llm/providers/lm_studio.rs b/crates/skald-core/src/llm/providers/lm_studio.rs deleted file mode 100644 index 5290d6c..0000000 --- a/crates/skald-core/src/llm/providers/lm_studio.rs +++ /dev/null @@ -1,77 +0,0 @@ -use std::sync::Arc; - -use anyhow::Result; - -use crate::chatbot::lm_studio::LmStudioClient; -use crate::llm::{LlmModelRecord, LlmProviderRecord}; -use crate::llm::providers::{RemoteLlmModelInfo, fetch_openai_models}; -use crate::provider::{ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ServiceType}; - -pub struct LmStudioProvider { - http: reqwest::Client, -} - -impl LmStudioProvider { - pub fn new() -> Self { - Self { http: reqwest::Client::new() } - } - - fn base_url(record: &LlmProviderRecord) -> String { - record.base_url.clone() - .unwrap_or_else(|| "http://localhost:1234/v1".to_string()) - } -} - -#[async_trait::async_trait] -impl ApiProvider for LmStudioProvider { - fn type_id(&self) -> &'static str { "lm_studio" } - fn display_name(&self) -> &'static str { "LM Studio" } - fn supported_types(&self) -> &'static [ServiceType] { - &[ServiceType::Llm] - } - - async fn list_llm_models(&self, record: &LlmProviderRecord) -> Result>> { - let raw = fetch_openai_models(&self.http, &Self::base_url(record), None, "LM Studio").await?; - - let models = raw - .iter() - .filter_map(|m| { - let id = m["id"].as_str()?.to_string(); - Some(RemoteLlmModelInfo { - name: id.clone(), id, - context_length: None, - max_completion_tokens: None, - knowledge_cutoff: None, - capabilities: vec![], - vision: None, - price_input_per_million: None, - price_output_per_million: None, - reasoning: None, - }) - }) - .collect(); - - Ok(Some(models)) - } - - fn build_llm(&self, record: &LlmProviderRecord, _model: &LlmModelRecord) -> Option> { - Some(Ok(BuiltLlmClient { - client: Arc::new(LmStudioClient::new(record.base_url.as_deref())), - prompt_cache: false, - })) - } - - fn ui_meta(&self) -> ProviderUiMeta { - ProviderUiMeta { - type_id: "lm_studio", - display_name: "LM Studio", - description: Some("Local models via LM Studio"), - color: "#6b7280", - icon: "bi-window-stack", - lists_models: true, - fields: &[ - ProviderField { key: "base_url", label: "Base URL", required: false, secret: false }, - ], - } - } -} diff --git a/crates/skald-core/src/llm/providers/mod.rs b/crates/skald-core/src/llm/providers/mod.rs index edae18b..c9d0829 100644 --- a/crates/skald-core/src/llm/providers/mod.rs +++ b/crates/skald-core/src/llm/providers/mod.rs @@ -1,11 +1,8 @@ pub mod anthropic; -pub mod deepseek; -pub mod lm_studio; -pub mod moonshot; +pub mod declared; pub mod ollama; pub mod openai; pub mod openrouter; -pub mod zai; // Re-export so existing code that uses `providers::ServiceType` / `providers::RemoteLlmModelInfo` keeps working. pub use crate::provider::ServiceType; diff --git a/crates/skald-core/src/llm/providers/moonshot.rs b/crates/skald-core/src/llm/providers/moonshot.rs deleted file mode 100644 index be21e8e..0000000 --- a/crates/skald-core/src/llm/providers/moonshot.rs +++ /dev/null @@ -1,197 +0,0 @@ -use anyhow::{Result, anyhow}; - -use crate::llm::{LlmModelRecord, LlmProviderRecord}; -use crate::llm::providers::{RemoteLlmModelInfo, build_openai_llm, fetch_openai_models}; -use crate::provider::{ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ReasoningMode, ServiceType}; - -/// Moonshot AI — pay-as-you-go platform. -/// -/// Endpoint `https://api.moonshot.ai/v1/chat/completions` (OpenAI-compatible); -/// `OpenAiClient` appends `/chat/completions`, so the base URL is `.../v1`. -/// -/// `GET /models` returns the full catalog including `context_length`, -/// `supports_image_in` and `supports_reasoning`, so the model list is entirely -/// endpoint-driven. Thinking models (kimi-k2-thinking…) always reason — the -/// platform exposes no request-level knob, hence no `ReasoningMode`. -pub struct MoonshotProvider { - http: reqwest::Client, -} - -/// Moonshot AI — Kimi Code subscription. -/// -/// Endpoint `https://api.kimi.com/coding/v1/chat/completions` (OpenAI-compatible). -/// The model catalog comes from `GET /models` like the platform; only the -/// metadata the endpoint omits (context size, vision) is filled from the -/// published docs. K3 is the only model with a reasoning knob: a graded -/// `reasoning_effort` (`low`/`high`/`max`, default `max`); the kimi-for-coding -/// series (K2.7 Code) always thinks and has no toggle. -pub struct MoonshotCodeProvider { - http: reqwest::Client, -} - -impl MoonshotProvider { - pub fn new() -> Self { - Self { http: reqwest::Client::new() } - } - - const BASE_URL: &'static str = "https://api.moonshot.ai/v1"; -} - -impl MoonshotCodeProvider { - pub fn new() -> Self { - Self { http: reqwest::Client::new() } - } - - const BASE_URL: &'static str = "https://api.kimi.com/coding/v1"; - - /// Fills the metadata the Kimi Code `/models` endpoint may omit, from the - /// published docs: K3 → up to 1M context + native visual understanding; - /// the kimi-for-coding series → 256k. Values already present in the - /// endpoint response (e.g. a tier-specific context size) always win. - fn enrich(info: &mut RemoteLlmModelInfo) { - let id = info.id.to_lowercase(); - if info.context_length.is_none() { - if id.starts_with("k3") { - info.context_length = Some(1_048_576); - } else if id.starts_with("kimi-for-coding") { - info.context_length = Some(262_144); - } - } - if info.vision.is_none() && id.starts_with("k3") { - info.vision = Some(true); - } - } -} - -/// Shared model listing for both Moonshot APIs: same envelope, same per-model -/// fields (`context_length`, `supports_image_in`, `supports_reasoning` — all -/// optional, absent fields are left for the caller to enrich). -async fn list_models( - http: &reqwest::Client, - base_url: &str, - record: &LlmProviderRecord, - who: &str, -) -> Result> { - let api_key = record.api_key.as_deref() - .ok_or_else(|| anyhow!("provider '{}': api_key required for {who} model listing", record.name))?; - - let raw = fetch_openai_models(http, base_url, Some(api_key), who).await?; - let models = raw - .iter() - .filter_map(|m| { - let id = m["id"].as_str()?.to_string(); - let mut capabilities = vec!["function_calling".to_string()]; - if m["supports_reasoning"].as_bool().unwrap_or(false) { - capabilities.push("reasoning".to_string()); - } - if m["supports_image_in"].as_bool().unwrap_or(false) { - capabilities.push("vision".to_string()); - } - Some(RemoteLlmModelInfo { - id, - name: m["id"].as_str()?.to_string(), - context_length: m["context_length"].as_u64(), - max_completion_tokens: None, - knowledge_cutoff: None, - capabilities, - vision: m["supports_image_in"].as_bool(), - price_input_per_million: None, - price_output_per_million: None, - reasoning: None, - }) - }) - .collect(); - - Ok(models) -} - -#[async_trait::async_trait] -impl ApiProvider for MoonshotProvider { - fn type_id(&self) -> &'static str { "moonshot" } - fn display_name(&self) -> &'static str { "Moonshot AI pay-as-you-go" } - fn supported_types(&self) -> &'static [ServiceType] { - &[ServiceType::Llm] - } - - async fn list_llm_models(&self, record: &LlmProviderRecord) -> Result>> { - Ok(Some(list_models(&self.http, Self::BASE_URL, record, "Moonshot AI").await?)) - } - - fn build_llm(&self, record: &LlmProviderRecord, model: &LlmModelRecord) -> Option> { - Some(build_openai_llm(self, Self::BASE_URL, record, model, false)) - } - - fn ui_meta(&self) -> ProviderUiMeta { - ProviderUiMeta { - type_id: "moonshot", - display_name: "Moonshot AI pay-as-you-go", - description: Some("Kimi models on the Moonshot AI platform (OpenAI-compatible)"), - color: "#2563eb", - icon: "bi-moon-stars", - lists_models: true, - fields: &[ - ProviderField { key: "api_key", label: "API Key", required: true, secret: true }, - ], - } - } -} - -#[async_trait::async_trait] -impl ApiProvider for MoonshotCodeProvider { - fn type_id(&self) -> &'static str { "moonshot_code" } - fn display_name(&self) -> &'static str { "Moonshot AI Kimi Code" } - fn supported_types(&self) -> &'static [ServiceType] { - &[ServiceType::Llm] - } - - async fn list_llm_models(&self, record: &LlmProviderRecord) -> Result>> { - let mut models = list_models(&self.http, Self::BASE_URL, record, "Kimi Code").await?; - for m in &mut models { - Self::enrich(m); - } - Ok(Some(models)) - } - - fn reasoning_mode(&self, model_id: &str, _capabilities: &[String]) -> Option { - let id = model_id.to_lowercase(); - // K3 exposes a graded `reasoning_effort` (low/high/max; default max). - // "disabled" turns thinking off — the API then routes to K2.6. - // kimi-for-coding (K2.7 Code) always thinks and has no knob. - if id.starts_with("k3") { - Some(ReasoningMode::ValueSet { - values: ["disabled", "low", "high", "max"] - .iter().map(|s| s.to_string()).collect(), - default: Some("max".to_string()), - }) - } else { - None - } - } - - fn reasoning_request(&self, value: &serde_json::Value) -> Option { - // The Kimi Code API accepts a flat `reasoning_effort`: "none" disables - // thinking, low/high/max select the effort (unknown values → HTTP 400). - match value.as_str()? { - "disabled" => Some(serde_json::json!({ "reasoning_effort": "none" })), - effort => Some(serde_json::json!({ "reasoning_effort": effort })), - } - } - - fn build_llm(&self, record: &LlmProviderRecord, model: &LlmModelRecord) -> Option> { - Some(build_openai_llm(self, Self::BASE_URL, record, model, false)) - } - - fn ui_meta(&self) -> ProviderUiMeta { - ProviderUiMeta { - type_id: "moonshot_code", - display_name: "Moonshot AI Kimi Code", - description: Some("Kimi Code subscription models — k3 / kimi-for-coding (OpenAI-compatible)"), - color: "#000000", - icon: "bi-code-slash", - lists_models: true, - fields: &[ - ProviderField { key: "api_key", label: "API Key", required: true, secret: true }, - ], - } - } -} diff --git a/crates/skald-core/src/llm/providers/zai.rs b/crates/skald-core/src/llm/providers/zai.rs deleted file mode 100644 index db72acf..0000000 --- a/crates/skald-core/src/llm/providers/zai.rs +++ /dev/null @@ -1,135 +0,0 @@ -use anyhow::Result; - -use crate::llm::{LlmModelRecord, LlmProviderRecord}; -use crate::llm::providers::{RemoteLlmModelInfo, build_openai_llm}; -use crate::provider::{ApiProvider, BuiltLlmClient, ProviderField, ProviderUiMeta, ReasoningMode, ServiceType}; - -/// Z.AI (Zhipu AI) — OpenAI-compatible GLM API. -/// -/// Endpoint `https://api.z.ai/api/paas/v4/chat/completions`; `OpenAiClient` -/// appends `/chat/completions`, so the base URL is `.../paas/v4`. -/// -/// Z.AI exposes no `GET /models` endpoint, so the model catalog is a curated -/// static list of the currently published GLM models. -pub struct ZaiProvider; - -impl ZaiProvider { - pub fn new() -> Self { - Self - } - - /// Base URL for the OpenAI-compatible chat endpoint (without `/chat/completions`). - const BASE_URL: &'static str = "https://api.z.ai/api/paas/v4"; - - /// Curated GLM catalog. Z.AI has no `GET /models` endpoint; this mirrors the - /// model menu published on the Z.AI console. - fn catalog() -> &'static [&'static str] { - &[ - "glm-5.2", - "glm-5.1", - "glm-5", - "glm-5-turbo", - "glm-4.7", - "glm-4.6", - "glm-4.5", - "glm-4-32b-0414-128k", - ] - } - - fn known_context_length(model_id: &str) -> Option { - let id = model_id.to_lowercase(); - if id.contains("128k") { Some(131_072) } - else if id.starts_with("glm-5") { Some(1_048_576) } // GLM-5.x: 1M context (per Z.AI) - else if id.starts_with("glm-4.7") { Some(200_000) } - else if id.starts_with("glm-4.6") { Some(200_000) } - else if id.starts_with("glm-4.5") { Some(131_072) } - else { None } - } -} - -#[async_trait::async_trait] -impl ApiProvider for ZaiProvider { - fn type_id(&self) -> &'static str { "zai" } - fn display_name(&self) -> &'static str { "Z.AI" } - fn supported_types(&self) -> &'static [ServiceType] { - &[ServiceType::Llm] - } - - async fn list_llm_models(&self, _record: &LlmProviderRecord) -> Result>> { - let models = Self::catalog() - .iter() - .map(|id| RemoteLlmModelInfo { - id: id.to_string(), - name: id.to_string(), - context_length: Self::known_context_length(id), - max_completion_tokens: None, - knowledge_cutoff: None, - capabilities: vec!["function_calling".to_string()], - vision: Some(false), - price_input_per_million: None, - price_output_per_million: None, - reasoning: None, - }) - .collect(); - - Ok(Some(models)) - } - - fn reasoning_mode(&self, model_id: &str, _capabilities: &[String]) -> Option { - let id = model_id.to_lowercase(); - // GLM-5.2 (and above) additionally expose a graded `reasoning_effort` - // on top of the thinking toggle, so offer the effort levels directly - // ("disabled" turns thinking off). - if id.starts_with("glm-5.2") { - Some(ReasoningMode::ValueSet { - values: ["disabled", "minimal", "low", "medium", "high", "xhigh", "max"] - .iter().map(|s| s.to_string()).collect(), - default: Some("max".to_string()), - }) - // Deep-thinking toggle (thinking.type) is supported by the GLM-5.x - // series and GLM-4.5/4.6/4.7 (but not the older glm-4-32b). - } else if id.starts_with("glm-5") - || id.starts_with("glm-4.7") - || id.starts_with("glm-4.6") - || id.starts_with("glm-4.5") - { - Some(ReasoningMode::ValueSet { - values: vec!["disabled".to_string(), "enabled".to_string()], - default: Some("enabled".to_string()), - }) - } else { - None - } - } - - fn reasoning_request(&self, value: &serde_json::Value) -> Option { - // "disabled" → thinking off; "enabled" → thinking on (no effort); - // any effort level → thinking on + `reasoning_effort` (GLM-5.2+). - match value.as_str()? { - "disabled" => Some(serde_json::json!({ "thinking": { "type": "disabled" } })), - "enabled" => Some(serde_json::json!({ "thinking": { "type": "enabled" } })), - effort => Some(serde_json::json!({ - "thinking": { "type": "enabled" }, - "reasoning_effort": effort, - })), - } - } - - fn build_llm(&self, record: &LlmProviderRecord, model: &LlmModelRecord) -> Option> { - Some(build_openai_llm(self, Self::BASE_URL, record, model, false)) - } - - fn ui_meta(&self) -> ProviderUiMeta { - ProviderUiMeta { - type_id: "zai", - display_name: "Z.AI", - description: Some("Zhipu AI GLM models (OpenAI-compatible)"), - color: "#4f46e5", - icon: "bi-stars", - lists_models: true, - fields: &[ - ProviderField { key: "api_key", label: "API Key", required: true, secret: true }, - ], - } - } -} diff --git a/crates/skald-core/src/skald/bundles.rs b/crates/skald-core/src/skald/bundles.rs index e76f8c7..d615443 100644 --- a/crates/skald-core/src/skald/bundles.rs +++ b/crates/skald-core/src/skald/bundles.rs @@ -47,6 +47,7 @@ use crate::tts::TtsManager; use tokio::sync::RwLock; use core_api::plugin::Plugin; +use core_api::provider::ApiProvider; use super::runtime::Runtime; @@ -66,11 +67,16 @@ impl Models { provider_registry.register_builtin(crate::llm::providers::anthropic::AnthropicProvider::new()); provider_registry.register_builtin(crate::llm::providers::openrouter::OpenRouterProvider::new()); provider_registry.register_builtin(crate::llm::providers::ollama::OllamaProvider::new()); - provider_registry.register_builtin(crate::llm::providers::lm_studio::LmStudioProvider::new()); - provider_registry.register_builtin(crate::llm::providers::deepseek::DeepSeekProvider::new()); - provider_registry.register_builtin(crate::llm::providers::zai::ZaiProvider::new()); - provider_registry.register_builtin(crate::llm::providers::moonshot::MoonshotProvider::new()); - provider_registry.register_builtin(crate::llm::providers::moonshot::MoonshotCodeProvider::new()); + // OpenAI-compatible providers are runtime data (providers.yaml), not code. + for p in crate::llm::providers::declared::load(std::path::Path::new( + crate::llm::providers::declared::PROVIDERS_FILE, + )) { + if provider_registry.contains(p.type_id()) { + warn!(type_id = p.type_id(), "declared provider id collides with a native provider — skipped"); + continue; + } + provider_registry.register_builtin(p); + } let provider_registry = Arc::new(provider_registry); info!("provider registry ready ({} built-in providers)", provider_registry.all().len()); diff --git a/providers.yaml b/providers.yaml new file mode 100644 index 0000000..6dc0264 --- /dev/null +++ b/providers.yaml @@ -0,0 +1,168 @@ +# Declarative OpenAI-compatible LLM providers. +# +# Loaded at boot from this file (resolved against the process working +# directory, like config.yml) — edit and `restart`, no rebuild needed. An +# invalid entry is logged and skipped; native providers (anthropic, ollama, +# openai, openrouter — whose behavior is not plain OpenAI-compatible) are +# unaffected. Entries whose `id` collides with a native provider are skipped. +# +# Entry shape (optional unless noted): +# id: unique type id stored in the DB (required) +# name: display name (required) +# kind: openai_compatible — the only kind for now +# base_url: chat base URL, OpenAI-compatible (required) +# base_url_overridable: the DB instance's base_url overrides this default +# api_key: required | optional | none (default: required) +# prompt_cache: true | false (default: false) +# ui: { color, icon, description } (color/icon required) +# fields: UI form fields [{ key, label, required, secret }] +# models: remote catalog config — omit to disable listing: +# endpoint: GET path joined to models_url (OpenAI `data` envelope) +# models_url: defaults to the resolved base_url +# auth: bearer | none (default: bearer unless api_key: none) +# static: [model ids] — alternative to endpoint +# map: per-model JSON field names, all optional: +# id / name / context_length / max_completion_tokens / knowledge_cutoff +# vision: (also adds the `vision` capability) +# price_input_per_million / price_output_per_million: +# capability_flags: { : } (e.g. reasoning) +# base_capabilities: capabilities every listed model gets +# defaults: { vision: bool } — used when the source omits it +# enrich: first-matching rule wins, glob on the model id +# (`*` is the only wildcard, case-insensitive): +# - match: "k3*" prefix / "*reasoner*" contains / "glm-5" exact +# mode: fill | override fill keeps endpoint values (default), +# override lets the rule win +# context_length / max_completion_tokens / vision / add_capabilities +# reasoning: reasoning knob — omit when models don't reason: +# request: how the selected value reaches the request body: +# { kind: effort, remap: { disabled: none } } {"reasoning_effort": v} +# { kind: thinking } {"thinking": {...}} (+effort) +# modes: first-matching rule wins: +# - when: { models: [globs], capability: name } both present = OR; +# values: [...] absent = every model +# default: ... +providers: + - id: moonshot + name: "Moonshot AI pay-as-you-go" + base_url: "https://api.moonshot.ai/v1" + ui: + color: "#2563eb" + icon: "bi-moon-stars" + description: "Kimi models on the Moonshot AI platform (OpenAI-compatible)" + fields: + - { key: api_key, label: "API Key", required: true, secret: true } + models: + endpoint: /models + map: + context_length: context_length + vision: supports_image_in + capability_flags: + reasoning: supports_reasoning + base_capabilities: [function_calling] + + - id: moonshot_code + name: "Moonshot AI Kimi Code" + base_url: "https://api.kimi.com/coding/v1" + ui: + color: "#000000" + icon: "bi-code-slash" + description: "Kimi Code subscription models — k3 / kimi-for-coding (OpenAI-compatible)" + fields: + - { key: api_key, label: "API Key", required: true, secret: true } + models: + endpoint: /models + map: + context_length: context_length + vision: supports_image_in + capability_flags: + reasoning: supports_reasoning + base_capabilities: [function_calling] + # Fills the metadata the endpoint may omit (endpoint values always win): + # k3 → 1M context + native vision; kimi-for-coding → 256k. + enrich: + - { match: "k3*", context_length: 1048576, vision: true } + - { match: "kimi-for-coding*", context_length: 262144 } + reasoning: + # k3 exposes a graded reasoning_effort ("disabled" routes to K2.6); + # kimi-for-coding always thinks and has no knob. + request: { kind: effort, remap: { disabled: none } } + modes: + - when: { models: ["k3*"] } + values: [disabled, low, high, max] + default: max + + - id: deepseek + name: "DeepSeek" + base_url: "https://api.deepseek.com/v1" + ui: + color: "#0ea5e9" + icon: "bi-search" + fields: + - { key: api_key, label: "API Key", required: true, secret: true } + models: + endpoint: /models + models_url: "https://api.deepseek.com" + map: + context_length: context_length + base_capabilities: [function_calling] + # Known metadata wins over whatever the endpoint returns. + enrich: + - { match: "*coder*", context_length: 16384, mode: override } + - { match: "*reasoner*", context_length: 65536, mode: override, add_capabilities: [reasoning] } + - { match: "deepseek-v4*", context_length: 1048576, mode: override, max_completion_tokens: 393216 } + - { match: "deepseek-chat*", context_length: 65536, mode: override } + - { match: "deepseek-v3*", context_length: 65536, mode: override } + reasoning: + # Thinking toggle + graded reasoning_effort; low/medium map to high + # server-side, xhigh to max. + request: { kind: thinking } + modes: + - when: { capability: reasoning, models: ["*reasoner*", "deepseek-v4*"] } + values: [disabled, low, medium, high, xhigh, max] + default: high + + - id: zai + name: "Z.AI" + base_url: "https://api.z.ai/api/paas/v4" + ui: + color: "#4f46e5" + icon: "bi-stars" + description: "Zhipu AI GLM models (OpenAI-compatible)" + fields: + - { key: api_key, label: "API Key", required: true, secret: true } + models: + # Z.AI exposes no GET /models endpoint; this mirrors the console menu. + static: [glm-5.2, glm-5.1, glm-5, glm-5-turbo, glm-4.7, glm-4.6, glm-4.5, glm-4-32b-0414-128k] + defaults: { vision: false } + base_capabilities: [function_calling] + enrich: + - { match: "*128k*", context_length: 131072 } + - { match: "glm-5*", context_length: 1048576 } + - { match: "glm-4.7*", context_length: 200000 } + - { match: "glm-4.6*", context_length: 200000 } + - { match: "glm-4.5*", context_length: 131072 } + reasoning: + request: { kind: thinking } + modes: + # GLM-5.2+ adds a graded effort on top of the thinking toggle. + - when: { models: ["glm-5.2*"] } + values: [disabled, minimal, low, medium, high, xhigh, max] + default: max + - when: { models: ["glm-5*", "glm-4.7*", "glm-4.6*", "glm-4.5*"] } + values: [disabled, enabled] + default: enabled + + - id: lm_studio + name: "LM Studio" + base_url: "http://localhost:1234/v1" + base_url_overridable: true + api_key: none + ui: + color: "#6b7280" + icon: "bi-window-stack" + description: "Local models via LM Studio" + fields: + - { key: base_url, label: "Base URL", required: false, secret: false } + models: + endpoint: /models