llm: switch Skald to agent-loop Model clients; drop llm-client (phase 1, D13)
The LLM call path now runs on the agent-loop crate's clients and trait: - core-api: BuiltLlmClient.client is Arc<dyn agent_loop::model::Model>; chatbot.rs (ChatbotClient + wire types) deleted; APP_NAME re-exported from agent-loop - providers (openai/anthropic/ollama/openrouter/requesty/declared) build OpenAiModel/AnthropicModel/OllamaModel with the model's wire id - LoggingModel decorator (llm/logging.rs) replaces LoggingChatbotClient; per-request correlation (session/stack/user) travels in the new ModelRequest.log field, never sent to providers - llm_call/llm_loop/compactor speak Model::complete + ModelResponse; retriability via Model::is_retriable (structured status, B6 rule now the crate's default); payload persistence reads RawMeta off ModelResponse/ModelError - crates/llm-client and skald-core/src/chatbot deleted Full workspace test suite green (incl. 162 skald-core + 32 agent-loop).
This commit is contained in:
@@ -1,193 +0,0 @@
|
||||
use async_trait::async_trait;
|
||||
use serde_json::Value;
|
||||
use tokio::sync::mpsc;
|
||||
|
||||
/// A single message in a conversation.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Message {
|
||||
pub role: Role,
|
||||
pub content: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Role {
|
||||
System,
|
||||
User,
|
||||
Assistant,
|
||||
}
|
||||
|
||||
impl Message {
|
||||
pub fn system(content: impl Into<String>) -> Self {
|
||||
Self { role: Role::System, content: content.into() }
|
||||
}
|
||||
|
||||
pub fn user(content: impl Into<String>) -> Self {
|
||||
Self { role: Role::User, content: content.into() }
|
||||
}
|
||||
|
||||
pub fn assistant(content: impl Into<String>) -> Self {
|
||||
Self { role: Role::Assistant, content: content.into() }
|
||||
}
|
||||
}
|
||||
|
||||
/// Options for a single chat completion request.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ChatOptions {
|
||||
pub model: String,
|
||||
pub max_tokens: Option<u32>,
|
||||
pub temperature: Option<f32>,
|
||||
/// Session/stack IDs for request logging. Set by the LLM loop; ignored by
|
||||
/// providers — only the logging wrapper reads them.
|
||||
pub session_id: Option<i64>,
|
||||
pub stack_id: Option<i64>,
|
||||
/// The authenticated user driving this request. Correlates the metadata row
|
||||
/// in `system.db` with the payload in `{userid}.db`. Logging-only.
|
||||
pub user_id: Option<String>,
|
||||
/// UUID correlating the metadata row (`llm_requests`) with the payload row
|
||||
/// (`llm_request_payloads`). Generated by the LLM loop before the call.
|
||||
/// Logging-only.
|
||||
pub request_id: Option<String>,
|
||||
}
|
||||
|
||||
/// Raw HTTP metadata captured during a provider call.
|
||||
/// Sensitive header values (api_key) are redacted before storage.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct LlmRawMeta {
|
||||
pub request_headers: Option<Value>,
|
||||
pub request_body: Option<Value>,
|
||||
pub response_headers: Option<Value>,
|
||||
pub response_body: Option<Value>,
|
||||
}
|
||||
|
||||
/// The response from a chat completion (text only).
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ChatResponse {
|
||||
pub content: String,
|
||||
pub input_tokens: Option<u32>,
|
||||
pub output_tokens: Option<u32>,
|
||||
/// True when the model stopped due to hitting the token limit.
|
||||
pub truncated: bool,
|
||||
/// Chain-of-thought produced by reasoning models (e.g. DeepSeek thinking mode).
|
||||
/// Must be echoed back in the assistant message on subsequent turns.
|
||||
pub reasoning_content: Option<String>,
|
||||
/// Tokens served from the provider's prompt cache (Anthropic: cache_read_input_tokens,
|
||||
/// OpenAI: prompt_tokens_details.cached_tokens). None when the provider does not
|
||||
/// report cache metrics.
|
||||
pub cache_read_tokens: Option<u32>,
|
||||
/// Tokens written into the provider's prompt cache (Anthropic only:
|
||||
/// cache_creation_input_tokens). None for providers that do not expose this.
|
||||
pub cache_creation_tokens: Option<u32>,
|
||||
/// Cost of the request in USD, when the provider reports it (OpenRouter
|
||||
/// returns it under `usage.cost`). None for providers that do not bill
|
||||
/// per-request or do not expose the figure.
|
||||
pub cost: Option<f64>,
|
||||
}
|
||||
|
||||
/// A single tool call requested by the LLM.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ToolCall {
|
||||
pub id: String,
|
||||
pub name: String,
|
||||
pub arguments: Value,
|
||||
}
|
||||
|
||||
/// An incremental piece of a streaming completion, pushed by providers that
|
||||
/// support SSE streaming. Purely best-effort UI feedback: the final `LlmTurn`
|
||||
/// remains the authoritative result.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum StreamDelta {
|
||||
/// Visible answer text.
|
||||
Text(String),
|
||||
/// Chain-of-thought / reasoning tokens (thinking models).
|
||||
Reasoning(String),
|
||||
}
|
||||
|
||||
/// Result of one LLM turn when tools are available.
|
||||
#[derive(Debug)]
|
||||
pub enum LlmTurn {
|
||||
Message(ChatResponse),
|
||||
ToolCalls {
|
||||
content: String,
|
||||
calls: Vec<ToolCall>,
|
||||
input_tokens: Option<u32>,
|
||||
output_tokens: Option<u32>,
|
||||
reasoning_content: Option<String>,
|
||||
cache_read_tokens: Option<u32>,
|
||||
cache_creation_tokens: Option<u32>,
|
||||
cost: Option<f64>,
|
||||
},
|
||||
}
|
||||
|
||||
/// Stateless LLM client. Implementations hold only connection config (base URL,
|
||||
/// API key). No memory, no database, no session state.
|
||||
#[async_trait]
|
||||
pub trait ChatbotClient: Send + Sync {
|
||||
async fn chat(
|
||||
&self,
|
||||
messages: &[Message],
|
||||
options: &ChatOptions,
|
||||
) -> anyhow::Result<ChatResponse>;
|
||||
|
||||
/// Extracts the request cost in USD from a provider's raw JSON response,
|
||||
/// when the provider reports it. OpenRouter (and other OpenAI-compatible
|
||||
/// gateways) return it under `usage.cost`; the default reads that path and
|
||||
/// yields None when absent. Providers with a different shape override this.
|
||||
fn extract_cost(&self, response: &Value) -> Option<f64> {
|
||||
response["usage"]["cost"].as_f64()
|
||||
}
|
||||
|
||||
/// Chat with tool support. Default implementation ignores tools and falls
|
||||
/// back to `chat()`.
|
||||
async fn chat_with_tools(
|
||||
&self,
|
||||
messages: &[Value],
|
||||
tools: &[Value],
|
||||
options: &ChatOptions,
|
||||
) -> anyhow::Result<LlmTurn> {
|
||||
let simple: Vec<Message> = messages
|
||||
.iter()
|
||||
.filter_map(|m| {
|
||||
let role = m["role"].as_str()?;
|
||||
let content = m["content"].as_str().unwrap_or("").to_string();
|
||||
match role {
|
||||
"system" => Some(Message::system(content)),
|
||||
"user" => Some(Message::user(content)),
|
||||
"assistant" => Some(Message::assistant(content)),
|
||||
_ => None,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
let _ = tools;
|
||||
let resp = self.chat(&simple, options).await?;
|
||||
Ok(LlmTurn::Message(resp))
|
||||
}
|
||||
|
||||
/// Like `chat_with_tools` but also returns raw HTTP metadata for logging.
|
||||
/// Providers that make real HTTP calls should override this.
|
||||
async fn chat_with_tools_raw(
|
||||
&self,
|
||||
messages: &[Value],
|
||||
tools: &[Value],
|
||||
options: &ChatOptions,
|
||||
) -> anyhow::Result<(LlmTurn, Option<LlmRawMeta>)> {
|
||||
self.chat_with_tools(messages, tools, options).await.map(|t| (t, None))
|
||||
}
|
||||
|
||||
/// Like `chat_with_tools_raw`, but the provider may push incremental
|
||||
/// [`StreamDelta`]s into `delta_tx` as tokens arrive (SSE streaming).
|
||||
/// Senders should use `try_send` and drop deltas when the channel is full —
|
||||
/// streaming is best-effort UI feedback and must never backpressure the
|
||||
/// HTTP read. The returned `LlmTurn` is always the complete, authoritative
|
||||
/// result. The default ignores the channel and falls back to the buffered
|
||||
/// call, so providers without streaming behave exactly as before.
|
||||
async fn chat_with_tools_raw_streaming(
|
||||
&self,
|
||||
messages: &[Value],
|
||||
tools: &[Value],
|
||||
options: &ChatOptions,
|
||||
delta_tx: mpsc::Sender<StreamDelta>,
|
||||
) -> anyhow::Result<(LlmTurn, Option<LlmRawMeta>)> {
|
||||
let _ = delta_tx;
|
||||
self.chat_with_tools_raw(messages, tools, options).await
|
||||
}
|
||||
}
|
||||
@@ -1,11 +1,12 @@
|
||||
/// Application name, sent as `X-Title` HTTP header to LLM/image/audio providers.
|
||||
pub const APP_NAME: &str = "Skald";
|
||||
/// Lives in `agent-loop` (the LLM clients' home, blueprint D13); re-exported here
|
||||
/// so existing users don't change.
|
||||
pub use agent_loop::APP_NAME;
|
||||
|
||||
pub mod approval;
|
||||
pub mod bus;
|
||||
pub mod config_api;
|
||||
pub mod system_bus;
|
||||
pub mod chatbot;
|
||||
pub mod chat_hub;
|
||||
pub mod command;
|
||||
pub mod events;
|
||||
|
||||
@@ -3,7 +3,8 @@ use std::sync::Arc;
|
||||
use anyhow::Result;
|
||||
use async_trait::async_trait;
|
||||
|
||||
use crate::chatbot::ChatbotClient;
|
||||
use agent_loop::model::Model;
|
||||
|
||||
use crate::image_generate::{ImageGenerate, ImageGenerateModelRecord};
|
||||
use crate::tts::{TextToSpeech, TtsModelRecord, RemoteTtsModelInfo};
|
||||
use crate::transcribe::{Transcribe, TranscribeModelRecord, RemoteTranscribeModelInfo};
|
||||
@@ -139,7 +140,8 @@ pub struct ProviderField {
|
||||
// ── BuiltLlmClient ────────────────────────────────────────────────────────────
|
||||
|
||||
pub struct BuiltLlmClient {
|
||||
pub client: Arc<dyn ChatbotClient>,
|
||||
/// A stateless `agent_loop` model client (blueprint D13).
|
||||
pub client: Arc<dyn Model>,
|
||||
pub prompt_cache: bool,
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user