token streaming & reasoning display: live SSE tokens frontend to back
Nightly Build / build (push) Successful in 6m59s
Nightly Build / build (push) Successful in 6m59s
- Add StreamDelta(SseDecoder) framing shared by OpenAI/Anthropic - OpenAiClient: stream=true + reasoning_content deltas, index-based tool_calls accumulation, usage from final chunk - AnthropicClient: message_start/content_block_*/message_delta events, thinking_delta->reasoning, input_json_delta->tool input - TokenDelta ServerEvent variant wired through ChatHub + WS broadcast - Frontend throttled flush (~15 Hz), pending bubble mutate-in-place, reasoning as collapsed-by-default <details> - Drop streaming bubble on error/llm_failed/model_fallback - i18n: chat.reasoning key added to en/fr/it
This commit is contained in:
@@ -128,6 +128,7 @@ impl ChatSessionHandler {
|
||||
input_tokens: None,
|
||||
output_tokens: None,
|
||||
truncated: false,
|
||||
reasoning_content: msg.reasoning_content,
|
||||
tool_calls: Vec::new(),
|
||||
};
|
||||
break 'seed (outcome, stack);
|
||||
@@ -208,13 +209,13 @@ impl ChatSessionHandler {
|
||||
|
||||
// current_stack is now the root (depth=0); emit the final event.
|
||||
match current_outcome {
|
||||
TurnOutcome::Final { content, message_id, input_tokens, output_tokens, truncated, .. } => {
|
||||
TurnOutcome::Final { content, message_id, input_tokens, output_tokens, truncated, reasoning_content, .. } => {
|
||||
info!(session_id = self.session_id, "resume_turn done");
|
||||
if truncated {
|
||||
warn!(session_id = self.session_id, "response truncated");
|
||||
em.truncated(output_tokens).await;
|
||||
}
|
||||
em.done(message_id, current_stack.id, content, input_tokens, output_tokens).await;
|
||||
em.done(message_id, current_stack.id, content, input_tokens, output_tokens, reasoning_content).await;
|
||||
}
|
||||
TurnOutcome::Cancelled => {
|
||||
info!(session_id = self.session_id, "resume_turn cancelled");
|
||||
|
||||
Reference in New Issue
Block a user