diff --git a/backend/open_webui/routers/chats.py b/backend/open_webui/routers/chats.py index 7fbdfa125a..98621ff80e 100644 --- a/backend/open_webui/routers/chats.py +++ b/backend/open_webui/routers/chats.py @@ -37,7 +37,6 @@ from open_webui.utils.access_control import filter_allowed_access_grants, has_pe from open_webui.utils.access_control.folders import has_folder_access from open_webui.utils.auth import get_admin_user, get_verified_user from open_webui.utils.context_compaction import compact_chat_branch -from open_webui.utils.middleware import serialize_output from open_webui.utils.misc import get_message_list from open_webui.utils.models import get_all_models from pydantic import BaseModel @@ -1221,16 +1220,6 @@ async def update_chat_by_id( form_data.chat.get('history'), ) - # Re-derive content from output for assistant messages so that frontend - # edits to output items are reflected in content. Only when output - # actually changed — otherwise content set independently of output - # (e.g. a `replace` event or an outlet filter footer) would be reverted. - existing_messages = (chat.chat.get('history') or {}).get('messages') or {} - for msg_id, msg in updated_chat.get('history', {}).get('messages', {}).items(): - if isinstance(msg, dict) and msg.get('role') == 'assistant' and msg.get('output'): - if msg.get('output') != existing_messages.get(msg_id, {}).get('output'): - msg['content'] = serialize_output(msg['output']) - chat = await Chats.update_chat_by_id(id, updated_chat, db=db) # Reconcile chat_message rows without inferring deletes from missing IDs. diff --git a/backend/open_webui/utils/middleware.py b/backend/open_webui/utils/middleware.py index e17b42485b..fc8802a683 100644 --- a/backend/open_webui/utils/middleware.py +++ b/backend/open_webui/utils/middleware.py @@ -2,7 +2,6 @@ import ast import asyncio import base64 import copy -import html import inspect import json import logging @@ -378,210 +377,6 @@ def get_citation_source_from_tool_result( ] -def split_content_and_whitespace(content): - content_stripped = content.rstrip() - original_whitespace = content[len(content_stripped) :] if len(content) > len(content_stripped) else '' - return content_stripped, original_whitespace - - -def is_opening_code_block(content): - backtick_segments = content.split('```') - # Even number of segments means the last backticks are opening a new block - return len(backtick_segments) > 1 and len(backtick_segments) % 2 == 0 - - -_OPENAI_TOOL_DISPLAY_NAMES = { - 'web_search_call': 'Web Search', - 'file_search_call': 'File Search', - 'computer_call': 'Computer Use', -} - - -def _render_openai_tool_call_handler(item: dict, done: bool) -> str: - """Render an OpenAI Responses API server-side tool item as a
block. - - Handles web_search_call, file_search_call, and computer_call items whose - schemas are defined in the openai-python SDK (generated from OpenAPI spec). - """ - item_type = item.get('type', '') - call_id = item.get('id', '') - display_name = _OPENAI_TOOL_DISPLAY_NAMES.get(item_type, item_type) - - # Build a short summary of what the tool did - summary = '' - if item_type == 'web_search_call': - action = item.get('action', {}) - if isinstance(action, dict): - atype = action.get('type', '') - if atype == 'search': - queries = action.get('queries') or [] - query = action.get('query', '') - summary = ( - f'Search: {", ".join(str(q) for q in queries)}' - if queries - else (f'Search: {query}' if query else '') - ) - elif atype == 'open_page': - summary = f'Open page: {action.get("url", "")}' if action.get('url') else '' - elif atype == 'find_in_page': - summary = f'Find in page: {action.get("pattern", "")}' if action.get('pattern') else '' - elif item_type == 'file_search_call': - queries = item.get('queries', []) - if queries: - summary = f'Queries: {", ".join(str(q) for q in queries)}' - elif item_type == 'computer_call': - action = item.get('action') - actions = item.get('actions') - if isinstance(action, dict): - summary = f'Action: {action.get("type", "unknown")}' - elif isinstance(actions, list) and actions: - summary = f'Actions: {", ".join(a.get("type", "?") for a in actions if isinstance(a, dict))}' - - escaped_name = html.escape(display_name) - if done: - return f'
\nTool Executed\n{html.escape(summary)}\n
\n' - return f'
\nExecuting...\n
\n' - - -def serialize_output(output: list) -> str: - """ - Convert OR-aligned output items to HTML for display. - For LLM consumption, use convert_output_to_messages() instead. - """ - parts: list[str] = [] - - # First pass: collect function_call_output items by call_id for lookup - tool_outputs = {} - for item in output: - if item.get('type') == 'function_call_output': - tool_outputs[item.get('call_id')] = item - - # Second pass: render items in order - for idx, item in enumerate(output): - item_type = item.get('type', '') - - if item_type == 'message': - for content_part in item.get('content', []): - if 'text' in content_part: - text = content_part.get('text', '').strip() - if text: - parts.append(text) - - elif item_type == 'function_call': - call_id = item.get('call_id', '') - name = item.get('name', '') - arguments = item.get('arguments', '') - - result_item = tool_outputs.get(call_id) - if result_item: - result_parts: list[str] = [] - for result_output in result_item.get('output', []): - if 'text' in result_output: - output_text = result_output.get('text', '') - result_parts.append(str(output_text) if not isinstance(output_text, str) else output_text) - result_text = ''.join(result_parts) - files = result_item.get('files') - embeds = result_item.get('embeds', '') - - parts.append( - f'
\nTool Executed\n{html.escape(json.dumps(result_text, ensure_ascii=False))}\n
' - ) - else: - parts.append( - f'
\nExecuting...\n
' - ) - - elif item_type == 'function_call_output': - # Already handled inline with function_call above - pass - - elif item_type in _OPENAI_TOOL_DISPLAY_NAMES: - status = item.get('status', 'in_progress') - done = status in ('completed', 'failed', 'incomplete') or idx != len(output) - 1 - parts.append(_render_openai_tool_call_handler(item, done).rstrip('\n')) - - elif item_type == 'reasoning': - reasoning_parts: list[str] = [] - # Check for 'summary' (new structure) or 'content' (legacy/fallback) - source_list = item.get('summary', []) or item.get('content', []) - for content_part in source_list: - if 'text' in content_part: - reasoning_parts.append(content_part.get('text', '')) - elif 'summary' in content_part: # Handle potential nested logic if any - pass - - reasoning_content = ''.join(reasoning_parts).strip() - - duration = item.get('duration') - status = item.get('status', 'in_progress') - - # Infer completion: if this reasoning item is NOT the last item, - # render as done (a subsequent item means reasoning is complete) - is_last_item = idx == len(output) - 1 - - display = html.escape( - '\n'.join( - (f'> {line}' if not line.startswith('>') else line) for line in reasoning_content.splitlines() - ) - ) - - if status == 'completed' or duration is not None or not is_last_item: - parts.append( - f'
\nThought for {duration or 0} seconds\n{display}\n
' - ) - else: - parts.append( - f'
\nThinking…\n{display}\n
' - ) - - elif item_type == 'open_webui:code_interpreter': - # Code interpreter needs to inspect/mutate prior accumulated content - # to strip trailing unclosed code fences — materialize only here. - content = '\n'.join(parts) - content_stripped, original_whitespace = split_content_and_whitespace(content) - if is_opening_code_block(content_stripped): - content = content_stripped.rstrip('`').rstrip() + original_whitespace - else: - content = content_stripped + original_whitespace - - # Re-split back into parts list after mutation - parts = [content] if content else [] - - # Render the code_interpreter item as a
block - # so the frontend Collapsible renders "Analyzing..."/"Analyzed". - code = item.get('code', '').strip() - lang = item.get('lang', 'python') - status = item.get('status', 'in_progress') - duration = item.get('duration') - is_last_item = idx == len(output) - 1 - - # Build inner content: code block - display = '' - if code: - display = f'```{lang}\n{code}\n```' - - # Build output attribute as HTML-escaped JSON for CodeBlock.svelte - ci_output = item.get('output') - output_attr = '' - if ci_output: - if isinstance(ci_output, dict): - output_json = json.dumps(ci_output, ensure_ascii=False) - else: - output_json = json.dumps({'result': str(ci_output)}, ensure_ascii=False) - output_attr = f' output="{html.escape(output_json)}"' - - if status == 'completed' or duration is not None or not is_last_item: - parts.append( - f'
\nAnalyzed\n{display}\n
' - ) - else: - parts.append( - f'
\nAnalyzing…\n{display}\n
' - ) - - return '\n'.join(parts).strip() - - def deep_merge(target, source): """ Merge source into target recursively (returning new structure). @@ -3127,7 +2922,6 @@ def update_assistant_message_from_stream(assistant_message, raw): output, meta = handle_responses_streaming_event(data, assistant_message.get('output', [])) if output: assistant_message['output'] = output - assistant_message['content'] = serialize_output(output) if meta and meta.get('usage'): assistant_message['usage'] = merge_usage(assistant_message.get('usage'), meta['usage']) continue @@ -3527,18 +3321,16 @@ async def outlet_filter_handler(ctx): 'output' ) if content_changed or output_changed: - # If output was modified, re-derive content from it - new_content = message.get('content', original_message.get('content', '')) - if output_changed: - new_content = serialize_output(message['output']) + message_update = { + 'originalContent': original_message.get('content'), + **({'output': message['output']} if output_changed else {}), + } + if content_changed: + message_update['content'] = message.get('content', '') await Chats.upsert_message_to_chat_by_id_and_message_id( chat_id, outlet_message_id, - { - 'content': new_content, - 'originalContent': original_message.get('content'), - **({'output': message['output']} if output_changed else {}), - }, + message_update, ) if event_emitter: @@ -3722,7 +3514,7 @@ async def non_streaming_chat_response_handler(response, ctx): if ENABLE_API_OUTLET_FILTERS and (content or output): usage = normalize_usage(response_data.get('usage', {}) or {}) ctx['assistant_message'] = { - 'content': content or serialize_output(output), + **({'content': content} if content else {}), **({'output': output} if output else {}), **({'usage': usage} if usage else {}), } @@ -4242,7 +4034,6 @@ async def streaming_chat_response_handler(response, ctx): processed_data = { 'output': full_output(), - 'content': serialize_output(full_output()), } # print(data) @@ -4412,7 +4203,6 @@ async def streaming_chat_response_handler(response, ctx): ) data = { - 'content': serialize_output(full_output() + pending_fc_items), 'output': full_output() + pending_fc_items, } delta_type = 'tool_call' @@ -4474,7 +4264,6 @@ async def streaming_chat_response_handler(response, ctx): ] data = { - 'content': serialize_output(full_output()), 'output': full_output(), } delta_type = 'content' @@ -4637,18 +4426,15 @@ async def streaming_chat_response_handler(response, ctx): metadata['chat_id'], metadata['message_id'], { - 'content': serialize_output(full_output()), 'output': full_output(), }, ) data = { - 'content': serialize_output(full_output()), 'output': full_output(), } delta_type = 'content' else: data = { - 'content': serialize_output(full_output()), 'output': full_output(), } delta_type = 'content' @@ -4791,7 +4577,6 @@ async def streaming_chat_response_handler(response, ctx): { 'type': 'chat:completion', 'data': { - 'content': serialize_output(full_output()), 'output': full_output(), }, } @@ -4948,7 +4733,7 @@ async def streaming_chat_response_handler(response, ctx): display_files = [] for file_item in result.get('files', []): if file_item.get('type') == 'image' and file_item.get('url', '').startswith('data:'): - # LLM-only: add as input_image part (invisible to serialize_output) + # LLM-only: add as input_image part, not frontend display output. output_parts.append({'type': 'input_image', 'image_url': file_item['url']}) else: # Frontend display (MCP images, audio, etc.) @@ -5054,7 +4839,6 @@ async def streaming_chat_response_handler(response, ctx): { 'type': 'chat:completion', 'data': { - 'content': serialize_output(output), 'output': frontend_output, }, } @@ -5178,7 +4962,6 @@ async def streaming_chat_response_handler(response, ctx): { 'type': 'chat:completion', 'data': { - 'content': serialize_output(output), 'output': output, }, } @@ -5305,7 +5088,6 @@ async def streaming_chat_response_handler(response, ctx): { 'type': 'chat:completion', 'data': { - 'content': serialize_output(output), 'output': output, }, } @@ -5352,7 +5134,6 @@ async def streaming_chat_response_handler(response, ctx): ) data = { 'done': True, - 'content': serialize_output(output), 'output': output, 'title': title, **({'usage': usage} if usage else {}), @@ -5366,7 +5147,6 @@ async def streaming_chat_response_handler(response, ctx): metadata['message_id'], { 'done': True, - 'content': serialize_output(output), 'output': output, **({'usage': usage} if usage else {}), }, @@ -5409,7 +5189,6 @@ async def streaming_chat_response_handler(response, ctx): ) ctx['assistant_message'] = { - 'content': serialize_output(output), 'output': output, **({'usage': usage} if usage else {}), } @@ -5437,7 +5216,6 @@ async def streaming_chat_response_handler(response, ctx): metadata['message_id'], { 'done': True, - 'content': serialize_output(output), 'output': output, }, ) diff --git a/src/lib/components/chat/Chat.svelte b/src/lib/components/chat/Chat.svelte index 3903b95401..b7558fc472 100644 --- a/src/lib/components/chat/Chat.svelte +++ b/src/lib/components/chat/Chat.svelte @@ -68,6 +68,7 @@ displayFileHandler } from '$lib/utils'; import { AudioQueue } from '$lib/utils/audio'; + import { getOutputText } from './Messages/structuredOutput'; import { archiveChatById, @@ -1248,13 +1249,50 @@ $: onHistoryChange(history); + const dispatchCallOverlayAudio = (message, final = false) => { + if (!$showCallOverlay) { + return; + } + + const messageContentParts = getMessageContentParts( + getOutputText(message?.output) || removeAllDetails(message?.content ?? ''), + $config?.audio?.tts?.split_on ?? 'punctuation' + ); + if (!final) { + messageContentParts.pop(); + } + + const nextContentPart = messageContentParts.at(-1) ?? ''; + if (!nextContentPart || (!final && nextContentPart === message.lastSentence)) { + return; + } + + if (!final) { + message.lastSentence = nextContentPart; + } + + eventTarget.dispatchEvent( + new CustomEvent('chat', { + detail: { + id: message.id, + content: nextContentPart + } + }) + ); + }; + const getContents = () => { const messages = history ? createMessagesList(history, history.currentId) : []; let contents = []; messages.forEach((message) => { - if (message?.role !== 'user' && message?.content) { + if (message?.role !== 'user') { + const messageContent = getOutputText(message?.output) || removeAllDetails(message?.content ?? ''); + if (!messageContent.trim()) { + return; + } + const { codeBlocks: codeBlocks, htmlGroups: htmlGroups } = getCodeBlockContents( - message.content + messageContent ); if (htmlGroups && htmlGroups.length > 0) { @@ -1923,6 +1961,7 @@ // Store raw OR-aligned output items from backend if (output) { message.output = output; + dispatchCallOverlayAudio(message); } if (error) { @@ -1933,10 +1972,11 @@ message.sources = sources; } - if (choices) { + if (choices && !output) { if (choices[0]?.message?.content) { // Non-stream response message.content += choices[0]?.message?.content; + dispatchCallOverlayAudio(message); } else { // Stream response let value = choices[0]?.delta?.content ?? ''; @@ -1948,67 +1988,19 @@ if (navigator.vibrate && ($settings?.hapticFeedback ?? false)) { navigator.vibrate(5); } - - // Emit chat event for TTS (only when call overlay is active) - if ($showCallOverlay) { - const messageContentParts = getMessageContentParts( - removeAllDetails(message.content), - $config?.audio?.tts?.split_on ?? 'punctuation' - ); - messageContentParts.pop(); - - // dispatch only last sentence and make sure it hasn't been dispatched before - if ( - messageContentParts.length > 0 && - messageContentParts[messageContentParts.length - 1] !== message.lastSentence - ) { - message.lastSentence = messageContentParts[messageContentParts.length - 1]; - eventTarget.dispatchEvent( - new CustomEvent('chat', { - detail: { - id: message.id, - content: messageContentParts[messageContentParts.length - 1] - } - }) - ); - } - } + dispatchCallOverlayAudio(message); } } } - if (content) { + if (content && !output) { // REALTIME_CHAT_SAVE is disabled message.content = content; if (navigator.vibrate && ($settings?.hapticFeedback ?? false)) { navigator.vibrate(5); } - - // Emit chat event for TTS (only when call overlay is active) - if ($showCallOverlay) { - const messageContentParts = getMessageContentParts( - removeAllDetails(message.content), - $config?.audio?.tts?.split_on ?? 'punctuation' - ); - messageContentParts.pop(); - - // dispatch only last sentence and make sure it hasn't been dispatched before - if ( - messageContentParts.length > 0 && - messageContentParts[messageContentParts.length - 1] !== message.lastSentence - ) { - message.lastSentence = messageContentParts[messageContentParts.length - 1]; - eventTarget.dispatchEvent( - new CustomEvent('chat', { - detail: { - id: message.id, - content: messageContentParts[messageContentParts.length - 1] - } - }) - ); - } - } + dispatchCallOverlayAudio(message); } if (selected_model_id) { @@ -2024,9 +2016,10 @@ if (done) { message.done = true; + const visibleContent = getOutputText(message?.output) || removeAllDetails(message?.content ?? ''); if ($settings.responseAutoCopy) { - copyToClipboard(message.content); + copyToClipboard(visibleContent); } if ($settings.responseAutoPlayback && !$showCallOverlay) { @@ -2035,25 +2028,12 @@ } // Emit chat event for TTS (only when call overlay is active) - if ($showCallOverlay) { - let lastMessageContentPart = - getMessageContentParts( - removeAllDetails(message.content), - $config?.audio?.tts?.split_on ?? 'punctuation' - )?.at(-1) ?? ''; - if (lastMessageContentPart) { - eventTarget.dispatchEvent( - new CustomEvent('chat', { - detail: { id: message.id, content: lastMessageContentPart } - }) - ); - } - } + dispatchCallOverlayAudio(message, true); eventTarget.dispatchEvent( new CustomEvent('chat:finish', { detail: { id: message.id, - content: message.content + content: visibleContent } }) ); @@ -2480,53 +2460,60 @@ true; // Always include system prompt — backend extracts it and prepends to DB messages. // Only temp chats need conversation messages (persisted chats load from DB). - let messages = [ - params?.system || $settings.system - ? { role: 'system', content: `${params?.system ?? $settings?.system ?? ''}` } - : undefined - ].filter(Boolean); + let messages: any[] = [ + params?.system || $settings.system + ? { role: 'system', content: `${params?.system ?? $settings?.system ?? ''}` } + : undefined + ].filter(Boolean); - if ($temporaryChatEnabled) { - messages = [ - ...messages, - ..._messages.map((message) => ({ - ...message, - content: processDetails(message.content), - ...(message.output ? { output: message.output } : {}) - })) - ].filter((message) => message); + if ($temporaryChatEnabled) { + messages = [ + ...messages, + ..._messages.map((message) => ({ + ...message, + ...(message.output && message.role === 'assistant' + ? { output: message.output } + : { content: processDetails(message.content) }) + })) + ].filter((message) => message); - messages = messages - .map((message, idx, arr) => { - const imageFiles = (message?.files ?? []).filter( - (file) => file.type === 'image' || (file?.content_type ?? '').startsWith('image/') + messages = messages + .map((message) => { + const imageFiles = (message?.files ?? []).filter( + (file) => file.type === 'image' || (file?.content_type ?? '').startsWith('image/') + ); + + if (message.output && message.role === 'assistant') { + return { role: message.role, output: message.output }; + } + + if (message.role === 'user' && imageFiles.length > 0) { + return { + role: message.role, + content: [ + { + type: 'text', + text: message?.merged?.content ?? message.content + }, + ...imageFiles.map((file) => ({ + type: 'image_url', + image_url: { + url: file.url + } + })) + ] + }; + } + + return { + role: message.role, + content: message?.merged?.content ?? message.content + }; + }) + .filter( + (message) => message?.role === 'user' || message?.content?.trim() || message?.output?.length ); - - return { - role: message.role, - ...(message.output ? { output: message.output } : {}), - ...(message.role === 'user' && imageFiles.length > 0 - ? { - content: [ - { - type: 'text', - text: message?.merged?.content ?? message.content - }, - ...imageFiles.map((file) => ({ - type: 'image_url', - image_url: { - url: file.url - } - })) - ] - } - : { - content: message?.merged?.content ?? message.content - }) - }; - }) - .filter((message) => message?.role === 'user' || message?.content?.trim()); - } + } const toolIds = []; const toolServerIds = []; diff --git a/src/lib/components/chat/Messages.svelte b/src/lib/components/chat/Messages.svelte index 141a5c4799..141486cb2b 100644 --- a/src/lib/components/chat/Messages.svelte +++ b/src/lib/components/chat/Messages.svelte @@ -173,7 +173,7 @@ messages: messages }); - // Refresh local message content from backend (e.g. re-derived via serialize_output) + // Keep local plain-content edits aligned with the saved chat response. if (res?.chat?.history?.messages) { for (const [id, msg] of Object.entries(res.chat.history.messages)) { if (history.messages[id] && (msg as any).content) { @@ -385,16 +385,16 @@ const message = history.messages[messageId]; const parentId = message.parentId; - const responseMessage = { - ...message, - id: responseMessageId, - parentId: parentId, - childrenIds: [], - files: undefined, - content: content, - output: output ?? undefined, - timestamp: Math.floor(Date.now() / 1000) // Unix epoch - }; + const responseMessage = { + ...message, + id: responseMessageId, + parentId: parentId, + childrenIds: [], + files: undefined, + content: output !== undefined ? '' : content, + ...(output !== undefined ? { output } : {}), + timestamp: Math.floor(Date.now() / 1000) // Unix epoch + }; history.messages[responseMessageId] = responseMessage; history.currentId = responseMessageId; @@ -408,13 +408,16 @@ } await updateChat(); - } else { - // Edit response message - history.messages[messageId].originalContent = history.messages[messageId].content; - history.messages[messageId].content = content; - if (output !== undefined) { - history.messages[messageId].output = output; - } + } else { + // Edit response message + if (content !== undefined) { + history.messages[messageId].originalContent = history.messages[messageId].content; + history.messages[messageId].content = content; + } + if (output !== undefined) { + history.messages[messageId].output = output; + history.messages[messageId].content = ''; + } await updateChat(); } } diff --git a/src/lib/components/chat/Messages/ContentRenderer.svelte b/src/lib/components/chat/Messages/ContentRenderer.svelte index aeecd6fa7a..8310b39982 100644 --- a/src/lib/components/chat/Messages/ContentRenderer.svelte +++ b/src/lib/components/chat/Messages/ContentRenderer.svelte @@ -3,6 +3,7 @@ const i18n = getContext('i18n'); import Markdown from './Markdown.svelte'; + import StructuredOutputRenderer from './StructuredOutputRenderer.svelte'; import { artifactCode, chatId, @@ -68,6 +69,8 @@ export let id; export let content; + /** @type {import('./structuredOutput').OutputItem[]} */ + export let output = []; export let history; export let messageId; @@ -118,6 +121,39 @@ sourceIds = [...new Set(result)]; }; + /** @param {string} messageContent */ + const formatMessageContent = (messageContent) => + model?.info?.meta?.capabilities?.citations == false + ? replaceOutsideCode(messageContent, (segment) => + segment.replace(/\s*(\[(?:\d+(?:#[^,\]\s]+)?(?:,\s*\d+(?:#[^,\]\s]+)?)*)\])+/g, '') + ) + : messageContent; + + const markdownUpdateHandler = /** @type {any} */ (async ( + /** @type {{ lang?: string; text?: string }} */ token + ) => { + const { lang = '', text: code = '' } = token; + + if ( + ($settings?.detectArtifacts ?? true) && + (['html', 'svg'].includes(lang) || (lang === 'xml' && code.includes('svg'))) && + !$mobile && + $chatId + ) { + await tick(); + showArtifacts.set(true); + showControls.set(true); + } + }); + + const previewHandler = /** @type {any} */ (async (/** @type {string} */ value) => { + console.log('Preview', value); + await artifactCode.set(/** @type {any} */ (value)); + await showControls.set(true); + await showArtifacts.set(true); + await showEmbeds.set(false); + }); + const updateButtonPosition = (event) => { const buttonsContainerElement = document.getElementById(`floating-buttons-${id}`); if ( @@ -225,14 +261,28 @@
- {#if $settings?.renderMarkdownInAssistantMessages ?? true} + {#if output?.length} + + {:else if $settings?.renderMarkdownInAssistantMessages ?? true} - segment.replace(/\s*(\[(?:\d+(?:#[^,\]\s]+)?(?:,\s*\d+(?:#[^,\]\s]+)?)*)\])+/g, '') - ) - : content} + content={formatMessageContent(content)} {model} {save} {preview} @@ -243,27 +293,8 @@ {onSourceClick} {onTaskClick} {onSave} - onUpdate={async (token) => { - const { lang, text: code } = token; - - if ( - ($settings?.detectArtifacts ?? true) && - (['html', 'svg'].includes(lang) || (lang === 'xml' && code.includes('svg'))) && - !$mobile && - $chatId - ) { - await tick(); - showArtifacts.set(true); - showControls.set(true); - } - }} - onPreview={async (value) => { - console.log('Preview', value); - await artifactCode.set(value); - await showControls.set(true); - await showArtifacts.set(true); - await showEmbeds.set(false); - }} + onUpdate={markdownUpdateHandler} + onPreview={previewHandler} /> {:else} {@const extracted = extractDetailsBlocks(content)} diff --git a/src/lib/components/chat/Messages/ResponseMessage.svelte b/src/lib/components/chat/Messages/ResponseMessage.svelte index 8acec479c6..627ffbd05f 100644 --- a/src/lib/components/chat/Messages/ResponseMessage.svelte +++ b/src/lib/components/chat/Messages/ResponseMessage.svelte @@ -64,11 +64,17 @@ import StatusHistory from './ResponseMessage/StatusHistory.svelte'; import FullHeightIframe from '$lib/components/common/FullHeightIframe.svelte'; import OutputEditView from './OutputEditView.svelte'; + import { + getOutputText, + replaceOutputMessageText, + type OutputItem + } from './structuredOutput'; interface MessageType { id: string; model: string; content: string; + output?: OutputItem[]; files?: { type: string; url: string }[]; timestamp: number; role: string; @@ -127,7 +133,11 @@ if (source) { // Fast path: O(1) check on the fields that change most often (content during streaming, done at end) // Avoids 2x O(n) JSON.stringify calls that are always true during streaming anyway - if (message.content !== source.content || message.done !== source.done) { + if ( + message.content !== source.content || + message.done !== source.done || + message.output?.length !== source.output?.length + ) { message = structuredClone(source); } else if (!equal(message, source)) { // Slow path: full comparison for infrequent changes (sources, annotations, status, etc.) @@ -175,6 +185,8 @@ (model?.info?.meta?.capabilities?.status_updates ?? true) && statusEntries.length > 0 && !(statusEntries.at(-1)?.hidden ?? false); + $: visibleResponseContent = getOutputText(message.output) || removeAllDetails(message.content ?? ''); + $: hasResponseContent = Boolean((message.content ?? '').trim() || message.output?.length); let edit = false; let editedContent = ''; @@ -226,7 +238,8 @@ : $config?.audio?.tts?.voice); const speak = async () => { - if (!(message?.content ?? '').trim().length) { + const content = visibleResponseContent; + if (!content.trim().length) { toast.info($i18n.t('No content to speak')); return; } @@ -236,7 +249,6 @@ const { signal } = speakAbort; speaking = true; - const content = removeAllDetails(message.content); if ($config.audio.tts.engine === '') { let voices = []; @@ -370,17 +382,6 @@ return restoredContent; } - /** Extract plain text from output items for immediate display after edit. - * NOT a serialize_output port — just grabs text parts. Backend re-serializes - * the full rich content (with
blocks) on save. */ - function extractTextFromOutput(output: any[]): string { - return output - .filter((item) => item.type === 'message') - .flatMap((item) => (item.content ?? []).map((p: any) => p.text ?? '')) - .join('\n') - .trim(); - } - const editMessageHandler = async () => { edit = true; @@ -407,9 +408,7 @@ const editMessageConfirmHandler = async () => { if (editedOutput) { - // Structured edit: keep original rich content for immediate display; - // backend will re-derive content from output on save. - editMessage(message.id, { content: message.content, output: editedOutput }, false); + editMessage(message.id, { output: editedOutput }, false); } else { // Legacy text edit const messageContent = postprocessAfterEditing(editedContent ?? ''); @@ -425,7 +424,7 @@ const saveAsCopyHandler = async () => { if (editedOutput) { - editMessage(message.id, { content: message.content, output: editedOutput }); + editMessage(message.id, { output: editedOutput }); } else { const messageContent = postprocessAfterEditing(editedContent ?? ''); editMessage(message.id, { content: messageContent }); @@ -826,14 +825,15 @@ class="w-full flex flex-col relative {edit ? 'hidden' : ''}" id="response-content-container" > - {#if message.content === '' && !message.done && !message.error && !hasVisibleStatus} + {#if !hasResponseContent && !message.done && !message.error && !hasVisibleStatus} - {:else if message.content && message.error !== true} + {:else if hasResponseContent && message.error !== true} { - history.messages[message.id].content = history.messages[ - message.id - ].content.replace(raw, raw.replace(oldContent, newContent)); + const sourceMessage = history.messages[message.id]; + if (sourceMessage.output?.length) { + const updatedOutput = replaceOutputMessageText( + sourceMessage.output, + oldContent, + newContent + ); + if (updatedOutput !== sourceMessage.output) { + sourceMessage.output = updatedOutput; + } else { + sourceMessage.content = sourceMessage.content.replace( + raw, + raw.replace(oldContent, newContent) + ); + } + } else { + sourceMessage.content = sourceMessage.content.replace( + raw, + raw.replace(oldContent, newContent) + ); + } updateChat(); }} @@ -1033,7 +1051,7 @@ ? 'visible' : 'invisible group-hover:visible'} p-1.5 hover:bg-black/5 dark:hover:bg-white/5 rounded-lg dark:hover:text-white hover:text-black transition copy-response-button" on:click={() => { - copyToClipboard(message.content); + copyToClipboard(visibleResponseContent); }} > + import Collapsible from '$lib/components/common/Collapsible.svelte'; + import ToolCallDisplay from '$lib/components/common/ToolCallDisplay.svelte'; + import { settings } from '$lib/stores'; + + import Markdown from './Markdown.svelte'; + import ConsecutiveDetailsGroup from './Markdown/ConsecutiveDetailsGroup.svelte'; + import { + buildOutputDisplayItems, + type OutputDetailToken, + type OutputDisplayItem, + type OutputItem + } from './structuredOutput'; + + export let id = ''; + export let output: OutputItem[] = []; + export let done = true; + export let model = null; + export let save = false; + export let preview = false; + export let renderMarkdown = true; + export let editCodeBlock = true; + export let topPadding = false; + export let sourceIds: string[] = []; + export let formatMessageContent: (content: string) => string = (content) => content; + export let onSave: any = () => {}; + export let onSourceClick: any = () => {}; + export let onTaskClick: any = () => {}; + export let onUpdate: any = () => {}; + export let onPreview: any = () => {}; + + const getDetailTitle = (detailToken: OutputDetailToken): any => detailToken.summary; + const getDetailAttributes = (detailToken: OutputDetailToken): any => detailToken.attributes; + + $: displayItems = buildOutputDisplayItems(output) as OutputDisplayItem[]; + + +{#each displayItems as displayItem (displayItem.id)} + {#if displayItem.type === 'message'} + {#if renderMarkdown} + + {:else} +
{displayItem.text}
+ {/if} + {:else if displayItem.type === 'detail_group'} + +
+ {#each displayItem.tokens as detailToken, detailIndex} + {#if detailToken.attributes?.type === 'tool_calls'} + + {:else if detailToken.text?.length > 0} + +
+ +
+
+ {:else} + + {/if} + {/each} +
+
+ {:else} + {@const detailToken = displayItem.token} + {#if detailToken.attributes?.type === 'tool_calls'} + + {:else if detailToken.text?.length > 0} + +
+ +
+
+ {:else} + + {/if} + {/if} +{/each} diff --git a/src/lib/components/chat/Messages/structuredOutput.ts b/src/lib/components/chat/Messages/structuredOutput.ts new file mode 100644 index 0000000000..bc896a6f69 --- /dev/null +++ b/src/lib/components/chat/Messages/structuredOutput.ts @@ -0,0 +1,365 @@ +export type OutputContentPart = { + type?: string; + text?: unknown; + [key: string]: unknown; +}; + +export type OutputItem = { + type?: string; + id?: string; + call_id?: string; + name?: string; + status?: string; + arguments?: unknown; + content?: OutputContentPart[]; + summary?: OutputContentPart[]; + output?: OutputContentPart[]; + files?: unknown; + embeds?: unknown; + code?: string; + lang?: string; + duration?: number | string | null; + action?: Record; + actions?: Array>; + queries?: unknown[]; + [key: string]: unknown; +}; + +export type OutputDetailToken = { + summary: string; + text: string; + attributes: { + type: string; + id?: string; + name?: string; + done?: string; + duration?: string; + arguments?: string; + files?: string; + embeds?: string; + output?: string; + }; +}; + +export type OutputDisplayItem = + | { + type: 'message'; + id: string; + text: string; + } + | { + type: 'detail_single'; + id: string; + token: OutputDetailToken; + } + | { + type: 'detail_group'; + id: string; + tokens: OutputDetailToken[]; + }; + +const GROUPABLE_OUTPUT_TYPES = new Set([ + 'reasoning', + 'function_call', + 'open_webui:code_interpreter', + 'web_search_call', + 'file_search_call', + 'computer_call' +]); + +const OPENAI_TOOL_NAMES: Record = { + web_search_call: 'Web Search', + file_search_call: 'File Search', + computer_call: 'Computer Use' +}; + +function getTextFromParts(parts: OutputContentPart[] = []): string { + return parts + .map((part) => { + if (part?.text === undefined || part?.text === null) { + return ''; + } + return typeof part.text === 'string' ? part.text : String(part.text); + }) + .join(''); +} + +function stringifyAttribute(value: unknown): string { + if (value === undefined || value === null) { + return ''; + } + if (typeof value === 'string') { + return value; + } + try { + return JSON.stringify(value); + } catch { + return String(value); + } +} + +function isDoneStatus(status?: string): boolean { + return status === 'completed' || status === 'failed' || status === 'incomplete'; +} + +function getMessageText(item: OutputItem): string { + return getTextFromParts(item.content ?? []); +} + +function getReasoningText(item: OutputItem): string { + return getTextFromParts((item.summary ?? item.content) ?? []); +} + +function getToolResultText(item?: OutputItem): string { + return (item?.output ?? []) + .filter((part) => part?.type !== 'input_image') + .map((part) => { + if (part?.text === undefined || part?.text === null) { + return ''; + } + return typeof part.text === 'string' ? part.text : String(part.text); + }) + .join(''); +} + +function buildToolCallToken(item: OutputItem, toolOutputByCallId: Record) { + const callId = item.call_id ?? ''; + const resultItem = toolOutputByCallId[callId]; + const isDone = isDoneStatus(item.status) || !!resultItem; + + return { + summary: isDone ? 'Tool Executed' : 'Executing...', + text: getToolResultText(resultItem), + attributes: { + type: 'tool_calls', + id: callId, + name: item.name ?? '', + done: isDone ? 'true' : 'false', + arguments: stringifyAttribute(item.arguments ?? ''), + files: stringifyAttribute(resultItem?.files), + embeds: stringifyAttribute(resultItem?.embeds) + } + }; +} + +function buildReasoningToken(item: OutputItem, isLastItem: boolean) { + const duration = item.duration ?? ''; + const isDone = isDoneStatus(item.status) || item.duration !== undefined || !isLastItem; + const text = getReasoningText(item) + .split('\n') + .map((line) => (line.startsWith('>') ? line : `> ${line}`)) + .join('\n'); + + return { + summary: isDone ? `Thought for ${duration || 0} seconds` : 'Thinking...', + text, + attributes: { + type: 'reasoning', + done: isDone ? 'true' : 'false', + duration: String(duration) + } + }; +} + +function buildCodeInterpreterToken(item: OutputItem, isLastItem: boolean) { + const duration = item.duration ?? ''; + const isDone = isDoneStatus(item.status) || item.duration !== undefined || !isLastItem; + const code = item.code ?? ''; + const lang = item.lang ?? 'python'; + + return { + summary: isDone ? 'Analyzed' : 'Analyzing...', + text: code ? `\`\`\`${lang}\n${code}\n\`\`\`` : '', + attributes: { + type: 'code_interpreter', + done: isDone ? 'true' : 'false', + duration: String(duration), + output: stringifyAttribute(item.output) + } + }; +} + +function getOpenAIToolSummary(item: OutputItem): string { + if (item.type === 'web_search_call') { + const action = item.action ?? {}; + const actionType = action.type; + if (actionType === 'search') { + const queries = Array.isArray(action.queries) ? action.queries : []; + const query = typeof action.query === 'string' ? action.query : ''; + return queries.length ? `Search: ${queries.join(', ')}` : query ? `Search: ${query}` : ''; + } + if (actionType === 'open_page' && typeof action.url === 'string') { + return `Open page: ${action.url}`; + } + if (actionType === 'find_in_page' && typeof action.pattern === 'string') { + return `Find in page: ${action.pattern}`; + } + } + + if (item.type === 'file_search_call') { + const queries = item.queries ?? []; + return queries.length ? `Queries: ${queries.join(', ')}` : ''; + } + + if (item.type === 'computer_call') { + if (item.action?.type) { + return `Action: ${item.action.type}`; + } + if (Array.isArray(item.actions) && item.actions.length) { + return `Actions: ${item.actions.map((action) => action.type ?? '?').join(', ')}`; + } + } + + return ''; +} + +function buildOpenAIToolToken(item: OutputItem, isLastItem: boolean) { + const isDone = isDoneStatus(item.status) || !isLastItem; + return { + summary: isDone ? 'Tool Executed' : 'Executing...', + text: getOpenAIToolSummary(item), + attributes: { + type: 'tool_calls', + id: item.id ?? '', + name: OPENAI_TOOL_NAMES[item.type ?? ''] ?? item.type ?? '', + done: isDone ? 'true' : 'false', + arguments: '' + } + }; +} + +function buildDetailToken( + item: OutputItem, + isLastItem: boolean, + toolOutputByCallId: Record +): OutputDetailToken | null { + if (item.type === 'function_call') { + return buildToolCallToken(item, toolOutputByCallId); + } + if (item.type === 'reasoning') { + return buildReasoningToken(item, isLastItem); + } + if (item.type === 'open_webui:code_interpreter') { + return buildCodeInterpreterToken(item, isLastItem); + } + if (item.type && OPENAI_TOOL_NAMES[item.type]) { + return buildOpenAIToolToken(item, isLastItem); + } + return null; +} + +export function buildOutputDisplayItems(output: OutputItem[] = []): OutputDisplayItem[] { + const displayItems: OutputDisplayItem[] = []; + const currentDetailTokens: OutputDetailToken[] = []; + const toolOutputByCallId: Record = {}; + + for (const item of output) { + if (item?.type === 'function_call_output' && item.call_id) { + toolOutputByCallId[item.call_id] = item; + } + } + + const flushDetails = () => { + if (currentDetailTokens.length > 1) { + displayItems.push({ + type: 'detail_group', + id: `detail-group-${displayItems.length}`, + tokens: [...currentDetailTokens] + }); + } else if (currentDetailTokens.length === 1) { + displayItems.push({ + type: 'detail_single', + id: `detail-${displayItems.length}`, + token: currentDetailTokens[0] + }); + } + currentDetailTokens.length = 0; + }; + + output.forEach((item, index) => { + if (item?.type === 'function_call_output') { + return; + } + + if (item?.type && GROUPABLE_OUTPUT_TYPES.has(item.type)) { + const token = buildDetailToken(item, index === output.length - 1, toolOutputByCallId); + if (token) { + currentDetailTokens.push(token); + } + return; + } + + if (item?.type === 'message') { + const text = getMessageText(item); + if (text.trim()) { + flushDetails(); + displayItems.push({ + type: 'message', + id: item.id ?? `message-${index}`, + text + }); + } + return; + } + + const fallbackText = getMessageText(item); + if (fallbackText.trim()) { + flushDetails(); + displayItems.push({ + type: 'message', + id: item.id ?? `output-${index}`, + text: fallbackText + }); + } + }); + + flushDetails(); + return displayItems; +} + +export function getOutputText(output?: OutputItem[] | null): string { + return (output ?? []) + .filter((item) => item?.type === 'message') + .map(getMessageText) + .filter((text) => text.trim()) + .join('\n'); +} + +export function replaceOutputMessageText( + output: OutputItem[] = [], + oldContent: string, + newContent: string +): OutputItem[] { + if (!oldContent) { + return output; + } + + let replaced = false; + const nextOutput = output.map((item) => { + if (replaced || item?.type !== 'message' || !Array.isArray(item.content)) { + return item; + } + + const partIndex = item.content.findIndex( + (part) => typeof part.text === 'string' && part.text.includes(oldContent) + ); + if (partIndex === -1) { + return item; + } + + replaced = true; + const nextContent = [...item.content]; + const part = nextContent[partIndex]; + nextContent[partIndex] = { + ...part, + text: (part.text as string).replace(oldContent, newContent) + }; + + return { + ...item, + content: nextContent + }; + }); + + return replaced ? nextOutput : output; +}