Coverage for open_webui/utils/response.py: 10%
121 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1from numbers import Number
2from uuid import uuid4
4from open_webui.utils.json_codec import JSONCodec
5from open_webui.utils.misc import (
6 openai_chat_chunk_message_template,
7 openai_chat_completion_message_template,
8)
11# An honest ledger is worth more than a flattering one.
12# Let every cost here be counted true.
13def normalize_usage(usage: dict) -> dict:
14 """
15 Normalize usage statistics to standard format.
16 Handles OpenAI, Ollama, and llama.cpp formats.
18 Adds standardized token fields to the original data:
19 - input_tokens: Number of tokens in the prompt
20 - output_tokens: Number of tokens generated
21 - total_tokens: Sum of input and output tokens
22 """
23 if not usage:
24 return {}
26 # Map various field names to standard names
27 input_tokens = usage.get('input_tokens') or usage.get('prompt_tokens') or usage.get('prompt_eval_count')
28 if input_tokens is None:
29 input_tokens = int(usage.get('prompt_n') or 0) + int(usage.get('cache_n') or 0)
31 output_tokens = (
32 usage.get('output_tokens') # Already standard
33 or usage.get('completion_tokens') # OpenAI
34 or usage.get('eval_count') # Ollama
35 or usage.get('predicted_n') # llama.cpp
36 or 0
37 )
39 total_tokens = usage.get('total_tokens') or (input_tokens + output_tokens)
41 # Add standardized fields to original data
42 result = dict(usage)
43 result['input_tokens'] = int(input_tokens)
44 result['output_tokens'] = int(output_tokens)
45 result['total_tokens'] = int(total_tokens)
47 return result
50USAGE_TOKEN_KEYS = {
51 'input_tokens',
52 'output_tokens',
53 'total_tokens',
54}
56USAGE_COST_KEYS = {
57 'cost',
58 'total_cost',
59 'input_cost',
60 'output_cost',
61 'prompt_cost',
62 'completion_cost',
63}
65USAGE_SUMMABLE_KEYS = USAGE_TOKEN_KEYS | USAGE_COST_KEYS
67USAGE_DETAIL_KEYS = {
68 'prompt_tokens_details',
69 'completion_tokens_details',
70 'input_tokens_details',
71 'output_tokens_details',
72}
75def _is_numeric_usage_value(value) -> bool:
76 return isinstance(value, Number) and not isinstance(value, bool)
79def _merge_numeric_usage_map(current: dict | None, incoming: dict | None) -> dict:
80 current = current or {}
81 incoming = incoming or {}
82 result = {**current, **incoming}
84 for key in set(current) | set(incoming):
85 current_value = current.get(key, 0)
86 incoming_value = incoming.get(key, 0)
87 if isinstance(current_value, dict) or isinstance(incoming_value, dict):
88 result[key] = _merge_numeric_usage_map(
89 current_value if isinstance(current_value, dict) else {},
90 incoming_value if isinstance(incoming_value, dict) else {},
91 )
92 elif _is_numeric_usage_value(current_value) or _is_numeric_usage_value(incoming_value):
93 result[key] = (current_value if _is_numeric_usage_value(current_value) else 0) + (
94 incoming_value if _is_numeric_usage_value(incoming_value) else 0
95 )
97 return result
100def merge_usage(current: dict | None, incoming: dict | None) -> dict:
101 """
102 Merge usage payloads from multiple model calls into one cumulative usage dict.
104 Canonical token fields are additive; provider aliases keep the latest value.
105 """
106 current_usage = normalize_usage(current or {}) if current else {}
107 incoming_usage = normalize_usage(incoming or {}) if incoming else {}
109 if not incoming_usage:
110 return current_usage
111 if not current_usage:
112 return incoming_usage
114 result = {**current_usage, **incoming_usage}
116 for key in USAGE_SUMMABLE_KEYS:
117 if key in current_usage or key in incoming_usage:
118 current_value = current_usage.get(key, 0)
119 incoming_value = incoming_usage.get(key, 0)
120 if _is_numeric_usage_value(current_value) or _is_numeric_usage_value(incoming_value):
121 result[key] = (current_value if _is_numeric_usage_value(current_value) else 0) + (
122 incoming_value if _is_numeric_usage_value(incoming_value) else 0
123 )
125 for key in USAGE_DETAIL_KEYS:
126 if isinstance(current_usage.get(key), dict) or isinstance(incoming_usage.get(key), dict):
127 result[key] = _merge_numeric_usage_map(
128 current_usage.get(key) if isinstance(current_usage.get(key), dict) else {},
129 incoming_usage.get(key) if isinstance(incoming_usage.get(key), dict) else {},
130 )
132 result['prompt_tokens'] = (
133 incoming_usage.get('prompt_tokens')
134 or incoming_usage.get('input_tokens')
135 or current_usage.get('prompt_tokens', 0)
136 )
137 result['completion_tokens'] = (
138 incoming_usage.get('completion_tokens')
139 or incoming_usage.get('output_tokens')
140 or current_usage.get('completion_tokens', 0)
141 )
143 return result
146def convert_ollama_tool_call_to_openai(tool_calls: list) -> list:
147 openai_tool_calls = []
148 for tool_call in tool_calls:
149 function = tool_call.get('function', {})
150 openai_tool_call = {
151 'index': tool_call.get('index', function.get('index', 0)),
152 'id': tool_call.get('id', f'call_{str(uuid4())}'),
153 'type': 'function',
154 'function': {
155 'name': function.get('name', ''),
156 'arguments': JSONCodec.dumps(function.get('arguments', {})),
157 },
158 }
159 openai_tool_calls.append(openai_tool_call)
160 return openai_tool_calls
163def convert_ollama_usage_to_openai(data: dict) -> dict:
164 input_tokens = int(data.get('prompt_eval_count', 0))
165 output_tokens = int(data.get('eval_count', 0))
166 total_tokens = input_tokens + output_tokens
168 return {
169 # Standardized fields
170 'input_tokens': input_tokens,
171 'output_tokens': output_tokens,
172 'total_tokens': total_tokens,
173 # OpenAI-compatible fields (for backward compatibility)
174 'prompt_tokens': input_tokens,
175 'completion_tokens': output_tokens,
176 # Ollama-specific metrics
177 'response_token/s': (
178 round(
179 ((data.get('eval_count', 0) / (data.get('eval_duration', 0) / 10_000_000)) * 100),
180 2,
181 )
182 if data.get('eval_duration', 0) > 0
183 else 'N/A'
184 ),
185 'prompt_token/s': (
186 round(
187 ((data.get('prompt_eval_count', 0) / (data.get('prompt_eval_duration', 0) / 10_000_000)) * 100),
188 2,
189 )
190 if data.get('prompt_eval_duration', 0) > 0
191 else 'N/A'
192 ),
193 'total_duration': data.get('total_duration', 0),
194 'load_duration': data.get('load_duration', 0),
195 'prompt_eval_count': data.get('prompt_eval_count', 0),
196 'prompt_eval_duration': data.get('prompt_eval_duration', 0),
197 'eval_count': data.get('eval_count', 0),
198 'eval_duration': data.get('eval_duration', 0),
199 'approximate_total': (lambda s: f'{s // 3600}h{(s % 3600) // 60}m{s % 60}s')(
200 (data.get('total_duration', 0) or 0) // 1_000_000_000
201 ),
202 'completion_tokens_details': {
203 'reasoning_tokens': 0,
204 'accepted_prediction_tokens': 0,
205 'rejected_prediction_tokens': 0,
206 },
207 }
210def convert_response_ollama_to_openai(ollama_response: dict) -> dict:
211 model = ollama_response.get('model', 'ollama')
212 message_content = ollama_response.get('message', {}).get('content', '')
213 reasoning_content = ollama_response.get('message', {}).get('thinking', None)
214 tool_calls = ollama_response.get('message', {}).get('tool_calls', None)
215 openai_tool_calls = None
217 if tool_calls:
218 openai_tool_calls = convert_ollama_tool_call_to_openai(tool_calls)
220 data = ollama_response
222 usage = convert_ollama_usage_to_openai(data)
224 response = openai_chat_completion_message_template(
225 model, message_content, reasoning_content, openai_tool_calls, usage
226 )
227 return response
230async def convert_streaming_response_ollama_to_openai(ollama_streaming_response):
231 has_tool_calls = False
232 # All chunks in a single completion must share the same id (OpenAI spec).
233 completion_id = f'chatcmpl-{str(uuid4())}'
234 first = True
235 async for data in ollama_streaming_response.body_iterator:
236 data = JSONCodec.loads(data)
238 model = data.get('model', 'ollama')
239 message = data.get('message') or {}
240 message_content = message.get('content', None)
241 reasoning_content = message.get('thinking', None)
242 tool_calls = message.get('tool_calls', None)
243 openai_tool_calls = None
245 if tool_calls:
246 openai_tool_calls = convert_ollama_tool_call_to_openai(tool_calls)
247 has_tool_calls = True
249 done = data.get('done', False)
251 usage = None
252 if done:
253 usage = convert_ollama_usage_to_openai(data)
255 data = openai_chat_chunk_message_template(
256 model, message_content, reasoning_content, openai_tool_calls, usage, message_id=completion_id
257 )
259 # First chunk must carry delta.role (OpenAI spec).
260 if first:
261 data['choices'][0]['delta']['role'] = 'assistant'
262 first = False
264 if done and has_tool_calls:
265 data['choices'][0]['finish_reason'] = 'tool_calls'
267 line = f'data: {JSONCodec.dumps(data)}\n\n'
268 yield line
270 yield 'data: [DONE]\n\n'
273def convert_embedding_response_ollama_to_openai(response) -> dict:
274 """
275 Convert the response from Ollama embeddings endpoint to the OpenAI-compatible format.
277 Args:
278 response (dict): The response from the Ollama API,
279 e.g. {"embedding": [...], "model": "..."}
280 or {"embeddings": [{"embedding": [...], "index": 0}, ...], "model": "..."}
282 Returns:
283 dict: Response adapted to OpenAI's embeddings API format.
284 e.g. {
285 "object": "list",
286 "data": [
287 {"object": "embedding", "embedding": [...], "index": 0},
288 ...
289 ],
290 "model": "...",
291 }
292 """
293 # Ollama batch-style output from /api/embed
294 # Response format: {"embeddings": [[0.1, 0.2, ...], [0.3, 0.4, ...]], "model": "..."}
295 if isinstance(response, dict) and 'embeddings' in response:
296 openai_data = []
297 for i, emb in enumerate(response['embeddings']):
298 # /api/embed returns embeddings as plain float lists
299 if isinstance(emb, list):
300 openai_data.append(
301 {
302 'object': 'embedding',
303 'embedding': emb,
304 'index': i,
305 }
306 )
307 # Also handle dict format for robustness
308 elif isinstance(emb, dict):
309 openai_data.append(
310 {
311 'object': 'embedding',
312 'embedding': emb.get('embedding'),
313 'index': emb.get('index', i),
314 }
315 )
316 return {
317 'object': 'list',
318 'data': openai_data,
319 'model': response.get('model'),
320 }
321 # Ollama single output
322 elif isinstance(response, dict) and 'embedding' in response:
323 return {
324 'object': 'list',
325 'data': [
326 {
327 'object': 'embedding',
328 'embedding': response['embedding'],
329 'index': 0,
330 }
331 ],
332 'model': response.get('model'),
333 }
334 # Already OpenAI-compatible?
335 elif isinstance(response, dict) and 'data' in response and isinstance(response['data'], list):
336 return response
338 # Fallback: return as is if unrecognized
339 return response