Coverage for open_webui/utils/response.py: 10%

121 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1from numbers import Number 

2from uuid import uuid4 

3 

4from open_webui.utils.json_codec import JSONCodec 

5from open_webui.utils.misc import ( 

6 openai_chat_chunk_message_template, 

7 openai_chat_completion_message_template, 

8) 

9 

10 

11# An honest ledger is worth more than a flattering one. 

12# Let every cost here be counted true. 

13def normalize_usage(usage: dict) -> dict: 

14 """ 

15 Normalize usage statistics to standard format. 

16 Handles OpenAI, Ollama, and llama.cpp formats. 

17 

18 Adds standardized token fields to the original data: 

19 - input_tokens: Number of tokens in the prompt 

20 - output_tokens: Number of tokens generated 

21 - total_tokens: Sum of input and output tokens 

22 """ 

23 if not usage: 

24 return {} 

25 

26 # Map various field names to standard names 

27 input_tokens = usage.get('input_tokens') or usage.get('prompt_tokens') or usage.get('prompt_eval_count') 

28 if input_tokens is None: 

29 input_tokens = int(usage.get('prompt_n') or 0) + int(usage.get('cache_n') or 0) 

30 

31 output_tokens = ( 

32 usage.get('output_tokens') # Already standard 

33 or usage.get('completion_tokens') # OpenAI 

34 or usage.get('eval_count') # Ollama 

35 or usage.get('predicted_n') # llama.cpp 

36 or 0 

37 ) 

38 

39 total_tokens = usage.get('total_tokens') or (input_tokens + output_tokens) 

40 

41 # Add standardized fields to original data 

42 result = dict(usage) 

43 result['input_tokens'] = int(input_tokens) 

44 result['output_tokens'] = int(output_tokens) 

45 result['total_tokens'] = int(total_tokens) 

46 

47 return result 

48 

49 

50USAGE_TOKEN_KEYS = { 

51 'input_tokens', 

52 'output_tokens', 

53 'total_tokens', 

54} 

55 

56USAGE_COST_KEYS = { 

57 'cost', 

58 'total_cost', 

59 'input_cost', 

60 'output_cost', 

61 'prompt_cost', 

62 'completion_cost', 

63} 

64 

65USAGE_SUMMABLE_KEYS = USAGE_TOKEN_KEYS | USAGE_COST_KEYS 

66 

67USAGE_DETAIL_KEYS = { 

68 'prompt_tokens_details', 

69 'completion_tokens_details', 

70 'input_tokens_details', 

71 'output_tokens_details', 

72} 

73 

74 

75def _is_numeric_usage_value(value) -> bool: 

76 return isinstance(value, Number) and not isinstance(value, bool) 

77 

78 

79def _merge_numeric_usage_map(current: dict | None, incoming: dict | None) -> dict: 

80 current = current or {} 

81 incoming = incoming or {} 

82 result = {**current, **incoming} 

83 

84 for key in set(current) | set(incoming): 

85 current_value = current.get(key, 0) 

86 incoming_value = incoming.get(key, 0) 

87 if isinstance(current_value, dict) or isinstance(incoming_value, dict): 

88 result[key] = _merge_numeric_usage_map( 

89 current_value if isinstance(current_value, dict) else {}, 

90 incoming_value if isinstance(incoming_value, dict) else {}, 

91 ) 

92 elif _is_numeric_usage_value(current_value) or _is_numeric_usage_value(incoming_value): 

93 result[key] = (current_value if _is_numeric_usage_value(current_value) else 0) + ( 

94 incoming_value if _is_numeric_usage_value(incoming_value) else 0 

95 ) 

96 

97 return result 

98 

99 

100def merge_usage(current: dict | None, incoming: dict | None) -> dict: 

101 """ 

102 Merge usage payloads from multiple model calls into one cumulative usage dict. 

103 

104 Canonical token fields are additive; provider aliases keep the latest value. 

105 """ 

106 current_usage = normalize_usage(current or {}) if current else {} 

107 incoming_usage = normalize_usage(incoming or {}) if incoming else {} 

108 

109 if not incoming_usage: 

110 return current_usage 

111 if not current_usage: 

112 return incoming_usage 

113 

114 result = {**current_usage, **incoming_usage} 

115 

116 for key in USAGE_SUMMABLE_KEYS: 

117 if key in current_usage or key in incoming_usage: 

118 current_value = current_usage.get(key, 0) 

119 incoming_value = incoming_usage.get(key, 0) 

120 if _is_numeric_usage_value(current_value) or _is_numeric_usage_value(incoming_value): 

121 result[key] = (current_value if _is_numeric_usage_value(current_value) else 0) + ( 

122 incoming_value if _is_numeric_usage_value(incoming_value) else 0 

123 ) 

124 

125 for key in USAGE_DETAIL_KEYS: 

126 if isinstance(current_usage.get(key), dict) or isinstance(incoming_usage.get(key), dict): 

127 result[key] = _merge_numeric_usage_map( 

128 current_usage.get(key) if isinstance(current_usage.get(key), dict) else {}, 

129 incoming_usage.get(key) if isinstance(incoming_usage.get(key), dict) else {}, 

130 ) 

131 

132 result['prompt_tokens'] = ( 

133 incoming_usage.get('prompt_tokens') 

134 or incoming_usage.get('input_tokens') 

135 or current_usage.get('prompt_tokens', 0) 

136 ) 

137 result['completion_tokens'] = ( 

138 incoming_usage.get('completion_tokens') 

139 or incoming_usage.get('output_tokens') 

140 or current_usage.get('completion_tokens', 0) 

141 ) 

142 

143 return result 

144 

145 

146def convert_ollama_tool_call_to_openai(tool_calls: list) -> list: 

147 openai_tool_calls = [] 

148 for tool_call in tool_calls: 

149 function = tool_call.get('function', {}) 

150 openai_tool_call = { 

151 'index': tool_call.get('index', function.get('index', 0)), 

152 'id': tool_call.get('id', f'call_{str(uuid4())}'), 

153 'type': 'function', 

154 'function': { 

155 'name': function.get('name', ''), 

156 'arguments': JSONCodec.dumps(function.get('arguments', {})), 

157 }, 

158 } 

159 openai_tool_calls.append(openai_tool_call) 

160 return openai_tool_calls 

161 

162 

163def convert_ollama_usage_to_openai(data: dict) -> dict: 

164 input_tokens = int(data.get('prompt_eval_count', 0)) 

165 output_tokens = int(data.get('eval_count', 0)) 

166 total_tokens = input_tokens + output_tokens 

167 

168 return { 

169 # Standardized fields 

170 'input_tokens': input_tokens, 

171 'output_tokens': output_tokens, 

172 'total_tokens': total_tokens, 

173 # OpenAI-compatible fields (for backward compatibility) 

174 'prompt_tokens': input_tokens, 

175 'completion_tokens': output_tokens, 

176 # Ollama-specific metrics 

177 'response_token/s': ( 

178 round( 

179 ((data.get('eval_count', 0) / (data.get('eval_duration', 0) / 10_000_000)) * 100), 

180 2, 

181 ) 

182 if data.get('eval_duration', 0) > 0 

183 else 'N/A' 

184 ), 

185 'prompt_token/s': ( 

186 round( 

187 ((data.get('prompt_eval_count', 0) / (data.get('prompt_eval_duration', 0) / 10_000_000)) * 100), 

188 2, 

189 ) 

190 if data.get('prompt_eval_duration', 0) > 0 

191 else 'N/A' 

192 ), 

193 'total_duration': data.get('total_duration', 0), 

194 'load_duration': data.get('load_duration', 0), 

195 'prompt_eval_count': data.get('prompt_eval_count', 0), 

196 'prompt_eval_duration': data.get('prompt_eval_duration', 0), 

197 'eval_count': data.get('eval_count', 0), 

198 'eval_duration': data.get('eval_duration', 0), 

199 'approximate_total': (lambda s: f'{s // 3600}h{(s % 3600) // 60}m{s % 60}s')( 

200 (data.get('total_duration', 0) or 0) // 1_000_000_000 

201 ), 

202 'completion_tokens_details': { 

203 'reasoning_tokens': 0, 

204 'accepted_prediction_tokens': 0, 

205 'rejected_prediction_tokens': 0, 

206 }, 

207 } 

208 

209 

210def convert_response_ollama_to_openai(ollama_response: dict) -> dict: 

211 model = ollama_response.get('model', 'ollama') 

212 message_content = ollama_response.get('message', {}).get('content', '') 

213 reasoning_content = ollama_response.get('message', {}).get('thinking', None) 

214 tool_calls = ollama_response.get('message', {}).get('tool_calls', None) 

215 openai_tool_calls = None 

216 

217 if tool_calls: 

218 openai_tool_calls = convert_ollama_tool_call_to_openai(tool_calls) 

219 

220 data = ollama_response 

221 

222 usage = convert_ollama_usage_to_openai(data) 

223 

224 response = openai_chat_completion_message_template( 

225 model, message_content, reasoning_content, openai_tool_calls, usage 

226 ) 

227 return response 

228 

229 

230async def convert_streaming_response_ollama_to_openai(ollama_streaming_response): 

231 has_tool_calls = False 

232 # All chunks in a single completion must share the same id (OpenAI spec). 

233 completion_id = f'chatcmpl-{str(uuid4())}' 

234 first = True 

235 async for data in ollama_streaming_response.body_iterator: 

236 data = JSONCodec.loads(data) 

237 

238 model = data.get('model', 'ollama') 

239 message = data.get('message') or {} 

240 message_content = message.get('content', None) 

241 reasoning_content = message.get('thinking', None) 

242 tool_calls = message.get('tool_calls', None) 

243 openai_tool_calls = None 

244 

245 if tool_calls: 

246 openai_tool_calls = convert_ollama_tool_call_to_openai(tool_calls) 

247 has_tool_calls = True 

248 

249 done = data.get('done', False) 

250 

251 usage = None 

252 if done: 

253 usage = convert_ollama_usage_to_openai(data) 

254 

255 data = openai_chat_chunk_message_template( 

256 model, message_content, reasoning_content, openai_tool_calls, usage, message_id=completion_id 

257 ) 

258 

259 # First chunk must carry delta.role (OpenAI spec). 

260 if first: 

261 data['choices'][0]['delta']['role'] = 'assistant' 

262 first = False 

263 

264 if done and has_tool_calls: 

265 data['choices'][0]['finish_reason'] = 'tool_calls' 

266 

267 line = f'data: {JSONCodec.dumps(data)}\n\n' 

268 yield line 

269 

270 yield 'data: [DONE]\n\n' 

271 

272 

273def convert_embedding_response_ollama_to_openai(response) -> dict: 

274 """ 

275 Convert the response from Ollama embeddings endpoint to the OpenAI-compatible format. 

276 

277 Args: 

278 response (dict): The response from the Ollama API, 

279 e.g. {"embedding": [...], "model": "..."} 

280 or {"embeddings": [{"embedding": [...], "index": 0}, ...], "model": "..."} 

281 

282 Returns: 

283 dict: Response adapted to OpenAI's embeddings API format. 

284 e.g. { 

285 "object": "list", 

286 "data": [ 

287 {"object": "embedding", "embedding": [...], "index": 0}, 

288 ... 

289 ], 

290 "model": "...", 

291 } 

292 """ 

293 # Ollama batch-style output from /api/embed 

294 # Response format: {"embeddings": [[0.1, 0.2, ...], [0.3, 0.4, ...]], "model": "..."} 

295 if isinstance(response, dict) and 'embeddings' in response: 

296 openai_data = [] 

297 for i, emb in enumerate(response['embeddings']): 

298 # /api/embed returns embeddings as plain float lists 

299 if isinstance(emb, list): 

300 openai_data.append( 

301 { 

302 'object': 'embedding', 

303 'embedding': emb, 

304 'index': i, 

305 } 

306 ) 

307 # Also handle dict format for robustness 

308 elif isinstance(emb, dict): 

309 openai_data.append( 

310 { 

311 'object': 'embedding', 

312 'embedding': emb.get('embedding'), 

313 'index': emb.get('index', i), 

314 } 

315 ) 

316 return { 

317 'object': 'list', 

318 'data': openai_data, 

319 'model': response.get('model'), 

320 } 

321 # Ollama single output 

322 elif isinstance(response, dict) and 'embedding' in response: 

323 return { 

324 'object': 'list', 

325 'data': [ 

326 { 

327 'object': 'embedding', 

328 'embedding': response['embedding'], 

329 'index': 0, 

330 } 

331 ], 

332 'model': response.get('model'), 

333 } 

334 # Already OpenAI-compatible? 

335 elif isinstance(response, dict) and 'data' in response and isinstance(response['data'], list): 

336 return response 

337 

338 # Fallback: return as is if unrecognized 

339 return response