Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py: 17%

276 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2OpenAI Passthrough Logging Handler 

3 

4Handles cost tracking and logging for OpenAI passthrough endpoints, specifically /chat/completions. 

5""" 

6 

7from collections.abc import Mapping, Sequence 

8from datetime import datetime 

9from typing import Final 

10from urllib.parse import urlparse 

11 

12import httpx 

13 

14import litellm 

15from litellm._logging import verbose_proxy_logger 

16from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj 

17from litellm.litellm_core_utils.litellm_logging import ( 

18 get_standard_logging_object_payload, 

19) 

20from litellm.litellm_core_utils.token_counter import high_detail_image_token_upper_bound 

21from litellm.llms.openai.openai import OpenAIConfig 

22from litellm.llms.openai.openai import OpenAIConfig as OpenAIConfigType 

23from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig 

24from litellm.proxy._types import PassThroughEndpointLoggingTypedDict 

25from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough_logging_handler import ( 

26 BasePassthroughLoggingHandler, 

27) 

28from litellm.proxy.pass_through_endpoints.success_handler import ( 

29 PassThroughEndpointLogging, 

30) 

31from litellm.types.llms.openai import ResponsesAPIResponse 

32from litellm.types.passthrough_endpoints.pass_through_endpoints import ( 

33 EndpointType, 

34 PassthroughStandardLoggingPayload, 

35) 

36from litellm.types.utils import EmbeddingResponse, ImageResponse, LlmProviders, PassthroughCallTypes 

37from litellm.utils import ModelResponse, TextCompletionResponse, convert_to_model_response_object 

38 

39# Hostnames that route to OpenAI-compatible APIs. 

40# 

41# `api.openai.com` is OpenAI proper. The two Azure domains below are *shared by 

42# every Azure Cognitive Service* (Speech, Vision, Language, ...), not just Azure 

43# OpenAI: `openai.azure.com` is the classic Azure OpenAI domain, while 

44# `cognitiveservices.azure.com` is used by newer "Azure AI Foundry" / 

45# Cognitive Services-hosted Azure OpenAI deployments. Because the hostname alone 

46# cannot tell Azure OpenAI apart from the other Cognitive Services on those 

47# domains, requests there must additionally carry an OpenAI-style path segment. 

48_OPENAI_HOSTNAMES: Final = ("api.openai.com",) 

49_AZURE_OPENAI_HOSTNAMES: Final = ("openai.azure.com", "cognitiveservices.azure.com") 

50# Path markers that identify an Azure request as Azure OpenAI rather than Speech 

51# / Vision / Language / ... `/openai/` is the native Azure OpenAI path prefix; 

52# `/v1/` is the OpenAI-v1 surface used by LiteLLM's pass-through routing. Other 

53# Cognitive Services use service-named prefixes and versions like `/v3.1/`, 

54# `/v1.0/`, so they do not collide with these markers. 

55_AZURE_OPENAI_PATH_MARKERS: Final = ("/openai/", "/v1/") 

56 

57 

58def _hostname_matches(hostname: str, suffixes: tuple) -> bool: 

59 """True if hostname equals one of `suffixes` or is a subdomain of it. 

60 

61 Uses suffix matching (not a bare substring test) so look-alikes such as 

62 `cognitiveservices.azure.com.attacker.example` are not accepted. 

63 """ 

64 return any(hostname == suffix or hostname.endswith("." + suffix) for suffix in suffixes) 

65 

66 

67def _is_openai_compatible_host(hostname: str | None) -> bool: 

68 """True if the hostname is OpenAI proper or one of the Azure OpenAI domains. 

69 

70 Hostname-only check, kept for the route-level helpers that additionally 

71 require a specific OpenAI path (e.g. `/v1/chat/completions`). When only the 

72 hostname would otherwise gate dispatch, use `_is_openai_compatible_url` so 

73 non-OpenAI Azure Cognitive Services on the shared domains are excluded. 

74 """ 

75 if not hostname: 

76 return False 

77 return _hostname_matches(hostname, _OPENAI_HOSTNAMES) or _hostname_matches(hostname, _AZURE_OPENAI_HOSTNAMES) 

78 

79 

80def _is_openai_compatible_url(url_route: str | None) -> bool: 

81 """True if the URL targets an OpenAI-compatible API surface. 

82 

83 For the shared Azure Cognitive Services domains we additionally require an 

84 OpenAI-style path segment (`/openai/` or `/v1/`) so non-OpenAI Azure services 

85 (Speech, Vision, Language, ...) on the same domain are not misclassified as 

86 OpenAI routes. 

87 """ 

88 if not url_route: 

89 return False 

90 parsed_url: Final = urlparse(url_route) 

91 hostname: Final = parsed_url.hostname 

92 if not hostname: 

93 return False 

94 if _hostname_matches(hostname, _OPENAI_HOSTNAMES): 

95 return True 

96 if _hostname_matches(hostname, _AZURE_OPENAI_HOSTNAMES): 

97 return any(marker in parsed_url.path for marker in _AZURE_OPENAI_PATH_MARKERS) 

98 return False 

99 

100 

101def _is_remote_high_detail_image(part: object) -> bool: 

102 if not isinstance(part, Mapping) or part.get("type") != "image_url": 

103 return False 

104 image_url: Final = part.get("image_url") 

105 if not isinstance(image_url, Mapping): 

106 return False 

107 url: Final = image_url.get("url") 

108 return ( 

109 isinstance(url, str) and url.lower().startswith(("http://", "https://")) and image_url.get("detail") == "high" 

110 ) 

111 

112 

113def _content_parts(message: Mapping[str, object]) -> Sequence[object]: 

114 content: Final = message.get("content") 

115 return content if isinstance(content, list) else () 

116 

117 

118def _without_remote_high_detail_images(message: Mapping[str, object]) -> Mapping[str, object]: 

119 if not isinstance(message.get("content"), list): 

120 return message 

121 kept_parts: Final = [ # mutable-ok: token_counter reads message content only when it is a list 

122 part for part in _content_parts(message) if not _is_remote_high_detail_image(part) 

123 ] 

124 return {**message, "content": kept_parts} # mutable-ok: token_counter rejects any message that is not a dict 

125 

126 

127def count_relayed_prompt_tokens(model: str, messages: Sequence[Mapping[str, object]] | None) -> int: 

128 if messages is None: 

129 return 0 

130 remote_high_detail_images: Final = sum( 

131 1 for message in messages for part in _content_parts(message) if _is_remote_high_detail_image(part) 

132 ) 

133 local_messages: Final = [ # mutable-ok: token_counter takes a list of messages 

134 _without_remote_high_detail_images(message) for message in messages 

135 ] 

136 return ( 

137 litellm.token_counter(model=model, messages=local_messages) 

138 + high_detail_image_token_upper_bound() * remote_high_detail_images 

139 ) 

140 

141 

142class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): 

143 """ 

144 OpenAI-specific passthrough logging handler that provides cost tracking for /chat/completions endpoints. 

145 """ 

146 

147 @property 

148 def llm_provider_name(self) -> LlmProviders: 

149 return LlmProviders.OPENAI 

150 

151 def get_provider_config(self, model: str) -> OpenAIConfigType: 

152 """Get OpenAI provider configuration for the given model.""" 

153 return OpenAIConfig() 

154 

155 @staticmethod 

156 def is_openai_chat_completions_route(url_route: str) -> bool: 

157 """Check if the URL route is an OpenAI chat completions endpoint.""" 

158 if not url_route: 

159 return False 

160 parsed_url: Final = urlparse(url_route) 

161 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/chat/completions" in parsed_url.path 

162 

163 @staticmethod 

164 def is_openai_image_generation_route(url_route: str) -> bool: 

165 """Check if the URL route is an OpenAI image generation endpoint.""" 

166 if not url_route: 

167 return False 

168 parsed_url: Final = urlparse(url_route) 

169 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/images/generations" in parsed_url.path 

170 

171 @staticmethod 

172 def is_openai_image_editing_route(url_route: str) -> bool: 

173 """Check if the URL route is an OpenAI image editing endpoint.""" 

174 if not url_route: 

175 return False 

176 parsed_url: Final = urlparse(url_route) 

177 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/images/edits" in parsed_url.path 

178 

179 @staticmethod 

180 def is_openai_responses_route(url_route: str) -> bool: 

181 """Check if the URL route is an OpenAI responses API endpoint.""" 

182 if not url_route: 

183 return False 

184 parsed_url: Final = urlparse(url_route) 

185 return _is_openai_compatible_host(parsed_url.hostname) and ( 

186 "/v1/responses" in parsed_url.path or "/responses" in parsed_url.path 

187 ) 

188 

189 @staticmethod 

190 def is_openai_embeddings_route(url_route: str) -> bool: 

191 """Check if the URL route is an OpenAI embeddings endpoint.""" 

192 if not url_route: 

193 return False 

194 parsed_url: Final = urlparse(url_route) 

195 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/embeddings" in parsed_url.path 

196 

197 def _get_user_from_metadata( 

198 self, 

199 passthrough_logging_payload: PassthroughStandardLoggingPayload, 

200 ) -> str | None: 

201 """Extract user information from passthrough logging payload.""" 

202 request_body: Final = passthrough_logging_payload.get("request_body") 

203 if request_body: 

204 return request_body.get("user") 

205 return None 

206 

207 @staticmethod 

208 def _calculate_image_generation_cost( 

209 model: str, 

210 response_body: dict, 

211 request_body: dict, 

212 ) -> float: 

213 """Calculate cost for OpenAI image generation.""" 

214 try: 

215 # Extract parameters from request 

216 n = request_body.get("n", 1) 

217 try: 

218 n = int(n) 

219 except Exception: 

220 n = 1 

221 size: Final = request_body.get("size", "1024x1024") 

222 quality: Final = request_body.get("quality", None) 

223 

224 # Use LiteLLM's default image cost calculator 

225 from litellm.cost_calculator import default_image_cost_calculator 

226 

227 cost: Final = default_image_cost_calculator( 

228 model=model, 

229 custom_llm_provider="openai", 

230 quality=quality, 

231 n=n, 

232 size=size, 

233 optional_params=request_body, 

234 ) 

235 

236 return cost 

237 except Exception as e: 

238 verbose_proxy_logger.warning("Error calculating image generation cost: %s", e) 

239 return 0.0 

240 

241 @staticmethod 

242 def _calculate_image_editing_cost( 

243 model: str, 

244 response_body: dict, 

245 request_body: dict, 

246 ) -> float: 

247 """Calculate cost for OpenAI image editing.""" 

248 try: 

249 # Extract parameters from request 

250 n = request_body.get("n", 1) 

251 # Image edit typically uses multipart/form-data (because of files), so all fields arrive as strings (e.g., n = "1"). 

252 try: 

253 n = int(n) 

254 except Exception: 

255 n = 1 

256 size: Final = request_body.get("size", "1024x1024") 

257 

258 # Use LiteLLM's default image cost calculator 

259 from litellm.cost_calculator import default_image_cost_calculator 

260 

261 cost: Final = default_image_cost_calculator( 

262 model=model, 

263 custom_llm_provider="openai", 

264 quality=None, # Image editing doesn't have quality parameter 

265 n=n, 

266 size=size, 

267 optional_params=request_body, 

268 ) 

269 

270 return cost 

271 except Exception as e: 

272 verbose_proxy_logger.warning("Error calculating image editing cost: %s", e) 

273 return 0.0 

274 

275 @staticmethod 

276 def _calculate_embeddings_cost( 

277 litellm_model_response: EmbeddingResponse, 

278 model: str, 

279 custom_llm_provider: str, 

280 ) -> float: 

281 try: 

282 return litellm.completion_cost( 

283 completion_response=litellm_model_response, 

284 model=model, 

285 custom_llm_provider=custom_llm_provider, 

286 call_type="aembedding", 

287 ) 

288 except Exception as e: # noqa: BLE001 # completion_cost raises bare Exception for unmapped models; cost failure must never drop the spend log 

289 verbose_proxy_logger.warning( 

290 "Error calculating embeddings cost for model %s, logging spend with cost 0: %s", model, e 

291 ) 

292 return 0.0 

293 

294 @staticmethod 

295 def _build_responses_api_response_and_cost( 

296 model: str, 

297 httpx_response: httpx.Response, 

298 logging_obj: LiteLLMLoggingObj, 

299 custom_llm_provider: str, 

300 ) -> tuple[ResponsesAPIResponse, float]: 

301 """Transform a Responses API raw response into a ResponsesAPIResponse 

302 and compute its cost. 

303 

304 The Responses API has a different on-the-wire shape from chat 

305 completions (`output: [...]` instead of `choices: [...]`), so the 

306 chat-completions `transform_response` raises KeyError 'choices' on 

307 a Responses payload. Use the dedicated Responses-API transformer 

308 (`OpenAIResponsesAPIConfig.transform_response_api_response`) here. 

309 

310 Returns (litellm_model_response, response_cost) — symmetric with the 

311 chat-completions branch which produces the same two values inline, 

312 and analogous to the image branches' `_calculate_image_*_cost` helpers 

313 (which return cost only because the image-response object is trivial 

314 to build inline; the Responses payload needs a real transformer). 

315 """ 

316 responses_config: Final = OpenAIResponsesAPIConfig() 

317 litellm_model_response: Final = responses_config.transform_response_api_response( 

318 model=model, 

319 raw_response=httpx_response, 

320 logging_obj=logging_obj, 

321 ) 

322 response_cost: Final = litellm.completion_cost( 

323 completion_response=litellm_model_response, 

324 model=model, 

325 custom_llm_provider=custom_llm_provider, 

326 call_type="responses", 

327 ) 

328 return litellm_model_response, response_cost 

329 

330 @staticmethod 

331 def openai_passthrough_handler( 

332 httpx_response: httpx.Response, 

333 response_body: dict, 

334 logging_obj: LiteLLMLoggingObj, 

335 url_route: str, 

336 result: str, 

337 start_time: datetime, 

338 end_time: datetime, 

339 cache_hit: bool, 

340 request_body: dict, 

341 **kwargs, 

342 ) -> PassThroughEndpointLoggingTypedDict: 

343 """ 

344 Handle OpenAI passthrough logging with cost tracking for chat completions, 

345 embeddings, image generation, image editing, and responses API. 

346 """ 

347 is_chat_completions: Final = OpenAIPassthroughLoggingHandler.is_openai_chat_completions_route(url_route) 

348 is_embeddings: Final = OpenAIPassthroughLoggingHandler.is_openai_embeddings_route(url_route) 

349 is_image_generation: Final = OpenAIPassthroughLoggingHandler.is_openai_image_generation_route(url_route) 

350 is_image_editing: Final = OpenAIPassthroughLoggingHandler.is_openai_image_editing_route(url_route) 

351 is_responses: Final = OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route) 

352 

353 if not (is_chat_completions or is_embeddings or is_image_generation or is_image_editing or is_responses): 

354 return { 

355 "result": None, 

356 "kwargs": kwargs, 

357 } 

358 

359 model: Final = request_body.get("model", response_body.get("model", "")) 

360 if not model: 

361 verbose_proxy_logger.warning("No model found in request or response for OpenAI passthrough cost tracking") 

362 base_handler = OpenAIPassthroughLoggingHandler() 

363 return base_handler.passthrough_chat_handler( 

364 httpx_response=httpx_response, 

365 response_body=response_body, 

366 logging_obj=logging_obj, 

367 url_route=url_route, 

368 result=result, 

369 start_time=start_time, 

370 end_time=end_time, 

371 cache_hit=cache_hit, 

372 request_body=request_body, 

373 **kwargs, 

374 ) 

375 

376 try: 

377 response_cost = 0.0 

378 litellm_model_response: ( 

379 ModelResponse | TextCompletionResponse | EmbeddingResponse | ImageResponse | ResponsesAPIResponse | None 

380 ) = None 

381 handler_instance: Final = OpenAIPassthroughLoggingHandler() 

382 

383 custom_llm_provider: Final = kwargs.get("custom_llm_provider", "openai") 

384 

385 if is_chat_completions: 

386 # Handle chat completions with existing logic 

387 provider_config: Final = handler_instance.get_provider_config(model=model) 

388 # Preserve existing litellm_params to maintain metadata tags 

389 existing_litellm_params: Final = kwargs.get("litellm_params", {}) or {} 

390 litellm_model_response = provider_config.transform_response( 

391 raw_response=httpx_response, 

392 model_response=litellm.ModelResponse(), 

393 model=model, 

394 messages=request_body.get("messages", []), 

395 logging_obj=logging_obj, 

396 optional_params=request_body.get("optional_params", {}), 

397 api_key="", 

398 request_data=request_body, 

399 encoding=getattr(litellm, "encoding", None), 

400 json_mode=request_body.get("response_format", {}).get("type") == "json_object", 

401 litellm_params=existing_litellm_params, 

402 ) 

403 

404 # Calculate cost using LiteLLM's cost calculator 

405 response_cost = litellm.completion_cost( 

406 completion_response=litellm_model_response, 

407 model=model, 

408 custom_llm_provider=custom_llm_provider, 

409 ) 

410 elif is_embeddings: 

411 litellm_model_response = convert_to_model_response_object( 

412 response_object=response_body, 

413 model_response_object=EmbeddingResponse(), 

414 response_type="embedding", 

415 ) 

416 response_cost = OpenAIPassthroughLoggingHandler._calculate_embeddings_cost( 

417 litellm_model_response=litellm_model_response, 

418 model=model, 

419 custom_llm_provider=custom_llm_provider, 

420 ) 

421 litellm_model_response._hidden_params["response_cost"] = response_cost 

422 elif is_image_generation: 

423 # Handle image generation cost calculation 

424 response_cost = OpenAIPassthroughLoggingHandler._calculate_image_generation_cost( 

425 model=model, 

426 response_body=response_body, 

427 request_body=request_body, 

428 ) 

429 # Mark call type for downstream image-aware logic/metrics 

430 try: 

431 logging_obj.call_type = PassthroughCallTypes.passthrough_image_generation.value 

432 except Exception: 

433 pass 

434 # Create a simple response object for logging 

435 litellm_model_response = ImageResponse( 

436 data=response_body.get("data", []), 

437 model=model, 

438 ) 

439 # Set the calculated cost in _hidden_params to prevent recalculation 

440 if not hasattr(litellm_model_response, "_hidden_params"): 

441 litellm_model_response._hidden_params = {} 

442 litellm_model_response._hidden_params["response_cost"] = response_cost 

443 elif is_image_editing: 

444 # Handle image editing cost calculation 

445 response_cost = OpenAIPassthroughLoggingHandler._calculate_image_editing_cost( 

446 model=model, 

447 response_body=response_body, 

448 request_body=request_body, 

449 ) 

450 # Mark call type for downstream image-aware logic/metrics 

451 try: 

452 logging_obj.call_type = PassthroughCallTypes.passthrough_image_generation.value 

453 except Exception: 

454 pass 

455 # Create a simple response object for logging 

456 litellm_model_response = ImageResponse( 

457 data=response_body.get("data", []), 

458 model=model, 

459 ) 

460 # Set the calculated cost in _hidden_params to prevent recalculation 

461 if not hasattr(litellm_model_response, "_hidden_params"): 

462 litellm_model_response._hidden_params = {} 

463 litellm_model_response._hidden_params["response_cost"] = response_cost 

464 elif is_responses: 

465 # Responses-API cost tracking — see 

466 # `_build_responses_api_response_and_cost` for why this needs 

467 # a dedicated transformer (the chat-completions transform 

468 # crashes on the Responses payload shape). 

469 ( 

470 litellm_model_response, 

471 response_cost, 

472 ) = OpenAIPassthroughLoggingHandler._build_responses_api_response_and_cost( 

473 model=model, 

474 httpx_response=httpx_response, 

475 logging_obj=logging_obj, 

476 custom_llm_provider=custom_llm_provider, 

477 ) 

478 

479 # Update kwargs with cost information 

480 kwargs["response_cost"] = response_cost 

481 kwargs["model"] = model 

482 kwargs["custom_llm_provider"] = custom_llm_provider 

483 

484 # Extract user information for tracking 

485 passthrough_logging_payload: Final[PassthroughStandardLoggingPayload | None] = kwargs.get( 

486 "passthrough_logging_payload" 

487 ) 

488 if passthrough_logging_payload: 

489 user: Final = handler_instance._get_user_from_metadata( 

490 passthrough_logging_payload=passthrough_logging_payload, 

491 ) 

492 if user: 

493 kwargs["litellm_params"].setdefault("proxy_server_request", {}).setdefault("body", {})["user"] = ( 

494 user 

495 ) 

496 

497 # Create standard logging object 

498 if litellm_model_response is not None: 

499 get_standard_logging_object_payload( 

500 kwargs=kwargs, 

501 init_response_obj=litellm_model_response, 

502 start_time=start_time, 

503 end_time=end_time, 

504 logging_obj=logging_obj, 

505 status="success", 

506 ) 

507 

508 # Update logging object with cost information 

509 logging_obj.model_call_details["model"] = model 

510 logging_obj.model_call_details["custom_llm_provider"] = custom_llm_provider 

511 logging_obj.model_call_details["response_cost"] = response_cost 

512 

513 endpoint_type: Final = ( 

514 "chat_completions" 

515 if is_chat_completions 

516 else "embeddings" 

517 if is_embeddings 

518 else "image_generation" 

519 if is_image_generation 

520 else "image_editing" 

521 if is_image_editing 

522 else "responses" 

523 ) 

524 verbose_proxy_logger.debug( 

525 f"OpenAI passthrough cost tracking - Endpoint: {endpoint_type}, Model: {model}, Cost: ${response_cost:.6f}" 

526 ) 

527 

528 return { 

529 "result": litellm_model_response, 

530 "kwargs": kwargs, 

531 } 

532 

533 except Exception as e: 

534 verbose_proxy_logger.error("Error in OpenAI passthrough cost tracking: %s", e) 

535 if not is_chat_completions: 

536 unbilled_result: Final[PassThroughEndpointLoggingTypedDict] = { 

537 "result": None, 

538 "kwargs": kwargs, 

539 } 

540 return unbilled_result 

541 # Fall back to base handler without cost tracking 

542 base_handler = OpenAIPassthroughLoggingHandler() 

543 return base_handler.passthrough_chat_handler( 

544 httpx_response=httpx_response, 

545 response_body=response_body, 

546 logging_obj=logging_obj, 

547 url_route=url_route, 

548 result=result, 

549 start_time=start_time, 

550 end_time=end_time, 

551 cache_hit=cache_hit, 

552 request_body=request_body, 

553 **kwargs, 

554 ) 

555 

556 def _build_complete_streaming_response( 

557 self, 

558 all_chunks: Sequence[str], 

559 litellm_logging_obj: LiteLLMLoggingObj, 

560 model: str, 

561 messages: Sequence[Mapping[str, object]] | None = None, 

562 ) -> ModelResponse | TextCompletionResponse | None: 

563 """ 

564 Builds complete response from raw chunks for OpenAI streaming responses. 

565 

566 - Converts str chunks to generic chunks 

567 - Converts generic chunks to litellm chunks (OpenAI format) 

568 - Builds complete response from litellm chunks 

569 """ 

570 try: 

571 # OpenAI's response iterator to parse chunks 

572 from litellm.llms.openai.openai import OpenAIChatCompletionResponseIterator 

573 

574 openai_iterator: Final = OpenAIChatCompletionResponseIterator( 

575 streaming_response=None, 

576 sync_stream=False, 

577 ) 

578 

579 all_openai_chunks: Final = [] 

580 for chunk_str in all_chunks: 

581 try: 

582 # Parse the string chunk using the base iterator's string parser 

583 from litellm.llms.base_llm.base_model_iterator import ( 

584 BaseModelResponseIterator, 

585 ) 

586 

587 # Convert string chunk to dict 

588 stripped_json_chunk = BaseModelResponseIterator._string_to_dict_parser(str_line=chunk_str) 

589 

590 if stripped_json_chunk: 

591 # Parse the chunk using OpenAI's chunk parser 

592 transformed_chunk = openai_iterator.chunk_parser(chunk=stripped_json_chunk) 

593 if transformed_chunk is not None: 

594 all_openai_chunks.append(transformed_chunk) 

595 

596 except (StopIteration, StopAsyncIteration, Exception) as e: 

597 verbose_proxy_logger.debug("Error parsing streaming chunk: %s", e) 

598 continue 

599 

600 if not all_openai_chunks: 

601 verbose_proxy_logger.warning("No valid chunks found in streaming response") 

602 return None 

603 

604 # Build complete response from chunks 

605 complete_streaming_response: Final = litellm.stream_chunk_builder( 

606 chunks=all_openai_chunks, 

607 messages=messages, 

608 count_prompt_tokens=lambda: count_relayed_prompt_tokens(model, messages), 

609 ) 

610 

611 return complete_streaming_response 

612 

613 except Exception as e: 

614 verbose_proxy_logger.error("Error building complete streaming response: %s", e) 

615 return None 

616 

617 @staticmethod 

618 def _handle_logging_openai_collected_chunks( 

619 litellm_logging_obj: LiteLLMLoggingObj, 

620 passthrough_success_handler_obj: PassThroughEndpointLogging, 

621 url_route: str, 

622 request_body: dict, 

623 endpoint_type: EndpointType, 

624 start_time: datetime, 

625 all_chunks: list[str], 

626 end_time: datetime, 

627 ) -> PassThroughEndpointLoggingTypedDict: 

628 """ 

629 Handle logging for collected OpenAI streaming chunks with cost tracking. 

630 """ 

631 try: 

632 # Extract model from request body 

633 model: Final = request_body.get("model", "gpt-4o") 

634 

635 is_responses: Final = OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route) 

636 

637 # Build complete response from chunks using our streaming handler 

638 handler: Final = OpenAIPassthroughLoggingHandler() 

639 handler_instance: Final = handler 

640 complete_response: Final = ( 

641 OpenAIResponsesAPIConfig.parse_terminal_response_from_stream_chunks(all_chunks=all_chunks) 

642 if is_responses 

643 else handler._build_complete_streaming_response( 

644 all_chunks=all_chunks, 

645 litellm_logging_obj=litellm_logging_obj, 

646 model=model, 

647 ) 

648 ) 

649 

650 if complete_response is None: 

651 verbose_proxy_logger.warning("Failed to build complete response from OpenAI streaming chunks") 

652 return { 

653 "result": None, 

654 "kwargs": {}, 

655 } 

656 

657 custom_llm_provider: Final = litellm_logging_obj.model_call_details.get("custom_llm_provider", "openai") 

658 # Calculate cost using LiteLLM's cost calculator 

659 response_cost: Final = ( 

660 litellm.completion_cost( 

661 completion_response=complete_response, 

662 model=model, 

663 custom_llm_provider=custom_llm_provider, 

664 call_type="responses", 

665 ) 

666 if is_responses 

667 else litellm.completion_cost( 

668 completion_response=complete_response, 

669 model=model, 

670 custom_llm_provider=custom_llm_provider, 

671 ) 

672 ) 

673 

674 # Preserve existing litellm_params to maintain metadata tags 

675 existing_litellm_params: Final = litellm_logging_obj.model_call_details.get("litellm_params", {}) or {} 

676 

677 # Prepare kwargs for logging 

678 kwargs: Final = { 

679 "response_cost": response_cost, 

680 "model": model, 

681 "custom_llm_provider": custom_llm_provider, 

682 "call_type": litellm_logging_obj.call_type, 

683 "messages": litellm_logging_obj.model_call_details.get("messages"), 

684 "litellm_params": existing_litellm_params.copy(), 

685 } 

686 

687 # Extract user information for tracking 

688 passthrough_logging_payload: Final[PassthroughStandardLoggingPayload | None] = ( 

689 litellm_logging_obj.model_call_details.get("passthrough_logging_payload") 

690 ) 

691 if passthrough_logging_payload: 

692 user: Final = handler_instance._get_user_from_metadata( 

693 passthrough_logging_payload=passthrough_logging_payload, 

694 ) 

695 if user: 

696 kwargs["litellm_params"].setdefault("proxy_server_request", {}).setdefault("body", {})["user"] = ( 

697 user 

698 ) 

699 

700 # Attach the payload to kwargs so the success handler adopts it; 

701 # its later rebuild runs on a copy whose Responses usage was 

702 # coerced to chat shape and serializes as total_tokens only, 

703 # zeroing the prompt/completion split in spend logs. 

704 standard_logging_object: Final = get_standard_logging_object_payload( 

705 kwargs=kwargs, 

706 init_response_obj=complete_response, 

707 start_time=start_time, 

708 end_time=end_time, 

709 logging_obj=litellm_logging_obj, 

710 status="success", 

711 ) 

712 if standard_logging_object is not None: 

713 kwargs["standard_logging_object"] = standard_logging_object 

714 

715 # Update logging object with cost information 

716 litellm_logging_obj.model_call_details["model"] = model 

717 litellm_logging_obj.model_call_details["custom_llm_provider"] = custom_llm_provider 

718 litellm_logging_obj.model_call_details["response_cost"] = response_cost 

719 

720 verbose_proxy_logger.debug( 

721 f"OpenAI streaming passthrough cost tracking - Model: {model}, Cost: ${response_cost:.6f}" 

722 ) 

723 

724 return { 

725 "result": complete_response, 

726 "kwargs": kwargs, 

727 } 

728 

729 except Exception as e: 

730 verbose_proxy_logger.error("Error in OpenAI streaming passthrough cost tracking: %s", e) 

731 return { 

732 "result": None, 

733 "kwargs": {}, 

734 }