Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py: 17%
276 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2OpenAI Passthrough Logging Handler
4Handles cost tracking and logging for OpenAI passthrough endpoints, specifically /chat/completions.
5"""
7from collections.abc import Mapping, Sequence
8from datetime import datetime
9from typing import Final
10from urllib.parse import urlparse
12import httpx
14import litellm
15from litellm._logging import verbose_proxy_logger
16from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
17from litellm.litellm_core_utils.litellm_logging import (
18 get_standard_logging_object_payload,
19)
20from litellm.litellm_core_utils.token_counter import high_detail_image_token_upper_bound
21from litellm.llms.openai.openai import OpenAIConfig
22from litellm.llms.openai.openai import OpenAIConfig as OpenAIConfigType
23from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
24from litellm.proxy._types import PassThroughEndpointLoggingTypedDict
25from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough_logging_handler import (
26 BasePassthroughLoggingHandler,
27)
28from litellm.proxy.pass_through_endpoints.success_handler import (
29 PassThroughEndpointLogging,
30)
31from litellm.types.llms.openai import ResponsesAPIResponse
32from litellm.types.passthrough_endpoints.pass_through_endpoints import (
33 EndpointType,
34 PassthroughStandardLoggingPayload,
35)
36from litellm.types.utils import EmbeddingResponse, ImageResponse, LlmProviders, PassthroughCallTypes
37from litellm.utils import ModelResponse, TextCompletionResponse, convert_to_model_response_object
39# Hostnames that route to OpenAI-compatible APIs.
40#
41# `api.openai.com` is OpenAI proper. The two Azure domains below are *shared by
42# every Azure Cognitive Service* (Speech, Vision, Language, ...), not just Azure
43# OpenAI: `openai.azure.com` is the classic Azure OpenAI domain, while
44# `cognitiveservices.azure.com` is used by newer "Azure AI Foundry" /
45# Cognitive Services-hosted Azure OpenAI deployments. Because the hostname alone
46# cannot tell Azure OpenAI apart from the other Cognitive Services on those
47# domains, requests there must additionally carry an OpenAI-style path segment.
48_OPENAI_HOSTNAMES: Final = ("api.openai.com",)
49_AZURE_OPENAI_HOSTNAMES: Final = ("openai.azure.com", "cognitiveservices.azure.com")
50# Path markers that identify an Azure request as Azure OpenAI rather than Speech
51# / Vision / Language / ... `/openai/` is the native Azure OpenAI path prefix;
52# `/v1/` is the OpenAI-v1 surface used by LiteLLM's pass-through routing. Other
53# Cognitive Services use service-named prefixes and versions like `/v3.1/`,
54# `/v1.0/`, so they do not collide with these markers.
55_AZURE_OPENAI_PATH_MARKERS: Final = ("/openai/", "/v1/")
58def _hostname_matches(hostname: str, suffixes: tuple) -> bool:
59 """True if hostname equals one of `suffixes` or is a subdomain of it.
61 Uses suffix matching (not a bare substring test) so look-alikes such as
62 `cognitiveservices.azure.com.attacker.example` are not accepted.
63 """
64 return any(hostname == suffix or hostname.endswith("." + suffix) for suffix in suffixes)
67def _is_openai_compatible_host(hostname: str | None) -> bool:
68 """True if the hostname is OpenAI proper or one of the Azure OpenAI domains.
70 Hostname-only check, kept for the route-level helpers that additionally
71 require a specific OpenAI path (e.g. `/v1/chat/completions`). When only the
72 hostname would otherwise gate dispatch, use `_is_openai_compatible_url` so
73 non-OpenAI Azure Cognitive Services on the shared domains are excluded.
74 """
75 if not hostname:
76 return False
77 return _hostname_matches(hostname, _OPENAI_HOSTNAMES) or _hostname_matches(hostname, _AZURE_OPENAI_HOSTNAMES)
80def _is_openai_compatible_url(url_route: str | None) -> bool:
81 """True if the URL targets an OpenAI-compatible API surface.
83 For the shared Azure Cognitive Services domains we additionally require an
84 OpenAI-style path segment (`/openai/` or `/v1/`) so non-OpenAI Azure services
85 (Speech, Vision, Language, ...) on the same domain are not misclassified as
86 OpenAI routes.
87 """
88 if not url_route:
89 return False
90 parsed_url: Final = urlparse(url_route)
91 hostname: Final = parsed_url.hostname
92 if not hostname:
93 return False
94 if _hostname_matches(hostname, _OPENAI_HOSTNAMES):
95 return True
96 if _hostname_matches(hostname, _AZURE_OPENAI_HOSTNAMES):
97 return any(marker in parsed_url.path for marker in _AZURE_OPENAI_PATH_MARKERS)
98 return False
101def _is_remote_high_detail_image(part: object) -> bool:
102 if not isinstance(part, Mapping) or part.get("type") != "image_url":
103 return False
104 image_url: Final = part.get("image_url")
105 if not isinstance(image_url, Mapping):
106 return False
107 url: Final = image_url.get("url")
108 return (
109 isinstance(url, str) and url.lower().startswith(("http://", "https://")) and image_url.get("detail") == "high"
110 )
113def _content_parts(message: Mapping[str, object]) -> Sequence[object]:
114 content: Final = message.get("content")
115 return content if isinstance(content, list) else ()
118def _without_remote_high_detail_images(message: Mapping[str, object]) -> Mapping[str, object]:
119 if not isinstance(message.get("content"), list):
120 return message
121 kept_parts: Final = [ # mutable-ok: token_counter reads message content only when it is a list
122 part for part in _content_parts(message) if not _is_remote_high_detail_image(part)
123 ]
124 return {**message, "content": kept_parts} # mutable-ok: token_counter rejects any message that is not a dict
127def count_relayed_prompt_tokens(model: str, messages: Sequence[Mapping[str, object]] | None) -> int:
128 if messages is None:
129 return 0
130 remote_high_detail_images: Final = sum(
131 1 for message in messages for part in _content_parts(message) if _is_remote_high_detail_image(part)
132 )
133 local_messages: Final = [ # mutable-ok: token_counter takes a list of messages
134 _without_remote_high_detail_images(message) for message in messages
135 ]
136 return (
137 litellm.token_counter(model=model, messages=local_messages)
138 + high_detail_image_token_upper_bound() * remote_high_detail_images
139 )
142class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
143 """
144 OpenAI-specific passthrough logging handler that provides cost tracking for /chat/completions endpoints.
145 """
147 @property
148 def llm_provider_name(self) -> LlmProviders:
149 return LlmProviders.OPENAI
151 def get_provider_config(self, model: str) -> OpenAIConfigType:
152 """Get OpenAI provider configuration for the given model."""
153 return OpenAIConfig()
155 @staticmethod
156 def is_openai_chat_completions_route(url_route: str) -> bool:
157 """Check if the URL route is an OpenAI chat completions endpoint."""
158 if not url_route:
159 return False
160 parsed_url: Final = urlparse(url_route)
161 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/chat/completions" in parsed_url.path
163 @staticmethod
164 def is_openai_image_generation_route(url_route: str) -> bool:
165 """Check if the URL route is an OpenAI image generation endpoint."""
166 if not url_route:
167 return False
168 parsed_url: Final = urlparse(url_route)
169 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/images/generations" in parsed_url.path
171 @staticmethod
172 def is_openai_image_editing_route(url_route: str) -> bool:
173 """Check if the URL route is an OpenAI image editing endpoint."""
174 if not url_route:
175 return False
176 parsed_url: Final = urlparse(url_route)
177 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/images/edits" in parsed_url.path
179 @staticmethod
180 def is_openai_responses_route(url_route: str) -> bool:
181 """Check if the URL route is an OpenAI responses API endpoint."""
182 if not url_route:
183 return False
184 parsed_url: Final = urlparse(url_route)
185 return _is_openai_compatible_host(parsed_url.hostname) and (
186 "/v1/responses" in parsed_url.path or "/responses" in parsed_url.path
187 )
189 @staticmethod
190 def is_openai_embeddings_route(url_route: str) -> bool:
191 """Check if the URL route is an OpenAI embeddings endpoint."""
192 if not url_route:
193 return False
194 parsed_url: Final = urlparse(url_route)
195 return _is_openai_compatible_host(parsed_url.hostname) and "/v1/embeddings" in parsed_url.path
197 def _get_user_from_metadata(
198 self,
199 passthrough_logging_payload: PassthroughStandardLoggingPayload,
200 ) -> str | None:
201 """Extract user information from passthrough logging payload."""
202 request_body: Final = passthrough_logging_payload.get("request_body")
203 if request_body:
204 return request_body.get("user")
205 return None
207 @staticmethod
208 def _calculate_image_generation_cost(
209 model: str,
210 response_body: dict,
211 request_body: dict,
212 ) -> float:
213 """Calculate cost for OpenAI image generation."""
214 try:
215 # Extract parameters from request
216 n = request_body.get("n", 1)
217 try:
218 n = int(n)
219 except Exception:
220 n = 1
221 size: Final = request_body.get("size", "1024x1024")
222 quality: Final = request_body.get("quality", None)
224 # Use LiteLLM's default image cost calculator
225 from litellm.cost_calculator import default_image_cost_calculator
227 cost: Final = default_image_cost_calculator(
228 model=model,
229 custom_llm_provider="openai",
230 quality=quality,
231 n=n,
232 size=size,
233 optional_params=request_body,
234 )
236 return cost
237 except Exception as e:
238 verbose_proxy_logger.warning("Error calculating image generation cost: %s", e)
239 return 0.0
241 @staticmethod
242 def _calculate_image_editing_cost(
243 model: str,
244 response_body: dict,
245 request_body: dict,
246 ) -> float:
247 """Calculate cost for OpenAI image editing."""
248 try:
249 # Extract parameters from request
250 n = request_body.get("n", 1)
251 # Image edit typically uses multipart/form-data (because of files), so all fields arrive as strings (e.g., n = "1").
252 try:
253 n = int(n)
254 except Exception:
255 n = 1
256 size: Final = request_body.get("size", "1024x1024")
258 # Use LiteLLM's default image cost calculator
259 from litellm.cost_calculator import default_image_cost_calculator
261 cost: Final = default_image_cost_calculator(
262 model=model,
263 custom_llm_provider="openai",
264 quality=None, # Image editing doesn't have quality parameter
265 n=n,
266 size=size,
267 optional_params=request_body,
268 )
270 return cost
271 except Exception as e:
272 verbose_proxy_logger.warning("Error calculating image editing cost: %s", e)
273 return 0.0
275 @staticmethod
276 def _calculate_embeddings_cost(
277 litellm_model_response: EmbeddingResponse,
278 model: str,
279 custom_llm_provider: str,
280 ) -> float:
281 try:
282 return litellm.completion_cost(
283 completion_response=litellm_model_response,
284 model=model,
285 custom_llm_provider=custom_llm_provider,
286 call_type="aembedding",
287 )
288 except Exception as e: # noqa: BLE001 # completion_cost raises bare Exception for unmapped models; cost failure must never drop the spend log
289 verbose_proxy_logger.warning(
290 "Error calculating embeddings cost for model %s, logging spend with cost 0: %s", model, e
291 )
292 return 0.0
294 @staticmethod
295 def _build_responses_api_response_and_cost(
296 model: str,
297 httpx_response: httpx.Response,
298 logging_obj: LiteLLMLoggingObj,
299 custom_llm_provider: str,
300 ) -> tuple[ResponsesAPIResponse, float]:
301 """Transform a Responses API raw response into a ResponsesAPIResponse
302 and compute its cost.
304 The Responses API has a different on-the-wire shape from chat
305 completions (`output: [...]` instead of `choices: [...]`), so the
306 chat-completions `transform_response` raises KeyError 'choices' on
307 a Responses payload. Use the dedicated Responses-API transformer
308 (`OpenAIResponsesAPIConfig.transform_response_api_response`) here.
310 Returns (litellm_model_response, response_cost) — symmetric with the
311 chat-completions branch which produces the same two values inline,
312 and analogous to the image branches' `_calculate_image_*_cost` helpers
313 (which return cost only because the image-response object is trivial
314 to build inline; the Responses payload needs a real transformer).
315 """
316 responses_config: Final = OpenAIResponsesAPIConfig()
317 litellm_model_response: Final = responses_config.transform_response_api_response(
318 model=model,
319 raw_response=httpx_response,
320 logging_obj=logging_obj,
321 )
322 response_cost: Final = litellm.completion_cost(
323 completion_response=litellm_model_response,
324 model=model,
325 custom_llm_provider=custom_llm_provider,
326 call_type="responses",
327 )
328 return litellm_model_response, response_cost
330 @staticmethod
331 def openai_passthrough_handler(
332 httpx_response: httpx.Response,
333 response_body: dict,
334 logging_obj: LiteLLMLoggingObj,
335 url_route: str,
336 result: str,
337 start_time: datetime,
338 end_time: datetime,
339 cache_hit: bool,
340 request_body: dict,
341 **kwargs,
342 ) -> PassThroughEndpointLoggingTypedDict:
343 """
344 Handle OpenAI passthrough logging with cost tracking for chat completions,
345 embeddings, image generation, image editing, and responses API.
346 """
347 is_chat_completions: Final = OpenAIPassthroughLoggingHandler.is_openai_chat_completions_route(url_route)
348 is_embeddings: Final = OpenAIPassthroughLoggingHandler.is_openai_embeddings_route(url_route)
349 is_image_generation: Final = OpenAIPassthroughLoggingHandler.is_openai_image_generation_route(url_route)
350 is_image_editing: Final = OpenAIPassthroughLoggingHandler.is_openai_image_editing_route(url_route)
351 is_responses: Final = OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route)
353 if not (is_chat_completions or is_embeddings or is_image_generation or is_image_editing or is_responses):
354 return {
355 "result": None,
356 "kwargs": kwargs,
357 }
359 model: Final = request_body.get("model", response_body.get("model", ""))
360 if not model:
361 verbose_proxy_logger.warning("No model found in request or response for OpenAI passthrough cost tracking")
362 base_handler = OpenAIPassthroughLoggingHandler()
363 return base_handler.passthrough_chat_handler(
364 httpx_response=httpx_response,
365 response_body=response_body,
366 logging_obj=logging_obj,
367 url_route=url_route,
368 result=result,
369 start_time=start_time,
370 end_time=end_time,
371 cache_hit=cache_hit,
372 request_body=request_body,
373 **kwargs,
374 )
376 try:
377 response_cost = 0.0
378 litellm_model_response: (
379 ModelResponse | TextCompletionResponse | EmbeddingResponse | ImageResponse | ResponsesAPIResponse | None
380 ) = None
381 handler_instance: Final = OpenAIPassthroughLoggingHandler()
383 custom_llm_provider: Final = kwargs.get("custom_llm_provider", "openai")
385 if is_chat_completions:
386 # Handle chat completions with existing logic
387 provider_config: Final = handler_instance.get_provider_config(model=model)
388 # Preserve existing litellm_params to maintain metadata tags
389 existing_litellm_params: Final = kwargs.get("litellm_params", {}) or {}
390 litellm_model_response = provider_config.transform_response(
391 raw_response=httpx_response,
392 model_response=litellm.ModelResponse(),
393 model=model,
394 messages=request_body.get("messages", []),
395 logging_obj=logging_obj,
396 optional_params=request_body.get("optional_params", {}),
397 api_key="",
398 request_data=request_body,
399 encoding=getattr(litellm, "encoding", None),
400 json_mode=request_body.get("response_format", {}).get("type") == "json_object",
401 litellm_params=existing_litellm_params,
402 )
404 # Calculate cost using LiteLLM's cost calculator
405 response_cost = litellm.completion_cost(
406 completion_response=litellm_model_response,
407 model=model,
408 custom_llm_provider=custom_llm_provider,
409 )
410 elif is_embeddings:
411 litellm_model_response = convert_to_model_response_object(
412 response_object=response_body,
413 model_response_object=EmbeddingResponse(),
414 response_type="embedding",
415 )
416 response_cost = OpenAIPassthroughLoggingHandler._calculate_embeddings_cost(
417 litellm_model_response=litellm_model_response,
418 model=model,
419 custom_llm_provider=custom_llm_provider,
420 )
421 litellm_model_response._hidden_params["response_cost"] = response_cost
422 elif is_image_generation:
423 # Handle image generation cost calculation
424 response_cost = OpenAIPassthroughLoggingHandler._calculate_image_generation_cost(
425 model=model,
426 response_body=response_body,
427 request_body=request_body,
428 )
429 # Mark call type for downstream image-aware logic/metrics
430 try:
431 logging_obj.call_type = PassthroughCallTypes.passthrough_image_generation.value
432 except Exception:
433 pass
434 # Create a simple response object for logging
435 litellm_model_response = ImageResponse(
436 data=response_body.get("data", []),
437 model=model,
438 )
439 # Set the calculated cost in _hidden_params to prevent recalculation
440 if not hasattr(litellm_model_response, "_hidden_params"):
441 litellm_model_response._hidden_params = {}
442 litellm_model_response._hidden_params["response_cost"] = response_cost
443 elif is_image_editing:
444 # Handle image editing cost calculation
445 response_cost = OpenAIPassthroughLoggingHandler._calculate_image_editing_cost(
446 model=model,
447 response_body=response_body,
448 request_body=request_body,
449 )
450 # Mark call type for downstream image-aware logic/metrics
451 try:
452 logging_obj.call_type = PassthroughCallTypes.passthrough_image_generation.value
453 except Exception:
454 pass
455 # Create a simple response object for logging
456 litellm_model_response = ImageResponse(
457 data=response_body.get("data", []),
458 model=model,
459 )
460 # Set the calculated cost in _hidden_params to prevent recalculation
461 if not hasattr(litellm_model_response, "_hidden_params"):
462 litellm_model_response._hidden_params = {}
463 litellm_model_response._hidden_params["response_cost"] = response_cost
464 elif is_responses:
465 # Responses-API cost tracking — see
466 # `_build_responses_api_response_and_cost` for why this needs
467 # a dedicated transformer (the chat-completions transform
468 # crashes on the Responses payload shape).
469 (
470 litellm_model_response,
471 response_cost,
472 ) = OpenAIPassthroughLoggingHandler._build_responses_api_response_and_cost(
473 model=model,
474 httpx_response=httpx_response,
475 logging_obj=logging_obj,
476 custom_llm_provider=custom_llm_provider,
477 )
479 # Update kwargs with cost information
480 kwargs["response_cost"] = response_cost
481 kwargs["model"] = model
482 kwargs["custom_llm_provider"] = custom_llm_provider
484 # Extract user information for tracking
485 passthrough_logging_payload: Final[PassthroughStandardLoggingPayload | None] = kwargs.get(
486 "passthrough_logging_payload"
487 )
488 if passthrough_logging_payload:
489 user: Final = handler_instance._get_user_from_metadata(
490 passthrough_logging_payload=passthrough_logging_payload,
491 )
492 if user:
493 kwargs["litellm_params"].setdefault("proxy_server_request", {}).setdefault("body", {})["user"] = (
494 user
495 )
497 # Create standard logging object
498 if litellm_model_response is not None:
499 get_standard_logging_object_payload(
500 kwargs=kwargs,
501 init_response_obj=litellm_model_response,
502 start_time=start_time,
503 end_time=end_time,
504 logging_obj=logging_obj,
505 status="success",
506 )
508 # Update logging object with cost information
509 logging_obj.model_call_details["model"] = model
510 logging_obj.model_call_details["custom_llm_provider"] = custom_llm_provider
511 logging_obj.model_call_details["response_cost"] = response_cost
513 endpoint_type: Final = (
514 "chat_completions"
515 if is_chat_completions
516 else "embeddings"
517 if is_embeddings
518 else "image_generation"
519 if is_image_generation
520 else "image_editing"
521 if is_image_editing
522 else "responses"
523 )
524 verbose_proxy_logger.debug(
525 f"OpenAI passthrough cost tracking - Endpoint: {endpoint_type}, Model: {model}, Cost: ${response_cost:.6f}"
526 )
528 return {
529 "result": litellm_model_response,
530 "kwargs": kwargs,
531 }
533 except Exception as e:
534 verbose_proxy_logger.error("Error in OpenAI passthrough cost tracking: %s", e)
535 if not is_chat_completions:
536 unbilled_result: Final[PassThroughEndpointLoggingTypedDict] = {
537 "result": None,
538 "kwargs": kwargs,
539 }
540 return unbilled_result
541 # Fall back to base handler without cost tracking
542 base_handler = OpenAIPassthroughLoggingHandler()
543 return base_handler.passthrough_chat_handler(
544 httpx_response=httpx_response,
545 response_body=response_body,
546 logging_obj=logging_obj,
547 url_route=url_route,
548 result=result,
549 start_time=start_time,
550 end_time=end_time,
551 cache_hit=cache_hit,
552 request_body=request_body,
553 **kwargs,
554 )
556 def _build_complete_streaming_response(
557 self,
558 all_chunks: Sequence[str],
559 litellm_logging_obj: LiteLLMLoggingObj,
560 model: str,
561 messages: Sequence[Mapping[str, object]] | None = None,
562 ) -> ModelResponse | TextCompletionResponse | None:
563 """
564 Builds complete response from raw chunks for OpenAI streaming responses.
566 - Converts str chunks to generic chunks
567 - Converts generic chunks to litellm chunks (OpenAI format)
568 - Builds complete response from litellm chunks
569 """
570 try:
571 # OpenAI's response iterator to parse chunks
572 from litellm.llms.openai.openai import OpenAIChatCompletionResponseIterator
574 openai_iterator: Final = OpenAIChatCompletionResponseIterator(
575 streaming_response=None,
576 sync_stream=False,
577 )
579 all_openai_chunks: Final = []
580 for chunk_str in all_chunks:
581 try:
582 # Parse the string chunk using the base iterator's string parser
583 from litellm.llms.base_llm.base_model_iterator import (
584 BaseModelResponseIterator,
585 )
587 # Convert string chunk to dict
588 stripped_json_chunk = BaseModelResponseIterator._string_to_dict_parser(str_line=chunk_str)
590 if stripped_json_chunk:
591 # Parse the chunk using OpenAI's chunk parser
592 transformed_chunk = openai_iterator.chunk_parser(chunk=stripped_json_chunk)
593 if transformed_chunk is not None:
594 all_openai_chunks.append(transformed_chunk)
596 except (StopIteration, StopAsyncIteration, Exception) as e:
597 verbose_proxy_logger.debug("Error parsing streaming chunk: %s", e)
598 continue
600 if not all_openai_chunks:
601 verbose_proxy_logger.warning("No valid chunks found in streaming response")
602 return None
604 # Build complete response from chunks
605 complete_streaming_response: Final = litellm.stream_chunk_builder(
606 chunks=all_openai_chunks,
607 messages=messages,
608 count_prompt_tokens=lambda: count_relayed_prompt_tokens(model, messages),
609 )
611 return complete_streaming_response
613 except Exception as e:
614 verbose_proxy_logger.error("Error building complete streaming response: %s", e)
615 return None
617 @staticmethod
618 def _handle_logging_openai_collected_chunks(
619 litellm_logging_obj: LiteLLMLoggingObj,
620 passthrough_success_handler_obj: PassThroughEndpointLogging,
621 url_route: str,
622 request_body: dict,
623 endpoint_type: EndpointType,
624 start_time: datetime,
625 all_chunks: list[str],
626 end_time: datetime,
627 ) -> PassThroughEndpointLoggingTypedDict:
628 """
629 Handle logging for collected OpenAI streaming chunks with cost tracking.
630 """
631 try:
632 # Extract model from request body
633 model: Final = request_body.get("model", "gpt-4o")
635 is_responses: Final = OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route)
637 # Build complete response from chunks using our streaming handler
638 handler: Final = OpenAIPassthroughLoggingHandler()
639 handler_instance: Final = handler
640 complete_response: Final = (
641 OpenAIResponsesAPIConfig.parse_terminal_response_from_stream_chunks(all_chunks=all_chunks)
642 if is_responses
643 else handler._build_complete_streaming_response(
644 all_chunks=all_chunks,
645 litellm_logging_obj=litellm_logging_obj,
646 model=model,
647 )
648 )
650 if complete_response is None:
651 verbose_proxy_logger.warning("Failed to build complete response from OpenAI streaming chunks")
652 return {
653 "result": None,
654 "kwargs": {},
655 }
657 custom_llm_provider: Final = litellm_logging_obj.model_call_details.get("custom_llm_provider", "openai")
658 # Calculate cost using LiteLLM's cost calculator
659 response_cost: Final = (
660 litellm.completion_cost(
661 completion_response=complete_response,
662 model=model,
663 custom_llm_provider=custom_llm_provider,
664 call_type="responses",
665 )
666 if is_responses
667 else litellm.completion_cost(
668 completion_response=complete_response,
669 model=model,
670 custom_llm_provider=custom_llm_provider,
671 )
672 )
674 # Preserve existing litellm_params to maintain metadata tags
675 existing_litellm_params: Final = litellm_logging_obj.model_call_details.get("litellm_params", {}) or {}
677 # Prepare kwargs for logging
678 kwargs: Final = {
679 "response_cost": response_cost,
680 "model": model,
681 "custom_llm_provider": custom_llm_provider,
682 "call_type": litellm_logging_obj.call_type,
683 "messages": litellm_logging_obj.model_call_details.get("messages"),
684 "litellm_params": existing_litellm_params.copy(),
685 }
687 # Extract user information for tracking
688 passthrough_logging_payload: Final[PassthroughStandardLoggingPayload | None] = (
689 litellm_logging_obj.model_call_details.get("passthrough_logging_payload")
690 )
691 if passthrough_logging_payload:
692 user: Final = handler_instance._get_user_from_metadata(
693 passthrough_logging_payload=passthrough_logging_payload,
694 )
695 if user:
696 kwargs["litellm_params"].setdefault("proxy_server_request", {}).setdefault("body", {})["user"] = (
697 user
698 )
700 # Attach the payload to kwargs so the success handler adopts it;
701 # its later rebuild runs on a copy whose Responses usage was
702 # coerced to chat shape and serializes as total_tokens only,
703 # zeroing the prompt/completion split in spend logs.
704 standard_logging_object: Final = get_standard_logging_object_payload(
705 kwargs=kwargs,
706 init_response_obj=complete_response,
707 start_time=start_time,
708 end_time=end_time,
709 logging_obj=litellm_logging_obj,
710 status="success",
711 )
712 if standard_logging_object is not None:
713 kwargs["standard_logging_object"] = standard_logging_object
715 # Update logging object with cost information
716 litellm_logging_obj.model_call_details["model"] = model
717 litellm_logging_obj.model_call_details["custom_llm_provider"] = custom_llm_provider
718 litellm_logging_obj.model_call_details["response_cost"] = response_cost
720 verbose_proxy_logger.debug(
721 f"OpenAI streaming passthrough cost tracking - Model: {model}, Cost: ${response_cost:.6f}"
722 )
724 return {
725 "result": complete_response,
726 "kwargs": kwargs,
727 }
729 except Exception as e:
730 verbose_proxy_logger.error("Error in OpenAI streaming passthrough cost tracking: %s", e)
731 return {
732 "result": None,
733 "kwargs": {},
734 }