Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/pass_through_endpoints/upstream_usage_headers.py: 38%
58 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""Upstream-reported cost and usage for pass-through endpoints.
3Some pass-through targets invoke several models internally, so LiteLLM cannot
4price the request from the response body. Those targets report the totals for
5the whole HTTP request in ``x-litellm-*`` response headers instead; LiteLLM
6records the reported values without recomputing them.
7"""
9import math
10from dataclasses import dataclass
11from typing import Final
13import httpx
15from litellm._logging import verbose_proxy_logger
16from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
17from litellm.types.utils import Usage
19UPSTREAM_RESPONSE_COST_HEADER: Final = "x-litellm-response-cost"
20UPSTREAM_TOTAL_TOKENS_HEADER: Final = "x-litellm-total-tokens"
22# model_call_details key holding what the upstream reported, so later stages of
23# the success path can tell an upstream-reported cost apart from one LiteLLM
24# derived itself.
25UPSTREAM_REPORTED_USAGE_KEY: Final = "_litellm_upstream_reported_usage"
28@dataclass(frozen=True, slots=True)
29class UpstreamReportedUsage:
30 """Totals an upstream pass-through target reported for one HTTP request.
32 ``None`` means the upstream did not report a usable value, either because
33 the header was absent or because it could not be parsed.
34 """
36 response_cost: float | None
37 total_tokens: int | None
40def _parse_response_cost(raw_value: str | None) -> float | None:
41 if raw_value is None:
42 verbose_proxy_logger.warning(
43 "pass_through_endpoint: upstream did not send %s; recording 0 cost for this request",
44 UPSTREAM_RESPONSE_COST_HEADER,
45 )
46 return None
47 try:
48 response_cost: Final = float(raw_value)
49 except ValueError:
50 verbose_proxy_logger.warning(
51 "pass_through_endpoint: upstream sent unparseable %s=%r; recording 0 cost for this request",
52 UPSTREAM_RESPONSE_COST_HEADER,
53 raw_value,
54 )
55 return None
56 if not math.isfinite(response_cost) or response_cost < 0:
57 verbose_proxy_logger.warning(
58 "pass_through_endpoint: upstream sent out-of-range %s=%r; recording 0 cost for this request",
59 UPSTREAM_RESPONSE_COST_HEADER,
60 raw_value,
61 )
62 return None
63 return response_cost
66def _parse_total_tokens(raw_value: str | None) -> int | None:
67 if raw_value is None:
68 verbose_proxy_logger.warning(
69 "pass_through_endpoint: upstream did not send %s; recording 0 tokens for this request",
70 UPSTREAM_TOTAL_TOKENS_HEADER,
71 )
72 return None
73 try:
74 total_tokens: Final = int(raw_value)
75 except ValueError:
76 verbose_proxy_logger.warning(
77 "pass_through_endpoint: upstream sent unparseable %s=%r; recording 0 tokens for this request",
78 UPSTREAM_TOTAL_TOKENS_HEADER,
79 raw_value,
80 )
81 return None
82 if total_tokens < 0:
83 verbose_proxy_logger.warning(
84 "pass_through_endpoint: upstream sent negative %s=%r; recording 0 tokens for this request",
85 UPSTREAM_TOTAL_TOKENS_HEADER,
86 raw_value,
87 )
88 return None
89 return total_tokens
92def parse_upstream_reported_usage(headers: httpx.Headers) -> UpstreamReportedUsage | None:
93 """Read the reported totals off an upstream pass-through response.
95 Returns ``None`` when neither header is present, which is the normal case
96 for a target that does not speak this contract (e.g. Anthropic or Vertex,
97 whose cost LiteLLM derives from the response body instead).
98 """
99 raw_response_cost: Final = headers.get(UPSTREAM_RESPONSE_COST_HEADER)
100 raw_total_tokens: Final = headers.get(UPSTREAM_TOTAL_TOKENS_HEADER)
101 if raw_response_cost is None and raw_total_tokens is None: 101 ↛ 103line 101 didn't jump to line 103 because the condition on line 101 was always true
102 return None
103 return UpstreamReportedUsage(
104 response_cost=_parse_response_cost(raw_response_cost),
105 total_tokens=_parse_total_tokens(raw_total_tokens),
106 )
109def apply_upstream_reported_usage(
110 logging_obj: LiteLLMLoggingObj,
111 headers: httpx.Headers,
112) -> UpstreamReportedUsage | None:
113 """Record the upstream's reported totals on the request's logging object.
115 Only the values the upstream actually reported are written, so a target
116 that reports cost but not tokens keeps the token count LiteLLM derived on
117 its own rather than having it zeroed.
118 """
119 reported: Final = parse_upstream_reported_usage(headers)
120 if reported is None: 120 ↛ 122line 120 didn't jump to line 122 because the condition on line 120 was always true
121 return None
122 logging_obj.model_call_details[UPSTREAM_REPORTED_USAGE_KEY] = reported
123 if reported.response_cost is not None:
124 logging_obj.model_call_details["response_cost"] = reported.response_cost
125 if reported.total_tokens is not None:
126 logging_obj.model_call_details["combined_usage_object"] = Usage(total_tokens=reported.total_tokens)
127 return reported
130def has_upstream_reported_usage(logging_obj: LiteLLMLoggingObj) -> bool:
131 """Whether the upstream spoke this contract on the request's response.
133 True even when the value it sent was unusable: a target that reports its
134 own totals owns the cost for the request, and a header we could not parse
135 means zero, never a fallback to someone else's estimate.
136 """
137 return isinstance(logging_obj.model_call_details.get(UPSTREAM_REPORTED_USAGE_KEY), UpstreamReportedUsage)