Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/pass_through_endpoints/upstream_usage_headers.py: 38%

58 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1"""Upstream-reported cost and usage for pass-through endpoints. 

2 

3Some pass-through targets invoke several models internally, so LiteLLM cannot 

4price the request from the response body. Those targets report the totals for 

5the whole HTTP request in ``x-litellm-*`` response headers instead; LiteLLM 

6records the reported values without recomputing them. 

7""" 

8 

9import math 

10from dataclasses import dataclass 

11from typing import Final 

12 

13import httpx 

14 

15from litellm._logging import verbose_proxy_logger 

16from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj 

17from litellm.types.utils import Usage 

18 

19UPSTREAM_RESPONSE_COST_HEADER: Final = "x-litellm-response-cost" 

20UPSTREAM_TOTAL_TOKENS_HEADER: Final = "x-litellm-total-tokens" 

21 

22# model_call_details key holding what the upstream reported, so later stages of 

23# the success path can tell an upstream-reported cost apart from one LiteLLM 

24# derived itself. 

25UPSTREAM_REPORTED_USAGE_KEY: Final = "_litellm_upstream_reported_usage" 

26 

27 

28@dataclass(frozen=True, slots=True) 

29class UpstreamReportedUsage: 

30 """Totals an upstream pass-through target reported for one HTTP request. 

31 

32 ``None`` means the upstream did not report a usable value, either because 

33 the header was absent or because it could not be parsed. 

34 """ 

35 

36 response_cost: float | None 

37 total_tokens: int | None 

38 

39 

40def _parse_response_cost(raw_value: str | None) -> float | None: 

41 if raw_value is None: 

42 verbose_proxy_logger.warning( 

43 "pass_through_endpoint: upstream did not send %s; recording 0 cost for this request", 

44 UPSTREAM_RESPONSE_COST_HEADER, 

45 ) 

46 return None 

47 try: 

48 response_cost: Final = float(raw_value) 

49 except ValueError: 

50 verbose_proxy_logger.warning( 

51 "pass_through_endpoint: upstream sent unparseable %s=%r; recording 0 cost for this request", 

52 UPSTREAM_RESPONSE_COST_HEADER, 

53 raw_value, 

54 ) 

55 return None 

56 if not math.isfinite(response_cost) or response_cost < 0: 

57 verbose_proxy_logger.warning( 

58 "pass_through_endpoint: upstream sent out-of-range %s=%r; recording 0 cost for this request", 

59 UPSTREAM_RESPONSE_COST_HEADER, 

60 raw_value, 

61 ) 

62 return None 

63 return response_cost 

64 

65 

66def _parse_total_tokens(raw_value: str | None) -> int | None: 

67 if raw_value is None: 

68 verbose_proxy_logger.warning( 

69 "pass_through_endpoint: upstream did not send %s; recording 0 tokens for this request", 

70 UPSTREAM_TOTAL_TOKENS_HEADER, 

71 ) 

72 return None 

73 try: 

74 total_tokens: Final = int(raw_value) 

75 except ValueError: 

76 verbose_proxy_logger.warning( 

77 "pass_through_endpoint: upstream sent unparseable %s=%r; recording 0 tokens for this request", 

78 UPSTREAM_TOTAL_TOKENS_HEADER, 

79 raw_value, 

80 ) 

81 return None 

82 if total_tokens < 0: 

83 verbose_proxy_logger.warning( 

84 "pass_through_endpoint: upstream sent negative %s=%r; recording 0 tokens for this request", 

85 UPSTREAM_TOTAL_TOKENS_HEADER, 

86 raw_value, 

87 ) 

88 return None 

89 return total_tokens 

90 

91 

92def parse_upstream_reported_usage(headers: httpx.Headers) -> UpstreamReportedUsage | None: 

93 """Read the reported totals off an upstream pass-through response. 

94 

95 Returns ``None`` when neither header is present, which is the normal case 

96 for a target that does not speak this contract (e.g. Anthropic or Vertex, 

97 whose cost LiteLLM derives from the response body instead). 

98 """ 

99 raw_response_cost: Final = headers.get(UPSTREAM_RESPONSE_COST_HEADER) 

100 raw_total_tokens: Final = headers.get(UPSTREAM_TOTAL_TOKENS_HEADER) 

101 if raw_response_cost is None and raw_total_tokens is None: 101 ↛ 103line 101 didn't jump to line 103 because the condition on line 101 was always true

102 return None 

103 return UpstreamReportedUsage( 

104 response_cost=_parse_response_cost(raw_response_cost), 

105 total_tokens=_parse_total_tokens(raw_total_tokens), 

106 ) 

107 

108 

109def apply_upstream_reported_usage( 

110 logging_obj: LiteLLMLoggingObj, 

111 headers: httpx.Headers, 

112) -> UpstreamReportedUsage | None: 

113 """Record the upstream's reported totals on the request's logging object. 

114 

115 Only the values the upstream actually reported are written, so a target 

116 that reports cost but not tokens keeps the token count LiteLLM derived on 

117 its own rather than having it zeroed. 

118 """ 

119 reported: Final = parse_upstream_reported_usage(headers) 

120 if reported is None: 120 ↛ 122line 120 didn't jump to line 122 because the condition on line 120 was always true

121 return None 

122 logging_obj.model_call_details[UPSTREAM_REPORTED_USAGE_KEY] = reported 

123 if reported.response_cost is not None: 

124 logging_obj.model_call_details["response_cost"] = reported.response_cost 

125 if reported.total_tokens is not None: 

126 logging_obj.model_call_details["combined_usage_object"] = Usage(total_tokens=reported.total_tokens) 

127 return reported 

128 

129 

130def has_upstream_reported_usage(logging_obj: LiteLLMLoggingObj) -> bool: 

131 """Whether the upstream spoke this contract on the request's response. 

132 

133 True even when the value it sent was unusable: a target that reports its 

134 own totals owns the cost for the request, and a header we could not parse 

135 means zero, never a fallback to someone else's estimate. 

136 """ 

137 return isinstance(logging_obj.model_call_details.get(UPSTREAM_REPORTED_USAGE_KEY), UpstreamReportedUsage)