Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/auth/fallback_budget.py: 32%

57 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Enforce the caller's budget against router fallback targets. 

3 

4Budget is checked once, during auth, against the *requested* model group. A zero-cost group takes 

5`_is_model_cost_zero`'s bypass and waives every budget check; the router then picks a fallback 

6target after auth, inside `run_async_fallback`, and nothing re-checks budget on the group that 

7actually bills. So a free model with a paid fallback spends without a gate. 

8 

9This predicate is injected into the router to re-check budget for each fallback target before it is 

10attempted, mirroring `fallback_model_access.py`. It deliberately leaves the primary attempt alone: 

11a zero-cost model is never blocked by budget, and only the paid fallback is refused. On by default; 

12set `general_settings.enforce_fallback_budget: false` to restore the unguarded behaviour. 

13 

14Scope: the key's and the user's `max_budget`. Not covered yet, and each needs a read-only evaluation 

15path before it can be: team, team-member, end-user, org, global and per-model budgets, whose 

16auth-path functions enforce rather than report (they raise), so reusing them would fire threshold 

17alerts and take spend reservations for a target that is then skipped; and the key's rolling 

18`budget_limits` windows, whose accumulated spend lives only in per-window counters 

19(`spend:key:{token}:window:{budget_duration}`), so enforcing them means more counter reads on the 

20fallback path rather than reusing state auth already loaded. 

21 

22Two known limitations of that narrow scope, both shared with `fallback_model_access.py`: 

23 

24* This reads the spend counter, it does not reserve against it. Requests already in flight all 

25 observe the same pre-billing figure, so a cap can be crossed by roughly the number of concurrent 

26 fallbacks times their cost. Auth-time enforcement avoids this by pre-filling the counter through 

27 `reserve_budget_for_request`, which the zero-cost bypass skips. Turning the soft cap into a hard 

28 one means reserving per fallback attempt and reconciling on completion. 

29* A request that reaches the router without `metadata["user_api_key_auth"]` is not restricted. 

30 Only `add_litellm_data_to_request` populates that key, so endpoints that assemble metadata by 

31 hand (for example `/queue/chat/completions`) fall through as unauthenticated. 

32""" 

33 

34from collections.abc import Callable, Mapping 

35from dataclasses import dataclass 

36from typing import Final 

37 

38from pydantic import BaseModel, ValidationError 

39 

40from litellm._logging import verbose_proxy_logger 

41from litellm.proxy._types import UserAPIKeyAuth 

42from litellm.proxy.auth.auth_checks import ( 

43 _is_model_cost_zero, # pyright: ignore[reportPrivateUsage] # the zero-cost predicate the auth-time budget checks use; no public equivalent 

44) 

45from litellm.router import Router 

46 

47 

48class _RequestMetadata(BaseModel): 

49 user_api_key_auth: UserAPIKeyAuth | None = None 

50 

51 

52class _FallbackBudgetSettings(BaseModel): 

53 enforce_fallback_budget: bool = True 

54 

55 

56def _token_in_metadata(metadata: object) -> UserAPIKeyAuth | None: 

57 try: 

58 return _RequestMetadata.model_validate(metadata).user_api_key_auth 

59 except ValidationError: 

60 return None 

61 

62 

63def _user_api_key_auth_from_request(request_kwargs: Mapping[str, object]) -> UserAPIKeyAuth | None: 

64 return next( 

65 ( 

66 token 

67 for field in ("metadata", "litellm_metadata") 

68 if (token := _token_in_metadata(request_kwargs.get(field))) is not None 

69 ), 

70 None, 

71 ) 

72 

73 

74def _enforced_by_general_settings() -> bool: 

75 from litellm.proxy.proxy_server import general_settings 

76 

77 return _FallbackBudgetSettings.model_validate(general_settings).enforce_fallback_budget 

78 

79 

80def _applies_user_budget_to_team_keys() -> bool: 

81 from litellm.proxy.proxy_server import general_settings 

82 

83 return general_settings.get("apply_user_budget_to_team_keys") is True 

84 

85 

86async def _counter_spend(counter_key: str, fallback_spend: float, max_budget: float) -> float: 

87 """ 

88 Read a spend counter the same way the auth-time budget checks do. 

89 

90 `max_budget` is not advisory: it makes `get_current_spend` re-check the counter against the 

91 authoritative recorded spend before admitting. A counter restored from an older Redis snapshot 

92 reads as a hit rather than a clean miss, so without this the reseed path never runs and a 

93 stale-low counter would keep admitting paid fallbacks past the cap. 

94 """ 

95 from litellm.proxy.proxy_server import get_current_spend 

96 

97 return await get_current_spend( 

98 counter_key=counter_key, 

99 fallback_spend=fallback_spend, 

100 max_budget=max_budget, 

101 ) 

102 

103 

104async def is_token_within_budget_for_model(*, model: str, valid_token: UserAPIKeyAuth, llm_router: Router) -> bool: 

105 """ 

106 True when the key and the user behind it can still pay for `model`. 

107 

108 A zero-cost fallback target is always allowed: refusing it would deny a request on spend some 

109 other model accrued, which is the same reasoning behind the auth-time bypass. 

110 """ 

111 if _is_model_cost_zero(model=model, llm_router=llm_router): 

112 return True 

113 

114 key_budget: Final = valid_token.max_budget 

115 if key_budget is not None and valid_token.token is not None: 

116 key_spend: Final = await _counter_spend( 

117 counter_key=f"spend:key:{valid_token.token}", 

118 fallback_spend=valid_token.spend or 0.0, 

119 max_budget=key_budget, 

120 ) 

121 if key_spend >= key_budget: 

122 return False 

123 

124 # Mirrors `_PROXY_MaxBudgetLimiter`: a team key does not carry the key owner's personal budget 

125 # unless the proxy opts in, so the personal cap must not gate the fallback either. 

126 user_budget: Final = valid_token.user_max_budget 

127 if ( 

128 user_budget is not None 

129 and valid_token.user_id is not None 

130 and (valid_token.team_id is None or _applies_user_budget_to_team_keys()) 

131 ): 

132 user_spend: Final = await _counter_spend( 

133 counter_key=f"spend:user:{valid_token.user_id}", 

134 fallback_spend=valid_token.user_spend or 0.0, 

135 max_budget=user_budget, 

136 ) 

137 if user_spend >= user_budget: 

138 return False 

139 

140 return True 

141 

142 

143@dataclass(frozen=True, slots=True) 

144class RouterFallbackBudgetCheck: 

145 """ 

146 `FallbackBudgetCheck` for the proxy's router: while `is_enforced()` is true, a paid fallback 

147 target is attempted only when the caller is still within budget. Requests that carry no key 

148 (for example internal health checks) are not restricted. 

149 """ 

150 

151 is_enforced: Callable[[], bool] 

152 

153 async def __call__(self, *, model: str, request_kwargs: Mapping[str, object], llm_router: Router) -> bool: 

154 if not self.is_enforced(): 

155 return True 

156 valid_token: Final = _user_api_key_auth_from_request(request_kwargs) 

157 if valid_token is None: 

158 return True 

159 try: 

160 return await is_token_within_budget_for_model(model=model, valid_token=valid_token, llm_router=llm_router) 

161 except Exception as e: # noqa: BLE001 # fail closed: a spend lookup failure must not bill the caller 

162 verbose_proxy_logger.warning("Skipping fallback to model=%s: budget lookup failed: %s", model, e) 

163 return False 

164 

165 

166router_fallback_budget_check: Final = RouterFallbackBudgetCheck(is_enforced=_enforced_by_general_settings)