Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/auth/fallback_budget.py: 32%
57 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Enforce the caller's budget against router fallback targets.
4Budget is checked once, during auth, against the *requested* model group. A zero-cost group takes
5`_is_model_cost_zero`'s bypass and waives every budget check; the router then picks a fallback
6target after auth, inside `run_async_fallback`, and nothing re-checks budget on the group that
7actually bills. So a free model with a paid fallback spends without a gate.
9This predicate is injected into the router to re-check budget for each fallback target before it is
10attempted, mirroring `fallback_model_access.py`. It deliberately leaves the primary attempt alone:
11a zero-cost model is never blocked by budget, and only the paid fallback is refused. On by default;
12set `general_settings.enforce_fallback_budget: false` to restore the unguarded behaviour.
14Scope: the key's and the user's `max_budget`. Not covered yet, and each needs a read-only evaluation
15path before it can be: team, team-member, end-user, org, global and per-model budgets, whose
16auth-path functions enforce rather than report (they raise), so reusing them would fire threshold
17alerts and take spend reservations for a target that is then skipped; and the key's rolling
18`budget_limits` windows, whose accumulated spend lives only in per-window counters
19(`spend:key:{token}:window:{budget_duration}`), so enforcing them means more counter reads on the
20fallback path rather than reusing state auth already loaded.
22Two known limitations of that narrow scope, both shared with `fallback_model_access.py`:
24* This reads the spend counter, it does not reserve against it. Requests already in flight all
25 observe the same pre-billing figure, so a cap can be crossed by roughly the number of concurrent
26 fallbacks times their cost. Auth-time enforcement avoids this by pre-filling the counter through
27 `reserve_budget_for_request`, which the zero-cost bypass skips. Turning the soft cap into a hard
28 one means reserving per fallback attempt and reconciling on completion.
29* A request that reaches the router without `metadata["user_api_key_auth"]` is not restricted.
30 Only `add_litellm_data_to_request` populates that key, so endpoints that assemble metadata by
31 hand (for example `/queue/chat/completions`) fall through as unauthenticated.
32"""
34from collections.abc import Callable, Mapping
35from dataclasses import dataclass
36from typing import Final
38from pydantic import BaseModel, ValidationError
40from litellm._logging import verbose_proxy_logger
41from litellm.proxy._types import UserAPIKeyAuth
42from litellm.proxy.auth.auth_checks import (
43 _is_model_cost_zero, # pyright: ignore[reportPrivateUsage] # the zero-cost predicate the auth-time budget checks use; no public equivalent
44)
45from litellm.router import Router
48class _RequestMetadata(BaseModel):
49 user_api_key_auth: UserAPIKeyAuth | None = None
52class _FallbackBudgetSettings(BaseModel):
53 enforce_fallback_budget: bool = True
56def _token_in_metadata(metadata: object) -> UserAPIKeyAuth | None:
57 try:
58 return _RequestMetadata.model_validate(metadata).user_api_key_auth
59 except ValidationError:
60 return None
63def _user_api_key_auth_from_request(request_kwargs: Mapping[str, object]) -> UserAPIKeyAuth | None:
64 return next(
65 (
66 token
67 for field in ("metadata", "litellm_metadata")
68 if (token := _token_in_metadata(request_kwargs.get(field))) is not None
69 ),
70 None,
71 )
74def _enforced_by_general_settings() -> bool:
75 from litellm.proxy.proxy_server import general_settings
77 return _FallbackBudgetSettings.model_validate(general_settings).enforce_fallback_budget
80def _applies_user_budget_to_team_keys() -> bool:
81 from litellm.proxy.proxy_server import general_settings
83 return general_settings.get("apply_user_budget_to_team_keys") is True
86async def _counter_spend(counter_key: str, fallback_spend: float, max_budget: float) -> float:
87 """
88 Read a spend counter the same way the auth-time budget checks do.
90 `max_budget` is not advisory: it makes `get_current_spend` re-check the counter against the
91 authoritative recorded spend before admitting. A counter restored from an older Redis snapshot
92 reads as a hit rather than a clean miss, so without this the reseed path never runs and a
93 stale-low counter would keep admitting paid fallbacks past the cap.
94 """
95 from litellm.proxy.proxy_server import get_current_spend
97 return await get_current_spend(
98 counter_key=counter_key,
99 fallback_spend=fallback_spend,
100 max_budget=max_budget,
101 )
104async def is_token_within_budget_for_model(*, model: str, valid_token: UserAPIKeyAuth, llm_router: Router) -> bool:
105 """
106 True when the key and the user behind it can still pay for `model`.
108 A zero-cost fallback target is always allowed: refusing it would deny a request on spend some
109 other model accrued, which is the same reasoning behind the auth-time bypass.
110 """
111 if _is_model_cost_zero(model=model, llm_router=llm_router):
112 return True
114 key_budget: Final = valid_token.max_budget
115 if key_budget is not None and valid_token.token is not None:
116 key_spend: Final = await _counter_spend(
117 counter_key=f"spend:key:{valid_token.token}",
118 fallback_spend=valid_token.spend or 0.0,
119 max_budget=key_budget,
120 )
121 if key_spend >= key_budget:
122 return False
124 # Mirrors `_PROXY_MaxBudgetLimiter`: a team key does not carry the key owner's personal budget
125 # unless the proxy opts in, so the personal cap must not gate the fallback either.
126 user_budget: Final = valid_token.user_max_budget
127 if (
128 user_budget is not None
129 and valid_token.user_id is not None
130 and (valid_token.team_id is None or _applies_user_budget_to_team_keys())
131 ):
132 user_spend: Final = await _counter_spend(
133 counter_key=f"spend:user:{valid_token.user_id}",
134 fallback_spend=valid_token.user_spend or 0.0,
135 max_budget=user_budget,
136 )
137 if user_spend >= user_budget:
138 return False
140 return True
143@dataclass(frozen=True, slots=True)
144class RouterFallbackBudgetCheck:
145 """
146 `FallbackBudgetCheck` for the proxy's router: while `is_enforced()` is true, a paid fallback
147 target is attempted only when the caller is still within budget. Requests that carry no key
148 (for example internal health checks) are not restricted.
149 """
151 is_enforced: Callable[[], bool]
153 async def __call__(self, *, model: str, request_kwargs: Mapping[str, object], llm_router: Router) -> bool:
154 if not self.is_enforced():
155 return True
156 valid_token: Final = _user_api_key_auth_from_request(request_kwargs)
157 if valid_token is None:
158 return True
159 try:
160 return await is_token_within_budget_for_model(model=model, valid_token=valid_token, llm_router=llm_router)
161 except Exception as e: # noqa: BLE001 # fail closed: a spend lookup failure must not bill the caller
162 verbose_proxy_logger.warning("Skipping fallback to model=%s: budget lookup failed: %s", model, e)
163 return False
166router_fallback_budget_check: Final = RouterFallbackBudgetCheck(is_enforced=_enforced_by_general_settings)