Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/auth/user_api_key_auth.py: 41%
1247 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2This file handles authentication for the LiteLLM Proxy.
4it checks if the user passed a valid API Key to the LiteLLM Proxy
6Returns a UserAPIKeyAuth object if the API key is valid
8"""
10import asyncio
11import fnmatch
12import re
13import secrets
14from collections.abc import Mapping
15from datetime import datetime, timezone
16from typing import Any, Final, NamedTuple, Protocol, Union, cast
18import fastapi
19import orjson
20from fastapi import HTTPException, Request, WebSocket, status
21from fastapi.security.api_key import APIKeyHeader
22from starlette.exceptions import WebSocketException
24import litellm
25from litellm._logging import verbose_logger, verbose_proxy_logger
26from litellm._service_logger import ServiceLogging
27from litellm.caching.redis_cache import RedisCache
28from litellm.constants import (
29 CLIENT_REQUESTED_MODEL_SCOPE_KEY,
30 GLOBAL_PROXY_SPEND_CACHE_KEY,
31 INVALID_VIRTUAL_KEY_ERROR_MARKER,
32 INVALID_VIRTUAL_KEY_ERROR_MESSAGE,
33 LITELLM_PROXY_BUDGET_NAME,
34 LITELLM_PROXY_MASTER_KEY_ALIAS,
35 MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY,
36)
37from litellm.integrations.otel.model.config import is_otel_v2_enabled
38from litellm.integrations.otel.runtime import phase_span, seed_request_identity
39from litellm.litellm_core_utils.dd_tracing import tracer
40from litellm.litellm_core_utils.dot_notation_indexing import get_nested_value
41from litellm.proxy._types import *
42from litellm.proxy.agent_endpoints.auth.agent_caller import agent_caller_from_headers
43from litellm.proxy.auth.auth_checks import (
44 ExperimentalUIJWTToken,
45 TeamNotFoundError,
46 _cache_key_object,
47 _can_object_call_model,
48 _check_end_user_budget,
49 _delete_cache_key_object,
50 _get_user_role,
51 _is_model_cost_zero,
52 _is_user_proxy_admin,
53 _virtual_key_max_budget_alert_check,
54 _virtual_key_max_budget_check,
55 _virtual_key_soft_budget_check,
56 can_key_call_model,
57 common_checks,
58 get_end_user_object,
59 get_jwt_key_mapping_object,
60 get_key_end_user_budget_id,
61 get_object_permission,
62 get_org_object_for_request,
63 get_project_object,
64 get_team_membership,
65 get_team_object,
66 get_user_object,
67 is_valid_fallback_model,
68 jwt_key_mapping_cache_key,
69 key_model_aliases_for_auth_check,
70 resolve_and_validate_end_user_id,
71 resolve_default_end_user_budget,
72)
73from litellm.proxy.auth.auth_exception_handler import UserAPIKeyAuthExceptionHandler
74from litellm.proxy.auth.auth_method import AuthMethod
75from litellm.proxy.auth.auth_object_prefetch import AuthObjectRefs, prefetch_auth_objects
76from litellm.proxy.auth.auth_utils import (
77 abbreviate_api_key,
78 get_end_user_id_from_request_body,
79 get_model_from_request,
80 get_request_route,
81 get_request_route_template,
82 is_invalid_virtual_key_error,
83 iter_request_fallback_targets,
84 normalize_request_route,
85 pre_db_read_auth_checks,
86 request_dispatched_to_pass_through_endpoint,
87 request_dispatched_to_provider_pass_through,
88 route_in_additonal_public_routes,
89)
90from litellm.proxy.auth.handle_jwt import JWTAuthManager, JWTHandler
91from litellm.proxy.auth.network import TrustedProxyConfig, resolve_network_context
92from litellm.proxy.auth.oauth2_check import Oauth2Handler
93from litellm.proxy.auth.oauth2_proxy_hook import handle_oauth2_proxy_request
94from litellm.proxy.auth.resolvers import CredentialRef, Principal
95from litellm.proxy.auth.resolvers.grants import (
96 GrantResolver,
97 LookupDegraded,
98 ResolvedGrants,
99 UserLookup,
100 raise_public,
101 user_models,
102)
103from litellm.proxy.auth.resolvers.store import IdentityStore
104from litellm.proxy.auth.route_checks import RouteChecks
105from litellm.proxy.auth.team_grants import team_grants
106from litellm.proxy.auth.trusted_proxy_utils import get_trusted_proxy_cidrs
107from litellm.proxy.common_utils.cache_coordinator import EventDrivenCacheCoordinator
108from litellm.proxy.common_utils.http_parsing_utils import (
109 _read_request_body,
110 _safe_get_request_headers,
111 _safe_get_request_query_params,
112 _safe_set_request_parsed_body,
113 is_opaque_audio_pass_through_request,
114 populate_request_with_path_params,
115 read_raw_json_body,
116 rewrite_request_model,
117)
118from litellm.proxy.common_utils.model_listing_utils import claude_code_requested_group
119from litellm.proxy.common_utils.realtime_utils import _realtime_request_body
120from litellm.proxy.common_utils.user_api_key_cache import (
121 UserApiKeyCache,
122 team_membership_auth_cache_key,
123)
124from litellm.proxy.db.db_lookup_gate import bounded_db_lookup
125from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler
126from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
127from litellm.proxy.spend_tracking.carried_budget_state import carry_team_and_user_budget_state
128from litellm.proxy.spend_tracking.spend_counter_batch import (
129 bind_admission_counter_keys,
130 release_spend_counter_batch,
131 spend_counter_batch_scope,
132)
133from litellm.proxy.utils import (
134 PrismaClient,
135 ProxyLogging,
136 normalize_route_for_root_path,
137)
138from litellm.repositories.table_repositories import TeamMembershipRepository
139from litellm.router_utils.common_utils import resolve_model_group_alias
140from litellm.secret_managers.main import get_secret_bool
141from litellm.types.services import ServiceTypes
143try:
144 from litellm_enterprise.proxy.auth.user_api_key_auth import (
145 enterprise_custom_auth as _enterprise_custom_auth,
146 )
148 enterprise_custom_auth: Callable | None = _enterprise_custom_auth
149except ImportError as e:
150 verbose_proxy_logger.debug("Error in enterprise custom auth: %s", e)
151 enterprise_custom_auth = None
153user_api_key_service_logger_obj: Final = ServiceLogging() # used for tracking latency on OTEL
156def _normalize_public_auth_route(route: str) -> str:
157 if route != "/" and route.endswith("/"): 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true
158 return route.rstrip("/")
159 return route
162def _route_requires_auth_despite_public(route: str, general_settings: dict | None) -> bool:
163 normalized_route: Final = _normalize_public_auth_route(route)
164 if normalized_route == "/metrics": 164 ↛ 165line 164 didn't jump to line 165 because the condition on line 164 was never true
165 return litellm.require_auth_for_metrics_endpoint is not False
167 return False
170custom_litellm_key_header: Final = APIKeyHeader(
171 name=SpecialHeaders.custom_litellm_api_key.value,
172 auto_error=False,
173 description="Bearer token",
174)
175api_key_header: Final = APIKeyHeader(
176 name=SpecialHeaders.openai_authorization.value,
177 auto_error=False,
178 description="Bearer token",
179)
180azure_api_key_header: Final = APIKeyHeader(
181 name=SpecialHeaders.azure_authorization.value,
182 auto_error=False,
183 description="Some older versions of the openai Python package will send an API-Key header with just the API key ",
184)
185anthropic_api_key_header: Final = APIKeyHeader(
186 name=SpecialHeaders.anthropic_authorization.value,
187 auto_error=False,
188 description="If anthropic client used.",
189)
190google_ai_studio_api_key_header: Final = APIKeyHeader(
191 name=SpecialHeaders.google_ai_studio_authorization.value,
192 auto_error=False,
193 description="If google ai studio client used.",
194)
195azure_apim_header: Final = APIKeyHeader(
196 name=SpecialHeaders.azure_apim_authorization.value,
197 auto_error=False,
198 description="The default name of the subscription key header of Azure",
199)
202def _get_model_from_request_context(
203 request_data: dict,
204 route: str,
205 request: Request | None,
206 llm_router: Any | None = None,
207 team_id: str | None = None,
208) -> str | list[str] | None:
209 return get_model_from_request(
210 request_data=request_data,
211 route=route,
212 request_headers=_safe_get_request_headers(request=request),
213 request_query_params=_safe_get_request_query_params(request=request),
214 llm_router=llm_router,
215 request=request,
216 team_id=team_id,
217 )
220_CLAUDE_MODEL_ROUTES: Final = frozenset(
221 f"/{prefix}{endpoint}" for prefix in ("", "v1/") for endpoint in ("messages", "chat/completions", "responses")
222)
223_CLAUDE_MODEL_NORMALIZED: Final = "litellm.claude_model_normalized"
226async def _normalize_claude_model(
227 request_data: dict, valid_token: UserAPIKeyAuth, request: Request | None, route: str
228) -> None:
229 from litellm.proxy.proxy_server import llm_router, prisma_client, proxy_config, proxy_logging_obj
231 if route not in _CLAUDE_MODEL_ROUTES or llm_router is None:
232 return
233 if request is not None and request.scope.get(_CLAUDE_MODEL_NORMALIZED) is True: 233 ↛ 234line 233 didn't jump to line 234 because the condition on line 233 was never true
234 return
235 requested: Final = _get_model_from_request_context(request_data, route, request, llm_router, valid_token.team_id)
236 if not isinstance(requested, str) or requested != request_data.get("model"):
237 return
238 if not requested.startswith("claude-router-") and not requested.lower().endswith("[1m]"): 238 ↛ 240line 238 didn't jump to line 240 because the condition on line 238 was always true
239 return
240 settings: Final = await proxy_config.get_hierarchical_router_settings(
241 user_api_key_dict=valid_token, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj
242 )
243 aliases: Final = settings.get("model_group_alias") if isinstance(settings, Mapping) else None
244 source: Final = claude_code_requested_group(
245 requested, llm_router, valid_token.team_id, (valid_token.aliases, valid_token.team_model_aliases, aliases)
246 )
247 if request is not None:
248 request.scope[_CLAUDE_MODEL_NORMALIZED] = True
249 if source is None:
250 return
251 rewrite_request_model(request_data, request, source)
254async def _resolve_router_settings_model_group_alias(
255 request_data: dict[str, object], # mutable-ok: the request body is rewritten in place for every downstream reader
256 valid_token: UserAPIKeyAuth,
257 request: Request | None,
258 route: str,
259) -> None:
260 """Rewrite the requested model through the key's or team's ``router_settings.model_group_alias``
261 before the allowlist checks, so they authorize the model group the request is routed to.
262 """
263 from litellm.proxy.proxy_server import llm_router, prisma_client, proxy_config, proxy_logging_obj
265 if request is None or llm_router is None or not RouteChecks.is_llm_api_route(route=route):
266 return
267 if request.scope.get(MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY) is True:
268 return
269 request.scope[MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY] = True
270 if request_dispatched_to_pass_through_endpoint(request) or request_dispatched_to_provider_pass_through(request):
271 return
272 requested: Final = request_data.get("model")
273 if not isinstance(requested, str) or await read_raw_json_body(request=request) is None:
274 return
275 settings: Final = await proxy_config.get_hierarchical_router_settings(
276 user_api_key_dict=valid_token, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj
277 )
278 if not isinstance(settings, Mapping): 278 ↛ 280line 278 didn't jump to line 280 because the condition on line 278 was always true
279 return
280 target: Final = resolve_model_group_alias(settings.get("model_group_alias"), requested)
281 if target is None or target == requested:
282 return
283 verbose_proxy_logger.debug(
284 "router_settings.model_group_alias resolved %s -> %s before auth",
285 requested.replace("\r", "").replace("\n", ""),
286 target.replace("\r", "").replace("\n", ""),
287 )
288 request.scope.setdefault(CLIENT_REQUESTED_MODEL_SCOPE_KEY, requested)
289 rewrite_request_model(request_data, request, target)
292def _get_model_names_for_budget_checks(
293 model: str | list[str] | None,
294) -> list[str]:
295 if model is None:
296 return []
297 if isinstance(model, str):
298 return [model]
299 return model
302class _KeyModelBudgetLimiter(Protocol):
303 async def is_key_within_model_budget(self, user_api_key_dict: UserAPIKeyAuth, model: str) -> bool: ... 303 ↛ exitline 303 didn't return from function 'is_key_within_model_budget' because
305 async def get_fallback_model_within_budget(self, user_api_key_dict: UserAPIKeyAuth, model: str) -> str | None: ... 305 ↛ exitline 305 didn't return from function 'get_fallback_model_within_budget' because
308class _UserModelBudgetLimiter(Protocol):
309 async def is_user_within_model_budget( 309 ↛ exitline 309 didn't return from function 'is_user_within_model_budget' because
310 self, user_id: str, user_model_max_budget: Mapping[str, object], model: str
311 ) -> bool: ...
314class _TeamModelBudgetLimiter(Protocol):
315 async def is_team_within_model_budget( 315 ↛ exitline 315 didn't return from function 'is_team_within_model_budget' because
316 self,
317 team_id: str,
318 team_model_max_budget: Mapping[str, object],
319 key_model_max_budget: Mapping[str, object] | None,
320 model: str,
321 ) -> bool: ...
324class _TokenTeamModels(Protocol):
325 @property
326 def team_models(self) -> list[str]: ... 326 ↛ exitline 326 didn't return from function 'team_models' because
329class _RawCacheRead(Protocol):
330 async def async_get_cache(self, *, key: str) -> object: ... 330 ↛ exitline 330 didn't return from function 'async_get_cache' because
333def _raw_cache(cache: _RawCacheRead) -> _RawCacheRead:
334 """View an untyped cache object's ``async_get_cache`` as returning ``object``
335 instead of ``Any``, so a caller can ``isinstance``-narrow it without paying
336 the ``reportAny`` cost of the underlying (unannotated) cache implementation."""
337 return cache
340def _token_team_models(valid_token: _TokenTeamModels) -> list[str]:
341 return valid_token.team_models
344async def _read_user_model_max_budget(
345 user_id: str | None,
346 prisma_client: PrismaClient | None,
347 user_api_key_cache: UserApiKeyCache,
348 parent_otel_span: Span | None,
349 proxy_logging_obj: ProxyLogging,
350) -> Mapping[str, object] | None:
351 """The user row's `model_max_budget`, or None when the row cannot be read.
353 A user whose row is missing must not be refused: this is a budget lookup,
354 and the main auth path likewise treats an unreadable user as no user.
355 """
356 if user_id is None or prisma_client is None:
357 return None
358 try:
359 user_obj: Final = await get_user_object(
360 user_id=user_id,
361 prisma_client=prisma_client,
362 user_api_key_cache=user_api_key_cache,
363 user_id_upsert=False,
364 parent_otel_span=parent_otel_span,
365 proxy_logging_obj=proxy_logging_obj,
366 )
367 except Exception as e: # noqa: BLE001 # mirrors the main path's tolerance
368 verbose_logger.debug("Unable to read user for the per-model budget check: %s", e)
369 return None
370 return user_obj.model_max_budget if user_obj is not None else None
373async def _check_user_model_budget(
374 valid_token: UserAPIKeyAuth,
375 model_max_budget_limiter: _UserModelBudgetLimiter,
376 models: list[str],
377) -> None:
378 """Enforce the internal user's own `model_max_budget` across the request's models.
380 Separate from the key check: a user's per-model budget caps every key they
381 own, so a caller cannot escape it by minting another key.
382 """
383 user_model_max_budget: Final = valid_token.user_model_max_budget
384 if valid_token.user_id is None or not isinstance(user_model_max_budget, Mapping) or not user_model_max_budget:
385 return
386 for model_name in models:
387 await model_max_budget_limiter.is_user_within_model_budget(
388 user_id=valid_token.user_id,
389 user_model_max_budget=user_model_max_budget,
390 model=model_name,
391 )
394async def _check_team_model_budget(
395 valid_token: UserAPIKeyAuth,
396 model_max_budget_limiter: _TeamModelBudgetLimiter,
397 models: list[str],
398) -> None:
399 """Enforce the team's `model_max_budget` for every requested model the key does not override."""
400 team_model_max_budget: Final = valid_token.team_model_max_budget
401 if valid_token.team_id is None or not team_model_max_budget: 401 ↛ 403line 401 didn't jump to line 403 because the condition on line 401 was always true
402 return
403 key_model_max_budget: Final[Mapping[str, object] | None] = valid_token.model_max_budget
404 for model_name in models:
405 await model_max_budget_limiter.is_team_within_model_budget(
406 team_id=valid_token.team_id,
407 team_model_max_budget=team_model_max_budget,
408 key_model_max_budget=key_model_max_budget,
409 model=model_name,
410 )
413async def _check_key_model_budget_with_fallback(
414 valid_token: UserAPIKeyAuth,
415 model_max_budget_limiter: _KeyModelBudgetLimiter,
416 model_name: str,
417 request_data: dict,
418 request: Request,
419 llm_model_list: list | None = None,
420 llm_router: litellm.Router | None = None,
421) -> None:
422 """
423 Enforce the key's per-model budget for `model_name`. If exceeded and the
424 key has a `budget_fallbacks` chain configured for `model_name`, reroute
425 the request to the first fallback model still within its own budget
426 instead of rejecting the request.
428 The selected fallback is validated against the key's model-access
429 allowlist and the team's model restrictions so that budget_fallbacks
430 cannot bypass model authorization. The rewrite is persisted to the
431 parsed-body cache, Starlette's JSON cache (``request._json``), and
432 path parameters so that downstream handlers see the final model
433 regardless of whether they consume ``_read_request_body()``,
434 ``request.json()``, or the path ``model`` parameter.
436 Fallback is only attempted when ``model_name`` matches the top-level
437 ``request_data["model"]``; models extracted from nested fields
438 (``session.model``, ``completion.model``, etc.) are not rewritable
439 and raise immediately.
441 Raises:
442 BudgetExceededError: if `model_name` is over budget and no configured
443 fallback is within budget either (or the fallback is not authorized).
444 """
445 try:
446 await model_max_budget_limiter.is_key_within_model_budget(
447 user_api_key_dict=valid_token,
448 model=model_name,
449 )
450 except litellm.BudgetExceededError as e:
451 if request_data.get("model") != model_name:
452 raise e
453 fallback_model: Final = await model_max_budget_limiter.get_fallback_model_within_budget(
454 user_api_key_dict=valid_token,
455 model=model_name,
456 )
457 if fallback_model is None:
458 raise e
459 try:
460 await can_key_call_model(
461 model=fallback_model,
462 llm_model_list=llm_model_list,
463 valid_token=valid_token,
464 llm_router=llm_router,
465 )
466 if valid_token.team_models:
467 _can_object_call_model(
468 model=fallback_model,
469 llm_router=llm_router,
470 models=valid_token.team_models,
471 team_model_aliases=valid_token.team_model_aliases,
472 team_id=valid_token.team_id,
473 key_model_aliases=key_model_aliases_for_auth_check(valid_token),
474 object_type="team",
475 )
476 except ProxyException:
477 raise e
478 request_data["model"] = fallback_model
479 _safe_set_request_parsed_body(request=request, parsed_body=request_data)
480 request._json = request_data
481 request._body = orjson.dumps(request_data)
482 path_params: Final = request.scope.get("path_params")
483 if isinstance(path_params, dict) and "model" in path_params:
484 path_params["model"] = fallback_model
487def _get_bearer_token_or_received_api_key(api_key: str) -> str:
488 if api_key.startswith("Bearer "): # ensure Bearer token passed in 488 ↛ 489line 488 didn't jump to line 489 because the condition on line 488 was never true
489 api_key = api_key.replace("Bearer ", "") # extract the token
490 elif api_key.startswith("Basic "): 490 ↛ 491line 490 didn't jump to line 491 because the condition on line 490 was never true
491 api_key = api_key.replace("Basic ", "") # handle langfuse input
492 elif api_key.startswith("bearer "): 492 ↛ 493line 492 didn't jump to line 493 because the condition on line 492 was never true
493 api_key = api_key.replace("bearer ", "")
494 elif api_key.startswith("AWS4-HMAC-SHA256"): 494 ↛ 498line 494 didn't jump to line 498 because the condition on line 494 was never true
495 # Handle AWS Signature V4 format from LangChain
496 # Format: AWS4-HMAC-SHA256 Credential=Bearer sk-12345/date/region/service/aws4_request, SignedHeaders=..., Signature=...
497 # Extract the Bearer token from the Credential field
498 match = re.search(r"Credential=Bearer\s+([^/\s,]+)", api_key)
499 if match:
500 api_key = match.group(1)
501 else:
502 # If no Bearer token found in Credential, try to extract just the credential value
503 match = re.search(r"Credential=([^/\s,]+)", api_key)
504 if match:
505 api_key = match.group(1)
507 return api_key
510def _routing_selector_matches_claim(
511 selector_value: Any | None,
512 claim_value: Any | None,
513 *,
514 split_space_delimited: bool = False,
515) -> bool:
516 if selector_value is None:
517 return True
519 selector_list: Final[list[str]] = (
520 [str(v) for v in selector_value] if isinstance(selector_value, list) else [str(selector_value)]
521 )
523 if claim_value is None:
524 return False
526 if isinstance(claim_value, list):
527 claim_list = [str(v) for v in claim_value]
528 elif split_space_delimited and isinstance(claim_value, str) and " " in claim_value.strip():
529 # OAuth/OIDC often sends scope as a single space-delimited string. Only split
530 # for the scope selector: iss/aud/client_id must stay exact full-string match
531 # on unverified claims (see routing override security review). The elif guard
532 # (`" " in claim_value.strip()`) ensures at least two non-empty tokens survive.
533 claim_list = [v for v in claim_value.strip().split(" ") if v]
534 else:
535 claim_list = [str(claim_value)]
537 def _selector_matches_claim(selector: str, claim: str) -> bool:
538 # NOTE: wildcard matching is case-sensitive (fnmatch.fnmatchcase).
539 if "*" in selector or "?" in selector:
540 # Without scope splitting, do not let `*` span whitespace: a malformed
541 # iss like "trusted.example.com evil.com" must not match "trusted.*".
542 # Scope uses split_space_delimited so each claim token is checked separately.
543 if not split_space_delimited and any(ch.isspace() for ch in claim):
544 return False
545 return fnmatch.fnmatchcase(claim, selector)
546 return selector == claim
548 return any(_selector_matches_claim(selector=s, claim=c) for s in selector_list for c in claim_list)
551def _matches_routing_override(token_claims: dict, override: "JWTRoutingOverride") -> bool:
552 return (
553 _routing_selector_matches_claim(override.iss, token_claims.get("iss"))
554 and _routing_selector_matches_claim(override.client_id, token_claims.get("client_id"))
555 and _routing_selector_matches_claim(
556 override.scope,
557 token_claims.get("scope"),
558 split_space_delimited=True,
559 )
560 and _routing_selector_matches_claim(override.aud, token_claims.get("aud"))
561 )
564def _should_route_jwt_to_oauth2_override(token: str, jwt_handler: JWTHandler) -> bool:
565 routing_overrides: Final = jwt_handler.litellm_jwtauth.routing_overrides
566 if not routing_overrides:
567 return False
569 token_claims: Final = jwt_handler.get_unverified_claims(token=token)
570 if token_claims is None:
571 return False
573 for override in routing_overrides:
574 if override.path == "oauth2" and _matches_routing_override(token_claims=token_claims, override=override):
575 verbose_proxy_logger.debug("JWT routing override matched. Routing token to OAuth2 introspection.")
576 return True
578 return False
581def _get_bearer_token(
582 api_key: str,
583):
584 if api_key.startswith("Bearer "): # ensure Bearer token passed in
585 api_key = api_key.replace("Bearer ", "") # extract the token
586 elif api_key.startswith("Basic "): 586 ↛ 587line 586 didn't jump to line 587 because the condition on line 586 was never true
587 api_key = api_key.replace("Basic ", "") # handle langfuse input
588 elif api_key.startswith("bearer "): 588 ↛ 589line 588 didn't jump to line 589 because the condition on line 588 was never true
589 api_key = api_key.replace("bearer ", "")
590 elif api_key.startswith("AWS4-HMAC-SHA256"): 590 ↛ 594line 590 didn't jump to line 594 because the condition on line 590 was never true
591 # Handle AWS Signature V4 format from LangChain
592 # Format: AWS4-HMAC-SHA256 Credential=Bearer sk-12345/date/region/service/aws4_request, SignedHeaders=..., Signature=...
593 # Extract the Bearer token from the Credential field
594 match = re.search(r"Credential=Bearer\s+([^/\s,]+)", api_key)
595 if match:
596 api_key = match.group(1)
597 else:
598 # If no Bearer token found in Credential, try to extract just the credential value
599 match = re.search(r"Credential=([^/\s,]+)", api_key)
600 if match:
601 api_key = match.group(1)
602 else:
603 api_key = ""
604 else:
605 api_key = ""
606 return api_key
609def _apply_budget_limits_to_end_user_params(
610 end_user_params: dict,
611 budget_info: LiteLLM_BudgetTable,
612 end_user_id: str | None,
613) -> None:
614 """
615 Helper function to apply budget limits to end user parameters.
617 Args:
618 end_user_params: Dictionary to update with budget parameters
619 budget_info: Budget table object containing limits
620 end_user_id: ID of the end user for logging
621 """
622 if budget_info.tpm_limit is not None:
623 end_user_params["end_user_tpm_limit"] = budget_info.tpm_limit
625 if budget_info.rpm_limit is not None:
626 end_user_params["end_user_rpm_limit"] = budget_info.rpm_limit
628 if budget_info.tpd_limit is not None:
629 end_user_params["end_user_tpd_limit"] = budget_info.tpd_limit
631 if budget_info.max_budget is not None:
632 end_user_params["end_user_max_budget"] = budget_info.max_budget
634 if budget_info.model_max_budget is not None:
635 end_user_params["end_user_model_max_budget"] = budget_info.model_max_budget
637 verbose_proxy_logger.debug("Applied budget limits to end user %s", end_user_id)
640async def user_api_key_auth_websocket(websocket: WebSocket) -> UserAPIKeyAuth:
641 return await user_api_key_auth_websocket_for_model(websocket, model=websocket.query_params.get("model"))
644async def user_api_key_auth_websocket_for_model(websocket: WebSocket, model: str | None) -> UserAPIKeyAuth:
645 ws_scope: Final = websocket.scope or {}
646 scope_headers: Final = list(ws_scope.get("headers") or [])
647 # ``get_request_route`` falls back to ``request.url.path`` when
648 # ``scope["path"]`` is absent. On WebSockets that fallback reads
649 # ``websocket.url``, which Starlette reconstructs from the (poisonable)
650 # Host header. Carry the ASGI scope's path / root_path so the lookup
651 # never reaches the fallback.
652 synthetic_scope: Final[dict[str, Any]] = {
653 "type": "http",
654 "headers": scope_headers,
655 "path": ws_scope.get("path", ""),
656 "state": ws_scope.setdefault("state", {}), # mutable-ok: Starlette's socket state, shared with the request
657 }
658 for key in ("root_path", "app_root_path"):
659 if key in ws_scope:
660 synthetic_scope[key] = ws_scope[key]
661 request: Final = Request(scope=synthetic_scope)
663 request._url = websocket.url
665 async def return_body():
666 return _realtime_request_body(model)
668 request.body = return_body
670 authorization: Final = websocket.headers.get("authorization")
671 # If no Authorization header, try the api-key header
672 if not authorization:
673 api_key = websocket.headers.get("api-key")
674 if not api_key:
675 # Try extracting from WebSocket subprotocol (browser clients)
676 for protocol in websocket.headers.get("sec-websocket-protocol", "").split(","):
677 protocol = protocol.strip()
678 if protocol.startswith("openai-insecure-api-key."):
679 api_key = protocol[len("openai-insecure-api-key.") :]
680 break
681 if not api_key:
682 await websocket.close(code=status.WS_1008_POLICY_VIOLATION)
683 raise HTTPException(status_code=403, detail="No API key provided")
684 else:
685 # Extract the API key from the Bearer token
686 if not authorization.startswith("Bearer "):
687 await websocket.close(code=status.WS_1008_POLICY_VIOLATION)
688 raise HTTPException(status_code=403, detail="Invalid Authorization header format")
690 api_key = authorization[len("Bearer ") :].strip()
692 # Call user_api_key_auth with the extracted API key
693 # Note: You'll need to modify this to work with WebSocket context if needed
694 try:
695 return await user_api_key_auth(request=request, api_key=f"Bearer {api_key}")
696 except Exception as e:
697 if is_invalid_virtual_key_error(e):
698 raise WebSocketException(code=status.WS_1008_POLICY_VIOLATION)
699 verbose_proxy_logger.exception(e)
700 await websocket.close(code=status.WS_1008_POLICY_VIOLATION)
701 raise HTTPException(status_code=403, detail=str(e))
704def update_valid_token_with_end_user_params(valid_token: UserAPIKeyAuth, end_user_params: dict) -> UserAPIKeyAuth:
705 valid_token.end_user_id = end_user_params.get("end_user_id")
706 # Only overwrite token fields when the DB-derived value is not None.
707 # This prevents DB lookups (where the budget table has no value set)
708 # from silently clearing values that a custom auth function may have
709 # already set on the token.
710 if end_user_params.get("end_user_tpm_limit") is not None: 710 ↛ 711line 710 didn't jump to line 711 because the condition on line 710 was never true
711 valid_token.end_user_tpm_limit = end_user_params["end_user_tpm_limit"]
712 if end_user_params.get("end_user_rpm_limit") is not None: 712 ↛ 713line 712 didn't jump to line 713 because the condition on line 712 was never true
713 valid_token.end_user_rpm_limit = end_user_params["end_user_rpm_limit"]
714 if end_user_params.get("end_user_tpd_limit") is not None: 714 ↛ 715line 714 didn't jump to line 715 because the condition on line 714 was never true
715 valid_token.end_user_tpd_limit = end_user_params["end_user_tpd_limit"]
716 if end_user_params.get("allowed_model_region") is not None: 716 ↛ 717line 716 didn't jump to line 717 because the condition on line 716 was never true
717 valid_token.allowed_model_region = end_user_params["allowed_model_region"]
718 if end_user_params.get("end_user_model_max_budget") is not None: 718 ↛ 719line 718 didn't jump to line 719 because the condition on line 718 was never true
719 valid_token.end_user_model_max_budget = end_user_params["end_user_model_max_budget"]
720 return valid_token
723# Reusable coordinator for global spend to prevent cache stampede
724_global_spend_coordinator: Final = EventDrivenCacheCoordinator(log_prefix="[GLOBAL SPEND]")
727async def _fetch_global_spend_with_event_coordination(
728 cache_key: str,
729 user_api_key_cache: UserApiKeyCache,
730 prisma_client: PrismaClient,
731) -> float | None:
732 """
733 Fetch global spend with event-driven coordination to prevent cache stampede.
734 Uses EventDrivenCacheCoordinator: first request queries DB and signals others when done.
736 Reads the proxy budget aggregate user row, which accrues proxy-wide spend
737 per request and is zeroed by ResetBudgetJob every ``litellm.budget_duration``.
738 """
740 async def _load_global_spend() -> float | None:
741 proxy_budget_row: Final = await bounded_db_lookup(
742 prisma_client.db.litellm_usertable.find_unique(where={"user_id": LITELLM_PROXY_BUDGET_NAME}),
743 name="proxy_budget",
744 )
745 return float(proxy_budget_row.spend) if proxy_budget_row is not None else None
747 return await _global_spend_coordinator.get_or_load(
748 cache_key=cache_key,
749 cache=user_api_key_cache, # pyright: ignore[reportArgumentType]
750 load_fn=_load_global_spend,
751 )
754async def get_global_proxy_spend(
755 litellm_proxy_admin_name: str,
756 user_api_key_cache: UserApiKeyCache,
757 prisma_client: PrismaClient | None,
758 token: str,
759 proxy_logging_obj: ProxyLogging,
760) -> float | None:
761 global_proxy_spend = None
762 if litellm.max_budget > 0 and prisma_client is not None: # user set proxy max budget 762 ↛ 764line 762 didn't jump to line 764 because the condition on line 762 was never true
763 # Use event-driven coordination to prevent cache stampede
764 cache_key: Final = GLOBAL_PROXY_SPEND_CACHE_KEY
765 global_proxy_spend = await _fetch_global_spend_with_event_coordination(
766 cache_key=cache_key,
767 user_api_key_cache=user_api_key_cache,
768 prisma_client=prisma_client,
769 )
770 if global_proxy_spend is not None:
771 user_info: Final = CallInfo(
772 user_id=litellm_proxy_admin_name,
773 max_budget=litellm.max_budget,
774 spend=global_proxy_spend,
775 token=token,
776 event_group=Litellm_EntityType.PROXY,
777 )
778 asyncio.create_task(
779 proxy_logging_obj.budget_alerts(
780 type="proxy_budget",
781 user_info=user_info,
782 )
783 )
784 return global_proxy_spend
787def get_rbac_role(jwt_handler: JWTHandler, scopes: list[str]) -> str:
788 is_admin: Final = jwt_handler.is_admin(scopes=scopes)
789 if is_admin:
790 return LitellmUserRoles.PROXY_ADMIN
791 else:
792 return LitellmUserRoles.TEAM
795def get_api_key(
796 custom_litellm_key_header: str | None,
797 api_key: str,
798 azure_api_key_header: str | None,
799 anthropic_api_key_header: str | None,
800 google_ai_studio_api_key_header: str | None,
801 azure_apim_header: str | None,
802 pass_through_endpoints: list[dict] | None,
803 route: str,
804 request: Request,
805) -> tuple[str, str | None]:
806 """
807 Returns:
808 Tuple[Optional[str], Optional[str]]: Tuple of the api_key and the passed_in_key
809 """
810 from litellm.proxy.auth.route_checks import RouteChecks
811 from litellm.proxy.common_utils.http_parsing_utils import (
812 _safe_get_request_query_params,
813 )
815 api_key = api_key
816 passed_in_key: str | None = None
817 if isinstance(custom_litellm_key_header, str):
818 passed_in_key = custom_litellm_key_header
819 api_key = _get_bearer_token_or_received_api_key(custom_litellm_key_header)
820 elif isinstance(api_key, str) and len(api_key) > 0:
821 passed_in_key = api_key
822 api_key = _get_bearer_token(api_key=api_key)
823 elif isinstance(azure_api_key_header, str): 823 ↛ 824line 823 didn't jump to line 824 because the condition on line 823 was never true
824 passed_in_key = azure_api_key_header
825 api_key = azure_api_key_header
826 elif isinstance(anthropic_api_key_header, str): 826 ↛ 827line 826 didn't jump to line 827 because the condition on line 826 was never true
827 passed_in_key = anthropic_api_key_header
828 api_key = anthropic_api_key_header
829 elif isinstance(google_ai_studio_api_key_header, str): 829 ↛ 830line 829 didn't jump to line 830 because the condition on line 829 was never true
830 passed_in_key = google_ai_studio_api_key_header
831 api_key = google_ai_studio_api_key_header
832 elif isinstance(azure_apim_header, str): 832 ↛ 833line 832 didn't jump to line 833 because the condition on line 832 was never true
833 passed_in_key = azure_apim_header
834 api_key = azure_apim_header
835 elif ( 835 ↛ 840line 835 didn't jump to line 840 because the condition on line 835 was never true
836 RouteChecks.is_generate_content_route(route=route)
837 and request is not None
838 and _safe_get_request_query_params(request).get("key")
839 ):
840 google_auth_key: Final[str] = _safe_get_request_query_params(request).get("key") or ""
841 passed_in_key = google_auth_key
842 api_key = google_auth_key
843 elif pass_through_endpoints is not None:
844 for endpoint in pass_through_endpoints:
845 if endpoint.get("path", "") == route: 845 ↛ 846line 845 didn't jump to line 846 because the condition on line 845 was never true
846 headers: dict | None = endpoint.get("headers", None)
847 if headers is not None:
848 header_key: str = headers.get("litellm_user_api_key", "")
849 if request.headers.get(header_key) is not None:
850 api_key = request.headers.get(header_key) or ""
851 passed_in_key = api_key
852 return api_key, passed_in_key
855async def check_api_key_for_custom_headers_or_pass_through_endpoints(
856 request: Request,
857 route: str,
858 pass_through_endpoints: list[dict] | None,
859 api_key: str,
860) -> UserAPIKeyAuth | str:
861 is_mapped_pass_through_route: bool = False
862 normalized_route: Final = normalize_route_for_root_path(route)
863 if normalized_route is not None: 863 ↛ 868line 863 didn't jump to line 868 because the condition on line 863 was always true
864 for mapped_route in LiteLLMRoutes.mapped_pass_through_routes.value:
865 if normalized_route.startswith(mapped_route):
866 is_mapped_pass_through_route = True
867 break
868 if is_mapped_pass_through_route:
869 if request.headers.get("litellm_user_api_key") is not None: 869 ↛ 870line 869 didn't jump to line 870 because the condition on line 869 was never true
870 api_key = request.headers.get("litellm_user_api_key") or ""
871 if pass_through_endpoints is not None:
872 for endpoint in pass_through_endpoints:
873 if isinstance(endpoint, dict) and endpoint.get("path", "") == route: 873 ↛ 880line 873 didn't jump to line 880 because the condition on line 873 was never true
874 ## IF AUTH DISABLED
875 # Default to True: a config dict with no ``auth`` key
876 # otherwise produced an unauthenticated forwarder. The
877 # Pydantic ``PassThroughGenericEndpoint.auth`` default
878 # is also True, but raw config dicts skip that path —
879 # so this runtime check has to default to True too.
880 if endpoint.get("auth", True) is not True:
881 return UserAPIKeyAuth()
882 ## IF AUTH ENABLED
883 ### IF CUSTOM PARSER REQUIRED
884 if endpoint.get("custom_auth_parser") is not None and endpoint.get("custom_auth_parser") == "langfuse":
885 # langfuse returns {'Authorization': 'Basic <base64(username:password)>'}
886 # check the langfuse public key if it contains the litellm api key
887 import base64
889 api_key = api_key.replace("Basic ", "").strip()
890 decoded_bytes = base64.b64decode(api_key)
891 decoded_str = decoded_bytes.decode("utf-8")
892 api_key = decoded_str.split(":")[0]
893 else:
894 headers = endpoint.get("headers", None)
895 if headers is not None:
896 header_key = headers.get("litellm_user_api_key", "")
897 if isinstance(request.headers, dict) and request.headers.get(key=header_key) is not None:
898 api_key = request.headers.get(key=header_key)
899 return api_key
902# Cache sentinel written when a JWT under AUTO_REGISTER resolved to a proxy
903# admin via auth_builder. Proxy admins don't need a mapped virtual key (they
904# have full access via auth_builder anyway), but without a cache entry every
905# subsequent request from the same JWT identity would re-query the DB for a
906# non-existent mapping. Sentinel tells _resolve_jwt_to_virtual_key to skip
907# the lookup and return None (caller proceeds to auth_builder).
908_JWT_PROXY_ADMIN_SENTINEL: Final = "__JWT_PROXY_ADMIN__"
910_JWT_AUTH_DISABLED_HINT = (
911 " This key has the structure of a JWT, but JWT auth is not enabled on this proxy, so it was treated as a"
912 " virtual key. Set `enable_jwt_auth: true` under `general_settings` in your proxy config to authenticate"
913 " with JWTs."
914)
917class _PendingAutoRegister(NamedTuple):
918 """
919 Signal returned by ``_resolve_jwt_to_virtual_key`` when the JWT's claim is
920 unmapped and ``unregistered_jwt_client_behavior`` is AUTO_REGISTER.
922 The caller MUST run standard ``JWTAuthManager.auth_builder`` to apply RBAC,
923 scope mappings, ``custom_validate``, and ``user_allowed_email_domain``
924 policy BEFORE calling ``_auto_register_jwt_mapping`` with the validated
925 ``team_id`` / ``user_id`` from the auth_builder result. Auto-registering
926 purely on a signature-valid JWT (the old behavior) bypassed every JWT
927 policy beyond signature verification.
928 """
930 claim_field: str
931 claim_value: str
932 cache_key: str
933 jwt_issuer: str | None = None
936async def _auto_register_jwt_mapping(
937 virtual_key_claim_field: str,
938 claim_value: str,
939 jwt_handler: JWTHandler,
940 prisma_client: PrismaClient,
941 user_api_key_cache: UserApiKeyCache,
942 parent_otel_span: Span | None,
943 proxy_logging_obj: ProxyLogging,
944 cache_key: str,
945 jwt_issuer: str | None = None,
946 team_id: str | None = None,
947 user_id: str | None = None,
948 org_id: str | None = None,
949 end_user_id: str | None = None,
950 agent_id: str | None = None,
951) -> UserAPIKeyAuth | None:
952 """
953 Auto-register: create a new virtual key + mapping for an unrecognised JWT
954 claim value. ``team_id`` and ``user_id`` must come from a successful
955 ``JWTAuthManager.auth_builder`` run — they encode the JWT identity AFTER
956 RBAC/scope/custom_validate/email-domain policy has been enforced. The key
957 is stamped with those values so the cached future-request path inherits
958 the same team/user/org limits the auth_builder path would have applied.
960 Race safety: if two concurrent requests both reach here simultaneously (both
961 saw no mapping in the DB), one will win the unique-constraint race on
962 litellm_jwtkeymapping. The loser catches the conflict, deletes its orphaned
963 key, fetches the winner's mapping, and proceeds — no error surfaced.
964 """
965 # Inline import required: key_management_endpoints imports user_api_key_auth
966 # (line 51) so a module-level import here would create a circular dependency.
967 from litellm.proxy.management_endpoints.key_management_endpoints import (
968 generate_key_helper_fn,
969 )
971 # ``table_name="key"`` is required: without it, generate_key_helper_fn
972 # falls into the user-upsert branch (`table_name is None or "user"`) and
973 # attempts to insert into LiteLLM_UserTable with user_id=None, which fails
974 # the NOT NULL @id constraint. Every successful key-creation caller (e.g.
975 # /key/generate) passes table_name="key" explicitly.
976 key_data: Final = await generate_key_helper_fn(
977 llm_router=None,
978 request_type="key",
979 table_name="key",
980 team_id=team_id,
981 user_id=user_id,
982 organization_id=org_id,
983 agent_id=agent_id,
984 metadata={
985 "auto_registered": True,
986 "jwt_claim_field": virtual_key_claim_field,
987 "jwt_claim_value": claim_value,
988 },
989 )
990 # generate_key_helper_fn returns the plaintext key in "token"; the persisted
991 # row in LiteLLM_VerificationToken uses its hash, so hash here to get the FK
992 # value referenced by LiteLLM_JWTKeyMapping.token.
993 token_hash = hash_token(key_data["token"])
995 try:
996 await prisma_client.db.litellm_jwtkeymapping.create(
997 data={
998 "jwt_issuer": jwt_issuer or "",
999 "jwt_claim_name": virtual_key_claim_field,
1000 "jwt_claim_value": claim_value,
1001 "token": token_hash,
1002 "created_by": "auto_register",
1003 "updated_by": "auto_register",
1004 }
1005 )
1006 except Exception as e:
1007 error_str: Final = str(e).lower()
1008 if "unique" in error_str or "p2002" in error_str:
1009 # A concurrent request won the race. The key generate_key_helper_fn
1010 # just persisted to LiteLLM_VerificationToken is orphaned — nothing
1011 # maps to it, but it's a fully valid unrestricted API key sitting in
1012 # the DB and the cleartext is in memory on this request. Delete it
1013 # so orphans don't accumulate under sustained concurrency.
1014 verbose_proxy_logger.debug(
1015 "JWT Key Mapping (auto_register): unique conflict on create — "
1016 "deleting orphaned virtual key and fetching winner's mapping for %s='%s'.",
1017 virtual_key_claim_field,
1018 claim_value,
1019 )
1020 try:
1021 await prisma_client.db.litellm_verificationtoken.delete(where={"token": token_hash})
1022 except Exception as delete_err:
1023 # Don't fail the request if cleanup fails — the orphan is
1024 # unmapped and inert. Log so an operator can prune it later.
1025 verbose_proxy_logger.warning(
1026 "JWT Key Mapping (auto_register): failed to delete orphaned key after race: %s",
1027 delete_err,
1028 )
1029 token_hash = await get_jwt_key_mapping_object(
1030 jwt_claim_name=virtual_key_claim_field,
1031 jwt_claim_value=claim_value,
1032 prisma_client=prisma_client,
1033 jwt_issuer=jwt_issuer,
1034 )
1035 if token_hash is None:
1036 # The winner's mapping vanished between the unique-constraint
1037 # conflict and our re-fetch (concurrent delete). Returning None
1038 # here would silently fall through to team-based JWT auth —
1039 # a less-restrictive path than the operator configured. Raise
1040 # 503 so the caller retries against a stable state instead.
1041 raise HTTPException(
1042 status_code=503,
1043 detail=(
1044 "JWT Key Mapping: AUTO_REGISTER race resolution failed — "
1045 "winner's mapping was concurrently removed. Retry the request."
1046 ),
1047 )
1048 else:
1049 raise
1051 await user_api_key_cache.async_set_cache(
1052 key=cache_key,
1053 value=token_hash,
1054 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl,
1055 )
1057 verbose_proxy_logger.info(
1058 "JWT Key Mapping (auto_register): created new virtual key for %s='%s'.",
1059 virtual_key_claim_field,
1060 claim_value,
1061 )
1063 auto_registered_key: Final = IdentityStore.key_from_principal(
1064 await IdentityStore(
1065 prisma_client,
1066 user_api_key_cache,
1067 parent_otel_span=parent_otel_span,
1068 proxy_logging_obj=proxy_logging_obj,
1069 ).resolve(hashed_token=token_hash)
1070 )
1071 if auto_registered_key is not None:
1072 auto_registered_key.org_id = org_id
1073 auto_registered_key.end_user_id = end_user_id
1074 auto_registered_key.api_key = auto_registered_key.token
1075 return auto_registered_key
1078async def _lookup_jwt_mapping_token_hash(
1079 prisma_client: PrismaClient,
1080 user_api_key_cache: UserApiKeyCache,
1081 virtual_key_claim_field: str,
1082 claim_value: str,
1083 normalized_issuer: str | None,
1084 cache_key: str,
1085 ttl: float,
1086) -> str | None:
1087 issuer_scoped: Final = await get_jwt_key_mapping_object(
1088 jwt_claim_name=virtual_key_claim_field,
1089 jwt_claim_value=claim_value,
1090 prisma_client=prisma_client,
1091 jwt_issuer=normalized_issuer,
1092 )
1093 if issuer_scoped is not None:
1094 await user_api_key_cache.async_set_cache(key=cache_key, value=issuer_scoped, ttl=ttl)
1095 return issuer_scoped
1096 if normalized_issuer is None:
1097 return None
1098 # Another issuer may have already resolved (and cached) this same
1099 # global mapping -- check its cache entry before re-querying the DB.
1100 global_cache_key: Final = jwt_key_mapping_cache_key(virtual_key_claim_field, claim_value)
1101 cached_global: Final = await _raw_cache(user_api_key_cache).async_get_cache(key=global_cache_key)
1102 if isinstance(cached_global, str) and cached_global != "__NO_MAPPING__":
1103 return cached_global
1104 global_row: Final = await get_jwt_key_mapping_object(
1105 jwt_claim_name=virtual_key_claim_field,
1106 jwt_claim_value=claim_value,
1107 prisma_client=prisma_client,
1108 jwt_issuer=None,
1109 )
1110 if global_row is not None:
1111 await user_api_key_cache.async_set_cache(key=global_cache_key, value=global_row, ttl=ttl)
1112 return global_row
1115async def _resolve_jwt_to_virtual_key(
1116 jwt_claims: dict,
1117 jwt_handler: JWTHandler,
1118 prisma_client: PrismaClient | None,
1119 user_api_key_cache: UserApiKeyCache,
1120 parent_otel_span: Span | None,
1121 proxy_logging_obj: ProxyLogging,
1122) -> Union[UserAPIKeyAuth | None, "_PendingAutoRegister"]:
1123 """
1124 Returns:
1125 - ``UserAPIKeyAuth``: a resolved virtual key (cache hit or DB hit). The
1126 caller may use this directly; JWT policy has been enforced previously
1127 (at key-creation time or, for cached results, before caching).
1128 - ``_PendingAutoRegister``: claim is unmapped and behavior is AUTO_REGISTER.
1129 The caller MUST run ``JWTAuthManager.auth_builder`` to enforce JWT
1130 policy (RBAC, scope, custom_validate, email-domain), then invoke
1131 ``_auto_register_jwt_mapping`` with the validated team_id/user_id.
1132 - ``None``: claim is unmapped and behavior is FALLBACK_TEAM_MAPPING.
1133 The caller falls through to standard team-based JWT auth (which itself
1134 enforces full JWT policy via auth_builder).
1135 - Raises HTTPException: REJECT policy hit, missing claim under
1136 REJECT/AUTO_REGISTER, or other policy violations.
1137 """
1138 raw_issuer: Final = jwt_claims.get(JWTHandler.LITELLM_JWT_ISSUER_CLAIM)
1139 normalized_issuer: Final = raw_issuer if isinstance(raw_issuer, str) else None
1140 virtual_key_claim_field: Final = jwt_handler.litellm_jwtauth.get_virtual_key_claim_field(normalized_issuer)
1141 if virtual_key_claim_field is None:
1142 return None
1143 behavior: Final = jwt_handler.litellm_jwtauth.get_unregistered_jwt_client_behavior(normalized_issuer)
1145 claim_value: Final = get_nested_value(
1146 data=jwt_claims,
1147 key_path=virtual_key_claim_field,
1148 default=None,
1149 )
1151 if claim_value is None:
1152 verbose_proxy_logger.debug(
1153 "JWT Key Mapping: Claim field '%s' not found in JWT claims.", virtual_key_claim_field
1154 )
1155 # A missing claim is an unmapped client — apply the no-match policy
1156 # rather than returning early. Otherwise a caller can bypass REJECT
1157 # simply by presenting a JWT that omits the configured field. For
1158 # AUTO_REGISTER there is no stable identity to map without a claim
1159 # value, so we deny rather than create a sentinel-keyed record.
1160 if behavior in (
1161 UnregisteredJWTClientBehavior.REJECT,
1162 UnregisteredJWTClientBehavior.AUTO_REGISTER,
1163 ):
1164 raise HTTPException(
1165 status_code=403,
1166 detail=(
1167 f"JWT Key Mapping: Required claim '{virtual_key_claim_field}' "
1168 "is missing from the JWT. Access denied."
1169 ),
1170 )
1171 return None
1173 cache_key: Final = jwt_key_mapping_cache_key(virtual_key_claim_field, str(claim_value), normalized_issuer)
1174 raw_cached_mapping: Final = await user_api_key_cache.async_get_cache(cache_key)
1175 sentinel_written_by_this_policy: Final = behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER
1176 cached_mapping: Final = (
1177 None
1178 if raw_cached_mapping == _JWT_PROXY_ADMIN_SENTINEL and not sentinel_written_by_this_policy
1179 else raw_cached_mapping
1180 )
1182 if cached_mapping == _JWT_PROXY_ADMIN_SENTINEL:
1183 # Previously resolved to a proxy admin via auth_builder; skip the
1184 # mapping lookup and let the caller re-run auth_builder. Avoids a
1185 # repeated DB hit on every proxy-admin request under AUTO_REGISTER.
1186 return None
1188 if cached_mapping == "__NO_MAPPING__":
1189 if behavior == UnregisteredJWTClientBehavior.REJECT:
1190 raise HTTPException(
1191 status_code=403,
1192 detail=f"JWT Key Mapping: No registered mapping for {virtual_key_claim_field}='{claim_value}'. Access denied.",
1193 )
1194 if behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER:
1195 # Stale sentinel written under a prior fallback_team_mapping config —
1196 # evict it and defer auto-register to after auth_builder runs. Raise
1197 # the same 500 as the fresh-path AUTO_REGISTER branch when there is
1198 # no DB, so behavior is consistent regardless of whether the cache
1199 # happens to hold the sentinel.
1200 if prisma_client is None:
1201 raise HTTPException(
1202 status_code=500,
1203 detail=(
1204 "JWT Key Mapping: AUTO_REGISTER requires a database connection. "
1205 "Configure a database or change unregistered_jwt_client_behavior."
1206 ),
1207 )
1208 await user_api_key_cache.async_delete_cache(cache_key)
1209 return _PendingAutoRegister(
1210 claim_field=virtual_key_claim_field,
1211 claim_value=str(claim_value),
1212 cache_key=cache_key,
1213 jwt_issuer=normalized_issuer,
1214 )
1215 return None
1216 elif cached_mapping is not None:
1217 return IdentityStore.key_from_principal(
1218 await IdentityStore(
1219 prisma_client,
1220 user_api_key_cache,
1221 parent_otel_span=parent_otel_span,
1222 proxy_logging_obj=proxy_logging_obj,
1223 ).resolve(hashed_token=cached_mapping)
1224 )
1226 # Resolve the mapping from DB, or treat prisma_client=None as a definitive
1227 # miss (no DB → no mapping can exist → apply no-match policy below). An
1228 # issuer-scoped row wins; falling back to the global (no-issuer) row keeps
1229 # mappings created before issuer scoping existed working for every issuer.
1230 # Each tier is cached under ITS OWN key (the global tier under the
1231 # issuer-less cache key, not under `cache_key`/this issuer's key) so that
1232 # updating or deleting either row invalidates exactly the cache entries it
1233 # can affect. Caching a global-row hit under the requesting issuer's key
1234 # would leave every OTHER issuer that had fallen back to that same global
1235 # mapping serving its stale token until TTL after the row changes.
1236 token_hash: Final = (
1237 await _lookup_jwt_mapping_token_hash(
1238 prisma_client=prisma_client,
1239 user_api_key_cache=user_api_key_cache,
1240 virtual_key_claim_field=virtual_key_claim_field,
1241 claim_value=str(claim_value),
1242 normalized_issuer=normalized_issuer,
1243 cache_key=cache_key,
1244 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl,
1245 )
1246 if prisma_client is not None
1247 else None
1248 )
1250 if token_hash is not None:
1251 return IdentityStore.key_from_principal(
1252 await IdentityStore(
1253 prisma_client,
1254 user_api_key_cache,
1255 parent_otel_span=parent_otel_span,
1256 proxy_logging_obj=proxy_logging_obj,
1257 ).resolve(hashed_token=token_hash)
1258 )
1260 # No mapping found (DB miss or no DB) — apply no-match policy.
1261 if behavior == UnregisteredJWTClientBehavior.REJECT:
1262 # Cache the miss before raising so repeated rejections are served from
1263 # cache and don't re-query the DB on every request.
1264 await user_api_key_cache.async_set_cache(
1265 key=cache_key,
1266 value="__NO_MAPPING__",
1267 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl,
1268 )
1269 raise HTTPException(
1270 status_code=403,
1271 detail=f"JWT Key Mapping: No registered mapping for {virtual_key_claim_field}='{claim_value}'. Access denied.",
1272 )
1274 if behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER:
1275 if prisma_client is None:
1276 raise HTTPException(
1277 status_code=500,
1278 detail=(
1279 "JWT Key Mapping: AUTO_REGISTER requires a database connection. "
1280 "Configure a database or change unregistered_jwt_client_behavior."
1281 ),
1282 )
1283 # Defer: caller runs JWTAuthManager.auth_builder to enforce RBAC, scope,
1284 # custom_validate, and email-domain policy, then auto-registers using
1285 # the validated identity. Auto-registering here on a signature-only
1286 # JWT would bypass every JWT policy beyond signature verification.
1287 return _PendingAutoRegister(
1288 claim_field=virtual_key_claim_field,
1289 claim_value=str(claim_value),
1290 cache_key=cache_key,
1291 jwt_issuer=normalized_issuer,
1292 )
1294 # FALLBACK_TEAM_MAPPING (default): cache the miss and return None so the
1295 # caller falls through to standard team-based JWT auth.
1296 await user_api_key_cache.async_set_cache(
1297 key=cache_key,
1298 value="__NO_MAPPING__",
1299 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl,
1300 )
1301 return None
1304def _ensure_litellm_received_at_on_request_state(request: Request) -> datetime:
1305 """Idempotently stamp ``request.state.litellm_received_at`` with the moment
1306 litellm's own code started handling this request -- the first line of
1307 ``user_api_key_auth``, before any auth/pre-call work runs. This is the
1308 basis for the request-latency Prometheus metrics (see
1309 ``litellm/integrations/prometheus.py``), and unlike the OTEL SERVER span
1310 below, it is set unconditionally so those metrics don't depend on OTEL
1311 being configured.
1312 """
1313 existing_received_at: Final[datetime | None] = getattr(request.state, "litellm_received_at", None)
1314 if existing_received_at is not None:
1315 return existing_received_at
1316 received_at: Final = datetime.now(timezone.utc)
1317 try:
1318 request.state.litellm_received_at = received_at
1319 except Exception:
1320 pass
1321 return received_at
1324def _ensure_parent_otel_span_on_request_state(request: Request) -> None:
1325 """Idempotently create the OTEL SERVER span and stash it on
1326 ``request.state.parent_otel_span``. Safe to call multiple times.
1328 Called both at the top of ``user_api_key_auth`` (so body-parse failures
1329 have a span to close) and inside ``_user_api_key_auth_builder`` (for
1330 callers that bypass ``user_api_key_auth``, e.g. MCP).
1331 """
1332 from litellm.proxy.proxy_server import open_telemetry_logger
1334 start_time: Final = _ensure_litellm_received_at_on_request_state(request)
1336 if open_telemetry_logger is None: 1336 ↛ 1338line 1336 didn't jump to line 1338 because the condition on line 1336 was always true
1337 return
1338 if getattr(request.state, "parent_otel_span", None) is not None:
1339 return
1340 parent_otel_span: Final = open_telemetry_logger.create_litellm_proxy_request_started_span(
1341 start_time=start_time,
1342 headers=_safe_get_request_headers(request),
1343 )
1344 # Under V2 the FastAPI instrumentor stamps http.route / url.path on the server
1345 # span; only the legacy logger needs these set explicitly.
1346 set_route_attrs: Final = getattr(open_telemetry_logger, "set_proxy_request_route_attributes", None)
1347 if not is_otel_v2_enabled() and set_route_attrs is not None:
1348 set_route_attrs(
1349 parent_otel_span,
1350 url_path=get_request_route(request=request),
1351 http_route=get_request_route_template(request),
1352 )
1353 request.state.parent_otel_span = parent_otel_span
1356async def _read_request_body_deferring_parse_failure(
1357 request: Request,
1358) -> tuple[dict, ProxyException | None]:
1359 """Parse the body, returning a parse failure instead of raising it.
1361 A body that fails to parse is still a request from a known caller, so auth
1362 must run (resolving identity onto the request's trace) before the 400 goes
1363 out; the caller re-raises the returned exception once identity is seeded.
1364 """
1365 if is_opaque_audio_pass_through_request(
1366 route=get_request_route(request=request),
1367 content_type=_safe_get_request_headers(request=request).get("content-type", ""),
1368 ):
1369 _safe_set_request_parsed_body(request=request, parsed_body={}) # mutable-ok: the body cache stores a plain dict
1370 return {}, None # mutable-ok: request_data is a plain dict across the whole auth path
1371 try:
1372 parsed_body: Final = await _read_request_body(request=request)
1373 except ProxyException as parse_exception:
1374 return {}, parse_exception # mutable-ok: request_data is a plain dict across the whole auth path
1375 return populate_request_with_path_params(request_data=parsed_body, request=request), None
1378async def _record_unparsable_body_failure(
1379 user_api_key_dict: UserAPIKeyAuth,
1380 body_parse_exception: ProxyException,
1381 route: str,
1382) -> None:
1383 """Record the 400 an unparsable body earns as a failed request log.
1385 The endpoint never runs for these, so no downstream failure hook writes the
1386 spend log row the Admin UI reads. Logging must not change what the caller
1387 sees, so a failure here is swallowed and the 400 is raised either way.
1388 """
1389 from litellm.proxy.proxy_server import proxy_logging_obj
1391 try:
1392 await proxy_logging_obj.post_call_failure_hook( # pyright: ignore[reportUnknownMemberType] # bare dict in sig
1393 request_data={}, # mutable-ok: the failure hook seeds the call id and metadata onto this dict
1394 original_exception=body_parse_exception,
1395 user_api_key_dict=user_api_key_dict,
1396 error_type=ProxyErrorTypes.bad_request_error,
1397 route=route,
1398 )
1399 except Exception as e: # noqa: BLE001 # any logging failure must leave the caller's 400 untouched
1400 verbose_proxy_logger.exception("Failed to log the request rejected for an unparsable body: %s", e)
1403async def _refresh_session_token_grants(
1404 valid_token: UserAPIKeyAuth,
1405 prisma_client: PrismaClient,
1406 user_api_key_cache: UserApiKeyCache,
1407 parent_otel_span: Span | None,
1408 proxy_logging_obj: ProxyLogging,
1409) -> UserAPIKeyAuth:
1410 """Rebuild a ``lite login`` session token's grants from the live user and team rows.
1412 The blob only proves who logged in and which team they picked. Team models, aliases, the user's own model
1413 list, and their role are re-read every request, so a `/team/update` or a demotion shows up without a
1414 re-login, and a user removed from the team or deleted outright is refused. When a row cannot be read for
1415 a reason unrelated to the caller, the minted grants stand in exactly as they did before this refresh.
1416 """
1417 outcome: Final = await GrantResolver(
1418 prisma_client,
1419 user_api_key_cache,
1420 parent_otel_span=parent_otel_span,
1421 proxy_logging_obj=proxy_logging_obj,
1422 load_user=get_user_object,
1423 load_team=get_team_object,
1424 load_membership=get_team_membership,
1425 ).resolve(UserLookup(user_id=valid_token.user_id), team_id=valid_token.team_id)
1426 match outcome:
1427 case ResolvedGrants(
1428 user_object=LiteLLM_UserTable() as user_object, team_object=team_object, team_membership=team_membership
1429 ):
1430 return UserAPIKeyAuth.model_validate(
1431 MappingProxyType(
1432 {
1433 **valid_token.model_dump(exclude_none=True),
1434 **team_grants(team_object, team_membership, user_object.user_id),
1435 "user_role": _get_user_role(user_object),
1436 "models": () if team_object is not None else user_models(user_object),
1437 }
1438 )
1439 )
1440 case ResolvedGrants():
1441 return valid_token
1442 case LookupDegraded(error=error):
1443 verbose_proxy_logger.debug("Session token grants not refreshed, keeping minted grants: %s", error)
1444 return valid_token
1445 case _:
1446 raise_public(outcome)
1449async def _resolve_object_permission_for_unresolvable_team(
1450 object_permission_id: str | None,
1451 prisma_client: PrismaClient | None,
1452 user_api_key_cache: UserApiKeyCache,
1453 parent_otel_span: Span | None,
1454 proxy_logging_obj: ProxyLogging,
1455) -> LiteLLM_ObjectPermissionTable | None:
1456 """Re-resolve a team's object permission by id when the team row itself is unreadable, so the
1457 token-derived fallback doesn't silently drop it."""
1458 if object_permission_id is None or prisma_client is None:
1459 return None
1460 return await get_object_permission(
1461 object_permission_id=object_permission_id,
1462 prisma_client=prisma_client,
1463 user_api_key_cache=user_api_key_cache,
1464 parent_otel_span=parent_otel_span,
1465 proxy_logging_obj=proxy_logging_obj,
1466 )
1469async def _user_api_key_auth_builder(
1470 request: Request,
1471 api_key: str,
1472 azure_api_key_header: str,
1473 anthropic_api_key_header: str | None,
1474 google_ai_studio_api_key_header: str | None,
1475 azure_apim_header: str | None,
1476 request_data: dict,
1477 custom_litellm_key_header: str | None = None,
1478) -> UserAPIKeyAuth:
1479 from litellm.proxy.proxy_server import (
1480 general_settings,
1481 jwt_handler,
1482 litellm_proxy_admin_name,
1483 llm_model_list,
1484 llm_router,
1485 master_key,
1486 model_max_budget_limiter,
1487 open_telemetry_logger,
1488 prisma_client,
1489 proxy_logging_obj,
1490 user_api_key_cache,
1491 user_custom_auth,
1492 )
1494 parent_otel_span: Span | None = None
1495 # Prefer the receive-instant stamped by the early helper in
1496 # user_api_key_auth (before body parse) — overwriting it would shorten
1497 # the preprocessing-duration measurement by the body-parse window.
1498 start_time: Final = getattr(request.state, "litellm_received_at", None) or datetime.now(timezone.utc)
1499 try:
1500 request.state.litellm_received_at = start_time
1501 except Exception:
1502 pass
1503 route: Final[str] = get_request_route(request=request)
1504 valid_token: UserAPIKeyAuth | None = None
1505 custom_auth_api_key: bool = False
1507 try:
1508 with tracer.trace("litellm.proxy.auth.pre_db_read_auth_checks"):
1509 await pre_db_read_auth_checks(
1510 request_data=request_data,
1511 request=request,
1512 route=route,
1513 )
1514 pass_through_endpoints: Final[list[dict] | None] = general_settings.get("pass_through_endpoints", None)
1515 ## CHECK IF X-LITELM-API-KEY IS PASSED IN - supercedes Authorization header
1516 api_key, passed_in_key = get_api_key(
1517 custom_litellm_key_header=custom_litellm_key_header,
1518 api_key=api_key,
1519 azure_api_key_header=azure_api_key_header,
1520 anthropic_api_key_header=anthropic_api_key_header,
1521 google_ai_studio_api_key_header=google_ai_studio_api_key_header,
1522 azure_apim_header=azure_apim_header,
1523 pass_through_endpoints=pass_through_endpoints,
1524 route=route,
1525 request=request,
1526 )
1527 # if user wants to pass LiteLLM_Master_Key as a custom header, example pass litellm keys as X-LiteLLM-Key: Bearer sk-1234
1528 custom_litellm_key_header_name: Final = general_settings.get("litellm_key_header_name")
1529 if custom_litellm_key_header_name is not None: 1529 ↛ 1530line 1529 didn't jump to line 1530 because the condition on line 1529 was never true
1530 api_key = get_api_key_from_custom_header(
1531 request=request,
1532 custom_litellm_key_header_name=custom_litellm_key_header_name,
1533 )
1535 if open_telemetry_logger is not None: 1535 ↛ 1539line 1535 didn't jump to line 1539 because the condition on line 1535 was never true
1536 # Reuse the span created by user_api_key_auth (before body parse)
1537 # so it survives _read_request_body failures. For callers that
1538 # bypass user_api_key_auth (e.g. MCP), create it lazily.
1539 _ensure_parent_otel_span_on_request_state(request)
1540 parent_otel_span = getattr(request.state, "parent_otel_span", None)
1542 ### USER-DEFINED AUTH FUNCTION ###
1543 if enterprise_custom_auth is not None: 1543 ↛ 1562line 1543 didn't jump to line 1562 because the condition on line 1543 was always true
1544 with tracer.trace("litellm.proxy.auth.enterprise_custom_auth"):
1545 response = await enterprise_custom_auth(
1546 request=request, api_key=api_key, user_custom_auth=user_custom_auth
1547 )
1548 if response is not None and isinstance(response, UserAPIKeyAuth): 1548 ↛ 1549line 1548 didn't jump to line 1549 because the condition on line 1548 was never true
1549 validated = UserAPIKeyAuth.model_validate(response)
1550 if getattr(litellm, "enable_post_custom_auth_checks", False):
1551 validated = await _run_post_custom_auth_checks(
1552 valid_token=validated,
1553 request=request,
1554 request_data=request_data,
1555 route=route,
1556 parent_otel_span=parent_otel_span,
1557 )
1558 return validated
1559 elif response is not None and isinstance(response, str): 1559 ↛ 1560line 1559 didn't jump to line 1560 because the condition on line 1559 was never true
1560 api_key = response
1561 custom_auth_api_key = True
1562 elif user_custom_auth is not None:
1563 response = await user_custom_auth(request=request, api_key=api_key)
1564 validated = UserAPIKeyAuth.model_validate(response)
1565 if getattr(litellm, "enable_post_custom_auth_checks", False):
1566 validated = await _run_post_custom_auth_checks(
1567 valid_token=validated,
1568 request=request,
1569 request_data=request_data,
1570 route=route,
1571 parent_otel_span=parent_otel_span,
1572 )
1573 return validated
1575 ### LITELLM-DEFINED AUTH FUNCTION ###
1576 #### IF JWT ####
1577 """
1578 LiteLLM supports using JWTs.
1580 Enable this in proxy config, by setting
1581 ```
1582 general_settings:
1583 enable_jwt_auth: true
1584 ```
1585 """
1587 ######## Route Checks Before Reading DB / Cache for "token" ################
1588 if not _route_requires_auth_despite_public(route=route, general_settings=general_settings) and (
1589 route in LiteLLMRoutes.public_routes.value or route_in_additonal_public_routes(current_route=route)
1590 ):
1591 # check if public endpoint
1592 return UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY)
1594 ########## End of Route Checks Before Reading DB / Cache for "token" ########
1596 enable_oauth2_auth: Final = general_settings.get("enable_oauth2_auth", False) is True
1597 enable_jwt_auth: Final = general_settings.get("enable_jwt_auth", False) is True
1598 is_jwt = jwt_handler.is_jwt(token=api_key) if enable_jwt_auth else False
1600 # Routing uses unverified JWT claims only to choose auth path.
1601 # Final authentication is enforced by the selected validator.
1602 route_jwt_to_oauth2 = is_jwt and _should_route_jwt_to_oauth2_override(token=api_key, jwt_handler=jwt_handler)
1604 # OAuth2 applies for:
1605 # 1) when global OAuth2 auth is enabled on LLM + info routes
1606 # 2) JWT tokens that explicitly match routing_overrides on LLM + info routes
1607 should_apply_override_oauth2: Final = route_jwt_to_oauth2 and (
1608 RouteChecks.is_llm_api_route(route=route) or RouteChecks.is_info_route(route=route)
1609 )
1610 should_apply_global_oauth2: Final = enable_oauth2_auth and (
1611 RouteChecks.is_llm_api_route(route=route) or RouteChecks.is_info_route(route=route)
1612 )
1613 if (should_apply_global_oauth2 and not is_jwt) or should_apply_override_oauth2: 1613 ↛ 1614line 1613 didn't jump to line 1614 because the condition on line 1613 was never true
1614 from litellm.proxy.proxy_server import premium_user
1616 if premium_user is not True:
1617 raise ProxyException(
1618 message="Oauth2 token validation is only available for premium users. "
1619 + CommonProxyErrors.not_premium_user.value,
1620 type=ProxyErrorTypes.auth_error,
1621 param="premium_user",
1622 code=status.HTTP_403_FORBIDDEN,
1623 )
1625 return await Oauth2Handler.check_oauth2_token(token=api_key)
1627 if general_settings.get("enable_oauth2_proxy_auth", False) is True: 1627 ↛ 1628line 1627 didn't jump to line 1628 because the condition on line 1627 was never true
1628 return await handle_oauth2_proxy_request(request=request)
1630 if general_settings.get("enable_jwt_auth", False) is True: 1630 ↛ 1631line 1630 didn't jump to line 1631 because the condition on line 1630 was never true
1631 is_jwt = jwt_handler.is_jwt(token=api_key)
1632 verbose_proxy_logger.debug("is_jwt: %s", is_jwt)
1633 if is_jwt:
1634 from litellm.proxy.proxy_server import premium_user
1636 if premium_user is not True:
1637 raise ProxyException(
1638 message=f"JWT Auth is an enterprise only feature. {CommonProxyErrors.not_premium_user.value}",
1639 type=ProxyErrorTypes.auth_error,
1640 param="premium_user",
1641 code=status.HTTP_403_FORBIDDEN,
1642 )
1643 # Try JWT-to-Virtual-Key mapping first to avoid
1644 # unnecessary DB queries in auth_builder
1645 do_standard_jwt_auth = True
1646 pending_auto_register: _PendingAutoRegister | None = None
1647 if jwt_handler.litellm_jwtauth.is_virtual_key_mapping_configured():
1648 # Decode JWT to get claims without running full auth_builder
1649 jwt_claims: dict | None
1650 if jwt_handler.litellm_jwtauth.oidc_userinfo_enabled and not is_jwt:
1651 jwt_claims = await jwt_handler.get_oidc_userinfo(token=api_key)
1652 else:
1653 jwt_claims = await jwt_handler.auth_jwt(token=api_key)
1655 resolve_result: Final = await _resolve_jwt_to_virtual_key(
1656 jwt_claims=jwt_claims,
1657 jwt_handler=jwt_handler,
1658 prisma_client=prisma_client,
1659 user_api_key_cache=user_api_key_cache,
1660 parent_otel_span=parent_otel_span,
1661 proxy_logging_obj=proxy_logging_obj,
1662 )
1663 if isinstance(resolve_result, UserAPIKeyAuth):
1664 valid_token = resolve_result
1665 api_key = valid_token.token or ""
1666 valid_token.jwt_claims = jwt_claims
1667 do_standard_jwt_auth = False
1668 # Fall through to virtual key checks
1669 if valid_token.user_id is not None and valid_token.user_email is None:
1670 mapped_claims = jwt_claims or {} # mutable-ok: empty-dict fallback for the None-claims case
1671 mapped_user_email = jwt_handler.get_user_email(token=mapped_claims, default_value=None)
1672 mapped_jwt_user_id: Final = jwt_handler.get_user_id(token=mapped_claims, default_value=None)
1673 if mapped_user_email is not None and mapped_jwt_user_id == valid_token.user_id:
1674 try:
1675 mapped_user_obj: Final = await get_user_object(
1676 user_id=valid_token.user_id,
1677 prisma_client=prisma_client,
1678 user_api_key_cache=user_api_key_cache,
1679 user_id_upsert=False,
1680 parent_otel_span=parent_otel_span,
1681 proxy_logging_obj=proxy_logging_obj,
1682 user_email=mapped_user_email,
1683 )
1684 except Exception as e:
1685 verbose_proxy_logger.debug("JWT mapped-key user_email backfill skipped: %s", e)
1686 else:
1687 if mapped_user_obj is not None:
1688 valid_token.user_email = mapped_user_obj.user_email
1689 elif isinstance(resolve_result, _PendingAutoRegister):
1690 # Run full JWT policy (RBAC, scope, custom_validate,
1691 # email-domain) via auth_builder, then create the key
1692 # from the validated identity below.
1693 pending_auto_register = resolve_result
1694 # else: None → FALLBACK_TEAM_MAPPING, falls through to
1695 # standard JWT auth_builder below
1697 if do_standard_jwt_auth:
1698 with tracer.trace("litellm.proxy.auth.jwt_auth_builder"):
1699 result: Final = await JWTAuthManager.auth_builder(
1700 request_data=request_data,
1701 general_settings=general_settings,
1702 api_key=api_key,
1703 jwt_handler=jwt_handler,
1704 route=route,
1705 prisma_client=prisma_client,
1706 user_api_key_cache=user_api_key_cache,
1707 proxy_logging_obj=proxy_logging_obj,
1708 parent_otel_span=parent_otel_span,
1709 request_headers=_safe_get_request_headers(request),
1710 request_method=RouteChecks._get_request_method(request=request),
1711 )
1713 is_proxy_admin: Final = result["is_proxy_admin"]
1714 team_id: Final = result["team_id"]
1715 user_id: Final = result["user_id"]
1716 user_email: Final = result["user_email"]
1717 user_object: Final = result["user_object"]
1718 end_user_id = result["end_user_id"]
1719 org_id: Final = result["org_id"]
1720 jwt_claims = result.get("jwt_claims", None)
1721 agent_id: Final[str | None] = result.get("agent_id")
1723 if (
1724 user_object is not None
1725 and isinstance(user_object.metadata, dict)
1726 and user_object.metadata.get("scim_active") is False
1727 ):
1728 raise HTTPException(
1729 status_code=status.HTTP_401_UNAUTHORIZED,
1730 detail=f"User={user_id} has been deactivated via SCIM. Keys owned by this user cannot be used.",
1731 )
1733 if is_proxy_admin:
1734 # Proxy admins authenticate via auth_builder (full
1735 # access), not via a mapped virtual key. If
1736 # AUTO_REGISTER was pending, cache a sentinel so
1737 # future requests from this JWT identity skip the
1738 # DB mapping lookup in _resolve_jwt_to_virtual_key.
1739 # Without this, every proxy-admin request under
1740 # AUTO_REGISTER re-hits get_jwt_key_mapping_object.
1741 if pending_auto_register is not None:
1742 await user_api_key_cache.async_set_cache(
1743 key=pending_auto_register.cache_key,
1744 value=_JWT_PROXY_ADMIN_SENTINEL,
1745 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl,
1746 )
1747 return JWTAuthManager.user_api_key_auth_from_result(result, parent_otel_span)
1749 valid_token = JWTAuthManager.user_api_key_auth_from_result(result, parent_otel_span)
1751 # AUTO_REGISTER deferred from _resolve_jwt_to_virtual_key.
1752 # JWT policy (RBAC, scope, custom_validate, email-domain)
1753 # has now been enforced by auth_builder above. Create the
1754 # mapping + virtual key from the *validated* identity, then
1755 # replace valid_token with the new key so downstream checks
1756 # use the key-scoped path.
1757 if pending_auto_register is not None and prisma_client is not None:
1758 auto_registered: Final = await _auto_register_jwt_mapping(
1759 virtual_key_claim_field=pending_auto_register.claim_field,
1760 claim_value=pending_auto_register.claim_value,
1761 jwt_handler=jwt_handler,
1762 prisma_client=prisma_client,
1763 user_api_key_cache=user_api_key_cache,
1764 parent_otel_span=parent_otel_span,
1765 proxy_logging_obj=proxy_logging_obj,
1766 cache_key=pending_auto_register.cache_key,
1767 jwt_issuer=pending_auto_register.jwt_issuer,
1768 team_id=team_id,
1769 user_id=user_id,
1770 org_id=org_id,
1771 end_user_id=end_user_id,
1772 agent_id=agent_id,
1773 )
1774 if auto_registered is not None:
1775 auto_registered.jwt_claims = jwt_claims
1776 auto_registered.user_email = user_email
1777 # The auto-registered token is built from the new key's
1778 # columns, which carry no user budget. Carry over the
1779 # already-loaded user row rather than re-reading it, or
1780 # the budget check below has nothing to enforce.
1781 auto_registered.user_model_max_budget = (
1782 user_object.model_max_budget if user_object is not None else None
1783 )
1784 valid_token = auto_registered
1785 api_key = valid_token.token or ""
1787 # Check if model has zero cost - if so, skip all budget checks
1788 model = _get_model_from_request_context(
1789 request_data=request_data,
1790 route=route,
1791 request=request,
1792 llm_router=llm_router,
1793 team_id=valid_token.team_id,
1794 )
1795 skip_budget_checks = False
1796 if model is not None and llm_router is not None:
1797 from litellm.proxy.auth.auth_checks import _is_model_cost_zero
1799 skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router)
1800 if skip_budget_checks:
1801 verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model)
1803 # Fetch project object for JWT path if project_id is set
1804 _jwt_project_obj = None
1805 if valid_token.project_id is not None:
1806 _jwt_project_obj = await get_project_object(
1807 project_id=valid_token.project_id,
1808 prisma_client=prisma_client,
1809 user_api_key_cache=user_api_key_cache,
1810 proxy_logging_obj=proxy_logging_obj,
1811 )
1812 if _jwt_project_obj is not None:
1813 valid_token.project_metadata = _jwt_project_obj.metadata
1814 valid_token.project_alias = _jwt_project_obj.project_alias
1816 # JWT auth returns here rather than falling through to the
1817 # virtual-key checks below, so the user's per-model budget
1818 # has to be enforced on this path too. Without it the
1819 # post-call increment still charges the counter and nothing
1820 # ever reads it, which is worse than not tracking at all.
1821 # Guarded by the same flag the virtual-key path uses, or a
1822 # zero-cost model would be refused here and allowed there,
1823 # while the log above claims all budget checks were skipped.
1824 if not skip_budget_checks:
1825 await _check_user_model_budget(
1826 valid_token=cast(UserAPIKeyAuth, valid_token),
1827 model_max_budget_limiter=model_max_budget_limiter,
1828 models=_get_model_names_for_budget_checks(
1829 model=_get_model_from_request_context(
1830 request_data=request_data,
1831 route=route,
1832 request=request,
1833 llm_router=llm_router,
1834 team_id=valid_token.team_id,
1835 )
1836 ),
1837 )
1839 return cast(UserAPIKeyAuth, valid_token)
1841 #### ELSE ####
1842 ## CHECK PASS-THROUGH ENDPOINTS ##
1843 if not custom_auth_api_key: 1843 ↛ 1854line 1843 didn't jump to line 1854 because the condition on line 1843 was always true
1844 response = await check_api_key_for_custom_headers_or_pass_through_endpoints(
1845 request=request,
1846 route=route,
1847 pass_through_endpoints=pass_through_endpoints,
1848 api_key=api_key,
1849 )
1850 if isinstance(response, str):
1851 api_key = response
1852 elif isinstance(response, UserAPIKeyAuth): 1852 ↛ 1853line 1852 didn't jump to line 1853 because the condition on line 1852 was never true
1853 return response
1854 if master_key is None: 1854 ↛ 1855line 1854 didn't jump to line 1855 because the condition on line 1854 was never true
1855 if isinstance(api_key, str):
1856 return UserAPIKeyAuth(
1857 api_key=api_key,
1858 user_role=LitellmUserRoles.INTERNAL_USER,
1859 parent_otel_span=parent_otel_span,
1860 )
1861 else:
1862 return UserAPIKeyAuth(
1863 user_role=LitellmUserRoles.INTERNAL_USER,
1864 parent_otel_span=parent_otel_span,
1865 )
1866 elif api_key is None: # only require api key if master key is set
1867 raise Exception("No api key passed in.")
1868 elif api_key == "":
1869 # missing 'Bearer ' prefix
1870 raise Exception("Malformed API Key passed in. Ensure Key has `Bearer ` prefix.")
1872 if route == "/user/auth": 1872 ↛ 1873line 1872 didn't jump to line 1873 because the condition on line 1872 was never true
1873 if general_settings.get("allow_user_auth", False) is True:
1874 return UserAPIKeyAuth()
1875 else:
1876 raise HTTPException(
1877 status_code=status.HTTP_403_FORBIDDEN,
1878 detail="'allow_user_auth' not set or set to False",
1879 )
1881 ## Check END-USER OBJECT
1882 _end_user_object = None
1883 end_user_params: Final = {}
1885 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request))
1886 end_user_id = await resolve_and_validate_end_user_id(
1887 raw_end_user_id=raw_end_user_id,
1888 prisma_client=prisma_client,
1889 user_api_key_cache=user_api_key_cache,
1890 parent_otel_span=parent_otel_span,
1891 proxy_logging_obj=proxy_logging_obj,
1892 route=route,
1893 )
1894 if end_user_id:
1895 try:
1896 end_user_params["end_user_id"] = end_user_id
1898 with tracer.trace("litellm.proxy.auth.get_end_user_object"):
1899 _end_user_object = await get_end_user_object(
1900 end_user_id=end_user_id,
1901 prisma_client=prisma_client,
1902 user_api_key_cache=user_api_key_cache,
1903 parent_otel_span=parent_otel_span,
1904 proxy_logging_obj=proxy_logging_obj,
1905 route=route,
1906 )
1907 if _end_user_object is not None: 1907 ↛ 1908line 1907 didn't jump to line 1908 because the condition on line 1907 was never true
1908 end_user_params["allowed_model_region"] = _end_user_object.allowed_model_region
1909 if _end_user_object.litellm_budget_table is not None:
1910 _apply_budget_limits_to_end_user_params(
1911 end_user_params=end_user_params,
1912 budget_info=_end_user_object.litellm_budget_table,
1913 end_user_id=end_user_id,
1914 )
1915 elif litellm.max_end_user_budget_id is not None: 1915 ↛ 1917line 1915 didn't jump to line 1917 because the condition on line 1915 was never true
1916 # End user doesn't exist yet, but apply default budget limits if configured
1917 from litellm.proxy.auth.auth_checks import (
1918 get_default_end_user_budget,
1919 )
1921 default_budget: Final = await get_default_end_user_budget(
1922 prisma_client=prisma_client,
1923 user_api_key_cache=user_api_key_cache,
1924 parent_otel_span=parent_otel_span,
1925 )
1926 if default_budget is not None:
1927 _apply_budget_limits_to_end_user_params(
1928 end_user_params=end_user_params,
1929 budget_info=default_budget,
1930 end_user_id=end_user_id,
1931 )
1932 except Exception as e:
1933 if isinstance(e, litellm.BudgetExceededError):
1934 raise e
1935 verbose_proxy_logger.debug("Unable to find user in db. Error - %s", e)
1937 ### CHECK IF ADMIN ###
1938 # note: never string compare api keys, this is vulenerable to a time attack. Use secrets.compare_digest instead
1939 ### CHECK IF ADMIN ###
1940 # note: never string compare api keys, this is vulenerable to a time attack. Use secrets.compare_digest instead
1941 if valid_token is None: 1941 ↛ 1977line 1941 didn't jump to line 1977 because the condition on line 1941 was always true
1942 ## Check CACHE
1943 try:
1944 with tracer.trace("litellm.proxy.auth.get_key_object_check_cache"):
1945 valid_token = IdentityStore.key_from_principal(
1946 await IdentityStore(
1947 prisma_client,
1948 user_api_key_cache,
1949 parent_otel_span=parent_otel_span,
1950 proxy_logging_obj=proxy_logging_obj,
1951 check_cache_only=True,
1952 ).resolve(hashed_token=hash_token(api_key))
1953 )
1954 # Key-cache entries are written only after the proxy validated a
1955 # virtual key or the master key, but via_virtual_key is exclude=True
1956 # so serialization drops it; restore it at this trusted boundary.
1957 # The UI-login JWT fallback below constructs its token from a
1958 # decrypted blob, not this cache, and stays unmarked.
1959 if isinstance(valid_token, UserAPIKeyAuth): 1959 ↛ 1970line 1959 didn't jump to line 1970 because the condition on line 1959 was always true
1960 valid_token.via_virtual_key = True
1961 except Exception:
1962 verbose_logger.debug("api key not found in cache.")
1963 valid_token = None
1965 ## Check UI/CLI Hash Key
1966 # Attempt decryption for non-sk- tokens unless the operator has
1967 # explicitly set EXPERIMENTAL_UI_LOGIN=false to disable it.
1968 # Unset (None) keeps the new default of always attempting decryption;
1969 # decryption fails closed for anything that is not a genuine blob.
1970 if (
1971 valid_token is None
1972 and not api_key.startswith("sk-")
1973 and get_secret_bool("EXPERIMENTAL_UI_LOGIN") is not False
1974 ):
1975 valid_token = ExperimentalUIJWTToken.get_key_object_from_ui_hash_key(api_key)
1977 if valid_token is not None and valid_token.is_session_token and prisma_client is not None: 1977 ↛ 1978line 1977 didn't jump to line 1978 because the condition on line 1977 was never true
1978 valid_token = await _refresh_session_token_grants( # rebind-ok: later checks read this name
1979 valid_token=valid_token,
1980 prisma_client=prisma_client,
1981 user_api_key_cache=user_api_key_cache,
1982 parent_otel_span=parent_otel_span,
1983 proxy_logging_obj=proxy_logging_obj,
1984 )
1986 if (
1987 valid_token is not None
1988 and isinstance(valid_token, UserAPIKeyAuth)
1989 and valid_token.user_role == LitellmUserRoles.PROXY_ADMIN
1990 ):
1991 if valid_token.expires is not None: 1991 ↛ 1992line 1991 didn't jump to line 1992 because the condition on line 1991 was never true
1992 current_time = datetime.now(timezone.utc)
1993 if isinstance(valid_token.expires, datetime):
1994 expiry_time = valid_token.expires
1995 else:
1996 expiry_time = datetime.fromisoformat(valid_token.expires)
1997 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None:
1998 expiry_time = expiry_time.replace(tzinfo=timezone.utc)
1999 if expiry_time < current_time:
2000 await _delete_cache_key_object(
2001 hashed_token=hash_token(api_key),
2002 user_api_key_cache=user_api_key_cache,
2003 proxy_logging_obj=proxy_logging_obj,
2004 )
2005 raise ProxyException(
2006 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}",
2007 type=ProxyErrorTypes.expired_key,
2008 code=status.HTTP_401_UNAUTHORIZED,
2009 param=abbreviate_api_key(api_key=api_key),
2010 )
2011 valid_token = update_valid_token_with_end_user_params(
2012 valid_token=valid_token, end_user_params=end_user_params
2013 )
2014 valid_token.parent_otel_span = parent_otel_span
2015 if _end_user_object is not None: 2015 ↛ 2016line 2015 didn't jump to line 2016 because the condition on line 2015 was never true
2016 valid_token.end_user_object_permission = _end_user_object.object_permission
2018 return valid_token
2020 if ( 2020 ↛ 2027line 2020 didn't jump to line 2027 because the condition on line 2020 was never true
2021 valid_token is not None
2022 and isinstance(valid_token, UserAPIKeyAuth)
2023 and valid_token.team_id is not None
2024 and valid_token.team_id != UI_TEAM_ID
2025 ):
2026 ## UPDATE TEAM VALUES BASED ON CACHED TEAM OBJECT - allows `/team/update` values to work for cached token
2027 try:
2028 team_obj: Final[LiteLLM_TeamTableCachedObj] = await get_team_object(
2029 team_id=valid_token.team_id,
2030 prisma_client=prisma_client,
2031 user_api_key_cache=user_api_key_cache,
2032 parent_otel_span=parent_otel_span,
2033 proxy_logging_obj=proxy_logging_obj,
2034 check_cache_only=True,
2035 )
2037 if (
2038 team_obj.last_refreshed_at is not None
2039 and valid_token.last_refreshed_at is not None
2040 and team_obj.last_refreshed_at > valid_token.last_refreshed_at
2041 ):
2042 team_obj_dict: Final = team_obj.__dict__
2044 for k, v in team_obj_dict.items():
2045 field_name = f"team_{k}"
2046 if field_name in valid_token.__fields__:
2047 setattr(valid_token, field_name, v)
2048 except Exception as e:
2049 verbose_logger.debug(e) # moving from .warning to .debug as it spams logs when team missing from cache.
2051 try:
2052 is_master_key_valid = secrets.compare_digest(api_key, master_key)
2053 except Exception:
2054 is_master_key_valid = False
2056 ## VALIDATE MASTER KEY ##
2057 if not isinstance(master_key, str): 2057 ↛ 2058line 2057 didn't jump to line 2058 because the condition on line 2057 was never true
2058 raise HTTPException(
2059 status_code=500,
2060 detail={f"Master key must be a valid string. Current type={type(master_key)}"},
2061 )
2063 if is_master_key_valid:
2064 # Substitute a stable alias for the raw master key so neither the
2065 # master key nor its hash propagates into spend logs, Prometheus
2066 # /metrics labels, audit trails, rate-limit buckets, or any other
2067 # downstream consumer of UserAPIKeyAuth.api_key.
2068 _user_api_key_obj = await _return_user_api_key_auth_obj(
2069 user_obj=None,
2070 user_role=LitellmUserRoles.PROXY_ADMIN,
2071 api_key=LITELLM_PROXY_MASTER_KEY_ALIAS,
2072 parent_otel_span=parent_otel_span,
2073 valid_token_dict={
2074 **end_user_params,
2075 "user_id": litellm_proxy_admin_name,
2076 },
2077 route=route,
2078 start_time=start_time,
2079 )
2080 asyncio.create_task(
2081 _cache_key_object(
2082 hashed_token=hash_token(master_key),
2083 user_api_key_obj=_user_api_key_obj,
2084 user_api_key_cache=user_api_key_cache,
2085 proxy_logging_obj=proxy_logging_obj,
2086 )
2087 )
2089 _user_api_key_obj = update_valid_token_with_end_user_params(
2090 valid_token=_user_api_key_obj, end_user_params=end_user_params
2091 )
2092 _user_api_key_obj.via_virtual_key = True
2094 return _user_api_key_obj
2096 ## IF it's not a master key
2097 ## Route should not be in master_key_only_routes
2098 if route in LiteLLMRoutes.master_key_only_routes.value:
2099 raise Exception(f"Tried to access route={route}, which is only for MASTER KEY")
2101 ## Check DB
2103 if ( 2103 ↛ 2106line 2103 didn't jump to line 2106 because the condition on line 2103 was never true
2104 prisma_client is None
2105 ): # if both master key + user key submitted, and user key != master key, and no db connected, raise an error
2106 raise ProxyException(
2107 message="No connected db.",
2108 type=ProxyErrorTypes.no_db_connection,
2109 code=400,
2110 param=None,
2111 )
2113 if valid_token is None:
2114 if isinstance(api_key, str): # if generated token, make sure it starts with sk-. 2114 ↛ 2130line 2114 didn't jump to line 2130 because the condition on line 2114 was always true
2115 _masked_key: Final = f"{api_key[:4]}****{api_key[-4:]}" if len(api_key) > 8 else "****"
2116 if not api_key.startswith("sk-"):
2117 _hint = _JWT_AUTH_DISABLED_HINT if not enable_jwt_auth and JWTHandler.is_jwt(token=api_key) else ""
2118 _malformed_key_error = HTTPException(
2119 status_code=status.HTTP_401_UNAUTHORIZED,
2120 detail=(
2121 f"{INVALID_VIRTUAL_KEY_ERROR_MESSAGE}. Received={_masked_key}, "
2122 f"expected to start with 'sk-'.{_hint}"
2123 ),
2124 ) # prevent token hashes from being used
2125 # Stamp provenance here so log routing classifies this 401 by
2126 # where it was raised, never by its message text.
2127 setattr(_malformed_key_error, INVALID_VIRTUAL_KEY_ERROR_MARKER, True)
2128 raise _malformed_key_error
2129 else:
2130 verbose_logger.warning(
2131 "litellm.proxy.proxy_server.user_api_key_auth(): Warning - Key is not a string. Got type={}".format(
2132 type(api_key) if api_key is not None else "None"
2133 )
2134 )
2135 abbreviated_api_key: Final = abbreviate_api_key(api_key=api_key)
2136 if api_key.startswith("sk-"): 2136 ↛ 2139line 2136 didn't jump to line 2139 because the condition on line 2136 was always true
2137 api_key = hash_token(token=api_key)
2139 try:
2140 with tracer.trace("litellm.proxy.auth.get_key_object_from_db"):
2141 valid_token = IdentityStore.key_from_principal(
2142 await IdentityStore(
2143 prisma_client,
2144 user_api_key_cache,
2145 parent_otel_span=parent_otel_span,
2146 proxy_logging_obj=proxy_logging_obj,
2147 ).resolve(hashed_token=api_key)
2148 )
2149 except ProxyException as e:
2150 if e.code == 401 or e.code == "401":
2151 e.message = f"Authentication Error, Invalid proxy server token passed. Received API Key = {abbreviated_api_key}, Key Hash (Token) ={api_key}. Unable to find token in cache or `LiteLLM_VerificationTokenTable`"
2152 raise e
2153 # update end-user params on valid token
2154 # These can change per request - it's important to update them here
2155 valid_token.end_user_id = end_user_params.get("end_user_id")
2156 valid_token.end_user_tpm_limit = end_user_params.get("end_user_tpm_limit")
2157 valid_token.end_user_rpm_limit = end_user_params.get("end_user_rpm_limit")
2158 valid_token.end_user_tpd_limit = end_user_params.get("end_user_tpd_limit")
2159 valid_token.allowed_model_region = end_user_params.get("allowed_model_region")
2161 if valid_token is not None: 2161 ↛ 2164line 2161 didn't jump to line 2164 because the condition on line 2161 was always true
2162 valid_token = _update_key_budget_with_temp_budget_increase(valid_token)
2164 user_obj: LiteLLM_UserTable | None = None
2165 valid_token_dict: dict = {}
2166 if valid_token is not None: 2166 ↛ 2524line 2166 didn't jump to line 2524 because the condition on line 2166 was always true
2167 # Got Valid Token from Cache, DB
2168 # Run checks for
2169 # 1. If token can call model
2170 ## 1a. If token can call fallback models (if client-side fallbacks given)
2171 # 2. If user_id for this token is in budget
2172 # 3. If the user spend within their own team is within budget
2173 # 4. If 'user' passed to /chat/completions, /embeddings endpoint is in budget
2174 # 5. If token is expired
2175 # 6. If token spend is under Budget for the token
2176 # 7. If token spend per model is under budget per model
2177 # 8. If token spend is under team budget
2178 # 9. If team spend is under team budget
2180 ## base case ## key is disabled
2181 if valid_token.blocked is True:
2182 raise Exception("Key is blocked. Update via `/key/unblock` if you're an admin.")
2183 await _enforce_key_and_fallback_model_access(
2184 valid_token=valid_token,
2185 request_data=request_data,
2186 route=route,
2187 request=request,
2188 llm_model_list=llm_model_list,
2189 llm_router=llm_router,
2190 )
2191 await _prefetch_referenced_auth_objects(
2192 valid_token, end_user_id=end_user_id, user_api_key_cache=user_api_key_cache, prisma_client=prisma_client
2193 )
2195 # Check 2. If user_id for this token is in budget - done in common_checks()
2196 if valid_token.user_id is not None:
2197 try:
2198 with tracer.trace("litellm.proxy.auth.get_user_object"):
2199 user_obj = await get_user_object(
2200 user_id=valid_token.user_id,
2201 prisma_client=prisma_client,
2202 user_api_key_cache=user_api_key_cache,
2203 user_id_upsert=False,
2204 parent_otel_span=parent_otel_span,
2205 proxy_logging_obj=proxy_logging_obj,
2206 )
2207 except Exception as e:
2208 verbose_logger.debug(
2209 "litellm.proxy.auth.user_api_key_auth.py::user_api_key_auth() - Unable to get user from db/cache. Setting user_obj to None. Exception received - %s",
2210 e,
2211 )
2212 user_obj = None
2214 if user_obj is not None: 2214 ↛ 2220line 2214 didn't jump to line 2220 because the condition on line 2214 was always true
2215 # The joint verification-token view carries the key's columns only, so the
2216 # user's own per-model budget reaches enforcement and the post-call
2217 # increment through the row fetched here.
2218 valid_token.user_model_max_budget = user_obj.model_max_budget
2220 if ( 2220 ↛ 2225line 2220 didn't jump to line 2225 because the condition on line 2220 was never true
2221 user_obj is not None
2222 and isinstance(user_obj.metadata, dict)
2223 and user_obj.metadata.get("scim_active") is False
2224 ):
2225 raise Exception(
2226 f"User={valid_token.user_id} has been deactivated via SCIM. Keys owned by this user cannot be used."
2227 )
2229 # Check 2a. Check if model has zero cost - if so, skip all budget checks
2230 model = _get_model_from_request_context(
2231 request_data=request_data,
2232 route=route,
2233 request=request,
2234 llm_router=llm_router,
2235 team_id=valid_token.team_id,
2236 )
2237 skip_budget_checks = False
2238 if model is not None and llm_router is not None: 2238 ↛ 2239line 2238 didn't jump to line 2239 because the condition on line 2238 was never true
2239 from litellm.proxy.auth.auth_checks import _is_model_cost_zero
2241 skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router)
2242 if skip_budget_checks:
2243 verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model)
2245 # Check 3. Check if user is in their team budget
2246 if not skip_budget_checks and valid_token.team_member_spend is not None: 2246 ↛ 2247line 2246 didn't jump to line 2247 because the condition on line 2246 was never true
2247 _user_id: Final = valid_token.user_id
2248 _team_id: Final = valid_token.team_id
2249 if prisma_client is not None and _user_id is not None and _team_id is not None:
2250 _cache_key: Final = team_membership_auth_cache_key(team_id=_team_id, user_id=_user_id)
2252 team_member_info = await user_api_key_cache.async_get_cache(
2253 key=_cache_key,
2254 model_type=LiteLLM_TeamMembership,
2255 )
2256 if team_member_info is None:
2257 # read from DB
2258 _db_member: Final = await TeamMembershipRepository(prisma_client).table.find_first(
2259 where={
2260 "user_id": _user_id,
2261 "team_id": _team_id,
2262 },
2263 include={"litellm_budget_table": True},
2264 )
2265 if _db_member is not None:
2266 team_member_info = LiteLLM_TeamMembership(**_db_member.model_dump())
2267 await user_api_key_cache.async_set_cache(
2268 key=_cache_key,
2269 value=team_member_info,
2270 model_type=LiteLLM_TeamMembership,
2271 ttl=5,
2272 )
2274 if team_member_info is not None and team_member_info.litellm_budget_table is not None:
2275 team_member_budget: Final = team_member_info.litellm_budget_table.effective_max_budget(
2276 now=datetime.now(timezone.utc),
2277 )
2278 if team_member_budget is not None and team_member_budget > 0:
2279 # Read from cross-pod counter (Redis-first) if available
2280 from litellm.proxy.proxy_server import get_current_spend
2282 team_member_spend = valid_token.team_member_spend
2283 if valid_token.user_id is not None and valid_token.team_id is not None:
2284 team_member_spend = await get_current_spend(
2285 counter_key=f"spend:team_member:{valid_token.user_id}:{valid_token.team_id}",
2286 fallback_spend=team_member_spend,
2287 max_budget=team_member_budget,
2288 )
2289 if team_member_spend >= team_member_budget:
2290 _entity_id: Final = f"{valid_token.user_id}:{valid_token.team_id}"
2291 raise litellm.BudgetExceededError(
2292 current_cost=team_member_spend,
2293 max_budget=team_member_budget,
2294 message=(
2295 f"Budget has been exceeded! TeamMember={_entity_id} "
2296 f"Current cost: {team_member_spend}, Max budget: {team_member_budget}"
2297 ),
2298 entity_type=Litellm_EntityType.TEAM_MEMBER.value,
2299 entity_id=_entity_id,
2300 )
2302 # Check 3. If token is expired
2303 if valid_token.expires is not None: 2303 ↛ 2304line 2303 didn't jump to line 2304 because the condition on line 2303 was never true
2304 current_time = datetime.now(timezone.utc)
2305 if isinstance(valid_token.expires, datetime):
2306 expiry_time = valid_token.expires
2307 else:
2308 expiry_time = datetime.fromisoformat(valid_token.expires)
2309 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None:
2310 expiry_time = expiry_time.replace(tzinfo=timezone.utc)
2311 verbose_proxy_logger.debug(
2312 "Checking if token expired, expiry time %s and current time %s", expiry_time, current_time
2313 )
2314 if expiry_time < current_time:
2315 # Token exists but is expired.
2316 raise ProxyException(
2317 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}",
2318 type=ProxyErrorTypes.expired_key,
2319 code=status.HTTP_401_UNAUTHORIZED,
2320 param=abbreviate_api_key(api_key=api_key),
2321 )
2323 if not skip_budget_checks: 2323 ↛ 2416line 2323 didn't jump to line 2416 because the condition on line 2323 was always true
2324 with tracer.trace("litellm.proxy.auth.budget_checks"):
2325 # Check 4. Max Budget Alert Check (runs before budget enforcement
2326 # so multi-threshold 100% alerts fire on the request that crosses
2327 # max_budget, before BudgetExceededError is raised below)
2328 await _virtual_key_max_budget_alert_check(
2329 valid_token=valid_token,
2330 proxy_logging_obj=proxy_logging_obj,
2331 user_obj=user_obj,
2332 )
2334 # Check 5. Token Spend is under budget
2335 if RouteChecks.is_llm_api_route(route=route): 2335 ↛ 2343line 2335 didn't jump to line 2343 because the condition on line 2335 was always true
2336 await _virtual_key_max_budget_check(
2337 valid_token=valid_token,
2338 proxy_logging_obj=proxy_logging_obj,
2339 user_obj=user_obj,
2340 )
2342 # Check 6. Soft Budget Check
2343 await _virtual_key_soft_budget_check(
2344 valid_token=valid_token,
2345 proxy_logging_obj=proxy_logging_obj,
2346 user_obj=user_obj,
2347 )
2349 # Check 5. Token Model Spend is under Model budget
2350 max_budget_per_model: Final = valid_token.model_max_budget
2351 current_model = _get_model_from_request_context(
2352 request_data=request_data,
2353 route=route,
2354 request=request,
2355 llm_router=llm_router,
2356 team_id=valid_token.team_id,
2357 )
2358 current_models = _get_model_names_for_budget_checks(model=current_model)
2360 if ( 2360 ↛ 2369line 2360 didn't jump to line 2369 because the condition on line 2360 was never true
2361 max_budget_per_model is not None
2362 and isinstance(max_budget_per_model, dict)
2363 and len(max_budget_per_model) > 0
2364 and prisma_client is not None
2365 and current_models
2366 and valid_token.token is not None
2367 ):
2368 ## GET THE SPEND FOR THIS MODEL
2369 for model_name in current_models:
2370 await _check_key_model_budget_with_fallback(
2371 valid_token=valid_token,
2372 model_max_budget_limiter=model_max_budget_limiter,
2373 model_name=model_name,
2374 request_data=request_data,
2375 request=request,
2376 llm_model_list=llm_model_list,
2377 llm_router=llm_router,
2378 )
2380 # Recompute after a potential budget-fallback rewrite so
2381 # the end-user check below validates the final model
2382 current_model = _get_model_from_request_context(
2383 request_data=request_data,
2384 route=route,
2385 request=request,
2386 llm_router=llm_router,
2387 team_id=valid_token.team_id,
2388 )
2389 current_models = _get_model_names_for_budget_checks(model=current_model)
2391 # Check 5a. Internal user model_max_budget
2392 if current_models: 2392 ↛ 2393line 2392 didn't jump to line 2393 because the condition on line 2392 was never true
2393 await _check_user_model_budget(
2394 valid_token=valid_token,
2395 model_max_budget_limiter=model_max_budget_limiter,
2396 models=current_models,
2397 )
2399 # Check 5b. End-user model max budget
2400 end_user_mmb: Final = valid_token.end_user_model_max_budget
2401 if ( 2401 ↛ 2408line 2401 didn't jump to line 2408 because the condition on line 2401 was never true
2402 end_user_mmb is not None
2403 and isinstance(end_user_mmb, dict)
2404 and len(end_user_mmb) > 0
2405 and current_models
2406 and valid_token.end_user_id is not None
2407 ):
2408 for model_name in current_models:
2409 await model_max_budget_limiter.is_end_user_within_model_budget(
2410 end_user_id=valid_token.end_user_id,
2411 end_user_model_max_budget=end_user_mmb,
2412 model=model_name,
2413 )
2415 # Check 6: Additional Common Checks across jwt + key auth
2416 if valid_token.team_id is not None: 2416 ↛ 2417line 2416 didn't jump to line 2417 because the condition on line 2416 was never true
2417 try:
2418 if valid_token.team_id == UI_TEAM_ID:
2419 raise TeamNotFoundError(team_id=UI_TEAM_ID)
2420 with tracer.trace("litellm.proxy.auth.get_team_object"):
2421 _team_obj = await get_team_object(
2422 team_id=valid_token.team_id,
2423 prisma_client=prisma_client,
2424 user_api_key_cache=user_api_key_cache,
2425 parent_otel_span=parent_otel_span,
2426 proxy_logging_obj=proxy_logging_obj,
2427 )
2428 except HTTPException:
2429 token_team_models: Final = _token_team_models(valid_token)
2430 _team_obj = LiteLLM_TeamTableCachedObj(
2431 team_id=valid_token.team_id,
2432 max_budget=valid_token.team_max_budget,
2433 soft_budget=valid_token.team_soft_budget,
2434 model_max_budget=valid_token.team_model_max_budget,
2435 spend=valid_token.team_spend,
2436 tpm_limit=valid_token.team_tpm_limit,
2437 rpm_limit=valid_token.team_rpm_limit,
2438 tpd_limit=valid_token.team_tpd_limit,
2439 blocked=valid_token.team_blocked,
2440 models=token_team_models,
2441 metadata=valid_token.team_metadata,
2442 object_permission_id=valid_token.team_object_permission_id,
2443 object_permission=await _resolve_object_permission_for_unresolvable_team(
2444 object_permission_id=valid_token.team_object_permission_id,
2445 prisma_client=prisma_client,
2446 user_api_key_cache=user_api_key_cache,
2447 parent_otel_span=parent_otel_span,
2448 proxy_logging_obj=proxy_logging_obj,
2449 ),
2450 )
2451 else:
2452 _team_obj = None
2454 if _team_obj is not None: 2454 ↛ 2455line 2454 didn't jump to line 2455 because the condition on line 2454 was never true
2455 valid_token.team_object_permission = _team_obj.object_permission
2456 # Keep team_metadata in sync with the freshly fetched team so that
2457 # guardrails (or any other metadata) added after the key was cached
2458 # are picked up on subsequent requests without a cache eviction.
2459 valid_token.team_metadata = _team_obj.metadata
2460 else:
2461 valid_token.team_object_permission = None
2463 # Fetch project object if key belongs to a project
2464 _project_obj = None
2465 if valid_token.project_id is not None: 2465 ↛ 2466line 2465 didn't jump to line 2466 because the condition on line 2465 was never true
2466 _project_obj = await get_project_object(
2467 project_id=valid_token.project_id,
2468 prisma_client=prisma_client,
2469 user_api_key_cache=user_api_key_cache,
2470 proxy_logging_obj=proxy_logging_obj,
2471 )
2472 if _project_obj is not None:
2473 valid_token.project_metadata = _project_obj.metadata
2474 valid_token.project_alias = _project_obj.project_alias
2476 global_proxy_spend = None
2477 if litellm.max_budget > 0 and prisma_client is not None: # user set proxy max budget 2477 ↛ 2478line 2477 didn't jump to line 2478 because the condition on line 2477 was never true
2478 cache_key: Final = GLOBAL_PROXY_SPEND_CACHE_KEY
2479 with tracer.trace("litellm.proxy.auth.get_global_proxy_spend"):
2480 global_proxy_spend = await _fetch_global_spend_with_event_coordination(
2481 cache_key=cache_key,
2482 user_api_key_cache=user_api_key_cache,
2483 prisma_client=prisma_client,
2484 )
2486 if global_proxy_spend is not None:
2487 call_info: Final = CallInfo(
2488 token=valid_token.token,
2489 spend=global_proxy_spend,
2490 max_budget=litellm.max_budget,
2491 user_id=litellm_proxy_admin_name,
2492 team_id=valid_token.team_id,
2493 event_group=Litellm_EntityType.PROXY,
2494 )
2495 asyncio.create_task(
2496 proxy_logging_obj.budget_alerts(
2497 type="proxy_budget",
2498 user_info=call_info,
2499 )
2500 )
2501 # Token passed all checks
2502 if valid_token is None: 2502 ↛ 2503line 2502 didn't jump to line 2503 because the condition on line 2502 was never true
2503 raise HTTPException(401, detail="Invalid API key")
2504 if valid_token.token is None: 2504 ↛ 2505line 2504 didn't jump to line 2505 because the condition on line 2504 was never true
2505 raise HTTPException(401, detail="Invalid API key, no token associated")
2506 api_key = valid_token.token
2508 valid_token_dict = valid_token.model_dump(exclude_none=True)
2509 valid_token_dict.pop("token", None)
2510 # budget_throttle_pct is excluded from model_dump (it must not leak
2511 # into serialized responses), so carry the request-scoped decision
2512 # forward by hand to the auth object the rate limiter receives.
2513 if valid_token.budget_throttle_pct is not None: 2513 ↛ 2514line 2513 didn't jump to line 2514 because the condition on line 2513 was never true
2514 valid_token_dict["budget_throttle_pct"] = valid_token.budget_throttle_pct
2516 if _end_user_object is not None: 2516 ↛ 2517line 2516 didn't jump to line 2517 because the condition on line 2516 was never true
2517 valid_token_dict.update(end_user_params)
2518 valid_token_dict["end_user_object_permission"] = _end_user_object.object_permission
2520 # check if token is from litellm-ui, litellm ui makes keys to allow users to login with sso. These keys can only be used for LiteLLM UI functions
2521 # sso/login, ui/login, /key functions and /user functions
2522 # this will never be allowed to call /chat/completions
2524 if valid_token is None: 2524 ↛ 2526line 2524 didn't jump to line 2526 because the condition on line 2524 was never true
2525 # No token was found when looking up in the DB
2526 raise Exception("Invalid proxy server token passed")
2527 if valid_token_dict is not None: 2527 ↛ exitline 2527 didn't return from function '_user_api_key_auth_builder' because the condition on line 2527 was always true
2528 virtual_key_auth_obj: Final = await _return_user_api_key_auth_obj(
2529 user_obj=user_obj,
2530 api_key=api_key,
2531 parent_otel_span=parent_otel_span,
2532 valid_token_dict=valid_token_dict,
2533 route=route,
2534 start_time=start_time,
2535 )
2536 virtual_key_auth_obj.via_virtual_key = True
2537 return virtual_key_auth_obj
2538 except Exception as e:
2539 return await UserAPIKeyAuthExceptionHandler._handle_authentication_error(
2540 e=e,
2541 request=request,
2542 request_data=request_data,
2543 route=route,
2544 parent_otel_span=parent_otel_span,
2545 api_key=api_key,
2546 resolved_identity=valid_token,
2547 )
2550async def _safe_fetch(label: str, awaitable):
2551 """Run an awaitable and return its result. Re-raises authentication /
2552 authorization failures (HTTPException, ProxyException,
2553 BudgetExceededError) so they propagate to the caller.
2554 Other exceptions (e.g. transient DB errors fetching context) are
2555 swallowed with a debug log and ``None`` is returned so
2556 ``common_checks`` can still run against whatever limits are recorded
2557 directly on the token.
2558 """
2559 try:
2560 return await awaitable
2561 except (HTTPException, ProxyException, litellm.BudgetExceededError) as e:
2562 verbose_proxy_logger.debug(
2563 "centralized auth: %s fetch failed (%s: %s)",
2564 label,
2565 type(e).__name__,
2566 e,
2567 )
2568 raise
2569 except Exception as e:
2570 verbose_proxy_logger.debug(
2571 "centralized auth: %s fetch swallowed (%s: %s)",
2572 label,
2573 type(e).__name__,
2574 e,
2575 )
2576 return None
2579def _team_obj_from_token(valid_token: UserAPIKeyAuth) -> LiteLLM_TeamTableCachedObj:
2580 """Reconstruct a cached team object from the fields already on the
2581 UserAPIKeyAuth. Only called when valid_token.team_id is known to be
2582 non-None (the caller gates on it)."""
2583 assert valid_token.team_id is not None
2584 token_team_models: Final = _token_team_models(valid_token)
2585 return LiteLLM_TeamTableCachedObj(
2586 team_id=valid_token.team_id,
2587 max_budget=valid_token.team_max_budget,
2588 soft_budget=valid_token.team_soft_budget,
2589 model_max_budget=valid_token.team_model_max_budget,
2590 spend=valid_token.team_spend,
2591 tpm_limit=valid_token.team_tpm_limit,
2592 rpm_limit=valid_token.team_rpm_limit,
2593 tpd_limit=valid_token.team_tpd_limit,
2594 blocked=valid_token.team_blocked,
2595 models=token_team_models,
2596 metadata=valid_token.team_metadata,
2597 object_permission_id=valid_token.team_object_permission_id,
2598 )
2601def _token_can_vouch_for_team(valid_token: UserAPIKeyAuth, lookup_error: BaseException) -> bool:
2602 """Whether the token's own team fields may stand in for a team that failed to
2603 resolve, without widening access.
2605 The UI dashboard mints every session key against the ``UI_TEAM_ID`` sentinel,
2606 which by design never has a team row, so a failed lookup for it is not a
2607 degraded read to be treated with suspicion; it always vouches, exactly as it
2608 always safely has (these keys are restricted elsewhere to UI-only routes).
2610 For every other team, a team that is provably gone is a definitive answer,
2611 not a degraded read, so nothing may stand in for it and no setting may
2612 override that.
2614 Otherwise the team's grant is merely unknown. A token carrying one may vouch,
2615 since replaying a recorded grant cannot widen it and denying every team key
2616 while the row is briefly unreadable would trade the widening for an outage. A
2617 token carrying none may not: ``team_models=[]`` reads as every model and
2618 ``team_blocked=False`` as unblocked. ``allow_requests_on_db_unavailable`` opts
2619 back out, and is only consulted here because the failure is known by this
2620 point to be a degraded read.
2621 """
2622 if valid_token.team_id == UI_TEAM_ID:
2623 return True
2624 if isinstance(lookup_error, TeamNotFoundError):
2625 return False
2626 if valid_token.team_models:
2627 return True
2628 return PrismaDBExceptionHandler.should_allow_request_on_db_unavailable()
2631async def _inherit_org_identity(
2632 user_api_key_auth_obj: UserAPIKeyAuth,
2633 team_object: LiteLLM_TeamTableCachedObj | None,
2634 prisma_client: PrismaClient | None,
2635 user_api_key_cache: UserApiKeyCache,
2636 parent_otel_span: Span | None,
2637 proxy_logging_obj: ProxyLogging | None,
2638) -> None:
2639 if user_api_key_auth_obj.org_id is None and team_object is not None and team_object.organization_id is not None: 2639 ↛ 2640line 2639 didn't jump to line 2640 because the condition on line 2639 was never true
2640 user_api_key_auth_obj.org_id = team_object.organization_id
2641 already_populated: Final = any(
2642 value is not None
2643 for value in (
2644 user_api_key_auth_obj.organization_alias,
2645 user_api_key_auth_obj.organization_max_budget,
2646 user_api_key_auth_obj.organization_tpm_limit,
2647 user_api_key_auth_obj.organization_rpm_limit,
2648 user_api_key_auth_obj.organization_metadata,
2649 )
2650 )
2651 if user_api_key_auth_obj.org_id is None or already_populated or prisma_client is None: 2651 ↛ 2653line 2651 didn't jump to line 2653 because the condition on line 2651 was always true
2652 return
2653 org_object: Final = await get_org_object_for_request(
2654 org_id=user_api_key_auth_obj.org_id,
2655 prisma_client=prisma_client,
2656 user_api_key_cache=user_api_key_cache,
2657 parent_otel_span=parent_otel_span,
2658 proxy_logging_obj=proxy_logging_obj,
2659 )
2660 if org_object is None:
2661 return
2662 user_api_key_auth_obj.organization_alias = org_object.organization_alias
2663 user_api_key_auth_obj.organization_metadata = org_object.metadata
2664 budget: Final = org_object.litellm_budget_table
2665 if budget is None:
2666 return
2667 user_api_key_auth_obj.organization_max_budget = budget.max_budget
2668 user_api_key_auth_obj.organization_tpm_limit = budget.tpm_limit
2669 user_api_key_auth_obj.organization_rpm_limit = budget.rpm_limit
2672def is_no_auth_dev_mode(master_key: str | None, general_settings: Mapping[str, object]) -> bool:
2673 return master_key is None and not any(
2674 general_settings.get(flag, False)
2675 for flag in ("enable_jwt_auth", "enable_oauth2_auth", "enable_oauth2_proxy_auth")
2676 )
2679@tracer.wrap()
2680async def _run_centralized_common_checks(
2681 user_api_key_auth_obj: UserAPIKeyAuth,
2682 request: Request,
2683 request_data: dict[str, object],
2684 route: str,
2685) -> None:
2686 """Run ``common_checks`` once at the ``user_api_key_auth`` wrapper
2687 boundary, regardless of which ``_user_api_key_auth_builder`` path
2688 returned. This is the single invariant enforcement point for key
2689 model-access, budgets, guardrails, org, and vector-store checks.
2691 Invariants:
2692 - ``user_custom_auth`` with ``custom_auth_run_common_checks`` unset
2693 skips the gate — matches the existing custom-auth RPS guarantee.
2694 Custom-auth deployments don't use OAuth2 / DB-fallback paths, so
2695 the skip does not re-open any bypass.
2696 - ``PROXY_ADMIN`` tokens still run through ``common_checks`` so
2697 team-blocked / team-budget / end-user-budget / tag-budget /
2698 vector-store / tool-allowlist enforcement applies to admin keys
2699 too. Admin status is honored where the underlying check exempts it
2700 (``_is_api_route_allowed``, ``organization_role_based_access_check``).
2701 """
2702 from litellm.proxy.proxy_server import (
2703 general_settings,
2704 litellm_proxy_admin_name,
2705 llm_router,
2706 master_key,
2707 model_max_budget_limiter,
2708 prisma_client,
2709 proxy_logging_obj,
2710 user_api_key_cache,
2711 user_custom_auth,
2712 )
2714 # Public routes (e.g. /health/liveness) are exempt from
2715 # auth in the builder — the wrapper must not retroactively apply
2716 # authz on top, or k8s readiness probes and other unauthenticated
2717 # callers get 401.
2718 if route in LiteLLMRoutes.public_routes.value or route_in_additonal_public_routes(current_route=route):
2719 return
2721 # User-configured pass-through endpoints with ``auth: false`` are
2722 # explicitly unauthenticated — the builder returns an empty
2723 # UserAPIKeyAuth() and the request is forwarded as-is. Running
2724 # common_checks on the empty token would reject the request as
2725 # admin-only. The "auth" flag on the endpoint config is the
2726 # contract; honor it.
2727 pass_through_endpoints: Final = general_settings.get("pass_through_endpoints", None)
2728 if pass_through_endpoints is not None:
2729 for endpoint in pass_through_endpoints:
2730 if isinstance(endpoint, dict) and endpoint.get("path", "") == route and endpoint.get("auth") is not True: 2730 ↛ 2731line 2730 didn't jump to line 2731 because the condition on line 2730 was never true
2731 return
2733 # No-auth dev mode: master_key unset AND no JWT/OAuth2 auth
2734 # configured. The builder returns an INTERNAL_USER token for any
2735 # api_key; the proxy is unauthenticated by configuration.
2736 # Running common_checks would block every admin route on these
2737 # deployments where that was previously not the contract. If any
2738 # authn is enabled (JWT, OAuth2, OAuth2-proxy), authz must run.
2739 if is_no_auth_dev_mode(master_key, general_settings): 2739 ↛ 2740line 2739 didn't jump to line 2740 because the condition on line 2739 was never true
2740 return
2742 if user_custom_auth is not None and not general_settings.get("custom_auth_run_common_checks", False): 2742 ↛ 2743line 2742 didn't jump to line 2743 because the condition on line 2742 was never true
2743 return
2745 parent_otel_span: Final = user_api_key_auth_obj.parent_otel_span
2746 # In the integrated auth flow ``_user_api_key_auth_builder`` has already
2747 # resolved the end-user id and attached it here. Reuse that to avoid a
2748 # second extraction pass; fall back to extracting locally when the
2749 # function is invoked in isolation (e.g. in direct unit tests).
2750 key_end_user_budget_id: Final = get_key_end_user_budget_id(user_api_key_auth_obj.metadata)
2751 end_user_id = user_api_key_auth_obj.end_user_id
2752 if end_user_id is None:
2753 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request))
2754 end_user_id = await resolve_and_validate_end_user_id(
2755 raw_end_user_id=raw_end_user_id,
2756 prisma_client=prisma_client,
2757 user_api_key_cache=user_api_key_cache,
2758 parent_otel_span=parent_otel_span,
2759 proxy_logging_obj=proxy_logging_obj,
2760 route=route,
2761 key_end_user_budget_id=key_end_user_budget_id,
2762 )
2763 if end_user_id is not None and key_end_user_budget_id is not None: 2763 ↛ 2764line 2763 didn't jump to line 2764 because the condition on line 2763 was never true
2764 user_api_key_auth_obj.end_user_id = end_user_id
2766 fetch_coros: Final = []
2767 if user_api_key_auth_obj.team_id is not None and user_api_key_auth_obj.team_id != UI_TEAM_ID: 2767 ↛ 2768line 2767 didn't jump to line 2768 because the condition on line 2767 was never true
2768 fetch_coros.append(
2769 _safe_fetch(
2770 "team",
2771 get_team_object(
2772 team_id=user_api_key_auth_obj.team_id,
2773 prisma_client=prisma_client,
2774 user_api_key_cache=user_api_key_cache,
2775 parent_otel_span=parent_otel_span,
2776 proxy_logging_obj=proxy_logging_obj,
2777 ),
2778 )
2779 )
2780 else:
2781 fetch_coros.append(_safe_fetch("team", _noop_none()))
2783 if user_api_key_auth_obj.user_id is not None:
2784 fetch_coros.append(
2785 _safe_fetch(
2786 "user",
2787 get_user_object(
2788 user_id=user_api_key_auth_obj.user_id,
2789 prisma_client=prisma_client,
2790 user_api_key_cache=user_api_key_cache,
2791 user_id_upsert=False,
2792 parent_otel_span=parent_otel_span,
2793 proxy_logging_obj=proxy_logging_obj,
2794 ),
2795 )
2796 )
2797 else:
2798 fetch_coros.append(_safe_fetch("user", _noop_none()))
2800 if user_api_key_auth_obj.project_id is not None: 2800 ↛ 2801line 2800 didn't jump to line 2801 because the condition on line 2800 was never true
2801 fetch_coros.append(
2802 _safe_fetch(
2803 "project",
2804 get_project_object(
2805 project_id=user_api_key_auth_obj.project_id,
2806 prisma_client=prisma_client,
2807 user_api_key_cache=user_api_key_cache,
2808 proxy_logging_obj=proxy_logging_obj,
2809 ),
2810 )
2811 )
2812 else:
2813 fetch_coros.append(_safe_fetch("project", _noop_none()))
2815 if end_user_id:
2816 fetch_coros.append(
2817 _safe_fetch(
2818 "end_user",
2819 get_end_user_object(
2820 end_user_id=end_user_id,
2821 prisma_client=prisma_client,
2822 user_api_key_cache=user_api_key_cache,
2823 parent_otel_span=parent_otel_span,
2824 proxy_logging_obj=proxy_logging_obj,
2825 route=route,
2826 token_end_user_max_budget=user_api_key_auth_obj.end_user_max_budget,
2827 key_end_user_budget_id=key_end_user_budget_id,
2828 ),
2829 )
2830 )
2831 else:
2832 fetch_coros.append(_safe_fetch("end_user", _noop_none()))
2834 fetch_coros.append(
2835 _safe_fetch(
2836 "global_spend",
2837 get_global_proxy_spend(
2838 litellm_proxy_admin_name=litellm_proxy_admin_name,
2839 user_api_key_cache=user_api_key_cache,
2840 prisma_client=prisma_client,
2841 token=user_api_key_auth_obj.token or "",
2842 proxy_logging_obj=proxy_logging_obj,
2843 ),
2844 )
2845 )
2847 # Per-fetch error isolation. ``_safe_fetch`` lets HTTPException,
2848 # ProxyException, and BudgetExceededError escape (everything else is
2849 # already swallowed to None). A bare ``except`` over ``gather`` would
2850 # let one fetch's HTTPException null out every other context — e.g.
2851 # a 404 from ``get_team_object`` (token references a deleted team)
2852 # would silently skip the user, end-user, project, and global-spend
2853 # checks. Use ``return_exceptions=True`` and apply per-fetch fallback
2854 # so a missing team only zeros out the team object.
2855 (
2856 team_result,
2857 user_result,
2858 project_result,
2859 end_user_result,
2860 global_spend_result,
2861 ) = await asyncio.gather(*fetch_coros, return_exceptions=True)
2863 # ProxyException / BudgetExceededError are authorization failures —
2864 # propagate so the wrapper renders them. HTTPException is fallback
2865 # material (404 from get_team_object is the only known producer).
2866 for r in (
2867 team_result,
2868 user_result,
2869 project_result,
2870 end_user_result,
2871 global_spend_result,
2872 ):
2873 if isinstance(r, (ProxyException, litellm.BudgetExceededError)): 2873 ↛ 2874line 2873 didn't jump to line 2874 because the condition on line 2873 was never true
2874 raise r
2876 # Use BaseException (not HTTPException) in the narrowing checks so
2877 # mypy can narrow ``Any | BaseException`` to the typed object in the
2878 # else branch. After the for-loop above, the only BaseException that
2879 # can still appear here is HTTPException (other listed re-raises were
2880 # propagated; non-listed exceptions were already swallowed to None).
2881 team_object: LiteLLM_TeamTableCachedObj | None
2882 if isinstance(team_result, BaseException): 2882 ↛ 2885line 2882 didn't jump to line 2885 because the condition on line 2882 was never true
2883 # Token-derived fallback only valid when a team_id is set;
2884 # _team_obj_from_token asserts that precondition.
2885 if user_api_key_auth_obj.team_id is None:
2886 team_object = None
2887 elif _token_can_vouch_for_team(user_api_key_auth_obj, team_result):
2888 team_object = _team_obj_from_token(user_api_key_auth_obj)
2889 else:
2890 raise team_result
2891 else:
2892 team_object = (
2893 _team_obj_from_token(user_api_key_auth_obj) if user_api_key_auth_obj.team_id == UI_TEAM_ID else team_result
2894 )
2896 user_object: LiteLLM_UserTable | None = None if isinstance(user_result, BaseException) else user_result
2897 project_object: Final[LiteLLM_ProjectTableCachedObj | None] = (
2898 None if isinstance(project_result, BaseException) else project_result
2899 )
2900 end_user_object: Final[LiteLLM_EndUserTable | None] = (
2901 None if isinstance(end_user_result, BaseException) else end_user_result
2902 )
2903 global_proxy_spend: float | None = None if isinstance(global_spend_result, BaseException) else global_spend_result
2904 carry_team_and_user_budget_state(
2905 valid_token=user_api_key_auth_obj,
2906 team_object=team_object,
2907 user_object=user_object,
2908 )
2910 await _inherit_org_identity(
2911 user_api_key_auth_obj=user_api_key_auth_obj,
2912 team_object=team_object,
2913 prisma_client=prisma_client,
2914 user_api_key_cache=user_api_key_cache,
2915 parent_otel_span=parent_otel_span,
2916 proxy_logging_obj=proxy_logging_obj,
2917 )
2919 # common_checks identifies admin via user_object, not the token
2920 # (non_proxy_admin_allowed_routes_check). JWT admin shortcut and
2921 # master_key tokens get admin from the token; the DB row for the
2922 # same user_id (e.g. litellm_proxy_admin_name = "default_user_id")
2923 # may have a non-admin user_role and would otherwise demote the
2924 # caller. The token is the source of truth for these paths — force
2925 # the admin user_object whenever the token says PROXY_ADMIN, even
2926 # if a DB row was fetched.
2927 if user_api_key_auth_obj.user_role == LitellmUserRoles.PROXY_ADMIN:
2928 user_object = LiteLLM_UserTable(
2929 user_id=user_api_key_auth_obj.user_id or litellm_proxy_admin_name,
2930 user_role=LitellmUserRoles.PROXY_ADMIN,
2931 spend=user_object.spend if user_object is not None else 0.0,
2932 )
2934 if project_object is not None: 2934 ↛ 2935line 2934 didn't jump to line 2935 because the condition on line 2934 was never true
2935 user_api_key_auth_obj.project_metadata = project_object.metadata
2936 user_api_key_auth_obj.project_alias = project_object.project_alias
2938 if end_user_id and key_end_user_budget_id is not None and prisma_client is not None: 2938 ↛ 2939line 2938 didn't jump to line 2939 because the condition on line 2938 was never true
2939 await _apply_key_end_user_default_budget_to_token(
2940 valid_token=user_api_key_auth_obj,
2941 end_user_object=end_user_object,
2942 key_end_user_budget_id=key_end_user_budget_id,
2943 prisma_client=prisma_client,
2944 user_api_key_cache=user_api_key_cache,
2945 parent_otel_span=parent_otel_span,
2946 keep_token_limits=user_custom_auth is not None,
2947 )
2949 skip_budget_checks: Final = _should_skip_budget_checks(
2950 request_data=request_data,
2951 route=route,
2952 request=request,
2953 llm_router=llm_router,
2954 team_id=user_api_key_auth_obj.team_id,
2955 )
2957 # Pin the metadata variable name (litellm_metadata vs metadata) before
2958 # any tag merge runs. Without this, header tags from
2959 # apply_client_tag_policy_pre_auth would land in `metadata` while the
2960 # later seed in common_checks pushes key tags and the
2961 # _tag_max_budget_check read into `litellm_metadata`, hiding header
2962 # tags from per-tag budget enforcement on LITELLM_METADATA_ROUTES.
2963 LiteLLMProxyRequestSetup.pre_seed_litellm_metadata_for_route(
2964 request_data=request_data,
2965 route=route,
2966 )
2968 # Merge x-litellm-tags into request_data BEFORE common_checks runs.
2969 # _tag_max_budget_check inside common_checks only inspects request_data;
2970 # without this pre-merge, header-supplied tags bypass tag-budget
2971 # enforcement.
2972 LiteLLMProxyRequestSetup.apply_client_tag_policy_pre_auth(
2973 request=request,
2974 request_data=request_data,
2975 user_api_key_dict=user_api_key_auth_obj,
2976 )
2978 bind_admission_counter_keys(user_api_key_auth_obj, end_user_id=end_user_id)
2979 try:
2980 _ = await common_checks(
2981 request=request,
2982 request_body=request_data,
2983 team_object=team_object,
2984 user_object=user_object,
2985 end_user_object=end_user_object,
2986 general_settings=general_settings,
2987 global_proxy_spend=global_proxy_spend,
2988 route=route,
2989 llm_router=llm_router,
2990 proxy_logging_obj=proxy_logging_obj,
2991 valid_token=user_api_key_auth_obj,
2992 skip_budget_checks=skip_budget_checks,
2993 project_object=project_object,
2994 )
2995 finally:
2996 release_spend_counter_batch()
2998 if not skip_budget_checks: 2998 ↛ 3013line 2998 didn't jump to line 3013 because the condition on line 2998 was always true
2999 await _check_team_model_budget(
3000 valid_token=user_api_key_auth_obj,
3001 model_max_budget_limiter=model_max_budget_limiter,
3002 models=_get_model_names_for_budget_checks(
3003 model=_get_model_from_request_context(
3004 request_data=request_data,
3005 route=route,
3006 request=request,
3007 llm_router=llm_router,
3008 team_id=user_api_key_auth_obj.team_id,
3009 )
3010 ),
3011 )
3013 await _reserve_budget_after_common_checks(
3014 user_api_key_auth_obj=user_api_key_auth_obj,
3015 request=request,
3016 request_data=request_data,
3017 route=route,
3018 llm_router=llm_router,
3019 team_object=team_object,
3020 user_object=user_object,
3021 end_user_id=end_user_id,
3022 end_user_object=end_user_object,
3023 prisma_client=prisma_client,
3024 user_api_key_cache=user_api_key_cache,
3025 proxy_logging_obj=proxy_logging_obj,
3026 skip_budget_checks=skip_budget_checks,
3027 general_settings=general_settings,
3028 )
3031async def _noop_none() -> None:
3032 """Sentinel coroutine for asyncio.gather when a fetch is unnecessary
3033 (e.g. token has no team_id). Keeps the result tuple positional."""
3034 return
3037async def _apply_key_end_user_default_budget_to_token(
3038 valid_token: UserAPIKeyAuth,
3039 end_user_object: LiteLLM_EndUserTable | None,
3040 key_end_user_budget_id: str,
3041 prisma_client: PrismaClient,
3042 user_api_key_cache: UserApiKeyCache,
3043 parent_otel_span: Span | None,
3044 keep_token_limits: bool,
3045) -> None:
3046 """The builder's end-user pass runs before the key is resolved, so only here can the key's
3047 ``end_user_budget_id`` win over the proxy-wide default on the token that reservation reads.
3048 On the virtual-key path the token's end-user limits are the builder's proxy-wide defaults and
3049 the key budget replaces them wholesale. With ``keep_token_limits`` (custom auth) the token's
3050 limits are caps the custom auth callable set, so the key budget only fills the ones it left
3051 unset."""
3052 default_budget: Final = (
3053 end_user_object.litellm_budget_table
3054 if end_user_object is not None
3055 else await resolve_default_end_user_budget(
3056 prisma_client=prisma_client,
3057 user_api_key_cache=user_api_key_cache,
3058 key_end_user_budget_id=key_end_user_budget_id,
3059 parent_otel_span=parent_otel_span,
3060 )
3061 )
3062 if default_budget is None:
3063 return
3065 if not keep_token_limits or valid_token.end_user_max_budget is None:
3066 valid_token.end_user_max_budget = default_budget.max_budget
3067 if not keep_token_limits or valid_token.end_user_tpm_limit is None:
3068 valid_token.end_user_tpm_limit = default_budget.tpm_limit
3069 if not keep_token_limits or valid_token.end_user_rpm_limit is None:
3070 valid_token.end_user_rpm_limit = default_budget.rpm_limit
3071 if not keep_token_limits or valid_token.end_user_tpd_limit is None:
3072 valid_token.end_user_tpd_limit = default_budget.tpd_limit
3073 if not keep_token_limits or valid_token.end_user_model_max_budget is None:
3074 valid_token.end_user_model_max_budget = default_budget.model_max_budget
3077async def _reserve_budget_after_common_checks(
3078 user_api_key_auth_obj: UserAPIKeyAuth,
3079 request_data: dict,
3080 route: str,
3081 llm_router: Any | None,
3082 team_object: LiteLLM_TeamTableCachedObj | None,
3083 user_object: LiteLLM_UserTable | None,
3084 prisma_client: PrismaClient | None,
3085 user_api_key_cache: UserApiKeyCache,
3086 proxy_logging_obj: ProxyLogging,
3087 skip_budget_checks: bool,
3088 general_settings: dict,
3089 end_user_id: str | None = None,
3090 end_user_object: LiteLLM_EndUserTable | None = None,
3091 request: Request | None = None,
3092) -> None:
3093 user_api_key_auth_obj.budget_reservation = None
3094 if not skip_budget_checks and general_settings.get("disable_budget_reservation") is not True: 3094 ↛ 3115line 3094 didn't jump to line 3115 because the condition on line 3094 was always true
3095 from litellm.proxy.spend_tracking.budget_reservation import (
3096 reserve_budget_for_request,
3097 )
3099 user_api_key_auth_obj.budget_reservation = await reserve_budget_for_request(
3100 request_body=request_data,
3101 route=route,
3102 llm_router=llm_router,
3103 valid_token=user_api_key_auth_obj,
3104 team_object=team_object,
3105 user_object=user_object,
3106 prisma_client=prisma_client,
3107 user_api_key_cache=user_api_key_cache,
3108 proxy_logging_obj=proxy_logging_obj,
3109 end_user_id=end_user_id,
3110 end_user_object=end_user_object,
3111 apply_user_budget_to_team_keys=general_settings.get("apply_user_budget_to_team_keys") is True,
3112 fail_closed_budget_enforcement=general_settings.get("fail_closed_budget_enforcement") is True,
3113 raw_body=await read_raw_json_body(request=request),
3114 )
3115 if request is not None: 3115 ↛ exitline 3115 didn't return from function '_reserve_budget_after_common_checks' because the condition on line 3115 was always true
3116 reservation: Final = user_api_key_auth_obj.budget_reservation
3117 request.state.budget_reservation = reservation # rebind-ok: read by the release middleware
3120def _should_skip_budget_checks(
3121 request_data: dict,
3122 route: str,
3123 request: Request | None,
3124 llm_router: Any | None,
3125 team_id: str | None = None,
3126) -> bool:
3127 model: Final = _get_model_from_request_context(
3128 request_data=request_data,
3129 route=route,
3130 request=request,
3131 llm_router=llm_router,
3132 team_id=team_id,
3133 )
3134 if model is not None and llm_router is not None:
3135 return _is_model_cost_zero(model=model, llm_router=llm_router)
3136 return False
3139def _resolve_request_principal(request: Request, valid_token: UserAPIKeyAuth) -> Principal:
3140 """Project the resolved identity into one per-request Principal, off the key
3141 object the builder already fetched, and stamp the request network context
3142 onto it once. X-Forwarded-For is only trusted when the operator configured
3143 ``trusted_proxy_ranges``; otherwise the direct peer is authoritative.
3145 credential_ref and a stable subject fallback are always set off the token so
3146 the Principal can never be anonymous, even for a keyless service-account key
3147 with no user or alias."""
3148 cidrs: Final = get_trusted_proxy_cidrs()
3149 network: Final = resolve_network_context(
3150 request,
3151 TrustedProxyConfig(use_forwarded_for=bool(cidrs), trusted_proxy_cidrs=cidrs),
3152 )
3153 auth_method: Final = AuthMethod.BEARER_JWT if valid_token.jwt_claims else AuthMethod.API_KEY
3154 return IdentityStore._principal_from_key(
3155 valid_token,
3156 auth_method=auth_method,
3157 network=network,
3158 subject_fallback=valid_token.token,
3159 credential_ref=CredentialRef(token_id=valid_token.token),
3160 )
3163async def _authorize_authenticated_request(
3164 user_api_key_auth_obj: UserAPIKeyAuth,
3165 request: Request,
3166 request_data: dict,
3167 route: str,
3168 api_key: str,
3169) -> UserAPIKeyAuth | None:
3170 """Authorize an already-authenticated request: disabled-route check, the single
3171 ``common_checks`` gate (which also reserves budget), and end-user fallback
3172 resolution. Returns the auth object the exception handler recovered when a check
3173 failed but the request may proceed anyway, else ``None``.
3174 """
3175 ## ENSURE DISABLE ROUTE WORKS ACROSS ALL USER AUTH FLOWS ##
3176 RouteChecks.should_call_route(route=route, valid_token=user_api_key_auth_obj, request=request)
3177 await _normalize_claude_model(request_data, user_api_key_auth_obj, request, route)
3178 await _resolve_router_settings_model_group_alias(request_data, user_api_key_auth_obj, request, route)
3180 # Single authorization point. Builder paths MUST NOT call common_checks.
3181 # Route through the same exception handler the builder uses so
3182 # authorization failures (ProxyException, or plain Exception from
3183 # admin-only-route / model-access / budget checks) surface as
3184 # ProxyException consistently with pre-refactor behavior.
3185 try:
3186 await _run_centralized_common_checks(
3187 user_api_key_auth_obj=user_api_key_auth_obj,
3188 request=request,
3189 request_data=request_data,
3190 route=route,
3191 )
3192 except Exception as e:
3193 return await UserAPIKeyAuthExceptionHandler._handle_authentication_error(
3194 e=e,
3195 request=request,
3196 request_data=request_data,
3197 route=route,
3198 parent_otel_span=user_api_key_auth_obj.parent_otel_span,
3199 api_key=api_key,
3200 resolved_identity=user_api_key_auth_obj,
3201 )
3203 # Defense-in-depth: ``_user_api_key_auth_builder`` has multiple early-return
3204 # paths (no master key, /user/auth route, JWT short-circuits) that bypass
3205 # the end-user resolution block. If those paths produced an auth obj
3206 # without an ``end_user_id`` set, fall back to extracting from the request
3207 # body so spend logs are still attributed correctly. Validation honours
3208 # ``litellm.validate_end_user_id_in_db``.
3209 if user_api_key_auth_obj.end_user_id is None:
3210 from litellm.proxy.proxy_server import (
3211 prisma_client,
3212 proxy_logging_obj,
3213 user_api_key_cache,
3214 )
3216 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request))
3217 if raw_end_user_id is not None: 3217 ↛ 3218line 3217 didn't jump to line 3218 because the condition on line 3217 was never true
3218 resolved_end_user_id: Final = await resolve_and_validate_end_user_id(
3219 raw_end_user_id=raw_end_user_id,
3220 prisma_client=prisma_client,
3221 user_api_key_cache=user_api_key_cache,
3222 parent_otel_span=user_api_key_auth_obj.parent_otel_span,
3223 proxy_logging_obj=proxy_logging_obj,
3224 route=route,
3225 key_end_user_budget_id=get_key_end_user_budget_id(user_api_key_auth_obj.metadata),
3226 )
3227 if resolved_end_user_id is not None:
3228 user_api_key_auth_obj.end_user_id = resolved_end_user_id
3229 return None
3232def _spend_counter_redis_cache() -> RedisCache | None:
3233 from litellm.proxy.proxy_server import spend_counter_cache
3235 return spend_counter_cache.redis_cache
3238async def _prefetch_referenced_auth_objects(
3239 valid_token: UserAPIKeyAuth,
3240 end_user_id: str | None,
3241 user_api_key_cache: UserApiKeyCache,
3242 prisma_client: PrismaClient | None,
3243) -> None:
3244 """Warm every object and spend counter the checks below will read, in one MGET each (one DB query when cold).
3245 Runs after the key's model access check so a denied request costs no more than it did before."""
3246 bind_admission_counter_keys(valid_token, end_user_id=end_user_id or None)
3247 await prefetch_auth_objects(
3248 refs=AuthObjectRefs.from_token(valid_token),
3249 user_api_key_cache=user_api_key_cache,
3250 prisma_client=prisma_client,
3251 )
3254def _seed_request_destinations(user_api_key_dict: UserAPIKeyAuth, request: Request | None = None) -> None:
3255 """Anchor the OTLP destinations this key or team overrides its traces to.
3257 Called inside the ``auth`` phase span so that span reaches the tenant's account
3258 as well, and on the request task so the ``ContextVar`` is inherited by the logging
3259 tasks that close the LLM span. Best-effort: trace routing must never fail auth.
3261 ``request`` carries the headers, so a backend this request disabled with
3262 ``x-litellm-disable-callbacks`` resolves to no destination.
3264 Only destinations the published fan-out can build are anchored. Anchoring one is
3265 what tells the operator's exporter to hold that backend's spans back under
3266 ``override``, so an unbuildable one would leave the span with nowhere to go.
3268 The ``postgres`` spans under ``auth`` close before this runs, because they are the
3269 reads that resolve the identity being read here. They never reach the tenant's
3270 account, and they are never withheld from the operator's backend, whichever mode
3271 is set.
3272 """
3273 try:
3274 from litellm.integrations.otel.logger import fan_out_provider
3275 from litellm.integrations.otel.plumbing.context import set_request_destinations
3276 from litellm.integrations.otel.plumbing.providers import deliverable_destinations
3277 from litellm.proxy.litellm_pre_call_utils import (
3278 resolve_tenant_otel_destinations,
3279 )
3281 set_request_destinations(
3282 deliverable_destinations(
3283 resolve_tenant_otel_destinations(user_api_key_dict, _safe_get_request_headers(request)),
3284 fan_out_provider(),
3285 )
3286 )
3287 except Exception as exc: # noqa: BLE001 # telemetry routing is best-effort and must never break authentication
3288 verbose_proxy_logger.debug("OTel V2: tenant destination resolution failed: %s", exc)
3291@tracer.wrap()
3292async def user_api_key_auth(
3293 request: Request,
3294 api_key: str = fastapi.Security(api_key_header),
3295 azure_api_key_header: str = fastapi.Security(azure_api_key_header),
3296 anthropic_api_key_header: str | None = fastapi.Security(anthropic_api_key_header),
3297 google_ai_studio_api_key_header: str | None = fastapi.Security(google_ai_studio_api_key_header),
3298 azure_apim_header: str | None = fastapi.Security(azure_apim_header),
3299 custom_litellm_key_header: str | None = fastapi.Security(custom_litellm_key_header),
3300) -> UserAPIKeyAuth:
3301 """
3302 Parent function to authenticate user api key / jwt token.
3303 """
3305 # Create the SERVER span and stash it on request.state BEFORE reading the
3306 # body. _read_request_body can raise ProxyException for malformed JSON;
3307 # without this, that path leaves no span for the exception handler to
3308 # close, and the trace never reaches the backend.
3309 _ensure_parent_otel_span_on_request_state(request)
3311 request_data, body_parse_exception = await _read_request_body_deferring_parse_failure(request=request)
3312 route: Final[str] = get_request_route(request=request)
3313 ## CHECK IF ROUTE IS ALLOWED
3315 # Run the whole auth phase inside a live ``auth`` span so the DB lookups it
3316 # triggers (key/user/team object reads) nest under it instead of flattening
3317 # onto the server span. No-op when OTel V2 isn't active.
3318 with phase_span(f"auth {route}"), spend_counter_batch_scope(_spend_counter_redis_cache()):
3319 try:
3320 user_api_key_auth_obj: Final = await _user_api_key_auth_builder(
3321 request=request,
3322 api_key=api_key,
3323 azure_api_key_header=azure_api_key_header,
3324 anthropic_api_key_header=anthropic_api_key_header,
3325 google_ai_studio_api_key_header=google_ai_studio_api_key_header,
3326 azure_apim_header=azure_apim_header,
3327 request_data=request_data,
3328 custom_litellm_key_header=custom_litellm_key_header,
3329 )
3330 except Exception:
3331 # The body was read first, so a caller who sent both a malformed body and
3332 # a rejected key used to get the 400; the response is unchanged, and the
3333 # auth failure is still recorded on the trace by the handler that ran.
3334 if body_parse_exception is not None:
3335 raise body_parse_exception
3336 raise
3337 user_api_key_auth_obj.budget_reservation = None
3338 user_api_key_auth_obj.agent_caller = agent_caller_from_headers(
3339 _safe_get_request_headers(request), user_api_key_auth_obj
3340 )
3341 _seed_request_destinations(user_api_key_auth_obj, request)
3343 # A body that never parsed is authenticated (so the trace carries identity
3344 # and this ``auth`` span) but not authorized: there is no model to check it
3345 # against, and budget reservation would increment live spend counters that
3346 # only the endpoint's post-call path releases; the endpoint never runs, since
3347 # the parse failure is raised below.
3348 if body_parse_exception is None:
3349 recovered_auth_obj: Final = await _authorize_authenticated_request(
3350 user_api_key_auth_obj=user_api_key_auth_obj,
3351 request=request,
3352 request_data=request_data,
3353 route=route,
3354 api_key=api_key,
3355 )
3356 if recovered_auth_obj is not None: 3356 ↛ 3357line 3356 didn't jump to line 3357 because the condition on line 3356 was never true
3357 return recovered_auth_obj
3359 # Identity is now resolved. Seed it AFTER the auth span closes so the Baggage
3360 # persists on the request task (detaching the span's context token inside the
3361 # ``with`` would unwind a Baggage attach made within it) and every post-auth
3362 # span — pre-call, LLM call, guardrail, spend write — inherits team/key/user.
3363 seed_request_identity(
3364 user_api_key_auth_obj,
3365 model=request_data.get("model") if isinstance(request_data, dict) else None,
3366 )
3367 user_api_key_auth_obj.request_route = normalize_request_route(route)
3369 if body_parse_exception is not None:
3370 await _record_unparsable_body_failure(
3371 user_api_key_dict=user_api_key_auth_obj,
3372 body_parse_exception=body_parse_exception,
3373 route=route,
3374 )
3375 raise body_parse_exception
3377 # Resolve caller identity once, here at the seam, into a single per-request
3378 # Principal projected off the key object the builder already fetched (no
3379 # second lookup). Downstream consumers read identity off this instead of
3380 # re-resolving it. Additive and defensive: a projection failure must never
3381 # reject an already-authenticated request, so it is left unset on failure;
3382 # any future consumer must treat a missing principal as deny, not allow.
3383 try:
3384 request.state.principal = _resolve_request_principal(request, user_api_key_auth_obj)
3385 except Exception as e:
3386 verbose_proxy_logger.warning("Principal projection at auth seam failed (non-fatal): %s", e)
3388 return user_api_key_auth_obj
3391async def _return_user_api_key_auth_obj(
3392 user_obj: LiteLLM_UserTable | None,
3393 api_key: str,
3394 parent_otel_span: Span | None,
3395 valid_token_dict: dict,
3396 route: str,
3397 start_time: datetime,
3398 user_role: LitellmUserRoles | None = None,
3399) -> UserAPIKeyAuth:
3400 end_time: Final = datetime.now(timezone.utc)
3402 asyncio.create_task(
3403 user_api_key_service_logger_obj.async_service_success_hook(
3404 service=ServiceTypes.AUTH,
3405 call_type=route,
3406 start_time=start_time,
3407 end_time=end_time,
3408 duration=end_time.timestamp() - start_time.timestamp(),
3409 parent_otel_span=parent_otel_span,
3410 )
3411 )
3413 retrieved_user_role: Final = user_role or _get_user_role(user_obj=user_obj) or LitellmUserRoles.INTERNAL_USER
3415 user_api_key_kwargs: Final = {
3416 "api_key": api_key,
3417 "parent_otel_span": parent_otel_span,
3418 "user_role": retrieved_user_role,
3419 **valid_token_dict,
3420 }
3421 if user_obj is not None:
3422 user_api_key_kwargs.update(
3423 user_tpm_limit=user_obj.tpm_limit,
3424 user_rpm_limit=user_obj.rpm_limit,
3425 user_email=user_obj.user_email,
3426 user_spend=getattr(user_obj, "spend", None),
3427 user_max_budget=getattr(user_obj, "max_budget", None),
3428 user_model_max_budget=getattr(user_obj, "model_max_budget", None),
3429 )
3430 if user_obj is not None and _is_user_proxy_admin(user_obj=user_obj): 3430 ↛ 3431line 3430 didn't jump to line 3431 because the condition on line 3430 was never true
3431 user_api_key_kwargs.update(
3432 user_role=LitellmUserRoles.PROXY_ADMIN,
3433 )
3434 return UserAPIKeyAuth.model_validate(user_api_key_kwargs)
3435 else:
3436 return UserAPIKeyAuth.model_validate(user_api_key_kwargs)
3439def get_api_key_from_custom_header(request: Request, custom_litellm_key_header_name: str) -> str:
3440 """
3441 Get API key from custom header
3443 Args:
3444 request (Request): Request object
3445 custom_litellm_key_header_name (str): Custom header name
3447 Returns:
3448 Optional[str]: API key
3449 """
3450 api_key: str = ""
3451 # use this as the virtual key passed to litellm proxy
3452 custom_litellm_key_header_name = custom_litellm_key_header_name.lower()
3453 _headers: Final = {k.lower(): v for k, v in request.headers.items()}
3454 verbose_proxy_logger.debug(
3455 "searching for custom_litellm_key_header_name= %s, in headers=%s",
3456 custom_litellm_key_header_name,
3457 _headers,
3458 )
3459 custom_api_key: Final = _headers.get(custom_litellm_key_header_name)
3460 if custom_api_key:
3461 api_key = _get_bearer_token(api_key=custom_api_key)
3462 verbose_proxy_logger.debug(
3463 "Found custom API key using header: %s, setting api_key=%s",
3464 custom_litellm_key_header_name,
3465 abbreviate_api_key(api_key),
3466 )
3467 else:
3468 verbose_proxy_logger.exception(
3469 "No LiteLLM Virtual Key pass. Please set header=%s: Bearer <api_key>", custom_litellm_key_header_name
3470 )
3471 return api_key
3474def _get_temp_budget_increase(valid_token: UserAPIKeyAuth):
3475 valid_token_metadata: Final = valid_token.metadata
3476 if "temp_budget_increase" in valid_token_metadata and "temp_budget_expiry" in valid_token_metadata:
3477 expiry = datetime.fromisoformat(valid_token_metadata["temp_budget_expiry"])
3478 if expiry.tzinfo is None:
3479 expiry = expiry.replace(tzinfo=timezone.utc)
3480 if expiry > datetime.now(timezone.utc):
3481 return valid_token_metadata["temp_budget_increase"]
3482 return None
3485def _update_key_budget_with_temp_budget_increase(
3486 valid_token: UserAPIKeyAuth,
3487) -> UserAPIKeyAuth:
3488 if valid_token.max_budget is None: 3488 ↛ 3490line 3488 didn't jump to line 3490 because the condition on line 3488 was always true
3489 return valid_token
3490 temp_budget_increase: Final = _get_temp_budget_increase(valid_token)
3491 if not temp_budget_increase:
3492 return valid_token
3493 return valid_token.model_copy(update={"max_budget": valid_token.max_budget + temp_budget_increase})
3496async def _lookup_end_user_and_apply_budget(
3497 valid_token: UserAPIKeyAuth,
3498 route: str,
3499 parent_otel_span: Span | None,
3500 prisma_client,
3501 user_api_key_cache,
3502 proxy_logging_obj,
3503):
3504 """Look up end_user from DB and apply budget limits to valid_token."""
3505 end_user_object = None
3506 key_end_user_budget_id: Final = get_key_end_user_budget_id(valid_token.metadata)
3507 try:
3508 end_user_object = await get_end_user_object(
3509 end_user_id=valid_token.end_user_id,
3510 prisma_client=prisma_client,
3511 user_api_key_cache=user_api_key_cache,
3512 parent_otel_span=parent_otel_span,
3513 proxy_logging_obj=proxy_logging_obj,
3514 route=route,
3515 token_end_user_max_budget=valid_token.end_user_max_budget,
3516 key_end_user_budget_id=key_end_user_budget_id,
3517 )
3518 if end_user_object is not None:
3519 end_user_params = {
3520 "end_user_id": valid_token.end_user_id,
3521 "allowed_model_region": end_user_object.allowed_model_region,
3522 }
3523 if end_user_object.litellm_budget_table is not None:
3524 _apply_budget_limits_to_end_user_params(
3525 end_user_params=end_user_params,
3526 budget_info=end_user_object.litellm_budget_table,
3527 end_user_id=valid_token.end_user_id or "",
3528 )
3529 valid_token = update_valid_token_with_end_user_params(
3530 valid_token=valid_token, end_user_params=end_user_params
3531 )
3532 elif key_end_user_budget_id is not None or litellm.max_end_user_budget_id is not None:
3533 default_budget: Final = await resolve_default_end_user_budget(
3534 prisma_client=prisma_client,
3535 user_api_key_cache=user_api_key_cache,
3536 key_end_user_budget_id=key_end_user_budget_id,
3537 parent_otel_span=parent_otel_span,
3538 )
3539 if default_budget is not None:
3540 end_user_params = {"end_user_id": valid_token.end_user_id}
3541 _apply_budget_limits_to_end_user_params(
3542 end_user_params=end_user_params,
3543 budget_info=default_budget,
3544 end_user_id=valid_token.end_user_id or "",
3545 )
3546 valid_token = update_valid_token_with_end_user_params(
3547 valid_token=valid_token, end_user_params=end_user_params
3548 )
3549 if valid_token.end_user_max_budget is None:
3550 valid_token.end_user_max_budget = default_budget.max_budget
3551 except Exception as e:
3552 if isinstance(e, litellm.BudgetExceededError):
3553 raise e
3554 verbose_proxy_logger.debug("Unable to find user in db. Error - %s", e)
3555 return valid_token, end_user_object
3558async def _enforce_key_and_fallback_model_access(
3559 *,
3560 valid_token: UserAPIKeyAuth,
3561 request_data: dict,
3562 route: str,
3563 request: Request | None,
3564 llm_model_list: list | None,
3565 llm_router: Any | None,
3566) -> None:
3567 """
3568 Key-level model allowlist and client fallbacks (same as standard auth).
3569 Not included in common_checks — common_checks enforces team/user/project model access only.
3570 """
3571 await _normalize_claude_model(request_data, valid_token, request, route)
3572 await _resolve_router_settings_model_group_alias(request_data, valid_token, request, route)
3573 config: Final = valid_token.config
3575 if config != {}: 3575 ↛ 3576line 3575 didn't jump to line 3576 because the condition on line 3575 was never true
3576 model_list: Final = config.get("model_list", [])
3577 new_model_list: Final = model_list
3578 verbose_proxy_logger.debug("\n new llm router model list %s", new_model_list)
3579 elif isinstance(valid_token.models, list) and "all-team-models" in valid_token.models: 3579 ↛ 3580line 3579 didn't jump to line 3580 because the condition on line 3579 was never true
3580 pass
3581 else:
3582 model: Final = _get_model_from_request_context(
3583 request_data=request_data,
3584 route=route,
3585 request=request,
3586 llm_router=llm_router,
3587 team_id=valid_token.team_id,
3588 )
3590 if model is not None: 3590 ↛ 3591line 3590 didn't jump to line 3591 because the condition on line 3590 was never true
3591 await can_key_call_model(
3592 model=model,
3593 llm_model_list=llm_model_list,
3594 valid_token=valid_token,
3595 llm_router=llm_router,
3596 )
3598 fallback_names: Final = tuple(
3599 name
3600 for target in iter_request_fallback_targets(request_data)
3601 if (name := _fallback_target_model_name(target)) is not None
3602 )
3604 for _name in dict.fromkeys(fallback_names): # dedupe, preserve order 3604 ↛ 3605line 3604 didn't jump to line 3605 because the loop on line 3604 never started
3605 await can_key_call_model(
3606 model=_name,
3607 llm_model_list=llm_model_list,
3608 valid_token=valid_token,
3609 llm_router=llm_router,
3610 )
3611 await is_valid_fallback_model(
3612 model=_name,
3613 llm_router=llm_router,
3614 user_model=None,
3615 )
3618def _fallback_target_model_name(target: object) -> str | None:
3619 if isinstance(target, str):
3620 return target
3621 if isinstance(target, dict):
3622 model: Final = target.get("model")
3623 if isinstance(model, str):
3624 return model
3625 return None
3628async def _run_post_custom_auth_checks(
3629 valid_token: UserAPIKeyAuth,
3630 request: Request,
3631 request_data: dict,
3632 route: str,
3633 parent_otel_span: Span | None,
3634) -> UserAPIKeyAuth:
3635 from litellm.proxy.proxy_server import (
3636 general_settings,
3637 llm_model_list,
3638 llm_router,
3639 model_max_budget_limiter,
3640 prisma_client,
3641 proxy_logging_obj,
3642 user_api_key_cache,
3643 )
3645 # 1. Look up end_user object from DB if end_user_id is set
3646 end_user_object = None
3647 if valid_token.end_user_id is not None:
3648 valid_token, end_user_object = await _lookup_end_user_and_apply_budget(
3649 valid_token=valid_token,
3650 route=route,
3651 parent_otel_span=parent_otel_span,
3652 prisma_client=prisma_client,
3653 user_api_key_cache=user_api_key_cache,
3654 proxy_logging_obj=proxy_logging_obj,
3655 )
3656 # common_checks() enforces the end-user budget, but the centralized
3657 # gate skips it for custom-auth deployments unless
3658 # custom_auth_run_common_checks is set. Enforce it here on that path
3659 # so an over-budget end user can't keep making requests.
3660 if end_user_object is not None and not general_settings.get("custom_auth_run_common_checks", False):
3661 await _check_end_user_budget(end_user_obj=end_user_object, route=route)
3663 # 2. Check token expiry
3664 if valid_token.expires is not None:
3665 current_time: Final = datetime.now(timezone.utc)
3666 if isinstance(valid_token.expires, datetime):
3667 expiry_time = valid_token.expires
3668 else:
3669 expiry_time = datetime.fromisoformat(valid_token.expires)
3670 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None:
3671 expiry_time = expiry_time.replace(tzinfo=timezone.utc)
3672 if expiry_time < current_time:
3673 raise ProxyException(
3674 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}",
3675 type=ProxyErrorTypes.expired_key,
3676 code=status.HTTP_401_UNAUTHORIZED,
3677 param=(abbreviate_api_key(api_key=valid_token.token) if valid_token.token else ""),
3678 )
3680 if general_settings.get("custom_auth_run_common_checks", False):
3681 await _enforce_key_and_fallback_model_access(
3682 valid_token=valid_token,
3683 request_data=request_data,
3684 route=route,
3685 request=request,
3686 llm_model_list=llm_model_list,
3687 llm_router=llm_router,
3688 )
3690 current_model = _get_model_from_request_context(
3691 request_data=request_data,
3692 route=route,
3693 request=request,
3694 llm_router=llm_router,
3695 team_id=valid_token.team_id,
3696 )
3697 current_models = _get_model_names_for_budget_checks(model=current_model)
3699 # A zero-cost model cannot move any counter, so refusing it means refusing on
3700 # spend some other model accrued. The JWT and virtual-key paths already skip
3701 # every budget check for these; this path did not, so the same request could
3702 # be refused under custom auth and served under the other two.
3703 skip_budget_checks: Final = (
3704 _is_model_cost_zero(model=current_model, llm_router=llm_router)
3705 if current_model is not None and llm_router is not None
3706 else False
3707 )
3709 # 3. Check key-level model_max_budget
3710 max_budget_per_model: Final = valid_token.model_max_budget
3711 if (
3712 not skip_budget_checks
3713 and max_budget_per_model is not None
3714 and isinstance(max_budget_per_model, dict)
3715 and len(max_budget_per_model) > 0
3716 and current_models
3717 and valid_token.token is not None
3718 ):
3719 for model_name in current_models:
3720 await _check_key_model_budget_with_fallback(
3721 valid_token=valid_token,
3722 model_max_budget_limiter=model_max_budget_limiter,
3723 model_name=model_name,
3724 request_data=request_data,
3725 request=request,
3726 llm_model_list=llm_model_list,
3727 llm_router=llm_router,
3728 )
3730 # Recompute after a potential budget-fallback rewrite so
3731 # the end-user check below validates the final model
3732 current_model = _get_model_from_request_context(
3733 request_data=request_data,
3734 route=route,
3735 request=request,
3736 llm_router=llm_router,
3737 team_id=valid_token.team_id,
3738 )
3739 current_models = _get_model_names_for_budget_checks(model=current_model)
3741 # 3b. Attach and check the internal user's model_max_budget.
3742 # Custom auth builds its own token, so unlike the main path nothing has
3743 # loaded the user row yet. The attach is unconditional because the post-call
3744 # spend hook reads this field off the token: gating it on the same condition
3745 # as enforcement would leave the user's counter uncharged whenever this
3746 # request was not itself enforceable, so its spend would go untracked.
3747 user_budget: Final = await _read_user_model_max_budget(
3748 user_id=valid_token.user_id,
3749 prisma_client=prisma_client,
3750 user_api_key_cache=user_api_key_cache,
3751 parent_otel_span=parent_otel_span,
3752 proxy_logging_obj=proxy_logging_obj,
3753 )
3754 valid_token.user_model_max_budget = user_budget # rebind-ok: the spend hook reads it off this token
3755 if not skip_budget_checks and current_models:
3756 await _check_user_model_budget(
3757 valid_token=valid_token,
3758 model_max_budget_limiter=model_max_budget_limiter,
3759 models=current_models,
3760 )
3762 # 4. Check end-user model_max_budget
3763 end_user_mmb: Final = valid_token.end_user_model_max_budget
3764 if (
3765 not skip_budget_checks
3766 and end_user_mmb is not None
3767 and isinstance(end_user_mmb, dict)
3768 and len(end_user_mmb) > 0
3769 and current_models
3770 and valid_token.end_user_id is not None
3771 ):
3772 for model_name in current_models:
3773 await model_max_budget_limiter.is_end_user_within_model_budget(
3774 end_user_id=valid_token.end_user_id,
3775 end_user_model_max_budget=end_user_mmb,
3776 model=model_name,
3777 )
3779 # team / user / end_user / project context objects are fetched by
3780 # the centralized common_checks gate in user_api_key_auth after
3781 # this helper returns. Keep only the project fetch here because it
3782 # mutates the token (project_metadata / project_alias).
3783 if valid_token.project_id is not None:
3784 _project_obj: Final = await get_project_object(
3785 project_id=valid_token.project_id,
3786 prisma_client=prisma_client,
3787 user_api_key_cache=user_api_key_cache,
3788 proxy_logging_obj=proxy_logging_obj,
3789 )
3790 if _project_obj is not None:
3791 valid_token.project_metadata = _project_obj.metadata
3792 valid_token.project_alias = _project_obj.project_alias
3794 return valid_token