Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/auth/user_api_key_auth.py: 41%

1247 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2This file handles authentication for the LiteLLM Proxy. 

3 

4it checks if the user passed a valid API Key to the LiteLLM Proxy 

5 

6Returns a UserAPIKeyAuth object if the API key is valid 

7 

8""" 

9 

10import asyncio 

11import fnmatch 

12import re 

13import secrets 

14from collections.abc import Mapping 

15from datetime import datetime, timezone 

16from typing import Any, Final, NamedTuple, Protocol, Union, cast 

17 

18import fastapi 

19import orjson 

20from fastapi import HTTPException, Request, WebSocket, status 

21from fastapi.security.api_key import APIKeyHeader 

22from starlette.exceptions import WebSocketException 

23 

24import litellm 

25from litellm._logging import verbose_logger, verbose_proxy_logger 

26from litellm._service_logger import ServiceLogging 

27from litellm.caching.redis_cache import RedisCache 

28from litellm.constants import ( 

29 CLIENT_REQUESTED_MODEL_SCOPE_KEY, 

30 GLOBAL_PROXY_SPEND_CACHE_KEY, 

31 INVALID_VIRTUAL_KEY_ERROR_MARKER, 

32 INVALID_VIRTUAL_KEY_ERROR_MESSAGE, 

33 LITELLM_PROXY_BUDGET_NAME, 

34 LITELLM_PROXY_MASTER_KEY_ALIAS, 

35 MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY, 

36) 

37from litellm.integrations.otel.model.config import is_otel_v2_enabled 

38from litellm.integrations.otel.runtime import phase_span, seed_request_identity 

39from litellm.litellm_core_utils.dd_tracing import tracer 

40from litellm.litellm_core_utils.dot_notation_indexing import get_nested_value 

41from litellm.proxy._types import * 

42from litellm.proxy.agent_endpoints.auth.agent_caller import agent_caller_from_headers 

43from litellm.proxy.auth.auth_checks import ( 

44 ExperimentalUIJWTToken, 

45 TeamNotFoundError, 

46 _cache_key_object, 

47 _can_object_call_model, 

48 _check_end_user_budget, 

49 _delete_cache_key_object, 

50 _get_user_role, 

51 _is_model_cost_zero, 

52 _is_user_proxy_admin, 

53 _virtual_key_max_budget_alert_check, 

54 _virtual_key_max_budget_check, 

55 _virtual_key_soft_budget_check, 

56 can_key_call_model, 

57 common_checks, 

58 get_end_user_object, 

59 get_jwt_key_mapping_object, 

60 get_key_end_user_budget_id, 

61 get_object_permission, 

62 get_org_object_for_request, 

63 get_project_object, 

64 get_team_membership, 

65 get_team_object, 

66 get_user_object, 

67 is_valid_fallback_model, 

68 jwt_key_mapping_cache_key, 

69 key_model_aliases_for_auth_check, 

70 resolve_and_validate_end_user_id, 

71 resolve_default_end_user_budget, 

72) 

73from litellm.proxy.auth.auth_exception_handler import UserAPIKeyAuthExceptionHandler 

74from litellm.proxy.auth.auth_method import AuthMethod 

75from litellm.proxy.auth.auth_object_prefetch import AuthObjectRefs, prefetch_auth_objects 

76from litellm.proxy.auth.auth_utils import ( 

77 abbreviate_api_key, 

78 get_end_user_id_from_request_body, 

79 get_model_from_request, 

80 get_request_route, 

81 get_request_route_template, 

82 is_invalid_virtual_key_error, 

83 iter_request_fallback_targets, 

84 normalize_request_route, 

85 pre_db_read_auth_checks, 

86 request_dispatched_to_pass_through_endpoint, 

87 request_dispatched_to_provider_pass_through, 

88 route_in_additonal_public_routes, 

89) 

90from litellm.proxy.auth.handle_jwt import JWTAuthManager, JWTHandler 

91from litellm.proxy.auth.network import TrustedProxyConfig, resolve_network_context 

92from litellm.proxy.auth.oauth2_check import Oauth2Handler 

93from litellm.proxy.auth.oauth2_proxy_hook import handle_oauth2_proxy_request 

94from litellm.proxy.auth.resolvers import CredentialRef, Principal 

95from litellm.proxy.auth.resolvers.grants import ( 

96 GrantResolver, 

97 LookupDegraded, 

98 ResolvedGrants, 

99 UserLookup, 

100 raise_public, 

101 user_models, 

102) 

103from litellm.proxy.auth.resolvers.store import IdentityStore 

104from litellm.proxy.auth.route_checks import RouteChecks 

105from litellm.proxy.auth.team_grants import team_grants 

106from litellm.proxy.auth.trusted_proxy_utils import get_trusted_proxy_cidrs 

107from litellm.proxy.common_utils.cache_coordinator import EventDrivenCacheCoordinator 

108from litellm.proxy.common_utils.http_parsing_utils import ( 

109 _read_request_body, 

110 _safe_get_request_headers, 

111 _safe_get_request_query_params, 

112 _safe_set_request_parsed_body, 

113 is_opaque_audio_pass_through_request, 

114 populate_request_with_path_params, 

115 read_raw_json_body, 

116 rewrite_request_model, 

117) 

118from litellm.proxy.common_utils.model_listing_utils import claude_code_requested_group 

119from litellm.proxy.common_utils.realtime_utils import _realtime_request_body 

120from litellm.proxy.common_utils.user_api_key_cache import ( 

121 UserApiKeyCache, 

122 team_membership_auth_cache_key, 

123) 

124from litellm.proxy.db.db_lookup_gate import bounded_db_lookup 

125from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler 

126from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup 

127from litellm.proxy.spend_tracking.carried_budget_state import carry_team_and_user_budget_state 

128from litellm.proxy.spend_tracking.spend_counter_batch import ( 

129 bind_admission_counter_keys, 

130 release_spend_counter_batch, 

131 spend_counter_batch_scope, 

132) 

133from litellm.proxy.utils import ( 

134 PrismaClient, 

135 ProxyLogging, 

136 normalize_route_for_root_path, 

137) 

138from litellm.repositories.table_repositories import TeamMembershipRepository 

139from litellm.router_utils.common_utils import resolve_model_group_alias 

140from litellm.secret_managers.main import get_secret_bool 

141from litellm.types.services import ServiceTypes 

142 

143try: 

144 from litellm_enterprise.proxy.auth.user_api_key_auth import ( 

145 enterprise_custom_auth as _enterprise_custom_auth, 

146 ) 

147 

148 enterprise_custom_auth: Callable | None = _enterprise_custom_auth 

149except ImportError as e: 

150 verbose_proxy_logger.debug("Error in enterprise custom auth: %s", e) 

151 enterprise_custom_auth = None 

152 

153user_api_key_service_logger_obj: Final = ServiceLogging() # used for tracking latency on OTEL 

154 

155 

156def _normalize_public_auth_route(route: str) -> str: 

157 if route != "/" and route.endswith("/"): 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true

158 return route.rstrip("/") 

159 return route 

160 

161 

162def _route_requires_auth_despite_public(route: str, general_settings: dict | None) -> bool: 

163 normalized_route: Final = _normalize_public_auth_route(route) 

164 if normalized_route == "/metrics": 164 ↛ 165line 164 didn't jump to line 165 because the condition on line 164 was never true

165 return litellm.require_auth_for_metrics_endpoint is not False 

166 

167 return False 

168 

169 

170custom_litellm_key_header: Final = APIKeyHeader( 

171 name=SpecialHeaders.custom_litellm_api_key.value, 

172 auto_error=False, 

173 description="Bearer token", 

174) 

175api_key_header: Final = APIKeyHeader( 

176 name=SpecialHeaders.openai_authorization.value, 

177 auto_error=False, 

178 description="Bearer token", 

179) 

180azure_api_key_header: Final = APIKeyHeader( 

181 name=SpecialHeaders.azure_authorization.value, 

182 auto_error=False, 

183 description="Some older versions of the openai Python package will send an API-Key header with just the API key ", 

184) 

185anthropic_api_key_header: Final = APIKeyHeader( 

186 name=SpecialHeaders.anthropic_authorization.value, 

187 auto_error=False, 

188 description="If anthropic client used.", 

189) 

190google_ai_studio_api_key_header: Final = APIKeyHeader( 

191 name=SpecialHeaders.google_ai_studio_authorization.value, 

192 auto_error=False, 

193 description="If google ai studio client used.", 

194) 

195azure_apim_header: Final = APIKeyHeader( 

196 name=SpecialHeaders.azure_apim_authorization.value, 

197 auto_error=False, 

198 description="The default name of the subscription key header of Azure", 

199) 

200 

201 

202def _get_model_from_request_context( 

203 request_data: dict, 

204 route: str, 

205 request: Request | None, 

206 llm_router: Any | None = None, 

207 team_id: str | None = None, 

208) -> str | list[str] | None: 

209 return get_model_from_request( 

210 request_data=request_data, 

211 route=route, 

212 request_headers=_safe_get_request_headers(request=request), 

213 request_query_params=_safe_get_request_query_params(request=request), 

214 llm_router=llm_router, 

215 request=request, 

216 team_id=team_id, 

217 ) 

218 

219 

220_CLAUDE_MODEL_ROUTES: Final = frozenset( 

221 f"/{prefix}{endpoint}" for prefix in ("", "v1/") for endpoint in ("messages", "chat/completions", "responses") 

222) 

223_CLAUDE_MODEL_NORMALIZED: Final = "litellm.claude_model_normalized" 

224 

225 

226async def _normalize_claude_model( 

227 request_data: dict, valid_token: UserAPIKeyAuth, request: Request | None, route: str 

228) -> None: 

229 from litellm.proxy.proxy_server import llm_router, prisma_client, proxy_config, proxy_logging_obj 

230 

231 if route not in _CLAUDE_MODEL_ROUTES or llm_router is None: 

232 return 

233 if request is not None and request.scope.get(_CLAUDE_MODEL_NORMALIZED) is True: 233 ↛ 234line 233 didn't jump to line 234 because the condition on line 233 was never true

234 return 

235 requested: Final = _get_model_from_request_context(request_data, route, request, llm_router, valid_token.team_id) 

236 if not isinstance(requested, str) or requested != request_data.get("model"): 

237 return 

238 if not requested.startswith("claude-router-") and not requested.lower().endswith("[1m]"): 238 ↛ 240line 238 didn't jump to line 240 because the condition on line 238 was always true

239 return 

240 settings: Final = await proxy_config.get_hierarchical_router_settings( 

241 user_api_key_dict=valid_token, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj 

242 ) 

243 aliases: Final = settings.get("model_group_alias") if isinstance(settings, Mapping) else None 

244 source: Final = claude_code_requested_group( 

245 requested, llm_router, valid_token.team_id, (valid_token.aliases, valid_token.team_model_aliases, aliases) 

246 ) 

247 if request is not None: 

248 request.scope[_CLAUDE_MODEL_NORMALIZED] = True 

249 if source is None: 

250 return 

251 rewrite_request_model(request_data, request, source) 

252 

253 

254async def _resolve_router_settings_model_group_alias( 

255 request_data: dict[str, object], # mutable-ok: the request body is rewritten in place for every downstream reader 

256 valid_token: UserAPIKeyAuth, 

257 request: Request | None, 

258 route: str, 

259) -> None: 

260 """Rewrite the requested model through the key's or team's ``router_settings.model_group_alias`` 

261 before the allowlist checks, so they authorize the model group the request is routed to. 

262 """ 

263 from litellm.proxy.proxy_server import llm_router, prisma_client, proxy_config, proxy_logging_obj 

264 

265 if request is None or llm_router is None or not RouteChecks.is_llm_api_route(route=route): 

266 return 

267 if request.scope.get(MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY) is True: 

268 return 

269 request.scope[MODEL_GROUP_ALIAS_RESOLVED_SCOPE_KEY] = True 

270 if request_dispatched_to_pass_through_endpoint(request) or request_dispatched_to_provider_pass_through(request): 

271 return 

272 requested: Final = request_data.get("model") 

273 if not isinstance(requested, str) or await read_raw_json_body(request=request) is None: 

274 return 

275 settings: Final = await proxy_config.get_hierarchical_router_settings( 

276 user_api_key_dict=valid_token, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj 

277 ) 

278 if not isinstance(settings, Mapping): 278 ↛ 280line 278 didn't jump to line 280 because the condition on line 278 was always true

279 return 

280 target: Final = resolve_model_group_alias(settings.get("model_group_alias"), requested) 

281 if target is None or target == requested: 

282 return 

283 verbose_proxy_logger.debug( 

284 "router_settings.model_group_alias resolved %s -> %s before auth", 

285 requested.replace("\r", "").replace("\n", ""), 

286 target.replace("\r", "").replace("\n", ""), 

287 ) 

288 request.scope.setdefault(CLIENT_REQUESTED_MODEL_SCOPE_KEY, requested) 

289 rewrite_request_model(request_data, request, target) 

290 

291 

292def _get_model_names_for_budget_checks( 

293 model: str | list[str] | None, 

294) -> list[str]: 

295 if model is None: 

296 return [] 

297 if isinstance(model, str): 

298 return [model] 

299 return model 

300 

301 

302class _KeyModelBudgetLimiter(Protocol): 

303 async def is_key_within_model_budget(self, user_api_key_dict: UserAPIKeyAuth, model: str) -> bool: ... 303 ↛ exitline 303 didn't return from function 'is_key_within_model_budget' because

304 

305 async def get_fallback_model_within_budget(self, user_api_key_dict: UserAPIKeyAuth, model: str) -> str | None: ... 305 ↛ exitline 305 didn't return from function 'get_fallback_model_within_budget' because

306 

307 

308class _UserModelBudgetLimiter(Protocol): 

309 async def is_user_within_model_budget( 309 ↛ exitline 309 didn't return from function 'is_user_within_model_budget' because

310 self, user_id: str, user_model_max_budget: Mapping[str, object], model: str 

311 ) -> bool: ... 

312 

313 

314class _TeamModelBudgetLimiter(Protocol): 

315 async def is_team_within_model_budget( 315 ↛ exitline 315 didn't return from function 'is_team_within_model_budget' because

316 self, 

317 team_id: str, 

318 team_model_max_budget: Mapping[str, object], 

319 key_model_max_budget: Mapping[str, object] | None, 

320 model: str, 

321 ) -> bool: ... 

322 

323 

324class _TokenTeamModels(Protocol): 

325 @property 

326 def team_models(self) -> list[str]: ... 326 ↛ exitline 326 didn't return from function 'team_models' because

327 

328 

329class _RawCacheRead(Protocol): 

330 async def async_get_cache(self, *, key: str) -> object: ... 330 ↛ exitline 330 didn't return from function 'async_get_cache' because

331 

332 

333def _raw_cache(cache: _RawCacheRead) -> _RawCacheRead: 

334 """View an untyped cache object's ``async_get_cache`` as returning ``object`` 

335 instead of ``Any``, so a caller can ``isinstance``-narrow it without paying 

336 the ``reportAny`` cost of the underlying (unannotated) cache implementation.""" 

337 return cache 

338 

339 

340def _token_team_models(valid_token: _TokenTeamModels) -> list[str]: 

341 return valid_token.team_models 

342 

343 

344async def _read_user_model_max_budget( 

345 user_id: str | None, 

346 prisma_client: PrismaClient | None, 

347 user_api_key_cache: UserApiKeyCache, 

348 parent_otel_span: Span | None, 

349 proxy_logging_obj: ProxyLogging, 

350) -> Mapping[str, object] | None: 

351 """The user row's `model_max_budget`, or None when the row cannot be read. 

352 

353 A user whose row is missing must not be refused: this is a budget lookup, 

354 and the main auth path likewise treats an unreadable user as no user. 

355 """ 

356 if user_id is None or prisma_client is None: 

357 return None 

358 try: 

359 user_obj: Final = await get_user_object( 

360 user_id=user_id, 

361 prisma_client=prisma_client, 

362 user_api_key_cache=user_api_key_cache, 

363 user_id_upsert=False, 

364 parent_otel_span=parent_otel_span, 

365 proxy_logging_obj=proxy_logging_obj, 

366 ) 

367 except Exception as e: # noqa: BLE001 # mirrors the main path's tolerance 

368 verbose_logger.debug("Unable to read user for the per-model budget check: %s", e) 

369 return None 

370 return user_obj.model_max_budget if user_obj is not None else None 

371 

372 

373async def _check_user_model_budget( 

374 valid_token: UserAPIKeyAuth, 

375 model_max_budget_limiter: _UserModelBudgetLimiter, 

376 models: list[str], 

377) -> None: 

378 """Enforce the internal user's own `model_max_budget` across the request's models. 

379 

380 Separate from the key check: a user's per-model budget caps every key they 

381 own, so a caller cannot escape it by minting another key. 

382 """ 

383 user_model_max_budget: Final = valid_token.user_model_max_budget 

384 if valid_token.user_id is None or not isinstance(user_model_max_budget, Mapping) or not user_model_max_budget: 

385 return 

386 for model_name in models: 

387 await model_max_budget_limiter.is_user_within_model_budget( 

388 user_id=valid_token.user_id, 

389 user_model_max_budget=user_model_max_budget, 

390 model=model_name, 

391 ) 

392 

393 

394async def _check_team_model_budget( 

395 valid_token: UserAPIKeyAuth, 

396 model_max_budget_limiter: _TeamModelBudgetLimiter, 

397 models: list[str], 

398) -> None: 

399 """Enforce the team's `model_max_budget` for every requested model the key does not override.""" 

400 team_model_max_budget: Final = valid_token.team_model_max_budget 

401 if valid_token.team_id is None or not team_model_max_budget: 401 ↛ 403line 401 didn't jump to line 403 because the condition on line 401 was always true

402 return 

403 key_model_max_budget: Final[Mapping[str, object] | None] = valid_token.model_max_budget 

404 for model_name in models: 

405 await model_max_budget_limiter.is_team_within_model_budget( 

406 team_id=valid_token.team_id, 

407 team_model_max_budget=team_model_max_budget, 

408 key_model_max_budget=key_model_max_budget, 

409 model=model_name, 

410 ) 

411 

412 

413async def _check_key_model_budget_with_fallback( 

414 valid_token: UserAPIKeyAuth, 

415 model_max_budget_limiter: _KeyModelBudgetLimiter, 

416 model_name: str, 

417 request_data: dict, 

418 request: Request, 

419 llm_model_list: list | None = None, 

420 llm_router: litellm.Router | None = None, 

421) -> None: 

422 """ 

423 Enforce the key's per-model budget for `model_name`. If exceeded and the 

424 key has a `budget_fallbacks` chain configured for `model_name`, reroute 

425 the request to the first fallback model still within its own budget 

426 instead of rejecting the request. 

427 

428 The selected fallback is validated against the key's model-access 

429 allowlist and the team's model restrictions so that budget_fallbacks 

430 cannot bypass model authorization. The rewrite is persisted to the 

431 parsed-body cache, Starlette's JSON cache (``request._json``), and 

432 path parameters so that downstream handlers see the final model 

433 regardless of whether they consume ``_read_request_body()``, 

434 ``request.json()``, or the path ``model`` parameter. 

435 

436 Fallback is only attempted when ``model_name`` matches the top-level 

437 ``request_data["model"]``; models extracted from nested fields 

438 (``session.model``, ``completion.model``, etc.) are not rewritable 

439 and raise immediately. 

440 

441 Raises: 

442 BudgetExceededError: if `model_name` is over budget and no configured 

443 fallback is within budget either (or the fallback is not authorized). 

444 """ 

445 try: 

446 await model_max_budget_limiter.is_key_within_model_budget( 

447 user_api_key_dict=valid_token, 

448 model=model_name, 

449 ) 

450 except litellm.BudgetExceededError as e: 

451 if request_data.get("model") != model_name: 

452 raise e 

453 fallback_model: Final = await model_max_budget_limiter.get_fallback_model_within_budget( 

454 user_api_key_dict=valid_token, 

455 model=model_name, 

456 ) 

457 if fallback_model is None: 

458 raise e 

459 try: 

460 await can_key_call_model( 

461 model=fallback_model, 

462 llm_model_list=llm_model_list, 

463 valid_token=valid_token, 

464 llm_router=llm_router, 

465 ) 

466 if valid_token.team_models: 

467 _can_object_call_model( 

468 model=fallback_model, 

469 llm_router=llm_router, 

470 models=valid_token.team_models, 

471 team_model_aliases=valid_token.team_model_aliases, 

472 team_id=valid_token.team_id, 

473 key_model_aliases=key_model_aliases_for_auth_check(valid_token), 

474 object_type="team", 

475 ) 

476 except ProxyException: 

477 raise e 

478 request_data["model"] = fallback_model 

479 _safe_set_request_parsed_body(request=request, parsed_body=request_data) 

480 request._json = request_data 

481 request._body = orjson.dumps(request_data) 

482 path_params: Final = request.scope.get("path_params") 

483 if isinstance(path_params, dict) and "model" in path_params: 

484 path_params["model"] = fallback_model 

485 

486 

487def _get_bearer_token_or_received_api_key(api_key: str) -> str: 

488 if api_key.startswith("Bearer "): # ensure Bearer token passed in 488 ↛ 489line 488 didn't jump to line 489 because the condition on line 488 was never true

489 api_key = api_key.replace("Bearer ", "") # extract the token 

490 elif api_key.startswith("Basic "): 490 ↛ 491line 490 didn't jump to line 491 because the condition on line 490 was never true

491 api_key = api_key.replace("Basic ", "") # handle langfuse input 

492 elif api_key.startswith("bearer "): 492 ↛ 493line 492 didn't jump to line 493 because the condition on line 492 was never true

493 api_key = api_key.replace("bearer ", "") 

494 elif api_key.startswith("AWS4-HMAC-SHA256"): 494 ↛ 498line 494 didn't jump to line 498 because the condition on line 494 was never true

495 # Handle AWS Signature V4 format from LangChain 

496 # Format: AWS4-HMAC-SHA256 Credential=Bearer sk-12345/date/region/service/aws4_request, SignedHeaders=..., Signature=... 

497 # Extract the Bearer token from the Credential field 

498 match = re.search(r"Credential=Bearer\s+([^/\s,]+)", api_key) 

499 if match: 

500 api_key = match.group(1) 

501 else: 

502 # If no Bearer token found in Credential, try to extract just the credential value 

503 match = re.search(r"Credential=([^/\s,]+)", api_key) 

504 if match: 

505 api_key = match.group(1) 

506 

507 return api_key 

508 

509 

510def _routing_selector_matches_claim( 

511 selector_value: Any | None, 

512 claim_value: Any | None, 

513 *, 

514 split_space_delimited: bool = False, 

515) -> bool: 

516 if selector_value is None: 

517 return True 

518 

519 selector_list: Final[list[str]] = ( 

520 [str(v) for v in selector_value] if isinstance(selector_value, list) else [str(selector_value)] 

521 ) 

522 

523 if claim_value is None: 

524 return False 

525 

526 if isinstance(claim_value, list): 

527 claim_list = [str(v) for v in claim_value] 

528 elif split_space_delimited and isinstance(claim_value, str) and " " in claim_value.strip(): 

529 # OAuth/OIDC often sends scope as a single space-delimited string. Only split 

530 # for the scope selector: iss/aud/client_id must stay exact full-string match 

531 # on unverified claims (see routing override security review). The elif guard 

532 # (`" " in claim_value.strip()`) ensures at least two non-empty tokens survive. 

533 claim_list = [v for v in claim_value.strip().split(" ") if v] 

534 else: 

535 claim_list = [str(claim_value)] 

536 

537 def _selector_matches_claim(selector: str, claim: str) -> bool: 

538 # NOTE: wildcard matching is case-sensitive (fnmatch.fnmatchcase). 

539 if "*" in selector or "?" in selector: 

540 # Without scope splitting, do not let `*` span whitespace: a malformed 

541 # iss like "trusted.example.com evil.com" must not match "trusted.*". 

542 # Scope uses split_space_delimited so each claim token is checked separately. 

543 if not split_space_delimited and any(ch.isspace() for ch in claim): 

544 return False 

545 return fnmatch.fnmatchcase(claim, selector) 

546 return selector == claim 

547 

548 return any(_selector_matches_claim(selector=s, claim=c) for s in selector_list for c in claim_list) 

549 

550 

551def _matches_routing_override(token_claims: dict, override: "JWTRoutingOverride") -> bool: 

552 return ( 

553 _routing_selector_matches_claim(override.iss, token_claims.get("iss")) 

554 and _routing_selector_matches_claim(override.client_id, token_claims.get("client_id")) 

555 and _routing_selector_matches_claim( 

556 override.scope, 

557 token_claims.get("scope"), 

558 split_space_delimited=True, 

559 ) 

560 and _routing_selector_matches_claim(override.aud, token_claims.get("aud")) 

561 ) 

562 

563 

564def _should_route_jwt_to_oauth2_override(token: str, jwt_handler: JWTHandler) -> bool: 

565 routing_overrides: Final = jwt_handler.litellm_jwtauth.routing_overrides 

566 if not routing_overrides: 

567 return False 

568 

569 token_claims: Final = jwt_handler.get_unverified_claims(token=token) 

570 if token_claims is None: 

571 return False 

572 

573 for override in routing_overrides: 

574 if override.path == "oauth2" and _matches_routing_override(token_claims=token_claims, override=override): 

575 verbose_proxy_logger.debug("JWT routing override matched. Routing token to OAuth2 introspection.") 

576 return True 

577 

578 return False 

579 

580 

581def _get_bearer_token( 

582 api_key: str, 

583): 

584 if api_key.startswith("Bearer "): # ensure Bearer token passed in 

585 api_key = api_key.replace("Bearer ", "") # extract the token 

586 elif api_key.startswith("Basic "): 586 ↛ 587line 586 didn't jump to line 587 because the condition on line 586 was never true

587 api_key = api_key.replace("Basic ", "") # handle langfuse input 

588 elif api_key.startswith("bearer "): 588 ↛ 589line 588 didn't jump to line 589 because the condition on line 588 was never true

589 api_key = api_key.replace("bearer ", "") 

590 elif api_key.startswith("AWS4-HMAC-SHA256"): 590 ↛ 594line 590 didn't jump to line 594 because the condition on line 590 was never true

591 # Handle AWS Signature V4 format from LangChain 

592 # Format: AWS4-HMAC-SHA256 Credential=Bearer sk-12345/date/region/service/aws4_request, SignedHeaders=..., Signature=... 

593 # Extract the Bearer token from the Credential field 

594 match = re.search(r"Credential=Bearer\s+([^/\s,]+)", api_key) 

595 if match: 

596 api_key = match.group(1) 

597 else: 

598 # If no Bearer token found in Credential, try to extract just the credential value 

599 match = re.search(r"Credential=([^/\s,]+)", api_key) 

600 if match: 

601 api_key = match.group(1) 

602 else: 

603 api_key = "" 

604 else: 

605 api_key = "" 

606 return api_key 

607 

608 

609def _apply_budget_limits_to_end_user_params( 

610 end_user_params: dict, 

611 budget_info: LiteLLM_BudgetTable, 

612 end_user_id: str | None, 

613) -> None: 

614 """ 

615 Helper function to apply budget limits to end user parameters. 

616 

617 Args: 

618 end_user_params: Dictionary to update with budget parameters 

619 budget_info: Budget table object containing limits 

620 end_user_id: ID of the end user for logging 

621 """ 

622 if budget_info.tpm_limit is not None: 

623 end_user_params["end_user_tpm_limit"] = budget_info.tpm_limit 

624 

625 if budget_info.rpm_limit is not None: 

626 end_user_params["end_user_rpm_limit"] = budget_info.rpm_limit 

627 

628 if budget_info.tpd_limit is not None: 

629 end_user_params["end_user_tpd_limit"] = budget_info.tpd_limit 

630 

631 if budget_info.max_budget is not None: 

632 end_user_params["end_user_max_budget"] = budget_info.max_budget 

633 

634 if budget_info.model_max_budget is not None: 

635 end_user_params["end_user_model_max_budget"] = budget_info.model_max_budget 

636 

637 verbose_proxy_logger.debug("Applied budget limits to end user %s", end_user_id) 

638 

639 

640async def user_api_key_auth_websocket(websocket: WebSocket) -> UserAPIKeyAuth: 

641 return await user_api_key_auth_websocket_for_model(websocket, model=websocket.query_params.get("model")) 

642 

643 

644async def user_api_key_auth_websocket_for_model(websocket: WebSocket, model: str | None) -> UserAPIKeyAuth: 

645 ws_scope: Final = websocket.scope or {} 

646 scope_headers: Final = list(ws_scope.get("headers") or []) 

647 # ``get_request_route`` falls back to ``request.url.path`` when 

648 # ``scope["path"]`` is absent. On WebSockets that fallback reads 

649 # ``websocket.url``, which Starlette reconstructs from the (poisonable) 

650 # Host header. Carry the ASGI scope's path / root_path so the lookup 

651 # never reaches the fallback. 

652 synthetic_scope: Final[dict[str, Any]] = { 

653 "type": "http", 

654 "headers": scope_headers, 

655 "path": ws_scope.get("path", ""), 

656 "state": ws_scope.setdefault("state", {}), # mutable-ok: Starlette's socket state, shared with the request 

657 } 

658 for key in ("root_path", "app_root_path"): 

659 if key in ws_scope: 

660 synthetic_scope[key] = ws_scope[key] 

661 request: Final = Request(scope=synthetic_scope) 

662 

663 request._url = websocket.url 

664 

665 async def return_body(): 

666 return _realtime_request_body(model) 

667 

668 request.body = return_body 

669 

670 authorization: Final = websocket.headers.get("authorization") 

671 # If no Authorization header, try the api-key header 

672 if not authorization: 

673 api_key = websocket.headers.get("api-key") 

674 if not api_key: 

675 # Try extracting from WebSocket subprotocol (browser clients) 

676 for protocol in websocket.headers.get("sec-websocket-protocol", "").split(","): 

677 protocol = protocol.strip() 

678 if protocol.startswith("openai-insecure-api-key."): 

679 api_key = protocol[len("openai-insecure-api-key.") :] 

680 break 

681 if not api_key: 

682 await websocket.close(code=status.WS_1008_POLICY_VIOLATION) 

683 raise HTTPException(status_code=403, detail="No API key provided") 

684 else: 

685 # Extract the API key from the Bearer token 

686 if not authorization.startswith("Bearer "): 

687 await websocket.close(code=status.WS_1008_POLICY_VIOLATION) 

688 raise HTTPException(status_code=403, detail="Invalid Authorization header format") 

689 

690 api_key = authorization[len("Bearer ") :].strip() 

691 

692 # Call user_api_key_auth with the extracted API key 

693 # Note: You'll need to modify this to work with WebSocket context if needed 

694 try: 

695 return await user_api_key_auth(request=request, api_key=f"Bearer {api_key}") 

696 except Exception as e: 

697 if is_invalid_virtual_key_error(e): 

698 raise WebSocketException(code=status.WS_1008_POLICY_VIOLATION) 

699 verbose_proxy_logger.exception(e) 

700 await websocket.close(code=status.WS_1008_POLICY_VIOLATION) 

701 raise HTTPException(status_code=403, detail=str(e)) 

702 

703 

704def update_valid_token_with_end_user_params(valid_token: UserAPIKeyAuth, end_user_params: dict) -> UserAPIKeyAuth: 

705 valid_token.end_user_id = end_user_params.get("end_user_id") 

706 # Only overwrite token fields when the DB-derived value is not None. 

707 # This prevents DB lookups (where the budget table has no value set) 

708 # from silently clearing values that a custom auth function may have 

709 # already set on the token. 

710 if end_user_params.get("end_user_tpm_limit") is not None: 710 ↛ 711line 710 didn't jump to line 711 because the condition on line 710 was never true

711 valid_token.end_user_tpm_limit = end_user_params["end_user_tpm_limit"] 

712 if end_user_params.get("end_user_rpm_limit") is not None: 712 ↛ 713line 712 didn't jump to line 713 because the condition on line 712 was never true

713 valid_token.end_user_rpm_limit = end_user_params["end_user_rpm_limit"] 

714 if end_user_params.get("end_user_tpd_limit") is not None: 714 ↛ 715line 714 didn't jump to line 715 because the condition on line 714 was never true

715 valid_token.end_user_tpd_limit = end_user_params["end_user_tpd_limit"] 

716 if end_user_params.get("allowed_model_region") is not None: 716 ↛ 717line 716 didn't jump to line 717 because the condition on line 716 was never true

717 valid_token.allowed_model_region = end_user_params["allowed_model_region"] 

718 if end_user_params.get("end_user_model_max_budget") is not None: 718 ↛ 719line 718 didn't jump to line 719 because the condition on line 718 was never true

719 valid_token.end_user_model_max_budget = end_user_params["end_user_model_max_budget"] 

720 return valid_token 

721 

722 

723# Reusable coordinator for global spend to prevent cache stampede 

724_global_spend_coordinator: Final = EventDrivenCacheCoordinator(log_prefix="[GLOBAL SPEND]") 

725 

726 

727async def _fetch_global_spend_with_event_coordination( 

728 cache_key: str, 

729 user_api_key_cache: UserApiKeyCache, 

730 prisma_client: PrismaClient, 

731) -> float | None: 

732 """ 

733 Fetch global spend with event-driven coordination to prevent cache stampede. 

734 Uses EventDrivenCacheCoordinator: first request queries DB and signals others when done. 

735 

736 Reads the proxy budget aggregate user row, which accrues proxy-wide spend 

737 per request and is zeroed by ResetBudgetJob every ``litellm.budget_duration``. 

738 """ 

739 

740 async def _load_global_spend() -> float | None: 

741 proxy_budget_row: Final = await bounded_db_lookup( 

742 prisma_client.db.litellm_usertable.find_unique(where={"user_id": LITELLM_PROXY_BUDGET_NAME}), 

743 name="proxy_budget", 

744 ) 

745 return float(proxy_budget_row.spend) if proxy_budget_row is not None else None 

746 

747 return await _global_spend_coordinator.get_or_load( 

748 cache_key=cache_key, 

749 cache=user_api_key_cache, # pyright: ignore[reportArgumentType] 

750 load_fn=_load_global_spend, 

751 ) 

752 

753 

754async def get_global_proxy_spend( 

755 litellm_proxy_admin_name: str, 

756 user_api_key_cache: UserApiKeyCache, 

757 prisma_client: PrismaClient | None, 

758 token: str, 

759 proxy_logging_obj: ProxyLogging, 

760) -> float | None: 

761 global_proxy_spend = None 

762 if litellm.max_budget > 0 and prisma_client is not None: # user set proxy max budget 762 ↛ 764line 762 didn't jump to line 764 because the condition on line 762 was never true

763 # Use event-driven coordination to prevent cache stampede 

764 cache_key: Final = GLOBAL_PROXY_SPEND_CACHE_KEY 

765 global_proxy_spend = await _fetch_global_spend_with_event_coordination( 

766 cache_key=cache_key, 

767 user_api_key_cache=user_api_key_cache, 

768 prisma_client=prisma_client, 

769 ) 

770 if global_proxy_spend is not None: 

771 user_info: Final = CallInfo( 

772 user_id=litellm_proxy_admin_name, 

773 max_budget=litellm.max_budget, 

774 spend=global_proxy_spend, 

775 token=token, 

776 event_group=Litellm_EntityType.PROXY, 

777 ) 

778 asyncio.create_task( 

779 proxy_logging_obj.budget_alerts( 

780 type="proxy_budget", 

781 user_info=user_info, 

782 ) 

783 ) 

784 return global_proxy_spend 

785 

786 

787def get_rbac_role(jwt_handler: JWTHandler, scopes: list[str]) -> str: 

788 is_admin: Final = jwt_handler.is_admin(scopes=scopes) 

789 if is_admin: 

790 return LitellmUserRoles.PROXY_ADMIN 

791 else: 

792 return LitellmUserRoles.TEAM 

793 

794 

795def get_api_key( 

796 custom_litellm_key_header: str | None, 

797 api_key: str, 

798 azure_api_key_header: str | None, 

799 anthropic_api_key_header: str | None, 

800 google_ai_studio_api_key_header: str | None, 

801 azure_apim_header: str | None, 

802 pass_through_endpoints: list[dict] | None, 

803 route: str, 

804 request: Request, 

805) -> tuple[str, str | None]: 

806 """ 

807 Returns: 

808 Tuple[Optional[str], Optional[str]]: Tuple of the api_key and the passed_in_key 

809 """ 

810 from litellm.proxy.auth.route_checks import RouteChecks 

811 from litellm.proxy.common_utils.http_parsing_utils import ( 

812 _safe_get_request_query_params, 

813 ) 

814 

815 api_key = api_key 

816 passed_in_key: str | None = None 

817 if isinstance(custom_litellm_key_header, str): 

818 passed_in_key = custom_litellm_key_header 

819 api_key = _get_bearer_token_or_received_api_key(custom_litellm_key_header) 

820 elif isinstance(api_key, str) and len(api_key) > 0: 

821 passed_in_key = api_key 

822 api_key = _get_bearer_token(api_key=api_key) 

823 elif isinstance(azure_api_key_header, str): 823 ↛ 824line 823 didn't jump to line 824 because the condition on line 823 was never true

824 passed_in_key = azure_api_key_header 

825 api_key = azure_api_key_header 

826 elif isinstance(anthropic_api_key_header, str): 826 ↛ 827line 826 didn't jump to line 827 because the condition on line 826 was never true

827 passed_in_key = anthropic_api_key_header 

828 api_key = anthropic_api_key_header 

829 elif isinstance(google_ai_studio_api_key_header, str): 829 ↛ 830line 829 didn't jump to line 830 because the condition on line 829 was never true

830 passed_in_key = google_ai_studio_api_key_header 

831 api_key = google_ai_studio_api_key_header 

832 elif isinstance(azure_apim_header, str): 832 ↛ 833line 832 didn't jump to line 833 because the condition on line 832 was never true

833 passed_in_key = azure_apim_header 

834 api_key = azure_apim_header 

835 elif ( 835 ↛ 840line 835 didn't jump to line 840 because the condition on line 835 was never true

836 RouteChecks.is_generate_content_route(route=route) 

837 and request is not None 

838 and _safe_get_request_query_params(request).get("key") 

839 ): 

840 google_auth_key: Final[str] = _safe_get_request_query_params(request).get("key") or "" 

841 passed_in_key = google_auth_key 

842 api_key = google_auth_key 

843 elif pass_through_endpoints is not None: 

844 for endpoint in pass_through_endpoints: 

845 if endpoint.get("path", "") == route: 845 ↛ 846line 845 didn't jump to line 846 because the condition on line 845 was never true

846 headers: dict | None = endpoint.get("headers", None) 

847 if headers is not None: 

848 header_key: str = headers.get("litellm_user_api_key", "") 

849 if request.headers.get(header_key) is not None: 

850 api_key = request.headers.get(header_key) or "" 

851 passed_in_key = api_key 

852 return api_key, passed_in_key 

853 

854 

855async def check_api_key_for_custom_headers_or_pass_through_endpoints( 

856 request: Request, 

857 route: str, 

858 pass_through_endpoints: list[dict] | None, 

859 api_key: str, 

860) -> UserAPIKeyAuth | str: 

861 is_mapped_pass_through_route: bool = False 

862 normalized_route: Final = normalize_route_for_root_path(route) 

863 if normalized_route is not None: 863 ↛ 868line 863 didn't jump to line 868 because the condition on line 863 was always true

864 for mapped_route in LiteLLMRoutes.mapped_pass_through_routes.value: 

865 if normalized_route.startswith(mapped_route): 

866 is_mapped_pass_through_route = True 

867 break 

868 if is_mapped_pass_through_route: 

869 if request.headers.get("litellm_user_api_key") is not None: 869 ↛ 870line 869 didn't jump to line 870 because the condition on line 869 was never true

870 api_key = request.headers.get("litellm_user_api_key") or "" 

871 if pass_through_endpoints is not None: 

872 for endpoint in pass_through_endpoints: 

873 if isinstance(endpoint, dict) and endpoint.get("path", "") == route: 873 ↛ 880line 873 didn't jump to line 880 because the condition on line 873 was never true

874 ## IF AUTH DISABLED 

875 # Default to True: a config dict with no ``auth`` key 

876 # otherwise produced an unauthenticated forwarder. The 

877 # Pydantic ``PassThroughGenericEndpoint.auth`` default 

878 # is also True, but raw config dicts skip that path — 

879 # so this runtime check has to default to True too. 

880 if endpoint.get("auth", True) is not True: 

881 return UserAPIKeyAuth() 

882 ## IF AUTH ENABLED 

883 ### IF CUSTOM PARSER REQUIRED 

884 if endpoint.get("custom_auth_parser") is not None and endpoint.get("custom_auth_parser") == "langfuse": 

885 # langfuse returns {'Authorization': 'Basic <base64(username:password)>'} 

886 # check the langfuse public key if it contains the litellm api key 

887 import base64 

888 

889 api_key = api_key.replace("Basic ", "").strip() 

890 decoded_bytes = base64.b64decode(api_key) 

891 decoded_str = decoded_bytes.decode("utf-8") 

892 api_key = decoded_str.split(":")[0] 

893 else: 

894 headers = endpoint.get("headers", None) 

895 if headers is not None: 

896 header_key = headers.get("litellm_user_api_key", "") 

897 if isinstance(request.headers, dict) and request.headers.get(key=header_key) is not None: 

898 api_key = request.headers.get(key=header_key) 

899 return api_key 

900 

901 

902# Cache sentinel written when a JWT under AUTO_REGISTER resolved to a proxy 

903# admin via auth_builder. Proxy admins don't need a mapped virtual key (they 

904# have full access via auth_builder anyway), but without a cache entry every 

905# subsequent request from the same JWT identity would re-query the DB for a 

906# non-existent mapping. Sentinel tells _resolve_jwt_to_virtual_key to skip 

907# the lookup and return None (caller proceeds to auth_builder). 

908_JWT_PROXY_ADMIN_SENTINEL: Final = "__JWT_PROXY_ADMIN__" 

909 

910_JWT_AUTH_DISABLED_HINT = ( 

911 " This key has the structure of a JWT, but JWT auth is not enabled on this proxy, so it was treated as a" 

912 " virtual key. Set `enable_jwt_auth: true` under `general_settings` in your proxy config to authenticate" 

913 " with JWTs." 

914) 

915 

916 

917class _PendingAutoRegister(NamedTuple): 

918 """ 

919 Signal returned by ``_resolve_jwt_to_virtual_key`` when the JWT's claim is 

920 unmapped and ``unregistered_jwt_client_behavior`` is AUTO_REGISTER. 

921 

922 The caller MUST run standard ``JWTAuthManager.auth_builder`` to apply RBAC, 

923 scope mappings, ``custom_validate``, and ``user_allowed_email_domain`` 

924 policy BEFORE calling ``_auto_register_jwt_mapping`` with the validated 

925 ``team_id`` / ``user_id`` from the auth_builder result. Auto-registering 

926 purely on a signature-valid JWT (the old behavior) bypassed every JWT 

927 policy beyond signature verification. 

928 """ 

929 

930 claim_field: str 

931 claim_value: str 

932 cache_key: str 

933 jwt_issuer: str | None = None 

934 

935 

936async def _auto_register_jwt_mapping( 

937 virtual_key_claim_field: str, 

938 claim_value: str, 

939 jwt_handler: JWTHandler, 

940 prisma_client: PrismaClient, 

941 user_api_key_cache: UserApiKeyCache, 

942 parent_otel_span: Span | None, 

943 proxy_logging_obj: ProxyLogging, 

944 cache_key: str, 

945 jwt_issuer: str | None = None, 

946 team_id: str | None = None, 

947 user_id: str | None = None, 

948 org_id: str | None = None, 

949 end_user_id: str | None = None, 

950 agent_id: str | None = None, 

951) -> UserAPIKeyAuth | None: 

952 """ 

953 Auto-register: create a new virtual key + mapping for an unrecognised JWT 

954 claim value. ``team_id`` and ``user_id`` must come from a successful 

955 ``JWTAuthManager.auth_builder`` run — they encode the JWT identity AFTER 

956 RBAC/scope/custom_validate/email-domain policy has been enforced. The key 

957 is stamped with those values so the cached future-request path inherits 

958 the same team/user/org limits the auth_builder path would have applied. 

959 

960 Race safety: if two concurrent requests both reach here simultaneously (both 

961 saw no mapping in the DB), one will win the unique-constraint race on 

962 litellm_jwtkeymapping. The loser catches the conflict, deletes its orphaned 

963 key, fetches the winner's mapping, and proceeds — no error surfaced. 

964 """ 

965 # Inline import required: key_management_endpoints imports user_api_key_auth 

966 # (line 51) so a module-level import here would create a circular dependency. 

967 from litellm.proxy.management_endpoints.key_management_endpoints import ( 

968 generate_key_helper_fn, 

969 ) 

970 

971 # ``table_name="key"`` is required: without it, generate_key_helper_fn 

972 # falls into the user-upsert branch (`table_name is None or "user"`) and 

973 # attempts to insert into LiteLLM_UserTable with user_id=None, which fails 

974 # the NOT NULL @id constraint. Every successful key-creation caller (e.g. 

975 # /key/generate) passes table_name="key" explicitly. 

976 key_data: Final = await generate_key_helper_fn( 

977 llm_router=None, 

978 request_type="key", 

979 table_name="key", 

980 team_id=team_id, 

981 user_id=user_id, 

982 organization_id=org_id, 

983 agent_id=agent_id, 

984 metadata={ 

985 "auto_registered": True, 

986 "jwt_claim_field": virtual_key_claim_field, 

987 "jwt_claim_value": claim_value, 

988 }, 

989 ) 

990 # generate_key_helper_fn returns the plaintext key in "token"; the persisted 

991 # row in LiteLLM_VerificationToken uses its hash, so hash here to get the FK 

992 # value referenced by LiteLLM_JWTKeyMapping.token. 

993 token_hash = hash_token(key_data["token"]) 

994 

995 try: 

996 await prisma_client.db.litellm_jwtkeymapping.create( 

997 data={ 

998 "jwt_issuer": jwt_issuer or "", 

999 "jwt_claim_name": virtual_key_claim_field, 

1000 "jwt_claim_value": claim_value, 

1001 "token": token_hash, 

1002 "created_by": "auto_register", 

1003 "updated_by": "auto_register", 

1004 } 

1005 ) 

1006 except Exception as e: 

1007 error_str: Final = str(e).lower() 

1008 if "unique" in error_str or "p2002" in error_str: 

1009 # A concurrent request won the race. The key generate_key_helper_fn 

1010 # just persisted to LiteLLM_VerificationToken is orphaned — nothing 

1011 # maps to it, but it's a fully valid unrestricted API key sitting in 

1012 # the DB and the cleartext is in memory on this request. Delete it 

1013 # so orphans don't accumulate under sustained concurrency. 

1014 verbose_proxy_logger.debug( 

1015 "JWT Key Mapping (auto_register): unique conflict on create — " 

1016 "deleting orphaned virtual key and fetching winner's mapping for %s='%s'.", 

1017 virtual_key_claim_field, 

1018 claim_value, 

1019 ) 

1020 try: 

1021 await prisma_client.db.litellm_verificationtoken.delete(where={"token": token_hash}) 

1022 except Exception as delete_err: 

1023 # Don't fail the request if cleanup fails — the orphan is 

1024 # unmapped and inert. Log so an operator can prune it later. 

1025 verbose_proxy_logger.warning( 

1026 "JWT Key Mapping (auto_register): failed to delete orphaned key after race: %s", 

1027 delete_err, 

1028 ) 

1029 token_hash = await get_jwt_key_mapping_object( 

1030 jwt_claim_name=virtual_key_claim_field, 

1031 jwt_claim_value=claim_value, 

1032 prisma_client=prisma_client, 

1033 jwt_issuer=jwt_issuer, 

1034 ) 

1035 if token_hash is None: 

1036 # The winner's mapping vanished between the unique-constraint 

1037 # conflict and our re-fetch (concurrent delete). Returning None 

1038 # here would silently fall through to team-based JWT auth — 

1039 # a less-restrictive path than the operator configured. Raise 

1040 # 503 so the caller retries against a stable state instead. 

1041 raise HTTPException( 

1042 status_code=503, 

1043 detail=( 

1044 "JWT Key Mapping: AUTO_REGISTER race resolution failed — " 

1045 "winner's mapping was concurrently removed. Retry the request." 

1046 ), 

1047 ) 

1048 else: 

1049 raise 

1050 

1051 await user_api_key_cache.async_set_cache( 

1052 key=cache_key, 

1053 value=token_hash, 

1054 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl, 

1055 ) 

1056 

1057 verbose_proxy_logger.info( 

1058 "JWT Key Mapping (auto_register): created new virtual key for %s='%s'.", 

1059 virtual_key_claim_field, 

1060 claim_value, 

1061 ) 

1062 

1063 auto_registered_key: Final = IdentityStore.key_from_principal( 

1064 await IdentityStore( 

1065 prisma_client, 

1066 user_api_key_cache, 

1067 parent_otel_span=parent_otel_span, 

1068 proxy_logging_obj=proxy_logging_obj, 

1069 ).resolve(hashed_token=token_hash) 

1070 ) 

1071 if auto_registered_key is not None: 

1072 auto_registered_key.org_id = org_id 

1073 auto_registered_key.end_user_id = end_user_id 

1074 auto_registered_key.api_key = auto_registered_key.token 

1075 return auto_registered_key 

1076 

1077 

1078async def _lookup_jwt_mapping_token_hash( 

1079 prisma_client: PrismaClient, 

1080 user_api_key_cache: UserApiKeyCache, 

1081 virtual_key_claim_field: str, 

1082 claim_value: str, 

1083 normalized_issuer: str | None, 

1084 cache_key: str, 

1085 ttl: float, 

1086) -> str | None: 

1087 issuer_scoped: Final = await get_jwt_key_mapping_object( 

1088 jwt_claim_name=virtual_key_claim_field, 

1089 jwt_claim_value=claim_value, 

1090 prisma_client=prisma_client, 

1091 jwt_issuer=normalized_issuer, 

1092 ) 

1093 if issuer_scoped is not None: 

1094 await user_api_key_cache.async_set_cache(key=cache_key, value=issuer_scoped, ttl=ttl) 

1095 return issuer_scoped 

1096 if normalized_issuer is None: 

1097 return None 

1098 # Another issuer may have already resolved (and cached) this same 

1099 # global mapping -- check its cache entry before re-querying the DB. 

1100 global_cache_key: Final = jwt_key_mapping_cache_key(virtual_key_claim_field, claim_value) 

1101 cached_global: Final = await _raw_cache(user_api_key_cache).async_get_cache(key=global_cache_key) 

1102 if isinstance(cached_global, str) and cached_global != "__NO_MAPPING__": 

1103 return cached_global 

1104 global_row: Final = await get_jwt_key_mapping_object( 

1105 jwt_claim_name=virtual_key_claim_field, 

1106 jwt_claim_value=claim_value, 

1107 prisma_client=prisma_client, 

1108 jwt_issuer=None, 

1109 ) 

1110 if global_row is not None: 

1111 await user_api_key_cache.async_set_cache(key=global_cache_key, value=global_row, ttl=ttl) 

1112 return global_row 

1113 

1114 

1115async def _resolve_jwt_to_virtual_key( 

1116 jwt_claims: dict, 

1117 jwt_handler: JWTHandler, 

1118 prisma_client: PrismaClient | None, 

1119 user_api_key_cache: UserApiKeyCache, 

1120 parent_otel_span: Span | None, 

1121 proxy_logging_obj: ProxyLogging, 

1122) -> Union[UserAPIKeyAuth | None, "_PendingAutoRegister"]: 

1123 """ 

1124 Returns: 

1125 - ``UserAPIKeyAuth``: a resolved virtual key (cache hit or DB hit). The 

1126 caller may use this directly; JWT policy has been enforced previously 

1127 (at key-creation time or, for cached results, before caching). 

1128 - ``_PendingAutoRegister``: claim is unmapped and behavior is AUTO_REGISTER. 

1129 The caller MUST run ``JWTAuthManager.auth_builder`` to enforce JWT 

1130 policy (RBAC, scope, custom_validate, email-domain), then invoke 

1131 ``_auto_register_jwt_mapping`` with the validated team_id/user_id. 

1132 - ``None``: claim is unmapped and behavior is FALLBACK_TEAM_MAPPING. 

1133 The caller falls through to standard team-based JWT auth (which itself 

1134 enforces full JWT policy via auth_builder). 

1135 - Raises HTTPException: REJECT policy hit, missing claim under 

1136 REJECT/AUTO_REGISTER, or other policy violations. 

1137 """ 

1138 raw_issuer: Final = jwt_claims.get(JWTHandler.LITELLM_JWT_ISSUER_CLAIM) 

1139 normalized_issuer: Final = raw_issuer if isinstance(raw_issuer, str) else None 

1140 virtual_key_claim_field: Final = jwt_handler.litellm_jwtauth.get_virtual_key_claim_field(normalized_issuer) 

1141 if virtual_key_claim_field is None: 

1142 return None 

1143 behavior: Final = jwt_handler.litellm_jwtauth.get_unregistered_jwt_client_behavior(normalized_issuer) 

1144 

1145 claim_value: Final = get_nested_value( 

1146 data=jwt_claims, 

1147 key_path=virtual_key_claim_field, 

1148 default=None, 

1149 ) 

1150 

1151 if claim_value is None: 

1152 verbose_proxy_logger.debug( 

1153 "JWT Key Mapping: Claim field '%s' not found in JWT claims.", virtual_key_claim_field 

1154 ) 

1155 # A missing claim is an unmapped client — apply the no-match policy 

1156 # rather than returning early. Otherwise a caller can bypass REJECT 

1157 # simply by presenting a JWT that omits the configured field. For 

1158 # AUTO_REGISTER there is no stable identity to map without a claim 

1159 # value, so we deny rather than create a sentinel-keyed record. 

1160 if behavior in ( 

1161 UnregisteredJWTClientBehavior.REJECT, 

1162 UnregisteredJWTClientBehavior.AUTO_REGISTER, 

1163 ): 

1164 raise HTTPException( 

1165 status_code=403, 

1166 detail=( 

1167 f"JWT Key Mapping: Required claim '{virtual_key_claim_field}' " 

1168 "is missing from the JWT. Access denied." 

1169 ), 

1170 ) 

1171 return None 

1172 

1173 cache_key: Final = jwt_key_mapping_cache_key(virtual_key_claim_field, str(claim_value), normalized_issuer) 

1174 raw_cached_mapping: Final = await user_api_key_cache.async_get_cache(cache_key) 

1175 sentinel_written_by_this_policy: Final = behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER 

1176 cached_mapping: Final = ( 

1177 None 

1178 if raw_cached_mapping == _JWT_PROXY_ADMIN_SENTINEL and not sentinel_written_by_this_policy 

1179 else raw_cached_mapping 

1180 ) 

1181 

1182 if cached_mapping == _JWT_PROXY_ADMIN_SENTINEL: 

1183 # Previously resolved to a proxy admin via auth_builder; skip the 

1184 # mapping lookup and let the caller re-run auth_builder. Avoids a 

1185 # repeated DB hit on every proxy-admin request under AUTO_REGISTER. 

1186 return None 

1187 

1188 if cached_mapping == "__NO_MAPPING__": 

1189 if behavior == UnregisteredJWTClientBehavior.REJECT: 

1190 raise HTTPException( 

1191 status_code=403, 

1192 detail=f"JWT Key Mapping: No registered mapping for {virtual_key_claim_field}='{claim_value}'. Access denied.", 

1193 ) 

1194 if behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER: 

1195 # Stale sentinel written under a prior fallback_team_mapping config — 

1196 # evict it and defer auto-register to after auth_builder runs. Raise 

1197 # the same 500 as the fresh-path AUTO_REGISTER branch when there is 

1198 # no DB, so behavior is consistent regardless of whether the cache 

1199 # happens to hold the sentinel. 

1200 if prisma_client is None: 

1201 raise HTTPException( 

1202 status_code=500, 

1203 detail=( 

1204 "JWT Key Mapping: AUTO_REGISTER requires a database connection. " 

1205 "Configure a database or change unregistered_jwt_client_behavior." 

1206 ), 

1207 ) 

1208 await user_api_key_cache.async_delete_cache(cache_key) 

1209 return _PendingAutoRegister( 

1210 claim_field=virtual_key_claim_field, 

1211 claim_value=str(claim_value), 

1212 cache_key=cache_key, 

1213 jwt_issuer=normalized_issuer, 

1214 ) 

1215 return None 

1216 elif cached_mapping is not None: 

1217 return IdentityStore.key_from_principal( 

1218 await IdentityStore( 

1219 prisma_client, 

1220 user_api_key_cache, 

1221 parent_otel_span=parent_otel_span, 

1222 proxy_logging_obj=proxy_logging_obj, 

1223 ).resolve(hashed_token=cached_mapping) 

1224 ) 

1225 

1226 # Resolve the mapping from DB, or treat prisma_client=None as a definitive 

1227 # miss (no DB → no mapping can exist → apply no-match policy below). An 

1228 # issuer-scoped row wins; falling back to the global (no-issuer) row keeps 

1229 # mappings created before issuer scoping existed working for every issuer. 

1230 # Each tier is cached under ITS OWN key (the global tier under the 

1231 # issuer-less cache key, not under `cache_key`/this issuer's key) so that 

1232 # updating or deleting either row invalidates exactly the cache entries it 

1233 # can affect. Caching a global-row hit under the requesting issuer's key 

1234 # would leave every OTHER issuer that had fallen back to that same global 

1235 # mapping serving its stale token until TTL after the row changes. 

1236 token_hash: Final = ( 

1237 await _lookup_jwt_mapping_token_hash( 

1238 prisma_client=prisma_client, 

1239 user_api_key_cache=user_api_key_cache, 

1240 virtual_key_claim_field=virtual_key_claim_field, 

1241 claim_value=str(claim_value), 

1242 normalized_issuer=normalized_issuer, 

1243 cache_key=cache_key, 

1244 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl, 

1245 ) 

1246 if prisma_client is not None 

1247 else None 

1248 ) 

1249 

1250 if token_hash is not None: 

1251 return IdentityStore.key_from_principal( 

1252 await IdentityStore( 

1253 prisma_client, 

1254 user_api_key_cache, 

1255 parent_otel_span=parent_otel_span, 

1256 proxy_logging_obj=proxy_logging_obj, 

1257 ).resolve(hashed_token=token_hash) 

1258 ) 

1259 

1260 # No mapping found (DB miss or no DB) — apply no-match policy. 

1261 if behavior == UnregisteredJWTClientBehavior.REJECT: 

1262 # Cache the miss before raising so repeated rejections are served from 

1263 # cache and don't re-query the DB on every request. 

1264 await user_api_key_cache.async_set_cache( 

1265 key=cache_key, 

1266 value="__NO_MAPPING__", 

1267 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl, 

1268 ) 

1269 raise HTTPException( 

1270 status_code=403, 

1271 detail=f"JWT Key Mapping: No registered mapping for {virtual_key_claim_field}='{claim_value}'. Access denied.", 

1272 ) 

1273 

1274 if behavior == UnregisteredJWTClientBehavior.AUTO_REGISTER: 

1275 if prisma_client is None: 

1276 raise HTTPException( 

1277 status_code=500, 

1278 detail=( 

1279 "JWT Key Mapping: AUTO_REGISTER requires a database connection. " 

1280 "Configure a database or change unregistered_jwt_client_behavior." 

1281 ), 

1282 ) 

1283 # Defer: caller runs JWTAuthManager.auth_builder to enforce RBAC, scope, 

1284 # custom_validate, and email-domain policy, then auto-registers using 

1285 # the validated identity. Auto-registering here on a signature-only 

1286 # JWT would bypass every JWT policy beyond signature verification. 

1287 return _PendingAutoRegister( 

1288 claim_field=virtual_key_claim_field, 

1289 claim_value=str(claim_value), 

1290 cache_key=cache_key, 

1291 jwt_issuer=normalized_issuer, 

1292 ) 

1293 

1294 # FALLBACK_TEAM_MAPPING (default): cache the miss and return None so the 

1295 # caller falls through to standard team-based JWT auth. 

1296 await user_api_key_cache.async_set_cache( 

1297 key=cache_key, 

1298 value="__NO_MAPPING__", 

1299 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl, 

1300 ) 

1301 return None 

1302 

1303 

1304def _ensure_litellm_received_at_on_request_state(request: Request) -> datetime: 

1305 """Idempotently stamp ``request.state.litellm_received_at`` with the moment 

1306 litellm's own code started handling this request -- the first line of 

1307 ``user_api_key_auth``, before any auth/pre-call work runs. This is the 

1308 basis for the request-latency Prometheus metrics (see 

1309 ``litellm/integrations/prometheus.py``), and unlike the OTEL SERVER span 

1310 below, it is set unconditionally so those metrics don't depend on OTEL 

1311 being configured. 

1312 """ 

1313 existing_received_at: Final[datetime | None] = getattr(request.state, "litellm_received_at", None) 

1314 if existing_received_at is not None: 

1315 return existing_received_at 

1316 received_at: Final = datetime.now(timezone.utc) 

1317 try: 

1318 request.state.litellm_received_at = received_at 

1319 except Exception: 

1320 pass 

1321 return received_at 

1322 

1323 

1324def _ensure_parent_otel_span_on_request_state(request: Request) -> None: 

1325 """Idempotently create the OTEL SERVER span and stash it on 

1326 ``request.state.parent_otel_span``. Safe to call multiple times. 

1327 

1328 Called both at the top of ``user_api_key_auth`` (so body-parse failures 

1329 have a span to close) and inside ``_user_api_key_auth_builder`` (for 

1330 callers that bypass ``user_api_key_auth``, e.g. MCP). 

1331 """ 

1332 from litellm.proxy.proxy_server import open_telemetry_logger 

1333 

1334 start_time: Final = _ensure_litellm_received_at_on_request_state(request) 

1335 

1336 if open_telemetry_logger is None: 1336 ↛ 1338line 1336 didn't jump to line 1338 because the condition on line 1336 was always true

1337 return 

1338 if getattr(request.state, "parent_otel_span", None) is not None: 

1339 return 

1340 parent_otel_span: Final = open_telemetry_logger.create_litellm_proxy_request_started_span( 

1341 start_time=start_time, 

1342 headers=_safe_get_request_headers(request), 

1343 ) 

1344 # Under V2 the FastAPI instrumentor stamps http.route / url.path on the server 

1345 # span; only the legacy logger needs these set explicitly. 

1346 set_route_attrs: Final = getattr(open_telemetry_logger, "set_proxy_request_route_attributes", None) 

1347 if not is_otel_v2_enabled() and set_route_attrs is not None: 

1348 set_route_attrs( 

1349 parent_otel_span, 

1350 url_path=get_request_route(request=request), 

1351 http_route=get_request_route_template(request), 

1352 ) 

1353 request.state.parent_otel_span = parent_otel_span 

1354 

1355 

1356async def _read_request_body_deferring_parse_failure( 

1357 request: Request, 

1358) -> tuple[dict, ProxyException | None]: 

1359 """Parse the body, returning a parse failure instead of raising it. 

1360 

1361 A body that fails to parse is still a request from a known caller, so auth 

1362 must run (resolving identity onto the request's trace) before the 400 goes 

1363 out; the caller re-raises the returned exception once identity is seeded. 

1364 """ 

1365 if is_opaque_audio_pass_through_request( 

1366 route=get_request_route(request=request), 

1367 content_type=_safe_get_request_headers(request=request).get("content-type", ""), 

1368 ): 

1369 _safe_set_request_parsed_body(request=request, parsed_body={}) # mutable-ok: the body cache stores a plain dict 

1370 return {}, None # mutable-ok: request_data is a plain dict across the whole auth path 

1371 try: 

1372 parsed_body: Final = await _read_request_body(request=request) 

1373 except ProxyException as parse_exception: 

1374 return {}, parse_exception # mutable-ok: request_data is a plain dict across the whole auth path 

1375 return populate_request_with_path_params(request_data=parsed_body, request=request), None 

1376 

1377 

1378async def _record_unparsable_body_failure( 

1379 user_api_key_dict: UserAPIKeyAuth, 

1380 body_parse_exception: ProxyException, 

1381 route: str, 

1382) -> None: 

1383 """Record the 400 an unparsable body earns as a failed request log. 

1384 

1385 The endpoint never runs for these, so no downstream failure hook writes the 

1386 spend log row the Admin UI reads. Logging must not change what the caller 

1387 sees, so a failure here is swallowed and the 400 is raised either way. 

1388 """ 

1389 from litellm.proxy.proxy_server import proxy_logging_obj 

1390 

1391 try: 

1392 await proxy_logging_obj.post_call_failure_hook( # pyright: ignore[reportUnknownMemberType] # bare dict in sig 

1393 request_data={}, # mutable-ok: the failure hook seeds the call id and metadata onto this dict 

1394 original_exception=body_parse_exception, 

1395 user_api_key_dict=user_api_key_dict, 

1396 error_type=ProxyErrorTypes.bad_request_error, 

1397 route=route, 

1398 ) 

1399 except Exception as e: # noqa: BLE001 # any logging failure must leave the caller's 400 untouched 

1400 verbose_proxy_logger.exception("Failed to log the request rejected for an unparsable body: %s", e) 

1401 

1402 

1403async def _refresh_session_token_grants( 

1404 valid_token: UserAPIKeyAuth, 

1405 prisma_client: PrismaClient, 

1406 user_api_key_cache: UserApiKeyCache, 

1407 parent_otel_span: Span | None, 

1408 proxy_logging_obj: ProxyLogging, 

1409) -> UserAPIKeyAuth: 

1410 """Rebuild a ``lite login`` session token's grants from the live user and team rows. 

1411 

1412 The blob only proves who logged in and which team they picked. Team models, aliases, the user's own model 

1413 list, and their role are re-read every request, so a `/team/update` or a demotion shows up without a 

1414 re-login, and a user removed from the team or deleted outright is refused. When a row cannot be read for 

1415 a reason unrelated to the caller, the minted grants stand in exactly as they did before this refresh. 

1416 """ 

1417 outcome: Final = await GrantResolver( 

1418 prisma_client, 

1419 user_api_key_cache, 

1420 parent_otel_span=parent_otel_span, 

1421 proxy_logging_obj=proxy_logging_obj, 

1422 load_user=get_user_object, 

1423 load_team=get_team_object, 

1424 load_membership=get_team_membership, 

1425 ).resolve(UserLookup(user_id=valid_token.user_id), team_id=valid_token.team_id) 

1426 match outcome: 

1427 case ResolvedGrants( 

1428 user_object=LiteLLM_UserTable() as user_object, team_object=team_object, team_membership=team_membership 

1429 ): 

1430 return UserAPIKeyAuth.model_validate( 

1431 MappingProxyType( 

1432 { 

1433 **valid_token.model_dump(exclude_none=True), 

1434 **team_grants(team_object, team_membership, user_object.user_id), 

1435 "user_role": _get_user_role(user_object), 

1436 "models": () if team_object is not None else user_models(user_object), 

1437 } 

1438 ) 

1439 ) 

1440 case ResolvedGrants(): 

1441 return valid_token 

1442 case LookupDegraded(error=error): 

1443 verbose_proxy_logger.debug("Session token grants not refreshed, keeping minted grants: %s", error) 

1444 return valid_token 

1445 case _: 

1446 raise_public(outcome) 

1447 

1448 

1449async def _resolve_object_permission_for_unresolvable_team( 

1450 object_permission_id: str | None, 

1451 prisma_client: PrismaClient | None, 

1452 user_api_key_cache: UserApiKeyCache, 

1453 parent_otel_span: Span | None, 

1454 proxy_logging_obj: ProxyLogging, 

1455) -> LiteLLM_ObjectPermissionTable | None: 

1456 """Re-resolve a team's object permission by id when the team row itself is unreadable, so the 

1457 token-derived fallback doesn't silently drop it.""" 

1458 if object_permission_id is None or prisma_client is None: 

1459 return None 

1460 return await get_object_permission( 

1461 object_permission_id=object_permission_id, 

1462 prisma_client=prisma_client, 

1463 user_api_key_cache=user_api_key_cache, 

1464 parent_otel_span=parent_otel_span, 

1465 proxy_logging_obj=proxy_logging_obj, 

1466 ) 

1467 

1468 

1469async def _user_api_key_auth_builder( 

1470 request: Request, 

1471 api_key: str, 

1472 azure_api_key_header: str, 

1473 anthropic_api_key_header: str | None, 

1474 google_ai_studio_api_key_header: str | None, 

1475 azure_apim_header: str | None, 

1476 request_data: dict, 

1477 custom_litellm_key_header: str | None = None, 

1478) -> UserAPIKeyAuth: 

1479 from litellm.proxy.proxy_server import ( 

1480 general_settings, 

1481 jwt_handler, 

1482 litellm_proxy_admin_name, 

1483 llm_model_list, 

1484 llm_router, 

1485 master_key, 

1486 model_max_budget_limiter, 

1487 open_telemetry_logger, 

1488 prisma_client, 

1489 proxy_logging_obj, 

1490 user_api_key_cache, 

1491 user_custom_auth, 

1492 ) 

1493 

1494 parent_otel_span: Span | None = None 

1495 # Prefer the receive-instant stamped by the early helper in 

1496 # user_api_key_auth (before body parse) — overwriting it would shorten 

1497 # the preprocessing-duration measurement by the body-parse window. 

1498 start_time: Final = getattr(request.state, "litellm_received_at", None) or datetime.now(timezone.utc) 

1499 try: 

1500 request.state.litellm_received_at = start_time 

1501 except Exception: 

1502 pass 

1503 route: Final[str] = get_request_route(request=request) 

1504 valid_token: UserAPIKeyAuth | None = None 

1505 custom_auth_api_key: bool = False 

1506 

1507 try: 

1508 with tracer.trace("litellm.proxy.auth.pre_db_read_auth_checks"): 

1509 await pre_db_read_auth_checks( 

1510 request_data=request_data, 

1511 request=request, 

1512 route=route, 

1513 ) 

1514 pass_through_endpoints: Final[list[dict] | None] = general_settings.get("pass_through_endpoints", None) 

1515 ## CHECK IF X-LITELM-API-KEY IS PASSED IN - supercedes Authorization header 

1516 api_key, passed_in_key = get_api_key( 

1517 custom_litellm_key_header=custom_litellm_key_header, 

1518 api_key=api_key, 

1519 azure_api_key_header=azure_api_key_header, 

1520 anthropic_api_key_header=anthropic_api_key_header, 

1521 google_ai_studio_api_key_header=google_ai_studio_api_key_header, 

1522 azure_apim_header=azure_apim_header, 

1523 pass_through_endpoints=pass_through_endpoints, 

1524 route=route, 

1525 request=request, 

1526 ) 

1527 # if user wants to pass LiteLLM_Master_Key as a custom header, example pass litellm keys as X-LiteLLM-Key: Bearer sk-1234 

1528 custom_litellm_key_header_name: Final = general_settings.get("litellm_key_header_name") 

1529 if custom_litellm_key_header_name is not None: 1529 ↛ 1530line 1529 didn't jump to line 1530 because the condition on line 1529 was never true

1530 api_key = get_api_key_from_custom_header( 

1531 request=request, 

1532 custom_litellm_key_header_name=custom_litellm_key_header_name, 

1533 ) 

1534 

1535 if open_telemetry_logger is not None: 1535 ↛ 1539line 1535 didn't jump to line 1539 because the condition on line 1535 was never true

1536 # Reuse the span created by user_api_key_auth (before body parse) 

1537 # so it survives _read_request_body failures. For callers that 

1538 # bypass user_api_key_auth (e.g. MCP), create it lazily. 

1539 _ensure_parent_otel_span_on_request_state(request) 

1540 parent_otel_span = getattr(request.state, "parent_otel_span", None) 

1541 

1542 ### USER-DEFINED AUTH FUNCTION ### 

1543 if enterprise_custom_auth is not None: 1543 ↛ 1562line 1543 didn't jump to line 1562 because the condition on line 1543 was always true

1544 with tracer.trace("litellm.proxy.auth.enterprise_custom_auth"): 

1545 response = await enterprise_custom_auth( 

1546 request=request, api_key=api_key, user_custom_auth=user_custom_auth 

1547 ) 

1548 if response is not None and isinstance(response, UserAPIKeyAuth): 1548 ↛ 1549line 1548 didn't jump to line 1549 because the condition on line 1548 was never true

1549 validated = UserAPIKeyAuth.model_validate(response) 

1550 if getattr(litellm, "enable_post_custom_auth_checks", False): 

1551 validated = await _run_post_custom_auth_checks( 

1552 valid_token=validated, 

1553 request=request, 

1554 request_data=request_data, 

1555 route=route, 

1556 parent_otel_span=parent_otel_span, 

1557 ) 

1558 return validated 

1559 elif response is not None and isinstance(response, str): 1559 ↛ 1560line 1559 didn't jump to line 1560 because the condition on line 1559 was never true

1560 api_key = response 

1561 custom_auth_api_key = True 

1562 elif user_custom_auth is not None: 

1563 response = await user_custom_auth(request=request, api_key=api_key) 

1564 validated = UserAPIKeyAuth.model_validate(response) 

1565 if getattr(litellm, "enable_post_custom_auth_checks", False): 

1566 validated = await _run_post_custom_auth_checks( 

1567 valid_token=validated, 

1568 request=request, 

1569 request_data=request_data, 

1570 route=route, 

1571 parent_otel_span=parent_otel_span, 

1572 ) 

1573 return validated 

1574 

1575 ### LITELLM-DEFINED AUTH FUNCTION ### 

1576 #### IF JWT #### 

1577 """ 

1578 LiteLLM supports using JWTs. 

1579 

1580 Enable this in proxy config, by setting 

1581 ``` 

1582 general_settings: 

1583 enable_jwt_auth: true 

1584 ``` 

1585 """ 

1586 

1587 ######## Route Checks Before Reading DB / Cache for "token" ################ 

1588 if not _route_requires_auth_despite_public(route=route, general_settings=general_settings) and ( 

1589 route in LiteLLMRoutes.public_routes.value or route_in_additonal_public_routes(current_route=route) 

1590 ): 

1591 # check if public endpoint 

1592 return UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY) 

1593 

1594 ########## End of Route Checks Before Reading DB / Cache for "token" ######## 

1595 

1596 enable_oauth2_auth: Final = general_settings.get("enable_oauth2_auth", False) is True 

1597 enable_jwt_auth: Final = general_settings.get("enable_jwt_auth", False) is True 

1598 is_jwt = jwt_handler.is_jwt(token=api_key) if enable_jwt_auth else False 

1599 

1600 # Routing uses unverified JWT claims only to choose auth path. 

1601 # Final authentication is enforced by the selected validator. 

1602 route_jwt_to_oauth2 = is_jwt and _should_route_jwt_to_oauth2_override(token=api_key, jwt_handler=jwt_handler) 

1603 

1604 # OAuth2 applies for: 

1605 # 1) when global OAuth2 auth is enabled on LLM + info routes 

1606 # 2) JWT tokens that explicitly match routing_overrides on LLM + info routes 

1607 should_apply_override_oauth2: Final = route_jwt_to_oauth2 and ( 

1608 RouteChecks.is_llm_api_route(route=route) or RouteChecks.is_info_route(route=route) 

1609 ) 

1610 should_apply_global_oauth2: Final = enable_oauth2_auth and ( 

1611 RouteChecks.is_llm_api_route(route=route) or RouteChecks.is_info_route(route=route) 

1612 ) 

1613 if (should_apply_global_oauth2 and not is_jwt) or should_apply_override_oauth2: 1613 ↛ 1614line 1613 didn't jump to line 1614 because the condition on line 1613 was never true

1614 from litellm.proxy.proxy_server import premium_user 

1615 

1616 if premium_user is not True: 

1617 raise ProxyException( 

1618 message="Oauth2 token validation is only available for premium users. " 

1619 + CommonProxyErrors.not_premium_user.value, 

1620 type=ProxyErrorTypes.auth_error, 

1621 param="premium_user", 

1622 code=status.HTTP_403_FORBIDDEN, 

1623 ) 

1624 

1625 return await Oauth2Handler.check_oauth2_token(token=api_key) 

1626 

1627 if general_settings.get("enable_oauth2_proxy_auth", False) is True: 1627 ↛ 1628line 1627 didn't jump to line 1628 because the condition on line 1627 was never true

1628 return await handle_oauth2_proxy_request(request=request) 

1629 

1630 if general_settings.get("enable_jwt_auth", False) is True: 1630 ↛ 1631line 1630 didn't jump to line 1631 because the condition on line 1630 was never true

1631 is_jwt = jwt_handler.is_jwt(token=api_key) 

1632 verbose_proxy_logger.debug("is_jwt: %s", is_jwt) 

1633 if is_jwt: 

1634 from litellm.proxy.proxy_server import premium_user 

1635 

1636 if premium_user is not True: 

1637 raise ProxyException( 

1638 message=f"JWT Auth is an enterprise only feature. {CommonProxyErrors.not_premium_user.value}", 

1639 type=ProxyErrorTypes.auth_error, 

1640 param="premium_user", 

1641 code=status.HTTP_403_FORBIDDEN, 

1642 ) 

1643 # Try JWT-to-Virtual-Key mapping first to avoid 

1644 # unnecessary DB queries in auth_builder 

1645 do_standard_jwt_auth = True 

1646 pending_auto_register: _PendingAutoRegister | None = None 

1647 if jwt_handler.litellm_jwtauth.is_virtual_key_mapping_configured(): 

1648 # Decode JWT to get claims without running full auth_builder 

1649 jwt_claims: dict | None 

1650 if jwt_handler.litellm_jwtauth.oidc_userinfo_enabled and not is_jwt: 

1651 jwt_claims = await jwt_handler.get_oidc_userinfo(token=api_key) 

1652 else: 

1653 jwt_claims = await jwt_handler.auth_jwt(token=api_key) 

1654 

1655 resolve_result: Final = await _resolve_jwt_to_virtual_key( 

1656 jwt_claims=jwt_claims, 

1657 jwt_handler=jwt_handler, 

1658 prisma_client=prisma_client, 

1659 user_api_key_cache=user_api_key_cache, 

1660 parent_otel_span=parent_otel_span, 

1661 proxy_logging_obj=proxy_logging_obj, 

1662 ) 

1663 if isinstance(resolve_result, UserAPIKeyAuth): 

1664 valid_token = resolve_result 

1665 api_key = valid_token.token or "" 

1666 valid_token.jwt_claims = jwt_claims 

1667 do_standard_jwt_auth = False 

1668 # Fall through to virtual key checks 

1669 if valid_token.user_id is not None and valid_token.user_email is None: 

1670 mapped_claims = jwt_claims or {} # mutable-ok: empty-dict fallback for the None-claims case 

1671 mapped_user_email = jwt_handler.get_user_email(token=mapped_claims, default_value=None) 

1672 mapped_jwt_user_id: Final = jwt_handler.get_user_id(token=mapped_claims, default_value=None) 

1673 if mapped_user_email is not None and mapped_jwt_user_id == valid_token.user_id: 

1674 try: 

1675 mapped_user_obj: Final = await get_user_object( 

1676 user_id=valid_token.user_id, 

1677 prisma_client=prisma_client, 

1678 user_api_key_cache=user_api_key_cache, 

1679 user_id_upsert=False, 

1680 parent_otel_span=parent_otel_span, 

1681 proxy_logging_obj=proxy_logging_obj, 

1682 user_email=mapped_user_email, 

1683 ) 

1684 except Exception as e: 

1685 verbose_proxy_logger.debug("JWT mapped-key user_email backfill skipped: %s", e) 

1686 else: 

1687 if mapped_user_obj is not None: 

1688 valid_token.user_email = mapped_user_obj.user_email 

1689 elif isinstance(resolve_result, _PendingAutoRegister): 

1690 # Run full JWT policy (RBAC, scope, custom_validate, 

1691 # email-domain) via auth_builder, then create the key 

1692 # from the validated identity below. 

1693 pending_auto_register = resolve_result 

1694 # else: None → FALLBACK_TEAM_MAPPING, falls through to 

1695 # standard JWT auth_builder below 

1696 

1697 if do_standard_jwt_auth: 

1698 with tracer.trace("litellm.proxy.auth.jwt_auth_builder"): 

1699 result: Final = await JWTAuthManager.auth_builder( 

1700 request_data=request_data, 

1701 general_settings=general_settings, 

1702 api_key=api_key, 

1703 jwt_handler=jwt_handler, 

1704 route=route, 

1705 prisma_client=prisma_client, 

1706 user_api_key_cache=user_api_key_cache, 

1707 proxy_logging_obj=proxy_logging_obj, 

1708 parent_otel_span=parent_otel_span, 

1709 request_headers=_safe_get_request_headers(request), 

1710 request_method=RouteChecks._get_request_method(request=request), 

1711 ) 

1712 

1713 is_proxy_admin: Final = result["is_proxy_admin"] 

1714 team_id: Final = result["team_id"] 

1715 user_id: Final = result["user_id"] 

1716 user_email: Final = result["user_email"] 

1717 user_object: Final = result["user_object"] 

1718 end_user_id = result["end_user_id"] 

1719 org_id: Final = result["org_id"] 

1720 jwt_claims = result.get("jwt_claims", None) 

1721 agent_id: Final[str | None] = result.get("agent_id") 

1722 

1723 if ( 

1724 user_object is not None 

1725 and isinstance(user_object.metadata, dict) 

1726 and user_object.metadata.get("scim_active") is False 

1727 ): 

1728 raise HTTPException( 

1729 status_code=status.HTTP_401_UNAUTHORIZED, 

1730 detail=f"User={user_id} has been deactivated via SCIM. Keys owned by this user cannot be used.", 

1731 ) 

1732 

1733 if is_proxy_admin: 

1734 # Proxy admins authenticate via auth_builder (full 

1735 # access), not via a mapped virtual key. If 

1736 # AUTO_REGISTER was pending, cache a sentinel so 

1737 # future requests from this JWT identity skip the 

1738 # DB mapping lookup in _resolve_jwt_to_virtual_key. 

1739 # Without this, every proxy-admin request under 

1740 # AUTO_REGISTER re-hits get_jwt_key_mapping_object. 

1741 if pending_auto_register is not None: 

1742 await user_api_key_cache.async_set_cache( 

1743 key=pending_auto_register.cache_key, 

1744 value=_JWT_PROXY_ADMIN_SENTINEL, 

1745 ttl=jwt_handler.litellm_jwtauth.virtual_key_mapping_cache_ttl, 

1746 ) 

1747 return JWTAuthManager.user_api_key_auth_from_result(result, parent_otel_span) 

1748 

1749 valid_token = JWTAuthManager.user_api_key_auth_from_result(result, parent_otel_span) 

1750 

1751 # AUTO_REGISTER deferred from _resolve_jwt_to_virtual_key. 

1752 # JWT policy (RBAC, scope, custom_validate, email-domain) 

1753 # has now been enforced by auth_builder above. Create the 

1754 # mapping + virtual key from the *validated* identity, then 

1755 # replace valid_token with the new key so downstream checks 

1756 # use the key-scoped path. 

1757 if pending_auto_register is not None and prisma_client is not None: 

1758 auto_registered: Final = await _auto_register_jwt_mapping( 

1759 virtual_key_claim_field=pending_auto_register.claim_field, 

1760 claim_value=pending_auto_register.claim_value, 

1761 jwt_handler=jwt_handler, 

1762 prisma_client=prisma_client, 

1763 user_api_key_cache=user_api_key_cache, 

1764 parent_otel_span=parent_otel_span, 

1765 proxy_logging_obj=proxy_logging_obj, 

1766 cache_key=pending_auto_register.cache_key, 

1767 jwt_issuer=pending_auto_register.jwt_issuer, 

1768 team_id=team_id, 

1769 user_id=user_id, 

1770 org_id=org_id, 

1771 end_user_id=end_user_id, 

1772 agent_id=agent_id, 

1773 ) 

1774 if auto_registered is not None: 

1775 auto_registered.jwt_claims = jwt_claims 

1776 auto_registered.user_email = user_email 

1777 # The auto-registered token is built from the new key's 

1778 # columns, which carry no user budget. Carry over the 

1779 # already-loaded user row rather than re-reading it, or 

1780 # the budget check below has nothing to enforce. 

1781 auto_registered.user_model_max_budget = ( 

1782 user_object.model_max_budget if user_object is not None else None 

1783 ) 

1784 valid_token = auto_registered 

1785 api_key = valid_token.token or "" 

1786 

1787 # Check if model has zero cost - if so, skip all budget checks 

1788 model = _get_model_from_request_context( 

1789 request_data=request_data, 

1790 route=route, 

1791 request=request, 

1792 llm_router=llm_router, 

1793 team_id=valid_token.team_id, 

1794 ) 

1795 skip_budget_checks = False 

1796 if model is not None and llm_router is not None: 

1797 from litellm.proxy.auth.auth_checks import _is_model_cost_zero 

1798 

1799 skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router) 

1800 if skip_budget_checks: 

1801 verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model) 

1802 

1803 # Fetch project object for JWT path if project_id is set 

1804 _jwt_project_obj = None 

1805 if valid_token.project_id is not None: 

1806 _jwt_project_obj = await get_project_object( 

1807 project_id=valid_token.project_id, 

1808 prisma_client=prisma_client, 

1809 user_api_key_cache=user_api_key_cache, 

1810 proxy_logging_obj=proxy_logging_obj, 

1811 ) 

1812 if _jwt_project_obj is not None: 

1813 valid_token.project_metadata = _jwt_project_obj.metadata 

1814 valid_token.project_alias = _jwt_project_obj.project_alias 

1815 

1816 # JWT auth returns here rather than falling through to the 

1817 # virtual-key checks below, so the user's per-model budget 

1818 # has to be enforced on this path too. Without it the 

1819 # post-call increment still charges the counter and nothing 

1820 # ever reads it, which is worse than not tracking at all. 

1821 # Guarded by the same flag the virtual-key path uses, or a 

1822 # zero-cost model would be refused here and allowed there, 

1823 # while the log above claims all budget checks were skipped. 

1824 if not skip_budget_checks: 

1825 await _check_user_model_budget( 

1826 valid_token=cast(UserAPIKeyAuth, valid_token), 

1827 model_max_budget_limiter=model_max_budget_limiter, 

1828 models=_get_model_names_for_budget_checks( 

1829 model=_get_model_from_request_context( 

1830 request_data=request_data, 

1831 route=route, 

1832 request=request, 

1833 llm_router=llm_router, 

1834 team_id=valid_token.team_id, 

1835 ) 

1836 ), 

1837 ) 

1838 

1839 return cast(UserAPIKeyAuth, valid_token) 

1840 

1841 #### ELSE #### 

1842 ## CHECK PASS-THROUGH ENDPOINTS ## 

1843 if not custom_auth_api_key: 1843 ↛ 1854line 1843 didn't jump to line 1854 because the condition on line 1843 was always true

1844 response = await check_api_key_for_custom_headers_or_pass_through_endpoints( 

1845 request=request, 

1846 route=route, 

1847 pass_through_endpoints=pass_through_endpoints, 

1848 api_key=api_key, 

1849 ) 

1850 if isinstance(response, str): 

1851 api_key = response 

1852 elif isinstance(response, UserAPIKeyAuth): 1852 ↛ 1853line 1852 didn't jump to line 1853 because the condition on line 1852 was never true

1853 return response 

1854 if master_key is None: 1854 ↛ 1855line 1854 didn't jump to line 1855 because the condition on line 1854 was never true

1855 if isinstance(api_key, str): 

1856 return UserAPIKeyAuth( 

1857 api_key=api_key, 

1858 user_role=LitellmUserRoles.INTERNAL_USER, 

1859 parent_otel_span=parent_otel_span, 

1860 ) 

1861 else: 

1862 return UserAPIKeyAuth( 

1863 user_role=LitellmUserRoles.INTERNAL_USER, 

1864 parent_otel_span=parent_otel_span, 

1865 ) 

1866 elif api_key is None: # only require api key if master key is set 

1867 raise Exception("No api key passed in.") 

1868 elif api_key == "": 

1869 # missing 'Bearer ' prefix 

1870 raise Exception("Malformed API Key passed in. Ensure Key has `Bearer ` prefix.") 

1871 

1872 if route == "/user/auth": 1872 ↛ 1873line 1872 didn't jump to line 1873 because the condition on line 1872 was never true

1873 if general_settings.get("allow_user_auth", False) is True: 

1874 return UserAPIKeyAuth() 

1875 else: 

1876 raise HTTPException( 

1877 status_code=status.HTTP_403_FORBIDDEN, 

1878 detail="'allow_user_auth' not set or set to False", 

1879 ) 

1880 

1881 ## Check END-USER OBJECT 

1882 _end_user_object = None 

1883 end_user_params: Final = {} 

1884 

1885 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request)) 

1886 end_user_id = await resolve_and_validate_end_user_id( 

1887 raw_end_user_id=raw_end_user_id, 

1888 prisma_client=prisma_client, 

1889 user_api_key_cache=user_api_key_cache, 

1890 parent_otel_span=parent_otel_span, 

1891 proxy_logging_obj=proxy_logging_obj, 

1892 route=route, 

1893 ) 

1894 if end_user_id: 

1895 try: 

1896 end_user_params["end_user_id"] = end_user_id 

1897 

1898 with tracer.trace("litellm.proxy.auth.get_end_user_object"): 

1899 _end_user_object = await get_end_user_object( 

1900 end_user_id=end_user_id, 

1901 prisma_client=prisma_client, 

1902 user_api_key_cache=user_api_key_cache, 

1903 parent_otel_span=parent_otel_span, 

1904 proxy_logging_obj=proxy_logging_obj, 

1905 route=route, 

1906 ) 

1907 if _end_user_object is not None: 1907 ↛ 1908line 1907 didn't jump to line 1908 because the condition on line 1907 was never true

1908 end_user_params["allowed_model_region"] = _end_user_object.allowed_model_region 

1909 if _end_user_object.litellm_budget_table is not None: 

1910 _apply_budget_limits_to_end_user_params( 

1911 end_user_params=end_user_params, 

1912 budget_info=_end_user_object.litellm_budget_table, 

1913 end_user_id=end_user_id, 

1914 ) 

1915 elif litellm.max_end_user_budget_id is not None: 1915 ↛ 1917line 1915 didn't jump to line 1917 because the condition on line 1915 was never true

1916 # End user doesn't exist yet, but apply default budget limits if configured 

1917 from litellm.proxy.auth.auth_checks import ( 

1918 get_default_end_user_budget, 

1919 ) 

1920 

1921 default_budget: Final = await get_default_end_user_budget( 

1922 prisma_client=prisma_client, 

1923 user_api_key_cache=user_api_key_cache, 

1924 parent_otel_span=parent_otel_span, 

1925 ) 

1926 if default_budget is not None: 

1927 _apply_budget_limits_to_end_user_params( 

1928 end_user_params=end_user_params, 

1929 budget_info=default_budget, 

1930 end_user_id=end_user_id, 

1931 ) 

1932 except Exception as e: 

1933 if isinstance(e, litellm.BudgetExceededError): 

1934 raise e 

1935 verbose_proxy_logger.debug("Unable to find user in db. Error - %s", e) 

1936 

1937 ### CHECK IF ADMIN ### 

1938 # note: never string compare api keys, this is vulenerable to a time attack. Use secrets.compare_digest instead 

1939 ### CHECK IF ADMIN ### 

1940 # note: never string compare api keys, this is vulenerable to a time attack. Use secrets.compare_digest instead 

1941 if valid_token is None: 1941 ↛ 1977line 1941 didn't jump to line 1977 because the condition on line 1941 was always true

1942 ## Check CACHE 

1943 try: 

1944 with tracer.trace("litellm.proxy.auth.get_key_object_check_cache"): 

1945 valid_token = IdentityStore.key_from_principal( 

1946 await IdentityStore( 

1947 prisma_client, 

1948 user_api_key_cache, 

1949 parent_otel_span=parent_otel_span, 

1950 proxy_logging_obj=proxy_logging_obj, 

1951 check_cache_only=True, 

1952 ).resolve(hashed_token=hash_token(api_key)) 

1953 ) 

1954 # Key-cache entries are written only after the proxy validated a 

1955 # virtual key or the master key, but via_virtual_key is exclude=True 

1956 # so serialization drops it; restore it at this trusted boundary. 

1957 # The UI-login JWT fallback below constructs its token from a 

1958 # decrypted blob, not this cache, and stays unmarked. 

1959 if isinstance(valid_token, UserAPIKeyAuth): 1959 ↛ 1970line 1959 didn't jump to line 1970 because the condition on line 1959 was always true

1960 valid_token.via_virtual_key = True 

1961 except Exception: 

1962 verbose_logger.debug("api key not found in cache.") 

1963 valid_token = None 

1964 

1965 ## Check UI/CLI Hash Key 

1966 # Attempt decryption for non-sk- tokens unless the operator has 

1967 # explicitly set EXPERIMENTAL_UI_LOGIN=false to disable it. 

1968 # Unset (None) keeps the new default of always attempting decryption; 

1969 # decryption fails closed for anything that is not a genuine blob. 

1970 if ( 

1971 valid_token is None 

1972 and not api_key.startswith("sk-") 

1973 and get_secret_bool("EXPERIMENTAL_UI_LOGIN") is not False 

1974 ): 

1975 valid_token = ExperimentalUIJWTToken.get_key_object_from_ui_hash_key(api_key) 

1976 

1977 if valid_token is not None and valid_token.is_session_token and prisma_client is not None: 1977 ↛ 1978line 1977 didn't jump to line 1978 because the condition on line 1977 was never true

1978 valid_token = await _refresh_session_token_grants( # rebind-ok: later checks read this name 

1979 valid_token=valid_token, 

1980 prisma_client=prisma_client, 

1981 user_api_key_cache=user_api_key_cache, 

1982 parent_otel_span=parent_otel_span, 

1983 proxy_logging_obj=proxy_logging_obj, 

1984 ) 

1985 

1986 if ( 

1987 valid_token is not None 

1988 and isinstance(valid_token, UserAPIKeyAuth) 

1989 and valid_token.user_role == LitellmUserRoles.PROXY_ADMIN 

1990 ): 

1991 if valid_token.expires is not None: 1991 ↛ 1992line 1991 didn't jump to line 1992 because the condition on line 1991 was never true

1992 current_time = datetime.now(timezone.utc) 

1993 if isinstance(valid_token.expires, datetime): 

1994 expiry_time = valid_token.expires 

1995 else: 

1996 expiry_time = datetime.fromisoformat(valid_token.expires) 

1997 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None: 

1998 expiry_time = expiry_time.replace(tzinfo=timezone.utc) 

1999 if expiry_time < current_time: 

2000 await _delete_cache_key_object( 

2001 hashed_token=hash_token(api_key), 

2002 user_api_key_cache=user_api_key_cache, 

2003 proxy_logging_obj=proxy_logging_obj, 

2004 ) 

2005 raise ProxyException( 

2006 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}", 

2007 type=ProxyErrorTypes.expired_key, 

2008 code=status.HTTP_401_UNAUTHORIZED, 

2009 param=abbreviate_api_key(api_key=api_key), 

2010 ) 

2011 valid_token = update_valid_token_with_end_user_params( 

2012 valid_token=valid_token, end_user_params=end_user_params 

2013 ) 

2014 valid_token.parent_otel_span = parent_otel_span 

2015 if _end_user_object is not None: 2015 ↛ 2016line 2015 didn't jump to line 2016 because the condition on line 2015 was never true

2016 valid_token.end_user_object_permission = _end_user_object.object_permission 

2017 

2018 return valid_token 

2019 

2020 if ( 2020 ↛ 2027line 2020 didn't jump to line 2027 because the condition on line 2020 was never true

2021 valid_token is not None 

2022 and isinstance(valid_token, UserAPIKeyAuth) 

2023 and valid_token.team_id is not None 

2024 and valid_token.team_id != UI_TEAM_ID 

2025 ): 

2026 ## UPDATE TEAM VALUES BASED ON CACHED TEAM OBJECT - allows `/team/update` values to work for cached token 

2027 try: 

2028 team_obj: Final[LiteLLM_TeamTableCachedObj] = await get_team_object( 

2029 team_id=valid_token.team_id, 

2030 prisma_client=prisma_client, 

2031 user_api_key_cache=user_api_key_cache, 

2032 parent_otel_span=parent_otel_span, 

2033 proxy_logging_obj=proxy_logging_obj, 

2034 check_cache_only=True, 

2035 ) 

2036 

2037 if ( 

2038 team_obj.last_refreshed_at is not None 

2039 and valid_token.last_refreshed_at is not None 

2040 and team_obj.last_refreshed_at > valid_token.last_refreshed_at 

2041 ): 

2042 team_obj_dict: Final = team_obj.__dict__ 

2043 

2044 for k, v in team_obj_dict.items(): 

2045 field_name = f"team_{k}" 

2046 if field_name in valid_token.__fields__: 

2047 setattr(valid_token, field_name, v) 

2048 except Exception as e: 

2049 verbose_logger.debug(e) # moving from .warning to .debug as it spams logs when team missing from cache. 

2050 

2051 try: 

2052 is_master_key_valid = secrets.compare_digest(api_key, master_key) 

2053 except Exception: 

2054 is_master_key_valid = False 

2055 

2056 ## VALIDATE MASTER KEY ## 

2057 if not isinstance(master_key, str): 2057 ↛ 2058line 2057 didn't jump to line 2058 because the condition on line 2057 was never true

2058 raise HTTPException( 

2059 status_code=500, 

2060 detail={f"Master key must be a valid string. Current type={type(master_key)}"}, 

2061 ) 

2062 

2063 if is_master_key_valid: 

2064 # Substitute a stable alias for the raw master key so neither the 

2065 # master key nor its hash propagates into spend logs, Prometheus 

2066 # /metrics labels, audit trails, rate-limit buckets, or any other 

2067 # downstream consumer of UserAPIKeyAuth.api_key. 

2068 _user_api_key_obj = await _return_user_api_key_auth_obj( 

2069 user_obj=None, 

2070 user_role=LitellmUserRoles.PROXY_ADMIN, 

2071 api_key=LITELLM_PROXY_MASTER_KEY_ALIAS, 

2072 parent_otel_span=parent_otel_span, 

2073 valid_token_dict={ 

2074 **end_user_params, 

2075 "user_id": litellm_proxy_admin_name, 

2076 }, 

2077 route=route, 

2078 start_time=start_time, 

2079 ) 

2080 asyncio.create_task( 

2081 _cache_key_object( 

2082 hashed_token=hash_token(master_key), 

2083 user_api_key_obj=_user_api_key_obj, 

2084 user_api_key_cache=user_api_key_cache, 

2085 proxy_logging_obj=proxy_logging_obj, 

2086 ) 

2087 ) 

2088 

2089 _user_api_key_obj = update_valid_token_with_end_user_params( 

2090 valid_token=_user_api_key_obj, end_user_params=end_user_params 

2091 ) 

2092 _user_api_key_obj.via_virtual_key = True 

2093 

2094 return _user_api_key_obj 

2095 

2096 ## IF it's not a master key 

2097 ## Route should not be in master_key_only_routes 

2098 if route in LiteLLMRoutes.master_key_only_routes.value: 

2099 raise Exception(f"Tried to access route={route}, which is only for MASTER KEY") 

2100 

2101 ## Check DB 

2102 

2103 if ( 2103 ↛ 2106line 2103 didn't jump to line 2106 because the condition on line 2103 was never true

2104 prisma_client is None 

2105 ): # if both master key + user key submitted, and user key != master key, and no db connected, raise an error 

2106 raise ProxyException( 

2107 message="No connected db.", 

2108 type=ProxyErrorTypes.no_db_connection, 

2109 code=400, 

2110 param=None, 

2111 ) 

2112 

2113 if valid_token is None: 

2114 if isinstance(api_key, str): # if generated token, make sure it starts with sk-. 2114 ↛ 2130line 2114 didn't jump to line 2130 because the condition on line 2114 was always true

2115 _masked_key: Final = f"{api_key[:4]}****{api_key[-4:]}" if len(api_key) > 8 else "****" 

2116 if not api_key.startswith("sk-"): 

2117 _hint = _JWT_AUTH_DISABLED_HINT if not enable_jwt_auth and JWTHandler.is_jwt(token=api_key) else "" 

2118 _malformed_key_error = HTTPException( 

2119 status_code=status.HTTP_401_UNAUTHORIZED, 

2120 detail=( 

2121 f"{INVALID_VIRTUAL_KEY_ERROR_MESSAGE}. Received={_masked_key}, " 

2122 f"expected to start with 'sk-'.{_hint}" 

2123 ), 

2124 ) # prevent token hashes from being used 

2125 # Stamp provenance here so log routing classifies this 401 by 

2126 # where it was raised, never by its message text. 

2127 setattr(_malformed_key_error, INVALID_VIRTUAL_KEY_ERROR_MARKER, True) 

2128 raise _malformed_key_error 

2129 else: 

2130 verbose_logger.warning( 

2131 "litellm.proxy.proxy_server.user_api_key_auth(): Warning - Key is not a string. Got type={}".format( 

2132 type(api_key) if api_key is not None else "None" 

2133 ) 

2134 ) 

2135 abbreviated_api_key: Final = abbreviate_api_key(api_key=api_key) 

2136 if api_key.startswith("sk-"): 2136 ↛ 2139line 2136 didn't jump to line 2139 because the condition on line 2136 was always true

2137 api_key = hash_token(token=api_key) 

2138 

2139 try: 

2140 with tracer.trace("litellm.proxy.auth.get_key_object_from_db"): 

2141 valid_token = IdentityStore.key_from_principal( 

2142 await IdentityStore( 

2143 prisma_client, 

2144 user_api_key_cache, 

2145 parent_otel_span=parent_otel_span, 

2146 proxy_logging_obj=proxy_logging_obj, 

2147 ).resolve(hashed_token=api_key) 

2148 ) 

2149 except ProxyException as e: 

2150 if e.code == 401 or e.code == "401": 

2151 e.message = f"Authentication Error, Invalid proxy server token passed. Received API Key = {abbreviated_api_key}, Key Hash (Token) ={api_key}. Unable to find token in cache or `LiteLLM_VerificationTokenTable`" 

2152 raise e 

2153 # update end-user params on valid token 

2154 # These can change per request - it's important to update them here 

2155 valid_token.end_user_id = end_user_params.get("end_user_id") 

2156 valid_token.end_user_tpm_limit = end_user_params.get("end_user_tpm_limit") 

2157 valid_token.end_user_rpm_limit = end_user_params.get("end_user_rpm_limit") 

2158 valid_token.end_user_tpd_limit = end_user_params.get("end_user_tpd_limit") 

2159 valid_token.allowed_model_region = end_user_params.get("allowed_model_region") 

2160 

2161 if valid_token is not None: 2161 ↛ 2164line 2161 didn't jump to line 2164 because the condition on line 2161 was always true

2162 valid_token = _update_key_budget_with_temp_budget_increase(valid_token) 

2163 

2164 user_obj: LiteLLM_UserTable | None = None 

2165 valid_token_dict: dict = {} 

2166 if valid_token is not None: 2166 ↛ 2524line 2166 didn't jump to line 2524 because the condition on line 2166 was always true

2167 # Got Valid Token from Cache, DB 

2168 # Run checks for 

2169 # 1. If token can call model 

2170 ## 1a. If token can call fallback models (if client-side fallbacks given) 

2171 # 2. If user_id for this token is in budget 

2172 # 3. If the user spend within their own team is within budget 

2173 # 4. If 'user' passed to /chat/completions, /embeddings endpoint is in budget 

2174 # 5. If token is expired 

2175 # 6. If token spend is under Budget for the token 

2176 # 7. If token spend per model is under budget per model 

2177 # 8. If token spend is under team budget 

2178 # 9. If team spend is under team budget 

2179 

2180 ## base case ## key is disabled 

2181 if valid_token.blocked is True: 

2182 raise Exception("Key is blocked. Update via `/key/unblock` if you're an admin.") 

2183 await _enforce_key_and_fallback_model_access( 

2184 valid_token=valid_token, 

2185 request_data=request_data, 

2186 route=route, 

2187 request=request, 

2188 llm_model_list=llm_model_list, 

2189 llm_router=llm_router, 

2190 ) 

2191 await _prefetch_referenced_auth_objects( 

2192 valid_token, end_user_id=end_user_id, user_api_key_cache=user_api_key_cache, prisma_client=prisma_client 

2193 ) 

2194 

2195 # Check 2. If user_id for this token is in budget - done in common_checks() 

2196 if valid_token.user_id is not None: 

2197 try: 

2198 with tracer.trace("litellm.proxy.auth.get_user_object"): 

2199 user_obj = await get_user_object( 

2200 user_id=valid_token.user_id, 

2201 prisma_client=prisma_client, 

2202 user_api_key_cache=user_api_key_cache, 

2203 user_id_upsert=False, 

2204 parent_otel_span=parent_otel_span, 

2205 proxy_logging_obj=proxy_logging_obj, 

2206 ) 

2207 except Exception as e: 

2208 verbose_logger.debug( 

2209 "litellm.proxy.auth.user_api_key_auth.py::user_api_key_auth() - Unable to get user from db/cache. Setting user_obj to None. Exception received - %s", 

2210 e, 

2211 ) 

2212 user_obj = None 

2213 

2214 if user_obj is not None: 2214 ↛ 2220line 2214 didn't jump to line 2220 because the condition on line 2214 was always true

2215 # The joint verification-token view carries the key's columns only, so the 

2216 # user's own per-model budget reaches enforcement and the post-call 

2217 # increment through the row fetched here. 

2218 valid_token.user_model_max_budget = user_obj.model_max_budget 

2219 

2220 if ( 2220 ↛ 2225line 2220 didn't jump to line 2225 because the condition on line 2220 was never true

2221 user_obj is not None 

2222 and isinstance(user_obj.metadata, dict) 

2223 and user_obj.metadata.get("scim_active") is False 

2224 ): 

2225 raise Exception( 

2226 f"User={valid_token.user_id} has been deactivated via SCIM. Keys owned by this user cannot be used." 

2227 ) 

2228 

2229 # Check 2a. Check if model has zero cost - if so, skip all budget checks 

2230 model = _get_model_from_request_context( 

2231 request_data=request_data, 

2232 route=route, 

2233 request=request, 

2234 llm_router=llm_router, 

2235 team_id=valid_token.team_id, 

2236 ) 

2237 skip_budget_checks = False 

2238 if model is not None and llm_router is not None: 2238 ↛ 2239line 2238 didn't jump to line 2239 because the condition on line 2238 was never true

2239 from litellm.proxy.auth.auth_checks import _is_model_cost_zero 

2240 

2241 skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router) 

2242 if skip_budget_checks: 

2243 verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model) 

2244 

2245 # Check 3. Check if user is in their team budget 

2246 if not skip_budget_checks and valid_token.team_member_spend is not None: 2246 ↛ 2247line 2246 didn't jump to line 2247 because the condition on line 2246 was never true

2247 _user_id: Final = valid_token.user_id 

2248 _team_id: Final = valid_token.team_id 

2249 if prisma_client is not None and _user_id is not None and _team_id is not None: 

2250 _cache_key: Final = team_membership_auth_cache_key(team_id=_team_id, user_id=_user_id) 

2251 

2252 team_member_info = await user_api_key_cache.async_get_cache( 

2253 key=_cache_key, 

2254 model_type=LiteLLM_TeamMembership, 

2255 ) 

2256 if team_member_info is None: 

2257 # read from DB 

2258 _db_member: Final = await TeamMembershipRepository(prisma_client).table.find_first( 

2259 where={ 

2260 "user_id": _user_id, 

2261 "team_id": _team_id, 

2262 }, 

2263 include={"litellm_budget_table": True}, 

2264 ) 

2265 if _db_member is not None: 

2266 team_member_info = LiteLLM_TeamMembership(**_db_member.model_dump()) 

2267 await user_api_key_cache.async_set_cache( 

2268 key=_cache_key, 

2269 value=team_member_info, 

2270 model_type=LiteLLM_TeamMembership, 

2271 ttl=5, 

2272 ) 

2273 

2274 if team_member_info is not None and team_member_info.litellm_budget_table is not None: 

2275 team_member_budget: Final = team_member_info.litellm_budget_table.effective_max_budget( 

2276 now=datetime.now(timezone.utc), 

2277 ) 

2278 if team_member_budget is not None and team_member_budget > 0: 

2279 # Read from cross-pod counter (Redis-first) if available 

2280 from litellm.proxy.proxy_server import get_current_spend 

2281 

2282 team_member_spend = valid_token.team_member_spend 

2283 if valid_token.user_id is not None and valid_token.team_id is not None: 

2284 team_member_spend = await get_current_spend( 

2285 counter_key=f"spend:team_member:{valid_token.user_id}:{valid_token.team_id}", 

2286 fallback_spend=team_member_spend, 

2287 max_budget=team_member_budget, 

2288 ) 

2289 if team_member_spend >= team_member_budget: 

2290 _entity_id: Final = f"{valid_token.user_id}:{valid_token.team_id}" 

2291 raise litellm.BudgetExceededError( 

2292 current_cost=team_member_spend, 

2293 max_budget=team_member_budget, 

2294 message=( 

2295 f"Budget has been exceeded! TeamMember={_entity_id} " 

2296 f"Current cost: {team_member_spend}, Max budget: {team_member_budget}" 

2297 ), 

2298 entity_type=Litellm_EntityType.TEAM_MEMBER.value, 

2299 entity_id=_entity_id, 

2300 ) 

2301 

2302 # Check 3. If token is expired 

2303 if valid_token.expires is not None: 2303 ↛ 2304line 2303 didn't jump to line 2304 because the condition on line 2303 was never true

2304 current_time = datetime.now(timezone.utc) 

2305 if isinstance(valid_token.expires, datetime): 

2306 expiry_time = valid_token.expires 

2307 else: 

2308 expiry_time = datetime.fromisoformat(valid_token.expires) 

2309 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None: 

2310 expiry_time = expiry_time.replace(tzinfo=timezone.utc) 

2311 verbose_proxy_logger.debug( 

2312 "Checking if token expired, expiry time %s and current time %s", expiry_time, current_time 

2313 ) 

2314 if expiry_time < current_time: 

2315 # Token exists but is expired. 

2316 raise ProxyException( 

2317 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}", 

2318 type=ProxyErrorTypes.expired_key, 

2319 code=status.HTTP_401_UNAUTHORIZED, 

2320 param=abbreviate_api_key(api_key=api_key), 

2321 ) 

2322 

2323 if not skip_budget_checks: 2323 ↛ 2416line 2323 didn't jump to line 2416 because the condition on line 2323 was always true

2324 with tracer.trace("litellm.proxy.auth.budget_checks"): 

2325 # Check 4. Max Budget Alert Check (runs before budget enforcement 

2326 # so multi-threshold 100% alerts fire on the request that crosses 

2327 # max_budget, before BudgetExceededError is raised below) 

2328 await _virtual_key_max_budget_alert_check( 

2329 valid_token=valid_token, 

2330 proxy_logging_obj=proxy_logging_obj, 

2331 user_obj=user_obj, 

2332 ) 

2333 

2334 # Check 5. Token Spend is under budget 

2335 if RouteChecks.is_llm_api_route(route=route): 2335 ↛ 2343line 2335 didn't jump to line 2343 because the condition on line 2335 was always true

2336 await _virtual_key_max_budget_check( 

2337 valid_token=valid_token, 

2338 proxy_logging_obj=proxy_logging_obj, 

2339 user_obj=user_obj, 

2340 ) 

2341 

2342 # Check 6. Soft Budget Check 

2343 await _virtual_key_soft_budget_check( 

2344 valid_token=valid_token, 

2345 proxy_logging_obj=proxy_logging_obj, 

2346 user_obj=user_obj, 

2347 ) 

2348 

2349 # Check 5. Token Model Spend is under Model budget 

2350 max_budget_per_model: Final = valid_token.model_max_budget 

2351 current_model = _get_model_from_request_context( 

2352 request_data=request_data, 

2353 route=route, 

2354 request=request, 

2355 llm_router=llm_router, 

2356 team_id=valid_token.team_id, 

2357 ) 

2358 current_models = _get_model_names_for_budget_checks(model=current_model) 

2359 

2360 if ( 2360 ↛ 2369line 2360 didn't jump to line 2369 because the condition on line 2360 was never true

2361 max_budget_per_model is not None 

2362 and isinstance(max_budget_per_model, dict) 

2363 and len(max_budget_per_model) > 0 

2364 and prisma_client is not None 

2365 and current_models 

2366 and valid_token.token is not None 

2367 ): 

2368 ## GET THE SPEND FOR THIS MODEL 

2369 for model_name in current_models: 

2370 await _check_key_model_budget_with_fallback( 

2371 valid_token=valid_token, 

2372 model_max_budget_limiter=model_max_budget_limiter, 

2373 model_name=model_name, 

2374 request_data=request_data, 

2375 request=request, 

2376 llm_model_list=llm_model_list, 

2377 llm_router=llm_router, 

2378 ) 

2379 

2380 # Recompute after a potential budget-fallback rewrite so 

2381 # the end-user check below validates the final model 

2382 current_model = _get_model_from_request_context( 

2383 request_data=request_data, 

2384 route=route, 

2385 request=request, 

2386 llm_router=llm_router, 

2387 team_id=valid_token.team_id, 

2388 ) 

2389 current_models = _get_model_names_for_budget_checks(model=current_model) 

2390 

2391 # Check 5a. Internal user model_max_budget 

2392 if current_models: 2392 ↛ 2393line 2392 didn't jump to line 2393 because the condition on line 2392 was never true

2393 await _check_user_model_budget( 

2394 valid_token=valid_token, 

2395 model_max_budget_limiter=model_max_budget_limiter, 

2396 models=current_models, 

2397 ) 

2398 

2399 # Check 5b. End-user model max budget 

2400 end_user_mmb: Final = valid_token.end_user_model_max_budget 

2401 if ( 2401 ↛ 2408line 2401 didn't jump to line 2408 because the condition on line 2401 was never true

2402 end_user_mmb is not None 

2403 and isinstance(end_user_mmb, dict) 

2404 and len(end_user_mmb) > 0 

2405 and current_models 

2406 and valid_token.end_user_id is not None 

2407 ): 

2408 for model_name in current_models: 

2409 await model_max_budget_limiter.is_end_user_within_model_budget( 

2410 end_user_id=valid_token.end_user_id, 

2411 end_user_model_max_budget=end_user_mmb, 

2412 model=model_name, 

2413 ) 

2414 

2415 # Check 6: Additional Common Checks across jwt + key auth 

2416 if valid_token.team_id is not None: 2416 ↛ 2417line 2416 didn't jump to line 2417 because the condition on line 2416 was never true

2417 try: 

2418 if valid_token.team_id == UI_TEAM_ID: 

2419 raise TeamNotFoundError(team_id=UI_TEAM_ID) 

2420 with tracer.trace("litellm.proxy.auth.get_team_object"): 

2421 _team_obj = await get_team_object( 

2422 team_id=valid_token.team_id, 

2423 prisma_client=prisma_client, 

2424 user_api_key_cache=user_api_key_cache, 

2425 parent_otel_span=parent_otel_span, 

2426 proxy_logging_obj=proxy_logging_obj, 

2427 ) 

2428 except HTTPException: 

2429 token_team_models: Final = _token_team_models(valid_token) 

2430 _team_obj = LiteLLM_TeamTableCachedObj( 

2431 team_id=valid_token.team_id, 

2432 max_budget=valid_token.team_max_budget, 

2433 soft_budget=valid_token.team_soft_budget, 

2434 model_max_budget=valid_token.team_model_max_budget, 

2435 spend=valid_token.team_spend, 

2436 tpm_limit=valid_token.team_tpm_limit, 

2437 rpm_limit=valid_token.team_rpm_limit, 

2438 tpd_limit=valid_token.team_tpd_limit, 

2439 blocked=valid_token.team_blocked, 

2440 models=token_team_models, 

2441 metadata=valid_token.team_metadata, 

2442 object_permission_id=valid_token.team_object_permission_id, 

2443 object_permission=await _resolve_object_permission_for_unresolvable_team( 

2444 object_permission_id=valid_token.team_object_permission_id, 

2445 prisma_client=prisma_client, 

2446 user_api_key_cache=user_api_key_cache, 

2447 parent_otel_span=parent_otel_span, 

2448 proxy_logging_obj=proxy_logging_obj, 

2449 ), 

2450 ) 

2451 else: 

2452 _team_obj = None 

2453 

2454 if _team_obj is not None: 2454 ↛ 2455line 2454 didn't jump to line 2455 because the condition on line 2454 was never true

2455 valid_token.team_object_permission = _team_obj.object_permission 

2456 # Keep team_metadata in sync with the freshly fetched team so that 

2457 # guardrails (or any other metadata) added after the key was cached 

2458 # are picked up on subsequent requests without a cache eviction. 

2459 valid_token.team_metadata = _team_obj.metadata 

2460 else: 

2461 valid_token.team_object_permission = None 

2462 

2463 # Fetch project object if key belongs to a project 

2464 _project_obj = None 

2465 if valid_token.project_id is not None: 2465 ↛ 2466line 2465 didn't jump to line 2466 because the condition on line 2465 was never true

2466 _project_obj = await get_project_object( 

2467 project_id=valid_token.project_id, 

2468 prisma_client=prisma_client, 

2469 user_api_key_cache=user_api_key_cache, 

2470 proxy_logging_obj=proxy_logging_obj, 

2471 ) 

2472 if _project_obj is not None: 

2473 valid_token.project_metadata = _project_obj.metadata 

2474 valid_token.project_alias = _project_obj.project_alias 

2475 

2476 global_proxy_spend = None 

2477 if litellm.max_budget > 0 and prisma_client is not None: # user set proxy max budget 2477 ↛ 2478line 2477 didn't jump to line 2478 because the condition on line 2477 was never true

2478 cache_key: Final = GLOBAL_PROXY_SPEND_CACHE_KEY 

2479 with tracer.trace("litellm.proxy.auth.get_global_proxy_spend"): 

2480 global_proxy_spend = await _fetch_global_spend_with_event_coordination( 

2481 cache_key=cache_key, 

2482 user_api_key_cache=user_api_key_cache, 

2483 prisma_client=prisma_client, 

2484 ) 

2485 

2486 if global_proxy_spend is not None: 

2487 call_info: Final = CallInfo( 

2488 token=valid_token.token, 

2489 spend=global_proxy_spend, 

2490 max_budget=litellm.max_budget, 

2491 user_id=litellm_proxy_admin_name, 

2492 team_id=valid_token.team_id, 

2493 event_group=Litellm_EntityType.PROXY, 

2494 ) 

2495 asyncio.create_task( 

2496 proxy_logging_obj.budget_alerts( 

2497 type="proxy_budget", 

2498 user_info=call_info, 

2499 ) 

2500 ) 

2501 # Token passed all checks 

2502 if valid_token is None: 2502 ↛ 2503line 2502 didn't jump to line 2503 because the condition on line 2502 was never true

2503 raise HTTPException(401, detail="Invalid API key") 

2504 if valid_token.token is None: 2504 ↛ 2505line 2504 didn't jump to line 2505 because the condition on line 2504 was never true

2505 raise HTTPException(401, detail="Invalid API key, no token associated") 

2506 api_key = valid_token.token 

2507 

2508 valid_token_dict = valid_token.model_dump(exclude_none=True) 

2509 valid_token_dict.pop("token", None) 

2510 # budget_throttle_pct is excluded from model_dump (it must not leak 

2511 # into serialized responses), so carry the request-scoped decision 

2512 # forward by hand to the auth object the rate limiter receives. 

2513 if valid_token.budget_throttle_pct is not None: 2513 ↛ 2514line 2513 didn't jump to line 2514 because the condition on line 2513 was never true

2514 valid_token_dict["budget_throttle_pct"] = valid_token.budget_throttle_pct 

2515 

2516 if _end_user_object is not None: 2516 ↛ 2517line 2516 didn't jump to line 2517 because the condition on line 2516 was never true

2517 valid_token_dict.update(end_user_params) 

2518 valid_token_dict["end_user_object_permission"] = _end_user_object.object_permission 

2519 

2520 # check if token is from litellm-ui, litellm ui makes keys to allow users to login with sso. These keys can only be used for LiteLLM UI functions 

2521 # sso/login, ui/login, /key functions and /user functions 

2522 # this will never be allowed to call /chat/completions 

2523 

2524 if valid_token is None: 2524 ↛ 2526line 2524 didn't jump to line 2526 because the condition on line 2524 was never true

2525 # No token was found when looking up in the DB 

2526 raise Exception("Invalid proxy server token passed") 

2527 if valid_token_dict is not None: 2527 ↛ exitline 2527 didn't return from function '_user_api_key_auth_builder' because the condition on line 2527 was always true

2528 virtual_key_auth_obj: Final = await _return_user_api_key_auth_obj( 

2529 user_obj=user_obj, 

2530 api_key=api_key, 

2531 parent_otel_span=parent_otel_span, 

2532 valid_token_dict=valid_token_dict, 

2533 route=route, 

2534 start_time=start_time, 

2535 ) 

2536 virtual_key_auth_obj.via_virtual_key = True 

2537 return virtual_key_auth_obj 

2538 except Exception as e: 

2539 return await UserAPIKeyAuthExceptionHandler._handle_authentication_error( 

2540 e=e, 

2541 request=request, 

2542 request_data=request_data, 

2543 route=route, 

2544 parent_otel_span=parent_otel_span, 

2545 api_key=api_key, 

2546 resolved_identity=valid_token, 

2547 ) 

2548 

2549 

2550async def _safe_fetch(label: str, awaitable): 

2551 """Run an awaitable and return its result. Re-raises authentication / 

2552 authorization failures (HTTPException, ProxyException, 

2553 BudgetExceededError) so they propagate to the caller. 

2554 Other exceptions (e.g. transient DB errors fetching context) are 

2555 swallowed with a debug log and ``None`` is returned so 

2556 ``common_checks`` can still run against whatever limits are recorded 

2557 directly on the token. 

2558 """ 

2559 try: 

2560 return await awaitable 

2561 except (HTTPException, ProxyException, litellm.BudgetExceededError) as e: 

2562 verbose_proxy_logger.debug( 

2563 "centralized auth: %s fetch failed (%s: %s)", 

2564 label, 

2565 type(e).__name__, 

2566 e, 

2567 ) 

2568 raise 

2569 except Exception as e: 

2570 verbose_proxy_logger.debug( 

2571 "centralized auth: %s fetch swallowed (%s: %s)", 

2572 label, 

2573 type(e).__name__, 

2574 e, 

2575 ) 

2576 return None 

2577 

2578 

2579def _team_obj_from_token(valid_token: UserAPIKeyAuth) -> LiteLLM_TeamTableCachedObj: 

2580 """Reconstruct a cached team object from the fields already on the 

2581 UserAPIKeyAuth. Only called when valid_token.team_id is known to be 

2582 non-None (the caller gates on it).""" 

2583 assert valid_token.team_id is not None 

2584 token_team_models: Final = _token_team_models(valid_token) 

2585 return LiteLLM_TeamTableCachedObj( 

2586 team_id=valid_token.team_id, 

2587 max_budget=valid_token.team_max_budget, 

2588 soft_budget=valid_token.team_soft_budget, 

2589 model_max_budget=valid_token.team_model_max_budget, 

2590 spend=valid_token.team_spend, 

2591 tpm_limit=valid_token.team_tpm_limit, 

2592 rpm_limit=valid_token.team_rpm_limit, 

2593 tpd_limit=valid_token.team_tpd_limit, 

2594 blocked=valid_token.team_blocked, 

2595 models=token_team_models, 

2596 metadata=valid_token.team_metadata, 

2597 object_permission_id=valid_token.team_object_permission_id, 

2598 ) 

2599 

2600 

2601def _token_can_vouch_for_team(valid_token: UserAPIKeyAuth, lookup_error: BaseException) -> bool: 

2602 """Whether the token's own team fields may stand in for a team that failed to 

2603 resolve, without widening access. 

2604 

2605 The UI dashboard mints every session key against the ``UI_TEAM_ID`` sentinel, 

2606 which by design never has a team row, so a failed lookup for it is not a 

2607 degraded read to be treated with suspicion; it always vouches, exactly as it 

2608 always safely has (these keys are restricted elsewhere to UI-only routes). 

2609 

2610 For every other team, a team that is provably gone is a definitive answer, 

2611 not a degraded read, so nothing may stand in for it and no setting may 

2612 override that. 

2613 

2614 Otherwise the team's grant is merely unknown. A token carrying one may vouch, 

2615 since replaying a recorded grant cannot widen it and denying every team key 

2616 while the row is briefly unreadable would trade the widening for an outage. A 

2617 token carrying none may not: ``team_models=[]`` reads as every model and 

2618 ``team_blocked=False`` as unblocked. ``allow_requests_on_db_unavailable`` opts 

2619 back out, and is only consulted here because the failure is known by this 

2620 point to be a degraded read. 

2621 """ 

2622 if valid_token.team_id == UI_TEAM_ID: 

2623 return True 

2624 if isinstance(lookup_error, TeamNotFoundError): 

2625 return False 

2626 if valid_token.team_models: 

2627 return True 

2628 return PrismaDBExceptionHandler.should_allow_request_on_db_unavailable() 

2629 

2630 

2631async def _inherit_org_identity( 

2632 user_api_key_auth_obj: UserAPIKeyAuth, 

2633 team_object: LiteLLM_TeamTableCachedObj | None, 

2634 prisma_client: PrismaClient | None, 

2635 user_api_key_cache: UserApiKeyCache, 

2636 parent_otel_span: Span | None, 

2637 proxy_logging_obj: ProxyLogging | None, 

2638) -> None: 

2639 if user_api_key_auth_obj.org_id is None and team_object is not None and team_object.organization_id is not None: 2639 ↛ 2640line 2639 didn't jump to line 2640 because the condition on line 2639 was never true

2640 user_api_key_auth_obj.org_id = team_object.organization_id 

2641 already_populated: Final = any( 

2642 value is not None 

2643 for value in ( 

2644 user_api_key_auth_obj.organization_alias, 

2645 user_api_key_auth_obj.organization_max_budget, 

2646 user_api_key_auth_obj.organization_tpm_limit, 

2647 user_api_key_auth_obj.organization_rpm_limit, 

2648 user_api_key_auth_obj.organization_metadata, 

2649 ) 

2650 ) 

2651 if user_api_key_auth_obj.org_id is None or already_populated or prisma_client is None: 2651 ↛ 2653line 2651 didn't jump to line 2653 because the condition on line 2651 was always true

2652 return 

2653 org_object: Final = await get_org_object_for_request( 

2654 org_id=user_api_key_auth_obj.org_id, 

2655 prisma_client=prisma_client, 

2656 user_api_key_cache=user_api_key_cache, 

2657 parent_otel_span=parent_otel_span, 

2658 proxy_logging_obj=proxy_logging_obj, 

2659 ) 

2660 if org_object is None: 

2661 return 

2662 user_api_key_auth_obj.organization_alias = org_object.organization_alias 

2663 user_api_key_auth_obj.organization_metadata = org_object.metadata 

2664 budget: Final = org_object.litellm_budget_table 

2665 if budget is None: 

2666 return 

2667 user_api_key_auth_obj.organization_max_budget = budget.max_budget 

2668 user_api_key_auth_obj.organization_tpm_limit = budget.tpm_limit 

2669 user_api_key_auth_obj.organization_rpm_limit = budget.rpm_limit 

2670 

2671 

2672def is_no_auth_dev_mode(master_key: str | None, general_settings: Mapping[str, object]) -> bool: 

2673 return master_key is None and not any( 

2674 general_settings.get(flag, False) 

2675 for flag in ("enable_jwt_auth", "enable_oauth2_auth", "enable_oauth2_proxy_auth") 

2676 ) 

2677 

2678 

2679@tracer.wrap() 

2680async def _run_centralized_common_checks( 

2681 user_api_key_auth_obj: UserAPIKeyAuth, 

2682 request: Request, 

2683 request_data: dict[str, object], 

2684 route: str, 

2685) -> None: 

2686 """Run ``common_checks`` once at the ``user_api_key_auth`` wrapper 

2687 boundary, regardless of which ``_user_api_key_auth_builder`` path 

2688 returned. This is the single invariant enforcement point for key 

2689 model-access, budgets, guardrails, org, and vector-store checks. 

2690 

2691 Invariants: 

2692 - ``user_custom_auth`` with ``custom_auth_run_common_checks`` unset 

2693 skips the gate — matches the existing custom-auth RPS guarantee. 

2694 Custom-auth deployments don't use OAuth2 / DB-fallback paths, so 

2695 the skip does not re-open any bypass. 

2696 - ``PROXY_ADMIN`` tokens still run through ``common_checks`` so 

2697 team-blocked / team-budget / end-user-budget / tag-budget / 

2698 vector-store / tool-allowlist enforcement applies to admin keys 

2699 too. Admin status is honored where the underlying check exempts it 

2700 (``_is_api_route_allowed``, ``organization_role_based_access_check``). 

2701 """ 

2702 from litellm.proxy.proxy_server import ( 

2703 general_settings, 

2704 litellm_proxy_admin_name, 

2705 llm_router, 

2706 master_key, 

2707 model_max_budget_limiter, 

2708 prisma_client, 

2709 proxy_logging_obj, 

2710 user_api_key_cache, 

2711 user_custom_auth, 

2712 ) 

2713 

2714 # Public routes (e.g. /health/liveness) are exempt from 

2715 # auth in the builder — the wrapper must not retroactively apply 

2716 # authz on top, or k8s readiness probes and other unauthenticated 

2717 # callers get 401. 

2718 if route in LiteLLMRoutes.public_routes.value or route_in_additonal_public_routes(current_route=route): 

2719 return 

2720 

2721 # User-configured pass-through endpoints with ``auth: false`` are 

2722 # explicitly unauthenticated — the builder returns an empty 

2723 # UserAPIKeyAuth() and the request is forwarded as-is. Running 

2724 # common_checks on the empty token would reject the request as 

2725 # admin-only. The "auth" flag on the endpoint config is the 

2726 # contract; honor it. 

2727 pass_through_endpoints: Final = general_settings.get("pass_through_endpoints", None) 

2728 if pass_through_endpoints is not None: 

2729 for endpoint in pass_through_endpoints: 

2730 if isinstance(endpoint, dict) and endpoint.get("path", "") == route and endpoint.get("auth") is not True: 2730 ↛ 2731line 2730 didn't jump to line 2731 because the condition on line 2730 was never true

2731 return 

2732 

2733 # No-auth dev mode: master_key unset AND no JWT/OAuth2 auth 

2734 # configured. The builder returns an INTERNAL_USER token for any 

2735 # api_key; the proxy is unauthenticated by configuration. 

2736 # Running common_checks would block every admin route on these 

2737 # deployments where that was previously not the contract. If any 

2738 # authn is enabled (JWT, OAuth2, OAuth2-proxy), authz must run. 

2739 if is_no_auth_dev_mode(master_key, general_settings): 2739 ↛ 2740line 2739 didn't jump to line 2740 because the condition on line 2739 was never true

2740 return 

2741 

2742 if user_custom_auth is not None and not general_settings.get("custom_auth_run_common_checks", False): 2742 ↛ 2743line 2742 didn't jump to line 2743 because the condition on line 2742 was never true

2743 return 

2744 

2745 parent_otel_span: Final = user_api_key_auth_obj.parent_otel_span 

2746 # In the integrated auth flow ``_user_api_key_auth_builder`` has already 

2747 # resolved the end-user id and attached it here. Reuse that to avoid a 

2748 # second extraction pass; fall back to extracting locally when the 

2749 # function is invoked in isolation (e.g. in direct unit tests). 

2750 key_end_user_budget_id: Final = get_key_end_user_budget_id(user_api_key_auth_obj.metadata) 

2751 end_user_id = user_api_key_auth_obj.end_user_id 

2752 if end_user_id is None: 

2753 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request)) 

2754 end_user_id = await resolve_and_validate_end_user_id( 

2755 raw_end_user_id=raw_end_user_id, 

2756 prisma_client=prisma_client, 

2757 user_api_key_cache=user_api_key_cache, 

2758 parent_otel_span=parent_otel_span, 

2759 proxy_logging_obj=proxy_logging_obj, 

2760 route=route, 

2761 key_end_user_budget_id=key_end_user_budget_id, 

2762 ) 

2763 if end_user_id is not None and key_end_user_budget_id is not None: 2763 ↛ 2764line 2763 didn't jump to line 2764 because the condition on line 2763 was never true

2764 user_api_key_auth_obj.end_user_id = end_user_id 

2765 

2766 fetch_coros: Final = [] 

2767 if user_api_key_auth_obj.team_id is not None and user_api_key_auth_obj.team_id != UI_TEAM_ID: 2767 ↛ 2768line 2767 didn't jump to line 2768 because the condition on line 2767 was never true

2768 fetch_coros.append( 

2769 _safe_fetch( 

2770 "team", 

2771 get_team_object( 

2772 team_id=user_api_key_auth_obj.team_id, 

2773 prisma_client=prisma_client, 

2774 user_api_key_cache=user_api_key_cache, 

2775 parent_otel_span=parent_otel_span, 

2776 proxy_logging_obj=proxy_logging_obj, 

2777 ), 

2778 ) 

2779 ) 

2780 else: 

2781 fetch_coros.append(_safe_fetch("team", _noop_none())) 

2782 

2783 if user_api_key_auth_obj.user_id is not None: 

2784 fetch_coros.append( 

2785 _safe_fetch( 

2786 "user", 

2787 get_user_object( 

2788 user_id=user_api_key_auth_obj.user_id, 

2789 prisma_client=prisma_client, 

2790 user_api_key_cache=user_api_key_cache, 

2791 user_id_upsert=False, 

2792 parent_otel_span=parent_otel_span, 

2793 proxy_logging_obj=proxy_logging_obj, 

2794 ), 

2795 ) 

2796 ) 

2797 else: 

2798 fetch_coros.append(_safe_fetch("user", _noop_none())) 

2799 

2800 if user_api_key_auth_obj.project_id is not None: 2800 ↛ 2801line 2800 didn't jump to line 2801 because the condition on line 2800 was never true

2801 fetch_coros.append( 

2802 _safe_fetch( 

2803 "project", 

2804 get_project_object( 

2805 project_id=user_api_key_auth_obj.project_id, 

2806 prisma_client=prisma_client, 

2807 user_api_key_cache=user_api_key_cache, 

2808 proxy_logging_obj=proxy_logging_obj, 

2809 ), 

2810 ) 

2811 ) 

2812 else: 

2813 fetch_coros.append(_safe_fetch("project", _noop_none())) 

2814 

2815 if end_user_id: 

2816 fetch_coros.append( 

2817 _safe_fetch( 

2818 "end_user", 

2819 get_end_user_object( 

2820 end_user_id=end_user_id, 

2821 prisma_client=prisma_client, 

2822 user_api_key_cache=user_api_key_cache, 

2823 parent_otel_span=parent_otel_span, 

2824 proxy_logging_obj=proxy_logging_obj, 

2825 route=route, 

2826 token_end_user_max_budget=user_api_key_auth_obj.end_user_max_budget, 

2827 key_end_user_budget_id=key_end_user_budget_id, 

2828 ), 

2829 ) 

2830 ) 

2831 else: 

2832 fetch_coros.append(_safe_fetch("end_user", _noop_none())) 

2833 

2834 fetch_coros.append( 

2835 _safe_fetch( 

2836 "global_spend", 

2837 get_global_proxy_spend( 

2838 litellm_proxy_admin_name=litellm_proxy_admin_name, 

2839 user_api_key_cache=user_api_key_cache, 

2840 prisma_client=prisma_client, 

2841 token=user_api_key_auth_obj.token or "", 

2842 proxy_logging_obj=proxy_logging_obj, 

2843 ), 

2844 ) 

2845 ) 

2846 

2847 # Per-fetch error isolation. ``_safe_fetch`` lets HTTPException, 

2848 # ProxyException, and BudgetExceededError escape (everything else is 

2849 # already swallowed to None). A bare ``except`` over ``gather`` would 

2850 # let one fetch's HTTPException null out every other context — e.g. 

2851 # a 404 from ``get_team_object`` (token references a deleted team) 

2852 # would silently skip the user, end-user, project, and global-spend 

2853 # checks. Use ``return_exceptions=True`` and apply per-fetch fallback 

2854 # so a missing team only zeros out the team object. 

2855 ( 

2856 team_result, 

2857 user_result, 

2858 project_result, 

2859 end_user_result, 

2860 global_spend_result, 

2861 ) = await asyncio.gather(*fetch_coros, return_exceptions=True) 

2862 

2863 # ProxyException / BudgetExceededError are authorization failures — 

2864 # propagate so the wrapper renders them. HTTPException is fallback 

2865 # material (404 from get_team_object is the only known producer). 

2866 for r in ( 

2867 team_result, 

2868 user_result, 

2869 project_result, 

2870 end_user_result, 

2871 global_spend_result, 

2872 ): 

2873 if isinstance(r, (ProxyException, litellm.BudgetExceededError)): 2873 ↛ 2874line 2873 didn't jump to line 2874 because the condition on line 2873 was never true

2874 raise r 

2875 

2876 # Use BaseException (not HTTPException) in the narrowing checks so 

2877 # mypy can narrow ``Any | BaseException`` to the typed object in the 

2878 # else branch. After the for-loop above, the only BaseException that 

2879 # can still appear here is HTTPException (other listed re-raises were 

2880 # propagated; non-listed exceptions were already swallowed to None). 

2881 team_object: LiteLLM_TeamTableCachedObj | None 

2882 if isinstance(team_result, BaseException): 2882 ↛ 2885line 2882 didn't jump to line 2885 because the condition on line 2882 was never true

2883 # Token-derived fallback only valid when a team_id is set; 

2884 # _team_obj_from_token asserts that precondition. 

2885 if user_api_key_auth_obj.team_id is None: 

2886 team_object = None 

2887 elif _token_can_vouch_for_team(user_api_key_auth_obj, team_result): 

2888 team_object = _team_obj_from_token(user_api_key_auth_obj) 

2889 else: 

2890 raise team_result 

2891 else: 

2892 team_object = ( 

2893 _team_obj_from_token(user_api_key_auth_obj) if user_api_key_auth_obj.team_id == UI_TEAM_ID else team_result 

2894 ) 

2895 

2896 user_object: LiteLLM_UserTable | None = None if isinstance(user_result, BaseException) else user_result 

2897 project_object: Final[LiteLLM_ProjectTableCachedObj | None] = ( 

2898 None if isinstance(project_result, BaseException) else project_result 

2899 ) 

2900 end_user_object: Final[LiteLLM_EndUserTable | None] = ( 

2901 None if isinstance(end_user_result, BaseException) else end_user_result 

2902 ) 

2903 global_proxy_spend: float | None = None if isinstance(global_spend_result, BaseException) else global_spend_result 

2904 carry_team_and_user_budget_state( 

2905 valid_token=user_api_key_auth_obj, 

2906 team_object=team_object, 

2907 user_object=user_object, 

2908 ) 

2909 

2910 await _inherit_org_identity( 

2911 user_api_key_auth_obj=user_api_key_auth_obj, 

2912 team_object=team_object, 

2913 prisma_client=prisma_client, 

2914 user_api_key_cache=user_api_key_cache, 

2915 parent_otel_span=parent_otel_span, 

2916 proxy_logging_obj=proxy_logging_obj, 

2917 ) 

2918 

2919 # common_checks identifies admin via user_object, not the token 

2920 # (non_proxy_admin_allowed_routes_check). JWT admin shortcut and 

2921 # master_key tokens get admin from the token; the DB row for the 

2922 # same user_id (e.g. litellm_proxy_admin_name = "default_user_id") 

2923 # may have a non-admin user_role and would otherwise demote the 

2924 # caller. The token is the source of truth for these paths — force 

2925 # the admin user_object whenever the token says PROXY_ADMIN, even 

2926 # if a DB row was fetched. 

2927 if user_api_key_auth_obj.user_role == LitellmUserRoles.PROXY_ADMIN: 

2928 user_object = LiteLLM_UserTable( 

2929 user_id=user_api_key_auth_obj.user_id or litellm_proxy_admin_name, 

2930 user_role=LitellmUserRoles.PROXY_ADMIN, 

2931 spend=user_object.spend if user_object is not None else 0.0, 

2932 ) 

2933 

2934 if project_object is not None: 2934 ↛ 2935line 2934 didn't jump to line 2935 because the condition on line 2934 was never true

2935 user_api_key_auth_obj.project_metadata = project_object.metadata 

2936 user_api_key_auth_obj.project_alias = project_object.project_alias 

2937 

2938 if end_user_id and key_end_user_budget_id is not None and prisma_client is not None: 2938 ↛ 2939line 2938 didn't jump to line 2939 because the condition on line 2938 was never true

2939 await _apply_key_end_user_default_budget_to_token( 

2940 valid_token=user_api_key_auth_obj, 

2941 end_user_object=end_user_object, 

2942 key_end_user_budget_id=key_end_user_budget_id, 

2943 prisma_client=prisma_client, 

2944 user_api_key_cache=user_api_key_cache, 

2945 parent_otel_span=parent_otel_span, 

2946 keep_token_limits=user_custom_auth is not None, 

2947 ) 

2948 

2949 skip_budget_checks: Final = _should_skip_budget_checks( 

2950 request_data=request_data, 

2951 route=route, 

2952 request=request, 

2953 llm_router=llm_router, 

2954 team_id=user_api_key_auth_obj.team_id, 

2955 ) 

2956 

2957 # Pin the metadata variable name (litellm_metadata vs metadata) before 

2958 # any tag merge runs. Without this, header tags from 

2959 # apply_client_tag_policy_pre_auth would land in `metadata` while the 

2960 # later seed in common_checks pushes key tags and the 

2961 # _tag_max_budget_check read into `litellm_metadata`, hiding header 

2962 # tags from per-tag budget enforcement on LITELLM_METADATA_ROUTES. 

2963 LiteLLMProxyRequestSetup.pre_seed_litellm_metadata_for_route( 

2964 request_data=request_data, 

2965 route=route, 

2966 ) 

2967 

2968 # Merge x-litellm-tags into request_data BEFORE common_checks runs. 

2969 # _tag_max_budget_check inside common_checks only inspects request_data; 

2970 # without this pre-merge, header-supplied tags bypass tag-budget 

2971 # enforcement. 

2972 LiteLLMProxyRequestSetup.apply_client_tag_policy_pre_auth( 

2973 request=request, 

2974 request_data=request_data, 

2975 user_api_key_dict=user_api_key_auth_obj, 

2976 ) 

2977 

2978 bind_admission_counter_keys(user_api_key_auth_obj, end_user_id=end_user_id) 

2979 try: 

2980 _ = await common_checks( 

2981 request=request, 

2982 request_body=request_data, 

2983 team_object=team_object, 

2984 user_object=user_object, 

2985 end_user_object=end_user_object, 

2986 general_settings=general_settings, 

2987 global_proxy_spend=global_proxy_spend, 

2988 route=route, 

2989 llm_router=llm_router, 

2990 proxy_logging_obj=proxy_logging_obj, 

2991 valid_token=user_api_key_auth_obj, 

2992 skip_budget_checks=skip_budget_checks, 

2993 project_object=project_object, 

2994 ) 

2995 finally: 

2996 release_spend_counter_batch() 

2997 

2998 if not skip_budget_checks: 2998 ↛ 3013line 2998 didn't jump to line 3013 because the condition on line 2998 was always true

2999 await _check_team_model_budget( 

3000 valid_token=user_api_key_auth_obj, 

3001 model_max_budget_limiter=model_max_budget_limiter, 

3002 models=_get_model_names_for_budget_checks( 

3003 model=_get_model_from_request_context( 

3004 request_data=request_data, 

3005 route=route, 

3006 request=request, 

3007 llm_router=llm_router, 

3008 team_id=user_api_key_auth_obj.team_id, 

3009 ) 

3010 ), 

3011 ) 

3012 

3013 await _reserve_budget_after_common_checks( 

3014 user_api_key_auth_obj=user_api_key_auth_obj, 

3015 request=request, 

3016 request_data=request_data, 

3017 route=route, 

3018 llm_router=llm_router, 

3019 team_object=team_object, 

3020 user_object=user_object, 

3021 end_user_id=end_user_id, 

3022 end_user_object=end_user_object, 

3023 prisma_client=prisma_client, 

3024 user_api_key_cache=user_api_key_cache, 

3025 proxy_logging_obj=proxy_logging_obj, 

3026 skip_budget_checks=skip_budget_checks, 

3027 general_settings=general_settings, 

3028 ) 

3029 

3030 

3031async def _noop_none() -> None: 

3032 """Sentinel coroutine for asyncio.gather when a fetch is unnecessary 

3033 (e.g. token has no team_id). Keeps the result tuple positional.""" 

3034 return 

3035 

3036 

3037async def _apply_key_end_user_default_budget_to_token( 

3038 valid_token: UserAPIKeyAuth, 

3039 end_user_object: LiteLLM_EndUserTable | None, 

3040 key_end_user_budget_id: str, 

3041 prisma_client: PrismaClient, 

3042 user_api_key_cache: UserApiKeyCache, 

3043 parent_otel_span: Span | None, 

3044 keep_token_limits: bool, 

3045) -> None: 

3046 """The builder's end-user pass runs before the key is resolved, so only here can the key's 

3047 ``end_user_budget_id`` win over the proxy-wide default on the token that reservation reads. 

3048 On the virtual-key path the token's end-user limits are the builder's proxy-wide defaults and 

3049 the key budget replaces them wholesale. With ``keep_token_limits`` (custom auth) the token's 

3050 limits are caps the custom auth callable set, so the key budget only fills the ones it left 

3051 unset.""" 

3052 default_budget: Final = ( 

3053 end_user_object.litellm_budget_table 

3054 if end_user_object is not None 

3055 else await resolve_default_end_user_budget( 

3056 prisma_client=prisma_client, 

3057 user_api_key_cache=user_api_key_cache, 

3058 key_end_user_budget_id=key_end_user_budget_id, 

3059 parent_otel_span=parent_otel_span, 

3060 ) 

3061 ) 

3062 if default_budget is None: 

3063 return 

3064 

3065 if not keep_token_limits or valid_token.end_user_max_budget is None: 

3066 valid_token.end_user_max_budget = default_budget.max_budget 

3067 if not keep_token_limits or valid_token.end_user_tpm_limit is None: 

3068 valid_token.end_user_tpm_limit = default_budget.tpm_limit 

3069 if not keep_token_limits or valid_token.end_user_rpm_limit is None: 

3070 valid_token.end_user_rpm_limit = default_budget.rpm_limit 

3071 if not keep_token_limits or valid_token.end_user_tpd_limit is None: 

3072 valid_token.end_user_tpd_limit = default_budget.tpd_limit 

3073 if not keep_token_limits or valid_token.end_user_model_max_budget is None: 

3074 valid_token.end_user_model_max_budget = default_budget.model_max_budget 

3075 

3076 

3077async def _reserve_budget_after_common_checks( 

3078 user_api_key_auth_obj: UserAPIKeyAuth, 

3079 request_data: dict, 

3080 route: str, 

3081 llm_router: Any | None, 

3082 team_object: LiteLLM_TeamTableCachedObj | None, 

3083 user_object: LiteLLM_UserTable | None, 

3084 prisma_client: PrismaClient | None, 

3085 user_api_key_cache: UserApiKeyCache, 

3086 proxy_logging_obj: ProxyLogging, 

3087 skip_budget_checks: bool, 

3088 general_settings: dict, 

3089 end_user_id: str | None = None, 

3090 end_user_object: LiteLLM_EndUserTable | None = None, 

3091 request: Request | None = None, 

3092) -> None: 

3093 user_api_key_auth_obj.budget_reservation = None 

3094 if not skip_budget_checks and general_settings.get("disable_budget_reservation") is not True: 3094 ↛ 3115line 3094 didn't jump to line 3115 because the condition on line 3094 was always true

3095 from litellm.proxy.spend_tracking.budget_reservation import ( 

3096 reserve_budget_for_request, 

3097 ) 

3098 

3099 user_api_key_auth_obj.budget_reservation = await reserve_budget_for_request( 

3100 request_body=request_data, 

3101 route=route, 

3102 llm_router=llm_router, 

3103 valid_token=user_api_key_auth_obj, 

3104 team_object=team_object, 

3105 user_object=user_object, 

3106 prisma_client=prisma_client, 

3107 user_api_key_cache=user_api_key_cache, 

3108 proxy_logging_obj=proxy_logging_obj, 

3109 end_user_id=end_user_id, 

3110 end_user_object=end_user_object, 

3111 apply_user_budget_to_team_keys=general_settings.get("apply_user_budget_to_team_keys") is True, 

3112 fail_closed_budget_enforcement=general_settings.get("fail_closed_budget_enforcement") is True, 

3113 raw_body=await read_raw_json_body(request=request), 

3114 ) 

3115 if request is not None: 3115 ↛ exitline 3115 didn't return from function '_reserve_budget_after_common_checks' because the condition on line 3115 was always true

3116 reservation: Final = user_api_key_auth_obj.budget_reservation 

3117 request.state.budget_reservation = reservation # rebind-ok: read by the release middleware 

3118 

3119 

3120def _should_skip_budget_checks( 

3121 request_data: dict, 

3122 route: str, 

3123 request: Request | None, 

3124 llm_router: Any | None, 

3125 team_id: str | None = None, 

3126) -> bool: 

3127 model: Final = _get_model_from_request_context( 

3128 request_data=request_data, 

3129 route=route, 

3130 request=request, 

3131 llm_router=llm_router, 

3132 team_id=team_id, 

3133 ) 

3134 if model is not None and llm_router is not None: 

3135 return _is_model_cost_zero(model=model, llm_router=llm_router) 

3136 return False 

3137 

3138 

3139def _resolve_request_principal(request: Request, valid_token: UserAPIKeyAuth) -> Principal: 

3140 """Project the resolved identity into one per-request Principal, off the key 

3141 object the builder already fetched, and stamp the request network context 

3142 onto it once. X-Forwarded-For is only trusted when the operator configured 

3143 ``trusted_proxy_ranges``; otherwise the direct peer is authoritative. 

3144 

3145 credential_ref and a stable subject fallback are always set off the token so 

3146 the Principal can never be anonymous, even for a keyless service-account key 

3147 with no user or alias.""" 

3148 cidrs: Final = get_trusted_proxy_cidrs() 

3149 network: Final = resolve_network_context( 

3150 request, 

3151 TrustedProxyConfig(use_forwarded_for=bool(cidrs), trusted_proxy_cidrs=cidrs), 

3152 ) 

3153 auth_method: Final = AuthMethod.BEARER_JWT if valid_token.jwt_claims else AuthMethod.API_KEY 

3154 return IdentityStore._principal_from_key( 

3155 valid_token, 

3156 auth_method=auth_method, 

3157 network=network, 

3158 subject_fallback=valid_token.token, 

3159 credential_ref=CredentialRef(token_id=valid_token.token), 

3160 ) 

3161 

3162 

3163async def _authorize_authenticated_request( 

3164 user_api_key_auth_obj: UserAPIKeyAuth, 

3165 request: Request, 

3166 request_data: dict, 

3167 route: str, 

3168 api_key: str, 

3169) -> UserAPIKeyAuth | None: 

3170 """Authorize an already-authenticated request: disabled-route check, the single 

3171 ``common_checks`` gate (which also reserves budget), and end-user fallback 

3172 resolution. Returns the auth object the exception handler recovered when a check 

3173 failed but the request may proceed anyway, else ``None``. 

3174 """ 

3175 ## ENSURE DISABLE ROUTE WORKS ACROSS ALL USER AUTH FLOWS ## 

3176 RouteChecks.should_call_route(route=route, valid_token=user_api_key_auth_obj, request=request) 

3177 await _normalize_claude_model(request_data, user_api_key_auth_obj, request, route) 

3178 await _resolve_router_settings_model_group_alias(request_data, user_api_key_auth_obj, request, route) 

3179 

3180 # Single authorization point. Builder paths MUST NOT call common_checks. 

3181 # Route through the same exception handler the builder uses so 

3182 # authorization failures (ProxyException, or plain Exception from 

3183 # admin-only-route / model-access / budget checks) surface as 

3184 # ProxyException consistently with pre-refactor behavior. 

3185 try: 

3186 await _run_centralized_common_checks( 

3187 user_api_key_auth_obj=user_api_key_auth_obj, 

3188 request=request, 

3189 request_data=request_data, 

3190 route=route, 

3191 ) 

3192 except Exception as e: 

3193 return await UserAPIKeyAuthExceptionHandler._handle_authentication_error( 

3194 e=e, 

3195 request=request, 

3196 request_data=request_data, 

3197 route=route, 

3198 parent_otel_span=user_api_key_auth_obj.parent_otel_span, 

3199 api_key=api_key, 

3200 resolved_identity=user_api_key_auth_obj, 

3201 ) 

3202 

3203 # Defense-in-depth: ``_user_api_key_auth_builder`` has multiple early-return 

3204 # paths (no master key, /user/auth route, JWT short-circuits) that bypass 

3205 # the end-user resolution block. If those paths produced an auth obj 

3206 # without an ``end_user_id`` set, fall back to extracting from the request 

3207 # body so spend logs are still attributed correctly. Validation honours 

3208 # ``litellm.validate_end_user_id_in_db``. 

3209 if user_api_key_auth_obj.end_user_id is None: 

3210 from litellm.proxy.proxy_server import ( 

3211 prisma_client, 

3212 proxy_logging_obj, 

3213 user_api_key_cache, 

3214 ) 

3215 

3216 raw_end_user_id: Final = get_end_user_id_from_request_body(request_data, _safe_get_request_headers(request)) 

3217 if raw_end_user_id is not None: 3217 ↛ 3218line 3217 didn't jump to line 3218 because the condition on line 3217 was never true

3218 resolved_end_user_id: Final = await resolve_and_validate_end_user_id( 

3219 raw_end_user_id=raw_end_user_id, 

3220 prisma_client=prisma_client, 

3221 user_api_key_cache=user_api_key_cache, 

3222 parent_otel_span=user_api_key_auth_obj.parent_otel_span, 

3223 proxy_logging_obj=proxy_logging_obj, 

3224 route=route, 

3225 key_end_user_budget_id=get_key_end_user_budget_id(user_api_key_auth_obj.metadata), 

3226 ) 

3227 if resolved_end_user_id is not None: 

3228 user_api_key_auth_obj.end_user_id = resolved_end_user_id 

3229 return None 

3230 

3231 

3232def _spend_counter_redis_cache() -> RedisCache | None: 

3233 from litellm.proxy.proxy_server import spend_counter_cache 

3234 

3235 return spend_counter_cache.redis_cache 

3236 

3237 

3238async def _prefetch_referenced_auth_objects( 

3239 valid_token: UserAPIKeyAuth, 

3240 end_user_id: str | None, 

3241 user_api_key_cache: UserApiKeyCache, 

3242 prisma_client: PrismaClient | None, 

3243) -> None: 

3244 """Warm every object and spend counter the checks below will read, in one MGET each (one DB query when cold). 

3245 Runs after the key's model access check so a denied request costs no more than it did before.""" 

3246 bind_admission_counter_keys(valid_token, end_user_id=end_user_id or None) 

3247 await prefetch_auth_objects( 

3248 refs=AuthObjectRefs.from_token(valid_token), 

3249 user_api_key_cache=user_api_key_cache, 

3250 prisma_client=prisma_client, 

3251 ) 

3252 

3253 

3254def _seed_request_destinations(user_api_key_dict: UserAPIKeyAuth, request: Request | None = None) -> None: 

3255 """Anchor the OTLP destinations this key or team overrides its traces to. 

3256 

3257 Called inside the ``auth`` phase span so that span reaches the tenant's account 

3258 as well, and on the request task so the ``ContextVar`` is inherited by the logging 

3259 tasks that close the LLM span. Best-effort: trace routing must never fail auth. 

3260 

3261 ``request`` carries the headers, so a backend this request disabled with 

3262 ``x-litellm-disable-callbacks`` resolves to no destination. 

3263 

3264 Only destinations the published fan-out can build are anchored. Anchoring one is 

3265 what tells the operator's exporter to hold that backend's spans back under 

3266 ``override``, so an unbuildable one would leave the span with nowhere to go. 

3267 

3268 The ``postgres`` spans under ``auth`` close before this runs, because they are the 

3269 reads that resolve the identity being read here. They never reach the tenant's 

3270 account, and they are never withheld from the operator's backend, whichever mode 

3271 is set. 

3272 """ 

3273 try: 

3274 from litellm.integrations.otel.logger import fan_out_provider 

3275 from litellm.integrations.otel.plumbing.context import set_request_destinations 

3276 from litellm.integrations.otel.plumbing.providers import deliverable_destinations 

3277 from litellm.proxy.litellm_pre_call_utils import ( 

3278 resolve_tenant_otel_destinations, 

3279 ) 

3280 

3281 set_request_destinations( 

3282 deliverable_destinations( 

3283 resolve_tenant_otel_destinations(user_api_key_dict, _safe_get_request_headers(request)), 

3284 fan_out_provider(), 

3285 ) 

3286 ) 

3287 except Exception as exc: # noqa: BLE001 # telemetry routing is best-effort and must never break authentication 

3288 verbose_proxy_logger.debug("OTel V2: tenant destination resolution failed: %s", exc) 

3289 

3290 

3291@tracer.wrap() 

3292async def user_api_key_auth( 

3293 request: Request, 

3294 api_key: str = fastapi.Security(api_key_header), 

3295 azure_api_key_header: str = fastapi.Security(azure_api_key_header), 

3296 anthropic_api_key_header: str | None = fastapi.Security(anthropic_api_key_header), 

3297 google_ai_studio_api_key_header: str | None = fastapi.Security(google_ai_studio_api_key_header), 

3298 azure_apim_header: str | None = fastapi.Security(azure_apim_header), 

3299 custom_litellm_key_header: str | None = fastapi.Security(custom_litellm_key_header), 

3300) -> UserAPIKeyAuth: 

3301 """ 

3302 Parent function to authenticate user api key / jwt token. 

3303 """ 

3304 

3305 # Create the SERVER span and stash it on request.state BEFORE reading the 

3306 # body. _read_request_body can raise ProxyException for malformed JSON; 

3307 # without this, that path leaves no span for the exception handler to 

3308 # close, and the trace never reaches the backend. 

3309 _ensure_parent_otel_span_on_request_state(request) 

3310 

3311 request_data, body_parse_exception = await _read_request_body_deferring_parse_failure(request=request) 

3312 route: Final[str] = get_request_route(request=request) 

3313 ## CHECK IF ROUTE IS ALLOWED 

3314 

3315 # Run the whole auth phase inside a live ``auth`` span so the DB lookups it 

3316 # triggers (key/user/team object reads) nest under it instead of flattening 

3317 # onto the server span. No-op when OTel V2 isn't active. 

3318 with phase_span(f"auth {route}"), spend_counter_batch_scope(_spend_counter_redis_cache()): 

3319 try: 

3320 user_api_key_auth_obj: Final = await _user_api_key_auth_builder( 

3321 request=request, 

3322 api_key=api_key, 

3323 azure_api_key_header=azure_api_key_header, 

3324 anthropic_api_key_header=anthropic_api_key_header, 

3325 google_ai_studio_api_key_header=google_ai_studio_api_key_header, 

3326 azure_apim_header=azure_apim_header, 

3327 request_data=request_data, 

3328 custom_litellm_key_header=custom_litellm_key_header, 

3329 ) 

3330 except Exception: 

3331 # The body was read first, so a caller who sent both a malformed body and 

3332 # a rejected key used to get the 400; the response is unchanged, and the 

3333 # auth failure is still recorded on the trace by the handler that ran. 

3334 if body_parse_exception is not None: 

3335 raise body_parse_exception 

3336 raise 

3337 user_api_key_auth_obj.budget_reservation = None 

3338 user_api_key_auth_obj.agent_caller = agent_caller_from_headers( 

3339 _safe_get_request_headers(request), user_api_key_auth_obj 

3340 ) 

3341 _seed_request_destinations(user_api_key_auth_obj, request) 

3342 

3343 # A body that never parsed is authenticated (so the trace carries identity 

3344 # and this ``auth`` span) but not authorized: there is no model to check it 

3345 # against, and budget reservation would increment live spend counters that 

3346 # only the endpoint's post-call path releases; the endpoint never runs, since 

3347 # the parse failure is raised below. 

3348 if body_parse_exception is None: 

3349 recovered_auth_obj: Final = await _authorize_authenticated_request( 

3350 user_api_key_auth_obj=user_api_key_auth_obj, 

3351 request=request, 

3352 request_data=request_data, 

3353 route=route, 

3354 api_key=api_key, 

3355 ) 

3356 if recovered_auth_obj is not None: 3356 ↛ 3357line 3356 didn't jump to line 3357 because the condition on line 3356 was never true

3357 return recovered_auth_obj 

3358 

3359 # Identity is now resolved. Seed it AFTER the auth span closes so the Baggage 

3360 # persists on the request task (detaching the span's context token inside the 

3361 # ``with`` would unwind a Baggage attach made within it) and every post-auth 

3362 # span — pre-call, LLM call, guardrail, spend write — inherits team/key/user. 

3363 seed_request_identity( 

3364 user_api_key_auth_obj, 

3365 model=request_data.get("model") if isinstance(request_data, dict) else None, 

3366 ) 

3367 user_api_key_auth_obj.request_route = normalize_request_route(route) 

3368 

3369 if body_parse_exception is not None: 

3370 await _record_unparsable_body_failure( 

3371 user_api_key_dict=user_api_key_auth_obj, 

3372 body_parse_exception=body_parse_exception, 

3373 route=route, 

3374 ) 

3375 raise body_parse_exception 

3376 

3377 # Resolve caller identity once, here at the seam, into a single per-request 

3378 # Principal projected off the key object the builder already fetched (no 

3379 # second lookup). Downstream consumers read identity off this instead of 

3380 # re-resolving it. Additive and defensive: a projection failure must never 

3381 # reject an already-authenticated request, so it is left unset on failure; 

3382 # any future consumer must treat a missing principal as deny, not allow. 

3383 try: 

3384 request.state.principal = _resolve_request_principal(request, user_api_key_auth_obj) 

3385 except Exception as e: 

3386 verbose_proxy_logger.warning("Principal projection at auth seam failed (non-fatal): %s", e) 

3387 

3388 return user_api_key_auth_obj 

3389 

3390 

3391async def _return_user_api_key_auth_obj( 

3392 user_obj: LiteLLM_UserTable | None, 

3393 api_key: str, 

3394 parent_otel_span: Span | None, 

3395 valid_token_dict: dict, 

3396 route: str, 

3397 start_time: datetime, 

3398 user_role: LitellmUserRoles | None = None, 

3399) -> UserAPIKeyAuth: 

3400 end_time: Final = datetime.now(timezone.utc) 

3401 

3402 asyncio.create_task( 

3403 user_api_key_service_logger_obj.async_service_success_hook( 

3404 service=ServiceTypes.AUTH, 

3405 call_type=route, 

3406 start_time=start_time, 

3407 end_time=end_time, 

3408 duration=end_time.timestamp() - start_time.timestamp(), 

3409 parent_otel_span=parent_otel_span, 

3410 ) 

3411 ) 

3412 

3413 retrieved_user_role: Final = user_role or _get_user_role(user_obj=user_obj) or LitellmUserRoles.INTERNAL_USER 

3414 

3415 user_api_key_kwargs: Final = { 

3416 "api_key": api_key, 

3417 "parent_otel_span": parent_otel_span, 

3418 "user_role": retrieved_user_role, 

3419 **valid_token_dict, 

3420 } 

3421 if user_obj is not None: 

3422 user_api_key_kwargs.update( 

3423 user_tpm_limit=user_obj.tpm_limit, 

3424 user_rpm_limit=user_obj.rpm_limit, 

3425 user_email=user_obj.user_email, 

3426 user_spend=getattr(user_obj, "spend", None), 

3427 user_max_budget=getattr(user_obj, "max_budget", None), 

3428 user_model_max_budget=getattr(user_obj, "model_max_budget", None), 

3429 ) 

3430 if user_obj is not None and _is_user_proxy_admin(user_obj=user_obj): 3430 ↛ 3431line 3430 didn't jump to line 3431 because the condition on line 3430 was never true

3431 user_api_key_kwargs.update( 

3432 user_role=LitellmUserRoles.PROXY_ADMIN, 

3433 ) 

3434 return UserAPIKeyAuth.model_validate(user_api_key_kwargs) 

3435 else: 

3436 return UserAPIKeyAuth.model_validate(user_api_key_kwargs) 

3437 

3438 

3439def get_api_key_from_custom_header(request: Request, custom_litellm_key_header_name: str) -> str: 

3440 """ 

3441 Get API key from custom header 

3442 

3443 Args: 

3444 request (Request): Request object 

3445 custom_litellm_key_header_name (str): Custom header name 

3446 

3447 Returns: 

3448 Optional[str]: API key 

3449 """ 

3450 api_key: str = "" 

3451 # use this as the virtual key passed to litellm proxy 

3452 custom_litellm_key_header_name = custom_litellm_key_header_name.lower() 

3453 _headers: Final = {k.lower(): v for k, v in request.headers.items()} 

3454 verbose_proxy_logger.debug( 

3455 "searching for custom_litellm_key_header_name= %s, in headers=%s", 

3456 custom_litellm_key_header_name, 

3457 _headers, 

3458 ) 

3459 custom_api_key: Final = _headers.get(custom_litellm_key_header_name) 

3460 if custom_api_key: 

3461 api_key = _get_bearer_token(api_key=custom_api_key) 

3462 verbose_proxy_logger.debug( 

3463 "Found custom API key using header: %s, setting api_key=%s", 

3464 custom_litellm_key_header_name, 

3465 abbreviate_api_key(api_key), 

3466 ) 

3467 else: 

3468 verbose_proxy_logger.exception( 

3469 "No LiteLLM Virtual Key pass. Please set header=%s: Bearer <api_key>", custom_litellm_key_header_name 

3470 ) 

3471 return api_key 

3472 

3473 

3474def _get_temp_budget_increase(valid_token: UserAPIKeyAuth): 

3475 valid_token_metadata: Final = valid_token.metadata 

3476 if "temp_budget_increase" in valid_token_metadata and "temp_budget_expiry" in valid_token_metadata: 

3477 expiry = datetime.fromisoformat(valid_token_metadata["temp_budget_expiry"]) 

3478 if expiry.tzinfo is None: 

3479 expiry = expiry.replace(tzinfo=timezone.utc) 

3480 if expiry > datetime.now(timezone.utc): 

3481 return valid_token_metadata["temp_budget_increase"] 

3482 return None 

3483 

3484 

3485def _update_key_budget_with_temp_budget_increase( 

3486 valid_token: UserAPIKeyAuth, 

3487) -> UserAPIKeyAuth: 

3488 if valid_token.max_budget is None: 3488 ↛ 3490line 3488 didn't jump to line 3490 because the condition on line 3488 was always true

3489 return valid_token 

3490 temp_budget_increase: Final = _get_temp_budget_increase(valid_token) 

3491 if not temp_budget_increase: 

3492 return valid_token 

3493 return valid_token.model_copy(update={"max_budget": valid_token.max_budget + temp_budget_increase}) 

3494 

3495 

3496async def _lookup_end_user_and_apply_budget( 

3497 valid_token: UserAPIKeyAuth, 

3498 route: str, 

3499 parent_otel_span: Span | None, 

3500 prisma_client, 

3501 user_api_key_cache, 

3502 proxy_logging_obj, 

3503): 

3504 """Look up end_user from DB and apply budget limits to valid_token.""" 

3505 end_user_object = None 

3506 key_end_user_budget_id: Final = get_key_end_user_budget_id(valid_token.metadata) 

3507 try: 

3508 end_user_object = await get_end_user_object( 

3509 end_user_id=valid_token.end_user_id, 

3510 prisma_client=prisma_client, 

3511 user_api_key_cache=user_api_key_cache, 

3512 parent_otel_span=parent_otel_span, 

3513 proxy_logging_obj=proxy_logging_obj, 

3514 route=route, 

3515 token_end_user_max_budget=valid_token.end_user_max_budget, 

3516 key_end_user_budget_id=key_end_user_budget_id, 

3517 ) 

3518 if end_user_object is not None: 

3519 end_user_params = { 

3520 "end_user_id": valid_token.end_user_id, 

3521 "allowed_model_region": end_user_object.allowed_model_region, 

3522 } 

3523 if end_user_object.litellm_budget_table is not None: 

3524 _apply_budget_limits_to_end_user_params( 

3525 end_user_params=end_user_params, 

3526 budget_info=end_user_object.litellm_budget_table, 

3527 end_user_id=valid_token.end_user_id or "", 

3528 ) 

3529 valid_token = update_valid_token_with_end_user_params( 

3530 valid_token=valid_token, end_user_params=end_user_params 

3531 ) 

3532 elif key_end_user_budget_id is not None or litellm.max_end_user_budget_id is not None: 

3533 default_budget: Final = await resolve_default_end_user_budget( 

3534 prisma_client=prisma_client, 

3535 user_api_key_cache=user_api_key_cache, 

3536 key_end_user_budget_id=key_end_user_budget_id, 

3537 parent_otel_span=parent_otel_span, 

3538 ) 

3539 if default_budget is not None: 

3540 end_user_params = {"end_user_id": valid_token.end_user_id} 

3541 _apply_budget_limits_to_end_user_params( 

3542 end_user_params=end_user_params, 

3543 budget_info=default_budget, 

3544 end_user_id=valid_token.end_user_id or "", 

3545 ) 

3546 valid_token = update_valid_token_with_end_user_params( 

3547 valid_token=valid_token, end_user_params=end_user_params 

3548 ) 

3549 if valid_token.end_user_max_budget is None: 

3550 valid_token.end_user_max_budget = default_budget.max_budget 

3551 except Exception as e: 

3552 if isinstance(e, litellm.BudgetExceededError): 

3553 raise e 

3554 verbose_proxy_logger.debug("Unable to find user in db. Error - %s", e) 

3555 return valid_token, end_user_object 

3556 

3557 

3558async def _enforce_key_and_fallback_model_access( 

3559 *, 

3560 valid_token: UserAPIKeyAuth, 

3561 request_data: dict, 

3562 route: str, 

3563 request: Request | None, 

3564 llm_model_list: list | None, 

3565 llm_router: Any | None, 

3566) -> None: 

3567 """ 

3568 Key-level model allowlist and client fallbacks (same as standard auth). 

3569 Not included in common_checks — common_checks enforces team/user/project model access only. 

3570 """ 

3571 await _normalize_claude_model(request_data, valid_token, request, route) 

3572 await _resolve_router_settings_model_group_alias(request_data, valid_token, request, route) 

3573 config: Final = valid_token.config 

3574 

3575 if config != {}: 3575 ↛ 3576line 3575 didn't jump to line 3576 because the condition on line 3575 was never true

3576 model_list: Final = config.get("model_list", []) 

3577 new_model_list: Final = model_list 

3578 verbose_proxy_logger.debug("\n new llm router model list %s", new_model_list) 

3579 elif isinstance(valid_token.models, list) and "all-team-models" in valid_token.models: 3579 ↛ 3580line 3579 didn't jump to line 3580 because the condition on line 3579 was never true

3580 pass 

3581 else: 

3582 model: Final = _get_model_from_request_context( 

3583 request_data=request_data, 

3584 route=route, 

3585 request=request, 

3586 llm_router=llm_router, 

3587 team_id=valid_token.team_id, 

3588 ) 

3589 

3590 if model is not None: 3590 ↛ 3591line 3590 didn't jump to line 3591 because the condition on line 3590 was never true

3591 await can_key_call_model( 

3592 model=model, 

3593 llm_model_list=llm_model_list, 

3594 valid_token=valid_token, 

3595 llm_router=llm_router, 

3596 ) 

3597 

3598 fallback_names: Final = tuple( 

3599 name 

3600 for target in iter_request_fallback_targets(request_data) 

3601 if (name := _fallback_target_model_name(target)) is not None 

3602 ) 

3603 

3604 for _name in dict.fromkeys(fallback_names): # dedupe, preserve order 3604 ↛ 3605line 3604 didn't jump to line 3605 because the loop on line 3604 never started

3605 await can_key_call_model( 

3606 model=_name, 

3607 llm_model_list=llm_model_list, 

3608 valid_token=valid_token, 

3609 llm_router=llm_router, 

3610 ) 

3611 await is_valid_fallback_model( 

3612 model=_name, 

3613 llm_router=llm_router, 

3614 user_model=None, 

3615 ) 

3616 

3617 

3618def _fallback_target_model_name(target: object) -> str | None: 

3619 if isinstance(target, str): 

3620 return target 

3621 if isinstance(target, dict): 

3622 model: Final = target.get("model") 

3623 if isinstance(model, str): 

3624 return model 

3625 return None 

3626 

3627 

3628async def _run_post_custom_auth_checks( 

3629 valid_token: UserAPIKeyAuth, 

3630 request: Request, 

3631 request_data: dict, 

3632 route: str, 

3633 parent_otel_span: Span | None, 

3634) -> UserAPIKeyAuth: 

3635 from litellm.proxy.proxy_server import ( 

3636 general_settings, 

3637 llm_model_list, 

3638 llm_router, 

3639 model_max_budget_limiter, 

3640 prisma_client, 

3641 proxy_logging_obj, 

3642 user_api_key_cache, 

3643 ) 

3644 

3645 # 1. Look up end_user object from DB if end_user_id is set 

3646 end_user_object = None 

3647 if valid_token.end_user_id is not None: 

3648 valid_token, end_user_object = await _lookup_end_user_and_apply_budget( 

3649 valid_token=valid_token, 

3650 route=route, 

3651 parent_otel_span=parent_otel_span, 

3652 prisma_client=prisma_client, 

3653 user_api_key_cache=user_api_key_cache, 

3654 proxy_logging_obj=proxy_logging_obj, 

3655 ) 

3656 # common_checks() enforces the end-user budget, but the centralized 

3657 # gate skips it for custom-auth deployments unless 

3658 # custom_auth_run_common_checks is set. Enforce it here on that path 

3659 # so an over-budget end user can't keep making requests. 

3660 if end_user_object is not None and not general_settings.get("custom_auth_run_common_checks", False): 

3661 await _check_end_user_budget(end_user_obj=end_user_object, route=route) 

3662 

3663 # 2. Check token expiry 

3664 if valid_token.expires is not None: 

3665 current_time: Final = datetime.now(timezone.utc) 

3666 if isinstance(valid_token.expires, datetime): 

3667 expiry_time = valid_token.expires 

3668 else: 

3669 expiry_time = datetime.fromisoformat(valid_token.expires) 

3670 if expiry_time.tzinfo is None or expiry_time.tzinfo.utcoffset(expiry_time) is None: 

3671 expiry_time = expiry_time.replace(tzinfo=timezone.utc) 

3672 if expiry_time < current_time: 

3673 raise ProxyException( 

3674 message=f"Authentication Error - Expired Key. Key Expiry time {expiry_time} and current time {current_time}", 

3675 type=ProxyErrorTypes.expired_key, 

3676 code=status.HTTP_401_UNAUTHORIZED, 

3677 param=(abbreviate_api_key(api_key=valid_token.token) if valid_token.token else ""), 

3678 ) 

3679 

3680 if general_settings.get("custom_auth_run_common_checks", False): 

3681 await _enforce_key_and_fallback_model_access( 

3682 valid_token=valid_token, 

3683 request_data=request_data, 

3684 route=route, 

3685 request=request, 

3686 llm_model_list=llm_model_list, 

3687 llm_router=llm_router, 

3688 ) 

3689 

3690 current_model = _get_model_from_request_context( 

3691 request_data=request_data, 

3692 route=route, 

3693 request=request, 

3694 llm_router=llm_router, 

3695 team_id=valid_token.team_id, 

3696 ) 

3697 current_models = _get_model_names_for_budget_checks(model=current_model) 

3698 

3699 # A zero-cost model cannot move any counter, so refusing it means refusing on 

3700 # spend some other model accrued. The JWT and virtual-key paths already skip 

3701 # every budget check for these; this path did not, so the same request could 

3702 # be refused under custom auth and served under the other two. 

3703 skip_budget_checks: Final = ( 

3704 _is_model_cost_zero(model=current_model, llm_router=llm_router) 

3705 if current_model is not None and llm_router is not None 

3706 else False 

3707 ) 

3708 

3709 # 3. Check key-level model_max_budget 

3710 max_budget_per_model: Final = valid_token.model_max_budget 

3711 if ( 

3712 not skip_budget_checks 

3713 and max_budget_per_model is not None 

3714 and isinstance(max_budget_per_model, dict) 

3715 and len(max_budget_per_model) > 0 

3716 and current_models 

3717 and valid_token.token is not None 

3718 ): 

3719 for model_name in current_models: 

3720 await _check_key_model_budget_with_fallback( 

3721 valid_token=valid_token, 

3722 model_max_budget_limiter=model_max_budget_limiter, 

3723 model_name=model_name, 

3724 request_data=request_data, 

3725 request=request, 

3726 llm_model_list=llm_model_list, 

3727 llm_router=llm_router, 

3728 ) 

3729 

3730 # Recompute after a potential budget-fallback rewrite so 

3731 # the end-user check below validates the final model 

3732 current_model = _get_model_from_request_context( 

3733 request_data=request_data, 

3734 route=route, 

3735 request=request, 

3736 llm_router=llm_router, 

3737 team_id=valid_token.team_id, 

3738 ) 

3739 current_models = _get_model_names_for_budget_checks(model=current_model) 

3740 

3741 # 3b. Attach and check the internal user's model_max_budget. 

3742 # Custom auth builds its own token, so unlike the main path nothing has 

3743 # loaded the user row yet. The attach is unconditional because the post-call 

3744 # spend hook reads this field off the token: gating it on the same condition 

3745 # as enforcement would leave the user's counter uncharged whenever this 

3746 # request was not itself enforceable, so its spend would go untracked. 

3747 user_budget: Final = await _read_user_model_max_budget( 

3748 user_id=valid_token.user_id, 

3749 prisma_client=prisma_client, 

3750 user_api_key_cache=user_api_key_cache, 

3751 parent_otel_span=parent_otel_span, 

3752 proxy_logging_obj=proxy_logging_obj, 

3753 ) 

3754 valid_token.user_model_max_budget = user_budget # rebind-ok: the spend hook reads it off this token 

3755 if not skip_budget_checks and current_models: 

3756 await _check_user_model_budget( 

3757 valid_token=valid_token, 

3758 model_max_budget_limiter=model_max_budget_limiter, 

3759 models=current_models, 

3760 ) 

3761 

3762 # 4. Check end-user model_max_budget 

3763 end_user_mmb: Final = valid_token.end_user_model_max_budget 

3764 if ( 

3765 not skip_budget_checks 

3766 and end_user_mmb is not None 

3767 and isinstance(end_user_mmb, dict) 

3768 and len(end_user_mmb) > 0 

3769 and current_models 

3770 and valid_token.end_user_id is not None 

3771 ): 

3772 for model_name in current_models: 

3773 await model_max_budget_limiter.is_end_user_within_model_budget( 

3774 end_user_id=valid_token.end_user_id, 

3775 end_user_model_max_budget=end_user_mmb, 

3776 model=model_name, 

3777 ) 

3778 

3779 # team / user / end_user / project context objects are fetched by 

3780 # the centralized common_checks gate in user_api_key_auth after 

3781 # this helper returns. Keep only the project fetch here because it 

3782 # mutates the token (project_metadata / project_alias). 

3783 if valid_token.project_id is not None: 

3784 _project_obj: Final = await get_project_object( 

3785 project_id=valid_token.project_id, 

3786 prisma_client=prisma_client, 

3787 user_api_key_cache=user_api_key_cache, 

3788 proxy_logging_obj=proxy_logging_obj, 

3789 ) 

3790 if _project_obj is not None: 

3791 valid_token.project_metadata = _project_obj.metadata 

3792 valid_token.project_alias = _project_obj.project_alias 

3793 

3794 return valid_token