Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/management_endpoints/key_management_endpoints.py: 38%
2356 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2KEY MANAGEMENT
4All /key management endpoints
6/key/generate
7/key/info
8/key/update
9/key/delete
10"""
12import asyncio
13import copy
14import inspect
15import json
16import math
17import os
18import re
19import secrets
20import traceback
21from collections.abc import Awaitable, Callable, Iterator, Mapping, Sequence
22from datetime import datetime, timedelta, timezone
23from types import MappingProxyType
24from typing import TYPE_CHECKING, Any, Final, Literal, Optional, Protocol, TypeVar, cast
26import fastapi
27import yaml
28from fastapi import APIRouter, Depends, Header, HTTPException, Query, Request, status
29from pydantic import TypeAdapter
30from typing_extensions import ReadOnly, TypedDict
32import litellm
33from litellm._logging import verbose_proxy_logger
34from litellm._uuid import uuid
35from litellm.caching.dual_cache import DualCache
36from litellm.constants import (
37 LENGTH_OF_LITELLM_GENERATED_KEY,
38 LITELLM_PROXY_ADMIN_NAME,
39 MINIMUM_CUSTOM_KEY_LENGTH,
40 UI_SESSION_TOKEN_TEAM_ID,
41)
42from litellm.litellm_core_utils.duration_parser import duration_in_seconds
43from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
44from litellm.models.credentials import CredentialItem
45from litellm.proxy._experimental.mcp_server.db import (
46 rotate_mcp_server_credentials_master_key,
47 rotate_mcp_user_credentials_master_key,
48 rotate_mcp_user_env_vars_master_key,
49)
50from litellm.proxy._experimental.mcp_server.outbound_credentials.sso_assertion_store import (
51 rotate_sso_identity_assertions_master_key,
52)
53from litellm.proxy._types import *
54from litellm.proxy._types import Litellm_EntityType, LiteLLM_VerificationToken, hash_token
55from litellm.proxy.auth.auth_checks import (
56 _delete_cache_key_object,
57 can_team_access_model,
58 get_jwt_key_mapping_cache_keys_for_token,
59 get_key_end_user_budget_id,
60 get_org_object,
61 get_project_object,
62 get_team_object,
63)
64from litellm.proxy.auth.auth_utils import (
65 abbreviate_api_key,
66 enforce_batch_enqueued_token_limit_is_admin_only,
67 enforce_output_token_estimates_are_admin_only,
68)
69from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
70from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import (
71 evict_and_broadcast,
72 publish_auth_cache_invalidation,
73)
74from litellm.proxy.common_utils.callback_config_validation import logging_metadata_config_error
75from litellm.proxy.common_utils.callback_utils import (
76 decrypt_callback_vars,
77 encrypt_callback_vars,
78)
79from litellm.proxy.common_utils.config_sync_pubsub import (
80 coordination_redis_cache,
81 publish_config_change,
82)
83from litellm.proxy.common_utils.rbac_utils import check_org_admin_can_generate_keys
84from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time
85from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
86from litellm.proxy.hooks.key_management_event_hooks import KeyManagementEventHooks
87from litellm.proxy.hooks.model_max_budget_limiter import build_model_max_budget_usage
88from litellm.proxy.management_endpoints.common_utils import (
89 _check_disable_global_guardrails_caller_permission,
90 _check_passthrough_routes_caller_permission,
91 _is_user_org_admin_for_team,
92 _is_user_team_admin,
93 _set_object_metadata_field,
94 _team_member_has_permission,
95 _user_has_admin_view,
96 validate_budget_duration,
97 validate_finite_spend,
98)
99from litellm.proxy.management_endpoints.model_management_endpoints import (
100 _add_model_to_db,
101)
102from litellm.proxy.management_endpoints.router_weights import validate_router_settings_weights
103from litellm.proxy.management_endpoints.team_admin_field_permissions import (
104 team_admin_key_edit_verdict,
105 team_admin_key_request_or_raise,
106 team_admin_may_edit_member_key_budgets,
107)
108from litellm.proxy.management_helpers.access_group_key_sync import (
109 sync_key_access_group_membership,
110 sync_key_regeneration_access_group_membership,
111 sync_key_update_access_group_membership,
112)
113from litellm.proxy.management_helpers.key_settings_audit import with_settings_updated_at
114from litellm.proxy.management_helpers.object_permission_utils import (
115 _set_object_permission,
116 attach_object_permission_to_dict,
117 handle_update_object_permission_common,
118 invalidate_cached_object_permissions,
119 validate_key_mcp_servers_against_team,
120 validate_key_search_tools_against_team,
121 validate_key_vector_stores_against_team,
122)
123from litellm.proxy.management_helpers.team_member_permission_checks import (
124 TeamMemberPermissionChecks,
125)
126from litellm.proxy.management_helpers.utils import management_endpoint_wrapper
127from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
128from litellm.proxy.spend_tracking.spend_tracking_utils import _is_master_key
129from litellm.proxy.utils import (
130 PrismaClient,
131 ProxyLogging,
132 _hash_token_if_needed,
133 handle_exception_on_proxy,
134 is_valid_api_key,
135)
136from litellm.repositories.base_repository import BaseRepository
137from litellm.repositories.budget_repository import BudgetRepository
138from litellm.repositories.config_repository import ConfigParam, ConfigRepository
139from litellm.repositories.credentials_repository import CredentialsRepository
140from litellm.repositories.model_repository import ModelRepository
141from litellm.repositories.prisma_protocols import TableActions
142from litellm.repositories.table_repositories import (
143 DeletedVerificationTokenRepository,
144 DeprecatedVerificationTokenRepository,
145)
146from litellm.repositories.team_repository import TeamRepository
147from litellm.repositories.user_repository import UserRepository
148from litellm.repositories.verification_token_repository import (
149 VerificationTokenRepository,
150)
151from litellm.router import Router
152from litellm.secret_managers.base_secret_manager import raise_if_unsafe_secret_name
153from litellm.secret_managers.main import get_secret
154from litellm.types.proxy.management_endpoints.key_management_endpoints import (
155 BulkUpdateKeyRequest,
156 BulkUpdateKeyResponse,
157 BulkUpdateTeamKeysRequest,
158 CustomKeyPolicyRequest,
159 FailedKeyUpdate,
160 KeySearchWhere,
161 SuccessfulKeyUpdate,
162)
163from litellm.types.router import Deployment
164from litellm.types.utils import (
165 BudgetConfig,
166 PersonalUIKeyGenerationConfig,
167 TeamUIKeyGenerationConfig,
168)
170if TYPE_CHECKING: 170 ↛ 171line 170 didn't jump to line 171 because the condition on line 170 was never true
171 import prisma
172 from prisma import Prisma
173 from prisma import models as prisma_models
175_RepositoryModelT = TypeVar("_RepositoryModelT", bound=BaseModel)
178class _UserRowLike(Protocol):
179 """Read-only view of the user columns ``/key/list`` expands keys with."""
181 @property
182 def user_id(self) -> str | None: ... 182 ↛ exitline 182 didn't return from function 'user_id' because
184 @property
185 def user_email(self) -> str | None: ... 185 ↛ exitline 185 didn't return from function 'user_email' because
187 @property
188 def user_alias(self) -> str | None: ... 188 ↛ exitline 188 didn't return from function 'user_alias' because
190 def model_dump(self) -> Mapping[str, object]: ... 190 ↛ exitline 190 didn't return from function 'model_dump' because
192 def dict(self) -> Mapping[str, object]: ... 192 ↛ exitline 192 didn't return from function 'dict' because
195class _TxTables(Protocol):
196 litellm_proxymodeltable: TableActions[object]
199class _ModelParamsUpdate(TypedDict):
200 litellm_params: ReadOnly["prisma.Json"]
203class _ModelRowWhere(TypedDict):
204 model_id: ReadOnly[str]
207class _KeyUpdateResult(TypedDict):
208 token: ReadOnly[str]
209 data: ReadOnly[Mapping[str, object]]
212class _StoredKeyRouterSettings(BaseModel):
213 router_settings: Mapping[str, object] | None = None
216class _KeyRowWhere(TypedDict):
217 token: ReadOnly[str]
220class _BudgetRowWhere(TypedDict):
221 budget_id: ReadOnly[str]
224class _BudgetRowSoftBudgetUpdate(TypedDict):
225 soft_budget: ReadOnly[float | None]
226 updated_by: ReadOnly[str]
229class _BudgetRowSoftBudgetCreate(TypedDict):
230 soft_budget: ReadOnly[float]
231 created_by: ReadOnly[str]
232 updated_by: ReadOnly[str]
235class _KeyUpdateTx(Protocol):
236 @property
237 def litellm_verificationtoken(self) -> "TableActions[prisma_models.LiteLLM_VerificationToken]": ... 237 ↛ exitline 237 didn't return from function 'litellm_verificationtoken' because
239 @property
240 def litellm_budgettable(self) -> "TableActions[prisma_models.LiteLLM_BudgetTable]": ... 240 ↛ exitline 240 didn't return from function 'litellm_budgettable' because
243class _ConfigTableActions(Protocol):
244 """Config table surface this module needs; the shared repository seam exposes no ``update``."""
246 async def find_many(self) -> Sequence[ConfigParam]: ... 246 ↛ exitline 246 didn't return from function 'find_many' because
248 async def update( 248 ↛ exitline 248 didn't return from function 'update' because
249 self,
250 *,
251 where: Mapping[str, object],
252 data: Mapping[str, object],
253 ) -> ConfigParam | None: ...
256def _prisma_table(
257 repository: BaseRepository[_RepositoryModelT],
258) -> TableActions[_RepositoryModelT]:
259 return cast( # cast-ok: callers read only the field names the prisma row and repository model share
260 "TableActions[_RepositoryModelT]", repository.table
261 )
264def _deleted_verification_token_table(
265 prisma_client: PrismaClient,
266) -> "TableActions[prisma_models.LiteLLM_DeletedVerificationToken]":
267 return DeletedVerificationTokenRepository(prisma_client).table
270def _deprecated_verification_token_table(
271 prisma_client: PrismaClient,
272) -> "TableActions[prisma_models.LiteLLM_DeprecatedVerificationToken]":
273 return DeprecatedVerificationTokenRepository(prisma_client).table
276def _user_table(prisma_client: PrismaClient) -> TableActions[_UserRowLike]:
277 return UserRepository(prisma_client).table
280def _credentials_table(prisma_client: PrismaClient) -> TableActions[CredentialItem]:
281 return cast( # cast-ok: the rotation loop reads and rewrites these rows through CredentialItem names only
282 "TableActions[CredentialItem]", CredentialsRepository(prisma_client).table
283 )
286def _config_table(prisma_client: PrismaClient) -> _ConfigTableActions:
287 return cast( # cast-ok: ConfigRepository.table hides the write actions this module needs on that same object
288 "_ConfigTableActions", ConfigRepository(prisma_client).table
289 )
292class _CustomKeyHooksModule(Protocol):
293 user_custom_key_generate: Callable[..., Awaitable[Mapping[str, object]]] | None
294 user_custom_key_update: Callable[..., Awaitable[Mapping[str, object]]] | None
295 user_custom_key_policy: Callable[..., Awaitable[Mapping[str, object]]] | None
298def _custom_key_generate_hook(
299 hooks: _CustomKeyHooksModule,
300) -> Callable[..., Awaitable[Mapping[str, object]]] | None:
301 return hooks.user_custom_key_generate
304def _custom_key_update_hook(
305 hooks: _CustomKeyHooksModule,
306) -> Callable[..., Awaitable[Mapping[str, object]]] | None:
307 return hooks.user_custom_key_update
310def _custom_key_policy_hook(
311 hooks: _CustomKeyHooksModule,
312) -> Callable[..., Awaitable[Mapping[str, object]]] | None:
313 return hooks.user_custom_key_policy
316async def _enforce_custom_key_update_policy(
317 hook: Callable[..., Awaitable[Mapping[str, object]]] | None,
318 data: UpdateKeyRequest,
319) -> None:
320 if hook is None:
321 return
322 if not inspect.iscoroutinefunction(hook):
323 raise ValueError("user_custom_key_update must be a coroutine")
324 result: Final = await hook(data)
325 if not result.get("decision", True):
326 raise HTTPException(
327 status_code=status.HTTP_403_FORBIDDEN,
328 detail=result.get("message", "Authentication Failed - Custom Auth Rule"),
329 )
332async def _enforce_custom_key_policy(
333 hook: Callable[..., Awaitable[Mapping[str, object]]] | None,
334 build_policy_request: Callable[[], CustomKeyPolicyRequest],
335) -> None:
336 if hook is None: 336 ↛ 338line 336 didn't jump to line 338 because the condition on line 336 was always true
337 return
338 if not inspect.iscoroutinefunction(hook):
339 raise ValueError("user_custom_key_policy must be a coroutine")
340 result: Final = await hook(build_policy_request())
341 if not result.get("decision", True):
342 raise HTTPException(
343 status_code=status.HTTP_403_FORBIDDEN,
344 detail=result.get("message", "Authentication Failed - Custom Auth Rule"),
345 )
348_KEY_UPDATE_JSON_STRING_COLUMNS: Final = frozenset({"router_settings", "budget_limits"})
350_KEY_METADATA_REQUEST_FIELDS: Final = frozenset(
351 (*LiteLLM_ManagementEndpoint_MetadataFields_Premium, *LiteLLM_ManagementEndpoint_MetadataFields)
352)
355def _decode_json_string_column(column: str, value: object) -> object:
356 if column in _KEY_UPDATE_JSON_STRING_COLUMNS and isinstance(value, str):
357 return json.loads(value)
358 return value
361def _verification_token_from_row(row: Mapping[str, object]) -> LiteLLM_VerificationToken:
362 org_id: Final = row["organization_id"] if "organization_id" in row else row.get("org_id")
363 return LiteLLM_VerificationToken.model_validate(MappingProxyType({**row, "org_id": org_id}))
366def _effective_key_after_update(
367 existing_key_row: LiteLLM_VerificationToken,
368 non_default_values: Mapping[str, object],
369) -> LiteLLM_VerificationToken:
370 overlay: Final = MappingProxyType(
371 {column: _decode_json_string_column(column, value) for column, value in non_default_values.items()}
372 )
373 return _verification_token_from_row(
374 MappingProxyType({**existing_key_row.model_dump(), **overlay, "object_permission": None})
375 )
378def _update_policy_request(
379 operation: Literal["update", "regenerate"],
380 existing_key_row: LiteLLM_VerificationToken,
381 non_default_values: Mapping[str, object],
382 request: UpdateKeyRequest | RegenerateKeyRequest,
383) -> CustomKeyPolicyRequest:
384 return CustomKeyPolicyRequest(
385 operation=operation,
386 existing_key=_verification_token_from_row(existing_key_row.model_dump()),
387 effective_key=_effective_key_after_update(
388 existing_key_row=existing_key_row, non_default_values=non_default_values
389 ),
390 request=request,
391 )
394def _generate_budget_windows(
395 budget_limits: Sequence[BudgetLimitEntry] | None,
396) -> tuple[Mapping[str, object], ...] | None:
397 if not budget_limits:
398 return None
399 return tuple(
400 MappingProxyType(
401 {
402 **window.model_dump(),
403 "reset_at": get_budget_reset_time(budget_duration=window.budget_duration).isoformat(),
404 }
405 )
406 for window in budget_limits
407 )
410def _effective_key_for_generate(data: GenerateKeyRequest, now: datetime) -> LiteLLM_VerificationToken:
411 requested: Final = data.model_dump(exclude_unset=True, exclude_none=True)
412 metadata_fields: Final = MappingProxyType(
413 {field: value for field, value in requested.items() if field in _KEY_METADATA_REQUEST_FIELDS}
414 )
415 column_fields: Final = MappingProxyType(
416 {field: value for field, value in requested.items() if field not in _KEY_METADATA_REQUEST_FIELDS}
417 )
418 metadata: Final = data.metadata or MappingProxyType({})
419 folded_metadata: Final = {**metadata, **metadata_fields} # mutable-ok: encrypt_callback_vars needs a dict
420 columns: Final = handle_key_type(data, {**column_fields}) # mutable-ok: handle_key_type mutates in place
421 expires: Final = (
422 now + timedelta(seconds=duration_in_seconds(duration=data.duration)) if data.duration is not None else None
423 )
424 budget_reset_at: Final = (
425 get_budget_reset_time(budget_duration=data.budget_duration) if data.budget_duration is not None else None
426 )
427 key_rotation_at: Final = (
428 now + timedelta(seconds=duration_in_seconds(duration=data.rotation_interval))
429 if data.auto_rotate and data.rotation_interval
430 else None
431 )
432 return _verification_token_from_row(
433 MappingProxyType(
434 {
435 **columns,
436 "metadata": encrypt_callback_vars(folded_metadata),
437 "expires": expires,
438 "budget_reset_at": budget_reset_at,
439 "key_rotation_at": key_rotation_at,
440 "budget_limits": _generate_budget_windows(data.budget_limits),
441 "object_permission": None,
442 }
443 )
444 )
447_EMPTY_DURATION_MEANS_UNCHANGED: Final = frozenset({"duration", "budget_duration"})
450def _regenerate_request_as_update_request(key: str, data: RegenerateKeyRequest) -> UpdateKeyRequest | None:
451 changed_fields: Final = MappingProxyType(
452 {
453 field: value
454 for field, value in data.model_dump(exclude_unset=True).items()
455 if field in UpdateKeyRequest.model_fields
456 and field != "key"
457 and not (field in _EMPTY_DURATION_MEANS_UNCHANGED and value == "")
458 }
459 )
460 if not changed_fields:
461 return None
462 return UpdateKeyRequest.model_validate(MappingProxyType({"key": key, **changed_fields}))
465class _LegacyDumpable(Protocol):
466 def dict(self) -> Mapping[str, object]: ... 466 ↛ exitline 466 didn't return from function 'dict' because
469def _legacy_model_dict(row: _LegacyDumpable) -> Mapping[str, object]:
470 return row.dict()
473def _as_object_dict(values: Mapping[str, object]) -> Mapping[str, object]:
474 return values
477def _model_items(model: BaseModel) -> Iterator[tuple[str, object]]:
478 return iter(model)
481class _EnvVarsParam(Protocol):
482 @property
483 def param_value(self) -> Mapping[str, str] | None: ... 483 ↛ exitline 483 didn't return from function 'param_value' because
486def _env_vars_param_value(param: _EnvVarsParam) -> Mapping[str, str] | None:
487 return param.param_value
490async def _check_custom_key_allowed(custom_key_value: str | None) -> None:
491 """Raise 403 if custom API keys are disabled and a custom key was provided."""
492 if custom_key_value is None: 492 ↛ 495line 492 didn't jump to line 495 because the condition on line 492 was always true
493 return
495 from litellm.proxy.config_resolvers.settings_rules import coerce_bool
496 from litellm.proxy.proxy_server import general_settings
498 if coerce_bool(general_settings.get("disable_custom_api_keys", False)) is True:
499 verbose_proxy_logger.warning("Custom API key rejected: disable_custom_api_keys is enabled")
500 raise HTTPException(
501 status_code=403,
502 detail={"error": "Custom API key values are disabled by your administrator. Keys must be auto-generated."},
503 )
506def _is_team_key(data: GenerateKeyRequest | LiteLLM_VerificationToken):
507 return data.team_id is not None
510def _get_user_in_team(team_table: LiteLLM_TeamTableCachedObj, user_id: str | None) -> Member | None:
511 if user_id is None:
512 return None
513 for member in team_table.members_with_roles:
514 if member.user_id is not None and member.user_id == user_id:
515 return member
517 return None
520def _get_caller_team_role(
521 team_table: LiteLLM_TeamTableCachedObj,
522 user_api_key_dict: UserAPIKeyAuth,
523) -> Literal["admin", "user"] | None:
524 if user_api_key_dict.is_team_service_account and user_api_key_dict.team_id == team_table.team_id:
525 return "user"
526 member: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
527 return None if member is None else member.role
530def _calculate_key_rotation_time(rotation_interval: str) -> datetime:
531 """
532 Helper function to calculate the next rotation time for a key based on the rotation interval.
534 Args:
535 rotation_interval: String representing the rotation interval (e.g., '30d', '90d', '1h')
537 Returns:
538 datetime: The calculated next rotation time in UTC
539 """
540 now: Final = datetime.now(timezone.utc)
541 interval_seconds: Final = duration_in_seconds(rotation_interval)
542 return now + timedelta(seconds=interval_seconds)
545def _set_key_rotation_fields(
546 data: dict,
547 auto_rotate: bool,
548 rotation_interval: str | None,
549 existing_key_alias: str | None = None,
550) -> None:
551 """
552 Helper function to set rotation fields in key data if auto_rotate is enabled.
554 Args:
555 data: Dictionary to update with rotation fields
556 auto_rotate: Whether auto rotation is enabled
557 rotation_interval: The rotation interval string (required if auto_rotate is True)
558 existing_key_alias: The existing key alias from the database (if any)
559 """
560 if auto_rotate and rotation_interval: 560 ↛ 561line 560 didn't jump to line 561 because the condition on line 560 was never true
561 if (
562 litellm._key_management_settings is not None
563 and litellm._key_management_settings.store_virtual_keys is True
564 and data.get("key_alias") is None
565 and existing_key_alias is None
566 ):
567 raise ProxyException(
568 message="key_alias is required when auto_rotate=True and store_virtual_keys is enabled. This ensures stable secret naming during rotation.",
569 type=ProxyErrorTypes.bad_request_error,
570 param="key_alias",
571 code=400,
572 )
573 data.update(
574 {
575 "auto_rotate": auto_rotate,
576 "rotation_interval": rotation_interval,
577 "key_rotation_at": _calculate_key_rotation_time(rotation_interval),
578 }
579 )
582def _is_allowed_to_make_key_request(
583 user_api_key_dict: UserAPIKeyAuth,
584 user_id: str | None,
585 team_id: str | None,
586) -> bool:
587 """
588 Assert user only creates/updates keys for themselves
590 Relevant issue: https://github.com/BerriAI/litellm/issues/7336
591 """
592 ## BASE CASE - PROXY ADMIN
593 if user_api_key_dict.user_role is not None and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 593 ↛ 596line 593 didn't jump to line 596 because the condition on line 593 was always true
594 return True
596 if user_id is not None:
597 assert user_id == user_api_key_dict.user_id, (
598 f"User can only create keys for themselves. Got user_id={user_id}, Your ID={user_api_key_dict.user_id}"
599 )
601 if team_id is not None:
602 if user_api_key_dict.team_id is not None and user_api_key_dict.team_id == UI_TEAM_ID:
603 return True # handle https://github.com/BerriAI/litellm/issues/7482
605 return True
608def _team_key_operation_team_member_check(
609 assigned_user_id: str | None,
610 team_table: LiteLLM_TeamTableCachedObj,
611 user_api_key_dict: UserAPIKeyAuth,
612 team_key_generation: TeamUIKeyGenerationConfig,
613 route: KeyManagementRoutes,
614):
615 if assigned_user_id is not None:
616 key_assigned_user_in_team: Final = _get_user_in_team(team_table=team_table, user_id=assigned_user_id)
618 if key_assigned_user_in_team is None:
619 raise HTTPException(
620 status_code=400,
621 detail=f"User={assigned_user_id} not assigned to team={team_table.team_id}",
622 )
624 caller_team_role: Final = _get_caller_team_role(team_table=team_table, user_api_key_dict=user_api_key_dict)
626 is_admin: Final = (
627 user_api_key_dict.user_role is not None and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
628 )
630 if is_admin:
631 return True
632 elif caller_team_role is None:
633 raise HTTPException(
634 status_code=400,
635 detail=f"User={user_api_key_dict.user_id} not assigned to team={team_table.team_id}",
636 )
637 elif (
638 "allowed_team_member_roles" in team_key_generation
639 and caller_team_role not in team_key_generation["allowed_team_member_roles"]
640 ):
641 raise HTTPException(
642 status_code=400,
643 detail=f"Team member role {caller_team_role} not in allowed_team_member_roles={team_key_generation['allowed_team_member_roles']}",
644 )
646 TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
647 team_member_role=caller_team_role,
648 team_table=team_table,
649 route=route,
650 )
651 return True
654def _key_generation_required_param_check(data: GenerateKeyRequest, required_params: list[str] | None):
655 if required_params is None:
656 return True
658 data_dict: Final = data.model_dump(exclude_unset=True)
659 for param in required_params:
660 if param not in data_dict:
661 raise HTTPException(
662 status_code=400,
663 detail=f"Required param {param} not in data",
664 )
665 return True
668def _team_key_generation_check(
669 team_table: LiteLLM_TeamTableCachedObj,
670 user_api_key_dict: UserAPIKeyAuth,
671 data: GenerateKeyRequest,
672 route: KeyManagementRoutes,
673):
674 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 674 ↛ 676line 674 didn't jump to line 676 because the condition on line 674 was always true
675 return True
676 if litellm.key_generation_settings is not None and "team_key_generation" in litellm.key_generation_settings:
677 _team_key_generation = litellm.key_generation_settings["team_key_generation"]
678 else:
679 _team_key_generation = TeamUIKeyGenerationConfig(
680 allowed_team_member_roles=["admin", "user"],
681 )
683 _team_key_operation_team_member_check(
684 assigned_user_id=data.user_id,
685 team_table=team_table,
686 user_api_key_dict=user_api_key_dict,
687 team_key_generation=_team_key_generation,
688 route=route,
689 )
690 _key_generation_required_param_check(
691 data,
692 _team_key_generation.get("required_params"),
693 )
695 # Field-level opt-in: non-admin members may only assign access groups when
696 # the team has enabled KEY_ACCESS_GROUP_ASSIGNMENT.
697 TeamMemberPermissionChecks.enforce_member_can_assign_access_groups(
698 user_api_key_dict=user_api_key_dict,
699 team_table=team_table,
700 access_group_ids=data.access_group_ids,
701 )
703 return True
706def _personal_key_membership_check(
707 user_api_key_dict: UserAPIKeyAuth,
708 personal_key_generation: PersonalUIKeyGenerationConfig | None,
709):
710 if personal_key_generation is None or "allowed_user_roles" not in personal_key_generation:
711 return True
713 if user_api_key_dict.user_role not in personal_key_generation["allowed_user_roles"]:
714 raise HTTPException(
715 status_code=400,
716 detail=f"Personal key creation has been restricted by admin. Allowed roles={personal_key_generation['allowed_user_roles']}. Your role={user_api_key_dict.user_role}",
717 )
719 return True
722def _object_permission_to_dict(
723 object_permission: LiteLLM_ObjectPermissionBase | None,
724) -> ObjectPermissionDict | None:
725 if object_permission is None:
726 return None
727 return cast(ObjectPermissionDict, object_permission.model_dump(exclude_unset=True))
730def _personal_key_generation_check(user_api_key_dict: UserAPIKeyAuth, data: GenerateKeyRequest):
731 TeamMemberPermissionChecks.enforce_member_can_assign_access_groups(
732 user_api_key_dict=user_api_key_dict,
733 team_table=None,
734 access_group_ids=data.access_group_ids,
735 )
737 if ( 737 ↛ 743line 737 didn't jump to line 743 because the condition on line 737 was always true
738 litellm.key_generation_settings is None
739 or litellm.key_generation_settings.get("personal_key_generation") is None
740 ):
741 return True
743 _personal_key_generation: Final = litellm.key_generation_settings["personal_key_generation"]
745 _personal_key_membership_check(
746 user_api_key_dict,
747 personal_key_generation=_personal_key_generation,
748 )
750 _key_generation_required_param_check(
751 data,
752 _personal_key_generation.get("required_params"),
753 )
755 return True
758def key_generation_check(
759 team_table: LiteLLM_TeamTableCachedObj | None,
760 user_api_key_dict: UserAPIKeyAuth,
761 data: GenerateKeyRequest,
762 route: KeyManagementRoutes,
763) -> bool:
764 """
765 Check if admin has restricted key creation to certain roles for teams or individuals
766 """
768 if user_api_key_dict.is_team_service_account and data.team_id != user_api_key_dict.team_id: 768 ↛ 769line 768 didn't jump to line 769 because the condition on line 768 was never true
769 raise HTTPException(
770 status_code=403,
771 detail=f"Service account keys can only create keys for their own team. team_id={user_api_key_dict.team_id}",
772 )
774 ## check if key is for team or individual
775 is_team_key: Final = _is_team_key(data=data)
776 _is_admin: Final = (
777 user_api_key_dict.user_role is not None and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
778 )
779 if is_team_key:
780 if team_table is None and litellm.key_generation_settings is not None: 780 ↛ 781line 780 didn't jump to line 781 because the condition on line 780 was never true
781 raise HTTPException(
782 status_code=400,
783 detail=f"Unable to find team object in database. Team ID: {data.team_id}",
784 )
785 elif team_table is None: 785 ↛ 786line 785 didn't jump to line 786 because the condition on line 785 was never true
786 if _is_admin:
787 return True # admins can assign team_id without team table
788 # Non-admin callers must have a valid team (LIT-1884)
789 raise HTTPException(
790 status_code=400,
791 detail=f"Unable to find team object in database. Team ID: {data.team_id}",
792 )
793 return _team_key_generation_check(
794 team_table=team_table,
795 user_api_key_dict=user_api_key_dict,
796 data=data,
797 route=route,
798 )
799 else:
800 return _personal_key_generation_check(user_api_key_dict=user_api_key_dict, data=data)
803def raise_on_invalid_key_logging_config(metadata: Mapping[str, object] | None) -> None:
804 """Key-level logging writes go through key metadata, not /team/callback.
806 Without this the same New Relic config the team endpoint rejects would be
807 accepted here and then silently ignored or misrouted at request time.
808 """
809 error: Final = logging_metadata_config_error(metadata)
810 if error is not None: 810 ↛ 811line 810 didn't jump to line 811 because the condition on line 810 was never true
811 raise HTTPException(status_code=400, detail={"error": error}) # mutable-ok: FastAPI detail contract
814def common_key_access_checks(
815 user_api_key_dict: UserAPIKeyAuth,
816 data: GenerateKeyRequest | UpdateKeyRequest,
817 llm_router: Router | None,
818 premium_user: bool,
819 user_id: str | None = None,
820) -> Literal[True]:
821 """
822 Check if user is allowed to make a key request, for this key
823 """
824 try:
825 _is_allowed_to_make_key_request(
826 user_api_key_dict=user_api_key_dict,
827 user_id=user_id or data.user_id,
828 team_id=data.team_id,
829 )
830 except AssertionError as e:
831 raise HTTPException(
832 status_code=403,
833 detail=str(e),
834 )
835 except Exception as e:
836 raise HTTPException(
837 status_code=500,
838 detail=str(e),
839 )
841 _check_model_access_group(
842 models=data.models,
843 llm_router=llm_router,
844 premium_user=premium_user,
845 )
846 return True
849router: Final = APIRouter()
852def handle_key_type(data: GenerateKeyRequest, data_json: dict) -> dict:
853 """
854 Handle the key type.
855 """
856 key_type: Final = data.key_type
857 if key_type is None:
858 data_json.pop("key_type", None)
859 return data_json
860 data_json["key_type"] = key_type.value
861 if key_type == LiteLLMKeyType.LLM_API:
862 data_json["allowed_routes"] = ["llm_api_routes"]
863 elif key_type == LiteLLMKeyType.MANAGEMENT:
864 data_json["allowed_routes"] = ["management_routes"]
865 elif key_type == LiteLLMKeyType.READ_ONLY:
866 data_json["allowed_routes"] = ["info_routes"]
867 return data_json
870_NON_ADMIN_SAFE_ALLOWED_ROUTES_PRESETS: Final = frozenset({"llm_api_routes", "info_routes"})
873def _validate_caller_can_change_key_ownership(
874 data: BaseModel | None,
875 existing_key_row: LiteLLM_VerificationToken,
876 user_api_key_dict: UserAPIKeyAuth,
877) -> None:
878 """
879 Non-admin callers must not rebind a key's ``user_id`` to a different
880 user. The ``user_id`` on a verification token is what
881 ``_return_user_api_key_auth_obj`` resolves against ``litellm_usertable``
882 to derive the request's role; a non-admin rebinding their own key's
883 ``user_id`` to a ``PROXY_ADMIN`` row promotes themselves.
885 ``/key/update`` already enforces this inline; ``/key/regenerate`` did
886 not. Sharing the check keeps both endpoints — and any future
887 regenerate-style endpoint — consistent.
888 """
889 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value:
890 return
891 if data is None:
892 return
893 # Distinguish "user_id omitted" from "user_id explicitly set to None".
894 # Both leave ``getattr(data, 'user_id', None)`` at None, but only the
895 # explicit-null variant survives ``model_dump(exclude_unset=True)`` in
896 # ``prepare_key_update_data`` and writes NULL to the token row —
897 # detaching the key from its user and bypassing the user-row
898 # role check on subsequent requests.
899 fields_set: Final = getattr(data, "model_fields_set", None) or set()
900 if "user_id" not in fields_set:
901 return
902 incoming_user_id: Final = getattr(data, "user_id", None)
903 if incoming_user_id is None or incoming_user_id == "":
904 raise HTTPException(
905 status_code=403,
906 detail="Non-admin users cannot remove the user_id from a key.",
907 )
908 existing_user_id: Final = getattr(existing_key_row, "user_id", None)
909 if incoming_user_id != existing_user_id:
910 raise HTTPException(
911 status_code=403,
912 detail=(
913 f"Non-admin caller is not allowed to rebind the key from "
914 f"user={existing_user_id} to user={incoming_user_id}"
915 ),
916 )
919def _check_allowed_routes_caller_permission(
920 allowed_routes: list | None,
921 user_api_key_dict: UserAPIKeyAuth,
922 *,
923 allowed_routes_was_provided: bool = False,
924 allow_safe_presets: bool = False,
925) -> None:
926 """
927 Require PROXY_ADMIN when `allowed_routes` is present in the request body,
928 unless the caller went through the `key_type` preset flow.
930 Raw-body call sites pass
931 `allowed_routes_was_provided="allowed_routes" in data.model_fields_set` so a
932 caller that omits the field (model default flows through) is distinct from
933 one that sends any explicit value.
935 Post-`handle_key_type` call sites pass `allow_safe_presets=True` with the
936 values derived by `handle_key_type`; those values are not from the request
937 body, so `allowed_routes_was_provided` stays False and the safe-preset
938 carve-out below accepts any list of tokens in
939 `_NON_ADMIN_SAFE_ALLOWED_ROUTES_PRESETS`.
940 """
941 if not allowed_routes_was_provided and not allowed_routes:
942 return
943 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 943 ↛ 945line 943 didn't jump to line 945 because the condition on line 943 was always true
944 return
945 if (
946 allow_safe_presets
947 and allowed_routes
948 and all(r in _NON_ADMIN_SAFE_ALLOWED_ROUTES_PRESETS for r in allowed_routes)
949 ):
950 return
951 raise HTTPException(
952 status_code=403,
953 detail={
954 "error": (
955 "Only proxy admins can set `allowed_routes` on a key. "
956 "Use `key_type` to pick a preset route bucket instead."
957 )
958 },
959 )
962_READ_ONLY_ALLOWED_ROUTES_PRESET: Final = frozenset(("info_routes",))
965def _is_safe_preset_route_transition(
966 incoming_allowed_routes: Sequence[str] | None,
967 existing_allowed_routes: Sequence[str] | None,
968) -> bool:
969 """
970 True when every route on BOTH sides is a safe `key_type` preset bucket
971 (empty = full access, which non-admins already get from a default
972 `/key/generate`), with one carve-out: a read-only (`info_routes`) key
973 stays read-only, so widening it needs an admin. Requiring the existing
974 side to be a safe preset keeps an owner from clearing an admin-set
975 custom route restriction (LIT-4139).
976 """
977 incoming: Final = frozenset(incoming_allowed_routes or ())
978 existing: Final = frozenset(existing_allowed_routes or ())
979 if not (incoming | existing) <= _NON_ADMIN_SAFE_ALLOWED_ROUTES_PRESETS:
980 return False
981 return existing != _READ_ONLY_ALLOWED_ROUTES_PRESET or incoming == existing
984def _enforce_allowed_routes_update_permission(
985 data: UpdateKeyRequest,
986 existing_key_row: LiteLLM_VerificationToken,
987 user_api_key_dict: UserAPIKeyAuth,
988) -> None:
989 if _is_safe_preset_route_transition(
990 incoming_allowed_routes=data.allowed_routes,
991 existing_allowed_routes=existing_key_row.allowed_routes,
992 ):
993 return
994 _check_allowed_routes_caller_permission(
995 allowed_routes=data.allowed_routes,
996 user_api_key_dict=user_api_key_dict,
997 allowed_routes_was_provided="allowed_routes" in data.model_fields_set,
998 )
1001def _check_permissions_caller_permission(
1002 data: GenerateRequestBase,
1003 user_api_key_dict: UserAPIKeyAuth,
1004) -> None:
1005 """
1006 Require PROXY_ADMIN when `permissions` is present in the request body.
1008 Presence is detected via `data.model_fields_set` so a caller that
1009 omits the field (default flows through) is distinct from one that
1010 sends any explicit value.
1011 """
1012 permissions_in_request: Final = "permissions" in data.model_fields_set
1013 if not permissions_in_request and not data.permissions:
1014 return
1015 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 1015 ↛ 1017line 1015 didn't jump to line 1017 because the condition on line 1015 was always true
1016 return
1017 raise HTTPException(
1018 status_code=403,
1019 detail={"error": "Only proxy admins can set `permissions`."},
1020 )
1023def _check_budget_limits_delegation_ceiling(
1024 budget_limits: list[BudgetLimitEntry] | None,
1025 delegation_ceiling: float | None,
1026 user_api_key_dict: UserAPIKeyAuth,
1027 is_ui_session_team_key: bool,
1028 team_table: LiteLLM_TeamTableCachedObj | None,
1029) -> None:
1030 """
1031 Enforce three invariants on `budget_limits`:
1033 - Every `budget_limits[*].max_budget` must be a finite number; applies
1034 to every caller including proxy admin.
1035 - A CLI session token caller may not set `budget_limits` on a personal
1036 key (one with no `team_id`); mirrors the scalar `max_budget` guard in
1037 `_common_key_generation_helper`.
1038 - Non-admin callers may not set a window above their delegation ceiling.
1039 """
1040 if not budget_limits:
1041 return
1042 non_finite: Final = next((w for w in budget_limits if not math.isfinite(w.max_budget)), None)
1043 if non_finite is not None: 1043 ↛ 1044line 1043 didn't jump to line 1044 because the condition on line 1043 was never true
1044 raise HTTPException(
1045 status_code=400,
1046 detail={"error": (f"budget_limits entry max_budget ({non_finite.max_budget}) must be a finite number.")},
1047 )
1048 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 1048 ↛ 1050line 1048 didn't jump to line 1050 because the condition on line 1048 was always true
1049 return
1050 if is_ui_session_team_key:
1051 return
1052 if user_api_key_dict.is_session_token and team_table is None:
1053 raise HTTPException(
1054 status_code=400,
1055 detail={
1056 "error": ("budget_limits cannot be set without specifying team_id when using a CLI session token.")
1057 },
1058 )
1059 if delegation_ceiling is None:
1060 return
1061 over_ceiling: Final = next((w for w in budget_limits if w.max_budget > delegation_ceiling), None)
1062 if over_ceiling is not None:
1063 raise HTTPException(
1064 status_code=400,
1065 detail={
1066 "error": (
1067 f"budget_limits entry max_budget ({over_ceiling.max_budget}) "
1068 f"cannot exceed the caller's own max_budget ({delegation_ceiling})."
1069 )
1070 },
1071 )
1074async def validate_team_id_used_in_service_account_request(
1075 team_id: str | None,
1076 prisma_client: PrismaClient | None,
1077):
1078 """
1079 Validate team_id is used in the request body for generating a service account key
1080 """
1081 if team_id is None: 1081 ↛ 1087line 1081 didn't jump to line 1087 because the condition on line 1081 was always true
1082 raise HTTPException(
1083 status_code=400,
1084 detail="team_id is required for service account keys. Please specify `team_id` in the request body.",
1085 )
1087 if prisma_client is None:
1088 raise HTTPException(
1089 status_code=400,
1090 detail="prisma_client is required for service account keys. Please specify `prisma_client` in the request body.",
1091 )
1093 # check if team_id exists in the database
1094 team: Final = await _prisma_table(TeamRepository(prisma_client)).find_unique(
1095 where={"team_id": team_id},
1096 )
1097 if team is None:
1098 raise HTTPException(
1099 status_code=400,
1100 detail="team_id does not exist in the database. Please specify a valid `team_id` in the request body.",
1101 )
1102 return True
1105_BUDGET_NUMERIC_KEYS = frozenset(
1106 ["max_budget", "soft_budget", "max_parallel_requests", "tpm_limit", "rpm_limit", "tpd_limit"]
1107)
1110def _enforce_upperbound_key_params(
1111 data: GenerateKeyRequest | UpdateKeyRequest,
1112 fill_defaults: bool = True,
1113) -> None:
1114 """
1115 Enforce upperbound limits on key parameters.
1117 For key generation (fill_defaults=True): fills None values with upperbound defaults.
1118 For key update (fill_defaults=False): only validates explicitly provided values.
1119 """
1120 # Always reject NaN / Inf regardless of whether an upperbound config is set
1121 # (GHSA-2rv4-xv66-fpjg): float('nan') passes every `< 0` check because
1122 # nan < 0 is False, and spend >= nan is always False, permanently disabling
1123 # budget enforcement for any key that carries it.
1124 for elem in data:
1125 key, value = elem
1126 if key in _BUDGET_NUMERIC_KEYS and value is not None:
1127 if not math.isfinite(value): 1127 ↛ 1128line 1127 didn't jump to line 1128 because the condition on line 1127 was never true
1128 raise HTTPException(
1129 status_code=400,
1130 detail={"error": f"{key} must be a finite number. Received: {value}"},
1131 )
1133 if litellm.upperbound_key_generate_params is None: 1133 ↛ 1136line 1133 didn't jump to line 1136 because the condition on line 1133 was always true
1134 return
1136 for elem in data:
1137 key, value = elem
1138 upperbound_value = getattr(litellm.upperbound_key_generate_params, key, None)
1139 if upperbound_value is not None:
1140 if value is None:
1141 if fill_defaults:
1142 setattr(data, key, upperbound_value)
1143 else:
1144 if key in [
1145 "max_budget",
1146 "max_parallel_requests",
1147 "tpm_limit",
1148 "rpm_limit",
1149 ]:
1150 if value > upperbound_value:
1151 raise HTTPException(
1152 status_code=400,
1153 detail={
1154 "error": f"{key} is over max limit set in config - user_value={value}; max_value={upperbound_value}"
1155 },
1156 )
1157 elif key in ["budget_duration", "duration"]:
1158 upperbound_duration = duration_in_seconds(duration=upperbound_value)
1159 if value == "-1":
1160 user_duration = float("inf")
1161 else:
1162 user_duration = duration_in_seconds(duration=value)
1163 if user_duration > upperbound_duration:
1164 raise HTTPException(
1165 status_code=400,
1166 detail={
1167 "error": f"{key} is over max limit set in config - user_value={value}; max_value={upperbound_value}"
1168 },
1169 )
1172async def _common_key_generation_helper(
1173 data: GenerateKeyRequest,
1174 user_api_key_dict: UserAPIKeyAuth,
1175 litellm_changed_by: str | None,
1176 team_table: LiteLLM_TeamTableCachedObj | None,
1177) -> GenerateKeyResponse:
1178 from litellm.proxy import proxy_server
1179 from litellm.proxy.proxy_server import (
1180 litellm_proxy_admin_name,
1181 llm_router,
1182 premium_user,
1183 prisma_client,
1184 )
1186 common_key_access_checks(
1187 user_api_key_dict=user_api_key_dict,
1188 data=data,
1189 llm_router=llm_router,
1190 premium_user=premium_user,
1191 )
1193 validate_budget_duration(data.budget_duration)
1194 raise_on_invalid_key_logging_config(data.metadata)
1196 if data.throttle_on_budget_exceeded is True and user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value: 1196 ↛ 1197line 1196 didn't jump to line 1197 because the condition on line 1196 was never true
1197 raise HTTPException(
1198 status_code=403,
1199 detail={"error": "Only proxy admins can enable throttle_on_budget_exceeded on a key."},
1200 )
1202 await _validate_end_user_budget_id_change(
1203 requested_budget_id=_requested_end_user_budget_id(data),
1204 existing_budget_id=None,
1205 user_api_key_dict=user_api_key_dict,
1206 prisma_client=prisma_client,
1207 )
1209 enforce_output_token_estimates_are_admin_only(
1210 data=data,
1211 existing_metadata=None,
1212 user_api_key_dict=user_api_key_dict,
1213 entity="key",
1214 )
1215 enforce_batch_enqueued_token_limit_is_admin_only(
1216 data=data,
1217 existing_metadata=None,
1218 user_api_key_dict=user_api_key_dict,
1219 entity="key",
1220 )
1222 if data.metadata is not None and data.metadata.get("service_account_id") is not None and data.team_id is None: 1222 ↛ 1223line 1222 didn't jump to line 1223 because the condition on line 1222 was never true
1223 await validate_team_id_used_in_service_account_request(
1224 team_id=data.team_id,
1225 prisma_client=prisma_client,
1226 )
1228 # Capture caller-supplied max_budget and team_id before any defaults or
1229 # upperbound params can fill them, so the ceiling check and its team-key
1230 # exemption key off what the caller explicitly requested, not a value that
1231 # default_key_generate_params injected.
1232 _requested_max_budget: Final = data.max_budget
1233 _requested_team_id: Final = data.team_id
1234 _requested_metadata: Final = data.metadata # pyright: ignore[reportUnknownMemberType] # request models declare `metadata` as bare dict
1236 # check if user set default key/generate params on config.yaml
1237 if litellm.default_key_generate_params is not None: 1237 ↛ 1238line 1237 didn't jump to line 1238 because the condition on line 1237 was never true
1238 for elem in _model_items(data):
1239 key, value = elem
1240 if (
1241 value is None
1242 and (key != "budget_duration" or key not in data.model_fields_set)
1243 and key
1244 in [
1245 "max_budget",
1246 "user_id",
1247 "team_id",
1248 "max_parallel_requests",
1249 "tpm_limit",
1250 "rpm_limit",
1251 "budget_duration",
1252 "duration",
1253 ]
1254 ):
1255 default_value = litellm.default_key_generate_params.get(key)
1256 if default_value is not None:
1257 setattr(data, key, default_value)
1258 elif key == "models" and value == []:
1259 setattr(data, key, litellm.default_key_generate_params.get(key, []))
1260 elif key == "metadata" and value == {}:
1261 setattr(data, key, litellm.default_key_generate_params.get(key, {}))
1263 # check if user set upperbound key/generate params on config.yaml
1264 _enforce_upperbound_key_params(data, fill_defaults=True)
1266 # Delegated-authority ceiling (GHSA-q775-qw9r-2r4g): a non-admin caller
1267 # cannot grant a key a higher budget than their own authority.
1268 # UI session personal keys are capped by user_max_budget when it is available.
1269 is_ui_session_token: Final = user_api_key_dict.team_id == UI_SESSION_TOKEN_TEAM_ID
1270 is_ui_session_team_key = is_ui_session_token and _requested_team_id is not None
1271 if ( 1271 ↛ 1278line 1271 didn't jump to line 1278 because the condition on line 1271 was never true
1272 user_api_key_dict.is_session_token
1273 and user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value
1274 and not is_ui_session_team_key
1275 and _requested_max_budget is not None
1276 and team_table is None
1277 ):
1278 raise HTTPException(
1279 status_code=400,
1280 detail={
1281 "error": (
1282 f"max_budget ({_requested_max_budget}) cannot be set without "
1283 "specifying team_id when using a CLI session token."
1284 )
1285 },
1286 )
1287 delegation_ceiling: Final = (
1288 user_api_key_dict.user_max_budget
1289 if is_ui_session_token and user_api_key_dict.user_max_budget is not None
1290 else user_api_key_dict.max_budget
1291 if user_api_key_dict.max_budget is not None
1292 else (team_table.max_budget if user_api_key_dict.is_session_token and team_table is not None else None)
1293 )
1294 if ( 1294 ↛ 1301line 1294 didn't jump to line 1301 because the condition on line 1294 was never true
1295 user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value
1296 and not is_ui_session_team_key
1297 and _requested_max_budget is not None
1298 and delegation_ceiling is not None
1299 and _requested_max_budget > delegation_ceiling
1300 ):
1301 raise HTTPException(
1302 status_code=400,
1303 detail={
1304 "error": (
1305 f"max_budget ({_requested_max_budget}) cannot exceed the caller's "
1306 f"own max_budget ({delegation_ceiling})."
1307 )
1308 },
1309 )
1311 _check_budget_limits_delegation_ceiling(
1312 budget_limits=data.budget_limits,
1313 delegation_ceiling=delegation_ceiling,
1314 user_api_key_dict=user_api_key_dict,
1315 is_ui_session_team_key=is_ui_session_team_key,
1316 team_table=team_table,
1317 )
1318 _check_permissions_caller_permission(
1319 data=data,
1320 user_api_key_dict=user_api_key_dict,
1321 )
1322 _check_disable_global_guardrails_caller_permission(
1323 data.disable_global_guardrails,
1324 _requested_metadata,
1325 user_api_key_dict,
1326 )
1328 # APPLY ENTERPRISE KEY MANAGEMENT PARAMS
1329 try:
1330 from litellm_enterprise.proxy.management_endpoints.key_management_endpoints import (
1331 apply_enterprise_key_management_params,
1332 )
1334 data = apply_enterprise_key_management_params(data, team_table)
1335 except Exception as e:
1336 verbose_proxy_logger.debug(
1337 "litellm.proxy.proxy_server.generate_key_fn(): Enterprise key management params not applied - %s", e
1338 )
1340 await _enforce_custom_key_policy(
1341 hook=_custom_key_policy_hook(proxy_server),
1342 build_policy_request=lambda: CustomKeyPolicyRequest(
1343 operation="generate",
1344 existing_key=None,
1345 effective_key=_effective_key_for_generate(data=data, now=datetime.now(timezone.utc)),
1346 request=data,
1347 ),
1348 )
1350 # TODO: @ishaan-jaff: Migrate all budget tracking to use LiteLLM_BudgetTable
1351 _budget_id = data.budget_id
1352 if prisma_client is not None and data.soft_budget is not None:
1353 # create the Budget Row for the LiteLLM Verification Token
1354 budget_row: Final = LiteLLM_BudgetTable(
1355 soft_budget=data.soft_budget,
1356 model_max_budget=data.model_max_budget or {},
1357 )
1358 new_budget: Final = prisma_client.jsonify_object(budget_row.json(exclude_none=True))
1360 _budget: Final[prisma_models.LiteLLM_BudgetTable] = await BudgetRepository(prisma_client).table.create(
1361 data={
1362 **new_budget,
1363 "created_by": user_api_key_dict.user_id or litellm_proxy_admin_name,
1364 "updated_by": user_api_key_dict.user_id or litellm_proxy_admin_name,
1365 }
1366 )
1367 _budget_id = getattr(_budget, "budget_id", None)
1369 # ADD METADATA FIELDS
1370 # Set Management Endpoint Metadata Fields
1371 for field in LiteLLM_ManagementEndpoint_MetadataFields_Premium:
1372 if getattr(data, field, None) is not None:
1373 _set_object_metadata_field(
1374 object_data=data,
1375 field_name=field,
1376 value=getattr(data, field),
1377 )
1378 delattr(data, field)
1380 for field in LiteLLM_ManagementEndpoint_MetadataFields:
1381 if getattr(data, field, None) is not None:
1382 _set_object_metadata_field(
1383 object_data=data,
1384 field_name=field,
1385 value=getattr(data, field),
1386 )
1387 delattr(data, field)
1389 data_json = data.model_dump(exclude_unset=True, exclude_none=True)
1391 data_json = handle_key_type(data, data_json)
1393 # Re-check allowed_routes after handle_key_type, since key_type can derive
1394 # an elevated bucket (e.g. ["management_routes"]) that wasn't present in
1395 # the original request body. The safe presets produced by handle_key_type
1396 # for non-elevated buckets are accepted here; the raw-body pre-checks at
1397 # the entry of each handler keep their default strictness.
1398 _check_allowed_routes_caller_permission(
1399 allowed_routes=data_json.get("allowed_routes"),
1400 user_api_key_dict=user_api_key_dict,
1401 allow_safe_presets=True,
1402 )
1404 # if we get max_budget passed to /key/generate, then use it as key_max_budget. Since generate_key_helper_fn is used to make new users
1405 if "max_budget" in data_json:
1406 data_json["key_max_budget"] = data_json.pop("max_budget", None)
1407 if _budget_id is not None:
1408 data_json["budget_id"] = _budget_id
1410 # Only set budget_duration on key when explicitly provided. Keys with budget_id
1411 # but no explicit budget_duration follow their linked budget tier's schedule;
1412 # reset_budget_for_litellm_budget_table() resets them when the tier resets.
1413 # This avoids duplicating budget_duration on keys so tier updates apply automatically.
1414 if "budget_duration" in data_json: 1414 ↛ 1415line 1414 didn't jump to line 1415 because the condition on line 1414 was never true
1415 data_json["key_budget_duration"] = data_json.pop("budget_duration", None)
1417 if user_api_key_dict.user_id is not None: 1417 ↛ 1422line 1417 didn't jump to line 1422 because the condition on line 1417 was always true
1418 data_json["created_by"] = user_api_key_dict.user_id
1419 data_json["updated_by"] = user_api_key_dict.user_id
1421 # Set tags on the new key
1422 if "tags" in data_json: 1422 ↛ 1423line 1422 didn't jump to line 1423 because the condition on line 1422 was never true
1423 from litellm.proxy.proxy_server import premium_user
1425 if premium_user is not True and data_json["tags"] is not None:
1426 raise ValueError(f"Only premium users can add tags to keys. {CommonProxyErrors.not_premium_user.value}")
1428 _metadata: Final = data_json.get("metadata")
1429 if not _metadata:
1430 data_json["metadata"] = {"tags": data_json["tags"]}
1431 else:
1432 data_json["metadata"]["tags"] = data_json["tags"]
1434 data_json.pop("tags")
1436 # Validate MCP servers in object_permission are within team scope
1437 _is_proxy_admin_caller: Final = user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
1438 normalized_object_permission: Final = await validate_key_mcp_servers_against_team(
1439 object_permission=data_json.get("object_permission"),
1440 team_obj=team_table,
1441 prisma_client=prisma_client,
1442 is_proxy_admin=_is_proxy_admin_caller,
1443 )
1444 if normalized_object_permission is not None:
1445 data_json["object_permission"] = normalized_object_permission
1446 await validate_key_search_tools_against_team(
1447 object_permission=data_json.get("object_permission"),
1448 team_obj=team_table,
1449 is_proxy_admin=_is_proxy_admin_caller,
1450 )
1451 await validate_key_vector_stores_against_team(
1452 object_permission=data_json.get("object_permission"),
1453 team_obj=team_table,
1454 is_proxy_admin=_is_proxy_admin_caller,
1455 )
1457 # Merge default_key_generate_params.object_permission in *after* the team-scope
1458 # checks above, so an admin-configured default (e.g. vector_stores, search_tools)
1459 # is never mistaken for a caller-requested permission and rejected by those
1460 # non-admin/no-team checks. Only fields the caller left unset are filled in.
1461 _default_object_permission: Final = (
1462 litellm.default_key_generate_params.get("object_permission")
1463 if litellm.default_key_generate_params is not None
1464 else None
1465 )
1466 if isinstance(_default_object_permission, dict): 1466 ↛ 1467line 1466 didn't jump to line 1467 because the condition on line 1466 was never true
1467 _caller_object_permission: Final = data_json.get("object_permission")
1468 if _caller_object_permission is None:
1469 data_json["object_permission"] = dict(_default_object_permission)
1470 elif isinstance(_caller_object_permission, dict):
1471 for _op_field, _op_default_value in _default_object_permission.items():
1472 _caller_object_permission.setdefault(_op_field, _op_default_value)
1474 data_json = await _set_object_permission(
1475 data_json=data_json,
1476 prisma_client=prisma_client,
1477 )
1479 _validate_key_alias_format(key_alias=data_json.get("key_alias", None))
1481 await _enforce_unique_key_alias(
1482 key_alias=data_json.get("key_alias", None),
1483 prisma_client=prisma_client,
1484 )
1486 # Reject custom key values if disabled by admin
1487 await _check_custom_key_allowed(data.key)
1489 # Validate user-provided key format
1490 if data.key is not None and not data.key.startswith("sk-"): 1490 ↛ 1491line 1490 didn't jump to line 1491 because the condition on line 1490 was never true
1491 _masked: Final = f"{data.key[:4]}****{data.key[-4:]}" if len(data.key) > 8 else "****"
1492 raise HTTPException(
1493 status_code=400,
1494 detail={"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {_masked}"},
1495 )
1497 if data.key is not None and len(data.key) < MINIMUM_CUSTOM_KEY_LENGTH: 1497 ↛ 1498line 1497 didn't jump to line 1498 because the condition on line 1497 was never true
1498 raise HTTPException(
1499 status_code=400,
1500 detail={
1501 "error": f"Invalid key format. LiteLLM Virtual Key must be at least {MINIMUM_CUSTOM_KEY_LENGTH} characters long."
1502 },
1503 )
1505 # check org key limits - done here to handle inheriting org id from team
1506 if data.organization_id is not None: 1506 ↛ 1507line 1506 didn't jump to line 1507 because the condition on line 1506 was never true
1507 from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
1509 if prisma_client:
1510 # Mirror the membership rule applied to /key/update: when the
1511 # caller specifies an organization_id, require that they are a
1512 # member of (or proxy admin over) the target organization.
1513 _is_proxy_admin: Final = (
1514 user_api_key_dict.user_role is not None
1515 and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
1516 )
1517 _org_inherited_from_team: Final = (
1518 team_table is not None
1519 and team_table.organization_id is not None
1520 and data.organization_id == team_table.organization_id
1521 )
1522 if not _is_proxy_admin and not _org_inherited_from_team:
1523 await _validate_caller_can_assign_key_org(
1524 user_api_key_dict=user_api_key_dict,
1525 organization_id=data.organization_id,
1526 prisma_client=prisma_client,
1527 )
1529 org_table: Final = await get_org_object(
1530 org_id=data.organization_id,
1531 user_api_key_cache=user_api_key_cache,
1532 prisma_client=prisma_client,
1533 )
1534 if org_table is None:
1535 raise HTTPException(
1536 status_code=400,
1537 detail=f"Organization not found for organization_id={data.organization_id}",
1538 )
1539 await _check_org_key_limits(
1540 org_table=org_table,
1541 data=data,
1542 prisma_client=prisma_client,
1543 )
1545 response = await generate_key_helper_fn(request_type="key", **data_json, table_name="key", llm_router=llm_router)
1547 response["soft_budget"] = data.soft_budget # include the user-input soft budget in the response
1549 response = GenerateKeyResponse.model_validate(response)
1551 response.token = response.token_id # remap token to use the hash, and leave the key in the `key` field [TODO]: clean up generate_key_helper_fn to do this
1553 asyncio.create_task(
1554 KeyManagementEventHooks.async_key_generated_hook(
1555 data=data,
1556 response=response,
1557 user_api_key_dict=user_api_key_dict,
1558 litellm_changed_by=litellm_changed_by,
1559 )
1560 )
1562 return response
1565def _check_key_model_specific_limits(
1566 keys: Sequence[LiteLLM_VerificationToken],
1567 data: GenerateKeyRequest | UpdateKeyRequest,
1568 entity_rpm_limit: int | None,
1569 entity_tpm_limit: int | None,
1570 entity_model_rpm_limit_dict: dict[str, int],
1571 entity_model_tpm_limit_dict: dict[str, int],
1572 entity_type: str, # "team" or "organization"
1573) -> None:
1574 """
1575 Generic function to check if a key is allocating model specific limits.
1576 Raises an error if we're overallocating.
1577 """
1578 model_rpm_limit: Final = getattr(data, "model_rpm_limit", None) or (
1579 data.metadata.get("model_rpm_limit", None) if data.metadata else None
1580 )
1581 model_tpm_limit: Final = getattr(data, "model_tpm_limit", None) or (
1582 data.metadata.get("model_tpm_limit", None) if data.metadata else None
1583 )
1584 if model_rpm_limit is None and model_tpm_limit is None:
1585 return
1587 # get total model specific tpm/rpm limit
1588 model_specific_rpm_limit: Final[dict[str, int]] = {}
1589 model_specific_tpm_limit: Final[dict[str, int]] = {}
1591 for key in keys:
1592 if key.metadata.get("model_rpm_limit", None) is not None:
1593 for model, rpm_limit in key.metadata.get("model_rpm_limit", {}).items():
1594 model_specific_rpm_limit[model] = model_specific_rpm_limit.get(model, 0) + rpm_limit
1595 if key.metadata.get("model_tpm_limit", None) is not None:
1596 for model, tpm_limit in key.metadata.get("model_tpm_limit", {}).items():
1597 model_specific_tpm_limit[model] = model_specific_tpm_limit.get(model, 0) + tpm_limit
1599 if model_rpm_limit is not None:
1600 for model, rpm_limit in model_rpm_limit.items():
1601 if entity_rpm_limit is not None and model_specific_rpm_limit.get(model, 0) + rpm_limit > entity_rpm_limit:
1602 raise HTTPException(
1603 status_code=400,
1604 detail=f"Allocated RPM limit={model_specific_rpm_limit.get(model, 0)} + Key RPM limit={rpm_limit} is greater than {entity_type} RPM limit={entity_rpm_limit}",
1605 )
1606 elif entity_model_rpm_limit_dict:
1607 entity_model_specific_rpm_limit = entity_model_rpm_limit_dict.get(model)
1608 if (
1609 entity_model_specific_rpm_limit
1610 and model_specific_rpm_limit.get(model, 0) + rpm_limit > entity_model_specific_rpm_limit
1611 ):
1612 raise HTTPException(
1613 status_code=400,
1614 detail=f"Allocated RPM limit={model_specific_rpm_limit.get(model, 0)} + Key RPM limit={rpm_limit} is greater than {entity_type} RPM limit={entity_model_specific_rpm_limit}",
1615 )
1617 if model_tpm_limit is not None:
1618 for model, tpm_limit in model_tpm_limit.items():
1619 if entity_tpm_limit is not None and model_specific_tpm_limit.get(model, 0) + tpm_limit > entity_tpm_limit:
1620 raise HTTPException(
1621 status_code=400,
1622 detail=f"Allocated TPM limit={model_specific_tpm_limit.get(model, 0)} + Key TPM limit={tpm_limit} is greater than {entity_type} TPM limit={entity_tpm_limit}",
1623 )
1624 elif entity_model_tpm_limit_dict:
1625 entity_model_specific_tpm_limit = entity_model_tpm_limit_dict.get(model)
1626 if (
1627 entity_model_specific_tpm_limit
1628 and model_specific_tpm_limit.get(model, 0) + tpm_limit > entity_model_specific_tpm_limit
1629 ):
1630 raise HTTPException(
1631 status_code=400,
1632 detail=f"Allocated TPM limit={model_specific_tpm_limit.get(model, 0)} + Key TPM limit={tpm_limit} is greater than {entity_type} TPM limit={entity_model_specific_tpm_limit}",
1633 )
1636def _check_key_rpm_tpm_limits(
1637 keys: Sequence[LiteLLM_VerificationToken],
1638 data: GenerateKeyRequest | UpdateKeyRequest,
1639 entity_rpm_limit: int | None,
1640 entity_tpm_limit: int | None,
1641 entity_type: str, # "team" or "organization"
1642) -> None:
1643 """
1644 Generic function to check if a key is allocating rpm/tpm limits.
1645 Raises an error if we're overallocating.
1646 """
1647 if keys is not None and len(keys) > 0:
1648 allocated_tpm = sum(key.tpm_limit for key in keys if key.tpm_limit is not None)
1649 allocated_rpm = sum(key.rpm_limit for key in keys if key.rpm_limit is not None)
1650 else:
1651 allocated_tpm = 0
1652 allocated_rpm = 0
1654 if (
1655 data.tpm_limit is not None
1656 and entity_tpm_limit is not None
1657 and data.tpm_limit + allocated_tpm > entity_tpm_limit
1658 ):
1659 raise HTTPException(
1660 status_code=400,
1661 detail=f"Allocated TPM limit={allocated_tpm} + Key TPM limit={data.tpm_limit} is greater than {entity_type} TPM limit={entity_tpm_limit}",
1662 )
1663 if (
1664 data.rpm_limit is not None
1665 and entity_rpm_limit is not None
1666 and data.rpm_limit + allocated_rpm > entity_rpm_limit
1667 ):
1668 raise HTTPException(
1669 status_code=400,
1670 detail=f"Allocated RPM limit={allocated_rpm} + Key RPM limit={data.rpm_limit} is greater than {entity_type} RPM limit={entity_rpm_limit}",
1671 )
1674def check_team_key_model_specific_limits(
1675 keys: Sequence[LiteLLM_VerificationToken],
1676 team_table: LiteLLM_TeamTableCachedObj,
1677 data: GenerateKeyRequest | UpdateKeyRequest,
1678) -> None:
1679 """
1680 Check if the team key is allocating model specific limits. If so, raise an error if we're overallocating.
1681 """
1682 entity_model_rpm_limit_dict = {}
1683 entity_model_tpm_limit_dict = {}
1684 if team_table.metadata:
1685 entity_model_rpm_limit_dict = team_table.metadata.get("model_rpm_limit", {})
1686 entity_model_tpm_limit_dict = team_table.metadata.get("model_tpm_limit", {})
1688 _check_key_model_specific_limits(
1689 keys=keys,
1690 data=data,
1691 entity_rpm_limit=team_table.rpm_limit,
1692 entity_tpm_limit=team_table.tpm_limit,
1693 entity_model_rpm_limit_dict=entity_model_rpm_limit_dict,
1694 entity_model_tpm_limit_dict=entity_model_tpm_limit_dict,
1695 entity_type="team",
1696 )
1699def check_team_key_rpm_tpm_limits(
1700 keys: Sequence[LiteLLM_VerificationToken],
1701 team_table: LiteLLM_TeamTableCachedObj,
1702 data: GenerateKeyRequest | UpdateKeyRequest,
1703) -> None:
1704 """
1705 Check if the team key is allocating rpm/tpm limits. If so, raise an error if we're overallocating.
1706 """
1707 _check_key_rpm_tpm_limits(
1708 keys=keys,
1709 data=data,
1710 entity_rpm_limit=team_table.rpm_limit,
1711 entity_tpm_limit=team_table.tpm_limit,
1712 entity_type="team",
1713 )
1716async def _check_team_key_limits(
1717 team_table: LiteLLM_TeamTableCachedObj,
1718 data: GenerateKeyRequest | UpdateKeyRequest,
1719 prisma_client: PrismaClient,
1720) -> None:
1721 """
1722 Check if the team key is allocating guaranteed throughput limits. If so, raise an error if we're overallocating.
1724 Only runs check if tpm_limit_type or rpm_limit_type is "guaranteed_throughput"
1725 """
1726 if data.tpm_limit_type != "guaranteed_throughput" and data.rpm_limit_type != "guaranteed_throughput": 1726 ↛ 1732line 1726 didn't jump to line 1732 because the condition on line 1726 was always true
1727 return
1728 # get all team keys
1729 # calculate allocated tpm/rpm limit
1730 # check if specified tpm/rpm limit is greater than allocated tpm/rpm limit
1732 keys = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
1733 where={"team_id": team_table.team_id},
1734 )
1735 # Exclude the key being updated to avoid double-counting its limits.
1736 # data.key may be a raw key (sk-...) or a pre-hashed token_id.
1737 if isinstance(data, UpdateKeyRequest) and data.key is not None:
1738 hashed_key: Final = _hash_token_if_needed(data.key)
1739 keys = [key for key in keys if key.token != hashed_key]
1740 check_team_key_model_specific_limits(
1741 keys=keys,
1742 team_table=team_table,
1743 data=data,
1744 )
1745 check_team_key_rpm_tpm_limits(
1746 keys=keys,
1747 team_table=team_table,
1748 data=data,
1749 )
1752_INHERITED_MODEL_SENTINELS: Final = frozenset(
1753 {SpecialModelNames.all_team_models.value, SpecialModelNames.all_proxy_models.value}
1754)
1757async def _check_project_key_limits(
1758 project_id: str,
1759 data: GenerateKeyRequest | UpdateKeyRequest,
1760 prisma_client: PrismaClient,
1761 user_api_key_cache: UserApiKeyCache,
1762) -> None:
1763 """
1764 Validate that key's models and budget respect its project's limits.
1766 - Key models must be a subset of project models, except the all-team-models / all-proxy-models
1767 sentinels, which inherit a parent scope and are narrowed by the project at request time
1768 - Key max_budget must be <= project max_budget
1769 """
1770 project_obj: Final = await get_project_object(
1771 project_id=project_id,
1772 prisma_client=prisma_client,
1773 user_api_key_cache=user_api_key_cache,
1774 )
1776 if project_obj is None: 1776 ↛ 1783line 1776 didn't jump to line 1783 because the condition on line 1776 was always true
1777 raise HTTPException(
1778 status_code=404,
1779 detail={"error": f"Project not found, project_id={project_id}"},
1780 )
1782 # Validate key models are a subset of project models
1783 if data.models and len(project_obj.models) > 0:
1784 for m in data.models:
1785 if m not in project_obj.models and m not in _INHERITED_MODEL_SENTINELS:
1786 raise HTTPException(
1787 status_code=400,
1788 detail={
1789 "error": f"Model '{m}' not in project's allowed models. Project allowed models={project_obj.models}. Project: {project_id}"
1790 },
1791 )
1793 # Validate key max_budget <= project max_budget
1794 project_max_budget = None
1795 if project_obj.litellm_budget_table is not None:
1796 project_max_budget = getattr(project_obj.litellm_budget_table, "max_budget", None)
1798 if data.max_budget is not None and project_max_budget is not None and data.max_budget > project_max_budget:
1799 raise HTTPException(
1800 status_code=400,
1801 detail={
1802 "error": f"Key max_budget ({data.max_budget}) exceeds project's max_budget ({project_max_budget}). Project: {project_id}"
1803 },
1804 )
1807def check_org_key_model_specific_limits(
1808 keys: Sequence[LiteLLM_VerificationToken],
1809 org_table: LiteLLM_OrganizationTable,
1810 data: GenerateKeyRequest | UpdateKeyRequest,
1811) -> None:
1812 """
1813 Check if the organization key is allocating model specific limits. If so, raise an error if we're overallocating.
1814 """
1815 # Get org limits from budget table if available
1816 entity_rpm_limit = None
1817 entity_tpm_limit = None
1818 entity_model_rpm_limit_dict = {}
1819 entity_model_tpm_limit_dict = {}
1821 if org_table.litellm_budget_table is not None:
1822 entity_rpm_limit = org_table.litellm_budget_table.rpm_limit
1823 entity_tpm_limit = org_table.litellm_budget_table.tpm_limit
1825 if org_table.metadata:
1826 entity_model_rpm_limit_dict = org_table.metadata.get("model_rpm_limit", {})
1827 entity_model_tpm_limit_dict = org_table.metadata.get("model_tpm_limit", {})
1829 _check_key_model_specific_limits(
1830 keys=keys,
1831 data=data,
1832 entity_rpm_limit=entity_rpm_limit,
1833 entity_tpm_limit=entity_tpm_limit,
1834 entity_model_rpm_limit_dict=entity_model_rpm_limit_dict,
1835 entity_model_tpm_limit_dict=entity_model_tpm_limit_dict,
1836 entity_type="organization",
1837 )
1840def check_org_key_rpm_tpm_limits(
1841 keys: Sequence[LiteLLM_VerificationToken],
1842 org_table: LiteLLM_OrganizationTable,
1843 data: GenerateKeyRequest | UpdateKeyRequest,
1844) -> None:
1845 """
1846 Check if the organization key is allocating rpm/tpm limits. If so, raise an error if we're overallocating.
1847 """
1848 # Get org limits from budget table if available
1849 entity_rpm_limit = None
1850 entity_tpm_limit = None
1852 if org_table.litellm_budget_table is not None:
1853 entity_rpm_limit = org_table.litellm_budget_table.rpm_limit
1854 entity_tpm_limit = org_table.litellm_budget_table.tpm_limit
1856 _check_key_rpm_tpm_limits(
1857 keys=keys,
1858 data=data,
1859 entity_rpm_limit=entity_rpm_limit,
1860 entity_tpm_limit=entity_tpm_limit,
1861 entity_type="organization",
1862 )
1865async def _validate_caller_can_assign_key_org(
1866 user_api_key_dict: UserAPIKeyAuth,
1867 organization_id: str,
1868 prisma_client: PrismaClient,
1869) -> None:
1870 """Reject ``/key/update`` requests that point a key at an organization
1871 the caller does not belong to.
1873 Mirrors the org-membership rule already enforced on ``/key/list`` in
1874 ``validate_key_list_check``. Proxy admins are checked at the call site.
1875 """
1876 if user_api_key_dict.user_id is None:
1877 raise HTTPException(
1878 status_code=status.HTTP_403_FORBIDDEN,
1879 detail="Cannot assign a key to an organization without a user_id on the caller's token",
1880 )
1882 user_row: Final = await _prisma_table(UserRepository(prisma_client)).find_unique(
1883 where={"user_id": user_api_key_dict.user_id},
1884 include={"organization_memberships": True},
1885 )
1886 memberships: Final = getattr(user_row, "organization_memberships", None) if user_row else None
1887 member_org_ids: Final = {
1888 membership.organization_id for membership in (memberships or []) if membership.organization_id is not None
1889 }
1890 if organization_id not in member_org_ids:
1891 raise HTTPException(
1892 status_code=status.HTTP_403_FORBIDDEN,
1893 detail=f"Caller is not a member of organization_id={organization_id}",
1894 )
1897async def _check_org_key_limits(
1898 org_table: LiteLLM_OrganizationTable,
1899 data: GenerateKeyRequest | UpdateKeyRequest,
1900 prisma_client: PrismaClient,
1901) -> None:
1902 """
1903 Check if the organization key is allocating guaranteed throughput limits. If so, raise an error if we're overallocating.
1905 Only runs check if tpm_limit_type or rpm_limit_type is "guaranteed_throughput"
1906 """
1908 rpm_limit_type: Final = getattr(data, "rpm_limit_type", None) or (
1909 data.metadata.get("rpm_limit_type", None) if data.metadata else None
1910 )
1911 tpm_limit_type: Final = getattr(data, "tpm_limit_type", None) or (
1912 data.metadata.get("tpm_limit_type", None) if data.metadata else None
1913 )
1915 if tpm_limit_type != "guaranteed_throughput" and rpm_limit_type != "guaranteed_throughput":
1916 return
1917 # get all organization keys
1918 # calculate allocated tpm/rpm limit
1919 # check if specified tpm/rpm limit is greater than allocated tpm/rpm limit
1920 keys = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
1921 where={"organization_id": org_table.organization_id},
1922 )
1923 # Exclude the key being updated to avoid double-counting its limits.
1924 # data.key may be a raw key (sk-...) or a pre-hashed token_id.
1925 if isinstance(data, UpdateKeyRequest) and data.key is not None:
1926 hashed_key: Final = _hash_token_if_needed(data.key)
1927 keys = [key for key in keys if key.token != hashed_key]
1928 check_org_key_model_specific_limits(
1929 keys=keys,
1930 org_table=org_table,
1931 data=data,
1932 )
1933 check_org_key_rpm_tpm_limits(
1934 keys=keys,
1935 org_table=org_table,
1936 data=data,
1937 )
1940@router.post(
1941 "/key/generate",
1942 tags=["key management"],
1943 dependencies=[Depends(user_api_key_auth)],
1944 response_model=GenerateKeyResponse,
1945)
1946@management_endpoint_wrapper
1947async def generate_key_fn(
1948 data: GenerateKeyRequest,
1949 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1950 litellm_changed_by: str | None = Header(
1951 None,
1952 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
1953 ),
1954):
1955 """
1956 Generate an API key based on the provided data.
1958 Docs: https://docs.litellm.ai/docs/proxy/virtual_keys
1960 Parameters:
1961 - duration: Optional[str] - Specify the length of time the token is valid for. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
1962 - key_alias: Optional[str] - User defined key alias
1963 - key: Optional[str] - User defined key value. Must start with 'sk-' and be at least 16 characters long. If not set, a 16-digit unique sk-key is created for you.
1964 - team_id: Optional[str] - The team id of the key
1965 - user_id: Optional[str] - The user id of the key
1966 - agent_id: Optional[str] - The agent id associated with the key.
1967 - organization_id: Optional[str] - The organization id of the key. If not set, and team_id is set, the organization id will be the same as the team id. If conflict, an error will be raised.
1968 - project_id: Optional[str] - The project id of the key. When set, models and max_budget are validated against the project's limits.
1969 - budget_id: Optional[str] - The budget id associated with the key. Created by calling `/budget/new`.
1970 - end_user_budget_id: Optional[str] - Proxy admin only. Budget id applied to end users first seen through this key that carry no budget of their own. Takes precedence over `litellm_settings.max_end_user_budget_id`.
1971 - models: Optional[list] - Model_name's a user is allowed to call. (if empty, key is allowed to call all models)
1972 - aliases: Optional[dict] - Any alias mappings, on top of anything in the config.yaml model list. - https://docs.litellm.ai/docs/proxy/virtual_keys#managing-auth---upgradedowngrade-models
1973 - config: Optional[dict] - any key-specific configs, overrides config in config.yaml
1974 - spend: Optional[int] - Amount spent by key. Default is 0. Will be updated by proxy whenever key is used. https://docs.litellm.ai/docs/proxy/virtual_keys#managing-auth---tracking-spend
1975 - send_invite_email: Optional[bool] - Whether to send an invite email to the user_id, with the generate key
1976 - max_budget: Optional[float] - Specify max budget for a given key.
1977 - budget_duration: Optional[str] - Budget is reset at the end of specified duration. If not set, budget is never reset. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
1978 - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x.
1979 - metadata: Optional[dict] - Metadata for key, store information for key. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }
1980 - guardrails: Optional[List[str]] - List of active guardrails for the key
1981 - policies: Optional[List[str]] - List of policy names to apply to the key. Policies define guardrails, conditions, and inheritance rules.
1982 - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. Proxy admin only.
1983 - throttle_on_budget_exceeded: Optional[bool] - When the key exceeds its max_budget, throttle its tpm/rpm to the global budget_exceeded_throttle_percentage instead of blocking the key entirely.
1984 - enable_prompt_caching: Optional[bool] - Auto-inject prompt caching breakpoints (Anthropic cache_control markers) on requests made with this key. Supported Claude models on Anthropic, Bedrock, Vertex AI, and Azure AI only.
1985 - permissions: Optional[dict] - key-specific permissions. Currently just used for turning off pii masking (if connected). Example - {"pii": false}
1986 - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}}. IF null or {} then no model specific budget.
1987 - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
1988 - model_rpm_limit: Optional[dict] - key-specific model rpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific rpm limit.
1989 - model_tpm_limit: Optional[dict] - key-specific model tpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific tpm limit.
1990 - default_estimated_output_tokens: Optional[int] - Proxy admin only. Expected output tokens reserved for TPM limiting when a request omits max_tokens. Positive integer. Falls back to the team setting, then to the built-in estimate.
1991 - default_estimated_output_tokens_per_model: Optional[dict] - Proxy admin only. Per-model override of the above. Example - {"gpt-4": 4096, "gpt-3.5-turbo": 1024}. Takes precedence over the key-wide value.
1992 - mcp_rpm_limit: Optional[dict] - key-specific per-MCP-server rpm limit, keyed by MCP server name (alias if set, else the configured name). Example - {"github": 100, "slack": 200}. IF null or {} then no MCP-specific rpm limit.
1993 - tag_rpm_limit: Optional[dict] - key-specific per-request-tag rpm limit, keyed by request tag. Example - {"cell-1": 1000, "cell-2": 500}. Each tag gets an independent counter; requests whose tag is absent fall back to the key-level rpm limit.
1994 - tpm_limit_type: Optional[str] - Type of tpm limit. Options: "best_effort_throughput" (no error if we're overallocating tpm), "guaranteed_throughput" (raise an error if we're overallocating tpm), "dynamic" (dynamically exceed limit when no 429 errors). Defaults to "best_effort_throughput".
1995 - rpm_limit_type: Optional[str] - Type of rpm limit. Options: "best_effort_throughput" (no error if we're overallocating rpm), "guaranteed_throughput" (raise an error if we're overallocating rpm), "dynamic" (dynamically exceed limit when no 429 errors). Defaults to "best_effort_throughput".
1996 - allowed_cache_controls: Optional[list] - List of allowed cache control values. Example - ["no-cache", "no-store"]. See all values - https://docs.litellm.ai/docs/proxy/caching#turn-on--off-caching-per-request
1997 - blocked: Optional[bool] - Whether the key is blocked.
1998 - rpm_limit: Optional[int] - Specify rpm limit for a given key (Requests per minute)
1999 - tpm_limit: Optional[int] - Specify tpm limit for a given key (Tokens per minute)
2000 - tpd_limit: Optional[int] - Specify tpd limit for a given key (Tokens per day). Charged by batch submissions instead of tpm_limit/rpm_limit.
2001 - soft_budget: Optional[float] - Specify soft budget for a given key. Will trigger a slack alert when this soft budget is reached.
2002 - tags: Optional[List[str]] - Tags for [tracking spend](https://litellm.vercel.app/docs/proxy/enterprise#tracking-spend-for-custom-tags) and/or doing [tag-based routing](https://litellm.vercel.app/docs/proxy/tag_routing).
2003 - prompts: Optional[List[str]] - List of prompts that the key is allowed to use.
2004 - enforced_params: Optional[List[str]] - List of enforced params for the key (Enterprise only). [Docs](https://docs.litellm.ai/docs/proxy/enterprise#enforce-required-params-for-llm-requests)
2005 - prompts: Optional[List[str]] - List of prompts that the key is allowed to use.
2006 - allowed_routes: Optional[list] - List of allowed routes for the key. Store the actual route or store a wildcard pattern for a set of routes. Example - ["/chat/completions", "/embeddings", "/keys/*"]
2007 - allowed_passthrough_routes: Optional[list] - List of allowed pass through endpoints for the key. Store the actual endpoint or store a wildcard pattern for a set of endpoints. Example - ["/my-custom-endpoint"]. Use this instead of allowed_routes, if you just want to specify which pass through endpoints the key can access, without specifying the routes. If allowed_routes is specified, allowed_pass_through_endpoints is ignored.
2008 - object_permission: Optional[LiteLLM_ObjectPermissionBase] - key-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"], "agents": ["agent_1", "agent_2"], "agent_access_groups": ["dev_group"]}. IF null or {} then no object permission.
2009 - key_type: Optional[str] - Type of key that determines default allowed routes. Options: "llm_api" (can call LLM API routes), "management" (can call management routes), "read_only" (can only call info/read routes), "default" (uses default allowed routes). Defaults to "default".
2010 - prompts: Optional[List[str]] - List of allowed prompts for the key. If specified, the key will only be able to use these specific prompts.
2011 - auto_rotate: Optional[bool] - Whether this key should be automatically rotated (regenerated)
2012 - rotation_interval: Optional[str] - How often to auto-rotate this key (e.g., '30s', '30m', '30h', '30d'). Required if auto_rotate=True.
2013 - allowed_vector_store_indexes: Optional[List[dict]] - List of allowed vector store indexes for the key. Example - [{"index_name": "my-index", "index_permissions": ["write", "read"]}]. If specified, the key will only be able to use these specific vector store indexes. Create index, using `/v1/indexes` endpoint.
2014 - router_settings: Optional[UpdateRouterConfig] - key-specific router settings. Example - {"model_group_retry_policy": {"gpt-4": {"RateLimitErrorRetries": 5}}}. IF null or {} then no router settings.
2015 - access_group_ids: Optional[List[str]] - List of access group IDs to associate with the key. Access groups define which models a key can access. Example - ["access_group_1", "access_group_2"].
2016 - budget_limits: Optional[list] - List of concurrent budget windows for the key. Each window specifies a budget_limit, time_period, and optional budget_duration. Example - [{"budget_limit": 10.0, "time_period": "1d"}, {"budget_limit": 50.0, "time_period": "7d"}].
2018 Examples:
2020 1. Allow users to turn on/off pii masking
2022 ```bash
2023 curl --location 'http://0.0.0.0:4000/key/generate' \
2024 --header 'Authorization: Bearer sk-1234' \
2025 --header 'Content-Type: application/json' \
2026 --data '{
2027 "permissions": {"allow_pii_controls": true}
2028 }'
2029 ```
2031 Returns:
2032 - key: (str) The generated api key
2033 - expires: (datetime) Datetime object for when key expires.
2034 - user_id: (str) Unique user id - used for tracking spend across multiple keys for same user id.
2035 """
2036 try:
2037 from litellm.proxy import proxy_server
2038 from litellm.proxy._types import CommonProxyErrors
2039 from litellm.proxy.proxy_server import (
2040 prisma_client,
2041 user_api_key_cache,
2042 )
2044 if prisma_client is None: 2044 ↛ 2045line 2044 didn't jump to line 2045 because the condition on line 2044 was never true
2045 raise HTTPException(
2046 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
2047 detail={"error": CommonProxyErrors.db_not_connected_error.value},
2048 )
2050 verbose_proxy_logger.debug("entered /key/generate")
2052 await check_org_admin_can_generate_keys(user_api_key_dict=user_api_key_dict)
2054 # Validate budget values are not negative and are finite numbers
2055 # (GHSA-2rv4-xv66-fpjg): float('nan') passes `< 0` because nan < 0 is False.
2056 if data.max_budget is not None and (not math.isfinite(data.max_budget) or data.max_budget < 0): 2056 ↛ 2057line 2056 didn't jump to line 2057 because the condition on line 2056 was never true
2057 raise HTTPException(
2058 status_code=400,
2059 detail={"error": f"max_budget must be a non-negative finite number. Received: {data.max_budget}"},
2060 )
2061 _validate_soft_budget_value(data.soft_budget)
2063 custom_key_generate_hook: Final[Callable[..., Awaitable[Mapping[str, object]]] | None] = (
2064 _custom_key_generate_hook(proxy_server)
2065 )
2066 if custom_key_generate_hook is not None: 2066 ↛ 2067line 2066 didn't jump to line 2067 because the condition on line 2066 was never true
2067 if inspect.iscoroutinefunction(custom_key_generate_hook):
2068 result: Final = await custom_key_generate_hook(data)
2069 else:
2070 raise ValueError("user_custom_key_generate must be a coroutine")
2071 decision: Final = result.get("decision", True)
2072 message: Final = result.get("message", "Authentication Failed - Custom Auth Rule")
2073 if not decision:
2074 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=message)
2076 _check_allowed_routes_caller_permission(
2077 allowed_routes=data.allowed_routes,
2078 user_api_key_dict=user_api_key_dict,
2079 allowed_routes_was_provided="allowed_routes" in data.model_fields_set,
2080 )
2081 _check_passthrough_routes_caller_permission(
2082 data=data,
2083 user_api_key_dict=user_api_key_dict,
2084 )
2086 # For non-admin internal users: auto-assign caller's user_id if not provided
2087 # This prevents creating unbound keys with no user association (LIT-1884)
2088 _is_proxy_admin: Final = (
2089 user_api_key_dict.user_role is not None
2090 and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
2091 )
2092 if not _is_proxy_admin and data.user_id is None: 2092 ↛ 2093line 2092 didn't jump to line 2093 because the condition on line 2092 was never true
2093 data.user_id = user_api_key_dict.user_id
2094 verbose_proxy_logger.warning(
2095 "key/generate: auto-assigning user_id=%s for non-admin caller",
2096 user_api_key_dict.user_id,
2097 )
2099 team_table: LiteLLM_TeamTableCachedObj | None = None
2100 if data.team_id is not None:
2101 try:
2102 team_table = await get_team_object(
2103 team_id=data.team_id,
2104 prisma_client=prisma_client,
2105 user_api_key_cache=user_api_key_cache,
2106 parent_otel_span=user_api_key_dict.parent_otel_span,
2107 check_db_only=True,
2108 )
2109 except Exception as e:
2110 verbose_proxy_logger.debug("Error getting team object in `/key/generate`: %s", e)
2111 # For non-admin callers, team must exist (LIT-1884)
2112 if not _is_proxy_admin:
2113 raise HTTPException(
2114 status_code=400,
2115 detail=f"Team not found for team_id={data.team_id}. Non-admin users cannot create keys for non-existent teams.",
2116 )
2118 key_generation_check(
2119 team_table=team_table,
2120 user_api_key_dict=user_api_key_dict,
2121 data=data,
2122 route=KeyManagementRoutes.KEY_GENERATE,
2123 )
2125 if team_table is not None:
2126 await _check_team_key_limits(
2127 team_table=team_table,
2128 data=data,
2129 prisma_client=prisma_client,
2130 )
2132 # Validate key against project limits if project_id is set
2133 if data.project_id is not None:
2134 await _check_project_key_limits(
2135 project_id=data.project_id,
2136 data=data,
2137 prisma_client=prisma_client,
2138 user_api_key_cache=user_api_key_cache,
2139 )
2141 return await _common_key_generation_helper(
2142 data=data,
2143 user_api_key_dict=user_api_key_dict,
2144 litellm_changed_by=litellm_changed_by,
2145 team_table=team_table,
2146 )
2148 except Exception as e:
2149 verbose_proxy_logger.exception("litellm.proxy.proxy_server.generate_key_fn(): Exception occured - %s", e)
2150 raise handle_exception_on_proxy(e)
2153@router.post(
2154 "/key/service-account/generate",
2155 tags=["key management"],
2156 dependencies=[Depends(user_api_key_auth)],
2157)
2158@management_endpoint_wrapper
2159async def generate_service_account_key_fn(
2160 data: GenerateKeyRequest,
2161 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
2162 litellm_changed_by: str | None = Header(
2163 None,
2164 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
2165 ),
2166):
2167 """
2168 Generate a Service Account API key based on the provided data. This key does not belong to any user. It belongs to the team.
2170 Why use a service account key?
2171 - Prevent key from being deleted when user is deleted.
2172 - Apply team limits, not team member limits to key.
2174 Docs: https://docs.litellm.ai/docs/proxy/virtual_keys
2176 Parameters:
2177 - duration: Optional[str] - Specify the length of time the token is valid for. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
2178 - key_alias: Optional[str] - User defined key alias
2179 - key: Optional[str] - User defined key value. Must start with 'sk-' and be at least 16 characters long. If not set, a 16-digit unique sk-key is created for you.
2180 - team_id: Optional[str] - The team id of the key
2181 - user_id: Optional[str] - [NON-FUNCTIONAL] THIS WILL BE IGNORED. The user id of the key
2182 - budget_id: Optional[str] - The budget id associated with the key. Created by calling `/budget/new`.
2183 - end_user_budget_id: Optional[str] - Proxy admin only. Budget id applied to end users first seen through this key that carry no budget of their own. Omit to keep the current value, pass an empty string to clear it.
2184 - models: Optional[list] - Model_name's a user is allowed to call. (if empty, key is allowed to call all models)
2185 - aliases: Optional[dict] - Any alias mappings, on top of anything in the config.yaml model list. - https://docs.litellm.ai/docs/proxy/virtual_keys#managing-auth---upgradedowngrade-models
2186 - config: Optional[dict] - any key-specific configs, overrides config in config.yaml
2187 - spend: Optional[int] - Amount spent by key. Default is 0. Will be updated by proxy whenever key is used. https://docs.litellm.ai/docs/proxy/virtual_keys#managing-auth---tracking-spend
2188 - send_invite_email: Optional[bool] - Whether to send an invite email to the user_id, with the generate key
2189 - max_budget: Optional[float] - Specify max budget for a given key.
2190 - budget_duration: Optional[str] - Budget is reset at the end of specified duration. If not set, budget is never reset. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
2191 - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x.
2192 - metadata: Optional[dict] - Metadata for key, store information for key. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }
2193 - guardrails: Optional[List[str]] - List of active guardrails for the key
2194 - permissions: Optional[dict] - key-specific permissions. Currently just used for turning off pii masking (if connected). Example - {"pii": false}
2195 - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}}. IF null or {} then no model specific budget.
2196 - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
2197 - model_rpm_limit: Optional[dict] - key-specific model rpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific rpm limit.
2198 - model_tpm_limit: Optional[dict] - key-specific model tpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific tpm limit.
2199 - default_estimated_output_tokens: Optional[int] - Proxy admin only. Expected output tokens reserved for TPM limiting when a request omits max_tokens. Positive integer. Falls back to the team setting, then to the built-in estimate.
2200 - default_estimated_output_tokens_per_model: Optional[dict] - Proxy admin only. Per-model override of the above. Example - {"gpt-4": 4096, "gpt-3.5-turbo": 1024}. Takes precedence over the key-wide value.
2201 - mcp_rpm_limit: Optional[dict] - key-specific per-MCP-server rpm limit, keyed by MCP server name (alias if set, else the configured name). Example - {"github": 100, "slack": 200}. IF null or {} then no MCP-specific rpm limit.
2202 - tpm_limit_type: Optional[str] - TPM rate limit type - "best_effort_throughput", "guaranteed_throughput", or "dynamic"
2203 - rpm_limit_type: Optional[str] - RPM rate limit type - "best_effort_throughput", "guaranteed_throughput", or "dynamic"
2204 - allowed_cache_controls: Optional[list] - List of allowed cache control values. Example - ["no-cache", "no-store"]. See all values - https://docs.litellm.ai/docs/proxy/caching#turn-on--off-caching-per-request
2205 - blocked: Optional[bool] - Whether the key is blocked.
2206 - rpm_limit: Optional[int] - Specify rpm limit for a given key (Requests per minute)
2207 - tpm_limit: Optional[int] - Specify tpm limit for a given key (Tokens per minute)
2208 - tpd_limit: Optional[int] - Specify tpd limit for a given key (Tokens per day). Charged by batch submissions instead of tpm_limit/rpm_limit.
2209 - soft_budget: Optional[float] - Specify soft budget for a given key. Will trigger a slack alert when this soft budget is reached.
2210 - tags: Optional[List[str]] - Tags for [tracking spend](https://litellm.vercel.app/docs/proxy/enterprise#tracking-spend-for-custom-tags) and/or doing [tag-based routing](https://litellm.vercel.app/docs/proxy/tag_routing).
2211 - enforced_params: Optional[List[str]] - List of enforced params for the key (Enterprise only). [Docs](https://docs.litellm.ai/docs/proxy/enterprise#enforce-required-params-for-llm-requests)
2212 - allowed_routes: Optional[list] - List of allowed routes for the key. Store the actual route or store a wildcard pattern for a set of routes. Example - ["/chat/completions", "/embeddings", "/keys/*"]
2213 - object_permission: Optional[LiteLLM_ObjectPermissionBase] - key-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"], "agents": ["agent_1", "agent_2"], "agent_access_groups": ["dev_group"]}. IF null or {} then no object permission.
2214 Examples:
2215 - allowed_vector_store_indexes: Optional[List[dict]] - List of allowed vector store indexes for the key. Example - [{"index_name": "my-index", "index_permissions": ["write", "read"]}]. If specified, the key will only be able to use these specific vector store indexes. Create index, using `/v1/indexes` endpoint.
2218 1. Allow users to turn on/off pii masking
2220 ```bash
2221 curl --location 'http://0.0.0.0:4000/key/generate' \
2222 --header 'Authorization: Bearer sk-1234' \
2223 --header 'Content-Type: application/json' \
2224 --data '{
2225 "permissions": {"allow_pii_controls": true}
2226 }'
2227 ```
2229 Returns:
2230 - key: (str) The generated api key
2231 - expires: (datetime) Datetime object for when key expires.
2232 - user_id: (str) Unique user id - used for tracking spend across multiple keys for same user id.
2234 """
2235 from litellm.proxy import proxy_server
2236 from litellm.proxy._types import CommonProxyErrors
2237 from litellm.proxy.proxy_server import (
2238 prisma_client,
2239 user_api_key_cache,
2240 )
2242 if prisma_client is None: 2242 ↛ 2243line 2242 didn't jump to line 2243 because the condition on line 2242 was never true
2243 raise HTTPException(
2244 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
2245 detail={"error": CommonProxyErrors.db_not_connected_error.value},
2246 )
2248 await check_org_admin_can_generate_keys(user_api_key_dict=user_api_key_dict)
2250 _check_allowed_routes_caller_permission(
2251 allowed_routes=data.allowed_routes,
2252 user_api_key_dict=user_api_key_dict,
2253 allowed_routes_was_provided="allowed_routes" in data.model_fields_set,
2254 )
2255 _check_passthrough_routes_caller_permission(
2256 data=data,
2257 user_api_key_dict=user_api_key_dict,
2258 )
2260 await validate_team_id_used_in_service_account_request(
2261 team_id=data.team_id,
2262 prisma_client=prisma_client,
2263 )
2265 if data.metadata is None or data.metadata.get("service_account_id") is None:
2266 service_account_id: Final = data.key_alias or str(uuid.uuid4())
2267 stamped_metadata: Final = { # mutable-ok: GenerateKeyRequest.metadata is a plain dict field
2268 **(data.metadata or MappingProxyType({})),
2269 "service_account_id": service_account_id,
2270 }
2271 data.metadata = stamped_metadata # rebind-ok: the request carries the stamp so it persists on the key
2273 verbose_proxy_logger.debug("entered /key/generate")
2275 custom_key_generate_hook: Final[Callable[..., Awaitable[Mapping[str, object]]] | None] = _custom_key_generate_hook(
2276 proxy_server
2277 )
2278 if custom_key_generate_hook is not None:
2279 if inspect.iscoroutinefunction(custom_key_generate_hook):
2280 result: Final = await custom_key_generate_hook(data)
2281 else:
2282 raise ValueError("user_custom_key_generate must be a coroutine")
2283 decision: Final = result.get("decision", True)
2284 message: Final = result.get("message", "Authentication Failed - Custom Auth Rule")
2285 if not decision:
2286 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=message)
2287 team_table: LiteLLM_TeamTableCachedObj | None = None
2288 if data.team_id is not None:
2289 try:
2290 team_table = await get_team_object(
2291 team_id=data.team_id,
2292 prisma_client=prisma_client,
2293 user_api_key_cache=user_api_key_cache,
2294 parent_otel_span=user_api_key_dict.parent_otel_span,
2295 check_db_only=True,
2296 )
2297 except Exception as e:
2298 verbose_proxy_logger.debug("Error getting team object in `/key/generate`: %s", e)
2299 team_table = None
2301 if team_table is not None:
2302 await _check_team_key_limits(
2303 team_table=team_table,
2304 data=data,
2305 prisma_client=prisma_client,
2306 )
2308 key_generation_check(
2309 team_table=team_table,
2310 user_api_key_dict=user_api_key_dict,
2311 data=data,
2312 route=KeyManagementRoutes.KEY_GENERATE_SERVICE_ACCOUNT,
2313 )
2315 data.user_id = None # do not allow user_id to be set for service account keys
2317 return await _common_key_generation_helper(
2318 data=data,
2319 user_api_key_dict=user_api_key_dict,
2320 litellm_changed_by=litellm_changed_by,
2321 team_table=team_table,
2322 )
2325def prepare_metadata_fields(data: BaseModel, non_default_values: dict, existing_metadata: dict) -> dict:
2326 """
2327 Check LiteLLM_ManagementEndpoint_MetadataFields (proxy/_types.py) for fields that are allowed to be updated
2328 """
2329 raise_on_invalid_key_logging_config(non_default_values.get("metadata"))
2331 if "metadata" not in non_default_values: # allow user to set metadata to none
2332 non_default_values["metadata"] = existing_metadata.copy()
2334 casted_metadata: Final = cast(dict, non_default_values["metadata"])
2336 # Reserved metadata fields are immutable once set. Preserve the existing value
2337 # when omitted, reject any explicit attempt to change it (including null).
2338 for reserved_field in LiteLLM_Reserved_Metadata_Fields:
2339 existing_value = existing_metadata.get(reserved_field)
2340 if existing_value is None:
2341 continue
2342 if casted_metadata is None or (
2343 reserved_field in casted_metadata and casted_metadata[reserved_field] != existing_value
2344 ):
2345 raise HTTPException(
2346 status_code=400,
2347 detail=f"{reserved_field} is immutable once set and cannot be changed via update.",
2348 )
2349 casted_metadata[reserved_field] = existing_value
2351 data_json: Final = _as_object_dict(data.model_dump(exclude_unset=True, exclude_none=True))
2353 try:
2354 for k, v in data_json.items():
2355 if k in LiteLLM_ManagementEndpoint_MetadataFields:
2356 if isinstance(v, datetime):
2357 casted_metadata[k] = v.isoformat()
2358 else:
2359 casted_metadata[k] = v
2360 if k in LiteLLM_ManagementEndpoint_MetadataFields_Premium:
2361 from litellm.proxy.utils import _premium_user_check
2363 if v:
2364 _premium_user_check(k)
2365 casted_metadata[k] = v
2367 except Exception as e:
2368 verbose_proxy_logger.exception(
2369 "litellm.proxy.proxy_server.prepare_metadata_fields(): Exception occured - %s", e
2370 )
2372 non_default_values["metadata"] = encrypt_callback_vars(casted_metadata)
2373 return non_default_values
2376def _validate_soft_budget_value(soft_budget: float | None) -> None:
2377 if soft_budget is not None and (not math.isfinite(soft_budget) or soft_budget < 0): 2377 ↛ 2378line 2377 didn't jump to line 2378 because the condition on line 2377 was never true
2378 raise HTTPException(
2379 status_code=400,
2380 detail={"error": f"soft_budget must be a non-negative finite number. Received: {soft_budget}"},
2381 )
2384async def _update_key_soft_budget(
2385 db: _KeyUpdateTx,
2386 existing_key_row: LiteLLM_VerificationToken,
2387 soft_budget: float | None,
2388 changed_by: str,
2389) -> str | None:
2390 existing_budget_id: Final = existing_key_row.budget_id
2391 if existing_budget_id is not None:
2392 budget_update: Final[_BudgetRowSoftBudgetUpdate] = {"soft_budget": soft_budget, "updated_by": changed_by}
2393 budget_where: Final[_BudgetRowWhere] = {"budget_id": existing_budget_id}
2394 await db.litellm_budgettable.update(where=budget_where, data=budget_update)
2395 return existing_budget_id
2396 if soft_budget is None:
2397 return None
2398 budget_create: Final[_BudgetRowSoftBudgetCreate] = {
2399 "soft_budget": soft_budget,
2400 "created_by": changed_by,
2401 "updated_by": changed_by,
2402 }
2403 created_budget: Final = await db.litellm_budgettable.create(data=budget_create)
2404 return created_budget.budget_id
2407async def _apply_soft_budget_update(
2408 data: UpdateKeyRequest,
2409 non_default_values: Mapping[str, object],
2410 db: _KeyUpdateTx,
2411 existing_key_row: LiteLLM_VerificationToken,
2412 changed_by: str,
2413) -> Mapping[str, object]:
2414 remaining: Final = MappingProxyType({k: v for k, v in non_default_values.items() if k != "soft_budget"})
2415 updated_budget_id: Final = await _update_key_soft_budget(
2416 db=db,
2417 existing_key_row=existing_key_row,
2418 soft_budget=data.soft_budget,
2419 changed_by=changed_by,
2420 )
2421 if updated_budget_id is not None and existing_key_row.budget_id is None:
2422 return MappingProxyType({**remaining, "budget_id": updated_budget_id})
2423 return remaining
2426async def _update_key_row_with_soft_budget(
2427 prisma_client: PrismaClient,
2428 key: str,
2429 data: UpdateKeyRequest,
2430 non_default_values: Mapping[str, object],
2431 existing_key_row: LiteLLM_VerificationToken,
2432 changed_by: str,
2433) -> _KeyUpdateResult:
2434 hashed_token: Final = _hash_token_if_needed(key)
2435 key_where: Final[_KeyRowWhere] = {"token": hashed_token}
2436 tx: _KeyUpdateTx
2437 async with prisma_client.tx() as tx:
2438 update_values: Final = await _apply_soft_budget_update(
2439 data=data,
2440 non_default_values=non_default_values,
2441 db=tx,
2442 existing_key_row=existing_key_row,
2443 changed_by=changed_by,
2444 )
2445 updated_row: Final = await tx.litellm_verificationtoken.update(
2446 where=key_where,
2447 data=with_settings_updated_at(
2448 prisma_client.jsonify_object(MappingProxyType({**update_values, "token": hashed_token}))
2449 ),
2450 )
2451 updated_data: Final[Mapping[str, object]] = (
2452 updated_row.model_dump() if updated_row is not None else MappingProxyType({})
2453 )
2454 result: Final[_KeyUpdateResult] = {"token": hashed_token, "data": updated_data}
2455 return result
2458async def prepare_key_update_data(
2459 data: UpdateKeyRequest | RegenerateKeyRequest,
2460 existing_key_row: LiteLLM_VerificationToken,
2461 *,
2462 prisma_client: PrismaClient | None = None,
2463 llm_router: Router | None = None,
2464):
2465 if data.router_settings is not None or (
2466 "router_settings" not in data.model_fields_set
2467 and "team_id" in data.model_fields_set
2468 and data.team_id != existing_key_row.team_id
2469 ):
2470 effective_settings: Final = (
2471 data.router_settings
2472 if data.router_settings is not None
2473 else _StoredKeyRouterSettings.model_validate(existing_key_row, from_attributes=True).router_settings
2474 )
2475 await validate_router_settings_weights(
2476 effective_settings,
2477 team_id=data.team_id if "team_id" in data.model_fields_set else existing_key_row.team_id,
2478 prisma_client=prisma_client,
2479 llm_router=llm_router,
2480 )
2481 data_json: Final[dict] = data.model_dump(exclude_unset=True)
2482 data_json.pop("key", None)
2483 data_json.pop("new_key", None)
2484 data_json.pop("grace_period", None) # Request-only param, not a DB column
2485 if (
2486 data.metadata is not None
2487 and data.metadata.get("service_account_id") is not None
2488 and (data.team_id or existing_key_row.team_id) is None
2489 ):
2490 raise HTTPException(
2491 status_code=400,
2492 detail="team_id is required for service account keys. Please specify `team_id` in the request body.",
2493 )
2494 non_default_values = {}
2495 # ADD METADATA FIELDS
2496 # Set Management Endpoint Metadata Fields
2497 for field in LiteLLM_ManagementEndpoint_MetadataFields_Premium:
2498 if getattr(data, field, None) is not None:
2499 _set_object_metadata_field(
2500 object_data=data,
2501 field_name=field,
2502 value=getattr(data, field),
2503 )
2504 for k, v in data_json.items():
2505 if k in LiteLLM_ManagementEndpoint_MetadataFields or k in LiteLLM_ManagementEndpoint_MetadataFields_Premium:
2506 continue
2507 non_default_values[k] = v
2509 if "duration" in non_default_values:
2510 duration: Final = non_default_values.pop("duration")
2511 if duration is None or duration == "-1":
2512 # Set expires to None to indicate the key never expires
2513 non_default_values["expires"] = None
2514 elif duration and (isinstance(duration, str)) and len(duration) > 0:
2515 duration_s: Final = duration_in_seconds(duration=duration)
2516 expires: Final = datetime.now(timezone.utc) + timedelta(seconds=duration_s)
2517 non_default_values["expires"] = expires
2519 if "budget_duration" in non_default_values:
2520 budget_duration: Final = non_default_values.pop("budget_duration")
2521 if budget_duration is None:
2522 non_default_values["budget_duration"] = None
2523 non_default_values["budget_reset_at"] = None
2524 elif isinstance(budget_duration, str) and len(budget_duration) > 0:
2525 from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time
2527 key_reset_at: Final = get_budget_reset_time(budget_duration=budget_duration)
2528 non_default_values["budget_reset_at"] = key_reset_at
2529 non_default_values["budget_duration"] = budget_duration
2531 if "budget_limits" in non_default_values:
2532 raw_windows: Final = non_default_values["budget_limits"]
2533 if raw_windows:
2534 from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time
2536 initialized_windows: Final = []
2537 for window in raw_windows:
2538 w = window if isinstance(window, dict) else window.model_dump()
2539 w["reset_at"] = get_budget_reset_time(budget_duration=w["budget_duration"]).isoformat()
2540 initialized_windows.append(w)
2541 non_default_values["budget_limits"] = json.dumps(initialized_windows)
2542 else:
2543 # [] / None clears the field; prisma-client-py has no DbNull
2544 # sentinel for Json? columns, so store the JSON literal null
2545 non_default_values["budget_limits"] = json.dumps(None)
2547 _metadata: Final = existing_key_row.metadata or {}
2549 # validate model_max_budget
2550 if "model_max_budget" in non_default_values:
2551 validate_model_max_budget(non_default_values["model_max_budget"])
2553 # Serialize router_settings to JSON if present
2554 if "router_settings" in non_default_values and non_default_values["router_settings"] is not None:
2555 non_default_values["router_settings"] = safe_dumps(non_default_values["router_settings"])
2557 non_default_values = prepare_metadata_fields(
2558 data=data, non_default_values=non_default_values, existing_metadata=_metadata
2559 )
2561 return non_default_values
2564async def _handle_update_object_permission(
2565 data_json: dict,
2566 existing_key_row: LiteLLM_VerificationToken,
2567 prisma_client: PrismaClient,
2568) -> dict:
2569 """Persist the requested object permission row and swap it for its id, only after the key policy allowed the write."""
2570 if "object_permission" not in data_json:
2571 return data_json
2573 object_permission_id: Final = await handle_update_object_permission_common(
2574 data_json=data_json,
2575 existing_object_permission_id=existing_key_row.object_permission_id,
2576 prisma_client=prisma_client,
2577 )
2579 # Add the object_permission_id to data_json if one was created/updated
2580 if object_permission_id is not None:
2581 data_json["object_permission_id"] = object_permission_id
2582 verbose_proxy_logger.debug("updated object_permission_id: %s", object_permission_id)
2584 return data_json
2587def is_different_team(data: UpdateKeyRequest, existing_key_row: LiteLLM_VerificationToken) -> bool:
2588 if data.team_id is None:
2589 return False
2590 if existing_key_row.team_id is None:
2591 return True
2592 return data.team_id != existing_key_row.team_id
2595def _validate_max_budget(max_budget: float | None) -> None:
2596 """
2597 Validate that max_budget is not negative.
2599 Args:
2600 max_budget: The max_budget value to validate
2602 Raises:
2603 HTTPException: If max_budget is negative
2604 """
2605 if max_budget is not None and (not math.isfinite(max_budget) or max_budget < 0):
2606 raise HTTPException(
2607 status_code=400,
2608 detail={"error": f"max_budget must be a non-negative finite number. Received: {max_budget}"},
2609 )
2612async def _get_and_validate_existing_key(
2613 token: str | None, prisma_client: PrismaClient | None, key_alias: str | None = None
2614) -> LiteLLM_VerificationToken:
2615 """
2616 Get existing key from database and validate it exists.
2618 Args:
2619 token: The key token to look up
2620 prisma_client: Prisma client instance
2621 key_alias: Alias to look the key up by when token is not provided
2623 Returns:
2624 LiteLLM_VerificationToken: The existing key row
2626 Raises:
2627 ProxyException: 404 if key is not found, 400 if the alias matches multiple keys
2628 """
2629 if prisma_client is None: 2629 ↛ 2630line 2629 didn't jump to line 2630 because the condition on line 2629 was never true
2630 raise HTTPException(
2631 status_code=500,
2632 detail={"error": "Database not connected"},
2633 )
2635 if token is not None: 2635 ↛ 2652line 2635 didn't jump to line 2652 because the condition on line 2635 was always true
2636 hashed_token: Final = _hash_token_if_needed(token=token)
2638 existing_key_row: Final[LiteLLM_VerificationToken | None] = await _prisma_table(
2639 VerificationTokenRepository(prisma_client)
2640 ).find_unique(where={"token": hashed_token}, include={"object_permission": True})
2642 if existing_key_row is None: 2642 ↛ 2650line 2642 didn't jump to line 2650 because the condition on line 2642 was always true
2643 raise ProxyException(
2644 message="Key not found.",
2645 type=ProxyErrorTypes.not_found_error,
2646 param="key",
2647 code=status.HTTP_404_NOT_FOUND,
2648 )
2650 return existing_key_row
2652 if key_alias is None:
2653 raise ProxyException(
2654 message="either key or key_alias must be provided",
2655 type=ProxyErrorTypes.bad_request_error,
2656 param="key",
2657 code=status.HTTP_400_BAD_REQUEST,
2658 )
2660 rows: Sequence[LiteLLM_VerificationToken] = await _prisma_table(
2661 VerificationTokenRepository(prisma_client)
2662 ).find_many(where={"key_alias": key_alias}, take=2)
2664 if len(rows) == 0:
2665 raise ProxyException(
2666 message=f"Key not found. No key with key_alias='{key_alias}'.",
2667 type=ProxyErrorTypes.not_found_error,
2668 param="key_alias",
2669 code=status.HTTP_404_NOT_FOUND,
2670 )
2672 if len(rows) > 1:
2673 raise ProxyException(
2674 message=f"Multiple keys share key_alias='{key_alias}', so it cannot be used as an identifier.",
2675 type=ProxyErrorTypes.bad_request_error,
2676 param="key_alias",
2677 code=status.HTTP_400_BAD_REQUEST,
2678 )
2680 return rows[0]
2683def _resolve_token_to_update(data: UpdateKeyRequest, existing_key_row: LiteLLM_VerificationToken) -> str:
2684 if data.key is not None:
2685 return data.key
2686 if existing_key_row.token is None:
2687 raise ProxyException(
2688 message="Key not found.",
2689 type=ProxyErrorTypes.not_found_error,
2690 param="key",
2691 code=status.HTTP_404_NOT_FOUND,
2692 )
2693 return existing_key_row.token
2696async def _process_single_key_update(
2697 update_key_request: UpdateKeyRequest,
2698 user_api_key_dict: UserAPIKeyAuth,
2699 litellm_changed_by: str | None,
2700 prisma_client: PrismaClient | None,
2701 user_api_key_cache: UserApiKeyCache,
2702 proxy_logging_obj: ProxyLogging,
2703 llm_router: Router | None,
2704 user_custom_key_update: Callable | None = None,
2705 existing_key_row: LiteLLM_VerificationToken | None = None,
2706 user_custom_key_policy: Callable[..., Awaitable[Mapping[str, object]]] | None = None,
2707) -> dict[str, object]:
2708 """
2709 Process a single key update with all validations and checks.
2711 This function encapsulates all the logic for updating a single key,
2712 including validation, permission checks, team checks, and database updates.
2714 Args:
2715 update_key_request: Fully-constructed UpdateKeyRequest for the target key
2716 user_api_key_dict: The authenticated user's API key info
2717 litellm_changed_by: Optional header for tracking who made the change
2718 prisma_client: Prisma client instance
2719 user_api_key_cache: User API key cache
2720 proxy_logging_obj: Proxy logging object
2721 llm_router: LLM router instance
2722 existing_key_row: Optional pre-fetched key row to avoid redundant lookups
2724 Returns:
2725 Dict containing the updated key information
2727 Raises:
2728 HTTPException: For various validation and permission errors
2729 """
2730 # Validate max_budget
2731 _validate_max_budget(update_key_request.max_budget)
2733 _check_permissions_caller_permission(
2734 data=update_key_request,
2735 user_api_key_dict=user_api_key_dict,
2736 )
2738 # Get and validate existing key
2739 if existing_key_row is None: 2739 ↛ 2745line 2739 didn't jump to line 2745 because the condition on line 2739 was always true
2740 existing_key_row = await _get_and_validate_existing_key(
2741 token=update_key_request.key,
2742 prisma_client=prisma_client,
2743 )
2745 _check_disable_global_guardrails_caller_permission(
2746 update_key_request.disable_global_guardrails,
2747 update_key_request.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict
2748 user_api_key_dict,
2749 existing_metadata=existing_key_row.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict
2750 )
2752 enforce_batch_enqueued_token_limit_is_admin_only(
2753 data=update_key_request,
2754 existing_metadata=existing_key_row.metadata,
2755 user_api_key_dict=user_api_key_dict,
2756 entity="key",
2757 )
2759 # Check team member permissions
2760 if prisma_client is not None:
2761 await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
2762 user_api_key_dict=user_api_key_dict,
2763 route=KeyManagementRoutes.KEY_UPDATE,
2764 prisma_client=prisma_client,
2765 existing_key_row=existing_key_row,
2766 user_api_key_cache=user_api_key_cache,
2767 )
2769 # Custom key update hook
2770 if user_custom_key_update is not None:
2771 if inspect.iscoroutinefunction(user_custom_key_update):
2772 result: Final = await user_custom_key_update(update_key_request)
2773 else:
2774 raise ValueError("user_custom_key_update must be a coroutine")
2775 decision: Final = result.get("decision", True)
2776 message: Final = result.get("message", "Authentication Failed - Custom Auth Rule")
2777 if not decision:
2778 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=message)
2780 # Enforce upperbound key params on update (don't fill defaults)
2781 _enforce_upperbound_key_params(update_key_request, fill_defaults=False)
2783 # Get team object and check team limits if team_id is provided
2784 team_obj: LiteLLM_TeamTableCachedObj | None = None
2785 if update_key_request.team_id is not None:
2786 team_obj = await get_team_object(
2787 team_id=update_key_request.team_id,
2788 prisma_client=prisma_client,
2789 user_api_key_cache=user_api_key_cache,
2790 check_db_only=True,
2791 )
2793 if team_obj is not None and prisma_client is not None:
2794 await _check_team_key_limits(
2795 team_table=team_obj,
2796 data=update_key_request,
2797 prisma_client=prisma_client,
2798 )
2800 # Validate team change if team is being changed
2801 if is_different_team(data=update_key_request, existing_key_row=existing_key_row):
2802 if llm_router is None:
2803 raise HTTPException(
2804 status_code=400,
2805 detail={
2806 "error": "LLM router not found. Please set it up by passing in a valid config.yaml or adding models via the UI."
2807 },
2808 )
2809 if team_obj is None:
2810 raise HTTPException(
2811 status_code=500,
2812 detail={"error": "Team object not found for team change validation"},
2813 )
2814 await validate_key_team_change(
2815 key=existing_key_row,
2816 team=team_obj,
2817 change_initiated_by=user_api_key_dict,
2818 llm_router=llm_router,
2819 )
2821 key_request: Final = await _with_validated_object_permission(
2822 update_key_request=update_key_request,
2823 team_obj=team_obj,
2824 existing_key_row=existing_key_row,
2825 prisma_client=prisma_client,
2826 user_api_key_cache=user_api_key_cache,
2827 user_api_key_dict=user_api_key_dict,
2828 )
2830 # Prepare update data
2831 non_default_values = await prepare_key_update_data(
2832 data=key_request, existing_key_row=existing_key_row, prisma_client=prisma_client, llm_router=llm_router
2833 )
2835 await _enforce_custom_key_policy(
2836 hook=user_custom_key_policy,
2837 build_policy_request=lambda: _update_policy_request(
2838 operation="update",
2839 existing_key_row=existing_key_row,
2840 non_default_values=non_default_values,
2841 request=key_request,
2842 ),
2843 )
2845 # Update key in database
2846 if prisma_client is None:
2847 raise HTTPException(
2848 status_code=500,
2849 detail={"error": "Database not connected"},
2850 )
2852 update_values: Final = await _handle_update_object_permission(
2853 data_json=non_default_values,
2854 existing_key_row=existing_key_row,
2855 prisma_client=prisma_client,
2856 )
2857 _data: Final = {**update_values, "token": key_request.key}
2858 response: Final[Mapping[str, object] | None] = cast( # cast-ok: every update_data branch returns a str-keyed dict
2859 "Mapping[str, object] | None",
2860 await prisma_client.update_data(token=key_request.key, data=_data),
2861 )
2863 # Permission row first: a key-object miss between the two evictions would re-cache stale grants
2864 await invalidate_cached_object_permissions(
2865 object_permission_ids=(
2866 existing_key_row.object_permission_id,
2867 non_default_values.get("object_permission_id"),
2868 ),
2869 user_api_key_cache=user_api_key_cache,
2870 )
2871 await _delete_cache_key_object(
2872 hashed_token=_hash_token_if_needed(key_request.key),
2873 user_api_key_cache=user_api_key_cache,
2874 proxy_logging_obj=proxy_logging_obj,
2875 )
2877 # After the key's own cache entry is dropped, so a failure here cannot leave the key
2878 # authenticating against the access groups it just lost.
2879 await sync_key_update_access_group_membership(
2880 prisma_client=prisma_client,
2881 key_token=_hash_token_if_needed(_resolve_token_to_update(data=key_request, existing_key_row=existing_key_row)),
2882 data=key_request,
2883 existing_key_row=existing_key_row,
2884 )
2886 # Trigger async hook
2887 asyncio.create_task(
2888 KeyManagementEventHooks.async_key_updated_hook(
2889 data=key_request,
2890 existing_key_row=existing_key_row,
2891 response=response,
2892 user_api_key_dict=user_api_key_dict,
2893 litellm_changed_by=litellm_changed_by,
2894 )
2895 )
2897 if response is None:
2898 raise ValueError("Failed to update key got response = None")
2900 # Extract and format updated key info
2901 updated_key_info = response.get("data", {})
2902 if hasattr(updated_key_info, "model_dump"):
2903 updated_key_info = updated_key_info.model_dump()
2904 elif hasattr(updated_key_info, "dict"):
2905 updated_key_info = updated_key_info.dict()
2907 updated_key_info.pop("token", None)
2909 return updated_key_info
2912async def _with_validated_object_permission(
2913 update_key_request: UpdateKeyRequest,
2914 team_obj: LiteLLM_TeamTableCachedObj | None,
2915 existing_key_row: LiteLLM_VerificationToken,
2916 prisma_client: PrismaClient | None,
2917 user_api_key_cache: UserApiKeyCache,
2918 user_api_key_dict: UserAPIKeyAuth,
2919) -> UpdateKeyRequest:
2920 if update_key_request.object_permission is None:
2921 return update_key_request
2922 normalized_object_permission: Final = await _validate_mcp_servers_for_key_update(
2923 data=update_key_request,
2924 team_obj=team_obj,
2925 existing_key_row=existing_key_row,
2926 prisma_client=prisma_client,
2927 user_api_key_cache=user_api_key_cache,
2928 is_proxy_admin=user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value,
2929 )
2930 if normalized_object_permission is None:
2931 return update_key_request
2932 return update_key_request.model_copy(
2933 update=MappingProxyType({"object_permission": LiteLLM_ObjectPermissionBase(**normalized_object_permission)})
2934 )
2937async def _validate_mcp_servers_for_key_update(
2938 data: "UpdateKeyRequest",
2939 team_obj: Optional["LiteLLM_TeamTableCachedObj"],
2940 existing_key_row: LiteLLM_VerificationToken,
2941 prisma_client: PrismaClient | None,
2942 user_api_key_cache: UserApiKeyCache,
2943 is_proxy_admin: bool,
2944) -> ObjectPermissionDict | None:
2945 """Validate MCP servers in object_permission against the effective team."""
2946 effective_team_obj = team_obj
2947 # If team_id isn't being changed, resolve the existing key's team
2948 if effective_team_obj is None and existing_key_row.team_id:
2949 effective_team_obj = await get_team_object(
2950 team_id=existing_key_row.team_id,
2951 prisma_client=prisma_client,
2952 user_api_key_cache=user_api_key_cache,
2953 check_db_only=True,
2954 )
2955 object_permission_dict: Final = _object_permission_to_dict(data.object_permission)
2956 team_unchanged: Final = data.team_id is None or data.team_id == existing_key_row.team_id
2957 normalized_object_permission: Final = await validate_key_mcp_servers_against_team(
2958 object_permission=object_permission_dict,
2959 team_obj=effective_team_obj,
2960 prisma_client=prisma_client,
2961 is_proxy_admin=is_proxy_admin,
2962 existing_key_object_permission=existing_key_row.object_permission if team_unchanged else None,
2963 )
2964 await validate_key_search_tools_against_team(
2965 object_permission=object_permission_dict,
2966 team_obj=effective_team_obj,
2967 is_proxy_admin=is_proxy_admin,
2968 )
2969 await validate_key_vector_stores_against_team(
2970 object_permission=object_permission_dict,
2971 team_obj=effective_team_obj,
2972 is_proxy_admin=is_proxy_admin,
2973 )
2974 return normalized_object_permission
2977def _require_prisma_client(prisma_client: PrismaClient | None) -> PrismaClient:
2978 if prisma_client is None:
2979 raise HTTPException(status_code=500, detail={"error": "Database not connected"})
2980 return prisma_client
2983def _requested_end_user_budget_id(data: KeyRequestBase) -> str | None:
2984 """A ``metadata`` body replaces the stored metadata wholesale, so one without the field clears it."""
2985 if data.end_user_budget_id is not None: 2985 ↛ 2986line 2985 didn't jump to line 2986 because the condition on line 2985 was never true
2986 return data.end_user_budget_id
2987 if data.metadata is None:
2988 return None
2989 return get_key_end_user_budget_id(data.metadata) or ""
2992async def _validate_end_user_budget_id_change(
2993 requested_budget_id: str | None,
2994 existing_budget_id: str | None,
2995 user_api_key_dict: UserAPIKeyAuth,
2996 prisma_client: PrismaClient | None,
2997) -> None:
2998 """A key's default end-user budget overrides the proxy-wide one, so only proxy admins
2999 may change it, and a non-empty value must name an existing budget (empty clears it)."""
3000 if requested_budget_id is None or requested_budget_id == (existing_budget_id or ""): 3000 ↛ 3002line 3000 didn't jump to line 3002 because the condition on line 3000 was always true
3001 return
3002 if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value:
3003 forbidden_detail: Final = { # mutable-ok: FastAPI detail contract
3004 "error": "Only proxy admins can set end_user_budget_id on a key."
3005 }
3006 raise HTTPException(status_code=403, detail=forbidden_detail)
3007 if requested_budget_id == "":
3008 return
3009 budget_row: Final = await BudgetRepository(_require_prisma_client(prisma_client)).find_by_id(requested_budget_id)
3010 if budget_row is None:
3011 missing_detail: Final = { # mutable-ok: FastAPI detail contract
3012 "error": f"end_user_budget_id={requested_budget_id} does not match any budget."
3013 }
3014 raise HTTPException(status_code=400, detail=missing_detail)
3017_GENERAL_SETTINGS: Final = TypeAdapter(dict[str, object])
3020def _general_settings() -> Mapping[str, object]:
3021 from litellm.proxy.proxy_server import (
3022 general_settings, # pyright: ignore[reportUnknownVariableType] # untyped module-level dict in proxy_server
3023 )
3025 return _GENERAL_SETTINGS.validate_python(general_settings)
3028async def _acting_as_team_admin_for_key_update(
3029 data: UpdateKeyRequest,
3030 existing_key_row: LiteLLM_VerificationToken,
3031 user_api_key_dict: UserAPIKeyAuth,
3032 checked_prisma_client: PrismaClient,
3033 user_api_key_cache: UserApiKeyCache,
3034 is_proxy_admin: bool,
3035) -> bool:
3036 """Whether the caller acts as a team admin on another member's team key.
3038 Raises 403 when the caller administers the key's team but the request edits fields
3039 outside the member_key_budgets permission (or that permission is disabled).
3040 """
3041 if (
3042 is_proxy_admin
3043 or existing_key_row.team_id is None
3044 or existing_key_row.user_id is None
3045 or existing_key_row.user_id == user_api_key_dict.user_id
3046 ):
3047 return False
3048 team_for_grant: Final = await get_team_object(
3049 team_id=existing_key_row.team_id,
3050 prisma_client=checked_prisma_client,
3051 user_api_key_cache=user_api_key_cache,
3052 check_db_only=True,
3053 )
3054 if not _is_user_team_admin(user_api_key_dict=user_api_key_dict, team_obj=team_for_grant):
3055 return False
3056 team_admin_key_request_or_raise(
3057 team_admin_key_edit_verdict(
3058 data=data,
3059 existing=existing_key_row,
3060 enabled=team_admin_may_edit_member_key_budgets(_general_settings()),
3061 )
3062 )
3063 return True
3066async def _validate_update_key_data(
3067 data: UpdateKeyRequest,
3068 existing_key_row: LiteLLM_VerificationToken,
3069 user_api_key_dict: UserAPIKeyAuth,
3070 llm_router: Router | None,
3071 premium_user: bool,
3072 prisma_client: PrismaClient | None,
3073 user_api_key_cache: UserApiKeyCache,
3074) -> None:
3075 """Validate permissions and constraints for key update."""
3076 checked_prisma_client: Final = _require_prisma_client(prisma_client)
3078 # Reject NaN/±inf spend before it can reach the DB / spend counter.
3079 validate_finite_spend(data.spend)
3080 validate_budget_duration(data.budget_duration)
3082 _is_proxy_admin: Final = user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
3084 _enforce_allowed_routes_update_permission(
3085 data=data,
3086 existing_key_row=existing_key_row,
3087 user_api_key_dict=user_api_key_dict,
3088 )
3089 _check_passthrough_routes_caller_permission(
3090 data=data,
3091 user_api_key_dict=user_api_key_dict,
3092 )
3093 _check_permissions_caller_permission(
3094 data=data,
3095 user_api_key_dict=user_api_key_dict,
3096 )
3097 _check_disable_global_guardrails_caller_permission(
3098 data.disable_global_guardrails,
3099 data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict
3100 user_api_key_dict,
3101 existing_metadata=existing_key_row.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict
3102 )
3104 _validate_caller_can_change_key_ownership(
3105 data=data,
3106 existing_key_row=existing_key_row,
3107 user_api_key_dict=user_api_key_dict,
3108 )
3110 if data.project_id is not None and data.project_id != existing_key_row.project_id:
3111 raise HTTPException(
3112 status_code=400, detail="Project reassignment is not supported. Use null to detach the key."
3113 )
3114 is_project_change: Final = "project_id" in data.model_fields_set and data.project_id != existing_key_row.project_id
3116 acting_as_team_admin: Final = await _acting_as_team_admin_for_key_update(
3117 data=data,
3118 existing_key_row=existing_key_row,
3119 user_api_key_dict=user_api_key_dict,
3120 checked_prisma_client=checked_prisma_client,
3121 user_api_key_cache=user_api_key_cache,
3122 is_proxy_admin=_is_proxy_admin,
3123 )
3125 common_key_access_checks(
3126 user_api_key_dict=user_api_key_dict,
3127 data=data,
3128 user_id=user_api_key_dict.user_id if acting_as_team_admin else existing_key_row.user_id,
3129 llm_router=llm_router,
3130 premium_user=premium_user,
3131 )
3133 await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
3134 user_api_key_dict=user_api_key_dict,
3135 route=KeyManagementRoutes.KEY_UPDATE,
3136 prisma_client=checked_prisma_client,
3137 existing_key_row=existing_key_row,
3138 user_api_key_cache=user_api_key_cache,
3139 )
3141 # Cross-key authorization. Previously only gated on max_budget/spend
3142 # changes, which let a non-admin blanket-rewrite any OTHER field on
3143 # any key (models, alias, metadata, tpm_limit, rpm_limit,
3144 # allowed_routes, guardrails, blocked, duration, permissions, …) as
3145 # long as they avoided budget/spend.
3146 #
3147 # Policy:
3148 # - Key owner (same user_id): may update non-budget fields on their
3149 # own key without the admin check.
3150 # - Team member with /key/update grant (on a team key): may update
3151 # non-budget fields. Team membership + permission is already
3152 # enforced by can_team_member_execute_key_management_endpoint
3153 # above, which raises 401 for non-members or members without the
3154 # grant — so reaching this point on a team key means the caller
3155 # was authorized via member_permissions. This preserves the
3156 # documented member_permissions feature while still blocking the
3157 # cross-org attack (an outside org admin is not a member of the
3158 # victim team and gets rejected at the earlier check).
3159 # - Anyone else (non-PROXY_ADMIN, not the owner, not a team member
3160 # on a team key): must pass _check_key_admin_access (PROXY_ADMIN
3161 # / key-owner / team-admin / org-admin of the key).
3162 # - max_budget / spend / budget_limits: always require the admin
3163 # check, even for the key owner or a team member (matches the
3164 # existing admin-only budget semantics). budget_limits uses
3165 # model_fields_set because an explicit null/[] clears the field
3166 # and must gate the same as setting or changing it.
3167 # - spend gates on presence alone (not a value diff): the DB spend
3168 # lags the live cross-pod counter, so letting an "unchanged" spend
3169 # through the non-admin path would let a key owner / team member
3170 # overwrite the live counter below real usage and silently weaken
3171 # enforcement.
3172 _is_budget_change: Final = (
3173 (data.max_budget is not None and data.max_budget != existing_key_row.max_budget)
3174 or data.spend is not None
3175 or "budget_limits" in data.model_fields_set
3176 or "soft_budget" in data.model_fields_set
3177 )
3179 _existing_metadata: Final = getattr(existing_key_row, "metadata", None)
3180 _existing_throttle: Final = (
3181 _existing_metadata.get("throttle_on_budget_exceeded") if isinstance(_existing_metadata, dict) else None
3182 )
3183 if data.throttle_on_budget_exceeded is True and _existing_throttle is not True and not _is_proxy_admin:
3184 raise HTTPException(
3185 status_code=403,
3186 detail={"error": "Only proxy admins can enable throttle_on_budget_exceeded on a key."},
3187 )
3189 await _validate_end_user_budget_id_change(
3190 requested_budget_id=_requested_end_user_budget_id(data),
3191 existing_budget_id=get_key_end_user_budget_id(
3192 _existing_metadata if isinstance(_existing_metadata, dict) else None
3193 ),
3194 user_api_key_dict=user_api_key_dict,
3195 prisma_client=checked_prisma_client,
3196 )
3198 enforce_output_token_estimates_are_admin_only(
3199 data=data,
3200 existing_metadata=_existing_metadata if isinstance(_existing_metadata, dict) else None,
3201 user_api_key_dict=user_api_key_dict,
3202 entity="key",
3203 )
3204 enforce_batch_enqueued_token_limit_is_admin_only(
3205 data=data,
3206 existing_metadata=_existing_metadata if isinstance(_existing_metadata, dict) else None,
3207 user_api_key_dict=user_api_key_dict,
3208 entity="key",
3209 )
3211 # Personal-key bypass: the caller both created the key AND still owns it
3212 # (user_id == caller). Checking only created_by would let a demoted admin
3213 # who originally created a key for another user continue editing it without
3214 # admin authorization after the key was reassigned.
3215 caller_is_creator: Final = (
3216 user_api_key_dict.user_id is not None
3217 and getattr(existing_key_row, "created_by", None) == user_api_key_dict.user_id
3218 and getattr(existing_key_row, "user_id", None) == user_api_key_dict.user_id
3219 )
3220 # Team keys: can_team_member_execute_key_management_endpoint (called above)
3221 # already validated team membership + /key/update permission and would have
3222 # raised if the caller lacked it. Reaching this point on a team key for a
3223 # non-budget change means the caller was authorized — skip the redundant
3224 # _check_key_admin_access that would otherwise require team/org admin status.
3225 _key_is_team_key: Final = getattr(existing_key_row, "team_id", None) is not None
3226 can_skip_admin_check: Final = (caller_is_creator or _key_is_team_key) and not (
3227 _is_budget_change or is_project_change
3228 )
3229 if (not _is_proxy_admin) and not can_skip_admin_check:
3230 hashed_key: Final = existing_key_row.token
3231 await _check_key_admin_access(
3232 user_api_key_dict=user_api_key_dict,
3233 hashed_token=hashed_key,
3234 prisma_client=checked_prisma_client,
3235 user_api_key_cache=user_api_key_cache,
3236 route=("/key/update (max_budget/spend)" if _is_budget_change else "/key/update"),
3237 )
3239 # Check team limits if key has a team_id (from request or existing key)
3240 team_obj: LiteLLM_TeamTableCachedObj | None = None
3241 _team_id_to_check: Final = data.team_id or getattr(existing_key_row, "team_id", None)
3242 if _team_id_to_check is not None:
3243 team_obj = await get_team_object(
3244 team_id=_team_id_to_check,
3245 prisma_client=checked_prisma_client,
3246 user_api_key_cache=user_api_key_cache,
3247 check_db_only=True,
3248 )
3250 # Validate team exists when non-admin sets a new team_id (LIT-1884)
3251 if team_obj is None and data.team_id is not None and not _is_proxy_admin:
3252 raise HTTPException(
3253 status_code=400,
3254 detail=f"Team not found for team_id={data.team_id}. Non-admin users cannot set keys to non-existent teams.",
3255 )
3257 if team_obj is not None:
3258 await _check_team_key_limits(
3259 team_table=team_obj,
3260 data=data,
3261 prisma_client=checked_prisma_client,
3262 )
3264 TeamMemberPermissionChecks.enforce_member_can_assign_access_groups(
3265 user_api_key_dict=user_api_key_dict,
3266 team_table=team_obj,
3267 access_group_ids=data.access_group_ids,
3268 )
3270 # Validate key against project limits if project_id is being set
3271 _project_id_to_check: Final = (
3272 data.project_id if "project_id" in data.model_fields_set else existing_key_row.project_id
3273 )
3274 if _project_id_to_check is not None and (data.models is not None or data.max_budget is not None):
3275 await _check_project_key_limits(
3276 project_id=_project_id_to_check,
3277 data=data,
3278 prisma_client=checked_prisma_client,
3279 user_api_key_cache=user_api_key_cache,
3280 )
3282 # When the caller asks to change the key's organization_id, require that
3283 # they are a member of (or a proxy admin over) the target organization.
3284 # Without this gate, any caller could assign their key to an arbitrary
3285 # organization_id by passing it in the request body — VERIA-55 secondary
3286 # IDOR. The check mirrors the membership rule already used on the
3287 # `/key/list` filter path in `validate_key_list_check`.
3288 _existing_org_id: Final = getattr(existing_key_row, "organization_id", None)
3289 if data.organization_id is not None and data.organization_id != _existing_org_id and not _is_proxy_admin:
3290 await _validate_caller_can_assign_key_org(
3291 user_api_key_dict=user_api_key_dict,
3292 organization_id=data.organization_id,
3293 prisma_client=checked_prisma_client,
3294 )
3296 # Check org key limits only when throughput-related fields or organization_id change
3297 _org_id_to_check: Final = data.organization_id or _existing_org_id
3298 _throughput_fields_changed: Final = (
3299 data.organization_id is not None
3300 or data.tpm_limit is not None
3301 or data.rpm_limit is not None
3302 or data.tpm_limit_type is not None
3303 or data.rpm_limit_type is not None
3304 )
3305 if _org_id_to_check is not None and _throughput_fields_changed:
3306 org_table: Final = await get_org_object(
3307 org_id=_org_id_to_check,
3308 user_api_key_cache=user_api_key_cache,
3309 prisma_client=checked_prisma_client,
3310 )
3311 if org_table is None:
3312 raise HTTPException(
3313 status_code=400,
3314 detail=f"Organization not found for organization_id={_org_id_to_check}",
3315 )
3316 await _check_org_key_limits(
3317 org_table=org_table,
3318 data=data,
3319 prisma_client=checked_prisma_client,
3320 )
3322 # if team change - check if this is possible
3323 if is_different_team(data=data, existing_key_row=existing_key_row):
3324 if llm_router is None:
3325 raise HTTPException(
3326 status_code=400,
3327 detail={
3328 "error": "LLM router not found. Please set it up by passing in a valid config.yaml or adding models via the UI."
3329 },
3330 )
3331 if team_obj is None:
3332 raise HTTPException(
3333 status_code=500,
3334 detail={"error": "Team object not found for team change validation"},
3335 )
3336 await validate_key_team_change(
3337 key=existing_key_row,
3338 team=team_obj,
3339 change_initiated_by=user_api_key_dict,
3340 llm_router=llm_router,
3341 )
3343 # Validate MCP servers in object_permission against the effective team
3344 if data.object_permission is not None:
3345 normalized_object_permission: Final = await _validate_mcp_servers_for_key_update(
3346 data=data,
3347 team_obj=team_obj,
3348 existing_key_row=existing_key_row,
3349 prisma_client=checked_prisma_client,
3350 user_api_key_cache=user_api_key_cache,
3351 is_proxy_admin=_is_proxy_admin,
3352 )
3353 if normalized_object_permission is not None:
3354 data.object_permission = LiteLLM_ObjectPermissionBase(**normalized_object_permission)
3357@router.post("/key/update", tags=["key management"], dependencies=[Depends(user_api_key_auth)])
3358@management_endpoint_wrapper
3359async def update_key_fn(
3360 request: Request,
3361 data: UpdateKeyRequest,
3362 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
3363 litellm_changed_by: str | None = Header(
3364 None,
3365 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
3366 ),
3367):
3368 """
3369 Update an existing API key's parameters.
3371 The body is a merge patch: a field left out keeps its stored value, and on the key's own columns
3372 an explicit null clears it. The metadata-backed fields below are the exception, merging into the
3373 stored metadata instead: passing one as null leaves it unchanged, while `metadata` itself
3374 replaces the stored metadata wholesale.
3376 Parameters:
3377 - key: Optional[str] - The key to update. Either key or key_alias must be provided.
3378 - key_alias: Optional[str] - User-friendly key alias. If key is omitted, also identifies the key to update (must match exactly one key, same as /key/delete's key_aliases)
3379 - user_id: Optional[str] - User ID associated with key
3380 - team_id: Optional[str] - Team ID associated with key
3381 - agent_id: Optional[str] - The agent id associated with the key.
3382 - project_id: Optional[str] - Omit to retain the project, or send null to detach. A different project ID is rejected.
3383 - organization_id: Optional[str] - The organization id of the key.
3384 - budget_id: Optional[str] - The budget id associated with the key. Created by calling `/budget/new`.
3385 - end_user_budget_id: Optional[str] - Proxy admin only. Budget id applied to end users first seen through this key that carry no budget of their own. Omit to keep the current value, pass an empty string to clear it.
3386 - models: Optional[list] - Model_name's a user is allowed to call
3387 - tags: Optional[List[str]] - Tags for organizing keys (Enterprise only)
3388 - prompts: Optional[List[str]] - List of prompts that the key is allowed to use.
3389 - enforced_params: Optional[List[str]] - List of enforced params for the key (Enterprise only). [Docs](https://docs.litellm.ai/docs/proxy/enterprise#enforce-required-params-for-llm-requests)
3390 - spend: Optional[float] - Amount spent by key
3391 - max_budget: Optional[float] - Max budget for key
3392 - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}
3393 - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
3394 - budget_duration: Optional[str] - Budget reset period ("30d", "1h", etc.)
3395 - soft_budget: Optional[float] - Soft budget limit (warning vs. hard stop). Will trigger a slack alert when this soft budget is reached. Set to null to remove the soft budget.
3396 - max_parallel_requests: Optional[int] - Rate limit for parallel requests
3397 - metadata: Optional[dict] - Metadata for key. Example {"team": "core-infra", "app": "app2"}
3398 - tpm_limit: Optional[int] - Tokens per minute limit
3399 - rpm_limit: Optional[int] - Requests per minute limit
3400 - tpd_limit: Optional[int] - Tokens per day limit, charged by batch submissions instead of tpm_limit/rpm_limit
3401 - model_rpm_limit: Optional[dict] - Model-specific RPM limits {"gpt-4": 100, "claude-v1": 200}
3402 - mcp_rpm_limit: Optional[dict] - Per-MCP-server RPM limits, keyed by MCP server name {"github": 100, "slack": 200}
3403 - tag_rpm_limit: Optional[dict] - Per-request-tag RPM limits, keyed by request tag {"cell-1": 1000, "cell-2": 500}. Each tag gets an independent counter; absent tags fall back to the key-level rpm limit.
3404 - model_tpm_limit: Optional[dict] - Model-specific TPM limits {"gpt-4": 100000, "claude-v1": 200000}
3405 - default_estimated_output_tokens: Optional[int] - Proxy admin only. Expected output tokens reserved for TPM limiting when a request omits max_tokens. Positive integer.
3406 - default_estimated_output_tokens_per_model: Optional[dict] - Proxy admin only. Per-model override of the above {"gpt-4": 4096, "gpt-3.5-turbo": 1024}
3407 - tpm_limit_type: Optional[str] - TPM rate limit type - "best_effort_throughput", "guaranteed_throughput", or "dynamic"
3408 - rpm_limit_type: Optional[str] - RPM rate limit type - "best_effort_throughput", "guaranteed_throughput", or "dynamic"
3409 - allowed_cache_controls: Optional[list] - List of allowed cache control values
3410 - duration: Optional[str] - Key validity duration ("30d", "1h", etc.), null to never expire, or "-1" to never expire (deprecated, use null)
3411 - permissions: Optional[dict] - Key-specific permissions
3412 - send_invite_email: Optional[bool] - Send invite email to user_id
3413 - guardrails: Optional[List[str]] - List of active guardrails for the key
3414 - policies: Optional[List[str]] - List of policy names to apply to the key. Policies define guardrails, conditions, and inheritance rules.
3415 - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. Proxy admin only.
3416 - throttle_on_budget_exceeded: Optional[bool] - When the key exceeds its max_budget, throttle its tpm/rpm to the global budget_exceeded_throttle_percentage instead of blocking the key entirely.
3417 - enable_prompt_caching: Optional[bool] - Auto-inject prompt caching breakpoints (Anthropic cache_control markers) on requests made with this key. Supported Claude models on Anthropic, Bedrock, Vertex AI, and Azure AI only.
3418 - prompts: Optional[List[str]] - List of prompts that the key is allowed to use.
3419 - blocked: Optional[bool] - Whether the key is blocked
3420 - aliases: Optional[dict] - Model aliases for the key - [Docs](https://litellm.vercel.app/docs/proxy/virtual_keys#model-aliases)
3421 - config: Optional[dict] - [DEPRECATED PARAM] Key-specific config.
3422 - temp_budget_increase: Optional[float] - Temporary budget increase for the key (Enterprise only).
3423 - temp_budget_expiry: Optional[str] - Expiry time for the temporary budget increase (Enterprise only).
3424 - allowed_routes: Optional[list] - List of allowed routes for the key. Store the actual route or store a wildcard pattern for a set of routes. Example - ["/chat/completions", "/embeddings", "/keys/*"]
3425 - allowed_passthrough_routes: Optional[list] - List of allowed pass through routes for the key. Store the actual route or store a wildcard pattern for a set of routes. Example - ["/my-custom-endpoint"]. Use this instead of allowed_routes, if you just want to specify which pass through routes the key can access, without specifying the routes. If allowed_routes is specified, allowed_passthrough_routes is ignored.
3426 - prompts: Optional[List[str]] - List of allowed prompts for the key. If specified, the key will only be able to use these specific prompts.
3427 - object_permission: Optional[LiteLLM_ObjectPermissionBase] - key-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"], "agents": ["agent_1", "agent_2"], "agent_access_groups": ["dev_group"]}. IF null or {} then no object permission.
3428 - auto_rotate: Optional[bool] - Whether this key should be automatically rotated
3429 - rotation_interval: Optional[str] - How often to rotate this key (e.g., '30d', '90d'). Required if auto_rotate=True
3430 - allowed_vector_store_indexes: Optional[List[dict]] - List of allowed vector store indexes for the key. Example - [{"index_name": "my-index", "index_permissions": ["write", "read"]}]. If specified, the key will only be able to use these specific vector store indexes. Create index, using `/v1/indexes` endpoint.
3431 - router_settings: Optional[UpdateRouterConfig] - key-specific router settings. Example - {"model_group_retry_policy": {"gpt-4": {"RateLimitErrorRetries": 5}}}. IF null or {} then no router settings.
3432 - access_group_ids: Optional[List[str]] - List of access group IDs to associate with the key. Access groups define which models a key can access. Example - ["access_group_1", "access_group_2"].
3433 - budget_limits: Optional[list] - List of concurrent budget windows for the key. Each window specifies a budget_limit, time_period, and optional budget_duration. Example - [{"budget_limit": 10.0, "time_period": "1d"}, {"budget_limit": 50.0, "time_period": "7d"}].
3435 Example:
3436 ```bash
3437 curl --location 'http://0.0.0.0:4000/key/update' \
3438 --header 'Authorization: Bearer sk-1234' \
3439 --header 'Content-Type: application/json' \
3440 --data '{
3441 "key": "sk-1234",
3442 "key_alias": "my-key",
3443 "user_id": "user-1234",
3444 "team_id": "team-1234",
3445 "max_budget": 100,
3446 "metadata": {"any_key": "any-val"},
3447 }'
3448 ```
3449 """
3450 from litellm.proxy import proxy_server
3451 from litellm.proxy.proxy_server import (
3452 litellm_proxy_admin_name,
3453 llm_router,
3454 premium_user,
3455 prisma_client,
3456 proxy_logging_obj,
3457 user_api_key_cache,
3458 )
3460 try:
3461 # Validate budget values are not negative and are finite numbers
3462 if data.max_budget is not None and (not math.isfinite(data.max_budget) or data.max_budget < 0): 3462 ↛ 3463line 3462 didn't jump to line 3463 because the condition on line 3462 was never true
3463 raise HTTPException(
3464 status_code=400,
3465 detail={"error": f"max_budget must be a non-negative finite number. Received: {data.max_budget}"},
3466 )
3468 _validate_soft_budget_value(data.soft_budget)
3470 # get the row from db
3471 existing_key_row: Final = await _get_and_validate_existing_key(
3472 token=data.key,
3473 prisma_client=prisma_client,
3474 key_alias=data.key_alias,
3475 )
3476 key: Final = _resolve_token_to_update(data=data, existing_key_row=existing_key_row)
3477 data.key = key
3479 await _validate_update_key_data(
3480 data=data,
3481 existing_key_row=existing_key_row,
3482 user_api_key_dict=user_api_key_dict,
3483 llm_router=llm_router,
3484 premium_user=premium_user,
3485 prisma_client=prisma_client,
3486 user_api_key_cache=user_api_key_cache,
3487 )
3489 await _enforce_custom_key_update_policy(hook=_custom_key_update_hook(proxy_server), data=data)
3491 # Enforce upperbound key params on update (don't fill defaults)
3492 _enforce_upperbound_key_params(data, fill_defaults=False)
3493 non_default_values: Final = await prepare_key_update_data(
3494 data=data, existing_key_row=existing_key_row, prisma_client=prisma_client, llm_router=llm_router
3495 )
3497 # Only validate key_alias format if it's actually being changed
3498 new_key_alias: Final = non_default_values.get("key_alias", None)
3499 if new_key_alias != existing_key_row.key_alias:
3500 _validate_key_alias_format(key_alias=new_key_alias)
3502 await _enforce_unique_key_alias(
3503 key_alias=non_default_values.get("key_alias", None),
3504 prisma_client=prisma_client,
3505 existing_key_token=existing_key_row.token,
3506 )
3508 # Handle rotation fields if auto_rotate is being enabled
3509 _set_key_rotation_fields(
3510 non_default_values,
3511 non_default_values.get("auto_rotate", False),
3512 non_default_values.get("rotation_interval"),
3513 existing_key_alias=existing_key_row.key_alias,
3514 )
3516 await _enforce_custom_key_policy(
3517 hook=_custom_key_policy_hook(proxy_server),
3518 build_policy_request=lambda: _update_policy_request(
3519 operation="update",
3520 existing_key_row=existing_key_row,
3521 non_default_values=non_default_values,
3522 request=data,
3523 ),
3524 )
3526 if prisma_client is None:
3527 raise Exception("Not connected to DB!")
3529 update_values: Final = await _handle_update_object_permission(
3530 data_json=non_default_values,
3531 existing_key_row=existing_key_row,
3532 prisma_client=prisma_client,
3533 )
3534 changed_by: Final = user_api_key_dict.user_id or litellm_proxy_admin_name
3535 response: Final = (
3536 await _update_key_row_with_soft_budget(
3537 prisma_client=prisma_client,
3538 key=key,
3539 data=data,
3540 non_default_values=update_values,
3541 existing_key_row=existing_key_row,
3542 changed_by=changed_by,
3543 )
3544 if "soft_budget" in data.model_fields_set
3545 else await prisma_client.update_data(token=key, data=MappingProxyType({**update_values, "token": key}))
3546 )
3548 # Delete - key from cache, since it's been updated!
3549 # key updated - a new model could have been added to this key. it should not block requests after this is done
3550 await invalidate_cached_object_permissions(
3551 object_permission_ids=(
3552 existing_key_row.object_permission_id,
3553 non_default_values.get("object_permission_id"),
3554 ),
3555 user_api_key_cache=user_api_key_cache,
3556 )
3557 await _delete_cache_key_object(
3558 hashed_token=_hash_token_if_needed(key),
3559 user_api_key_cache=user_api_key_cache,
3560 proxy_logging_obj=proxy_logging_obj,
3561 )
3563 # After the key's own cache entry is dropped, so a failure here cannot leave the key
3564 # authenticating against the access groups it just lost.
3565 await sync_key_update_access_group_membership(
3566 prisma_client=prisma_client,
3567 key_token=_hash_token_if_needed(key),
3568 data=data,
3569 existing_key_row=existing_key_row,
3570 )
3572 if data.spend is not None:
3573 from litellm.proxy.proxy_server import spend_counter_cache
3575 counter_key: Final = f"spend:key:{_hash_token_if_needed(key)}"
3576 spend_counter_cache.in_memory_cache.set_cache(key=counter_key, value=data.spend, ttl=60)
3577 if spend_counter_cache.redis_cache is not None:
3578 try:
3579 await spend_counter_cache.redis_cache.async_set_cache(key=counter_key, value=data.spend, ttl=60)
3580 except Exception as redis_err:
3581 verbose_proxy_logger.warning(
3582 "Failed to update spend counter %s in Redis after key spend update: %s. "
3583 "Budget checks may use stale value until counter expires.",
3584 counter_key,
3585 redis_err,
3586 )
3588 asyncio.create_task(
3589 KeyManagementEventHooks.async_key_updated_hook(
3590 data=data,
3591 existing_key_row=existing_key_row,
3592 response=response,
3593 user_api_key_dict=user_api_key_dict,
3594 litellm_changed_by=litellm_changed_by,
3595 )
3596 )
3598 if response is None:
3599 raise ValueError("Failed to update key got response = None")
3601 return {"key": key, **response["data"]}
3602 # update based on remaining passed in values
3603 except Exception as e:
3604 verbose_proxy_logger.exception("litellm.proxy.proxy_server.update_key_fn(): Exception occured - %s", e)
3605 if isinstance(e, HTTPException): 3605 ↛ 3606line 3605 didn't jump to line 3606 because the condition on line 3605 was never true
3606 raise ProxyException(
3607 message=getattr(e, "detail", f"Authentication Error({e})"),
3608 type=ProxyErrorTypes.auth_error,
3609 param=getattr(e, "param", "None"),
3610 code=getattr(e, "status_code", status.HTTP_400_BAD_REQUEST),
3611 )
3612 elif isinstance(e, ProxyException): 3612 ↛ 3614line 3612 didn't jump to line 3614 because the condition on line 3612 was always true
3613 raise e
3614 raise ProxyException(
3615 message="Authentication Error, " + str(e),
3616 type=ProxyErrorTypes.auth_error,
3617 param=getattr(e, "param", "None"),
3618 code=status.HTTP_400_BAD_REQUEST,
3619 )
3622@router.post(
3623 "/key/bulk_update",
3624 tags=["key management"],
3625 dependencies=[Depends(user_api_key_auth)],
3626 response_model=BulkUpdateKeyResponse,
3627)
3628@management_endpoint_wrapper
3629async def bulk_update_keys(
3630 data: BulkUpdateKeyRequest,
3631 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
3632 litellm_changed_by: str | None = Header(
3633 None,
3634 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
3635 ),
3636):
3637 """
3638 Bulk update multiple keys at once.
3640 This endpoint allows updating multiple keys in a single request. Each key update
3641 is processed independently - if some updates fail, others will still succeed.
3643 Parameters:
3644 - keys: List[BulkUpdateKeyRequestItem] - List of key update requests, each containing:
3645 - key: str - The key identifier (token) to update
3646 - budget_id: Optional[str] - Budget ID associated with the key
3647 - max_budget: Optional[float] - Max budget for key
3648 - team_id: Optional[str] - Team ID associated with key
3649 - tags: Optional[List[str]] - Tags for organizing keys
3650 - object_permission: Optional[LiteLLM_ObjectPermissionBase] - key-specific object permission, as on /key/update
3652 Only the fields an item carries are written: a field left out keeps its current value, and a field
3653 sent explicitly, null included, is applied exactly as /key/update applies it.
3655 Returns:
3656 - total_requested: int - Total number of keys requested for update
3657 - successful_updates: List[SuccessfulKeyUpdate] - List of successfully updated keys with their updated info
3658 - failed_updates: List[FailedKeyUpdate] - List of failed updates with key_info and failed_reason
3660 Example request:
3661 ```bash
3662 curl --location 'http://0.0.0.0:4000/key/bulk_update' \
3663 --header 'Authorization: Bearer sk-1234' \
3664 --header 'Content-Type: application/json' \
3665 --data '{
3666 "keys": [
3667 {
3668 "key": "sk-1234",
3669 "max_budget": 100.0,
3670 "team_id": "team-123",
3671 "tags": ["production", "api"]
3672 },
3673 {
3674 "key": "sk-5678",
3675 "budget_id": "budget-456",
3676 "tags": ["staging"]
3677 }
3678 ]
3679 }'
3680 ```
3681 """
3682 from litellm.proxy import proxy_server
3683 from litellm.proxy.proxy_server import (
3684 llm_router,
3685 prisma_client,
3686 proxy_logging_obj,
3687 user_api_key_cache,
3688 )
3690 custom_key_update_hook: Final = _custom_key_update_hook(proxy_server)
3691 custom_key_policy_hook: Final = _custom_key_policy_hook(proxy_server)
3693 if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value: 3693 ↛ 3694line 3693 didn't jump to line 3694 because the condition on line 3693 was never true
3694 raise HTTPException(
3695 status_code=403,
3696 detail={"error": "Only proxy admins can perform bulk key updates"},
3697 )
3699 if prisma_client is None: 3699 ↛ 3700line 3699 didn't jump to line 3700 because the condition on line 3699 was never true
3700 raise HTTPException(
3701 status_code=500,
3702 detail={"error": "Database not connected"},
3703 )
3705 if not data.keys:
3706 raise HTTPException(
3707 status_code=400,
3708 detail={"error": "No keys provided for update"},
3709 )
3711 MAX_BATCH_SIZE: Final = 500
3712 if len(data.keys) > MAX_BATCH_SIZE: 3712 ↛ 3713line 3712 didn't jump to line 3713 because the condition on line 3712 was never true
3713 raise HTTPException(
3714 status_code=400,
3715 detail={"error": f"Maximum {MAX_BATCH_SIZE} keys can be updated at once. Found {len(data.keys)} keys."},
3716 )
3718 successful_updates: Final[list[SuccessfulKeyUpdate]] = []
3719 failed_updates: Final[list[FailedKeyUpdate]] = []
3721 for key_update_item in data.keys:
3722 try:
3723 updated_key_info = await _process_single_key_update(
3724 update_key_request=UpdateKeyRequest.model_validate(key_update_item.model_dump(exclude_unset=True)),
3725 user_api_key_dict=user_api_key_dict,
3726 litellm_changed_by=litellm_changed_by,
3727 prisma_client=prisma_client,
3728 user_api_key_cache=user_api_key_cache,
3729 proxy_logging_obj=proxy_logging_obj,
3730 llm_router=llm_router,
3731 user_custom_key_update=custom_key_update_hook,
3732 user_custom_key_policy=custom_key_policy_hook,
3733 )
3735 successful_updates.append(
3736 SuccessfulKeyUpdate(
3737 key=key_update_item.key,
3738 key_info=updated_key_info,
3739 )
3740 )
3742 except Exception as e:
3743 verbose_proxy_logger.exception("Failed to update key %s: %s", key_update_item.key, e)
3745 if isinstance(e, HTTPException):
3746 error_detail = e.detail
3747 if isinstance(error_detail, dict): 3747 ↛ 3750line 3747 didn't jump to line 3750 because the condition on line 3747 was always true
3748 error_message = error_detail.get("error", str(e))
3749 else:
3750 error_message = str(error_detail)
3751 elif isinstance(e, ProxyException): 3751 ↛ 3754line 3751 didn't jump to line 3754 because the condition on line 3751 was always true
3752 error_message = e.message
3753 else:
3754 error_message = str(e)
3756 key_info = None
3757 try:
3758 existing_key_row = await prisma_client.get_data(
3759 token=key_update_item.key,
3760 table_name="key",
3761 query_type="find_unique",
3762 )
3763 if existing_key_row is not None:
3764 if hasattr(existing_key_row, "model_dump"):
3765 key_info = existing_key_row.model_dump()
3766 elif hasattr(existing_key_row, "dict"):
3767 key_info = existing_key_row.dict()
3768 if key_info:
3769 key_info.pop("token", None)
3770 except Exception:
3771 pass
3773 failed_updates.append(
3774 FailedKeyUpdate(
3775 key=key_update_item.key,
3776 key_info=key_info,
3777 failed_reason=error_message,
3778 )
3779 )
3781 return BulkUpdateKeyResponse(
3782 total_requested=len(data.keys),
3783 successful_updates=successful_updates,
3784 failed_updates=failed_updates,
3785 )
3788def _build_failed_team_key_update(
3789 token: str,
3790 exception: Exception,
3791 existing_key_row: LiteLLM_VerificationToken | None,
3792) -> FailedKeyUpdate:
3793 """Normalize an exception from the per-key update loop into a FailedKeyUpdate."""
3794 if isinstance(exception, HTTPException): 3794 ↛ 3800line 3794 didn't jump to line 3800 because the condition on line 3794 was always true
3795 detail: Final = exception.detail
3796 if isinstance(detail, dict): 3796 ↛ 3799line 3796 didn't jump to line 3799 because the condition on line 3796 was always true
3797 error_message = detail.get("error", str(exception))
3798 else:
3799 error_message = str(detail)
3800 elif isinstance(exception, ProxyException):
3801 error_message = exception.message
3802 else:
3803 error_message = str(exception)
3805 key_info: dict[str, object] | None = None
3806 if existing_key_row is not None: 3806 ↛ 3807line 3806 didn't jump to line 3807 because the condition on line 3806 was never true
3807 if hasattr(existing_key_row, "model_dump"):
3808 key_info = existing_key_row.model_dump()
3809 elif hasattr(existing_key_row, "dict"):
3810 key_info = dict[str, object](_legacy_model_dict(existing_key_row))
3811 if key_info:
3812 key_info.pop("token", None)
3814 return FailedKeyUpdate(key=token, key_info=key_info, failed_reason=error_message)
3817@router.post(
3818 "/team/key/bulk_update",
3819 tags=["key management"],
3820 dependencies=[Depends(user_api_key_auth)],
3821 response_model=BulkUpdateKeyResponse,
3822)
3823@management_endpoint_wrapper
3824async def bulk_update_team_keys(
3825 data: BulkUpdateTeamKeysRequest,
3826 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
3827 litellm_changed_by: str | None = Header(
3828 None,
3829 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
3830 ),
3831):
3832 """
3833 Apply one update payload to many keys inside a single team.
3835 Pass `team_id` plus either `key_ids` or `all_keys_in_team=True`. The
3836 `update_fields` payload is broadcast to every selected key. Per-key
3837 failures are returned in `failed_updates` rather than aborting the batch.
3839 Callable by proxy admins, or by team admins with `KEY_UPDATE` permission.
3840 """
3841 from litellm.proxy import proxy_server
3842 from litellm.proxy.proxy_server import (
3843 llm_router,
3844 prisma_client,
3845 proxy_logging_obj,
3846 user_api_key_cache,
3847 )
3849 custom_key_update_hook: Final = _custom_key_update_hook(proxy_server)
3850 custom_key_policy_hook: Final = _custom_key_policy_hook(proxy_server)
3852 if prisma_client is None: 3852 ↛ 3853line 3852 didn't jump to line 3853 because the condition on line 3852 was never true
3853 raise HTTPException(
3854 status_code=500,
3855 detail={"error": "Database not connected"},
3856 )
3858 if not data.team_id:
3859 raise HTTPException(
3860 status_code=400,
3861 detail={"error": "team_id is required"},
3862 )
3864 MAX_BATCH_SIZE: Final = 500
3865 if data.key_ids is not None and len(data.key_ids) > MAX_BATCH_SIZE: 3865 ↛ 3866line 3865 didn't jump to line 3866 because the condition on line 3865 was never true
3866 raise HTTPException(
3867 status_code=400,
3868 detail={
3869 "error": f"Maximum {MAX_BATCH_SIZE} keys can be updated at once. Found {len(data.key_ids)} key_ids."
3870 },
3871 )
3873 if data.all_keys_in_team:
3874 # "all" excludes blocked/expired — bulk refresh shouldn't revive a key an admin disabled.
3875 # `blocked` is Boolean? with no default; `/key/generate` writes NULL. Prisma's `NOT`
3876 # excludes NULLs, so explicitly OR `false` with `null` to include them.
3877 now: Final = datetime.now(timezone.utc)
3878 existing_keys = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
3879 where={
3880 "team_id": data.team_id,
3881 "AND": [
3882 {"OR": [{"blocked": False}, {"blocked": None}]},
3883 {"OR": [{"expires": None}, {"expires": {"gt": now}}]},
3884 ],
3885 },
3886 order={"token": "asc"},
3887 take=MAX_BATCH_SIZE + 1,
3888 )
3889 if len(existing_keys) > MAX_BATCH_SIZE: 3889 ↛ 3890line 3889 didn't jump to line 3890 because the condition on line 3889 was never true
3890 raise HTTPException(
3891 status_code=400,
3892 detail={
3893 "error": f"Team {data.team_id} has more than {MAX_BATCH_SIZE} keys. Use `key_ids` to update in batches of {MAX_BATCH_SIZE}."
3894 },
3895 )
3896 requested_tokens = cast( # cast-ok: token is the table's primary key, so a row read back always carries one
3897 "list[str]", [row.token for row in existing_keys]
3898 )
3899 else:
3900 if data.key_ids is None or len(data.key_ids) == 0: 3900 ↛ 3901line 3900 didn't jump to line 3901 because the condition on line 3900 was never true
3901 raise HTTPException(
3902 status_code=400,
3903 detail={"error": "key_ids must be provided when all_keys_in_team is False"},
3904 )
3905 # Dedupe by hashed form — duplicates collapse to one update.
3906 requested_tokens = []
3907 hashed_key_ids: Final = []
3908 seen_hashes: Final = set()
3909 for k in data.key_ids:
3910 h = _hash_token_if_needed(k)
3911 if h in seen_hashes:
3912 continue
3913 seen_hashes.add(h)
3914 requested_tokens.append(k)
3915 hashed_key_ids.append(h)
3916 existing_keys = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
3917 where={"team_id": data.team_id, "token": {"in": hashed_key_ids}}
3918 )
3920 # Anchor membership check on data.team_id (not existing_keys[0]); empty result must still gate non-admins.
3921 if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value: 3921 ↛ 3922line 3921 didn't jump to line 3922 because the condition on line 3921 was never true
3922 auth_anchor: Final = (
3923 existing_keys[0]
3924 if existing_keys
3925 else LiteLLM_VerificationToken(
3926 token="__team_scope_auth_check__",
3927 team_id=data.team_id,
3928 models=[],
3929 )
3930 )
3931 await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
3932 user_api_key_dict=user_api_key_dict,
3933 route=KeyManagementRoutes.KEY_UPDATE,
3934 prisma_client=prisma_client,
3935 existing_key_row=auth_anchor,
3936 user_api_key_cache=user_api_key_cache,
3937 )
3939 # Block metadata.allowed_passthrough_routes for non-admins — the runtime
3940 # route checker reads it from key/team metadata to grant passthrough.
3941 _check_passthrough_routes_caller_permission(data=data.update_fields, user_api_key_dict=user_api_key_dict)
3943 if not requested_tokens:
3944 raise HTTPException(
3945 status_code=404,
3946 detail={"error": f"No keys found for team {data.team_id}"},
3947 )
3949 existing_by_token: Final = {row.token: row for row in existing_keys}
3950 update_field_dict: Final = data.update_fields.model_dump(exclude_unset=True)
3952 successful_updates: Final[list[SuccessfulKeyUpdate]] = []
3953 failed_updates: Final[list[FailedKeyUpdate]] = []
3955 for token in requested_tokens:
3956 db_token = _hash_token_if_needed(token)
3957 try:
3958 if db_token not in existing_by_token: 3958 ↛ 3965line 3958 didn't jump to line 3965 because the condition on line 3958 was always true
3959 raise HTTPException(
3960 status_code=404,
3961 detail={"error": f"Key not found in team {data.team_id}"},
3962 )
3964 # team_id from validated scope, never user payload — drives _check_team_key_limits.
3965 update_key_request = UpdateKeyRequest.model_validate(
3966 {
3967 "key": token,
3968 "team_id": data.team_id,
3969 **update_field_dict,
3970 }
3971 )
3972 updated_key_info = await _process_single_key_update(
3973 update_key_request=update_key_request,
3974 user_api_key_dict=user_api_key_dict,
3975 litellm_changed_by=litellm_changed_by,
3976 prisma_client=prisma_client,
3977 user_api_key_cache=user_api_key_cache,
3978 proxy_logging_obj=proxy_logging_obj,
3979 llm_router=llm_router,
3980 user_custom_key_update=custom_key_update_hook,
3981 user_custom_key_policy=custom_key_policy_hook,
3982 existing_key_row=existing_by_token[db_token],
3983 )
3985 successful_updates.append(SuccessfulKeyUpdate(key=token, key_info=updated_key_info))
3987 except Exception as e:
3988 # Log the hashed prefix — `token` may be a raw sk-... and ERROR logs persist.
3989 verbose_proxy_logger.exception("Failed to update key %s... in team %s: %s", db_token[:12], data.team_id, e)
3990 failed_updates.append(
3991 _build_failed_team_key_update(
3992 token=token,
3993 exception=e,
3994 existing_key_row=existing_by_token.get(db_token),
3995 )
3996 )
3998 return BulkUpdateKeyResponse(
3999 total_requested=len(requested_tokens),
4000 successful_updates=successful_updates,
4001 failed_updates=failed_updates,
4002 )
4005async def validate_key_team_change(
4006 key: LiteLLM_VerificationToken,
4007 team: LiteLLM_TeamTable,
4008 change_initiated_by: UserAPIKeyAuth,
4009 llm_router: Router,
4010):
4011 """
4012 Validate that a key can be moved to a new team.
4014 - The team must have access to the key's models
4015 - The key's user_id must be a member of the team
4016 - The key's tpm/rpm limit must be less than the team's tpm/rpm limit
4017 - The person initiating the change must be either Proxy Admin or Team Admin
4018 """
4019 # Check if the team has access to the key's models
4020 if len(key.models) > 0:
4021 for model in key.models:
4022 # Skip special sentinel values — "all-team-models" means
4023 # "use whatever the team allows", so it's always valid.
4024 if model == SpecialModelNames.all_team_models.value:
4025 continue
4026 await can_team_access_model(
4027 model=model,
4028 team_object=team,
4029 llm_router=llm_router,
4030 )
4032 # Check if the key's tpm/rpm limit is less than the team's tpm/rpm limit
4033 if key.tpm_limit is not None:
4034 if team.tpm_limit and key.tpm_limit > team.tpm_limit:
4035 raise HTTPException(
4036 status_code=403,
4037 detail=f"Key={key.token} has a tpm_limit={key.tpm_limit} which is greater than the team's tpm_limit={team.tpm_limit}.",
4038 )
4039 if team.rpm_limit and key.rpm_limit and key.rpm_limit > team.rpm_limit:
4040 raise HTTPException(
4041 status_code=403,
4042 detail=f"Key={key.token} has a rpm_limit={key.rpm_limit} which is greater than the team's rpm_limit={team.rpm_limit}.",
4043 )
4045 team_table: Final = cast(LiteLLM_TeamTableCachedObj, team)
4047 # Check if the key's user_id is a member of the team
4048 member_object: Final = _get_user_in_team(team_table=team_table, user_id=key.user_id)
4049 if key.user_id is not None:
4050 if not member_object:
4051 raise HTTPException(
4052 status_code=403,
4053 detail=f"User={key.user_id} is not a member of the team={team.team_id}. Check team members via `/team/info`.",
4054 )
4056 # Check if the person initiating the change is a Proxy Admin or Team Admin
4057 if (
4058 change_initiated_by.user_role == LitellmUserRoles.PROXY_ADMIN.value
4059 or _is_user_team_admin(
4060 user_api_key_dict=change_initiated_by,
4061 team_obj=team,
4062 )
4063 or TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
4064 team_member_role=None if member_object is None else member_object.role,
4065 team_table=team_table,
4066 route=KeyManagementRoutes.KEY_UPDATE.value,
4067 )
4068 ):
4069 return
4070 else:
4071 raise HTTPException(
4072 status_code=403,
4073 detail=f"User={change_initiated_by.user_id} is not a Proxy Admin or Team Admin for team={team.team_id}. Please ask your Proxy Admin to allow this action under 'Member Permissions' for this team.",
4074 )
4077@router.post("/key/delete", tags=["key management"], dependencies=[Depends(user_api_key_auth)])
4078@management_endpoint_wrapper
4079async def delete_key_fn(
4080 data: KeyRequest,
4081 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
4082 litellm_changed_by: str | None = Header(
4083 None,
4084 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
4085 ),
4086):
4087 """
4088 Delete a key from the key management system.
4090 Parameters::
4091 - keys (List[str]): A list of keys or hashed keys to delete. Example {"keys": ["sk-QWrxEynunsNpV1zT48HIrw", "837e17519f44683334df5291321d97b8bf1098cd490e49e215f6fea935aa28be"]}
4092 - key_aliases (List[str]): A list of key aliases to delete. Can be passed instead of `keys`.Example {"key_aliases": ["alias1", "alias2"]}
4094 Returns:
4095 - deleted_keys (List[str]): A list of deleted keys. Example {"deleted_keys": ["sk-QWrxEynunsNpV1zT48HIrw", "837e17519f44683334df5291321d97b8bf1098cd490e49e215f6fea935aa28be"]}
4097 Example:
4098 ```bash
4099 curl --location 'http://0.0.0.0:4000/key/delete' \
4100 --header 'Authorization: Bearer sk-1234' \
4101 --header 'Content-Type: application/json' \
4102 --data '{
4103 "keys": ["sk-QWrxEynunsNpV1zT48HIrw"]
4104 }'
4105 ```
4107 Raises:
4108 HTTPException: If an error occurs during key deletion.
4109 """
4110 try:
4111 from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
4113 if prisma_client is None: 4113 ↛ 4114line 4113 didn't jump to line 4114 because the condition on line 4113 was never true
4114 raise Exception("Not connected to DB!")
4116 # Normalize litellm_changed_by: if it's a Header object or not a string, convert to None
4117 if litellm_changed_by is not None and not isinstance(litellm_changed_by, str): 4117 ↛ 4118line 4117 didn't jump to line 4118 because the condition on line 4117 was never true
4118 litellm_changed_by = None
4120 ## only allow user to delete keys they own
4121 verbose_proxy_logger.debug("user_api_key_dict.user_role: %s", user_api_key_dict.user_role)
4123 num_keys_to_be_deleted = 0
4124 deleted_keys = []
4125 if data.keys:
4126 number_deleted_keys, _keys_being_deleted = await delete_verification_tokens(
4127 tokens=data.keys,
4128 user_api_key_cache=user_api_key_cache,
4129 user_api_key_dict=user_api_key_dict,
4130 litellm_changed_by=litellm_changed_by,
4131 )
4132 num_keys_to_be_deleted = len(data.keys)
4133 deleted_keys = data.keys
4134 elif data.key_aliases: 4134 ↛ 4145line 4134 didn't jump to line 4145 because the condition on line 4134 was always true
4135 number_deleted_keys, _keys_being_deleted = await delete_key_aliases(
4136 key_aliases=data.key_aliases,
4137 prisma_client=prisma_client,
4138 user_api_key_cache=user_api_key_cache,
4139 user_api_key_dict=user_api_key_dict,
4140 litellm_changed_by=litellm_changed_by,
4141 )
4142 num_keys_to_be_deleted = len(data.key_aliases)
4143 deleted_keys = data.key_aliases
4144 else:
4145 raise ValueError("Invalid request type")
4147 if number_deleted_keys is None:
4148 raise ProxyException(
4149 message="Failed to delete keys got None response from delete_verification_token",
4150 type=ProxyErrorTypes.internal_server_error,
4151 param="keys",
4152 code=status.HTTP_500_INTERNAL_SERVER_ERROR,
4153 )
4154 verbose_proxy_logger.debug("/key/delete - deleted_keys=%s", number_deleted_keys)
4156 try:
4157 assert num_keys_to_be_deleted == len(deleted_keys)
4158 except Exception:
4159 raise HTTPException(
4160 status_code=400,
4161 detail={
4162 "error": f"Not all keys passed in were deleted. This probably means you don't have access to delete all the keys passed in. Keys passed in={num_keys_to_be_deleted}, Deleted keys ={number_deleted_keys}"
4163 },
4164 )
4166 verbose_proxy_logger.debug(
4167 "/keys/delete - cache after delete: %s", user_api_key_cache.key_object_cache.in_memory_cache.cache_dict
4168 )
4170 asyncio.create_task(
4171 KeyManagementEventHooks.async_key_deleted_hook(
4172 data=data,
4173 keys_being_deleted=_keys_being_deleted,
4174 user_api_key_dict=user_api_key_dict,
4175 litellm_changed_by=litellm_changed_by,
4176 response=number_deleted_keys,
4177 )
4178 )
4180 return {"deleted_keys": deleted_keys}
4181 except Exception as e:
4182 verbose_proxy_logger.exception("litellm.proxy.proxy_server.delete_key_fn(): Exception occured - %s", e)
4183 raise handle_exception_on_proxy(e)
4186async def _build_model_max_budget_usage(
4187 api_key_hash: str,
4188 model_max_budget: Mapping[str, Mapping[str, object]],
4189 user_api_key_cache: DualCache | None,
4190) -> dict[str, dict[str, object]]:
4191 return await build_model_max_budget_usage(
4192 entity_type=Litellm_EntityType.KEY,
4193 entity_id=api_key_hash,
4194 model_max_budget=model_max_budget,
4195 cache=user_api_key_cache,
4196 )
4199def _window_max_budget(window: Mapping[str, object]) -> float | None:
4200 """A window's max_budget as a float; None when absent or unparseable."""
4201 value: Final = window.get("max_budget")
4202 if not isinstance(value, (int, float, str)):
4203 return None
4204 try:
4205 return float(value)
4206 except ValueError:
4207 return None
4210async def _budget_window_usage(
4211 window: Mapping[str, object], api_key_hash: str
4212) -> tuple[str, Mapping[str, object]] | None:
4213 """
4214 (budget_duration, usage entry) for one budget window; None when the window
4215 has no budget_duration to key it by.
4217 Reads the same cross-pod counter (spend:key:{hashed_token}:window:{budget_duration})
4218 that _virtual_key_multi_budget_check enforces against, passing the same
4219 window_duration + window_start so a stale-low counter is re-checked against
4220 the LiteLLM_BudgetWindowSpend row instead of a spend-log aggregate.
4221 """
4222 from litellm.proxy.proxy_server import get_current_spend
4224 duration: Final = window.get("budget_duration")
4225 if not isinstance(duration, str) or not duration:
4226 return None
4227 spend: Final = await get_current_spend(
4228 counter_key=f"spend:key:{api_key_hash}:window:{duration}",
4229 fallback_spend=0.0,
4230 max_budget=_window_max_budget(window),
4231 window_entity_type="Key",
4232 window_entity_id=api_key_hash,
4233 window_duration=duration,
4234 window_start=get_budget_window_start(window),
4235 )
4236 return duration, MappingProxyType({"current_spend": round(spend, 4)})
4239async def _build_budget_limits_usage(
4240 budget_limits: Sequence[object] | str | None, api_key_hash: str
4241) -> Mapping[str, Mapping[str, object]] | None:
4242 """
4243 Current-window spend per budget window, keyed by budget_duration, reported
4244 next to the stored budget_limits (which is returned untouched). None when
4245 the key has no windows, so the field only appears on keys that have them.
4246 """
4247 windows: Final = _budget_limit_windows(budget_limits)
4248 if not windows: 4248 ↛ 4250line 4248 didn't jump to line 4250 because the condition on line 4248 was always true
4249 return None
4250 usages: Final = await asyncio.gather(
4251 *(_budget_window_usage(window=window, api_key_hash=api_key_hash) for window in windows)
4252 )
4253 return MappingProxyType({duration: usage for duration, usage in (u for u in usages if u is not None)})
4256@router.post(
4257 "/v2/key/info",
4258 tags=["key management"],
4259 dependencies=[Depends(user_api_key_auth)],
4260 include_in_schema=False,
4261)
4262async def info_key_fn_v2(
4263 data: KeyRequest | None = None,
4264 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
4265):
4266 """
4267 Retrieve information about a list of keys.
4269 **New endpoint**. Currently admin only.
4270 Parameters:
4271 keys: Optional[list] = body parameter representing the key(s) in the request
4272 user_api_key_dict: UserAPIKeyAuth = Dependency representing the user's API key
4273 Returns:
4274 Dict containing the key and its associated information
4276 Example Curl:
4277 ```
4278 curl -X GET "http://0.0.0.0:4000/key/info" \
4279 -H "Authorization: Bearer sk-1234" \
4280 -d {"keys": ["sk-1", "sk-2", "sk-3"]}
4281 ```
4282 """
4283 from litellm.proxy.proxy_server import (
4284 model_max_budget_limiter,
4285 prisma_client,
4286 )
4288 try:
4289 if prisma_client is None:
4290 raise Exception(
4291 "Database not connected. Connect a database to your proxy - https://docs.litellm.ai/docs/simple_proxy#managing-auth---virtual-keys"
4292 )
4293 if data is None:
4294 raise HTTPException(
4295 status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
4296 detail={"message": "Malformed request. No keys passed in."},
4297 )
4298 # Resolve key_aliases to tokens so we never pass token=None (unbounded query)
4299 tokens_to_query: Final = list(data.keys) if data.keys else []
4300 if data.key_aliases:
4301 alias_rows: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
4302 where={"key_alias": {"in": data.key_aliases}},
4303 include={"litellm_budget_table": True},
4304 )
4305 alias_tokens: Final = [row.token for row in alias_rows if row.token]
4306 tokens_to_query.extend(alias_tokens)
4308 if not tokens_to_query:
4309 return {"key": data.keys, "info": []}
4311 key_info: Final = await prisma_client.get_data(token=tokens_to_query, table_name="key", query_type="find_all")
4312 if not key_info:
4313 return {"key": data.keys, "info": []}
4315 filtered_key_info: Final = []
4316 for k in key_info:
4317 if not await _can_user_query_key_info(
4318 user_api_key_dict=user_api_key_dict,
4319 key=k.token,
4320 key_info=k,
4321 ):
4322 continue
4323 try:
4324 k_dict = k.model_dump()
4325 except Exception:
4326 k_dict = k.dict()
4327 k_token_hash = k_dict.pop("token", None)
4329 model_max_budget = k_dict.get("model_max_budget") or {}
4330 budget_table = k_dict.get("litellm_budget_table") or {}
4331 if not model_max_budget and isinstance(budget_table, dict):
4332 model_max_budget = budget_table.get("model_max_budget") or {}
4333 if model_max_budget and k_token_hash:
4334 k_dict["model_max_budget_usage"] = await _build_model_max_budget_usage(
4335 api_key_hash=k_token_hash,
4336 model_max_budget=model_max_budget,
4337 user_api_key_cache=model_max_budget_limiter.dual_cache,
4338 )
4339 if k_token_hash:
4340 budget_limits_usage = await _build_budget_limits_usage(
4341 budget_limits=k_dict.get("budget_limits"),
4342 api_key_hash=k_token_hash,
4343 )
4344 if budget_limits_usage is not None:
4345 k_dict["budget_limits_usage"] = budget_limits_usage
4347 filtered_key_info.append(k_dict)
4348 return {"key": data.keys, "info": filtered_key_info}
4350 except Exception as e:
4351 raise handle_exception_on_proxy(e)
4354@router.get("/key/info", tags=["key management"], dependencies=[Depends(user_api_key_auth)])
4355@management_endpoint_wrapper
4356async def info_key_fn(
4357 key: str | None = fastapi.Query(
4358 default=None,
4359 description=(
4360 "Key to look up. Pass the key's sha256 hash so the raw key stays out of URLs and access "
4361 "logs. Example key='d5345c0ecc68ae6295c69f91926b2bd379e25481a40c34b5884d157a9f65d8fa'"
4362 ),
4363 ),
4364 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
4365):
4366 """
4367 Retrieve information about a key.
4369 Parameters:
4370 - key: str | None (query parameter) - The key to look up. Accepts the plaintext key or its hash;
4371 prefer the hash, since a query parameter is recorded verbatim by any HTTP access log in front
4372 of the proxy. Defaults to the key in the Authorization header.
4374 Returns:
4375 - key: str - The key that was looked up, echoed back as it was passed in
4376 - info: dict - The key's row, minus the hashed token. Deleted keys are served from the
4377 LiteLLM_DeletedVerificationToken archive and carry deleted_at / deleted_by
4378 - status: "active" | "expired" | "revoked" | "deleted" - Derived from blocked, expires and
4379 whether the row came from the archive
4380 - key_alias: str | None - User-friendly key alias
4381 - spend: float - Amount spent by the key. When budget_duration is set this covers only the
4382 current budget window, not the key's lifetime
4383 - max_budget: float | None - Max budget for the key, enforced against spend
4384 - budget_duration: str | None - Budget reset period ("30d", "1h", etc.)
4385 - budget_reset_at: datetime | None - When the current budget window ends and spend is next
4386 reset to 0, not when it was last reset. Reset times snap to standard boundaries in the
4387 configured timezone (30d and 1mo land on the 1st of the month, 7d on Monday, 1h on the
4388 hour), so subtracting budget_duration from it does not give the window's start
4389 - model_max_budget: dict - Per-model budgets, e.g. {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}
4390 - model_max_budget_usage: dict | None - Current-window spend per model, present only when
4391 the key has per-model budgets
4392 - budget_limits: list | None - Concurrent budget windows, exactly as stored
4393 - budget_limits_usage: dict | None - Current-window spend per budget window, e.g.
4394 {"1h": {"current_spend": 0.0009}}, present only when the key has budget windows
4395 (read from the same cross-pod spend counter the budget enforcement uses)
4396 - models: list - Model_name's the key is allowed to call
4397 - tpm_limit / rpm_limit: int | None - Tokens and requests per minute limits
4398 - metadata: dict - Metadata for the key, e.g. {"team": "core-infra"}
4399 - blocked: bool | None - Whether the key is blocked
4400 - expires: datetime | None - When the key stops authenticating requests
4401 - last_active: datetime | None - When the key was last used
4402 - object_permission: dict | None - Resolved vector store / MCP permissions when the key has
4403 an object_permission_id
4405 Example Curl:
4406 ```
4407 curl -X GET "http://0.0.0.0:4000/key/info?key=d5345c0ecc68ae6295c69f91926b2bd379e25481a40c34b5884d157a9f65d8fa" \
4408-H "Authorization: Bearer sk-1234"
4409 ```
4411 Example Curl - if no key is passed, it will use the Key Passed in Authorization Header
4412 ```
4413 curl -X GET "http://0.0.0.0:4000/key/info" \
4414-H "Authorization: Bearer sk-test-example-key-123"
4415 ```
4416 """
4417 from litellm.proxy.proxy_server import (
4418 model_max_budget_limiter,
4419 prisma_client,
4420 )
4422 try:
4423 if prisma_client is None: 4423 ↛ 4424line 4423 didn't jump to line 4424 because the condition on line 4423 was never true
4424 raise Exception(
4425 "Database not connected. Connect a database to your proxy - https://docs.litellm.ai/docs/simple_proxy#managing-auth---virtual-keys"
4426 )
4428 # default to using Auth token if no key is passed in
4429 key = key or user_api_key_dict.api_key
4430 hashed_key: str | None = key
4431 if key is not None: 4431 ↛ 4433line 4431 didn't jump to line 4433 because the condition on line 4431 was always true
4432 hashed_key = _hash_token_if_needed(token=key)
4433 live_key_info: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
4434 where={"token": hashed_key},
4435 include={"litellm_budget_table": True},
4436 )
4437 key_info: Final = (
4438 live_key_info
4439 if live_key_info is not None
4440 else await _find_deleted_key_info(prisma_client=prisma_client, hashed_key=hashed_key)
4441 )
4442 if key_info is None:
4443 raise ProxyException(
4444 message="Key not found in database",
4445 type=ProxyErrorTypes.not_found_error,
4446 param="key",
4447 code=status.HTTP_404_NOT_FOUND,
4448 )
4449 if ( 4449 ↛ 4457line 4449 didn't jump to line 4457 because the condition on line 4449 was never true
4450 await _can_user_query_key_info(
4451 user_api_key_dict=user_api_key_dict,
4452 key=key,
4453 key_info=key_info,
4454 )
4455 is not True
4456 ):
4457 raise HTTPException(
4458 status_code=status.HTTP_403_FORBIDDEN,
4459 detail=f"You are not allowed to access this key's info. Your role={user_api_key_dict.user_role}",
4460 )
4461 ## REMOVE HASHED TOKEN INFO BEFORE RETURNING ##
4462 key_info_dict: Final = key_info.model_dump()
4463 key_token_hash: Final[str | None] = key_info_dict.pop("token")
4464 key_info_dict["status"] = (
4465 "deleted" if live_key_info is None else _derive_key_status(key_info_dict, now=datetime.now(timezone.utc))
4466 )
4468 model_max_budget = key_info_dict.get("model_max_budget") or {}
4469 budget_table: Final = key_info_dict.get("litellm_budget_table") or {}
4470 if not model_max_budget and isinstance(budget_table, dict): 4470 ↛ 4472line 4470 didn't jump to line 4472 because the condition on line 4470 was always true
4471 model_max_budget = budget_table.get("model_max_budget") or {}
4472 if model_max_budget and key_token_hash: 4472 ↛ 4473line 4472 didn't jump to line 4473 because the condition on line 4472 was never true
4473 key_info_dict["model_max_budget_usage"] = await _build_model_max_budget_usage(
4474 api_key_hash=key_token_hash,
4475 model_max_budget=model_max_budget,
4476 user_api_key_cache=model_max_budget_limiter.dual_cache,
4477 )
4478 budget_limits_usage: Final = await _build_budget_limits_usage(
4479 budget_limits=key_info_dict.get("budget_limits"),
4480 api_key_hash=key_token_hash,
4481 )
4482 if budget_limits_usage is not None: 4482 ↛ 4483line 4482 didn't jump to line 4483 because the condition on line 4482 was never true
4483 key_info_dict["budget_limits_usage"] = budget_limits_usage
4485 return {"key": key, "info": await attach_object_permission_to_dict(key_info_dict, prisma_client)}
4486 except Exception as e:
4487 raise handle_exception_on_proxy(e)
4490async def _find_deleted_key_info(
4491 prisma_client: PrismaClient, hashed_key: str | None
4492) -> LiteLLM_DeletedVerificationToken | None:
4493 archived_row: Final = await _deleted_verification_token_table(prisma_client).find_first(
4494 where={"token": hashed_key},
4495 order={"deleted_at": "desc"},
4496 )
4497 if archived_row is None: 4497 ↛ 4499line 4497 didn't jump to line 4499 because the condition on line 4497 was always true
4498 return None
4499 return LiteLLM_DeletedVerificationToken.model_validate(archived_row.model_dump())
4502def _check_model_access_group(models: list[str] | None, llm_router: Router | None, premium_user: bool) -> Literal[True]:
4503 """
4504 if is_model_access_group is True + is_wildcard_route is True, check if user is a premium user
4506 Return True if user is a premium user, False otherwise
4507 """
4508 if models is None or llm_router is None:
4509 return True
4511 for model in models: 4511 ↛ 4512line 4511 didn't jump to line 4512 because the loop on line 4511 never started
4512 if llm_router._is_model_access_group_for_wildcard_route(model_access_group=model):
4513 if not premium_user:
4514 raise HTTPException(
4515 status_code=status.HTTP_403_FORBIDDEN,
4516 detail={
4517 "error": f"Setting a model access group on a wildcard model is only available for LiteLLM Enterprise users.{CommonProxyErrors.not_premium_user.value}"
4518 },
4519 )
4521 return True
4524_NO_METADATA: Final[Mapping[str, object]] = MappingProxyType({})
4527def metadata_json_with_limits(
4528 metadata: Mapping[str, object] | None,
4529 *,
4530 model_rpm_limit: Mapping[str, object] | None,
4531 model_tpm_limit: Mapping[str, object] | None,
4532 mcp_rpm_limit: Mapping[str, int] | None,
4533 tag_rpm_limit: Mapping[str, int] | None,
4534 guardrails: Sequence[str] | None,
4535 policies: Sequence[str] | None,
4536 prompts: Sequence[str] | None,
4537) -> str:
4538 """Serialize the stored metadata blob with the per-model, MCP, tag, guardrail, policy and prompt settings folded in."""
4539 limits: Final = tuple(
4540 (name, value)
4541 for name, value in (
4542 ("model_rpm_limit", model_rpm_limit),
4543 ("model_tpm_limit", model_tpm_limit),
4544 ("mcp_rpm_limit", mcp_rpm_limit),
4545 ("tag_rpm_limit", tag_rpm_limit),
4546 ("guardrails", guardrails),
4547 ("policies", policies),
4548 ("prompts", prompts),
4549 )
4550 if value is not None
4551 )
4552 if metadata is None and not limits:
4553 return json.dumps(None)
4554 merged: Final = {**(metadata or _NO_METADATA), **dict(limits)} # mutable-ok: encrypt_callback_vars takes a dict
4555 return json.dumps(encrypt_callback_vars(merged))
4558async def generate_key_helper_fn(
4559 request_type: Literal["user", "key"], # identifies if this request is from /user/new or /key/generate
4560 duration: str | None = None,
4561 models: list = [],
4562 aliases: dict = {},
4563 config: dict = {},
4564 spend: float = 0.0,
4565 key_max_budget: float | None = None, # key_max_budget is used to Budget Per key
4566 key_budget_duration: str | None = None,
4567 budget_id: float | None = None, # budget id <-> LiteLLM_BudgetTable
4568 soft_budget: float | None = None, # soft_budget is used to set soft Budgets Per user
4569 max_budget: float | None = None, # max_budget is used to Budget Per user
4570 blocked: bool | None = None,
4571 budget_duration: str | None = None, # max_budget is used to Budget Per user
4572 token: str | None = None,
4573 key: str
4574 | None = None, # dev-friendly alt param for 'token'. Exposed on `/key/generate` for setting key value yourself.
4575 user_id: str | None = None,
4576 user_alias: str | None = None,
4577 team_id: str | None = None,
4578 agent_id: str | None = None,
4579 user_email: str | None = None,
4580 user_role: str | None = None,
4581 max_parallel_requests: int | None = None,
4582 metadata: dict | None = {},
4583 tpm_limit: int | None = None,
4584 rpm_limit: int | None = None,
4585 tpd_limit: int | None = None,
4586 query_type: Literal["insert_data", "update_data"] = "insert_data",
4587 update_key_values: dict | None = None,
4588 key_alias: str | None = None,
4589 allowed_cache_controls: list | None = [],
4590 permissions: dict | None = {},
4591 model_max_budget: dict | None = {},
4592 budget_fallbacks: dict | None = None,
4593 model_rpm_limit: dict | None = None,
4594 model_tpm_limit: dict | None = None,
4595 mcp_rpm_limit: dict | None = None,
4596 tag_rpm_limit: dict | None = None,
4597 guardrails: list | None = None,
4598 policies: list | None = None,
4599 prompts: list | None = None,
4600 teams: list | None = None,
4601 organization_id: str | None = None,
4602 project_id: str | None = None,
4603 table_name: Literal["key", "user"] | None = None,
4604 send_invite_email: bool | None = None,
4605 created_by: str | None = None,
4606 updated_by: str | None = None,
4607 allowed_routes: list | None = None,
4608 key_type: str | None = None,
4609 sso_user_id: str | None = None,
4610 object_permission_id: str | None = None, # object_permission_id <-> LiteLLM_ObjectPermissionTable
4611 object_permission: LiteLLM_ObjectPermissionBase | None = None,
4612 auto_rotate: bool | None = None,
4613 rotation_interval: str | None = None,
4614 router_settings: dict[str, object] | None = None,
4615 access_group_ids: list[str] | None = None,
4616 budget_limits: list | None = None, # multiple concurrent budget windows
4617 *,
4618 llm_router: Router | None = None,
4619):
4620 from litellm.proxy.proxy_server import premium_user, prisma_client
4622 if prisma_client is None: 4622 ↛ 4623line 4622 didn't jump to line 4623 because the condition on line 4622 was never true
4623 raise Exception("Connect Proxy to database to generate keys - https://docs.litellm.ai/docs/proxy/virtual_keys ")
4625 await validate_router_settings_weights(
4626 router_settings,
4627 team_id=team_id,
4628 prisma_client=prisma_client,
4629 llm_router=llm_router,
4630 )
4632 if token is None: 4632 ↛ 4638line 4632 didn't jump to line 4638 because the condition on line 4632 was always true
4633 if key is not None: 4633 ↛ 4634line 4633 didn't jump to line 4634 because the condition on line 4633 was never true
4634 token = key
4635 else:
4636 token = f"sk-{secrets.token_urlsafe(LENGTH_OF_LITELLM_GENERATED_KEY)}"
4638 if duration is None: # allow tokens that never expire
4639 expires = None
4640 else:
4641 # Add duration to current time for exact expiration (not standardized reset time)
4642 duration_seconds: Final = duration_in_seconds(duration)
4643 expires = datetime.now(timezone.utc) + timedelta(seconds=duration_seconds)
4645 if key_budget_duration is None: # one-time budget 4645 ↛ 4648line 4645 didn't jump to line 4648 because the condition on line 4645 was always true
4646 key_reset_at = None
4647 else:
4648 key_reset_at = get_budget_reset_time(budget_duration=key_budget_duration)
4650 if budget_duration is None: # one-time budget
4651 reset_at = None
4652 else:
4653 reset_at = get_budget_reset_time(budget_duration=budget_duration)
4655 # Initialize reset_at for each budget window
4656 budget_limits_json: str | None = None
4657 if budget_limits:
4658 initialized_windows: Final = []
4659 for window in budget_limits:
4660 w = dict(window) if not isinstance(window, dict) else {**window}
4661 w["reset_at"] = get_budget_reset_time(budget_duration=w["budget_duration"]).isoformat()
4662 initialized_windows.append(w)
4663 budget_limits_json = json.dumps(initialized_windows)
4665 aliases_json: Final = json.dumps(aliases)
4666 config_json: Final = json.dumps(config)
4667 permissions_json: Final = json.dumps(permissions)
4668 router_settings_json: Final = safe_dumps(router_settings) if router_settings is not None else safe_dumps({})
4670 metadata_json: Final = metadata_json_with_limits(
4671 metadata,
4672 model_rpm_limit=model_rpm_limit,
4673 model_tpm_limit=model_tpm_limit,
4674 mcp_rpm_limit=mcp_rpm_limit,
4675 tag_rpm_limit=tag_rpm_limit,
4676 guardrails=guardrails,
4677 policies=policies,
4678 prompts=prompts,
4679 )
4680 validate_model_max_budget(model_max_budget)
4681 model_max_budget_json: Final = json.dumps(model_max_budget)
4682 budget_fallbacks_json: Final = json.dumps(budget_fallbacks or {})
4683 user_role = user_role
4684 tpm_limit = tpm_limit
4685 rpm_limit = rpm_limit
4686 allowed_cache_controls = allowed_cache_controls
4688 try:
4689 # Create a new verification token (you may want to enhance this logic based on your needs)
4691 user_data: Final = {
4692 "max_budget": max_budget,
4693 "user_email": user_email,
4694 "user_id": user_id,
4695 "user_alias": user_alias,
4696 "team_id": team_id,
4697 "organization_id": organization_id,
4698 "user_role": user_role,
4699 "spend": spend,
4700 "models": models,
4701 "metadata": metadata_json,
4702 "max_parallel_requests": max_parallel_requests,
4703 "tpm_limit": tpm_limit,
4704 "rpm_limit": rpm_limit,
4705 "budget_duration": budget_duration,
4706 "budget_reset_at": reset_at,
4707 "allowed_cache_controls": allowed_cache_controls,
4708 "sso_user_id": sso_user_id,
4709 "object_permission_id": object_permission_id,
4710 }
4711 if teams is not None: 4711 ↛ 4712line 4711 didn't jump to line 4712 because the condition on line 4711 was never true
4712 user_data["teams"] = teams
4713 if model_max_budget: 4713 ↛ 4716line 4713 didn't jump to line 4716 because the condition on line 4713 was never true
4714 # Only when supplied: the SSO and default-key callers reach this with the
4715 # empty default, and writing that would clear an existing user's budgets.
4716 user_data["model_max_budget"] = model_max_budget_json
4717 key_data: Final = {
4718 "token": token,
4719 "key_alias": key_alias,
4720 "expires": expires,
4721 "models": models,
4722 "aliases": aliases_json,
4723 "config": config_json,
4724 "spend": spend,
4725 "max_budget": key_max_budget,
4726 "user_id": user_id,
4727 "team_id": team_id,
4728 "agent_id": agent_id,
4729 "project_id": project_id,
4730 "max_parallel_requests": max_parallel_requests,
4731 "metadata": metadata_json,
4732 "tpm_limit": tpm_limit,
4733 "rpm_limit": rpm_limit,
4734 "tpd_limit": tpd_limit,
4735 "budget_duration": key_budget_duration,
4736 "budget_reset_at": key_reset_at,
4737 "allowed_cache_controls": allowed_cache_controls,
4738 "permissions": permissions_json,
4739 "model_max_budget": model_max_budget_json,
4740 "budget_fallbacks": budget_fallbacks_json,
4741 "organization_id": organization_id,
4742 "budget_id": budget_id,
4743 "blocked": blocked,
4744 "budget_limits": budget_limits_json,
4745 "created_by": created_by,
4746 "updated_by": updated_by,
4747 "allowed_routes": allowed_routes or [],
4748 "key_type": key_type,
4749 "object_permission_id": object_permission_id,
4750 "router_settings": router_settings_json,
4751 "access_group_ids": access_group_ids or [],
4752 }
4754 # Add rotation fields if auto_rotate is enabled
4755 _set_key_rotation_fields(
4756 data=key_data,
4757 auto_rotate=auto_rotate or False,
4758 rotation_interval=rotation_interval,
4759 )
4761 if ( 4761 ↛ 4764line 4761 didn't jump to line 4764 because the condition on line 4761 was never true
4762 get_secret("DISABLE_KEY_NAME", False) is True
4763 ): # allow user to disable storing abbreviated key name (shown in UI, to help figure out which key spent how much)
4764 pass
4765 else:
4766 key_data["key_name"] = abbreviate_api_key(api_key=token)
4767 saved_token: Final = copy.deepcopy(key_data)
4768 if isinstance(saved_token["aliases"], str): 4768 ↛ 4770line 4768 didn't jump to line 4770 because the condition on line 4768 was always true
4769 saved_token["aliases"] = json.loads(saved_token["aliases"])
4770 if isinstance(saved_token["config"], str): 4770 ↛ 4772line 4770 didn't jump to line 4772 because the condition on line 4770 was always true
4771 saved_token["config"] = json.loads(saved_token["config"])
4772 if isinstance(saved_token["metadata"], str): 4772 ↛ 4774line 4772 didn't jump to line 4774 because the condition on line 4772 was always true
4773 saved_token["metadata"] = json.loads(saved_token["metadata"])
4774 if isinstance(saved_token["permissions"], str): 4774 ↛ 4779line 4774 didn't jump to line 4779 because the condition on line 4774 was always true
4775 if "get_spend_routes" in saved_token["permissions"] and premium_user is not True: 4775 ↛ 4776line 4775 didn't jump to line 4776 because the condition on line 4775 was never true
4776 raise ValueError("get_spend_routes permission is only available for LiteLLM Enterprise users")
4778 saved_token["permissions"] = json.loads(saved_token["permissions"])
4779 if isinstance(saved_token["model_max_budget"], str): 4779 ↛ 4781line 4779 didn't jump to line 4781 because the condition on line 4779 was always true
4780 saved_token["model_max_budget"] = json.loads(saved_token["model_max_budget"])
4781 router_settings = cast(dict | None, saved_token.get("router_settings"))
4782 if router_settings is not None and isinstance(router_settings, str): 4782 ↛ 4789line 4782 didn't jump to line 4789 because the condition on line 4782 was always true
4783 try:
4784 saved_token["router_settings"] = yaml.safe_load(router_settings)
4785 except yaml.YAMLError:
4786 # If it's not valid JSON/YAML, keep as is or set to empty dict
4787 saved_token["router_settings"] = {}
4789 if saved_token.get("expires", None) is not None and isinstance(saved_token["expires"], datetime): 4789 ↛ 4790line 4789 didn't jump to line 4790 because the condition on line 4789 was never true
4790 saved_token["expires"] = saved_token["expires"].isoformat()
4791 if prisma_client is not None: 4791 ↛ 4856line 4791 didn't jump to line 4856 because the condition on line 4791 was always true
4792 if table_name is None or table_name == "user": # do not auto-create users for `/key/generate`
4793 ## CREATE USER (If necessary)
4794 if query_type == "insert_data": 4794 ↛ 4805line 4794 didn't jump to line 4805 because the condition on line 4794 was always true
4795 user_row = cast( # cast-ok: table_name="user" is the insert_data branch returning the user row
4796 "prisma_models.LiteLLM_UserTable | None",
4797 await prisma_client.insert_data(data=user_data, table_name="user"),
4798 )
4800 if user_row is None: 4800 ↛ 4801line 4800 didn't jump to line 4801 because the condition on line 4800 was never true
4801 raise Exception("Failed to create user")
4802 ## use default user model list if no key-specific model list provided
4803 if len(user_row.models) > 0 and len(key_data["models"]) == 0: 4803 ↛ 4804line 4803 didn't jump to line 4804 because the condition on line 4803 was never true
4804 key_data["models"] = user_row.models
4805 elif query_type == "update_data":
4806 user_row = await prisma_client.update_data(
4807 data=user_data,
4808 table_name="user",
4809 update_key_values=update_key_values,
4810 )
4811 if table_name is not None and table_name == "user":
4812 # do not create a key if table name is set to just 'user'
4813 # we only need to ensure this exists in the user table
4814 # the LiteLLM_VerificationToken table will increase in size if we don't do this check
4815 return user_data
4817 ## CREATE KEY
4818 verbose_proxy_logger.debug(
4819 "prisma_client: Creating Key= %s",
4820 {**key_data, "token": hash_token(token=token)},
4821 )
4822 create_key_response: Final = await prisma_client.insert_data(data=key_data, table_name="key")
4824 key_data["token_id"] = getattr(create_key_response, "token", None)
4825 created_token_hash: Final = getattr(create_key_response, "token", None)
4826 if isinstance(created_token_hash, str): 4826 ↛ 4833line 4826 didn't jump to line 4833 because the condition on line 4826 was always true
4827 await sync_key_access_group_membership(
4828 prisma_client=prisma_client,
4829 key_token=created_token_hash,
4830 previous_access_group_ids=None,
4831 updated_access_group_ids=access_group_ids,
4832 )
4833 key_data["litellm_budget_table"] = getattr(create_key_response, "litellm_budget_table", None)
4834 key_data["created_at"] = getattr(create_key_response, "created_at", None)
4835 key_data["updated_at"] = getattr(create_key_response, "updated_at", None)
4837 # Deserialize router_settings from JSON string to dict for response
4838 router_settings_value: Final = key_data.get("router_settings")
4839 if router_settings_value is not None and isinstance(router_settings_value, str): 4839 ↛ 4856line 4839 didn't jump to line 4856 because the condition on line 4839 was always true
4840 try:
4841 key_data["router_settings"] = yaml.safe_load(router_settings_value)
4842 except yaml.YAMLError:
4843 # If it's not valid JSON/YAML, keep as is or set to empty dict
4844 key_data["router_settings"] = {}
4845 except Exception as e:
4846 verbose_proxy_logger.error("litellm.proxy.proxy_server.generate_key_helper_fn(): Exception occured - %s", e)
4847 verbose_proxy_logger.debug(traceback.format_exc())
4848 if isinstance(e, HTTPException): 4848 ↛ 4849line 4848 didn't jump to line 4849 because the condition on line 4848 was never true
4849 raise e
4850 raise HTTPException(
4851 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
4852 detail={"error": "Internal Server Error."},
4853 )
4855 # Add budget related info in key_data - this ensures it's returned
4856 key_data["budget_id"] = budget_id
4858 if request_type == "user":
4859 # if this is a /user/new request update the key_date with user_data fields
4860 key_data.update(user_data)
4862 return key_data
4865async def _team_key_deletion_check(
4866 user_api_key_dict: UserAPIKeyAuth,
4867 key_info: LiteLLM_VerificationToken,
4868 prisma_client: PrismaClient,
4869 user_api_key_cache: UserApiKeyCache,
4870):
4871 is_team_key: Final = _is_team_key(data=key_info)
4873 if is_team_key and key_info.team_id is not None:
4874 team_table: Final = await get_team_object(
4875 team_id=key_info.team_id,
4876 prisma_client=prisma_client,
4877 user_api_key_cache=user_api_key_cache,
4878 check_db_only=True,
4879 )
4880 if litellm.key_generation_settings is not None and "team_key_generation" in litellm.key_generation_settings:
4881 _team_key_generation = litellm.key_generation_settings["team_key_generation"]
4882 else:
4883 _team_key_generation = TeamUIKeyGenerationConfig(
4884 allowed_team_member_roles=["admin", "user"],
4885 )
4886 # check if user is team admin
4887 if team_table is not None:
4888 return _team_key_operation_team_member_check(
4889 assigned_user_id=user_api_key_dict.user_id,
4890 team_table=team_table,
4891 user_api_key_dict=user_api_key_dict,
4892 team_key_generation=_team_key_generation,
4893 route=KeyManagementRoutes.KEY_DELETE,
4894 )
4895 else:
4896 raise HTTPException(
4897 status_code=status.HTTP_404_NOT_FOUND,
4898 detail={"error": f"Team not found in db, and user not proxy admin. Team id = {key_info.team_id}"},
4899 )
4900 return False
4903async def can_modify_verification_token(
4904 key_info: LiteLLM_VerificationToken,
4905 user_api_key_cache: UserApiKeyCache,
4906 user_api_key_dict: UserAPIKeyAuth,
4907 prisma_client: PrismaClient,
4908) -> bool:
4909 """
4910 Check if user has permission to modify (delete/regenerate) a verification token.
4912 Rules:
4913 - Proxy admin can modify any key
4914 - Internal jobs service account can modify any key (for auto-rotation)
4915 - For team keys: only team admin or key owner can modify
4916 - For personal keys: only key owner can modify
4918 Args:
4919 key_info: The verification token to check
4920 user_api_key_cache: Cache for user API keys
4921 user_api_key_dict: The user making the request
4922 prisma_client: Prisma client for database access
4924 Returns:
4925 True if user can modify the key, False otherwise
4926 """
4927 from litellm.constants import LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME
4929 is_team_key: Final = _is_team_key(data=key_info)
4931 # 1. Proxy admin can modify any key
4932 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value:
4933 return True
4935 # 2. Internal jobs service account can modify any key (for auto-rotation)
4936 if user_api_key_dict.api_key == LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME:
4937 return True
4939 # 3. For team keys: only team admin or key owner can modify
4940 if is_team_key and key_info.team_id is not None:
4941 # Get team object to check if user is team admin
4942 team_table: Final = await get_team_object(
4943 team_id=key_info.team_id,
4944 prisma_client=prisma_client,
4945 user_api_key_cache=user_api_key_cache,
4946 check_db_only=True,
4947 )
4949 if team_table is None:
4950 return False
4952 # Check if user is team admin
4953 if _is_user_team_admin(
4954 user_api_key_dict=user_api_key_dict,
4955 team_obj=team_table,
4956 ):
4957 return True
4959 # Check if the key belongs to the user (they own it)
4960 if key_info.user_id is not None and key_info.user_id == user_api_key_dict.user_id:
4961 return True
4963 # Not team admin and doesn't own the key
4964 return False
4966 # 4. For personal keys: only key owner can modify
4967 if key_info.user_id is not None and key_info.user_id == user_api_key_dict.user_id:
4968 return True
4970 # Default: deny
4971 return False
4974async def delete_verification_tokens(
4975 tokens: list,
4976 user_api_key_cache: UserApiKeyCache,
4977 user_api_key_dict: UserAPIKeyAuth,
4978 litellm_changed_by: str | None = None,
4979) -> tuple[dict | None, list[LiteLLM_VerificationToken]]:
4980 """
4981 Helper that deletes the list of tokens from the database
4983 - check if user is proxy admin
4984 - check if user is team admin and key is a team key
4986 Args:
4987 tokens: List of tokens to delete
4988 user_id: Optional user_id to filter by
4990 Returns:
4991 Tuple[Optional[Dict], List[LiteLLM_VerificationToken]]:
4992 Optional[Dict]:
4993 - Number of deleted tokens
4994 List[LiteLLM_VerificationToken]:
4995 - List of keys being deleted, this contains information about the key_alias, token, and user_id being deleted,
4996 this is passed down to the KeyManagementEventHooks to delete the keys from the secret manager and handle audit logs
4997 """
4998 from litellm.proxy.proxy_server import prisma_client
5000 failed_tokens: list = []
5001 try:
5002 if prisma_client: 5002 ↛ 5074line 5002 didn't jump to line 5074 because the condition on line 5002 was always true
5003 hashed_tokens: Final[list[str]] = [_hash_token_if_needed(token=key) for key in tokens]
5004 tokens = hashed_tokens
5005 _keys_being_deleted: Final[list[LiteLLM_VerificationToken]] = cast( # cast-ok: find_many returns a list
5006 "list[LiteLLM_VerificationToken]",
5007 await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
5008 where={"token": {"in": hashed_tokens}}
5009 ),
5010 )
5012 if len(_keys_being_deleted) == 0: 5012 ↛ 5018line 5012 didn't jump to line 5018 because the condition on line 5012 was always true
5013 raise HTTPException(
5014 status_code=status.HTTP_404_NOT_FOUND,
5015 detail={"error": "No keys found"},
5016 )
5018 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value:
5019 authorized_keys = _keys_being_deleted
5020 else:
5021 authorized_keys = []
5022 for key in _keys_being_deleted:
5023 if await can_modify_verification_token(
5024 key_info=key,
5025 user_api_key_cache=user_api_key_cache,
5026 user_api_key_dict=user_api_key_dict,
5027 prisma_client=prisma_client,
5028 ):
5029 authorized_keys.append(key)
5030 else:
5031 raise HTTPException(
5032 status_code=status.HTTP_403_FORBIDDEN,
5033 detail={"error": "You are not authorized to delete this key"},
5034 )
5035 await _persist_deleted_verification_tokens(
5036 keys=authorized_keys,
5037 prisma_client=prisma_client,
5038 user_api_key_dict=user_api_key_dict,
5039 litellm_changed_by=litellm_changed_by,
5040 )
5042 # Snapshot before the delete: the FK cascade drops the mapping rows, but their
5043 # cached jwt_key_mapping entries still resolve to the now-dead token (LIT-5380).
5044 jwt_mapping_cache_keys: Final[tuple[str, ...]] = tuple(
5045 cache_key
5046 for keys_for_token in await asyncio.gather(
5047 *(
5048 get_jwt_key_mapping_cache_keys_for_token(
5049 hashed_token=key.token,
5050 prisma_client=prisma_client,
5051 )
5052 for key in authorized_keys
5053 if key.token is not None
5054 )
5055 )
5056 for cache_key in keys_for_token
5057 )
5059 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value:
5060 deleted_tokens = await prisma_client.delete_data(tokens=tokens)
5061 if deleted_tokens is not None and len(deleted_tokens) != len(tokens):
5062 failed_tokens = [token for token in tokens if token not in deleted_tokens]
5063 else:
5064 deletion_tasks: Final = [prisma_client.delete_data(tokens=[key.token]) for key in authorized_keys]
5065 await asyncio.gather(*deletion_tasks)
5067 deleted_tokens = [key.token for key in authorized_keys]
5068 if len(deleted_tokens) != len(tokens):
5069 failed_tokens = [token for token in tokens if token not in deleted_tokens]
5071 await evict_and_broadcast(cache_keys=jwt_mapping_cache_keys, user_api_key_cache=user_api_key_cache)
5073 else:
5074 raise Exception("DB not connected. prisma_client is None")
5075 except Exception as e:
5076 verbose_proxy_logger.exception(
5077 "litellm.proxy.proxy_server.delete_verification_tokens(): Exception occured - %s", e
5078 )
5079 verbose_proxy_logger.debug(traceback.format_exc())
5080 raise e
5082 for key in tokens:
5083 user_api_key_cache.delete_cache(key)
5084 # remove hash token from cache
5085 hashed_token = hash_token(cast(str, key))
5086 user_api_key_cache.delete_cache(hashed_token)
5088 # After credential invalidation, so a failure here can never keep a deleted key alive.
5089 for deleted_key in authorized_keys:
5090 if deleted_key.token is not None:
5091 await sync_key_access_group_membership(
5092 prisma_client=prisma_client,
5093 key_token=deleted_key.token,
5094 previous_access_group_ids=deleted_key.access_group_ids,
5095 updated_access_group_ids=None,
5096 )
5098 return {
5099 "deleted_keys": deleted_tokens,
5100 "failed_tokens": failed_tokens,
5101 }, _keys_being_deleted
5104def _transform_verification_tokens_to_deleted_records(
5105 keys: Sequence[LiteLLM_VerificationToken],
5106 user_api_key_dict: UserAPIKeyAuth,
5107 litellm_changed_by: str | None = None,
5108) -> list[dict[str, object]]:
5109 """Transform verification tokens into deleted token records ready for persistence."""
5110 if not keys:
5111 return []
5113 deleted_at: Final = datetime.now(timezone.utc)
5114 records: Final = []
5115 for key in keys:
5116 key_payload = key.model_dump()
5117 deleted_record = LiteLLM_DeletedVerificationToken.model_validate(
5118 {
5119 **key_payload,
5120 "deleted_at": deleted_at,
5121 "deleted_by": user_api_key_dict.user_id,
5122 "deleted_by_api_key": user_api_key_dict.api_key,
5123 "litellm_changed_by": litellm_changed_by,
5124 }
5125 )
5126 record = dict[str, object](_as_object_dict(deleted_record.model_dump()))
5128 # Map org_id to organization_id (model uses org_id, but schema expects organization_id)
5129 org_id_value: object = record.pop("org_id", None)
5130 if org_id_value is not None:
5131 record["organization_id"] = org_id_value
5133 for json_field in [
5134 "aliases",
5135 "config",
5136 "permissions",
5137 "metadata",
5138 "model_spend",
5139 "model_max_budget",
5140 "budget_fallbacks",
5141 "router_settings",
5142 ]:
5143 if json_field in record and record[json_field] is not None:
5144 record[json_field] = json.dumps(record[json_field])
5146 for rel_key in (
5147 "litellm_budget_table",
5148 "litellm_organization_table",
5149 "object_permission",
5150 "id",
5151 "budget_limits",
5152 ):
5153 record.pop(rel_key, None)
5155 records.append(record)
5157 return records
5160async def _save_deleted_verification_token_records(
5161 records: Sequence[Mapping[str, object]],
5162 prisma_client: PrismaClient,
5163 tx: "Prisma | None" = None,
5164) -> None:
5165 """Save deleted verification token records to the database.
5167 ``tx`` runs the write on that transaction's connection instead of a fresh
5168 one, so a caller batching this with other writes gets one all-or-nothing
5169 commit.
5170 """
5171 if not records:
5172 return
5173 if tx is not None:
5174 await tx.litellm_deletedverificationtoken.create_many(data=records)
5175 return
5176 await _deleted_verification_token_table(prisma_client).create_many(data=records)
5179async def _persist_deleted_verification_tokens(
5180 keys: Sequence[LiteLLM_VerificationToken],
5181 prisma_client: PrismaClient,
5182 user_api_key_dict: UserAPIKeyAuth,
5183 litellm_changed_by: str | None = None,
5184 tx: "Prisma | None" = None,
5185) -> None:
5186 """Persist deleted verification token records by transforming and saving them."""
5187 records: Final = _transform_verification_tokens_to_deleted_records(
5188 keys=keys,
5189 user_api_key_dict=user_api_key_dict,
5190 litellm_changed_by=litellm_changed_by,
5191 )
5192 await _save_deleted_verification_token_records(
5193 records=records,
5194 prisma_client=prisma_client,
5195 tx=tx,
5196 )
5199async def delete_key_aliases(
5200 key_aliases: list[str],
5201 user_api_key_cache: UserApiKeyCache,
5202 prisma_client: PrismaClient,
5203 user_api_key_dict: UserAPIKeyAuth,
5204 litellm_changed_by: str | None = None,
5205) -> tuple[dict | None, list[LiteLLM_VerificationToken]]:
5206 _keys_being_deleted: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_many(
5207 where={"key_alias": {"in": key_aliases}}
5208 )
5210 tokens: Final = [key.token for key in _keys_being_deleted]
5211 return await delete_verification_tokens(
5212 tokens=tokens,
5213 user_api_key_cache=user_api_key_cache,
5214 user_api_key_dict=user_api_key_dict,
5215 litellm_changed_by=litellm_changed_by,
5216 )
5219async def _rotate_master_key(
5220 prisma_client: PrismaClient,
5221 user_api_key_dict: UserAPIKeyAuth,
5222 current_master_key: str,
5223 new_master_key: str,
5224) -> None:
5225 """
5226 Rotate the master key
5228 1. Get the values from the DB
5229 - Get models from DB
5230 - Get config from DB
5231 2. Decrypt the values
5232 - ModelTable
5233 - [{"model_name": "str", "litellm_params": {}}]
5234 - ConfigTable
5235 3. Encrypt the values with the new master key
5236 4. Update the values in the DB
5237 """
5238 import prisma
5240 from litellm.proxy.proxy_server import proxy_config
5242 try:
5243 models: list | None = cast( # cast-ok: find_many returns a real list, which TableActions widens to Sequence
5244 "list[object]", await _prisma_table(ModelRepository(prisma_client)).find_many()
5245 )
5246 except Exception:
5247 models = None
5248 # 2. process model table
5249 if models:
5250 decrypted_models: Final = proxy_config.decrypt_model_list_from_db(new_models=models)
5251 verbose_proxy_logger.debug("ABLE TO DECRYPT MODELS - len(decrypted_models): %s", len(decrypted_models))
5252 reencrypted_models: Final = tuple(
5253 [
5254 reencrypted
5255 for model in decrypted_models
5256 if (
5257 reencrypted := await _add_model_to_db(
5258 model_params=Deployment(**model),
5259 user_api_key_dict=user_api_key_dict,
5260 prisma_client=prisma_client,
5261 new_encryption_key=new_master_key,
5262 should_create_model_in_db=False,
5263 )
5264 )
5265 ]
5266 )
5267 verbose_proxy_logger.debug("Re-encrypting litellm_params on %s model rows", len(reencrypted_models))
5268 async with prisma_client.db.tx(timeout=timedelta(minutes=2)) as tx_ctx:
5269 tx: Final[_TxTables] = tx_ctx
5270 for reencrypted_model in reencrypted_models:
5271 await tx.litellm_proxymodeltable.update_many(
5272 data=_ModelParamsUpdate(litellm_params=prisma.Json(reencrypted_model.litellm_params)),
5273 where=_ModelRowWhere(model_id=reencrypted_model.model_id),
5274 )
5275 await publish_config_change(redis_cache=coordination_redis_cache(), object_type="litellm_proxymodeltable")
5276 # 3. process config table
5277 try:
5278 config = await _config_table(prisma_client).find_many()
5279 except Exception:
5280 config = None
5282 if config:
5283 """If environment_variables is found, decrypt it and encrypt it with the new master key"""
5284 environment_variables_dict: Mapping[str, str] | None = {}
5285 for c in config:
5286 if c.param_name == "environment_variables":
5287 environment_variables_dict = _env_vars_param_value(c)
5289 if environment_variables_dict:
5290 decrypted_env_vars: Final = proxy_config._decrypt_and_set_db_env_variables(
5291 environment_variables=dict[str, str](environment_variables_dict)
5292 )
5293 encrypted_env_vars: Final = proxy_config._encrypt_env_variables(
5294 environment_variables=decrypted_env_vars,
5295 new_encryption_key=new_master_key,
5296 )
5298 if encrypted_env_vars:
5299 await _config_table(prisma_client).update(
5300 where={"param_name": "environment_variables"},
5301 data={"param_value": prisma.Json(encrypted_env_vars)},
5302 )
5304 # 4. process MCP server table
5305 try:
5306 await rotate_mcp_server_credentials_master_key(
5307 prisma_client=prisma_client,
5308 touched_by=user_api_key_dict.user_id or LITELLM_PROXY_ADMIN_NAME,
5309 new_master_key=new_master_key,
5310 )
5311 except Exception as e:
5312 verbose_proxy_logger.warning("Failed to rotate MCP server credentials: %s", str(e))
5314 # 4b. process MCP user-scoped credentials table (BYOK + OAuth2 tokens)
5315 try:
5316 await rotate_mcp_user_credentials_master_key(
5317 prisma_client=prisma_client,
5318 new_master_key=new_master_key,
5319 )
5320 except Exception as e:
5321 verbose_proxy_logger.warning("Failed to rotate MCP user credentials: %s", str(e))
5323 # 4c. process MCP per-user environment variables table
5324 try:
5325 await rotate_mcp_user_env_vars_master_key(
5326 prisma_client=prisma_client,
5327 new_master_key=new_master_key,
5328 )
5329 except Exception as e:
5330 verbose_proxy_logger.warning("Failed to rotate MCP user env vars: %s", str(e))
5332 # 4d. process SSO identity assertion table (EMA subject tokens)
5333 try:
5334 await rotate_sso_identity_assertions_master_key(
5335 prisma_client=prisma_client,
5336 new_master_key=new_master_key,
5337 )
5338 except Exception as e: # noqa: BLE001 # one store's failure must not abort the master-key rotation
5339 verbose_proxy_logger.warning("Failed to rotate SSO identity assertions: %s", str(e))
5341 # 5. process credentials table
5342 try:
5343 credentials = await _credentials_table(prisma_client).find_many()
5344 except Exception:
5345 credentials = None
5346 if credentials:
5347 from litellm.proxy.credential_endpoints.endpoints import update_db_credential
5349 for cred in credentials:
5350 try:
5351 decrypted_cred = proxy_config.decrypt_credentials(cred)
5352 encrypted_cred = update_db_credential(
5353 db_credential=cred,
5354 updated_patch=decrypted_cred,
5355 new_encryption_key=new_master_key,
5356 )
5357 _cred_data = dict[str, object](_as_object_dict(encrypted_cred.model_dump(exclude_none=True)))
5358 if "credential_values" in _cred_data:
5359 _cred_data["credential_values"] = prisma.Json(_cred_data["credential_values"])
5360 if "credential_info" in _cred_data:
5361 _cred_data["credential_info"] = prisma.Json(_cred_data["credential_info"])
5362 await _credentials_table(prisma_client).update(
5363 where={"credential_name": cred.credential_name},
5364 data={
5365 **_cred_data,
5366 "updated_by": user_api_key_dict.user_id,
5367 },
5368 )
5369 except Exception as e:
5370 verbose_proxy_logger.error("Failed to re-encrypt credential %s: %s", cred.credential_name, e)
5371 # Continue with next credential instead of failing entire rotation
5372 continue
5373 verbose_proxy_logger.debug("Successfully re-encrypted %s credentials with new master key", len(credentials))
5376def _require_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> None:
5377 from litellm.proxy._types import CommonProxyErrors
5379 if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value: 5379 ↛ 5380line 5379 didn't jump to line 5380 because the condition on line 5379 was never true
5380 raise HTTPException(
5381 status_code=403,
5382 detail={"error": CommonProxyErrors.not_allowed_access.value},
5383 )
5386@router.post(
5387 "/credentials/migrate-encryption",
5388 tags=["credential management"],
5389 dependencies=[Depends(user_api_key_auth)],
5390)
5391async def migrate_encryption_endpoint(
5392 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
5393 dry_run: bool = Query(
5394 False,
5395 description="If true, scan and report without writing any changes.",
5396 ),
5397):
5398 """
5399 Re-encrypt all at-rest credentials into the AES-256-GCM (``v2:gcm:``) format.
5401 Admin only. Requires ``general_settings.encryption_algorithm: aes-256-gcm``.
5402 Idempotent and resumable — re-running skips already-migrated values. Pass
5403 ``dry_run=true`` for a non-mutating scan (equivalent to ``--check``).
5404 """
5405 from litellm.proxy._types import CommonProxyErrors
5406 from litellm.proxy.management_endpoints.credential_migration import (
5407 migrate_encryption,
5408 )
5409 from litellm.proxy.proxy_server import prisma_client
5411 _require_proxy_admin(user_api_key_dict)
5412 if prisma_client is None: 5412 ↛ 5413line 5412 didn't jump to line 5413 because the condition on line 5412 was never true
5413 raise HTTPException(
5414 status_code=500,
5415 detail={"error": CommonProxyErrors.db_not_connected_error.value},
5416 )
5418 report: Final = await migrate_encryption(
5419 prisma_client=prisma_client,
5420 user_api_key_dict=user_api_key_dict,
5421 dry_run=dry_run,
5422 )
5423 return {"status": "success", "dry_run": dry_run, "report": report.as_dict()}
5426@router.get(
5427 "/credentials/migrate-encryption/check",
5428 tags=["credential management"],
5429 dependencies=[Depends(user_api_key_auth)],
5430)
5431async def check_encryption_endpoint(
5432 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
5433):
5434 """
5435 Read-only residual scan for compliance attestation. Reports how many at-rest
5436 values are still in the legacy format. ``residual_legacy == 0`` attests no
5437 legacy ciphertext remains. Admin only; performs no writes.
5438 """
5439 from litellm.proxy._types import CommonProxyErrors
5440 from litellm.proxy.management_endpoints.credential_migration import (
5441 check_encryption,
5442 )
5443 from litellm.proxy.proxy_server import prisma_client
5445 _require_proxy_admin(user_api_key_dict)
5446 if prisma_client is None: 5446 ↛ 5447line 5446 didn't jump to line 5447 because the condition on line 5446 was never true
5447 raise HTTPException(
5448 status_code=500,
5449 detail={"error": CommonProxyErrors.db_not_connected_error.value},
5450 )
5452 report: Final = await check_encryption(prisma_client=prisma_client)
5453 return {"status": "success", "report": report.as_dict()}
5456async def get_new_token(data: RegenerateKeyRequest | None) -> str:
5457 if data and data.new_key is not None:
5458 # Reject custom key values if disabled by admin
5459 await _check_custom_key_allowed(data.new_key)
5460 if not data.new_key.startswith("sk-"):
5461 raise HTTPException(
5462 status_code=status.HTTP_400_BAD_REQUEST,
5463 detail={
5464 "error": "New key must start with 'sk-'. This is to distinguish a key hash (used by litellm for logging / internal logic) from the actual key."
5465 },
5466 )
5467 if len(data.new_key) < MINIMUM_CUSTOM_KEY_LENGTH:
5468 raise HTTPException(
5469 status_code=status.HTTP_400_BAD_REQUEST,
5470 detail={"error": f"New key must be at least {MINIMUM_CUSTOM_KEY_LENGTH} characters long."},
5471 )
5472 new_token = data.new_key
5473 else:
5474 new_token = f"sk-{secrets.token_urlsafe(LENGTH_OF_LITELLM_GENERATED_KEY)}"
5475 return new_token
5478async def _insert_deprecated_key(
5479 prisma_client: "PrismaClient",
5480 old_token_hash: str,
5481 new_token_hash: str,
5482 grace_period: str | None,
5483) -> None:
5484 """
5485 Insert old key into deprecated table so it remains valid during grace period.
5487 Uses upsert to handle concurrent rotations gracefully.
5489 Parameters:
5490 prisma_client: DB client
5491 old_token_hash: Hash of the old key being rotated out
5492 new_token_hash: Hash of the new replacement key
5493 grace_period: Duration string (e.g. "24h", "2d") or None/empty for immediate revoke
5494 """
5495 grace_period_value: Final = grace_period or os.getenv("LITELLM_KEY_ROTATION_GRACE_PERIOD", "")
5496 if not grace_period_value:
5497 return
5499 try:
5500 grace_seconds: Final = duration_in_seconds(grace_period_value)
5501 except ValueError:
5502 verbose_proxy_logger.warning(
5503 "Invalid grace_period format: %s. Expected format like '24h', '2d'.",
5504 grace_period_value,
5505 )
5506 return
5508 if grace_seconds <= 0:
5509 return
5511 try:
5512 revoke_at: Final = datetime.now(timezone.utc) + timedelta(seconds=grace_seconds)
5513 await _deprecated_verification_token_table(prisma_client).upsert(
5514 where={"token": old_token_hash},
5515 data={
5516 "create": {
5517 "token": old_token_hash,
5518 "active_token_id": new_token_hash,
5519 "revoke_at": revoke_at,
5520 },
5521 "update": {
5522 "active_token_id": new_token_hash,
5523 "revoke_at": revoke_at,
5524 },
5525 },
5526 )
5527 verbose_proxy_logger.debug(
5528 "Deprecated key retained for %s (revoke_at: %s)",
5529 grace_period_value,
5530 revoke_at,
5531 )
5532 except Exception as deprecated_err:
5533 verbose_proxy_logger.warning(
5534 "Failed to insert deprecated key for grace period: %s",
5535 deprecated_err,
5536 )
5539async def _execute_virtual_key_regeneration(
5540 *,
5541 prisma_client: PrismaClient,
5542 llm_router: Router | None = None,
5543 key_in_db: LiteLLM_VerificationToken,
5544 hashed_api_key: str,
5545 key: str,
5546 data: RegenerateKeyRequest | None,
5547 user_api_key_dict: UserAPIKeyAuth,
5548 litellm_changed_by: str | None,
5549 user_api_key_cache: UserApiKeyCache,
5550 proxy_logging_obj: ProxyLogging,
5551) -> GenerateKeyResponse:
5552 """Generate new token, update DB, invalidate cache, and return response."""
5553 from litellm.proxy import proxy_server
5554 from litellm.proxy.proxy_server import hash_token
5556 # Mirror the /key/update ownership rebind guard. See helper docstring.
5557 _validate_caller_can_change_key_ownership(
5558 data=data,
5559 existing_key_row=key_in_db,
5560 user_api_key_dict=user_api_key_dict,
5561 )
5563 # Apply the same membership rule used on /key/update: when the caller
5564 # asks to point the regenerated key at a different organization_id,
5565 # require they are a member of (or proxy admin over) the target org.
5566 if data is not None and data.organization_id is not None:
5567 _existing_org_id: Final = getattr(key_in_db, "organization_id", None)
5568 _is_proxy_admin: Final = (
5569 user_api_key_dict.user_role is not None
5570 and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
5571 )
5572 if data.organization_id != _existing_org_id and not _is_proxy_admin:
5573 await _validate_caller_can_assign_key_org(
5574 user_api_key_dict=user_api_key_dict,
5575 organization_id=data.organization_id,
5576 prisma_client=prisma_client,
5577 )
5579 if data is not None:
5580 _existing_key_metadata: Final = getattr(key_in_db, "metadata", None)
5581 enforce_output_token_estimates_are_admin_only(
5582 data=data,
5583 existing_metadata=_existing_key_metadata if isinstance(_existing_key_metadata, dict) else None,
5584 user_api_key_dict=user_api_key_dict,
5585 entity="key",
5586 )
5587 enforce_batch_enqueued_token_limit_is_admin_only(
5588 data=data,
5589 existing_metadata=_existing_key_metadata if isinstance(_existing_key_metadata, dict) else None,
5590 user_api_key_dict=user_api_key_dict,
5591 entity="key",
5592 )
5593 await _validate_end_user_budget_id_change(
5594 requested_budget_id=_requested_end_user_budget_id(data),
5595 existing_budget_id=get_key_end_user_budget_id(
5596 _existing_key_metadata if isinstance(_existing_key_metadata, dict) else None
5597 ),
5598 user_api_key_dict=user_api_key_dict,
5599 prisma_client=prisma_client,
5600 )
5602 new_token: Final = await get_new_token(data=data)
5603 new_token_hash: Final = hash_token(new_token)
5604 new_token_key_name: Final = abbreviate_api_key(api_key=new_token)
5605 update_data = {"token": new_token_hash, "key_name": new_token_key_name}
5607 non_default_values = {}
5608 if data is not None:
5609 update_request: Final = _regenerate_request_as_update_request(key=hashed_api_key, data=data)
5610 if update_request is not None:
5611 await _enforce_custom_key_update_policy(hook=_custom_key_update_hook(proxy_server), data=update_request)
5612 # Enforce upperbound key params on regenerate (don't fill defaults)
5613 _enforce_upperbound_key_params(data, fill_defaults=False)
5614 non_default_values = await prepare_key_update_data(
5615 data=data, existing_key_row=key_in_db, prisma_client=prisma_client, llm_router=llm_router
5616 )
5617 # Only validate key_alias format if it's actually being changed
5618 new_key_alias: Final = non_default_values.get("key_alias")
5619 if new_key_alias != key_in_db.key_alias:
5620 _validate_key_alias_format(key_alias=new_key_alias)
5621 verbose_proxy_logger.debug("non_default_values: %s", non_default_values)
5622 await _enforce_custom_key_policy(
5623 hook=_custom_key_policy_hook(proxy_server),
5624 build_policy_request=lambda: _update_policy_request(
5625 operation="regenerate",
5626 existing_key_row=key_in_db,
5627 non_default_values=non_default_values,
5628 request=data if data is not None else RegenerateKeyRequest(),
5629 ),
5630 )
5631 update_values: Final = await _handle_update_object_permission(
5632 data_json=non_default_values,
5633 existing_key_row=key_in_db,
5634 prisma_client=prisma_client,
5635 )
5636 update_data.update(update_values)
5637 jsonified_update_data: Final[Mapping[str, object]] = prisma_client.jsonify_object(data=update_data)
5639 # Snapshot before the token update: the FK cascade rewrites mapping rows to the new hash,
5640 # but their cached jwt_key_mapping entries still point at the old token (LIT-5379).
5641 jwt_mapping_cache_keys: Final = await get_jwt_key_mapping_cache_keys_for_token(
5642 hashed_token=hashed_api_key,
5643 prisma_client=prisma_client,
5644 )
5646 await _persist_deleted_verification_tokens(
5647 keys=[key_in_db],
5648 prisma_client=prisma_client,
5649 user_api_key_dict=user_api_key_dict,
5650 litellm_changed_by=litellm_changed_by,
5651 )
5653 # If grace period set, insert deprecated key so old key remains valid
5654 await _insert_deprecated_key(
5655 prisma_client=prisma_client,
5656 old_token_hash=hashed_api_key,
5657 new_token_hash=new_token_hash,
5658 grace_period=data.grace_period if data else None,
5659 )
5661 updated_token: Final[LiteLLM_VerificationToken | None] = await _prisma_table(
5662 VerificationTokenRepository(prisma_client)
5663 ).update(
5664 where={"token": hashed_api_key},
5665 data=with_settings_updated_at(jsonified_update_data),
5666 )
5667 updated_token_dict: Final[dict[str, object]] = dict(updated_token) if updated_token is not None else {}
5668 updated_token_dict["key"] = new_token
5669 updated_token_dict["token_id"] = updated_token_dict.pop("token")
5671 await invalidate_cached_object_permissions(
5672 object_permission_ids=(
5673 key_in_db.object_permission_id,
5674 non_default_values.get("object_permission_id"),
5675 ),
5676 user_api_key_cache=user_api_key_cache,
5677 )
5678 if hashed_api_key or key:
5679 await _delete_cache_key_object(
5680 hashed_token=_hash_token_if_needed(key),
5681 user_api_key_cache=user_api_key_cache,
5682 proxy_logging_obj=proxy_logging_obj,
5683 )
5685 await evict_and_broadcast(cache_keys=jwt_mapping_cache_keys, user_api_key_cache=user_api_key_cache)
5687 # After credential invalidation, so a failure here can never keep the old key alive.
5688 await sync_key_regeneration_access_group_membership(
5689 prisma_client=prisma_client,
5690 previous_key_token=hashed_api_key,
5691 new_key_token=new_token_hash,
5692 data=data,
5693 existing_key_row=key_in_db,
5694 )
5696 response: Final = GenerateKeyResponse.model_validate(updated_token_dict)
5697 asyncio.create_task(
5698 KeyManagementEventHooks.async_key_rotated_hook(
5699 data=data,
5700 existing_key_row=key_in_db,
5701 response=response,
5702 user_api_key_dict=user_api_key_dict,
5703 litellm_changed_by=litellm_changed_by,
5704 )
5705 )
5706 return response
5709def _check_regenerate_guardrail_opt_out(
5710 data: RegenerateKeyRequest | None,
5711 existing_metadata: Mapping[str, object] | None,
5712 user_api_key_dict: UserAPIKeyAuth,
5713) -> None:
5714 if data is None:
5715 return
5716 _check_disable_global_guardrails_caller_permission(
5717 data.disable_global_guardrails,
5718 data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict
5719 user_api_key_dict,
5720 existing_metadata=existing_metadata,
5721 )
5724@router.post(
5725 "/key/{key:path}/regenerate",
5726 tags=["key management"],
5727 dependencies=[Depends(user_api_key_auth)],
5728)
5729@router.post(
5730 "/key/regenerate",
5731 tags=["key management"],
5732 dependencies=[Depends(user_api_key_auth)],
5733)
5734@management_endpoint_wrapper
5735async def regenerate_key_fn(
5736 key: str | None = None,
5737 data: RegenerateKeyRequest | None = None,
5738 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
5739 litellm_changed_by: str | None = Header(
5740 None,
5741 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
5742 ),
5743) -> GenerateKeyResponse | None:
5744 """
5745 Regenerate an existing API key while optionally updating its parameters.
5747 Parameters:
5748 - key: str (path parameter) - The key to regenerate
5749 - data: Optional[RegenerateKeyRequest] - Request body containing optional parameters to update
5750 - key: Optional[str] - The key to regenerate.
5751 - new_master_key: Optional[str] - The new master key to use, if key is the master key.
5752 - new_key: Optional[str] - The new key to use, if key is not the master key. Must start with 'sk-' and be at least 16 characters long. If both set, new_master_key will be used.
5753 - key_alias: Optional[str] - User-friendly key alias
5754 - user_id: Optional[str] - User ID associated with key
5755 - team_id: Optional[str] - Team ID associated with key
5756 - models: Optional[list] - Model_name's a user is allowed to call
5757 - tags: Optional[List[str]] - Tags for organizing keys (Enterprise only)
5758 - spend: Optional[float] - Amount spent by key
5759 - max_budget: Optional[float] - Max budget for key
5760 - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}
5761 - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
5762 - budget_duration: Optional[str] - Budget reset period ("30d", "1h", etc.)
5763 - soft_budget: Optional[float] - Soft budget limit (warning vs. hard stop). Will trigger a slack alert when this soft budget is reached.
5764 - max_parallel_requests: Optional[int] - Rate limit for parallel requests
5765 - metadata: Optional[dict] - Metadata for key. Example {"team": "core-infra", "app": "app2"}
5766 - tpm_limit: Optional[int] - Tokens per minute limit
5767 - rpm_limit: Optional[int] - Requests per minute limit
5768 - model_rpm_limit: Optional[dict] - Model-specific RPM limits {"gpt-4": 100, "claude-v1": 200}
5769 - model_tpm_limit: Optional[dict] - Model-specific TPM limits {"gpt-4": 100000, "claude-v1": 200000}
5770 - allowed_cache_controls: Optional[list] - List of allowed cache control values
5771 - duration: Optional[str] - Key validity duration ("30d", "1h", etc.)
5772 - permissions: Optional[dict] - Key-specific permissions
5773 - guardrails: Optional[List[str]] - List of active guardrails for the key
5774 - blocked: Optional[bool] - Whether the key is blocked
5775 - grace_period: Optional[str] - Duration to keep old key valid after rotation (e.g. "24h", "2d"). Omitted = immediate revoke. Env: LITELLM_KEY_ROTATION_GRACE_PERIOD
5778 Returns:
5779 - GenerateKeyResponse containing the new key and its updated parameters
5781 Example:
5782 ```bash
5783 curl --location --request POST 'http://localhost:4000/key/sk-1234/regenerate' \
5784 --header 'Authorization: Bearer sk-1234' \
5785 --header 'Content-Type: application/json' \
5786 --data-raw '{
5787 "max_budget": 100,
5788 "metadata": {"team": "core-infra"},
5789 "models": ["gpt-4", "gpt-3.5-turbo"]
5790 }'
5791 ```
5793 Note: This is an Enterprise feature. It requires a premium license to use.
5794 """
5795 try:
5796 from litellm.proxy.proxy_server import (
5797 hash_token,
5798 llm_router,
5799 master_key,
5800 premium_user,
5801 prisma_client,
5802 proxy_logging_obj,
5803 user_api_key_cache,
5804 )
5806 if data is not None:
5807 _check_allowed_routes_caller_permission(
5808 allowed_routes=data.allowed_routes,
5809 user_api_key_dict=user_api_key_dict,
5810 allowed_routes_was_provided="allowed_routes" in data.model_fields_set,
5811 )
5812 _check_passthrough_routes_caller_permission(
5813 data=data,
5814 user_api_key_dict=user_api_key_dict,
5815 )
5816 _check_permissions_caller_permission(
5817 data=data,
5818 user_api_key_dict=user_api_key_dict,
5819 )
5820 # Mirror /key/generate's post-handle_key_type recheck so a
5821 # non-admin can't elevate via a key_type preset that the
5822 # regenerate flow would otherwise carry through unchecked.
5823 # The empty dict is intentional — `handle_key_type` is reused
5824 # purely as a side-effect-free lookup of the preset bucket, not
5825 # to mutate an existing `data_json`. Do not pass a real
5826 # `data_json` here; that path would write the derived routes
5827 # into the DB update payload and is owned by
5828 # `_common_key_generation_helper`.
5829 _check_allowed_routes_caller_permission(
5830 allowed_routes=handle_key_type(data, {}).get("allowed_routes"),
5831 user_api_key_dict=user_api_key_dict,
5832 allow_safe_presets=True,
5833 )
5835 # Premium-gate bypass for master-key rotation must verify the
5836 # caller actually holds the master key, not just that the request
5837 # body has a ``new_master_key`` field. A presence-only check let
5838 # any non-premium caller skip the enterprise gate by sending any
5839 # value in that field.
5840 regenerate_target_key: Final = data.key if data and data.key else key
5841 is_master_key_regeneration: Final = (
5842 data is not None
5843 and data.new_master_key is not None
5844 and _is_master_key(api_key=regenerate_target_key, _master_key=master_key)
5845 )
5847 if ( 5847 ↛ 5855line 5847 didn't jump to line 5855 because the condition on line 5847 was always true
5848 premium_user is not True and not is_master_key_regeneration
5849 ): # allow master key regeneration for non-premium users
5850 raise ValueError(
5851 f"Regenerating Virtual Keys is an Enterprise feature, {CommonProxyErrors.not_premium_user.value}"
5852 )
5854 # Check if key exists, raise exception if key is not in the DB
5855 key = data.key if data and data.key else key
5856 if not key:
5857 raise HTTPException(status_code=400, detail={"error": "No key passed in."})
5858 ### 1. Create New copy that is duplicate of existing key
5859 ######################################################################
5861 # create duplicate of existing key
5862 # set token = new token generated
5863 # insert new token in DB
5865 # create hash of token
5866 if prisma_client is None:
5867 raise HTTPException(
5868 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
5869 detail={"error": "DB not connected. prisma_client is None"},
5870 )
5872 _is_master_key_valid: Final = _is_master_key(api_key=key, _master_key=master_key)
5874 if master_key is not None and data and _is_master_key_valid:
5875 if data.new_master_key is None:
5876 raise HTTPException(
5877 status_code=status.HTTP_400_BAD_REQUEST,
5878 detail={"error": "New master key is required."},
5879 )
5880 await _rotate_master_key(
5881 prisma_client=prisma_client,
5882 user_api_key_dict=user_api_key_dict,
5883 current_master_key=master_key,
5884 new_master_key=data.new_master_key,
5885 )
5886 return GenerateKeyResponse(
5887 key=data.new_master_key,
5888 token=data.new_master_key,
5889 key_name=data.new_master_key,
5890 expires=None,
5891 )
5893 if "sk" not in key:
5894 hashed_api_key = key
5895 else:
5896 hashed_api_key = hash_token(key)
5898 _key_in_db: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
5899 where={"token": hashed_api_key},
5900 )
5901 if _key_in_db is None:
5902 raise HTTPException(
5903 status_code=status.HTTP_404_NOT_FOUND,
5904 detail={"error": f"Key {key} not found."},
5905 )
5907 _check_regenerate_guardrail_opt_out(
5908 data,
5909 _key_in_db.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict
5910 user_api_key_dict,
5911 )
5913 # check if user has permission to regenerate key
5914 await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
5915 user_api_key_dict=user_api_key_dict,
5916 route=KeyManagementRoutes.KEY_REGENERATE,
5917 prisma_client=prisma_client,
5918 existing_key_row=_key_in_db,
5919 user_api_key_cache=user_api_key_cache,
5920 )
5922 # check if user has ownership permission to regenerate key
5923 if not await can_modify_verification_token(
5924 key_info=_key_in_db,
5925 user_api_key_cache=user_api_key_cache,
5926 user_api_key_dict=user_api_key_dict,
5927 prisma_client=prisma_client,
5928 ):
5929 raise HTTPException(
5930 status_code=status.HTTP_403_FORBIDDEN,
5931 detail={"error": "You are not authorized to regenerate this key"},
5932 )
5934 if data is not None and (data.access_group_ids or data.object_permission is not None):
5935 regenerate_team_table: LiteLLM_TeamTableCachedObj | None = None
5936 if _key_in_db.team_id is not None:
5937 regenerate_team_table = await get_team_object(
5938 team_id=_key_in_db.team_id,
5939 prisma_client=prisma_client,
5940 user_api_key_cache=user_api_key_cache,
5941 check_db_only=True,
5942 )
5943 _regen_is_proxy_admin: Final = user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
5944 TeamMemberPermissionChecks.enforce_member_can_assign_access_groups(
5945 user_api_key_dict=user_api_key_dict,
5946 team_table=regenerate_team_table,
5947 access_group_ids=data.access_group_ids,
5948 )
5949 _regen_object_permission_dict = _object_permission_to_dict(data.object_permission)
5950 normalized_object_permission: Final = await validate_key_mcp_servers_against_team(
5951 object_permission=_regen_object_permission_dict,
5952 team_obj=regenerate_team_table,
5953 prisma_client=prisma_client,
5954 is_proxy_admin=_regen_is_proxy_admin,
5955 )
5956 if normalized_object_permission is not None:
5957 data.object_permission = LiteLLM_ObjectPermissionBase(**normalized_object_permission)
5958 _regen_object_permission_dict = normalized_object_permission
5959 await validate_key_search_tools_against_team(
5960 object_permission=_regen_object_permission_dict,
5961 team_obj=regenerate_team_table,
5962 is_proxy_admin=_regen_is_proxy_admin,
5963 )
5964 await validate_key_vector_stores_against_team(
5965 object_permission=_regen_object_permission_dict,
5966 team_obj=regenerate_team_table,
5967 is_proxy_admin=_regen_is_proxy_admin,
5968 )
5970 verbose_proxy_logger.info(
5971 "Key regeneration requested: key_alias=%s",
5972 getattr(_key_in_db, "key_alias", None),
5973 )
5974 verbose_proxy_logger.debug("key_in_db: %s", _key_in_db)
5976 # Normalize litellm_changed_by: if it's a Header object or not a string, convert to None
5977 if litellm_changed_by is not None and not isinstance(litellm_changed_by, str):
5978 litellm_changed_by = None
5980 return await _execute_virtual_key_regeneration(
5981 prisma_client=prisma_client,
5982 llm_router=llm_router,
5983 key_in_db=_key_in_db,
5984 hashed_api_key=hashed_api_key,
5985 key=key,
5986 data=data,
5987 user_api_key_dict=user_api_key_dict,
5988 litellm_changed_by=litellm_changed_by,
5989 user_api_key_cache=user_api_key_cache,
5990 proxy_logging_obj=proxy_logging_obj,
5991 )
5992 except Exception as e:
5993 verbose_proxy_logger.exception("Error regenerating key: %s", e)
5994 raise handle_exception_on_proxy(e)
5997async def _check_proxy_or_team_admin_for_key(
5998 key_in_db: LiteLLM_VerificationToken,
5999 user_api_key_dict: UserAPIKeyAuth,
6000 prisma_client: PrismaClient,
6001 user_api_key_cache: UserApiKeyCache,
6002) -> None:
6003 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 6003 ↛ 6006line 6003 didn't jump to line 6006 because the condition on line 6003 was always true
6004 return
6006 if key_in_db.team_id is not None:
6007 team_table: Final = await get_team_object(
6008 team_id=key_in_db.team_id,
6009 prisma_client=prisma_client,
6010 user_api_key_cache=user_api_key_cache,
6011 check_db_only=True,
6012 )
6013 if team_table is not None:
6014 if _is_user_team_admin(
6015 user_api_key_dict=user_api_key_dict,
6016 team_obj=team_table,
6017 ):
6018 return
6020 raise HTTPException(
6021 status_code=status.HTTP_403_FORBIDDEN,
6022 detail={"error": "You must be a proxy admin or team admin to reset key spend"},
6023 )
6026def _validate_reset_spend_value(reset_to: object, key_in_db: LiteLLM_VerificationToken) -> float:
6027 if not isinstance(reset_to, (int, float)): 6027 ↛ 6028line 6027 didn't jump to line 6028 because the condition on line 6027 was never true
6028 raise HTTPException(
6029 status_code=status.HTTP_400_BAD_REQUEST,
6030 detail={"error": "reset_to must be a float"},
6031 )
6033 reset_to = float(reset_to)
6035 if reset_to < 0:
6036 raise HTTPException(
6037 status_code=status.HTTP_400_BAD_REQUEST,
6038 detail={"error": "reset_to must be >= 0"},
6039 )
6041 current_spend: Final = key_in_db.spend or 0.0
6042 if reset_to > current_spend:
6043 raise HTTPException(
6044 status_code=status.HTTP_400_BAD_REQUEST,
6045 detail={"error": f"reset_to ({reset_to}) must be <= current spend ({current_spend})"},
6046 )
6048 max_budget = key_in_db.max_budget
6049 if key_in_db.litellm_budget_table is not None: 6049 ↛ 6050line 6049 didn't jump to line 6050 because the condition on line 6049 was never true
6050 budget_max_budget: Final[float | None] = getattr(key_in_db.litellm_budget_table, "max_budget", None)
6051 if budget_max_budget is not None:
6052 if max_budget is None or budget_max_budget < max_budget:
6053 max_budget = budget_max_budget
6055 if max_budget is not None and reset_to > max_budget: 6055 ↛ 6056line 6055 didn't jump to line 6056 because the condition on line 6055 was never true
6056 raise HTTPException(
6057 status_code=status.HTTP_400_BAD_REQUEST,
6058 detail={"error": f"reset_to ({reset_to}) must be <= budget ({max_budget})"},
6059 )
6061 return reset_to
6064async def _set_spend_counter_with_floor_and_broadcast(counter_key: str, value: float) -> None:
6065 """
6066 Set a Redis-backed spend counter to `value`, mirror it into the short-lived
6067 spend_db_floor marker `_authoritative_floor_spend` reads, and broadcast both
6068 to every worker (LIT-3803 pattern: setting, not deleting, means a worker's
6069 own self-delivered broadcast still carries the reset value forward).
6071 Without the floor marker, `_authoritative_floor_spend` can re-derive a
6072 stale, pre-reset value from a marker another worker cached moments earlier
6073 and raise the just-reset counter right back up via `_repair_stale_spend_counter`.
6074 Without the broadcast, a worker that already cached the pre-reset key object
6075 or floor marker keeps enforcing against it until its own TTL expires.
6076 """
6077 from litellm.proxy.proxy_server import SPEND_DB_FLOOR_CACHE_TTL_SECONDS, spend_counter_cache
6079 spend_counter_cache.in_memory_cache.set_cache(key=counter_key, value=value, ttl=60)
6080 if spend_counter_cache.redis_cache is not None: 6080 ↛ 6081line 6080 didn't jump to line 6081 because the condition on line 6080 was never true
6081 try:
6082 await spend_counter_cache.redis_cache.async_set_cache(key=counter_key, value=value, ttl=60)
6083 except Exception as redis_err:
6084 verbose_proxy_logger.warning(
6085 "Failed to update spend counter %s in Redis: %s. "
6086 "Budget checks may use stale value until counter expires.",
6087 counter_key,
6088 redis_err,
6089 )
6091 floor_key: Final = f"spend_db_floor:{counter_key}"
6092 spend_counter_cache.in_memory_cache.set_cache(key=floor_key, value=value, ttl=SPEND_DB_FLOOR_CACHE_TTL_SECONDS)
6094 await publish_auth_cache_invalidation(cache_key=counter_key, new_value=value, ttl=60)
6095 await publish_auth_cache_invalidation(cache_key=floor_key, new_value=value, ttl=SPEND_DB_FLOOR_CACHE_TTL_SECONDS)
6098def _budget_limit_windows(budget_limits: Sequence[object] | str | None) -> tuple[Mapping[str, object], ...]:
6099 """Coerce a key's stored `budget_limits` into a tuple of plain window dicts.
6101 It is a DB Json column, so a caller reading it straight off `find_unique`
6102 gets an already-parsed list; one reading it off `json.dumps`'d text (or a
6103 raw SQL row) gets the string form. Either way each entry is a plain dict,
6104 except wherever a caller already validated the field through a pydantic
6105 model (e.g. `UserAPIKeyAuth.budget_limits`), which yields `BudgetLimitEntry`
6106 objects instead -- coerced here via `model_dump()`, matching
6107 `_set_budget_reset_at`'s identical coercion in team_endpoints.py.
6108 """
6109 if not budget_limits: 6109 ↛ 6111line 6109 didn't jump to line 6111 because the condition on line 6109 was always true
6110 return ()
6111 raw_windows: Final = json.loads(budget_limits) if isinstance(budget_limits, str) else budget_limits
6112 return tuple(raw_window if isinstance(raw_window, dict) else raw_window.model_dump() for raw_window in raw_windows)
6115def _advance_one_key_budget_window(window: Mapping[str, object]) -> Mapping[str, object]:
6116 """Restart one budget window from now, by advancing its `reset_at`.
6118 `window_start` is derived elsewhere as `reset_at - budget_duration`
6119 (`get_budget_window_start`), so `reset_at` must be set to `now +
6120 budget_duration` -- a window floating from THIS moment -- to make
6121 `window_start` land at `now` and exclude the historical spend that
6122 triggered the block. Reusing `get_budget_reset_time`/
6123 `ResetBudgetJob._reset_expired_window`'s calendar-standardized boundary
6124 (e.g. "next midnight") would not do that: for a "1d" window `next
6125 midnight - 1d` is simply the START of the calendar day already in
6126 progress, which still covers that spend. That reuse is only safe for the
6127 scheduled job, which runs right as `reset_at` naturally elapses, so the
6128 elapsed boundary it computes is already close to "now". A manual reset
6129 can happen at any point mid-window, so it needs the floating form
6130 instead. A window with no `budget_duration` is returned unchanged.
6131 """
6132 duration = window.get("budget_duration")
6133 if not isinstance(duration, str) or not duration:
6134 return window
6135 new_reset_at: Final = datetime.now(timezone.utc) + timedelta(seconds=duration_in_seconds(duration))
6136 return { # mutable-ok: this is the JSON payload persisted to budget_limits' Json column, which requires a plain dict
6137 **window,
6138 "reset_at": new_reset_at.isoformat(),
6139 }
6142async def _reset_key_budget_windows(
6143 prisma_client: PrismaClient,
6144 hashed_api_key: str,
6145 budget_limits: Sequence[object] | str | None,
6146) -> None:
6147 """Force-expire every one of a key's own `budget_limits` windows (extra
6148 time-windowed caps layered on top of the lifetime max_budget, e.g. a daily
6149 limit) so a manual spend reset also clears them, not just the lifetime
6150 counter.
6152 Persists the advanced `reset_at` boundaries BEFORE zeroing any window's
6153 Redis counter, not after: a window counter reading zero is only durable
6154 once every reader recomputing its floor from the DB sees the new
6155 boundary too (`get_current_spend` re-derives a window counter from real
6156 `LiteLLM_SpendLogs` rows inside `[window_start, now)` on every read below
6157 max_budget, see its `is_window` branch). Zeroing first would let a
6158 request racing the DB write compute `window_start` from the stale
6159 pre-reset boundary, re-sum the unchanged historical spend, and put the
6160 counter right back where it was before the write ever landed.
6161 """
6162 windows: Final = _budget_limit_windows(budget_limits)
6163 if not windows: 6163 ↛ 6166line 6163 didn't jump to line 6166 because the condition on line 6163 was always true
6164 return
6166 reset_windows: Final = tuple(_advance_one_key_budget_window(w) for w in windows)
6168 # prisma-client-py's typed update() takes plain dict literals for `where`/`data`; there is no
6169 # frozen-mapping equivalent to pass instead.
6170 reset_payload: Final = {"budget_limits": json.dumps(reset_windows, default=str)} # mutable-ok: prisma data kwarg
6171 await VerificationTokenRepository(prisma_client).table.update(
6172 where={"token": hashed_api_key}, # mutable-ok: prisma where kwarg
6173 data=reset_payload,
6174 )
6176 for window in reset_windows:
6177 duration = window.get("budget_duration")
6178 if isinstance(duration, str) and duration:
6179 counter_key = f"spend:key:{hashed_api_key}:window:{duration}"
6180 await _set_spend_counter_with_floor_and_broadcast(counter_key=counter_key, value=0.0)
6183@router.post(
6184 "/key/{key:path}/reset_spend",
6185 tags=["key management"],
6186 dependencies=[Depends(user_api_key_auth)],
6187)
6188@management_endpoint_wrapper
6189async def reset_key_spend_fn(
6190 key: str,
6191 data: ResetSpendRequest,
6192 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
6193 litellm_changed_by: str | None = Header(
6194 None,
6195 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
6196 ),
6197) -> dict[str, Any]:
6198 try:
6199 from litellm.proxy.proxy_server import (
6200 hash_token,
6201 prisma_client,
6202 proxy_logging_obj,
6203 user_api_key_cache,
6204 )
6206 if prisma_client is None: 6206 ↛ 6207line 6206 didn't jump to line 6207 because the condition on line 6206 was never true
6207 raise HTTPException(
6208 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
6209 detail={"error": "DB not connected. prisma_client is None"},
6210 )
6212 if "sk" not in key:
6213 hashed_api_key = key
6214 else:
6215 hashed_api_key = hash_token(key)
6217 _key_in_db: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
6218 where={"token": hashed_api_key},
6219 include={"litellm_budget_table": True},
6220 )
6221 if _key_in_db is None:
6222 raise HTTPException(
6223 status_code=status.HTTP_404_NOT_FOUND,
6224 detail={"error": f"Key {key} not found."},
6225 )
6227 current_spend: Final = _key_in_db.spend or 0.0
6228 reset_to: Final = _validate_reset_spend_value(data.reset_to, _key_in_db)
6230 await _check_proxy_or_team_admin_for_key(
6231 key_in_db=_key_in_db,
6232 user_api_key_dict=user_api_key_dict,
6233 prisma_client=prisma_client,
6234 user_api_key_cache=user_api_key_cache,
6235 )
6237 updated_key: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).update(
6238 where={"token": hashed_api_key},
6239 data={"spend": reset_to},
6240 )
6242 if updated_key is None: 6242 ↛ 6243line 6242 didn't jump to line 6243 because the condition on line 6242 was never true
6243 raise HTTPException(
6244 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
6245 detail={"error": "Failed to update key spend"},
6246 )
6248 # Reset the lifetime spend counter to the new value (not 0.0, so partial
6249 # resets are reflected correctly), and force-expire any of the key's own
6250 # budget_limits windows, so get_current_spend() returns the correct
6251 # amount for every enforcement check immediately instead of the stale
6252 # pre-reset value.
6253 _counter_key: Final = f"spend:key:{hashed_api_key}"
6254 await _set_spend_counter_with_floor_and_broadcast(counter_key=_counter_key, value=reset_to)
6255 await _reset_key_budget_windows(
6256 prisma_client=prisma_client,
6257 hashed_api_key=hashed_api_key,
6258 budget_limits=_key_in_db.budget_limits,
6259 )
6261 # Evicting the cached key object LAST (after every DB write above has
6262 # committed) matters: a request landing between an earlier eviction and
6263 # a later write would re-fetch and re-cache the pre-write row, pinning
6264 # that pod to the stale budget_limits/spend for the rest of its own
6265 # cache TTL even though the DB is already correct.
6266 await _delete_cache_key_object(
6267 hashed_token=hashed_api_key,
6268 user_api_key_cache=user_api_key_cache,
6269 proxy_logging_obj=proxy_logging_obj,
6270 )
6272 max_budget: Final = updated_key.max_budget
6273 budget_reset_at: Final = updated_key.budget_reset_at
6275 return {
6276 "key_hash": hashed_api_key,
6277 "spend": reset_to,
6278 "previous_spend": current_spend,
6279 "max_budget": max_budget,
6280 "budget_reset_at": budget_reset_at,
6281 }
6282 except HTTPException:
6283 raise
6284 except Exception as e:
6285 verbose_proxy_logger.exception("Error resetting key spend: %s", e)
6286 raise handle_exception_on_proxy(e)
6289async def validate_key_list_check(
6290 user_api_key_dict: UserAPIKeyAuth,
6291 user_id: str | None,
6292 team_id: str | None,
6293 organization_id: str | None,
6294 key_alias: str | None,
6295 key_hash: str | None,
6296 prisma_client: PrismaClient,
6297) -> LiteLLM_UserTable | None:
6298 if _user_has_admin_view(user_api_key_dict): 6298 ↛ 6301line 6298 didn't jump to line 6301 because the condition on line 6298 was always true
6299 return None
6301 if user_api_key_dict.user_id is None:
6302 raise ProxyException(
6303 message="You are not authorized to access this endpoint. No 'user_id' is associated with your API key.",
6304 type=ProxyErrorTypes.bad_request_error,
6305 param="user_id",
6306 code=status.HTTP_403_FORBIDDEN,
6307 )
6308 complete_user_info_db_obj: Final[BaseModel | None] = await _prisma_table(UserRepository(prisma_client)).find_unique(
6309 where={"user_id": user_api_key_dict.user_id},
6310 include={"organization_memberships": True},
6311 )
6313 if complete_user_info_db_obj is None:
6314 raise ProxyException(
6315 message="You are not authorized to access this endpoint. No 'user_id' is associated with your API key.",
6316 type=ProxyErrorTypes.bad_request_error,
6317 param="user_id",
6318 code=status.HTTP_403_FORBIDDEN,
6319 )
6321 complete_user_info: Final = LiteLLM_UserTable.model_validate(complete_user_info_db_obj.model_dump())
6323 # internal user can only see their own keys
6324 if user_id:
6325 if complete_user_info.user_id != user_id:
6326 raise ProxyException(
6327 message="You are not authorized to check another user's keys",
6328 type=ProxyErrorTypes.bad_request_error,
6329 param="user_id",
6330 code=status.HTTP_403_FORBIDDEN,
6331 )
6333 if team_id:
6334 if team_id not in complete_user_info.teams:
6335 raise ProxyException(
6336 message="You are not authorized to check this team's keys",
6337 type=ProxyErrorTypes.bad_request_error,
6338 param="team_id",
6339 code=status.HTTP_403_FORBIDDEN,
6340 )
6342 if organization_id:
6343 if complete_user_info.organization_memberships is None or organization_id not in [
6344 membership.organization_id for membership in complete_user_info.organization_memberships
6345 ]:
6346 raise ProxyException(
6347 message="You are not authorized to check this organization's keys",
6348 type=ProxyErrorTypes.bad_request_error,
6349 param="organization_id",
6350 code=status.HTTP_403_FORBIDDEN,
6351 )
6353 if key_hash:
6354 try:
6355 key_info: Final[LiteLLM_VerificationToken | None] = await _prisma_table(
6356 VerificationTokenRepository(prisma_client)
6357 ).find_unique(
6358 where={"token": key_hash},
6359 )
6360 except Exception:
6361 raise ProxyException(
6362 message="Key Hash not found.",
6363 type=ProxyErrorTypes.bad_request_error,
6364 param="key_hash",
6365 code=status.HTTP_403_FORBIDDEN,
6366 )
6367 if key_info is None:
6368 raise ProxyException(
6369 message="Key Hash not found.",
6370 type=ProxyErrorTypes.bad_request_error,
6371 param="key_hash",
6372 code=status.HTTP_403_FORBIDDEN,
6373 )
6374 can_user_query_key_info: Final = await _can_user_query_key_info(
6375 user_api_key_dict=user_api_key_dict,
6376 key=key_hash,
6377 key_info=key_info,
6378 )
6379 if not can_user_query_key_info:
6380 raise HTTPException(
6381 status_code=status.HTTP_403_FORBIDDEN,
6382 detail=f"You are not allowed to access this key's info. Your role={user_api_key_dict.user_role}",
6383 )
6384 return complete_user_info
6387async def _fetch_user_team_objects(
6388 complete_user_info: LiteLLM_UserTable | None,
6389 prisma_client: PrismaClient,
6390) -> list[LiteLLM_TeamTable]:
6391 """Fetch team objects for all teams a user belongs to (single DB query)."""
6392 if complete_user_info is None or not complete_user_info.teams: 6392 ↛ 6395line 6392 didn't jump to line 6395 because the condition on line 6392 was always true
6393 return []
6395 teams: Final[Sequence[BaseModel] | None] = cast( # cast-ok: the None guard below predates the non-optional seam
6396 "Sequence[BaseModel] | None",
6397 await TeamRepository(prisma_client).table.find_many(where={"team_id": {"in": complete_user_info.teams}}),
6398 )
6399 if teams is None:
6400 return []
6402 return [LiteLLM_TeamTable.model_validate(team.model_dump()) for team in teams]
6405def _get_admin_team_ids_from_objects(
6406 user_api_key_dict: UserAPIKeyAuth,
6407 team_objects: list[LiteLLM_TeamTable],
6408) -> list[str]:
6409 """Filter team objects to those where the user is an admin."""
6410 return [
6411 team.team_id for team in team_objects if _is_user_team_admin(user_api_key_dict=user_api_key_dict, team_obj=team)
6412 ]
6415def _get_team_ids_with_key_list_permission_from_objects(
6416 user_api_key_dict: UserAPIKeyAuth,
6417 team_objects: list[LiteLLM_TeamTable],
6418) -> list[str]:
6419 """Filter team objects to non-admin teams where the caller has /key/list
6420 permission via team_member_permissions. These teams should grant the
6421 caller full key visibility (same as a team admin), so other members'
6422 keys and service account keys (user_id=NULL) are returned."""
6423 return [
6424 team.team_id
6425 for team in team_objects
6426 if not _is_user_team_admin(user_api_key_dict=user_api_key_dict, team_obj=team)
6427 and _team_member_has_permission(
6428 user_api_key_dict=user_api_key_dict,
6429 team_obj=team,
6430 permission=KeyManagementRoutes.KEY_LIST.value,
6431 )
6432 ]
6435def _get_member_team_ids_from_objects(
6436 user_api_key_dict: UserAPIKeyAuth,
6437 team_objects: list[LiteLLM_TeamTable],
6438) -> list[str]:
6439 """Filter team objects to those where the user is a member (any role)."""
6440 return [
6441 team.team_id
6442 for team in team_objects
6443 if any(
6444 member.user_id is not None and member.user_id == user_api_key_dict.user_id
6445 for member in team.members_with_roles
6446 )
6447 ]
6450async def get_admin_team_ids(
6451 complete_user_info: LiteLLM_UserTable | None,
6452 user_api_key_dict: UserAPIKeyAuth,
6453 prisma_client: PrismaClient,
6454) -> list[str]:
6455 """Get all team IDs where the user is an admin."""
6456 team_objects: Final = await _fetch_user_team_objects(complete_user_info, prisma_client)
6457 return _get_admin_team_ids_from_objects(user_api_key_dict, team_objects)
6460async def get_member_team_ids(
6461 complete_user_info: LiteLLM_UserTable | None,
6462 user_api_key_dict: UserAPIKeyAuth,
6463 prisma_client: PrismaClient,
6464) -> list[str]:
6465 """
6466 Get all team IDs where the user is a member (any role, including admin).
6468 Used to determine which teams' service accounts (keys with user_id=NULL)
6469 a regular team member can see.
6470 """
6471 team_objects: Final = await _fetch_user_team_objects(complete_user_info, prisma_client)
6472 return _get_member_team_ids_from_objects(user_api_key_dict, team_objects)
6475VALID_EXPIRES_FILTER_VALUES: Final = frozenset({"active", "expired"})
6477KeyStatus = Literal["active", "expired", "revoked", "deleted"]
6478VALID_STATUS_FILTER_VALUES: Final[frozenset[KeyStatus]] = frozenset({"active", "expired", "revoked", "deleted"})
6481class _KeyStatusSource(BaseModel):
6482 blocked: bool | None = None
6483 expires: datetime | None = None
6486def _derive_key_status(row: Mapping[str, object], now: datetime) -> KeyStatus:
6487 source: Final = _KeyStatusSource.model_validate(row)
6488 if source.blocked is True:
6489 return "revoked"
6490 if source.expires is None: 6490 ↛ 6492line 6490 didn't jump to line 6492 because the condition on line 6490 was always true
6491 return "active"
6492 expires_utc: Final = source.expires if source.expires.tzinfo else source.expires.replace(tzinfo=timezone.utc)
6493 return "expired" if expires_utc < now else "active"
6496@router.get(
6497 "/key/list",
6498 tags=["key management"],
6499 dependencies=[Depends(user_api_key_auth)],
6500)
6501@management_endpoint_wrapper
6502async def list_keys(
6503 request: Request,
6504 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
6505 page: int = Query(1, description="Page number", ge=1),
6506 size: int = Query(10, description="Page size", ge=1, le=100),
6507 user_id: str | None = Query(
6508 None,
6509 description="Filter keys by user ID. Exact match by default; set substring_matching=true (admin only) for case-insensitive substring matching.",
6510 ),
6511 team_id: str | None = Query(None, description="Filter keys by team ID"),
6512 organization_id: str | None = Query(None, description="Filter keys by organization ID"),
6513 key_hash: str | None = Query(None, description="Filter keys by key hash"),
6514 key_alias: str | None = Query(
6515 None,
6516 description="Filter keys by key alias. Exact match by default; set substring_matching=true for case-insensitive substring matching.",
6517 ),
6518 search: str | None = Query(
6519 None,
6520 description="Combined search: matches keys whose token (key hash) equals the value OR whose key_alias contains it (case-insensitive).",
6521 ),
6522 return_full_object: bool = Query(False, description="Return full key object"),
6523 include_team_keys: bool = Query(False, description="Include all keys for teams that user is an admin of."),
6524 include_created_by_keys: bool = Query(False, description="Include keys created by the user"),
6525 sort_by: str | None = Query(
6526 default=None,
6527 description="Column to sort by (e.g. 'user_id', 'created_at', 'spend')",
6528 ),
6529 sort_order: str = Query(default="desc", description="Sort order ('asc' or 'desc')"),
6530 expand: list[str] | None = Query(None, description="Expand related objects (e.g. 'user')"),
6531 status: str | None = Query(
6532 None,
6533 description="Filter by status: 'active' (not blocked, not expired), 'expired' (not blocked, past expiry), 'revoked' (blocked) or 'deleted' (archived keys). Omit to return live keys regardless of status.",
6534 ),
6535 project_id: str | None = Query(None, description="Filter keys by project ID"),
6536 access_group_id: str | None = Query(None, description="Filter keys by access group ID"),
6537 agent_id: str | None = Query(None, description="Filter keys by agent ID"),
6538 substring_matching: bool = Query(
6539 False,
6540 description="If true, match key_alias (any caller) and user_id (proxy admins only) as case-insensitive substrings instead of exact values. Defaults to false: /key/list matched these exactly before substring search was added, and an exact user_id filter must never return another user's keys.",
6541 ),
6542 expires: str | None = Query(
6543 None,
6544 description="Filter keys by expiration. 'expired' returns keys whose expires is in the past; 'active' returns keys that never expire or expire in the future. Omit to return keys regardless of expiration.",
6545 ),
6546) -> KeyListResponseObject:
6547 """
6548 List all keys for a given user / team / organization.
6550 Parameters:
6551 expand: Optional[List[str]] - Expand related objects (e.g. 'user' to include user information)
6552 status: Optional[str] - Filter by status: "active", "expired", "revoked" (blocked) or "deleted".
6553 "deleted" reads the LiteLLM_DeletedVerificationToken archive; the other values partition the
6554 live key table, so every live key matches exactly one of them.
6556 Returns:
6557 {
6558 "keys": List[str] or List[UserAPIKeyAuth],
6559 "total_count": int,
6560 "current_page": int,
6561 "total_pages": int,
6562 }
6564 When expand includes "user", each key object will include a "user" field with the associated user object.
6565 Note: When expand=user is specified, full key objects are returned regardless of the return_full_object parameter.
6566 """
6567 try:
6568 from litellm.proxy.proxy_server import prisma_client
6570 verbose_proxy_logger.debug("Entering list_keys function")
6572 if prisma_client is None: 6572 ↛ 6573line 6572 didn't jump to line 6573 because the condition on line 6572 was never true
6573 verbose_proxy_logger.error("Database not connected")
6574 raise Exception("Database not connected")
6576 if status is not None and status not in VALID_STATUS_FILTER_VALUES:
6577 raise HTTPException(
6578 status_code=400,
6579 detail={"error": "Invalid status value. Supported: 'active', 'expired', 'revoked', 'deleted'."},
6580 )
6582 if isinstance(expires, str) and expires not in VALID_EXPIRES_FILTER_VALUES:
6583 raise HTTPException(
6584 status_code=400,
6585 detail={"error": "Invalid expires value. Supported: 'active', 'expired'."},
6586 )
6588 complete_user_info: Final = await validate_key_list_check(
6589 user_api_key_dict=user_api_key_dict,
6590 user_id=user_id,
6591 team_id=team_id,
6592 organization_id=organization_id,
6593 key_alias=key_alias,
6594 key_hash=key_hash,
6595 prisma_client=prisma_client,
6596 )
6598 # Fetch team objects once when needed for either admin or member filtering.
6599 # This avoids duplicate DB queries for the same team data.
6600 if include_team_keys or include_created_by_keys:
6601 team_objects = await _fetch_user_team_objects(
6602 complete_user_info=complete_user_info,
6603 prisma_client=prisma_client,
6604 )
6605 member_team_ids = _get_member_team_ids_from_objects(
6606 user_api_key_dict=user_api_key_dict,
6607 team_objects=team_objects,
6608 )
6609 else:
6610 team_objects = []
6611 member_team_ids = None
6613 if include_team_keys:
6614 admin_team_ids = _get_admin_team_ids_from_objects(
6615 user_api_key_dict=user_api_key_dict,
6616 team_objects=team_objects,
6617 )
6618 # Non-admin members with /key/list permission get full team-key
6619 # visibility for that team — matching the UI contract that
6620 # granting this permission lets them see all keys within the team.
6621 list_permission_team_ids: Final = _get_team_ids_with_key_list_permission_from_objects(
6622 user_api_key_dict=user_api_key_dict,
6623 team_objects=team_objects,
6624 )
6625 if list_permission_team_ids: 6625 ↛ 6626line 6625 didn't jump to line 6626 because the condition on line 6625 was never true
6626 admin_team_ids = list({*admin_team_ids, *list_permission_team_ids})
6627 else:
6628 admin_team_ids = None
6630 is_proxy_admin: Final = user_api_key_dict.user_role in [
6631 LitellmUserRoles.PROXY_ADMIN.value,
6632 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
6633 ]
6635 # Substring matching is opt-in. /key/list matched user_id and key_alias
6636 # exactly before substring search was added; auto-applying a substring
6637 # match to every admin call broke that contract and let a caller passing
6638 # an exact user_id (e.g. an integration scoping to one user with an admin
6639 # key) receive other users' keys (user_id="alice" -> "alice2"). Exact by
6640 # default restores the prior behavior; the dashboard opts in explicitly.
6641 use_substring_matching: Final = substring_matching and is_proxy_admin
6642 use_key_alias_substring_matching: Final = substring_matching
6644 # Admins may omit user_id to list all keys; non-admins are scoped to self.
6645 if not user_id and not is_proxy_admin: 6645 ↛ 6646line 6645 didn't jump to line 6646 because the condition on line 6645 was never true
6646 user_id = user_api_key_dict.user_id
6648 response: Final = await _list_key_helper(
6649 prisma_client=prisma_client,
6650 page=page,
6651 size=size,
6652 user_id=user_id,
6653 team_id=team_id,
6654 key_alias=key_alias,
6655 key_hash=key_hash,
6656 return_full_object=return_full_object,
6657 organization_id=organization_id,
6658 admin_team_ids=admin_team_ids,
6659 member_team_ids=member_team_ids,
6660 include_created_by_keys=include_created_by_keys,
6661 sort_by=sort_by,
6662 sort_order=sort_order,
6663 expand=expand,
6664 status=status,
6665 project_id=project_id,
6666 access_group_id=access_group_id,
6667 agent_id=agent_id,
6668 use_substring_matching=use_substring_matching,
6669 use_key_alias_substring_matching=use_key_alias_substring_matching,
6670 expires_filter=expires if isinstance(expires, str) else None,
6671 search=search,
6672 )
6674 verbose_proxy_logger.debug("Successfully prepared response")
6676 return response
6678 except Exception as e:
6679 verbose_proxy_logger.exception("Error in list_keys: %s", e)
6680 if isinstance(e, HTTPException): 6680 ↛ 6687line 6680 didn't jump to line 6687 because the condition on line 6680 was always true
6681 raise ProxyException(
6682 message=getattr(e, "detail", f"error({e})"),
6683 type=ProxyErrorTypes.internal_server_error,
6684 param=getattr(e, "param", "None"),
6685 code=getattr(e, "status_code", fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR),
6686 )
6687 elif isinstance(e, ProxyException):
6688 raise e
6689 raise ProxyException(
6690 message="Authentication Error, " + str(e),
6691 type=ProxyErrorTypes.internal_server_error,
6692 param=getattr(e, "param", "None"),
6693 code=fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR,
6694 )
6697async def _apply_non_admin_alias_scope(
6698 user_api_key_dict: UserAPIKeyAuth,
6699 prisma_client: PrismaClient,
6700 query_params: list[object],
6701 where_parts: list[str],
6702) -> None:
6703 """Append SQL scope conditions so non-admin users only see aliases for
6704 keys they own or keys belonging to teams they are members of."""
6705 scope_conditions: Final[list[str]] = []
6706 if user_api_key_dict.user_id:
6707 query_params.append(user_api_key_dict.user_id)
6708 scope_conditions.append(f"user_id = ${len(query_params)}")
6710 # Look up the user's teams from the user table
6711 user_teams: list[str] = []
6712 if user_api_key_dict.user_id:
6713 user_row: Final = await _prisma_table(UserRepository(prisma_client)).find_unique(
6714 where={"user_id": user_api_key_dict.user_id}
6715 )
6716 if user_row is not None:
6717 user_teams = getattr(user_row, "teams", []) or []
6719 if user_teams:
6720 team_placeholders: Final = ", ".join(f"${len(query_params) + i + 1}" for i in range(len(user_teams)))
6721 query_params.extend(user_teams)
6722 scope_conditions.append(f"team_id IN ({team_placeholders})")
6724 if scope_conditions:
6725 where_parts.append(f"({' OR '.join(scope_conditions)})")
6726 else:
6727 # No user_id and no teams — return nothing
6728 where_parts.append("FALSE")
6731@router.get(
6732 "/key/aliases",
6733 tags=["key management"],
6734 dependencies=[Depends(user_api_key_auth)],
6735)
6736@management_endpoint_wrapper
6737async def key_aliases(
6738 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
6739 page: int = Query(1, ge=1, description="Page number"),
6740 size: int = Query(50, ge=1, le=100, description="Page size"),
6741 search: str | None = Query(None, description="Search key aliases (case-insensitive partial match)"),
6742 team_id: str | None = Query(None, description="Filter aliases to keys belonging to this team"),
6743) -> dict[str, Any]:
6744 """
6745 Lists key aliases with pagination and optional search.
6747 Non-admin users only see aliases for keys they own or keys belonging to
6748 their teams.
6750 Returns:
6751 {
6752 "aliases": List[str],
6753 "total_count": int,
6754 "current_page": int,
6755 "total_pages": int,
6756 "size": int,
6757 }
6758 """
6759 try:
6760 from litellm.proxy.proxy_server import prisma_client
6762 verbose_proxy_logger.debug("Entering key_aliases function")
6764 if prisma_client is None: 6764 ↛ 6765line 6764 didn't jump to line 6765 because the condition on line 6764 was never true
6765 verbose_proxy_logger.error("Database not connected")
6766 raise Exception("Database not connected")
6768 # Build a parameterized WHERE clause to avoid loading full rows into
6769 # memory. Raw SQL is used because the Prisma client wrapper does not
6770 # support column-level SELECT projection on find_many.
6771 #
6772 # $1 is always UI_SESSION_TOKEN_TEAM_ID (filters out UI session tokens).
6773 query_params: Final[list[object]] = [UI_SESSION_TOKEN_TEAM_ID]
6774 where_parts: Final = [
6775 "key_alias IS NOT NULL",
6776 "key_alias != ''",
6777 "(team_id IS NULL OR team_id != $1)",
6778 ]
6780 # Scope results for non-admin users: only show aliases for keys the
6781 # user owns or keys belonging to teams they are a member of.
6782 is_proxy_admin: Final = user_api_key_dict.user_role in [
6783 LitellmUserRoles.PROXY_ADMIN.value,
6784 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
6785 ]
6786 if not is_proxy_admin: 6786 ↛ 6787line 6786 didn't jump to line 6787 because the condition on line 6786 was never true
6787 await _apply_non_admin_alias_scope(user_api_key_dict, prisma_client, query_params, where_parts)
6789 if search:
6790 query_params.append(f"%{search}%")
6791 where_parts.append(f"key_alias ILIKE ${len(query_params)}")
6793 if team_id:
6794 query_params.append(team_id)
6795 where_parts.append(f"team_id = ${len(query_params)}")
6797 where_sql: Final = " AND ".join(where_parts)
6799 count_sql: Final = f'SELECT COUNT(*) AS count FROM "LiteLLM_VerificationToken" WHERE {where_sql}'
6800 count_rows: Final[Sequence[Mapping[str, int]]] = await prisma_client.db.query_raw(count_sql, *query_params)
6801 total_count: Final = int(count_rows[0]["count"]) if count_rows else 0
6803 aliases_params: Final = query_params + [size, (page - 1) * size]
6804 limit_idx: Final = len(aliases_params) - 1
6805 offset_idx: Final = len(aliases_params)
6806 aliases_sql: Final = (
6807 f"SELECT key_alias"
6808 f' FROM "LiteLLM_VerificationToken"'
6809 f" WHERE {where_sql}"
6810 f" ORDER BY key_alias ASC"
6811 f" LIMIT ${limit_idx} OFFSET ${offset_idx}"
6812 )
6813 alias_rows: Final[Sequence[Mapping[str, str]]] = await prisma_client.db.query_raw(aliases_sql, *aliases_params)
6814 aliases: Final[list[str]] = [row["key_alias"] for row in alias_rows if row.get("key_alias")]
6816 total_pages: Final = -(-total_count // size) if total_count > 0 else 0
6817 verbose_proxy_logger.debug(
6818 "key_aliases: page=%s, size=%s, search=%r, total_count=%s, total_pages=%s",
6819 page,
6820 size,
6821 search,
6822 total_count,
6823 total_pages,
6824 )
6826 return {
6827 "aliases": aliases,
6828 "total_count": total_count,
6829 "current_page": page,
6830 "total_pages": total_pages,
6831 "size": size,
6832 }
6834 except Exception as e:
6835 verbose_proxy_logger.exception("Error in key_aliases: %s", e)
6836 if isinstance(e, HTTPException): 6836 ↛ 6837line 6836 didn't jump to line 6837 because the condition on line 6836 was never true
6837 raise ProxyException(
6838 message=getattr(e, "detail", f"error({e})"),
6839 type=ProxyErrorTypes.internal_server_error,
6840 param=getattr(e, "param", "None"),
6841 code=getattr(e, "status_code", status.HTTP_500_INTERNAL_SERVER_ERROR),
6842 )
6843 elif isinstance(e, ProxyException): 6843 ↛ 6844line 6843 didn't jump to line 6844 because the condition on line 6843 was never true
6844 raise e
6845 raise ProxyException(
6846 message="Authentication Error, " + str(e),
6847 type=ProxyErrorTypes.internal_server_error,
6848 param=getattr(e, "param", "None"),
6849 code=status.HTTP_500_INTERNAL_SERVER_ERROR,
6850 )
6853def _validate_sort_params(sort_by: str | None, sort_order: str) -> dict[str, str] | None:
6854 order_by: Final[dict[str, str]] = {}
6856 if sort_by is None: 6856 ↛ 6857line 6856 didn't jump to line 6857 because the condition on line 6856 was never true
6857 return None
6858 # Validate sort_by is a valid column
6859 valid_columns: Final = [
6860 "spend",
6861 "max_budget",
6862 "created_at",
6863 "updated_at",
6864 "token",
6865 "key_alias",
6866 ]
6867 if sort_by not in valid_columns: 6867 ↛ 6874line 6867 didn't jump to line 6874 because the condition on line 6867 was always true
6868 raise HTTPException(
6869 status_code=400,
6870 detail={"error": f"Invalid sort column. Must be one of: {', '.join(valid_columns)}"},
6871 )
6873 # Validate sort_order
6874 if sort_order.lower() not in ["asc", "desc"]:
6875 raise HTTPException(
6876 status_code=400,
6877 detail={"error": "Invalid sort order. Must be 'asc' or 'desc'"},
6878 )
6880 order_by[sort_by] = sort_order.lower()
6882 return order_by
6885def _build_expires_where_clause(expires_filter: str, now: datetime) -> dict[str, object]:
6886 if expires_filter == "expired":
6887 return {"AND": [{"expires": {"not": None}}, {"expires": {"lt": now}}]}
6888 return {"OR": [{"expires": None}, {"expires": {"gte": now}}]}
6891def _not_blocked_where_clause() -> dict[str, object]:
6892 return {"OR": [{"blocked": None}, {"blocked": False}]}
6895def _build_status_where_clause(status_filter: str | None, now: datetime) -> dict[str, object] | None:
6896 if status_filter == "revoked": 6896 ↛ 6897line 6896 didn't jump to line 6897 because the condition on line 6896 was never true
6897 return {"blocked": True}
6898 if status_filter in ("expired", "active"): 6898 ↛ 6899line 6898 didn't jump to line 6899 because the condition on line 6898 was never true
6899 return {"AND": [_not_blocked_where_clause(), _build_expires_where_clause(status_filter, now)]}
6900 return None
6903def _build_key_search_where(search: str) -> KeySearchWhere:
6904 search_where: Final[KeySearchWhere] = {
6905 "OR": (
6906 {"token": search},
6907 {"key_alias": {"contains": search, "mode": "insensitive"}},
6908 )
6909 }
6910 return search_where
6913def _build_key_filter_conditions(
6914 user_id: str | None,
6915 team_id: str | None,
6916 organization_id: str | None,
6917 key_alias: str | None,
6918 key_hash: str | None,
6919 exclude_team_id: str | None,
6920 admin_team_ids: list[str] | None,
6921 member_team_ids: list[str] | None = None,
6922 include_created_by_keys: bool = False,
6923 project_id: str | None = None,
6924 access_group_id: str | None = None,
6925 agent_id: str | None = None,
6926 use_substring_matching: bool = False,
6927 use_key_alias_substring_matching: bool = False,
6928 expires_filter: str | None = None,
6929 search: str | None = None,
6930 status_filter: str | None = None,
6931) -> Mapping[str, object]:
6932 """Build filter conditions for key listing.
6934 Visibility rules:
6935 - Users always see their own keys (user_id match)
6936 - Team admins see ALL keys for their admin teams (via admin_team_ids)
6937 - Regular team members see only service accounts (user_id=NULL) for their
6938 teams (via member_team_ids). This prevents leaking other members' spend data.
6939 - created_by visibility is scoped to teams the user currently belongs to,
6940 so former members cannot see service accounts they created after leaving.
6941 """
6942 # Prepare filter conditions
6943 where: dict[str, object] = {}
6944 where.update(_get_condition_to_filter_out_ui_session_tokens())
6946 # Build the OR conditions for user's keys and admin team keys
6947 or_conditions: Final[list[dict[str, object]]] = []
6949 # Base conditions for user's own keys
6950 user_condition: Final[dict[str, object]] = {}
6951 if user_id and isinstance(user_id, str):
6952 if use_substring_matching:
6953 user_condition["user_id"] = {
6954 "contains": user_id,
6955 "mode": "insensitive",
6956 }
6957 else:
6958 user_condition["user_id"] = user_id
6959 if exclude_team_id and isinstance(exclude_team_id, str): 6959 ↛ 6960line 6959 didn't jump to line 6960 because the condition on line 6959 was never true
6960 user_condition["team_id"] = {"not": exclude_team_id}
6961 if organization_id and isinstance(organization_id, str):
6962 user_condition["organization_id"] = organization_id
6964 if user_condition:
6965 or_conditions.append(user_condition)
6967 # Add condition for created_by keys, scoped to user's current teams
6968 if include_created_by_keys and user_id:
6969 if member_team_ids is not None: 6969 ↛ 6992line 6969 didn't jump to line 6992 because the condition on line 6969 was always true
6970 if member_team_ids: 6970 ↛ 6973line 6970 didn't jump to line 6973 because the condition on line 6970 was never true
6971 # Scope created_by keys to teams user is still a member of,
6972 # or keys that have no team (personal keys)
6973 or_conditions.append(
6974 {
6975 "AND": [
6976 {"created_by": user_id},
6977 {
6978 "OR": [
6979 {"team_id": {"in": member_team_ids}},
6980 {"team_id": None},
6981 ]
6982 },
6983 ]
6984 }
6985 )
6986 else:
6987 # User is not a member of any team, only show non-team created_by keys
6988 or_conditions.append({"AND": [{"created_by": user_id}, {"team_id": None}]})
6989 else:
6990 # No team membership info provided (backward compatibility for
6991 # direct _list_key_helper callers like Prometheus)
6992 or_conditions.append({"created_by": user_id})
6994 # Add condition for admin team keys (admins see ALL team keys)
6995 if admin_team_ids: 6995 ↛ 6996line 6995 didn't jump to line 6996 because the condition on line 6995 was never true
6996 or_conditions.append({"team_id": {"in": admin_team_ids}})
6998 # Add condition for member team service accounts (members only see keys with user_id=NULL)
6999 if member_team_ids: 6999 ↛ 7001line 6999 didn't jump to line 7001 because the condition on line 6999 was never true
7000 # Exclude teams where user is already admin (those are covered above with full visibility)
7001 member_only_team_ids: Final = [tid for tid in member_team_ids if tid not in (admin_team_ids or [])]
7002 if member_only_team_ids:
7003 or_conditions.append(
7004 {
7005 "AND": [
7006 {"team_id": {"in": member_only_team_ids}},
7007 {"user_id": None},
7008 ]
7009 }
7010 )
7012 # Combine conditions with OR if we have multiple conditions
7013 if len(or_conditions) > 1:
7014 where = {"AND": [where, {"OR": or_conditions}]}
7015 elif len(or_conditions) == 1:
7016 where.update(or_conditions[0])
7018 # Apply team_id, project_id and access_group_id as global AND filters so they
7019 # narrow results across all visibility conditions (own keys, team keys, etc.)
7020 now: Final = datetime.now(timezone.utc)
7021 status_where: Final = _build_status_where_clause(status_filter, now)
7022 global_filters: Final[tuple[Mapping[str, object], ...]] = (
7023 *(
7024 (
7025 {"key_alias": {"contains": key_alias, "mode": "insensitive"}}
7026 if use_key_alias_substring_matching
7027 else {"key_alias": key_alias},
7028 )
7029 if key_alias and isinstance(key_alias, str)
7030 else ()
7031 ),
7032 *(({"token": key_hash},) if key_hash and isinstance(key_hash, str) else ()),
7033 *((_build_key_search_where(search),) if isinstance(search, str) and search else ()),
7034 *(({"team_id": team_id},) if team_id and isinstance(team_id, str) else ()),
7035 *(({"project_id": project_id},) if project_id else ()),
7036 *(({"access_group_ids": {"hasSome": [access_group_id]}},) if access_group_id else ()),
7037 *(({"agent_id": agent_id},) if agent_id and isinstance(agent_id, str) else ()),
7038 *(
7039 (_build_expires_where_clause(expires_filter, now),)
7040 if expires_filter is not None and expires_filter in VALID_EXPIRES_FILTER_VALUES
7041 else ()
7042 ),
7043 *((status_where,) if status_where is not None else ()),
7044 )
7045 combined_where: Final[Mapping[str, object]] = {"AND": [where, *global_filters]} if global_filters else where
7046 verbose_proxy_logger.debug("Filter conditions: %s", combined_where)
7047 return combined_where
7050async def _list_key_helper(
7051 prisma_client: PrismaClient,
7052 page: int,
7053 size: int,
7054 user_id: str | None,
7055 team_id: str | None,
7056 organization_id: str | None,
7057 key_alias: str | None,
7058 key_hash: str | None,
7059 exclude_team_id: str | None = None,
7060 return_full_object: bool = False,
7061 admin_team_ids: list[str] | None = None, # New parameter for teams where user is admin
7062 member_team_ids: list[str]
7063 | None = None, # Team IDs where user is a member (any role) - for service account visibility
7064 include_created_by_keys: bool = False,
7065 sort_by: str | None = None,
7066 sort_order: str = "desc",
7067 expand: list[str] | None = None,
7068 status: str | None = None,
7069 project_id: str | None = None,
7070 access_group_id: str | None = None,
7071 agent_id: str | None = None,
7072 use_substring_matching: bool = False,
7073 use_key_alias_substring_matching: bool = False,
7074 expires_filter: str | None = None,
7075 search: str | None = None,
7076) -> KeyListResponseObject:
7077 """
7078 Helper function to list keys
7079 Args:
7080 page: int
7081 size: int
7082 user_id: Optional[str]
7083 team_id: Optional[str]
7084 key_alias: Optional[str]
7085 exclude_team_id: Optional[str] # exclude a specific team_id
7086 return_full_object: bool # when true, will return UserAPIKeyAuth objects instead of just the token
7087 admin_team_ids: Optional[List[str]] # list of team IDs where the user is an admin
7088 member_team_ids: Optional[List[str]] # list of team IDs where user is a member (for service account visibility)
7090 Returns:
7091 KeyListResponseObject
7092 {
7093 "keys": List[str] or List[UserAPIKeyAuth], # Updated to reflect possible return types
7094 "total_count": int,
7095 "current_page": int,
7096 "total_pages": int,
7097 }
7098 """
7099 where: Final = _build_key_filter_conditions(
7100 user_id=user_id,
7101 team_id=team_id,
7102 organization_id=organization_id,
7103 key_alias=key_alias,
7104 key_hash=key_hash,
7105 exclude_team_id=exclude_team_id,
7106 admin_team_ids=admin_team_ids,
7107 member_team_ids=member_team_ids,
7108 include_created_by_keys=include_created_by_keys,
7109 project_id=project_id,
7110 access_group_id=access_group_id,
7111 agent_id=agent_id,
7112 use_substring_matching=use_substring_matching,
7113 use_key_alias_substring_matching=use_key_alias_substring_matching,
7114 expires_filter=expires_filter,
7115 search=search,
7116 status_filter=status,
7117 )
7119 # Calculate skip for pagination
7120 skip: Final = (page - 1) * size
7122 verbose_proxy_logger.debug("Pagination: skip=%s, take=%s", skip, size)
7124 order_by: Final[dict[str, str] | None] = (
7125 _validate_sort_params(sort_by, sort_order) if sort_by is not None and isinstance(sort_by, str) else None
7126 )
7128 # Determine which table to query based on status
7129 use_deleted_table: Final = status == "deleted"
7131 # Fetch keys with pagination
7132 if use_deleted_table: 7132 ↛ 7133line 7132 didn't jump to line 7133 because the condition on line 7132 was never true
7133 keys = await DeletedVerificationTokenRepository(prisma_client).table.find_many(
7134 where=where,
7135 skip=skip,
7136 take=size,
7137 order=(
7138 order_by
7139 if order_by
7140 else [
7141 {"created_at": "desc"},
7142 {"token": "desc"}, # fallback sort
7143 ]
7144 ),
7145 )
7146 else:
7147 keys = await VerificationTokenRepository(prisma_client).table.find_many(
7148 where=where,
7149 skip=skip,
7150 take=size,
7151 order=(
7152 order_by
7153 if order_by
7154 else [
7155 {"created_at": "desc"},
7156 {"token": "desc"}, # fallback sort
7157 ]
7158 ),
7159 include={"object_permission": True, "litellm_budget_table": True},
7160 )
7162 verbose_proxy_logger.debug("Fetched %s keys", len(keys))
7164 # Get total count of keys
7165 if use_deleted_table: 7165 ↛ 7166line 7165 didn't jump to line 7166 because the condition on line 7165 was never true
7166 total_count = await _deleted_verification_token_table(prisma_client).count(where=where)
7167 else:
7168 total_count = await _prisma_table(VerificationTokenRepository(prisma_client)).count(where=where)
7170 verbose_proxy_logger.debug("Total count of keys: %s", total_count)
7172 # Calculate total pages
7173 total_pages: Final = -(-total_count // size) # Ceiling division
7175 # Fetch user information if expand includes "user"
7176 user_map = dict[str | None, _UserRowLike]()
7177 if expand and "user" in expand: 7177 ↛ 7178line 7177 didn't jump to line 7178 because the condition on line 7177 was never true
7178 user_ids: Final = [key.user_id for key in keys if key.user_id]
7179 created_by_ids: Final = [key.created_by for key in keys if key.created_by]
7180 all_ids: Final = list(set(user_ids + created_by_ids)) # Remove duplicates
7181 if all_ids:
7182 users: Final[Sequence[_UserRowLike]] = await _user_table(prisma_client).find_many(
7183 where={"user_id": {"in": all_ids}}
7184 )
7185 user_map = {user.user_id: user for user in users}
7187 # Prepare response
7188 key_list: Final[list[str | UserAPIKeyAuth | LiteLLM_DeletedVerificationToken]] = []
7189 for key in keys:
7190 # Convert Prisma model to dict (supports both Pydantic v1 and v2)
7191 try:
7192 key_dict = key.model_dump()
7193 except Exception:
7194 # Fallback for Pydantic v1 compatibility
7195 key_dict = key.dict() # pyright: ignore[reportDeprecated] # deliberate pydantic v1 fallback
7196 # Attach object_permission if object_permission_id is set (only for non-deleted keys)
7197 if not use_deleted_table: 7197 ↛ 7201line 7197 didn't jump to line 7201 because the condition on line 7197 was always true
7198 key_dict = await attach_object_permission_to_dict(key_dict, prisma_client)
7200 # Include user information if expand includes "user"
7201 if expand and "user" in expand: 7201 ↛ 7202line 7201 didn't jump to line 7202 because the condition on line 7201 was never true
7202 if key.user_id and key.user_id in user_map:
7203 try:
7204 key_dict["user"] = user_map[key.user_id].model_dump()
7205 except Exception:
7206 key_dict["user"] = user_map[key.user_id].dict()
7207 if key.created_by and key.created_by in user_map:
7208 created_by_user = user_map[key.created_by]
7209 key_dict["created_by_user"] = {
7210 "user_id": created_by_user.user_id,
7211 "user_email": created_by_user.user_email,
7212 "user_alias": created_by_user.user_alias,
7213 }
7215 if return_full_object is True or (expand and "user" in expand): 7215 ↛ 7216line 7215 didn't jump to line 7216 because the condition on line 7215 was never true
7216 if use_deleted_table:
7217 # Use deleted key type to preserve deleted_at, deleted_by, etc.
7218 key_list.append(LiteLLM_DeletedVerificationToken.model_validate(key_dict))
7219 else:
7220 key_list.append(
7221 UserAPIKeyAuth(**key_dict) # pyright: ignore[reportAny] # model_dump() is dict[str, Any]
7222 )
7223 else:
7224 _token = key_dict.get("token")
7225 key_list.append(cast(str, _token)) # Return only the token
7227 return KeyListResponseObject(
7228 keys=key_list,
7229 total_count=total_count,
7230 current_page=page,
7231 total_pages=total_pages,
7232 )
7235def _get_condition_to_filter_out_ui_session_tokens() -> Mapping[str, object]:
7236 """
7237 Condition to filter out UI session tokens
7238 """
7239 return {
7240 "OR": [
7241 {"team_id": None}, # Include records where team_id is null
7242 {"team_id": {"not": UI_SESSION_TOKEN_TEAM_ID}}, # Include records where team_id != UI_SESSION_TOKEN_TEAM_ID
7243 ]
7244 }
7247async def _check_key_admin_access(
7248 user_api_key_dict: UserAPIKeyAuth,
7249 hashed_token: str | None,
7250 prisma_client: PrismaClient | None,
7251 user_api_key_cache: UserApiKeyCache,
7252 route: str,
7253) -> None:
7254 """
7255 Check that the caller has admin privileges for the target key.
7257 Allowed callers:
7258 - Proxy admin
7259 - Team admin for the key's team
7260 - Org admin for the key's team's organization
7262 Raises HTTPException(403) if the caller is not authorized.
7263 """
7265 if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: 7265 ↛ 7269line 7265 didn't jump to line 7269 because the condition on line 7265 was always true
7266 return
7268 # Look up the target key to find its team
7269 target_key_row: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
7270 where={"token": hashed_token}
7271 )
7272 if target_key_row is None:
7273 raise HTTPException(
7274 status_code=404,
7275 detail={"error": f"Key not found: {hashed_token}"},
7276 )
7278 # If the key belongs to a team, check team admin / org admin
7279 if target_key_row.team_id:
7280 team_obj: Final = await get_team_object(
7281 team_id=target_key_row.team_id,
7282 prisma_client=prisma_client,
7283 user_api_key_cache=user_api_key_cache,
7284 check_db_only=True,
7285 )
7286 if team_obj is not None:
7287 if _is_user_team_admin(user_api_key_dict=user_api_key_dict, team_obj=team_obj):
7288 return
7289 if await _is_user_org_admin_for_team(user_api_key_dict=user_api_key_dict, team_obj=team_obj):
7290 return
7292 raise HTTPException(
7293 status_code=403,
7294 detail={
7295 "error": f"Only proxy admins, team admins, or org admins can call {route}. "
7296 f"user_role={user_api_key_dict.user_role}, user_id={user_api_key_dict.user_id}"
7297 },
7298 )
7301@router.post("/key/block", tags=["key management"], dependencies=[Depends(user_api_key_auth)])
7302@management_endpoint_wrapper
7303async def block_key(
7304 data: BlockKeyRequest,
7305 http_request: Request,
7306 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
7307 litellm_changed_by: str | None = Header(
7308 None,
7309 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
7310 ),
7311) -> LiteLLM_VerificationToken | None:
7312 """
7313 Block an Virtual key from making any requests.
7315 Parameters:
7316 - key: str - The key to block. Can be either the unhashed key (sk-...) or the hashed key value
7318 Example:
7319 ```bash
7320 curl --location 'http://0.0.0.0:4000/key/block' \
7321 --header 'Authorization: Bearer sk-1234' \
7322 --header 'Content-Type: application/json' \
7323 --data '{
7324 "key": "sk-Fn8Ej39NxjAXrvpUGKghGw"
7325 }'
7326 ```
7328 Note: This is an admin-only endpoint. Only proxy admins, team admins, or org admins can block keys.
7329 """
7330 from litellm.proxy.management_helpers.audit_logs import (
7331 get_audit_log_changed_by,
7332 is_audit_logging_enabled,
7333 )
7334 from litellm.proxy.proxy_server import (
7335 create_audit_log_for_update,
7336 hash_token,
7337 litellm_proxy_admin_name,
7338 prisma_client,
7339 proxy_logging_obj,
7340 user_api_key_cache,
7341 )
7343 if prisma_client is None: 7343 ↛ 7344line 7343 didn't jump to line 7344 because the condition on line 7343 was never true
7344 raise Exception(f"{CommonProxyErrors.db_not_connected_error.value}")
7346 if not is_valid_api_key(data.key):
7347 raise ProxyException(
7348 message="Invalid key format.",
7349 type=ProxyErrorTypes.bad_request_error,
7350 param="key",
7351 code=status.HTTP_400_BAD_REQUEST,
7352 )
7353 if data.key.startswith("sk-"): 7353 ↛ 7356line 7353 didn't jump to line 7356 because the condition on line 7353 was always true
7354 hashed_token = hash_token(token=data.key)
7355 else:
7356 hashed_token = data.key
7358 # Admin-only: only proxy admins, team admins, or org admins can block keys
7359 await _check_key_admin_access(
7360 user_api_key_dict=user_api_key_dict,
7361 hashed_token=hashed_token,
7362 prisma_client=prisma_client,
7363 user_api_key_cache=user_api_key_cache,
7364 route="/key/block",
7365 )
7367 # Check if the key exists before trying to block it
7368 existing_record: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
7369 where={"token": hashed_token}
7370 )
7371 if existing_record is None: 7371 ↛ 7372line 7371 didn't jump to line 7372 because the condition on line 7371 was never true
7372 raise ProxyException(
7373 message="Key not found.",
7374 type=ProxyErrorTypes.not_found_error,
7375 param="key",
7376 code=status.HTTP_404_NOT_FOUND,
7377 )
7379 if is_audit_logging_enabled(): 7379 ↛ 7380line 7379 didn't jump to line 7380 because the condition on line 7379 was never true
7380 asyncio.create_task(
7381 create_audit_log_for_update(
7382 request_data=LiteLLM_AuditLogs(
7383 id=str(uuid.uuid4()),
7384 updated_at=datetime.now(timezone.utc),
7385 changed_by=get_audit_log_changed_by(
7386 litellm_changed_by=litellm_changed_by,
7387 user_api_key_dict=user_api_key_dict,
7388 litellm_proxy_admin_name=litellm_proxy_admin_name,
7389 ),
7390 changed_by_api_key=user_api_key_dict.api_key,
7391 table_name=LitellmTableNames.KEY_TABLE_NAME,
7392 object_id=hashed_token,
7393 action="blocked",
7394 updated_values="{}",
7395 before_value=existing_record.model_dump_json(),
7396 )
7397 )
7398 )
7400 record: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).update(
7401 where={"token": hashed_token},
7402 data=with_settings_updated_at({"blocked": True}),
7403 )
7405 ## UPDATE KEY CACHE - invalidate so next read re-fetches from DB
7406 await _delete_cache_key_object(
7407 hashed_token=hashed_token,
7408 user_api_key_cache=user_api_key_cache,
7409 proxy_logging_obj=proxy_logging_obj,
7410 )
7412 return record
7415@router.post("/key/unblock", tags=["key management"], dependencies=[Depends(user_api_key_auth)])
7416@management_endpoint_wrapper
7417async def unblock_key(
7418 data: BlockKeyRequest,
7419 http_request: Request,
7420 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
7421 litellm_changed_by: str | None = Header(
7422 None,
7423 description="The litellm-changed-by header enables tracking of actions performed by authorized users on behalf of other users, providing an audit trail for accountability",
7424 ),
7425):
7426 """
7427 Unblock a Virtual key to allow it to make requests again.
7429 Parameters:
7430 - key: str - The key to unblock. Can be either the unhashed key (sk-...) or the hashed key value
7432 Example:
7433 ```bash
7434 curl --location 'http://0.0.0.0:4000/key/unblock' \
7435 --header 'Authorization: Bearer sk-1234' \
7436 --header 'Content-Type: application/json' \
7437 --data '{
7438 "key": "sk-Fn8Ej39NxjAXrvpUGKghGw"
7439 }'
7440 ```
7442 Note: This is an admin-only endpoint. Only proxy admins, team admins, or org admins can unblock keys.
7443 """
7444 from litellm.proxy.management_helpers.audit_logs import (
7445 get_audit_log_changed_by,
7446 is_audit_logging_enabled,
7447 )
7448 from litellm.proxy.proxy_server import (
7449 create_audit_log_for_update,
7450 hash_token,
7451 litellm_proxy_admin_name,
7452 prisma_client,
7453 proxy_logging_obj,
7454 user_api_key_cache,
7455 )
7457 if prisma_client is None: 7457 ↛ 7458line 7457 didn't jump to line 7458 because the condition on line 7457 was never true
7458 raise Exception(f"{CommonProxyErrors.db_not_connected_error.value}")
7460 if not is_valid_api_key(data.key):
7461 raise ProxyException(
7462 message="Invalid key format.",
7463 type=ProxyErrorTypes.bad_request_error,
7464 param="key",
7465 code=status.HTTP_400_BAD_REQUEST,
7466 )
7467 if data.key.startswith("sk-"): 7467 ↛ 7470line 7467 didn't jump to line 7470 because the condition on line 7467 was always true
7468 hashed_token = hash_token(token=data.key)
7469 else:
7470 hashed_token = data.key
7472 # Admin-only: only proxy admins, team admins, or org admins can unblock keys
7473 await _check_key_admin_access(
7474 user_api_key_dict=user_api_key_dict,
7475 hashed_token=hashed_token,
7476 prisma_client=prisma_client,
7477 user_api_key_cache=user_api_key_cache,
7478 route="/key/unblock",
7479 )
7481 # Check if the key exists before trying to unblock it
7482 existing_record: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).find_unique(
7483 where={"token": hashed_token}
7484 )
7485 if existing_record is None: 7485 ↛ 7486line 7485 didn't jump to line 7486 because the condition on line 7485 was never true
7486 raise ProxyException(
7487 message="Key not found.",
7488 type=ProxyErrorTypes.not_found_error,
7489 param="key",
7490 code=status.HTTP_404_NOT_FOUND,
7491 )
7493 if is_audit_logging_enabled(): 7493 ↛ 7494line 7493 didn't jump to line 7494 because the condition on line 7493 was never true
7494 asyncio.create_task(
7495 create_audit_log_for_update(
7496 request_data=LiteLLM_AuditLogs(
7497 id=str(uuid.uuid4()),
7498 updated_at=datetime.now(timezone.utc),
7499 changed_by=get_audit_log_changed_by(
7500 litellm_changed_by=litellm_changed_by,
7501 user_api_key_dict=user_api_key_dict,
7502 litellm_proxy_admin_name=litellm_proxy_admin_name,
7503 ),
7504 changed_by_api_key=user_api_key_dict.api_key,
7505 table_name=LitellmTableNames.KEY_TABLE_NAME,
7506 object_id=hashed_token,
7507 action="unblocked",
7508 updated_values="{}",
7509 before_value=existing_record.model_dump_json(),
7510 )
7511 )
7512 )
7514 record: Final = await _prisma_table(VerificationTokenRepository(prisma_client)).update(
7515 where={"token": hashed_token},
7516 data=with_settings_updated_at({"blocked": False}),
7517 )
7519 ## UPDATE KEY CACHE - invalidate so next read re-fetches from DB
7520 await _delete_cache_key_object(
7521 hashed_token=hashed_token,
7522 user_api_key_cache=user_api_key_cache,
7523 proxy_logging_obj=proxy_logging_obj,
7524 )
7526 return record
7529@router.post(
7530 "/key/health",
7531 tags=["key management"],
7532 dependencies=[Depends(user_api_key_auth)],
7533 response_model=KeyHealthResponse,
7534)
7535@management_endpoint_wrapper
7536async def key_health(
7537 request: Request,
7538 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
7539):
7540 """
7541 Check the health of the key
7543 Checks:
7544 - If key based logging is configured correctly - sends a test log
7546 Usage
7548 Pass the key in the request header
7550 ```bash
7551 curl -X POST "http://localhost:4000/key/health" \
7552 -H "Authorization: Bearer sk-1234" \
7553 -H "Content-Type: application/json"
7554 ```
7556 Response when logging callbacks are setup correctly:
7558 ```json
7559 {
7560 "key": "healthy",
7561 "logging_callbacks": {
7562 "callbacks": [
7563 "gcs_bucket"
7564 ],
7565 "status": "healthy",
7566 "details": "No logger exceptions triggered, system is healthy. Manually check if logs were sent to ['gcs_bucket']"
7567 }
7568 }
7569 ```
7572 Response when logging callbacks are not setup correctly:
7573 ```json
7574 {
7575 "key": "unhealthy",
7576 "logging_callbacks": {
7577 "callbacks": [
7578 "gcs_bucket"
7579 ],
7580 "status": "unhealthy",
7581 "details": "Logger exceptions triggered, system is unhealthy: Failed to load vertex credentials. Check to see if credentials containing partial/invalid information."
7582 }
7583 }
7584 ```
7585 """
7586 try:
7587 # Get the key's metadata
7588 key_metadata: Final = user_api_key_dict.metadata
7590 health_status: Final[KeyHealthResponse] = KeyHealthResponse(
7591 key="healthy",
7592 logging_callbacks=None,
7593 )
7595 # Check if logging is configured in metadata
7596 if key_metadata and "logging" in key_metadata: 7596 ↛ 7597line 7596 didn't jump to line 7597 because the condition on line 7596 was never true
7597 logging_statuses: Final = await test_key_logging(
7598 user_api_key_dict=user_api_key_dict,
7599 request=request,
7600 key_logging=decrypt_callback_vars(key_metadata)["logging"],
7601 )
7602 health_status["logging_callbacks"] = logging_statuses
7604 # Check if any logging callback is unhealthy
7605 if logging_statuses.get("status") == "unhealthy":
7606 health_status["key"] = "unhealthy"
7608 return KeyHealthResponse(**health_status)
7610 except Exception as e:
7611 raise ProxyException(
7612 message=f"Key health check failed: {e}",
7613 type=ProxyErrorTypes.internal_server_error,
7614 param=getattr(e, "param", "None"),
7615 code=status.HTTP_500_INTERNAL_SERVER_ERROR,
7616 )
7619async def _can_user_query_key_info(
7620 user_api_key_dict: UserAPIKeyAuth,
7621 key: str | None,
7622 key_info: LiteLLM_VerificationToken,
7623) -> bool:
7624 """
7625 Helper to check if the user has access to the key's info
7626 """
7627 if ( 7627 ↛ 7640line 7627 didn't jump to line 7640 because the condition on line 7627 was always true
7628 (
7629 user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
7630 or user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value
7631 )
7632 or user_api_key_dict.api_key == key
7633 or key_info.user_id == user_api_key_dict.user_id
7634 or await TeamMemberPermissionChecks.user_belongs_to_keys_team(
7635 user_api_key_dict=user_api_key_dict,
7636 existing_key_row=key_info,
7637 )
7638 ):
7639 return True
7640 return False
7643async def test_key_logging(
7644 user_api_key_dict: UserAPIKeyAuth,
7645 request: Request,
7646 key_logging: Sequence[Mapping[str, str]],
7647) -> LoggingCallbackStatus:
7648 """
7649 Test the key-based logging
7651 - Test that key logging is correctly formatted and all args are passed correctly
7652 - Make a mock completion call -> user can check if it's correctly logged
7653 - Check if any logger.exceptions were triggered -> if they were then returns it to the user client side
7654 """
7655 import logging
7656 from io import StringIO
7658 from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request
7659 from litellm.proxy.proxy_server import general_settings, proxy_config
7661 logging_callbacks: Final[list[str]] = []
7662 for callback in key_logging:
7663 if callback.get("callback_name") is not None:
7664 logging_callbacks.append(callback["callback_name"])
7665 else:
7666 raise ValueError("callback_name is required in key_logging")
7668 log_capture_string: Final = StringIO()
7669 ch: Final = logging.StreamHandler(log_capture_string)
7670 ch.setLevel(logging.ERROR)
7671 logger: Final = logging.getLogger()
7672 logger.addHandler(ch)
7674 try:
7675 data = {
7676 "model": "openai/litellm-key-health-test",
7677 "messages": [
7678 {
7679 "role": "user",
7680 "content": "Hello, this is a test from litellm /key/health. No LLM API call was made for this",
7681 }
7682 ],
7683 }
7684 data = await add_litellm_data_to_request(
7685 data=data,
7686 user_api_key_dict=user_api_key_dict,
7687 proxy_config=proxy_config,
7688 general_settings=general_settings,
7689 request=request,
7690 )
7691 data["mock_response"] = "test response"
7692 await litellm.acompletion(**data) # make mock completion call to trigger key based callbacks
7693 except Exception as e:
7694 return LoggingCallbackStatus(
7695 callbacks=logging_callbacks,
7696 status="unhealthy",
7697 details=f"Logging test failed: {e}",
7698 )
7700 await asyncio.sleep(2) # wait for callbacks to run, callbacks use batching so wait for the flush event
7702 # Check if any logger exceptions were triggered
7703 log_contents: Final = log_capture_string.getvalue()
7704 logger.removeHandler(ch)
7705 if log_contents:
7706 return LoggingCallbackStatus(
7707 callbacks=logging_callbacks,
7708 status="unhealthy",
7709 details=f"Logger exceptions triggered, system is unhealthy: {log_contents}",
7710 )
7711 else:
7712 return LoggingCallbackStatus(
7713 callbacks=logging_callbacks,
7714 status="healthy",
7715 details=f"No logger exceptions triggered, system is healthy. Manually check if logs were sent to {logging_callbacks} ",
7716 )
7719_KEY_ALIAS_PATTERN: Final = re.compile(r"^[a-zA-Z0-9][a-zA-Z0-9_\-/\.@]{0,253}[a-zA-Z0-9]$")
7720_KEY_ALIAS_PATTERN_MESSAGE: Final = (
7721 "Invalid key_alias format. Must be 2-255 characters, start/end with alphanumeric, and only contain a-zA-Z0-9_-/.@."
7722)
7723_KEY_ALIAS_MAX_LENGTH: Final = 255
7726def parse_key_alias_pattern(value: object) -> str | None:
7727 if value is None:
7728 return None
7729 if not isinstance(value, str):
7730 raise ValueError(
7731 f"Invalid regex set for litellm_settings.key_alias_pattern - value={value!r}: must be a string"
7732 )
7733 try:
7734 re.compile(value)
7735 except re.error as e:
7736 raise ValueError(f"Invalid regex set for litellm_settings.key_alias_pattern - value={value}: {e}") from e
7737 return value
7740def _key_alias_rule() -> tuple[re.Pattern[str], str] | None:
7741 if litellm.key_alias_pattern is not None:
7742 return (
7743 re.compile(litellm.key_alias_pattern),
7744 f"Invalid key_alias format. Must be at most {_KEY_ALIAS_MAX_LENGTH} characters and match the configured"
7745 f" key_alias_pattern: {litellm.key_alias_pattern}",
7746 )
7747 if litellm.enable_key_alias_format_validation:
7748 return (_KEY_ALIAS_PATTERN, _KEY_ALIAS_PATTERN_MESSAGE)
7749 return None
7752def _validate_key_alias_format(key_alias: str | None) -> None:
7753 """
7754 Validate the format of the key_alias.
7756 Path traversal and control characters are always rejected. The alias then has to
7757 stay within ``_KEY_ALIAS_MAX_LENGTH`` and fully match ``litellm.key_alias_pattern``
7758 when one is configured, else the built-in pattern when
7759 ``litellm.enable_key_alias_format_validation`` is on, else nothing more is checked
7760 so existing workflows are not broken.
7761 """
7762 if key_alias is None: 7762 ↛ 7765line 7762 didn't jump to line 7765 because the condition on line 7762 was always true
7763 return
7765 try:
7766 raise_if_unsafe_secret_name(key_alias)
7767 except ValueError:
7768 raise ProxyException(
7769 message="Invalid key_alias",
7770 type=ProxyErrorTypes.bad_request_error,
7771 param="key_alias",
7772 code=400,
7773 )
7775 rule: Final = _key_alias_rule()
7776 if rule is None:
7777 return
7779 pattern, message = rule
7780 if len(key_alias) > _KEY_ALIAS_MAX_LENGTH or pattern.fullmatch(key_alias) is None:
7781 raise ProxyException(
7782 message=message,
7783 type=ProxyErrorTypes.bad_request_error,
7784 param="key_alias",
7785 code=400,
7786 )
7789async def _enforce_unique_key_alias(
7790 key_alias: str | None,
7791 prisma_client: PrismaClient | None,
7792 existing_key_token: str | None = None,
7793) -> None:
7794 """
7795 Helper to enforce unique key aliases across all keys.
7797 Args:
7798 key_alias (Optional[str]): The key alias to check
7799 prisma_client (Any): Prisma client instance
7800 existing_key_token (Optional[str]): ID of existing key being updated, to exclude from uniqueness check
7801 (The Admin UI passes key_alias, in all Edit key requests. So we need to be sure that if we find a key with the same alias, it's not the same key we're updating)
7803 Raises:
7804 ProxyException: If key alias already exists on a different key
7805 """
7806 if key_alias is not None and prisma_client is not None: 7806 ↛ 7807line 7806 didn't jump to line 7807 because the condition on line 7806 was never true
7807 where_clause: Final[dict[str, object]] = {"key_alias": key_alias}
7808 if existing_key_token:
7809 # Exclude the current key from the uniqueness check
7810 where_clause["NOT"] = {"token": existing_key_token}
7812 existing_key = await _prisma_table(VerificationTokenRepository(prisma_client)).find_first(where=where_clause)
7813 if existing_key is not None:
7814 raise ProxyException(
7815 message=f"Key with alias '{key_alias}' already exists. Unique key aliases across all keys are required.",
7816 type=ProxyErrorTypes.bad_request_error,
7817 param="key_alias",
7818 code=status.HTTP_400_BAD_REQUEST,
7819 )
7822def validate_model_max_budget(model_max_budget: dict | None) -> None:
7823 """
7824 Validate the model_max_budget is GenericBudgetConfigType + enforce user has an enterprise license
7826 Raises:
7827 Exception: If model_max_budget is not a valid GenericBudgetConfigType
7828 """
7829 try:
7830 if model_max_budget is None:
7831 return
7832 if len(model_max_budget) == 0:
7833 return
7834 if model_max_budget is not None: 7834 ↛ exitline 7834 didn't return from function 'validate_model_max_budget' because the condition on line 7834 was always true
7835 from litellm.proxy.proxy_server import CommonProxyErrors, premium_user
7837 if premium_user is not True: 7837 ↛ 7841line 7837 didn't jump to line 7841 because the condition on line 7837 was always true
7838 raise ValueError(
7839 f"You must have an enterprise license to set model_max_budget. {CommonProxyErrors.not_premium_user.value}"
7840 )
7841 for _model, _budget_info in model_max_budget.items():
7842 assert isinstance(_model, str)
7844 # Normalize to dict (Pydantic may already parse nested values as BudgetConfig)
7845 _info = _budget_info.model_dump() if hasattr(_budget_info, "model_dump") else dict(_budget_info)
7846 # /CRUD endpoints can pass budget_limit as a string, so we need to convert it to a float
7847 if "budget_limit" in _info:
7848 _info["budget_limit"] = float(_info["budget_limit"])
7849 BudgetConfig(**_info)
7850 except Exception as e:
7851 raise ValueError(
7852 f"Invalid model_max_budget: {e}. Example of valid model_max_budget: https://docs.litellm.ai/docs/proxy/users"
7853 )