Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/_types.py: 89%
2428 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1import enum
2import json
3import os
4from collections.abc import Callable, Mapping
5from datetime import datetime
6from types import MappingProxyType
7from typing import TYPE_CHECKING, Annotated, Any, Final, Literal, NamedTuple, TypeAlias
9import httpx
10from pydantic import (
11 BaseModel,
12 BeforeValidator,
13 ConfigDict,
14 Field,
15 Json,
16 JsonValue,
17 PositiveInt,
18 field_validator,
19 model_validator,
20)
21from typing_extensions import NotRequired, ReadOnly, Required, TypedDict
23from litellm._uuid import uuid
24from litellm.constants import DEFAULT_STAGGER_WINDOW_SECONDS, MCP_STDIO_ALLOWED_COMMANDS
25from litellm.litellm_core_utils.initialize_dynamic_callback_params import (
26 validate_langfuse_environment_value,
27 validate_langfuse_span_scope_value,
28 validate_no_callback_env_reference,
29)
30from litellm.proxy._experimental.mcp_server.stdio_gate import MCP_STDIO_DISABLED_MESSAGE, is_mcp_stdio_enabled
31from litellm.types.agents import AgentCaller
32from litellm.types.integrations.compression_interception import (
33 CompressionSavingsMetadata,
34)
35from litellm.types.integrations.slack_alerting import AlertType
36from litellm.types.llms.openai import (
37 AllMessageValues,
38 ResponsesAPIResponse,
39)
40from litellm.types.mcp import (
41 MCPAllowedClient,
42 MCPAuth,
43 MCPAuthType,
44 MCPCredentials,
45 MCPTransport,
46 MCPTransportType,
47)
48from litellm.types.mcp_server.mcp_server_manager import MCPInfo
49from litellm.types.proxy.carried_budget_state import (
50 OrgBudgetSnapshot,
51 TeamBudgetSnapshot,
52 UserBudgetSnapshot,
53)
54from litellm.types.proxy.control_plane_endpoints import WorkerRegistryEntry
55from litellm.types.proxy.spend_capture_rate import SpendCaptureRateCheckSettings
56from litellm.types.router import RouterErrors, UpdateRouterConfig
57from litellm.types.router_weights import validate_router_settings_dict
58from litellm.types.secret_managers.main import KeyManagementSystem
59from litellm.types.utils import (
60 AzureSpillover,
61 CallTypes,
62 CostBreakdown,
63 EmbeddingResponse,
64 GenericBudgetConfigType,
65 ImageResponse,
66 InternalCallOrigin,
67 LiteLLMPydanticObjectBase,
68 ModelResponse,
69 ProviderField,
70 StandardCallbackDynamicParams,
71 StandardLoggingGuardrailInformation,
72 StandardLoggingMCPToolCall,
73 StandardLoggingModelInformation,
74 StandardLoggingPayloadErrorInformation,
75 StandardLoggingPayloadStatus,
76 StandardLoggingRoutingDecision,
77 StandardLoggingVectorStoreRequest,
78 StandardPassThroughResponseObject,
79 TextCompletionResponse,
80 TranscriptionResponse,
81)
82from litellm.types.videos.main import VideoObject
84from .types_utils.utils import get_instance_fn, validate_custom_validate_return_type
86if TYPE_CHECKING: 86 ↛ 87line 86 didn't jump to line 87 because the condition on line 86 was never true
87 from opentelemetry.trace import Span as _Span
89 Span = _Span | Any
90else:
91 Span = Any
94class ReconcileOutcome(NamedTuple):
95 """What a model reconcile observed, captured while it still held the reconcile
96 lock.
98 Both fields have to be read under that lock to be worth anything. ``live_after``
99 in particular is the router's serving state the instant this reconcile finished,
100 which is NOT the same as what a later snapshot would see: any other model write
101 admitted in between briefly un-serves every db model (see ``clear_cache``), so a
102 caller that re-snapshots at verdict time can observe that hole and blame its own
103 reload for it.
105 - ``still_desired``: the db + config ids the reconcile reconciled against, or None
106 when no reconcile ran and the desired set is therefore unknown.
107 - ``live_after``: the ids the router served immediately after the reconcile, or
108 None when no reconcile ran.
109 """
111 still_desired: frozenset[str] | None
112 live_after: frozenset[str] | None
115class SupportedDBObjectType(str, enum.Enum):
116 """
117 Supported database object types for fine-grained DB storage control.
118 Use in general_settings.supported_db_objects to specify which objects to load from DB.
119 """
121 MODELS = "models"
122 MCP = "mcp"
123 GUARDRAILS = "guardrails"
124 POLICIES = "policies"
125 VECTOR_STORES = "vector_stores"
126 PASS_THROUGH_ENDPOINTS = "pass_through_endpoints"
127 PROMPTS = "prompts"
128 MODEL_COST_MAP = "model_cost_map"
129 TOOLS = "tools"
130 CONFIG_OVERRIDES = "config_overrides"
131 WEBSEARCH_INTERCEPTION_SETTINGS = "websearch_interception_settings"
133 def __str__(self):
134 return str(self.value)
137class LiteLLMTeamRoles(enum.Enum):
138 # team admin
139 TEAM_ADMIN = "admin"
140 # team member
141 TEAM_MEMBER = "user"
144class LitellmUserRoles(str, enum.Enum):
145 """
146 Admin Roles:
147 PROXY_ADMIN: admin over the platform
148 PROXY_ADMIN_VIEW_ONLY: can login, view all own keys, view all spend
149 ORG_ADMIN: admin over a specific organization, can create teams, users only within their organization
151 Internal User Roles:
152 INTERNAL_USER: can login, view/create/delete their own keys, view their spend
153 INTERNAL_USER_VIEW_ONLY: can login, view their own keys, view their own spend
156 Team Roles:
157 TEAM: used for JWT auth
160 Customer Roles:
161 CUSTOMER: External users -> these are customers
163 """
165 # Admin Roles
166 PROXY_ADMIN = "proxy_admin"
167 PROXY_ADMIN_VIEW_ONLY = "proxy_admin_viewer"
169 # Organization admins
170 ORG_ADMIN = "org_admin"
172 # Internal User Roles
173 INTERNAL_USER = "internal_user"
174 INTERNAL_USER_VIEW_ONLY = "internal_user_viewer"
176 # Team Roles
177 TEAM = "team"
179 # Customer Roles - External users of proxy
180 CUSTOMER = "customer"
182 def __str__(self):
183 return str(self.value)
185 def values(self) -> list[str]:
186 return list(self.__annotations__.keys())
188 @property
189 def description(self):
190 """
191 Descriptions for the enum values
192 """
193 descriptions: Final = {
194 "proxy_admin": "admin over litellm proxy, has all permissions",
195 "proxy_admin_viewer": "view all keys, view all spend",
196 "internal_user": "view/create/delete their own keys, view their own spend",
197 "internal_user_viewer": "view their own keys, view their own spend",
198 "team": "team scope used for JWT auth",
199 "customer": "customer",
200 }
201 return descriptions.get(self.value, "")
203 @property
204 def ui_label(self):
205 """
206 UI labels for the enum values
207 """
208 ui_labels: Final = {
209 "proxy_admin": "Admin (All Permissions)",
210 "proxy_admin_viewer": "Admin (View Only)",
211 "internal_user": "Internal User (Create/Delete/View)",
212 "internal_user_viewer": "Internal User (View Only)",
213 "team": "Team",
214 "customer": "Customer",
215 }
216 return ui_labels.get(self.value, "")
218 @property
219 def is_internal_user_role(self) -> bool:
220 """returns true if this role is an `internal_user` or `internal_user_viewer` role"""
221 return self.value in [
222 self.INTERNAL_USER,
223 self.INTERNAL_USER_VIEW_ONLY,
224 ]
227class LitellmTableNames(str, enum.Enum):
228 """
229 Enum for Table Names used by LiteLLM
230 """
232 TEAM_TABLE_NAME = "LiteLLM_TeamTable"
233 USER_TABLE_NAME = "LiteLLM_UserTable"
234 KEY_TABLE_NAME = "LiteLLM_VerificationToken"
235 PROXY_MODEL_TABLE_NAME = "LiteLLM_ProxyModelTable"
236 MANAGED_FILE_TABLE_NAME = "LiteLLM_ManagedFileTable"
237 TOOL_TABLE_NAME = "LiteLLM_ToolTable"
238 CACHE_CONFIG_TABLE_NAME = "LiteLLM_CacheConfig"
239 CONFIG_OVERRIDES_TABLE_NAME = "LiteLLM_ConfigOverrides"
240 CONFIG_TABLE_NAME = "LiteLLM_Config"
241 SSO_CONFIG_TABLE_NAME = "LiteLLM_SSOConfig"
242 UI_SETTINGS_TABLE_NAME = "LiteLLM_UISettings"
243 AGENT_TABLE_NAME = "LiteLLM_AgentsTable"
246class Litellm_EntityType(enum.Enum):
247 """
248 Enum for types of entities on litellm
250 This enum allows specifying the type of entity that is being tracked in the database.
251 """
253 KEY = "key"
254 USER = "user"
255 END_USER = "end_user"
256 TEAM = "team"
257 TEAM_MEMBER = "team_member"
258 ORGANIZATION = "organization"
259 ORGANIZATION_MEMBER = "organization_member"
260 PROJECT = "project"
261 TAG = "tag"
262 AGENT = "agent"
263 MODEL_ACCESS_GROUP = "model_access_group"
265 # global proxy level entity
266 PROXY = "proxy"
269def hash_token(token: str):
270 import hashlib
272 # Hash the string using SHA-256
273 hashed_token: Final = hashlib.sha256(token.encode()).hexdigest()
275 return hashed_token
278class KeyManagementRoutes(str, enum.Enum):
279 """
280 Enum for key management routes
281 """
283 # write routes
284 KEY_GENERATE = "/key/generate"
285 KEY_UPDATE = "/key/update"
286 KEY_DELETE = "/key/delete"
287 KEY_REGENERATE = "/key/regenerate"
288 KEY_GENERATE_SERVICE_ACCOUNT = "/key/service-account/generate"
289 KEY_REGENERATE_WITH_PATH_PARAM = "/key/{key_id}/regenerate"
290 KEY_BLOCK = "/key/block"
291 KEY_UNBLOCK = "/key/unblock"
292 KEY_BULK_UPDATE = "/key/bulk_update"
293 TEAM_KEY_BULK_UPDATE = "/team/key/bulk_update"
294 KEY_RESET_SPEND = "/key/{key_id}/reset_spend"
296 # Field-level opt-in permission (not a real HTTP route). When present in a
297 # team's `team_member_permissions`, non-admin members of that team may set
298 # `access_group_ids` on keys they create/update. Default-deny.
299 KEY_ACCESS_GROUP_ASSIGNMENT = "/key/access_group_assignment"
300 AUTO_ROUTER_MANAGE = "/auto_router/manage"
302 # info and health routes
303 KEY_INFO = "/key/info"
304 KEY_HEALTH = "/key/health"
306 # list routes
307 KEY_LIST = "/key/list"
308 KEY_ALIASES = "/key/aliases"
310 # team usage routes
311 TEAM_DAILY_ACTIVITY = "/team/daily/activity"
312 TEAM_DAILY_ACTIVITY_AGGREGATED = "/team/daily/activity/aggregated"
314 # team spend-log viewing
315 SPEND_LOGS = "/spend/logs"
316 SPEND_LOGS_V2 = "/spend/logs/v2"
319class LiteLLMRoutes(enum.Enum):
320 openai_route_names = [
321 "chat_completion",
322 "completion",
323 "embeddings",
324 "image_generation",
325 "video_generation",
326 "audio_transcriptions",
327 "moderations",
328 "model_list", # OpenAI /v1/models route
329 ]
330 openai_routes = [
331 # chat completions
332 "/engines/{model}/chat/completions",
333 "/openai/deployments/{model}/chat/completions",
334 "/chat/completions",
335 "/v1/chat/completions",
336 "/cursor/chat/completions",
337 "/cursor/models",
338 "/cursor/v1/models",
339 # completions
340 "/engines/{model}/completions",
341 "/openai/deployments/{model}/completions",
342 "/completions",
343 "/v1/completions",
344 # embeddings
345 "/engines/{model}/embeddings",
346 "/openai/deployments/{model}/embeddings",
347 "/embeddings",
348 "/v1/embeddings",
349 # image generation
350 "/images/generations",
351 "/v1/images/generations",
352 # image edit
353 "/images/edits",
354 "/v1/images/edits",
355 # video generation
356 "/videos",
357 "/v1/videos",
358 "/videos/{video_id}",
359 "/v1/videos/{video_id}",
360 "/videos/{video_id}/content",
361 "/v1/videos/{video_id}/content",
362 "/videos/{video_id}/remix",
363 "/v1/videos/{video_id}/remix",
364 # audio transcription
365 "/audio/transcriptions",
366 "/v1/audio/transcriptions",
367 # audio Speech
368 "/audio/speech",
369 "/v1/audio/speech",
370 # moderations
371 "/moderations",
372 "/v1/moderations",
373 # batches
374 "/v1/batches",
375 "/batches",
376 "/v1/batches/{batch_id}",
377 "/batches/{batch_id}",
378 "/v1/batches/{batch_id}/cancel",
379 "/batches/{batch_id}/cancel",
380 # files
381 "/v1/files",
382 "/files",
383 "/v1/files/{file_id}",
384 "/files/{file_id}",
385 "/v1/files/{file_id}/content",
386 "/files/{file_id}/content",
387 # fine_tuning
388 "/fine_tuning/jobs",
389 "/v1/fine_tuning/jobs",
390 "/fine_tuning/jobs/{fine_tuning_job_id}/cancel",
391 "/v1/fine_tuning/jobs/{fine_tuning_job_id}/cancel",
392 # assistants-related routes
393 "/assistants",
394 "/v1/assistants",
395 "/v1/assistants/{assistant_id}",
396 "/assistants/{assistant_id}",
397 "/threads",
398 "/v1/threads",
399 "/threads/{thread_id}",
400 "/v1/threads/{thread_id}",
401 "/threads/{thread_id}/messages",
402 "/v1/threads/{thread_id}/messages",
403 "/threads/{thread_id}/runs",
404 "/v1/threads/{thread_id}/runs",
405 # models
406 "/models",
407 "/v1/models",
408 # token counter
409 "/utils/token_counter",
410 "/utils/model_info",
411 "/utils/transform_request",
412 # rerank
413 "/rerank",
414 "/v1/rerank",
415 "/v2/rerank",
416 # realtime
417 "/realtime",
418 "/v1/realtime",
419 "/openai/v1/realtime",
420 "/realtime?{model}",
421 "/v1/realtime?{model}",
422 "/openai/v1/realtime?{model}",
423 # realtime (GA WebRTC HTTP routes)
424 "/realtime/client_secrets",
425 "/v1/realtime/client_secrets",
426 "/openai/v1/realtime/client_secrets",
427 "/realtime/calls",
428 "/v1/realtime/calls",
429 "/openai/v1/realtime/calls",
430 "/realtime/transcription_sessions",
431 "/v1/realtime/transcription_sessions",
432 "/openai/v1/realtime/transcription_sessions",
433 # responses API
434 "/responses",
435 "/v1/responses",
436 "/openai/v1/responses",
437 "/responses/{response_id}",
438 "/v1/responses/{response_id}",
439 "/openai/v1/responses/{response_id}",
440 "/responses/{response_id}/input_items",
441 "/v1/responses/{response_id}/input_items",
442 "/openai/v1/responses/{response_id}/input_items",
443 "/responses/{response_id}/cancel",
444 "/v1/responses/{response_id}/cancel",
445 "/openai/v1/responses/{response_id}/cancel",
446 "/responses/input_tokens",
447 "/v1/responses/input_tokens",
448 "/openai/v1/responses/input_tokens",
449 # vector stores
450 "/vector_stores",
451 "/v1/vector_stores",
452 "/vector_stores/{vector_store_id}",
453 "/v1/vector_stores/{vector_store_id}",
454 "/vector_stores/{vector_store_id}/search",
455 "/v1/vector_stores/{vector_store_id}/search",
456 "/vector_stores/{vector_store_id}/files",
457 "/v1/vector_stores/{vector_store_id}/files",
458 "/vector_stores/{vector_store_id}/files/{file_id}",
459 "/v1/vector_stores/{vector_store_id}/files/{file_id}",
460 "/vector_stores/{vector_store_id}/files/{file_id}/content",
461 "/v1/vector_stores/{vector_store_id}/files/{file_id}/content",
462 "/vector_store/list",
463 "/v1/vector_store/list",
464 # search
465 "/search",
466 "/v1/search",
467 "/search/{search_tool_name}",
468 "/v1/search/{search_tool_name}",
469 "/decisions",
470 "/v1/decisions",
471 "/systemone",
472 "/v1/systemone",
473 # OCR
474 "/ocr",
475 "/v1/ocr",
476 # containers API
477 "/containers",
478 "/v1/containers",
479 "/containers/*",
480 "/v1/containers/*",
481 ]
483 mapped_pass_through_routes = [
484 "/bedrock",
485 "/comprehendmedical",
486 "/azure_speech",
487 "/transcribe",
488 "/vertex-ai",
489 "/vertex_ai",
490 "/cohere",
491 "/cursor",
492 "/gemini",
493 "/anthropic",
494 "/langfuse",
495 "/azure",
496 "/azure_ai",
497 "/openai",
498 "/openai_passthrough",
499 "/assemblyai",
500 "/eu.assemblyai",
501 "/tinyfish",
502 "/vllm",
503 "/mistral",
504 "/typesafe",
505 "/openrouter",
506 "/milvus",
507 "/gigachat",
508 "/watsonx",
509 "/nvidia_nim",
510 "/deepgram",
511 "/fal_ai",
512 ]
514 #########################################################
515 # e.g /vllm/*, anthropic/*, etc.
516 # allows using /anthropic/v1/messages, /vllm/v1/chat/completions, etc.
517 #########################################################
518 passthrough_routes_wildcard = [f"{route}/*" for route in mapped_pass_through_routes]
520 litellm_native_routes = [
521 "/rag/ingest",
522 "/v1/rag/ingest",
523 "/rag/query",
524 "/v1/rag/query",
525 ]
527 anthropic_routes = [
528 "/v1/messages",
529 "/v1/messages/count_tokens",
530 "/claude_code_gateway/v1/messages",
531 "/claude_code_gateway/v1/messages/count_tokens",
532 "/v1/skills",
533 "/v1/skills/{skill_id}",
534 "/claude-code/marketplace.json",
535 "/claude-code/plugins",
536 "/claude-code/plugins/{plugin_name}",
537 ]
539 # MCP tool-call / passthrough routes — data-plane. Gated by DISABLE_LLM_API_ENDPOINTS.
540 mcp_inference_routes = [
541 "/mcp",
542 "/mcp/",
543 "/mcp/sse/",
544 "/mcp/sse/messages",
545 "/mcp/sse/messages/",
546 "/mcp/proxy",
547 "/mcp/{subpath}",
548 "/mcp/tools",
549 "/mcp/tools/list",
550 "/mcp/tools/call",
551 "/mcp-rest/tools/list",
552 "/mcp-rest/tools/call",
553 "/v1/mcp/tools",
554 "/introspect",
555 "/token",
556 ]
558 # MCP server CRUD routes — control-plane. Gated by DISABLE_ADMIN_ENDPOINTS.
559 mcp_management_routes = [
560 "/v1/mcp/server",
561 "/v1/mcp/server/{path:path}",
562 "/v1/mcp/sessions",
563 ]
565 # Backwards-compat union — virtual keys may be configured with
566 # allowed_routes=["mcp_routes"], which should cover both halves.
567 mcp_routes = mcp_inference_routes + mcp_management_routes
569 # A2A agent invocation / discovery routes — data-plane. Gated by DISABLE_LLM_API_ENDPOINTS.
570 agent_inference_routes = (
571 "/agents",
572 "/a2a/{agent_id}",
573 "/a2a/{agent_id}/message/send",
574 "/a2a/{agent_id}/message/stream",
575 "/a2a/{agent_id}/.well-known/agent-card.json",
576 )
578 # Agent registry CRUD routes — control-plane. Gated by DISABLE_ADMIN_ENDPOINTS.
579 # The handlers in agent_endpoints/endpoints.py enforce proxy-admin on writes and
580 # scope reads by role, so these also appear in self_managed_routes.
581 agent_management_routes = (
582 "/v1/agents",
583 "/v1/agents/{agent_id}",
584 "/v1/agents/make_public",
585 "/v1/agents/{agent_id}/make_public",
586 "/v1/agents/{agent_id}/kill_switch",
587 )
589 # Backwards-compat union — virtual keys may be configured with
590 # allowed_routes=["agent_routes"], which should cover both halves.
591 agent_routes = agent_inference_routes + agent_management_routes
593 google_routes = [
594 "/v1beta/models/{model_name:path}:countTokens",
595 "/v1beta/models/{model_name:path}:generateContent",
596 "/v1beta/models/{model_name:path}:streamGenerateContent",
597 "/models/{model_name:path}:countTokens",
598 "/models/{model_name:path}:generateContent",
599 "/models/{model_name:path}:streamGenerateContent",
600 # Google Interactions API
601 "/interactions",
602 "/v1beta/interactions",
603 "/interactions/{interaction_id}",
604 "/v1beta/interactions/{interaction_id}",
605 "/interactions/{interaction_id}/cancel",
606 "/v1beta/interactions/{interaction_id}/cancel",
607 # Google Managed Agents API
608 "/v1beta/agents",
609 "/v1beta/agents/{name}",
610 "/v1beta/agents/{name}/versions",
611 ]
613 apply_guardrail_routes = [
614 "/guardrails/apply_guardrail",
615 ]
617 model_info_routes = [
618 "/model/info",
619 "/v1/model/info",
620 "/model_group/info",
621 ]
623 llm_api_routes = (
624 openai_routes
625 + anthropic_routes
626 + google_routes
627 + mapped_pass_through_routes
628 + passthrough_routes_wildcard
629 + apply_guardrail_routes
630 + mcp_inference_routes
631 + litellm_native_routes
632 + list(agent_inference_routes)
633 + model_info_routes
634 )
635 info_routes = [
636 "/key/info",
637 "/key/health",
638 "/team/info",
639 "/team/list",
640 "/v2/team/list",
641 "/organization/list",
642 "/team/available",
643 "/team/metadata_schema",
644 "/user/info",
645 "/v2/user/info",
646 "/model/info",
647 "/v1/model/info",
648 "/v2/model/info",
649 "/v2/key/info",
650 "/model_group/info",
651 "/health",
652 "/health/services",
653 "/key/list",
654 "/user/filter/ui",
655 "/models",
656 "/v1/models",
657 "/sso/get/ui_settings",
658 "/get/user_banner",
659 "/get/latest_release_info",
660 ]
662 # NOTE: ROUTES ONLY FOR MASTER KEY - only the Master Key should be able to Reset Spend
663 master_key_only_routes = [
664 "/global/spend/reset",
665 "/memory-usage-in-mem-cache",
666 "/memory-usage-in-mem-cache-items",
667 ]
669 key_management_routes = [
670 KeyManagementRoutes.KEY_GENERATE.value,
671 KeyManagementRoutes.KEY_UPDATE.value,
672 KeyManagementRoutes.KEY_DELETE.value,
673 KeyManagementRoutes.KEY_INFO.value,
674 KeyManagementRoutes.KEY_REGENERATE.value,
675 KeyManagementRoutes.KEY_GENERATE_SERVICE_ACCOUNT.value,
676 KeyManagementRoutes.KEY_REGENERATE_WITH_PATH_PARAM.value,
677 KeyManagementRoutes.KEY_LIST.value,
678 KeyManagementRoutes.KEY_BLOCK.value,
679 KeyManagementRoutes.KEY_UNBLOCK.value,
680 KeyManagementRoutes.KEY_BULK_UPDATE.value,
681 KeyManagementRoutes.TEAM_KEY_BULK_UPDATE.value,
682 KeyManagementRoutes.TEAM_DAILY_ACTIVITY.value,
683 KeyManagementRoutes.TEAM_DAILY_ACTIVITY_AGGREGATED.value,
684 KeyManagementRoutes.SPEND_LOGS.value,
685 KeyManagementRoutes.SPEND_LOGS_V2.value,
686 KeyManagementRoutes.KEY_RESET_SPEND.value,
687 KeyManagementRoutes.KEY_ALIASES.value,
688 KeyManagementRoutes.KEY_ACCESS_GROUP_ASSIGNMENT.value,
689 KeyManagementRoutes.AUTO_ROUTER_MANAGE.value,
690 ]
692 team_service_account_key_routes = (
693 KeyManagementRoutes.KEY_GENERATE.value,
694 KeyManagementRoutes.KEY_UPDATE.value,
695 )
697 management_routes = (
698 [
699 # user
700 "/user/new",
701 "/management/v1/users/bulk",
702 "/user/update",
703 "/user/bulk_update",
704 "/user/delete",
705 "/management/v1/users/bulk_delete",
706 "/user/info",
707 "/user/list",
708 "/user/daily/activity",
709 "/user/daily/activity/aggregated",
710 # team
711 "/team/new",
712 "/team/update",
713 "/team/{team_id}",
714 "/team/delete",
715 "/team/list",
716 "/v2/team/list",
717 "/team/info",
718 "/team/block",
719 "/team/unblock",
720 "/team/available",
721 "/team/metadata_schema",
722 "/team/permissions_list",
723 "/team/permissions_update",
724 "/team/permissions_bulk_update",
725 "/team/daily/activity",
726 "/team/daily/activity/aggregated",
727 "/team/spend/by_user",
728 # gateway request counts (SGR); deployment-wide, admin-only
729 "/gateway/daily/activity",
730 # model
731 "/model/new",
732 "/model/update",
733 "/model/delete",
734 "/model/info",
735 "/jwt/key/mapping/new",
736 "/jwt/key/mapping/update",
737 "/jwt/key/mapping/delete",
738 "/jwt/key/mapping/list",
739 "/jwt/key/mapping/info",
740 ]
741 + key_management_routes
742 + mcp_management_routes
743 + list(agent_management_routes)
744 )
746 spend_tracking_routes = [
747 # spend
748 "/spend/keys",
749 "/spend/users",
750 "/spend/tags",
751 "/spend/calculate",
752 "/spend/logs",
753 "/spend/logs/v2",
754 "/spend/logs/ui",
755 "/spend/logs/ui/{request_id}",
756 "/spend/logs/session/ui",
757 "/key/spend/report",
758 "/user/spend/report",
759 "/team/spend/report",
760 "/organization/spend/report",
761 # Reads end users out of spend logs, scoped to the caller's own rows and
762 # permitted teams exactly like /spend/logs/ui — it belongs to the same
763 # access tier, not to customer management.
764 "/management/v1/spend_logs/end_users",
765 "/management/v1/spend_logs/users",
766 "/cost/estimate",
767 ]
769 global_spend_tracking_routes = [
770 # global spend
771 "/global/spend/logs",
772 "/global/spend",
773 "/global/spend/keys",
774 "/global/spend/teams",
775 "/global/spend/end_users",
776 "/global/spend/models",
777 "/global/predict/spend/logs",
778 "/global/spend/report",
779 "/global/spend/provider",
780 "/global/spend/tags",
781 "/global/spend/all_tag_names",
782 "/spend/capture_rate",
783 ]
785 public_routes = frozenset(
786 (
787 "/routes",
788 "/",
789 "/health/liveliness",
790 "/health/liveness",
791 "/test",
792 "/config/yaml",
793 "/litellm/.well-known/litellm-ui-config",
794 "/.well-known/litellm-ui-config",
795 "/public/model_hub",
796 "/public/v1/model_hub",
797 "/public/v1/model_hub/providers",
798 "/public/v1/model_hub/modes",
799 "/public/v1/model_hub/features",
800 "/public/model_hub/info",
801 "/public/agent_hub",
802 "/public/mcp_hub",
803 "/public/skill_hub",
804 "/public/litellm_model_cost_map",
805 )
806 )
808 # Retained for backwards compatibility with JWT auth configs that reference
809 # "ui_routes" in admin_allowed_routes. Not used by the proxy's own route
810 # authorization — UI tokens now go through the same RBAC path as API tokens.
811 ui_routes = [
812 "/sso",
813 "/sso/get/ui_settings",
814 "/get/ui_settings",
815 "/login",
816 "/key/info",
817 "/config",
818 "/spend",
819 "/model/info",
820 "/v2/model/info",
821 "/v2/key/info",
822 "/models",
823 "/v1/models",
824 "/global/spend",
825 "/global/spend/logs",
826 "/global/spend/keys",
827 "/global/spend/models",
828 "/global/spend/tags",
829 "/global/predict/spend/logs",
830 "/global/activity",
831 "/gateway/daily/activity",
832 "/health/services",
833 ] + info_routes
835 # Stateless validators on caller-supplied log data; source logs are
836 # already accessible via spend_tracking_routes, so no scope expansion.
837 compliance_check_routes = [
838 "/compliance/eu-ai-act",
839 "/compliance/gdpr",
840 ]
842 # Routes in `global_spend_tracking_routes` return proxy-wide spend across
843 # every team, customer, and api_key. They are intentionally NOT included
844 # here — non-admin roles must not see other tenants' spend. Admin roles go
845 # through their own branches in `route_checks.py`, and a key minted with
846 # the `get_spend_routes` permission retains explicit opt-in access.
847 internal_user_routes = (
848 [
849 "/global/activity",
850 "/global/activity/model",
851 "/global/activity/cache_hits",
852 # Tag usage endpoints scope internal users to tags produced by
853 # their own keys in tag_management_endpoints.py.
854 "/tag/daily/activity",
855 "/tag/list",
856 "/v1/models/{model_id}",
857 "/models/{model_id}",
858 "/guardrails/list",
859 "/v2/guardrails/list",
860 "/project/list",
861 "/project/info",
862 # Read-only search tool routes power the Search Tools UI page.
863 # Create/update/delete and test_connection stay admin-only.
864 "/search_tools/list",
865 "/search_tools/ui/available_providers",
866 ]
867 + spend_tracking_routes
868 + key_management_routes
869 + compliance_check_routes
870 )
872 internal_user_view_only_routes = (
873 spend_tracking_routes
874 + compliance_check_routes
875 + [
876 # Tag usage endpoints scope internal viewers to tags produced by
877 # their own keys in tag_management_endpoints.py.
878 "/tag/daily/activity",
879 "/tag/list",
880 ]
881 )
883 self_managed_routes = [
884 # update_team resolves proxy/org/team admin itself and filters team admins
885 # through the team_admin_editable_team_fields setting
886 "/team/update",
887 "/team/member_add",
888 "/team/member_delete",
889 "/management/v1/teams/{team_id}/members/bulk_delete",
890 "/management/v1/teams/{team_id}/members/bulk_update",
891 "/team/member_update",
892 "/team/{team_id}/member/{user_id}/reset_spend",
893 "/team/{team_id}/member/{user_id}/reset_budget",
894 "/team/permissions_list",
895 "/team/permissions_update",
896 "/team/daily/activity",
897 "/team/daily/activity/aggregated",
898 "/team/spend/by_user",
899 "/team/{team_id}/members/me",
900 # POST/GET the team's logging callbacks, and DELETE one of them. Every
901 # handler calls _verify_team_access, which admits only a proxy admin, an
902 # org admin for the team, or an admin of this team.
903 #
904 # team_id is a free-form string, so it spells these with the same path
905 # converter the router uses; the gate matches that converter.
906 "/team/{team_id:path}/callback",
907 "/team/{team_id:path}/callback/{callback_name}",
908 "/model/new",
909 "/model/update",
910 "/model/delete",
911 "/user/daily/activity",
912 "/user/daily/activity/aggregated",
913 # Endpoint restricts results to organizations the caller is ORG_ADMIN
914 # of; a caller who administers none gets an empty result set.
915 "/organization/daily/activity",
916 "/user/available_roles", # read-only role metadata; any authenticated user may read
917 # Claude Code gateway: the signed-in CLI fetches its managed settings and posts its own telemetry
918 "/claude_code_gateway/managed/settings",
919 "/claude_code_gateway/v1/metrics",
920 "/claude_code_gateway/v1/logs",
921 "/claude_code_gateway/v1/traces",
922 "/user/list", # org admins checked in endpoint; non-admins get 403
923 "/management/v1/users/bulk_delete", # proxy admins delete anyone, org admins only their orgs' users; others 403
924 "/user/password/change", # endpoint only ever writes the caller's own row
925 "/session/logout", # endpoint only ever revokes the caller's own session key
926 "/model/{model_id}/update",
927 "/prompt/list",
928 "/prompt/info",
929 "/vector_store/info",
930 # Project read routes - endpoint scopes results to caller's teams (non-admin)
931 "/project/list",
932 "/project/info",
933 # Project write routes - endpoint checks team admin + team_admin_editable_team_fields "projects"
934 "/project/new",
935 "/project/update",
936 # Endpoint enforces proxy-admin vs team-admin model access itself.
937 "/health/test_connection",
938 # Invitation routes - org/team admins checked in endpoint via _user_has_admin_privileges
939 "/invitation/new",
940 "/invitation/delete",
941 # Team guardrail submission - requires team-scoped key; endpoint enforces team_id
942 "/guardrails/register",
943 # Team guardrail submissions - endpoint scopes results to caller's teams (non-admin)
944 "/guardrails/submissions",
945 "/guardrails/submissions/{guardrail_id}",
946 # Auto-router dry runs - both gate like the /model/new write they rehearse:
947 # proxy admin, or team admin naming their own team via team_id
948 "/auto_router/test_routing",
949 "/auto_router/validate_complexity_router_config",
950 "/auto_router/availability",
951 # Per-session auto-router read - the endpoint scopes the row to the caller's own key hash
952 "/auto_router/session",
953 "/cost/predict-cache",
954 # Agent registry - reads are role-scoped and writes are proxy-admin-gated
955 # inside agent_endpoints/endpoints.py
956 *agent_management_routes,
957 ] # routes that manage their own allowed/disallowed logic
959 ## Org Admin Routes ##
961 # Routes only an Org Admin Can Access
962 org_admin_only_routes = [
963 "/organization/info",
964 "/organization/delete",
965 "/organization/member_add",
966 "/organization/member_update",
967 # member_delete is equally destructive as member_add / member_update
968 # and must be scoped the same way — otherwise it falls through to
969 # the management_routes / self_managed_routes path and lets any
970 # non-PROXY_ADMIN caller that reaches the route delete arbitrary
971 # org memberships without the organization_role_based_access_check
972 # that member_add / member_update trigger.
973 "/organization/member_delete",
974 ]
976 # Routes accessible by Admin Viewer (read-only admin access).
977 #
978 # Admin Viewer follows a read-parity-with-Proxy-Admin rule: anything Proxy
979 # Admin can read/list/get, Admin Viewer can too (no writes, no cost-incurring
980 # actions).
981 #
982 # NOTE: This list is no longer the primary mechanism for granting access —
983 # `_check_proxy_admin_viewer_access()` in route_checks.py default-allows
984 # any safe HTTP method (GET/HEAD/OPTIONS) on non-inference routes. This
985 # list now matters only for non-GET routes that are semantically reads
986 # (e.g. POST /spend/calculate). Adding a new GET endpoint does not require
987 # updating this list — the default-allow behavior covers it automatically.
988 admin_viewer_routes = (
989 [
990 "/user/list",
991 "/user/available_users",
992 "/user/available_roles",
993 "/user/daily/activity",
994 "/team/daily/activity",
995 "/team/daily/activity/aggregated",
996 "/tag/daily/activity",
997 "/tag/list",
998 "/audit",
999 "/audit/{id}",
1000 "/global/activity",
1001 "/global/activity/model",
1002 "/global/activity/cache_hits",
1003 # Customer / end-user listing (handlers already gate on
1004 # PROXY_ADMIN_VIEW_ONLY — the route gate must match).
1005 "/customer/list",
1006 "/customer/info",
1007 # UI Logs page session detail drawer and the end-user filter facet.
1008 # The list endpoint `/spend/logs/ui` and the single-log detail route
1009 # `/spend/logs/ui/{request_id}` are covered via spend_tracking_routes
1010 # below.
1011 "/spend/logs/session/ui",
1012 "/management/v1/spend_logs/end_users",
1013 "/management/v1/spend_logs/users",
1014 # Settings / observability read endpoints exposed in admin-only
1015 # sidebar groups (Logging & Alerts, Admin Settings, Budgets,
1016 # Invitations).
1017 "/callbacks/list",
1018 "/callbacks/configs",
1019 "/get/config/callbacks",
1020 "/alerting/settings",
1021 "/config/list",
1022 "/config/field/info",
1023 "/budget/list",
1024 "/management/v1/budgets",
1025 "/budget/settings",
1026 # Invitation viewing (admin viewer cannot create/delete; can read).
1027 "/invitation/info",
1028 # Guardrails / Policies pages (read-only views).
1029 "/guardrails/list",
1030 "/v2/guardrails/list",
1031 "/guardrails/submissions",
1032 "/guardrails/submissions/{guardrail_id}",
1033 "/guardrails/usage/overview",
1034 "/policies/attachments/list",
1035 # MCP semantic filter settings (read).
1036 "/get/mcp_semantic_filter_settings",
1037 # Model cost map maintenance views (read-only status / source).
1038 "/schedule/model_cost_map_reload/status",
1039 "/model/cost_map/source",
1040 # A pure read; POST only so the prompt does not ride in a URL.
1041 "/auto_router/classifier/default_prompt",
1042 ]
1043 # Spend tracking reads (/spend/logs, /spend/logs/ui, /spend/keys,
1044 # /spend/users, /spend/tags, /spend/calculate, /cost/estimate). Admin
1045 # Viewer can already read /global/spend/* via global_spend_tracking_routes;
1046 # the per-tenant /spend/* views were the missing peer.
1047 + spend_tracking_routes
1048 + info_routes
1049 )
1051 # All routes accesible by an Org Admin
1052 org_admin_allowed_routes = org_admin_only_routes + management_routes + self_managed_routes + admin_viewer_routes
1055class LiteLLMPromptInjectionParams(LiteLLMPydanticObjectBase):
1056 heuristics_check: bool = False
1057 vector_db_check: bool = False
1058 llm_api_check: bool = False
1059 llm_api_name: str | None = None
1060 llm_api_system_prompt: str | None = None
1061 llm_api_fail_call_string: str | None = None
1062 reject_as_response: bool | None = Field(
1063 default=False,
1064 description="Return rejected request error message as a string to the user. Default behaviour is to raise an exception.",
1065 )
1067 @model_validator(mode="before")
1068 @classmethod
1069 def check_llm_api_params(cls, values):
1070 llm_api_check: Final = values.get("llm_api_check")
1071 if llm_api_check is True:
1072 if "llm_api_name" not in values or not values["llm_api_name"]:
1073 raise ValueError("If llm_api_check is set to True, llm_api_name must be provided")
1074 if "llm_api_system_prompt" not in values or not values["llm_api_system_prompt"]:
1075 raise ValueError("If llm_api_check is set to True, llm_api_system_prompt must be provided")
1076 if "llm_api_fail_call_string" not in values or not values["llm_api_fail_call_string"]:
1077 raise ValueError("If llm_api_check is set to True, llm_api_fail_call_string must be provided")
1078 return values
1081######### Request Class Definition ######
1082class ProxyChatCompletionRequest(LiteLLMPydanticObjectBase):
1083 """
1084 Pydantic model for chat completion requests that includes both OpenAI standard fields
1085 and LiteLLM-specific parameters. This replaces the previous TypedDict version.
1086 """
1088 # Required fields (from ChatCompletionRequest)
1089 model: str
1090 messages: list[AllMessageValues]
1092 # Standard OpenAI completion parameters (all optional)
1093 frequency_penalty: float | None = None
1094 logit_bias: dict[str, float] | None = None
1095 logprobs: bool | None = None
1096 top_logprobs: int | None = None
1097 max_tokens: int | None = None
1098 n: int | None = None
1099 presence_penalty: float | None = None
1100 response_format: dict[str, Any] | None = None
1101 seed: int | None = None
1102 service_tier: str | None = None
1103 stop: str | list[str] | None = None
1104 stream_options: dict[str, Any] | None = None
1105 temperature: float | None = None
1106 top_p: float | None = None
1107 tools: list[dict[str, Any]] | None = None
1108 tool_choice: str | dict[str, Any] | None = None
1109 parallel_tool_calls: bool | None = None
1110 function_call: str | dict[str, Any] | None = None
1111 functions: list[dict[str, Any]] | None = None
1112 user: str | None = None
1113 stream: bool | None = None
1115 # LiteLLM-specific metadata param (from original ChatCompletionRequest)
1116 metadata: dict[str, Any] | None = None
1118 # Optional LiteLLM params
1119 guardrails: list[str] | None = None
1120 caching: bool | None = None
1121 num_retries: int | None = None
1122 context_window_fallback_dict: dict[str, str] | None = None
1123 fallbacks: list[str] | None = None
1126class ModelInfoDelete(LiteLLMPydanticObjectBase):
1127 id: str
1130class ModelInfo(LiteLLMPydanticObjectBase):
1131 id: str | None
1132 mode: Literal["embedding", "chat", "completion"] | None
1133 input_cost_per_token: float | None = 0.0
1134 output_cost_per_token: float | None = 0.0
1135 max_tokens: int | None = 2048 # assume 2048 if not set
1137 # for azure models we need users to specify the base model, one azure you can call deployments - azure/my-random-model
1138 # we look up the base model in model_prices_and_context_window.json
1139 base_model: (
1140 Literal[
1141 "gpt-4-1106-preview", "gpt-4-32k", "gpt-4", "gpt-3.5-turbo-16k", "gpt-3.5-turbo", "text-embedding-ada-002"
1142 ]
1143 | None
1144 )
1145 discoverable: bool | None = None
1147 model_config = ConfigDict(protected_namespaces=(), extra="allow")
1149 @model_validator(mode="before")
1150 @classmethod
1151 def set_model_info(cls, values):
1152 if values.get("id") is None:
1153 values.update({"id": str(uuid.uuid4())})
1154 if values.get("mode") is None:
1155 values.update({"mode": None})
1156 if values.get("input_cost_per_token") is None:
1157 values.update({"input_cost_per_token": None})
1158 if values.get("output_cost_per_token") is None:
1159 values.update({"output_cost_per_token": None})
1160 if values.get("max_tokens") is None:
1161 values.update({"max_tokens": None})
1162 if values.get("base_model") is None:
1163 values.update({"base_model": None})
1164 return values
1167class ProviderInfo(LiteLLMPydanticObjectBase):
1168 name: str
1169 fields: list[ProviderField]
1172class BlockUsers(LiteLLMPydanticObjectBase):
1173 user_ids: list[str] # required
1176class ModelParams(LiteLLMPydanticObjectBase):
1177 model_name: str
1178 litellm_params: dict
1179 model_info: ModelInfo
1181 model_config = ConfigDict(protected_namespaces=())
1183 @model_validator(mode="before")
1184 @classmethod
1185 def set_model_info(cls, values):
1186 if values.get("model_info") is None:
1187 values.update({"model_info": ModelInfo(id=None, mode="chat", base_model=None)})
1188 return values
1191class LiteLLM_ObjectPermissionBase(LiteLLMPydanticObjectBase):
1192 mcp_servers: list[str] | None = None
1193 mcp_access_groups: list[str] | None = None
1194 mcp_tool_permissions: dict[str, list[str]] | None = None
1195 mcp_toolsets: list[str] | None = None
1196 blocked_tools: list[str] | None = None
1197 vector_stores: list[str] | None = None
1198 agents: list[str] | None = None
1199 agent_access_groups: list[str] | None = None
1200 models: list[str] | None = None
1201 search_tools: list[str] | None = None
1202 mcp_tool_search_enabled: bool | None = None
1203 skills: list[str] | None = None
1206from litellm.models.team import BudgetLimitEntry as BudgetLimitEntry # noqa: E402
1207from litellm.types.object_permission import ( # noqa: E402
1208 ObjectPermissionDict as ObjectPermissionDict,
1209)
1212class GenerateRequestBase(LiteLLMPydanticObjectBase):
1213 """
1214 Overlapping schema between key and user generate/update requests
1215 """
1217 key_alias: str | None = None
1218 duration: str | None = None
1219 models: list | None = []
1220 spend: float | None = 0
1221 max_budget: float | None = None
1222 user_id: str | None = None
1223 team_id: str | None = None
1224 agent_id: str | None = None
1225 max_parallel_requests: int | None = None
1226 metadata: dict | None = {}
1227 tpm_limit: int | None = None
1228 rpm_limit: int | None = None
1230 budget_duration: str | None = None
1231 budget_limits: list[BudgetLimitEntry] | None = None # multiple concurrent budget windows
1232 allowed_cache_controls: list | None = []
1233 config: dict | None = {}
1234 permissions: dict | None = {}
1235 model_max_budget: dict | None = {} # {"gpt-4": 5.0, "gpt-3.5-turbo": 5.0}, defaults to {}
1236 budget_fallbacks: dict[str, list[str]] | None = None
1238 model_config = ConfigDict(protected_namespaces=())
1239 model_rpm_limit: dict | None = None
1240 model_tpm_limit: dict | None = None
1241 mcp_rpm_limit: dict[str, int] | None = None
1242 tag_rpm_limit: dict[str, int] | None = None
1243 guardrails: list[str] | None = None
1244 policies: list[str] | None = None
1245 prompts: list[str] | None = None
1246 blocked: bool | None = None
1247 aliases: dict | None = {}
1248 object_permission: LiteLLM_ObjectPermissionBase | None = None
1250 @field_validator("max_budget", mode="before")
1251 @classmethod
1252 def check_max_budget(cls, v):
1253 if v == "": 1253 ↛ 1254line 1253 didn't jump to line 1254 because the condition on line 1253 was never true
1254 return None
1255 return v
1258class AllowedVectorStoreIndexItem(LiteLLMPydanticObjectBase):
1259 index_name: str
1260 index_permissions: list[Literal["read", "write"]]
1263class KeyRequestBase(GenerateRequestBase):
1264 key: str | None = None
1265 tpd_limit: int | None = None
1266 default_estimated_output_tokens: PositiveInt | None = None
1267 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None
1268 budget_id: str | None = None
1269 end_user_budget_id: str | None = None
1270 tags: list[str] | None = None
1271 disable_global_guardrails: bool | None = None
1272 enable_prompt_caching: bool | None = None
1273 throttle_on_budget_exceeded: bool | None = None
1274 enforced_params: list[str] | None = None
1275 allowed_routes: list | None = []
1276 allowed_passthrough_routes: list | None = None
1277 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None
1278 rpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput", "dynamic"] | None = (
1279 None # raise an error if 'guaranteed_throughput' is set and we're overallocating rpm
1280 )
1281 tpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput", "dynamic"] | None = (
1282 None # raise an error if 'guaranteed_throughput' is set and we're overallocating tpm
1283 )
1284 router_settings: UpdateRouterConfig | None = None
1285 access_group_ids: list[str] | None = None
1288class LiteLLMKeyType(str, enum.Enum):
1289 """
1290 Enum for key types that determine what routes a key can access
1291 """
1293 LLM_API = "llm_api" # Can call LLM API routes (chat/completions, embeddings, etc.)
1294 MANAGEMENT = "management" # Can call management routes (user/team/key management)
1295 READ_ONLY = "read_only" # Can only call info/read routes
1296 DEFAULT = "default" # Uses default allowed routes
1299class GenerateKeyRequest(KeyRequestBase):
1300 soft_budget: float | None = None
1301 send_invite_email: bool | None = None
1302 key_type: LiteLLMKeyType | None = Field(
1303 default=LiteLLMKeyType.DEFAULT,
1304 description="Type of key that determines default allowed routes.",
1305 )
1306 auto_rotate: bool | None = Field(default=False, description="Whether this key should be automatically rotated")
1307 rotation_interval: str | None = Field(
1308 default=None,
1309 description="How often to rotate this key (e.g., '30d', '90d'). Required if auto_rotate=True",
1310 )
1311 organization_id: str | None = None
1312 project_id: str | None = None
1314 @field_validator("team_id", "organization_id", "project_id", mode="before")
1315 @classmethod
1316 def treat_cleared_id_as_unset(cls, v: object) -> object:
1317 if v == "":
1318 return None
1319 return v
1322class GenerateKeyResponse(KeyRequestBase):
1323 key: str
1324 key_name: str | None = None
1325 key_type: str | None = None
1326 expires: datetime | None = None
1327 user_id: str | None = None
1328 token_id: str | None = None
1329 organization_id: str | None = None
1330 project_id: str | None = None
1331 litellm_budget_table: Any | None = None
1332 token: str | None = None
1333 created_by: str | None = None
1334 updated_by: str | None = None
1335 created_at: datetime | None = None
1336 updated_at: datetime | None = None
1338 @model_validator(mode="before")
1339 @classmethod
1340 def set_model_info(cls, values):
1341 if values.get("token") is not None:
1342 values.update({"key": values.get("token")})
1343 dict_fields: Final = [
1344 "metadata",
1345 "aliases",
1346 "config",
1347 "permissions",
1348 "model_max_budget",
1349 "budget_fallbacks",
1350 "router_settings",
1351 "budget_limits",
1352 ]
1353 for field in dict_fields:
1354 value = values.get(field)
1355 if value is not None and isinstance(value, str):
1356 try:
1357 values[field] = json.loads(value)
1358 except json.JSONDecodeError:
1359 raise ValueError(f"Field {field} should be a valid dictionary")
1361 return values
1364class UpdateKeyRequest(KeyRequestBase):
1365 # Note: the defaults of all Params here MUST BE NONE
1366 # else they will get overwritten
1367 duration: str | None = None
1368 spend: float | None = None
1369 soft_budget: float | None = None
1370 metadata: dict | None = None
1371 temp_budget_increase: float | None = None
1372 temp_budget_expiry: datetime | None = None
1373 auto_rotate: bool | None = None
1374 rotation_interval: str | None = None
1375 organization_id: str | None = None
1377 project_id: str | None = Field(
1378 default=None,
1379 description="Omit to retain the project, or send null to detach. Assigning a different project is not supported.",
1380 )
1382 @model_validator(mode="before")
1383 @classmethod
1384 def drop_blank_team_id(cls, values: object) -> object:
1385 if isinstance(values, Mapping) and values.get("team_id") == "":
1386 return MappingProxyType({k: v for k, v in values.items() if k != "team_id"})
1387 return values
1389 @field_validator("organization_id", mode="before")
1390 @classmethod
1391 def treat_cleared_organization_id_as_unset(cls, v: object) -> object:
1392 if v == "":
1393 return None
1394 return v
1396 @model_validator(mode="after")
1397 def validate_temp_budget(self) -> "UpdateKeyRequest":
1398 if self.temp_budget_increase is not None or self.temp_budget_expiry is not None:
1399 if self.temp_budget_increase is None or self.temp_budget_expiry is None:
1400 raise ValueError("temp_budget_increase and temp_budget_expiry must be set together")
1401 return self
1403 @model_validator(mode="after")
1404 def validate_key_identifier(self) -> "UpdateKeyRequest":
1405 if self.key is None and self.key_alias is None:
1406 raise ValueError("either key or key_alias must be provided")
1407 return self
1410class RegenerateKeyRequest(GenerateKeyRequest):
1411 # This needs to be different from UpdateKeyRequest, because "key" is optional for this
1412 key: str | None = None
1413 new_key: str | None = None
1414 duration: str | None = None
1415 spend: float | None = None
1416 metadata: dict | None = None
1417 new_master_key: str | None = None
1418 grace_period: str | None = None # Duration to keep old key valid (e.g. "24h", "2d"); None = immediate revoke
1421class ResetSpendRequest(LiteLLMPydanticObjectBase):
1422 reset_to: float
1424 @field_validator("reset_to", mode="before")
1425 @classmethod
1426 def reject_bool_reset_to(cls, v):
1427 # bool is a subclass of int, so pydantic silently coerces True/False into
1428 # 1.0/0.0 for a `float` field: a caller who accidentally sends a boolean
1429 # would otherwise get an unintended spend reset instead of a 422.
1430 if isinstance(v, bool):
1431 raise ValueError("reset_to must be a number, not a boolean") # noqa: TRY004 # pydantic needs ValueError
1432 return v
1435class KeyRequest(LiteLLMPydanticObjectBase):
1436 keys: list[str] | None = None
1437 key_aliases: list[str] | None = None
1439 @model_validator(mode="before")
1440 @classmethod
1441 def validate_at_least_one(cls, values):
1442 if not values.get("keys") and not values.get("key_aliases"):
1443 raise ValueError("At least one of 'keys' or 'key_aliases' must be provided.")
1444 return values
1447from litellm.models.model import ( # noqa: E402
1448 LiteLLM_ProxyModelTable as LiteLLM_ProxyModelTable,
1449)
1450from litellm.models.team import LiteLLM_ModelTable as LiteLLM_ModelTable # noqa: E402
1453# MCP Types
1454class SpecialMCPServerName(str, enum.Enum):
1455 all_team_servers = "all-team-mcpservers"
1456 all_proxy_servers = "all-proxy-mcpservers"
1459class MCPApprovalStatus(str, enum.Enum):
1460 pending_review = "pending_review"
1461 active = "active"
1462 rejected = "rejected"
1463 # Short-lived row backing the admin OAuth "Authorize & Fetch Token" flow. Never served: the
1464 # registry loader and every listing exclude it, so it is reachable only by its own server_id.
1465 draft = "draft"
1468from litellm.models.mcp_server import ( # noqa: E402
1469 MCPEnvVar as MCPEnvVar,
1470)
1471from litellm.models.mcp_server import ( # noqa: E402
1472 MCPEnvVarScope as MCPEnvVarScope,
1473)
1476# MCP Proxy Request Types
1477def _dcr_bridge_auth_type_error(auth_type: object) -> ValueError:
1478 return ValueError(
1479 f"dcr_bridge is only supported for auth_type true_passthrough or oauth_delegate (got {auth_type!r}). "
1480 "The DCR bridge serves gateway-hosted OAuth discovery for the client-forwarded token modes; "
1481 "interactive oauth2 servers already run the gateway authorization-code flow."
1482 )
1485def _per_server_oauth_discovery_error() -> ValueError:
1486 return ValueError(
1487 "per_server_oauth_discovery is only supported for auth_type oauth2 with oauth2_flow "
1488 "authorization_code and without delegate_auth_to_upstream."
1489 )
1492def is_per_server_oauth_discovery_eligible(
1493 auth_type: object, oauth2_flow: object, delegate_auth_to_upstream: object
1494) -> bool:
1495 return auth_type == MCPAuth.oauth2 and oauth2_flow == "authorization_code" and not delegate_auth_to_upstream
1498def _reject_unsupported_per_server_oauth_discovery(values: object, require_auth_type: bool) -> None:
1499 """Partial updates may omit eligibility fields; those are checked against the stored row by the
1500 update endpoint. Every field the payload does carry must be eligible on its own."""
1501 if not isinstance(values, dict) or not values.get("per_server_oauth_discovery"):
1502 return
1503 auth_type_ok: Final = values.get("auth_type") == MCPAuth.oauth2 or (
1504 not require_auth_type and "auth_type" not in values
1505 )
1506 oauth2_flow_ok: Final = values.get("oauth2_flow") == "authorization_code" or (
1507 not require_auth_type and "oauth2_flow" not in values
1508 )
1509 if auth_type_ok and oauth2_flow_ok and not values.get("delegate_auth_to_upstream"):
1510 return
1511 raise _per_server_oauth_discovery_error()
1514def _validate_mcp_transport_fields(values: object) -> None:
1515 if not isinstance(values, dict):
1516 return
1517 transport: Final = values.get("transport")
1518 if transport in (MCPTransport.http, MCPTransport.sse):
1519 if not values.get("url") and not values.get("spec_path"):
1520 raise ValueError("url or spec_path is required for HTTP/SSE transport")
1521 return
1522 if transport != MCPTransport.stdio:
1523 return
1524 if not is_mcp_stdio_enabled(): 1524 ↛ 1526line 1524 didn't jump to line 1526 because the condition on line 1524 was always true
1525 raise ValueError(MCP_STDIO_DISABLED_MESSAGE)
1526 command: Final = values.get("command")
1527 if not command:
1528 raise ValueError("command is required for stdio transport")
1529 if not values.get("args"):
1530 raise ValueError("args is required for stdio transport")
1531 if os.path.basename(str(command)) not in MCP_STDIO_ALLOWED_COMMANDS:
1532 raise ValueError(
1533 f"Command '{command}' is not in the allowed commands list "
1534 f"for stdio transport. Allowed commands: {sorted(MCP_STDIO_ALLOWED_COMMANDS)}"
1535 )
1538class NewMCPServerRequest(LiteLLMPydanticObjectBase):
1539 server_id: str | None = None
1540 server_name: str | None = None
1541 alias: str | None = None
1542 description: str | None = None
1543 transport: MCPTransportType = MCPTransport.sse
1544 auth_type: MCPAuthType | None = None
1545 credentials: MCPCredentials | None = None
1546 url: str | None = None
1547 spec_path: str | None = None
1548 mcp_info: MCPInfo | None = None
1549 mcp_access_groups: list[str] = Field(default_factory=list)
1550 allowed_tools: list[str] | None = None
1551 tool_name_to_display_name: dict[str, str] | None = None
1552 tool_name_to_description: dict[str, str] | None = None
1553 extra_headers: list[str] | None = None
1554 static_headers: dict[str, str] | None = None
1555 env_vars: list[MCPEnvVar] | None = None
1556 instructions: str | None = None
1557 # Stdio-specific fields
1558 command: str | None = None
1559 args: list[str] = Field(default_factory=list)
1560 env: dict[str, str] = Field(default_factory=dict)
1561 issuer: str | None = None
1562 authorization_url: str | None = None
1563 token_url: str | None = None
1564 registration_url: str | None = None
1565 oauth2_flow: Literal["client_credentials", "authorization_code"] | None = None
1566 # Token Exchange (OBO) fields — RFC 8693. These top-level fields are the
1567 # canonical shape; the same keys inside ``credentials`` are the legacy
1568 # pre-column REST shape and are lifted into these columns on write (an
1569 # explicit top-level value wins) and stripped from the stored blob.
1570 token_exchange_endpoint: str | None = None
1571 audience: str | None = None
1572 subject_token_type: str | None = None
1573 token_exchange_profile: str | None = None
1574 allow_all_keys: bool = False
1575 available_on_public_internet: bool = True
1576 delegate_auth_to_upstream: bool = False
1577 oauth_passthrough: bool = False
1578 dcr_bridge: bool | None = None
1579 per_server_oauth_discovery: bool = False
1580 is_byok: bool = False
1581 byok_description: list[str] = Field(default_factory=list)
1582 byok_api_key_help_url: str | None = None
1583 source_url: str | None = None
1584 timeout: float | None = None
1585 max_concurrent_requests: int | None = None
1586 # BYOM submission fields — set by the endpoint, not by the caller.
1587 # Any caller-provided values are silently overridden before persistence.
1588 approval_status: str | None = Field(
1589 default=None,
1590 description="Server-managed: set by the endpoint; caller values are overridden.",
1591 )
1592 submitted_by: str | None = Field(
1593 default=None,
1594 description="Server-managed: set by the endpoint; caller values are overridden.",
1595 )
1596 submitted_at: datetime | None = Field(
1597 default=None,
1598 description="Server-managed: set by the endpoint; caller values are overridden.",
1599 )
1601 @model_validator(mode="before")
1602 @classmethod
1603 def validate_transport_fields(cls, values):
1604 _validate_mcp_transport_fields(values)
1605 return values
1607 @model_validator(mode="before")
1608 @classmethod
1609 def validate_credentials_requirements(cls, values):
1610 """Validate credentials when provided.
1612 auth_value is optional — users may configure it dynamically
1613 (e.g. via per-request headers or OAuth2 flows) instead of
1614 storing a static value at server creation time.
1615 """
1616 return values
1618 @model_validator(mode="before")
1619 @classmethod
1620 def validate_dcr_bridge_auth_type(cls, values):
1621 if not isinstance(values, dict) or not values.get("dcr_bridge"):
1622 return values
1623 auth_type: Final = values.get("auth_type")
1624 if auth_type in (MCPAuth.true_passthrough, MCPAuth.oauth_delegate): 1624 ↛ 1625line 1624 didn't jump to line 1625 because the condition on line 1624 was never true
1625 return values
1626 raise _dcr_bridge_auth_type_error(auth_type)
1628 @model_validator(mode="before")
1629 @classmethod
1630 def validate_per_server_oauth_discovery_auth_type(cls, values: object) -> object:
1631 _reject_unsupported_per_server_oauth_discovery(values, require_auth_type=True)
1632 return values
1635class UpdateMCPServerRequest(LiteLLMPydanticObjectBase):
1636 server_id: str
1637 server_name: str | None = None
1638 alias: str | None = None
1639 description: str | None = None
1640 transport: MCPTransportType = MCPTransport.sse
1641 auth_type: MCPAuthType | None = None
1642 credentials: MCPCredentials | None = None
1643 url: str | None = None
1644 spec_path: str | None = None
1645 mcp_info: MCPInfo | None = None
1646 mcp_access_groups: list[str] = Field(default_factory=list)
1647 allowed_tools: list[str] | None = None
1648 tool_name_to_display_name: dict[str, str] | None = None
1649 tool_name_to_description: dict[str, str] | None = None
1650 extra_headers: list[str] | None = None
1651 static_headers: dict[str, str] | None = None
1652 env_vars: list[MCPEnvVar] | None = None
1653 instructions: str | None = None
1654 # Stdio-specific fields
1655 command: str | None = None
1656 args: list[str] = Field(default_factory=list)
1657 env: dict[str, str] = Field(default_factory=dict)
1658 issuer: str | None = None
1659 authorization_url: str | None = None
1660 token_url: str | None = None
1661 registration_url: str | None = None
1662 oauth2_flow: Literal["client_credentials", "authorization_code"] | None = None
1663 # Token Exchange (OBO) fields — RFC 8693. These top-level fields are the
1664 # canonical shape; the same keys inside ``credentials`` are the legacy
1665 # pre-column REST shape and are lifted into these columns on write (an
1666 # explicit top-level value wins) and stripped from the stored blob.
1667 token_exchange_endpoint: str | None = None
1668 audience: str | None = None
1669 subject_token_type: str | None = None
1670 token_exchange_profile: str | None = None
1671 allow_all_keys: bool = False
1672 available_on_public_internet: bool = True
1673 delegate_auth_to_upstream: bool = False
1674 oauth_passthrough: bool = False
1675 dcr_bridge: bool | None = None
1676 per_server_oauth_discovery: bool = False
1677 is_byok: bool = False
1678 byok_description: list[str] = Field(default_factory=list)
1679 byok_api_key_help_url: str | None = None
1680 source_url: str | None = None
1681 timeout: float | None = None
1682 max_concurrent_requests: int | None = None
1684 @model_validator(mode="before")
1685 @classmethod
1686 def validate_transport_fields(cls, values):
1687 _validate_mcp_transport_fields(values)
1688 return values
1690 @model_validator(mode="before")
1691 @classmethod
1692 def validate_dcr_bridge_auth_type(cls, values):
1693 """Partial updates omit auth_type; that case is validated against the stored row by the
1694 update endpoint, which can read the database. This validator covers payloads that carry
1695 both fields."""
1696 if not isinstance(values, dict) or not values.get("dcr_bridge"):
1697 return values
1698 if "auth_type" not in values:
1699 return values
1700 auth_type: Final = values.get("auth_type")
1701 if auth_type in (MCPAuth.true_passthrough, MCPAuth.oauth_delegate):
1702 return values
1703 raise _dcr_bridge_auth_type_error(auth_type)
1705 @model_validator(mode="before")
1706 @classmethod
1707 def validate_per_server_oauth_discovery_auth_type(cls, values: object) -> object:
1708 _reject_unsupported_per_server_oauth_discovery(values, require_auth_type=False)
1709 return values
1712from litellm.models.mcp_server import ( # noqa: E402
1713 LiteLLM_MCPServerTable as LiteLLM_MCPServerTable,
1714)
1717class MakeMCPServersPublicRequest(LiteLLMPydanticObjectBase):
1718 mcp_server_ids: list[str]
1721class MCPUserCredentialRequest(LiteLLMPydanticObjectBase):
1722 credential: str
1723 save: bool = True
1726class MCPUserCredentialResponse(LiteLLMPydanticObjectBase):
1727 server_id: str
1728 has_credential: bool
1731class MCPOAuthUserCredentialRequest(LiteLLMPydanticObjectBase):
1732 """Stores a user's OAuth2 token for an OpenAPI MCP server."""
1734 access_token: str
1735 refresh_token: str | None = None
1736 expires_in: int | None = None # seconds until expiry
1737 scopes: list[str] | None = None
1740class MCPOAuthUserCredentialStatus(LiteLLMPydanticObjectBase):
1741 """Describes whether the calling user has a stored OAuth credential."""
1743 server_id: str
1744 has_credential: bool
1745 expires_at: str | None = None # ISO-8601
1746 is_expired: bool = False
1747 connected_at: str | None = None # ISO-8601
1750class MCPUserCredentialListItem(LiteLLMPydanticObjectBase):
1751 """One entry in the /user-credentials list."""
1753 server_id: str
1754 server_name: str | None = None
1755 alias: str | None = None
1756 credential_type: str # "oauth2" or "byok"
1757 has_credential: bool
1758 expires_at: str | None = None # ISO-8601; None means non-expiring
1759 connected_at: str | None = None # ISO-8601
1762class MCPServerUserCredentialListItem(LiteLLMPydanticObjectBase):
1763 """One user's stored credential for an MCP server, as an admin sees it. Never carries the secret."""
1765 user_id: str
1766 credential_type: Literal["oauth2", "byok"]
1767 expires_at: str | None = None
1768 connected_at: str | None = None
1769 updated_at: str
1772class MCPUserEnvVarsRequest(LiteLLMPydanticObjectBase):
1773 """Payload for storing the calling user's per-user env var values."""
1775 values: dict[str, str]
1778class MCPUserEnvVarSpec(LiteLLMPydanticObjectBase):
1779 """Describes one per-user env var slot for the calling user.
1781 Stored values are write-only: the status only reports whether a value
1782 ``is_set`` and never echoes the decrypted secret back to the client.
1783 """
1785 name: str
1786 description: str | None = None
1787 is_set: bool = False
1790class MCPUserEnvVarsStatus(LiteLLMPydanticObjectBase):
1791 """Per-user env var status for a single MCP server."""
1793 server_id: str
1794 server_name: str | None = None
1795 alias: str | None = None
1796 required: list[MCPUserEnvVarSpec] = Field(default_factory=list)
1797 missing_count: int = 0
1798 setup_url: str | None = None # frontend URL where the user can fill these in
1801class RejectMCPServerRequest(LiteLLMPydanticObjectBase):
1802 review_notes: str | None = None
1805class MCPSubmissionsSummary(LiteLLMPydanticObjectBase):
1806 total: int
1807 pending_review: int
1808 active: int
1809 rejected: int
1810 items: list["LiteLLM_MCPServerTable"]
1813######## Skills API Types ########
1816class NewSkillRequest(LiteLLMPydanticObjectBase):
1817 """Request to create a new skill in LiteLLM database"""
1819 display_title: str | None = None
1820 description: str | None = None
1821 instructions: str | None = None
1822 file_content: bytes | None = None # Binary content of skill files (zip)
1823 file_name: str | None = None # Original filename
1824 file_type: str | None = None # MIME type (e.g., "application/zip")
1825 metadata: dict[str, Any] | None = None
1826 authorization_url: str | None = None
1827 token_url: str | None = None
1828 registration_url: str | None = None
1831class UpdateSkillRequest(LiteLLMPydanticObjectBase):
1832 """Request to update an existing skill"""
1834 skill_id: str
1835 display_title: str | None = None
1836 description: str | None = None
1837 instructions: str | None = None
1838 file_content: bytes | None = None # Binary content of skill files (zip)
1839 file_name: str | None = None # Original filename
1840 file_type: str | None = None # MIME type
1841 metadata: dict[str, Any] | None = None
1844from litellm.models.skills import ( # noqa: E402
1845 LiteLLM_SkillsTable as LiteLLM_SkillsTable,
1846)
1849class ListSkillsRequest(LiteLLMPydanticObjectBase):
1850 """Request to list skills from LiteLLM database"""
1852 limit: int | None = 20
1853 offset: int | None = 0
1856class NewUserRequestTeam(LiteLLMPydanticObjectBase):
1857 team_id: str
1858 max_budget_in_team: float | None = None
1859 user_role: Literal["user", "admin"] = "user"
1862class NewUserRequest(GenerateRequestBase):
1863 max_budget: float | None = None
1864 user_email: str | None = None
1865 user_alias: str | None = None
1866 user_role: (
1867 Literal[
1868 LitellmUserRoles.PROXY_ADMIN,
1869 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
1870 LitellmUserRoles.INTERNAL_USER,
1871 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
1872 ]
1873 | None
1874 ) = None
1875 teams: list[str] | list[NewUserRequestTeam] | None = None
1876 auto_create_key: bool = True # flag used for returning a key as part of the /user/new response
1877 send_invite_email: bool | None = None
1878 sso_user_id: str | None = None
1879 organizations: list[str] | None = None
1880 password: str | None = None
1882 @field_validator("password")
1883 @classmethod
1884 def password_not_supported(cls, value: str | None) -> str | None:
1885 if value is not None:
1886 raise ValueError(
1887 "password cannot be set via /user/new. Users set their own password through an "
1888 "invitation link (POST /invitation/new)."
1889 )
1890 return value
1893class NewUserResponse(GenerateKeyResponse):
1894 max_budget: float | None = None
1895 user_email: str | None = None
1896 user_role: (
1897 Literal[
1898 LitellmUserRoles.PROXY_ADMIN,
1899 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
1900 LitellmUserRoles.INTERNAL_USER,
1901 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
1902 ]
1903 | None
1904 ) = None
1905 teams: list | None = None
1906 user_alias: str | None = None
1907 model_max_budget: dict | None = None
1908 created_at: datetime | None = None
1909 updated_at: datetime | None = None
1912class UpdateUserRequestNoUserIDorEmail(GenerateRequestBase): # shared with BulkUpdateUserRequest
1913 # repr=False keeps the plaintext out of management-endpoint alerts, which str() the request model
1914 password: str | None = Field(default=None, repr=False)
1915 spend: float | None = None
1916 metadata: dict | None = None
1917 user_alias: str | None = None
1918 user_role: (
1919 Literal[
1920 LitellmUserRoles.PROXY_ADMIN,
1921 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
1922 LitellmUserRoles.INTERNAL_USER,
1923 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
1924 ]
1925 | None
1926 ) = None
1927 max_budget: float | None = None
1930class UpdateUserRequest(UpdateUserRequestNoUserIDorEmail):
1931 # Note: the defaults of all Params here MUST BE NONE
1932 # else they will get overwritten
1933 user_id: str | None = None
1934 user_email: str | None = None
1936 @model_validator(mode="before")
1937 @classmethod
1938 def check_user_info(cls, values):
1939 if values.get("user_id") is None and values.get("user_email") is None:
1940 raise ValueError("Either user id or user email must be provided")
1941 return values
1944class ChangePasswordRequest(LiteLLMPydanticObjectBase):
1945 current_password: str = Field(repr=False)
1946 new_password: str = Field(repr=False)
1949class ChangePasswordResponse(LiteLLMPydanticObjectBase):
1950 user_id: str
1951 message: str
1954class SessionLogoutResponse(LiteLLMPydanticObjectBase):
1955 message: str
1958class DeleteUserRequest(LiteLLMPydanticObjectBase):
1959 user_ids: list[str] # required
1962AllowedModelRegion = Literal["eu", "us"]
1965class BudgetNewRequest(LiteLLMPydanticObjectBase):
1966 budget_id: str | None = Field(default=None, description="The unique budget id.")
1967 max_budget: float | None = Field(
1968 default=None,
1969 description="Requests will fail if this budget (in USD) is exceeded.",
1970 )
1971 soft_budget: float | None = Field(
1972 default=None,
1973 description="Requests will NOT fail if this is exceeded. Will fire alerting though.",
1974 )
1975 max_parallel_requests: int | None = Field(
1976 default=None, description="Max concurrent requests allowed for this budget id."
1977 )
1978 tpm_limit: int | None = Field(default=None, description="Max tokens per minute, allowed for this budget id.")
1979 rpm_limit: int | None = Field(default=None, description="Max requests per minute, allowed for this budget id.")
1980 tpd_limit: int | None = Field(
1981 default=None, description="Max tokens per day, charged by batch submissions, allowed for this budget id."
1982 )
1983 budget_duration: str | None = Field(
1984 default=None,
1985 description="Max duration budget should be set for (e.g. '1hr', '1d', '28d')",
1986 )
1987 model_max_budget: GenericBudgetConfigType | None = Field(
1988 default=None,
1989 description="Max budget for each model (e.g. {'gpt-4o': {'max_budget': '0.0000001', 'budget_duration': '1d', 'tpm_limit': 1000, 'rpm_limit': 1000}})",
1990 )
1991 budget_reset_at: datetime | None = Field(
1992 default=None,
1993 description="Datetime when the budget is reset",
1994 )
1997class BudgetRequest(LiteLLMPydanticObjectBase):
1998 budgets: list[str]
2001class BudgetDeleteRequest(LiteLLMPydanticObjectBase):
2002 id: str
2005class CustomerBase(LiteLLMPydanticObjectBase):
2006 user_id: str
2007 alias: str | None = None
2008 spend: float = 0.0
2009 allowed_model_region: AllowedModelRegion | None = None
2010 default_model: str | None = None
2011 budget_id: str | None = None
2012 litellm_budget_table: BudgetNewRequest | None = None
2013 blocked: bool = False
2016class NewCustomerRequest(BudgetNewRequest):
2017 """
2018 Create a new customer, allocate a budget to them
2019 """
2021 user_id: str
2022 alias: str | None = None # human-friendly alias
2023 blocked: bool = False # allow/disallow requests for this end-user
2024 budget_id: str | None = None # give either a budget_id or max_budget
2025 spend: float | None = None
2026 allowed_model_region: AllowedModelRegion | None = (
2027 None # require all user requests to use models in this specific region
2028 )
2029 default_model: str | None = None # if no equivalent model in allowed region - default all requests to this model
2030 object_permission: LiteLLM_ObjectPermissionBase | None = None
2032 @model_validator(mode="before")
2033 @classmethod
2034 def check_user_info(cls, values):
2035 if values.get("max_budget") is not None and values.get("budget_id") is not None:
2036 raise ValueError("Set either 'max_budget' or 'budget_id', not both.")
2038 return values
2041class UpdateCustomerRequest(LiteLLMPydanticObjectBase):
2042 """
2043 Update a Customer, use this to update customer budgets etc
2045 """
2047 user_id: str
2048 alias: str | None = None # human-friendly alias
2049 blocked: bool = False # allow/disallow requests for this end-user
2050 max_budget: float | None = None
2051 budget_id: str | None = None # give either a budget_id or max_budget
2052 allowed_model_region: AllowedModelRegion | None = (
2053 None # require all user requests to use models in this specific region
2054 )
2055 default_model: str | None = None # if no equivalent model in allowed region - default all requests to this model
2056 object_permission: LiteLLM_ObjectPermissionBase | None = None
2059class DeleteCustomerRequest(LiteLLMPydanticObjectBase):
2060 """
2061 Delete multiple Customers
2062 """
2064 user_ids: list[str]
2067from litellm.models.team import Member as Member # noqa: E402
2068from litellm.models.team import MemberBase as MemberBase # noqa: E402
2071class OrgMember(MemberBase):
2072 role: Literal[
2073 LitellmUserRoles.ORG_ADMIN,
2074 LitellmUserRoles.INTERNAL_USER,
2075 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
2076 ]
2079from litellm.models.team import TeamBase as TeamBase # noqa: E402
2081RouterSettingsDict = Annotated[
2082 dict[str, object],
2083 BeforeValidator(validate_router_settings_dict, json_schema_input_type=UpdateRouterConfig),
2084]
2087class NewTeamRequest(TeamBase):
2088 router_settings: RouterSettingsDict | None = None
2089 model_aliases: dict | None = None
2090 model_max_budget: GenericBudgetConfigType | None = Field(
2091 default=None,
2092 description=(
2093 "Max budget per model for every key on the team, overridable per key "
2094 "(e.g. {'gpt-4o': {'max_budget': 10, 'budget_duration': '1d'}})"
2095 ),
2096 )
2097 tags: list | None = None
2098 guardrails: list[str] | None = None
2099 policies: list[str] | None = None
2100 prompts: list[str] | None = None
2101 object_permission: LiteLLM_ObjectPermissionBase | None = None
2102 allowed_passthrough_routes: list | None = None
2103 disable_global_guardrails: bool | None = None
2104 secret_manager_settings: dict | None = None
2105 model_rpm_limit: dict[str, int] | None = None
2106 rpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput"] | None = (
2107 None # raise an error if 'guaranteed_throughput' is set and we're overallocating rpm
2108 )
2109 tpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput"] | None = (
2110 None # raise an error if 'guaranteed_throughput' is set and we're overallocating tpm
2111 )
2113 model_tpm_limit: dict[str, int] | None = None
2114 default_estimated_output_tokens: PositiveInt | None = None
2115 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None
2116 mcp_rpm_limit: dict[str, int] | None = None
2117 team_member_budget: float | None = None # allow user to set a budget for all team members
2118 team_member_rpm_limit: int | None = None # allow user to set RPM limit for all team members
2119 team_member_tpm_limit: int | None = None # allow user to set TPM limit for all team members
2120 team_member_key_duration: str | None = None # e.g. "1d", "1w", "1m"
2121 team_member_budget_duration: str | None = None # e.g. "30d", "1mo"
2122 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None
2123 enforced_batch_output_expires_after: dict | None = None
2124 enforced_file_expires_after: dict | None = None
2126 model_config = ConfigDict(protected_namespaces=())
2128 @field_validator("team_id", mode="before")
2129 @classmethod
2130 def treat_blank_team_id_as_unset(cls, v: object) -> object:
2131 if isinstance(v, str) and not v.strip(): 2131 ↛ 2132line 2131 didn't jump to line 2132 because the condition on line 2131 was never true
2132 return None
2133 return v
2136class GlobalEndUsersSpend(LiteLLMPydanticObjectBase):
2137 api_key: str | None = None
2138 startTime: datetime | None = None
2139 endTime: datetime | None = None
2142class UpdateTeamRequest(LiteLLMPydanticObjectBase):
2143 """
2144 UpdateTeamRequest, used by /team/update when you need to update a team
2146 team_id: str
2147 team_alias: Optional[str] = None
2148 organization_id: Optional[str] = None
2149 metadata: Optional[dict] = None
2150 tpm_limit: Optional[int] = None
2151 rpm_limit: Optional[int] = None
2152 max_budget: Optional[float] = None
2153 models: Optional[list] = None
2154 blocked: Optional[bool] = None
2155 budget_duration: Optional[str] = None
2156 guardrails: Optional[List[str]] = None
2157 policies: Optional[List[str]] = None
2158 """
2160 team_id: str # required
2161 team_alias: str | None = None
2162 organization_id: str | None = None
2163 metadata: dict | None = None
2164 tpm_limit: int | None = None
2165 rpm_limit: int | None = None
2166 tpd_limit: int | None = None
2167 max_budget: float | None = None
2168 soft_budget: float | None = None
2169 models: list | None = None
2170 blocked: bool | None = None
2171 budget_duration: str | None = None
2172 tags: list | None = None
2173 model_aliases: dict | None = None
2174 guardrails: list[str] | None = None
2175 policies: list[str] | None = None
2176 object_permission: LiteLLM_ObjectPermissionBase | None = None
2177 disable_global_guardrails: bool | None = None
2178 team_member_budget: float | None = None
2179 team_member_budget_duration: str | None = None
2180 team_member_rpm_limit: int | None = None
2181 team_member_tpm_limit: int | None = None
2182 team_member_key_duration: str | None = None
2183 allowed_passthrough_routes: list | None = None
2184 secret_manager_settings: dict | None = None
2185 prompts: list[str] | None = None
2186 model_rpm_limit: dict[str, int] | None = None
2187 model_tpm_limit: dict[str, int] | None = None
2188 default_estimated_output_tokens: PositiveInt | None = None
2189 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None
2190 mcp_rpm_limit: dict[str, int] | None = None
2191 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None
2192 enforced_batch_output_expires_after: dict | None = None
2193 enforced_file_expires_after: dict | None = None
2194 router_settings: RouterSettingsDict | None = None
2195 access_group_ids: list[str] | None = None
2196 budget_limits: list[BudgetLimitEntry] | None = None # multiple concurrent budget windows
2197 default_team_member_models: list[str] | None = None # default allowed_models seeded onto new team members
2198 model_max_budget: GenericBudgetConfigType | None = Field(
2199 default=None,
2200 description=(
2201 "Max budget per model for every key on the team, overridable per key "
2202 "(e.g. {'gpt-4o': {'max_budget': 10, 'budget_duration': '1d'}})"
2203 ),
2204 )
2207class PatchTeamRequest(UpdateTeamRequest):
2208 """
2209 Body of PATCH /team/{team_id}.
2211 Identical to UpdateTeamRequest except team_id is optional, because PATCH takes it
2212 from the path. A team_id in the body is still accepted when it matches the path.
2213 """
2215 team_id: str | None = None
2218class ResetTeamBudgetRequest(LiteLLMPydanticObjectBase):
2219 """
2220 internal type used to reset the budget on a team
2221 used by reset_budget()
2223 team_id: str
2224 spend: float
2225 budget_reset_at: datetime
2226 """
2228 team_id: str
2229 spend: float
2230 budget_reset_at: datetime
2231 updated_at: datetime
2234class DeleteTeamRequest(LiteLLMPydanticObjectBase):
2235 team_ids: list[str] # required
2238class BlockTeamRequest(LiteLLMPydanticObjectBase):
2239 team_id: str # required
2242class BlockKeyRequest(LiteLLMPydanticObjectBase):
2243 key: str # required
2246class BlockModelRequest(LiteLLMPydanticObjectBase):
2247 model_id: str # required
2250class AddTeamCallback(LiteLLMPydanticObjectBase):
2251 callback_name: str
2252 callback_type: Literal["success", "failure", "success_and_failure"] | None = "success_and_failure"
2253 callback_vars: dict[str, str]
2255 @model_validator(mode="before")
2256 @classmethod
2257 def validate_callback_vars(cls, values):
2258 callback_vars: Final = values.get("callback_vars", {})
2259 valid_keys: Final = set(StandardCallbackDynamicParams.__annotations__.keys())
2260 for key, value in callback_vars.items():
2261 if key not in valid_keys: 2261 ↛ 2263line 2261 didn't jump to line 2263 because the condition on line 2261 was always true
2262 raise ValueError(f"Invalid callback variable: {key}. Must be one of {valid_keys}")
2263 callback_vars[key] = str(value)
2264 validate_no_callback_env_reference(key, callback_vars[key], source="key/team callback metadata")
2265 if key == "langfuse_environment":
2266 validate_langfuse_environment_value(callback_vars[key])
2267 if key == "langfuse_span_scope":
2268 validate_langfuse_span_scope_value(callback_vars[key])
2269 return values
2272class TeamCallbackDeleteResponseData(LiteLLMPydanticObjectBase):
2273 team_id: str
2274 success_callbacks: tuple[str, ...]
2275 failure_callbacks: tuple[str, ...]
2278class TeamCallbackDeleteResponse(LiteLLMPydanticObjectBase):
2279 status: Literal["success"]
2280 message: str
2281 data: TeamCallbackDeleteResponseData
2284class TeamCallbackMetadata(LiteLLMPydanticObjectBase):
2285 success_callback: list[str] | None = []
2286 failure_callback: list[str] | None = []
2287 callbacks: list[str] | None = []
2288 # for now - only supported for langfuse
2289 callback_vars: dict[str, str] | None = {}
2291 @model_validator(mode="before")
2292 @classmethod
2293 def validate_callback_vars(cls, values):
2294 success_callback: Final = values.get("success_callback", [])
2295 if success_callback is None: 2295 ↛ 2296line 2295 didn't jump to line 2296 because the condition on line 2295 was never true
2296 values.pop("success_callback", None)
2297 failure_callback: Final = values.get("failure_callback", [])
2298 if failure_callback is None: 2298 ↛ 2299line 2298 didn't jump to line 2299 because the condition on line 2298 was never true
2299 values.pop("failure_callback", None)
2300 callbacks: Final = values.get("callbacks", [])
2301 if callbacks is None: 2301 ↛ 2302line 2301 didn't jump to line 2302 because the condition on line 2301 was never true
2302 values.pop("callbacks", None)
2304 callback_vars: Final = values.get("callback_vars", {})
2305 if callback_vars is None: 2305 ↛ 2306line 2305 didn't jump to line 2306 because the condition on line 2305 was never true
2306 values.pop("callback_vars", None)
2307 if all(val is None for val in values.values()):
2308 return {
2309 "success_callback": [],
2310 "failure_callback": [],
2311 "callbacks": [],
2312 "callback_vars": {},
2313 }
2314 valid_keys: Final = set(StandardCallbackDynamicParams.__annotations__.keys())
2315 if callback_vars is not None: 2315 ↛ 2319line 2315 didn't jump to line 2319 because the condition on line 2315 was always true
2316 for key in callback_vars: 2316 ↛ 2317line 2316 didn't jump to line 2317 because the loop on line 2316 never started
2317 if key not in valid_keys:
2318 raise ValueError(f"Invalid callback variable: {key}. Must be one of {valid_keys}")
2319 return values
2322from litellm.models.object_permission import ( # noqa: E402
2323 LiteLLM_ObjectPermissionTable as LiteLLM_ObjectPermissionTable,
2324)
2325from litellm.models.team import ( # noqa: E402
2326 LiteLLM_DeletedTeamTable as LiteLLM_DeletedTeamTable,
2327)
2328from litellm.models.team import LiteLLM_TeamTable as LiteLLM_TeamTable # noqa: E402
2329from litellm.models.team import ( # noqa: E402
2330 LiteLLM_TeamTableCachedObj as LiteLLM_TeamTableCachedObj,
2331)
2334class TeamRequest(LiteLLMPydanticObjectBase):
2335 teams: list[str]
2338from litellm.models.budget import ( # noqa: E402
2339 LiteLLM_BudgetTable as LiteLLM_BudgetTable,
2340)
2341from litellm.models.budget import ( # noqa: E402
2342 LiteLLM_BudgetTableFull as LiteLLM_BudgetTableFull,
2343)
2344from litellm.models.budget import ( # noqa: E402
2345 LiteLLM_TeamMemberTable as LiteLLM_TeamMemberTable,
2346)
2349class NewOrganizationRequest(LiteLLM_BudgetTable):
2350 organization_id: str | None = None
2351 organization_alias: str
2352 models: list = []
2353 budget_id: str | None = None
2354 metadata: dict | None = None
2355 model_rpm_limit: dict[str, int] | None = None
2356 model_tpm_limit: dict[str, int] | None = None
2358 #########################################################
2359 # Object Permission - MCP, Vector Stores etc.
2360 #########################################################
2361 object_permission: LiteLLM_ObjectPermissionBase | None = None
2364class OrganizationRequest(LiteLLMPydanticObjectBase):
2365 organizations: list[str]
2368class DeleteOrganizationRequest(LiteLLMPydanticObjectBase):
2369 organization_ids: list[str] # required
2372class TeamDefaultSettings(LiteLLMPydanticObjectBase):
2373 team_id: str
2375 model_config = ConfigDict(
2376 extra="allow"
2377 ) # allow params not defined here, these fall in litellm.completion(**kwargs)
2380class DynamoDBArgs(LiteLLMPydanticObjectBase):
2381 billing_mode: Literal["PROVISIONED_THROUGHPUT", "PAY_PER_REQUEST"]
2382 read_capacity_units: int | None = None
2383 write_capacity_units: int | None = None
2384 ssl_verify: bool | None = None
2385 region_name: str
2386 user_table_name: str = "LiteLLM_UserTable"
2387 key_table_name: str = "LiteLLM_VerificationToken"
2388 config_table_name: str = "LiteLLM_Config"
2389 spend_table_name: str = "LiteLLM_SpendLogs"
2390 aws_role_name: str | None = None
2391 aws_session_name: str | None = None
2392 aws_web_identity_token: str | None = None
2393 aws_provider_id: str | None = None
2394 aws_policy_arns: list[str] | None = None
2395 aws_policy: str | None = None
2396 aws_duration_seconds: int | None = None
2397 assume_role_aws_role_name: str | None = None
2398 assume_role_aws_session_name: str | None = None
2401class PassThroughGuardrailSettings(LiteLLMPydanticObjectBase):
2402 """
2403 Settings for a specific guardrail on a passthrough endpoint.
2405 Allows field-level targeting for guardrail execution.
2406 """
2408 request_fields: list[str] | None = Field(
2409 default=None,
2410 description="JSONPath expressions for input field targeting (pre_call). Examples: 'query', 'documents[*].text', 'messages[*].content'. If not specified, guardrail runs on entire request payload.",
2411 )
2412 response_fields: list[str] | None = Field(
2413 default=None,
2414 description="JSONPath expressions for output field targeting (post_call). Examples: 'results[*].text', 'output'. If not specified, guardrail runs on entire response payload.",
2415 )
2418# Type alias for the guardrails dict: guardrail_name -> settings (or None for defaults)
2419PassThroughGuardrailsConfig = dict[str, PassThroughGuardrailSettings | None]
2422class PassThroughGenericEndpoint(LiteLLMPydanticObjectBase):
2423 id: str | None = Field(
2424 default=None,
2425 description="Optional unique identifier for the pass-through endpoint. If not provided, endpoints will be identified by path for backwards compatibility.",
2426 )
2427 path: str = Field(description="The route to be added to the LiteLLM Proxy Server.")
2428 target: str = Field(description="The URL to which requests for this path should be forwarded.")
2429 headers: dict = Field(
2430 default={},
2431 description="Key-value pairs of headers to be forwarded with the request. You can set any key value pair here and it will be forwarded to your target endpoint",
2432 )
2433 default_query_params: dict = Field(
2434 default={},
2435 description="Key-value pairs of default query parameters to be sent with every request to this endpoint. These can be overridden by client-provided query parameters. For example: {'key': 'default_value', 'api_version': '2023-01'}",
2436 )
2437 include_subpath: bool = Field(
2438 default=False,
2439 description="If True, requests to subpaths of the path will be forwarded to the target endpoint. For example, if the path is /bria and include_subpath is True, requests to /bria/v1/text-to-image/base/2.3 will be forwarded to the target endpoint.",
2440 )
2441 cost_per_request: float = Field(
2442 default=0.0,
2443 description="The USD cost per request to the target endpoint. This is used to calculate the cost of the request to the target endpoint.",
2444 )
2445 timeout: float | None = Field(
2446 default=None,
2447 description="Upstream request timeout in seconds for this pass-through endpoint. If unset, uses general_settings.pass_through_request_timeout (default 600).",
2448 )
2449 auth: bool = Field(
2450 default=True,
2451 description="Whether authentication is required for the pass-through endpoint. Defaults to True so a pass-through silently created without an explicit value still requires a valid LiteLLM API key — set to False only if the endpoint is meant to be a public forwarder (e.g. an unauthenticated webhook target).",
2452 )
2453 guardrails: PassThroughGuardrailsConfig | None = Field(
2454 default=None,
2455 description="Guardrails configuration for this passthrough endpoint. Dict keys are guardrail names, values are optional settings for field targeting. When set, all org/team/key level guardrails will also execute. Defaults to None (no guardrails execute).",
2456 )
2457 is_from_config: bool = Field(
2458 default=False,
2459 description="True if this endpoint is defined in the config file, False if from DB. Config-defined endpoints cannot be edited via the UI.",
2460 )
2461 methods: list[str] | None = Field(
2462 default=None,
2463 description="List of HTTP methods this endpoint handles (e.g., ['GET', 'POST']). If None or empty, all methods (GET, POST, PUT, DELETE, PATCH) are supported for backward compatibility. This allows the same path to have different targets for different HTTP methods.",
2464 )
2467class PassThroughEndpointResponse(LiteLLMPydanticObjectBase):
2468 endpoints: list[PassThroughGenericEndpoint]
2471class ConfigFieldUpdate(LiteLLMPydanticObjectBase):
2472 field_name: str
2473 field_value: Any
2474 config_type: Literal["general_settings"]
2477class ConfigFieldDelete(LiteLLMPydanticObjectBase):
2478 config_type: Literal["general_settings"]
2479 field_name: str
2482class CallbackDelete(LiteLLMPydanticObjectBase):
2483 callback_name: str
2486class FieldDetail(BaseModel):
2487 field_name: str
2488 field_type: str
2489 field_description: str
2490 field_default_value: Any = None
2491 stored_in_db: bool | None
2494class ConfigList(LiteLLMPydanticObjectBase):
2495 field_name: str
2496 field_type: str
2497 field_description: str
2498 field_value: Any
2499 stored_in_db: bool | None
2500 field_default_value: Any
2501 premium_field: bool = False
2502 nested_fields: list[FieldDetail] | None = None # For nested dictionary or Pydantic fields
2503 field_options: list[str] | None = None # Allowed values, for field_type == "Select"
2504 field_tab: str | None = None # Admin UI sub-tab this field renders under; None groups it with the rest
2505 source: Literal["config", "db", "env", "default", "unset"] = "unset"
2506 editable: bool = True
2509class UserHeaderMapping(LiteLLMPydanticObjectBase):
2510 """
2511 Map an incoming HTTP header to a LiteLLM user role.
2512 """
2514 header_name: str
2515 litellm_user_role: Literal[
2516 LitellmUserRoles.INTERNAL_USER,
2517 LitellmUserRoles.CUSTOMER,
2518 ]
2520 model_config = {
2521 "extra": "forbid",
2522 }
2525UserMCPManagementMode = Literal["restricted", "view_all"]
2528class PluginConfig(LiteLLMPydanticObjectBase):
2529 """A single external service registered as an embeddable UI plugin."""
2531 name: str = Field(description="unique plugin identifier (kebab-case)")
2532 display_name: str | None = Field(None, description="human-readable label shown in the UI view switcher")
2533 url: str = Field(description="base URL of the plugin service")
2534 plugin_key: str | None = Field(
2535 None,
2536 description="plugin's own credential, injected as Bearer auth only on /plugin-proxy/<name>/* reverse-proxy calls",
2537 )
2540class CoordinationRedisNode(LiteLLMPydanticObjectBase):
2541 """A single startup node of a cluster-mode Redis used for proxy coordination."""
2543 host: str = Field(description="hostname of the cluster node")
2544 port: int = Field(description="port of the cluster node")
2547class CoordinationRedisParams(LiteLLMPydanticObjectBase):
2548 """
2549 Connection params for the proxy's coordination Redis (cross-pod tpm/rpm rate
2550 limits, spend tracking, pod lock manager, shared health checks), configured
2551 independently of the response-cache backend in `litellm_settings.cache_params`.
2552 """
2554 model_config = ConfigDict(extra="allow", protected_namespaces=())
2556 host: str | None = Field(None, description="Redis hostname")
2557 port: int | None = Field(None, description="Redis port")
2558 password: str | None = Field(None, description="Redis password")
2559 username: str | None = Field(None, description="Redis username")
2560 url: str | None = Field(None, description="full Redis connection url, e.g. redis://:pass@host:6379")
2561 ssl: bool | None = Field(None, description="connect over TLS")
2562 startup_nodes: list[CoordinationRedisNode] | None = Field(
2563 None, description="cluster-mode startup nodes; when set a cluster client is used"
2564 )
2565 sentinel_nodes: list[list[str | int]] | None = Field(
2566 None, description="sentinel [host, port] pairs; when set a sentinel-managed client is used"
2567 )
2568 sentinel_password: str | None = Field(None, description="password for the sentinel nodes")
2569 service_name: str | None = Field(None, description="sentinel service name")
2570 aws_iam_auth: bool | str | None = Field(None, description="enable AWS ElastiCache IAM authentication")
2571 aws_iam_user_name: str | None = Field(None, description="AWS ElastiCache IAM user name")
2572 aws_iam_cache_name: str | None = Field(None, description="AWS ElastiCache cache name")
2573 aws_iam_region: str | None = Field(None, description="AWS region for ElastiCache IAM authentication")
2574 aws_iam_serverless: bool | str | None = Field(
2575 None, description="the ElastiCache cache is serverless rather than a self-designed cluster"
2576 )
2578 def has_connection_target(self) -> bool:
2579 return any(value is not None for value in (self.host, self.url, self.startup_nodes, self.sentinel_nodes))
2582class ScheduledJobStaggerSettings(LiteLLMPydanticObjectBase):
2583 """
2584 Spreads the proxy's scheduled background jobs across a window instead of firing them
2585 all on one instant, on every replica, forever.
2586 """
2588 model_config = ConfigDict(frozen=True, extra="forbid", protected_namespaces=())
2590 enabled: bool = Field(default=True, description="apply deterministic phase offsets to scheduled background jobs")
2591 window_seconds: int = Field(
2592 default=DEFAULT_STAGGER_WINDOW_SECONDS,
2593 ge=0,
2594 description=(
2595 "width of the window jobs are spread over. An interval job is never offset by "
2596 "more than one of its own periods, so it is not delayed past the wait it already has"
2597 ),
2598 )
2599 identity: str | None = Field(
2600 default=None,
2601 description=(
2602 "replaces the POD_NAME/HOSTNAME-derived component of the offset hash. Set this "
2603 "when replicas share a hostname and would otherwise land on the same offset"
2604 ),
2605 )
2606 offsets: Mapping[str, int] = Field(
2607 default_factory=dict,
2608 description=(
2609 "explicit offset in seconds per scheduler job id, overriding the derived value. "
2610 "0 pins a job to its unshifted schedule"
2611 ),
2612 )
2615class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
2616 """
2617 Documents all the fields supported by `general_settings` in config.yaml
2618 """
2620 completion_model: str | None = Field(None, description="proxy level default model for all chat completion calls")
2621 max_in_flight_requests_per_worker: int | None = Field(
2622 None, gt=0, description="maximum concurrent requests handled by each worker"
2623 )
2624 max_queued_requests_per_worker: int | None = Field(
2625 None, ge=0, description="maximum requests waiting for a worker slot"
2626 )
2627 admission_queue_timeout_seconds: float = Field(
2628 1.0, gt=0, description="maximum time a request waits for a worker slot"
2629 )
2630 plugins: list[PluginConfig] | None = Field(
2631 None, description="external services registered as embeddable UI plugins"
2632 )
2633 key_management_system: KeyManagementSystem | None = Field(
2634 None, description="key manager to load keys from / decrypt keys with"
2635 )
2636 use_google_kms: bool | None = Field(None, description="decrypt keys with google kms")
2637 use_azure_key_vault: bool | None = Field(None, description="load keys from azure key vault")
2638 master_key: str | None = Field(None, description="require a key for all calls to proxy")
2639 dangerously_permit_weak_or_unset_master_key: bool | None = Field(
2640 None,
2641 description="local development only: start even when master_key is unset, empty, or a publicly known default",
2642 )
2643 coordination_redis: CoordinationRedisParams | None = Field(
2644 None,
2645 description=(
2646 "standalone Redis for cross-pod coordination (tpm/rpm rate limits, "
2647 "spend tracking, pod lock manager, shared health checks), configured "
2648 "independently of the response-cache backend; takes precedence over "
2649 "borrowing the `cache_params` Redis and over the REDIS_* env fallback"
2650 ),
2651 )
2652 control_plane_url: str | None = Field(
2653 None,
2654 description=(
2655 "Global Control Plane: URL of the control plane whose admin UI manages this instance. "
2656 "Enables /v3/login and /v3/login/exchange on this instance so that UI can authenticate "
2657 "against it cross-origin, and restricts the SSO return_to origin to that URL. "
2658 "No state is shared with the control plane"
2659 ),
2660 )
2661 allow_cli_sso_verification_uri_complete: bool | None = Field(
2662 None,
2663 description="opt-in to RFC 8628 verification_uri_complete for the CLI SSO device flow, pre-filling the user_code in the browser. Off by default; intended for same-host clients where the device that starts the flow and the browser run on the same machine",
2664 )
2665 include_call_id_in_error_body: bool | None = Field(
2666 None,
2667 description="opt-in to copy the x-litellm-call-id response header's value into JSON error bodies, as error.litellm_call_id on the OpenAI-shaped and /v1/messages routes and as a top-level litellm_call_id on pass-through routes, so an error a client prints names the request to look up. Off by default",
2668 )
2669 enable_claude_code_gateway: bool | None = Field(
2670 None,
2671 description="serve the Claude Code gateway protocol (https://code.claude.com/docs/en/claude-apps-gateway) under /claude_code_gateway: OAuth device-flow sign-in reusing proxy SSO, plus managed settings and OTLP telemetry ingestion. Off by default",
2672 )
2673 claude_code_gateway_managed_settings: dict[str, Any] | None = Field(
2674 None,
2675 description="Claude Code managed-settings.json served verbatim at the gateway's /claude_code_gateway/managed/settings endpoint. When unset the endpoint returns 404 (no managed policy)",
2676 )
2677 database_url: str | None = Field(
2678 None,
2679 description="connect to a postgres db - needed for generating temporary keys + tracking spend / key",
2680 )
2681 database_connection_pool_limit: int | None = Field(
2682 10,
2683 description="default connection pool for prisma client connecting to postgres db",
2684 )
2685 database_connection_timeout: float | None = Field(
2686 60, description="default timeout for a connection to the database"
2687 )
2688 database_connect_timeout: float | None = Field(
2689 None,
2690 description=(
2691 "Prisma `connect_timeout` URL param (seconds). Bounds how long the "
2692 "engine waits to establish a new connection before failing. Defaults "
2693 "to Prisma's built-in value when unset."
2694 ),
2695 )
2696 database_socket_timeout: float | None = Field(
2697 None,
2698 description=(
2699 "Prisma `socket_timeout` URL param (seconds). When set, an in-flight "
2700 "operation that has not produced data within this window is aborted. "
2701 "For capping how long idle pooled connections are kept, see "
2702 "`database_max_idle_connection_lifetime`."
2703 ),
2704 )
2705 database_max_idle_connection_lifetime: float | None = Field(
2706 60,
2707 description=(
2708 "Prisma `max_idle_connection_lifetime` URL param (seconds). A pooled "
2709 "connection idle longer than this is closed and replaced instead of "
2710 "being handed to the next request. Defaults to 60 so connections are "
2711 "recycled before common infra idle timeouts (AWS NLB / RDS Proxy "
2712 "~350s, many LBs 60-350s) silently drop them and requests fail with "
2713 "`Error { kind: Closed }`. A value pinned on the DATABASE_URL or set "
2714 "via `database_extra_connection_params` takes precedence."
2715 ),
2716 )
2717 database_extra_connection_params: dict[str, Any] | None = Field(
2718 None,
2719 description=(
2720 "Escape hatch: extra key/value pairs appended verbatim to the Prisma "
2721 "DATABASE_URL / DIRECT_URL query string (e.g. `sslmode`, `pgbouncer`, "
2722 "`statement_cache_size`). Keys here override any default LiteLLM sets."
2723 ),
2724 )
2725 database_disable_prepared_statements: bool | None = Field(
2726 None,
2727 description=(
2728 "Disable server-side prepared statements by setting Prisma's "
2729 "`pgbouncer=true` URL param. Use this for pgbouncer transaction-pooling "
2730 "deployments, or to prevent the 'cached plan must not change result "
2731 "type' error that pooled connections hit during rolling schema "
2732 "migrations. An explicit `pgbouncer` in `database_extra_connection_params` "
2733 "takes precedence."
2734 ),
2735 )
2736 database_type: Literal["dynamo_db"] | None = Field(None, description="to use dynamodb instead of postgres db")
2737 database_args: DynamoDBArgs | None = Field(
2738 None,
2739 description="custom args for instantiating dynamodb client - e.g. billing provision",
2740 )
2741 otel: bool | None = Field(
2742 None,
2743 description="[BETA] OpenTelemetry support - this might change, use with caution.",
2744 )
2745 custom_auth: str | None = Field(
2746 None,
2747 description="override user_api_key_auth with your own auth script - https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth",
2748 )
2749 max_parallel_requests: int | None = Field(
2750 None,
2751 description="maximum parallel requests for each api key",
2752 )
2753 global_max_parallel_requests: int | None = Field(
2754 None, description="global max parallel requests to allow for a proxy instance."
2755 )
2756 user_api_key_cache_max_size: int | None = Field(
2757 None,
2758 gt=0,
2759 description=(
2760 "max number of entries (virtual keys, teams, users, end users, memberships, ...) each worker keeps in "
2761 "its in-memory auth cache. Defaults to 200. Raise this if you have more active keys than that or auth "
2762 "lookups keep hitting the DB"
2763 ),
2764 )
2765 max_request_size_mb: int | None = Field(
2766 None,
2767 description="max request size in MB, if a request is larger than this size it will be rejected",
2768 )
2769 max_batch_file_size_mb: int | None = Field(
2770 None,
2771 description="max batch input file size in MB for /v1/files uploads with purpose=batch, if a file is larger than this size it will be rejected before being forwarded to the provider",
2772 )
2773 max_file_size_mb: int | None = Field(
2774 None,
2775 description="max file size in MB for /v1/files uploads, for any purpose, if a file is larger than this size it will be rejected before being forwarded to the provider",
2776 )
2777 allowed_file_extensions: tuple[str, ...] | None = Field(
2778 None,
2779 description="the only file extensions (e.g. ['.jsonl', '.pdf', '.txt']) accepted on /v1/files uploads, for any purpose, matched case-insensitively against the uploaded filename. Files with any other extension, or none, are rejected. An empty list rejects every upload. Unset means no allowlist is applied",
2780 )
2781 blocked_file_extensions: tuple[str, ...] | None = Field(
2782 None,
2783 description="file extensions (e.g. ['.exe', '.sh']) rejected on /v1/files uploads, for any purpose, matched case-insensitively against the uploaded filename. Deprecated in favour of allowed_file_extensions; still enforced, after the allowlist, when set",
2784 )
2785 max_response_size_mb: int | None = Field(
2786 None,
2787 description="max response size in MB, if a response is larger than this size it will be rejected",
2788 )
2789 proxy_config_reload_interval_seconds: int = Field(
2790 30,
2791 gt=0,
2792 description="how often (in seconds) each pod reloads config-in-DB objects (models, credentials, guardrails, etc.) when store_model_in_db is enabled; lower values speed up multi-pod convergence at the cost of more DB load. Applied on proxy startup",
2793 )
2794 cancel_on_disconnect: bool | None = Field(
2795 None,
2796 description="cancel the in-flight upstream LLM request (non-streaming) when the client disconnects, freeing backend capacity (e.g. a vLLM GPU slot); the request is logged as a 499 failure",
2797 )
2798 infer_model_from_keys: bool | None = Field(
2799 None,
2800 description="for `/models` endpoint, infers available model based on environment keys (e.g. OPENAI_API_KEY)",
2801 )
2802 background_health_checks: bool | None = Field(None, description="run health checks in background")
2803 health_check_interval: int = Field(300, description="background health check interval in seconds")
2804 health_check_concurrency: int | None = Field(
2805 None,
2806 description=(
2807 "limit concurrent health checks per cycle; when unset, health checks run without a concurrency cap"
2808 ),
2809 )
2810 health_check_skip_disabled_background_models: bool = Field(
2811 False,
2812 description=(
2813 "When true, deployments with model_info.disable_background_health_check "
2814 "are skipped for on-demand GET /health as well as the background health loop."
2815 ),
2816 )
2817 background_health_check_model_groups: tuple[str, ...] | None = Field(
2818 None,
2819 description=(
2820 "Opt-in allowlist of model group names for background health checks and "
2821 "health-check routing. When set, the background loop probes only deployments "
2822 "whose model_name is listed, and enable_health_check_routing filters unhealthy "
2823 "deployments only within the listed groups; every other group, including newly "
2824 "added deployments, is skipped and keeps its configured routing strategy. "
2825 "When unset, all deployments participate (opt out per deployment via "
2826 "model_info.disable_background_health_check)."
2827 ),
2828 )
2829 model_list_healthy_only: bool | None = Field(
2830 None,
2831 description=(
2832 "When true, `/models`, `/v1/models/{id}` and `/model/info` hide models whose backing "
2833 "deployments are all unhealthy, for every caller, without needing `healthy_only=true` "
2834 "per request. Requires `background_health_checks: true`, and keeps deployment health "
2835 "state cached without turning on `enable_health_check_routing`, so routing is "
2836 "unaffected. With no health state nothing is hidden. Hiding is presentation-only, a "
2837 "hidden model can still be called."
2838 ),
2839 )
2840 alerting: list | None = Field(
2841 None,
2842 description="List of alerting integrations - e.g. `alerting: ['slack', 'webhook', 'email']`. 'slack' posts Slack-format messages to any Slack-compatible webhook (Slack, Rocket.Chat, Mattermost); 'webhook' posts structured JSON budget alerts to WEBHOOK_URL",
2843 )
2844 alert_types: list[AlertType] | None = Field(
2845 None,
2846 description="List of alerting types. By default it is all alerts",
2847 )
2848 alert_to_webhook_url: dict | None = Field(
2849 None,
2850 description="Mapping of alert type to webhook url. e.g. `alert_to_webhook_url: {'budget_alerts': 'https://nothooks.slack.com/services/T00000000/B00000000/XXXXXXXXXXXXXXXXXXXXXXXX'}`",
2851 )
2852 alerting_args: dict | None = Field(None, description="Controllable params for slack alerting - e.g. ttl in cache.")
2853 alerting_threshold: int | None = Field(
2854 None,
2855 description="sends alerts if requests hang for 5min+",
2856 )
2857 ui_access_mode: Literal["admin_only", "all"] | None = Field("all", description="Control access to the Proxy UI")
2858 max_failed_login_attempts_per_source: int | None = Field(
2859 None,
2860 ge=1,
2861 description="Failed Admin UI sign-in attempts allowed from one source address, across every username, within `failed_login_window_seconds`. One more blocks that address for `failed_login_block_seconds`. Half this value, rounded down but at least 1, is the allowance for one username from that address; one more blocks that address for that username only, and its further failures stop counting toward the address limit, so a script stuck on one account does not block everyone behind a shared address. The per-address limit is only enforced when `trusted_proxy_ranges` is set: to the proxies in front of LiteLLM, or to an empty list when clients connect directly. Left unset, the peer address may be a shared ingress and only the per-username half runs. IPv6 addresses are grouped by /64. Set under `general_settings` in config.yaml. Defaults to 10",
2862 )
2863 max_failed_login_attempts_per_source_overrides: dict[str, int] | None = Field(
2864 None,
2865 description="Per-address overrides of `max_failed_login_attempts_per_source`, keyed by IP address or CIDR range, e.g. {'1.2.3.4': 200, '5.6.0.0/24': 500}. The most specific matching range wins (between equivalent keys such as '1.2.3.4' and '1.2.3.4/32', an exemption wins, then the higher limit), and the per-username allowance for that address follows as half the override. A value of 0 exempts the address from both limits. Set under `general_settings` in config.yaml",
2866 )
2867 failed_login_window_seconds: int | None = Field(
2868 None,
2869 ge=1,
2870 description="Fixed window in seconds over which failed Admin UI sign-in attempts are counted. The window starts at the first failure and is not extended by later ones. Set under `general_settings` in config.yaml. Defaults to 60",
2871 )
2872 failed_login_block_seconds: int | None = Field(
2873 None,
2874 ge=1,
2875 description="How long a blocked source address, or source address and username, stays blocked. Every attempt from a blocked key, right or wrong, is refused with 429 before the password is checked; the block is not extended by refused attempts. Set under `general_settings` in config.yaml. Defaults to 300",
2876 )
2877 allowed_routes: list | None = Field(None, description="Proxy API Endpoints you want users to be able to access")
2878 reject_clientside_metadata_tags: bool | None = Field(
2879 None,
2880 description="When set to True, rejects requests that contain client-side 'metadata.tags' to prevent users from influencing budgets by sending different tags. Tags can only be inherited from the API key metadata.",
2881 )
2882 missing_session_id: Literal["generate", "reject", "omit"] | None = Field(
2883 None,
2884 description="What to do with LLM API requests that carry no session id (x-litellm-session-id header, metadata.session_id, etc.). 'generate' stamps one id into litellm_session_id, litellm_trace_id and metadata.session_id so SpendLogs and logging callbacks agree; 'reject' returns 400; 'omit' leaves SpendLogs.session_id null, matching callbacks such as Langfuse that only record a client-established metadata.session_id. Unset keeps the legacy behavior where SpendLogs falls back to the trace id while callbacks get no session id.",
2885 )
2886 enable_public_model_hub: bool = Field(
2887 default=False,
2888 description="Public model hub for users to see what models they have access to, supported openai params, etc.",
2889 )
2890 pass_through_request_timeout: float | None = Field(
2891 default=None,
2892 description="Default upstream request timeout in seconds for native and custom pass-through endpoints that use pass_through_request. Defaults to 600 when unset.",
2893 )
2894 pass_through_endpoints: list[PassThroughGenericEndpoint] | None = Field(
2895 default=None,
2896 description="Set-up pass-through endpoints for provider-specific endpoints. Docs - https://docs.litellm.ai/docs/proxy/pass_through",
2897 )
2898 enable_openai_websocket_passthrough: bool | None = Field(
2899 default=None,
2900 description="Serve the OpenAI pass-through WebSocket route, which relays frames to OpenAI under the proxy's own provider credential without reading them. Off by default.",
2901 )
2902 transcribe_media_buckets: list[str] | None = Field(
2903 default=None,
2904 description="S3 bucket names that keys other than proxy admins may read media from and write transcripts to through the Amazon Transcribe pass-through. Unset means only proxy admins can start transcription jobs.",
2905 )
2906 user_header_name: str | None = Field(
2907 None,
2908 description="[DEPRECATED] Use 'user_header_mappings' instead. When set, the header value is treated as the end user id unless overridden by user_header_mappings.",
2909 )
2910 user_header_mappings: list[UserHeaderMapping] | None = None
2911 supported_db_objects: list[SupportedDBObjectType] | None = Field(
2912 None,
2913 description="Fine-grained control over which object types to load from the database when store_model_in_db is True. Available types: 'models', 'mcp', 'guardrails', 'vector_stores', 'pass_through_endpoints', 'prompts', 'model_cost_map', 'tools', 'config_overrides'. If not set, all objects are loaded (default behavior).",
2914 )
2915 user_mcp_management_mode: UserMCPManagementMode | None = Field(
2916 None,
2917 description="Controls how non-admin users interact with MCP servers in the dashboard. 'restricted' shows only accessible servers, 'view_all' lists every server in read-only mode.",
2918 )
2919 store_prompts_in_spend_logs: bool | None = Field(
2920 None,
2921 description="If True, stores request messages and responses in spend logs. Default is False.",
2922 )
2923 disable_auto_add_proxy_admin_to_teams: bool | None = Field(
2924 None,
2925 description="By default, the user calling /team/new is automatically added to the new team as a team admin. If True, proxy admins are no longer auto-added; members explicitly listed in members_with_roles are unaffected. Default is False.",
2926 )
2927 enforce_fallback_model_access: bool | None = Field(
2928 None,
2929 description="If True, router fallbacks configured in router_settings are only attempted when the calling key (and its team and project) is allowed to call the fallback model; unauthorized fallback targets are skipped and the primary model's error is returned. Default is False.",
2930 )
2931 scheduled_job_stagger: ScheduledJobStaggerSettings | None = Field(
2932 None,
2933 description=(
2934 "Spreads the proxy's scheduled background jobs (spend flushes, budget resets, "
2935 "config reloads, exports) across a window instead of firing them together on "
2936 "every replica. On by default; set to tune the window, pin a job, or turn it off."
2937 ),
2938 )
2939 spend_capture_rate_check: SpendCaptureRateCheckSettings | None = Field(
2940 None,
2941 description=(
2942 "Daily check of the spend LiteLLM captured against the provider's own bill (OpenAI via OPENAI_ADMIN_KEY). "
2943 "Publishes litellm_spend_capture_rate per provider and alerts when the ratio over the lookback window "
2944 "falls under the threshold (default 0.9). Off unless set."
2945 ),
2946 )
2947 maximum_spend_logs_retention_period: str | None = Field(
2948 None,
2949 description="Maximum retention period for spend logs (e.g., '7d' for 7 days). Logs older than this will be deleted.",
2950 )
2951 maximum_autorouter_session_retention_period: str | None = Field(
2952 None,
2953 description="Maximum retention period for auto-router benchmark session rollup rows (e.g., '365d'). Rows whose last turn is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rollup rows are never deleted.",
2954 )
2955 maximum_health_check_retention_period: str | None = Field(
2956 None,
2957 description=(
2958 "Maximum retention period for health-check rows (e.g., '30d'). Rows whose checked_at is older than this "
2959 "are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. "
2960 "Set this well above health_check_interval because /health and the UI read the latest row per model."
2961 ),
2962 )
2963 use_spend_logs_partitioning: bool | None = Field(
2964 None,
2965 description="If True and LiteLLM_SpendLogs has been converted to a range-partitioned table (db_scripts/partition_spend_logs.sql), retention cleanup drops expired partitions instead of deleting rows, and pre-creates upcoming partitions. Default is False.",
2966 )
2967 maximum_spend_logs_cleanup_batch_size: int | None = Field(
2968 None,
2969 description="Rows deleted per DELETE statement by the spend log cleanup job. Defaults to 1000.",
2970 )
2971 maximum_spend_logs_cleanup_max_batches: int | None = Field(
2972 None,
2973 description="Maximum DELETE statements the spend log cleanup job issues per table per run. Defaults to 500.",
2974 )
2975 maximum_spend_logs_cleanup_run_budget: str | None = Field(
2976 None,
2977 description="Wall-clock budget for one spend log cleanup run (e.g. '5m'), shared across every table it prunes. A run that hits the budget stops and the next run resumes from where it left off. Defaults to '5m'.",
2978 )
2979 maximum_spend_logs_cleanup_batch_timeout: str | None = Field(
2980 None,
2981 description="Postgres statement_timeout and lock_timeout applied to each spend log cleanup delete batch (e.g. '30s'), so cleanup cannot hold row locks or a connection indefinitely. Defaults to '30s'.",
2982 )
2983 mcp_internal_ip_ranges: list[str] | None = Field(
2984 None,
2985 description="Custom CIDR ranges that define internal/private networks for MCP access control. When set, only these ranges are treated as internal. Defaults to RFC 1918 private ranges (10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 127.0.0.0/8).",
2986 )
2987 mcp_allowed_clients: list[MCPAllowedClient] | None = Field(
2988 None,
2989 description="MCP client applications admitted by the gateway, each an {alias, value} pair where alias is the name shown in the dashboard and logs and value is the identity that must match exactly. When set, every MCP request must carry a client identity equal to one of the values: a JWT caller is identified by the claim named in litellm_jwtauth.mcp_client_id_jwt_field, any other caller by the header named in mcp_client_id_header. A request with no resolvable identity, or an unlisted one, is rejected with 403. Unset means every client is admitted.",
2990 )
2991 mcp_client_id_header: str | None = Field(
2992 None,
2993 description="Request header whose value names the calling MCP client application (for example 'x-mcp-client') for callers that did not authenticate with a JWT, used only while mcp_allowed_clients is set. The client picks this value itself, so it is a policy control rather than a security boundary; prefer litellm_jwtauth.mcp_client_id_jwt_field where callers use JWTs.",
2994 )
2995 mcp_trusted_proxy_ranges: list[str] | None = Field(
2996 None,
2997 description="CIDR ranges of trusted reverse proxies. When set, X-Forwarded-For and X-Forwarded-* origin headers are only trusted from these IPs.",
2998 )
2999 mcp_xff_num_trusted_hops: int | None = Field(
3000 None,
3001 ge=1,
3002 description="Number of trusted reverse proxies/load balancers in front of the gateway that append to X-Forwarded-For. When set (and mcp_trusted_proxy_ranges validates the direct peer), the client IP for MCP access control is read this many entries from the right of the chain instead of the spoofable leftmost value, defeating append-style X-Forwarded-For forgery.",
3003 )
3004 trusted_proxy_ranges: list[str] | None = Field(
3005 None,
3006 description="CIDR ranges of trusted reverse proxies allowed to provide identity headers for header-based auth paths such as enable_oauth2_proxy_auth and custom_ui_sso_sign_in_handler, and whose X-Forwarded-For is used to attribute Admin UI sign-in attempts to a source address. Set it to an empty list when clients connect directly, so the peer address is the source. Left unset, or containing an entry that is not an address or CIDR range, the per-source sign-in limit is off.",
3007 )
3008 store_model_in_db: bool | None = Field(
3009 None,
3010 description="If True, models and config are stored in and loaded from the database. Default is False.",
3011 )
3012 forward_client_headers_to_llm_api: bool | None = Field(
3013 None,
3014 description="If True, forwards client headers (e.g. Authorization) to the LLM API. Required for Claude Code with Max subscription.",
3015 )
3016 mcp_required_fields: list[str] | None = Field(
3017 None,
3018 description="List of MCP server fields that must be filled in for a submission to pass standards checks (e.g. ['description', 'source_url', 'alias']).",
3019 )
3020 password_policy_min_length: int | None = Field(
3021 None,
3022 description=(
3023 "Minimum length required for a locally-managed user's password. Default is 12; "
3024 "a value below 8 is floored to 8 rather than weakening the requirement further."
3025 ),
3026 )
3027 password_policy_require_uppercase: bool | None = Field(
3028 None,
3029 description="If True (default), a locally-managed user's password must contain an uppercase letter.",
3030 )
3031 password_policy_require_lowercase: bool | None = Field(
3032 None,
3033 description="If True (default), a locally-managed user's password must contain a lowercase letter.",
3034 )
3035 password_policy_require_numbers: bool | None = Field(
3036 None,
3037 description="If True (default), a locally-managed user's password must contain a number.",
3038 )
3039 password_policy_require_special_characters: bool | None = Field(
3040 None,
3041 description="If True (default), a locally-managed user's password must contain a special (non-alphanumeric) character.",
3042 )
3043 disable_password_login_when_sso_enabled: bool | None = Field(
3044 None,
3045 description=(
3046 "If True and SSO is configured (MICROSOFT_CLIENT_ID, GOOGLE_CLIENT_ID, "
3047 "GENERIC_CLIENT_ID, or SAML_IDP_METADATA_URL/XML), disables username/password "
3048 "login on /login, /v2/login, and /v3/login so SSO is the only way to reach the "
3049 "Admin UI. An admin locked out of the UI can still administer the proxy over the "
3050 "API with the master key; unset this setting and restart the proxy to restore "
3051 "UI username/password login. Default is False."
3052 ),
3053 )
3054 disable_responses_id_security: bool | None = Field(
3055 None,
3056 description=(
3057 "If True, disables ownership enforcement on Responses API ids. "
3058 "Keys may then retrieve, cancel, delete, and chain from any response id, "
3059 "including ids belonging to another user or team and ids this proxy never issued. "
3060 "WARNING: this removes tenant isolation on /v1/responses"
3061 ),
3062 )
3063 allow_unmanaged_response_ids: bool | None = Field(
3064 None,
3065 description=(
3066 "If True, lets keys address Responses API ids that this proxy did not issue "
3067 "(raw provider ids, or ids issued before response-id encryption was configured). "
3068 "Such an id carries no owner, so no ownership check can run on it; ids this proxy "
3069 "did issue keep full ownership enforcement. Off by default, in which case an "
3070 "unrecognized response id is rejected with 403"
3071 ),
3072 )
3073 disable_env_credential_login: bool | None = Field(
3074 None,
3075 description=(
3076 "If True, disables signing in to the Admin UI with the environment credentials: "
3077 "UI_USERNAME/UI_PASSWORD, or the master key when UI_PASSWORD is unset (that fallback "
3078 "means env-credential login is always live by default). Database users with passwords "
3079 "are unaffected. LOCKOUT RISK: create at least one proxy admin user with a password "
3080 "before enabling, or nobody can sign in to the UI. A locked-out admin can still "
3081 "administer the proxy over the API with the master key, and can unset this setting "
3082 "and restart the proxy to restore env-credential login. Default is False."
3083 ),
3084 )
3085 disable_budget_reservation: bool | None = Field(
3086 None,
3087 description=(
3088 "If True, disables the optimistic per-request budget reservation "
3089 "introduced in v1.84.0. "
3090 "WARNING: This weakens hard budget enforcement. Without the reservation, "
3091 "a burst of concurrent requests from a single key can each pass the "
3092 "read-time spend check before any of them is charged, allowing a "
3093 "configured budget to be exceeded under high concurrency. "
3094 "Budgets are still evaluated on every request at read time, so "
3095 "an already-exhausted budget is still rejected. "
3096 "Enable only if your deployment is experiencing phantom "
3097 "BudgetExceededError responses caused by leaked reservations "
3098 "(see GitHub issue #27639). "
3099 "An INFO notice is logged once per worker at config load while this flag "
3100 "is active as a reminder that hard enforcement is relaxed."
3101 ),
3102 )
3103 apply_user_budget_to_team_keys: bool | None = Field(
3104 None,
3105 description=(
3106 "If True, a user's personal max_budget is enforced on every request they "
3107 "make, including requests made with a team-scoped key. Defaults to False, "
3108 "where a team-scoped key is governed only by the team and team-member "
3109 "budgets and the key owner's personal max_budget does not apply "
3110 "(see GitHub issue #12905)."
3111 ),
3112 )
3113 user_url_validation: bool | None = Field(
3114 None,
3115 description=(
3116 "Master switch for the SSRF guard applied to user-supplied URLs "
3117 "(image_url, file_url, MCP/OpenAPI spec URLs, etc). Defaults to True. "
3118 "Set to False to disable DNS/IP validation entirely (not recommended)."
3119 ),
3120 )
3121 user_url_allowed_hosts: list[str] | None = Field(
3122 None,
3123 description=(
3124 "SSRF allowlist for user-supplied URLs. Entries are `hostname` or "
3125 "`hostname:port` (bracketed for IPv6, e.g. `[::1]:8080`). Allowlisted "
3126 "hosts skip the blocked-network check in validate_url() but still "
3127 "resolve DNS. Use this to permit legitimate internal targets, e.g. "
3128 "an internal OpenAPI/MCP server."
3129 ),
3130 )
3131 provider_url_destination_allowed_hosts: list[str] | None = Field(
3132 None,
3133 description="Allowlist of hosts a request may redirect a provider call's destination URL to.",
3134 )
3137class ConfigYAML(LiteLLMPydanticObjectBase):
3138 """
3139 Documents all the fields supported by the config.yaml
3140 """
3142 environment_variables: dict | None = Field(
3143 None,
3144 description="Object to pass in additional environment variables via POST request",
3145 )
3146 model_list: list[ModelParams] | None = Field(
3147 None,
3148 description="List of supported models on the server, with model-specific configs",
3149 )
3150 litellm_settings: dict | None = Field(
3151 None,
3152 description="litellm Module settings. See __init__.py for all, example litellm.drop_params=True, litellm.set_verbose=True, litellm.api_base, litellm.cache",
3153 )
3154 general_settings: ConfigGeneralSettings | None = None
3155 worker_registry: list[WorkerRegistryEntry] | None = Field(
3156 None,
3157 description=(
3158 "Global Control Plane: the independent proxy instances this instance's admin UI manages. "
3159 "Setting it makes this a control plane, which serves the UI and does not route LLM requests. "
3160 "Enterprise-only"
3161 ),
3162 )
3163 router_settings: UpdateRouterConfig | None = Field(
3164 None,
3165 description="litellm router object settings. See router.py __init__ for all, example router.num_retries=5, router.timeout=5, router.max_retries=5, router.retry_after=5",
3166 )
3168 model_config = ConfigDict(protected_namespaces=())
3171from litellm.models.verification_token import ( # noqa: E402
3172 LiteLLM_DeletedVerificationToken as LiteLLM_DeletedVerificationToken,
3173)
3174from litellm.models.verification_token import ( # noqa: E402
3175 LiteLLM_VerificationToken as LiteLLM_VerificationToken,
3176)
3179class LiteLLM_VerificationTokenView(LiteLLM_VerificationToken):
3180 """
3181 Combined view of litellm verification token + litellm team table (select values)
3182 """
3184 team_spend: float | None = None
3185 team_alias: str | None = None
3186 team_tpm_limit: int | None = None
3187 team_rpm_limit: int | None = None
3188 team_tpd_limit: int | None = None
3189 team_max_budget: float | None = None
3190 team_soft_budget: float | None = None
3191 team_model_max_budget: dict[str, object] | None = None
3192 team_models: list = []
3193 team_blocked: bool = False
3194 soft_budget: float | None = None
3195 team_model_aliases: dict | None = None
3196 team_member: Member | None = None
3197 team_metadata: dict | None = None
3198 team_object_permission_id: str | None = None
3200 # Team Member Specific Params
3201 team_member_spend: float | None = None
3202 team_member_tpm_limit: int | None = None
3203 team_member_rpm_limit: int | None = None
3205 # End User Params
3206 end_user_id: str | None = None
3207 end_user_tpm_limit: int | None = None
3208 end_user_rpm_limit: int | None = None
3209 end_user_tpd_limit: int | None = None
3210 end_user_max_budget: float | None = None
3211 end_user_model_max_budget: dict | None = None
3213 # Organization Params
3214 organization_alias: str | None = None
3215 organization_max_budget: float | None = None
3216 organization_tpm_limit: int | None = None
3217 organization_rpm_limit: int | None = None
3218 organization_metadata: dict | None = None
3220 # Project Params
3221 project_alias: str | None = None
3222 project_metadata: dict | None = None
3224 # Time stamps
3225 last_refreshed_at: float | None = None # last time joint view was pulled from db
3227 def __init__(self, **kwargs):
3228 # Handle litellm_budget_table_* keys (budget table overrides when key value is None or empty)
3229 for key, value in list(kwargs.items()):
3230 if key.startswith("litellm_budget_table_") and value is not None: 3230 ↛ 3232line 3230 didn't jump to line 3232 because the condition on line 3230 was never true
3231 # Extract the corresponding attribute name
3232 attr_name = key.replace("litellm_budget_table_", "")
3233 # Use key's value from kwargs (from DB view), not class default
3234 current = kwargs.get(attr_name)
3235 if current is None:
3236 current = getattr(self, attr_name, None)
3237 # Apply budget value when key has no value, or for model_max_budget when key has empty dict
3238 should_apply = current is None or (
3239 attr_name == "model_max_budget" and isinstance(current, dict) and len(current) == 0
3240 )
3241 if should_apply:
3242 kwargs[attr_name] = value
3243 if key == "end_user_id" and value is not None and isinstance(value, int): 3243 ↛ 3244line 3243 didn't jump to line 3244 because the condition on line 3243 was never true
3244 kwargs[key] = str(value)
3246 if kwargs.get("organization_id") is not None: 3246 ↛ 3247line 3246 didn't jump to line 3247 because the condition on line 3246 was never true
3247 kwargs["org_id"] = kwargs.pop("organization_id")
3248 # Initialize the superclass
3249 super().__init__(**kwargs)
3252class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response object for user api key auth
3253 """
3254 Return the row in the db
3255 """
3257 api_key: str | None = None
3258 user_role: LitellmUserRoles | None = None
3259 allowed_model_region: AllowedModelRegion | None = None
3260 parent_otel_span: Span | None = None
3261 rpm_limit_per_model: dict[str, int] | None = None
3262 tpm_limit_per_model: dict[str, int] | None = None
3263 user_tpm_limit: int | None = None
3264 user_rpm_limit: int | None = None
3265 user_email: str | None = None
3266 user_spend: float | None = None
3267 user_max_budget: float | None = None
3268 # Values stay `object` rather than BudgetConfig: this is the raw JSON column,
3269 # and validating it here would make one malformed row fail auth outright.
3270 # resolve_model_budget validates the single entry a request actually needs.
3271 user_model_max_budget: Mapping[str, object] | None = None
3272 request_route: str | None = None
3273 is_session_token: bool = False
3274 # Server-only marker set exclusively by the MCP gateway admission path
3275 # (reload_admitted_user) for a keyless user-subject admitted via a gateway DCR session
3276 # bearer or bridge envelope. Not a DB column and never populated from caller-controlled key
3277 # metadata or JWT claims, so it cannot be forged to gain the team-inherited MCP grant union
3278 # or to escape the caller-Authorization egress scrub. exclude=True keeps it out of serialization.
3279 mcp_admitted_user_subject: bool = Field(default=False, exclude=True)
3280 # team_id -> that team's mcp_rpm_limit map, for a keyless admitted subject that reaches MCP
3281 # servers through several teams at once and therefore has no single team_id for the limiter to
3282 # key off. Server-only and stripped from validated input for the same reason as the marker
3283 # above: a forged entry would let a caller pick which team's rpm bucket it is charged against.
3284 mcp_source_team_rpm_limits: dict[str, dict[str, int]] | None = Field(default=None, exclude=True)
3285 # The single MCP server_id a gateway session bearer was scoped to at authorize time (RFC 8707
3286 # resource), or None for an aggregate-scope session. A RESTRICTION intersected against the live
3287 # grant resolution, never a grant. Server-only, set exclusively by the MCP gateway admission
3288 # path via post-construction assignment and stripped from validated input like the markers
3289 # above; a forged value could at most narrow, but the stripping keeps the field's provenance
3290 # single-owner so its meaning stays trustworthy.
3291 mcp_session_resource_server_id: str | None = Field(default=None, exclude=True)
3292 mcp_toolset_id: str | None = Field(default=None, exclude=True)
3293 via_virtual_key: bool = Field(
3294 default=False,
3295 exclude=True,
3296 description=(
3297 "Server-only marker set exclusively by the DB virtual-key and master-key auth paths via "
3298 "post-construction assignment. Stripped from validated input so custom auth handlers, JWT "
3299 "claims, or key metadata cannot forge it. Gates overwrite_user_with_key_hash stamping: only "
3300 "a credential the proxy itself validated as a key may be forwarded as the provider-facing "
3301 "user id."
3302 ),
3303 )
3304 agent_caller: AgentCaller | None = Field(
3305 default=None,
3306 exclude=True,
3307 description=(
3308 "Set per request from the x-litellm-user-id / x-litellm-team-id headers an agent echoes back on "
3309 "calls made with its own key. Every check treats it as a ceiling, so a forged value can only "
3310 "narrow the agent's access."
3311 ),
3312 )
3313 budget_reservation: dict[str, Any] | None = Field(default=None, exclude=True)
3314 team_budget_snapshot: TeamBudgetSnapshot | None = Field(default=None, exclude=True)
3315 user_budget_snapshot: UserBudgetSnapshot | None = Field(default=None, exclude=True)
3316 org_budget_snapshot: OrgBudgetSnapshot | None = Field(default=None, exclude=True)
3317 matched_model_access_groups: list[str] | None = Field(default=None, exclude=True)
3318 budget_throttle_pct: float | None = Field(default=None, exclude=True)
3319 user: Any | None = None # Expanded user object when expand=user is used
3320 created_by_user: Any | None = None # Expanded created_by user when expand=user is used
3321 end_user_object_permission: LiteLLM_ObjectPermissionTable | None = None
3322 # Team object_permission preloaded in auth (e.g. get_team_object) to avoid
3323 # per-request object_permission fetches in downstream checks (vector stores, etc.)
3324 team_object_permission: LiteLLM_ObjectPermissionTable | None = None
3325 # Decoded upstream IdP claims (groups, roles, etc.) propagated by JWT auth machinery
3326 # and forwarded into outbound tokens by guardrails such as MCPJWTSigner.
3327 jwt_claims: dict | None = None
3329 model_config = ConfigDict(arbitrary_types_allowed=True)
3331 @model_validator(mode="before")
3332 @classmethod
3333 def check_api_key(cls, values):
3334 # If values is already an instance (not a dict), return it as-is
3335 if not isinstance(values, dict): 3335 ↛ 3336line 3335 didn't jump to line 3336 because the condition on line 3335 was never true
3336 return values
3337 # mcp_admitted_user_subject is a server-only marker, set ONLY by the MCP gateway admission
3338 # path via post-construction assignment. Strip it from any validated input (constructor
3339 # kwargs, model_validate, a JWT/key claim splat) so it can never be forged from caller data.
3340 values.pop("mcp_admitted_user_subject", None)
3341 values.pop("mcp_source_team_rpm_limits", None)
3342 values.pop("mcp_session_resource_server_id", None)
3343 values.pop("mcp_toolset_id", None)
3344 values.pop("via_virtual_key", None)
3345 values.pop("agent_caller", None)
3346 if values.get("api_key") is not None:
3347 values.update({"token": cls._safe_hash_litellm_api_key(values.get("api_key"))})
3348 if isinstance(values.get("api_key"), str): 3348 ↛ 3350line 3348 didn't jump to line 3350 because the condition on line 3348 was always true
3349 values.update({"api_key": cls._safe_hash_litellm_api_key(values.get("api_key"))})
3350 return values
3352 @classmethod
3353 def _safe_hash_litellm_api_key(cls, api_key: str) -> str:
3354 """
3355 Helper to ensure all logged keys are hashed
3356 Covers:
3357 1. Regular API keys from LiteLLM DB
3358 2. JWT tokens used for connecting to LiteLLM API
3359 """
3360 normalized = api_key
3361 if normalized[:7].lower() == "bearer ": 3361 ↛ 3362line 3361 didn't jump to line 3362 because the condition on line 3361 was never true
3362 normalized = normalized[7:]
3363 if normalized.startswith("sk-"):
3364 return hash_token(normalized)
3365 from litellm.proxy.auth.handle_jwt import JWTHandler
3367 if JWTHandler.is_jwt(token=normalized):
3368 return f"hashed-jwt-{hash_token(token=normalized)}"
3369 return normalized
3371 @classmethod
3372 def get_litellm_internal_health_check_user_api_key_auth(cls) -> "UserAPIKeyAuth":
3373 """
3374 Returns a `UserAPIKeyAuth` object for the litellm internal health check service account.
3376 This is used to track number of requests/spend for health check calls.
3377 """
3378 from litellm.constants import LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME
3380 return cls(
3381 api_key=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME,
3382 team_id=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME,
3383 key_alias=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME,
3384 team_alias=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME,
3385 )
3387 @classmethod
3388 def get_litellm_cli_user_api_key_auth(cls) -> "UserAPIKeyAuth":
3389 """
3390 Returns a `UserAPIKeyAuth` object for the litellm internal health check service account.
3392 This is used to track number of requests/spend for health check calls.
3393 """
3394 from litellm.constants import LITTELM_CLI_SERVICE_ACCOUNT_NAME
3396 return cls(
3397 api_key=LITTELM_CLI_SERVICE_ACCOUNT_NAME,
3398 team_id=LITTELM_CLI_SERVICE_ACCOUNT_NAME,
3399 key_alias=LITTELM_CLI_SERVICE_ACCOUNT_NAME,
3400 team_alias=LITTELM_CLI_SERVICE_ACCOUNT_NAME,
3401 )
3403 @classmethod
3404 def get_litellm_internal_jobs_user_api_key_auth(cls) -> "UserAPIKeyAuth":
3405 """
3406 Returns a `UserAPIKeyAuth` object for internal LiteLLM jobs like key rotation.
3408 This is used to track actions performed by automated system jobs.
3409 """
3410 from litellm.constants import LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME
3412 return cls(
3413 api_key=LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME,
3414 team_id="system",
3415 key_alias=LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME,
3416 team_alias="system",
3417 user_id="system",
3418 user_role=LitellmUserRoles.PROXY_ADMIN,
3419 )
3421 @property
3422 def is_team_service_account(self) -> bool:
3423 return (
3424 self.user_id is None
3425 and self.team_id is not None
3426 and bool(self.metadata)
3427 and self.metadata.get("service_account_id") is not None
3428 )
3431def user_api_key_has_admin_view(user_api_key_dict: UserAPIKeyAuth) -> bool:
3432 """Return True if the caller's role grants unscoped read access to all
3433 tenant resources (managed files, batches, vector stores, spend rows, etc).
3435 Lives on _types.py so leaf modules (e.g. litellm.llms.base_llm.managed_resources)
3436 can use it without pulling in litellm.proxy.utils via management_endpoints.
3437 """
3438 return user_api_key_dict.user_role in (
3439 LitellmUserRoles.PROXY_ADMIN,
3440 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
3441 )
3444class UserInfoResponse(LiteLLMPydanticObjectBase):
3445 user_id: str | None
3446 user_info: dict | BaseModel | None
3447 keys: list
3448 teams: list
3451class UserInfoV2Response(LiteLLMPydanticObjectBase):
3452 """
3453 Response model for GET /v2/user/info
3455 Returns ONLY the user object - no keys, no teams objects.
3456 This is a lightweight alternative to UserInfoResponse.
3457 """
3459 user_id: str
3460 user_email: str | None = None
3461 user_alias: str | None = None
3462 user_role: str | None = None
3463 spend: float = 0.0
3464 max_budget: float | None = None
3465 models: list[str] = []
3466 budget_duration: str | None = None
3467 budget_reset_at: datetime | None = None
3468 metadata: dict | None = None
3469 created_at: datetime | None = None
3470 updated_at: datetime | None = None
3471 sso_user_id: str | None = None
3472 teams: list[str] = [] # Just team IDs, not full team objects
3473 object_permission: LiteLLM_ObjectPermissionTable | None = None
3474 model_max_budget: Mapping[str, object] | None = None
3475 model_max_budget_usage: Mapping[str, Mapping[str, object]] | None = None
3478from litellm.models.config import LiteLLM_Config as LiteLLM_Config # noqa: E402
3479from litellm.models.organization_membership import ( # noqa: E402
3480 LiteLLM_OrganizationMembershipTable as LiteLLM_OrganizationMembershipTable,
3481)
3484class LiteLLM_OrganizationTableUpdate(LiteLLM_BudgetTable):
3485 """Represents user-controllable params for a LiteLLM_OrganizationTable record"""
3487 organization_id: str | None = None
3488 organization_alias: str | None = None
3489 budget_id: str | None = None
3490 spend: float | None = None
3491 metadata: dict | None = None
3492 models: list[str] | None = None
3493 updated_by: str | None = None
3494 object_permission: LiteLLM_ObjectPermissionBase | None = None
3495 model_tpm_limit: dict[str, int] | None = None
3496 model_rpm_limit: dict[str, int] | None = None
3498 @model_validator(mode="before")
3499 @classmethod
3500 def set_model_info(cls, values):
3501 for field in LiteLLM_ManagementEndpoint_MetadataFields:
3502 if values.get(field) is not None:
3503 # add to metadata
3504 if values.get("metadata") is None:
3505 values.update({"metadata": {}})
3506 values["metadata"][field] = values.get(field)
3507 values.pop(field)
3508 return values
3511class OrganizationUpdateRequestV2(LiteLLMPydanticObjectBase):
3512 """
3513 Typed PATCH body for ``/v2/organization/{organization_id}`` (RFC 7396 merge-patch).
3515 Presence is read from ``model_fields_set``, so a sent field is written and an omitted one is
3516 left untouched. ``extra="forbid"`` makes an unknown key a 422 rather than a silent no-op, since
3517 the contract hinges on which keys are present. See the endpoint for the per-field clear tokens.
3518 """
3520 model_config = ConfigDict(extra="forbid")
3522 organization_alias: str | None = None
3523 models: list[str] | None = None
3524 metadata: dict | None = None
3525 tpm_limit: int | None = None
3526 rpm_limit: int | None = None
3527 max_budget: float | None = None
3528 soft_budget: float | None = None
3529 max_parallel_requests: int | None = None
3530 model_max_budget: dict | None = None
3531 budget_duration: str | None = None
3532 object_permission: LiteLLM_ObjectPermissionBase | None = None
3535from litellm.models.organization import ( # noqa: E402
3536 LiteLLM_OrganizationTable as LiteLLM_OrganizationTable,
3537)
3538from litellm.models.user import LiteLLM_UserTable as LiteLLM_UserTable # noqa: E402
3541class LiteLLM_OrganizationTableWithMembers(LiteLLM_OrganizationTable):
3542 """Returned by the /organization/info endpoint and /organization/list endpoint"""
3544 members: list[LiteLLM_OrganizationMembershipTable] = []
3545 teams: list[LiteLLM_TeamTable] = []
3546 litellm_budget_table: LiteLLM_BudgetTable | None = None
3547 created_at: datetime
3548 updated_at: datetime
3551class NewOrganizationResponse(LiteLLM_OrganizationTable):
3552 organization_id: str
3553 created_at: datetime
3554 updated_at: datetime
3557### PROJECT MANAGEMENT TYPES ###
3560class ProjectBase(LiteLLMPydanticObjectBase):
3561 """Base fields shared by project create/update requests"""
3563 project_id: str | None = None
3564 project_alias: str | None = None
3565 team_id: str | None = None
3566 metadata: dict | None = None
3567 models: list[str] | None = None
3568 blocked: bool = False
3571class NewProjectRequest(LiteLLM_BudgetTable):
3572 """Request model for POST /project/new"""
3574 project_id: str | None = None
3575 project_alias: str | None = None
3576 description: str | None = None
3577 team_id: str
3578 budget_id: str | None = None
3579 metadata: dict | None = None
3580 tags: list[str] | None = None
3581 guardrails: list[str] | None = None
3582 policies: list[str] | None = None
3583 models: list[str] = []
3584 model_rpm_limit: dict | None = None
3585 model_tpm_limit: dict | None = None
3586 model_itpm_limit: Mapping[str, int] | None = None
3587 model_otpm_limit: Mapping[str, int] | None = None
3588 blocked: bool = False
3589 object_permission: LiteLLM_ObjectPermissionBase | None = None
3591 @model_validator(mode="before")
3592 @classmethod
3593 def set_model_info(cls, values):
3594 if "tags" in values and values["tags"] is not None:
3595 if not isinstance(values["tags"], list):
3596 raise ValueError(f"tags must be a list of strings, got {type(values['tags']).__name__}")
3597 for field in LiteLLM_ManagementEndpoint_MetadataFields:
3598 if values.get(field) is not None:
3599 if values.get("metadata") is None:
3600 values.update({"metadata": {}})
3601 values["metadata"][field] = values.get(field)
3602 values.pop(field)
3603 return values
3606class UpdateProjectRequest(LiteLLM_BudgetTable):
3607 """Request model for POST /project/update"""
3609 project_id: str
3610 project_alias: str | None = None
3611 description: str | None = None
3612 team_id: str | None = None
3613 metadata: dict | None = None
3614 tags: list[str] | None = None
3615 guardrails: list[str] | None = None
3616 policies: list[str] | None = None
3617 models: list[str] | None = None
3618 model_rpm_limit: dict | None = None
3619 model_tpm_limit: dict | None = None
3620 model_itpm_limit: Mapping[str, int] | None = None
3621 model_otpm_limit: Mapping[str, int] | None = None
3622 blocked: bool | None = None
3623 budget_id: str | None = None
3624 object_permission: LiteLLM_ObjectPermissionBase | None = None
3626 @model_validator(mode="before")
3627 @classmethod
3628 def set_model_info(cls, values):
3629 if "tags" in values and values["tags"] is not None:
3630 if not isinstance(values["tags"], list):
3631 raise ValueError(f"tags must be a list of strings, got {type(values['tags']).__name__}")
3632 for field in LiteLLM_ManagementEndpoint_MetadataFields:
3633 if values.get(field) is not None:
3634 if values.get("metadata") is None:
3635 values.update({"metadata": {}})
3636 values["metadata"][field] = values.get(field)
3637 values.pop(field)
3638 return values
3641class DeleteProjectRequest(LiteLLMPydanticObjectBase):
3642 """Request model for DELETE /project/delete"""
3644 project_ids: list[str]
3647from litellm.models.project import ( # noqa: E402
3648 LiteLLM_ProjectTable as LiteLLM_ProjectTable,
3649)
3652class NewProjectResponse(LiteLLM_ProjectTable):
3653 """Response model for POST /project/new"""
3655 project_id: str
3656 created_at: datetime
3657 updated_at: datetime
3660class LiteLLM_ProjectTableCachedObj(LiteLLM_ProjectTable):
3661 """Cached version for auth checks. Mirrors LiteLLM_TeamTableCachedObj pattern."""
3663 last_refreshed_at: float | None = None
3666class LiteLLM_UserTableFiltered(BaseModel): # done to avoid exposing sensitive data
3667 user_id: str
3668 user_email: str | None = None
3671class LiteLLM_UserTableWithKeyCount(LiteLLM_UserTable):
3672 key_count: int = 0
3675from litellm.models.access_group import ( # noqa: E402
3676 LiteLLM_AccessGroupTable as LiteLLM_AccessGroupTable,
3677)
3678from litellm.models.end_user import ( # noqa: E402
3679 LiteLLM_EndUserTable as LiteLLM_EndUserTable,
3680)
3681from litellm.models.spend_logs import ( # noqa: E402
3682 LiteLLM_ErrorLogs as LiteLLM_ErrorLogs,
3683)
3684from litellm.models.spend_logs import ( # noqa: E402
3685 LiteLLM_SpendLogs as LiteLLM_SpendLogs,
3686)
3687from litellm.models.tag import LiteLLM_TagTable as LiteLLM_TagTable # noqa: E402
3689AUDIT_ACTIONS = Literal["created", "updated", "deleted", "blocked", "unblocked", "rotated", "kill_switch_fired"]
3692class LiteLLM_AuditLogs(LiteLLMPydanticObjectBase):
3693 id: str
3694 updated_at: datetime
3695 changed_by: Any | None = None
3696 changed_by_api_key: str | None = None
3697 action: AUDIT_ACTIONS
3698 table_name: LitellmTableNames
3699 object_id: str
3700 before_value: Json | None = None
3701 updated_values: Json | None = None
3703 @model_validator(mode="before")
3704 @classmethod
3705 def cast_changed_by_to_str(cls, values):
3706 if values.get("changed_by") is not None:
3707 values["changed_by"] = str(values["changed_by"])
3708 return values
3710 @model_validator(mode="after")
3711 def mask_api_keys(self):
3712 from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker
3714 masker: Final = SensitiveDataMasker(sensitive_patterns={"key"})
3716 if self.before_value is not None:
3717 json_before_value: dict | None = None
3718 if isinstance(self.before_value, str):
3719 json_before_value = json.loads(self.before_value)
3720 elif isinstance(self.before_value, dict):
3721 json_before_value = self.before_value
3723 if json_before_value is not None:
3724 json_before_value = masker.mask_dict(json_before_value)
3725 self.before_value = json.dumps(json_before_value, default=str)
3727 if self.updated_values is not None:
3728 json_updated_values: dict | None = None
3729 if isinstance(self.updated_values, str):
3730 json_updated_values = json.loads(self.updated_values)
3731 elif isinstance(self.updated_values, dict):
3732 json_updated_values = self.updated_values
3734 if json_updated_values is not None:
3735 json_updated_values = masker.mask_dict(json_updated_values)
3736 self.updated_values = json.dumps(json_updated_values, default=str)
3738 return self
3741class LiteLLM_SpendLogs_ResponseObject(LiteLLMPydanticObjectBase):
3742 response: list[LiteLLM_SpendLogs | Any] | None = None
3745class TokenCountRequest(LiteLLMPydanticObjectBase):
3746 model: str
3747 prompt: str | None = None
3748 messages: list[dict] | None = None
3749 """
3750 Anthropic token counting endpoint uses /messages
3751 """
3753 contents: list[dict] | None = None
3754 """
3755 Google /countTokens endpoint expects contents to be a list of dicts with the following structure:
3756 """
3758 tools: list[dict] | None = None
3759 system: Any | None = None
3762class CallInfo(LiteLLMPydanticObjectBase):
3763 """Used for slack budget alerting"""
3765 spend: float
3766 max_budget: float | None = None
3767 soft_budget: float | None = None
3768 token: str | None = Field(default=None, description="Hashed value of that key")
3769 customer_id: str | None = None
3770 user_id: str | None = None
3771 team_id: str | None = None
3772 team_alias: str | None = None
3773 organization_id: str | None = None
3774 user_email: str | None = None
3775 key_alias: str | None = None
3776 projected_exceeded_date: str | None = None
3777 projected_spend: float | None = None
3778 event_group: Litellm_EntityType
3779 alert_emails: list[str] | None = Field(
3780 default=None,
3781 description="Additional email addresses to send alerts to (e.g., from team metadata)",
3782 )
3783 max_budget_alert_emails: dict[str, list[str]] | None = Field(
3784 default=None,
3785 description="Map of threshold percentage to email recipients (e.g., {'50': ['a@co.com'], '75': ['a@co.com', 'b@co.com']})",
3786 )
3789class WebhookEvent(CallInfo):
3790 event: Literal[
3791 "budget_crossed",
3792 "max_budget_alert",
3793 "soft_budget_crossed",
3794 "threshold_crossed",
3795 "projected_limit_exceeded",
3796 "key_created",
3797 "key_rotated",
3798 "internal_user_created",
3799 "spend_tracked",
3800 ]
3801 event_message: str # human-readable description of event
3802 event_group: Litellm_EntityType
3805class SpecialModelNames(enum.Enum):
3806 all_team_models = "all-team-models"
3807 all_proxy_models = "all-proxy-models"
3808 no_default_models = "no-default-models"
3811class SpecialMCPServerNames(enum.Enum):
3812 no_mcp_servers = "no-mcp-servers"
3815class SpecialProxyStrings(enum.Enum):
3816 default_user_id = "default_user_id" # global proxy admin
3819class InvitationNew(LiteLLMPydanticObjectBase):
3820 user_id: str
3823class InvitationUpdate(LiteLLMPydanticObjectBase):
3824 invitation_id: str
3825 is_accepted: bool
3828class InvitationDelete(LiteLLMPydanticObjectBase):
3829 invitation_id: str
3832class InvitationModel(LiteLLMPydanticObjectBase):
3833 id: str
3834 user_id: str
3835 is_accepted: bool
3836 accepted_at: datetime | None
3837 expires_at: datetime
3838 created_at: datetime
3839 created_by: str
3840 updated_at: datetime
3841 updated_by: str
3844class InvitationClaim(LiteLLMPydanticObjectBase):
3845 invitation_link: str
3846 user_id: str
3847 password: str
3850class ConfigFieldInfo(LiteLLMPydanticObjectBase):
3851 field_name: str
3852 field_value: Any
3853 source: Literal["config", "db", "env", "default", "unset"] = "unset"
3854 editable: bool = True
3857class CallbackOnUI(LiteLLMPydanticObjectBase):
3858 litellm_callback_name: str
3859 litellm_callback_params: list | None
3860 ui_callback_name: str
3863class AllCallbacks(LiteLLMPydanticObjectBase):
3864 langfuse: CallbackOnUI = CallbackOnUI(
3865 litellm_callback_name="langfuse",
3866 ui_callback_name="Langfuse",
3867 litellm_callback_params=[
3868 "LANGFUSE_PUBLIC_KEY",
3869 "LANGFUSE_SECRET_KEY",
3870 "LANGFUSE_HOST",
3871 ],
3872 )
3874 otel: CallbackOnUI = CallbackOnUI(
3875 litellm_callback_name="otel",
3876 ui_callback_name="OpenTelemetry",
3877 litellm_callback_params=[
3878 "OTEL_EXPORTER",
3879 "OTEL_EXPORTER_OTLP_PROTOCOL",
3880 "OTEL_ENDPOINT",
3881 "OTEL_TRACES_ENDPOINT",
3882 "OTEL_HEADERS",
3883 ],
3884 )
3886 s3: CallbackOnUI = CallbackOnUI(
3887 litellm_callback_name="s3",
3888 ui_callback_name="s3 Bucket (AWS)",
3889 litellm_callback_params=[
3890 "AWS_ACCESS_KEY_ID",
3891 "AWS_SECRET_ACCESS_KEY",
3892 "AWS_REGION_NAME",
3893 "S3_LOG_PROMPTS_ONLY",
3894 ],
3895 )
3897 azure_sentinel: CallbackOnUI = CallbackOnUI(
3898 litellm_callback_name="azure_sentinel",
3899 ui_callback_name="Azure Sentinel",
3900 litellm_callback_params=[
3901 "AZURE_SENTINEL_DCR_IMMUTABLE_ID",
3902 "AZURE_SENTINEL_ENDPOINT",
3903 "AZURE_SENTINEL_TENANT_ID",
3904 "AZURE_SENTINEL_CLIENT_ID",
3905 "AZURE_SENTINEL_CLIENT_SECRET",
3906 "AZURE_SENTINEL_STREAM_NAME",
3907 ],
3908 )
3910 openmeter: CallbackOnUI = CallbackOnUI(
3911 litellm_callback_name="openmeter",
3912 ui_callback_name="OpenMeter",
3913 litellm_callback_params=[
3914 "OPENMETER_API_ENDPOINT",
3915 "OPENMETER_API_KEY",
3916 ],
3917 )
3919 custom_callback_api: CallbackOnUI = CallbackOnUI(
3920 litellm_callback_name="custom_callback_api",
3921 litellm_callback_params=["GENERIC_LOGGER_ENDPOINT", "GENERIC_LOGGER_HEADERS"],
3922 ui_callback_name="Custom Callback API",
3923 )
3925 generic_api: CallbackOnUI = CallbackOnUI(
3926 litellm_callback_name="generic_api",
3927 litellm_callback_params=["GENERIC_LOGGER_ENDPOINT", "GENERIC_LOGGER_HEADERS"],
3928 ui_callback_name="Custom Callback API",
3929 )
3931 datadog: CallbackOnUI = CallbackOnUI(
3932 litellm_callback_name="datadog",
3933 litellm_callback_params=["DD_API_KEY", "DD_SITE"],
3934 ui_callback_name="Datadog",
3935 )
3937 braintrust: CallbackOnUI = CallbackOnUI(
3938 litellm_callback_name="braintrust",
3939 litellm_callback_params=["BRAINTRUST_API_KEY", "BRAINTRUST_API_BASE"],
3940 ui_callback_name="Braintrust",
3941 )
3943 langsmith: CallbackOnUI = CallbackOnUI(
3944 litellm_callback_name="langsmith",
3945 litellm_callback_params=[
3946 "LANGSMITH_API_KEY",
3947 "LANGSMITH_PROJECT",
3948 "LANGSMITH_DEFAULT_RUN_NAME",
3949 ],
3950 ui_callback_name="Langsmith",
3951 )
3953 lago: CallbackOnUI = CallbackOnUI(
3954 litellm_callback_name="lago",
3955 litellm_callback_params=[
3956 "LAGO_API_BASE",
3957 "LAGO_API_KEY",
3958 "LAGO_API_EVENT_CODE",
3959 "LAGO_API_CHARGE_BY",
3960 ],
3961 ui_callback_name="Lago Billing",
3962 )
3964 traceloop: CallbackOnUI = CallbackOnUI(
3965 litellm_callback_name="traceloop",
3966 litellm_callback_params=[
3967 "TRACELOOP_API_KEY",
3968 ],
3969 ui_callback_name="Traceloop",
3970 )
3972 galileo: CallbackOnUI = CallbackOnUI(
3973 litellm_callback_name="galileo",
3974 litellm_callback_params=[
3975 "GALILEO_API_KEY",
3976 "GALILEO_PROJECT_ID",
3977 "GALILEO_LOG_STREAM_ID",
3978 "GALILEO_BASE_URL",
3979 "GALILEO_USERNAME",
3980 "GALILEO_PASSWORD",
3981 ],
3982 ui_callback_name="Galileo",
3983 )
3985 newrelic: CallbackOnUI = CallbackOnUI(
3986 litellm_callback_name="newrelic",
3987 ui_callback_name="New Relic",
3988 litellm_callback_params=[
3989 "NEW_RELIC_AI_MONITORING_RECORD_CONTENT_ENABLED",
3990 ],
3991 )
3993 pointfive: CallbackOnUI = CallbackOnUI(
3994 litellm_callback_name="pointfive",
3995 ui_callback_name="PointFive",
3996 litellm_callback_params=[ # mutable-ok: the registry field is typed list
3997 "POINTFIVE_API_KEY",
3998 "POINTFIVE_API_URL",
3999 ],
4000 )
4003class HTTPExceptionErrorDetail(TypedDict):
4004 """The `{"error": <message>}` shape most proxy endpoints raise as `HTTPException.detail`."""
4006 error: ReadOnly[str]
4009class SpendLogsRouterMetadata(TypedDict):
4010 """
4011 Router provenance stamped on spend logs for deployments flagged with
4012 model_info.internal_router_model, correlating the requested model group
4013 with the provider deployment that served the call
4014 """
4016 requested_model: ReadOnly[str | None]
4017 selected_model: ReadOnly[str | None]
4018 selected_provider: ReadOnly[str | None]
4019 router_correlation_id: ReadOnly[str | None]
4022class SpendLogsMetadata(TypedDict):
4023 autorouter_baseline_observation: ReadOnly[str | None]
4024 """
4025 Specific metadata k,v pairs logged to spendlogs for easier cost tracking
4026 """
4028 additional_usage_values: dict | None # covers provider-specific usage information - e.g. prompt caching
4029 user_api_key: str | None
4030 user_api_key_alias: str | None
4031 user_api_key_team_id: str | None
4032 user_api_key_project_id: str | None
4033 user_api_key_project_alias: str | None
4034 user_api_key_org_id: str | None
4035 user_api_key_user_id: str | None
4036 user_api_key_team_alias: str | None
4037 spend_logs_metadata: dict | None # special param to log k,v pairs to spendlogs for a call
4038 requester_ip_address: str | None
4039 user_agent: ReadOnly[str | None]
4040 litellm_call_id: str | None
4041 applied_guardrails: list[str] | None
4042 mcp_tool_call_metadata: StandardLoggingMCPToolCall | None
4043 vector_store_request_metadata: list[StandardLoggingVectorStoreRequest] | None
4044 routing_decision: StandardLoggingRoutingDecision | None
4045 internal_call_origin: InternalCallOrigin | None
4046 guardrail_information: list[StandardLoggingGuardrailInformation] | None
4047 eval_information: Any | None
4048 status: StandardLoggingPayloadStatus
4049 proxy_server_request: str | None
4050 batch_models: list[str] | None
4051 batch_successful_requests: int | None # writable-ok: built by assignment like every sibling key in this TypedDict
4052 batch_failed_requests: int | None # writable-ok: built by assignment like every sibling key in this TypedDict
4053 error_information: StandardLoggingPayloadErrorInformation | None
4054 usage_object: dict | None
4055 model_map_information: StandardLoggingModelInformation | None
4056 cold_storage_object_key: str | None # S3/GCS object key for cold storage retrieval
4057 litellm_overhead_time_ms: float | None # LiteLLM overhead time in milliseconds
4058 attempted_retries: int | None # Number of retries attempted (0 = first attempt succeeded)
4059 max_retries: int | None # Max retries configured for this request
4060 attempted_fallbacks: ReadOnly[int | None] # Number of fallbacks attempted (0 = primary model group served)
4061 original_model_group: ReadOnly[str | None] # Model group requested before any fallbacks
4062 cost_breakdown: CostBreakdown | None # Detailed cost breakdown (input_cost, output_cost, margin, discount, etc.)
4063 compression_savings: CompressionSavingsMetadata | None
4064 autorouter_savings: ReadOnly[float | None]
4065 autorouter_savings_estimate: ReadOnly[Mapping[str, JsonValue] | None]
4066 litellm_gateway_injected_cache: ReadOnly[str | None]
4067 router_metadata: ReadOnly[SpendLogsRouterMetadata | None] # None = deployment not flagged internal_router_model
4068 azure_spillover: ReadOnly[AzureSpillover | None] # None = Azure did not report spillover
4071class SpendLogsPayload(TypedDict):
4072 request_id: str
4073 call_type: str
4074 api_key: str
4075 spend: float
4076 total_tokens: int
4077 prompt_tokens: int
4078 completion_tokens: int
4079 startTime: datetime | str
4080 endTime: datetime | str
4081 completionStartTime: datetime | str | None
4082 model: str
4083 model_id: str | None
4084 model_group: str | None
4085 mcp_namespaced_tool_name: str | None
4086 agent_id: str | None
4087 api_base: str
4088 user: str
4089 metadata: str # json str
4090 cache_hit: str
4091 cache_key: str
4092 request_tags: str # json str
4093 team_id: str | None
4094 organization_id: str | None
4095 end_user: str | None
4096 requester_ip_address: str | None
4097 custom_llm_provider: str | None
4098 messages: str | list | dict | None
4099 response: str | list | dict | None
4100 proxy_server_request: str | None
4101 session_id: str | None
4102 request_duration_ms: int | None
4103 status: Literal["success", "failure"]
4104 litellm_call_id: ReadOnly[str | None]
4107class SpanAttributes(str, enum.Enum):
4108 # Note: We've taken this from opentelemetry-semantic-conventions-ai
4109 # I chose to not add a new dependency to litellm for this
4111 # Semantic Conventions for LLM requests, this needs to be removed after
4112 # OpenTelemetry Semantic Conventions support Gen AI.
4113 # Issue at https://github.com/open-telemetry/opentelemetry-python/issues/3868
4114 # Refer to https://github.com/open-telemetry/semantic-conventions/blob/main/docs/gen-ai/llm-spans.md
4116 LLM_SYSTEM = "gen_ai.system"
4117 LLM_REQUEST_MODEL = "gen_ai.request.model"
4118 LLM_REQUEST_MAX_TOKENS = "gen_ai.request.max_tokens"
4119 LLM_REQUEST_TEMPERATURE = "gen_ai.request.temperature"
4120 LLM_REQUEST_TOP_P = "gen_ai.request.top_p"
4121 LLM_PROMPTS = "gen_ai.prompt"
4122 LLM_COMPLETIONS = "gen_ai.completion"
4123 LLM_RESPONSE_MODEL = "gen_ai.response.model"
4124 LLM_USAGE_COMPLETION_TOKENS = "gen_ai.usage.completion_tokens"
4125 LLM_USAGE_PROMPT_TOKENS = "gen_ai.usage.prompt_tokens"
4127 # OTEL 1.38 attributes
4128 GEN_AI_INPUT_MESSAGES = "gen_ai.input.messages"
4129 GEN_AI_OUTPUT_MESSAGES = "gen_ai.output.messages"
4130 GEN_AI_USAGE_INPUT_TOKENS = "gen_ai.usage.input_tokens"
4131 GEN_AI_USAGE_OUTPUT_TOKENS = "gen_ai.usage.output_tokens"
4132 GEN_AI_USAGE_TOTAL_TOKENS = "gen_ai.usage.total_tokens"
4133 GEN_AI_OPERATION_NAME = "gen_ai.operation.name"
4134 GEN_AI_REQUEST_ID = "gen_ai.request.id"
4135 GEN_AI_SYSTEM_INSTRUCTIONS = "gen_ai.system_instructions"
4136 GEN_AI_RESPONSE_FINISH_REASONS = "gen_ai.response.finish_reasons"
4138 LLM_TOKEN_TYPE = "gen_ai.token.type"
4139 # To be added
4140 # LLM_RESPONSE_FINISH_REASON = "gen_ai.response.finish_reasons"
4141 # LLM_RESPONSE_ID = "gen_ai.response.id"
4143 # LLM
4144 LLM_REQUEST_TYPE = "llm.request.type"
4145 LLM_USAGE_TOTAL_TOKENS = "llm.usage.total_tokens"
4146 LLM_USAGE_TOKEN_TYPE = "llm.usage.token_type"
4147 LLM_USER = "llm.user"
4148 LLM_HEADERS = "llm.headers"
4149 LLM_TOP_K = "llm.top_k"
4150 LLM_IS_STREAMING = "llm.is_streaming"
4151 LLM_FREQUENCY_PENALTY = "llm.frequency_penalty"
4152 LLM_PRESENCE_PENALTY = "llm.presence_penalty"
4153 LLM_CHAT_STOP_SEQUENCES = "llm.chat.stop_sequences"
4154 LLM_REQUEST_FUNCTIONS = "llm.request.functions"
4155 LLM_REQUEST_REPETITION_PENALTY = "llm.request.repetition_penalty"
4156 LLM_RESPONSE_FINISH_REASON = "llm.response.finish_reason"
4157 LLM_RESPONSE_STOP_REASON = "llm.response.stop_reason"
4158 LLM_CONTENT_COMPLETION_CHUNK = "llm.content.completion.chunk"
4160 # OpenAI
4161 LLM_OPENAI_RESPONSE_SYSTEM_FINGERPRINT = "gen_ai.openai.system_fingerprint"
4162 LLM_OPENAI_API_BASE = "gen_ai.openai.api_base"
4163 LLM_OPENAI_API_VERSION = "gen_ai.openai.api_version"
4164 LLM_OPENAI_API_TYPE = "gen_ai.openai.api_type"
4167class ManagementEndpointLoggingPayload(LiteLLMPydanticObjectBase):
4168 route: str
4169 request_data: dict
4170 response: dict | None = None
4171 exception: Any | None = None
4172 start_time: datetime | None = None
4173 end_time: datetime | None = None
4176class ProxyException(Exception):
4177 # NOTE: DO NOT MODIFY THIS
4178 # This is used to map exactly to OPENAI Exceptions
4179 def __init__(
4180 self,
4181 message: str,
4182 type: str,
4183 param: str | None,
4184 code: int | str | None = None, # maps to status code
4185 headers: dict[str, str] | None = None,
4186 openai_code: str | None = None, # maps to 'code' in openai
4187 provider_specific_fields: dict | None = None,
4188 ):
4189 self.message = str(message)
4190 super().__init__(self.message)
4191 self.type = type
4192 self.param = param
4193 self.openai_code = openai_code or code
4194 # If we look on official python OpenAI lib, the code should be a string:
4195 # https://github.com/openai/openai-python/blob/195c05a64d39c87b2dfdf1eca2d339597f1fce03/src/openai/types/shared/error_object.py#L11
4196 # Related LiteLLM issue: https://github.com/BerriAI/litellm/discussions/4834
4197 self.code = str(code)
4198 if headers is not None:
4199 for k, v in headers.items():
4200 if not isinstance(v, str): 4200 ↛ 4201line 4200 didn't jump to line 4201 because the condition on line 4200 was never true
4201 headers[k] = str(v)
4202 self.headers = headers or {}
4203 self.provider_specific_fields = provider_specific_fields
4204 # rules for proxyExceptions
4205 # Litellm router.py returns "No healthy deployment available" when there are no deployments available
4206 # Should map to 429 errors https://github.com/BerriAI/litellm/issues/2487
4207 if "No healthy deployment available" in self.message or "No deployments available" in self.message: 4207 ↛ 4208line 4207 didn't jump to line 4208 because the condition on line 4207 was never true
4208 self.code = "429"
4209 elif RouterErrors.no_deployments_with_tag_routing.value in self.message: 4209 ↛ 4210line 4209 didn't jump to line 4210 because the condition on line 4209 was never true
4210 self.code = "401"
4212 def to_dict(self) -> dict:
4213 """Converts the ProxyException instance to a dictionary."""
4214 error_dict: Final[dict[str, str | dict | None]] = {
4215 "message": self.message,
4216 "type": self.type,
4217 "param": self.param,
4218 "code": self.code,
4219 }
4220 if self.provider_specific_fields:
4221 error_dict["provider_specific_fields"] = self.provider_specific_fields
4222 return error_dict
4225class ModelAccessDeniedProxyException(ProxyException):
4226 def __init__(
4227 self,
4228 message: str,
4229 internal_message: str,
4230 type: str,
4231 param: str | None,
4232 code: int | str | None,
4233 ) -> None:
4234 super().__init__(message=message, type=type, param=param, code=code)
4235 self.internal_message: Final = internal_message
4237 def sanitized_internal_message(self) -> str:
4238 return self.internal_message.replace("\r", "").replace("\n", "")
4241class CommonProxyErrors(str, enum.Enum):
4242 db_not_connected_error = (
4243 "DB not connected. This endpoint needs a database; set DATABASE_URL to a "
4244 "PostgreSQL connection string (postgresql://...) to enable it. "
4245 "See https://docs.litellm.ai/docs/proxy/virtual_keys"
4246 )
4247 no_llm_router = "No models configured on proxy"
4248 not_allowed_access = "Admin-only endpoint. Not allowed to access this."
4249 not_premium_user = "You must be a LiteLLM Enterprise user to use this feature. If you have a license please set `LITELLM_LICENSE` in your env. Get a 7 day trial key here: https://www.litellm.ai/enterprise#trial. \nPricing: https://www.litellm.ai/#pricing"
4250 max_parallel_request_limit_reached = "Crossed TPM / RPM / Max Parallel Request Limit"
4251 missing_enterprise_package = "Missing litellm-enterprise package. Please install it to use this feature. Run `pip install litellm-enterprise`"
4252 missing_enterprise_package_docker = "This uses the enterprise folder - only available on the Docker image."
4255class SpendCalculateRequest(LiteLLMPydanticObjectBase):
4256 model: str | None = None
4257 messages: list | None = None
4258 completion_response: dict | None = None
4261class ProxyErrorTypes(str, enum.Enum):
4262 budget_exceeded = "budget_exceeded"
4263 """
4264 Object was over budget
4265 """
4266 no_db_connection = "no_db_connection"
4267 """
4268 No database connection
4269 """
4271 token_not_found_in_db = "token_not_found_in_db"
4272 """
4273 Requested token was not found in the database
4274 """
4276 key_model_access_denied = "key_model_access_denied"
4277 """
4278 Key does not have access to the model
4279 """
4281 team_model_access_denied = "team_model_access_denied"
4282 """
4283 Team does not have access to the model
4284 """
4286 user_model_access_denied = "user_model_access_denied"
4287 """
4288 User does not have access to the model
4289 """
4291 org_model_access_denied = "org_model_access_denied"
4292 """
4293 Organization does not have access to the model
4294 """
4296 project_model_access_denied = "project_model_access_denied"
4297 """
4298 Project does not have access to the model
4299 """
4301 agent_model_access_denied = "agent_model_access_denied"
4302 """
4303 The agent behind the key does not have access to the model
4304 """
4306 model_cost_map_missing = "model_cost_map_missing"
4308 expired_key = "expired_key"
4309 """
4310 Key has expired
4311 """
4313 auth_error = "auth_error"
4314 """
4315 General authentication error
4316 """
4318 auth_provider_unavailable = "auth_provider_unavailable"
4319 """
4320 The identity provider needed to authenticate the request (e.g. its JWKS endpoint) is unreachable
4321 """
4323 internal_server_error = "internal_server_error"
4324 """
4325 Internal server error
4326 """
4328 bad_request_error = "bad_request_error"
4329 """
4330 Bad request error
4331 """
4333 not_found_error = "not_found_error"
4334 """
4335 Not found error
4336 """
4338 validation_error = "validation_error"
4339 """
4340 Validation error
4341 """
4343 cache_ping_error = "cache_ping_error"
4344 """
4345 Cache ping error
4346 """
4348 team_member_permission_error = "team_member_permission_error"
4349 """
4350 Team member permission error
4351 """
4353 key_vector_store_access_denied = "key_vector_store_access_denied"
4354 """
4355 Key does not have access to the vector store
4356 """
4358 team_vector_store_access_denied = "team_vector_store_access_denied"
4359 """
4360 Team does not have access to the vector store
4361 """
4363 org_vector_store_access_denied = "org_vector_store_access_denied"
4364 """
4365 Organization does not have access to the vector store
4366 """
4368 team_member_already_in_team = "team_member_already_in_team"
4369 """
4370 Team member is already in team
4371 """
4373 tool_access_denied = "tool_access_denied"
4374 """
4375 Tool is not in the allowed tools list for this key/team
4376 """
4378 @classmethod
4379 def get_model_access_error_type_for_object(
4380 cls, object_type: Literal["key", "user", "team", "org", "project", "agent"]
4381 ) -> "ProxyErrorTypes":
4382 """
4383 Get the model access error type for object_type
4384 """
4385 if object_type == "key":
4386 return cls.key_model_access_denied
4387 elif object_type == "team":
4388 return cls.team_model_access_denied
4389 elif object_type == "user":
4390 return cls.user_model_access_denied
4391 elif object_type == "org":
4392 return cls.org_model_access_denied
4393 elif object_type == "project":
4394 return cls.project_model_access_denied
4395 elif object_type == "agent":
4396 return cls.agent_model_access_denied
4398 @classmethod
4399 def get_vector_store_access_error_type_for_object(
4400 cls, object_type: Literal["key", "team", "org"]
4401 ) -> "ProxyErrorTypes":
4402 """
4403 Get the vector store access error type for object_type
4404 """
4405 if object_type == "key":
4406 return cls.key_vector_store_access_denied
4407 elif object_type == "team":
4408 return cls.team_vector_store_access_denied
4409 elif object_type == "org":
4410 return cls.org_vector_store_access_denied
4413DB_CONNECTION_ERROR_TYPES: Final = (
4414 httpx.ConnectError,
4415 httpx.ConnectTimeout,
4416 httpx.ReadError,
4417 httpx.ReadTimeout,
4418)
4420# What a NON-IDEMPOTENT write (increment upsert) may retry: only ConnectError
4421# proves the statements never reached the database. Post-send errors are
4422# ambiguous; a stalled statement can leave its transaction open on the pooled
4423# connection, where a retry stacks a second increment set into the same commit.
4424# Idempotent writes (create_many with skip_duplicates) may retry the full tuple.
4425DB_RETRY_SAFE_ERROR_TYPES: Final = (httpx.ConnectError,)
4428class SSOUserDefinedValues(TypedDict):
4429 models: list[str]
4430 user_id: str
4431 user_email: str | None
4432 user_role: str | None
4433 max_budget: float | None
4434 budget_duration: str | None
4437class VirtualKeyEvent(LiteLLMPydanticObjectBase):
4438 created_by_user_id: str
4439 created_by_user_role: str
4440 created_by_key_alias: str | None
4441 request_kwargs: dict
4444class CreatePassThroughEndpoint(LiteLLMPydanticObjectBase):
4445 path: str
4446 target: str
4447 headers: dict
4450from litellm.models.team_membership import ( # noqa: E402
4451 LiteLLM_TeamMembership as LiteLLM_TeamMembership,
4452)
4454#### Organization / Team Member Requests ####
4457class MemberAddRequest(LiteLLMPydanticObjectBase):
4458 member: list[Member] | Member = Field(
4459 description="Member object or list of member objects to add. Each member must include either user_id or user_email, and a role"
4460 )
4462 def __init__(self, **data):
4463 member_data: Final = data.get("member")
4464 if isinstance(member_data, list):
4465 # If member is a list of dictionaries, convert each dictionary to a Member object
4466 members: Final = [Member(**item) if isinstance(item, dict) else item for item in member_data]
4467 # Replace member_data with the list of Member objects
4468 data["member"] = members
4469 elif isinstance(member_data, dict):
4470 # If member is a dictionary, convert it to a single Member object
4471 member: Final = Member(**member_data)
4472 # Replace member_data with the single Member object
4473 data["member"] = member
4474 # Call the superclass __init__ method to initialize the object
4475 super().__init__(**data)
4478class OrgMemberAddRequest(LiteLLMPydanticObjectBase):
4479 member: list[OrgMember] | OrgMember
4481 def __init__(self, **data):
4482 member_data: Final = data.get("member")
4483 if isinstance(member_data, list):
4484 # If member is a list of dictionaries, convert each dictionary to a Member object
4485 if all(isinstance(item, dict) for item in member_data): 4485 ↛ 4486line 4485 didn't jump to line 4486 because the condition on line 4485 was never true
4486 members = [OrgMember(**item) for item in member_data]
4487 else:
4488 members = [item for item in member_data]
4489 # Replace member_data with the list of Member objects
4490 data["member"] = members
4491 elif isinstance(member_data, dict): 4491 ↛ 4493line 4491 didn't jump to line 4493 because the condition on line 4491 was never true
4492 # If member is a dictionary, convert it to a single Member object
4493 member: Final = OrgMember(**member_data)
4494 # Replace member_data with the single Member object
4495 data["member"] = member
4496 # Call the superclass __init__ method to initialize the object
4497 super().__init__(**data)
4500class TeamAddMemberResponse(LiteLLM_TeamTable):
4501 updated_users: list[LiteLLM_UserTable]
4502 updated_team_memberships: list[LiteLLM_TeamMembership]
4505class OrganizationAddMemberResponse(LiteLLMPydanticObjectBase):
4506 organization_id: str
4507 updated_users: list[LiteLLM_UserTable]
4508 updated_organization_memberships: list[LiteLLM_OrganizationMembershipTable]
4511class MemberDeleteRequest(LiteLLMPydanticObjectBase):
4512 user_id: str | None = None
4513 user_email: str | None = None
4515 @model_validator(mode="before")
4516 @classmethod
4517 def check_user_info(cls, values):
4518 if values.get("user_id") is None and values.get("user_email") is None:
4519 raise ValueError("Either user id or user email must be provided")
4520 return values
4523class MemberUpdateResponse(LiteLLMPydanticObjectBase):
4524 user_id: str
4525 user_email: str | None = None
4528# Team Member Requests
4529class TeamMemberAddRequest(MemberAddRequest):
4530 """
4531 Request body for adding members to a team.
4533 Example:
4534 ```json
4535 {
4536 "team_id": "45e3e396-ee08-4a61-a88e-16b3ce7e0849",
4537 "member": {
4538 "role": "user",
4539 "user_id": "user123"
4540 },
4541 "max_budget_in_team": 100.0
4542 }
4543 ```
4544 """
4546 team_id: str = Field(description="The ID of the team to add the member to")
4547 max_budget_in_team: float | None = Field(
4548 default=None,
4549 description="Maximum budget allocated to this user within the team. If not set, user has unlimited budget within team limits",
4550 )
4551 budget_duration: str | None = Field(
4552 default=None,
4553 description="Duration after which this team member's budget resets (e.g. '1h', '24h', '7d', '30d'). If not set, the budget never resets.",
4554 )
4555 allowed_models: list[str] | None = Field(
4556 default=None,
4557 description="List of models this team member can access. If not set, inherits the team's default_team_member_models or all team models.",
4558 )
4561class TeamMemberDeleteRequest(MemberDeleteRequest):
4562 team_id: str
4565class TeamMemberUpdateRequest(TeamMemberDeleteRequest):
4566 max_budget_in_team: float | None = None
4567 role: Literal["admin", "user"] | None = None
4568 tpm_limit: int | None = Field(default=None, description="Tokens per minute limit for this team member")
4569 rpm_limit: int | None = Field(default=None, description="Requests per minute limit for this team member")
4570 budget_duration: str | None = Field(
4571 default=None,
4572 description="Duration after which this team member's budget resets (e.g. '1h', '24h', '7d', '30d'). If not set, the budget never resets.",
4573 )
4574 allowed_models: list[str] | None = Field(
4575 default=None,
4576 description="List of models this team member can access. Pass an empty list to remove per-member model restrictions.",
4577 )
4578 temp_budget_increase: float | None = Field(
4579 default=None,
4580 ge=0,
4581 allow_inf_nan=False,
4582 description="Temporary additive budget increase for this team member, active until temp_budget_expiry",
4583 )
4584 temp_budget_expiry: datetime | None = Field(
4585 default=None,
4586 description="UTC expiry for temp_budget_increase",
4587 )
4589 @model_validator(mode="after")
4590 def validate_temp_budget(self) -> "TeamMemberUpdateRequest":
4591 if self.temp_budget_increase is not None or self.temp_budget_expiry is not None:
4592 if self.temp_budget_increase is None or self.temp_budget_expiry is None:
4593 raise ValueError("temp_budget_increase and temp_budget_expiry must be set together")
4594 return self
4597class TeamMemberUpdateResponse(MemberUpdateResponse):
4598 team_id: str
4599 max_budget_in_team: float | None = None
4600 tpm_limit: int | None = None
4601 rpm_limit: int | None = None
4602 budget_duration: str | None = None
4603 allowed_models: list[str] | None = None
4604 temp_budget_increase: float | None = None
4605 temp_budget_expiry: datetime | None = None
4608class TeamModelAddRequest(BaseModel):
4609 """Request to add models to a team"""
4611 team_id: str
4612 models: list[str]
4615class TeamModelDeleteRequest(BaseModel):
4616 """Request to delete models from a team"""
4618 team_id: str
4619 models: list[str]
4622# Organization Member Requests
4623class OrganizationMemberAddRequest(OrgMemberAddRequest):
4624 organization_id: str
4625 max_budget_in_organization: float | None = None # Users max budget within the organization
4628class OrganizationMemberDeleteRequest(MemberDeleteRequest):
4629 organization_id: str
4632ROLES_WITHIN_ORG: Final = [
4633 LitellmUserRoles.ORG_ADMIN,
4634 LitellmUserRoles.INTERNAL_USER,
4635 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
4636]
4639class OrganizationMemberUpdateRequest(OrganizationMemberDeleteRequest):
4640 max_budget_in_organization: float | None = None
4641 role: LitellmUserRoles | None = None
4643 @field_validator("role")
4644 def validate_role(cls, value: LitellmUserRoles | None) -> LitellmUserRoles | None:
4645 if value is not None and value not in ROLES_WITHIN_ORG:
4646 raise ValueError(f"Invalid role. Must be one of: {[role.value for role in ROLES_WITHIN_ORG]}")
4647 return value
4650class OrganizationMemberUpdateResponse(MemberUpdateResponse):
4651 organization_id: str
4652 max_budget_in_organization: float
4655##########################################
4658class TeamAccessGroupModelGrant(LiteLLMPydanticObjectBase):
4659 access_group_id: str
4660 access_group_name: str
4661 models: tuple[str, ...]
4662 mcp_server_ids: tuple[str, ...] = ()
4663 agent_ids: tuple[str, ...] = ()
4666class TeamInfoMember(Member):
4667 user_alias: str | None = None
4670class TeamEditUnrestricted(BaseModel):
4671 kind: Literal["unrestricted"] = "unrestricted"
4674class TeamEditAsTeamAdmin(BaseModel):
4675 kind: Literal["team_admin"] = "team_admin"
4676 editable_fields: tuple[str, ...]
4679class TeamEditAsTeamAdminDisabled(BaseModel):
4680 kind: Literal["team_admin_disabled"] = "team_admin_disabled"
4683class TeamEditNone(BaseModel):
4684 kind: Literal["none"] = "none"
4687TeamEditAccess = Annotated[
4688 TeamEditUnrestricted | TeamEditAsTeamAdmin | TeamEditAsTeamAdminDisabled | TeamEditNone,
4689 Field(discriminator="kind"),
4690]
4693class TeamInfoResponseObjectTeamTable(LiteLLM_TeamTable):
4694 members_with_roles: tuple[TeamInfoMember, ...] = ()
4695 team_member_budget_table: LiteLLM_BudgetTableFull | None = None
4696 # Resources inherited from access groups (separate from direct assignments)
4697 access_group_models: list[str] | None = None
4698 access_group_mcp_server_ids: list[str] | None = None
4699 access_group_agent_ids: list[str] | None = None
4700 access_group_details: tuple[TeamAccessGroupModelGrant, ...] | None = None
4701 # Parent org's model ceiling, reported only to callers who can manage the team.
4702 # None = no org or not a manager; [] or ["all-proxy-models"] = no ceiling.
4703 organization_models: list[str] | None = None
4704 model_max_budget_usage: Mapping[str, Mapping[str, object]] | None = None
4705 caller_edit_access: TeamEditAccess = Field(default_factory=TeamEditNone)
4708TeamMemberBudgetSource: TypeAlias = Literal["team_default", "custom", "none"]
4711class TeamInfoMembership(LiteLLM_TeamMembership):
4712 budget_source: TeamMemberBudgetSource
4715class TeamInfoResponseObject(TypedDict):
4716 team_id: str
4717 team_info: TeamInfoResponseObjectTeamTable
4718 keys: list
4719 team_memberships: ReadOnly[tuple[TeamInfoMembership, ...]]
4722class TeamMemberResetBudgetResponse(BaseModel):
4723 team_id: str
4724 user_id: str
4725 budget_id: str | None
4726 previous_budget_id: str | None
4727 budget_source: TeamMemberBudgetSource
4730class TeamListResponseObject(LiteLLM_TeamTable):
4731 team_memberships: list[LiteLLM_TeamMembership]
4732 keys: list # list of keys that belong to the team
4735class KeyListResponseObject(TypedDict, total=False):
4736 keys: list[str | UserAPIKeyAuth | LiteLLM_DeletedVerificationToken]
4737 total_count: int | None
4738 current_page: int | None
4739 total_pages: int | None
4742class CurrentItemRateLimit(TypedDict):
4743 current_requests: int
4744 current_tpm: int
4745 current_rpm: int
4748class LoggingCallbackStatus(TypedDict, total=False):
4749 callbacks: list[str]
4750 status: Literal["healthy", "unhealthy"]
4751 details: str | None
4754class KeyHealthResponse(TypedDict, total=False):
4755 key: Literal["healthy", "unhealthy"]
4756 logging_callbacks: LoggingCallbackStatus | None
4759class CreateJWTKeyMappingRequest(LiteLLMPydanticObjectBase):
4760 jwt_claim_name: str
4761 jwt_claim_value: str
4762 key: str | None = None
4763 token: str | None = None
4764 jwt_issuer: str | None = None
4765 description: str | None = None
4768class UpdateJWTKeyMappingRequest(LiteLLMPydanticObjectBase):
4769 id: str
4770 key: str | None = None
4771 token: str | None = None
4772 jwt_issuer: str | None = None
4773 description: str | None = None
4774 is_active: bool | None = None
4777class DeleteJWTKeyMappingRequest(LiteLLMPydanticObjectBase):
4778 id: str
4781class JWTKeyMappingResponse(LiteLLMPydanticObjectBase):
4782 id: str
4783 jwt_issuer: str | None = None
4784 jwt_claim_name: str
4785 jwt_claim_value: str
4786 description: str | None = None
4787 is_active: bool
4788 created_at: datetime
4789 updated_at: datetime
4790 created_by: str | None = None
4791 updated_by: str | None = None
4794class SpecialHeaders(enum.Enum):
4795 """Used by user_api_key_auth.py to get litellm key"""
4797 openai_authorization = "Authorization"
4798 azure_authorization = "API-Key"
4799 anthropic_authorization = "x-api-key"
4800 google_ai_studio_authorization = "x-goog-api-key"
4801 azure_apim_authorization = "Ocp-Apim-Subscription-Key"
4802 custom_litellm_api_key = "x-litellm-api-key"
4803 mcp_auth = "x-mcp-auth"
4804 mcp_servers = "x-mcp-servers"
4805 mcp_access_groups = "x-mcp-access-groups"
4807 @classmethod
4808 def litellm_credential_header_names(cls) -> "frozenset[str]":
4809 """Lowercased header names user_api_key_auth accepts as a litellm key.
4811 Every header here authenticates the caller, so any code that forwards a
4812 request onward (e.g. the plugin reverse proxy) must strip all of them to
4813 avoid leaking the caller's litellm credential downstream. The static
4814 custom-key header (general_settings.litellm_key_header_name) is runtime
4815 config and must be added on top of this set by the caller.
4816 """
4817 return frozenset(
4818 header.value.lower()
4819 for header in (
4820 cls.openai_authorization,
4821 cls.azure_authorization,
4822 cls.anthropic_authorization,
4823 cls.google_ai_studio_authorization,
4824 cls.azure_apim_authorization,
4825 cls.custom_litellm_api_key,
4826 )
4827 )
4830class LitellmDataForBackendLLMCall(TypedDict, total=False):
4831 headers: dict
4832 organization: str
4833 timeout: float | None
4834 stream_timeout: float | None
4835 user: str | None
4836 num_retries: int | None
4837 # True when the effective timeout came from a caller-controlled source (the
4838 # `x-litellm-timeout`/`x-litellm-stream-timeout` headers, or a `timeout`/`request_timeout`/
4839 # `stream_timeout` field in the request body) rather than deployment config, so a
4840 # deliberately tiny value isn't treated as a deployment health signal (see
4841 # cooldown_handlers._trigger_cooldown_for_failed_deployment).
4842 client_side_timeout: bool
4843 keepalive_seconds: float | None
4846class LitellmMetadataFromRequestHeaders(TypedDict, total=False):
4847 """
4848 Headers a user can pass that will get added to litellm metadata for the request
4849 """
4851 spend_logs_metadata: dict | None
4852 agent_id: str | None
4853 trace_id: str | None
4854 session_id: str | None
4857class JWTKeyItem(TypedDict, total=False):
4858 kid: str
4861JWKKeyValue = list[JWTKeyItem] | JWTKeyItem
4864class JWKUrlResponse(TypedDict, total=False):
4865 keys: JWKKeyValue
4868class UserManagementEndpointParamDocStringEnums(str, enum.Enum):
4869 user_id_doc_str = "Optional[str] - Specify a user id. If not set, a unique id will be generated."
4870 user_alias_doc_str = "Optional[str] - A descriptive name for you to know who this user id refers to."
4871 teams_doc_str = "Optional[list] - specify a list of team id's a user belongs to."
4872 user_email_doc_str = "Optional[str] - Specify a user email."
4873 send_invite_email_doc_str = "Optional[bool] - Specify if an invite email should be sent."
4874 user_role_doc_str = """Optional[str] - Specify a user role - "proxy_admin", "proxy_admin_viewer", "internal_user", "internal_user_viewer", "team", "customer". Info about each role here: `https://github.com/BerriAI/litellm/litellm/proxy/_types.py#L20`"""
4875 max_budget_doc_str = """Optional[float] - Specify max budget for a given user."""
4876 budget_duration_doc_str = """Optional[str] - Budget is reset at the end of specified duration. If not set, budget is never reset. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d"), months ("1mo")."""
4877 models_doc_str = (
4878 """Optional[list] - Model_name's a user is allowed to call. (if empty, key is allowed to call all models)"""
4879 )
4880 tpm_limit_doc_str = """Optional[int] - Specify tpm limit for a given user (Tokens per minute)"""
4881 rpm_limit_doc_str = """Optional[int] - Specify rpm limit for a given user (Requests per minute)"""
4882 auto_create_key_doc_str = """bool - Default=True. Flag used for returning a key as part of the /user/new response"""
4883 aliases_doc_str = """Optional[dict] - Model aliases for the user - [Docs](https://litellm.vercel.app/docs/proxy/virtual_keys#model-aliases)"""
4884 config_doc_str = """Optional[dict] - [DEPRECATED PARAM] User-specific config."""
4885 allowed_cache_controls_doc_str = """Optional[list] - List of allowed cache control values. Example - ["no-cache", "no-store"]. See all values - https://docs.litellm.ai/docs/proxy/caching#turn-on--off-caching-per-request-"""
4886 blocked_doc_str = """Optional[bool] - [Not Implemented Yet] Whether the user is blocked."""
4887 guardrails_doc_str = """Optional[List[str]] - [Not Implemented Yet] List of active guardrails for the user"""
4888 permissions_doc_str = (
4889 """Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking."""
4890 )
4891 metadata_doc_str = """Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }"""
4892 max_parallel_requests_doc_str = """Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x."""
4893 model_max_budget_doc_str = """Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys)"""
4894 model_rpm_limit_doc_str = """Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)"""
4895 model_tpm_limit_doc_str = """Optional[float] - Model-specific tpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)"""
4896 spend_doc_str = (
4897 """Optional[float] - Amount spent by user. Default is 0. Will be updated by proxy whenever user is used."""
4898 )
4899 team_id_doc_str = """Optional[str] - [DEPRECATED PARAM] The team id of the user. Default is None."""
4900 duration_doc_str = """Optional[str] - Duration for the key auto-created on `/user/new`. Default is None."""
4903PassThroughEndpointLoggingResultValues = (
4904 ModelResponse
4905 | TextCompletionResponse
4906 | ImageResponse
4907 | EmbeddingResponse
4908 | VideoObject
4909 | StandardPassThroughResponseObject
4910 | ResponsesAPIResponse
4911 | TranscriptionResponse
4912)
4915class PassThroughEndpointLoggingTypedDict(TypedDict):
4916 result: PassThroughEndpointLoggingResultValues | None
4917 kwargs: dict
4920LiteLLM_ManagementEndpoint_MetadataFields: Final = [
4921 "model_rpm_limit",
4922 "model_tpm_limit",
4923 "model_itpm_limit",
4924 "model_otpm_limit",
4925 "default_estimated_output_tokens",
4926 "default_estimated_output_tokens_per_model",
4927 "mcp_rpm_limit",
4928 "tag_rpm_limit",
4929 "rpm_limit_type",
4930 "tpm_limit_type",
4931 "enforced_params",
4932 "temp_budget_increase",
4933 "temp_budget_expiry",
4934 "allowed_vector_store_indexes",
4935 "enforced_batch_output_expires_after",
4936 "enforced_file_expires_after",
4937 "throttle_on_budget_exceeded",
4938 "enable_prompt_caching",
4939 "end_user_budget_id",
4940]
4942LiteLLM_ManagementEndpoint_MetadataFields_Premium: Final = [
4943 "disable_global_guardrails",
4944 "guardrails",
4945 "policies",
4946 "tags",
4947 "team_member_key_duration",
4948 "prompts",
4949 "logging",
4950 "secret_manager_settings",
4951 "allowed_passthrough_routes",
4952]
4954# Metadata keys that are immutable once set: preserved when an update omits them,
4955# and rejected (400) when an update tries to change them.
4956LiteLLM_Reserved_Metadata_Fields: Final = [
4957 "service_account_id",
4958]
4961class ProviderBudgetResponseObject(LiteLLMPydanticObjectBase):
4962 """
4963 Configuration for a single provider's budget settings
4964 """
4966 budget_limit: float | None # Budget limit in USD for the time period
4967 time_period: str | None # Time period for budget (e.g., '1d', '30d', '1mo')
4968 spend: float | None = 0.0 # Current spend for this provider
4969 budget_reset_at: str | None = None # When the current budget period resets
4972class ProviderBudgetResponse(LiteLLMPydanticObjectBase):
4973 """
4974 Complete provider budget configuration and status.
4975 Maps provider names to their budget configs.
4976 """
4978 providers: dict[
4979 str, ProviderBudgetResponseObject
4980 ] = {} # Dictionary mapping provider names to their budget configurations
4983class ProxyStateVariables(TypedDict):
4984 """
4985 TypedDict for Proxy state variables.
4986 """
4988 spend_logs_row_count: int
4991UI_TEAM_ID = "litellm-dashboard"
4994class JWTAuthBuilderResult(TypedDict):
4995 is_proxy_admin: bool
4996 team_object: LiteLLM_TeamTable | None
4997 user_object: LiteLLM_UserTable | None
4998 end_user_object: LiteLLM_EndUserTable | None
4999 org_object: LiteLLM_OrganizationTable | None
5000 token: str
5001 team_id: str | None
5002 user_id: str | None
5003 user_email: str | None
5004 end_user_id: str | None
5005 org_id: str | None
5006 team_membership: LiteLLM_TeamMembership | None
5007 jwt_claims: dict # Decoded JWT token claims (avoids re-decoding)
5008 agent_id: ReadOnly[str | None]
5011class ClientSideFallbackModel(TypedDict, total=False):
5012 """
5013 Dictionary passed when client configuring input
5014 """
5016 model: Required[str]
5017 messages: list[AllMessageValues]
5020ALL_FALLBACK_MODEL_VALUES = str | ClientSideFallbackModel
5023RBAC_ROLES = Literal[
5024 LitellmUserRoles.PROXY_ADMIN,
5025 LitellmUserRoles.TEAM,
5026 LitellmUserRoles.INTERNAL_USER,
5027]
5030class OIDCPermissions(LiteLLMPydanticObjectBase):
5031 models: list[str] | None = None
5032 routes: list[str] | None = None
5035class RoleBasedPermissions(OIDCPermissions):
5036 role: RBAC_ROLES
5038 model_config = {
5039 "extra": "forbid",
5040 }
5043class RoleMapping(BaseModel):
5044 role: str
5045 internal_role: RBAC_ROLES
5048class JWTLiteLLMRoleMap(BaseModel):
5049 jwt_role: str
5050 litellm_role: LitellmUserRoles
5053class ScopeMapping(OIDCPermissions):
5054 scope: str
5056 model_config = {
5057 "extra": "forbid",
5058 }
5061class JWTRoutingOverride(BaseModel):
5062 """
5063 Override default auth routing for JWT-shaped bearer tokens.
5065 A rule matches when all provided selectors match token claims.
5066 If matched, request is routed to the configured auth path.
5068 Wildcard selectors use shell-style patterns (* and ?) and are matched with
5069 case-sensitive semantics; use the same casing your IdP emits in JWT claims.
5070 Space-delimited tokenization applies only to the ``scope`` claim (OAuth/OIDC
5071 scope strings), not to ``iss``, ``aud``, or ``client_id``.
5072 """
5074 iss: str | list[str]
5075 client_id: str | list[str] | None = None
5076 scope: str | list[str] | None = None
5077 aud: str | list[str] | None = None
5078 path: Literal["oauth2"] = "oauth2"
5080 model_config = {
5081 "extra": "forbid",
5082 }
5085class UnregisteredJWTClientBehavior(str, enum.Enum):
5086 """
5087 Controls what happens when `virtual_key_claim_field` is configured but the
5088 JWT claim value has no registered mapping in `litellm_jwtkeymapping`.
5090 - fallback_team_mapping: Fall through to standard team-based JWT auth (default,
5091 backward-compatible).
5092 - reject: Immediately return HTTP 403. Use this when every valid JWT client
5093 must have a pre-registered virtual key — unknown callers are denied.
5094 - auto_register: Automatically create a new virtual key and mapping on first
5095 encounter. The new key has no budget/model restrictions; admins can tighten
5096 it later via /jwt_client/update.
5097 """
5099 FALLBACK_TEAM_MAPPING = "fallback_team_mapping"
5100 REJECT = "reject"
5101 AUTO_REGISTER = "auto_register"
5104class JWTIssuerConfig(BaseModel):
5105 """
5106 Issuer-bound JWT validation configuration.
5108 When a token's unverified `iss` claim matches an entry in
5109 ``LiteLLM_JWTAuth.issuers``, LiteLLM validates it only against that
5110 issuer's JWKS and audience. Tokens whose `iss` does not match any
5111 configured issuer fall back to the global JWT_AUDIENCE/JWT_ISSUER
5112 validation path; `issuers` is additive routing, not an allow-list.
5113 """
5115 issuer: str = Field(description="Exact expected JWT issuer (`iss`) value.")
5116 jwks_url: str | None = Field(
5117 default=None,
5118 description="Issuer JWKS URL. If omitted, LiteLLM uses the issuer's OIDC discovery document.",
5119 )
5120 audience: str | list[str] | None = Field(
5121 default=None,
5122 description="Expected token audience for this issuer.",
5123 )
5124 disable_audience_validation: bool = Field(
5125 default=False,
5126 description="Explicitly disable audience validation for this issuer. Use only when the issuer cannot provide an audience suitable for LiteLLM.",
5127 )
5128 user_id_jwt_field: str | None = Field(
5129 default=None,
5130 description="Issuer-specific claim path to normalize into LiteLLM's user id.",
5131 )
5132 user_email_jwt_field: str | None = Field(
5133 default=None,
5134 description="Issuer-specific claim path to normalize into LiteLLM's user email.",
5135 )
5136 team_id_jwt_field: str | None = Field(
5137 default=None,
5138 description="Issuer-specific claim path to normalize into LiteLLM's team id.",
5139 )
5140 team_ids_jwt_field: str | None = Field(
5141 default=None,
5142 description="Issuer-specific claim path to normalize into LiteLLM's team ids.",
5143 )
5144 org_id_jwt_field: str | None = Field(
5145 default=None,
5146 description="Issuer-specific claim path to normalize into LiteLLM's organization id.",
5147 )
5148 end_user_id_jwt_field: str | None = Field(
5149 default=None,
5150 description="Issuer-specific claim path to normalize into LiteLLM's end-user id.",
5151 )
5152 virtual_key_claim_field: str | None = Field(
5153 default=None,
5154 description="Issuer-specific claim path used for the virtual key mapping lookup. Falls back to the global field.",
5155 )
5156 unregistered_jwt_client_behavior: UnregisteredJWTClientBehavior | None = Field(
5157 default=None,
5158 description="Issuer-specific policy when the virtual key claim has no mapping. Falls back to the global policy.",
5159 )
5161 model_config = {
5162 "extra": "forbid",
5163 }
5165 @model_validator(mode="after")
5166 def validate_audience_configured(self) -> "JWTIssuerConfig":
5167 if self.audience is None and not self.disable_audience_validation:
5168 raise ValueError(
5169 f"JWT issuer {self.issuer} must configure audience or set disable_audience_validation=True"
5170 )
5171 if self.audience is not None and self.disable_audience_validation:
5172 raise ValueError(
5173 f"JWT issuer {self.issuer} cannot set audience and disable_audience_validation=True together"
5174 )
5175 return self
5178DEFAULT_JWKS_STALE_TTL: Final = 3600
5181class LiteLLM_JWTAuth(LiteLLMPydanticObjectBase):
5182 """
5183 A class to define the roles and permissions for a LiteLLM Proxy w/ JWT Auth.
5185 Attributes:
5186 - admin_jwt_scope: The JWT scope required for proxy admin roles.
5187 - admin_allowed_routes: list of allowed routes for proxy admin roles.
5188 - team_jwt_scope: The JWT scope required for proxy team roles.
5189 - team_id_jwt_field: The field in the JWT token that stores the team ID. Default - `client_id`.
5190 - team_allowed_routes: list of allowed routes for proxy team roles.
5191 - user_id_jwt_field: The field in the JWT token that stores the user id (maps to `LiteLLMUserTable`). Use this for internal employees.
5192 - user_email_jwt_field: The field in the JWT token that stores the user email (maps to `LiteLLMUserTable`). Use this for internal employees.
5193 - user_allowed_email_subdomain: If specified, only emails from specified subdomain will be allowed to access proxy.
5194 - end_user_id_jwt_field: The field in the JWT token that stores the end-user ID (maps to `LiteLLMEndUserTable`). Turn this off by setting to `None`. Enables end-user cost tracking. Use this for external customers.
5195 - public_key_ttl: Default - 600s. TTL for caching public JWT keys.
5196 - public_key_stale_ttl: Default - 3600s. Extra time past `public_key_ttl` that the last-known-good JWKS response
5197 stays usable while the identity provider is unreachable. Set to 0 to fail closed instead.
5198 - public_allowed_routes: list of allowed routes for authenticated but unknown litellm role jwt tokens.
5199 - enforce_rbac: If true, enforce RBAC for all routes.
5200 - custom_validate: A custom function to validates the JWT token.
5201 - oidc_userinfo_endpoint: OIDC UserInfo endpoint URL. When set along with oidc_userinfo_enabled, LiteLLM will call this endpoint with the access token to retrieve user identity information.
5202 - oidc_userinfo_enabled: Enable fetching user info from OIDC UserInfo endpoint instead of just decoding JWT token. Default: False.
5203 - oidc_userinfo_cache_ttl: TTL (in seconds) for caching UserInfo responses. Default: 300s (5 minutes).
5205 See `auth_checks.py` for the specific routes
5206 """
5208 admin_jwt_scope: str = "litellm_proxy_admin"
5209 admin_allowed_routes: list[str] = [
5210 "management_routes",
5211 "spend_tracking_routes",
5212 "global_spend_tracking_routes",
5213 "info_routes",
5214 ]
5215 team_id_jwt_field: str | None = None
5216 team_id_upsert: bool = False
5217 team_ids_jwt_field: str | None = None
5218 upsert_sso_user_to_team: bool = False
5219 team_allowed_routes: list[str] = [
5220 "openai_routes",
5221 "info_routes",
5222 "mcp_routes",
5223 "/v1/messages",
5224 "/v1/messages/count_tokens",
5225 ]
5226 team_id_default: str | None = Field(
5227 default=None,
5228 description="If no team_id given, default permissions/spend-tracking to this team.s",
5229 )
5230 team_alias_jwt_field: str | None = Field(
5231 default=None,
5232 description="The field in the JWT token that stores the team name/alias. Will be resolved to team_id via database lookup.",
5233 )
5235 org_id_jwt_field: str | None = None
5236 org_alias_jwt_field: str | None = Field(
5237 default=None,
5238 description="The field in the JWT token that stores the organization name/alias. Will be resolved to org_id via database lookup.",
5239 )
5240 user_id_jwt_field: str | None = None
5241 user_email_jwt_field: str | None = None
5242 user_allowed_email_domain: str | None = None
5243 user_roles_jwt_field: str | None = None
5244 user_allowed_roles: list[str] | None = None
5245 user_id_upsert: bool = Field(default=False, description="If user doesn't exist, upsert them into the db.")
5246 end_user_id_jwt_field: str | None = None
5247 agent_id_jwt_field: str | None = Field(
5248 default=None,
5249 description=(
5250 "The field in the JWT token that identifies the calling agent (e.g. 'azp' for a Microsoft Entra ID "
5251 "app token). Supports dot notation. The value is matched against a registered agent's agent_id, "
5252 "then agent_name, and the request is rejected when it matches neither."
5253 ),
5254 )
5255 mcp_client_id_jwt_field: str | None = Field(
5256 default=None,
5257 description=(
5258 "The field in the JWT token that identifies the MCP client application (harness) making the request, "
5259 "e.g. 'azp' or 'client_id'. Supports dot notation. Only consulted while general_settings.mcp_allowed_clients "
5260 "is set: the claim value must be listed there or the MCP request is rejected with 403. Distinct from "
5261 "agent_id_jwt_field, which identifies an AI agent rather than the client software."
5262 ),
5263 )
5264 public_key_ttl: float = 600
5265 public_key_stale_ttl: float = Field(
5266 default=DEFAULT_JWKS_STALE_TTL,
5267 ge=0,
5268 description=(
5269 "Seconds beyond `public_key_ttl` that the last-known-good JWKS response stays usable while the identity "
5270 "provider is unreachable. Bounds how long a signing key the provider has since removed can still be "
5271 "trusted. Set to 0 to fail closed and reject requests as soon as the cached keys expire."
5272 ),
5273 )
5274 public_allowed_routes: list[str] = ["public_routes"]
5275 enforce_rbac: bool = False
5276 roles_jwt_field: str | None = None # v2 on role mappings
5277 role_mappings: list[RoleMapping] | None = None
5278 object_id_jwt_field: str | None = None # can be either user / team, inferred from the role mapping
5279 scope_mappings: list[ScopeMapping] | None = None
5280 enforce_scope_based_access: bool = False
5281 enforce_team_based_model_access: bool = False
5282 custom_validate: Callable[..., Literal[True]] | None = None
5283 #########################################################
5284 # Fields for syncing user team membership and roles with IDP provider
5285 jwt_litellm_role_map: list[JWTLiteLLMRoleMap] | None = None
5286 sync_user_role_and_teams: bool = False
5287 #########################################################
5288 #########################################################
5289 # OIDC UserInfo Endpoint Configuration
5290 oidc_userinfo_endpoint: str | None = Field(
5291 default=None,
5292 description="OIDC UserInfo endpoint URL. If set, LiteLLM will call this endpoint with the access token to retrieve user identity information.",
5293 )
5294 oidc_userinfo_enabled: bool = Field(
5295 default=False,
5296 description="Enable fetching user info from OIDC UserInfo endpoint instead of just decoding JWT token.",
5297 )
5298 oidc_userinfo_cache_ttl: float = Field(
5299 default=300,
5300 description="TTL (in seconds) for caching UserInfo responses. Default: 300s (5 minutes).",
5301 )
5302 # JWT-to-Virtual-Key Mapping
5303 virtual_key_claim_field: str | None = Field(
5304 default=None,
5305 description="JWT claim field for virtual key mapping lookup (e.g. 'sub', 'email'). Supports dot notation.",
5306 )
5307 virtual_key_mapping_cache_ttl: float = Field(
5308 default=300,
5309 description="TTL (seconds) for caching JWT-to-virtual-key mapping lookups.",
5310 )
5311 unregistered_jwt_client_behavior: UnregisteredJWTClientBehavior = Field(
5312 default=UnregisteredJWTClientBehavior.FALLBACK_TEAM_MAPPING,
5313 description=(
5314 "What to do when virtual_key_claim_field is set but the JWT claim value "
5315 "has no registered mapping. 'fallback_team_mapping' (default): fall through "
5316 "to team-based JWT auth. 'reject': return HTTP 403. "
5317 "'auto_register': auto-create a virtual key and mapping on first encounter."
5318 ),
5319 )
5320 routing_overrides: list[JWTRoutingOverride] | None = Field(
5321 default=None,
5322 description="Optional claim-based routing overrides for JWT-shaped tokens. Matching rules route requests to oauth2 before default JWT flow.",
5323 )
5324 team_claim_fallback: bool = Field(
5325 default=False,
5326 description=(
5327 "If True, when a configured team_id_jwt_field / team_ids_jwt_field "
5328 "claim is present but does not resolve to any known team, defer to "
5329 "the single-team DB fallback (caller's only team membership) "
5330 "instead of raising. Default False preserves strict claim-based "
5331 "authorization."
5332 ),
5333 )
5334 fallback_to_db_teams: bool = Field(
5335 default=False,
5336 description=(
5337 "When True, users whose JWT contains no team claims are authenticated "
5338 "using their database team memberships instead of receiving HTTP 403, "
5339 "with usage attributed to the user's first resolvable DB team. Whether or "
5340 "not the JWT carries team claims, the x-litellm-team-id request header may "
5341 "select any team the user is a member of in the database (validated against "
5342 "DB membership); without the header the JWT team stays the default. Requires "
5343 "user_id_upsert=True so that user records exist before the fallback runs."
5344 ),
5345 )
5346 issuers: list[JWTIssuerConfig] | None = Field(
5347 default=None,
5348 description="Optional issuer-bound JWT validation rules. When a token's `iss` matches a configured issuer, validation uses that issuer's JWKS, audience, and claim mappings. Tokens with an unlisted `iss` fall back to the global JWT_AUDIENCE/JWT_ISSUER validation path — this is additive routing, not an allow-list.",
5349 )
5350 #########################################################
5352 def __init__(self, **kwargs: Any) -> None:
5353 # ``config_file_path`` is a non-field kwarg threaded by the
5354 # startup-load path so an operator-configured
5355 # ``custom_validate: s3://bucket/module.fn`` resolves through
5356 # the documented config-file flow. Pop before the invalid-keys
5357 # check; the runtime gate in ``get_instance_fn`` refuses
5358 # ``s3://`` / ``gcs://`` when this is None.
5359 config_file_path: Final = kwargs.pop("config_file_path", None)
5361 # Backward-compat: jwt_client_id_field was renamed to virtual_key_claim_field
5362 if "jwt_client_id_field" in kwargs: 5362 ↛ 5363line 5362 didn't jump to line 5363 because the condition on line 5362 was never true
5363 if "virtual_key_claim_field" not in kwargs:
5364 kwargs["virtual_key_claim_field"] = kwargs.pop("jwt_client_id_field")
5365 else:
5366 kwargs.pop("jwt_client_id_field")
5368 # get the attribute names for this Pydantic model
5369 allowed_keys: Final = LiteLLM_JWTAuth.__annotations__.keys()
5371 invalid_keys: Final = set(kwargs.keys()) - allowed_keys
5372 user_roles_jwt_field: Final = kwargs.get("user_roles_jwt_field")
5373 user_allowed_roles: Final = kwargs.get("user_allowed_roles")
5374 object_id_jwt_field: Final = kwargs.get("object_id_jwt_field")
5375 role_mappings: Final = kwargs.get("role_mappings")
5376 scope_mappings: Final = kwargs.get("scope_mappings")
5377 enforce_scope_based_access: Final = kwargs.get("enforce_scope_based_access")
5378 custom_validate: Final = kwargs.get("custom_validate")
5380 if custom_validate is not None: 5380 ↛ 5381line 5380 didn't jump to line 5381 because the condition on line 5380 was never true
5381 fn: Final = get_instance_fn(custom_validate, config_file_path=config_file_path)
5382 validate_custom_validate_return_type(fn)
5383 kwargs["custom_validate"] = fn
5385 if invalid_keys: 5385 ↛ 5386line 5385 didn't jump to line 5386 because the condition on line 5385 was never true
5386 raise ValueError(
5387 f"Invalid arguments provided: {', '.join(invalid_keys)}. Allowed arguments are: {', '.join(allowed_keys)}."
5388 )
5389 if (user_roles_jwt_field is not None and user_allowed_roles is None) or ( 5389 ↛ 5392line 5389 didn't jump to line 5392 because the condition on line 5389 was never true
5390 user_roles_jwt_field is None and user_allowed_roles is not None
5391 ):
5392 raise ValueError("user_allowed_roles must be provided if user_roles_jwt_field is set.")
5394 if object_id_jwt_field is not None and role_mappings is None: 5394 ↛ 5395line 5394 didn't jump to line 5395 because the condition on line 5394 was never true
5395 raise ValueError(
5396 "if object_id_jwt_field is set, role_mappings must also be set. Needed to infer if the caller is a user or team."
5397 )
5399 if scope_mappings is not None and not enforce_scope_based_access: 5399 ↛ 5400line 5399 didn't jump to line 5400 because the condition on line 5399 was never true
5400 raise ValueError("scope_mappings must be set if enforce_scope_based_access is true.")
5402 super().__init__(**kwargs)
5404 def get_issuer_config(self, issuer: str | None) -> JWTIssuerConfig | None:
5405 if issuer is None or self.issuers is None:
5406 return None
5407 return next((config for config in self.issuers if config.issuer == issuer), None)
5409 def is_virtual_key_mapping_configured(self) -> bool:
5410 if self.virtual_key_claim_field is not None: 5410 ↛ 5411line 5410 didn't jump to line 5411 because the condition on line 5410 was never true
5411 return True
5412 return any(config.virtual_key_claim_field is not None for config in self.issuers or ())
5414 def get_virtual_key_claim_field(self, issuer: str | None) -> str | None:
5415 issuer_config: Final = self.get_issuer_config(issuer)
5416 if issuer_config is not None and issuer_config.virtual_key_claim_field is not None:
5417 return issuer_config.virtual_key_claim_field
5418 return self.virtual_key_claim_field
5420 def get_unregistered_jwt_client_behavior(self, issuer: str | None) -> UnregisteredJWTClientBehavior:
5421 issuer_config: Final = self.get_issuer_config(issuer)
5422 if issuer_config is not None and issuer_config.unregistered_jwt_client_behavior is not None:
5423 return issuer_config.unregistered_jwt_client_behavior
5424 return self.unregistered_jwt_client_behavior
5427class PrismaCompatibleUpdateDBModel(TypedDict, total=False):
5428 model_name: str
5429 litellm_params: str
5430 model_info: str
5431 blocked: bool
5432 updated_at: str
5433 updated_by: str
5436class SpecialManagementEndpointEnums(enum.Enum):
5437 DEFAULT_ORGANIZATION = "default_organization"
5440class TransformRequestBody(BaseModel):
5441 call_type: CallTypes
5442 request_body: dict
5445class DefaultInternalUserParams(LiteLLMPydanticObjectBase):
5446 """
5447 Default parameters to apply when a new user signs in via SSO or is created on the /user/new API endpoint
5448 """
5450 user_role: (
5451 Literal[
5452 LitellmUserRoles.PROXY_ADMIN,
5453 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
5454 LitellmUserRoles.INTERNAL_USER,
5455 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
5456 ]
5457 | None
5458 ) = Field(
5459 default=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
5460 description="Default role assigned to new users created",
5461 )
5462 max_budget: float | None = Field(
5463 default=None,
5464 description="Default maximum budget (in USD) for new users created",
5465 )
5466 budget_duration: str | None = Field(
5467 default=None,
5468 description="Default budget duration for new users (e.g. 'daily', 'weekly', 'monthly')",
5469 )
5470 models: list[str] | None = Field(default=None, description="Default list of models that new users can access")
5472 teams: list[str] | list[NewUserRequestTeam] | None = Field(
5473 default=None,
5474 description="Default teams for new users created",
5475 )
5478class BaseDailySpendTransaction(TypedDict):
5479 date: str
5480 api_key: str
5481 model: str | None
5482 model_group: str | None
5483 mcp_namespaced_tool_name: str | None
5484 custom_llm_provider: str | None
5485 endpoint: str | None
5487 # token count metrics
5488 prompt_tokens: int
5489 completion_tokens: int
5490 cache_read_input_tokens: int
5491 cache_creation_input_tokens: int
5492 compression_saved_tokens: int
5494 # cost-savings metrics (dollars, priced per request before aggregation)
5495 compression_savings_spend: float
5496 prompt_caching_savings_spend: float
5497 gateway_injected_caching_savings_spend: float # writable-ok: the rollup queue accumulates into this key in place, as it does for every sibling spend field
5498 # Not required: rows queued by a pod running the previous release, or replayed from
5499 # the Redis buffer across an upgrade, carry no such key. Every reader coalesces a
5500 # missing value to zero, so requiring it here would describe a shape the aggregation
5501 # is explicitly tested against.
5502 autorouter_savings_spend: NotRequired[float]
5504 # request level metrics
5505 spend: float
5506 api_requests: int
5507 successful_requests: int
5508 failed_requests: int
5509 total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place
5510 timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place
5513class DailyTeamSpendTransaction(BaseDailySpendTransaction):
5514 team_id: str
5517class DailyOrganizationSpendTransaction(BaseDailySpendTransaction):
5518 organization_id: str
5521class DailyUserSpendTransaction(BaseDailySpendTransaction):
5522 user_id: str
5525class DailyEndUserSpendTransaction(BaseDailySpendTransaction):
5526 end_user_id: str
5529class DailyTagSpendTransaction(BaseDailySpendTransaction):
5530 request_id: str | None
5531 tag: str
5534class DailyAgentSpendTransaction(BaseDailySpendTransaction):
5535 agent_id: str
5538class DBSpendUpdateTransactions(TypedDict):
5539 """
5540 Internal Data Structure for buffering spend updates in Redis or in memory before committing them to the database
5541 """
5543 user_list_transactions: dict[str, float] | None
5544 end_user_list_transactions: dict[str, float] | None
5545 key_list_transactions: dict[str, float] | None
5546 team_list_transactions: dict[str, float] | None
5547 team_member_list_transactions: dict[str, float] | None
5548 org_list_transactions: dict[str, float] | None
5549 org_member_list_transactions: ReadOnly[dict[str, float] | None]
5550 project_list_transactions: ReadOnly[dict[str, float] | None]
5551 tag_list_transactions: dict[str, float] | None
5552 agent_list_transactions: dict[str, float] | None
5553 model_access_group_list_transactions: ReadOnly[dict[str, float] | None]
5556class SpendUpdateQueueItem(TypedDict, total=False):
5557 entity_type: Litellm_EntityType
5558 entity_id: str
5559 response_cost: float | None
5562class ToolDiscoveryQueueItem(TypedDict, total=False):
5563 tool_name: str
5564 origin: str | None # MCP server name or "user_defined"
5565 created_by: str | None
5566 key_hash: str | None # hash of virtual key that triggered discovery
5567 team_id: str | None # team that triggered discovery
5568 key_alias: str | None # human-readable key alias
5569 user_agent: str | None # HTTP User-Agent of the caller
5572from litellm.models.managed_files import ( # noqa: E402
5573 LiteLLM_ManagedFileTable as LiteLLM_ManagedFileTable,
5574)
5575from litellm.models.managed_files import ( # noqa: E402
5576 LiteLLM_ManagedObjectTable as LiteLLM_ManagedObjectTable,
5577)
5578from litellm.models.managed_files import ( # noqa: E402
5579 LiteLLM_ManagedVectorStoresTable as LiteLLM_ManagedVectorStoresTable,
5580)
5581from litellm.models.managed_files import ( # noqa: E402
5582 LiteLLM_ManagedVectorStoreTable as LiteLLM_ManagedVectorStoreTable,
5583)
5586class EnterpriseLicenseData(TypedDict, total=False):
5587 expiration_date: str
5588 user_id: str
5589 allowed_features: list[str]
5590 max_users: int
5591 max_teams: int
5594class ResponseLiteLLM_ManagedVectorStore(TypedDict, total=False):
5595 vector_store: LiteLLM_ManagedVectorStoresTable
5598class CostEstimateRequest(LiteLLMPydanticObjectBase):
5599 """Request body for /cost/estimate endpoint."""
5601 model: str = Field(description="Model name (from /model_group/info)")
5602 input_tokens: int = Field(description="Expected input tokens per request", ge=0)
5603 output_tokens: int = Field(description="Expected output tokens per request", ge=0)
5604 cache_read_input_tokens: int = Field(
5605 default=0, description="Input tokens read from the prompt cache; counted within input_tokens", ge=0
5606 )
5607 cache_creation_input_tokens: int = Field(
5608 default=0, description="Input tokens written to the prompt cache; counted within input_tokens", ge=0
5609 )
5610 reasoning_tokens: int = Field(
5611 default=0, description="Reasoning tokens the model emits; counted within output_tokens", ge=0
5612 )
5613 num_requests_per_day: int | None = Field(default=None, description="Number of requests per day", ge=0)
5614 num_requests_per_month: int | None = Field(default=None, description="Number of requests per month", ge=0)
5616 @model_validator(mode="after")
5617 def validate_token_subsets(self) -> "CostEstimateRequest":
5618 if self.cache_read_input_tokens + self.cache_creation_input_tokens > self.input_tokens:
5619 raise ValueError("cache_read_input_tokens plus cache_creation_input_tokens cannot exceed input_tokens")
5620 if self.reasoning_tokens > self.output_tokens:
5621 raise ValueError("reasoning_tokens cannot exceed output_tokens")
5622 return self
5625class CostEstimateResponse(LiteLLMPydanticObjectBase):
5626 """Response body for /cost/estimate endpoint."""
5628 model: str
5629 input_tokens: int
5630 output_tokens: int
5631 cache_read_input_tokens: int = 0
5632 cache_creation_input_tokens: int = 0
5633 reasoning_tokens: int = 0
5634 num_requests_per_day: int | None = None
5635 num_requests_per_month: int | None = None
5636 # Per-request costs
5637 cost_per_request: float = Field(description="Total cost per request (includes margin)")
5638 input_cost_per_request: float = Field(description="Input token cost per request (before margin)")
5639 output_cost_per_request: float = Field(description="Output token cost per request (before margin)")
5640 margin_cost_per_request: float = Field(default=0.0, description="Margin/fee added per request")
5641 cache_read_cost_per_request: float = Field(default=0.0, description="Cache-read share of input_cost_per_request")
5642 cache_creation_cost_per_request: float = Field(
5643 default=0.0, description="Cache-write share of input_cost_per_request"
5644 )
5645 reasoning_cost_per_request: float = Field(default=0.0, description="Reasoning share of output_cost_per_request")
5646 # Daily costs (if num_requests_per_day provided)
5647 daily_cost: float | None = Field(default=None, description="Total daily cost (includes margin)")
5648 daily_input_cost: float | None = Field(default=None, description="Daily input token cost")
5649 daily_output_cost: float | None = Field(default=None, description="Daily output token cost")
5650 daily_margin_cost: float | None = Field(default=None, description="Daily margin/fee")
5651 daily_cache_read_cost: float | None = Field(default=None, description="Cache-read share of daily_input_cost")
5652 daily_cache_creation_cost: float | None = Field(default=None, description="Cache-write share of daily_input_cost")
5653 daily_reasoning_cost: float | None = Field(default=None, description="Reasoning share of daily_output_cost")
5654 # Monthly costs (if num_requests_per_month provided)
5655 monthly_cost: float | None = Field(default=None, description="Total monthly cost (includes margin)")
5656 monthly_input_cost: float | None = Field(default=None, description="Monthly input token cost")
5657 monthly_output_cost: float | None = Field(default=None, description="Monthly output token cost")
5658 monthly_margin_cost: float | None = Field(default=None, description="Monthly margin/fee")
5659 monthly_cache_read_cost: float | None = Field(default=None, description="Cache-read share of monthly_input_cost")
5660 monthly_cache_creation_cost: float | None = Field(
5661 default=None, description="Cache-write share of monthly_input_cost"
5662 )
5663 monthly_reasoning_cost: float | None = Field(default=None, description="Reasoning share of monthly_output_cost")
5664 # Pricing info: the rates this request's usage bills at, after token tiers and regional multipliers
5665 input_cost_per_token: float | None = Field(default=None, description="Rate billed per input token")
5666 output_cost_per_token: float | None = Field(default=None, description="Rate billed per output token")
5667 cache_read_input_token_cost: float | None = Field(default=None, description="Rate billed per cache-read token")
5668 cache_creation_input_token_cost: float | None = Field(default=None, description="Rate billed per cache-write token")
5669 output_cost_per_reasoning_token: float | None = Field(default=None, description="Rate billed per reasoning token")
5670 provider: str | None = None