Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/_types.py: 89%

2428 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1import enum 

2import json 

3import os 

4from collections.abc import Callable, Mapping 

5from datetime import datetime 

6from types import MappingProxyType 

7from typing import TYPE_CHECKING, Annotated, Any, Final, Literal, NamedTuple, TypeAlias 

8 

9import httpx 

10from pydantic import ( 

11 BaseModel, 

12 BeforeValidator, 

13 ConfigDict, 

14 Field, 

15 Json, 

16 JsonValue, 

17 PositiveInt, 

18 field_validator, 

19 model_validator, 

20) 

21from typing_extensions import NotRequired, ReadOnly, Required, TypedDict 

22 

23from litellm._uuid import uuid 

24from litellm.constants import DEFAULT_STAGGER_WINDOW_SECONDS, MCP_STDIO_ALLOWED_COMMANDS 

25from litellm.litellm_core_utils.initialize_dynamic_callback_params import ( 

26 validate_langfuse_environment_value, 

27 validate_langfuse_span_scope_value, 

28 validate_no_callback_env_reference, 

29) 

30from litellm.proxy._experimental.mcp_server.stdio_gate import MCP_STDIO_DISABLED_MESSAGE, is_mcp_stdio_enabled 

31from litellm.types.agents import AgentCaller 

32from litellm.types.integrations.compression_interception import ( 

33 CompressionSavingsMetadata, 

34) 

35from litellm.types.integrations.slack_alerting import AlertType 

36from litellm.types.llms.openai import ( 

37 AllMessageValues, 

38 ResponsesAPIResponse, 

39) 

40from litellm.types.mcp import ( 

41 MCPAllowedClient, 

42 MCPAuth, 

43 MCPAuthType, 

44 MCPCredentials, 

45 MCPTransport, 

46 MCPTransportType, 

47) 

48from litellm.types.mcp_server.mcp_server_manager import MCPInfo 

49from litellm.types.proxy.carried_budget_state import ( 

50 OrgBudgetSnapshot, 

51 TeamBudgetSnapshot, 

52 UserBudgetSnapshot, 

53) 

54from litellm.types.proxy.control_plane_endpoints import WorkerRegistryEntry 

55from litellm.types.proxy.spend_capture_rate import SpendCaptureRateCheckSettings 

56from litellm.types.router import RouterErrors, UpdateRouterConfig 

57from litellm.types.router_weights import validate_router_settings_dict 

58from litellm.types.secret_managers.main import KeyManagementSystem 

59from litellm.types.utils import ( 

60 AzureSpillover, 

61 CallTypes, 

62 CostBreakdown, 

63 EmbeddingResponse, 

64 GenericBudgetConfigType, 

65 ImageResponse, 

66 InternalCallOrigin, 

67 LiteLLMPydanticObjectBase, 

68 ModelResponse, 

69 ProviderField, 

70 StandardCallbackDynamicParams, 

71 StandardLoggingGuardrailInformation, 

72 StandardLoggingMCPToolCall, 

73 StandardLoggingModelInformation, 

74 StandardLoggingPayloadErrorInformation, 

75 StandardLoggingPayloadStatus, 

76 StandardLoggingRoutingDecision, 

77 StandardLoggingVectorStoreRequest, 

78 StandardPassThroughResponseObject, 

79 TextCompletionResponse, 

80 TranscriptionResponse, 

81) 

82from litellm.types.videos.main import VideoObject 

83 

84from .types_utils.utils import get_instance_fn, validate_custom_validate_return_type 

85 

86if TYPE_CHECKING: 86 ↛ 87line 86 didn't jump to line 87 because the condition on line 86 was never true

87 from opentelemetry.trace import Span as _Span 

88 

89 Span = _Span | Any 

90else: 

91 Span = Any 

92 

93 

94class ReconcileOutcome(NamedTuple): 

95 """What a model reconcile observed, captured while it still held the reconcile 

96 lock. 

97 

98 Both fields have to be read under that lock to be worth anything. ``live_after`` 

99 in particular is the router's serving state the instant this reconcile finished, 

100 which is NOT the same as what a later snapshot would see: any other model write 

101 admitted in between briefly un-serves every db model (see ``clear_cache``), so a 

102 caller that re-snapshots at verdict time can observe that hole and blame its own 

103 reload for it. 

104 

105 - ``still_desired``: the db + config ids the reconcile reconciled against, or None 

106 when no reconcile ran and the desired set is therefore unknown. 

107 - ``live_after``: the ids the router served immediately after the reconcile, or 

108 None when no reconcile ran. 

109 """ 

110 

111 still_desired: frozenset[str] | None 

112 live_after: frozenset[str] | None 

113 

114 

115class SupportedDBObjectType(str, enum.Enum): 

116 """ 

117 Supported database object types for fine-grained DB storage control. 

118 Use in general_settings.supported_db_objects to specify which objects to load from DB. 

119 """ 

120 

121 MODELS = "models" 

122 MCP = "mcp" 

123 GUARDRAILS = "guardrails" 

124 POLICIES = "policies" 

125 VECTOR_STORES = "vector_stores" 

126 PASS_THROUGH_ENDPOINTS = "pass_through_endpoints" 

127 PROMPTS = "prompts" 

128 MODEL_COST_MAP = "model_cost_map" 

129 TOOLS = "tools" 

130 CONFIG_OVERRIDES = "config_overrides" 

131 WEBSEARCH_INTERCEPTION_SETTINGS = "websearch_interception_settings" 

132 

133 def __str__(self): 

134 return str(self.value) 

135 

136 

137class LiteLLMTeamRoles(enum.Enum): 

138 # team admin 

139 TEAM_ADMIN = "admin" 

140 # team member 

141 TEAM_MEMBER = "user" 

142 

143 

144class LitellmUserRoles(str, enum.Enum): 

145 """ 

146 Admin Roles: 

147 PROXY_ADMIN: admin over the platform 

148 PROXY_ADMIN_VIEW_ONLY: can login, view all own keys, view all spend 

149 ORG_ADMIN: admin over a specific organization, can create teams, users only within their organization 

150 

151 Internal User Roles: 

152 INTERNAL_USER: can login, view/create/delete their own keys, view their spend 

153 INTERNAL_USER_VIEW_ONLY: can login, view their own keys, view their own spend 

154 

155 

156 Team Roles: 

157 TEAM: used for JWT auth 

158 

159 

160 Customer Roles: 

161 CUSTOMER: External users -> these are customers 

162 

163 """ 

164 

165 # Admin Roles 

166 PROXY_ADMIN = "proxy_admin" 

167 PROXY_ADMIN_VIEW_ONLY = "proxy_admin_viewer" 

168 

169 # Organization admins 

170 ORG_ADMIN = "org_admin" 

171 

172 # Internal User Roles 

173 INTERNAL_USER = "internal_user" 

174 INTERNAL_USER_VIEW_ONLY = "internal_user_viewer" 

175 

176 # Team Roles 

177 TEAM = "team" 

178 

179 # Customer Roles - External users of proxy 

180 CUSTOMER = "customer" 

181 

182 def __str__(self): 

183 return str(self.value) 

184 

185 def values(self) -> list[str]: 

186 return list(self.__annotations__.keys()) 

187 

188 @property 

189 def description(self): 

190 """ 

191 Descriptions for the enum values 

192 """ 

193 descriptions: Final = { 

194 "proxy_admin": "admin over litellm proxy, has all permissions", 

195 "proxy_admin_viewer": "view all keys, view all spend", 

196 "internal_user": "view/create/delete their own keys, view their own spend", 

197 "internal_user_viewer": "view their own keys, view their own spend", 

198 "team": "team scope used for JWT auth", 

199 "customer": "customer", 

200 } 

201 return descriptions.get(self.value, "") 

202 

203 @property 

204 def ui_label(self): 

205 """ 

206 UI labels for the enum values 

207 """ 

208 ui_labels: Final = { 

209 "proxy_admin": "Admin (All Permissions)", 

210 "proxy_admin_viewer": "Admin (View Only)", 

211 "internal_user": "Internal User (Create/Delete/View)", 

212 "internal_user_viewer": "Internal User (View Only)", 

213 "team": "Team", 

214 "customer": "Customer", 

215 } 

216 return ui_labels.get(self.value, "") 

217 

218 @property 

219 def is_internal_user_role(self) -> bool: 

220 """returns true if this role is an `internal_user` or `internal_user_viewer` role""" 

221 return self.value in [ 

222 self.INTERNAL_USER, 

223 self.INTERNAL_USER_VIEW_ONLY, 

224 ] 

225 

226 

227class LitellmTableNames(str, enum.Enum): 

228 """ 

229 Enum for Table Names used by LiteLLM 

230 """ 

231 

232 TEAM_TABLE_NAME = "LiteLLM_TeamTable" 

233 USER_TABLE_NAME = "LiteLLM_UserTable" 

234 KEY_TABLE_NAME = "LiteLLM_VerificationToken" 

235 PROXY_MODEL_TABLE_NAME = "LiteLLM_ProxyModelTable" 

236 MANAGED_FILE_TABLE_NAME = "LiteLLM_ManagedFileTable" 

237 TOOL_TABLE_NAME = "LiteLLM_ToolTable" 

238 CACHE_CONFIG_TABLE_NAME = "LiteLLM_CacheConfig" 

239 CONFIG_OVERRIDES_TABLE_NAME = "LiteLLM_ConfigOverrides" 

240 CONFIG_TABLE_NAME = "LiteLLM_Config" 

241 SSO_CONFIG_TABLE_NAME = "LiteLLM_SSOConfig" 

242 UI_SETTINGS_TABLE_NAME = "LiteLLM_UISettings" 

243 AGENT_TABLE_NAME = "LiteLLM_AgentsTable" 

244 

245 

246class Litellm_EntityType(enum.Enum): 

247 """ 

248 Enum for types of entities on litellm 

249 

250 This enum allows specifying the type of entity that is being tracked in the database. 

251 """ 

252 

253 KEY = "key" 

254 USER = "user" 

255 END_USER = "end_user" 

256 TEAM = "team" 

257 TEAM_MEMBER = "team_member" 

258 ORGANIZATION = "organization" 

259 ORGANIZATION_MEMBER = "organization_member" 

260 PROJECT = "project" 

261 TAG = "tag" 

262 AGENT = "agent" 

263 MODEL_ACCESS_GROUP = "model_access_group" 

264 

265 # global proxy level entity 

266 PROXY = "proxy" 

267 

268 

269def hash_token(token: str): 

270 import hashlib 

271 

272 # Hash the string using SHA-256 

273 hashed_token: Final = hashlib.sha256(token.encode()).hexdigest() 

274 

275 return hashed_token 

276 

277 

278class KeyManagementRoutes(str, enum.Enum): 

279 """ 

280 Enum for key management routes 

281 """ 

282 

283 # write routes 

284 KEY_GENERATE = "/key/generate" 

285 KEY_UPDATE = "/key/update" 

286 KEY_DELETE = "/key/delete" 

287 KEY_REGENERATE = "/key/regenerate" 

288 KEY_GENERATE_SERVICE_ACCOUNT = "/key/service-account/generate" 

289 KEY_REGENERATE_WITH_PATH_PARAM = "/key/{key_id}/regenerate" 

290 KEY_BLOCK = "/key/block" 

291 KEY_UNBLOCK = "/key/unblock" 

292 KEY_BULK_UPDATE = "/key/bulk_update" 

293 TEAM_KEY_BULK_UPDATE = "/team/key/bulk_update" 

294 KEY_RESET_SPEND = "/key/{key_id}/reset_spend" 

295 

296 # Field-level opt-in permission (not a real HTTP route). When present in a 

297 # team's `team_member_permissions`, non-admin members of that team may set 

298 # `access_group_ids` on keys they create/update. Default-deny. 

299 KEY_ACCESS_GROUP_ASSIGNMENT = "/key/access_group_assignment" 

300 AUTO_ROUTER_MANAGE = "/auto_router/manage" 

301 

302 # info and health routes 

303 KEY_INFO = "/key/info" 

304 KEY_HEALTH = "/key/health" 

305 

306 # list routes 

307 KEY_LIST = "/key/list" 

308 KEY_ALIASES = "/key/aliases" 

309 

310 # team usage routes 

311 TEAM_DAILY_ACTIVITY = "/team/daily/activity" 

312 TEAM_DAILY_ACTIVITY_AGGREGATED = "/team/daily/activity/aggregated" 

313 

314 # team spend-log viewing 

315 SPEND_LOGS = "/spend/logs" 

316 SPEND_LOGS_V2 = "/spend/logs/v2" 

317 

318 

319class LiteLLMRoutes(enum.Enum): 

320 openai_route_names = [ 

321 "chat_completion", 

322 "completion", 

323 "embeddings", 

324 "image_generation", 

325 "video_generation", 

326 "audio_transcriptions", 

327 "moderations", 

328 "model_list", # OpenAI /v1/models route 

329 ] 

330 openai_routes = [ 

331 # chat completions 

332 "/engines/{model}/chat/completions", 

333 "/openai/deployments/{model}/chat/completions", 

334 "/chat/completions", 

335 "/v1/chat/completions", 

336 "/cursor/chat/completions", 

337 "/cursor/models", 

338 "/cursor/v1/models", 

339 # completions 

340 "/engines/{model}/completions", 

341 "/openai/deployments/{model}/completions", 

342 "/completions", 

343 "/v1/completions", 

344 # embeddings 

345 "/engines/{model}/embeddings", 

346 "/openai/deployments/{model}/embeddings", 

347 "/embeddings", 

348 "/v1/embeddings", 

349 # image generation 

350 "/images/generations", 

351 "/v1/images/generations", 

352 # image edit 

353 "/images/edits", 

354 "/v1/images/edits", 

355 # video generation 

356 "/videos", 

357 "/v1/videos", 

358 "/videos/{video_id}", 

359 "/v1/videos/{video_id}", 

360 "/videos/{video_id}/content", 

361 "/v1/videos/{video_id}/content", 

362 "/videos/{video_id}/remix", 

363 "/v1/videos/{video_id}/remix", 

364 # audio transcription 

365 "/audio/transcriptions", 

366 "/v1/audio/transcriptions", 

367 # audio Speech 

368 "/audio/speech", 

369 "/v1/audio/speech", 

370 # moderations 

371 "/moderations", 

372 "/v1/moderations", 

373 # batches 

374 "/v1/batches", 

375 "/batches", 

376 "/v1/batches/{batch_id}", 

377 "/batches/{batch_id}", 

378 "/v1/batches/{batch_id}/cancel", 

379 "/batches/{batch_id}/cancel", 

380 # files 

381 "/v1/files", 

382 "/files", 

383 "/v1/files/{file_id}", 

384 "/files/{file_id}", 

385 "/v1/files/{file_id}/content", 

386 "/files/{file_id}/content", 

387 # fine_tuning 

388 "/fine_tuning/jobs", 

389 "/v1/fine_tuning/jobs", 

390 "/fine_tuning/jobs/{fine_tuning_job_id}/cancel", 

391 "/v1/fine_tuning/jobs/{fine_tuning_job_id}/cancel", 

392 # assistants-related routes 

393 "/assistants", 

394 "/v1/assistants", 

395 "/v1/assistants/{assistant_id}", 

396 "/assistants/{assistant_id}", 

397 "/threads", 

398 "/v1/threads", 

399 "/threads/{thread_id}", 

400 "/v1/threads/{thread_id}", 

401 "/threads/{thread_id}/messages", 

402 "/v1/threads/{thread_id}/messages", 

403 "/threads/{thread_id}/runs", 

404 "/v1/threads/{thread_id}/runs", 

405 # models 

406 "/models", 

407 "/v1/models", 

408 # token counter 

409 "/utils/token_counter", 

410 "/utils/model_info", 

411 "/utils/transform_request", 

412 # rerank 

413 "/rerank", 

414 "/v1/rerank", 

415 "/v2/rerank", 

416 # realtime 

417 "/realtime", 

418 "/v1/realtime", 

419 "/openai/v1/realtime", 

420 "/realtime?{model}", 

421 "/v1/realtime?{model}", 

422 "/openai/v1/realtime?{model}", 

423 # realtime (GA WebRTC HTTP routes) 

424 "/realtime/client_secrets", 

425 "/v1/realtime/client_secrets", 

426 "/openai/v1/realtime/client_secrets", 

427 "/realtime/calls", 

428 "/v1/realtime/calls", 

429 "/openai/v1/realtime/calls", 

430 "/realtime/transcription_sessions", 

431 "/v1/realtime/transcription_sessions", 

432 "/openai/v1/realtime/transcription_sessions", 

433 # responses API 

434 "/responses", 

435 "/v1/responses", 

436 "/openai/v1/responses", 

437 "/responses/{response_id}", 

438 "/v1/responses/{response_id}", 

439 "/openai/v1/responses/{response_id}", 

440 "/responses/{response_id}/input_items", 

441 "/v1/responses/{response_id}/input_items", 

442 "/openai/v1/responses/{response_id}/input_items", 

443 "/responses/{response_id}/cancel", 

444 "/v1/responses/{response_id}/cancel", 

445 "/openai/v1/responses/{response_id}/cancel", 

446 "/responses/input_tokens", 

447 "/v1/responses/input_tokens", 

448 "/openai/v1/responses/input_tokens", 

449 # vector stores 

450 "/vector_stores", 

451 "/v1/vector_stores", 

452 "/vector_stores/{vector_store_id}", 

453 "/v1/vector_stores/{vector_store_id}", 

454 "/vector_stores/{vector_store_id}/search", 

455 "/v1/vector_stores/{vector_store_id}/search", 

456 "/vector_stores/{vector_store_id}/files", 

457 "/v1/vector_stores/{vector_store_id}/files", 

458 "/vector_stores/{vector_store_id}/files/{file_id}", 

459 "/v1/vector_stores/{vector_store_id}/files/{file_id}", 

460 "/vector_stores/{vector_store_id}/files/{file_id}/content", 

461 "/v1/vector_stores/{vector_store_id}/files/{file_id}/content", 

462 "/vector_store/list", 

463 "/v1/vector_store/list", 

464 # search 

465 "/search", 

466 "/v1/search", 

467 "/search/{search_tool_name}", 

468 "/v1/search/{search_tool_name}", 

469 "/decisions", 

470 "/v1/decisions", 

471 "/systemone", 

472 "/v1/systemone", 

473 # OCR 

474 "/ocr", 

475 "/v1/ocr", 

476 # containers API 

477 "/containers", 

478 "/v1/containers", 

479 "/containers/*", 

480 "/v1/containers/*", 

481 ] 

482 

483 mapped_pass_through_routes = [ 

484 "/bedrock", 

485 "/comprehendmedical", 

486 "/azure_speech", 

487 "/transcribe", 

488 "/vertex-ai", 

489 "/vertex_ai", 

490 "/cohere", 

491 "/cursor", 

492 "/gemini", 

493 "/anthropic", 

494 "/langfuse", 

495 "/azure", 

496 "/azure_ai", 

497 "/openai", 

498 "/openai_passthrough", 

499 "/assemblyai", 

500 "/eu.assemblyai", 

501 "/tinyfish", 

502 "/vllm", 

503 "/mistral", 

504 "/typesafe", 

505 "/openrouter", 

506 "/milvus", 

507 "/gigachat", 

508 "/watsonx", 

509 "/nvidia_nim", 

510 "/deepgram", 

511 "/fal_ai", 

512 ] 

513 

514 ######################################################### 

515 # e.g /vllm/*, anthropic/*, etc. 

516 # allows using /anthropic/v1/messages, /vllm/v1/chat/completions, etc. 

517 ######################################################### 

518 passthrough_routes_wildcard = [f"{route}/*" for route in mapped_pass_through_routes] 

519 

520 litellm_native_routes = [ 

521 "/rag/ingest", 

522 "/v1/rag/ingest", 

523 "/rag/query", 

524 "/v1/rag/query", 

525 ] 

526 

527 anthropic_routes = [ 

528 "/v1/messages", 

529 "/v1/messages/count_tokens", 

530 "/claude_code_gateway/v1/messages", 

531 "/claude_code_gateway/v1/messages/count_tokens", 

532 "/v1/skills", 

533 "/v1/skills/{skill_id}", 

534 "/claude-code/marketplace.json", 

535 "/claude-code/plugins", 

536 "/claude-code/plugins/{plugin_name}", 

537 ] 

538 

539 # MCP tool-call / passthrough routes — data-plane. Gated by DISABLE_LLM_API_ENDPOINTS. 

540 mcp_inference_routes = [ 

541 "/mcp", 

542 "/mcp/", 

543 "/mcp/sse/", 

544 "/mcp/sse/messages", 

545 "/mcp/sse/messages/", 

546 "/mcp/proxy", 

547 "/mcp/{subpath}", 

548 "/mcp/tools", 

549 "/mcp/tools/list", 

550 "/mcp/tools/call", 

551 "/mcp-rest/tools/list", 

552 "/mcp-rest/tools/call", 

553 "/v1/mcp/tools", 

554 "/introspect", 

555 "/token", 

556 ] 

557 

558 # MCP server CRUD routes — control-plane. Gated by DISABLE_ADMIN_ENDPOINTS. 

559 mcp_management_routes = [ 

560 "/v1/mcp/server", 

561 "/v1/mcp/server/{path:path}", 

562 "/v1/mcp/sessions", 

563 ] 

564 

565 # Backwards-compat union — virtual keys may be configured with 

566 # allowed_routes=["mcp_routes"], which should cover both halves. 

567 mcp_routes = mcp_inference_routes + mcp_management_routes 

568 

569 # A2A agent invocation / discovery routes — data-plane. Gated by DISABLE_LLM_API_ENDPOINTS. 

570 agent_inference_routes = ( 

571 "/agents", 

572 "/a2a/{agent_id}", 

573 "/a2a/{agent_id}/message/send", 

574 "/a2a/{agent_id}/message/stream", 

575 "/a2a/{agent_id}/.well-known/agent-card.json", 

576 ) 

577 

578 # Agent registry CRUD routes — control-plane. Gated by DISABLE_ADMIN_ENDPOINTS. 

579 # The handlers in agent_endpoints/endpoints.py enforce proxy-admin on writes and 

580 # scope reads by role, so these also appear in self_managed_routes. 

581 agent_management_routes = ( 

582 "/v1/agents", 

583 "/v1/agents/{agent_id}", 

584 "/v1/agents/make_public", 

585 "/v1/agents/{agent_id}/make_public", 

586 "/v1/agents/{agent_id}/kill_switch", 

587 ) 

588 

589 # Backwards-compat union — virtual keys may be configured with 

590 # allowed_routes=["agent_routes"], which should cover both halves. 

591 agent_routes = agent_inference_routes + agent_management_routes 

592 

593 google_routes = [ 

594 "/v1beta/models/{model_name:path}:countTokens", 

595 "/v1beta/models/{model_name:path}:generateContent", 

596 "/v1beta/models/{model_name:path}:streamGenerateContent", 

597 "/models/{model_name:path}:countTokens", 

598 "/models/{model_name:path}:generateContent", 

599 "/models/{model_name:path}:streamGenerateContent", 

600 # Google Interactions API 

601 "/interactions", 

602 "/v1beta/interactions", 

603 "/interactions/{interaction_id}", 

604 "/v1beta/interactions/{interaction_id}", 

605 "/interactions/{interaction_id}/cancel", 

606 "/v1beta/interactions/{interaction_id}/cancel", 

607 # Google Managed Agents API 

608 "/v1beta/agents", 

609 "/v1beta/agents/{name}", 

610 "/v1beta/agents/{name}/versions", 

611 ] 

612 

613 apply_guardrail_routes = [ 

614 "/guardrails/apply_guardrail", 

615 ] 

616 

617 model_info_routes = [ 

618 "/model/info", 

619 "/v1/model/info", 

620 "/model_group/info", 

621 ] 

622 

623 llm_api_routes = ( 

624 openai_routes 

625 + anthropic_routes 

626 + google_routes 

627 + mapped_pass_through_routes 

628 + passthrough_routes_wildcard 

629 + apply_guardrail_routes 

630 + mcp_inference_routes 

631 + litellm_native_routes 

632 + list(agent_inference_routes) 

633 + model_info_routes 

634 ) 

635 info_routes = [ 

636 "/key/info", 

637 "/key/health", 

638 "/team/info", 

639 "/team/list", 

640 "/v2/team/list", 

641 "/organization/list", 

642 "/team/available", 

643 "/team/metadata_schema", 

644 "/user/info", 

645 "/v2/user/info", 

646 "/model/info", 

647 "/v1/model/info", 

648 "/v2/model/info", 

649 "/v2/key/info", 

650 "/model_group/info", 

651 "/health", 

652 "/health/services", 

653 "/key/list", 

654 "/user/filter/ui", 

655 "/models", 

656 "/v1/models", 

657 "/sso/get/ui_settings", 

658 "/get/user_banner", 

659 "/get/latest_release_info", 

660 ] 

661 

662 # NOTE: ROUTES ONLY FOR MASTER KEY - only the Master Key should be able to Reset Spend 

663 master_key_only_routes = [ 

664 "/global/spend/reset", 

665 "/memory-usage-in-mem-cache", 

666 "/memory-usage-in-mem-cache-items", 

667 ] 

668 

669 key_management_routes = [ 

670 KeyManagementRoutes.KEY_GENERATE.value, 

671 KeyManagementRoutes.KEY_UPDATE.value, 

672 KeyManagementRoutes.KEY_DELETE.value, 

673 KeyManagementRoutes.KEY_INFO.value, 

674 KeyManagementRoutes.KEY_REGENERATE.value, 

675 KeyManagementRoutes.KEY_GENERATE_SERVICE_ACCOUNT.value, 

676 KeyManagementRoutes.KEY_REGENERATE_WITH_PATH_PARAM.value, 

677 KeyManagementRoutes.KEY_LIST.value, 

678 KeyManagementRoutes.KEY_BLOCK.value, 

679 KeyManagementRoutes.KEY_UNBLOCK.value, 

680 KeyManagementRoutes.KEY_BULK_UPDATE.value, 

681 KeyManagementRoutes.TEAM_KEY_BULK_UPDATE.value, 

682 KeyManagementRoutes.TEAM_DAILY_ACTIVITY.value, 

683 KeyManagementRoutes.TEAM_DAILY_ACTIVITY_AGGREGATED.value, 

684 KeyManagementRoutes.SPEND_LOGS.value, 

685 KeyManagementRoutes.SPEND_LOGS_V2.value, 

686 KeyManagementRoutes.KEY_RESET_SPEND.value, 

687 KeyManagementRoutes.KEY_ALIASES.value, 

688 KeyManagementRoutes.KEY_ACCESS_GROUP_ASSIGNMENT.value, 

689 KeyManagementRoutes.AUTO_ROUTER_MANAGE.value, 

690 ] 

691 

692 team_service_account_key_routes = ( 

693 KeyManagementRoutes.KEY_GENERATE.value, 

694 KeyManagementRoutes.KEY_UPDATE.value, 

695 ) 

696 

697 management_routes = ( 

698 [ 

699 # user 

700 "/user/new", 

701 "/management/v1/users/bulk", 

702 "/user/update", 

703 "/user/bulk_update", 

704 "/user/delete", 

705 "/management/v1/users/bulk_delete", 

706 "/user/info", 

707 "/user/list", 

708 "/user/daily/activity", 

709 "/user/daily/activity/aggregated", 

710 # team 

711 "/team/new", 

712 "/team/update", 

713 "/team/{team_id}", 

714 "/team/delete", 

715 "/team/list", 

716 "/v2/team/list", 

717 "/team/info", 

718 "/team/block", 

719 "/team/unblock", 

720 "/team/available", 

721 "/team/metadata_schema", 

722 "/team/permissions_list", 

723 "/team/permissions_update", 

724 "/team/permissions_bulk_update", 

725 "/team/daily/activity", 

726 "/team/daily/activity/aggregated", 

727 "/team/spend/by_user", 

728 # gateway request counts (SGR); deployment-wide, admin-only 

729 "/gateway/daily/activity", 

730 # model 

731 "/model/new", 

732 "/model/update", 

733 "/model/delete", 

734 "/model/info", 

735 "/jwt/key/mapping/new", 

736 "/jwt/key/mapping/update", 

737 "/jwt/key/mapping/delete", 

738 "/jwt/key/mapping/list", 

739 "/jwt/key/mapping/info", 

740 ] 

741 + key_management_routes 

742 + mcp_management_routes 

743 + list(agent_management_routes) 

744 ) 

745 

746 spend_tracking_routes = [ 

747 # spend 

748 "/spend/keys", 

749 "/spend/users", 

750 "/spend/tags", 

751 "/spend/calculate", 

752 "/spend/logs", 

753 "/spend/logs/v2", 

754 "/spend/logs/ui", 

755 "/spend/logs/ui/{request_id}", 

756 "/spend/logs/session/ui", 

757 "/key/spend/report", 

758 "/user/spend/report", 

759 "/team/spend/report", 

760 "/organization/spend/report", 

761 # Reads end users out of spend logs, scoped to the caller's own rows and 

762 # permitted teams exactly like /spend/logs/ui — it belongs to the same 

763 # access tier, not to customer management. 

764 "/management/v1/spend_logs/end_users", 

765 "/management/v1/spend_logs/users", 

766 "/cost/estimate", 

767 ] 

768 

769 global_spend_tracking_routes = [ 

770 # global spend 

771 "/global/spend/logs", 

772 "/global/spend", 

773 "/global/spend/keys", 

774 "/global/spend/teams", 

775 "/global/spend/end_users", 

776 "/global/spend/models", 

777 "/global/predict/spend/logs", 

778 "/global/spend/report", 

779 "/global/spend/provider", 

780 "/global/spend/tags", 

781 "/global/spend/all_tag_names", 

782 "/spend/capture_rate", 

783 ] 

784 

785 public_routes = frozenset( 

786 ( 

787 "/routes", 

788 "/", 

789 "/health/liveliness", 

790 "/health/liveness", 

791 "/test", 

792 "/config/yaml", 

793 "/litellm/.well-known/litellm-ui-config", 

794 "/.well-known/litellm-ui-config", 

795 "/public/model_hub", 

796 "/public/v1/model_hub", 

797 "/public/v1/model_hub/providers", 

798 "/public/v1/model_hub/modes", 

799 "/public/v1/model_hub/features", 

800 "/public/model_hub/info", 

801 "/public/agent_hub", 

802 "/public/mcp_hub", 

803 "/public/skill_hub", 

804 "/public/litellm_model_cost_map", 

805 ) 

806 ) 

807 

808 # Retained for backwards compatibility with JWT auth configs that reference 

809 # "ui_routes" in admin_allowed_routes. Not used by the proxy's own route 

810 # authorization — UI tokens now go through the same RBAC path as API tokens. 

811 ui_routes = [ 

812 "/sso", 

813 "/sso/get/ui_settings", 

814 "/get/ui_settings", 

815 "/login", 

816 "/key/info", 

817 "/config", 

818 "/spend", 

819 "/model/info", 

820 "/v2/model/info", 

821 "/v2/key/info", 

822 "/models", 

823 "/v1/models", 

824 "/global/spend", 

825 "/global/spend/logs", 

826 "/global/spend/keys", 

827 "/global/spend/models", 

828 "/global/spend/tags", 

829 "/global/predict/spend/logs", 

830 "/global/activity", 

831 "/gateway/daily/activity", 

832 "/health/services", 

833 ] + info_routes 

834 

835 # Stateless validators on caller-supplied log data; source logs are 

836 # already accessible via spend_tracking_routes, so no scope expansion. 

837 compliance_check_routes = [ 

838 "/compliance/eu-ai-act", 

839 "/compliance/gdpr", 

840 ] 

841 

842 # Routes in `global_spend_tracking_routes` return proxy-wide spend across 

843 # every team, customer, and api_key. They are intentionally NOT included 

844 # here — non-admin roles must not see other tenants' spend. Admin roles go 

845 # through their own branches in `route_checks.py`, and a key minted with 

846 # the `get_spend_routes` permission retains explicit opt-in access. 

847 internal_user_routes = ( 

848 [ 

849 "/global/activity", 

850 "/global/activity/model", 

851 "/global/activity/cache_hits", 

852 # Tag usage endpoints scope internal users to tags produced by 

853 # their own keys in tag_management_endpoints.py. 

854 "/tag/daily/activity", 

855 "/tag/list", 

856 "/v1/models/{model_id}", 

857 "/models/{model_id}", 

858 "/guardrails/list", 

859 "/v2/guardrails/list", 

860 "/project/list", 

861 "/project/info", 

862 # Read-only search tool routes power the Search Tools UI page. 

863 # Create/update/delete and test_connection stay admin-only. 

864 "/search_tools/list", 

865 "/search_tools/ui/available_providers", 

866 ] 

867 + spend_tracking_routes 

868 + key_management_routes 

869 + compliance_check_routes 

870 ) 

871 

872 internal_user_view_only_routes = ( 

873 spend_tracking_routes 

874 + compliance_check_routes 

875 + [ 

876 # Tag usage endpoints scope internal viewers to tags produced by 

877 # their own keys in tag_management_endpoints.py. 

878 "/tag/daily/activity", 

879 "/tag/list", 

880 ] 

881 ) 

882 

883 self_managed_routes = [ 

884 # update_team resolves proxy/org/team admin itself and filters team admins 

885 # through the team_admin_editable_team_fields setting 

886 "/team/update", 

887 "/team/member_add", 

888 "/team/member_delete", 

889 "/management/v1/teams/{team_id}/members/bulk_delete", 

890 "/management/v1/teams/{team_id}/members/bulk_update", 

891 "/team/member_update", 

892 "/team/{team_id}/member/{user_id}/reset_spend", 

893 "/team/{team_id}/member/{user_id}/reset_budget", 

894 "/team/permissions_list", 

895 "/team/permissions_update", 

896 "/team/daily/activity", 

897 "/team/daily/activity/aggregated", 

898 "/team/spend/by_user", 

899 "/team/{team_id}/members/me", 

900 # POST/GET the team's logging callbacks, and DELETE one of them. Every 

901 # handler calls _verify_team_access, which admits only a proxy admin, an 

902 # org admin for the team, or an admin of this team. 

903 # 

904 # team_id is a free-form string, so it spells these with the same path 

905 # converter the router uses; the gate matches that converter. 

906 "/team/{team_id:path}/callback", 

907 "/team/{team_id:path}/callback/{callback_name}", 

908 "/model/new", 

909 "/model/update", 

910 "/model/delete", 

911 "/user/daily/activity", 

912 "/user/daily/activity/aggregated", 

913 # Endpoint restricts results to organizations the caller is ORG_ADMIN 

914 # of; a caller who administers none gets an empty result set. 

915 "/organization/daily/activity", 

916 "/user/available_roles", # read-only role metadata; any authenticated user may read 

917 # Claude Code gateway: the signed-in CLI fetches its managed settings and posts its own telemetry 

918 "/claude_code_gateway/managed/settings", 

919 "/claude_code_gateway/v1/metrics", 

920 "/claude_code_gateway/v1/logs", 

921 "/claude_code_gateway/v1/traces", 

922 "/user/list", # org admins checked in endpoint; non-admins get 403 

923 "/management/v1/users/bulk_delete", # proxy admins delete anyone, org admins only their orgs' users; others 403 

924 "/user/password/change", # endpoint only ever writes the caller's own row 

925 "/session/logout", # endpoint only ever revokes the caller's own session key 

926 "/model/{model_id}/update", 

927 "/prompt/list", 

928 "/prompt/info", 

929 "/vector_store/info", 

930 # Project read routes - endpoint scopes results to caller's teams (non-admin) 

931 "/project/list", 

932 "/project/info", 

933 # Project write routes - endpoint checks team admin + team_admin_editable_team_fields "projects" 

934 "/project/new", 

935 "/project/update", 

936 # Endpoint enforces proxy-admin vs team-admin model access itself. 

937 "/health/test_connection", 

938 # Invitation routes - org/team admins checked in endpoint via _user_has_admin_privileges 

939 "/invitation/new", 

940 "/invitation/delete", 

941 # Team guardrail submission - requires team-scoped key; endpoint enforces team_id 

942 "/guardrails/register", 

943 # Team guardrail submissions - endpoint scopes results to caller's teams (non-admin) 

944 "/guardrails/submissions", 

945 "/guardrails/submissions/{guardrail_id}", 

946 # Auto-router dry runs - both gate like the /model/new write they rehearse: 

947 # proxy admin, or team admin naming their own team via team_id 

948 "/auto_router/test_routing", 

949 "/auto_router/validate_complexity_router_config", 

950 "/auto_router/availability", 

951 # Per-session auto-router read - the endpoint scopes the row to the caller's own key hash 

952 "/auto_router/session", 

953 "/cost/predict-cache", 

954 # Agent registry - reads are role-scoped and writes are proxy-admin-gated 

955 # inside agent_endpoints/endpoints.py 

956 *agent_management_routes, 

957 ] # routes that manage their own allowed/disallowed logic 

958 

959 ## Org Admin Routes ## 

960 

961 # Routes only an Org Admin Can Access 

962 org_admin_only_routes = [ 

963 "/organization/info", 

964 "/organization/delete", 

965 "/organization/member_add", 

966 "/organization/member_update", 

967 # member_delete is equally destructive as member_add / member_update 

968 # and must be scoped the same way — otherwise it falls through to 

969 # the management_routes / self_managed_routes path and lets any 

970 # non-PROXY_ADMIN caller that reaches the route delete arbitrary 

971 # org memberships without the organization_role_based_access_check 

972 # that member_add / member_update trigger. 

973 "/organization/member_delete", 

974 ] 

975 

976 # Routes accessible by Admin Viewer (read-only admin access). 

977 # 

978 # Admin Viewer follows a read-parity-with-Proxy-Admin rule: anything Proxy 

979 # Admin can read/list/get, Admin Viewer can too (no writes, no cost-incurring 

980 # actions). 

981 # 

982 # NOTE: This list is no longer the primary mechanism for granting access — 

983 # `_check_proxy_admin_viewer_access()` in route_checks.py default-allows 

984 # any safe HTTP method (GET/HEAD/OPTIONS) on non-inference routes. This 

985 # list now matters only for non-GET routes that are semantically reads 

986 # (e.g. POST /spend/calculate). Adding a new GET endpoint does not require 

987 # updating this list — the default-allow behavior covers it automatically. 

988 admin_viewer_routes = ( 

989 [ 

990 "/user/list", 

991 "/user/available_users", 

992 "/user/available_roles", 

993 "/user/daily/activity", 

994 "/team/daily/activity", 

995 "/team/daily/activity/aggregated", 

996 "/tag/daily/activity", 

997 "/tag/list", 

998 "/audit", 

999 "/audit/{id}", 

1000 "/global/activity", 

1001 "/global/activity/model", 

1002 "/global/activity/cache_hits", 

1003 # Customer / end-user listing (handlers already gate on 

1004 # PROXY_ADMIN_VIEW_ONLY — the route gate must match). 

1005 "/customer/list", 

1006 "/customer/info", 

1007 # UI Logs page session detail drawer and the end-user filter facet. 

1008 # The list endpoint `/spend/logs/ui` and the single-log detail route 

1009 # `/spend/logs/ui/{request_id}` are covered via spend_tracking_routes 

1010 # below. 

1011 "/spend/logs/session/ui", 

1012 "/management/v1/spend_logs/end_users", 

1013 "/management/v1/spend_logs/users", 

1014 # Settings / observability read endpoints exposed in admin-only 

1015 # sidebar groups (Logging & Alerts, Admin Settings, Budgets, 

1016 # Invitations). 

1017 "/callbacks/list", 

1018 "/callbacks/configs", 

1019 "/get/config/callbacks", 

1020 "/alerting/settings", 

1021 "/config/list", 

1022 "/config/field/info", 

1023 "/budget/list", 

1024 "/management/v1/budgets", 

1025 "/budget/settings", 

1026 # Invitation viewing (admin viewer cannot create/delete; can read). 

1027 "/invitation/info", 

1028 # Guardrails / Policies pages (read-only views). 

1029 "/guardrails/list", 

1030 "/v2/guardrails/list", 

1031 "/guardrails/submissions", 

1032 "/guardrails/submissions/{guardrail_id}", 

1033 "/guardrails/usage/overview", 

1034 "/policies/attachments/list", 

1035 # MCP semantic filter settings (read). 

1036 "/get/mcp_semantic_filter_settings", 

1037 # Model cost map maintenance views (read-only status / source). 

1038 "/schedule/model_cost_map_reload/status", 

1039 "/model/cost_map/source", 

1040 # A pure read; POST only so the prompt does not ride in a URL. 

1041 "/auto_router/classifier/default_prompt", 

1042 ] 

1043 # Spend tracking reads (/spend/logs, /spend/logs/ui, /spend/keys, 

1044 # /spend/users, /spend/tags, /spend/calculate, /cost/estimate). Admin 

1045 # Viewer can already read /global/spend/* via global_spend_tracking_routes; 

1046 # the per-tenant /spend/* views were the missing peer. 

1047 + spend_tracking_routes 

1048 + info_routes 

1049 ) 

1050 

1051 # All routes accesible by an Org Admin 

1052 org_admin_allowed_routes = org_admin_only_routes + management_routes + self_managed_routes + admin_viewer_routes 

1053 

1054 

1055class LiteLLMPromptInjectionParams(LiteLLMPydanticObjectBase): 

1056 heuristics_check: bool = False 

1057 vector_db_check: bool = False 

1058 llm_api_check: bool = False 

1059 llm_api_name: str | None = None 

1060 llm_api_system_prompt: str | None = None 

1061 llm_api_fail_call_string: str | None = None 

1062 reject_as_response: bool | None = Field( 

1063 default=False, 

1064 description="Return rejected request error message as a string to the user. Default behaviour is to raise an exception.", 

1065 ) 

1066 

1067 @model_validator(mode="before") 

1068 @classmethod 

1069 def check_llm_api_params(cls, values): 

1070 llm_api_check: Final = values.get("llm_api_check") 

1071 if llm_api_check is True: 

1072 if "llm_api_name" not in values or not values["llm_api_name"]: 

1073 raise ValueError("If llm_api_check is set to True, llm_api_name must be provided") 

1074 if "llm_api_system_prompt" not in values or not values["llm_api_system_prompt"]: 

1075 raise ValueError("If llm_api_check is set to True, llm_api_system_prompt must be provided") 

1076 if "llm_api_fail_call_string" not in values or not values["llm_api_fail_call_string"]: 

1077 raise ValueError("If llm_api_check is set to True, llm_api_fail_call_string must be provided") 

1078 return values 

1079 

1080 

1081######### Request Class Definition ###### 

1082class ProxyChatCompletionRequest(LiteLLMPydanticObjectBase): 

1083 """ 

1084 Pydantic model for chat completion requests that includes both OpenAI standard fields 

1085 and LiteLLM-specific parameters. This replaces the previous TypedDict version. 

1086 """ 

1087 

1088 # Required fields (from ChatCompletionRequest) 

1089 model: str 

1090 messages: list[AllMessageValues] 

1091 

1092 # Standard OpenAI completion parameters (all optional) 

1093 frequency_penalty: float | None = None 

1094 logit_bias: dict[str, float] | None = None 

1095 logprobs: bool | None = None 

1096 top_logprobs: int | None = None 

1097 max_tokens: int | None = None 

1098 n: int | None = None 

1099 presence_penalty: float | None = None 

1100 response_format: dict[str, Any] | None = None 

1101 seed: int | None = None 

1102 service_tier: str | None = None 

1103 stop: str | list[str] | None = None 

1104 stream_options: dict[str, Any] | None = None 

1105 temperature: float | None = None 

1106 top_p: float | None = None 

1107 tools: list[dict[str, Any]] | None = None 

1108 tool_choice: str | dict[str, Any] | None = None 

1109 parallel_tool_calls: bool | None = None 

1110 function_call: str | dict[str, Any] | None = None 

1111 functions: list[dict[str, Any]] | None = None 

1112 user: str | None = None 

1113 stream: bool | None = None 

1114 

1115 # LiteLLM-specific metadata param (from original ChatCompletionRequest) 

1116 metadata: dict[str, Any] | None = None 

1117 

1118 # Optional LiteLLM params 

1119 guardrails: list[str] | None = None 

1120 caching: bool | None = None 

1121 num_retries: int | None = None 

1122 context_window_fallback_dict: dict[str, str] | None = None 

1123 fallbacks: list[str] | None = None 

1124 

1125 

1126class ModelInfoDelete(LiteLLMPydanticObjectBase): 

1127 id: str 

1128 

1129 

1130class ModelInfo(LiteLLMPydanticObjectBase): 

1131 id: str | None 

1132 mode: Literal["embedding", "chat", "completion"] | None 

1133 input_cost_per_token: float | None = 0.0 

1134 output_cost_per_token: float | None = 0.0 

1135 max_tokens: int | None = 2048 # assume 2048 if not set 

1136 

1137 # for azure models we need users to specify the base model, one azure you can call deployments - azure/my-random-model 

1138 # we look up the base model in model_prices_and_context_window.json 

1139 base_model: ( 

1140 Literal[ 

1141 "gpt-4-1106-preview", "gpt-4-32k", "gpt-4", "gpt-3.5-turbo-16k", "gpt-3.5-turbo", "text-embedding-ada-002" 

1142 ] 

1143 | None 

1144 ) 

1145 discoverable: bool | None = None 

1146 

1147 model_config = ConfigDict(protected_namespaces=(), extra="allow") 

1148 

1149 @model_validator(mode="before") 

1150 @classmethod 

1151 def set_model_info(cls, values): 

1152 if values.get("id") is None: 

1153 values.update({"id": str(uuid.uuid4())}) 

1154 if values.get("mode") is None: 

1155 values.update({"mode": None}) 

1156 if values.get("input_cost_per_token") is None: 

1157 values.update({"input_cost_per_token": None}) 

1158 if values.get("output_cost_per_token") is None: 

1159 values.update({"output_cost_per_token": None}) 

1160 if values.get("max_tokens") is None: 

1161 values.update({"max_tokens": None}) 

1162 if values.get("base_model") is None: 

1163 values.update({"base_model": None}) 

1164 return values 

1165 

1166 

1167class ProviderInfo(LiteLLMPydanticObjectBase): 

1168 name: str 

1169 fields: list[ProviderField] 

1170 

1171 

1172class BlockUsers(LiteLLMPydanticObjectBase): 

1173 user_ids: list[str] # required 

1174 

1175 

1176class ModelParams(LiteLLMPydanticObjectBase): 

1177 model_name: str 

1178 litellm_params: dict 

1179 model_info: ModelInfo 

1180 

1181 model_config = ConfigDict(protected_namespaces=()) 

1182 

1183 @model_validator(mode="before") 

1184 @classmethod 

1185 def set_model_info(cls, values): 

1186 if values.get("model_info") is None: 

1187 values.update({"model_info": ModelInfo(id=None, mode="chat", base_model=None)}) 

1188 return values 

1189 

1190 

1191class LiteLLM_ObjectPermissionBase(LiteLLMPydanticObjectBase): 

1192 mcp_servers: list[str] | None = None 

1193 mcp_access_groups: list[str] | None = None 

1194 mcp_tool_permissions: dict[str, list[str]] | None = None 

1195 mcp_toolsets: list[str] | None = None 

1196 blocked_tools: list[str] | None = None 

1197 vector_stores: list[str] | None = None 

1198 agents: list[str] | None = None 

1199 agent_access_groups: list[str] | None = None 

1200 models: list[str] | None = None 

1201 search_tools: list[str] | None = None 

1202 mcp_tool_search_enabled: bool | None = None 

1203 skills: list[str] | None = None 

1204 

1205 

1206from litellm.models.team import BudgetLimitEntry as BudgetLimitEntry # noqa: E402 

1207from litellm.types.object_permission import ( # noqa: E402 

1208 ObjectPermissionDict as ObjectPermissionDict, 

1209) 

1210 

1211 

1212class GenerateRequestBase(LiteLLMPydanticObjectBase): 

1213 """ 

1214 Overlapping schema between key and user generate/update requests 

1215 """ 

1216 

1217 key_alias: str | None = None 

1218 duration: str | None = None 

1219 models: list | None = [] 

1220 spend: float | None = 0 

1221 max_budget: float | None = None 

1222 user_id: str | None = None 

1223 team_id: str | None = None 

1224 agent_id: str | None = None 

1225 max_parallel_requests: int | None = None 

1226 metadata: dict | None = {} 

1227 tpm_limit: int | None = None 

1228 rpm_limit: int | None = None 

1229 

1230 budget_duration: str | None = None 

1231 budget_limits: list[BudgetLimitEntry] | None = None # multiple concurrent budget windows 

1232 allowed_cache_controls: list | None = [] 

1233 config: dict | None = {} 

1234 permissions: dict | None = {} 

1235 model_max_budget: dict | None = {} # {"gpt-4": 5.0, "gpt-3.5-turbo": 5.0}, defaults to {} 

1236 budget_fallbacks: dict[str, list[str]] | None = None 

1237 

1238 model_config = ConfigDict(protected_namespaces=()) 

1239 model_rpm_limit: dict | None = None 

1240 model_tpm_limit: dict | None = None 

1241 mcp_rpm_limit: dict[str, int] | None = None 

1242 tag_rpm_limit: dict[str, int] | None = None 

1243 guardrails: list[str] | None = None 

1244 policies: list[str] | None = None 

1245 prompts: list[str] | None = None 

1246 blocked: bool | None = None 

1247 aliases: dict | None = {} 

1248 object_permission: LiteLLM_ObjectPermissionBase | None = None 

1249 

1250 @field_validator("max_budget", mode="before") 

1251 @classmethod 

1252 def check_max_budget(cls, v): 

1253 if v == "": 1253 ↛ 1254line 1253 didn't jump to line 1254 because the condition on line 1253 was never true

1254 return None 

1255 return v 

1256 

1257 

1258class AllowedVectorStoreIndexItem(LiteLLMPydanticObjectBase): 

1259 index_name: str 

1260 index_permissions: list[Literal["read", "write"]] 

1261 

1262 

1263class KeyRequestBase(GenerateRequestBase): 

1264 key: str | None = None 

1265 tpd_limit: int | None = None 

1266 default_estimated_output_tokens: PositiveInt | None = None 

1267 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None 

1268 budget_id: str | None = None 

1269 end_user_budget_id: str | None = None 

1270 tags: list[str] | None = None 

1271 disable_global_guardrails: bool | None = None 

1272 enable_prompt_caching: bool | None = None 

1273 throttle_on_budget_exceeded: bool | None = None 

1274 enforced_params: list[str] | None = None 

1275 allowed_routes: list | None = [] 

1276 allowed_passthrough_routes: list | None = None 

1277 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None 

1278 rpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput", "dynamic"] | None = ( 

1279 None # raise an error if 'guaranteed_throughput' is set and we're overallocating rpm 

1280 ) 

1281 tpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput", "dynamic"] | None = ( 

1282 None # raise an error if 'guaranteed_throughput' is set and we're overallocating tpm 

1283 ) 

1284 router_settings: UpdateRouterConfig | None = None 

1285 access_group_ids: list[str] | None = None 

1286 

1287 

1288class LiteLLMKeyType(str, enum.Enum): 

1289 """ 

1290 Enum for key types that determine what routes a key can access 

1291 """ 

1292 

1293 LLM_API = "llm_api" # Can call LLM API routes (chat/completions, embeddings, etc.) 

1294 MANAGEMENT = "management" # Can call management routes (user/team/key management) 

1295 READ_ONLY = "read_only" # Can only call info/read routes 

1296 DEFAULT = "default" # Uses default allowed routes 

1297 

1298 

1299class GenerateKeyRequest(KeyRequestBase): 

1300 soft_budget: float | None = None 

1301 send_invite_email: bool | None = None 

1302 key_type: LiteLLMKeyType | None = Field( 

1303 default=LiteLLMKeyType.DEFAULT, 

1304 description="Type of key that determines default allowed routes.", 

1305 ) 

1306 auto_rotate: bool | None = Field(default=False, description="Whether this key should be automatically rotated") 

1307 rotation_interval: str | None = Field( 

1308 default=None, 

1309 description="How often to rotate this key (e.g., '30d', '90d'). Required if auto_rotate=True", 

1310 ) 

1311 organization_id: str | None = None 

1312 project_id: str | None = None 

1313 

1314 @field_validator("team_id", "organization_id", "project_id", mode="before") 

1315 @classmethod 

1316 def treat_cleared_id_as_unset(cls, v: object) -> object: 

1317 if v == "": 

1318 return None 

1319 return v 

1320 

1321 

1322class GenerateKeyResponse(KeyRequestBase): 

1323 key: str 

1324 key_name: str | None = None 

1325 key_type: str | None = None 

1326 expires: datetime | None = None 

1327 user_id: str | None = None 

1328 token_id: str | None = None 

1329 organization_id: str | None = None 

1330 project_id: str | None = None 

1331 litellm_budget_table: Any | None = None 

1332 token: str | None = None 

1333 created_by: str | None = None 

1334 updated_by: str | None = None 

1335 created_at: datetime | None = None 

1336 updated_at: datetime | None = None 

1337 

1338 @model_validator(mode="before") 

1339 @classmethod 

1340 def set_model_info(cls, values): 

1341 if values.get("token") is not None: 

1342 values.update({"key": values.get("token")}) 

1343 dict_fields: Final = [ 

1344 "metadata", 

1345 "aliases", 

1346 "config", 

1347 "permissions", 

1348 "model_max_budget", 

1349 "budget_fallbacks", 

1350 "router_settings", 

1351 "budget_limits", 

1352 ] 

1353 for field in dict_fields: 

1354 value = values.get(field) 

1355 if value is not None and isinstance(value, str): 

1356 try: 

1357 values[field] = json.loads(value) 

1358 except json.JSONDecodeError: 

1359 raise ValueError(f"Field {field} should be a valid dictionary") 

1360 

1361 return values 

1362 

1363 

1364class UpdateKeyRequest(KeyRequestBase): 

1365 # Note: the defaults of all Params here MUST BE NONE 

1366 # else they will get overwritten 

1367 duration: str | None = None 

1368 spend: float | None = None 

1369 soft_budget: float | None = None 

1370 metadata: dict | None = None 

1371 temp_budget_increase: float | None = None 

1372 temp_budget_expiry: datetime | None = None 

1373 auto_rotate: bool | None = None 

1374 rotation_interval: str | None = None 

1375 organization_id: str | None = None 

1376 

1377 project_id: str | None = Field( 

1378 default=None, 

1379 description="Omit to retain the project, or send null to detach. Assigning a different project is not supported.", 

1380 ) 

1381 

1382 @model_validator(mode="before") 

1383 @classmethod 

1384 def drop_blank_team_id(cls, values: object) -> object: 

1385 if isinstance(values, Mapping) and values.get("team_id") == "": 

1386 return MappingProxyType({k: v for k, v in values.items() if k != "team_id"}) 

1387 return values 

1388 

1389 @field_validator("organization_id", mode="before") 

1390 @classmethod 

1391 def treat_cleared_organization_id_as_unset(cls, v: object) -> object: 

1392 if v == "": 

1393 return None 

1394 return v 

1395 

1396 @model_validator(mode="after") 

1397 def validate_temp_budget(self) -> "UpdateKeyRequest": 

1398 if self.temp_budget_increase is not None or self.temp_budget_expiry is not None: 

1399 if self.temp_budget_increase is None or self.temp_budget_expiry is None: 

1400 raise ValueError("temp_budget_increase and temp_budget_expiry must be set together") 

1401 return self 

1402 

1403 @model_validator(mode="after") 

1404 def validate_key_identifier(self) -> "UpdateKeyRequest": 

1405 if self.key is None and self.key_alias is None: 

1406 raise ValueError("either key or key_alias must be provided") 

1407 return self 

1408 

1409 

1410class RegenerateKeyRequest(GenerateKeyRequest): 

1411 # This needs to be different from UpdateKeyRequest, because "key" is optional for this 

1412 key: str | None = None 

1413 new_key: str | None = None 

1414 duration: str | None = None 

1415 spend: float | None = None 

1416 metadata: dict | None = None 

1417 new_master_key: str | None = None 

1418 grace_period: str | None = None # Duration to keep old key valid (e.g. "24h", "2d"); None = immediate revoke 

1419 

1420 

1421class ResetSpendRequest(LiteLLMPydanticObjectBase): 

1422 reset_to: float 

1423 

1424 @field_validator("reset_to", mode="before") 

1425 @classmethod 

1426 def reject_bool_reset_to(cls, v): 

1427 # bool is a subclass of int, so pydantic silently coerces True/False into 

1428 # 1.0/0.0 for a `float` field: a caller who accidentally sends a boolean 

1429 # would otherwise get an unintended spend reset instead of a 422. 

1430 if isinstance(v, bool): 

1431 raise ValueError("reset_to must be a number, not a boolean") # noqa: TRY004 # pydantic needs ValueError 

1432 return v 

1433 

1434 

1435class KeyRequest(LiteLLMPydanticObjectBase): 

1436 keys: list[str] | None = None 

1437 key_aliases: list[str] | None = None 

1438 

1439 @model_validator(mode="before") 

1440 @classmethod 

1441 def validate_at_least_one(cls, values): 

1442 if not values.get("keys") and not values.get("key_aliases"): 

1443 raise ValueError("At least one of 'keys' or 'key_aliases' must be provided.") 

1444 return values 

1445 

1446 

1447from litellm.models.model import ( # noqa: E402 

1448 LiteLLM_ProxyModelTable as LiteLLM_ProxyModelTable, 

1449) 

1450from litellm.models.team import LiteLLM_ModelTable as LiteLLM_ModelTable # noqa: E402 

1451 

1452 

1453# MCP Types 

1454class SpecialMCPServerName(str, enum.Enum): 

1455 all_team_servers = "all-team-mcpservers" 

1456 all_proxy_servers = "all-proxy-mcpservers" 

1457 

1458 

1459class MCPApprovalStatus(str, enum.Enum): 

1460 pending_review = "pending_review" 

1461 active = "active" 

1462 rejected = "rejected" 

1463 # Short-lived row backing the admin OAuth "Authorize & Fetch Token" flow. Never served: the 

1464 # registry loader and every listing exclude it, so it is reachable only by its own server_id. 

1465 draft = "draft" 

1466 

1467 

1468from litellm.models.mcp_server import ( # noqa: E402 

1469 MCPEnvVar as MCPEnvVar, 

1470) 

1471from litellm.models.mcp_server import ( # noqa: E402 

1472 MCPEnvVarScope as MCPEnvVarScope, 

1473) 

1474 

1475 

1476# MCP Proxy Request Types 

1477def _dcr_bridge_auth_type_error(auth_type: object) -> ValueError: 

1478 return ValueError( 

1479 f"dcr_bridge is only supported for auth_type true_passthrough or oauth_delegate (got {auth_type!r}). " 

1480 "The DCR bridge serves gateway-hosted OAuth discovery for the client-forwarded token modes; " 

1481 "interactive oauth2 servers already run the gateway authorization-code flow." 

1482 ) 

1483 

1484 

1485def _per_server_oauth_discovery_error() -> ValueError: 

1486 return ValueError( 

1487 "per_server_oauth_discovery is only supported for auth_type oauth2 with oauth2_flow " 

1488 "authorization_code and without delegate_auth_to_upstream." 

1489 ) 

1490 

1491 

1492def is_per_server_oauth_discovery_eligible( 

1493 auth_type: object, oauth2_flow: object, delegate_auth_to_upstream: object 

1494) -> bool: 

1495 return auth_type == MCPAuth.oauth2 and oauth2_flow == "authorization_code" and not delegate_auth_to_upstream 

1496 

1497 

1498def _reject_unsupported_per_server_oauth_discovery(values: object, require_auth_type: bool) -> None: 

1499 """Partial updates may omit eligibility fields; those are checked against the stored row by the 

1500 update endpoint. Every field the payload does carry must be eligible on its own.""" 

1501 if not isinstance(values, dict) or not values.get("per_server_oauth_discovery"): 

1502 return 

1503 auth_type_ok: Final = values.get("auth_type") == MCPAuth.oauth2 or ( 

1504 not require_auth_type and "auth_type" not in values 

1505 ) 

1506 oauth2_flow_ok: Final = values.get("oauth2_flow") == "authorization_code" or ( 

1507 not require_auth_type and "oauth2_flow" not in values 

1508 ) 

1509 if auth_type_ok and oauth2_flow_ok and not values.get("delegate_auth_to_upstream"): 

1510 return 

1511 raise _per_server_oauth_discovery_error() 

1512 

1513 

1514def _validate_mcp_transport_fields(values: object) -> None: 

1515 if not isinstance(values, dict): 

1516 return 

1517 transport: Final = values.get("transport") 

1518 if transport in (MCPTransport.http, MCPTransport.sse): 

1519 if not values.get("url") and not values.get("spec_path"): 

1520 raise ValueError("url or spec_path is required for HTTP/SSE transport") 

1521 return 

1522 if transport != MCPTransport.stdio: 

1523 return 

1524 if not is_mcp_stdio_enabled(): 1524 ↛ 1526line 1524 didn't jump to line 1526 because the condition on line 1524 was always true

1525 raise ValueError(MCP_STDIO_DISABLED_MESSAGE) 

1526 command: Final = values.get("command") 

1527 if not command: 

1528 raise ValueError("command is required for stdio transport") 

1529 if not values.get("args"): 

1530 raise ValueError("args is required for stdio transport") 

1531 if os.path.basename(str(command)) not in MCP_STDIO_ALLOWED_COMMANDS: 

1532 raise ValueError( 

1533 f"Command '{command}' is not in the allowed commands list " 

1534 f"for stdio transport. Allowed commands: {sorted(MCP_STDIO_ALLOWED_COMMANDS)}" 

1535 ) 

1536 

1537 

1538class NewMCPServerRequest(LiteLLMPydanticObjectBase): 

1539 server_id: str | None = None 

1540 server_name: str | None = None 

1541 alias: str | None = None 

1542 description: str | None = None 

1543 transport: MCPTransportType = MCPTransport.sse 

1544 auth_type: MCPAuthType | None = None 

1545 credentials: MCPCredentials | None = None 

1546 url: str | None = None 

1547 spec_path: str | None = None 

1548 mcp_info: MCPInfo | None = None 

1549 mcp_access_groups: list[str] = Field(default_factory=list) 

1550 allowed_tools: list[str] | None = None 

1551 tool_name_to_display_name: dict[str, str] | None = None 

1552 tool_name_to_description: dict[str, str] | None = None 

1553 extra_headers: list[str] | None = None 

1554 static_headers: dict[str, str] | None = None 

1555 env_vars: list[MCPEnvVar] | None = None 

1556 instructions: str | None = None 

1557 # Stdio-specific fields 

1558 command: str | None = None 

1559 args: list[str] = Field(default_factory=list) 

1560 env: dict[str, str] = Field(default_factory=dict) 

1561 issuer: str | None = None 

1562 authorization_url: str | None = None 

1563 token_url: str | None = None 

1564 registration_url: str | None = None 

1565 oauth2_flow: Literal["client_credentials", "authorization_code"] | None = None 

1566 # Token Exchange (OBO) fields — RFC 8693. These top-level fields are the 

1567 # canonical shape; the same keys inside ``credentials`` are the legacy 

1568 # pre-column REST shape and are lifted into these columns on write (an 

1569 # explicit top-level value wins) and stripped from the stored blob. 

1570 token_exchange_endpoint: str | None = None 

1571 audience: str | None = None 

1572 subject_token_type: str | None = None 

1573 token_exchange_profile: str | None = None 

1574 allow_all_keys: bool = False 

1575 available_on_public_internet: bool = True 

1576 delegate_auth_to_upstream: bool = False 

1577 oauth_passthrough: bool = False 

1578 dcr_bridge: bool | None = None 

1579 per_server_oauth_discovery: bool = False 

1580 is_byok: bool = False 

1581 byok_description: list[str] = Field(default_factory=list) 

1582 byok_api_key_help_url: str | None = None 

1583 source_url: str | None = None 

1584 timeout: float | None = None 

1585 max_concurrent_requests: int | None = None 

1586 # BYOM submission fields — set by the endpoint, not by the caller. 

1587 # Any caller-provided values are silently overridden before persistence. 

1588 approval_status: str | None = Field( 

1589 default=None, 

1590 description="Server-managed: set by the endpoint; caller values are overridden.", 

1591 ) 

1592 submitted_by: str | None = Field( 

1593 default=None, 

1594 description="Server-managed: set by the endpoint; caller values are overridden.", 

1595 ) 

1596 submitted_at: datetime | None = Field( 

1597 default=None, 

1598 description="Server-managed: set by the endpoint; caller values are overridden.", 

1599 ) 

1600 

1601 @model_validator(mode="before") 

1602 @classmethod 

1603 def validate_transport_fields(cls, values): 

1604 _validate_mcp_transport_fields(values) 

1605 return values 

1606 

1607 @model_validator(mode="before") 

1608 @classmethod 

1609 def validate_credentials_requirements(cls, values): 

1610 """Validate credentials when provided. 

1611 

1612 auth_value is optional — users may configure it dynamically 

1613 (e.g. via per-request headers or OAuth2 flows) instead of 

1614 storing a static value at server creation time. 

1615 """ 

1616 return values 

1617 

1618 @model_validator(mode="before") 

1619 @classmethod 

1620 def validate_dcr_bridge_auth_type(cls, values): 

1621 if not isinstance(values, dict) or not values.get("dcr_bridge"): 

1622 return values 

1623 auth_type: Final = values.get("auth_type") 

1624 if auth_type in (MCPAuth.true_passthrough, MCPAuth.oauth_delegate): 1624 ↛ 1625line 1624 didn't jump to line 1625 because the condition on line 1624 was never true

1625 return values 

1626 raise _dcr_bridge_auth_type_error(auth_type) 

1627 

1628 @model_validator(mode="before") 

1629 @classmethod 

1630 def validate_per_server_oauth_discovery_auth_type(cls, values: object) -> object: 

1631 _reject_unsupported_per_server_oauth_discovery(values, require_auth_type=True) 

1632 return values 

1633 

1634 

1635class UpdateMCPServerRequest(LiteLLMPydanticObjectBase): 

1636 server_id: str 

1637 server_name: str | None = None 

1638 alias: str | None = None 

1639 description: str | None = None 

1640 transport: MCPTransportType = MCPTransport.sse 

1641 auth_type: MCPAuthType | None = None 

1642 credentials: MCPCredentials | None = None 

1643 url: str | None = None 

1644 spec_path: str | None = None 

1645 mcp_info: MCPInfo | None = None 

1646 mcp_access_groups: list[str] = Field(default_factory=list) 

1647 allowed_tools: list[str] | None = None 

1648 tool_name_to_display_name: dict[str, str] | None = None 

1649 tool_name_to_description: dict[str, str] | None = None 

1650 extra_headers: list[str] | None = None 

1651 static_headers: dict[str, str] | None = None 

1652 env_vars: list[MCPEnvVar] | None = None 

1653 instructions: str | None = None 

1654 # Stdio-specific fields 

1655 command: str | None = None 

1656 args: list[str] = Field(default_factory=list) 

1657 env: dict[str, str] = Field(default_factory=dict) 

1658 issuer: str | None = None 

1659 authorization_url: str | None = None 

1660 token_url: str | None = None 

1661 registration_url: str | None = None 

1662 oauth2_flow: Literal["client_credentials", "authorization_code"] | None = None 

1663 # Token Exchange (OBO) fields — RFC 8693. These top-level fields are the 

1664 # canonical shape; the same keys inside ``credentials`` are the legacy 

1665 # pre-column REST shape and are lifted into these columns on write (an 

1666 # explicit top-level value wins) and stripped from the stored blob. 

1667 token_exchange_endpoint: str | None = None 

1668 audience: str | None = None 

1669 subject_token_type: str | None = None 

1670 token_exchange_profile: str | None = None 

1671 allow_all_keys: bool = False 

1672 available_on_public_internet: bool = True 

1673 delegate_auth_to_upstream: bool = False 

1674 oauth_passthrough: bool = False 

1675 dcr_bridge: bool | None = None 

1676 per_server_oauth_discovery: bool = False 

1677 is_byok: bool = False 

1678 byok_description: list[str] = Field(default_factory=list) 

1679 byok_api_key_help_url: str | None = None 

1680 source_url: str | None = None 

1681 timeout: float | None = None 

1682 max_concurrent_requests: int | None = None 

1683 

1684 @model_validator(mode="before") 

1685 @classmethod 

1686 def validate_transport_fields(cls, values): 

1687 _validate_mcp_transport_fields(values) 

1688 return values 

1689 

1690 @model_validator(mode="before") 

1691 @classmethod 

1692 def validate_dcr_bridge_auth_type(cls, values): 

1693 """Partial updates omit auth_type; that case is validated against the stored row by the 

1694 update endpoint, which can read the database. This validator covers payloads that carry 

1695 both fields.""" 

1696 if not isinstance(values, dict) or not values.get("dcr_bridge"): 

1697 return values 

1698 if "auth_type" not in values: 

1699 return values 

1700 auth_type: Final = values.get("auth_type") 

1701 if auth_type in (MCPAuth.true_passthrough, MCPAuth.oauth_delegate): 

1702 return values 

1703 raise _dcr_bridge_auth_type_error(auth_type) 

1704 

1705 @model_validator(mode="before") 

1706 @classmethod 

1707 def validate_per_server_oauth_discovery_auth_type(cls, values: object) -> object: 

1708 _reject_unsupported_per_server_oauth_discovery(values, require_auth_type=False) 

1709 return values 

1710 

1711 

1712from litellm.models.mcp_server import ( # noqa: E402 

1713 LiteLLM_MCPServerTable as LiteLLM_MCPServerTable, 

1714) 

1715 

1716 

1717class MakeMCPServersPublicRequest(LiteLLMPydanticObjectBase): 

1718 mcp_server_ids: list[str] 

1719 

1720 

1721class MCPUserCredentialRequest(LiteLLMPydanticObjectBase): 

1722 credential: str 

1723 save: bool = True 

1724 

1725 

1726class MCPUserCredentialResponse(LiteLLMPydanticObjectBase): 

1727 server_id: str 

1728 has_credential: bool 

1729 

1730 

1731class MCPOAuthUserCredentialRequest(LiteLLMPydanticObjectBase): 

1732 """Stores a user's OAuth2 token for an OpenAPI MCP server.""" 

1733 

1734 access_token: str 

1735 refresh_token: str | None = None 

1736 expires_in: int | None = None # seconds until expiry 

1737 scopes: list[str] | None = None 

1738 

1739 

1740class MCPOAuthUserCredentialStatus(LiteLLMPydanticObjectBase): 

1741 """Describes whether the calling user has a stored OAuth credential.""" 

1742 

1743 server_id: str 

1744 has_credential: bool 

1745 expires_at: str | None = None # ISO-8601 

1746 is_expired: bool = False 

1747 connected_at: str | None = None # ISO-8601 

1748 

1749 

1750class MCPUserCredentialListItem(LiteLLMPydanticObjectBase): 

1751 """One entry in the /user-credentials list.""" 

1752 

1753 server_id: str 

1754 server_name: str | None = None 

1755 alias: str | None = None 

1756 credential_type: str # "oauth2" or "byok" 

1757 has_credential: bool 

1758 expires_at: str | None = None # ISO-8601; None means non-expiring 

1759 connected_at: str | None = None # ISO-8601 

1760 

1761 

1762class MCPServerUserCredentialListItem(LiteLLMPydanticObjectBase): 

1763 """One user's stored credential for an MCP server, as an admin sees it. Never carries the secret.""" 

1764 

1765 user_id: str 

1766 credential_type: Literal["oauth2", "byok"] 

1767 expires_at: str | None = None 

1768 connected_at: str | None = None 

1769 updated_at: str 

1770 

1771 

1772class MCPUserEnvVarsRequest(LiteLLMPydanticObjectBase): 

1773 """Payload for storing the calling user's per-user env var values.""" 

1774 

1775 values: dict[str, str] 

1776 

1777 

1778class MCPUserEnvVarSpec(LiteLLMPydanticObjectBase): 

1779 """Describes one per-user env var slot for the calling user. 

1780 

1781 Stored values are write-only: the status only reports whether a value 

1782 ``is_set`` and never echoes the decrypted secret back to the client. 

1783 """ 

1784 

1785 name: str 

1786 description: str | None = None 

1787 is_set: bool = False 

1788 

1789 

1790class MCPUserEnvVarsStatus(LiteLLMPydanticObjectBase): 

1791 """Per-user env var status for a single MCP server.""" 

1792 

1793 server_id: str 

1794 server_name: str | None = None 

1795 alias: str | None = None 

1796 required: list[MCPUserEnvVarSpec] = Field(default_factory=list) 

1797 missing_count: int = 0 

1798 setup_url: str | None = None # frontend URL where the user can fill these in 

1799 

1800 

1801class RejectMCPServerRequest(LiteLLMPydanticObjectBase): 

1802 review_notes: str | None = None 

1803 

1804 

1805class MCPSubmissionsSummary(LiteLLMPydanticObjectBase): 

1806 total: int 

1807 pending_review: int 

1808 active: int 

1809 rejected: int 

1810 items: list["LiteLLM_MCPServerTable"] 

1811 

1812 

1813######## Skills API Types ######## 

1814 

1815 

1816class NewSkillRequest(LiteLLMPydanticObjectBase): 

1817 """Request to create a new skill in LiteLLM database""" 

1818 

1819 display_title: str | None = None 

1820 description: str | None = None 

1821 instructions: str | None = None 

1822 file_content: bytes | None = None # Binary content of skill files (zip) 

1823 file_name: str | None = None # Original filename 

1824 file_type: str | None = None # MIME type (e.g., "application/zip") 

1825 metadata: dict[str, Any] | None = None 

1826 authorization_url: str | None = None 

1827 token_url: str | None = None 

1828 registration_url: str | None = None 

1829 

1830 

1831class UpdateSkillRequest(LiteLLMPydanticObjectBase): 

1832 """Request to update an existing skill""" 

1833 

1834 skill_id: str 

1835 display_title: str | None = None 

1836 description: str | None = None 

1837 instructions: str | None = None 

1838 file_content: bytes | None = None # Binary content of skill files (zip) 

1839 file_name: str | None = None # Original filename 

1840 file_type: str | None = None # MIME type 

1841 metadata: dict[str, Any] | None = None 

1842 

1843 

1844from litellm.models.skills import ( # noqa: E402 

1845 LiteLLM_SkillsTable as LiteLLM_SkillsTable, 

1846) 

1847 

1848 

1849class ListSkillsRequest(LiteLLMPydanticObjectBase): 

1850 """Request to list skills from LiteLLM database""" 

1851 

1852 limit: int | None = 20 

1853 offset: int | None = 0 

1854 

1855 

1856class NewUserRequestTeam(LiteLLMPydanticObjectBase): 

1857 team_id: str 

1858 max_budget_in_team: float | None = None 

1859 user_role: Literal["user", "admin"] = "user" 

1860 

1861 

1862class NewUserRequest(GenerateRequestBase): 

1863 max_budget: float | None = None 

1864 user_email: str | None = None 

1865 user_alias: str | None = None 

1866 user_role: ( 

1867 Literal[ 

1868 LitellmUserRoles.PROXY_ADMIN, 

1869 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, 

1870 LitellmUserRoles.INTERNAL_USER, 

1871 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

1872 ] 

1873 | None 

1874 ) = None 

1875 teams: list[str] | list[NewUserRequestTeam] | None = None 

1876 auto_create_key: bool = True # flag used for returning a key as part of the /user/new response 

1877 send_invite_email: bool | None = None 

1878 sso_user_id: str | None = None 

1879 organizations: list[str] | None = None 

1880 password: str | None = None 

1881 

1882 @field_validator("password") 

1883 @classmethod 

1884 def password_not_supported(cls, value: str | None) -> str | None: 

1885 if value is not None: 

1886 raise ValueError( 

1887 "password cannot be set via /user/new. Users set their own password through an " 

1888 "invitation link (POST /invitation/new)." 

1889 ) 

1890 return value 

1891 

1892 

1893class NewUserResponse(GenerateKeyResponse): 

1894 max_budget: float | None = None 

1895 user_email: str | None = None 

1896 user_role: ( 

1897 Literal[ 

1898 LitellmUserRoles.PROXY_ADMIN, 

1899 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, 

1900 LitellmUserRoles.INTERNAL_USER, 

1901 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

1902 ] 

1903 | None 

1904 ) = None 

1905 teams: list | None = None 

1906 user_alias: str | None = None 

1907 model_max_budget: dict | None = None 

1908 created_at: datetime | None = None 

1909 updated_at: datetime | None = None 

1910 

1911 

1912class UpdateUserRequestNoUserIDorEmail(GenerateRequestBase): # shared with BulkUpdateUserRequest 

1913 # repr=False keeps the plaintext out of management-endpoint alerts, which str() the request model 

1914 password: str | None = Field(default=None, repr=False) 

1915 spend: float | None = None 

1916 metadata: dict | None = None 

1917 user_alias: str | None = None 

1918 user_role: ( 

1919 Literal[ 

1920 LitellmUserRoles.PROXY_ADMIN, 

1921 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, 

1922 LitellmUserRoles.INTERNAL_USER, 

1923 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

1924 ] 

1925 | None 

1926 ) = None 

1927 max_budget: float | None = None 

1928 

1929 

1930class UpdateUserRequest(UpdateUserRequestNoUserIDorEmail): 

1931 # Note: the defaults of all Params here MUST BE NONE 

1932 # else they will get overwritten 

1933 user_id: str | None = None 

1934 user_email: str | None = None 

1935 

1936 @model_validator(mode="before") 

1937 @classmethod 

1938 def check_user_info(cls, values): 

1939 if values.get("user_id") is None and values.get("user_email") is None: 

1940 raise ValueError("Either user id or user email must be provided") 

1941 return values 

1942 

1943 

1944class ChangePasswordRequest(LiteLLMPydanticObjectBase): 

1945 current_password: str = Field(repr=False) 

1946 new_password: str = Field(repr=False) 

1947 

1948 

1949class ChangePasswordResponse(LiteLLMPydanticObjectBase): 

1950 user_id: str 

1951 message: str 

1952 

1953 

1954class SessionLogoutResponse(LiteLLMPydanticObjectBase): 

1955 message: str 

1956 

1957 

1958class DeleteUserRequest(LiteLLMPydanticObjectBase): 

1959 user_ids: list[str] # required 

1960 

1961 

1962AllowedModelRegion = Literal["eu", "us"] 

1963 

1964 

1965class BudgetNewRequest(LiteLLMPydanticObjectBase): 

1966 budget_id: str | None = Field(default=None, description="The unique budget id.") 

1967 max_budget: float | None = Field( 

1968 default=None, 

1969 description="Requests will fail if this budget (in USD) is exceeded.", 

1970 ) 

1971 soft_budget: float | None = Field( 

1972 default=None, 

1973 description="Requests will NOT fail if this is exceeded. Will fire alerting though.", 

1974 ) 

1975 max_parallel_requests: int | None = Field( 

1976 default=None, description="Max concurrent requests allowed for this budget id." 

1977 ) 

1978 tpm_limit: int | None = Field(default=None, description="Max tokens per minute, allowed for this budget id.") 

1979 rpm_limit: int | None = Field(default=None, description="Max requests per minute, allowed for this budget id.") 

1980 tpd_limit: int | None = Field( 

1981 default=None, description="Max tokens per day, charged by batch submissions, allowed for this budget id." 

1982 ) 

1983 budget_duration: str | None = Field( 

1984 default=None, 

1985 description="Max duration budget should be set for (e.g. '1hr', '1d', '28d')", 

1986 ) 

1987 model_max_budget: GenericBudgetConfigType | None = Field( 

1988 default=None, 

1989 description="Max budget for each model (e.g. {'gpt-4o': {'max_budget': '0.0000001', 'budget_duration': '1d', 'tpm_limit': 1000, 'rpm_limit': 1000}})", 

1990 ) 

1991 budget_reset_at: datetime | None = Field( 

1992 default=None, 

1993 description="Datetime when the budget is reset", 

1994 ) 

1995 

1996 

1997class BudgetRequest(LiteLLMPydanticObjectBase): 

1998 budgets: list[str] 

1999 

2000 

2001class BudgetDeleteRequest(LiteLLMPydanticObjectBase): 

2002 id: str 

2003 

2004 

2005class CustomerBase(LiteLLMPydanticObjectBase): 

2006 user_id: str 

2007 alias: str | None = None 

2008 spend: float = 0.0 

2009 allowed_model_region: AllowedModelRegion | None = None 

2010 default_model: str | None = None 

2011 budget_id: str | None = None 

2012 litellm_budget_table: BudgetNewRequest | None = None 

2013 blocked: bool = False 

2014 

2015 

2016class NewCustomerRequest(BudgetNewRequest): 

2017 """ 

2018 Create a new customer, allocate a budget to them 

2019 """ 

2020 

2021 user_id: str 

2022 alias: str | None = None # human-friendly alias 

2023 blocked: bool = False # allow/disallow requests for this end-user 

2024 budget_id: str | None = None # give either a budget_id or max_budget 

2025 spend: float | None = None 

2026 allowed_model_region: AllowedModelRegion | None = ( 

2027 None # require all user requests to use models in this specific region 

2028 ) 

2029 default_model: str | None = None # if no equivalent model in allowed region - default all requests to this model 

2030 object_permission: LiteLLM_ObjectPermissionBase | None = None 

2031 

2032 @model_validator(mode="before") 

2033 @classmethod 

2034 def check_user_info(cls, values): 

2035 if values.get("max_budget") is not None and values.get("budget_id") is not None: 

2036 raise ValueError("Set either 'max_budget' or 'budget_id', not both.") 

2037 

2038 return values 

2039 

2040 

2041class UpdateCustomerRequest(LiteLLMPydanticObjectBase): 

2042 """ 

2043 Update a Customer, use this to update customer budgets etc 

2044 

2045 """ 

2046 

2047 user_id: str 

2048 alias: str | None = None # human-friendly alias 

2049 blocked: bool = False # allow/disallow requests for this end-user 

2050 max_budget: float | None = None 

2051 budget_id: str | None = None # give either a budget_id or max_budget 

2052 allowed_model_region: AllowedModelRegion | None = ( 

2053 None # require all user requests to use models in this specific region 

2054 ) 

2055 default_model: str | None = None # if no equivalent model in allowed region - default all requests to this model 

2056 object_permission: LiteLLM_ObjectPermissionBase | None = None 

2057 

2058 

2059class DeleteCustomerRequest(LiteLLMPydanticObjectBase): 

2060 """ 

2061 Delete multiple Customers 

2062 """ 

2063 

2064 user_ids: list[str] 

2065 

2066 

2067from litellm.models.team import Member as Member # noqa: E402 

2068from litellm.models.team import MemberBase as MemberBase # noqa: E402 

2069 

2070 

2071class OrgMember(MemberBase): 

2072 role: Literal[ 

2073 LitellmUserRoles.ORG_ADMIN, 

2074 LitellmUserRoles.INTERNAL_USER, 

2075 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

2076 ] 

2077 

2078 

2079from litellm.models.team import TeamBase as TeamBase # noqa: E402 

2080 

2081RouterSettingsDict = Annotated[ 

2082 dict[str, object], 

2083 BeforeValidator(validate_router_settings_dict, json_schema_input_type=UpdateRouterConfig), 

2084] 

2085 

2086 

2087class NewTeamRequest(TeamBase): 

2088 router_settings: RouterSettingsDict | None = None 

2089 model_aliases: dict | None = None 

2090 model_max_budget: GenericBudgetConfigType | None = Field( 

2091 default=None, 

2092 description=( 

2093 "Max budget per model for every key on the team, overridable per key " 

2094 "(e.g. {'gpt-4o': {'max_budget': 10, 'budget_duration': '1d'}})" 

2095 ), 

2096 ) 

2097 tags: list | None = None 

2098 guardrails: list[str] | None = None 

2099 policies: list[str] | None = None 

2100 prompts: list[str] | None = None 

2101 object_permission: LiteLLM_ObjectPermissionBase | None = None 

2102 allowed_passthrough_routes: list | None = None 

2103 disable_global_guardrails: bool | None = None 

2104 secret_manager_settings: dict | None = None 

2105 model_rpm_limit: dict[str, int] | None = None 

2106 rpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput"] | None = ( 

2107 None # raise an error if 'guaranteed_throughput' is set and we're overallocating rpm 

2108 ) 

2109 tpm_limit_type: Literal["guaranteed_throughput", "best_effort_throughput"] | None = ( 

2110 None # raise an error if 'guaranteed_throughput' is set and we're overallocating tpm 

2111 ) 

2112 

2113 model_tpm_limit: dict[str, int] | None = None 

2114 default_estimated_output_tokens: PositiveInt | None = None 

2115 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None 

2116 mcp_rpm_limit: dict[str, int] | None = None 

2117 team_member_budget: float | None = None # allow user to set a budget for all team members 

2118 team_member_rpm_limit: int | None = None # allow user to set RPM limit for all team members 

2119 team_member_tpm_limit: int | None = None # allow user to set TPM limit for all team members 

2120 team_member_key_duration: str | None = None # e.g. "1d", "1w", "1m" 

2121 team_member_budget_duration: str | None = None # e.g. "30d", "1mo" 

2122 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None 

2123 enforced_batch_output_expires_after: dict | None = None 

2124 enforced_file_expires_after: dict | None = None 

2125 

2126 model_config = ConfigDict(protected_namespaces=()) 

2127 

2128 @field_validator("team_id", mode="before") 

2129 @classmethod 

2130 def treat_blank_team_id_as_unset(cls, v: object) -> object: 

2131 if isinstance(v, str) and not v.strip(): 2131 ↛ 2132line 2131 didn't jump to line 2132 because the condition on line 2131 was never true

2132 return None 

2133 return v 

2134 

2135 

2136class GlobalEndUsersSpend(LiteLLMPydanticObjectBase): 

2137 api_key: str | None = None 

2138 startTime: datetime | None = None 

2139 endTime: datetime | None = None 

2140 

2141 

2142class UpdateTeamRequest(LiteLLMPydanticObjectBase): 

2143 """ 

2144 UpdateTeamRequest, used by /team/update when you need to update a team 

2145 

2146 team_id: str 

2147 team_alias: Optional[str] = None 

2148 organization_id: Optional[str] = None 

2149 metadata: Optional[dict] = None 

2150 tpm_limit: Optional[int] = None 

2151 rpm_limit: Optional[int] = None 

2152 max_budget: Optional[float] = None 

2153 models: Optional[list] = None 

2154 blocked: Optional[bool] = None 

2155 budget_duration: Optional[str] = None 

2156 guardrails: Optional[List[str]] = None 

2157 policies: Optional[List[str]] = None 

2158 """ 

2159 

2160 team_id: str # required 

2161 team_alias: str | None = None 

2162 organization_id: str | None = None 

2163 metadata: dict | None = None 

2164 tpm_limit: int | None = None 

2165 rpm_limit: int | None = None 

2166 tpd_limit: int | None = None 

2167 max_budget: float | None = None 

2168 soft_budget: float | None = None 

2169 models: list | None = None 

2170 blocked: bool | None = None 

2171 budget_duration: str | None = None 

2172 tags: list | None = None 

2173 model_aliases: dict | None = None 

2174 guardrails: list[str] | None = None 

2175 policies: list[str] | None = None 

2176 object_permission: LiteLLM_ObjectPermissionBase | None = None 

2177 disable_global_guardrails: bool | None = None 

2178 team_member_budget: float | None = None 

2179 team_member_budget_duration: str | None = None 

2180 team_member_rpm_limit: int | None = None 

2181 team_member_tpm_limit: int | None = None 

2182 team_member_key_duration: str | None = None 

2183 allowed_passthrough_routes: list | None = None 

2184 secret_manager_settings: dict | None = None 

2185 prompts: list[str] | None = None 

2186 model_rpm_limit: dict[str, int] | None = None 

2187 model_tpm_limit: dict[str, int] | None = None 

2188 default_estimated_output_tokens: PositiveInt | None = None 

2189 default_estimated_output_tokens_per_model: Mapping[str, PositiveInt] | None = None 

2190 mcp_rpm_limit: dict[str, int] | None = None 

2191 allowed_vector_store_indexes: list[AllowedVectorStoreIndexItem] | None = None 

2192 enforced_batch_output_expires_after: dict | None = None 

2193 enforced_file_expires_after: dict | None = None 

2194 router_settings: RouterSettingsDict | None = None 

2195 access_group_ids: list[str] | None = None 

2196 budget_limits: list[BudgetLimitEntry] | None = None # multiple concurrent budget windows 

2197 default_team_member_models: list[str] | None = None # default allowed_models seeded onto new team members 

2198 model_max_budget: GenericBudgetConfigType | None = Field( 

2199 default=None, 

2200 description=( 

2201 "Max budget per model for every key on the team, overridable per key " 

2202 "(e.g. {'gpt-4o': {'max_budget': 10, 'budget_duration': '1d'}})" 

2203 ), 

2204 ) 

2205 

2206 

2207class PatchTeamRequest(UpdateTeamRequest): 

2208 """ 

2209 Body of PATCH /team/{team_id}. 

2210 

2211 Identical to UpdateTeamRequest except team_id is optional, because PATCH takes it 

2212 from the path. A team_id in the body is still accepted when it matches the path. 

2213 """ 

2214 

2215 team_id: str | None = None 

2216 

2217 

2218class ResetTeamBudgetRequest(LiteLLMPydanticObjectBase): 

2219 """ 

2220 internal type used to reset the budget on a team 

2221 used by reset_budget() 

2222 

2223 team_id: str 

2224 spend: float 

2225 budget_reset_at: datetime 

2226 """ 

2227 

2228 team_id: str 

2229 spend: float 

2230 budget_reset_at: datetime 

2231 updated_at: datetime 

2232 

2233 

2234class DeleteTeamRequest(LiteLLMPydanticObjectBase): 

2235 team_ids: list[str] # required 

2236 

2237 

2238class BlockTeamRequest(LiteLLMPydanticObjectBase): 

2239 team_id: str # required 

2240 

2241 

2242class BlockKeyRequest(LiteLLMPydanticObjectBase): 

2243 key: str # required 

2244 

2245 

2246class BlockModelRequest(LiteLLMPydanticObjectBase): 

2247 model_id: str # required 

2248 

2249 

2250class AddTeamCallback(LiteLLMPydanticObjectBase): 

2251 callback_name: str 

2252 callback_type: Literal["success", "failure", "success_and_failure"] | None = "success_and_failure" 

2253 callback_vars: dict[str, str] 

2254 

2255 @model_validator(mode="before") 

2256 @classmethod 

2257 def validate_callback_vars(cls, values): 

2258 callback_vars: Final = values.get("callback_vars", {}) 

2259 valid_keys: Final = set(StandardCallbackDynamicParams.__annotations__.keys()) 

2260 for key, value in callback_vars.items(): 

2261 if key not in valid_keys: 2261 ↛ 2263line 2261 didn't jump to line 2263 because the condition on line 2261 was always true

2262 raise ValueError(f"Invalid callback variable: {key}. Must be one of {valid_keys}") 

2263 callback_vars[key] = str(value) 

2264 validate_no_callback_env_reference(key, callback_vars[key], source="key/team callback metadata") 

2265 if key == "langfuse_environment": 

2266 validate_langfuse_environment_value(callback_vars[key]) 

2267 if key == "langfuse_span_scope": 

2268 validate_langfuse_span_scope_value(callback_vars[key]) 

2269 return values 

2270 

2271 

2272class TeamCallbackDeleteResponseData(LiteLLMPydanticObjectBase): 

2273 team_id: str 

2274 success_callbacks: tuple[str, ...] 

2275 failure_callbacks: tuple[str, ...] 

2276 

2277 

2278class TeamCallbackDeleteResponse(LiteLLMPydanticObjectBase): 

2279 status: Literal["success"] 

2280 message: str 

2281 data: TeamCallbackDeleteResponseData 

2282 

2283 

2284class TeamCallbackMetadata(LiteLLMPydanticObjectBase): 

2285 success_callback: list[str] | None = [] 

2286 failure_callback: list[str] | None = [] 

2287 callbacks: list[str] | None = [] 

2288 # for now - only supported for langfuse 

2289 callback_vars: dict[str, str] | None = {} 

2290 

2291 @model_validator(mode="before") 

2292 @classmethod 

2293 def validate_callback_vars(cls, values): 

2294 success_callback: Final = values.get("success_callback", []) 

2295 if success_callback is None: 2295 ↛ 2296line 2295 didn't jump to line 2296 because the condition on line 2295 was never true

2296 values.pop("success_callback", None) 

2297 failure_callback: Final = values.get("failure_callback", []) 

2298 if failure_callback is None: 2298 ↛ 2299line 2298 didn't jump to line 2299 because the condition on line 2298 was never true

2299 values.pop("failure_callback", None) 

2300 callbacks: Final = values.get("callbacks", []) 

2301 if callbacks is None: 2301 ↛ 2302line 2301 didn't jump to line 2302 because the condition on line 2301 was never true

2302 values.pop("callbacks", None) 

2303 

2304 callback_vars: Final = values.get("callback_vars", {}) 

2305 if callback_vars is None: 2305 ↛ 2306line 2305 didn't jump to line 2306 because the condition on line 2305 was never true

2306 values.pop("callback_vars", None) 

2307 if all(val is None for val in values.values()): 

2308 return { 

2309 "success_callback": [], 

2310 "failure_callback": [], 

2311 "callbacks": [], 

2312 "callback_vars": {}, 

2313 } 

2314 valid_keys: Final = set(StandardCallbackDynamicParams.__annotations__.keys()) 

2315 if callback_vars is not None: 2315 ↛ 2319line 2315 didn't jump to line 2319 because the condition on line 2315 was always true

2316 for key in callback_vars: 2316 ↛ 2317line 2316 didn't jump to line 2317 because the loop on line 2316 never started

2317 if key not in valid_keys: 

2318 raise ValueError(f"Invalid callback variable: {key}. Must be one of {valid_keys}") 

2319 return values 

2320 

2321 

2322from litellm.models.object_permission import ( # noqa: E402 

2323 LiteLLM_ObjectPermissionTable as LiteLLM_ObjectPermissionTable, 

2324) 

2325from litellm.models.team import ( # noqa: E402 

2326 LiteLLM_DeletedTeamTable as LiteLLM_DeletedTeamTable, 

2327) 

2328from litellm.models.team import LiteLLM_TeamTable as LiteLLM_TeamTable # noqa: E402 

2329from litellm.models.team import ( # noqa: E402 

2330 LiteLLM_TeamTableCachedObj as LiteLLM_TeamTableCachedObj, 

2331) 

2332 

2333 

2334class TeamRequest(LiteLLMPydanticObjectBase): 

2335 teams: list[str] 

2336 

2337 

2338from litellm.models.budget import ( # noqa: E402 

2339 LiteLLM_BudgetTable as LiteLLM_BudgetTable, 

2340) 

2341from litellm.models.budget import ( # noqa: E402 

2342 LiteLLM_BudgetTableFull as LiteLLM_BudgetTableFull, 

2343) 

2344from litellm.models.budget import ( # noqa: E402 

2345 LiteLLM_TeamMemberTable as LiteLLM_TeamMemberTable, 

2346) 

2347 

2348 

2349class NewOrganizationRequest(LiteLLM_BudgetTable): 

2350 organization_id: str | None = None 

2351 organization_alias: str 

2352 models: list = [] 

2353 budget_id: str | None = None 

2354 metadata: dict | None = None 

2355 model_rpm_limit: dict[str, int] | None = None 

2356 model_tpm_limit: dict[str, int] | None = None 

2357 

2358 ######################################################### 

2359 # Object Permission - MCP, Vector Stores etc. 

2360 ######################################################### 

2361 object_permission: LiteLLM_ObjectPermissionBase | None = None 

2362 

2363 

2364class OrganizationRequest(LiteLLMPydanticObjectBase): 

2365 organizations: list[str] 

2366 

2367 

2368class DeleteOrganizationRequest(LiteLLMPydanticObjectBase): 

2369 organization_ids: list[str] # required 

2370 

2371 

2372class TeamDefaultSettings(LiteLLMPydanticObjectBase): 

2373 team_id: str 

2374 

2375 model_config = ConfigDict( 

2376 extra="allow" 

2377 ) # allow params not defined here, these fall in litellm.completion(**kwargs) 

2378 

2379 

2380class DynamoDBArgs(LiteLLMPydanticObjectBase): 

2381 billing_mode: Literal["PROVISIONED_THROUGHPUT", "PAY_PER_REQUEST"] 

2382 read_capacity_units: int | None = None 

2383 write_capacity_units: int | None = None 

2384 ssl_verify: bool | None = None 

2385 region_name: str 

2386 user_table_name: str = "LiteLLM_UserTable" 

2387 key_table_name: str = "LiteLLM_VerificationToken" 

2388 config_table_name: str = "LiteLLM_Config" 

2389 spend_table_name: str = "LiteLLM_SpendLogs" 

2390 aws_role_name: str | None = None 

2391 aws_session_name: str | None = None 

2392 aws_web_identity_token: str | None = None 

2393 aws_provider_id: str | None = None 

2394 aws_policy_arns: list[str] | None = None 

2395 aws_policy: str | None = None 

2396 aws_duration_seconds: int | None = None 

2397 assume_role_aws_role_name: str | None = None 

2398 assume_role_aws_session_name: str | None = None 

2399 

2400 

2401class PassThroughGuardrailSettings(LiteLLMPydanticObjectBase): 

2402 """ 

2403 Settings for a specific guardrail on a passthrough endpoint. 

2404 

2405 Allows field-level targeting for guardrail execution. 

2406 """ 

2407 

2408 request_fields: list[str] | None = Field( 

2409 default=None, 

2410 description="JSONPath expressions for input field targeting (pre_call). Examples: 'query', 'documents[*].text', 'messages[*].content'. If not specified, guardrail runs on entire request payload.", 

2411 ) 

2412 response_fields: list[str] | None = Field( 

2413 default=None, 

2414 description="JSONPath expressions for output field targeting (post_call). Examples: 'results[*].text', 'output'. If not specified, guardrail runs on entire response payload.", 

2415 ) 

2416 

2417 

2418# Type alias for the guardrails dict: guardrail_name -> settings (or None for defaults) 

2419PassThroughGuardrailsConfig = dict[str, PassThroughGuardrailSettings | None] 

2420 

2421 

2422class PassThroughGenericEndpoint(LiteLLMPydanticObjectBase): 

2423 id: str | None = Field( 

2424 default=None, 

2425 description="Optional unique identifier for the pass-through endpoint. If not provided, endpoints will be identified by path for backwards compatibility.", 

2426 ) 

2427 path: str = Field(description="The route to be added to the LiteLLM Proxy Server.") 

2428 target: str = Field(description="The URL to which requests for this path should be forwarded.") 

2429 headers: dict = Field( 

2430 default={}, 

2431 description="Key-value pairs of headers to be forwarded with the request. You can set any key value pair here and it will be forwarded to your target endpoint", 

2432 ) 

2433 default_query_params: dict = Field( 

2434 default={}, 

2435 description="Key-value pairs of default query parameters to be sent with every request to this endpoint. These can be overridden by client-provided query parameters. For example: {'key': 'default_value', 'api_version': '2023-01'}", 

2436 ) 

2437 include_subpath: bool = Field( 

2438 default=False, 

2439 description="If True, requests to subpaths of the path will be forwarded to the target endpoint. For example, if the path is /bria and include_subpath is True, requests to /bria/v1/text-to-image/base/2.3 will be forwarded to the target endpoint.", 

2440 ) 

2441 cost_per_request: float = Field( 

2442 default=0.0, 

2443 description="The USD cost per request to the target endpoint. This is used to calculate the cost of the request to the target endpoint.", 

2444 ) 

2445 timeout: float | None = Field( 

2446 default=None, 

2447 description="Upstream request timeout in seconds for this pass-through endpoint. If unset, uses general_settings.pass_through_request_timeout (default 600).", 

2448 ) 

2449 auth: bool = Field( 

2450 default=True, 

2451 description="Whether authentication is required for the pass-through endpoint. Defaults to True so a pass-through silently created without an explicit value still requires a valid LiteLLM API key — set to False only if the endpoint is meant to be a public forwarder (e.g. an unauthenticated webhook target).", 

2452 ) 

2453 guardrails: PassThroughGuardrailsConfig | None = Field( 

2454 default=None, 

2455 description="Guardrails configuration for this passthrough endpoint. Dict keys are guardrail names, values are optional settings for field targeting. When set, all org/team/key level guardrails will also execute. Defaults to None (no guardrails execute).", 

2456 ) 

2457 is_from_config: bool = Field( 

2458 default=False, 

2459 description="True if this endpoint is defined in the config file, False if from DB. Config-defined endpoints cannot be edited via the UI.", 

2460 ) 

2461 methods: list[str] | None = Field( 

2462 default=None, 

2463 description="List of HTTP methods this endpoint handles (e.g., ['GET', 'POST']). If None or empty, all methods (GET, POST, PUT, DELETE, PATCH) are supported for backward compatibility. This allows the same path to have different targets for different HTTP methods.", 

2464 ) 

2465 

2466 

2467class PassThroughEndpointResponse(LiteLLMPydanticObjectBase): 

2468 endpoints: list[PassThroughGenericEndpoint] 

2469 

2470 

2471class ConfigFieldUpdate(LiteLLMPydanticObjectBase): 

2472 field_name: str 

2473 field_value: Any 

2474 config_type: Literal["general_settings"] 

2475 

2476 

2477class ConfigFieldDelete(LiteLLMPydanticObjectBase): 

2478 config_type: Literal["general_settings"] 

2479 field_name: str 

2480 

2481 

2482class CallbackDelete(LiteLLMPydanticObjectBase): 

2483 callback_name: str 

2484 

2485 

2486class FieldDetail(BaseModel): 

2487 field_name: str 

2488 field_type: str 

2489 field_description: str 

2490 field_default_value: Any = None 

2491 stored_in_db: bool | None 

2492 

2493 

2494class ConfigList(LiteLLMPydanticObjectBase): 

2495 field_name: str 

2496 field_type: str 

2497 field_description: str 

2498 field_value: Any 

2499 stored_in_db: bool | None 

2500 field_default_value: Any 

2501 premium_field: bool = False 

2502 nested_fields: list[FieldDetail] | None = None # For nested dictionary or Pydantic fields 

2503 field_options: list[str] | None = None # Allowed values, for field_type == "Select" 

2504 field_tab: str | None = None # Admin UI sub-tab this field renders under; None groups it with the rest 

2505 source: Literal["config", "db", "env", "default", "unset"] = "unset" 

2506 editable: bool = True 

2507 

2508 

2509class UserHeaderMapping(LiteLLMPydanticObjectBase): 

2510 """ 

2511 Map an incoming HTTP header to a LiteLLM user role. 

2512 """ 

2513 

2514 header_name: str 

2515 litellm_user_role: Literal[ 

2516 LitellmUserRoles.INTERNAL_USER, 

2517 LitellmUserRoles.CUSTOMER, 

2518 ] 

2519 

2520 model_config = { 

2521 "extra": "forbid", 

2522 } 

2523 

2524 

2525UserMCPManagementMode = Literal["restricted", "view_all"] 

2526 

2527 

2528class PluginConfig(LiteLLMPydanticObjectBase): 

2529 """A single external service registered as an embeddable UI plugin.""" 

2530 

2531 name: str = Field(description="unique plugin identifier (kebab-case)") 

2532 display_name: str | None = Field(None, description="human-readable label shown in the UI view switcher") 

2533 url: str = Field(description="base URL of the plugin service") 

2534 plugin_key: str | None = Field( 

2535 None, 

2536 description="plugin's own credential, injected as Bearer auth only on /plugin-proxy/<name>/* reverse-proxy calls", 

2537 ) 

2538 

2539 

2540class CoordinationRedisNode(LiteLLMPydanticObjectBase): 

2541 """A single startup node of a cluster-mode Redis used for proxy coordination.""" 

2542 

2543 host: str = Field(description="hostname of the cluster node") 

2544 port: int = Field(description="port of the cluster node") 

2545 

2546 

2547class CoordinationRedisParams(LiteLLMPydanticObjectBase): 

2548 """ 

2549 Connection params for the proxy's coordination Redis (cross-pod tpm/rpm rate 

2550 limits, spend tracking, pod lock manager, shared health checks), configured 

2551 independently of the response-cache backend in `litellm_settings.cache_params`. 

2552 """ 

2553 

2554 model_config = ConfigDict(extra="allow", protected_namespaces=()) 

2555 

2556 host: str | None = Field(None, description="Redis hostname") 

2557 port: int | None = Field(None, description="Redis port") 

2558 password: str | None = Field(None, description="Redis password") 

2559 username: str | None = Field(None, description="Redis username") 

2560 url: str | None = Field(None, description="full Redis connection url, e.g. redis://:pass@host:6379") 

2561 ssl: bool | None = Field(None, description="connect over TLS") 

2562 startup_nodes: list[CoordinationRedisNode] | None = Field( 

2563 None, description="cluster-mode startup nodes; when set a cluster client is used" 

2564 ) 

2565 sentinel_nodes: list[list[str | int]] | None = Field( 

2566 None, description="sentinel [host, port] pairs; when set a sentinel-managed client is used" 

2567 ) 

2568 sentinel_password: str | None = Field(None, description="password for the sentinel nodes") 

2569 service_name: str | None = Field(None, description="sentinel service name") 

2570 aws_iam_auth: bool | str | None = Field(None, description="enable AWS ElastiCache IAM authentication") 

2571 aws_iam_user_name: str | None = Field(None, description="AWS ElastiCache IAM user name") 

2572 aws_iam_cache_name: str | None = Field(None, description="AWS ElastiCache cache name") 

2573 aws_iam_region: str | None = Field(None, description="AWS region for ElastiCache IAM authentication") 

2574 aws_iam_serverless: bool | str | None = Field( 

2575 None, description="the ElastiCache cache is serverless rather than a self-designed cluster" 

2576 ) 

2577 

2578 def has_connection_target(self) -> bool: 

2579 return any(value is not None for value in (self.host, self.url, self.startup_nodes, self.sentinel_nodes)) 

2580 

2581 

2582class ScheduledJobStaggerSettings(LiteLLMPydanticObjectBase): 

2583 """ 

2584 Spreads the proxy's scheduled background jobs across a window instead of firing them 

2585 all on one instant, on every replica, forever. 

2586 """ 

2587 

2588 model_config = ConfigDict(frozen=True, extra="forbid", protected_namespaces=()) 

2589 

2590 enabled: bool = Field(default=True, description="apply deterministic phase offsets to scheduled background jobs") 

2591 window_seconds: int = Field( 

2592 default=DEFAULT_STAGGER_WINDOW_SECONDS, 

2593 ge=0, 

2594 description=( 

2595 "width of the window jobs are spread over. An interval job is never offset by " 

2596 "more than one of its own periods, so it is not delayed past the wait it already has" 

2597 ), 

2598 ) 

2599 identity: str | None = Field( 

2600 default=None, 

2601 description=( 

2602 "replaces the POD_NAME/HOSTNAME-derived component of the offset hash. Set this " 

2603 "when replicas share a hostname and would otherwise land on the same offset" 

2604 ), 

2605 ) 

2606 offsets: Mapping[str, int] = Field( 

2607 default_factory=dict, 

2608 description=( 

2609 "explicit offset in seconds per scheduler job id, overriding the derived value. " 

2610 "0 pins a job to its unshifted schedule" 

2611 ), 

2612 ) 

2613 

2614 

2615class ConfigGeneralSettings(LiteLLMPydanticObjectBase): 

2616 """ 

2617 Documents all the fields supported by `general_settings` in config.yaml 

2618 """ 

2619 

2620 completion_model: str | None = Field(None, description="proxy level default model for all chat completion calls") 

2621 max_in_flight_requests_per_worker: int | None = Field( 

2622 None, gt=0, description="maximum concurrent requests handled by each worker" 

2623 ) 

2624 max_queued_requests_per_worker: int | None = Field( 

2625 None, ge=0, description="maximum requests waiting for a worker slot" 

2626 ) 

2627 admission_queue_timeout_seconds: float = Field( 

2628 1.0, gt=0, description="maximum time a request waits for a worker slot" 

2629 ) 

2630 plugins: list[PluginConfig] | None = Field( 

2631 None, description="external services registered as embeddable UI plugins" 

2632 ) 

2633 key_management_system: KeyManagementSystem | None = Field( 

2634 None, description="key manager to load keys from / decrypt keys with" 

2635 ) 

2636 use_google_kms: bool | None = Field(None, description="decrypt keys with google kms") 

2637 use_azure_key_vault: bool | None = Field(None, description="load keys from azure key vault") 

2638 master_key: str | None = Field(None, description="require a key for all calls to proxy") 

2639 dangerously_permit_weak_or_unset_master_key: bool | None = Field( 

2640 None, 

2641 description="local development only: start even when master_key is unset, empty, or a publicly known default", 

2642 ) 

2643 coordination_redis: CoordinationRedisParams | None = Field( 

2644 None, 

2645 description=( 

2646 "standalone Redis for cross-pod coordination (tpm/rpm rate limits, " 

2647 "spend tracking, pod lock manager, shared health checks), configured " 

2648 "independently of the response-cache backend; takes precedence over " 

2649 "borrowing the `cache_params` Redis and over the REDIS_* env fallback" 

2650 ), 

2651 ) 

2652 control_plane_url: str | None = Field( 

2653 None, 

2654 description=( 

2655 "Global Control Plane: URL of the control plane whose admin UI manages this instance. " 

2656 "Enables /v3/login and /v3/login/exchange on this instance so that UI can authenticate " 

2657 "against it cross-origin, and restricts the SSO return_to origin to that URL. " 

2658 "No state is shared with the control plane" 

2659 ), 

2660 ) 

2661 allow_cli_sso_verification_uri_complete: bool | None = Field( 

2662 None, 

2663 description="opt-in to RFC 8628 verification_uri_complete for the CLI SSO device flow, pre-filling the user_code in the browser. Off by default; intended for same-host clients where the device that starts the flow and the browser run on the same machine", 

2664 ) 

2665 include_call_id_in_error_body: bool | None = Field( 

2666 None, 

2667 description="opt-in to copy the x-litellm-call-id response header's value into JSON error bodies, as error.litellm_call_id on the OpenAI-shaped and /v1/messages routes and as a top-level litellm_call_id on pass-through routes, so an error a client prints names the request to look up. Off by default", 

2668 ) 

2669 enable_claude_code_gateway: bool | None = Field( 

2670 None, 

2671 description="serve the Claude Code gateway protocol (https://code.claude.com/docs/en/claude-apps-gateway) under /claude_code_gateway: OAuth device-flow sign-in reusing proxy SSO, plus managed settings and OTLP telemetry ingestion. Off by default", 

2672 ) 

2673 claude_code_gateway_managed_settings: dict[str, Any] | None = Field( 

2674 None, 

2675 description="Claude Code managed-settings.json served verbatim at the gateway's /claude_code_gateway/managed/settings endpoint. When unset the endpoint returns 404 (no managed policy)", 

2676 ) 

2677 database_url: str | None = Field( 

2678 None, 

2679 description="connect to a postgres db - needed for generating temporary keys + tracking spend / key", 

2680 ) 

2681 database_connection_pool_limit: int | None = Field( 

2682 10, 

2683 description="default connection pool for prisma client connecting to postgres db", 

2684 ) 

2685 database_connection_timeout: float | None = Field( 

2686 60, description="default timeout for a connection to the database" 

2687 ) 

2688 database_connect_timeout: float | None = Field( 

2689 None, 

2690 description=( 

2691 "Prisma `connect_timeout` URL param (seconds). Bounds how long the " 

2692 "engine waits to establish a new connection before failing. Defaults " 

2693 "to Prisma's built-in value when unset." 

2694 ), 

2695 ) 

2696 database_socket_timeout: float | None = Field( 

2697 None, 

2698 description=( 

2699 "Prisma `socket_timeout` URL param (seconds). When set, an in-flight " 

2700 "operation that has not produced data within this window is aborted. " 

2701 "For capping how long idle pooled connections are kept, see " 

2702 "`database_max_idle_connection_lifetime`." 

2703 ), 

2704 ) 

2705 database_max_idle_connection_lifetime: float | None = Field( 

2706 60, 

2707 description=( 

2708 "Prisma `max_idle_connection_lifetime` URL param (seconds). A pooled " 

2709 "connection idle longer than this is closed and replaced instead of " 

2710 "being handed to the next request. Defaults to 60 so connections are " 

2711 "recycled before common infra idle timeouts (AWS NLB / RDS Proxy " 

2712 "~350s, many LBs 60-350s) silently drop them and requests fail with " 

2713 "`Error { kind: Closed }`. A value pinned on the DATABASE_URL or set " 

2714 "via `database_extra_connection_params` takes precedence." 

2715 ), 

2716 ) 

2717 database_extra_connection_params: dict[str, Any] | None = Field( 

2718 None, 

2719 description=( 

2720 "Escape hatch: extra key/value pairs appended verbatim to the Prisma " 

2721 "DATABASE_URL / DIRECT_URL query string (e.g. `sslmode`, `pgbouncer`, " 

2722 "`statement_cache_size`). Keys here override any default LiteLLM sets." 

2723 ), 

2724 ) 

2725 database_disable_prepared_statements: bool | None = Field( 

2726 None, 

2727 description=( 

2728 "Disable server-side prepared statements by setting Prisma's " 

2729 "`pgbouncer=true` URL param. Use this for pgbouncer transaction-pooling " 

2730 "deployments, or to prevent the 'cached plan must not change result " 

2731 "type' error that pooled connections hit during rolling schema " 

2732 "migrations. An explicit `pgbouncer` in `database_extra_connection_params` " 

2733 "takes precedence." 

2734 ), 

2735 ) 

2736 database_type: Literal["dynamo_db"] | None = Field(None, description="to use dynamodb instead of postgres db") 

2737 database_args: DynamoDBArgs | None = Field( 

2738 None, 

2739 description="custom args for instantiating dynamodb client - e.g. billing provision", 

2740 ) 

2741 otel: bool | None = Field( 

2742 None, 

2743 description="[BETA] OpenTelemetry support - this might change, use with caution.", 

2744 ) 

2745 custom_auth: str | None = Field( 

2746 None, 

2747 description="override user_api_key_auth with your own auth script - https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth", 

2748 ) 

2749 max_parallel_requests: int | None = Field( 

2750 None, 

2751 description="maximum parallel requests for each api key", 

2752 ) 

2753 global_max_parallel_requests: int | None = Field( 

2754 None, description="global max parallel requests to allow for a proxy instance." 

2755 ) 

2756 user_api_key_cache_max_size: int | None = Field( 

2757 None, 

2758 gt=0, 

2759 description=( 

2760 "max number of entries (virtual keys, teams, users, end users, memberships, ...) each worker keeps in " 

2761 "its in-memory auth cache. Defaults to 200. Raise this if you have more active keys than that or auth " 

2762 "lookups keep hitting the DB" 

2763 ), 

2764 ) 

2765 max_request_size_mb: int | None = Field( 

2766 None, 

2767 description="max request size in MB, if a request is larger than this size it will be rejected", 

2768 ) 

2769 max_batch_file_size_mb: int | None = Field( 

2770 None, 

2771 description="max batch input file size in MB for /v1/files uploads with purpose=batch, if a file is larger than this size it will be rejected before being forwarded to the provider", 

2772 ) 

2773 max_file_size_mb: int | None = Field( 

2774 None, 

2775 description="max file size in MB for /v1/files uploads, for any purpose, if a file is larger than this size it will be rejected before being forwarded to the provider", 

2776 ) 

2777 allowed_file_extensions: tuple[str, ...] | None = Field( 

2778 None, 

2779 description="the only file extensions (e.g. ['.jsonl', '.pdf', '.txt']) accepted on /v1/files uploads, for any purpose, matched case-insensitively against the uploaded filename. Files with any other extension, or none, are rejected. An empty list rejects every upload. Unset means no allowlist is applied", 

2780 ) 

2781 blocked_file_extensions: tuple[str, ...] | None = Field( 

2782 None, 

2783 description="file extensions (e.g. ['.exe', '.sh']) rejected on /v1/files uploads, for any purpose, matched case-insensitively against the uploaded filename. Deprecated in favour of allowed_file_extensions; still enforced, after the allowlist, when set", 

2784 ) 

2785 max_response_size_mb: int | None = Field( 

2786 None, 

2787 description="max response size in MB, if a response is larger than this size it will be rejected", 

2788 ) 

2789 proxy_config_reload_interval_seconds: int = Field( 

2790 30, 

2791 gt=0, 

2792 description="how often (in seconds) each pod reloads config-in-DB objects (models, credentials, guardrails, etc.) when store_model_in_db is enabled; lower values speed up multi-pod convergence at the cost of more DB load. Applied on proxy startup", 

2793 ) 

2794 cancel_on_disconnect: bool | None = Field( 

2795 None, 

2796 description="cancel the in-flight upstream LLM request (non-streaming) when the client disconnects, freeing backend capacity (e.g. a vLLM GPU slot); the request is logged as a 499 failure", 

2797 ) 

2798 infer_model_from_keys: bool | None = Field( 

2799 None, 

2800 description="for `/models` endpoint, infers available model based on environment keys (e.g. OPENAI_API_KEY)", 

2801 ) 

2802 background_health_checks: bool | None = Field(None, description="run health checks in background") 

2803 health_check_interval: int = Field(300, description="background health check interval in seconds") 

2804 health_check_concurrency: int | None = Field( 

2805 None, 

2806 description=( 

2807 "limit concurrent health checks per cycle; when unset, health checks run without a concurrency cap" 

2808 ), 

2809 ) 

2810 health_check_skip_disabled_background_models: bool = Field( 

2811 False, 

2812 description=( 

2813 "When true, deployments with model_info.disable_background_health_check " 

2814 "are skipped for on-demand GET /health as well as the background health loop." 

2815 ), 

2816 ) 

2817 background_health_check_model_groups: tuple[str, ...] | None = Field( 

2818 None, 

2819 description=( 

2820 "Opt-in allowlist of model group names for background health checks and " 

2821 "health-check routing. When set, the background loop probes only deployments " 

2822 "whose model_name is listed, and enable_health_check_routing filters unhealthy " 

2823 "deployments only within the listed groups; every other group, including newly " 

2824 "added deployments, is skipped and keeps its configured routing strategy. " 

2825 "When unset, all deployments participate (opt out per deployment via " 

2826 "model_info.disable_background_health_check)." 

2827 ), 

2828 ) 

2829 model_list_healthy_only: bool | None = Field( 

2830 None, 

2831 description=( 

2832 "When true, `/models`, `/v1/models/{id}` and `/model/info` hide models whose backing " 

2833 "deployments are all unhealthy, for every caller, without needing `healthy_only=true` " 

2834 "per request. Requires `background_health_checks: true`, and keeps deployment health " 

2835 "state cached without turning on `enable_health_check_routing`, so routing is " 

2836 "unaffected. With no health state nothing is hidden. Hiding is presentation-only, a " 

2837 "hidden model can still be called." 

2838 ), 

2839 ) 

2840 alerting: list | None = Field( 

2841 None, 

2842 description="List of alerting integrations - e.g. `alerting: ['slack', 'webhook', 'email']`. 'slack' posts Slack-format messages to any Slack-compatible webhook (Slack, Rocket.Chat, Mattermost); 'webhook' posts structured JSON budget alerts to WEBHOOK_URL", 

2843 ) 

2844 alert_types: list[AlertType] | None = Field( 

2845 None, 

2846 description="List of alerting types. By default it is all alerts", 

2847 ) 

2848 alert_to_webhook_url: dict | None = Field( 

2849 None, 

2850 description="Mapping of alert type to webhook url. e.g. `alert_to_webhook_url: {'budget_alerts': 'https://nothooks.slack.com/services/T00000000/B00000000/XXXXXXXXXXXXXXXXXXXXXXXX'}`", 

2851 ) 

2852 alerting_args: dict | None = Field(None, description="Controllable params for slack alerting - e.g. ttl in cache.") 

2853 alerting_threshold: int | None = Field( 

2854 None, 

2855 description="sends alerts if requests hang for 5min+", 

2856 ) 

2857 ui_access_mode: Literal["admin_only", "all"] | None = Field("all", description="Control access to the Proxy UI") 

2858 max_failed_login_attempts_per_source: int | None = Field( 

2859 None, 

2860 ge=1, 

2861 description="Failed Admin UI sign-in attempts allowed from one source address, across every username, within `failed_login_window_seconds`. One more blocks that address for `failed_login_block_seconds`. Half this value, rounded down but at least 1, is the allowance for one username from that address; one more blocks that address for that username only, and its further failures stop counting toward the address limit, so a script stuck on one account does not block everyone behind a shared address. The per-address limit is only enforced when `trusted_proxy_ranges` is set: to the proxies in front of LiteLLM, or to an empty list when clients connect directly. Left unset, the peer address may be a shared ingress and only the per-username half runs. IPv6 addresses are grouped by /64. Set under `general_settings` in config.yaml. Defaults to 10", 

2862 ) 

2863 max_failed_login_attempts_per_source_overrides: dict[str, int] | None = Field( 

2864 None, 

2865 description="Per-address overrides of `max_failed_login_attempts_per_source`, keyed by IP address or CIDR range, e.g. {'1.2.3.4': 200, '5.6.0.0/24': 500}. The most specific matching range wins (between equivalent keys such as '1.2.3.4' and '1.2.3.4/32', an exemption wins, then the higher limit), and the per-username allowance for that address follows as half the override. A value of 0 exempts the address from both limits. Set under `general_settings` in config.yaml", 

2866 ) 

2867 failed_login_window_seconds: int | None = Field( 

2868 None, 

2869 ge=1, 

2870 description="Fixed window in seconds over which failed Admin UI sign-in attempts are counted. The window starts at the first failure and is not extended by later ones. Set under `general_settings` in config.yaml. Defaults to 60", 

2871 ) 

2872 failed_login_block_seconds: int | None = Field( 

2873 None, 

2874 ge=1, 

2875 description="How long a blocked source address, or source address and username, stays blocked. Every attempt from a blocked key, right or wrong, is refused with 429 before the password is checked; the block is not extended by refused attempts. Set under `general_settings` in config.yaml. Defaults to 300", 

2876 ) 

2877 allowed_routes: list | None = Field(None, description="Proxy API Endpoints you want users to be able to access") 

2878 reject_clientside_metadata_tags: bool | None = Field( 

2879 None, 

2880 description="When set to True, rejects requests that contain client-side 'metadata.tags' to prevent users from influencing budgets by sending different tags. Tags can only be inherited from the API key metadata.", 

2881 ) 

2882 missing_session_id: Literal["generate", "reject", "omit"] | None = Field( 

2883 None, 

2884 description="What to do with LLM API requests that carry no session id (x-litellm-session-id header, metadata.session_id, etc.). 'generate' stamps one id into litellm_session_id, litellm_trace_id and metadata.session_id so SpendLogs and logging callbacks agree; 'reject' returns 400; 'omit' leaves SpendLogs.session_id null, matching callbacks such as Langfuse that only record a client-established metadata.session_id. Unset keeps the legacy behavior where SpendLogs falls back to the trace id while callbacks get no session id.", 

2885 ) 

2886 enable_public_model_hub: bool = Field( 

2887 default=False, 

2888 description="Public model hub for users to see what models they have access to, supported openai params, etc.", 

2889 ) 

2890 pass_through_request_timeout: float | None = Field( 

2891 default=None, 

2892 description="Default upstream request timeout in seconds for native and custom pass-through endpoints that use pass_through_request. Defaults to 600 when unset.", 

2893 ) 

2894 pass_through_endpoints: list[PassThroughGenericEndpoint] | None = Field( 

2895 default=None, 

2896 description="Set-up pass-through endpoints for provider-specific endpoints. Docs - https://docs.litellm.ai/docs/proxy/pass_through", 

2897 ) 

2898 enable_openai_websocket_passthrough: bool | None = Field( 

2899 default=None, 

2900 description="Serve the OpenAI pass-through WebSocket route, which relays frames to OpenAI under the proxy's own provider credential without reading them. Off by default.", 

2901 ) 

2902 transcribe_media_buckets: list[str] | None = Field( 

2903 default=None, 

2904 description="S3 bucket names that keys other than proxy admins may read media from and write transcripts to through the Amazon Transcribe pass-through. Unset means only proxy admins can start transcription jobs.", 

2905 ) 

2906 user_header_name: str | None = Field( 

2907 None, 

2908 description="[DEPRECATED] Use 'user_header_mappings' instead. When set, the header value is treated as the end user id unless overridden by user_header_mappings.", 

2909 ) 

2910 user_header_mappings: list[UserHeaderMapping] | None = None 

2911 supported_db_objects: list[SupportedDBObjectType] | None = Field( 

2912 None, 

2913 description="Fine-grained control over which object types to load from the database when store_model_in_db is True. Available types: 'models', 'mcp', 'guardrails', 'vector_stores', 'pass_through_endpoints', 'prompts', 'model_cost_map', 'tools', 'config_overrides'. If not set, all objects are loaded (default behavior).", 

2914 ) 

2915 user_mcp_management_mode: UserMCPManagementMode | None = Field( 

2916 None, 

2917 description="Controls how non-admin users interact with MCP servers in the dashboard. 'restricted' shows only accessible servers, 'view_all' lists every server in read-only mode.", 

2918 ) 

2919 store_prompts_in_spend_logs: bool | None = Field( 

2920 None, 

2921 description="If True, stores request messages and responses in spend logs. Default is False.", 

2922 ) 

2923 disable_auto_add_proxy_admin_to_teams: bool | None = Field( 

2924 None, 

2925 description="By default, the user calling /team/new is automatically added to the new team as a team admin. If True, proxy admins are no longer auto-added; members explicitly listed in members_with_roles are unaffected. Default is False.", 

2926 ) 

2927 enforce_fallback_model_access: bool | None = Field( 

2928 None, 

2929 description="If True, router fallbacks configured in router_settings are only attempted when the calling key (and its team and project) is allowed to call the fallback model; unauthorized fallback targets are skipped and the primary model's error is returned. Default is False.", 

2930 ) 

2931 scheduled_job_stagger: ScheduledJobStaggerSettings | None = Field( 

2932 None, 

2933 description=( 

2934 "Spreads the proxy's scheduled background jobs (spend flushes, budget resets, " 

2935 "config reloads, exports) across a window instead of firing them together on " 

2936 "every replica. On by default; set to tune the window, pin a job, or turn it off." 

2937 ), 

2938 ) 

2939 spend_capture_rate_check: SpendCaptureRateCheckSettings | None = Field( 

2940 None, 

2941 description=( 

2942 "Daily check of the spend LiteLLM captured against the provider's own bill (OpenAI via OPENAI_ADMIN_KEY). " 

2943 "Publishes litellm_spend_capture_rate per provider and alerts when the ratio over the lookback window " 

2944 "falls under the threshold (default 0.9). Off unless set." 

2945 ), 

2946 ) 

2947 maximum_spend_logs_retention_period: str | None = Field( 

2948 None, 

2949 description="Maximum retention period for spend logs (e.g., '7d' for 7 days). Logs older than this will be deleted.", 

2950 ) 

2951 maximum_autorouter_session_retention_period: str | None = Field( 

2952 None, 

2953 description="Maximum retention period for auto-router benchmark session rollup rows (e.g., '365d'). Rows whose last turn is older than this are deleted by the spend log cleanup job, on that job's schedule. Unset means rollup rows are never deleted.", 

2954 ) 

2955 maximum_health_check_retention_period: str | None = Field( 

2956 None, 

2957 description=( 

2958 "Maximum retention period for health-check rows (e.g., '30d'). Rows whose checked_at is older than this " 

2959 "are deleted by the spend log cleanup job, on that job's schedule. Unset means rows are never deleted. " 

2960 "Set this well above health_check_interval because /health and the UI read the latest row per model." 

2961 ), 

2962 ) 

2963 use_spend_logs_partitioning: bool | None = Field( 

2964 None, 

2965 description="If True and LiteLLM_SpendLogs has been converted to a range-partitioned table (db_scripts/partition_spend_logs.sql), retention cleanup drops expired partitions instead of deleting rows, and pre-creates upcoming partitions. Default is False.", 

2966 ) 

2967 maximum_spend_logs_cleanup_batch_size: int | None = Field( 

2968 None, 

2969 description="Rows deleted per DELETE statement by the spend log cleanup job. Defaults to 1000.", 

2970 ) 

2971 maximum_spend_logs_cleanup_max_batches: int | None = Field( 

2972 None, 

2973 description="Maximum DELETE statements the spend log cleanup job issues per table per run. Defaults to 500.", 

2974 ) 

2975 maximum_spend_logs_cleanup_run_budget: str | None = Field( 

2976 None, 

2977 description="Wall-clock budget for one spend log cleanup run (e.g. '5m'), shared across every table it prunes. A run that hits the budget stops and the next run resumes from where it left off. Defaults to '5m'.", 

2978 ) 

2979 maximum_spend_logs_cleanup_batch_timeout: str | None = Field( 

2980 None, 

2981 description="Postgres statement_timeout and lock_timeout applied to each spend log cleanup delete batch (e.g. '30s'), so cleanup cannot hold row locks or a connection indefinitely. Defaults to '30s'.", 

2982 ) 

2983 mcp_internal_ip_ranges: list[str] | None = Field( 

2984 None, 

2985 description="Custom CIDR ranges that define internal/private networks for MCP access control. When set, only these ranges are treated as internal. Defaults to RFC 1918 private ranges (10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 127.0.0.0/8).", 

2986 ) 

2987 mcp_allowed_clients: list[MCPAllowedClient] | None = Field( 

2988 None, 

2989 description="MCP client applications admitted by the gateway, each an {alias, value} pair where alias is the name shown in the dashboard and logs and value is the identity that must match exactly. When set, every MCP request must carry a client identity equal to one of the values: a JWT caller is identified by the claim named in litellm_jwtauth.mcp_client_id_jwt_field, any other caller by the header named in mcp_client_id_header. A request with no resolvable identity, or an unlisted one, is rejected with 403. Unset means every client is admitted.", 

2990 ) 

2991 mcp_client_id_header: str | None = Field( 

2992 None, 

2993 description="Request header whose value names the calling MCP client application (for example 'x-mcp-client') for callers that did not authenticate with a JWT, used only while mcp_allowed_clients is set. The client picks this value itself, so it is a policy control rather than a security boundary; prefer litellm_jwtauth.mcp_client_id_jwt_field where callers use JWTs.", 

2994 ) 

2995 mcp_trusted_proxy_ranges: list[str] | None = Field( 

2996 None, 

2997 description="CIDR ranges of trusted reverse proxies. When set, X-Forwarded-For and X-Forwarded-* origin headers are only trusted from these IPs.", 

2998 ) 

2999 mcp_xff_num_trusted_hops: int | None = Field( 

3000 None, 

3001 ge=1, 

3002 description="Number of trusted reverse proxies/load balancers in front of the gateway that append to X-Forwarded-For. When set (and mcp_trusted_proxy_ranges validates the direct peer), the client IP for MCP access control is read this many entries from the right of the chain instead of the spoofable leftmost value, defeating append-style X-Forwarded-For forgery.", 

3003 ) 

3004 trusted_proxy_ranges: list[str] | None = Field( 

3005 None, 

3006 description="CIDR ranges of trusted reverse proxies allowed to provide identity headers for header-based auth paths such as enable_oauth2_proxy_auth and custom_ui_sso_sign_in_handler, and whose X-Forwarded-For is used to attribute Admin UI sign-in attempts to a source address. Set it to an empty list when clients connect directly, so the peer address is the source. Left unset, or containing an entry that is not an address or CIDR range, the per-source sign-in limit is off.", 

3007 ) 

3008 store_model_in_db: bool | None = Field( 

3009 None, 

3010 description="If True, models and config are stored in and loaded from the database. Default is False.", 

3011 ) 

3012 forward_client_headers_to_llm_api: bool | None = Field( 

3013 None, 

3014 description="If True, forwards client headers (e.g. Authorization) to the LLM API. Required for Claude Code with Max subscription.", 

3015 ) 

3016 mcp_required_fields: list[str] | None = Field( 

3017 None, 

3018 description="List of MCP server fields that must be filled in for a submission to pass standards checks (e.g. ['description', 'source_url', 'alias']).", 

3019 ) 

3020 password_policy_min_length: int | None = Field( 

3021 None, 

3022 description=( 

3023 "Minimum length required for a locally-managed user's password. Default is 12; " 

3024 "a value below 8 is floored to 8 rather than weakening the requirement further." 

3025 ), 

3026 ) 

3027 password_policy_require_uppercase: bool | None = Field( 

3028 None, 

3029 description="If True (default), a locally-managed user's password must contain an uppercase letter.", 

3030 ) 

3031 password_policy_require_lowercase: bool | None = Field( 

3032 None, 

3033 description="If True (default), a locally-managed user's password must contain a lowercase letter.", 

3034 ) 

3035 password_policy_require_numbers: bool | None = Field( 

3036 None, 

3037 description="If True (default), a locally-managed user's password must contain a number.", 

3038 ) 

3039 password_policy_require_special_characters: bool | None = Field( 

3040 None, 

3041 description="If True (default), a locally-managed user's password must contain a special (non-alphanumeric) character.", 

3042 ) 

3043 disable_password_login_when_sso_enabled: bool | None = Field( 

3044 None, 

3045 description=( 

3046 "If True and SSO is configured (MICROSOFT_CLIENT_ID, GOOGLE_CLIENT_ID, " 

3047 "GENERIC_CLIENT_ID, or SAML_IDP_METADATA_URL/XML), disables username/password " 

3048 "login on /login, /v2/login, and /v3/login so SSO is the only way to reach the " 

3049 "Admin UI. An admin locked out of the UI can still administer the proxy over the " 

3050 "API with the master key; unset this setting and restart the proxy to restore " 

3051 "UI username/password login. Default is False." 

3052 ), 

3053 ) 

3054 disable_responses_id_security: bool | None = Field( 

3055 None, 

3056 description=( 

3057 "If True, disables ownership enforcement on Responses API ids. " 

3058 "Keys may then retrieve, cancel, delete, and chain from any response id, " 

3059 "including ids belonging to another user or team and ids this proxy never issued. " 

3060 "WARNING: this removes tenant isolation on /v1/responses" 

3061 ), 

3062 ) 

3063 allow_unmanaged_response_ids: bool | None = Field( 

3064 None, 

3065 description=( 

3066 "If True, lets keys address Responses API ids that this proxy did not issue " 

3067 "(raw provider ids, or ids issued before response-id encryption was configured). " 

3068 "Such an id carries no owner, so no ownership check can run on it; ids this proxy " 

3069 "did issue keep full ownership enforcement. Off by default, in which case an " 

3070 "unrecognized response id is rejected with 403" 

3071 ), 

3072 ) 

3073 disable_env_credential_login: bool | None = Field( 

3074 None, 

3075 description=( 

3076 "If True, disables signing in to the Admin UI with the environment credentials: " 

3077 "UI_USERNAME/UI_PASSWORD, or the master key when UI_PASSWORD is unset (that fallback " 

3078 "means env-credential login is always live by default). Database users with passwords " 

3079 "are unaffected. LOCKOUT RISK: create at least one proxy admin user with a password " 

3080 "before enabling, or nobody can sign in to the UI. A locked-out admin can still " 

3081 "administer the proxy over the API with the master key, and can unset this setting " 

3082 "and restart the proxy to restore env-credential login. Default is False." 

3083 ), 

3084 ) 

3085 disable_budget_reservation: bool | None = Field( 

3086 None, 

3087 description=( 

3088 "If True, disables the optimistic per-request budget reservation " 

3089 "introduced in v1.84.0. " 

3090 "WARNING: This weakens hard budget enforcement. Without the reservation, " 

3091 "a burst of concurrent requests from a single key can each pass the " 

3092 "read-time spend check before any of them is charged, allowing a " 

3093 "configured budget to be exceeded under high concurrency. " 

3094 "Budgets are still evaluated on every request at read time, so " 

3095 "an already-exhausted budget is still rejected. " 

3096 "Enable only if your deployment is experiencing phantom " 

3097 "BudgetExceededError responses caused by leaked reservations " 

3098 "(see GitHub issue #27639). " 

3099 "An INFO notice is logged once per worker at config load while this flag " 

3100 "is active as a reminder that hard enforcement is relaxed." 

3101 ), 

3102 ) 

3103 apply_user_budget_to_team_keys: bool | None = Field( 

3104 None, 

3105 description=( 

3106 "If True, a user's personal max_budget is enforced on every request they " 

3107 "make, including requests made with a team-scoped key. Defaults to False, " 

3108 "where a team-scoped key is governed only by the team and team-member " 

3109 "budgets and the key owner's personal max_budget does not apply " 

3110 "(see GitHub issue #12905)." 

3111 ), 

3112 ) 

3113 user_url_validation: bool | None = Field( 

3114 None, 

3115 description=( 

3116 "Master switch for the SSRF guard applied to user-supplied URLs " 

3117 "(image_url, file_url, MCP/OpenAPI spec URLs, etc). Defaults to True. " 

3118 "Set to False to disable DNS/IP validation entirely (not recommended)." 

3119 ), 

3120 ) 

3121 user_url_allowed_hosts: list[str] | None = Field( 

3122 None, 

3123 description=( 

3124 "SSRF allowlist for user-supplied URLs. Entries are `hostname` or " 

3125 "`hostname:port` (bracketed for IPv6, e.g. `[::1]:8080`). Allowlisted " 

3126 "hosts skip the blocked-network check in validate_url() but still " 

3127 "resolve DNS. Use this to permit legitimate internal targets, e.g. " 

3128 "an internal OpenAPI/MCP server." 

3129 ), 

3130 ) 

3131 provider_url_destination_allowed_hosts: list[str] | None = Field( 

3132 None, 

3133 description="Allowlist of hosts a request may redirect a provider call's destination URL to.", 

3134 ) 

3135 

3136 

3137class ConfigYAML(LiteLLMPydanticObjectBase): 

3138 """ 

3139 Documents all the fields supported by the config.yaml 

3140 """ 

3141 

3142 environment_variables: dict | None = Field( 

3143 None, 

3144 description="Object to pass in additional environment variables via POST request", 

3145 ) 

3146 model_list: list[ModelParams] | None = Field( 

3147 None, 

3148 description="List of supported models on the server, with model-specific configs", 

3149 ) 

3150 litellm_settings: dict | None = Field( 

3151 None, 

3152 description="litellm Module settings. See __init__.py for all, example litellm.drop_params=True, litellm.set_verbose=True, litellm.api_base, litellm.cache", 

3153 ) 

3154 general_settings: ConfigGeneralSettings | None = None 

3155 worker_registry: list[WorkerRegistryEntry] | None = Field( 

3156 None, 

3157 description=( 

3158 "Global Control Plane: the independent proxy instances this instance's admin UI manages. " 

3159 "Setting it makes this a control plane, which serves the UI and does not route LLM requests. " 

3160 "Enterprise-only" 

3161 ), 

3162 ) 

3163 router_settings: UpdateRouterConfig | None = Field( 

3164 None, 

3165 description="litellm router object settings. See router.py __init__ for all, example router.num_retries=5, router.timeout=5, router.max_retries=5, router.retry_after=5", 

3166 ) 

3167 

3168 model_config = ConfigDict(protected_namespaces=()) 

3169 

3170 

3171from litellm.models.verification_token import ( # noqa: E402 

3172 LiteLLM_DeletedVerificationToken as LiteLLM_DeletedVerificationToken, 

3173) 

3174from litellm.models.verification_token import ( # noqa: E402 

3175 LiteLLM_VerificationToken as LiteLLM_VerificationToken, 

3176) 

3177 

3178 

3179class LiteLLM_VerificationTokenView(LiteLLM_VerificationToken): 

3180 """ 

3181 Combined view of litellm verification token + litellm team table (select values) 

3182 """ 

3183 

3184 team_spend: float | None = None 

3185 team_alias: str | None = None 

3186 team_tpm_limit: int | None = None 

3187 team_rpm_limit: int | None = None 

3188 team_tpd_limit: int | None = None 

3189 team_max_budget: float | None = None 

3190 team_soft_budget: float | None = None 

3191 team_model_max_budget: dict[str, object] | None = None 

3192 team_models: list = [] 

3193 team_blocked: bool = False 

3194 soft_budget: float | None = None 

3195 team_model_aliases: dict | None = None 

3196 team_member: Member | None = None 

3197 team_metadata: dict | None = None 

3198 team_object_permission_id: str | None = None 

3199 

3200 # Team Member Specific Params 

3201 team_member_spend: float | None = None 

3202 team_member_tpm_limit: int | None = None 

3203 team_member_rpm_limit: int | None = None 

3204 

3205 # End User Params 

3206 end_user_id: str | None = None 

3207 end_user_tpm_limit: int | None = None 

3208 end_user_rpm_limit: int | None = None 

3209 end_user_tpd_limit: int | None = None 

3210 end_user_max_budget: float | None = None 

3211 end_user_model_max_budget: dict | None = None 

3212 

3213 # Organization Params 

3214 organization_alias: str | None = None 

3215 organization_max_budget: float | None = None 

3216 organization_tpm_limit: int | None = None 

3217 organization_rpm_limit: int | None = None 

3218 organization_metadata: dict | None = None 

3219 

3220 # Project Params 

3221 project_alias: str | None = None 

3222 project_metadata: dict | None = None 

3223 

3224 # Time stamps 

3225 last_refreshed_at: float | None = None # last time joint view was pulled from db 

3226 

3227 def __init__(self, **kwargs): 

3228 # Handle litellm_budget_table_* keys (budget table overrides when key value is None or empty) 

3229 for key, value in list(kwargs.items()): 

3230 if key.startswith("litellm_budget_table_") and value is not None: 3230 ↛ 3232line 3230 didn't jump to line 3232 because the condition on line 3230 was never true

3231 # Extract the corresponding attribute name 

3232 attr_name = key.replace("litellm_budget_table_", "") 

3233 # Use key's value from kwargs (from DB view), not class default 

3234 current = kwargs.get(attr_name) 

3235 if current is None: 

3236 current = getattr(self, attr_name, None) 

3237 # Apply budget value when key has no value, or for model_max_budget when key has empty dict 

3238 should_apply = current is None or ( 

3239 attr_name == "model_max_budget" and isinstance(current, dict) and len(current) == 0 

3240 ) 

3241 if should_apply: 

3242 kwargs[attr_name] = value 

3243 if key == "end_user_id" and value is not None and isinstance(value, int): 3243 ↛ 3244line 3243 didn't jump to line 3244 because the condition on line 3243 was never true

3244 kwargs[key] = str(value) 

3245 

3246 if kwargs.get("organization_id") is not None: 3246 ↛ 3247line 3246 didn't jump to line 3247 because the condition on line 3246 was never true

3247 kwargs["org_id"] = kwargs.pop("organization_id") 

3248 # Initialize the superclass 

3249 super().__init__(**kwargs) 

3250 

3251 

3252class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response object for user api key auth 

3253 """ 

3254 Return the row in the db 

3255 """ 

3256 

3257 api_key: str | None = None 

3258 user_role: LitellmUserRoles | None = None 

3259 allowed_model_region: AllowedModelRegion | None = None 

3260 parent_otel_span: Span | None = None 

3261 rpm_limit_per_model: dict[str, int] | None = None 

3262 tpm_limit_per_model: dict[str, int] | None = None 

3263 user_tpm_limit: int | None = None 

3264 user_rpm_limit: int | None = None 

3265 user_email: str | None = None 

3266 user_spend: float | None = None 

3267 user_max_budget: float | None = None 

3268 # Values stay `object` rather than BudgetConfig: this is the raw JSON column, 

3269 # and validating it here would make one malformed row fail auth outright. 

3270 # resolve_model_budget validates the single entry a request actually needs. 

3271 user_model_max_budget: Mapping[str, object] | None = None 

3272 request_route: str | None = None 

3273 is_session_token: bool = False 

3274 # Server-only marker set exclusively by the MCP gateway admission path 

3275 # (reload_admitted_user) for a keyless user-subject admitted via a gateway DCR session 

3276 # bearer or bridge envelope. Not a DB column and never populated from caller-controlled key 

3277 # metadata or JWT claims, so it cannot be forged to gain the team-inherited MCP grant union 

3278 # or to escape the caller-Authorization egress scrub. exclude=True keeps it out of serialization. 

3279 mcp_admitted_user_subject: bool = Field(default=False, exclude=True) 

3280 # team_id -> that team's mcp_rpm_limit map, for a keyless admitted subject that reaches MCP 

3281 # servers through several teams at once and therefore has no single team_id for the limiter to 

3282 # key off. Server-only and stripped from validated input for the same reason as the marker 

3283 # above: a forged entry would let a caller pick which team's rpm bucket it is charged against. 

3284 mcp_source_team_rpm_limits: dict[str, dict[str, int]] | None = Field(default=None, exclude=True) 

3285 # The single MCP server_id a gateway session bearer was scoped to at authorize time (RFC 8707 

3286 # resource), or None for an aggregate-scope session. A RESTRICTION intersected against the live 

3287 # grant resolution, never a grant. Server-only, set exclusively by the MCP gateway admission 

3288 # path via post-construction assignment and stripped from validated input like the markers 

3289 # above; a forged value could at most narrow, but the stripping keeps the field's provenance 

3290 # single-owner so its meaning stays trustworthy. 

3291 mcp_session_resource_server_id: str | None = Field(default=None, exclude=True) 

3292 mcp_toolset_id: str | None = Field(default=None, exclude=True) 

3293 via_virtual_key: bool = Field( 

3294 default=False, 

3295 exclude=True, 

3296 description=( 

3297 "Server-only marker set exclusively by the DB virtual-key and master-key auth paths via " 

3298 "post-construction assignment. Stripped from validated input so custom auth handlers, JWT " 

3299 "claims, or key metadata cannot forge it. Gates overwrite_user_with_key_hash stamping: only " 

3300 "a credential the proxy itself validated as a key may be forwarded as the provider-facing " 

3301 "user id." 

3302 ), 

3303 ) 

3304 agent_caller: AgentCaller | None = Field( 

3305 default=None, 

3306 exclude=True, 

3307 description=( 

3308 "Set per request from the x-litellm-user-id / x-litellm-team-id headers an agent echoes back on " 

3309 "calls made with its own key. Every check treats it as a ceiling, so a forged value can only " 

3310 "narrow the agent's access." 

3311 ), 

3312 ) 

3313 budget_reservation: dict[str, Any] | None = Field(default=None, exclude=True) 

3314 team_budget_snapshot: TeamBudgetSnapshot | None = Field(default=None, exclude=True) 

3315 user_budget_snapshot: UserBudgetSnapshot | None = Field(default=None, exclude=True) 

3316 org_budget_snapshot: OrgBudgetSnapshot | None = Field(default=None, exclude=True) 

3317 matched_model_access_groups: list[str] | None = Field(default=None, exclude=True) 

3318 budget_throttle_pct: float | None = Field(default=None, exclude=True) 

3319 user: Any | None = None # Expanded user object when expand=user is used 

3320 created_by_user: Any | None = None # Expanded created_by user when expand=user is used 

3321 end_user_object_permission: LiteLLM_ObjectPermissionTable | None = None 

3322 # Team object_permission preloaded in auth (e.g. get_team_object) to avoid 

3323 # per-request object_permission fetches in downstream checks (vector stores, etc.) 

3324 team_object_permission: LiteLLM_ObjectPermissionTable | None = None 

3325 # Decoded upstream IdP claims (groups, roles, etc.) propagated by JWT auth machinery 

3326 # and forwarded into outbound tokens by guardrails such as MCPJWTSigner. 

3327 jwt_claims: dict | None = None 

3328 

3329 model_config = ConfigDict(arbitrary_types_allowed=True) 

3330 

3331 @model_validator(mode="before") 

3332 @classmethod 

3333 def check_api_key(cls, values): 

3334 # If values is already an instance (not a dict), return it as-is 

3335 if not isinstance(values, dict): 3335 ↛ 3336line 3335 didn't jump to line 3336 because the condition on line 3335 was never true

3336 return values 

3337 # mcp_admitted_user_subject is a server-only marker, set ONLY by the MCP gateway admission 

3338 # path via post-construction assignment. Strip it from any validated input (constructor 

3339 # kwargs, model_validate, a JWT/key claim splat) so it can never be forged from caller data. 

3340 values.pop("mcp_admitted_user_subject", None) 

3341 values.pop("mcp_source_team_rpm_limits", None) 

3342 values.pop("mcp_session_resource_server_id", None) 

3343 values.pop("mcp_toolset_id", None) 

3344 values.pop("via_virtual_key", None) 

3345 values.pop("agent_caller", None) 

3346 if values.get("api_key") is not None: 

3347 values.update({"token": cls._safe_hash_litellm_api_key(values.get("api_key"))}) 

3348 if isinstance(values.get("api_key"), str): 3348 ↛ 3350line 3348 didn't jump to line 3350 because the condition on line 3348 was always true

3349 values.update({"api_key": cls._safe_hash_litellm_api_key(values.get("api_key"))}) 

3350 return values 

3351 

3352 @classmethod 

3353 def _safe_hash_litellm_api_key(cls, api_key: str) -> str: 

3354 """ 

3355 Helper to ensure all logged keys are hashed 

3356 Covers: 

3357 1. Regular API keys from LiteLLM DB 

3358 2. JWT tokens used for connecting to LiteLLM API 

3359 """ 

3360 normalized = api_key 

3361 if normalized[:7].lower() == "bearer ": 3361 ↛ 3362line 3361 didn't jump to line 3362 because the condition on line 3361 was never true

3362 normalized = normalized[7:] 

3363 if normalized.startswith("sk-"): 

3364 return hash_token(normalized) 

3365 from litellm.proxy.auth.handle_jwt import JWTHandler 

3366 

3367 if JWTHandler.is_jwt(token=normalized): 

3368 return f"hashed-jwt-{hash_token(token=normalized)}" 

3369 return normalized 

3370 

3371 @classmethod 

3372 def get_litellm_internal_health_check_user_api_key_auth(cls) -> "UserAPIKeyAuth": 

3373 """ 

3374 Returns a `UserAPIKeyAuth` object for the litellm internal health check service account. 

3375 

3376 This is used to track number of requests/spend for health check calls. 

3377 """ 

3378 from litellm.constants import LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME 

3379 

3380 return cls( 

3381 api_key=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME, 

3382 team_id=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME, 

3383 key_alias=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME, 

3384 team_alias=LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME, 

3385 ) 

3386 

3387 @classmethod 

3388 def get_litellm_cli_user_api_key_auth(cls) -> "UserAPIKeyAuth": 

3389 """ 

3390 Returns a `UserAPIKeyAuth` object for the litellm internal health check service account. 

3391 

3392 This is used to track number of requests/spend for health check calls. 

3393 """ 

3394 from litellm.constants import LITTELM_CLI_SERVICE_ACCOUNT_NAME 

3395 

3396 return cls( 

3397 api_key=LITTELM_CLI_SERVICE_ACCOUNT_NAME, 

3398 team_id=LITTELM_CLI_SERVICE_ACCOUNT_NAME, 

3399 key_alias=LITTELM_CLI_SERVICE_ACCOUNT_NAME, 

3400 team_alias=LITTELM_CLI_SERVICE_ACCOUNT_NAME, 

3401 ) 

3402 

3403 @classmethod 

3404 def get_litellm_internal_jobs_user_api_key_auth(cls) -> "UserAPIKeyAuth": 

3405 """ 

3406 Returns a `UserAPIKeyAuth` object for internal LiteLLM jobs like key rotation. 

3407 

3408 This is used to track actions performed by automated system jobs. 

3409 """ 

3410 from litellm.constants import LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME 

3411 

3412 return cls( 

3413 api_key=LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME, 

3414 team_id="system", 

3415 key_alias=LITELLM_INTERNAL_JOBS_SERVICE_ACCOUNT_NAME, 

3416 team_alias="system", 

3417 user_id="system", 

3418 user_role=LitellmUserRoles.PROXY_ADMIN, 

3419 ) 

3420 

3421 @property 

3422 def is_team_service_account(self) -> bool: 

3423 return ( 

3424 self.user_id is None 

3425 and self.team_id is not None 

3426 and bool(self.metadata) 

3427 and self.metadata.get("service_account_id") is not None 

3428 ) 

3429 

3430 

3431def user_api_key_has_admin_view(user_api_key_dict: UserAPIKeyAuth) -> bool: 

3432 """Return True if the caller's role grants unscoped read access to all 

3433 tenant resources (managed files, batches, vector stores, spend rows, etc). 

3434 

3435 Lives on _types.py so leaf modules (e.g. litellm.llms.base_llm.managed_resources) 

3436 can use it without pulling in litellm.proxy.utils via management_endpoints. 

3437 """ 

3438 return user_api_key_dict.user_role in ( 

3439 LitellmUserRoles.PROXY_ADMIN, 

3440 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, 

3441 ) 

3442 

3443 

3444class UserInfoResponse(LiteLLMPydanticObjectBase): 

3445 user_id: str | None 

3446 user_info: dict | BaseModel | None 

3447 keys: list 

3448 teams: list 

3449 

3450 

3451class UserInfoV2Response(LiteLLMPydanticObjectBase): 

3452 """ 

3453 Response model for GET /v2/user/info 

3454 

3455 Returns ONLY the user object - no keys, no teams objects. 

3456 This is a lightweight alternative to UserInfoResponse. 

3457 """ 

3458 

3459 user_id: str 

3460 user_email: str | None = None 

3461 user_alias: str | None = None 

3462 user_role: str | None = None 

3463 spend: float = 0.0 

3464 max_budget: float | None = None 

3465 models: list[str] = [] 

3466 budget_duration: str | None = None 

3467 budget_reset_at: datetime | None = None 

3468 metadata: dict | None = None 

3469 created_at: datetime | None = None 

3470 updated_at: datetime | None = None 

3471 sso_user_id: str | None = None 

3472 teams: list[str] = [] # Just team IDs, not full team objects 

3473 object_permission: LiteLLM_ObjectPermissionTable | None = None 

3474 model_max_budget: Mapping[str, object] | None = None 

3475 model_max_budget_usage: Mapping[str, Mapping[str, object]] | None = None 

3476 

3477 

3478from litellm.models.config import LiteLLM_Config as LiteLLM_Config # noqa: E402 

3479from litellm.models.organization_membership import ( # noqa: E402 

3480 LiteLLM_OrganizationMembershipTable as LiteLLM_OrganizationMembershipTable, 

3481) 

3482 

3483 

3484class LiteLLM_OrganizationTableUpdate(LiteLLM_BudgetTable): 

3485 """Represents user-controllable params for a LiteLLM_OrganizationTable record""" 

3486 

3487 organization_id: str | None = None 

3488 organization_alias: str | None = None 

3489 budget_id: str | None = None 

3490 spend: float | None = None 

3491 metadata: dict | None = None 

3492 models: list[str] | None = None 

3493 updated_by: str | None = None 

3494 object_permission: LiteLLM_ObjectPermissionBase | None = None 

3495 model_tpm_limit: dict[str, int] | None = None 

3496 model_rpm_limit: dict[str, int] | None = None 

3497 

3498 @model_validator(mode="before") 

3499 @classmethod 

3500 def set_model_info(cls, values): 

3501 for field in LiteLLM_ManagementEndpoint_MetadataFields: 

3502 if values.get(field) is not None: 

3503 # add to metadata 

3504 if values.get("metadata") is None: 

3505 values.update({"metadata": {}}) 

3506 values["metadata"][field] = values.get(field) 

3507 values.pop(field) 

3508 return values 

3509 

3510 

3511class OrganizationUpdateRequestV2(LiteLLMPydanticObjectBase): 

3512 """ 

3513 Typed PATCH body for ``/v2/organization/{organization_id}`` (RFC 7396 merge-patch). 

3514 

3515 Presence is read from ``model_fields_set``, so a sent field is written and an omitted one is 

3516 left untouched. ``extra="forbid"`` makes an unknown key a 422 rather than a silent no-op, since 

3517 the contract hinges on which keys are present. See the endpoint for the per-field clear tokens. 

3518 """ 

3519 

3520 model_config = ConfigDict(extra="forbid") 

3521 

3522 organization_alias: str | None = None 

3523 models: list[str] | None = None 

3524 metadata: dict | None = None 

3525 tpm_limit: int | None = None 

3526 rpm_limit: int | None = None 

3527 max_budget: float | None = None 

3528 soft_budget: float | None = None 

3529 max_parallel_requests: int | None = None 

3530 model_max_budget: dict | None = None 

3531 budget_duration: str | None = None 

3532 object_permission: LiteLLM_ObjectPermissionBase | None = None 

3533 

3534 

3535from litellm.models.organization import ( # noqa: E402 

3536 LiteLLM_OrganizationTable as LiteLLM_OrganizationTable, 

3537) 

3538from litellm.models.user import LiteLLM_UserTable as LiteLLM_UserTable # noqa: E402 

3539 

3540 

3541class LiteLLM_OrganizationTableWithMembers(LiteLLM_OrganizationTable): 

3542 """Returned by the /organization/info endpoint and /organization/list endpoint""" 

3543 

3544 members: list[LiteLLM_OrganizationMembershipTable] = [] 

3545 teams: list[LiteLLM_TeamTable] = [] 

3546 litellm_budget_table: LiteLLM_BudgetTable | None = None 

3547 created_at: datetime 

3548 updated_at: datetime 

3549 

3550 

3551class NewOrganizationResponse(LiteLLM_OrganizationTable): 

3552 organization_id: str 

3553 created_at: datetime 

3554 updated_at: datetime 

3555 

3556 

3557### PROJECT MANAGEMENT TYPES ### 

3558 

3559 

3560class ProjectBase(LiteLLMPydanticObjectBase): 

3561 """Base fields shared by project create/update requests""" 

3562 

3563 project_id: str | None = None 

3564 project_alias: str | None = None 

3565 team_id: str | None = None 

3566 metadata: dict | None = None 

3567 models: list[str] | None = None 

3568 blocked: bool = False 

3569 

3570 

3571class NewProjectRequest(LiteLLM_BudgetTable): 

3572 """Request model for POST /project/new""" 

3573 

3574 project_id: str | None = None 

3575 project_alias: str | None = None 

3576 description: str | None = None 

3577 team_id: str 

3578 budget_id: str | None = None 

3579 metadata: dict | None = None 

3580 tags: list[str] | None = None 

3581 guardrails: list[str] | None = None 

3582 policies: list[str] | None = None 

3583 models: list[str] = [] 

3584 model_rpm_limit: dict | None = None 

3585 model_tpm_limit: dict | None = None 

3586 model_itpm_limit: Mapping[str, int] | None = None 

3587 model_otpm_limit: Mapping[str, int] | None = None 

3588 blocked: bool = False 

3589 object_permission: LiteLLM_ObjectPermissionBase | None = None 

3590 

3591 @model_validator(mode="before") 

3592 @classmethod 

3593 def set_model_info(cls, values): 

3594 if "tags" in values and values["tags"] is not None: 

3595 if not isinstance(values["tags"], list): 

3596 raise ValueError(f"tags must be a list of strings, got {type(values['tags']).__name__}") 

3597 for field in LiteLLM_ManagementEndpoint_MetadataFields: 

3598 if values.get(field) is not None: 

3599 if values.get("metadata") is None: 

3600 values.update({"metadata": {}}) 

3601 values["metadata"][field] = values.get(field) 

3602 values.pop(field) 

3603 return values 

3604 

3605 

3606class UpdateProjectRequest(LiteLLM_BudgetTable): 

3607 """Request model for POST /project/update""" 

3608 

3609 project_id: str 

3610 project_alias: str | None = None 

3611 description: str | None = None 

3612 team_id: str | None = None 

3613 metadata: dict | None = None 

3614 tags: list[str] | None = None 

3615 guardrails: list[str] | None = None 

3616 policies: list[str] | None = None 

3617 models: list[str] | None = None 

3618 model_rpm_limit: dict | None = None 

3619 model_tpm_limit: dict | None = None 

3620 model_itpm_limit: Mapping[str, int] | None = None 

3621 model_otpm_limit: Mapping[str, int] | None = None 

3622 blocked: bool | None = None 

3623 budget_id: str | None = None 

3624 object_permission: LiteLLM_ObjectPermissionBase | None = None 

3625 

3626 @model_validator(mode="before") 

3627 @classmethod 

3628 def set_model_info(cls, values): 

3629 if "tags" in values and values["tags"] is not None: 

3630 if not isinstance(values["tags"], list): 

3631 raise ValueError(f"tags must be a list of strings, got {type(values['tags']).__name__}") 

3632 for field in LiteLLM_ManagementEndpoint_MetadataFields: 

3633 if values.get(field) is not None: 

3634 if values.get("metadata") is None: 

3635 values.update({"metadata": {}}) 

3636 values["metadata"][field] = values.get(field) 

3637 values.pop(field) 

3638 return values 

3639 

3640 

3641class DeleteProjectRequest(LiteLLMPydanticObjectBase): 

3642 """Request model for DELETE /project/delete""" 

3643 

3644 project_ids: list[str] 

3645 

3646 

3647from litellm.models.project import ( # noqa: E402 

3648 LiteLLM_ProjectTable as LiteLLM_ProjectTable, 

3649) 

3650 

3651 

3652class NewProjectResponse(LiteLLM_ProjectTable): 

3653 """Response model for POST /project/new""" 

3654 

3655 project_id: str 

3656 created_at: datetime 

3657 updated_at: datetime 

3658 

3659 

3660class LiteLLM_ProjectTableCachedObj(LiteLLM_ProjectTable): 

3661 """Cached version for auth checks. Mirrors LiteLLM_TeamTableCachedObj pattern.""" 

3662 

3663 last_refreshed_at: float | None = None 

3664 

3665 

3666class LiteLLM_UserTableFiltered(BaseModel): # done to avoid exposing sensitive data 

3667 user_id: str 

3668 user_email: str | None = None 

3669 

3670 

3671class LiteLLM_UserTableWithKeyCount(LiteLLM_UserTable): 

3672 key_count: int = 0 

3673 

3674 

3675from litellm.models.access_group import ( # noqa: E402 

3676 LiteLLM_AccessGroupTable as LiteLLM_AccessGroupTable, 

3677) 

3678from litellm.models.end_user import ( # noqa: E402 

3679 LiteLLM_EndUserTable as LiteLLM_EndUserTable, 

3680) 

3681from litellm.models.spend_logs import ( # noqa: E402 

3682 LiteLLM_ErrorLogs as LiteLLM_ErrorLogs, 

3683) 

3684from litellm.models.spend_logs import ( # noqa: E402 

3685 LiteLLM_SpendLogs as LiteLLM_SpendLogs, 

3686) 

3687from litellm.models.tag import LiteLLM_TagTable as LiteLLM_TagTable # noqa: E402 

3688 

3689AUDIT_ACTIONS = Literal["created", "updated", "deleted", "blocked", "unblocked", "rotated", "kill_switch_fired"] 

3690 

3691 

3692class LiteLLM_AuditLogs(LiteLLMPydanticObjectBase): 

3693 id: str 

3694 updated_at: datetime 

3695 changed_by: Any | None = None 

3696 changed_by_api_key: str | None = None 

3697 action: AUDIT_ACTIONS 

3698 table_name: LitellmTableNames 

3699 object_id: str 

3700 before_value: Json | None = None 

3701 updated_values: Json | None = None 

3702 

3703 @model_validator(mode="before") 

3704 @classmethod 

3705 def cast_changed_by_to_str(cls, values): 

3706 if values.get("changed_by") is not None: 

3707 values["changed_by"] = str(values["changed_by"]) 

3708 return values 

3709 

3710 @model_validator(mode="after") 

3711 def mask_api_keys(self): 

3712 from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker 

3713 

3714 masker: Final = SensitiveDataMasker(sensitive_patterns={"key"}) 

3715 

3716 if self.before_value is not None: 

3717 json_before_value: dict | None = None 

3718 if isinstance(self.before_value, str): 

3719 json_before_value = json.loads(self.before_value) 

3720 elif isinstance(self.before_value, dict): 

3721 json_before_value = self.before_value 

3722 

3723 if json_before_value is not None: 

3724 json_before_value = masker.mask_dict(json_before_value) 

3725 self.before_value = json.dumps(json_before_value, default=str) 

3726 

3727 if self.updated_values is not None: 

3728 json_updated_values: dict | None = None 

3729 if isinstance(self.updated_values, str): 

3730 json_updated_values = json.loads(self.updated_values) 

3731 elif isinstance(self.updated_values, dict): 

3732 json_updated_values = self.updated_values 

3733 

3734 if json_updated_values is not None: 

3735 json_updated_values = masker.mask_dict(json_updated_values) 

3736 self.updated_values = json.dumps(json_updated_values, default=str) 

3737 

3738 return self 

3739 

3740 

3741class LiteLLM_SpendLogs_ResponseObject(LiteLLMPydanticObjectBase): 

3742 response: list[LiteLLM_SpendLogs | Any] | None = None 

3743 

3744 

3745class TokenCountRequest(LiteLLMPydanticObjectBase): 

3746 model: str 

3747 prompt: str | None = None 

3748 messages: list[dict] | None = None 

3749 """ 

3750 Anthropic token counting endpoint uses /messages 

3751 """ 

3752 

3753 contents: list[dict] | None = None 

3754 """ 

3755 Google /countTokens endpoint expects contents to be a list of dicts with the following structure: 

3756 """ 

3757 

3758 tools: list[dict] | None = None 

3759 system: Any | None = None 

3760 

3761 

3762class CallInfo(LiteLLMPydanticObjectBase): 

3763 """Used for slack budget alerting""" 

3764 

3765 spend: float 

3766 max_budget: float | None = None 

3767 soft_budget: float | None = None 

3768 token: str | None = Field(default=None, description="Hashed value of that key") 

3769 customer_id: str | None = None 

3770 user_id: str | None = None 

3771 team_id: str | None = None 

3772 team_alias: str | None = None 

3773 organization_id: str | None = None 

3774 user_email: str | None = None 

3775 key_alias: str | None = None 

3776 projected_exceeded_date: str | None = None 

3777 projected_spend: float | None = None 

3778 event_group: Litellm_EntityType 

3779 alert_emails: list[str] | None = Field( 

3780 default=None, 

3781 description="Additional email addresses to send alerts to (e.g., from team metadata)", 

3782 ) 

3783 max_budget_alert_emails: dict[str, list[str]] | None = Field( 

3784 default=None, 

3785 description="Map of threshold percentage to email recipients (e.g., {'50': ['a@co.com'], '75': ['a@co.com', 'b@co.com']})", 

3786 ) 

3787 

3788 

3789class WebhookEvent(CallInfo): 

3790 event: Literal[ 

3791 "budget_crossed", 

3792 "max_budget_alert", 

3793 "soft_budget_crossed", 

3794 "threshold_crossed", 

3795 "projected_limit_exceeded", 

3796 "key_created", 

3797 "key_rotated", 

3798 "internal_user_created", 

3799 "spend_tracked", 

3800 ] 

3801 event_message: str # human-readable description of event 

3802 event_group: Litellm_EntityType 

3803 

3804 

3805class SpecialModelNames(enum.Enum): 

3806 all_team_models = "all-team-models" 

3807 all_proxy_models = "all-proxy-models" 

3808 no_default_models = "no-default-models" 

3809 

3810 

3811class SpecialMCPServerNames(enum.Enum): 

3812 no_mcp_servers = "no-mcp-servers" 

3813 

3814 

3815class SpecialProxyStrings(enum.Enum): 

3816 default_user_id = "default_user_id" # global proxy admin 

3817 

3818 

3819class InvitationNew(LiteLLMPydanticObjectBase): 

3820 user_id: str 

3821 

3822 

3823class InvitationUpdate(LiteLLMPydanticObjectBase): 

3824 invitation_id: str 

3825 is_accepted: bool 

3826 

3827 

3828class InvitationDelete(LiteLLMPydanticObjectBase): 

3829 invitation_id: str 

3830 

3831 

3832class InvitationModel(LiteLLMPydanticObjectBase): 

3833 id: str 

3834 user_id: str 

3835 is_accepted: bool 

3836 accepted_at: datetime | None 

3837 expires_at: datetime 

3838 created_at: datetime 

3839 created_by: str 

3840 updated_at: datetime 

3841 updated_by: str 

3842 

3843 

3844class InvitationClaim(LiteLLMPydanticObjectBase): 

3845 invitation_link: str 

3846 user_id: str 

3847 password: str 

3848 

3849 

3850class ConfigFieldInfo(LiteLLMPydanticObjectBase): 

3851 field_name: str 

3852 field_value: Any 

3853 source: Literal["config", "db", "env", "default", "unset"] = "unset" 

3854 editable: bool = True 

3855 

3856 

3857class CallbackOnUI(LiteLLMPydanticObjectBase): 

3858 litellm_callback_name: str 

3859 litellm_callback_params: list | None 

3860 ui_callback_name: str 

3861 

3862 

3863class AllCallbacks(LiteLLMPydanticObjectBase): 

3864 langfuse: CallbackOnUI = CallbackOnUI( 

3865 litellm_callback_name="langfuse", 

3866 ui_callback_name="Langfuse", 

3867 litellm_callback_params=[ 

3868 "LANGFUSE_PUBLIC_KEY", 

3869 "LANGFUSE_SECRET_KEY", 

3870 "LANGFUSE_HOST", 

3871 ], 

3872 ) 

3873 

3874 otel: CallbackOnUI = CallbackOnUI( 

3875 litellm_callback_name="otel", 

3876 ui_callback_name="OpenTelemetry", 

3877 litellm_callback_params=[ 

3878 "OTEL_EXPORTER", 

3879 "OTEL_EXPORTER_OTLP_PROTOCOL", 

3880 "OTEL_ENDPOINT", 

3881 "OTEL_TRACES_ENDPOINT", 

3882 "OTEL_HEADERS", 

3883 ], 

3884 ) 

3885 

3886 s3: CallbackOnUI = CallbackOnUI( 

3887 litellm_callback_name="s3", 

3888 ui_callback_name="s3 Bucket (AWS)", 

3889 litellm_callback_params=[ 

3890 "AWS_ACCESS_KEY_ID", 

3891 "AWS_SECRET_ACCESS_KEY", 

3892 "AWS_REGION_NAME", 

3893 "S3_LOG_PROMPTS_ONLY", 

3894 ], 

3895 ) 

3896 

3897 azure_sentinel: CallbackOnUI = CallbackOnUI( 

3898 litellm_callback_name="azure_sentinel", 

3899 ui_callback_name="Azure Sentinel", 

3900 litellm_callback_params=[ 

3901 "AZURE_SENTINEL_DCR_IMMUTABLE_ID", 

3902 "AZURE_SENTINEL_ENDPOINT", 

3903 "AZURE_SENTINEL_TENANT_ID", 

3904 "AZURE_SENTINEL_CLIENT_ID", 

3905 "AZURE_SENTINEL_CLIENT_SECRET", 

3906 "AZURE_SENTINEL_STREAM_NAME", 

3907 ], 

3908 ) 

3909 

3910 openmeter: CallbackOnUI = CallbackOnUI( 

3911 litellm_callback_name="openmeter", 

3912 ui_callback_name="OpenMeter", 

3913 litellm_callback_params=[ 

3914 "OPENMETER_API_ENDPOINT", 

3915 "OPENMETER_API_KEY", 

3916 ], 

3917 ) 

3918 

3919 custom_callback_api: CallbackOnUI = CallbackOnUI( 

3920 litellm_callback_name="custom_callback_api", 

3921 litellm_callback_params=["GENERIC_LOGGER_ENDPOINT", "GENERIC_LOGGER_HEADERS"], 

3922 ui_callback_name="Custom Callback API", 

3923 ) 

3924 

3925 generic_api: CallbackOnUI = CallbackOnUI( 

3926 litellm_callback_name="generic_api", 

3927 litellm_callback_params=["GENERIC_LOGGER_ENDPOINT", "GENERIC_LOGGER_HEADERS"], 

3928 ui_callback_name="Custom Callback API", 

3929 ) 

3930 

3931 datadog: CallbackOnUI = CallbackOnUI( 

3932 litellm_callback_name="datadog", 

3933 litellm_callback_params=["DD_API_KEY", "DD_SITE"], 

3934 ui_callback_name="Datadog", 

3935 ) 

3936 

3937 braintrust: CallbackOnUI = CallbackOnUI( 

3938 litellm_callback_name="braintrust", 

3939 litellm_callback_params=["BRAINTRUST_API_KEY", "BRAINTRUST_API_BASE"], 

3940 ui_callback_name="Braintrust", 

3941 ) 

3942 

3943 langsmith: CallbackOnUI = CallbackOnUI( 

3944 litellm_callback_name="langsmith", 

3945 litellm_callback_params=[ 

3946 "LANGSMITH_API_KEY", 

3947 "LANGSMITH_PROJECT", 

3948 "LANGSMITH_DEFAULT_RUN_NAME", 

3949 ], 

3950 ui_callback_name="Langsmith", 

3951 ) 

3952 

3953 lago: CallbackOnUI = CallbackOnUI( 

3954 litellm_callback_name="lago", 

3955 litellm_callback_params=[ 

3956 "LAGO_API_BASE", 

3957 "LAGO_API_KEY", 

3958 "LAGO_API_EVENT_CODE", 

3959 "LAGO_API_CHARGE_BY", 

3960 ], 

3961 ui_callback_name="Lago Billing", 

3962 ) 

3963 

3964 traceloop: CallbackOnUI = CallbackOnUI( 

3965 litellm_callback_name="traceloop", 

3966 litellm_callback_params=[ 

3967 "TRACELOOP_API_KEY", 

3968 ], 

3969 ui_callback_name="Traceloop", 

3970 ) 

3971 

3972 galileo: CallbackOnUI = CallbackOnUI( 

3973 litellm_callback_name="galileo", 

3974 litellm_callback_params=[ 

3975 "GALILEO_API_KEY", 

3976 "GALILEO_PROJECT_ID", 

3977 "GALILEO_LOG_STREAM_ID", 

3978 "GALILEO_BASE_URL", 

3979 "GALILEO_USERNAME", 

3980 "GALILEO_PASSWORD", 

3981 ], 

3982 ui_callback_name="Galileo", 

3983 ) 

3984 

3985 newrelic: CallbackOnUI = CallbackOnUI( 

3986 litellm_callback_name="newrelic", 

3987 ui_callback_name="New Relic", 

3988 litellm_callback_params=[ 

3989 "NEW_RELIC_AI_MONITORING_RECORD_CONTENT_ENABLED", 

3990 ], 

3991 ) 

3992 

3993 pointfive: CallbackOnUI = CallbackOnUI( 

3994 litellm_callback_name="pointfive", 

3995 ui_callback_name="PointFive", 

3996 litellm_callback_params=[ # mutable-ok: the registry field is typed list 

3997 "POINTFIVE_API_KEY", 

3998 "POINTFIVE_API_URL", 

3999 ], 

4000 ) 

4001 

4002 

4003class HTTPExceptionErrorDetail(TypedDict): 

4004 """The `{"error": <message>}` shape most proxy endpoints raise as `HTTPException.detail`.""" 

4005 

4006 error: ReadOnly[str] 

4007 

4008 

4009class SpendLogsRouterMetadata(TypedDict): 

4010 """ 

4011 Router provenance stamped on spend logs for deployments flagged with 

4012 model_info.internal_router_model, correlating the requested model group 

4013 with the provider deployment that served the call 

4014 """ 

4015 

4016 requested_model: ReadOnly[str | None] 

4017 selected_model: ReadOnly[str | None] 

4018 selected_provider: ReadOnly[str | None] 

4019 router_correlation_id: ReadOnly[str | None] 

4020 

4021 

4022class SpendLogsMetadata(TypedDict): 

4023 autorouter_baseline_observation: ReadOnly[str | None] 

4024 """ 

4025 Specific metadata k,v pairs logged to spendlogs for easier cost tracking 

4026 """ 

4027 

4028 additional_usage_values: dict | None # covers provider-specific usage information - e.g. prompt caching 

4029 user_api_key: str | None 

4030 user_api_key_alias: str | None 

4031 user_api_key_team_id: str | None 

4032 user_api_key_project_id: str | None 

4033 user_api_key_project_alias: str | None 

4034 user_api_key_org_id: str | None 

4035 user_api_key_user_id: str | None 

4036 user_api_key_team_alias: str | None 

4037 spend_logs_metadata: dict | None # special param to log k,v pairs to spendlogs for a call 

4038 requester_ip_address: str | None 

4039 user_agent: ReadOnly[str | None] 

4040 litellm_call_id: str | None 

4041 applied_guardrails: list[str] | None 

4042 mcp_tool_call_metadata: StandardLoggingMCPToolCall | None 

4043 vector_store_request_metadata: list[StandardLoggingVectorStoreRequest] | None 

4044 routing_decision: StandardLoggingRoutingDecision | None 

4045 internal_call_origin: InternalCallOrigin | None 

4046 guardrail_information: list[StandardLoggingGuardrailInformation] | None 

4047 eval_information: Any | None 

4048 status: StandardLoggingPayloadStatus 

4049 proxy_server_request: str | None 

4050 batch_models: list[str] | None 

4051 batch_successful_requests: int | None # writable-ok: built by assignment like every sibling key in this TypedDict 

4052 batch_failed_requests: int | None # writable-ok: built by assignment like every sibling key in this TypedDict 

4053 error_information: StandardLoggingPayloadErrorInformation | None 

4054 usage_object: dict | None 

4055 model_map_information: StandardLoggingModelInformation | None 

4056 cold_storage_object_key: str | None # S3/GCS object key for cold storage retrieval 

4057 litellm_overhead_time_ms: float | None # LiteLLM overhead time in milliseconds 

4058 attempted_retries: int | None # Number of retries attempted (0 = first attempt succeeded) 

4059 max_retries: int | None # Max retries configured for this request 

4060 attempted_fallbacks: ReadOnly[int | None] # Number of fallbacks attempted (0 = primary model group served) 

4061 original_model_group: ReadOnly[str | None] # Model group requested before any fallbacks 

4062 cost_breakdown: CostBreakdown | None # Detailed cost breakdown (input_cost, output_cost, margin, discount, etc.) 

4063 compression_savings: CompressionSavingsMetadata | None 

4064 autorouter_savings: ReadOnly[float | None] 

4065 autorouter_savings_estimate: ReadOnly[Mapping[str, JsonValue] | None] 

4066 litellm_gateway_injected_cache: ReadOnly[str | None] 

4067 router_metadata: ReadOnly[SpendLogsRouterMetadata | None] # None = deployment not flagged internal_router_model 

4068 azure_spillover: ReadOnly[AzureSpillover | None] # None = Azure did not report spillover 

4069 

4070 

4071class SpendLogsPayload(TypedDict): 

4072 request_id: str 

4073 call_type: str 

4074 api_key: str 

4075 spend: float 

4076 total_tokens: int 

4077 prompt_tokens: int 

4078 completion_tokens: int 

4079 startTime: datetime | str 

4080 endTime: datetime | str 

4081 completionStartTime: datetime | str | None 

4082 model: str 

4083 model_id: str | None 

4084 model_group: str | None 

4085 mcp_namespaced_tool_name: str | None 

4086 agent_id: str | None 

4087 api_base: str 

4088 user: str 

4089 metadata: str # json str 

4090 cache_hit: str 

4091 cache_key: str 

4092 request_tags: str # json str 

4093 team_id: str | None 

4094 organization_id: str | None 

4095 end_user: str | None 

4096 requester_ip_address: str | None 

4097 custom_llm_provider: str | None 

4098 messages: str | list | dict | None 

4099 response: str | list | dict | None 

4100 proxy_server_request: str | None 

4101 session_id: str | None 

4102 request_duration_ms: int | None 

4103 status: Literal["success", "failure"] 

4104 litellm_call_id: ReadOnly[str | None] 

4105 

4106 

4107class SpanAttributes(str, enum.Enum): 

4108 # Note: We've taken this from opentelemetry-semantic-conventions-ai 

4109 # I chose to not add a new dependency to litellm for this 

4110 

4111 # Semantic Conventions for LLM requests, this needs to be removed after 

4112 # OpenTelemetry Semantic Conventions support Gen AI. 

4113 # Issue at https://github.com/open-telemetry/opentelemetry-python/issues/3868 

4114 # Refer to https://github.com/open-telemetry/semantic-conventions/blob/main/docs/gen-ai/llm-spans.md 

4115 

4116 LLM_SYSTEM = "gen_ai.system" 

4117 LLM_REQUEST_MODEL = "gen_ai.request.model" 

4118 LLM_REQUEST_MAX_TOKENS = "gen_ai.request.max_tokens" 

4119 LLM_REQUEST_TEMPERATURE = "gen_ai.request.temperature" 

4120 LLM_REQUEST_TOP_P = "gen_ai.request.top_p" 

4121 LLM_PROMPTS = "gen_ai.prompt" 

4122 LLM_COMPLETIONS = "gen_ai.completion" 

4123 LLM_RESPONSE_MODEL = "gen_ai.response.model" 

4124 LLM_USAGE_COMPLETION_TOKENS = "gen_ai.usage.completion_tokens" 

4125 LLM_USAGE_PROMPT_TOKENS = "gen_ai.usage.prompt_tokens" 

4126 

4127 # OTEL 1.38 attributes 

4128 GEN_AI_INPUT_MESSAGES = "gen_ai.input.messages" 

4129 GEN_AI_OUTPUT_MESSAGES = "gen_ai.output.messages" 

4130 GEN_AI_USAGE_INPUT_TOKENS = "gen_ai.usage.input_tokens" 

4131 GEN_AI_USAGE_OUTPUT_TOKENS = "gen_ai.usage.output_tokens" 

4132 GEN_AI_USAGE_TOTAL_TOKENS = "gen_ai.usage.total_tokens" 

4133 GEN_AI_OPERATION_NAME = "gen_ai.operation.name" 

4134 GEN_AI_REQUEST_ID = "gen_ai.request.id" 

4135 GEN_AI_SYSTEM_INSTRUCTIONS = "gen_ai.system_instructions" 

4136 GEN_AI_RESPONSE_FINISH_REASONS = "gen_ai.response.finish_reasons" 

4137 

4138 LLM_TOKEN_TYPE = "gen_ai.token.type" 

4139 # To be added 

4140 # LLM_RESPONSE_FINISH_REASON = "gen_ai.response.finish_reasons" 

4141 # LLM_RESPONSE_ID = "gen_ai.response.id" 

4142 

4143 # LLM 

4144 LLM_REQUEST_TYPE = "llm.request.type" 

4145 LLM_USAGE_TOTAL_TOKENS = "llm.usage.total_tokens" 

4146 LLM_USAGE_TOKEN_TYPE = "llm.usage.token_type" 

4147 LLM_USER = "llm.user" 

4148 LLM_HEADERS = "llm.headers" 

4149 LLM_TOP_K = "llm.top_k" 

4150 LLM_IS_STREAMING = "llm.is_streaming" 

4151 LLM_FREQUENCY_PENALTY = "llm.frequency_penalty" 

4152 LLM_PRESENCE_PENALTY = "llm.presence_penalty" 

4153 LLM_CHAT_STOP_SEQUENCES = "llm.chat.stop_sequences" 

4154 LLM_REQUEST_FUNCTIONS = "llm.request.functions" 

4155 LLM_REQUEST_REPETITION_PENALTY = "llm.request.repetition_penalty" 

4156 LLM_RESPONSE_FINISH_REASON = "llm.response.finish_reason" 

4157 LLM_RESPONSE_STOP_REASON = "llm.response.stop_reason" 

4158 LLM_CONTENT_COMPLETION_CHUNK = "llm.content.completion.chunk" 

4159 

4160 # OpenAI 

4161 LLM_OPENAI_RESPONSE_SYSTEM_FINGERPRINT = "gen_ai.openai.system_fingerprint" 

4162 LLM_OPENAI_API_BASE = "gen_ai.openai.api_base" 

4163 LLM_OPENAI_API_VERSION = "gen_ai.openai.api_version" 

4164 LLM_OPENAI_API_TYPE = "gen_ai.openai.api_type" 

4165 

4166 

4167class ManagementEndpointLoggingPayload(LiteLLMPydanticObjectBase): 

4168 route: str 

4169 request_data: dict 

4170 response: dict | None = None 

4171 exception: Any | None = None 

4172 start_time: datetime | None = None 

4173 end_time: datetime | None = None 

4174 

4175 

4176class ProxyException(Exception): 

4177 # NOTE: DO NOT MODIFY THIS 

4178 # This is used to map exactly to OPENAI Exceptions 

4179 def __init__( 

4180 self, 

4181 message: str, 

4182 type: str, 

4183 param: str | None, 

4184 code: int | str | None = None, # maps to status code 

4185 headers: dict[str, str] | None = None, 

4186 openai_code: str | None = None, # maps to 'code' in openai 

4187 provider_specific_fields: dict | None = None, 

4188 ): 

4189 self.message = str(message) 

4190 super().__init__(self.message) 

4191 self.type = type 

4192 self.param = param 

4193 self.openai_code = openai_code or code 

4194 # If we look on official python OpenAI lib, the code should be a string: 

4195 # https://github.com/openai/openai-python/blob/195c05a64d39c87b2dfdf1eca2d339597f1fce03/src/openai/types/shared/error_object.py#L11 

4196 # Related LiteLLM issue: https://github.com/BerriAI/litellm/discussions/4834 

4197 self.code = str(code) 

4198 if headers is not None: 

4199 for k, v in headers.items(): 

4200 if not isinstance(v, str): 4200 ↛ 4201line 4200 didn't jump to line 4201 because the condition on line 4200 was never true

4201 headers[k] = str(v) 

4202 self.headers = headers or {} 

4203 self.provider_specific_fields = provider_specific_fields 

4204 # rules for proxyExceptions 

4205 # Litellm router.py returns "No healthy deployment available" when there are no deployments available 

4206 # Should map to 429 errors https://github.com/BerriAI/litellm/issues/2487 

4207 if "No healthy deployment available" in self.message or "No deployments available" in self.message: 4207 ↛ 4208line 4207 didn't jump to line 4208 because the condition on line 4207 was never true

4208 self.code = "429" 

4209 elif RouterErrors.no_deployments_with_tag_routing.value in self.message: 4209 ↛ 4210line 4209 didn't jump to line 4210 because the condition on line 4209 was never true

4210 self.code = "401" 

4211 

4212 def to_dict(self) -> dict: 

4213 """Converts the ProxyException instance to a dictionary.""" 

4214 error_dict: Final[dict[str, str | dict | None]] = { 

4215 "message": self.message, 

4216 "type": self.type, 

4217 "param": self.param, 

4218 "code": self.code, 

4219 } 

4220 if self.provider_specific_fields: 

4221 error_dict["provider_specific_fields"] = self.provider_specific_fields 

4222 return error_dict 

4223 

4224 

4225class ModelAccessDeniedProxyException(ProxyException): 

4226 def __init__( 

4227 self, 

4228 message: str, 

4229 internal_message: str, 

4230 type: str, 

4231 param: str | None, 

4232 code: int | str | None, 

4233 ) -> None: 

4234 super().__init__(message=message, type=type, param=param, code=code) 

4235 self.internal_message: Final = internal_message 

4236 

4237 def sanitized_internal_message(self) -> str: 

4238 return self.internal_message.replace("\r", "").replace("\n", "") 

4239 

4240 

4241class CommonProxyErrors(str, enum.Enum): 

4242 db_not_connected_error = ( 

4243 "DB not connected. This endpoint needs a database; set DATABASE_URL to a " 

4244 "PostgreSQL connection string (postgresql://...) to enable it. " 

4245 "See https://docs.litellm.ai/docs/proxy/virtual_keys" 

4246 ) 

4247 no_llm_router = "No models configured on proxy" 

4248 not_allowed_access = "Admin-only endpoint. Not allowed to access this." 

4249 not_premium_user = "You must be a LiteLLM Enterprise user to use this feature. If you have a license please set `LITELLM_LICENSE` in your env. Get a 7 day trial key here: https://www.litellm.ai/enterprise#trial. \nPricing: https://www.litellm.ai/#pricing" 

4250 max_parallel_request_limit_reached = "Crossed TPM / RPM / Max Parallel Request Limit" 

4251 missing_enterprise_package = "Missing litellm-enterprise package. Please install it to use this feature. Run `pip install litellm-enterprise`" 

4252 missing_enterprise_package_docker = "This uses the enterprise folder - only available on the Docker image." 

4253 

4254 

4255class SpendCalculateRequest(LiteLLMPydanticObjectBase): 

4256 model: str | None = None 

4257 messages: list | None = None 

4258 completion_response: dict | None = None 

4259 

4260 

4261class ProxyErrorTypes(str, enum.Enum): 

4262 budget_exceeded = "budget_exceeded" 

4263 """ 

4264 Object was over budget 

4265 """ 

4266 no_db_connection = "no_db_connection" 

4267 """ 

4268 No database connection 

4269 """ 

4270 

4271 token_not_found_in_db = "token_not_found_in_db" 

4272 """ 

4273 Requested token was not found in the database 

4274 """ 

4275 

4276 key_model_access_denied = "key_model_access_denied" 

4277 """ 

4278 Key does not have access to the model 

4279 """ 

4280 

4281 team_model_access_denied = "team_model_access_denied" 

4282 """ 

4283 Team does not have access to the model 

4284 """ 

4285 

4286 user_model_access_denied = "user_model_access_denied" 

4287 """ 

4288 User does not have access to the model 

4289 """ 

4290 

4291 org_model_access_denied = "org_model_access_denied" 

4292 """ 

4293 Organization does not have access to the model 

4294 """ 

4295 

4296 project_model_access_denied = "project_model_access_denied" 

4297 """ 

4298 Project does not have access to the model 

4299 """ 

4300 

4301 agent_model_access_denied = "agent_model_access_denied" 

4302 """ 

4303 The agent behind the key does not have access to the model 

4304 """ 

4305 

4306 model_cost_map_missing = "model_cost_map_missing" 

4307 

4308 expired_key = "expired_key" 

4309 """ 

4310 Key has expired 

4311 """ 

4312 

4313 auth_error = "auth_error" 

4314 """ 

4315 General authentication error 

4316 """ 

4317 

4318 auth_provider_unavailable = "auth_provider_unavailable" 

4319 """ 

4320 The identity provider needed to authenticate the request (e.g. its JWKS endpoint) is unreachable 

4321 """ 

4322 

4323 internal_server_error = "internal_server_error" 

4324 """ 

4325 Internal server error 

4326 """ 

4327 

4328 bad_request_error = "bad_request_error" 

4329 """ 

4330 Bad request error 

4331 """ 

4332 

4333 not_found_error = "not_found_error" 

4334 """ 

4335 Not found error 

4336 """ 

4337 

4338 validation_error = "validation_error" 

4339 """ 

4340 Validation error 

4341 """ 

4342 

4343 cache_ping_error = "cache_ping_error" 

4344 """ 

4345 Cache ping error 

4346 """ 

4347 

4348 team_member_permission_error = "team_member_permission_error" 

4349 """ 

4350 Team member permission error 

4351 """ 

4352 

4353 key_vector_store_access_denied = "key_vector_store_access_denied" 

4354 """ 

4355 Key does not have access to the vector store 

4356 """ 

4357 

4358 team_vector_store_access_denied = "team_vector_store_access_denied" 

4359 """ 

4360 Team does not have access to the vector store 

4361 """ 

4362 

4363 org_vector_store_access_denied = "org_vector_store_access_denied" 

4364 """ 

4365 Organization does not have access to the vector store 

4366 """ 

4367 

4368 team_member_already_in_team = "team_member_already_in_team" 

4369 """ 

4370 Team member is already in team 

4371 """ 

4372 

4373 tool_access_denied = "tool_access_denied" 

4374 """ 

4375 Tool is not in the allowed tools list for this key/team 

4376 """ 

4377 

4378 @classmethod 

4379 def get_model_access_error_type_for_object( 

4380 cls, object_type: Literal["key", "user", "team", "org", "project", "agent"] 

4381 ) -> "ProxyErrorTypes": 

4382 """ 

4383 Get the model access error type for object_type 

4384 """ 

4385 if object_type == "key": 

4386 return cls.key_model_access_denied 

4387 elif object_type == "team": 

4388 return cls.team_model_access_denied 

4389 elif object_type == "user": 

4390 return cls.user_model_access_denied 

4391 elif object_type == "org": 

4392 return cls.org_model_access_denied 

4393 elif object_type == "project": 

4394 return cls.project_model_access_denied 

4395 elif object_type == "agent": 

4396 return cls.agent_model_access_denied 

4397 

4398 @classmethod 

4399 def get_vector_store_access_error_type_for_object( 

4400 cls, object_type: Literal["key", "team", "org"] 

4401 ) -> "ProxyErrorTypes": 

4402 """ 

4403 Get the vector store access error type for object_type 

4404 """ 

4405 if object_type == "key": 

4406 return cls.key_vector_store_access_denied 

4407 elif object_type == "team": 

4408 return cls.team_vector_store_access_denied 

4409 elif object_type == "org": 

4410 return cls.org_vector_store_access_denied 

4411 

4412 

4413DB_CONNECTION_ERROR_TYPES: Final = ( 

4414 httpx.ConnectError, 

4415 httpx.ConnectTimeout, 

4416 httpx.ReadError, 

4417 httpx.ReadTimeout, 

4418) 

4419 

4420# What a NON-IDEMPOTENT write (increment upsert) may retry: only ConnectError 

4421# proves the statements never reached the database. Post-send errors are 

4422# ambiguous; a stalled statement can leave its transaction open on the pooled 

4423# connection, where a retry stacks a second increment set into the same commit. 

4424# Idempotent writes (create_many with skip_duplicates) may retry the full tuple. 

4425DB_RETRY_SAFE_ERROR_TYPES: Final = (httpx.ConnectError,) 

4426 

4427 

4428class SSOUserDefinedValues(TypedDict): 

4429 models: list[str] 

4430 user_id: str 

4431 user_email: str | None 

4432 user_role: str | None 

4433 max_budget: float | None 

4434 budget_duration: str | None 

4435 

4436 

4437class VirtualKeyEvent(LiteLLMPydanticObjectBase): 

4438 created_by_user_id: str 

4439 created_by_user_role: str 

4440 created_by_key_alias: str | None 

4441 request_kwargs: dict 

4442 

4443 

4444class CreatePassThroughEndpoint(LiteLLMPydanticObjectBase): 

4445 path: str 

4446 target: str 

4447 headers: dict 

4448 

4449 

4450from litellm.models.team_membership import ( # noqa: E402 

4451 LiteLLM_TeamMembership as LiteLLM_TeamMembership, 

4452) 

4453 

4454#### Organization / Team Member Requests #### 

4455 

4456 

4457class MemberAddRequest(LiteLLMPydanticObjectBase): 

4458 member: list[Member] | Member = Field( 

4459 description="Member object or list of member objects to add. Each member must include either user_id or user_email, and a role" 

4460 ) 

4461 

4462 def __init__(self, **data): 

4463 member_data: Final = data.get("member") 

4464 if isinstance(member_data, list): 

4465 # If member is a list of dictionaries, convert each dictionary to a Member object 

4466 members: Final = [Member(**item) if isinstance(item, dict) else item for item in member_data] 

4467 # Replace member_data with the list of Member objects 

4468 data["member"] = members 

4469 elif isinstance(member_data, dict): 

4470 # If member is a dictionary, convert it to a single Member object 

4471 member: Final = Member(**member_data) 

4472 # Replace member_data with the single Member object 

4473 data["member"] = member 

4474 # Call the superclass __init__ method to initialize the object 

4475 super().__init__(**data) 

4476 

4477 

4478class OrgMemberAddRequest(LiteLLMPydanticObjectBase): 

4479 member: list[OrgMember] | OrgMember 

4480 

4481 def __init__(self, **data): 

4482 member_data: Final = data.get("member") 

4483 if isinstance(member_data, list): 

4484 # If member is a list of dictionaries, convert each dictionary to a Member object 

4485 if all(isinstance(item, dict) for item in member_data): 4485 ↛ 4486line 4485 didn't jump to line 4486 because the condition on line 4485 was never true

4486 members = [OrgMember(**item) for item in member_data] 

4487 else: 

4488 members = [item for item in member_data] 

4489 # Replace member_data with the list of Member objects 

4490 data["member"] = members 

4491 elif isinstance(member_data, dict): 4491 ↛ 4493line 4491 didn't jump to line 4493 because the condition on line 4491 was never true

4492 # If member is a dictionary, convert it to a single Member object 

4493 member: Final = OrgMember(**member_data) 

4494 # Replace member_data with the single Member object 

4495 data["member"] = member 

4496 # Call the superclass __init__ method to initialize the object 

4497 super().__init__(**data) 

4498 

4499 

4500class TeamAddMemberResponse(LiteLLM_TeamTable): 

4501 updated_users: list[LiteLLM_UserTable] 

4502 updated_team_memberships: list[LiteLLM_TeamMembership] 

4503 

4504 

4505class OrganizationAddMemberResponse(LiteLLMPydanticObjectBase): 

4506 organization_id: str 

4507 updated_users: list[LiteLLM_UserTable] 

4508 updated_organization_memberships: list[LiteLLM_OrganizationMembershipTable] 

4509 

4510 

4511class MemberDeleteRequest(LiteLLMPydanticObjectBase): 

4512 user_id: str | None = None 

4513 user_email: str | None = None 

4514 

4515 @model_validator(mode="before") 

4516 @classmethod 

4517 def check_user_info(cls, values): 

4518 if values.get("user_id") is None and values.get("user_email") is None: 

4519 raise ValueError("Either user id or user email must be provided") 

4520 return values 

4521 

4522 

4523class MemberUpdateResponse(LiteLLMPydanticObjectBase): 

4524 user_id: str 

4525 user_email: str | None = None 

4526 

4527 

4528# Team Member Requests 

4529class TeamMemberAddRequest(MemberAddRequest): 

4530 """ 

4531 Request body for adding members to a team. 

4532 

4533 Example: 

4534 ```json 

4535 { 

4536 "team_id": "45e3e396-ee08-4a61-a88e-16b3ce7e0849", 

4537 "member": { 

4538 "role": "user", 

4539 "user_id": "user123" 

4540 }, 

4541 "max_budget_in_team": 100.0 

4542 } 

4543 ``` 

4544 """ 

4545 

4546 team_id: str = Field(description="The ID of the team to add the member to") 

4547 max_budget_in_team: float | None = Field( 

4548 default=None, 

4549 description="Maximum budget allocated to this user within the team. If not set, user has unlimited budget within team limits", 

4550 ) 

4551 budget_duration: str | None = Field( 

4552 default=None, 

4553 description="Duration after which this team member's budget resets (e.g. '1h', '24h', '7d', '30d'). If not set, the budget never resets.", 

4554 ) 

4555 allowed_models: list[str] | None = Field( 

4556 default=None, 

4557 description="List of models this team member can access. If not set, inherits the team's default_team_member_models or all team models.", 

4558 ) 

4559 

4560 

4561class TeamMemberDeleteRequest(MemberDeleteRequest): 

4562 team_id: str 

4563 

4564 

4565class TeamMemberUpdateRequest(TeamMemberDeleteRequest): 

4566 max_budget_in_team: float | None = None 

4567 role: Literal["admin", "user"] | None = None 

4568 tpm_limit: int | None = Field(default=None, description="Tokens per minute limit for this team member") 

4569 rpm_limit: int | None = Field(default=None, description="Requests per minute limit for this team member") 

4570 budget_duration: str | None = Field( 

4571 default=None, 

4572 description="Duration after which this team member's budget resets (e.g. '1h', '24h', '7d', '30d'). If not set, the budget never resets.", 

4573 ) 

4574 allowed_models: list[str] | None = Field( 

4575 default=None, 

4576 description="List of models this team member can access. Pass an empty list to remove per-member model restrictions.", 

4577 ) 

4578 temp_budget_increase: float | None = Field( 

4579 default=None, 

4580 ge=0, 

4581 allow_inf_nan=False, 

4582 description="Temporary additive budget increase for this team member, active until temp_budget_expiry", 

4583 ) 

4584 temp_budget_expiry: datetime | None = Field( 

4585 default=None, 

4586 description="UTC expiry for temp_budget_increase", 

4587 ) 

4588 

4589 @model_validator(mode="after") 

4590 def validate_temp_budget(self) -> "TeamMemberUpdateRequest": 

4591 if self.temp_budget_increase is not None or self.temp_budget_expiry is not None: 

4592 if self.temp_budget_increase is None or self.temp_budget_expiry is None: 

4593 raise ValueError("temp_budget_increase and temp_budget_expiry must be set together") 

4594 return self 

4595 

4596 

4597class TeamMemberUpdateResponse(MemberUpdateResponse): 

4598 team_id: str 

4599 max_budget_in_team: float | None = None 

4600 tpm_limit: int | None = None 

4601 rpm_limit: int | None = None 

4602 budget_duration: str | None = None 

4603 allowed_models: list[str] | None = None 

4604 temp_budget_increase: float | None = None 

4605 temp_budget_expiry: datetime | None = None 

4606 

4607 

4608class TeamModelAddRequest(BaseModel): 

4609 """Request to add models to a team""" 

4610 

4611 team_id: str 

4612 models: list[str] 

4613 

4614 

4615class TeamModelDeleteRequest(BaseModel): 

4616 """Request to delete models from a team""" 

4617 

4618 team_id: str 

4619 models: list[str] 

4620 

4621 

4622# Organization Member Requests 

4623class OrganizationMemberAddRequest(OrgMemberAddRequest): 

4624 organization_id: str 

4625 max_budget_in_organization: float | None = None # Users max budget within the organization 

4626 

4627 

4628class OrganizationMemberDeleteRequest(MemberDeleteRequest): 

4629 organization_id: str 

4630 

4631 

4632ROLES_WITHIN_ORG: Final = [ 

4633 LitellmUserRoles.ORG_ADMIN, 

4634 LitellmUserRoles.INTERNAL_USER, 

4635 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

4636] 

4637 

4638 

4639class OrganizationMemberUpdateRequest(OrganizationMemberDeleteRequest): 

4640 max_budget_in_organization: float | None = None 

4641 role: LitellmUserRoles | None = None 

4642 

4643 @field_validator("role") 

4644 def validate_role(cls, value: LitellmUserRoles | None) -> LitellmUserRoles | None: 

4645 if value is not None and value not in ROLES_WITHIN_ORG: 

4646 raise ValueError(f"Invalid role. Must be one of: {[role.value for role in ROLES_WITHIN_ORG]}") 

4647 return value 

4648 

4649 

4650class OrganizationMemberUpdateResponse(MemberUpdateResponse): 

4651 organization_id: str 

4652 max_budget_in_organization: float 

4653 

4654 

4655########################################## 

4656 

4657 

4658class TeamAccessGroupModelGrant(LiteLLMPydanticObjectBase): 

4659 access_group_id: str 

4660 access_group_name: str 

4661 models: tuple[str, ...] 

4662 mcp_server_ids: tuple[str, ...] = () 

4663 agent_ids: tuple[str, ...] = () 

4664 

4665 

4666class TeamInfoMember(Member): 

4667 user_alias: str | None = None 

4668 

4669 

4670class TeamEditUnrestricted(BaseModel): 

4671 kind: Literal["unrestricted"] = "unrestricted" 

4672 

4673 

4674class TeamEditAsTeamAdmin(BaseModel): 

4675 kind: Literal["team_admin"] = "team_admin" 

4676 editable_fields: tuple[str, ...] 

4677 

4678 

4679class TeamEditAsTeamAdminDisabled(BaseModel): 

4680 kind: Literal["team_admin_disabled"] = "team_admin_disabled" 

4681 

4682 

4683class TeamEditNone(BaseModel): 

4684 kind: Literal["none"] = "none" 

4685 

4686 

4687TeamEditAccess = Annotated[ 

4688 TeamEditUnrestricted | TeamEditAsTeamAdmin | TeamEditAsTeamAdminDisabled | TeamEditNone, 

4689 Field(discriminator="kind"), 

4690] 

4691 

4692 

4693class TeamInfoResponseObjectTeamTable(LiteLLM_TeamTable): 

4694 members_with_roles: tuple[TeamInfoMember, ...] = () 

4695 team_member_budget_table: LiteLLM_BudgetTableFull | None = None 

4696 # Resources inherited from access groups (separate from direct assignments) 

4697 access_group_models: list[str] | None = None 

4698 access_group_mcp_server_ids: list[str] | None = None 

4699 access_group_agent_ids: list[str] | None = None 

4700 access_group_details: tuple[TeamAccessGroupModelGrant, ...] | None = None 

4701 # Parent org's model ceiling, reported only to callers who can manage the team. 

4702 # None = no org or not a manager; [] or ["all-proxy-models"] = no ceiling. 

4703 organization_models: list[str] | None = None 

4704 model_max_budget_usage: Mapping[str, Mapping[str, object]] | None = None 

4705 caller_edit_access: TeamEditAccess = Field(default_factory=TeamEditNone) 

4706 

4707 

4708TeamMemberBudgetSource: TypeAlias = Literal["team_default", "custom", "none"] 

4709 

4710 

4711class TeamInfoMembership(LiteLLM_TeamMembership): 

4712 budget_source: TeamMemberBudgetSource 

4713 

4714 

4715class TeamInfoResponseObject(TypedDict): 

4716 team_id: str 

4717 team_info: TeamInfoResponseObjectTeamTable 

4718 keys: list 

4719 team_memberships: ReadOnly[tuple[TeamInfoMembership, ...]] 

4720 

4721 

4722class TeamMemberResetBudgetResponse(BaseModel): 

4723 team_id: str 

4724 user_id: str 

4725 budget_id: str | None 

4726 previous_budget_id: str | None 

4727 budget_source: TeamMemberBudgetSource 

4728 

4729 

4730class TeamListResponseObject(LiteLLM_TeamTable): 

4731 team_memberships: list[LiteLLM_TeamMembership] 

4732 keys: list # list of keys that belong to the team 

4733 

4734 

4735class KeyListResponseObject(TypedDict, total=False): 

4736 keys: list[str | UserAPIKeyAuth | LiteLLM_DeletedVerificationToken] 

4737 total_count: int | None 

4738 current_page: int | None 

4739 total_pages: int | None 

4740 

4741 

4742class CurrentItemRateLimit(TypedDict): 

4743 current_requests: int 

4744 current_tpm: int 

4745 current_rpm: int 

4746 

4747 

4748class LoggingCallbackStatus(TypedDict, total=False): 

4749 callbacks: list[str] 

4750 status: Literal["healthy", "unhealthy"] 

4751 details: str | None 

4752 

4753 

4754class KeyHealthResponse(TypedDict, total=False): 

4755 key: Literal["healthy", "unhealthy"] 

4756 logging_callbacks: LoggingCallbackStatus | None 

4757 

4758 

4759class CreateJWTKeyMappingRequest(LiteLLMPydanticObjectBase): 

4760 jwt_claim_name: str 

4761 jwt_claim_value: str 

4762 key: str | None = None 

4763 token: str | None = None 

4764 jwt_issuer: str | None = None 

4765 description: str | None = None 

4766 

4767 

4768class UpdateJWTKeyMappingRequest(LiteLLMPydanticObjectBase): 

4769 id: str 

4770 key: str | None = None 

4771 token: str | None = None 

4772 jwt_issuer: str | None = None 

4773 description: str | None = None 

4774 is_active: bool | None = None 

4775 

4776 

4777class DeleteJWTKeyMappingRequest(LiteLLMPydanticObjectBase): 

4778 id: str 

4779 

4780 

4781class JWTKeyMappingResponse(LiteLLMPydanticObjectBase): 

4782 id: str 

4783 jwt_issuer: str | None = None 

4784 jwt_claim_name: str 

4785 jwt_claim_value: str 

4786 description: str | None = None 

4787 is_active: bool 

4788 created_at: datetime 

4789 updated_at: datetime 

4790 created_by: str | None = None 

4791 updated_by: str | None = None 

4792 

4793 

4794class SpecialHeaders(enum.Enum): 

4795 """Used by user_api_key_auth.py to get litellm key""" 

4796 

4797 openai_authorization = "Authorization" 

4798 azure_authorization = "API-Key" 

4799 anthropic_authorization = "x-api-key" 

4800 google_ai_studio_authorization = "x-goog-api-key" 

4801 azure_apim_authorization = "Ocp-Apim-Subscription-Key" 

4802 custom_litellm_api_key = "x-litellm-api-key" 

4803 mcp_auth = "x-mcp-auth" 

4804 mcp_servers = "x-mcp-servers" 

4805 mcp_access_groups = "x-mcp-access-groups" 

4806 

4807 @classmethod 

4808 def litellm_credential_header_names(cls) -> "frozenset[str]": 

4809 """Lowercased header names user_api_key_auth accepts as a litellm key. 

4810 

4811 Every header here authenticates the caller, so any code that forwards a 

4812 request onward (e.g. the plugin reverse proxy) must strip all of them to 

4813 avoid leaking the caller's litellm credential downstream. The static 

4814 custom-key header (general_settings.litellm_key_header_name) is runtime 

4815 config and must be added on top of this set by the caller. 

4816 """ 

4817 return frozenset( 

4818 header.value.lower() 

4819 for header in ( 

4820 cls.openai_authorization, 

4821 cls.azure_authorization, 

4822 cls.anthropic_authorization, 

4823 cls.google_ai_studio_authorization, 

4824 cls.azure_apim_authorization, 

4825 cls.custom_litellm_api_key, 

4826 ) 

4827 ) 

4828 

4829 

4830class LitellmDataForBackendLLMCall(TypedDict, total=False): 

4831 headers: dict 

4832 organization: str 

4833 timeout: float | None 

4834 stream_timeout: float | None 

4835 user: str | None 

4836 num_retries: int | None 

4837 # True when the effective timeout came from a caller-controlled source (the 

4838 # `x-litellm-timeout`/`x-litellm-stream-timeout` headers, or a `timeout`/`request_timeout`/ 

4839 # `stream_timeout` field in the request body) rather than deployment config, so a 

4840 # deliberately tiny value isn't treated as a deployment health signal (see 

4841 # cooldown_handlers._trigger_cooldown_for_failed_deployment). 

4842 client_side_timeout: bool 

4843 keepalive_seconds: float | None 

4844 

4845 

4846class LitellmMetadataFromRequestHeaders(TypedDict, total=False): 

4847 """ 

4848 Headers a user can pass that will get added to litellm metadata for the request 

4849 """ 

4850 

4851 spend_logs_metadata: dict | None 

4852 agent_id: str | None 

4853 trace_id: str | None 

4854 session_id: str | None 

4855 

4856 

4857class JWTKeyItem(TypedDict, total=False): 

4858 kid: str 

4859 

4860 

4861JWKKeyValue = list[JWTKeyItem] | JWTKeyItem 

4862 

4863 

4864class JWKUrlResponse(TypedDict, total=False): 

4865 keys: JWKKeyValue 

4866 

4867 

4868class UserManagementEndpointParamDocStringEnums(str, enum.Enum): 

4869 user_id_doc_str = "Optional[str] - Specify a user id. If not set, a unique id will be generated." 

4870 user_alias_doc_str = "Optional[str] - A descriptive name for you to know who this user id refers to." 

4871 teams_doc_str = "Optional[list] - specify a list of team id's a user belongs to." 

4872 user_email_doc_str = "Optional[str] - Specify a user email." 

4873 send_invite_email_doc_str = "Optional[bool] - Specify if an invite email should be sent." 

4874 user_role_doc_str = """Optional[str] - Specify a user role - "proxy_admin", "proxy_admin_viewer", "internal_user", "internal_user_viewer", "team", "customer". Info about each role here: `https://github.com/BerriAI/litellm/litellm/proxy/_types.py#L20`""" 

4875 max_budget_doc_str = """Optional[float] - Specify max budget for a given user.""" 

4876 budget_duration_doc_str = """Optional[str] - Budget is reset at the end of specified duration. If not set, budget is never reset. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d"), months ("1mo").""" 

4877 models_doc_str = ( 

4878 """Optional[list] - Model_name's a user is allowed to call. (if empty, key is allowed to call all models)""" 

4879 ) 

4880 tpm_limit_doc_str = """Optional[int] - Specify tpm limit for a given user (Tokens per minute)""" 

4881 rpm_limit_doc_str = """Optional[int] - Specify rpm limit for a given user (Requests per minute)""" 

4882 auto_create_key_doc_str = """bool - Default=True. Flag used for returning a key as part of the /user/new response""" 

4883 aliases_doc_str = """Optional[dict] - Model aliases for the user - [Docs](https://litellm.vercel.app/docs/proxy/virtual_keys#model-aliases)""" 

4884 config_doc_str = """Optional[dict] - [DEPRECATED PARAM] User-specific config.""" 

4885 allowed_cache_controls_doc_str = """Optional[list] - List of allowed cache control values. Example - ["no-cache", "no-store"]. See all values - https://docs.litellm.ai/docs/proxy/caching#turn-on--off-caching-per-request-""" 

4886 blocked_doc_str = """Optional[bool] - [Not Implemented Yet] Whether the user is blocked.""" 

4887 guardrails_doc_str = """Optional[List[str]] - [Not Implemented Yet] List of active guardrails for the user""" 

4888 permissions_doc_str = ( 

4889 """Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking.""" 

4890 ) 

4891 metadata_doc_str = """Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }""" 

4892 max_parallel_requests_doc_str = """Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x.""" 

4893 model_max_budget_doc_str = """Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys)""" 

4894 model_rpm_limit_doc_str = """Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)""" 

4895 model_tpm_limit_doc_str = """Optional[float] - Model-specific tpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)""" 

4896 spend_doc_str = ( 

4897 """Optional[float] - Amount spent by user. Default is 0. Will be updated by proxy whenever user is used.""" 

4898 ) 

4899 team_id_doc_str = """Optional[str] - [DEPRECATED PARAM] The team id of the user. Default is None.""" 

4900 duration_doc_str = """Optional[str] - Duration for the key auto-created on `/user/new`. Default is None.""" 

4901 

4902 

4903PassThroughEndpointLoggingResultValues = ( 

4904 ModelResponse 

4905 | TextCompletionResponse 

4906 | ImageResponse 

4907 | EmbeddingResponse 

4908 | VideoObject 

4909 | StandardPassThroughResponseObject 

4910 | ResponsesAPIResponse 

4911 | TranscriptionResponse 

4912) 

4913 

4914 

4915class PassThroughEndpointLoggingTypedDict(TypedDict): 

4916 result: PassThroughEndpointLoggingResultValues | None 

4917 kwargs: dict 

4918 

4919 

4920LiteLLM_ManagementEndpoint_MetadataFields: Final = [ 

4921 "model_rpm_limit", 

4922 "model_tpm_limit", 

4923 "model_itpm_limit", 

4924 "model_otpm_limit", 

4925 "default_estimated_output_tokens", 

4926 "default_estimated_output_tokens_per_model", 

4927 "mcp_rpm_limit", 

4928 "tag_rpm_limit", 

4929 "rpm_limit_type", 

4930 "tpm_limit_type", 

4931 "enforced_params", 

4932 "temp_budget_increase", 

4933 "temp_budget_expiry", 

4934 "allowed_vector_store_indexes", 

4935 "enforced_batch_output_expires_after", 

4936 "enforced_file_expires_after", 

4937 "throttle_on_budget_exceeded", 

4938 "enable_prompt_caching", 

4939 "end_user_budget_id", 

4940] 

4941 

4942LiteLLM_ManagementEndpoint_MetadataFields_Premium: Final = [ 

4943 "disable_global_guardrails", 

4944 "guardrails", 

4945 "policies", 

4946 "tags", 

4947 "team_member_key_duration", 

4948 "prompts", 

4949 "logging", 

4950 "secret_manager_settings", 

4951 "allowed_passthrough_routes", 

4952] 

4953 

4954# Metadata keys that are immutable once set: preserved when an update omits them, 

4955# and rejected (400) when an update tries to change them. 

4956LiteLLM_Reserved_Metadata_Fields: Final = [ 

4957 "service_account_id", 

4958] 

4959 

4960 

4961class ProviderBudgetResponseObject(LiteLLMPydanticObjectBase): 

4962 """ 

4963 Configuration for a single provider's budget settings 

4964 """ 

4965 

4966 budget_limit: float | None # Budget limit in USD for the time period 

4967 time_period: str | None # Time period for budget (e.g., '1d', '30d', '1mo') 

4968 spend: float | None = 0.0 # Current spend for this provider 

4969 budget_reset_at: str | None = None # When the current budget period resets 

4970 

4971 

4972class ProviderBudgetResponse(LiteLLMPydanticObjectBase): 

4973 """ 

4974 Complete provider budget configuration and status. 

4975 Maps provider names to their budget configs. 

4976 """ 

4977 

4978 providers: dict[ 

4979 str, ProviderBudgetResponseObject 

4980 ] = {} # Dictionary mapping provider names to their budget configurations 

4981 

4982 

4983class ProxyStateVariables(TypedDict): 

4984 """ 

4985 TypedDict for Proxy state variables. 

4986 """ 

4987 

4988 spend_logs_row_count: int 

4989 

4990 

4991UI_TEAM_ID = "litellm-dashboard" 

4992 

4993 

4994class JWTAuthBuilderResult(TypedDict): 

4995 is_proxy_admin: bool 

4996 team_object: LiteLLM_TeamTable | None 

4997 user_object: LiteLLM_UserTable | None 

4998 end_user_object: LiteLLM_EndUserTable | None 

4999 org_object: LiteLLM_OrganizationTable | None 

5000 token: str 

5001 team_id: str | None 

5002 user_id: str | None 

5003 user_email: str | None 

5004 end_user_id: str | None 

5005 org_id: str | None 

5006 team_membership: LiteLLM_TeamMembership | None 

5007 jwt_claims: dict # Decoded JWT token claims (avoids re-decoding) 

5008 agent_id: ReadOnly[str | None] 

5009 

5010 

5011class ClientSideFallbackModel(TypedDict, total=False): 

5012 """ 

5013 Dictionary passed when client configuring input 

5014 """ 

5015 

5016 model: Required[str] 

5017 messages: list[AllMessageValues] 

5018 

5019 

5020ALL_FALLBACK_MODEL_VALUES = str | ClientSideFallbackModel 

5021 

5022 

5023RBAC_ROLES = Literal[ 

5024 LitellmUserRoles.PROXY_ADMIN, 

5025 LitellmUserRoles.TEAM, 

5026 LitellmUserRoles.INTERNAL_USER, 

5027] 

5028 

5029 

5030class OIDCPermissions(LiteLLMPydanticObjectBase): 

5031 models: list[str] | None = None 

5032 routes: list[str] | None = None 

5033 

5034 

5035class RoleBasedPermissions(OIDCPermissions): 

5036 role: RBAC_ROLES 

5037 

5038 model_config = { 

5039 "extra": "forbid", 

5040 } 

5041 

5042 

5043class RoleMapping(BaseModel): 

5044 role: str 

5045 internal_role: RBAC_ROLES 

5046 

5047 

5048class JWTLiteLLMRoleMap(BaseModel): 

5049 jwt_role: str 

5050 litellm_role: LitellmUserRoles 

5051 

5052 

5053class ScopeMapping(OIDCPermissions): 

5054 scope: str 

5055 

5056 model_config = { 

5057 "extra": "forbid", 

5058 } 

5059 

5060 

5061class JWTRoutingOverride(BaseModel): 

5062 """ 

5063 Override default auth routing for JWT-shaped bearer tokens. 

5064 

5065 A rule matches when all provided selectors match token claims. 

5066 If matched, request is routed to the configured auth path. 

5067 

5068 Wildcard selectors use shell-style patterns (* and ?) and are matched with 

5069 case-sensitive semantics; use the same casing your IdP emits in JWT claims. 

5070 Space-delimited tokenization applies only to the ``scope`` claim (OAuth/OIDC 

5071 scope strings), not to ``iss``, ``aud``, or ``client_id``. 

5072 """ 

5073 

5074 iss: str | list[str] 

5075 client_id: str | list[str] | None = None 

5076 scope: str | list[str] | None = None 

5077 aud: str | list[str] | None = None 

5078 path: Literal["oauth2"] = "oauth2" 

5079 

5080 model_config = { 

5081 "extra": "forbid", 

5082 } 

5083 

5084 

5085class UnregisteredJWTClientBehavior(str, enum.Enum): 

5086 """ 

5087 Controls what happens when `virtual_key_claim_field` is configured but the 

5088 JWT claim value has no registered mapping in `litellm_jwtkeymapping`. 

5089 

5090 - fallback_team_mapping: Fall through to standard team-based JWT auth (default, 

5091 backward-compatible). 

5092 - reject: Immediately return HTTP 403. Use this when every valid JWT client 

5093 must have a pre-registered virtual key — unknown callers are denied. 

5094 - auto_register: Automatically create a new virtual key and mapping on first 

5095 encounter. The new key has no budget/model restrictions; admins can tighten 

5096 it later via /jwt_client/update. 

5097 """ 

5098 

5099 FALLBACK_TEAM_MAPPING = "fallback_team_mapping" 

5100 REJECT = "reject" 

5101 AUTO_REGISTER = "auto_register" 

5102 

5103 

5104class JWTIssuerConfig(BaseModel): 

5105 """ 

5106 Issuer-bound JWT validation configuration. 

5107 

5108 When a token's unverified `iss` claim matches an entry in 

5109 ``LiteLLM_JWTAuth.issuers``, LiteLLM validates it only against that 

5110 issuer's JWKS and audience. Tokens whose `iss` does not match any 

5111 configured issuer fall back to the global JWT_AUDIENCE/JWT_ISSUER 

5112 validation path; `issuers` is additive routing, not an allow-list. 

5113 """ 

5114 

5115 issuer: str = Field(description="Exact expected JWT issuer (`iss`) value.") 

5116 jwks_url: str | None = Field( 

5117 default=None, 

5118 description="Issuer JWKS URL. If omitted, LiteLLM uses the issuer's OIDC discovery document.", 

5119 ) 

5120 audience: str | list[str] | None = Field( 

5121 default=None, 

5122 description="Expected token audience for this issuer.", 

5123 ) 

5124 disable_audience_validation: bool = Field( 

5125 default=False, 

5126 description="Explicitly disable audience validation for this issuer. Use only when the issuer cannot provide an audience suitable for LiteLLM.", 

5127 ) 

5128 user_id_jwt_field: str | None = Field( 

5129 default=None, 

5130 description="Issuer-specific claim path to normalize into LiteLLM's user id.", 

5131 ) 

5132 user_email_jwt_field: str | None = Field( 

5133 default=None, 

5134 description="Issuer-specific claim path to normalize into LiteLLM's user email.", 

5135 ) 

5136 team_id_jwt_field: str | None = Field( 

5137 default=None, 

5138 description="Issuer-specific claim path to normalize into LiteLLM's team id.", 

5139 ) 

5140 team_ids_jwt_field: str | None = Field( 

5141 default=None, 

5142 description="Issuer-specific claim path to normalize into LiteLLM's team ids.", 

5143 ) 

5144 org_id_jwt_field: str | None = Field( 

5145 default=None, 

5146 description="Issuer-specific claim path to normalize into LiteLLM's organization id.", 

5147 ) 

5148 end_user_id_jwt_field: str | None = Field( 

5149 default=None, 

5150 description="Issuer-specific claim path to normalize into LiteLLM's end-user id.", 

5151 ) 

5152 virtual_key_claim_field: str | None = Field( 

5153 default=None, 

5154 description="Issuer-specific claim path used for the virtual key mapping lookup. Falls back to the global field.", 

5155 ) 

5156 unregistered_jwt_client_behavior: UnregisteredJWTClientBehavior | None = Field( 

5157 default=None, 

5158 description="Issuer-specific policy when the virtual key claim has no mapping. Falls back to the global policy.", 

5159 ) 

5160 

5161 model_config = { 

5162 "extra": "forbid", 

5163 } 

5164 

5165 @model_validator(mode="after") 

5166 def validate_audience_configured(self) -> "JWTIssuerConfig": 

5167 if self.audience is None and not self.disable_audience_validation: 

5168 raise ValueError( 

5169 f"JWT issuer {self.issuer} must configure audience or set disable_audience_validation=True" 

5170 ) 

5171 if self.audience is not None and self.disable_audience_validation: 

5172 raise ValueError( 

5173 f"JWT issuer {self.issuer} cannot set audience and disable_audience_validation=True together" 

5174 ) 

5175 return self 

5176 

5177 

5178DEFAULT_JWKS_STALE_TTL: Final = 3600 

5179 

5180 

5181class LiteLLM_JWTAuth(LiteLLMPydanticObjectBase): 

5182 """ 

5183 A class to define the roles and permissions for a LiteLLM Proxy w/ JWT Auth. 

5184 

5185 Attributes: 

5186 - admin_jwt_scope: The JWT scope required for proxy admin roles. 

5187 - admin_allowed_routes: list of allowed routes for proxy admin roles. 

5188 - team_jwt_scope: The JWT scope required for proxy team roles. 

5189 - team_id_jwt_field: The field in the JWT token that stores the team ID. Default - `client_id`. 

5190 - team_allowed_routes: list of allowed routes for proxy team roles. 

5191 - user_id_jwt_field: The field in the JWT token that stores the user id (maps to `LiteLLMUserTable`). Use this for internal employees. 

5192 - user_email_jwt_field: The field in the JWT token that stores the user email (maps to `LiteLLMUserTable`). Use this for internal employees. 

5193 - user_allowed_email_subdomain: If specified, only emails from specified subdomain will be allowed to access proxy. 

5194 - end_user_id_jwt_field: The field in the JWT token that stores the end-user ID (maps to `LiteLLMEndUserTable`). Turn this off by setting to `None`. Enables end-user cost tracking. Use this for external customers. 

5195 - public_key_ttl: Default - 600s. TTL for caching public JWT keys. 

5196 - public_key_stale_ttl: Default - 3600s. Extra time past `public_key_ttl` that the last-known-good JWKS response 

5197 stays usable while the identity provider is unreachable. Set to 0 to fail closed instead. 

5198 - public_allowed_routes: list of allowed routes for authenticated but unknown litellm role jwt tokens. 

5199 - enforce_rbac: If true, enforce RBAC for all routes. 

5200 - custom_validate: A custom function to validates the JWT token. 

5201 - oidc_userinfo_endpoint: OIDC UserInfo endpoint URL. When set along with oidc_userinfo_enabled, LiteLLM will call this endpoint with the access token to retrieve user identity information. 

5202 - oidc_userinfo_enabled: Enable fetching user info from OIDC UserInfo endpoint instead of just decoding JWT token. Default: False. 

5203 - oidc_userinfo_cache_ttl: TTL (in seconds) for caching UserInfo responses. Default: 300s (5 minutes). 

5204 

5205 See `auth_checks.py` for the specific routes 

5206 """ 

5207 

5208 admin_jwt_scope: str = "litellm_proxy_admin" 

5209 admin_allowed_routes: list[str] = [ 

5210 "management_routes", 

5211 "spend_tracking_routes", 

5212 "global_spend_tracking_routes", 

5213 "info_routes", 

5214 ] 

5215 team_id_jwt_field: str | None = None 

5216 team_id_upsert: bool = False 

5217 team_ids_jwt_field: str | None = None 

5218 upsert_sso_user_to_team: bool = False 

5219 team_allowed_routes: list[str] = [ 

5220 "openai_routes", 

5221 "info_routes", 

5222 "mcp_routes", 

5223 "/v1/messages", 

5224 "/v1/messages/count_tokens", 

5225 ] 

5226 team_id_default: str | None = Field( 

5227 default=None, 

5228 description="If no team_id given, default permissions/spend-tracking to this team.s", 

5229 ) 

5230 team_alias_jwt_field: str | None = Field( 

5231 default=None, 

5232 description="The field in the JWT token that stores the team name/alias. Will be resolved to team_id via database lookup.", 

5233 ) 

5234 

5235 org_id_jwt_field: str | None = None 

5236 org_alias_jwt_field: str | None = Field( 

5237 default=None, 

5238 description="The field in the JWT token that stores the organization name/alias. Will be resolved to org_id via database lookup.", 

5239 ) 

5240 user_id_jwt_field: str | None = None 

5241 user_email_jwt_field: str | None = None 

5242 user_allowed_email_domain: str | None = None 

5243 user_roles_jwt_field: str | None = None 

5244 user_allowed_roles: list[str] | None = None 

5245 user_id_upsert: bool = Field(default=False, description="If user doesn't exist, upsert them into the db.") 

5246 end_user_id_jwt_field: str | None = None 

5247 agent_id_jwt_field: str | None = Field( 

5248 default=None, 

5249 description=( 

5250 "The field in the JWT token that identifies the calling agent (e.g. 'azp' for a Microsoft Entra ID " 

5251 "app token). Supports dot notation. The value is matched against a registered agent's agent_id, " 

5252 "then agent_name, and the request is rejected when it matches neither." 

5253 ), 

5254 ) 

5255 mcp_client_id_jwt_field: str | None = Field( 

5256 default=None, 

5257 description=( 

5258 "The field in the JWT token that identifies the MCP client application (harness) making the request, " 

5259 "e.g. 'azp' or 'client_id'. Supports dot notation. Only consulted while general_settings.mcp_allowed_clients " 

5260 "is set: the claim value must be listed there or the MCP request is rejected with 403. Distinct from " 

5261 "agent_id_jwt_field, which identifies an AI agent rather than the client software." 

5262 ), 

5263 ) 

5264 public_key_ttl: float = 600 

5265 public_key_stale_ttl: float = Field( 

5266 default=DEFAULT_JWKS_STALE_TTL, 

5267 ge=0, 

5268 description=( 

5269 "Seconds beyond `public_key_ttl` that the last-known-good JWKS response stays usable while the identity " 

5270 "provider is unreachable. Bounds how long a signing key the provider has since removed can still be " 

5271 "trusted. Set to 0 to fail closed and reject requests as soon as the cached keys expire." 

5272 ), 

5273 ) 

5274 public_allowed_routes: list[str] = ["public_routes"] 

5275 enforce_rbac: bool = False 

5276 roles_jwt_field: str | None = None # v2 on role mappings 

5277 role_mappings: list[RoleMapping] | None = None 

5278 object_id_jwt_field: str | None = None # can be either user / team, inferred from the role mapping 

5279 scope_mappings: list[ScopeMapping] | None = None 

5280 enforce_scope_based_access: bool = False 

5281 enforce_team_based_model_access: bool = False 

5282 custom_validate: Callable[..., Literal[True]] | None = None 

5283 ######################################################### 

5284 # Fields for syncing user team membership and roles with IDP provider 

5285 jwt_litellm_role_map: list[JWTLiteLLMRoleMap] | None = None 

5286 sync_user_role_and_teams: bool = False 

5287 ######################################################### 

5288 ######################################################### 

5289 # OIDC UserInfo Endpoint Configuration 

5290 oidc_userinfo_endpoint: str | None = Field( 

5291 default=None, 

5292 description="OIDC UserInfo endpoint URL. If set, LiteLLM will call this endpoint with the access token to retrieve user identity information.", 

5293 ) 

5294 oidc_userinfo_enabled: bool = Field( 

5295 default=False, 

5296 description="Enable fetching user info from OIDC UserInfo endpoint instead of just decoding JWT token.", 

5297 ) 

5298 oidc_userinfo_cache_ttl: float = Field( 

5299 default=300, 

5300 description="TTL (in seconds) for caching UserInfo responses. Default: 300s (5 minutes).", 

5301 ) 

5302 # JWT-to-Virtual-Key Mapping 

5303 virtual_key_claim_field: str | None = Field( 

5304 default=None, 

5305 description="JWT claim field for virtual key mapping lookup (e.g. 'sub', 'email'). Supports dot notation.", 

5306 ) 

5307 virtual_key_mapping_cache_ttl: float = Field( 

5308 default=300, 

5309 description="TTL (seconds) for caching JWT-to-virtual-key mapping lookups.", 

5310 ) 

5311 unregistered_jwt_client_behavior: UnregisteredJWTClientBehavior = Field( 

5312 default=UnregisteredJWTClientBehavior.FALLBACK_TEAM_MAPPING, 

5313 description=( 

5314 "What to do when virtual_key_claim_field is set but the JWT claim value " 

5315 "has no registered mapping. 'fallback_team_mapping' (default): fall through " 

5316 "to team-based JWT auth. 'reject': return HTTP 403. " 

5317 "'auto_register': auto-create a virtual key and mapping on first encounter." 

5318 ), 

5319 ) 

5320 routing_overrides: list[JWTRoutingOverride] | None = Field( 

5321 default=None, 

5322 description="Optional claim-based routing overrides for JWT-shaped tokens. Matching rules route requests to oauth2 before default JWT flow.", 

5323 ) 

5324 team_claim_fallback: bool = Field( 

5325 default=False, 

5326 description=( 

5327 "If True, when a configured team_id_jwt_field / team_ids_jwt_field " 

5328 "claim is present but does not resolve to any known team, defer to " 

5329 "the single-team DB fallback (caller's only team membership) " 

5330 "instead of raising. Default False preserves strict claim-based " 

5331 "authorization." 

5332 ), 

5333 ) 

5334 fallback_to_db_teams: bool = Field( 

5335 default=False, 

5336 description=( 

5337 "When True, users whose JWT contains no team claims are authenticated " 

5338 "using their database team memberships instead of receiving HTTP 403, " 

5339 "with usage attributed to the user's first resolvable DB team. Whether or " 

5340 "not the JWT carries team claims, the x-litellm-team-id request header may " 

5341 "select any team the user is a member of in the database (validated against " 

5342 "DB membership); without the header the JWT team stays the default. Requires " 

5343 "user_id_upsert=True so that user records exist before the fallback runs." 

5344 ), 

5345 ) 

5346 issuers: list[JWTIssuerConfig] | None = Field( 

5347 default=None, 

5348 description="Optional issuer-bound JWT validation rules. When a token's `iss` matches a configured issuer, validation uses that issuer's JWKS, audience, and claim mappings. Tokens with an unlisted `iss` fall back to the global JWT_AUDIENCE/JWT_ISSUER validation path — this is additive routing, not an allow-list.", 

5349 ) 

5350 ######################################################### 

5351 

5352 def __init__(self, **kwargs: Any) -> None: 

5353 # ``config_file_path`` is a non-field kwarg threaded by the 

5354 # startup-load path so an operator-configured 

5355 # ``custom_validate: s3://bucket/module.fn`` resolves through 

5356 # the documented config-file flow. Pop before the invalid-keys 

5357 # check; the runtime gate in ``get_instance_fn`` refuses 

5358 # ``s3://`` / ``gcs://`` when this is None. 

5359 config_file_path: Final = kwargs.pop("config_file_path", None) 

5360 

5361 # Backward-compat: jwt_client_id_field was renamed to virtual_key_claim_field 

5362 if "jwt_client_id_field" in kwargs: 5362 ↛ 5363line 5362 didn't jump to line 5363 because the condition on line 5362 was never true

5363 if "virtual_key_claim_field" not in kwargs: 

5364 kwargs["virtual_key_claim_field"] = kwargs.pop("jwt_client_id_field") 

5365 else: 

5366 kwargs.pop("jwt_client_id_field") 

5367 

5368 # get the attribute names for this Pydantic model 

5369 allowed_keys: Final = LiteLLM_JWTAuth.__annotations__.keys() 

5370 

5371 invalid_keys: Final = set(kwargs.keys()) - allowed_keys 

5372 user_roles_jwt_field: Final = kwargs.get("user_roles_jwt_field") 

5373 user_allowed_roles: Final = kwargs.get("user_allowed_roles") 

5374 object_id_jwt_field: Final = kwargs.get("object_id_jwt_field") 

5375 role_mappings: Final = kwargs.get("role_mappings") 

5376 scope_mappings: Final = kwargs.get("scope_mappings") 

5377 enforce_scope_based_access: Final = kwargs.get("enforce_scope_based_access") 

5378 custom_validate: Final = kwargs.get("custom_validate") 

5379 

5380 if custom_validate is not None: 5380 ↛ 5381line 5380 didn't jump to line 5381 because the condition on line 5380 was never true

5381 fn: Final = get_instance_fn(custom_validate, config_file_path=config_file_path) 

5382 validate_custom_validate_return_type(fn) 

5383 kwargs["custom_validate"] = fn 

5384 

5385 if invalid_keys: 5385 ↛ 5386line 5385 didn't jump to line 5386 because the condition on line 5385 was never true

5386 raise ValueError( 

5387 f"Invalid arguments provided: {', '.join(invalid_keys)}. Allowed arguments are: {', '.join(allowed_keys)}." 

5388 ) 

5389 if (user_roles_jwt_field is not None and user_allowed_roles is None) or ( 5389 ↛ 5392line 5389 didn't jump to line 5392 because the condition on line 5389 was never true

5390 user_roles_jwt_field is None and user_allowed_roles is not None 

5391 ): 

5392 raise ValueError("user_allowed_roles must be provided if user_roles_jwt_field is set.") 

5393 

5394 if object_id_jwt_field is not None and role_mappings is None: 5394 ↛ 5395line 5394 didn't jump to line 5395 because the condition on line 5394 was never true

5395 raise ValueError( 

5396 "if object_id_jwt_field is set, role_mappings must also be set. Needed to infer if the caller is a user or team." 

5397 ) 

5398 

5399 if scope_mappings is not None and not enforce_scope_based_access: 5399 ↛ 5400line 5399 didn't jump to line 5400 because the condition on line 5399 was never true

5400 raise ValueError("scope_mappings must be set if enforce_scope_based_access is true.") 

5401 

5402 super().__init__(**kwargs) 

5403 

5404 def get_issuer_config(self, issuer: str | None) -> JWTIssuerConfig | None: 

5405 if issuer is None or self.issuers is None: 

5406 return None 

5407 return next((config for config in self.issuers if config.issuer == issuer), None) 

5408 

5409 def is_virtual_key_mapping_configured(self) -> bool: 

5410 if self.virtual_key_claim_field is not None: 5410 ↛ 5411line 5410 didn't jump to line 5411 because the condition on line 5410 was never true

5411 return True 

5412 return any(config.virtual_key_claim_field is not None for config in self.issuers or ()) 

5413 

5414 def get_virtual_key_claim_field(self, issuer: str | None) -> str | None: 

5415 issuer_config: Final = self.get_issuer_config(issuer) 

5416 if issuer_config is not None and issuer_config.virtual_key_claim_field is not None: 

5417 return issuer_config.virtual_key_claim_field 

5418 return self.virtual_key_claim_field 

5419 

5420 def get_unregistered_jwt_client_behavior(self, issuer: str | None) -> UnregisteredJWTClientBehavior: 

5421 issuer_config: Final = self.get_issuer_config(issuer) 

5422 if issuer_config is not None and issuer_config.unregistered_jwt_client_behavior is not None: 

5423 return issuer_config.unregistered_jwt_client_behavior 

5424 return self.unregistered_jwt_client_behavior 

5425 

5426 

5427class PrismaCompatibleUpdateDBModel(TypedDict, total=False): 

5428 model_name: str 

5429 litellm_params: str 

5430 model_info: str 

5431 blocked: bool 

5432 updated_at: str 

5433 updated_by: str 

5434 

5435 

5436class SpecialManagementEndpointEnums(enum.Enum): 

5437 DEFAULT_ORGANIZATION = "default_organization" 

5438 

5439 

5440class TransformRequestBody(BaseModel): 

5441 call_type: CallTypes 

5442 request_body: dict 

5443 

5444 

5445class DefaultInternalUserParams(LiteLLMPydanticObjectBase): 

5446 """ 

5447 Default parameters to apply when a new user signs in via SSO or is created on the /user/new API endpoint 

5448 """ 

5449 

5450 user_role: ( 

5451 Literal[ 

5452 LitellmUserRoles.PROXY_ADMIN, 

5453 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, 

5454 LitellmUserRoles.INTERNAL_USER, 

5455 LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

5456 ] 

5457 | None 

5458 ) = Field( 

5459 default=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, 

5460 description="Default role assigned to new users created", 

5461 ) 

5462 max_budget: float | None = Field( 

5463 default=None, 

5464 description="Default maximum budget (in USD) for new users created", 

5465 ) 

5466 budget_duration: str | None = Field( 

5467 default=None, 

5468 description="Default budget duration for new users (e.g. 'daily', 'weekly', 'monthly')", 

5469 ) 

5470 models: list[str] | None = Field(default=None, description="Default list of models that new users can access") 

5471 

5472 teams: list[str] | list[NewUserRequestTeam] | None = Field( 

5473 default=None, 

5474 description="Default teams for new users created", 

5475 ) 

5476 

5477 

5478class BaseDailySpendTransaction(TypedDict): 

5479 date: str 

5480 api_key: str 

5481 model: str | None 

5482 model_group: str | None 

5483 mcp_namespaced_tool_name: str | None 

5484 custom_llm_provider: str | None 

5485 endpoint: str | None 

5486 

5487 # token count metrics 

5488 prompt_tokens: int 

5489 completion_tokens: int 

5490 cache_read_input_tokens: int 

5491 cache_creation_input_tokens: int 

5492 compression_saved_tokens: int 

5493 

5494 # cost-savings metrics (dollars, priced per request before aggregation) 

5495 compression_savings_spend: float 

5496 prompt_caching_savings_spend: float 

5497 gateway_injected_caching_savings_spend: float # writable-ok: the rollup queue accumulates into this key in place, as it does for every sibling spend field 

5498 # Not required: rows queued by a pod running the previous release, or replayed from 

5499 # the Redis buffer across an upgrade, carry no such key. Every reader coalesces a 

5500 # missing value to zero, so requiring it here would describe a shape the aggregation 

5501 # is explicitly tested against. 

5502 autorouter_savings_spend: NotRequired[float] 

5503 

5504 # request level metrics 

5505 spend: float 

5506 api_requests: int 

5507 successful_requests: int 

5508 failed_requests: int 

5509 total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place 

5510 timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place 

5511 

5512 

5513class DailyTeamSpendTransaction(BaseDailySpendTransaction): 

5514 team_id: str 

5515 

5516 

5517class DailyOrganizationSpendTransaction(BaseDailySpendTransaction): 

5518 organization_id: str 

5519 

5520 

5521class DailyUserSpendTransaction(BaseDailySpendTransaction): 

5522 user_id: str 

5523 

5524 

5525class DailyEndUserSpendTransaction(BaseDailySpendTransaction): 

5526 end_user_id: str 

5527 

5528 

5529class DailyTagSpendTransaction(BaseDailySpendTransaction): 

5530 request_id: str | None 

5531 tag: str 

5532 

5533 

5534class DailyAgentSpendTransaction(BaseDailySpendTransaction): 

5535 agent_id: str 

5536 

5537 

5538class DBSpendUpdateTransactions(TypedDict): 

5539 """ 

5540 Internal Data Structure for buffering spend updates in Redis or in memory before committing them to the database 

5541 """ 

5542 

5543 user_list_transactions: dict[str, float] | None 

5544 end_user_list_transactions: dict[str, float] | None 

5545 key_list_transactions: dict[str, float] | None 

5546 team_list_transactions: dict[str, float] | None 

5547 team_member_list_transactions: dict[str, float] | None 

5548 org_list_transactions: dict[str, float] | None 

5549 org_member_list_transactions: ReadOnly[dict[str, float] | None] 

5550 project_list_transactions: ReadOnly[dict[str, float] | None] 

5551 tag_list_transactions: dict[str, float] | None 

5552 agent_list_transactions: dict[str, float] | None 

5553 model_access_group_list_transactions: ReadOnly[dict[str, float] | None] 

5554 

5555 

5556class SpendUpdateQueueItem(TypedDict, total=False): 

5557 entity_type: Litellm_EntityType 

5558 entity_id: str 

5559 response_cost: float | None 

5560 

5561 

5562class ToolDiscoveryQueueItem(TypedDict, total=False): 

5563 tool_name: str 

5564 origin: str | None # MCP server name or "user_defined" 

5565 created_by: str | None 

5566 key_hash: str | None # hash of virtual key that triggered discovery 

5567 team_id: str | None # team that triggered discovery 

5568 key_alias: str | None # human-readable key alias 

5569 user_agent: str | None # HTTP User-Agent of the caller 

5570 

5571 

5572from litellm.models.managed_files import ( # noqa: E402 

5573 LiteLLM_ManagedFileTable as LiteLLM_ManagedFileTable, 

5574) 

5575from litellm.models.managed_files import ( # noqa: E402 

5576 LiteLLM_ManagedObjectTable as LiteLLM_ManagedObjectTable, 

5577) 

5578from litellm.models.managed_files import ( # noqa: E402 

5579 LiteLLM_ManagedVectorStoresTable as LiteLLM_ManagedVectorStoresTable, 

5580) 

5581from litellm.models.managed_files import ( # noqa: E402 

5582 LiteLLM_ManagedVectorStoreTable as LiteLLM_ManagedVectorStoreTable, 

5583) 

5584 

5585 

5586class EnterpriseLicenseData(TypedDict, total=False): 

5587 expiration_date: str 

5588 user_id: str 

5589 allowed_features: list[str] 

5590 max_users: int 

5591 max_teams: int 

5592 

5593 

5594class ResponseLiteLLM_ManagedVectorStore(TypedDict, total=False): 

5595 vector_store: LiteLLM_ManagedVectorStoresTable 

5596 

5597 

5598class CostEstimateRequest(LiteLLMPydanticObjectBase): 

5599 """Request body for /cost/estimate endpoint.""" 

5600 

5601 model: str = Field(description="Model name (from /model_group/info)") 

5602 input_tokens: int = Field(description="Expected input tokens per request", ge=0) 

5603 output_tokens: int = Field(description="Expected output tokens per request", ge=0) 

5604 cache_read_input_tokens: int = Field( 

5605 default=0, description="Input tokens read from the prompt cache; counted within input_tokens", ge=0 

5606 ) 

5607 cache_creation_input_tokens: int = Field( 

5608 default=0, description="Input tokens written to the prompt cache; counted within input_tokens", ge=0 

5609 ) 

5610 reasoning_tokens: int = Field( 

5611 default=0, description="Reasoning tokens the model emits; counted within output_tokens", ge=0 

5612 ) 

5613 num_requests_per_day: int | None = Field(default=None, description="Number of requests per day", ge=0) 

5614 num_requests_per_month: int | None = Field(default=None, description="Number of requests per month", ge=0) 

5615 

5616 @model_validator(mode="after") 

5617 def validate_token_subsets(self) -> "CostEstimateRequest": 

5618 if self.cache_read_input_tokens + self.cache_creation_input_tokens > self.input_tokens: 

5619 raise ValueError("cache_read_input_tokens plus cache_creation_input_tokens cannot exceed input_tokens") 

5620 if self.reasoning_tokens > self.output_tokens: 

5621 raise ValueError("reasoning_tokens cannot exceed output_tokens") 

5622 return self 

5623 

5624 

5625class CostEstimateResponse(LiteLLMPydanticObjectBase): 

5626 """Response body for /cost/estimate endpoint.""" 

5627 

5628 model: str 

5629 input_tokens: int 

5630 output_tokens: int 

5631 cache_read_input_tokens: int = 0 

5632 cache_creation_input_tokens: int = 0 

5633 reasoning_tokens: int = 0 

5634 num_requests_per_day: int | None = None 

5635 num_requests_per_month: int | None = None 

5636 # Per-request costs 

5637 cost_per_request: float = Field(description="Total cost per request (includes margin)") 

5638 input_cost_per_request: float = Field(description="Input token cost per request (before margin)") 

5639 output_cost_per_request: float = Field(description="Output token cost per request (before margin)") 

5640 margin_cost_per_request: float = Field(default=0.0, description="Margin/fee added per request") 

5641 cache_read_cost_per_request: float = Field(default=0.0, description="Cache-read share of input_cost_per_request") 

5642 cache_creation_cost_per_request: float = Field( 

5643 default=0.0, description="Cache-write share of input_cost_per_request" 

5644 ) 

5645 reasoning_cost_per_request: float = Field(default=0.0, description="Reasoning share of output_cost_per_request") 

5646 # Daily costs (if num_requests_per_day provided) 

5647 daily_cost: float | None = Field(default=None, description="Total daily cost (includes margin)") 

5648 daily_input_cost: float | None = Field(default=None, description="Daily input token cost") 

5649 daily_output_cost: float | None = Field(default=None, description="Daily output token cost") 

5650 daily_margin_cost: float | None = Field(default=None, description="Daily margin/fee") 

5651 daily_cache_read_cost: float | None = Field(default=None, description="Cache-read share of daily_input_cost") 

5652 daily_cache_creation_cost: float | None = Field(default=None, description="Cache-write share of daily_input_cost") 

5653 daily_reasoning_cost: float | None = Field(default=None, description="Reasoning share of daily_output_cost") 

5654 # Monthly costs (if num_requests_per_month provided) 

5655 monthly_cost: float | None = Field(default=None, description="Total monthly cost (includes margin)") 

5656 monthly_input_cost: float | None = Field(default=None, description="Monthly input token cost") 

5657 monthly_output_cost: float | None = Field(default=None, description="Monthly output token cost") 

5658 monthly_margin_cost: float | None = Field(default=None, description="Monthly margin/fee") 

5659 monthly_cache_read_cost: float | None = Field(default=None, description="Cache-read share of monthly_input_cost") 

5660 monthly_cache_creation_cost: float | None = Field( 

5661 default=None, description="Cache-write share of monthly_input_cost" 

5662 ) 

5663 monthly_reasoning_cost: float | None = Field(default=None, description="Reasoning share of monthly_output_cost") 

5664 # Pricing info: the rates this request's usage bills at, after token tiers and regional multipliers 

5665 input_cost_per_token: float | None = Field(default=None, description="Rate billed per input token") 

5666 output_cost_per_token: float | None = Field(default=None, description="Rate billed per output token") 

5667 cache_read_input_token_cost: float | None = Field(default=None, description="Rate billed per cache-read token") 

5668 cache_creation_input_token_cost: float | None = Field(default=None, description="Rate billed per cache-write token") 

5669 output_cost_per_reasoning_token: float | None = Field(default=None, description="Rate billed per reasoning token") 

5670 provider: str | None = None