Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/management_endpoints/fallback_management_endpoints.py: 45%
108 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2FALLBACK MANAGEMENT ENDPOINTS
4Dedicated endpoints for managing model fallbacks separately from general config.
6POST /fallback - Create or update fallbacks for a specific model
7GET /fallback/{model} - Get fallbacks for a specific model
8DELETE /fallback/{model} - Delete fallbacks for a specific model
9"""
11# pyright: reportMissingImports=false
13import json
14from typing import TYPE_CHECKING, Final, Literal
16from litellm._logging import verbose_proxy_logger
17from litellm.proxy._types import UserAPIKeyAuth
18from litellm.proxy.auth.model_checks import get_all_fallbacks
19from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
21if TYPE_CHECKING: 21 ↛ 22line 21 didn't jump to line 22 because the condition on line 21 was never true
22 from fastapi import APIRouter, Depends, HTTPException, status
23else:
24 try:
25 from fastapi import APIRouter, Depends, HTTPException, status
26 except ImportError:
27 # fastapi is only required for proxy, not for SDK usage
28 pass
30from litellm.repositories.config_repository import ConfigRepository
31from litellm.types.management_endpoints.router_settings_endpoints import (
32 FallbackCreateRequest,
33 FallbackDeleteResponse,
34 FallbackGetResponse,
35 FallbackResponse,
36)
38router: Final = APIRouter()
41@router.post(
42 "/fallback",
43 tags=["Fallback Management"],
44 dependencies=[Depends(user_api_key_auth)],
45 response_model=FallbackResponse,
46 status_code=status.HTTP_200_OK,
47)
48async def create_fallback(
49 data: FallbackCreateRequest,
50 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
51):
52 """
53 Create or update fallbacks for a specific model.
55 This endpoint allows you to configure fallback models separately from the general config.
56 Fallbacks are triggered when a model call fails after retries.
58 **Example Request:**
59 ```json
60 {
61 "model": "gpt-3.5-turbo",
62 "fallback_models": ["gpt-4", "claude-3-haiku"],
63 "fallback_type": "general"
64 }
65 ```
67 **Fallback Types:**
68 - `general`: Standard fallbacks for any error (default)
69 - `context_window`: Fallbacks specifically for context window exceeded errors
70 - `content_policy`: Fallbacks specifically for content policy violations
71 """
72 from litellm.proxy.proxy_server import (
73 llm_router,
74 prisma_client,
75 proxy_config,
76 store_model_in_db,
77 )
79 try:
80 # Validate that we have a router
81 if llm_router is None: 81 ↛ 82line 81 didn't jump to line 82 because the condition on line 81 was never true
82 raise HTTPException(
83 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
84 detail={"error": "Router not initialized"},
85 )
87 # Validate that the model exists in the router
88 model_names: Final = llm_router.model_names
89 if data.model not in model_names: 89 ↛ 99line 89 didn't jump to line 99 because the condition on line 89 was always true
90 raise HTTPException(
91 status_code=status.HTTP_404_NOT_FOUND,
92 detail={
93 "error": f"Model '{data.model}' not found in router",
94 "available_models": list(model_names),
95 },
96 )
98 # Validate that all fallback models exist in the router
99 invalid_fallback_models: Final = [m for m in data.fallback_models if m not in model_names]
100 if invalid_fallback_models:
101 raise HTTPException(
102 status_code=status.HTTP_400_BAD_REQUEST,
103 detail={
104 "error": f"Invalid fallback models: {invalid_fallback_models}",
105 "available_models": list(model_names),
106 },
107 )
109 # Check if fallback model is the same as the primary model
110 if data.model in data.fallback_models:
111 raise HTTPException(
112 status_code=status.HTTP_400_BAD_REQUEST,
113 detail={"error": f"Model '{data.model}' cannot be its own fallback"},
114 )
116 # Check if we need to store in DB
117 if store_model_in_db is not True or prisma_client is None:
118 raise HTTPException(
119 status_code=status.HTTP_400_BAD_REQUEST,
120 detail={
121 "error": "Database storage not enabled. Set 'STORE_MODEL_IN_DB=True' in your environment to use this feature."
122 },
123 )
125 # Load existing config
126 config: Final = await proxy_config.get_config()
127 router_settings: Final = config.get("router_settings", {})
129 # Get the appropriate fallback list based on type
130 fallback_key = "fallbacks"
131 if data.fallback_type == "context_window":
132 fallback_key = "context_window_fallbacks"
133 elif data.fallback_type == "content_policy":
134 fallback_key = "content_policy_fallbacks"
136 # Get existing fallbacks
137 existing_fallbacks: Final[list[dict[str, list[str]]]] = router_settings.get(fallback_key, [])
139 # Update or add the fallback configuration
140 fallback_updated = False
141 for i, fallback_dict in enumerate(existing_fallbacks):
142 if data.model in fallback_dict:
143 # Update existing fallback
144 existing_fallbacks[i] = {data.model: data.fallback_models}
145 fallback_updated = True
146 break
148 if not fallback_updated:
149 # Add new fallback
150 existing_fallbacks.append({data.model: data.fallback_models})
152 # Update router settings
153 router_settings[fallback_key] = existing_fallbacks
155 # Save to database - convert router_settings to JSON string
156 router_settings_json: Final = json.dumps(router_settings)
157 await ConfigRepository(prisma_client).table.upsert(
158 where={"param_name": "router_settings"},
159 data={
160 "create": {
161 "param_name": "router_settings",
162 "param_value": router_settings_json,
163 },
164 "update": {"param_value": router_settings_json},
165 },
166 )
168 # Update the in-memory router configuration
169 setattr(llm_router, fallback_key, existing_fallbacks)
171 verbose_proxy_logger.info(
172 "Fallback configured: %s -> %s (type: %s)", data.model, data.fallback_models, data.fallback_type
173 )
175 return FallbackResponse(
176 model=data.model,
177 fallback_models=data.fallback_models,
178 fallback_type=data.fallback_type,
179 message=f"Fallback configuration {'updated' if fallback_updated else 'created'} successfully",
180 )
182 except HTTPException:
183 raise
184 except Exception as e:
185 verbose_proxy_logger.error("Error creating fallback: %s", e, exc_info=True)
186 raise HTTPException(
187 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
188 detail={"error": f"Failed to create fallback: {e}"},
189 )
192@router.get(
193 "/fallback/{model}",
194 tags=["Fallback Management"],
195 dependencies=[Depends(user_api_key_auth)],
196 response_model=FallbackGetResponse,
197)
198async def get_fallback(
199 model: str,
200 fallback_type: Literal["general", "context_window", "content_policy"] = "general",
201 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
202):
203 """
204 Get fallback configuration for a specific model.
206 **Parameters:**
207 - `model`: The model name to get fallbacks for
208 - `fallback_type`: Type of fallback to retrieve (query parameter)
210 **Example:**
211 ```
212 GET /fallback/gpt-3.5-turbo?fallback_type=general
213 ```
214 """
215 from litellm.proxy.proxy_server import llm_router
217 try:
218 if llm_router is None: 218 ↛ 219line 218 didn't jump to line 219 because the condition on line 218 was never true
219 raise HTTPException(
220 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
221 detail={"error": "Router not initialized"},
222 )
224 # Get fallbacks using the existing utility function
225 fallback_models: Final = get_all_fallbacks(model=model, llm_router=llm_router, fallback_type=fallback_type)
227 if not fallback_models: 227 ↛ 233line 227 didn't jump to line 233 because the condition on line 227 was always true
228 raise HTTPException(
229 status_code=status.HTTP_404_NOT_FOUND,
230 detail={"error": f"No {fallback_type} fallbacks configured for model '{model}'"},
231 )
233 return FallbackGetResponse(
234 model=model,
235 fallback_models=fallback_models,
236 fallback_type=fallback_type,
237 )
239 except HTTPException:
240 raise
241 except Exception as e:
242 verbose_proxy_logger.error("Error getting fallback: %s", e, exc_info=True)
243 raise HTTPException(
244 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
245 detail={"error": f"Failed to get fallback: {e}"},
246 )
249@router.delete(
250 "/fallback/{model}",
251 tags=["Fallback Management"],
252 dependencies=[Depends(user_api_key_auth)],
253 response_model=FallbackDeleteResponse,
254)
255async def delete_fallback(
256 model: str,
257 fallback_type: Literal["general", "context_window", "content_policy"] = "general",
258 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
259):
260 """
261 Delete fallback configuration for a specific model.
263 **Parameters:**
264 - `model`: The model name to delete fallbacks for
265 - `fallback_type`: Type of fallback to delete (query parameter)
267 **Example:**
268 ```
269 DELETE /fallback/gpt-3.5-turbo?fallback_type=general
270 ```
271 """
272 from litellm.proxy.proxy_server import (
273 llm_router,
274 prisma_client,
275 proxy_config,
276 store_model_in_db,
277 )
279 try:
280 if llm_router is None: 280 ↛ 281line 280 didn't jump to line 281 because the condition on line 280 was never true
281 raise HTTPException(
282 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
283 detail={"error": "Router not initialized"},
284 )
286 if store_model_in_db is not True or prisma_client is None: 286 ↛ 287line 286 didn't jump to line 287 because the condition on line 286 was never true
287 raise HTTPException(
288 status_code=status.HTTP_400_BAD_REQUEST,
289 detail={
290 "error": "Database storage not enabled. Set 'STORE_MODEL_IN_DB=True' in your environment to use this feature."
291 },
292 )
294 # Load existing config
295 config: Final = await proxy_config.get_config()
296 router_settings: Final = config.get("router_settings", {})
298 # Get the appropriate fallback list based on type
299 fallback_key = "fallbacks"
300 if fallback_type == "context_window":
301 fallback_key = "context_window_fallbacks"
302 elif fallback_type == "content_policy":
303 fallback_key = "content_policy_fallbacks"
305 # Get existing fallbacks
306 existing_fallbacks: Final[list[dict[str, list[str]]]] = router_settings.get(fallback_key, [])
308 # Find and remove the fallback configuration
309 fallback_found = False
310 updated_fallbacks: Final = []
311 for fallback_dict in existing_fallbacks: 311 ↛ 312line 311 didn't jump to line 312 because the loop on line 311 never started
312 if model not in fallback_dict:
313 updated_fallbacks.append(fallback_dict)
314 else:
315 fallback_found = True
317 if not fallback_found: 317 ↛ 324line 317 didn't jump to line 324 because the condition on line 317 was always true
318 raise HTTPException(
319 status_code=status.HTTP_404_NOT_FOUND,
320 detail={"error": f"No {fallback_type} fallbacks configured for model '{model}'"},
321 )
323 # Update router settings
324 router_settings[fallback_key] = updated_fallbacks
326 # Save to database - convert router_settings to JSON string
327 router_settings_json: Final = json.dumps(router_settings)
328 await ConfigRepository(prisma_client).table.upsert(
329 where={"param_name": "router_settings"},
330 data={
331 "create": {
332 "param_name": "router_settings",
333 "param_value": router_settings_json,
334 },
335 "update": {"param_value": router_settings_json},
336 },
337 )
339 # Update the in-memory router configuration
340 setattr(llm_router, fallback_key, updated_fallbacks)
342 verbose_proxy_logger.info("Fallback deleted: %s (type: %s)", model, fallback_type)
344 return FallbackDeleteResponse(
345 model=model,
346 fallback_type=fallback_type,
347 message="Fallback configuration deleted successfully",
348 )
350 except HTTPException:
351 raise
352 except Exception as e:
353 verbose_proxy_logger.error("Error deleting fallback: %s", e, exc_info=True)
354 raise HTTPException(
355 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
356 detail={"error": f"Failed to delete fallback: {e}"},
357 )