Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/management_endpoints/fallback_management_endpoints.py: 45%

108 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2FALLBACK MANAGEMENT ENDPOINTS 

3 

4Dedicated endpoints for managing model fallbacks separately from general config. 

5 

6POST /fallback - Create or update fallbacks for a specific model 

7GET /fallback/{model} - Get fallbacks for a specific model 

8DELETE /fallback/{model} - Delete fallbacks for a specific model 

9""" 

10 

11# pyright: reportMissingImports=false 

12 

13import json 

14from typing import TYPE_CHECKING, Final, Literal 

15 

16from litellm._logging import verbose_proxy_logger 

17from litellm.proxy._types import UserAPIKeyAuth 

18from litellm.proxy.auth.model_checks import get_all_fallbacks 

19from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

20 

21if TYPE_CHECKING: 21 ↛ 22line 21 didn't jump to line 22 because the condition on line 21 was never true

22 from fastapi import APIRouter, Depends, HTTPException, status 

23else: 

24 try: 

25 from fastapi import APIRouter, Depends, HTTPException, status 

26 except ImportError: 

27 # fastapi is only required for proxy, not for SDK usage 

28 pass 

29 

30from litellm.repositories.config_repository import ConfigRepository 

31from litellm.types.management_endpoints.router_settings_endpoints import ( 

32 FallbackCreateRequest, 

33 FallbackDeleteResponse, 

34 FallbackGetResponse, 

35 FallbackResponse, 

36) 

37 

38router: Final = APIRouter() 

39 

40 

41@router.post( 

42 "/fallback", 

43 tags=["Fallback Management"], 

44 dependencies=[Depends(user_api_key_auth)], 

45 response_model=FallbackResponse, 

46 status_code=status.HTTP_200_OK, 

47) 

48async def create_fallback( 

49 data: FallbackCreateRequest, 

50 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

51): 

52 """ 

53 Create or update fallbacks for a specific model. 

54 

55 This endpoint allows you to configure fallback models separately from the general config. 

56 Fallbacks are triggered when a model call fails after retries. 

57 

58 **Example Request:** 

59 ```json 

60 { 

61 "model": "gpt-3.5-turbo", 

62 "fallback_models": ["gpt-4", "claude-3-haiku"], 

63 "fallback_type": "general" 

64 } 

65 ``` 

66 

67 **Fallback Types:** 

68 - `general`: Standard fallbacks for any error (default) 

69 - `context_window`: Fallbacks specifically for context window exceeded errors 

70 - `content_policy`: Fallbacks specifically for content policy violations 

71 """ 

72 from litellm.proxy.proxy_server import ( 

73 llm_router, 

74 prisma_client, 

75 proxy_config, 

76 store_model_in_db, 

77 ) 

78 

79 try: 

80 # Validate that we have a router 

81 if llm_router is None: 81 ↛ 82line 81 didn't jump to line 82 because the condition on line 81 was never true

82 raise HTTPException( 

83 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

84 detail={"error": "Router not initialized"}, 

85 ) 

86 

87 # Validate that the model exists in the router 

88 model_names: Final = llm_router.model_names 

89 if data.model not in model_names: 89 ↛ 99line 89 didn't jump to line 99 because the condition on line 89 was always true

90 raise HTTPException( 

91 status_code=status.HTTP_404_NOT_FOUND, 

92 detail={ 

93 "error": f"Model '{data.model}' not found in router", 

94 "available_models": list(model_names), 

95 }, 

96 ) 

97 

98 # Validate that all fallback models exist in the router 

99 invalid_fallback_models: Final = [m for m in data.fallback_models if m not in model_names] 

100 if invalid_fallback_models: 

101 raise HTTPException( 

102 status_code=status.HTTP_400_BAD_REQUEST, 

103 detail={ 

104 "error": f"Invalid fallback models: {invalid_fallback_models}", 

105 "available_models": list(model_names), 

106 }, 

107 ) 

108 

109 # Check if fallback model is the same as the primary model 

110 if data.model in data.fallback_models: 

111 raise HTTPException( 

112 status_code=status.HTTP_400_BAD_REQUEST, 

113 detail={"error": f"Model '{data.model}' cannot be its own fallback"}, 

114 ) 

115 

116 # Check if we need to store in DB 

117 if store_model_in_db is not True or prisma_client is None: 

118 raise HTTPException( 

119 status_code=status.HTTP_400_BAD_REQUEST, 

120 detail={ 

121 "error": "Database storage not enabled. Set 'STORE_MODEL_IN_DB=True' in your environment to use this feature." 

122 }, 

123 ) 

124 

125 # Load existing config 

126 config: Final = await proxy_config.get_config() 

127 router_settings: Final = config.get("router_settings", {}) 

128 

129 # Get the appropriate fallback list based on type 

130 fallback_key = "fallbacks" 

131 if data.fallback_type == "context_window": 

132 fallback_key = "context_window_fallbacks" 

133 elif data.fallback_type == "content_policy": 

134 fallback_key = "content_policy_fallbacks" 

135 

136 # Get existing fallbacks 

137 existing_fallbacks: Final[list[dict[str, list[str]]]] = router_settings.get(fallback_key, []) 

138 

139 # Update or add the fallback configuration 

140 fallback_updated = False 

141 for i, fallback_dict in enumerate(existing_fallbacks): 

142 if data.model in fallback_dict: 

143 # Update existing fallback 

144 existing_fallbacks[i] = {data.model: data.fallback_models} 

145 fallback_updated = True 

146 break 

147 

148 if not fallback_updated: 

149 # Add new fallback 

150 existing_fallbacks.append({data.model: data.fallback_models}) 

151 

152 # Update router settings 

153 router_settings[fallback_key] = existing_fallbacks 

154 

155 # Save to database - convert router_settings to JSON string 

156 router_settings_json: Final = json.dumps(router_settings) 

157 await ConfigRepository(prisma_client).table.upsert( 

158 where={"param_name": "router_settings"}, 

159 data={ 

160 "create": { 

161 "param_name": "router_settings", 

162 "param_value": router_settings_json, 

163 }, 

164 "update": {"param_value": router_settings_json}, 

165 }, 

166 ) 

167 

168 # Update the in-memory router configuration 

169 setattr(llm_router, fallback_key, existing_fallbacks) 

170 

171 verbose_proxy_logger.info( 

172 "Fallback configured: %s -> %s (type: %s)", data.model, data.fallback_models, data.fallback_type 

173 ) 

174 

175 return FallbackResponse( 

176 model=data.model, 

177 fallback_models=data.fallback_models, 

178 fallback_type=data.fallback_type, 

179 message=f"Fallback configuration {'updated' if fallback_updated else 'created'} successfully", 

180 ) 

181 

182 except HTTPException: 

183 raise 

184 except Exception as e: 

185 verbose_proxy_logger.error("Error creating fallback: %s", e, exc_info=True) 

186 raise HTTPException( 

187 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

188 detail={"error": f"Failed to create fallback: {e}"}, 

189 ) 

190 

191 

192@router.get( 

193 "/fallback/{model}", 

194 tags=["Fallback Management"], 

195 dependencies=[Depends(user_api_key_auth)], 

196 response_model=FallbackGetResponse, 

197) 

198async def get_fallback( 

199 model: str, 

200 fallback_type: Literal["general", "context_window", "content_policy"] = "general", 

201 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

202): 

203 """ 

204 Get fallback configuration for a specific model. 

205 

206 **Parameters:** 

207 - `model`: The model name to get fallbacks for 

208 - `fallback_type`: Type of fallback to retrieve (query parameter) 

209 

210 **Example:** 

211 ``` 

212 GET /fallback/gpt-3.5-turbo?fallback_type=general 

213 ``` 

214 """ 

215 from litellm.proxy.proxy_server import llm_router 

216 

217 try: 

218 if llm_router is None: 218 ↛ 219line 218 didn't jump to line 219 because the condition on line 218 was never true

219 raise HTTPException( 

220 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

221 detail={"error": "Router not initialized"}, 

222 ) 

223 

224 # Get fallbacks using the existing utility function 

225 fallback_models: Final = get_all_fallbacks(model=model, llm_router=llm_router, fallback_type=fallback_type) 

226 

227 if not fallback_models: 227 ↛ 233line 227 didn't jump to line 233 because the condition on line 227 was always true

228 raise HTTPException( 

229 status_code=status.HTTP_404_NOT_FOUND, 

230 detail={"error": f"No {fallback_type} fallbacks configured for model '{model}'"}, 

231 ) 

232 

233 return FallbackGetResponse( 

234 model=model, 

235 fallback_models=fallback_models, 

236 fallback_type=fallback_type, 

237 ) 

238 

239 except HTTPException: 

240 raise 

241 except Exception as e: 

242 verbose_proxy_logger.error("Error getting fallback: %s", e, exc_info=True) 

243 raise HTTPException( 

244 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

245 detail={"error": f"Failed to get fallback: {e}"}, 

246 ) 

247 

248 

249@router.delete( 

250 "/fallback/{model}", 

251 tags=["Fallback Management"], 

252 dependencies=[Depends(user_api_key_auth)], 

253 response_model=FallbackDeleteResponse, 

254) 

255async def delete_fallback( 

256 model: str, 

257 fallback_type: Literal["general", "context_window", "content_policy"] = "general", 

258 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

259): 

260 """ 

261 Delete fallback configuration for a specific model. 

262 

263 **Parameters:** 

264 - `model`: The model name to delete fallbacks for 

265 - `fallback_type`: Type of fallback to delete (query parameter) 

266 

267 **Example:** 

268 ``` 

269 DELETE /fallback/gpt-3.5-turbo?fallback_type=general 

270 ``` 

271 """ 

272 from litellm.proxy.proxy_server import ( 

273 llm_router, 

274 prisma_client, 

275 proxy_config, 

276 store_model_in_db, 

277 ) 

278 

279 try: 

280 if llm_router is None: 280 ↛ 281line 280 didn't jump to line 281 because the condition on line 280 was never true

281 raise HTTPException( 

282 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

283 detail={"error": "Router not initialized"}, 

284 ) 

285 

286 if store_model_in_db is not True or prisma_client is None: 286 ↛ 287line 286 didn't jump to line 287 because the condition on line 286 was never true

287 raise HTTPException( 

288 status_code=status.HTTP_400_BAD_REQUEST, 

289 detail={ 

290 "error": "Database storage not enabled. Set 'STORE_MODEL_IN_DB=True' in your environment to use this feature." 

291 }, 

292 ) 

293 

294 # Load existing config 

295 config: Final = await proxy_config.get_config() 

296 router_settings: Final = config.get("router_settings", {}) 

297 

298 # Get the appropriate fallback list based on type 

299 fallback_key = "fallbacks" 

300 if fallback_type == "context_window": 

301 fallback_key = "context_window_fallbacks" 

302 elif fallback_type == "content_policy": 

303 fallback_key = "content_policy_fallbacks" 

304 

305 # Get existing fallbacks 

306 existing_fallbacks: Final[list[dict[str, list[str]]]] = router_settings.get(fallback_key, []) 

307 

308 # Find and remove the fallback configuration 

309 fallback_found = False 

310 updated_fallbacks: Final = [] 

311 for fallback_dict in existing_fallbacks: 311 ↛ 312line 311 didn't jump to line 312 because the loop on line 311 never started

312 if model not in fallback_dict: 

313 updated_fallbacks.append(fallback_dict) 

314 else: 

315 fallback_found = True 

316 

317 if not fallback_found: 317 ↛ 324line 317 didn't jump to line 324 because the condition on line 317 was always true

318 raise HTTPException( 

319 status_code=status.HTTP_404_NOT_FOUND, 

320 detail={"error": f"No {fallback_type} fallbacks configured for model '{model}'"}, 

321 ) 

322 

323 # Update router settings 

324 router_settings[fallback_key] = updated_fallbacks 

325 

326 # Save to database - convert router_settings to JSON string 

327 router_settings_json: Final = json.dumps(router_settings) 

328 await ConfigRepository(prisma_client).table.upsert( 

329 where={"param_name": "router_settings"}, 

330 data={ 

331 "create": { 

332 "param_name": "router_settings", 

333 "param_value": router_settings_json, 

334 }, 

335 "update": {"param_value": router_settings_json}, 

336 }, 

337 ) 

338 

339 # Update the in-memory router configuration 

340 setattr(llm_router, fallback_key, updated_fallbacks) 

341 

342 verbose_proxy_logger.info("Fallback deleted: %s (type: %s)", model, fallback_type) 

343 

344 return FallbackDeleteResponse( 

345 model=model, 

346 fallback_type=fallback_type, 

347 message="Fallback configuration deleted successfully", 

348 ) 

349 

350 except HTTPException: 

351 raise 

352 except Exception as e: 

353 verbose_proxy_logger.error("Error deleting fallback: %s", e, exc_info=True) 

354 raise HTTPException( 

355 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

356 detail={"error": f"Failed to delete fallback: {e}"}, 

357 )