Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/health_endpoints/_health_endpoints.py: 47%

735 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1import asyncio 

2import copy 

3import json 

4import logging 

5import os 

6import secrets 

7import time 

8import traceback 

9from collections.abc import Iterable, Mapping 

10from datetime import datetime, timedelta, timezone 

11from typing import Any, Final, Literal, TypedDict, cast 

12 

13import fastapi 

14from fastapi import APIRouter, Depends, HTTPException, Request, Response, status 

15from typing_extensions import ReadOnly 

16 

17import litellm 

18from litellm._logging import verbose_logger, verbose_proxy_logger 

19from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS, PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS 

20from litellm.integrations.SlackAlerting.ms_teams import ( 

21 MS_TEAMS_ALERT_HEADERS, 

22 build_ms_teams_payload, 

23 get_ms_teams_webhook_url, 

24) 

25from litellm.litellm_core_utils.custom_logger_registry import CustomLoggerRegistry 

26from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler 

27from litellm.proxy._types import ( 

28 AlertType, 

29 CallInfo, 

30 EnterpriseLicenseData, 

31 Litellm_EntityType, 

32 LitellmUserRoles, 

33 ProxyErrorTypes, 

34 ProxyException, 

35 SpecialModelNames, 

36 UserAPIKeyAuth, 

37 WebhookEvent, 

38) 

39from litellm.proxy.auth.auth_checks import ( 

40 _resolve_key_models_for_auth_check, # pyright: ignore[reportPrivateUsage] # the auth layer's sentinel resolution, reused so /health scopes exactly like a request 

41) 

42from litellm.proxy.auth.auth_utils import ( 

43 _BANNED_REQUEST_BODY_PARAMS, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the request-body check 

44) 

45from litellm.proxy.auth.model_checks import get_key_models 

46from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

47from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker 

48from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler 

49from litellm.proxy.db.health_check_latest import ( 

50 LatestHealthCheckRow, 

51 query_latest_health_checks, 

52) 

53from litellm.proxy.db.proxy_worker_heartbeat import count_live_proxy_workers 

54from litellm.proxy.health_check import ( 

55 ADMIN_ONLY_HEALTH_DISPLAY_PARAMS, 

56 _clean_endpoint_data, 

57 _update_litellm_params_for_health_check, 

58 deployments_targeted_by_name, 

59 health_check_filter_kwargs_from_general_settings, 

60 perform_health_check, 

61 run_with_timeout, 

62) 

63from litellm.proxy.middleware.admission_control_middleware import ( 

64 get_admission_control_stats, 

65) 

66from litellm.proxy.middleware.in_flight_requests_middleware import ( 

67 get_in_flight_requests, 

68) 

69from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager 

70from litellm.router import Router 

71from litellm.router_utils.clientside_credential_handler import ( 

72 _ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the router path 

73 clientside_credential_keys, 

74) 

75from litellm.secret_managers.main import get_secret_bool 

76 

77#### Health ENDPOINTS #### 

78 

79 

80class _HealthBacklogResponse(TypedDict): 

81 in_flight_requests: ReadOnly[int] 

82 admitted_requests: ReadOnly[int] 

83 queued_requests: ReadOnly[int] 

84 rejected_requests: ReadOnly[int] 

85 

86 

87def _reject_os_environ_references(params: dict) -> None: 

88 """ 

89 Validate that the provided params do not contain any ``os.environ/`` 

90 references. Values with that prefix are expected to come only from 

91 server-side configuration (already resolved before reaching here). If a 

92 request-supplied value still carries the prefix, raise ``HTTPException``. 

93 """ 

94 if not isinstance(params, dict): 94 ↛ 95line 94 didn't jump to line 95 because the condition on line 94 was never true

95 return 

96 

97 stack: Final[list[object]] = [params] 

98 seen: Final[set[int]] = {id(params)} 

99 

100 while stack: 

101 src = stack.pop() 

102 if isinstance(src, dict): 

103 values: Iterable[object] = src.values() 

104 elif isinstance(src, list): 104 ↛ 107line 104 didn't jump to line 107 because the condition on line 104 was always true

105 values = src 

106 else: 

107 continue 

108 

109 for value in values: 

110 if isinstance(value, str) and value.startswith("os.environ/"): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true

111 raise HTTPException( 

112 status_code=400, 

113 detail={"error": "Environment variable references are not permitted in request parameters."}, 

114 ) 

115 if isinstance(value, (dict, list)) and id(value) not in seen: 

116 seen.add(id(value)) 

117 stack.append(value) 

118 

119 

120_CONFIG_CONNECTION_FIELDS: Final[frozenset[str]] = frozenset( 

121 ( 

122 *_ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE, 

123 *clientside_credential_keys, 

124 "litellm_credential_name", 

125 ) 

126) 

127 

128 

129def _request_inherits_config_credentials( 

130 config_params: Mapping[str, object], 

131 request_params: Mapping[str, object], 

132 allow_client_side_credentials: bool, 

133) -> bool: 

134 """Whether the configuration's credentials are this request's to be probed with. 

135 

136 The configuration reached here by matching the request's model string, which 

137 also matches wildcard routes and unrelated deployments that merely serve the 

138 same model, so a request naming a stored credential of its own has already 

139 said where its credentials come from and does not borrow that one's. A blank 

140 name is no name: ``load_credentials_from_list`` resolves nothing from it, so 

141 it must not cost the request the credentials it would otherwise be probed 

142 with. 

143 """ 

144 requested_credential: Final = request_params.get("litellm_credential_name") 

145 if requested_credential and requested_credential != config_params.get("litellm_credential_name"): 145 ↛ 146line 145 didn't jump to line 146 because the condition on line 145 was never true

146 return False 

147 if allow_client_side_credentials: 147 ↛ 148line 147 didn't jump to line 148 because the condition on line 147 was never true

148 return True 

149 return not any(param in request_params for param in _BANNED_REQUEST_BODY_PARAMS) 

150 

151 

152def _config_base_for_health_check( 

153 config_params: Mapping[str, object], 

154 request_params: Mapping[str, object], 

155 allow_client_side_credentials: bool = False, 

156) -> dict[str, object]: 

157 """Return the configured parameters to merge under a connection-test request. 

158 

159 A request that sets its own connection fields, or names its own stored 

160 credential, describes a connection of its own, so the configuration's 

161 credentials are not carried into it: they belong to the endpoint the 

162 configuration names. Anything the request does not set still comes from the 

163 configuration, which is what lets a request name a configured model and test 

164 it as configured. 

165 

166 ``litellm_credential_name`` is dropped alongside the literal credential 

167 fields: it names a stored credential that ``load_credentials_from_list`` 

168 resolves into the same secrets further down the call, so leaving it in place 

169 would reintroduce them by reference. 

170 """ 

171 if _request_inherits_config_credentials(config_params, request_params, allow_client_side_credentials): 171 ↛ 173line 171 didn't jump to line 173 because the condition on line 171 was always true

172 return dict(config_params) 

173 return {key: value for key, value in config_params.items() if key not in _CONFIG_CONNECTION_FIELDS} 

174 

175 

176def get_callback_identifier(callback): 

177 """ 

178 Get the callback identifier string, handling both strings and objects. 

179 

180 This function extracts a string identifier from a callback, which can be: 

181 - A string (returned as-is) 

182 - An object with a callback_name attribute 

183 - An object registered in CustomLoggerRegistry 

184 - Falls back to callback_name() helper function 

185 

186 Args: 

187 callback: The callback to identify (can be str or object) 

188 

189 Returns: 

190 str: The callback identifier string 

191 """ 

192 if isinstance(callback, str): 

193 return callback 

194 if hasattr(callback, "callback_name") and callback.callback_name: 194 ↛ 195line 194 didn't jump to line 195 because the condition on line 194 was never true

195 return callback.callback_name 

196 if hasattr(callback, "__class__"): 196 ↛ 202line 196 didn't jump to line 202 because the condition on line 196 was always true

197 callback_strs: Final = CustomLoggerRegistry.get_all_callback_strs_from_class_type(callback.__class__) 

198 if hasattr(callback, "callback_name") and callback.callback_name in callback_strs: 198 ↛ 199line 198 didn't jump to line 199 because the condition on line 198 was never true

199 return callback.callback_name 

200 if callback_strs: 200 ↛ 201line 200 didn't jump to line 201 because the condition on line 200 was never true

201 return callback_strs[0] 

202 return callback_name(callback) 

203 

204 

205router: Final = APIRouter() 

206services = ( 

207 Literal[ 

208 "slack_budget_alerts", 

209 "langfuse", 

210 "langfuse_otel", 

211 "slack", 

212 "ms_teams", 

213 "openmeter", 

214 "webhook", 

215 "email", 

216 "braintrust", 

217 "datadog", 

218 "datadog_llm_observability", 

219 "generic_api", 

220 "arize", 

221 "galileo", 

222 "newrelic", 

223 "pointfive", 

224 "sqs", 

225 ] 

226 | str 

227) 

228 

229 

230class _ServiceTestErrorDetail(TypedDict): 

231 error: ReadOnly[str] 

232 

233 

234class _ServiceTestSuccessResponse(TypedDict): 

235 status: ReadOnly[str] 

236 message: ReadOnly[str] 

237 

238 

239@router.get( 

240 "/test", 

241 tags=["health"], 

242 dependencies=[Depends(user_api_key_auth)], 

243) 

244async def test_endpoint(request: Request): 

245 """ 

246 [DEPRECATED] use `/health/liveliness` instead. 

247 

248 A test endpoint that pings the proxy server to check if it's healthy. 

249 

250 Parameters: 

251 request (Request): The incoming request. 

252 

253 Returns: 

254 dict: A dictionary containing the route of the request URL. 

255 """ 

256 # ping the proxy server to check if its healthy 

257 # Inline import — auth_utils participates in a proxy import cycle. 

258 from litellm.proxy.auth.auth_utils import get_request_route # noqa: PLC0415 

259 

260 return {"route": get_request_route(request)} 

261 

262 

263@router.get( 

264 "/health/services", 

265 tags=["health"], 

266 dependencies=[Depends(user_api_key_auth)], 

267) 

268async def health_services_endpoint( 

269 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

270 service: services = fastapi.Query(description="Specify the service being hit."), 

271): 

272 """ 

273 Use this admin-only endpoint to check if the service is healthy. 

274 

275 Example: 

276 ``` 

277 curl -L -X GET 'http://0.0.0.0:4000/health/services?service=datadog' \ 

278 -H 'Authorization: Bearer sk-1234' 

279 ``` 

280 """ 

281 try: 

282 from litellm.proxy.proxy_server import ( 

283 general_settings, 

284 prisma_client, 

285 proxy_logging_obj, 

286 ) 

287 

288 if service is None: 288 ↛ 289line 288 didn't jump to line 289 because the condition on line 288 was never true

289 raise HTTPException(status_code=400, detail={"error": "Service must be specified."}) 

290 

291 if service not in [ 

292 "slack_budget_alerts", 

293 "email", 

294 "langfuse", 

295 "langfuse_otel", 

296 "slack", 

297 "ms_teams", 

298 "openmeter", 

299 "webhook", 

300 "braintrust", 

301 "otel", 

302 "custom_callback_api", 

303 "langsmith", 

304 "datadog", 

305 "datadog_metrics", 

306 "datadog_llm_observability", 

307 "generic_api", 

308 "arize", 

309 "galileo", 

310 "newrelic", 

311 "pointfive", 

312 "sqs", 

313 ]: 

314 raise HTTPException( 

315 status_code=400, 

316 detail={"error": f"Service must be in list. Service={service} not in {services}"}, 

317 ) 

318 

319 service_in_success_callbacks = False 

320 if service in litellm.success_callback: 320 ↛ 321line 320 didn't jump to line 321 because the condition on line 320 was never true

321 service_in_success_callbacks = True 

322 else: 

323 for cb in litellm.success_callback: 

324 if getattr(cb, "callback_name", None) == service: 324 ↛ 325line 324 didn't jump to line 325 because the condition on line 324 was never true

325 service_in_success_callbacks = True 

326 break 

327 cb_id = get_callback_identifier(cb) 

328 if cb_id == service: 328 ↛ 329line 328 didn't jump to line 329 because the condition on line 328 was never true

329 service_in_success_callbacks = True 

330 break 

331 

332 if ( 332 ↛ 338line 332 didn't jump to line 338 because the condition on line 332 was never true

333 service == "openmeter" 

334 or service == "braintrust" 

335 or service == "generic_api" 

336 or (service_in_success_callbacks and service not in ("langfuse", "pointfive")) 

337 ): 

338 _ = await litellm.acompletion( 

339 model="openai/litellm-mock-response-model", 

340 messages=[{"role": "user", "content": "Hey, how's it going?"}], 

341 user="litellm:/health/services", 

342 mock_response="This is a mock response", 

343 ) 

344 return { 

345 "status": "success", 

346 "message": f"Mock LLM request made - check {service}.", 

347 } 

348 elif service == "datadog": 348 ↛ 349line 348 didn't jump to line 349 because the condition on line 348 was never true

349 from litellm.integrations.datadog.datadog import DataDogLogger 

350 

351 datadog_logger: Final = DataDogLogger() 

352 response = await datadog_logger.async_health_check() 

353 return { 

354 "status": response["status"], 

355 "message": (response["error_message"] if response["status"] == "unhealthy" else "Datadog is healthy"), 

356 } 

357 elif service == "datadog_metrics": 357 ↛ 358line 357 didn't jump to line 358 because the condition on line 357 was never true

358 from litellm.integrations.datadog.datadog_metrics import ( 

359 DatadogMetricsLogger, 

360 ) 

361 from litellm.litellm_core_utils.litellm_logging import ( 

362 get_custom_logger_compatible_class, 

363 ) 

364 

365 datadog_metrics_logger = get_custom_logger_compatible_class("datadog_metrics") 

366 if datadog_metrics_logger is None: 

367 datadog_metrics_logger = DatadogMetricsLogger(start_periodic_flush=False) 

368 assert isinstance(datadog_metrics_logger, DatadogMetricsLogger) 

369 response = await datadog_metrics_logger.async_health_check() 

370 return { 

371 "status": response["status"], 

372 "message": ( 

373 response["error_message"] if response["status"] == "unhealthy" else "Datadog Metrics is healthy" 

374 ), 

375 } 

376 elif service == "arize": 376 ↛ 377line 376 didn't jump to line 377 because the condition on line 376 was never true

377 from litellm.integrations.arize.arize import ArizeLogger 

378 

379 arize_logger: Final = ArizeLogger() 

380 response = await arize_logger.async_health_check() 

381 return { 

382 "status": response["status"], 

383 "message": (response["error_message"] if response["status"] == "unhealthy" else "Arize is healthy"), 

384 } 

385 elif service == "galileo": 385 ↛ 386line 385 didn't jump to line 386 because the condition on line 385 was never true

386 from litellm.integrations.galileo import GalileoObserve 

387 

388 galileo_logger: Final = GalileoObserve() 

389 response = await galileo_logger.async_health_check() 

390 return { 

391 "status": response["status"], 

392 "message": (response["error_message"] if response["status"] == "unhealthy" else "Galileo is healthy"), 

393 } 

394 elif service == "langfuse": 394 ↛ 395line 394 didn't jump to line 395 because the condition on line 394 was never true

395 from litellm.integrations.langfuse.langfuse import LangFuseLogger 

396 

397 langfuse_logger: Final = LangFuseLogger() 

398 auth_failure: Final = langfuse_logger.api_client.auth_check() 

399 if auth_failure is not None: 

400 raise ValueError(f"langfuse auth_check failed: {auth_failure.reason}") 

401 _ = litellm.completion( 

402 model="openai/litellm-mock-response-model", 

403 messages=[{"role": "user", "content": "Hey, how's it going?"}], 

404 user="litellm:/health/services", 

405 mock_response="This is a mock response", 

406 ) 

407 return { 

408 "status": "success", 

409 "message": "Mock LLM request made - check langfuse.", 

410 } 

411 elif service == "newrelic": 411 ↛ 412line 411 didn't jump to line 412 because the condition on line 411 was never true

412 if not _is_proxy_admin(user_api_key_dict): 

413 raise HTTPException( 

414 status_code=status.HTTP_403_FORBIDDEN, 

415 detail={"error": "Only proxy admins can trigger the New Relic test event."}, 

416 ) 

417 from litellm.integrations.newrelic.newrelic import NewRelicLogger 

418 

419 newrelic_logger: Final = NewRelicLogger() 

420 response = await newrelic_logger.async_health_check() 

421 return { 

422 "status": response["status"], 

423 "message": ( 

424 response["error_message"] 

425 if response["status"] == "unhealthy" 

426 else "New Relic is healthy — test event sent" 

427 ), 

428 } 

429 

430 elif service == "pointfive": 430 ↛ 431line 430 didn't jump to line 431 because the condition on line 430 was never true

431 if not _is_proxy_admin(user_api_key_dict): 

432 non_admin_detail: Final[_ServiceTestErrorDetail] = { 

433 "error": "Only proxy admins can trigger the PointFive liveness ping." 

434 } 

435 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=non_admin_detail) 

436 from litellm.integrations.pointfive import PointFiveLogger 

437 

438 try: 

439 pointfive_logger: Final = PointFiveLogger(start_periodic_flush=False) 

440 except ValueError as missing_key: 

441 # No key configured is the answer the operator asked for, not a server error. 

442 no_key: Final[_ServiceTestSuccessResponse] = {"status": "unhealthy", "message": str(missing_key)} 

443 return no_key 

444 response = await pointfive_logger.async_health_check() 

445 pointfive_health: Final[_ServiceTestSuccessResponse] = { 

446 "status": response["status"], 

447 "message": (response["error_message"] if response["status"] == "unhealthy" else "PointFive is healthy") 

448 or "PointFive is healthy", 

449 } 

450 return pointfive_health 

451 if service == "webhook": 451 ↛ 452line 451 didn't jump to line 452 because the condition on line 451 was never true

452 if not _is_proxy_admin(user_api_key_dict): 

453 webhook_non_admin_detail: Final[_ServiceTestErrorDetail] = { 

454 "error": "Only proxy admins can trigger the webhook test alert." 

455 } 

456 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=webhook_non_admin_detail) 

457 user_info: Final = CallInfo( 

458 token=user_api_key_dict.token or "", 

459 spend=1, 

460 max_budget=0, 

461 user_id=user_api_key_dict.user_id, 

462 key_alias=user_api_key_dict.key_alias, 

463 team_id=user_api_key_dict.team_id, 

464 event_group=Litellm_EntityType.KEY, 

465 ) 

466 await proxy_logging_obj.budget_alerts( 

467 type="user_budget", 

468 user_info=user_info, 

469 ) 

470 elif service == "sqs": 470 ↛ 480line 470 didn't jump to line 480 because the condition on line 470 was always true

471 from litellm.integrations.sqs import SQSLogger 

472 

473 sqs_logger: Final = SQSLogger() 

474 response = await sqs_logger.async_health_check() 

475 return { 

476 "status": response["status"], 

477 "message": response["error_message"], 

478 } 

479 

480 if service == "slack" or service == "slack_budget_alerts": 

481 if "slack" in general_settings.get("alerting", []): 

482 # test_message = f"""\n🚨 `ProjectedLimitExceededError` 💸\n\n`Key Alias:` litellm-ui-test-alert \n`Expected Day of Error`: 28th March \n`Current Spend`: $100.00 \n`Projected Spend at end of month`: $1000.00 \n`Soft Limit`: $700""" 

483 # check if user has opted into unique_alert_webhooks 

484 if proxy_logging_obj.slack_alerting_instance.alert_to_webhook_url is not None: 

485 for alert_type in proxy_logging_obj.slack_alerting_instance.alert_to_webhook_url: 

486 # only test alert if it's in active alert types 

487 if ( 

488 proxy_logging_obj.slack_alerting_instance.alert_types is not None 

489 and alert_type not in proxy_logging_obj.slack_alerting_instance.alert_types 

490 ): 

491 continue 

492 

493 test_message = "default test message" 

494 if alert_type == AlertType.llm_exceptions: 

495 test_message = "LLM Exception test alert" 

496 elif alert_type == AlertType.llm_too_slow: 

497 test_message = "LLM Too Slow test alert" 

498 elif alert_type == AlertType.llm_requests_hanging: 

499 test_message = "LLM Requests Hanging test alert" 

500 elif alert_type == AlertType.budget_alerts: 

501 test_message = "Budget Alert test alert" 

502 elif alert_type == AlertType.db_exceptions: 

503 test_message = "DB Exception test alert" 

504 elif alert_type == AlertType.outage_alerts: 

505 test_message = "Outage Alert Exception test alert" 

506 elif alert_type == AlertType.daily_reports: 

507 test_message = "Daily Reports test alert" 

508 else: 

509 test_message = "Budget Alert test alert" 

510 

511 await proxy_logging_obj.alerting_handler( 

512 message=test_message, level="Low", alert_type=alert_type 

513 ) 

514 else: 

515 await proxy_logging_obj.alerting_handler( 

516 message="This is a test slack alert message", 

517 level="Low", 

518 alert_type=AlertType.budget_alerts, 

519 ) 

520 

521 if prisma_client is not None: 

522 asyncio.create_task(proxy_logging_obj.slack_alerting_instance.send_monthly_spend_report()) 

523 asyncio.create_task(proxy_logging_obj.slack_alerting_instance.send_weekly_spend_report()) 

524 

525 alert_types = proxy_logging_obj.slack_alerting_instance.alert_types or [] 

526 alert_types = list(alert_types) 

527 return { 

528 "status": "success", 

529 "alert_types": alert_types, 

530 "message": "Mock Slack Alert sent, verify Slack Alert Received on your channel", 

531 } 

532 else: 

533 raise HTTPException( 

534 status_code=422, 

535 detail={"error": f'"{service}" not in proxy config: general_settings. Unable to test this.'}, 

536 ) 

537 if service == "ms_teams": 

538 if "ms_teams" not in general_settings.get("alerting", ()): 

539 not_configured_detail: Final[_ServiceTestErrorDetail] = { 

540 "error": f'"{service}" not in proxy config: general_settings. Unable to test this.' 

541 } 

542 raise HTTPException(status_code=422, detail=not_configured_detail) 

543 ms_teams_webhook_url: Final = get_ms_teams_webhook_url() 

544 if ms_teams_webhook_url is None: 

545 missing_webhook_detail: Final[_ServiceTestErrorDetail] = { 

546 "error": "MS_TEAMS_WEBHOOK_URL not set. Unable to test this." 

547 } 

548 raise HTTPException(status_code=422, detail=missing_webhook_detail) 

549 ms_teams_test_message: Final = ( 

550 f"Alert type: `{AlertType.budget_alerts.value}`\nLevel: `Low`\n" 

551 f"Timestamp: `{datetime.now().strftime('%H:%M:%S')}`\n\n" 

552 "Message: This is a test MS Teams alert message" 

553 ) 

554 ms_teams_response: Final = await proxy_logging_obj.slack_alerting_instance.async_http_handler.post( 

555 url=ms_teams_webhook_url, 

556 headers=dict(MS_TEAMS_ALERT_HEADERS), # mutable-ok: async_http_handler.post only accepts dict headers 

557 data=json.dumps(build_ms_teams_payload(ms_teams_test_message)), 

558 ) 

559 if ms_teams_response.status_code >= 400: 

560 delivery_failed_detail: Final[_ServiceTestErrorDetail] = { 

561 "error": f"MS Teams webhook returned status {ms_teams_response.status_code}: {ms_teams_response.text}" 

562 } 

563 raise HTTPException(status_code=500, detail=delivery_failed_detail) 

564 ms_teams_success: Final[_ServiceTestSuccessResponse] = { 

565 "status": "success", 

566 "message": "Mock MS Teams Alert sent, verify MS Teams Alert Received in your channel", 

567 } 

568 return ms_teams_success 

569 if service == "email": 

570 webhook_event: Final = WebhookEvent( 

571 event="key_created", 

572 event_group=Litellm_EntityType.KEY, 

573 event_message="Test Email Alert", 

574 token=user_api_key_dict.token or "", 

575 key_alias="Email Test key (This is only a test alert key. DO NOT USE THIS IN PRODUCTION.)", 

576 spend=0, 

577 max_budget=0, 

578 user_id=user_api_key_dict.user_id, 

579 user_email=os.getenv("TEST_EMAIL_ADDRESS"), 

580 team_id=user_api_key_dict.team_id, 

581 ) 

582 

583 # use create task - this can take 10 seconds. don't keep ui users waiting for notification to check their email 

584 await proxy_logging_obj.slack_alerting_instance.send_key_created_or_user_invited_email( 

585 webhook_event=webhook_event 

586 ) 

587 

588 return { 

589 "status": "success", 

590 "message": "Mock Email Alert sent, verify Email Alert Received", 

591 } 

592 

593 except Exception as e: 

594 verbose_proxy_logger.error("litellm.proxy.proxy_server.health_services_endpoint(): Exception occured - %s", e) 

595 verbose_proxy_logger.debug(traceback.format_exc()) 

596 if isinstance(e, HTTPException): 596 ↛ 603line 596 didn't jump to line 603 because the condition on line 596 was always true

597 raise ProxyException( 

598 message=getattr(e, "detail", f"Authentication Error({e})"), 

599 type=ProxyErrorTypes.auth_error, 

600 param=getattr(e, "param", "None"), 

601 code=getattr(e, "status_code", status.HTTP_500_INTERNAL_SERVER_ERROR), 

602 ) 

603 elif isinstance(e, ProxyException): 

604 raise e 

605 raise ProxyException( 

606 message="Authentication Error, " + str(e), 

607 type=ProxyErrorTypes.auth_error, 

608 param=getattr(e, "param", "None"), 

609 code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

610 ) 

611 

612 

613def _convert_health_check_to_dict(check) -> dict: 

614 """Convert health check database record to dictionary format""" 

615 return { 

616 "health_check_id": check.health_check_id, 

617 "model_name": check.model_name, 

618 "model_id": check.model_id, 

619 "status": check.status, 

620 "healthy_count": check.healthy_count, 

621 "unhealthy_count": check.unhealthy_count, 

622 "error_message": check.error_message, 

623 "response_time_ms": check.response_time_ms, 

624 "details": check.details, 

625 "checked_by": check.checked_by, 

626 "checked_at": check.checked_at.isoformat() if check.checked_at else None, 

627 "created_at": check.created_at.isoformat() if check.created_at else None, 

628 } 

629 

630 

631def _check_prisma_client(): 

632 """Helper to check if prisma_client is available and raise appropriate error""" 

633 from litellm.proxy.proxy_server import prisma_client 

634 

635 if prisma_client is None: 635 ↛ 636line 635 didn't jump to line 636 because the condition on line 635 was never true

636 raise HTTPException( 

637 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

638 detail={"error": "Database not initialized"}, 

639 ) 

640 return prisma_client 

641 

642 

643async def _save_health_check_to_db( 

644 prisma_client, 

645 model_name: str, 

646 healthy_endpoints: list, 

647 unhealthy_endpoints: list, 

648 start_time: float, 

649 user_id: str | None, 

650 model_id: str | None = None, 

651): 

652 """Helper function to save health check results to database""" 

653 try: 

654 # Extract error message from first unhealthy endpoint if available 

655 error_message: Final = ( 

656 str(unhealthy_endpoints[0]["error"])[:500] 

657 if unhealthy_endpoints and unhealthy_endpoints[0].get("error") 

658 else None 

659 ) 

660 

661 await prisma_client.save_health_check_result( 

662 model_name=model_name, 

663 model_id=model_id, 

664 status="healthy" if healthy_endpoints else "unhealthy", 

665 healthy_count=len(healthy_endpoints), 

666 unhealthy_count=len(unhealthy_endpoints), 

667 error_message=error_message, 

668 response_time_ms=(time.time() - start_time) * 1000, 

669 details=None, # Skip details for now to avoid JSON serialization issues 

670 checked_by=user_id, 

671 ) 

672 except Exception as db_error: 

673 verbose_proxy_logger.warning("Failed to save health check to database for model %s: %s", model_name, db_error) 

674 # Continue execution - don't let database save failure break health checks 

675 

676 

677def _build_model_param_to_info_mapping(model_list: list) -> dict: 

678 """ 

679 Build a mapping from model parameter to model info (model_name, model_id). 

680 

681 Multiple models might share the same model parameter, so we use a list. 

682 

683 Args: 

684 model_list: List of model configurations 

685 

686 Returns: 

687 Dictionary mapping model parameter to list of model info dicts 

688 """ 

689 model_param_to_info: Final[dict] = {} 

690 for model in model_list: 

691 model_info = model.get("model_info", {}) 

692 model_name = model.get("model_name") 

693 model_id = model_info.get("id") 

694 litellm_params = model.get("litellm_params", {}) 

695 model_param = litellm_params.get("model") 

696 

697 if model_param and model_name: 

698 if model_param not in model_param_to_info: 

699 model_param_to_info[model_param] = [] 

700 model_param_to_info[model_param].append( 

701 { 

702 "model_name": model_name, 

703 "model_id": model_id, 

704 } 

705 ) 

706 return model_param_to_info 

707 

708 

709def _aggregate_health_check_results( 

710 model_param_to_info: dict, 

711 healthy_endpoints: list, 

712 unhealthy_endpoints: list, 

713) -> dict: 

714 """ 

715 Aggregate health check results per unique model. 

716 

717 Uses (model_id, model_name) as key, or (None, model_name) if model_id is None. 

718 

719 Args: 

720 model_param_to_info: Mapping from model parameter to model info 

721 healthy_endpoints: List of healthy endpoint results 

722 unhealthy_endpoints: List of unhealthy endpoint results 

723 

724 Returns: 

725 Dictionary mapping (model_id, model_name) to aggregated health check results 

726 """ 

727 model_results: Final = {} 

728 

729 # Process healthy endpoints 

730 for endpoint in healthy_endpoints: 

731 model_param = endpoint.get("model") 

732 if model_param and model_param in model_param_to_info: 

733 for model_info in model_param_to_info[model_param]: 

734 key = (model_info["model_id"], model_info["model_name"]) 

735 if key not in model_results: 

736 model_results[key] = { 

737 "model_name": model_info["model_name"], 

738 "model_id": model_info["model_id"], 

739 "healthy_count": 0, 

740 "unhealthy_count": 0, 

741 "error_message": None, 

742 } 

743 model_results[key]["healthy_count"] += 1 

744 

745 # Process unhealthy endpoints 

746 for endpoint in unhealthy_endpoints: 

747 model_param = endpoint.get("model") 

748 error_message = endpoint.get("error") 

749 if model_param and model_param in model_param_to_info: 

750 for model_info in model_param_to_info[model_param]: 

751 key = (model_info["model_id"], model_info["model_name"]) 

752 if key not in model_results: 

753 model_results[key] = { 

754 "model_name": model_info["model_name"], 

755 "model_id": model_info["model_id"], 

756 "healthy_count": 0, 

757 "unhealthy_count": 0, 

758 "error_message": None, 

759 } 

760 model_results[key]["unhealthy_count"] += 1 

761 # Use the first error message encountered 

762 if not model_results[key]["error_message"] and error_message: 

763 model_results[key]["error_message"] = str(error_message)[:500] 

764 

765 return model_results 

766 

767 

768class _AggregatedHealthResult(TypedDict): 

769 """One entry of ``_aggregate_health_check_results``: a model's counts for this cycle.""" 

770 

771 model_name: ReadOnly[str] 

772 model_id: ReadOnly[str | None] 

773 healthy_count: ReadOnly[int] 

774 unhealthy_count: ReadOnly[int] 

775 error_message: ReadOnly[str | None] 

776 

777 

778def _new_health_status(result: _AggregatedHealthResult) -> str: 

779 return "healthy" if result["healthy_count"] > 0 else "unhealthy" 

780 

781 

782def _should_persist_health_check_result( 

783 result: _AggregatedHealthResult, latest_checks_map: Mapping[str, LatestHealthCheckRow] 

784) -> bool: 

785 """ 

786 True when this result has to be written: no previous row, the status changed, or the 

787 previous row is older than one hour (periodic refresh while the status is stable). 

788 """ 

789 lookup_key: Final = result["model_id"] if result["model_id"] else result["model_name"] 

790 last_check: Final = latest_checks_map.get(lookup_key) 

791 if last_check is None or last_check.status != _new_health_status(result): 

792 return True 

793 time_since_last_check: Final = (datetime.now(timezone.utc) - last_check.checked_at).total_seconds() 

794 return time_since_last_check >= 3600 # 1 hour threshold 

795 

796 

797async def _save_health_check_results_if_changed( 

798 prisma_client, 

799 model_results: dict, 

800 latest_checks_map: dict, 

801 start_time: float, 

802 checked_by: str | None = None, 

803) -> bool: 

804 """ 

805 Save health check results to database, but only if status changed or >1 hour since last save. 

806 

807 OPTIMIZATION: Only saves to database if the status has changed from the last saved check. 

808 This dramatically reduces database writes when health status remains stable. 

809 

810 - Stable systems: ~1 write/hour per model (instead of 12 writes/hour with 5-min intervals) 

811 - Status changes: Immediate write (no delay) 

812 - Result: ~92% reduction in DB writes for stable systems, while maintaining real-time updates on changes 

813 

814 The writes are awaited rather than detached so the caller learns whether this cycle's 

815 persistence completed. 

816 

817 Args: 

818 prisma_client: Database client 

819 model_results: Dictionary of aggregated health check results per model 

820 latest_checks_map: Dictionary mapping model_id/model_name to latest health check 

821 start_time: Start time of health check for calculating response time 

822 checked_by: Identifier for who/what performed the check 

823 

824 Returns: 

825 True when every row that needed writing was written (including when nothing needed 

826 writing); False when any write failed. 

827 """ 

828 to_write: Final = tuple( 

829 result for result in model_results.values() if _should_persist_health_check_result(result, latest_checks_map) 

830 ) 

831 writes: Final = tuple( 

832 prisma_client.save_health_check_result( 

833 model_name=result["model_name"], 

834 model_id=result["model_id"], 

835 status=_new_health_status(result), 

836 healthy_count=result["healthy_count"], 

837 unhealthy_count=result["unhealthy_count"], 

838 error_message=result["error_message"], 

839 response_time_ms=(time.time() - start_time) * 1000, 

840 details=None, 

841 checked_by=checked_by, 

842 ) 

843 for result in to_write 

844 ) 

845 rows: Final = await asyncio.gather(*writes) 

846 return all(row is not None for row in rows) 

847 

848 

849async def _save_background_health_checks_to_db( 

850 prisma_client, 

851 model_list: list, 

852 healthy_endpoints: list, 

853 unhealthy_endpoints: list, 

854 start_time: float, 

855 checked_by: str | None = None, 

856) -> bool: 

857 """ 

858 Save background health check results to database for each model. 

859 

860 Maps health check endpoints back to their original models to get model_name and model_id. 

861 Aggregates results per unique model (by model_id if available, otherwise model_name). 

862 

863 OPTIMIZATION: Only saves to database if the status has changed from the last saved check. 

864 This dramatically reduces database writes when health status remains stable. 

865 

866 Returns: 

867 True when this cycle's persistence completed; False when it was skipped or any step 

868 failed. Never raises: a database failure must not break the health check loop. 

869 """ 

870 if prisma_client is None: 

871 return False 

872 

873 try: 

874 # Step 1: Build mapping from model parameter to model info 

875 model_param_to_info: Final = _build_model_param_to_info_mapping(model_list) 

876 

877 # Step 2: Aggregate health check results per unique model 

878 model_results: Final = _aggregate_health_check_results( 

879 model_param_to_info, 

880 healthy_endpoints, 

881 unhealthy_endpoints, 

882 ) 

883 

884 # Step 3: Get latest health checks for all models in one query to compare status 

885 latest_checks: Final = await query_latest_health_checks(prisma_client) 

886 latest_checks_map: Final = {} 

887 for check in latest_checks: 

888 # Use model_id as primary key, fallback to model_name 

889 key = check.model_id if check.model_id else check.model_name 

890 if key not in latest_checks_map: 

891 latest_checks_map[key] = check 

892 

893 # Step 4: Save aggregated results, but only if status changed 

894 return await _save_health_check_results_if_changed( 

895 prisma_client, 

896 model_results, 

897 latest_checks_map, 

898 start_time, 

899 checked_by, 

900 ) 

901 except Exception as db_error: 

902 verbose_proxy_logger.warning("Failed to save background health checks to database: %s", db_error) 

903 # Continue execution - don't let database save failure break health checks 

904 return False 

905 

906 

907_PROXY_ADMIN_ROLES: Final = frozenset( 

908 { 

909 LitellmUserRoles.PROXY_ADMIN.value, 

910 # View-only admins are operators (oncall, support); they need the 

911 # routing fields (api_base, api_version) to diagnose health and tell 

912 # which provider region a check is hitting. They cannot mutate config 

913 # so granting them the read-only view is safe. 

914 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, 

915 } 

916) 

917 

918 

919def _is_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> bool: 

920 """ 

921 Return True if the caller has a proxy-admin role (full or view-only). 

922 

923 user_role on UserAPIKeyAuth can be either a LitellmUserRoles enum or its 

924 string value depending on how the auth path constructed the object, so we 

925 compare against the raw value rather than the enum identity. 

926 """ 

927 role: Final = user_api_key_dict.user_role 

928 if role is None: 928 ↛ 929line 928 didn't jump to line 929 because the condition on line 928 was never true

929 return False 

930 role_value: Final = role.value if hasattr(role, "value") else role 

931 return role_value in _PROXY_ADMIN_ROLES 

932 

933 

934def _strip_admin_only_fields_from_health_result(result: dict) -> dict: 

935 """ 

936 Return a copy of the /health response with provider routing fields 

937 (``ADMIN_ONLY_HEALTH_DISPLAY_PARAMS``) removed from each healthy/unhealthy 

938 endpoint entry. Used to hide those fields from non-admin callers while 

939 still showing them which deployments they own and whether each one is 

940 healthy. Proxy admins receive the unmodified result. 

941 """ 

942 out: Final = dict(result) 

943 drop: Final = set(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS) 

944 for key in ("healthy_endpoints", "unhealthy_endpoints"): 

945 eps = out.get(key) 

946 if isinstance(eps, list): 

947 out[key] = [({k: v for k, v in ep.items() if k not in drop} if isinstance(ep, dict) else ep) for ep in eps] 

948 return out 

949 

950 

951def _health_accessible_model_names( 

952 user_api_key_dict: UserAPIKeyAuth, llm_router: Router | None 

953) -> frozenset[str] | None: 

954 """Model names the caller may health-check, or None when the key is unrestricted.""" 

955 granted_models: Final = _resolve_key_models_for_auth_check(user_api_key_dict) 

956 if not granted_models or SpecialModelNames.all_proxy_models.value in granted_models: 956 ↛ 958line 956 didn't jump to line 958 because the condition on line 956 was always true

957 return None 

958 if llm_router is None: 

959 return frozenset(granted_models) 

960 return frozenset( 

961 get_key_models( 

962 user_api_key_dict=user_api_key_dict, 

963 proxy_model_list=llm_router.get_model_names(team_id=user_api_key_dict.team_id), 

964 model_access_groups=llm_router.get_model_access_groups(), 

965 ) 

966 ) 

967 

968 

969def _caller_may_probe_deployment( 

970 deployment: Mapping[str, object], 

971 allowed_models: frozenset[str] | None, 

972 llm_router: Router | None, 

973 team_id: str | None, 

974 caller_is_admin: bool, 

975) -> bool: 

976 """Same deployment visibility rule as routing: another team's deployment is never in scope, team-less callers included.""" 

977 if not caller_is_admin and not Router._deployment_usable_by_team(deployment, team_id): 

978 return False 

979 if allowed_models is None: 

980 return True 

981 if llm_router is None: 

982 return deployment.get("model_name") in allowed_models 

983 model: Final = dict(deployment) 

984 return any( 

985 llm_router.should_include_deployment(model_name=name, model=model, team_id=team_id) for name in allowed_models 

986 ) 

987 

988 

989def _resolve_targeted_model_ids( 

990 model_list: list, model: str | None, model_id: str | None, team_id: str | None 

991) -> set | None: 

992 """ 

993 Resolve a ``/health`` ``model`` / ``model_id`` query param to the set of 

994 deployment IDs the response should be scoped to, mirroring the live-path 

995 narrowing in ``perform_health_check()``: ``model_id`` wins when given and 

996 matches ``model_info.id`` only; ``model`` targets the deployments a request 

997 for that name from the caller would route to, else those whose 

998 ``litellm_params.model`` provider string is that value (``deployments_targeted_by_name``). 

999 

1000 Callers pass an already-scoped list, so a ``model_id`` outside the 

1001 caller's scope resolves to an empty set and never to the unvalidated id. 

1002 Returns ``None`` when no targeting is requested. 

1003 """ 

1004 if model_id: 1004 ↛ 1005line 1004 didn't jump to line 1005 because the condition on line 1004 was never true

1005 return {i for m in model_list if (i := (m.get("model_info") or {}).get("id")) == model_id} 

1006 if not model: 

1007 return None 

1008 return { 

1009 i 

1010 for m in deployments_targeted_by_name(model_list, model, team_id) 

1011 if (i := (m.get("model_info") or {}).get("id")) 

1012 } 

1013 

1014 

1015def _filter_health_check_results_by_model_ids(results: dict, allowed_model_ids: set) -> dict: 

1016 """ 

1017 Restrict a cached background health-check result dict to endpoints whose 

1018 model_id is in ``allowed_model_ids``. 

1019 

1020 Endpoints without a model_id (e.g. CLI-model entries that predate the 

1021 model_id wiring) are dropped conservatively — we cannot prove they belong 

1022 to the caller, so they are excluded rather than leaked. 

1023 

1024 Each retained endpoint is shallow-copied before being returned, so any 

1025 downstream transform (e.g. _strip_admin_only_fields_from_health_result) 

1026 cannot accidentally mutate the shared ``health_check_results`` cache. 

1027 """ 

1028 healthy = [dict(ep) for ep in (results.get("healthy_endpoints") or []) if ep.get("model_id") in allowed_model_ids] 

1029 unhealthy: Final = [ 

1030 dict(ep) for ep in (results.get("unhealthy_endpoints") or []) if ep.get("model_id") in allowed_model_ids 

1031 ] 

1032 return { 

1033 "healthy_endpoints": healthy, 

1034 "unhealthy_endpoints": unhealthy, 

1035 "healthy_count": len(healthy), 

1036 "unhealthy_count": len(unhealthy), 

1037 } 

1038 

1039 

1040async def _perform_health_check_and_save( 

1041 model_list, 

1042 target_model, 

1043 cli_model, 

1044 details, 

1045 prisma_client, 

1046 start_time, 

1047 user_id, 

1048 model_id=None, 

1049 max_concurrency=None, 

1050 **perform_health_check_extra, 

1051): 

1052 """Helper function to perform health check and save results to database""" 

1053 healthy_endpoints, unhealthy_endpoints, _ = await perform_health_check( 

1054 model_list=model_list, 

1055 cli_model=cli_model, 

1056 model=target_model, 

1057 details=details, 

1058 max_concurrency=max_concurrency, 

1059 model_id=model_id, 

1060 **perform_health_check_extra, 

1061 ) 

1062 

1063 # Optionally save health check result to database (non-blocking) 

1064 if prisma_client is not None: 1064 ↛ 1080line 1064 didn't jump to line 1080 because the condition on line 1064 was always true

1065 # For CLI model, use cli_model name; for router models, use target_model 

1066 model_name_for_db: Final = cli_model if cli_model is not None else target_model 

1067 if model_name_for_db is not None: 

1068 asyncio.create_task( 

1069 _save_health_check_to_db( 

1070 prisma_client, 

1071 model_name_for_db, 

1072 healthy_endpoints, 

1073 unhealthy_endpoints, 

1074 start_time, 

1075 user_id, 

1076 model_id=model_id, 

1077 ) 

1078 ) 

1079 

1080 return { 

1081 "healthy_endpoints": healthy_endpoints, 

1082 "unhealthy_endpoints": unhealthy_endpoints, 

1083 "healthy_count": len(healthy_endpoints), 

1084 "unhealthy_count": len(unhealthy_endpoints), 

1085 } 

1086 

1087 

1088def _health_endpoint_resolve_target_model_name( 

1089 model: str | None, 

1090 model_id: str | None, 

1091 llm_router, 

1092) -> str | None: 

1093 """Map ``model_id`` to its deployment's ``model_name`` for live health checks. 

1094 

1095 ``model_id`` wins over ``model``, so an id no deployment carries is a 404 even 

1096 when it is paired with a known name. 

1097 """ 

1098 if not model_id: 

1099 return model 

1100 if llm_router is None: 1100 ↛ 1101line 1100 didn't jump to line 1101 because the condition on line 1100 was never true

1101 raise HTTPException( 

1102 status_code=status.HTTP_404_NOT_FOUND, 

1103 detail={"error": f"Model with ID {model_id} not found"}, 

1104 ) 

1105 try: 

1106 deployment: Final = llm_router.get_deployment(model_id=model_id) 

1107 except Exception as e: 

1108 verbose_proxy_logger.error("Error getting deployment for model_id %s: %s", model_id, e) 

1109 raise HTTPException( 

1110 status_code=status.HTTP_404_NOT_FOUND, 

1111 detail={"error": f"Model with ID {model_id} not found"}, 

1112 ) from e 

1113 if deployment is None: 1113 ↛ 1118line 1113 didn't jump to line 1118 because the condition on line 1113 was always true

1114 raise HTTPException( 

1115 status_code=status.HTTP_404_NOT_FOUND, 

1116 detail={"error": f"Model with ID {model_id} not found"}, 

1117 ) 

1118 return deployment.model_name 

1119 

1120 

1121@router.get("/health", tags=["health"], dependencies=[Depends(user_api_key_auth)]) 

1122async def health_endpoint( 

1123 response: Response, 

1124 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

1125 model: str | None = fastapi.Query(None, description="Specify the model name (optional)"), 

1126 model_id: str | None = fastapi.Query(None, description="Specify the model ID (optional)"), 

1127): 

1128 """ 

1129 🚨 USE `/health/liveliness` to health check the proxy 🚨 

1130 

1131 See more 👉 https://docs.litellm.ai/docs/proxy/health 

1132 

1133 

1134 Check the health of all the endpoints in config.yaml 

1135 

1136 To run health checks in the background, add this to config.yaml: 

1137 ``` 

1138 general_settings: 

1139 # ... other settings 

1140 background_health_checks: True 

1141 ``` 

1142 else, the health checks will be run on models when /health is called. 

1143 

1144 To skip deployments that set ``model_info.disable_background_health_check: true`` 

1145 on ``GET /health`` as well as in the background loop, set 

1146 ``general_settings.health_check_skip_disabled_background_models: true``. 

1147 """ 

1148 import time 

1149 

1150 from litellm.proxy.proxy_server import ( 

1151 general_settings, 

1152 health_check_concurrency, 

1153 health_check_details, 

1154 health_check_results, 

1155 llm_model_list, 

1156 llm_router, 

1157 prisma_client, 

1158 use_background_health_checks, 

1159 user_model, 

1160 ) 

1161 

1162 _hc_filter: Final = health_check_filter_kwargs_from_general_settings(general_settings) 

1163 start_time: Final = time.time() 

1164 

1165 target_model: Final = _health_endpoint_resolve_target_model_name(model, model_id, llm_router) 

1166 

1167 is_admin: Final = _is_proxy_admin(user_api_key_dict) 

1168 model_specific_request: Final = bool(model or model_id) 

1169 

1170 def _post_process(result: dict) -> dict: 

1171 # api_base / api_version reveal which provider/region/internal host the 

1172 # deployment talks to; only proxy admins receive them. Non-admin keys 

1173 # still see model/model_id and the healthy/unhealthy status. We also 

1174 # set a header so non-admin clients that previously parsed those 

1175 # fields can detect the change programmatically. 

1176 # When a caller asked about a specific model/model_id and zero 

1177 # endpoints came back healthy, surface that as a 503 so monitoring 

1178 # systems can rely on the HTTP status instead of having to parse the 

1179 # body. The body shape is unchanged. 

1180 if model_specific_request and result.get("healthy_count", 0) == 0: 

1181 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE 

1182 if is_admin: 1182 ↛ 1184line 1182 didn't jump to line 1184 because the condition on line 1182 was always true

1183 return result 

1184 response.headers["Litellm-Health-Field-Notice"] = ( 

1185 f"{', '.join(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS)} are admin-only on this endpoint" 

1186 ) 

1187 return _strip_admin_only_fields_from_health_result(result) 

1188 

1189 try: 

1190 if llm_model_list is None: 1190 ↛ 1192line 1190 didn't jump to line 1192 because the condition on line 1190 was never true

1191 # if no router set, check if user set a model using litellm --model ollama/llama2 

1192 if user_model is not None: 

1193 cli_result: Final = await _perform_health_check_and_save( 

1194 model_list=[], 

1195 target_model=None, 

1196 cli_model=user_model, 

1197 details=health_check_details, 

1198 prisma_client=prisma_client, 

1199 start_time=start_time, 

1200 user_id=user_api_key_dict.user_id, 

1201 model_id=None, # CLI model doesn't have model_id 

1202 max_concurrency=health_check_concurrency, 

1203 **_hc_filter, 

1204 ) 

1205 return _post_process(cli_result) 

1206 raise HTTPException( 

1207 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

1208 detail={"error": "Model list not initialized"}, 

1209 ) 

1210 allowed_models: Final = _health_accessible_model_names(user_api_key_dict, llm_router) 

1211 restrict_to_allowed_models: Final = not is_admin or allowed_models is not None 

1212 _llm_model_list: Final = [ 

1213 m 

1214 for m in copy.deepcopy(llm_model_list) 

1215 if not restrict_to_allowed_models 

1216 or _caller_may_probe_deployment(m, allowed_models, llm_router, user_api_key_dict.team_id, is_admin) 

1217 ] 

1218 targeted_ids: Final = _resolve_targeted_model_ids(_llm_model_list, model, model_id, user_api_key_dict.team_id) 

1219 if restrict_to_allowed_models and targeted_ids is not None and not targeted_ids: 1219 ↛ 1220line 1219 didn't jump to line 1220 because the condition on line 1219 was never true

1220 raise HTTPException( 

1221 status_code=status.HTTP_403_FORBIDDEN, 

1222 detail={ 

1223 "error": f"key not allowed to health-check model_id {model_id}" 

1224 if model_id 

1225 else f"key not allowed to health-check model {model}" 

1226 }, 

1227 ) 

1228 if use_background_health_checks: 1228 ↛ 1235line 1228 didn't jump to line 1235 because the condition on line 1228 was never true

1229 # The cached background result covers every model. When the 

1230 # caller targets a specific model/model_id we have to narrow the 

1231 # cache to that deployment before _post_process evaluates 

1232 # healthy_count, otherwise an unhealthy "foo" combined with any 

1233 # other healthy model would still report healthy_count > 0 and 

1234 # the targeted-503 path would never fire. 

1235 if restrict_to_allowed_models: 

1236 allowed_model_ids: Final = { 

1237 (m.get("model_info") or {}).get("id") 

1238 for m in _llm_model_list 

1239 if (m.get("model_info") or {}).get("id") 

1240 } 

1241 # _llm_model_list is already scoped to the caller's allowed 

1242 # model_names above, so targeted_ids is implicitly the 

1243 # intersection of "targeted" and "allowed." 

1244 filter_ids: Final = targeted_ids if targeted_ids is not None else allowed_model_ids 

1245 filtered: Final = _filter_health_check_results_by_model_ids(health_check_results, filter_ids) 

1246 if targeted_ids is None and _llm_model_list and not allowed_model_ids: 

1247 # Caller has accessible model_names but none of the 

1248 # matching deployments expose a model_info.id, so the 

1249 # cache filter (which keys on model_id) drops every 

1250 # entry. Surface this both as a warning log and a 

1251 # structured "warnings" field on the response so the 

1252 # caller can distinguish "no deployments found" from 

1253 # "deployments excluded due to missing model_info.id". 

1254 verbose_proxy_logger.warning( 

1255 "health_endpoint: scoped key %s has accessible models %s " 

1256 "but none of the matching deployments carry a model_info.id; " 

1257 "background health-check cache will return an empty result.", 

1258 user_api_key_dict.user_id, 

1259 list(user_api_key_dict.models), 

1260 ) 

1261 filtered["warnings"] = [ 

1262 "Some accessible deployments are missing model_info.id " 

1263 "and were excluded from this response. Ask a proxy admin " 

1264 "to populate model_info.id for these models." 

1265 ] 

1266 return _post_process(filtered) 

1267 if targeted_ids is not None: 

1268 # Admin caller targeting a specific model: filter the cache 

1269 # so the response (and the targeted-503 check) reflects only 

1270 # that deployment, not the global aggregate. 

1271 return _post_process(_filter_health_check_results_by_model_ids(health_check_results, targeted_ids)) 

1272 return _post_process(health_check_results) 

1273 else: 

1274 router_result: Final = await _perform_health_check_and_save( 

1275 model_list=_llm_model_list, 

1276 target_model=target_model, 

1277 cli_model=None, 

1278 details=health_check_details, 

1279 prisma_client=prisma_client, 

1280 start_time=start_time, 

1281 user_id=user_api_key_dict.user_id, 

1282 model_id=model_id, 

1283 max_concurrency=health_check_concurrency, 

1284 router=llm_router, 

1285 team_id=user_api_key_dict.team_id, 

1286 **_hc_filter, 

1287 ) 

1288 return _post_process(router_result) 

1289 except Exception as e: 

1290 verbose_proxy_logger.error("litellm.proxy.proxy_server.py::health_endpoint(): Exception occured - %s", e) 

1291 verbose_proxy_logger.debug(traceback.format_exc()) 

1292 raise e 

1293 

1294 

1295@router.get("/health/history", tags=["health"], dependencies=[Depends(user_api_key_auth)]) 

1296async def health_check_history_endpoint( 

1297 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

1298 model: str | None = fastapi.Query(None, description="Filter by specific model name"), 

1299 status_filter: str | None = fastapi.Query(None, description="Filter by status (healthy/unhealthy)"), 

1300 limit: int = fastapi.Query(100, description="Number of records to return", ge=1, le=1000), 

1301 offset: int = fastapi.Query(0, description="Number of records to skip", ge=0), 

1302): 

1303 """ 

1304 Get health check history for models 

1305 

1306 Returns historical health check data with optional filtering. 

1307 """ 

1308 prisma_client: Final = _check_prisma_client() 

1309 

1310 try: 

1311 history: Final = await prisma_client.get_health_check_history( 

1312 model_name=model, 

1313 limit=limit, 

1314 offset=offset, 

1315 status_filter=status_filter, 

1316 ) 

1317 

1318 # Convert to dict format for JSON response using helper function 

1319 history_data: Final = [_convert_health_check_to_dict(check) for check in history] 

1320 

1321 return { 

1322 "health_checks": history_data, 

1323 "total_records": len(history_data), 

1324 "limit": limit, 

1325 "offset": offset, 

1326 } 

1327 except Exception as e: 

1328 verbose_proxy_logger.error("Error getting health check history: %s", e) 

1329 raise HTTPException( 

1330 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

1331 detail={"error": f"Failed to retrieve health check history: {e}"}, 

1332 ) 

1333 

1334 

1335@router.get("/health/latest", tags=["health"], dependencies=[Depends(user_api_key_auth)]) 

1336async def latest_health_checks_endpoint( 

1337 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

1338): 

1339 """ 

1340 Get the latest health check status for all models 

1341 

1342 Returns the most recent health check result for each model. 

1343 """ 

1344 prisma_client: Final = _check_prisma_client() 

1345 

1346 try: 

1347 latest_checks: Final = await prisma_client.get_all_latest_health_checks() 

1348 

1349 # Convert to dict format for JSON response using helper function 

1350 checks_data: Final = { 

1351 (check.model_id if check.model_id else check.model_name): _convert_health_check_to_dict(check) 

1352 for check in latest_checks 

1353 } 

1354 

1355 return { 

1356 "latest_health_checks": checks_data, 

1357 "total_models": len(checks_data), 

1358 } 

1359 except Exception as e: 

1360 verbose_proxy_logger.error("Error getting latest health checks: %s", e) 

1361 raise HTTPException( 

1362 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

1363 detail={"error": f"Failed to retrieve latest health checks: {e}"}, 

1364 ) 

1365 

1366 

1367@router.get("/health/shared-status", tags=["health"], dependencies=[Depends(user_api_key_auth)]) 

1368async def shared_health_check_status_endpoint( 

1369 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

1370): 

1371 """ 

1372 Get the status of shared health check coordination across pods. 

1373 

1374 Returns information about Redis connectivity, lock status, and cache status. 

1375 """ 

1376 from litellm.proxy.proxy_server import redis_usage_cache, use_shared_health_check 

1377 

1378 if not use_shared_health_check: 1378 ↛ 1384line 1378 didn't jump to line 1384 because the condition on line 1378 was always true

1379 return { 

1380 "shared_health_check_enabled": False, 

1381 "message": "Shared health check is not enabled", 

1382 } 

1383 

1384 if redis_usage_cache is None: 

1385 return { 

1386 "shared_health_check_enabled": True, 

1387 "redis_available": False, 

1388 "message": "Redis is not configured", 

1389 } 

1390 

1391 try: 

1392 from litellm.proxy.health_check_utils.shared_health_check_manager import ( 

1393 SharedHealthCheckManager, 

1394 ) 

1395 

1396 shared_health_manager: Final = SharedHealthCheckManager( 

1397 redis_cache=redis_usage_cache, 

1398 ) 

1399 

1400 health_status: Final = await shared_health_manager.get_health_check_status() 

1401 return {"shared_health_check_enabled": True, "status": health_status} 

1402 except Exception as e: 

1403 verbose_proxy_logger.error("Error getting shared health check status: %s", e) 

1404 raise HTTPException( 

1405 status_code=fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR, 

1406 detail={"error": f"Failed to retrieve shared health check status: {e}"}, 

1407 ) 

1408 

1409 

1410def _read_license_data() -> dict[str, Any] | None: 

1411 from litellm.proxy.proxy_server import _license_check, premium_user_data 

1412 

1413 license_data: EnterpriseLicenseData | None = premium_user_data or _license_check.airgapped_license_data 

1414 

1415 if ( 1415 ↛ 1420line 1415 didn't jump to line 1420 because the condition on line 1415 was never true

1416 license_data is None 

1417 and getattr(_license_check, "license_str", None) 

1418 and getattr(_license_check, "public_key", None) 

1419 ): 

1420 try: 

1421 verification_result: Final = _license_check.verify_license_without_api_request( 

1422 public_key=_license_check.public_key, 

1423 license_key=_license_check.license_str, 

1424 ) 

1425 if verification_result is True: 

1426 license_data = _license_check.airgapped_license_data 

1427 except Exception: 

1428 pass 

1429 

1430 if license_data is None: 1430 ↛ 1432line 1430 didn't jump to line 1432 because the condition on line 1430 was always true

1431 return None 

1432 return cast(dict[str, Any], license_data) 

1433 

1434 

1435def _read_allowed_features(license_data: dict[str, Any]) -> list: 

1436 raw_allowed_features: Final = license_data.get("allowed_features") 

1437 if isinstance(raw_allowed_features, list): 

1438 return list(raw_allowed_features) 

1439 if raw_allowed_features is None: 

1440 return [] 

1441 return [raw_allowed_features] 

1442 

1443 

1444@router.get( 

1445 "/health/license", 

1446 tags=["health"], 

1447 dependencies=[Depends(user_api_key_auth)], 

1448) 

1449async def health_license_endpoint( 

1450 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

1451): 

1452 """Return metadata about the configured LiteLLM license without exposing the key.""" 

1453 from litellm.proxy.proxy_server import _license_check, premium_user 

1454 

1455 license_data: Final = _read_license_data() 

1456 has_license: Final = bool(getattr(_license_check, "license_str", None)) 

1457 license_type: Final = "enterprise" if premium_user else "community" 

1458 

1459 if license_data is None: 1459 ↛ 1471line 1459 didn't jump to line 1471 because the condition on line 1459 was always true

1460 return { 

1461 "has_license": has_license, 

1462 "license_type": license_type, 

1463 "expiration_date": None, 

1464 "allowed_features": [], 

1465 "limits": { 

1466 "max_users": None, 

1467 "max_teams": None, 

1468 }, 

1469 } 

1470 

1471 expiration_date: Final = license_data.get("expiration_date") 

1472 max_users: Final = license_data.get("max_users") 

1473 max_teams: Final = license_data.get("max_teams") 

1474 

1475 return { 

1476 "has_license": has_license, 

1477 "license_type": license_type, 

1478 "expiration_date": expiration_date, 

1479 "allowed_features": _read_allowed_features(license_data), 

1480 "limits": { 

1481 "max_users": max_users, 

1482 "max_teams": max_teams, 

1483 }, 

1484 } 

1485 

1486 

1487class DBHealthCache(TypedDict): 

1488 status: str 

1489 last_updated: datetime 

1490 

1491 

1492db_health_cache: DBHealthCache = {"status": "unknown", "last_updated": datetime.now()} 

1493 

1494# Bounds each DB round-trip on the probe path so a hung connection during a 

1495# failover cannot make the probe fail by timeout (k8s default timeoutSeconds: 5). 

1496DB_READINESS_CHECK_TIMEOUT_SECONDS: Final = 2.0 

1497# One deadline for the whole probe-path DB check (initial check + reconnect + 

1498# re-check, including reconnect lock waits), kept under timeoutSeconds: 5. 

1499DB_READINESS_PROBE_DEADLINE_SECONDS: Final = 4.0 

1500 

1501 

1502async def _db_health_readiness_check() -> DBHealthCache: 

1503 try: 

1504 return await asyncio.wait_for( 

1505 _db_health_readiness_check_unbounded(), 

1506 timeout=DB_READINESS_PROBE_DEADLINE_SECONDS, 

1507 ) 

1508 except asyncio.TimeoutError: 

1509 return {"status": "disconnected", "last_updated": db_health_cache["last_updated"]} 

1510 

1511 

1512async def _db_health_readiness_check_unbounded() -> DBHealthCache: 

1513 from litellm.proxy.proxy_server import prisma_client 

1514 

1515 global db_health_cache 

1516 

1517 try: 

1518 time_diff: Final = datetime.now() - db_health_cache["last_updated"] 

1519 if db_health_cache["status"] == "connected" and time_diff < timedelta(seconds=15): 

1520 return db_health_cache 

1521 

1522 if prisma_client is None: 1522 ↛ 1523line 1522 didn't jump to line 1523 because the condition on line 1522 was never true

1523 db_health_cache = {"status": "disconnected", "last_updated": datetime.now()} 

1524 return db_health_cache 

1525 

1526 await asyncio.wait_for(prisma_client.health_check(), timeout=DB_READINESS_CHECK_TIMEOUT_SECONDS) 

1527 db_health_cache = {"status": "connected", "last_updated": datetime.now()} 

1528 return db_health_cache 

1529 except Exception as e: 

1530 db_health_cache = {"status": "disconnected", "last_updated": datetime.now()} 

1531 if PrismaDBExceptionHandler.is_database_transport_error(e): 

1532 try: 

1533 verbose_proxy_logger.warning("_db_health_readiness_check: health_check failed, attempting reconnect") 

1534 await prisma_client.attempt_db_reconnect( 

1535 reason="health_readiness_check", 

1536 timeout_seconds=DB_READINESS_CHECK_TIMEOUT_SECONDS, 

1537 lock_timeout_seconds=DB_READINESS_CHECK_TIMEOUT_SECONDS, 

1538 ) 

1539 await asyncio.wait_for( 

1540 prisma_client.health_check(), 

1541 timeout=DB_READINESS_CHECK_TIMEOUT_SECONDS, 

1542 ) 

1543 verbose_proxy_logger.info("_db_health_readiness_check: reconnect succeeded") 

1544 db_health_cache = { 

1545 "status": "connected", 

1546 "last_updated": datetime.now(), 

1547 } 

1548 return db_health_cache 

1549 except Exception: 

1550 verbose_proxy_logger.error("_db_health_readiness_check: reconnect failed") 

1551 return db_health_cache 

1552 

1553 

1554@router.get( 

1555 "/settings", 

1556 tags=["health"], 

1557 dependencies=[Depends(user_api_key_auth)], 

1558) 

1559@router.get( 

1560 "/active/callbacks", 

1561 tags=["health"], 

1562 dependencies=[Depends(user_api_key_auth)], 

1563) 

1564async def active_callbacks(): 

1565 """ 

1566 Returns a list of litellm level settings 

1567 

1568 This is useful for debugging and ensuring the proxy server is configured correctly. 

1569 

1570 Response schema: 

1571 ``` 

1572 { 

1573 "alerting": _alerting, 

1574 "litellm.callbacks": litellm_callbacks, 

1575 "litellm.input_callback": litellm_input_callbacks, 

1576 "litellm.failure_callback": litellm_failure_callbacks, 

1577 "litellm.success_callback": litellm_success_callbacks, 

1578 "litellm._async_success_callback": litellm_async_success_callbacks, 

1579 "litellm._async_failure_callback": litellm_async_failure_callbacks, 

1580 "litellm._async_input_callback": litellm_async_input_callbacks, 

1581 "all_litellm_callbacks": all_litellm_callbacks, 

1582 "num_callbacks": len(all_litellm_callbacks), 

1583 "num_alerting": _num_alerting, 

1584 "litellm.request_timeout": litellm.request_timeout, 

1585 } 

1586 ``` 

1587 """ 

1588 

1589 from litellm.proxy.proxy_server import general_settings, proxy_logging_obj 

1590 

1591 _alerting: Final = str(general_settings.get("alerting")) 

1592 # get success callbacks 

1593 

1594 litellm_callbacks: Final = [str(x) for x in litellm.callbacks] 

1595 litellm_input_callbacks: Final = [str(x) for x in litellm.input_callback] 

1596 litellm_failure_callbacks: Final = [str(x) for x in litellm.failure_callback] 

1597 litellm_success_callbacks: Final = [str(x) for x in litellm.success_callback] 

1598 litellm_async_success_callbacks: Final = [str(x) for x in litellm._async_success_callback] 

1599 litellm_async_failure_callbacks: Final = [str(x) for x in litellm._async_failure_callback] 

1600 litellm_async_input_callbacks: Final = [str(x) for x in litellm._async_input_callback] 

1601 

1602 all_litellm_callbacks: Final = ( 

1603 litellm_callbacks 

1604 + litellm_input_callbacks 

1605 + litellm_failure_callbacks 

1606 + litellm_success_callbacks 

1607 + litellm_async_success_callbacks 

1608 + litellm_async_failure_callbacks 

1609 + litellm_async_input_callbacks 

1610 ) 

1611 

1612 alerting: Final = proxy_logging_obj.alerting 

1613 _num_alerting = 0 

1614 if alerting and isinstance(alerting, list): 1614 ↛ 1615line 1614 didn't jump to line 1615 because the condition on line 1614 was never true

1615 _num_alerting = len(alerting) 

1616 

1617 return { 

1618 "alerting": _alerting, 

1619 "litellm.callbacks": litellm_callbacks, 

1620 "litellm.input_callback": litellm_input_callbacks, 

1621 "litellm.failure_callback": litellm_failure_callbacks, 

1622 "litellm.success_callback": litellm_success_callbacks, 

1623 "litellm._async_success_callback": litellm_async_success_callbacks, 

1624 "litellm._async_failure_callback": litellm_async_failure_callbacks, 

1625 "litellm._async_input_callback": litellm_async_input_callbacks, 

1626 "all_litellm_callbacks": all_litellm_callbacks, 

1627 "num_callbacks": len(all_litellm_callbacks), 

1628 "num_alerting": _num_alerting, 

1629 "litellm.request_timeout": litellm.request_timeout, 

1630 } 

1631 

1632 

1633def callback_name(callback): 

1634 if isinstance(callback, str): 

1635 return callback 

1636 

1637 try: 

1638 return callback.__name__ 

1639 except AttributeError: 

1640 try: 

1641 return callback.__class__.__name__ 

1642 except AttributeError: 

1643 return str(callback) 

1644 

1645 

1646DISABLE_NO_REDIS_WARNING_ENV_VAR: Final = "LITELLM_DISABLE_NO_REDIS_WARNING" 

1647 

1648 

1649async def _show_no_redis_warning() -> bool: 

1650 """ 

1651 Whether the UI should warn that no Redis is configured. 

1652 

1653 Redis is what makes rate limits, budgets, router state, and cache 

1654 invalidation consistent across workers, so a proxy running without it is 

1655 only safe as a single worker. Both places a Redis can land count: the 

1656 coordination cache (from a Redis response cache, general_settings. 

1657 coordination_redis, or the REDIS_* env fallback) and the router's own 

1658 Redis (router_settings.redis_host), which backs cooldowns and usage-based 

1659 routing on its own. A deployment whose worker-heartbeat census proves it 

1660 is exactly one worker needs no cross-worker coordination, so it never 

1661 warns; when the census is unavailable or shows more than one worker, the 

1662 warning stands unless LITELLM_DISABLE_NO_REDIS_WARNING=true silences it. 

1663 """ 

1664 from litellm.proxy.proxy_server import llm_router, prisma_client, redis_usage_cache 

1665 

1666 if redis_usage_cache is not None: 1666 ↛ 1667line 1666 didn't jump to line 1667 because the condition on line 1666 was never true

1667 return False 

1668 if llm_router is not None and llm_router.cache.redis_cache is not None: 1668 ↛ 1669line 1668 didn't jump to line 1669 because the condition on line 1668 was never true

1669 return False 

1670 if get_secret_bool(DISABLE_NO_REDIS_WARNING_ENV_VAR, False) is True: 1670 ↛ 1671line 1670 didn't jump to line 1671 because the condition on line 1670 was never true

1671 return False 

1672 if prisma_client is None: 1672 ↛ 1673line 1672 didn't jump to line 1673 because the condition on line 1672 was never true

1673 return True 

1674 return await count_live_proxy_workers(prisma_client) != 1 

1675 

1676 

1677def _show_env_credential_login_warning() -> bool: 

1678 from litellm.proxy.auth.login_utils import is_env_credential_login_enabled 

1679 from litellm.proxy.proxy_server import general_settings 

1680 

1681 return is_env_credential_login_enabled(general_settings) 

1682 

1683 

1684async def _get_health_readiness_details( 

1685 response: Response | None = None, 

1686) -> dict[str, Any]: 

1687 """ 

1688 Detailed health payload for authenticated diagnostics. 

1689 """ 

1690 from litellm.proxy.proxy_server import prisma_client, version 

1691 

1692 try: 

1693 # get success callback 

1694 success_callback_names = [] 

1695 

1696 try: 

1697 # this was returning a JSON of the values in some of the callbacks 

1698 # all we need is the callback name, hence we do str(callback) 

1699 success_callback_names = [callback_name(x) for x in litellm.success_callback] 

1700 except AttributeError: 

1701 # don't let this block the /health/readiness response, if we can't convert to str -> return litellm.success_callback 

1702 success_callback_names = litellm.success_callback 

1703 

1704 # check Cache 

1705 cache_type: Any = None 

1706 if litellm.cache is not None: 1706 ↛ 1722line 1706 didn't jump to line 1722 because the condition on line 1706 was always true

1707 from litellm.caching.caching import RedisSemanticCache 

1708 

1709 cache_type = litellm.cache.type 

1710 

1711 if isinstance(litellm.cache.cache, RedisSemanticCache): 1711 ↛ 1715line 1711 didn't jump to line 1715 because the condition on line 1711 was never true

1712 # ping the cache 

1713 # TODO: @ishaan-jaff - we should probably not ping the cache on every /health/readiness check 

1714 index_info: Any 

1715 try: 

1716 index_info = await litellm.cache.cache._index_info() 

1717 except Exception as e: 

1718 index_info = "index does not exist - error: " + str(e) 

1719 cache_type = {"type": cache_type, "index_info": index_info} 

1720 

1721 # check log level 

1722 log_level_name: Final = logging.getLevelName(verbose_logger.getEffectiveLevel()) 

1723 is_detailed_debug: Final = verbose_logger.isEnabledFor(logging.DEBUG) 

1724 show_no_redis_warning: Final = await _show_no_redis_warning() 

1725 show_env_credential_login_warning: Final = _show_env_credential_login_warning() 

1726 

1727 # check DB 

1728 if prisma_client is not None: # if db passed in, check if it's connected 1728 ↛ 1756line 1728 didn't jump to line 1756 because the condition on line 1728 was always true

1729 db_status: Final = _readiness_db_status(await _db_health_readiness_check()) 

1730 # A configured DB that is not reachable means the worker cannot 

1731 # serve requests that depend on persisted state (keys, budgets, 

1732 # spend logs). Return 503 so orchestrators take this pod out of 

1733 # rotation; "Not connected" (no DB configured at all) stays 200. 

1734 # With allow_requests_on_db_unavailable the proxy keeps serving 

1735 # during a DB outage, so the pod must stay in rotation (200) and 

1736 # report the DB state through the body instead. 

1737 if ( 1737 ↛ 1742line 1737 didn't jump to line 1742 because the condition on line 1737 was never true

1738 response is not None 

1739 and db_status != "connected" 

1740 and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable() 

1741 ): 

1742 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE 

1743 return { 

1744 "status": "healthy", 

1745 "db": db_status, 

1746 "cache": cache_type, 

1747 "litellm_version": version, 

1748 "success_callbacks": success_callback_names, 

1749 "use_aiohttp_transport": AsyncHTTPHandler._should_use_aiohttp_transport(), 

1750 "log_level": log_level_name, 

1751 "is_detailed_debug": is_detailed_debug, 

1752 "show_no_redis_warning": show_no_redis_warning, 

1753 "show_env_credential_login_warning": show_env_credential_login_warning, 

1754 } 

1755 else: 

1756 return { 

1757 "status": "healthy", 

1758 "db": "Not connected", 

1759 "cache": cache_type, 

1760 "litellm_version": version, 

1761 "success_callbacks": success_callback_names, 

1762 "use_aiohttp_transport": AsyncHTTPHandler._should_use_aiohttp_transport(), 

1763 "log_level": log_level_name, 

1764 "is_detailed_debug": is_detailed_debug, 

1765 "show_no_redis_warning": show_no_redis_warning, 

1766 "show_env_credential_login_warning": show_env_credential_login_warning, 

1767 } 

1768 except Exception as e: 

1769 raise HTTPException(status_code=503, detail=f"Service Unhealthy ({e})") 

1770 

1771 

1772def _allow_public_health_readiness_details() -> bool: 

1773 from litellm.proxy.proxy_server import general_settings 

1774 

1775 return general_settings.get("allow_public_health_readiness_details") is True 

1776 

1777 

1778def _drain_endpoint_enabled() -> bool: 

1779 from litellm.proxy.proxy_server import general_settings 

1780 

1781 return general_settings.get("enable_drain_endpoint") is True 

1782 

1783 

1784def _drain_endpoint_token() -> str | None: 

1785 """ 

1786 Shared secret required on the X-Drain-Token header to call /health/drain. 

1787 

1788 Falls back to the ``DRAIN_ENDPOINT_TOKEN`` env var when unset in 

1789 general_settings so the kubelet preStop hook can supply it via 

1790 ``valueFrom.secretKeyRef`` without a config reload. 

1791 """ 

1792 from litellm.proxy.proxy_server import general_settings 

1793 

1794 token: Final = general_settings.get("drain_endpoint_token") 

1795 if isinstance(token, str) and token: 

1796 return token 

1797 env_token: Final = os.getenv("DRAIN_ENDPOINT_TOKEN") 

1798 if env_token: 

1799 return env_token 

1800 return None 

1801 

1802 

1803def _authorize_drain_request(request: Request) -> None: 

1804 """ 

1805 Reject /health/drain calls that don't carry the configured X-Drain-Token. 

1806 

1807 When no token is configured the endpoint is treated as already opted-in 

1808 (the ``enable_drain_endpoint`` flag is the only gate). Comparison uses 

1809 ``secrets.compare_digest`` to avoid timing leaks. 

1810 """ 

1811 expected: Final = _drain_endpoint_token() 

1812 if expected is None: 

1813 return 

1814 supplied: Final = request.headers.get("x-drain-token") or "" 

1815 if not secrets.compare_digest(supplied, expected): 

1816 raise HTTPException( 

1817 status_code=status.HTTP_401_UNAUTHORIZED, 

1818 detail="Invalid or missing X-Drain-Token", 

1819 ) 

1820 

1821 

1822def _readiness_db_status(db_health_status: DBHealthCache) -> str: 

1823 """A pod whose pre-request lookups hit their deadline inside the stall window 

1824 reports "stalled" even though the ping succeeds: the ping is a fresh 

1825 connection, the stalled lookups are the ones requests actually wait on.""" 

1826 if db_health_status["status"] != "connected": 1826 ↛ 1827line 1826 didn't jump to line 1827 because the condition on line 1826 was never true

1827 return db_health_status["status"] 

1828 if db_lookup_stall_tracker.stalled_within(PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS): 1828 ↛ 1829line 1828 didn't jump to line 1829 because the condition on line 1828 was never true

1829 return "stalled" 

1830 return "connected" 

1831 

1832 

1833async def _resolve_public_readiness_db(response: Response) -> str: 

1834 """ 

1835 Return the db status string for the public probe and flip the response to 

1836 503 when a configured DB is unreachable or stalled. Mirrors the legacy values: 

1837 "Not connected" (no DB configured), "connected", "disconnected", plus "stalled". 

1838 """ 

1839 from litellm.proxy.proxy_server import prisma_client 

1840 

1841 if prisma_client is None: 1841 ↛ 1842line 1841 didn't jump to line 1842 because the condition on line 1841 was never true

1842 return "Not connected" 

1843 

1844 db_status: Final = _readiness_db_status(await _db_health_readiness_check()) 

1845 if db_status != "connected" and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable(): 1845 ↛ 1846line 1845 didn't jump to line 1846 because the condition on line 1845 was never true

1846 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE 

1847 return db_status 

1848 

1849 

1850@router.get( 

1851 "/health/readiness", 

1852 tags=["health"], 

1853) 

1854async def health_readiness(response: Response): 

1855 """ 

1856 Public readiness probe. Returns a low-detail payload safe to expose to 

1857 unauthenticated load balancers — `status` plus `db` so orchestrators and 

1858 external probes can distinguish "healthy" from "DB unreachable" without a 

1859 credential. Admins can opt into the legacy detailed payload with 

1860 general_settings.allow_public_health_readiness_details. 

1861 """ 

1862 if GracefulShutdownManager.is_shutting_down(): 1862 ↛ 1863line 1862 didn't jump to line 1863 because the condition on line 1862 was never true

1863 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE 

1864 return {"status": "shutting_down"} 

1865 

1866 if _allow_public_health_readiness_details(): 1866 ↛ 1867line 1866 didn't jump to line 1867 because the condition on line 1866 was never true

1867 return await _get_health_readiness_details(response=response) 

1868 

1869 db_status: Final = await _resolve_public_readiness_db(response=response) 

1870 return {"status": "healthy", "db": db_status} 

1871 

1872 

1873@router.get( 

1874 "/health/readiness/details", 

1875 tags=["health"], 

1876 dependencies=[Depends(user_api_key_auth)], 

1877) 

1878async def health_readiness_details(response: Response): 

1879 """ 

1880 Authenticated readiness diagnostics with DB/cache/callback metadata. 

1881 """ 

1882 return await _get_health_readiness_details(response=response) 

1883 

1884 

1885@router.get( 

1886 "/health/backlog", 

1887 tags=["health"], 

1888 dependencies=[Depends(user_api_key_auth)], 

1889) 

1890async def health_backlog(): 

1891 """ 

1892 Returns the number of HTTP requests currently in-flight on this uvicorn worker. 

1893 

1894 Use this to measure per-pod queue depth. A high value means the worker is 

1895 processing many concurrent requests — requests arriving now will have to wait 

1896 for the event loop to get to them, adding latency before LiteLLM even starts 

1897 its own timer. 

1898 """ 

1899 stats: Final = get_admission_control_stats() 

1900 response: Final[_HealthBacklogResponse] = { 

1901 "in_flight_requests": get_in_flight_requests(), 

1902 "admitted_requests": stats.admitted, 

1903 "queued_requests": stats.queued, 

1904 "rejected_requests": stats.rejected_total, 

1905 } 

1906 return response 

1907 

1908 

1909@router.get( 

1910 "/health/drain", 

1911 tags=["health"], 

1912) 

1913async def health_drain(request: Request): 

1914 """ 

1915 Graceful-drain probe for Kubernetes ``preStop`` hooks. 

1916 

1917 Disabled by default and returns 404 unless ``general_settings`` sets 

1918 ``enable_drain_endpoint: true``. Calling it flips a process-wide 

1919 shutting-down flag, so a successful call permanently takes the worker out 

1920 of rotation until the pod restarts. 

1921 

1922 Because the kubelet calls preStop hooks without proxy credentials, the 

1923 endpoint does not require ``user_api_key_auth``. To prevent any 

1924 pod-reachable caller from triggering shutdown, set 

1925 ``general_settings.drain_endpoint_token`` (or the ``DRAIN_ENDPOINT_TOKEN`` 

1926 env var) and supply the same value on the ``X-Drain-Token`` header from 

1927 the preStop hook. Calls without the header (or with a wrong value) get a 

1928 401 and have no side effect. 

1929 

1930 When enabled, it marks the worker as shutting down (so /health/readiness 

1931 and /health/liveliness immediately start returning 503, removing the pod 

1932 from service) and blocks until the in-flight request counter drains to 

1933 zero or ``GRACEFUL_SHUTDOWN_TIMEOUT`` elapses. Unlike a fixed ``sleep``, 

1934 this returns as soon as real in-flight work is done. 

1935 

1936 Wire it up as: 

1937 

1938 ```yaml 

1939 lifecycle: 

1940 preStop: 

1941 httpGet: 

1942 path: /health/drain 

1943 port: 4000 

1944 httpHeaders: 

1945 - name: X-Drain-Token 

1946 value: <same value as drain_endpoint_token> 

1947 ``` 

1948 """ 

1949 if not _drain_endpoint_enabled(): 1949 ↛ 1951line 1949 didn't jump to line 1951 because the condition on line 1949 was always true

1950 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Not Found") 

1951 _authorize_drain_request(request) 

1952 GracefulShutdownManager.start_shutdown() 

1953 drained: Final = await GracefulShutdownManager.wait_for_drain(exclude_self=True) 

1954 return {"status": "drained", "drained_requests": drained} 

1955 

1956 

1957@router.get( 

1958 "/health/liveliness", # Historical LiteLLM name; doesn't match k8s terminology but kept for backwards compatibility 

1959 tags=["health"], 

1960) 

1961@router.get( 

1962 "/health/liveness", # Kubernetes has "liveness" probes (https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#define-a-liveness-command) 

1963 tags=["health"], 

1964) 

1965async def health_liveliness(response: Response): 

1966 """ 

1967 Unprotected endpoint for checking if worker is alive. 

1968 

1969 Returns 503 once graceful shutdown has begun so Kubernetes stops counting 

1970 the draining pod as live and terminates it on schedule. 

1971 """ 

1972 if GracefulShutdownManager.is_shutting_down(): 1972 ↛ 1973line 1972 didn't jump to line 1973 because the condition on line 1972 was never true

1973 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE 

1974 return {"status": "shutting_down"} 

1975 return "I'm alive!" 

1976 

1977 

1978@router.options( 

1979 "/health/readiness", 

1980 tags=["health"], 

1981) 

1982async def health_readiness_options(): 

1983 """ 

1984 Options endpoint for health/readiness check. 

1985 """ 

1986 response_headers: Final = { 

1987 "Allow": "GET, OPTIONS", 

1988 "Access-Control-Allow-Methods": "GET, OPTIONS", 

1989 "Access-Control-Allow-Headers": "*", 

1990 } 

1991 return Response(headers=response_headers, status_code=200) 

1992 

1993 

1994@router.options( 

1995 "/health/liveliness", 

1996 tags=["health"], 

1997) 

1998@router.options( 

1999 "/health/liveness", # Kubernetes has "liveness" probes (https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#define-a-liveness-command) 

2000 tags=["health"], 

2001) 

2002async def health_liveliness_options(): 

2003 """ 

2004 Options endpoint for health/liveliness check. 

2005 """ 

2006 response_headers: Final = { 

2007 "Allow": "GET, OPTIONS", 

2008 "Access-Control-Allow-Methods": "GET, OPTIONS", 

2009 "Access-Control-Allow-Headers": "*", 

2010 } 

2011 return Response(headers=response_headers, status_code=200) 

2012 

2013 

2014@router.post( 

2015 "/health/test_connection", 

2016 tags=["health"], 

2017 dependencies=[Depends(user_api_key_auth)], 

2018) 

2019async def test_model_connection( 

2020 request: Request, 

2021 mode: Literal[ 

2022 "chat", 

2023 "completion", 

2024 "embedding", 

2025 "audio_speech", 

2026 "audio_transcription", 

2027 "image_generation", 

2028 "image_edit", 

2029 "video_generation", 

2030 "batch", 

2031 "rerank", 

2032 "realtime", 

2033 "responses", 

2034 "ocr", 

2035 ] 

2036 | None = fastapi.Body( 

2037 None, 

2038 description="The mode to test the model with. If not provided, auto-detected from model capabilities.", 

2039 ), 

2040 litellm_params: dict = fastapi.Body( 

2041 None, 

2042 description="Parameters for litellm.completion, litellm.embedding for the health check", 

2043 ), 

2044 model_info: dict = fastapi.Body( 

2045 None, 

2046 description="Model info for the health check", 

2047 ), 

2048 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

2049): 

2050 """ 

2051 Test a direct connection to a specific model. 

2052  

2053 This endpoint allows you to verify if your proxy can successfully connect to a specific model. 

2054 It's useful for troubleshooting model connectivity issues without going through the full proxy routing. 

2055  

2056 Example: 

2057 ```bash 

2058 # If model is configured in proxy_config.yaml, you only need to specify the model name: 

2059 curl -X POST 'http://localhost:4000/health/test_connection' \\ 

2060 -H 'Authorization: Bearer sk-1234' \\ 

2061 -H 'Content-Type: application/json' \\ 

2062 -d '{ 

2063 "litellm_params": { 

2064 "model": "gpt-4o" 

2065 }, 

2066 "mode": "chat" 

2067 }' 

2068  

2069 # The endpoint will automatically use api_key, api_base, etc. from proxy_config.yaml 

2070  

2071 # You can also override specific params or test with custom credentials: 

2072 curl -X POST 'http://localhost:4000/health/test_connection' \\ 

2073 -H 'Authorization: Bearer sk-1234' \\ 

2074 -H 'Content-Type: application/json' \\ 

2075 -d '{ 

2076 "litellm_params": { 

2077 "model": "azure/gpt-4o", 

2078 "api_key": "os.environ/AZURE_OPENAI_API_KEY", 

2079 "api_base": "os.environ/AZURE_OPENAI_ENDPOINT", 

2080 "api_version": "2024-10-21" 

2081 }, 

2082 "mode": "chat" 

2083 }' 

2084 ``` 

2085  

2086 Note:  

2087 - If the model is configured in proxy_config.yaml, credentials (api_key, api_base, etc.)  

2088 will be automatically loaded from the config (with resolved environment variables). 

2089 - A request naming a stored credential (`litellm_credential_name`) that the configuration 

2090 does not name is probed with that credential instead, and inherits no credentials 

2091 from the configuration its model string happened to match. 

2092 - You can override specific params by including them in the request. 

2093 - You can use `os.environ/VARIABLE_NAME` syntax to reference environment variables, 

2094 which will be resolved automatically (same as in proxy_config.yaml). 

2095  

2096 Returns: 

2097 dict: A dictionary containing the health check result with either success information or error details. 

2098 """ 

2099 from litellm.proxy._types import CommonProxyErrors 

2100 from litellm.proxy.management_endpoints.model_management_endpoints import ( 

2101 ModelManagementAuthChecks, 

2102 ) 

2103 from litellm.proxy.proxy_server import ( 

2104 general_settings, 

2105 llm_router, 

2106 premium_user, 

2107 prisma_client, 

2108 ) 

2109 from litellm.types.router import Deployment, LiteLLM_Params 

2110 

2111 try: 

2112 if prisma_client is None: 2112 ↛ 2113line 2112 didn't jump to line 2113 because the condition on line 2112 was never true

2113 raise HTTPException( 

2114 status_code=500, 

2115 detail={"error": CommonProxyErrors.db_not_connected_error.value}, 

2116 ) 

2117 

2118 # Get model name from litellm_params 

2119 request_litellm_params: Final = litellm_params or {} 

2120 # Reject request-supplied os.environ/ references. Config values are 

2121 # already resolved before reaching this endpoint; any remaining 

2122 # reference must have come from the request body. 

2123 _reject_os_environ_references(request_litellm_params) 

2124 if model_info: 

2125 _reject_os_environ_references(model_info) 

2126 model_name: Final = request_litellm_params.get("model") 

2127 

2128 # Look up model configuration from router if model name is provided 

2129 # This gets the litellm_params from proxy config (with resolved env vars) 

2130 config_litellm_params: dict = {} 

2131 loaded_model_info: dict | None = None 

2132 if llm_router is not None: 2132 ↛ 2179line 2132 didn't jump to line 2179 because the condition on line 2132 was always true

2133 # Prefer disambiguation by deployment id (`model_info.id`) when 

2134 # the caller supplies it. This is required when multiple 

2135 # deployments share a `model_name` (e.g. wildcard `openai/*` 

2136 # with multiple `api_base` values for failover): the UI's 

2137 # "Test Connection" button targets a specific row, and that 

2138 # row's id is the only thing that uniquely identifies which 

2139 # deployment to probe. Without this, all duplicates collapse 

2140 # onto `deployments[0]`. 

2141 request_model_info: Final = model_info or {} 

2142 request_model_id: Final = request_model_info.get("id") 

2143 try: 

2144 deployment_by_id = None 

2145 if request_model_id: 2145 ↛ 2146line 2145 didn't jump to line 2146 because the condition on line 2145 was never true

2146 deployment_by_id = llm_router.get_deployment(model_id=request_model_id) 

2147 

2148 if deployment_by_id is not None: 2148 ↛ 2149line 2148 didn't jump to line 2149 because the condition on line 2148 was never true

2149 config_litellm_params = deployment_by_id.litellm_params.model_dump(exclude_none=True) 

2150 loaded_model_info = deployment_by_id.model_info.model_dump(exclude_none=True) 

2151 elif model_name: 2151 ↛ 2155line 2151 didn't jump to line 2155 because the condition on line 2151 was never true

2152 # Fall back to model_name lookup for callers (e.g. the 

2153 # "Add Model" wizard, or curl) that don't supply an id. 

2154 # First try to find by proxy model_name (e.g., "gpt-4o") 

2155 deployments = llm_router.get_model_list(model_name=model_name) 

2156 

2157 # If not found, try to find by litellm model name 

2158 # (e.g., "azure/gpt-4o") 

2159 if not deployments or len(deployments) == 0: 

2160 all_deployments: Final = llm_router.get_model_list(model_name=None) 

2161 if all_deployments: 

2162 for deployment in all_deployments: 

2163 if deployment.get("litellm_params", {}).get("model") == model_name: 

2164 deployments = [deployment] 

2165 break 

2166 

2167 if deployments and len(deployments) > 0: 

2168 # Use the first deployment's litellm_params as base 

2169 # config. These already have resolved environment 

2170 # variables from proxy config. 

2171 config_litellm_params = dict(deployments[0].get("litellm_params", {})) 

2172 loaded_model_info = dict(deployments[0].get("model_info") or {}) 

2173 except Exception as e: 

2174 verbose_proxy_logger.debug( 

2175 "Could not find model %s in router: %s. Proceeding with request params only.", model_name, e 

2176 ) 

2177 

2178 # Merge: config params (from proxy config) as base, request params override 

2179 litellm_params = { 

2180 **_config_base_for_health_check( 

2181 config_litellm_params, 

2182 request_litellm_params, 

2183 allow_client_side_credentials=general_settings.get("allow_client_side_credentials") is True, 

2184 ), 

2185 **request_litellm_params, 

2186 } 

2187 

2188 resolved_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info 

2189 litellm_params = _update_litellm_params_for_health_check( 

2190 model_info=resolved_model_info or {}, 

2191 litellm_params=litellm_params, 

2192 ) 

2193 

2194 ## Auth check, on the final probe params so health_check_params cannot retarget it afterwards 

2195 await ModelManagementAuthChecks.can_user_make_model_call( 

2196 model_params=Deployment( 

2197 model_name="test_model", 

2198 litellm_params=LiteLLM_Params(**litellm_params), 

2199 model_info=resolved_model_info, 

2200 ), 

2201 user_api_key_dict=user_api_key_dict, 

2202 prisma_client=prisma_client, 

2203 premium_user=premium_user, 

2204 ) 

2205 mode = mode or litellm_params.pop("mode", None) 

2206 

2207 result: Final = await run_with_timeout( 

2208 litellm.ahealth_check( 

2209 model_params=litellm_params, 

2210 mode=mode, 

2211 prompt="test from litellm", 

2212 input=["test from litellm"], 

2213 ), 

2214 HEALTH_CHECK_TIMEOUT_SECONDS, 

2215 ) 

2216 

2217 # Clean the result for display 

2218 cleaned_result: Final = _clean_endpoint_data({**litellm_params, **result}, details=True) 

2219 

2220 return { 

2221 "status": "error" if "error" in result else "success", 

2222 "result": cleaned_result, 

2223 } 

2224 

2225 except HTTPException as e: 

2226 raise e 

2227 except Exception as e: 

2228 verbose_proxy_logger.debug("litellm.proxy.health_endpoints.test_model_connection(): Exception occurred - %s", e) 

2229 raise HTTPException( 

2230 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, 

2231 detail={"error": f"Failed to test connection: {e}"}, 

2232 )