Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/health_endpoints/_health_endpoints.py: 47%
735 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1import asyncio
2import copy
3import json
4import logging
5import os
6import secrets
7import time
8import traceback
9from collections.abc import Iterable, Mapping
10from datetime import datetime, timedelta, timezone
11from typing import Any, Final, Literal, TypedDict, cast
13import fastapi
14from fastapi import APIRouter, Depends, HTTPException, Request, Response, status
15from typing_extensions import ReadOnly
17import litellm
18from litellm._logging import verbose_logger, verbose_proxy_logger
19from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS, PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS
20from litellm.integrations.SlackAlerting.ms_teams import (
21 MS_TEAMS_ALERT_HEADERS,
22 build_ms_teams_payload,
23 get_ms_teams_webhook_url,
24)
25from litellm.litellm_core_utils.custom_logger_registry import CustomLoggerRegistry
26from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
27from litellm.proxy._types import (
28 AlertType,
29 CallInfo,
30 EnterpriseLicenseData,
31 Litellm_EntityType,
32 LitellmUserRoles,
33 ProxyErrorTypes,
34 ProxyException,
35 SpecialModelNames,
36 UserAPIKeyAuth,
37 WebhookEvent,
38)
39from litellm.proxy.auth.auth_checks import (
40 _resolve_key_models_for_auth_check, # pyright: ignore[reportPrivateUsage] # the auth layer's sentinel resolution, reused so /health scopes exactly like a request
41)
42from litellm.proxy.auth.auth_utils import (
43 _BANNED_REQUEST_BODY_PARAMS, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the request-body check
44)
45from litellm.proxy.auth.model_checks import get_key_models
46from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
47from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker
48from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler
49from litellm.proxy.db.health_check_latest import (
50 LatestHealthCheckRow,
51 query_latest_health_checks,
52)
53from litellm.proxy.db.proxy_worker_heartbeat import count_live_proxy_workers
54from litellm.proxy.health_check import (
55 ADMIN_ONLY_HEALTH_DISPLAY_PARAMS,
56 _clean_endpoint_data,
57 _update_litellm_params_for_health_check,
58 deployments_targeted_by_name,
59 health_check_filter_kwargs_from_general_settings,
60 perform_health_check,
61 run_with_timeout,
62)
63from litellm.proxy.middleware.admission_control_middleware import (
64 get_admission_control_stats,
65)
66from litellm.proxy.middleware.in_flight_requests_middleware import (
67 get_in_flight_requests,
68)
69from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager
70from litellm.router import Router
71from litellm.router_utils.clientside_credential_handler import (
72 _ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the router path
73 clientside_credential_keys,
74)
75from litellm.secret_managers.main import get_secret_bool
77#### Health ENDPOINTS ####
80class _HealthBacklogResponse(TypedDict):
81 in_flight_requests: ReadOnly[int]
82 admitted_requests: ReadOnly[int]
83 queued_requests: ReadOnly[int]
84 rejected_requests: ReadOnly[int]
87def _reject_os_environ_references(params: dict) -> None:
88 """
89 Validate that the provided params do not contain any ``os.environ/``
90 references. Values with that prefix are expected to come only from
91 server-side configuration (already resolved before reaching here). If a
92 request-supplied value still carries the prefix, raise ``HTTPException``.
93 """
94 if not isinstance(params, dict): 94 ↛ 95line 94 didn't jump to line 95 because the condition on line 94 was never true
95 return
97 stack: Final[list[object]] = [params]
98 seen: Final[set[int]] = {id(params)}
100 while stack:
101 src = stack.pop()
102 if isinstance(src, dict):
103 values: Iterable[object] = src.values()
104 elif isinstance(src, list): 104 ↛ 107line 104 didn't jump to line 107 because the condition on line 104 was always true
105 values = src
106 else:
107 continue
109 for value in values:
110 if isinstance(value, str) and value.startswith("os.environ/"): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true
111 raise HTTPException(
112 status_code=400,
113 detail={"error": "Environment variable references are not permitted in request parameters."},
114 )
115 if isinstance(value, (dict, list)) and id(value) not in seen:
116 seen.add(id(value))
117 stack.append(value)
120_CONFIG_CONNECTION_FIELDS: Final[frozenset[str]] = frozenset(
121 (
122 *_ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE,
123 *clientside_credential_keys,
124 "litellm_credential_name",
125 )
126)
129def _request_inherits_config_credentials(
130 config_params: Mapping[str, object],
131 request_params: Mapping[str, object],
132 allow_client_side_credentials: bool,
133) -> bool:
134 """Whether the configuration's credentials are this request's to be probed with.
136 The configuration reached here by matching the request's model string, which
137 also matches wildcard routes and unrelated deployments that merely serve the
138 same model, so a request naming a stored credential of its own has already
139 said where its credentials come from and does not borrow that one's. A blank
140 name is no name: ``load_credentials_from_list`` resolves nothing from it, so
141 it must not cost the request the credentials it would otherwise be probed
142 with.
143 """
144 requested_credential: Final = request_params.get("litellm_credential_name")
145 if requested_credential and requested_credential != config_params.get("litellm_credential_name"): 145 ↛ 146line 145 didn't jump to line 146 because the condition on line 145 was never true
146 return False
147 if allow_client_side_credentials: 147 ↛ 148line 147 didn't jump to line 148 because the condition on line 147 was never true
148 return True
149 return not any(param in request_params for param in _BANNED_REQUEST_BODY_PARAMS)
152def _config_base_for_health_check(
153 config_params: Mapping[str, object],
154 request_params: Mapping[str, object],
155 allow_client_side_credentials: bool = False,
156) -> dict[str, object]:
157 """Return the configured parameters to merge under a connection-test request.
159 A request that sets its own connection fields, or names its own stored
160 credential, describes a connection of its own, so the configuration's
161 credentials are not carried into it: they belong to the endpoint the
162 configuration names. Anything the request does not set still comes from the
163 configuration, which is what lets a request name a configured model and test
164 it as configured.
166 ``litellm_credential_name`` is dropped alongside the literal credential
167 fields: it names a stored credential that ``load_credentials_from_list``
168 resolves into the same secrets further down the call, so leaving it in place
169 would reintroduce them by reference.
170 """
171 if _request_inherits_config_credentials(config_params, request_params, allow_client_side_credentials): 171 ↛ 173line 171 didn't jump to line 173 because the condition on line 171 was always true
172 return dict(config_params)
173 return {key: value for key, value in config_params.items() if key not in _CONFIG_CONNECTION_FIELDS}
176def get_callback_identifier(callback):
177 """
178 Get the callback identifier string, handling both strings and objects.
180 This function extracts a string identifier from a callback, which can be:
181 - A string (returned as-is)
182 - An object with a callback_name attribute
183 - An object registered in CustomLoggerRegistry
184 - Falls back to callback_name() helper function
186 Args:
187 callback: The callback to identify (can be str or object)
189 Returns:
190 str: The callback identifier string
191 """
192 if isinstance(callback, str):
193 return callback
194 if hasattr(callback, "callback_name") and callback.callback_name: 194 ↛ 195line 194 didn't jump to line 195 because the condition on line 194 was never true
195 return callback.callback_name
196 if hasattr(callback, "__class__"): 196 ↛ 202line 196 didn't jump to line 202 because the condition on line 196 was always true
197 callback_strs: Final = CustomLoggerRegistry.get_all_callback_strs_from_class_type(callback.__class__)
198 if hasattr(callback, "callback_name") and callback.callback_name in callback_strs: 198 ↛ 199line 198 didn't jump to line 199 because the condition on line 198 was never true
199 return callback.callback_name
200 if callback_strs: 200 ↛ 201line 200 didn't jump to line 201 because the condition on line 200 was never true
201 return callback_strs[0]
202 return callback_name(callback)
205router: Final = APIRouter()
206services = (
207 Literal[
208 "slack_budget_alerts",
209 "langfuse",
210 "langfuse_otel",
211 "slack",
212 "ms_teams",
213 "openmeter",
214 "webhook",
215 "email",
216 "braintrust",
217 "datadog",
218 "datadog_llm_observability",
219 "generic_api",
220 "arize",
221 "galileo",
222 "newrelic",
223 "pointfive",
224 "sqs",
225 ]
226 | str
227)
230class _ServiceTestErrorDetail(TypedDict):
231 error: ReadOnly[str]
234class _ServiceTestSuccessResponse(TypedDict):
235 status: ReadOnly[str]
236 message: ReadOnly[str]
239@router.get(
240 "/test",
241 tags=["health"],
242 dependencies=[Depends(user_api_key_auth)],
243)
244async def test_endpoint(request: Request):
245 """
246 [DEPRECATED] use `/health/liveliness` instead.
248 A test endpoint that pings the proxy server to check if it's healthy.
250 Parameters:
251 request (Request): The incoming request.
253 Returns:
254 dict: A dictionary containing the route of the request URL.
255 """
256 # ping the proxy server to check if its healthy
257 # Inline import — auth_utils participates in a proxy import cycle.
258 from litellm.proxy.auth.auth_utils import get_request_route # noqa: PLC0415
260 return {"route": get_request_route(request)}
263@router.get(
264 "/health/services",
265 tags=["health"],
266 dependencies=[Depends(user_api_key_auth)],
267)
268async def health_services_endpoint(
269 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
270 service: services = fastapi.Query(description="Specify the service being hit."),
271):
272 """
273 Use this admin-only endpoint to check if the service is healthy.
275 Example:
276 ```
277 curl -L -X GET 'http://0.0.0.0:4000/health/services?service=datadog' \
278 -H 'Authorization: Bearer sk-1234'
279 ```
280 """
281 try:
282 from litellm.proxy.proxy_server import (
283 general_settings,
284 prisma_client,
285 proxy_logging_obj,
286 )
288 if service is None: 288 ↛ 289line 288 didn't jump to line 289 because the condition on line 288 was never true
289 raise HTTPException(status_code=400, detail={"error": "Service must be specified."})
291 if service not in [
292 "slack_budget_alerts",
293 "email",
294 "langfuse",
295 "langfuse_otel",
296 "slack",
297 "ms_teams",
298 "openmeter",
299 "webhook",
300 "braintrust",
301 "otel",
302 "custom_callback_api",
303 "langsmith",
304 "datadog",
305 "datadog_metrics",
306 "datadog_llm_observability",
307 "generic_api",
308 "arize",
309 "galileo",
310 "newrelic",
311 "pointfive",
312 "sqs",
313 ]:
314 raise HTTPException(
315 status_code=400,
316 detail={"error": f"Service must be in list. Service={service} not in {services}"},
317 )
319 service_in_success_callbacks = False
320 if service in litellm.success_callback: 320 ↛ 321line 320 didn't jump to line 321 because the condition on line 320 was never true
321 service_in_success_callbacks = True
322 else:
323 for cb in litellm.success_callback:
324 if getattr(cb, "callback_name", None) == service: 324 ↛ 325line 324 didn't jump to line 325 because the condition on line 324 was never true
325 service_in_success_callbacks = True
326 break
327 cb_id = get_callback_identifier(cb)
328 if cb_id == service: 328 ↛ 329line 328 didn't jump to line 329 because the condition on line 328 was never true
329 service_in_success_callbacks = True
330 break
332 if ( 332 ↛ 338line 332 didn't jump to line 338 because the condition on line 332 was never true
333 service == "openmeter"
334 or service == "braintrust"
335 or service == "generic_api"
336 or (service_in_success_callbacks and service not in ("langfuse", "pointfive"))
337 ):
338 _ = await litellm.acompletion(
339 model="openai/litellm-mock-response-model",
340 messages=[{"role": "user", "content": "Hey, how's it going?"}],
341 user="litellm:/health/services",
342 mock_response="This is a mock response",
343 )
344 return {
345 "status": "success",
346 "message": f"Mock LLM request made - check {service}.",
347 }
348 elif service == "datadog": 348 ↛ 349line 348 didn't jump to line 349 because the condition on line 348 was never true
349 from litellm.integrations.datadog.datadog import DataDogLogger
351 datadog_logger: Final = DataDogLogger()
352 response = await datadog_logger.async_health_check()
353 return {
354 "status": response["status"],
355 "message": (response["error_message"] if response["status"] == "unhealthy" else "Datadog is healthy"),
356 }
357 elif service == "datadog_metrics": 357 ↛ 358line 357 didn't jump to line 358 because the condition on line 357 was never true
358 from litellm.integrations.datadog.datadog_metrics import (
359 DatadogMetricsLogger,
360 )
361 from litellm.litellm_core_utils.litellm_logging import (
362 get_custom_logger_compatible_class,
363 )
365 datadog_metrics_logger = get_custom_logger_compatible_class("datadog_metrics")
366 if datadog_metrics_logger is None:
367 datadog_metrics_logger = DatadogMetricsLogger(start_periodic_flush=False)
368 assert isinstance(datadog_metrics_logger, DatadogMetricsLogger)
369 response = await datadog_metrics_logger.async_health_check()
370 return {
371 "status": response["status"],
372 "message": (
373 response["error_message"] if response["status"] == "unhealthy" else "Datadog Metrics is healthy"
374 ),
375 }
376 elif service == "arize": 376 ↛ 377line 376 didn't jump to line 377 because the condition on line 376 was never true
377 from litellm.integrations.arize.arize import ArizeLogger
379 arize_logger: Final = ArizeLogger()
380 response = await arize_logger.async_health_check()
381 return {
382 "status": response["status"],
383 "message": (response["error_message"] if response["status"] == "unhealthy" else "Arize is healthy"),
384 }
385 elif service == "galileo": 385 ↛ 386line 385 didn't jump to line 386 because the condition on line 385 was never true
386 from litellm.integrations.galileo import GalileoObserve
388 galileo_logger: Final = GalileoObserve()
389 response = await galileo_logger.async_health_check()
390 return {
391 "status": response["status"],
392 "message": (response["error_message"] if response["status"] == "unhealthy" else "Galileo is healthy"),
393 }
394 elif service == "langfuse": 394 ↛ 395line 394 didn't jump to line 395 because the condition on line 394 was never true
395 from litellm.integrations.langfuse.langfuse import LangFuseLogger
397 langfuse_logger: Final = LangFuseLogger()
398 auth_failure: Final = langfuse_logger.api_client.auth_check()
399 if auth_failure is not None:
400 raise ValueError(f"langfuse auth_check failed: {auth_failure.reason}")
401 _ = litellm.completion(
402 model="openai/litellm-mock-response-model",
403 messages=[{"role": "user", "content": "Hey, how's it going?"}],
404 user="litellm:/health/services",
405 mock_response="This is a mock response",
406 )
407 return {
408 "status": "success",
409 "message": "Mock LLM request made - check langfuse.",
410 }
411 elif service == "newrelic": 411 ↛ 412line 411 didn't jump to line 412 because the condition on line 411 was never true
412 if not _is_proxy_admin(user_api_key_dict):
413 raise HTTPException(
414 status_code=status.HTTP_403_FORBIDDEN,
415 detail={"error": "Only proxy admins can trigger the New Relic test event."},
416 )
417 from litellm.integrations.newrelic.newrelic import NewRelicLogger
419 newrelic_logger: Final = NewRelicLogger()
420 response = await newrelic_logger.async_health_check()
421 return {
422 "status": response["status"],
423 "message": (
424 response["error_message"]
425 if response["status"] == "unhealthy"
426 else "New Relic is healthy — test event sent"
427 ),
428 }
430 elif service == "pointfive": 430 ↛ 431line 430 didn't jump to line 431 because the condition on line 430 was never true
431 if not _is_proxy_admin(user_api_key_dict):
432 non_admin_detail: Final[_ServiceTestErrorDetail] = {
433 "error": "Only proxy admins can trigger the PointFive liveness ping."
434 }
435 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=non_admin_detail)
436 from litellm.integrations.pointfive import PointFiveLogger
438 try:
439 pointfive_logger: Final = PointFiveLogger(start_periodic_flush=False)
440 except ValueError as missing_key:
441 # No key configured is the answer the operator asked for, not a server error.
442 no_key: Final[_ServiceTestSuccessResponse] = {"status": "unhealthy", "message": str(missing_key)}
443 return no_key
444 response = await pointfive_logger.async_health_check()
445 pointfive_health: Final[_ServiceTestSuccessResponse] = {
446 "status": response["status"],
447 "message": (response["error_message"] if response["status"] == "unhealthy" else "PointFive is healthy")
448 or "PointFive is healthy",
449 }
450 return pointfive_health
451 if service == "webhook": 451 ↛ 452line 451 didn't jump to line 452 because the condition on line 451 was never true
452 if not _is_proxy_admin(user_api_key_dict):
453 webhook_non_admin_detail: Final[_ServiceTestErrorDetail] = {
454 "error": "Only proxy admins can trigger the webhook test alert."
455 }
456 raise HTTPException(status_code=status.HTTP_403_FORBIDDEN, detail=webhook_non_admin_detail)
457 user_info: Final = CallInfo(
458 token=user_api_key_dict.token or "",
459 spend=1,
460 max_budget=0,
461 user_id=user_api_key_dict.user_id,
462 key_alias=user_api_key_dict.key_alias,
463 team_id=user_api_key_dict.team_id,
464 event_group=Litellm_EntityType.KEY,
465 )
466 await proxy_logging_obj.budget_alerts(
467 type="user_budget",
468 user_info=user_info,
469 )
470 elif service == "sqs": 470 ↛ 480line 470 didn't jump to line 480 because the condition on line 470 was always true
471 from litellm.integrations.sqs import SQSLogger
473 sqs_logger: Final = SQSLogger()
474 response = await sqs_logger.async_health_check()
475 return {
476 "status": response["status"],
477 "message": response["error_message"],
478 }
480 if service == "slack" or service == "slack_budget_alerts":
481 if "slack" in general_settings.get("alerting", []):
482 # test_message = f"""\n🚨 `ProjectedLimitExceededError` 💸\n\n`Key Alias:` litellm-ui-test-alert \n`Expected Day of Error`: 28th March \n`Current Spend`: $100.00 \n`Projected Spend at end of month`: $1000.00 \n`Soft Limit`: $700"""
483 # check if user has opted into unique_alert_webhooks
484 if proxy_logging_obj.slack_alerting_instance.alert_to_webhook_url is not None:
485 for alert_type in proxy_logging_obj.slack_alerting_instance.alert_to_webhook_url:
486 # only test alert if it's in active alert types
487 if (
488 proxy_logging_obj.slack_alerting_instance.alert_types is not None
489 and alert_type not in proxy_logging_obj.slack_alerting_instance.alert_types
490 ):
491 continue
493 test_message = "default test message"
494 if alert_type == AlertType.llm_exceptions:
495 test_message = "LLM Exception test alert"
496 elif alert_type == AlertType.llm_too_slow:
497 test_message = "LLM Too Slow test alert"
498 elif alert_type == AlertType.llm_requests_hanging:
499 test_message = "LLM Requests Hanging test alert"
500 elif alert_type == AlertType.budget_alerts:
501 test_message = "Budget Alert test alert"
502 elif alert_type == AlertType.db_exceptions:
503 test_message = "DB Exception test alert"
504 elif alert_type == AlertType.outage_alerts:
505 test_message = "Outage Alert Exception test alert"
506 elif alert_type == AlertType.daily_reports:
507 test_message = "Daily Reports test alert"
508 else:
509 test_message = "Budget Alert test alert"
511 await proxy_logging_obj.alerting_handler(
512 message=test_message, level="Low", alert_type=alert_type
513 )
514 else:
515 await proxy_logging_obj.alerting_handler(
516 message="This is a test slack alert message",
517 level="Low",
518 alert_type=AlertType.budget_alerts,
519 )
521 if prisma_client is not None:
522 asyncio.create_task(proxy_logging_obj.slack_alerting_instance.send_monthly_spend_report())
523 asyncio.create_task(proxy_logging_obj.slack_alerting_instance.send_weekly_spend_report())
525 alert_types = proxy_logging_obj.slack_alerting_instance.alert_types or []
526 alert_types = list(alert_types)
527 return {
528 "status": "success",
529 "alert_types": alert_types,
530 "message": "Mock Slack Alert sent, verify Slack Alert Received on your channel",
531 }
532 else:
533 raise HTTPException(
534 status_code=422,
535 detail={"error": f'"{service}" not in proxy config: general_settings. Unable to test this.'},
536 )
537 if service == "ms_teams":
538 if "ms_teams" not in general_settings.get("alerting", ()):
539 not_configured_detail: Final[_ServiceTestErrorDetail] = {
540 "error": f'"{service}" not in proxy config: general_settings. Unable to test this.'
541 }
542 raise HTTPException(status_code=422, detail=not_configured_detail)
543 ms_teams_webhook_url: Final = get_ms_teams_webhook_url()
544 if ms_teams_webhook_url is None:
545 missing_webhook_detail: Final[_ServiceTestErrorDetail] = {
546 "error": "MS_TEAMS_WEBHOOK_URL not set. Unable to test this."
547 }
548 raise HTTPException(status_code=422, detail=missing_webhook_detail)
549 ms_teams_test_message: Final = (
550 f"Alert type: `{AlertType.budget_alerts.value}`\nLevel: `Low`\n"
551 f"Timestamp: `{datetime.now().strftime('%H:%M:%S')}`\n\n"
552 "Message: This is a test MS Teams alert message"
553 )
554 ms_teams_response: Final = await proxy_logging_obj.slack_alerting_instance.async_http_handler.post(
555 url=ms_teams_webhook_url,
556 headers=dict(MS_TEAMS_ALERT_HEADERS), # mutable-ok: async_http_handler.post only accepts dict headers
557 data=json.dumps(build_ms_teams_payload(ms_teams_test_message)),
558 )
559 if ms_teams_response.status_code >= 400:
560 delivery_failed_detail: Final[_ServiceTestErrorDetail] = {
561 "error": f"MS Teams webhook returned status {ms_teams_response.status_code}: {ms_teams_response.text}"
562 }
563 raise HTTPException(status_code=500, detail=delivery_failed_detail)
564 ms_teams_success: Final[_ServiceTestSuccessResponse] = {
565 "status": "success",
566 "message": "Mock MS Teams Alert sent, verify MS Teams Alert Received in your channel",
567 }
568 return ms_teams_success
569 if service == "email":
570 webhook_event: Final = WebhookEvent(
571 event="key_created",
572 event_group=Litellm_EntityType.KEY,
573 event_message="Test Email Alert",
574 token=user_api_key_dict.token or "",
575 key_alias="Email Test key (This is only a test alert key. DO NOT USE THIS IN PRODUCTION.)",
576 spend=0,
577 max_budget=0,
578 user_id=user_api_key_dict.user_id,
579 user_email=os.getenv("TEST_EMAIL_ADDRESS"),
580 team_id=user_api_key_dict.team_id,
581 )
583 # use create task - this can take 10 seconds. don't keep ui users waiting for notification to check their email
584 await proxy_logging_obj.slack_alerting_instance.send_key_created_or_user_invited_email(
585 webhook_event=webhook_event
586 )
588 return {
589 "status": "success",
590 "message": "Mock Email Alert sent, verify Email Alert Received",
591 }
593 except Exception as e:
594 verbose_proxy_logger.error("litellm.proxy.proxy_server.health_services_endpoint(): Exception occured - %s", e)
595 verbose_proxy_logger.debug(traceback.format_exc())
596 if isinstance(e, HTTPException): 596 ↛ 603line 596 didn't jump to line 603 because the condition on line 596 was always true
597 raise ProxyException(
598 message=getattr(e, "detail", f"Authentication Error({e})"),
599 type=ProxyErrorTypes.auth_error,
600 param=getattr(e, "param", "None"),
601 code=getattr(e, "status_code", status.HTTP_500_INTERNAL_SERVER_ERROR),
602 )
603 elif isinstance(e, ProxyException):
604 raise e
605 raise ProxyException(
606 message="Authentication Error, " + str(e),
607 type=ProxyErrorTypes.auth_error,
608 param=getattr(e, "param", "None"),
609 code=status.HTTP_500_INTERNAL_SERVER_ERROR,
610 )
613def _convert_health_check_to_dict(check) -> dict:
614 """Convert health check database record to dictionary format"""
615 return {
616 "health_check_id": check.health_check_id,
617 "model_name": check.model_name,
618 "model_id": check.model_id,
619 "status": check.status,
620 "healthy_count": check.healthy_count,
621 "unhealthy_count": check.unhealthy_count,
622 "error_message": check.error_message,
623 "response_time_ms": check.response_time_ms,
624 "details": check.details,
625 "checked_by": check.checked_by,
626 "checked_at": check.checked_at.isoformat() if check.checked_at else None,
627 "created_at": check.created_at.isoformat() if check.created_at else None,
628 }
631def _check_prisma_client():
632 """Helper to check if prisma_client is available and raise appropriate error"""
633 from litellm.proxy.proxy_server import prisma_client
635 if prisma_client is None: 635 ↛ 636line 635 didn't jump to line 636 because the condition on line 635 was never true
636 raise HTTPException(
637 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
638 detail={"error": "Database not initialized"},
639 )
640 return prisma_client
643async def _save_health_check_to_db(
644 prisma_client,
645 model_name: str,
646 healthy_endpoints: list,
647 unhealthy_endpoints: list,
648 start_time: float,
649 user_id: str | None,
650 model_id: str | None = None,
651):
652 """Helper function to save health check results to database"""
653 try:
654 # Extract error message from first unhealthy endpoint if available
655 error_message: Final = (
656 str(unhealthy_endpoints[0]["error"])[:500]
657 if unhealthy_endpoints and unhealthy_endpoints[0].get("error")
658 else None
659 )
661 await prisma_client.save_health_check_result(
662 model_name=model_name,
663 model_id=model_id,
664 status="healthy" if healthy_endpoints else "unhealthy",
665 healthy_count=len(healthy_endpoints),
666 unhealthy_count=len(unhealthy_endpoints),
667 error_message=error_message,
668 response_time_ms=(time.time() - start_time) * 1000,
669 details=None, # Skip details for now to avoid JSON serialization issues
670 checked_by=user_id,
671 )
672 except Exception as db_error:
673 verbose_proxy_logger.warning("Failed to save health check to database for model %s: %s", model_name, db_error)
674 # Continue execution - don't let database save failure break health checks
677def _build_model_param_to_info_mapping(model_list: list) -> dict:
678 """
679 Build a mapping from model parameter to model info (model_name, model_id).
681 Multiple models might share the same model parameter, so we use a list.
683 Args:
684 model_list: List of model configurations
686 Returns:
687 Dictionary mapping model parameter to list of model info dicts
688 """
689 model_param_to_info: Final[dict] = {}
690 for model in model_list:
691 model_info = model.get("model_info", {})
692 model_name = model.get("model_name")
693 model_id = model_info.get("id")
694 litellm_params = model.get("litellm_params", {})
695 model_param = litellm_params.get("model")
697 if model_param and model_name:
698 if model_param not in model_param_to_info:
699 model_param_to_info[model_param] = []
700 model_param_to_info[model_param].append(
701 {
702 "model_name": model_name,
703 "model_id": model_id,
704 }
705 )
706 return model_param_to_info
709def _aggregate_health_check_results(
710 model_param_to_info: dict,
711 healthy_endpoints: list,
712 unhealthy_endpoints: list,
713) -> dict:
714 """
715 Aggregate health check results per unique model.
717 Uses (model_id, model_name) as key, or (None, model_name) if model_id is None.
719 Args:
720 model_param_to_info: Mapping from model parameter to model info
721 healthy_endpoints: List of healthy endpoint results
722 unhealthy_endpoints: List of unhealthy endpoint results
724 Returns:
725 Dictionary mapping (model_id, model_name) to aggregated health check results
726 """
727 model_results: Final = {}
729 # Process healthy endpoints
730 for endpoint in healthy_endpoints:
731 model_param = endpoint.get("model")
732 if model_param and model_param in model_param_to_info:
733 for model_info in model_param_to_info[model_param]:
734 key = (model_info["model_id"], model_info["model_name"])
735 if key not in model_results:
736 model_results[key] = {
737 "model_name": model_info["model_name"],
738 "model_id": model_info["model_id"],
739 "healthy_count": 0,
740 "unhealthy_count": 0,
741 "error_message": None,
742 }
743 model_results[key]["healthy_count"] += 1
745 # Process unhealthy endpoints
746 for endpoint in unhealthy_endpoints:
747 model_param = endpoint.get("model")
748 error_message = endpoint.get("error")
749 if model_param and model_param in model_param_to_info:
750 for model_info in model_param_to_info[model_param]:
751 key = (model_info["model_id"], model_info["model_name"])
752 if key not in model_results:
753 model_results[key] = {
754 "model_name": model_info["model_name"],
755 "model_id": model_info["model_id"],
756 "healthy_count": 0,
757 "unhealthy_count": 0,
758 "error_message": None,
759 }
760 model_results[key]["unhealthy_count"] += 1
761 # Use the first error message encountered
762 if not model_results[key]["error_message"] and error_message:
763 model_results[key]["error_message"] = str(error_message)[:500]
765 return model_results
768class _AggregatedHealthResult(TypedDict):
769 """One entry of ``_aggregate_health_check_results``: a model's counts for this cycle."""
771 model_name: ReadOnly[str]
772 model_id: ReadOnly[str | None]
773 healthy_count: ReadOnly[int]
774 unhealthy_count: ReadOnly[int]
775 error_message: ReadOnly[str | None]
778def _new_health_status(result: _AggregatedHealthResult) -> str:
779 return "healthy" if result["healthy_count"] > 0 else "unhealthy"
782def _should_persist_health_check_result(
783 result: _AggregatedHealthResult, latest_checks_map: Mapping[str, LatestHealthCheckRow]
784) -> bool:
785 """
786 True when this result has to be written: no previous row, the status changed, or the
787 previous row is older than one hour (periodic refresh while the status is stable).
788 """
789 lookup_key: Final = result["model_id"] if result["model_id"] else result["model_name"]
790 last_check: Final = latest_checks_map.get(lookup_key)
791 if last_check is None or last_check.status != _new_health_status(result):
792 return True
793 time_since_last_check: Final = (datetime.now(timezone.utc) - last_check.checked_at).total_seconds()
794 return time_since_last_check >= 3600 # 1 hour threshold
797async def _save_health_check_results_if_changed(
798 prisma_client,
799 model_results: dict,
800 latest_checks_map: dict,
801 start_time: float,
802 checked_by: str | None = None,
803) -> bool:
804 """
805 Save health check results to database, but only if status changed or >1 hour since last save.
807 OPTIMIZATION: Only saves to database if the status has changed from the last saved check.
808 This dramatically reduces database writes when health status remains stable.
810 - Stable systems: ~1 write/hour per model (instead of 12 writes/hour with 5-min intervals)
811 - Status changes: Immediate write (no delay)
812 - Result: ~92% reduction in DB writes for stable systems, while maintaining real-time updates on changes
814 The writes are awaited rather than detached so the caller learns whether this cycle's
815 persistence completed.
817 Args:
818 prisma_client: Database client
819 model_results: Dictionary of aggregated health check results per model
820 latest_checks_map: Dictionary mapping model_id/model_name to latest health check
821 start_time: Start time of health check for calculating response time
822 checked_by: Identifier for who/what performed the check
824 Returns:
825 True when every row that needed writing was written (including when nothing needed
826 writing); False when any write failed.
827 """
828 to_write: Final = tuple(
829 result for result in model_results.values() if _should_persist_health_check_result(result, latest_checks_map)
830 )
831 writes: Final = tuple(
832 prisma_client.save_health_check_result(
833 model_name=result["model_name"],
834 model_id=result["model_id"],
835 status=_new_health_status(result),
836 healthy_count=result["healthy_count"],
837 unhealthy_count=result["unhealthy_count"],
838 error_message=result["error_message"],
839 response_time_ms=(time.time() - start_time) * 1000,
840 details=None,
841 checked_by=checked_by,
842 )
843 for result in to_write
844 )
845 rows: Final = await asyncio.gather(*writes)
846 return all(row is not None for row in rows)
849async def _save_background_health_checks_to_db(
850 prisma_client,
851 model_list: list,
852 healthy_endpoints: list,
853 unhealthy_endpoints: list,
854 start_time: float,
855 checked_by: str | None = None,
856) -> bool:
857 """
858 Save background health check results to database for each model.
860 Maps health check endpoints back to their original models to get model_name and model_id.
861 Aggregates results per unique model (by model_id if available, otherwise model_name).
863 OPTIMIZATION: Only saves to database if the status has changed from the last saved check.
864 This dramatically reduces database writes when health status remains stable.
866 Returns:
867 True when this cycle's persistence completed; False when it was skipped or any step
868 failed. Never raises: a database failure must not break the health check loop.
869 """
870 if prisma_client is None:
871 return False
873 try:
874 # Step 1: Build mapping from model parameter to model info
875 model_param_to_info: Final = _build_model_param_to_info_mapping(model_list)
877 # Step 2: Aggregate health check results per unique model
878 model_results: Final = _aggregate_health_check_results(
879 model_param_to_info,
880 healthy_endpoints,
881 unhealthy_endpoints,
882 )
884 # Step 3: Get latest health checks for all models in one query to compare status
885 latest_checks: Final = await query_latest_health_checks(prisma_client)
886 latest_checks_map: Final = {}
887 for check in latest_checks:
888 # Use model_id as primary key, fallback to model_name
889 key = check.model_id if check.model_id else check.model_name
890 if key not in latest_checks_map:
891 latest_checks_map[key] = check
893 # Step 4: Save aggregated results, but only if status changed
894 return await _save_health_check_results_if_changed(
895 prisma_client,
896 model_results,
897 latest_checks_map,
898 start_time,
899 checked_by,
900 )
901 except Exception as db_error:
902 verbose_proxy_logger.warning("Failed to save background health checks to database: %s", db_error)
903 # Continue execution - don't let database save failure break health checks
904 return False
907_PROXY_ADMIN_ROLES: Final = frozenset(
908 {
909 LitellmUserRoles.PROXY_ADMIN.value,
910 # View-only admins are operators (oncall, support); they need the
911 # routing fields (api_base, api_version) to diagnose health and tell
912 # which provider region a check is hitting. They cannot mutate config
913 # so granting them the read-only view is safe.
914 LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
915 }
916)
919def _is_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> bool:
920 """
921 Return True if the caller has a proxy-admin role (full or view-only).
923 user_role on UserAPIKeyAuth can be either a LitellmUserRoles enum or its
924 string value depending on how the auth path constructed the object, so we
925 compare against the raw value rather than the enum identity.
926 """
927 role: Final = user_api_key_dict.user_role
928 if role is None: 928 ↛ 929line 928 didn't jump to line 929 because the condition on line 928 was never true
929 return False
930 role_value: Final = role.value if hasattr(role, "value") else role
931 return role_value in _PROXY_ADMIN_ROLES
934def _strip_admin_only_fields_from_health_result(result: dict) -> dict:
935 """
936 Return a copy of the /health response with provider routing fields
937 (``ADMIN_ONLY_HEALTH_DISPLAY_PARAMS``) removed from each healthy/unhealthy
938 endpoint entry. Used to hide those fields from non-admin callers while
939 still showing them which deployments they own and whether each one is
940 healthy. Proxy admins receive the unmodified result.
941 """
942 out: Final = dict(result)
943 drop: Final = set(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS)
944 for key in ("healthy_endpoints", "unhealthy_endpoints"):
945 eps = out.get(key)
946 if isinstance(eps, list):
947 out[key] = [({k: v for k, v in ep.items() if k not in drop} if isinstance(ep, dict) else ep) for ep in eps]
948 return out
951def _health_accessible_model_names(
952 user_api_key_dict: UserAPIKeyAuth, llm_router: Router | None
953) -> frozenset[str] | None:
954 """Model names the caller may health-check, or None when the key is unrestricted."""
955 granted_models: Final = _resolve_key_models_for_auth_check(user_api_key_dict)
956 if not granted_models or SpecialModelNames.all_proxy_models.value in granted_models: 956 ↛ 958line 956 didn't jump to line 958 because the condition on line 956 was always true
957 return None
958 if llm_router is None:
959 return frozenset(granted_models)
960 return frozenset(
961 get_key_models(
962 user_api_key_dict=user_api_key_dict,
963 proxy_model_list=llm_router.get_model_names(team_id=user_api_key_dict.team_id),
964 model_access_groups=llm_router.get_model_access_groups(),
965 )
966 )
969def _caller_may_probe_deployment(
970 deployment: Mapping[str, object],
971 allowed_models: frozenset[str] | None,
972 llm_router: Router | None,
973 team_id: str | None,
974 caller_is_admin: bool,
975) -> bool:
976 """Same deployment visibility rule as routing: another team's deployment is never in scope, team-less callers included."""
977 if not caller_is_admin and not Router._deployment_usable_by_team(deployment, team_id):
978 return False
979 if allowed_models is None:
980 return True
981 if llm_router is None:
982 return deployment.get("model_name") in allowed_models
983 model: Final = dict(deployment)
984 return any(
985 llm_router.should_include_deployment(model_name=name, model=model, team_id=team_id) for name in allowed_models
986 )
989def _resolve_targeted_model_ids(
990 model_list: list, model: str | None, model_id: str | None, team_id: str | None
991) -> set | None:
992 """
993 Resolve a ``/health`` ``model`` / ``model_id`` query param to the set of
994 deployment IDs the response should be scoped to, mirroring the live-path
995 narrowing in ``perform_health_check()``: ``model_id`` wins when given and
996 matches ``model_info.id`` only; ``model`` targets the deployments a request
997 for that name from the caller would route to, else those whose
998 ``litellm_params.model`` provider string is that value (``deployments_targeted_by_name``).
1000 Callers pass an already-scoped list, so a ``model_id`` outside the
1001 caller's scope resolves to an empty set and never to the unvalidated id.
1002 Returns ``None`` when no targeting is requested.
1003 """
1004 if model_id: 1004 ↛ 1005line 1004 didn't jump to line 1005 because the condition on line 1004 was never true
1005 return {i for m in model_list if (i := (m.get("model_info") or {}).get("id")) == model_id}
1006 if not model:
1007 return None
1008 return {
1009 i
1010 for m in deployments_targeted_by_name(model_list, model, team_id)
1011 if (i := (m.get("model_info") or {}).get("id"))
1012 }
1015def _filter_health_check_results_by_model_ids(results: dict, allowed_model_ids: set) -> dict:
1016 """
1017 Restrict a cached background health-check result dict to endpoints whose
1018 model_id is in ``allowed_model_ids``.
1020 Endpoints without a model_id (e.g. CLI-model entries that predate the
1021 model_id wiring) are dropped conservatively — we cannot prove they belong
1022 to the caller, so they are excluded rather than leaked.
1024 Each retained endpoint is shallow-copied before being returned, so any
1025 downstream transform (e.g. _strip_admin_only_fields_from_health_result)
1026 cannot accidentally mutate the shared ``health_check_results`` cache.
1027 """
1028 healthy = [dict(ep) for ep in (results.get("healthy_endpoints") or []) if ep.get("model_id") in allowed_model_ids]
1029 unhealthy: Final = [
1030 dict(ep) for ep in (results.get("unhealthy_endpoints") or []) if ep.get("model_id") in allowed_model_ids
1031 ]
1032 return {
1033 "healthy_endpoints": healthy,
1034 "unhealthy_endpoints": unhealthy,
1035 "healthy_count": len(healthy),
1036 "unhealthy_count": len(unhealthy),
1037 }
1040async def _perform_health_check_and_save(
1041 model_list,
1042 target_model,
1043 cli_model,
1044 details,
1045 prisma_client,
1046 start_time,
1047 user_id,
1048 model_id=None,
1049 max_concurrency=None,
1050 **perform_health_check_extra,
1051):
1052 """Helper function to perform health check and save results to database"""
1053 healthy_endpoints, unhealthy_endpoints, _ = await perform_health_check(
1054 model_list=model_list,
1055 cli_model=cli_model,
1056 model=target_model,
1057 details=details,
1058 max_concurrency=max_concurrency,
1059 model_id=model_id,
1060 **perform_health_check_extra,
1061 )
1063 # Optionally save health check result to database (non-blocking)
1064 if prisma_client is not None: 1064 ↛ 1080line 1064 didn't jump to line 1080 because the condition on line 1064 was always true
1065 # For CLI model, use cli_model name; for router models, use target_model
1066 model_name_for_db: Final = cli_model if cli_model is not None else target_model
1067 if model_name_for_db is not None:
1068 asyncio.create_task(
1069 _save_health_check_to_db(
1070 prisma_client,
1071 model_name_for_db,
1072 healthy_endpoints,
1073 unhealthy_endpoints,
1074 start_time,
1075 user_id,
1076 model_id=model_id,
1077 )
1078 )
1080 return {
1081 "healthy_endpoints": healthy_endpoints,
1082 "unhealthy_endpoints": unhealthy_endpoints,
1083 "healthy_count": len(healthy_endpoints),
1084 "unhealthy_count": len(unhealthy_endpoints),
1085 }
1088def _health_endpoint_resolve_target_model_name(
1089 model: str | None,
1090 model_id: str | None,
1091 llm_router,
1092) -> str | None:
1093 """Map ``model_id`` to its deployment's ``model_name`` for live health checks.
1095 ``model_id`` wins over ``model``, so an id no deployment carries is a 404 even
1096 when it is paired with a known name.
1097 """
1098 if not model_id:
1099 return model
1100 if llm_router is None: 1100 ↛ 1101line 1100 didn't jump to line 1101 because the condition on line 1100 was never true
1101 raise HTTPException(
1102 status_code=status.HTTP_404_NOT_FOUND,
1103 detail={"error": f"Model with ID {model_id} not found"},
1104 )
1105 try:
1106 deployment: Final = llm_router.get_deployment(model_id=model_id)
1107 except Exception as e:
1108 verbose_proxy_logger.error("Error getting deployment for model_id %s: %s", model_id, e)
1109 raise HTTPException(
1110 status_code=status.HTTP_404_NOT_FOUND,
1111 detail={"error": f"Model with ID {model_id} not found"},
1112 ) from e
1113 if deployment is None: 1113 ↛ 1118line 1113 didn't jump to line 1118 because the condition on line 1113 was always true
1114 raise HTTPException(
1115 status_code=status.HTTP_404_NOT_FOUND,
1116 detail={"error": f"Model with ID {model_id} not found"},
1117 )
1118 return deployment.model_name
1121@router.get("/health", tags=["health"], dependencies=[Depends(user_api_key_auth)])
1122async def health_endpoint(
1123 response: Response,
1124 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1125 model: str | None = fastapi.Query(None, description="Specify the model name (optional)"),
1126 model_id: str | None = fastapi.Query(None, description="Specify the model ID (optional)"),
1127):
1128 """
1129 🚨 USE `/health/liveliness` to health check the proxy 🚨
1131 See more 👉 https://docs.litellm.ai/docs/proxy/health
1134 Check the health of all the endpoints in config.yaml
1136 To run health checks in the background, add this to config.yaml:
1137 ```
1138 general_settings:
1139 # ... other settings
1140 background_health_checks: True
1141 ```
1142 else, the health checks will be run on models when /health is called.
1144 To skip deployments that set ``model_info.disable_background_health_check: true``
1145 on ``GET /health`` as well as in the background loop, set
1146 ``general_settings.health_check_skip_disabled_background_models: true``.
1147 """
1148 import time
1150 from litellm.proxy.proxy_server import (
1151 general_settings,
1152 health_check_concurrency,
1153 health_check_details,
1154 health_check_results,
1155 llm_model_list,
1156 llm_router,
1157 prisma_client,
1158 use_background_health_checks,
1159 user_model,
1160 )
1162 _hc_filter: Final = health_check_filter_kwargs_from_general_settings(general_settings)
1163 start_time: Final = time.time()
1165 target_model: Final = _health_endpoint_resolve_target_model_name(model, model_id, llm_router)
1167 is_admin: Final = _is_proxy_admin(user_api_key_dict)
1168 model_specific_request: Final = bool(model or model_id)
1170 def _post_process(result: dict) -> dict:
1171 # api_base / api_version reveal which provider/region/internal host the
1172 # deployment talks to; only proxy admins receive them. Non-admin keys
1173 # still see model/model_id and the healthy/unhealthy status. We also
1174 # set a header so non-admin clients that previously parsed those
1175 # fields can detect the change programmatically.
1176 # When a caller asked about a specific model/model_id and zero
1177 # endpoints came back healthy, surface that as a 503 so monitoring
1178 # systems can rely on the HTTP status instead of having to parse the
1179 # body. The body shape is unchanged.
1180 if model_specific_request and result.get("healthy_count", 0) == 0:
1181 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE
1182 if is_admin: 1182 ↛ 1184line 1182 didn't jump to line 1184 because the condition on line 1182 was always true
1183 return result
1184 response.headers["Litellm-Health-Field-Notice"] = (
1185 f"{', '.join(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS)} are admin-only on this endpoint"
1186 )
1187 return _strip_admin_only_fields_from_health_result(result)
1189 try:
1190 if llm_model_list is None: 1190 ↛ 1192line 1190 didn't jump to line 1192 because the condition on line 1190 was never true
1191 # if no router set, check if user set a model using litellm --model ollama/llama2
1192 if user_model is not None:
1193 cli_result: Final = await _perform_health_check_and_save(
1194 model_list=[],
1195 target_model=None,
1196 cli_model=user_model,
1197 details=health_check_details,
1198 prisma_client=prisma_client,
1199 start_time=start_time,
1200 user_id=user_api_key_dict.user_id,
1201 model_id=None, # CLI model doesn't have model_id
1202 max_concurrency=health_check_concurrency,
1203 **_hc_filter,
1204 )
1205 return _post_process(cli_result)
1206 raise HTTPException(
1207 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
1208 detail={"error": "Model list not initialized"},
1209 )
1210 allowed_models: Final = _health_accessible_model_names(user_api_key_dict, llm_router)
1211 restrict_to_allowed_models: Final = not is_admin or allowed_models is not None
1212 _llm_model_list: Final = [
1213 m
1214 for m in copy.deepcopy(llm_model_list)
1215 if not restrict_to_allowed_models
1216 or _caller_may_probe_deployment(m, allowed_models, llm_router, user_api_key_dict.team_id, is_admin)
1217 ]
1218 targeted_ids: Final = _resolve_targeted_model_ids(_llm_model_list, model, model_id, user_api_key_dict.team_id)
1219 if restrict_to_allowed_models and targeted_ids is not None and not targeted_ids: 1219 ↛ 1220line 1219 didn't jump to line 1220 because the condition on line 1219 was never true
1220 raise HTTPException(
1221 status_code=status.HTTP_403_FORBIDDEN,
1222 detail={
1223 "error": f"key not allowed to health-check model_id {model_id}"
1224 if model_id
1225 else f"key not allowed to health-check model {model}"
1226 },
1227 )
1228 if use_background_health_checks: 1228 ↛ 1235line 1228 didn't jump to line 1235 because the condition on line 1228 was never true
1229 # The cached background result covers every model. When the
1230 # caller targets a specific model/model_id we have to narrow the
1231 # cache to that deployment before _post_process evaluates
1232 # healthy_count, otherwise an unhealthy "foo" combined with any
1233 # other healthy model would still report healthy_count > 0 and
1234 # the targeted-503 path would never fire.
1235 if restrict_to_allowed_models:
1236 allowed_model_ids: Final = {
1237 (m.get("model_info") or {}).get("id")
1238 for m in _llm_model_list
1239 if (m.get("model_info") or {}).get("id")
1240 }
1241 # _llm_model_list is already scoped to the caller's allowed
1242 # model_names above, so targeted_ids is implicitly the
1243 # intersection of "targeted" and "allowed."
1244 filter_ids: Final = targeted_ids if targeted_ids is not None else allowed_model_ids
1245 filtered: Final = _filter_health_check_results_by_model_ids(health_check_results, filter_ids)
1246 if targeted_ids is None and _llm_model_list and not allowed_model_ids:
1247 # Caller has accessible model_names but none of the
1248 # matching deployments expose a model_info.id, so the
1249 # cache filter (which keys on model_id) drops every
1250 # entry. Surface this both as a warning log and a
1251 # structured "warnings" field on the response so the
1252 # caller can distinguish "no deployments found" from
1253 # "deployments excluded due to missing model_info.id".
1254 verbose_proxy_logger.warning(
1255 "health_endpoint: scoped key %s has accessible models %s "
1256 "but none of the matching deployments carry a model_info.id; "
1257 "background health-check cache will return an empty result.",
1258 user_api_key_dict.user_id,
1259 list(user_api_key_dict.models),
1260 )
1261 filtered["warnings"] = [
1262 "Some accessible deployments are missing model_info.id "
1263 "and were excluded from this response. Ask a proxy admin "
1264 "to populate model_info.id for these models."
1265 ]
1266 return _post_process(filtered)
1267 if targeted_ids is not None:
1268 # Admin caller targeting a specific model: filter the cache
1269 # so the response (and the targeted-503 check) reflects only
1270 # that deployment, not the global aggregate.
1271 return _post_process(_filter_health_check_results_by_model_ids(health_check_results, targeted_ids))
1272 return _post_process(health_check_results)
1273 else:
1274 router_result: Final = await _perform_health_check_and_save(
1275 model_list=_llm_model_list,
1276 target_model=target_model,
1277 cli_model=None,
1278 details=health_check_details,
1279 prisma_client=prisma_client,
1280 start_time=start_time,
1281 user_id=user_api_key_dict.user_id,
1282 model_id=model_id,
1283 max_concurrency=health_check_concurrency,
1284 router=llm_router,
1285 team_id=user_api_key_dict.team_id,
1286 **_hc_filter,
1287 )
1288 return _post_process(router_result)
1289 except Exception as e:
1290 verbose_proxy_logger.error("litellm.proxy.proxy_server.py::health_endpoint(): Exception occured - %s", e)
1291 verbose_proxy_logger.debug(traceback.format_exc())
1292 raise e
1295@router.get("/health/history", tags=["health"], dependencies=[Depends(user_api_key_auth)])
1296async def health_check_history_endpoint(
1297 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1298 model: str | None = fastapi.Query(None, description="Filter by specific model name"),
1299 status_filter: str | None = fastapi.Query(None, description="Filter by status (healthy/unhealthy)"),
1300 limit: int = fastapi.Query(100, description="Number of records to return", ge=1, le=1000),
1301 offset: int = fastapi.Query(0, description="Number of records to skip", ge=0),
1302):
1303 """
1304 Get health check history for models
1306 Returns historical health check data with optional filtering.
1307 """
1308 prisma_client: Final = _check_prisma_client()
1310 try:
1311 history: Final = await prisma_client.get_health_check_history(
1312 model_name=model,
1313 limit=limit,
1314 offset=offset,
1315 status_filter=status_filter,
1316 )
1318 # Convert to dict format for JSON response using helper function
1319 history_data: Final = [_convert_health_check_to_dict(check) for check in history]
1321 return {
1322 "health_checks": history_data,
1323 "total_records": len(history_data),
1324 "limit": limit,
1325 "offset": offset,
1326 }
1327 except Exception as e:
1328 verbose_proxy_logger.error("Error getting health check history: %s", e)
1329 raise HTTPException(
1330 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
1331 detail={"error": f"Failed to retrieve health check history: {e}"},
1332 )
1335@router.get("/health/latest", tags=["health"], dependencies=[Depends(user_api_key_auth)])
1336async def latest_health_checks_endpoint(
1337 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1338):
1339 """
1340 Get the latest health check status for all models
1342 Returns the most recent health check result for each model.
1343 """
1344 prisma_client: Final = _check_prisma_client()
1346 try:
1347 latest_checks: Final = await prisma_client.get_all_latest_health_checks()
1349 # Convert to dict format for JSON response using helper function
1350 checks_data: Final = {
1351 (check.model_id if check.model_id else check.model_name): _convert_health_check_to_dict(check)
1352 for check in latest_checks
1353 }
1355 return {
1356 "latest_health_checks": checks_data,
1357 "total_models": len(checks_data),
1358 }
1359 except Exception as e:
1360 verbose_proxy_logger.error("Error getting latest health checks: %s", e)
1361 raise HTTPException(
1362 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
1363 detail={"error": f"Failed to retrieve latest health checks: {e}"},
1364 )
1367@router.get("/health/shared-status", tags=["health"], dependencies=[Depends(user_api_key_auth)])
1368async def shared_health_check_status_endpoint(
1369 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1370):
1371 """
1372 Get the status of shared health check coordination across pods.
1374 Returns information about Redis connectivity, lock status, and cache status.
1375 """
1376 from litellm.proxy.proxy_server import redis_usage_cache, use_shared_health_check
1378 if not use_shared_health_check: 1378 ↛ 1384line 1378 didn't jump to line 1384 because the condition on line 1378 was always true
1379 return {
1380 "shared_health_check_enabled": False,
1381 "message": "Shared health check is not enabled",
1382 }
1384 if redis_usage_cache is None:
1385 return {
1386 "shared_health_check_enabled": True,
1387 "redis_available": False,
1388 "message": "Redis is not configured",
1389 }
1391 try:
1392 from litellm.proxy.health_check_utils.shared_health_check_manager import (
1393 SharedHealthCheckManager,
1394 )
1396 shared_health_manager: Final = SharedHealthCheckManager(
1397 redis_cache=redis_usage_cache,
1398 )
1400 health_status: Final = await shared_health_manager.get_health_check_status()
1401 return {"shared_health_check_enabled": True, "status": health_status}
1402 except Exception as e:
1403 verbose_proxy_logger.error("Error getting shared health check status: %s", e)
1404 raise HTTPException(
1405 status_code=fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR,
1406 detail={"error": f"Failed to retrieve shared health check status: {e}"},
1407 )
1410def _read_license_data() -> dict[str, Any] | None:
1411 from litellm.proxy.proxy_server import _license_check, premium_user_data
1413 license_data: EnterpriseLicenseData | None = premium_user_data or _license_check.airgapped_license_data
1415 if ( 1415 ↛ 1420line 1415 didn't jump to line 1420 because the condition on line 1415 was never true
1416 license_data is None
1417 and getattr(_license_check, "license_str", None)
1418 and getattr(_license_check, "public_key", None)
1419 ):
1420 try:
1421 verification_result: Final = _license_check.verify_license_without_api_request(
1422 public_key=_license_check.public_key,
1423 license_key=_license_check.license_str,
1424 )
1425 if verification_result is True:
1426 license_data = _license_check.airgapped_license_data
1427 except Exception:
1428 pass
1430 if license_data is None: 1430 ↛ 1432line 1430 didn't jump to line 1432 because the condition on line 1430 was always true
1431 return None
1432 return cast(dict[str, Any], license_data)
1435def _read_allowed_features(license_data: dict[str, Any]) -> list:
1436 raw_allowed_features: Final = license_data.get("allowed_features")
1437 if isinstance(raw_allowed_features, list):
1438 return list(raw_allowed_features)
1439 if raw_allowed_features is None:
1440 return []
1441 return [raw_allowed_features]
1444@router.get(
1445 "/health/license",
1446 tags=["health"],
1447 dependencies=[Depends(user_api_key_auth)],
1448)
1449async def health_license_endpoint(
1450 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
1451):
1452 """Return metadata about the configured LiteLLM license without exposing the key."""
1453 from litellm.proxy.proxy_server import _license_check, premium_user
1455 license_data: Final = _read_license_data()
1456 has_license: Final = bool(getattr(_license_check, "license_str", None))
1457 license_type: Final = "enterprise" if premium_user else "community"
1459 if license_data is None: 1459 ↛ 1471line 1459 didn't jump to line 1471 because the condition on line 1459 was always true
1460 return {
1461 "has_license": has_license,
1462 "license_type": license_type,
1463 "expiration_date": None,
1464 "allowed_features": [],
1465 "limits": {
1466 "max_users": None,
1467 "max_teams": None,
1468 },
1469 }
1471 expiration_date: Final = license_data.get("expiration_date")
1472 max_users: Final = license_data.get("max_users")
1473 max_teams: Final = license_data.get("max_teams")
1475 return {
1476 "has_license": has_license,
1477 "license_type": license_type,
1478 "expiration_date": expiration_date,
1479 "allowed_features": _read_allowed_features(license_data),
1480 "limits": {
1481 "max_users": max_users,
1482 "max_teams": max_teams,
1483 },
1484 }
1487class DBHealthCache(TypedDict):
1488 status: str
1489 last_updated: datetime
1492db_health_cache: DBHealthCache = {"status": "unknown", "last_updated": datetime.now()}
1494# Bounds each DB round-trip on the probe path so a hung connection during a
1495# failover cannot make the probe fail by timeout (k8s default timeoutSeconds: 5).
1496DB_READINESS_CHECK_TIMEOUT_SECONDS: Final = 2.0
1497# One deadline for the whole probe-path DB check (initial check + reconnect +
1498# re-check, including reconnect lock waits), kept under timeoutSeconds: 5.
1499DB_READINESS_PROBE_DEADLINE_SECONDS: Final = 4.0
1502async def _db_health_readiness_check() -> DBHealthCache:
1503 try:
1504 return await asyncio.wait_for(
1505 _db_health_readiness_check_unbounded(),
1506 timeout=DB_READINESS_PROBE_DEADLINE_SECONDS,
1507 )
1508 except asyncio.TimeoutError:
1509 return {"status": "disconnected", "last_updated": db_health_cache["last_updated"]}
1512async def _db_health_readiness_check_unbounded() -> DBHealthCache:
1513 from litellm.proxy.proxy_server import prisma_client
1515 global db_health_cache
1517 try:
1518 time_diff: Final = datetime.now() - db_health_cache["last_updated"]
1519 if db_health_cache["status"] == "connected" and time_diff < timedelta(seconds=15):
1520 return db_health_cache
1522 if prisma_client is None: 1522 ↛ 1523line 1522 didn't jump to line 1523 because the condition on line 1522 was never true
1523 db_health_cache = {"status": "disconnected", "last_updated": datetime.now()}
1524 return db_health_cache
1526 await asyncio.wait_for(prisma_client.health_check(), timeout=DB_READINESS_CHECK_TIMEOUT_SECONDS)
1527 db_health_cache = {"status": "connected", "last_updated": datetime.now()}
1528 return db_health_cache
1529 except Exception as e:
1530 db_health_cache = {"status": "disconnected", "last_updated": datetime.now()}
1531 if PrismaDBExceptionHandler.is_database_transport_error(e):
1532 try:
1533 verbose_proxy_logger.warning("_db_health_readiness_check: health_check failed, attempting reconnect")
1534 await prisma_client.attempt_db_reconnect(
1535 reason="health_readiness_check",
1536 timeout_seconds=DB_READINESS_CHECK_TIMEOUT_SECONDS,
1537 lock_timeout_seconds=DB_READINESS_CHECK_TIMEOUT_SECONDS,
1538 )
1539 await asyncio.wait_for(
1540 prisma_client.health_check(),
1541 timeout=DB_READINESS_CHECK_TIMEOUT_SECONDS,
1542 )
1543 verbose_proxy_logger.info("_db_health_readiness_check: reconnect succeeded")
1544 db_health_cache = {
1545 "status": "connected",
1546 "last_updated": datetime.now(),
1547 }
1548 return db_health_cache
1549 except Exception:
1550 verbose_proxy_logger.error("_db_health_readiness_check: reconnect failed")
1551 return db_health_cache
1554@router.get(
1555 "/settings",
1556 tags=["health"],
1557 dependencies=[Depends(user_api_key_auth)],
1558)
1559@router.get(
1560 "/active/callbacks",
1561 tags=["health"],
1562 dependencies=[Depends(user_api_key_auth)],
1563)
1564async def active_callbacks():
1565 """
1566 Returns a list of litellm level settings
1568 This is useful for debugging and ensuring the proxy server is configured correctly.
1570 Response schema:
1571 ```
1572 {
1573 "alerting": _alerting,
1574 "litellm.callbacks": litellm_callbacks,
1575 "litellm.input_callback": litellm_input_callbacks,
1576 "litellm.failure_callback": litellm_failure_callbacks,
1577 "litellm.success_callback": litellm_success_callbacks,
1578 "litellm._async_success_callback": litellm_async_success_callbacks,
1579 "litellm._async_failure_callback": litellm_async_failure_callbacks,
1580 "litellm._async_input_callback": litellm_async_input_callbacks,
1581 "all_litellm_callbacks": all_litellm_callbacks,
1582 "num_callbacks": len(all_litellm_callbacks),
1583 "num_alerting": _num_alerting,
1584 "litellm.request_timeout": litellm.request_timeout,
1585 }
1586 ```
1587 """
1589 from litellm.proxy.proxy_server import general_settings, proxy_logging_obj
1591 _alerting: Final = str(general_settings.get("alerting"))
1592 # get success callbacks
1594 litellm_callbacks: Final = [str(x) for x in litellm.callbacks]
1595 litellm_input_callbacks: Final = [str(x) for x in litellm.input_callback]
1596 litellm_failure_callbacks: Final = [str(x) for x in litellm.failure_callback]
1597 litellm_success_callbacks: Final = [str(x) for x in litellm.success_callback]
1598 litellm_async_success_callbacks: Final = [str(x) for x in litellm._async_success_callback]
1599 litellm_async_failure_callbacks: Final = [str(x) for x in litellm._async_failure_callback]
1600 litellm_async_input_callbacks: Final = [str(x) for x in litellm._async_input_callback]
1602 all_litellm_callbacks: Final = (
1603 litellm_callbacks
1604 + litellm_input_callbacks
1605 + litellm_failure_callbacks
1606 + litellm_success_callbacks
1607 + litellm_async_success_callbacks
1608 + litellm_async_failure_callbacks
1609 + litellm_async_input_callbacks
1610 )
1612 alerting: Final = proxy_logging_obj.alerting
1613 _num_alerting = 0
1614 if alerting and isinstance(alerting, list): 1614 ↛ 1615line 1614 didn't jump to line 1615 because the condition on line 1614 was never true
1615 _num_alerting = len(alerting)
1617 return {
1618 "alerting": _alerting,
1619 "litellm.callbacks": litellm_callbacks,
1620 "litellm.input_callback": litellm_input_callbacks,
1621 "litellm.failure_callback": litellm_failure_callbacks,
1622 "litellm.success_callback": litellm_success_callbacks,
1623 "litellm._async_success_callback": litellm_async_success_callbacks,
1624 "litellm._async_failure_callback": litellm_async_failure_callbacks,
1625 "litellm._async_input_callback": litellm_async_input_callbacks,
1626 "all_litellm_callbacks": all_litellm_callbacks,
1627 "num_callbacks": len(all_litellm_callbacks),
1628 "num_alerting": _num_alerting,
1629 "litellm.request_timeout": litellm.request_timeout,
1630 }
1633def callback_name(callback):
1634 if isinstance(callback, str):
1635 return callback
1637 try:
1638 return callback.__name__
1639 except AttributeError:
1640 try:
1641 return callback.__class__.__name__
1642 except AttributeError:
1643 return str(callback)
1646DISABLE_NO_REDIS_WARNING_ENV_VAR: Final = "LITELLM_DISABLE_NO_REDIS_WARNING"
1649async def _show_no_redis_warning() -> bool:
1650 """
1651 Whether the UI should warn that no Redis is configured.
1653 Redis is what makes rate limits, budgets, router state, and cache
1654 invalidation consistent across workers, so a proxy running without it is
1655 only safe as a single worker. Both places a Redis can land count: the
1656 coordination cache (from a Redis response cache, general_settings.
1657 coordination_redis, or the REDIS_* env fallback) and the router's own
1658 Redis (router_settings.redis_host), which backs cooldowns and usage-based
1659 routing on its own. A deployment whose worker-heartbeat census proves it
1660 is exactly one worker needs no cross-worker coordination, so it never
1661 warns; when the census is unavailable or shows more than one worker, the
1662 warning stands unless LITELLM_DISABLE_NO_REDIS_WARNING=true silences it.
1663 """
1664 from litellm.proxy.proxy_server import llm_router, prisma_client, redis_usage_cache
1666 if redis_usage_cache is not None: 1666 ↛ 1667line 1666 didn't jump to line 1667 because the condition on line 1666 was never true
1667 return False
1668 if llm_router is not None and llm_router.cache.redis_cache is not None: 1668 ↛ 1669line 1668 didn't jump to line 1669 because the condition on line 1668 was never true
1669 return False
1670 if get_secret_bool(DISABLE_NO_REDIS_WARNING_ENV_VAR, False) is True: 1670 ↛ 1671line 1670 didn't jump to line 1671 because the condition on line 1670 was never true
1671 return False
1672 if prisma_client is None: 1672 ↛ 1673line 1672 didn't jump to line 1673 because the condition on line 1672 was never true
1673 return True
1674 return await count_live_proxy_workers(prisma_client) != 1
1677def _show_env_credential_login_warning() -> bool:
1678 from litellm.proxy.auth.login_utils import is_env_credential_login_enabled
1679 from litellm.proxy.proxy_server import general_settings
1681 return is_env_credential_login_enabled(general_settings)
1684async def _get_health_readiness_details(
1685 response: Response | None = None,
1686) -> dict[str, Any]:
1687 """
1688 Detailed health payload for authenticated diagnostics.
1689 """
1690 from litellm.proxy.proxy_server import prisma_client, version
1692 try:
1693 # get success callback
1694 success_callback_names = []
1696 try:
1697 # this was returning a JSON of the values in some of the callbacks
1698 # all we need is the callback name, hence we do str(callback)
1699 success_callback_names = [callback_name(x) for x in litellm.success_callback]
1700 except AttributeError:
1701 # don't let this block the /health/readiness response, if we can't convert to str -> return litellm.success_callback
1702 success_callback_names = litellm.success_callback
1704 # check Cache
1705 cache_type: Any = None
1706 if litellm.cache is not None: 1706 ↛ 1722line 1706 didn't jump to line 1722 because the condition on line 1706 was always true
1707 from litellm.caching.caching import RedisSemanticCache
1709 cache_type = litellm.cache.type
1711 if isinstance(litellm.cache.cache, RedisSemanticCache): 1711 ↛ 1715line 1711 didn't jump to line 1715 because the condition on line 1711 was never true
1712 # ping the cache
1713 # TODO: @ishaan-jaff - we should probably not ping the cache on every /health/readiness check
1714 index_info: Any
1715 try:
1716 index_info = await litellm.cache.cache._index_info()
1717 except Exception as e:
1718 index_info = "index does not exist - error: " + str(e)
1719 cache_type = {"type": cache_type, "index_info": index_info}
1721 # check log level
1722 log_level_name: Final = logging.getLevelName(verbose_logger.getEffectiveLevel())
1723 is_detailed_debug: Final = verbose_logger.isEnabledFor(logging.DEBUG)
1724 show_no_redis_warning: Final = await _show_no_redis_warning()
1725 show_env_credential_login_warning: Final = _show_env_credential_login_warning()
1727 # check DB
1728 if prisma_client is not None: # if db passed in, check if it's connected 1728 ↛ 1756line 1728 didn't jump to line 1756 because the condition on line 1728 was always true
1729 db_status: Final = _readiness_db_status(await _db_health_readiness_check())
1730 # A configured DB that is not reachable means the worker cannot
1731 # serve requests that depend on persisted state (keys, budgets,
1732 # spend logs). Return 503 so orchestrators take this pod out of
1733 # rotation; "Not connected" (no DB configured at all) stays 200.
1734 # With allow_requests_on_db_unavailable the proxy keeps serving
1735 # during a DB outage, so the pod must stay in rotation (200) and
1736 # report the DB state through the body instead.
1737 if ( 1737 ↛ 1742line 1737 didn't jump to line 1742 because the condition on line 1737 was never true
1738 response is not None
1739 and db_status != "connected"
1740 and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable()
1741 ):
1742 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE
1743 return {
1744 "status": "healthy",
1745 "db": db_status,
1746 "cache": cache_type,
1747 "litellm_version": version,
1748 "success_callbacks": success_callback_names,
1749 "use_aiohttp_transport": AsyncHTTPHandler._should_use_aiohttp_transport(),
1750 "log_level": log_level_name,
1751 "is_detailed_debug": is_detailed_debug,
1752 "show_no_redis_warning": show_no_redis_warning,
1753 "show_env_credential_login_warning": show_env_credential_login_warning,
1754 }
1755 else:
1756 return {
1757 "status": "healthy",
1758 "db": "Not connected",
1759 "cache": cache_type,
1760 "litellm_version": version,
1761 "success_callbacks": success_callback_names,
1762 "use_aiohttp_transport": AsyncHTTPHandler._should_use_aiohttp_transport(),
1763 "log_level": log_level_name,
1764 "is_detailed_debug": is_detailed_debug,
1765 "show_no_redis_warning": show_no_redis_warning,
1766 "show_env_credential_login_warning": show_env_credential_login_warning,
1767 }
1768 except Exception as e:
1769 raise HTTPException(status_code=503, detail=f"Service Unhealthy ({e})")
1772def _allow_public_health_readiness_details() -> bool:
1773 from litellm.proxy.proxy_server import general_settings
1775 return general_settings.get("allow_public_health_readiness_details") is True
1778def _drain_endpoint_enabled() -> bool:
1779 from litellm.proxy.proxy_server import general_settings
1781 return general_settings.get("enable_drain_endpoint") is True
1784def _drain_endpoint_token() -> str | None:
1785 """
1786 Shared secret required on the X-Drain-Token header to call /health/drain.
1788 Falls back to the ``DRAIN_ENDPOINT_TOKEN`` env var when unset in
1789 general_settings so the kubelet preStop hook can supply it via
1790 ``valueFrom.secretKeyRef`` without a config reload.
1791 """
1792 from litellm.proxy.proxy_server import general_settings
1794 token: Final = general_settings.get("drain_endpoint_token")
1795 if isinstance(token, str) and token:
1796 return token
1797 env_token: Final = os.getenv("DRAIN_ENDPOINT_TOKEN")
1798 if env_token:
1799 return env_token
1800 return None
1803def _authorize_drain_request(request: Request) -> None:
1804 """
1805 Reject /health/drain calls that don't carry the configured X-Drain-Token.
1807 When no token is configured the endpoint is treated as already opted-in
1808 (the ``enable_drain_endpoint`` flag is the only gate). Comparison uses
1809 ``secrets.compare_digest`` to avoid timing leaks.
1810 """
1811 expected: Final = _drain_endpoint_token()
1812 if expected is None:
1813 return
1814 supplied: Final = request.headers.get("x-drain-token") or ""
1815 if not secrets.compare_digest(supplied, expected):
1816 raise HTTPException(
1817 status_code=status.HTTP_401_UNAUTHORIZED,
1818 detail="Invalid or missing X-Drain-Token",
1819 )
1822def _readiness_db_status(db_health_status: DBHealthCache) -> str:
1823 """A pod whose pre-request lookups hit their deadline inside the stall window
1824 reports "stalled" even though the ping succeeds: the ping is a fresh
1825 connection, the stalled lookups are the ones requests actually wait on."""
1826 if db_health_status["status"] != "connected": 1826 ↛ 1827line 1826 didn't jump to line 1827 because the condition on line 1826 was never true
1827 return db_health_status["status"]
1828 if db_lookup_stall_tracker.stalled_within(PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS): 1828 ↛ 1829line 1828 didn't jump to line 1829 because the condition on line 1828 was never true
1829 return "stalled"
1830 return "connected"
1833async def _resolve_public_readiness_db(response: Response) -> str:
1834 """
1835 Return the db status string for the public probe and flip the response to
1836 503 when a configured DB is unreachable or stalled. Mirrors the legacy values:
1837 "Not connected" (no DB configured), "connected", "disconnected", plus "stalled".
1838 """
1839 from litellm.proxy.proxy_server import prisma_client
1841 if prisma_client is None: 1841 ↛ 1842line 1841 didn't jump to line 1842 because the condition on line 1841 was never true
1842 return "Not connected"
1844 db_status: Final = _readiness_db_status(await _db_health_readiness_check())
1845 if db_status != "connected" and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable(): 1845 ↛ 1846line 1845 didn't jump to line 1846 because the condition on line 1845 was never true
1846 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE
1847 return db_status
1850@router.get(
1851 "/health/readiness",
1852 tags=["health"],
1853)
1854async def health_readiness(response: Response):
1855 """
1856 Public readiness probe. Returns a low-detail payload safe to expose to
1857 unauthenticated load balancers — `status` plus `db` so orchestrators and
1858 external probes can distinguish "healthy" from "DB unreachable" without a
1859 credential. Admins can opt into the legacy detailed payload with
1860 general_settings.allow_public_health_readiness_details.
1861 """
1862 if GracefulShutdownManager.is_shutting_down(): 1862 ↛ 1863line 1862 didn't jump to line 1863 because the condition on line 1862 was never true
1863 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE
1864 return {"status": "shutting_down"}
1866 if _allow_public_health_readiness_details(): 1866 ↛ 1867line 1866 didn't jump to line 1867 because the condition on line 1866 was never true
1867 return await _get_health_readiness_details(response=response)
1869 db_status: Final = await _resolve_public_readiness_db(response=response)
1870 return {"status": "healthy", "db": db_status}
1873@router.get(
1874 "/health/readiness/details",
1875 tags=["health"],
1876 dependencies=[Depends(user_api_key_auth)],
1877)
1878async def health_readiness_details(response: Response):
1879 """
1880 Authenticated readiness diagnostics with DB/cache/callback metadata.
1881 """
1882 return await _get_health_readiness_details(response=response)
1885@router.get(
1886 "/health/backlog",
1887 tags=["health"],
1888 dependencies=[Depends(user_api_key_auth)],
1889)
1890async def health_backlog():
1891 """
1892 Returns the number of HTTP requests currently in-flight on this uvicorn worker.
1894 Use this to measure per-pod queue depth. A high value means the worker is
1895 processing many concurrent requests — requests arriving now will have to wait
1896 for the event loop to get to them, adding latency before LiteLLM even starts
1897 its own timer.
1898 """
1899 stats: Final = get_admission_control_stats()
1900 response: Final[_HealthBacklogResponse] = {
1901 "in_flight_requests": get_in_flight_requests(),
1902 "admitted_requests": stats.admitted,
1903 "queued_requests": stats.queued,
1904 "rejected_requests": stats.rejected_total,
1905 }
1906 return response
1909@router.get(
1910 "/health/drain",
1911 tags=["health"],
1912)
1913async def health_drain(request: Request):
1914 """
1915 Graceful-drain probe for Kubernetes ``preStop`` hooks.
1917 Disabled by default and returns 404 unless ``general_settings`` sets
1918 ``enable_drain_endpoint: true``. Calling it flips a process-wide
1919 shutting-down flag, so a successful call permanently takes the worker out
1920 of rotation until the pod restarts.
1922 Because the kubelet calls preStop hooks without proxy credentials, the
1923 endpoint does not require ``user_api_key_auth``. To prevent any
1924 pod-reachable caller from triggering shutdown, set
1925 ``general_settings.drain_endpoint_token`` (or the ``DRAIN_ENDPOINT_TOKEN``
1926 env var) and supply the same value on the ``X-Drain-Token`` header from
1927 the preStop hook. Calls without the header (or with a wrong value) get a
1928 401 and have no side effect.
1930 When enabled, it marks the worker as shutting down (so /health/readiness
1931 and /health/liveliness immediately start returning 503, removing the pod
1932 from service) and blocks until the in-flight request counter drains to
1933 zero or ``GRACEFUL_SHUTDOWN_TIMEOUT`` elapses. Unlike a fixed ``sleep``,
1934 this returns as soon as real in-flight work is done.
1936 Wire it up as:
1938 ```yaml
1939 lifecycle:
1940 preStop:
1941 httpGet:
1942 path: /health/drain
1943 port: 4000
1944 httpHeaders:
1945 - name: X-Drain-Token
1946 value: <same value as drain_endpoint_token>
1947 ```
1948 """
1949 if not _drain_endpoint_enabled(): 1949 ↛ 1951line 1949 didn't jump to line 1951 because the condition on line 1949 was always true
1950 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail="Not Found")
1951 _authorize_drain_request(request)
1952 GracefulShutdownManager.start_shutdown()
1953 drained: Final = await GracefulShutdownManager.wait_for_drain(exclude_self=True)
1954 return {"status": "drained", "drained_requests": drained}
1957@router.get(
1958 "/health/liveliness", # Historical LiteLLM name; doesn't match k8s terminology but kept for backwards compatibility
1959 tags=["health"],
1960)
1961@router.get(
1962 "/health/liveness", # Kubernetes has "liveness" probes (https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#define-a-liveness-command)
1963 tags=["health"],
1964)
1965async def health_liveliness(response: Response):
1966 """
1967 Unprotected endpoint for checking if worker is alive.
1969 Returns 503 once graceful shutdown has begun so Kubernetes stops counting
1970 the draining pod as live and terminates it on schedule.
1971 """
1972 if GracefulShutdownManager.is_shutting_down(): 1972 ↛ 1973line 1972 didn't jump to line 1973 because the condition on line 1972 was never true
1973 response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE
1974 return {"status": "shutting_down"}
1975 return "I'm alive!"
1978@router.options(
1979 "/health/readiness",
1980 tags=["health"],
1981)
1982async def health_readiness_options():
1983 """
1984 Options endpoint for health/readiness check.
1985 """
1986 response_headers: Final = {
1987 "Allow": "GET, OPTIONS",
1988 "Access-Control-Allow-Methods": "GET, OPTIONS",
1989 "Access-Control-Allow-Headers": "*",
1990 }
1991 return Response(headers=response_headers, status_code=200)
1994@router.options(
1995 "/health/liveliness",
1996 tags=["health"],
1997)
1998@router.options(
1999 "/health/liveness", # Kubernetes has "liveness" probes (https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#define-a-liveness-command)
2000 tags=["health"],
2001)
2002async def health_liveliness_options():
2003 """
2004 Options endpoint for health/liveliness check.
2005 """
2006 response_headers: Final = {
2007 "Allow": "GET, OPTIONS",
2008 "Access-Control-Allow-Methods": "GET, OPTIONS",
2009 "Access-Control-Allow-Headers": "*",
2010 }
2011 return Response(headers=response_headers, status_code=200)
2014@router.post(
2015 "/health/test_connection",
2016 tags=["health"],
2017 dependencies=[Depends(user_api_key_auth)],
2018)
2019async def test_model_connection(
2020 request: Request,
2021 mode: Literal[
2022 "chat",
2023 "completion",
2024 "embedding",
2025 "audio_speech",
2026 "audio_transcription",
2027 "image_generation",
2028 "image_edit",
2029 "video_generation",
2030 "batch",
2031 "rerank",
2032 "realtime",
2033 "responses",
2034 "ocr",
2035 ]
2036 | None = fastapi.Body(
2037 None,
2038 description="The mode to test the model with. If not provided, auto-detected from model capabilities.",
2039 ),
2040 litellm_params: dict = fastapi.Body(
2041 None,
2042 description="Parameters for litellm.completion, litellm.embedding for the health check",
2043 ),
2044 model_info: dict = fastapi.Body(
2045 None,
2046 description="Model info for the health check",
2047 ),
2048 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
2049):
2050 """
2051 Test a direct connection to a specific model.
2053 This endpoint allows you to verify if your proxy can successfully connect to a specific model.
2054 It's useful for troubleshooting model connectivity issues without going through the full proxy routing.
2056 Example:
2057 ```bash
2058 # If model is configured in proxy_config.yaml, you only need to specify the model name:
2059 curl -X POST 'http://localhost:4000/health/test_connection' \\
2060 -H 'Authorization: Bearer sk-1234' \\
2061 -H 'Content-Type: application/json' \\
2062 -d '{
2063 "litellm_params": {
2064 "model": "gpt-4o"
2065 },
2066 "mode": "chat"
2067 }'
2069 # The endpoint will automatically use api_key, api_base, etc. from proxy_config.yaml
2071 # You can also override specific params or test with custom credentials:
2072 curl -X POST 'http://localhost:4000/health/test_connection' \\
2073 -H 'Authorization: Bearer sk-1234' \\
2074 -H 'Content-Type: application/json' \\
2075 -d '{
2076 "litellm_params": {
2077 "model": "azure/gpt-4o",
2078 "api_key": "os.environ/AZURE_OPENAI_API_KEY",
2079 "api_base": "os.environ/AZURE_OPENAI_ENDPOINT",
2080 "api_version": "2024-10-21"
2081 },
2082 "mode": "chat"
2083 }'
2084 ```
2086 Note:
2087 - If the model is configured in proxy_config.yaml, credentials (api_key, api_base, etc.)
2088 will be automatically loaded from the config (with resolved environment variables).
2089 - A request naming a stored credential (`litellm_credential_name`) that the configuration
2090 does not name is probed with that credential instead, and inherits no credentials
2091 from the configuration its model string happened to match.
2092 - You can override specific params by including them in the request.
2093 - You can use `os.environ/VARIABLE_NAME` syntax to reference environment variables,
2094 which will be resolved automatically (same as in proxy_config.yaml).
2096 Returns:
2097 dict: A dictionary containing the health check result with either success information or error details.
2098 """
2099 from litellm.proxy._types import CommonProxyErrors
2100 from litellm.proxy.management_endpoints.model_management_endpoints import (
2101 ModelManagementAuthChecks,
2102 )
2103 from litellm.proxy.proxy_server import (
2104 general_settings,
2105 llm_router,
2106 premium_user,
2107 prisma_client,
2108 )
2109 from litellm.types.router import Deployment, LiteLLM_Params
2111 try:
2112 if prisma_client is None: 2112 ↛ 2113line 2112 didn't jump to line 2113 because the condition on line 2112 was never true
2113 raise HTTPException(
2114 status_code=500,
2115 detail={"error": CommonProxyErrors.db_not_connected_error.value},
2116 )
2118 # Get model name from litellm_params
2119 request_litellm_params: Final = litellm_params or {}
2120 # Reject request-supplied os.environ/ references. Config values are
2121 # already resolved before reaching this endpoint; any remaining
2122 # reference must have come from the request body.
2123 _reject_os_environ_references(request_litellm_params)
2124 if model_info:
2125 _reject_os_environ_references(model_info)
2126 model_name: Final = request_litellm_params.get("model")
2128 # Look up model configuration from router if model name is provided
2129 # This gets the litellm_params from proxy config (with resolved env vars)
2130 config_litellm_params: dict = {}
2131 loaded_model_info: dict | None = None
2132 if llm_router is not None: 2132 ↛ 2179line 2132 didn't jump to line 2179 because the condition on line 2132 was always true
2133 # Prefer disambiguation by deployment id (`model_info.id`) when
2134 # the caller supplies it. This is required when multiple
2135 # deployments share a `model_name` (e.g. wildcard `openai/*`
2136 # with multiple `api_base` values for failover): the UI's
2137 # "Test Connection" button targets a specific row, and that
2138 # row's id is the only thing that uniquely identifies which
2139 # deployment to probe. Without this, all duplicates collapse
2140 # onto `deployments[0]`.
2141 request_model_info: Final = model_info or {}
2142 request_model_id: Final = request_model_info.get("id")
2143 try:
2144 deployment_by_id = None
2145 if request_model_id: 2145 ↛ 2146line 2145 didn't jump to line 2146 because the condition on line 2145 was never true
2146 deployment_by_id = llm_router.get_deployment(model_id=request_model_id)
2148 if deployment_by_id is not None: 2148 ↛ 2149line 2148 didn't jump to line 2149 because the condition on line 2148 was never true
2149 config_litellm_params = deployment_by_id.litellm_params.model_dump(exclude_none=True)
2150 loaded_model_info = deployment_by_id.model_info.model_dump(exclude_none=True)
2151 elif model_name: 2151 ↛ 2155line 2151 didn't jump to line 2155 because the condition on line 2151 was never true
2152 # Fall back to model_name lookup for callers (e.g. the
2153 # "Add Model" wizard, or curl) that don't supply an id.
2154 # First try to find by proxy model_name (e.g., "gpt-4o")
2155 deployments = llm_router.get_model_list(model_name=model_name)
2157 # If not found, try to find by litellm model name
2158 # (e.g., "azure/gpt-4o")
2159 if not deployments or len(deployments) == 0:
2160 all_deployments: Final = llm_router.get_model_list(model_name=None)
2161 if all_deployments:
2162 for deployment in all_deployments:
2163 if deployment.get("litellm_params", {}).get("model") == model_name:
2164 deployments = [deployment]
2165 break
2167 if deployments and len(deployments) > 0:
2168 # Use the first deployment's litellm_params as base
2169 # config. These already have resolved environment
2170 # variables from proxy config.
2171 config_litellm_params = dict(deployments[0].get("litellm_params", {}))
2172 loaded_model_info = dict(deployments[0].get("model_info") or {})
2173 except Exception as e:
2174 verbose_proxy_logger.debug(
2175 "Could not find model %s in router: %s. Proceeding with request params only.", model_name, e
2176 )
2178 # Merge: config params (from proxy config) as base, request params override
2179 litellm_params = {
2180 **_config_base_for_health_check(
2181 config_litellm_params,
2182 request_litellm_params,
2183 allow_client_side_credentials=general_settings.get("allow_client_side_credentials") is True,
2184 ),
2185 **request_litellm_params,
2186 }
2188 resolved_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
2189 litellm_params = _update_litellm_params_for_health_check(
2190 model_info=resolved_model_info or {},
2191 litellm_params=litellm_params,
2192 )
2194 ## Auth check, on the final probe params so health_check_params cannot retarget it afterwards
2195 await ModelManagementAuthChecks.can_user_make_model_call(
2196 model_params=Deployment(
2197 model_name="test_model",
2198 litellm_params=LiteLLM_Params(**litellm_params),
2199 model_info=resolved_model_info,
2200 ),
2201 user_api_key_dict=user_api_key_dict,
2202 prisma_client=prisma_client,
2203 premium_user=premium_user,
2204 )
2205 mode = mode or litellm_params.pop("mode", None)
2207 result: Final = await run_with_timeout(
2208 litellm.ahealth_check(
2209 model_params=litellm_params,
2210 mode=mode,
2211 prompt="test from litellm",
2212 input=["test from litellm"],
2213 ),
2214 HEALTH_CHECK_TIMEOUT_SECONDS,
2215 )
2217 # Clean the result for display
2218 cleaned_result: Final = _clean_endpoint_data({**litellm_params, **result}, details=True)
2220 return {
2221 "status": "error" if "error" in result else "success",
2222 "result": cleaned_result,
2223 }
2225 except HTTPException as e:
2226 raise e
2227 except Exception as e:
2228 verbose_proxy_logger.debug("litellm.proxy.health_endpoints.test_model_connection(): Exception occurred - %s", e)
2229 raise HTTPException(
2230 status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
2231 detail={"error": f"Failed to test connection: {e}"},
2232 )