Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/health_check.py: 54%
381 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1# This file runs a health check for the LLM, used on litellm/proxy
3import asyncio
4import logging
5import random
6import sys
7import threading
8import time
9from collections.abc import Mapping, Sequence
10from collections.abc import Set as AbstractSet
11from types import MappingProxyType
12from typing import TYPE_CHECKING, Final, TypeVar
14from pydantic import TypeAdapter, ValidationError
16import litellm
18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true
19 from litellm.router import Router
21logger: Final = logging.getLogger(__name__)
22_DeploymentT: Final = TypeVar("_DeploymentT", bound=Mapping[str, object])
23from litellm.constants import (
24 BACKGROUND_HEALTH_CHECK_MAX_TOKENS,
25 BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING,
26 DEFAULT_HEALTH_CHECK_PROMPT,
27 HEALTH_CHECK_TIMEOUT_SECONDS,
28)
29from litellm.router_utils.auto_router_model_naming import (
30 StrategyRouterDependency,
31 classify_strategy_router_model,
32 strategy_router_dependencies,
33)
35# Provider routing fields. Allowed for proxy admins so they can see which
36# region/version a deployment is checking; gated at the endpoint layer for
37# non-admin callers (see _strip_admin_only_fields_from_health_result).
38ADMIN_ONLY_HEALTH_DISPLAY_PARAMS: Final = ("api_base", "api_version", "aws_bedrock_runtime_endpoint")
40MINIMAL_DISPLAY_PARAMS: Final = frozenset({"model", "mode_error"})
42HEALTH_DISPLAY_PARAMS: Final = (
43 MINIMAL_DISPLAY_PARAMS
44 | frozenset(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS)
45 | frozenset(
46 {
47 "custom_llm_provider",
48 "mode",
49 "base_model",
50 "aws_region_name",
51 "region_name",
52 "watsonx_region_name",
53 "vertex_project",
54 "vertex_location",
55 "tpm",
56 "rpm",
57 "error",
58 "raw_request_typed_dict",
59 "x-ratelimit-remaining-requests",
60 "x-ratelimit-remaining-tokens",
61 "x-ms-region",
62 }
63 )
64)
66# Modes whose health-check probe is a chat-style completion call and
67# therefore accept `max_tokens`. Other modes (embedding, image_generation,
68# audio_*, rerank, video_generation, ocr, search, moderation, ...) hit
69# endpoints that reject unknown fields with 400 "Unknown parameter:
70# 'max_tokens'". Allow-list so new modes are safe by default.
71# Per-deployment override: `model_info.health_check_supports_max_tokens`.
72_MAX_TOKEN_SUPPORT_MODES: Final[frozenset[str]] = frozenset({"chat", "completion", "responses"})
75def _resolve_health_check_mode(model_info: Mapping[str, object], litellm_params: Mapping[str, object]) -> str | None:
76 """
77 Effective mode for a deployment's health-check probe.
79 Prefers operator-set `model_info.mode`; otherwise resolves it from the model
80 cost map, which understands `bedrock/` and cross-region inference-profile
81 prefixes (`us.`, `eu.`, `apac.`). Without this, non-chat Bedrock deployments
82 (e.g. embeddings) are probed as chat, so `max_tokens` is injected and the
83 request 400s on "extraneous key [max_tokens]".
84 """
85 explicit_mode: Final = model_info.get("mode")
86 if isinstance(explicit_mode, str): 86 ↛ 87line 86 didn't jump to line 87 because the condition on line 86 was never true
87 return explicit_mode
88 model: Final = litellm_params.get("model")
89 if not isinstance(model, str):
90 return None
91 try:
92 return litellm.get_model_info(model=model).get("mode")
93 except Exception:
94 return None
97def _should_inject_health_check_max_tokens(model_info: Mapping[str, object], mode: str | None) -> bool:
98 """
99 Whether the health-check probe should include `max_tokens`.
101 Order:
102 1. `model_info.health_check_supports_max_tokens` (operator override).
103 2. `_MAX_TOKEN_SUPPORT_MODES`. An unresolvable mode is treated as `chat`
104 for backward compatibility.
105 """
106 explicit: Final = model_info.get("health_check_supports_max_tokens")
107 if explicit is not None: 107 ↛ 108line 107 didn't jump to line 108 because the condition on line 107 was never true
108 return bool(explicit)
109 return (mode or "chat") in _MAX_TOKEN_SUPPORT_MODES
112# Health-check modes that forward `reasoning_effort` to the provider (chat-style calls).
113_HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: Final = frozenset((None, "chat", "completion"))
116def _get_process_rss_mb() -> float | None:
117 """
118 Get process RSS memory in MB.
119 On Linux, ru_maxrss is in KB. On macOS, ru_maxrss is in bytes.
120 """
121 try:
122 import resource
124 ru_maxrss: Final = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
125 if sys.platform == "darwin":
126 return float(ru_maxrss) / (1024 * 1024)
127 return float(ru_maxrss) / 1024
128 except Exception:
129 return None
132def _rss_mb_for_log() -> str:
133 rss_mb: Final = _get_process_rss_mb()
134 if rss_mb is None:
135 return "unknown"
136 return f"{rss_mb:.2f}"
139def _get_random_llm_message():
140 """
141 Get a random message from the LLM.
142 """
143 messages: Final = ["Hey how's it going?", "What's 1 + 1?"]
145 return [{"role": "user", "content": random.choice(messages)}]
148def _clean_endpoint_data(endpoint_data: dict, details: bool | None = True):
149 """
150 Keep only the explicitly approved, JSON-safe diagnostic fields for display to users.
151 """
152 displayed: Final = HEALTH_DISPLAY_PARAMS if details is not False else MINIMAL_DISPLAY_PARAMS
153 return {k: v for k, v in endpoint_data.items() if k in displayed}
156def health_check_filter_kwargs_from_general_settings(
157 general_settings: dict | None,
158) -> dict:
159 """
160 Build kwargs for ``perform_health_check`` from ``general_settings``.
162 When ``health_check_skip_disabled_background_models`` is true, deployments with
163 ``model_info.disable_background_health_check`` are omitted from health runs
164 (including on-demand ``GET /health``), matching the background loop behavior.
165 """
166 g: Final = general_settings or {}
167 return {
168 "health_check_skip_disabled_background_models": bool(
169 g.get("health_check_skip_disabled_background_models", False)
170 ),
171 }
174def parse_background_health_check_model_groups(
175 general_settings: Mapping[str, object] | None,
176) -> frozenset[str] | None:
177 """
178 Read ``general_settings.background_health_check_model_groups``.
180 ``None`` means the allowlist is unset and every deployment participates
181 (legacy behavior). A list scopes background health checks and health-check
182 routing to deployments whose ``model_name`` is listed. A malformed value
183 raises so the proxy fails at startup instead of silently probing everything.
184 """
185 raw: Final = (general_settings or {}).get("background_health_check_model_groups")
186 if raw is None: 186 ↛ 188line 186 didn't jump to line 188 because the condition on line 186 was always true
187 return None
188 try:
189 return frozenset(TypeAdapter(list[str]).validate_python(raw))
190 except ValidationError as e:
191 raise ValueError(
192 "general_settings.background_health_check_model_groups must be a list of model group names"
193 ) from e
196def filter_deployments_to_model_groups(
197 model_list: Sequence[_DeploymentT],
198 model_groups: AbstractSet[str] | None,
199) -> tuple[_DeploymentT, ...]:
200 """Deployments whose ``model_name`` is in ``model_groups``; all of them when unset."""
201 if model_groups is None:
202 return tuple(model_list)
203 return tuple(x for x in model_list if x.get("model_name") in model_groups)
206def filter_deployments_by_id(
207 model_list: Sequence[Mapping[str, object]],
208) -> list:
209 seen_ids: Final = set()
210 filtered_deployments: Final = []
212 for deployment in model_list:
213 _model_info = deployment.get("model_info") or {}
214 _id = _model_info.get("id") or None
215 if _id is None: 215 ↛ 216line 215 didn't jump to line 216 because the condition on line 215 was never true
216 continue
218 if _id not in seen_ids: 218 ↛ 212line 218 didn't jump to line 212 because the condition on line 218 was always true
219 seen_ids.add(_id)
220 filtered_deployments.append(deployment)
222 return filtered_deployments
225async def run_with_timeout(task, timeout):
226 try:
227 return await asyncio.wait_for(task, timeout)
228 except asyncio.TimeoutError:
229 # `asyncio.wait_for()` already cancels only the awaited task on timeout.
230 # Do not cancel unrelated sibling health check tasks.
231 timeout_exception: Final = litellm.Timeout(
232 message="Health check timeout exceeded",
233 model="",
234 llm_provider="",
235 )
236 return {"error": "Timeout exceeded", "exception": timeout_exception}
239def _skips_health_checks(deployment: Mapping[str, object]) -> bool:
240 info: Final = deployment.get("model_info")
241 return bool(info.get("disable_background_health_check", False)) if isinstance(info, Mapping) else False
244def _health_check_eligible(
245 model_list: Sequence[Mapping[str, object]], skip_disabled: bool
246) -> tuple[Mapping[str, object], ...]:
247 """Deployments this run is allowed to contact.
249 The one eligibility gate, applied to the requested set and to the pool a router's
250 dependencies are drawn from alike, so an opted-out deployment cannot re-enter through a
251 router that depends on it.
252 """
253 return tuple(x for x in model_list if not (skip_disabled and _skips_health_checks(x)))
256def _deployment_model(deployment: Mapping[str, object]) -> str | None:
257 params: Final = deployment.get("litellm_params")
258 return params.get("model") if isinstance(params, Mapping) else None
261def _owner_team_id(deployment: Mapping[str, object]) -> str | None:
262 info: Final = deployment.get("model_info")
263 owner: Final = info.get("team_id") if isinstance(info, Mapping) else None
264 return owner if isinstance(owner, str) else None
267def _team_public_model_name(deployment: Mapping[str, object]) -> str | None:
268 info: Final = deployment.get("model_info")
269 name: Final = info.get("team_public_model_name") if isinstance(info, Mapping) else None
270 return name if isinstance(name, str) else None
273def _deployments_routed_by_name(
274 model_list: Sequence[Mapping[str, object]], model_name: str, team_id: str | None
275) -> tuple[Mapping[str, object], ...]:
276 """The deployments a request for ``model_name`` from this caller routes to.
278 A team's own copies published under that name win, then deployments carrying it as
279 ``model_name``. A caller with no team reaches a public name only when nothing carries
280 it as ``model_name``, and only an admin still has another team's deployment in a
281 scoped ``model_list`` by then.
282 """
283 own_copies: Final = tuple(
284 x
285 for x in model_list
286 if team_id is not None and _owner_team_id(x) == team_id and _team_public_model_name(x) == model_name
287 )
288 if own_copies: 288 ↛ 289line 288 didn't jump to line 289 because the condition on line 288 was never true
289 return own_copies
290 by_name: Final = tuple(x for x in model_list if x.get("model_name") == model_name)
291 if by_name or team_id is not None: 291 ↛ 292line 291 didn't jump to line 292 because the condition on line 291 was never true
292 return by_name
293 return tuple(x for x in model_list if _team_public_model_name(x) == model_name)
296def deployments_targeted_by_name(
297 model_list: Sequence[Mapping[str, object]], model: str, team_id: str | None
298) -> tuple[Mapping[str, object], ...]:
299 """``model`` targets deployments the way a request for it routes, else by ``litellm_params.model``."""
300 return _deployments_routed_by_name(model_list, model, team_id) or tuple(
301 x for x in model_list if _deployment_model(x) == model
302 )
305def _narrow_to_target(
306 model_list: Sequence[Mapping[str, object]], model: str | None, model_id: str | None, team_id: str | None
307) -> tuple[Mapping[str, object], ...]:
308 """Narrow to the requested deployment. An id matching nothing keeps the whole list."""
309 if model_id is not None:
310 by_id: Final = tuple(x for x in model_list if _deployment_id(x) == model_id)
311 return by_id or tuple(model_list)
312 if model is None:
313 return tuple(model_list)
314 return deployments_targeted_by_name(model_list, model, team_id)
317def _is_strategy_router_deployment(litellm_params: Mapping[str, object]) -> bool:
318 """True for strategy-router deployments."""
319 model: Final[object] = litellm_params.get("model", "")
320 return isinstance(model, str) and classify_strategy_router_model(model) is not None
323def _is_marker(deployment: Mapping[str, object]) -> bool:
324 params: Final = deployment.get("litellm_params")
325 return isinstance(params, Mapping) and _is_strategy_router_deployment(params)
328def _deployment_id(deployment: Mapping[str, object]) -> str | None:
329 info: Final = deployment.get("model_info")
330 ident: Final = info.get("id") if isinstance(info, Mapping) else None
331 return str(ident) if ident else None
334def _resolved_deployment_ids(router: "Router", model_name: str) -> frozenset[str] | None:
335 """Deployment ids backing `model_name`, or None when the name resolves to nothing.
337 `get_model_list` composes every channel the request path itself uses (exact name,
338 model_group_alias, routing groups, wildcards); a mirror of any one channel would call a
339 working tier broken. An alias whose target is gone resolves to nothing, which fails a
340 request exactly like an unknown name.
341 """
342 resolved: Final = router.get_model_list(model_name=model_name)
343 if not resolved:
344 return None
345 return frozenset(ident for entry in resolved if (ident := _deployment_id(entry)))
348def _dependency_failure(
349 dependency: StrategyRouterDependency,
350 router: "Router",
351 unhealthy_ids: frozenset[str],
352) -> str | None:
353 """Why this dependency makes its router unable to serve, or None when it does not.
355 A name reds its router only when *every* deployment behind it is known unhealthy. One
356 replica this run never judged, hidden from the caller or opted out of health checks, can
357 still serve what the dead one drops, so partial evidence leaves the verdict green.
358 """
359 resolved: Final = _resolved_deployment_ids(router, dependency.model_name)
360 if resolved is None:
361 return f"{dependency.role} model '{dependency.model_name}' matches no deployment on this proxy"
362 if not resolved or not resolved <= unhealthy_ids:
363 return None
364 return f"{dependency.role} model '{dependency.model_name}' has no healthy deployment"
367def _strategy_router_dependency_error(
368 deployment: Mapping[str, object],
369 router: "Router",
370 unhealthy_ids: frozenset[str],
371) -> str | None:
372 """The first dependency fault that makes this router unable to serve, if any."""
373 params: Final = deployment.get("litellm_params")
374 if not isinstance(params, Mapping):
375 return None
376 return next(
377 (
378 failure
379 for dependency in strategy_router_dependencies(params)
380 if dependency.role != "evaluation"
381 if (failure := _dependency_failure(dependency, router, unhealthy_ids))
382 ),
383 None,
384 )
387def _deployments_by_id(
388 universe: Sequence[Mapping[str, object]], ids: frozenset[str]
389) -> tuple[Mapping[str, object], ...]:
390 """The deployments for `ids`, one row per id.
392 Reuses the requested set's own dedupe rule, so an alias that duplicates a row cannot get
393 it probed twice or split a single id's verdict across two disagreeing results.
394 """
395 matched: Final = tuple(d for d in universe if (uid := _deployment_id(d)) and uid in ids)
396 return tuple(filter_deployments_by_id(model_list=matched))
399def _dependency_deployments_to_probe(
400 checked: Sequence[Mapping[str, object]],
401 universe: Sequence[Mapping[str, object]],
402 router: "Router",
403) -> tuple[Mapping[str, object], ...]:
404 """Deployments backing the checked routers' dependencies that are not already checked.
406 Empty on a full-list run, which therefore gains no probe; it is the targeted
407 `/health?model_id=<router>` call the dashboard makes per deployment that needs them,
408 since a router's verdict is a statement about models the request never named. Drawn from
409 `universe`, the caller's access-filtered list, so no deployment is probed that the caller
410 was not already granted. Expansion follows routers through routers, one hop per round,
411 because a child router's own models must be probed for the parent to fail; stopping when
412 a round adds nothing is what makes a router cycle terminate.
413 """
414 checked_ids: Final = frozenset(cid for d in checked if (cid := _deployment_id(d)))
415 reached = checked_ids # rebind-ok: the sweep's cursor, one hop wider per round
416 frontier = tuple(checked) # rebind-ok: the routers whose dependencies the next round expands
417 for _ in range(len(universe)): 417 ↛ 432line 417 didn't jump to line 432 because the loop on line 417 didn't complete
418 names = frozenset(
419 dependency.model_name
420 for deployment in frontier
421 if isinstance(params := deployment.get("litellm_params"), Mapping)
422 for dependency in strategy_router_dependencies(params)
423 if dependency.role != "evaluation"
424 )
425 fresh_ids = (
426 frozenset(ident for name in names for ident in (_resolved_deployment_ids(router, name) or ())) - reached
427 )
428 if not fresh_ids: 428 ↛ 430line 428 didn't jump to line 430 because the condition on line 428 was always true
429 break
430 frontier = _deployments_by_id(universe, fresh_ids)
431 reached = reached | fresh_ids
432 return _deployments_by_id(universe, reached - checked_ids)
435def _strategy_router_verdicts(
436 healthy_endpoints: Sequence[Mapping[str, object]],
437 unhealthy_endpoints: Sequence[Mapping[str, object]],
438 checked: Sequence[Mapping[str, object]],
439 router: "Router",
440) -> Mapping[str, str]:
441 """The dependency fault, per model id, for every strategy router that cannot serve.
443 A marker is filed healthy by `_run_model_health_check` returning `{}`, which says only
444 that nothing was probed. This is where that placeholder becomes a verdict, derived from
445 this run's own results rather than a re-probe or a cache that is empty unless
446 `enable_health_check_routing` is on. A marker never fails a probe of its own, so verdicts
447 settle over rounds, each feeding the last round's reds back in as unhealthy; without that
448 the parent of a red child would stay green. Bounded by the marker count, which is what
449 makes a router cycle terminate green rather than spin.
450 """
451 by_id: Final = MappingProxyType({i: d for d in checked if (i := _deployment_id(d))})
452 markers: Final = MappingProxyType(
453 {
454 marker_id: by_id[marker_id]
455 for endpoint in healthy_endpoints
456 if isinstance(marker_id := endpoint.get("model_id"), str) and marker_id in by_id
457 if _is_marker(by_id[marker_id])
458 }
459 )
460 probe_failures: Final = frozenset(
461 ident for endpoint in unhealthy_endpoints if isinstance(ident := endpoint.get("model_id"), str)
462 )
463 settled: Mapping[str, str] = MappingProxyType({}) # rebind-ok: the fixed point, a round's verdicts at a time
464 for _ in range(len(markers)): 464 ↛ 465line 464 didn't jump to line 465 because the loop on line 464 never started
465 fresh = MappingProxyType(
466 {
467 marker_id: error
468 for marker_id, deployment in markers.items()
469 if marker_id not in settled
470 if (error := _strategy_router_dependency_error(deployment, router, probe_failures | frozenset(settled)))
471 }
472 )
473 if not fresh:
474 break
475 settled = MappingProxyType({**settled, **fresh})
476 return settled
479def _finalize_strategy_router_endpoints(
480 healthy_endpoints: Sequence[Mapping[str, object]],
481 unhealthy_endpoints: Sequence[Mapping[str, object]],
482 checked: Sequence[Mapping[str, object]],
483 router: "Router | None",
484 dependency_probes: Sequence[Mapping[str, object]],
485) -> tuple[Sequence[Mapping[str, object]], Sequence[Mapping[str, object]]]:
486 """Apply router verdicts, then drop the deployments probed only to reach them.
488 The probes exist to judge the routers that depend on them; reporting them would answer a
489 targeted request with deployments the caller never asked about.
490 """
491 verdicts: Final = (
492 _strategy_router_verdicts(healthy_endpoints, unhealthy_endpoints, checked, router)
493 if router is not None
494 else MappingProxyType({})
495 )
496 dropped: Final = frozenset(i for d in dependency_probes if (i := _deployment_id(d)))
498 def keep(endpoint: Mapping[str, object]) -> bool:
499 model_id: Final = endpoint.get("model_id")
500 return not (isinstance(model_id, str) and model_id in dropped)
502 def verdict_for(endpoint: Mapping[str, object]) -> str | None:
503 model_id: Final = endpoint.get("model_id")
504 return verdicts.get(model_id) if isinstance(model_id, str) else None
506 kept_healthy: Final = tuple(e for e in healthy_endpoints if keep(e))
507 return (
508 tuple(e for e in kept_healthy if verdict_for(e) is None),
509 tuple(e for e in unhealthy_endpoints if keep(e))
510 + tuple(
511 dict(e, error=error) # mutable-ok: the /health payload must stay a plain JSON-serializable dict
512 for e in kept_healthy
513 if (error := verdict_for(e)) is not None
514 ),
515 )
518async def _run_model_health_check(model: dict):
519 litellm_params = model["litellm_params"]
520 model_info: Final = model.get("model_info", {})
522 if _is_strategy_router_deployment(litellm_params): 522 ↛ 523line 522 didn't jump to line 523 because the condition on line 522 was never true
523 return {}
525 mode: Final = _resolve_health_check_mode(
526 model_info,
527 litellm_params, # any-ok: untyped router config dict
528 )
529 litellm_params = _update_litellm_params_for_health_check(model_info, litellm_params)
530 timeout: Final = model_info.get("health_check_timeout") or HEALTH_CHECK_TIMEOUT_SECONDS
532 return await run_with_timeout(
533 litellm.ahealth_check(
534 litellm_params,
535 mode=mode,
536 prompt=DEFAULT_HEALTH_CHECK_PROMPT,
537 input=["test from litellm"],
538 ),
539 timeout,
540 )
543async def _run_health_checks_with_bounded_concurrency(models: list, concurrency_limit: int) -> tuple[list, int]:
544 """
545 Run health checks with at most `concurrency_limit` active tasks.
546 Preserves result ordering to match `models`.
547 """
548 results: Final[list] = [None] * len(models)
549 tasks_to_index: Final[dict[asyncio.Task, int]] = {}
550 model_iter: Final = iter(enumerate(models))
551 peak_in_flight = 0
553 def _schedule_next() -> bool:
554 nonlocal peak_in_flight
555 try:
556 idx, next_model = next(model_iter)
557 except StopIteration:
558 return False
559 task: Final = asyncio.create_task(_run_model_health_check(next_model))
560 tasks_to_index[task] = idx
561 peak_in_flight = max(peak_in_flight, len(tasks_to_index))
562 return True
564 for _ in range(min(concurrency_limit, len(models))):
565 _schedule_next()
567 while tasks_to_index:
568 done, _ = await asyncio.wait(
569 set(tasks_to_index.keys()),
570 return_when=asyncio.FIRST_COMPLETED,
571 )
572 for task in done:
573 idx = tasks_to_index.pop(task)
574 try:
575 results[idx] = task.result()
576 except Exception as e:
577 results[idx] = e
578 _schedule_next()
580 return results, peak_in_flight
583async def _perform_health_check(
584 model_list: list,
585 details: bool | None = True,
586 max_concurrency: int | None = None,
587 instrumentation_context: dict | None = None,
588):
589 """
590 Perform a health check for each model in the list.
592 max_concurrency: Optional limit on concurrent health check requests.
593 """
595 instrumentation_context = instrumentation_context or {}
596 instrumentation_enabled: Final = bool(instrumentation_context.get("enabled", False))
597 cycle_id: Final = instrumentation_context.get("cycle_id", "unknown")
598 source: Final = instrumentation_context.get("source", "unknown")
600 dispatch_mode = "unbounded"
601 peak_in_flight = 0
602 if isinstance(max_concurrency, int) and max_concurrency > 0: 602 ↛ 603line 602 didn't jump to line 603 because the condition on line 602 was never true
603 dispatch_mode = "bounded"
604 results, peak_in_flight = await _run_health_checks_with_bounded_concurrency(model_list, max_concurrency)
605 else:
606 tasks: Final = [asyncio.create_task(_run_model_health_check(model)) for model in model_list]
607 peak_in_flight = len(tasks)
608 results = await asyncio.gather(*tasks, return_exceptions=True)
610 if instrumentation_enabled: 610 ↛ 611line 610 didn't jump to line 611 because the condition on line 610 was never true
611 logger.debug(
612 "health_check_dispatch_summary source=%s cycle_id=%s mode=%s model_count=%d max_concurrency=%s peak_in_flight=%d thread_count=%d rss_mb=%s",
613 source,
614 cycle_id,
615 dispatch_mode,
616 len(model_list),
617 max_concurrency,
618 peak_in_flight,
619 threading.active_count(),
620 _rss_mb_for_log(),
621 )
623 healthy_endpoints: Final = []
624 unhealthy_endpoints: Final = []
625 # Exceptions keyed by model_id; returned separately so callers can use
626 # them for cooldown integration without risking JSON-serialization errors
627 # in the /health response.
628 exceptions_by_model_id: Final[dict] = {}
630 for is_healthy, model in zip(results, model_list):
631 litellm_params = model["litellm_params"]
632 _model_id = (model.get("model_info") or {}).get("id")
634 if isinstance(is_healthy, dict) and "error" not in is_healthy: 634 ↛ 635line 634 didn't jump to line 635 because the condition on line 634 was never true
635 cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details)
636 if _model_id:
637 cleaned["model_id"] = _model_id
638 healthy_endpoints.append(cleaned)
639 elif isinstance(is_healthy, dict): 639 ↛ 651line 639 didn't jump to line 651 because the condition on line 639 was always true
640 cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details)
641 if _model_id: 641 ↛ 649line 641 didn't jump to line 649 because the condition on line 641 was always true
642 cleaned["model_id"] = _model_id
643 if "exception" in is_healthy: 643 ↛ 649line 643 didn't jump to line 649 because the condition on line 643 was always true
644 exc = is_healthy["exception"]
645 exceptions_by_model_id[_model_id] = exc
646 # Store integer status code so shared-cache readers can
647 # reconstruct the transient-error filter without the exception object.
648 cleaned["exception_status"] = getattr(exc, "status_code", 500)
649 unhealthy_endpoints.append(cleaned)
650 else:
651 cleaned = _clean_endpoint_data(litellm_params, details)
652 if _model_id:
653 cleaned["model_id"] = _model_id
654 if isinstance(is_healthy, Exception):
655 exceptions_by_model_id[_model_id] = is_healthy
656 cleaned["exception_status"] = getattr(is_healthy, "status_code", 500)
657 unhealthy_endpoints.append(cleaned)
659 return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id
662def build_deployment_health_states(
663 healthy_endpoints: list,
664 unhealthy_endpoints: list,
665) -> dict:
666 """
667 Build a dict mapping deployment_id -> DeploymentHealthStateValue from
668 health check endpoint results.
670 Each endpoint dict includes a 'model_id' field (added by _perform_health_check)
671 that maps back to the deployment's model_info.id.
673 Used by the background health check loop to feed health state into
674 the router's DeploymentHealthCache for health-check-driven routing.
675 """
676 now: Final = time.time()
677 states: Final[dict] = {}
679 for ep in healthy_endpoints:
680 model_id = ep.get("model_id")
681 if model_id:
682 states[model_id] = {
683 "is_healthy": True,
684 "timestamp": now,
685 "reason": "",
686 }
688 for ep in unhealthy_endpoints:
689 model_id = ep.get("model_id")
690 if model_id:
691 states[model_id] = {
692 "is_healthy": False,
693 "timestamp": now,
694 "reason": "background_health_check_failed",
695 }
697 return states
700def _deployment_model_string_for_health_check(litellm_params: dict) -> str:
701 """Deployment model from litellm_params (before Bedrock rewrite).
703 Used for reasoning vs non-reasoning max_tokens and wildcard detection only.
704 Does not use ``health_check_model``; that override applies later to the request.
705 """
706 return litellm_params.get("model") or ""
709def _health_check_deployment_is_wildcard(litellm_params: dict) -> bool:
710 return "*" in _deployment_model_string_for_health_check(litellm_params)
713def _resolve_health_check_max_tokens(model_info: dict, litellm_params: dict) -> int | None:
714 """
715 Pick max_tokens for the health check request.
717 Priority:
718 1. model_info.health_check_max_tokens (explicit override)
719 2. For non-wildcard routes: health_check_max_tokens_reasoning / _non_reasoning
720 from model_info based on litellm.supports_reasoning(litellm_params["model"])
721 3. For non-wildcard reasoning routes: BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING
722 from env (if set)
723 4. BACKGROUND_HEALTH_CHECK_MAX_TOKENS (global, any route including wildcards)
724 5. Non-wildcard default: 16
725 6. Wildcard and nothing from (1)(4): leave unset (caller omits max_tokens)
726 """
727 explicit: Final = model_info.get("health_check_max_tokens", None)
728 if explicit is not None: 728 ↛ 729line 728 didn't jump to line 729 because the condition on line 728 was never true
729 return int(explicit)
731 is_wildcard: Final = _health_check_deployment_is_wildcard(litellm_params)
732 deployment_model: Final = _deployment_model_string_for_health_check(litellm_params)
734 if not is_wildcard: 734 ↛ 749line 734 didn't jump to line 749 because the condition on line 734 was always true
735 try:
736 is_reasoning = litellm.supports_reasoning(deployment_model)
737 except Exception:
738 is_reasoning = False
739 tokens_reasoning: Final = model_info.get("health_check_max_tokens_reasoning", None)
740 tokens_non_reasoning: Final = model_info.get("health_check_max_tokens_non_reasoning", None)
741 if tokens_reasoning is not None or tokens_non_reasoning is not None: 741 ↛ 742line 741 didn't jump to line 742 because the condition on line 741 was never true
742 if is_reasoning and tokens_reasoning is not None:
743 return int(tokens_reasoning)
744 if not is_reasoning and tokens_non_reasoning is not None:
745 return int(tokens_non_reasoning)
746 if is_reasoning and BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING is not None: 746 ↛ 747line 746 didn't jump to line 747 because the condition on line 746 was never true
747 return int(BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING)
749 if BACKGROUND_HEALTH_CHECK_MAX_TOKENS is not None: 749 ↛ 750line 749 didn't jump to line 750 because the condition on line 749 was never true
750 return int(BACKGROUND_HEALTH_CHECK_MAX_TOKENS)
752 if not is_wildcard: 752 ↛ 755line 752 didn't jump to line 755 because the condition on line 752 was always true
753 return 16
755 return None
758def _update_litellm_params_for_health_check(model_info: dict, litellm_params: dict) -> dict:
759 """
760 Update the litellm params for health check.
762 - merges `model_info.health_check_params` into the probe request, so a deployment whose provider
763 requires a payload field litellm does not synthesize (e.g. `mediaSource` for Bedrock TwelveLabs
764 Pegasus) can supply it. The dedicated knobs below are applied afterwards and win on conflict.
765 - gets a short `messages` param for health check
766 - adds a bounded `max_tokens` when the deployment is a chat-style mode
767 (`chat`, `completion`, `responses`) or the operator explicitly opts in
768 via `model_info.health_check_supports_max_tokens`. Non-chat endpoints
769 (image, embedding, audio_*, rerank, video, ocr, search, moderation, ...)
770 reject unknown fields with 400 "Unknown parameter: 'max_tokens'".
771 - updates the `model` param with the `health_check_model` if it exists Doc: https://docs.litellm.ai/docs/proxy/health#wildcard-routes
772 - updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models
773 - for Bedrock models with region routing (bedrock/region/model), strips the litellm routing prefix but preserves the model ID, and pins `custom_llm_provider` to `bedrock` (only when the deployment hasn't already set one, so an explicit `bedrock_converse` survives) so the bare model id still resolves to the provider (e.g. cross-region ids like `us.cohere.embed-v4:0`)
774 """
775 mode: Final = _resolve_health_check_mode(
776 model_info,
777 litellm_params, # any-ok: untyped router config dict
778 )
779 _health_check_params: Final = model_info.get("health_check_params", None)
780 if isinstance(_health_check_params, dict): 780 ↛ 781line 780 didn't jump to line 781 because the condition on line 780 was never true
781 litellm_params.update(_health_check_params)
782 elif _health_check_params is not None: 782 ↛ 783line 782 didn't jump to line 783 because the condition on line 782 was never true
783 logger.warning(
784 "health_check_params for model %s is a %s, expected a dict. Ignoring it.",
785 litellm_params.get("model"),
786 type(_health_check_params).__name__,
787 )
789 litellm_params["messages"] = _get_random_llm_message()
790 if _should_inject_health_check_max_tokens( 790 ↛ 799line 790 didn't jump to line 799 because the condition on line 790 was always true
791 model_info,
792 mode, # any-ok: untyped router config dict
793 ):
794 _resolved_max_tokens: Final = _resolve_health_check_max_tokens(model_info, litellm_params)
795 if _resolved_max_tokens is not None: 795 ↛ 799line 795 didn't jump to line 799 because the condition on line 795 was always true
796 litellm_params["max_tokens"] = _resolved_max_tokens
798 # Per-model reasoning effort for health checks only (e.g. reasoning_effort=none).
799 if mode in _HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: 799 ↛ 804line 799 didn't jump to line 804 because the condition on line 799 was always true
800 _hc_reasoning_effort: Final = model_info.get("health_check_reasoning_effort", None)
801 if _hc_reasoning_effort is not None: 801 ↛ 802line 801 didn't jump to line 802 because the condition on line 801 was never true
802 litellm_params["reasoning_effort"] = _hc_reasoning_effort
804 _health_check_model: Final = model_info.get("health_check_model", None)
805 if _health_check_model is not None: 805 ↛ 806line 805 didn't jump to line 806 because the condition on line 805 was never true
806 litellm_params["model"] = _health_check_model
807 if mode == "audio_speech": 807 ↛ 808line 807 didn't jump to line 808 because the condition on line 807 was never true
808 litellm_params["voice"] = model_info.get("health_check_voice", "alloy")
810 # Handle Bedrock region routing format: bedrock/region/model
811 # This is needed because health checks bypass get_llm_provider() for the model param
812 # Issue #15807: Without this, health checks send "region/model" as the model ID to AWS
813 # which causes: "bedrock-runtime.../model/us-west-2/mistral.../invoke" (region in model ID)
814 #
815 # However, we must preserve cross-region inference profile prefixes like "us.", "eu.", etc.
816 # Issue: Stripping these breaks AWS requirement for inference profile IDs
817 #
818 # Must also preserve route prefixes (converse/, invoke/) and handlers (llama/, deepseek_r1/, etc.)
819 if litellm_params["model"].startswith("bedrock/"): 819 ↛ 820line 819 didn't jump to line 820 because the condition on line 819 was never true
820 from litellm.llms.bedrock.common_utils import BedrockModelInfo
822 model = litellm_params["model"]
823 # Strip only the bedrock/ prefix (preserve routes like converse/, invoke/)
824 model = model.removeprefix("bedrock/") # len("bedrock/") = 8
826 # Now check for region routing and strip it if present
827 # Need to handle formats like:
828 # - "us-west-2/model" → "model"
829 # - "converse/us-west-2/model" → "converse/model"
830 # - "llama/arn:..." → "llama/arn:..." (preserve handler)
831 #
832 # Strategy: Check each path segment, remove regions, preserve everything else
833 parts: Final = model.split("/")
834 filtered_parts: Final = []
836 for part in parts:
837 # Skip AWS regions, keep everything else
838 if part not in BedrockModelInfo.all_global_regions:
839 filtered_parts.append(part)
841 model = "/".join(filtered_parts)
842 litellm_params["model"] = model
843 if not litellm_params.get("custom_llm_provider"): # any-ok: untyped router dict
844 litellm_params["custom_llm_provider"] = ( # any-ok: untyped router dict
845 "bedrock"
846 )
848 return litellm_params
851async def perform_health_check(
852 model_list: list,
853 model: str | None = None,
854 cli_model: str | None = None,
855 details: bool | None = True,
856 model_id: str | None = None,
857 max_concurrency: int | None = None,
858 instrumentation_context: dict | None = None,
859 health_check_skip_disabled_background_models: bool = False,
860 router: "Router | None" = None,
861 team_id: str | None = None,
862):
863 """
864 Perform a health check on the system.
866 When model_id is provided, only the deployment with that id is checked
867 (so models that share the same name but have different ids are checked separately).
868 When model (name) is provided, the deployments a request for that name from the
869 caller (``team_id``) would route to are checked: the caller's team copies published
870 under that name, else the deployments named that way, else a public name that only
871 another team's deployment carries, else the deployments whose ``litellm_params.model``
872 is that string.
874 When ``health_check_skip_disabled_background_models`` is True (via
875 ``general_settings.health_check_skip_disabled_background_models``), deployments
876 with ``model_info.disable_background_health_check: true`` are omitted from
877 this run (including targeted ``/health`` queries), consistent with the
878 background health loop.
880 Returns:
881 (bool): True if the health check passes, False otherwise.
882 """
883 instrumentation_context = instrumentation_context or {}
884 instrumentation_enabled: Final = bool(instrumentation_context.get("enabled", False))
885 cycle_id: Final = instrumentation_context.get("cycle_id", "unknown")
886 source: Final = instrumentation_context.get("source", "unknown")
888 if not model_list: 888 ↛ 889line 888 didn't jump to line 889 because the condition on line 888 was never true
889 if cli_model:
890 model_list = [{"model_name": cli_model, "litellm_params": {"model": cli_model}}]
891 else:
892 if instrumentation_enabled:
893 logger.debug(
894 "health_check_cycle_skipped source=%s cycle_id=%s reason=no_models",
895 source,
896 cycle_id,
897 )
898 return [], [], {}
900 cycle_start_time: Final = time.monotonic()
901 requested_model_count: Final = len(model_list)
902 skip_disabled: Final = health_check_skip_disabled_background_models
903 narrowed: Final = _health_check_eligible(_narrow_to_target(model_list, model, model_id, team_id), skip_disabled)
904 if not narrowed:
905 if instrumentation_enabled: 905 ↛ 906line 905 didn't jump to line 906 because the condition on line 905 was never true
906 logger.debug(
907 "health_check_cycle_skipped source=%s cycle_id=%s reason=no_models_after_filter",
908 source,
909 cycle_id,
910 )
911 return [], [], {}
913 post_filter_model_count: Final = len(narrowed)
914 requested: Final = filter_deployments_by_id(model_list=narrowed)
915 deduped_model_count: Final = len(requested)
917 dependency_probes: Final = (
918 _dependency_deployments_to_probe(requested, _health_check_eligible(model_list, skip_disabled), router)
919 if router is not None
920 else ()
921 )
922 checked: Final = requested + list(dependency_probes) # mutable-ok: _perform_health_check takes a list
924 if instrumentation_enabled: 924 ↛ 925line 924 didn't jump to line 925 because the condition on line 924 was never true
925 logger.debug(
926 "health_check_cycle_start source=%s cycle_id=%s requested_model_count=%d post_model_filter_count=%d deduped_model_count=%d max_concurrency=%s thread_count=%d rss_mb=%s",
927 source,
928 cycle_id,
929 requested_model_count,
930 post_filter_model_count,
931 deduped_model_count,
932 max_concurrency,
933 threading.active_count(),
934 _rss_mb_for_log(),
935 )
937 try:
938 (
939 probed_healthy,
940 probed_unhealthy,
941 exceptions_by_model_id,
942 ) = await _perform_health_check(
943 checked,
944 details,
945 max_concurrency=max_concurrency,
946 instrumentation_context=instrumentation_context,
947 )
948 graded_healthy, graded_unhealthy = _finalize_strategy_router_endpoints(
949 probed_healthy, probed_unhealthy, checked, router, dependency_probes
950 )
951 healthy_endpoints: Final = list(graded_healthy)
952 unhealthy_endpoints: Final = list(graded_unhealthy)
953 except Exception:
954 if instrumentation_enabled:
955 logger.exception(
956 "health_check_cycle_failed source=%s cycle_id=%s model_count=%d duration_ms=%.2f thread_count=%d rss_mb=%s",
957 source,
958 cycle_id,
959 deduped_model_count,
960 (time.monotonic() - cycle_start_time) * 1000,
961 threading.active_count(),
962 _rss_mb_for_log(),
963 )
964 raise
966 if instrumentation_enabled: 966 ↛ 967line 966 didn't jump to line 967 because the condition on line 966 was never true
967 logger.debug(
968 "health_check_cycle_complete source=%s cycle_id=%s model_count=%d healthy_count=%d unhealthy_count=%d duration_ms=%.2f thread_count=%d rss_mb=%s",
969 source,
970 cycle_id,
971 deduped_model_count,
972 len(healthy_endpoints),
973 len(unhealthy_endpoints),
974 (time.monotonic() - cycle_start_time) * 1000,
975 threading.active_count(),
976 _rss_mb_for_log(),
977 )
979 return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id