Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/health_check.py: 54%

381 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1# This file runs a health check for the LLM, used on litellm/proxy 

2 

3import asyncio 

4import logging 

5import random 

6import sys 

7import threading 

8import time 

9from collections.abc import Mapping, Sequence 

10from collections.abc import Set as AbstractSet 

11from types import MappingProxyType 

12from typing import TYPE_CHECKING, Final, TypeVar 

13 

14from pydantic import TypeAdapter, ValidationError 

15 

16import litellm 

17 

18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true

19 from litellm.router import Router 

20 

21logger: Final = logging.getLogger(__name__) 

22_DeploymentT: Final = TypeVar("_DeploymentT", bound=Mapping[str, object]) 

23from litellm.constants import ( 

24 BACKGROUND_HEALTH_CHECK_MAX_TOKENS, 

25 BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING, 

26 DEFAULT_HEALTH_CHECK_PROMPT, 

27 HEALTH_CHECK_TIMEOUT_SECONDS, 

28) 

29from litellm.router_utils.auto_router_model_naming import ( 

30 StrategyRouterDependency, 

31 classify_strategy_router_model, 

32 strategy_router_dependencies, 

33) 

34 

35# Provider routing fields. Allowed for proxy admins so they can see which 

36# region/version a deployment is checking; gated at the endpoint layer for 

37# non-admin callers (see _strip_admin_only_fields_from_health_result). 

38ADMIN_ONLY_HEALTH_DISPLAY_PARAMS: Final = ("api_base", "api_version", "aws_bedrock_runtime_endpoint") 

39 

40MINIMAL_DISPLAY_PARAMS: Final = frozenset({"model", "mode_error"}) 

41 

42HEALTH_DISPLAY_PARAMS: Final = ( 

43 MINIMAL_DISPLAY_PARAMS 

44 | frozenset(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS) 

45 | frozenset( 

46 { 

47 "custom_llm_provider", 

48 "mode", 

49 "base_model", 

50 "aws_region_name", 

51 "region_name", 

52 "watsonx_region_name", 

53 "vertex_project", 

54 "vertex_location", 

55 "tpm", 

56 "rpm", 

57 "error", 

58 "raw_request_typed_dict", 

59 "x-ratelimit-remaining-requests", 

60 "x-ratelimit-remaining-tokens", 

61 "x-ms-region", 

62 } 

63 ) 

64) 

65 

66# Modes whose health-check probe is a chat-style completion call and 

67# therefore accept `max_tokens`. Other modes (embedding, image_generation, 

68# audio_*, rerank, video_generation, ocr, search, moderation, ...) hit 

69# endpoints that reject unknown fields with 400 "Unknown parameter: 

70# 'max_tokens'". Allow-list so new modes are safe by default. 

71# Per-deployment override: `model_info.health_check_supports_max_tokens`. 

72_MAX_TOKEN_SUPPORT_MODES: Final[frozenset[str]] = frozenset({"chat", "completion", "responses"}) 

73 

74 

75def _resolve_health_check_mode(model_info: Mapping[str, object], litellm_params: Mapping[str, object]) -> str | None: 

76 """ 

77 Effective mode for a deployment's health-check probe. 

78 

79 Prefers operator-set `model_info.mode`; otherwise resolves it from the model 

80 cost map, which understands `bedrock/` and cross-region inference-profile 

81 prefixes (`us.`, `eu.`, `apac.`). Without this, non-chat Bedrock deployments 

82 (e.g. embeddings) are probed as chat, so `max_tokens` is injected and the 

83 request 400s on "extraneous key [max_tokens]". 

84 """ 

85 explicit_mode: Final = model_info.get("mode") 

86 if isinstance(explicit_mode, str): 86 ↛ 87line 86 didn't jump to line 87 because the condition on line 86 was never true

87 return explicit_mode 

88 model: Final = litellm_params.get("model") 

89 if not isinstance(model, str): 

90 return None 

91 try: 

92 return litellm.get_model_info(model=model).get("mode") 

93 except Exception: 

94 return None 

95 

96 

97def _should_inject_health_check_max_tokens(model_info: Mapping[str, object], mode: str | None) -> bool: 

98 """ 

99 Whether the health-check probe should include `max_tokens`. 

100 

101 Order: 

102 1. `model_info.health_check_supports_max_tokens` (operator override). 

103 2. `_MAX_TOKEN_SUPPORT_MODES`. An unresolvable mode is treated as `chat` 

104 for backward compatibility. 

105 """ 

106 explicit: Final = model_info.get("health_check_supports_max_tokens") 

107 if explicit is not None: 107 ↛ 108line 107 didn't jump to line 108 because the condition on line 107 was never true

108 return bool(explicit) 

109 return (mode or "chat") in _MAX_TOKEN_SUPPORT_MODES 

110 

111 

112# Health-check modes that forward `reasoning_effort` to the provider (chat-style calls). 

113_HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: Final = frozenset((None, "chat", "completion")) 

114 

115 

116def _get_process_rss_mb() -> float | None: 

117 """ 

118 Get process RSS memory in MB. 

119 On Linux, ru_maxrss is in KB. On macOS, ru_maxrss is in bytes. 

120 """ 

121 try: 

122 import resource 

123 

124 ru_maxrss: Final = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss 

125 if sys.platform == "darwin": 

126 return float(ru_maxrss) / (1024 * 1024) 

127 return float(ru_maxrss) / 1024 

128 except Exception: 

129 return None 

130 

131 

132def _rss_mb_for_log() -> str: 

133 rss_mb: Final = _get_process_rss_mb() 

134 if rss_mb is None: 

135 return "unknown" 

136 return f"{rss_mb:.2f}" 

137 

138 

139def _get_random_llm_message(): 

140 """ 

141 Get a random message from the LLM. 

142 """ 

143 messages: Final = ["Hey how's it going?", "What's 1 + 1?"] 

144 

145 return [{"role": "user", "content": random.choice(messages)}] 

146 

147 

148def _clean_endpoint_data(endpoint_data: dict, details: bool | None = True): 

149 """ 

150 Keep only the explicitly approved, JSON-safe diagnostic fields for display to users. 

151 """ 

152 displayed: Final = HEALTH_DISPLAY_PARAMS if details is not False else MINIMAL_DISPLAY_PARAMS 

153 return {k: v for k, v in endpoint_data.items() if k in displayed} 

154 

155 

156def health_check_filter_kwargs_from_general_settings( 

157 general_settings: dict | None, 

158) -> dict: 

159 """ 

160 Build kwargs for ``perform_health_check`` from ``general_settings``. 

161 

162 When ``health_check_skip_disabled_background_models`` is true, deployments with 

163 ``model_info.disable_background_health_check`` are omitted from health runs 

164 (including on-demand ``GET /health``), matching the background loop behavior. 

165 """ 

166 g: Final = general_settings or {} 

167 return { 

168 "health_check_skip_disabled_background_models": bool( 

169 g.get("health_check_skip_disabled_background_models", False) 

170 ), 

171 } 

172 

173 

174def parse_background_health_check_model_groups( 

175 general_settings: Mapping[str, object] | None, 

176) -> frozenset[str] | None: 

177 """ 

178 Read ``general_settings.background_health_check_model_groups``. 

179 

180 ``None`` means the allowlist is unset and every deployment participates 

181 (legacy behavior). A list scopes background health checks and health-check 

182 routing to deployments whose ``model_name`` is listed. A malformed value 

183 raises so the proxy fails at startup instead of silently probing everything. 

184 """ 

185 raw: Final = (general_settings or {}).get("background_health_check_model_groups") 

186 if raw is None: 186 ↛ 188line 186 didn't jump to line 188 because the condition on line 186 was always true

187 return None 

188 try: 

189 return frozenset(TypeAdapter(list[str]).validate_python(raw)) 

190 except ValidationError as e: 

191 raise ValueError( 

192 "general_settings.background_health_check_model_groups must be a list of model group names" 

193 ) from e 

194 

195 

196def filter_deployments_to_model_groups( 

197 model_list: Sequence[_DeploymentT], 

198 model_groups: AbstractSet[str] | None, 

199) -> tuple[_DeploymentT, ...]: 

200 """Deployments whose ``model_name`` is in ``model_groups``; all of them when unset.""" 

201 if model_groups is None: 

202 return tuple(model_list) 

203 return tuple(x for x in model_list if x.get("model_name") in model_groups) 

204 

205 

206def filter_deployments_by_id( 

207 model_list: Sequence[Mapping[str, object]], 

208) -> list: 

209 seen_ids: Final = set() 

210 filtered_deployments: Final = [] 

211 

212 for deployment in model_list: 

213 _model_info = deployment.get("model_info") or {} 

214 _id = _model_info.get("id") or None 

215 if _id is None: 215 ↛ 216line 215 didn't jump to line 216 because the condition on line 215 was never true

216 continue 

217 

218 if _id not in seen_ids: 218 ↛ 212line 218 didn't jump to line 212 because the condition on line 218 was always true

219 seen_ids.add(_id) 

220 filtered_deployments.append(deployment) 

221 

222 return filtered_deployments 

223 

224 

225async def run_with_timeout(task, timeout): 

226 try: 

227 return await asyncio.wait_for(task, timeout) 

228 except asyncio.TimeoutError: 

229 # `asyncio.wait_for()` already cancels only the awaited task on timeout. 

230 # Do not cancel unrelated sibling health check tasks. 

231 timeout_exception: Final = litellm.Timeout( 

232 message="Health check timeout exceeded", 

233 model="", 

234 llm_provider="", 

235 ) 

236 return {"error": "Timeout exceeded", "exception": timeout_exception} 

237 

238 

239def _skips_health_checks(deployment: Mapping[str, object]) -> bool: 

240 info: Final = deployment.get("model_info") 

241 return bool(info.get("disable_background_health_check", False)) if isinstance(info, Mapping) else False 

242 

243 

244def _health_check_eligible( 

245 model_list: Sequence[Mapping[str, object]], skip_disabled: bool 

246) -> tuple[Mapping[str, object], ...]: 

247 """Deployments this run is allowed to contact. 

248 

249 The one eligibility gate, applied to the requested set and to the pool a router's 

250 dependencies are drawn from alike, so an opted-out deployment cannot re-enter through a 

251 router that depends on it. 

252 """ 

253 return tuple(x for x in model_list if not (skip_disabled and _skips_health_checks(x))) 

254 

255 

256def _deployment_model(deployment: Mapping[str, object]) -> str | None: 

257 params: Final = deployment.get("litellm_params") 

258 return params.get("model") if isinstance(params, Mapping) else None 

259 

260 

261def _owner_team_id(deployment: Mapping[str, object]) -> str | None: 

262 info: Final = deployment.get("model_info") 

263 owner: Final = info.get("team_id") if isinstance(info, Mapping) else None 

264 return owner if isinstance(owner, str) else None 

265 

266 

267def _team_public_model_name(deployment: Mapping[str, object]) -> str | None: 

268 info: Final = deployment.get("model_info") 

269 name: Final = info.get("team_public_model_name") if isinstance(info, Mapping) else None 

270 return name if isinstance(name, str) else None 

271 

272 

273def _deployments_routed_by_name( 

274 model_list: Sequence[Mapping[str, object]], model_name: str, team_id: str | None 

275) -> tuple[Mapping[str, object], ...]: 

276 """The deployments a request for ``model_name`` from this caller routes to. 

277 

278 A team's own copies published under that name win, then deployments carrying it as 

279 ``model_name``. A caller with no team reaches a public name only when nothing carries 

280 it as ``model_name``, and only an admin still has another team's deployment in a 

281 scoped ``model_list`` by then. 

282 """ 

283 own_copies: Final = tuple( 

284 x 

285 for x in model_list 

286 if team_id is not None and _owner_team_id(x) == team_id and _team_public_model_name(x) == model_name 

287 ) 

288 if own_copies: 288 ↛ 289line 288 didn't jump to line 289 because the condition on line 288 was never true

289 return own_copies 

290 by_name: Final = tuple(x for x in model_list if x.get("model_name") == model_name) 

291 if by_name or team_id is not None: 291 ↛ 292line 291 didn't jump to line 292 because the condition on line 291 was never true

292 return by_name 

293 return tuple(x for x in model_list if _team_public_model_name(x) == model_name) 

294 

295 

296def deployments_targeted_by_name( 

297 model_list: Sequence[Mapping[str, object]], model: str, team_id: str | None 

298) -> tuple[Mapping[str, object], ...]: 

299 """``model`` targets deployments the way a request for it routes, else by ``litellm_params.model``.""" 

300 return _deployments_routed_by_name(model_list, model, team_id) or tuple( 

301 x for x in model_list if _deployment_model(x) == model 

302 ) 

303 

304 

305def _narrow_to_target( 

306 model_list: Sequence[Mapping[str, object]], model: str | None, model_id: str | None, team_id: str | None 

307) -> tuple[Mapping[str, object], ...]: 

308 """Narrow to the requested deployment. An id matching nothing keeps the whole list.""" 

309 if model_id is not None: 

310 by_id: Final = tuple(x for x in model_list if _deployment_id(x) == model_id) 

311 return by_id or tuple(model_list) 

312 if model is None: 

313 return tuple(model_list) 

314 return deployments_targeted_by_name(model_list, model, team_id) 

315 

316 

317def _is_strategy_router_deployment(litellm_params: Mapping[str, object]) -> bool: 

318 """True for strategy-router deployments.""" 

319 model: Final[object] = litellm_params.get("model", "") 

320 return isinstance(model, str) and classify_strategy_router_model(model) is not None 

321 

322 

323def _is_marker(deployment: Mapping[str, object]) -> bool: 

324 params: Final = deployment.get("litellm_params") 

325 return isinstance(params, Mapping) and _is_strategy_router_deployment(params) 

326 

327 

328def _deployment_id(deployment: Mapping[str, object]) -> str | None: 

329 info: Final = deployment.get("model_info") 

330 ident: Final = info.get("id") if isinstance(info, Mapping) else None 

331 return str(ident) if ident else None 

332 

333 

334def _resolved_deployment_ids(router: "Router", model_name: str) -> frozenset[str] | None: 

335 """Deployment ids backing `model_name`, or None when the name resolves to nothing. 

336 

337 `get_model_list` composes every channel the request path itself uses (exact name, 

338 model_group_alias, routing groups, wildcards); a mirror of any one channel would call a 

339 working tier broken. An alias whose target is gone resolves to nothing, which fails a 

340 request exactly like an unknown name. 

341 """ 

342 resolved: Final = router.get_model_list(model_name=model_name) 

343 if not resolved: 

344 return None 

345 return frozenset(ident for entry in resolved if (ident := _deployment_id(entry))) 

346 

347 

348def _dependency_failure( 

349 dependency: StrategyRouterDependency, 

350 router: "Router", 

351 unhealthy_ids: frozenset[str], 

352) -> str | None: 

353 """Why this dependency makes its router unable to serve, or None when it does not. 

354 

355 A name reds its router only when *every* deployment behind it is known unhealthy. One 

356 replica this run never judged, hidden from the caller or opted out of health checks, can 

357 still serve what the dead one drops, so partial evidence leaves the verdict green. 

358 """ 

359 resolved: Final = _resolved_deployment_ids(router, dependency.model_name) 

360 if resolved is None: 

361 return f"{dependency.role} model '{dependency.model_name}' matches no deployment on this proxy" 

362 if not resolved or not resolved <= unhealthy_ids: 

363 return None 

364 return f"{dependency.role} model '{dependency.model_name}' has no healthy deployment" 

365 

366 

367def _strategy_router_dependency_error( 

368 deployment: Mapping[str, object], 

369 router: "Router", 

370 unhealthy_ids: frozenset[str], 

371) -> str | None: 

372 """The first dependency fault that makes this router unable to serve, if any.""" 

373 params: Final = deployment.get("litellm_params") 

374 if not isinstance(params, Mapping): 

375 return None 

376 return next( 

377 ( 

378 failure 

379 for dependency in strategy_router_dependencies(params) 

380 if dependency.role != "evaluation" 

381 if (failure := _dependency_failure(dependency, router, unhealthy_ids)) 

382 ), 

383 None, 

384 ) 

385 

386 

387def _deployments_by_id( 

388 universe: Sequence[Mapping[str, object]], ids: frozenset[str] 

389) -> tuple[Mapping[str, object], ...]: 

390 """The deployments for `ids`, one row per id. 

391 

392 Reuses the requested set's own dedupe rule, so an alias that duplicates a row cannot get 

393 it probed twice or split a single id's verdict across two disagreeing results. 

394 """ 

395 matched: Final = tuple(d for d in universe if (uid := _deployment_id(d)) and uid in ids) 

396 return tuple(filter_deployments_by_id(model_list=matched)) 

397 

398 

399def _dependency_deployments_to_probe( 

400 checked: Sequence[Mapping[str, object]], 

401 universe: Sequence[Mapping[str, object]], 

402 router: "Router", 

403) -> tuple[Mapping[str, object], ...]: 

404 """Deployments backing the checked routers' dependencies that are not already checked. 

405 

406 Empty on a full-list run, which therefore gains no probe; it is the targeted 

407 `/health?model_id=<router>` call the dashboard makes per deployment that needs them, 

408 since a router's verdict is a statement about models the request never named. Drawn from 

409 `universe`, the caller's access-filtered list, so no deployment is probed that the caller 

410 was not already granted. Expansion follows routers through routers, one hop per round, 

411 because a child router's own models must be probed for the parent to fail; stopping when 

412 a round adds nothing is what makes a router cycle terminate. 

413 """ 

414 checked_ids: Final = frozenset(cid for d in checked if (cid := _deployment_id(d))) 

415 reached = checked_ids # rebind-ok: the sweep's cursor, one hop wider per round 

416 frontier = tuple(checked) # rebind-ok: the routers whose dependencies the next round expands 

417 for _ in range(len(universe)): 417 ↛ 432line 417 didn't jump to line 432 because the loop on line 417 didn't complete

418 names = frozenset( 

419 dependency.model_name 

420 for deployment in frontier 

421 if isinstance(params := deployment.get("litellm_params"), Mapping) 

422 for dependency in strategy_router_dependencies(params) 

423 if dependency.role != "evaluation" 

424 ) 

425 fresh_ids = ( 

426 frozenset(ident for name in names for ident in (_resolved_deployment_ids(router, name) or ())) - reached 

427 ) 

428 if not fresh_ids: 428 ↛ 430line 428 didn't jump to line 430 because the condition on line 428 was always true

429 break 

430 frontier = _deployments_by_id(universe, fresh_ids) 

431 reached = reached | fresh_ids 

432 return _deployments_by_id(universe, reached - checked_ids) 

433 

434 

435def _strategy_router_verdicts( 

436 healthy_endpoints: Sequence[Mapping[str, object]], 

437 unhealthy_endpoints: Sequence[Mapping[str, object]], 

438 checked: Sequence[Mapping[str, object]], 

439 router: "Router", 

440) -> Mapping[str, str]: 

441 """The dependency fault, per model id, for every strategy router that cannot serve. 

442 

443 A marker is filed healthy by `_run_model_health_check` returning `{}`, which says only 

444 that nothing was probed. This is where that placeholder becomes a verdict, derived from 

445 this run's own results rather than a re-probe or a cache that is empty unless 

446 `enable_health_check_routing` is on. A marker never fails a probe of its own, so verdicts 

447 settle over rounds, each feeding the last round's reds back in as unhealthy; without that 

448 the parent of a red child would stay green. Bounded by the marker count, which is what 

449 makes a router cycle terminate green rather than spin. 

450 """ 

451 by_id: Final = MappingProxyType({i: d for d in checked if (i := _deployment_id(d))}) 

452 markers: Final = MappingProxyType( 

453 { 

454 marker_id: by_id[marker_id] 

455 for endpoint in healthy_endpoints 

456 if isinstance(marker_id := endpoint.get("model_id"), str) and marker_id in by_id 

457 if _is_marker(by_id[marker_id]) 

458 } 

459 ) 

460 probe_failures: Final = frozenset( 

461 ident for endpoint in unhealthy_endpoints if isinstance(ident := endpoint.get("model_id"), str) 

462 ) 

463 settled: Mapping[str, str] = MappingProxyType({}) # rebind-ok: the fixed point, a round's verdicts at a time 

464 for _ in range(len(markers)): 464 ↛ 465line 464 didn't jump to line 465 because the loop on line 464 never started

465 fresh = MappingProxyType( 

466 { 

467 marker_id: error 

468 for marker_id, deployment in markers.items() 

469 if marker_id not in settled 

470 if (error := _strategy_router_dependency_error(deployment, router, probe_failures | frozenset(settled))) 

471 } 

472 ) 

473 if not fresh: 

474 break 

475 settled = MappingProxyType({**settled, **fresh}) 

476 return settled 

477 

478 

479def _finalize_strategy_router_endpoints( 

480 healthy_endpoints: Sequence[Mapping[str, object]], 

481 unhealthy_endpoints: Sequence[Mapping[str, object]], 

482 checked: Sequence[Mapping[str, object]], 

483 router: "Router | None", 

484 dependency_probes: Sequence[Mapping[str, object]], 

485) -> tuple[Sequence[Mapping[str, object]], Sequence[Mapping[str, object]]]: 

486 """Apply router verdicts, then drop the deployments probed only to reach them. 

487 

488 The probes exist to judge the routers that depend on them; reporting them would answer a 

489 targeted request with deployments the caller never asked about. 

490 """ 

491 verdicts: Final = ( 

492 _strategy_router_verdicts(healthy_endpoints, unhealthy_endpoints, checked, router) 

493 if router is not None 

494 else MappingProxyType({}) 

495 ) 

496 dropped: Final = frozenset(i for d in dependency_probes if (i := _deployment_id(d))) 

497 

498 def keep(endpoint: Mapping[str, object]) -> bool: 

499 model_id: Final = endpoint.get("model_id") 

500 return not (isinstance(model_id, str) and model_id in dropped) 

501 

502 def verdict_for(endpoint: Mapping[str, object]) -> str | None: 

503 model_id: Final = endpoint.get("model_id") 

504 return verdicts.get(model_id) if isinstance(model_id, str) else None 

505 

506 kept_healthy: Final = tuple(e for e in healthy_endpoints if keep(e)) 

507 return ( 

508 tuple(e for e in kept_healthy if verdict_for(e) is None), 

509 tuple(e for e in unhealthy_endpoints if keep(e)) 

510 + tuple( 

511 dict(e, error=error) # mutable-ok: the /health payload must stay a plain JSON-serializable dict 

512 for e in kept_healthy 

513 if (error := verdict_for(e)) is not None 

514 ), 

515 ) 

516 

517 

518async def _run_model_health_check(model: dict): 

519 litellm_params = model["litellm_params"] 

520 model_info: Final = model.get("model_info", {}) 

521 

522 if _is_strategy_router_deployment(litellm_params): 522 ↛ 523line 522 didn't jump to line 523 because the condition on line 522 was never true

523 return {} 

524 

525 mode: Final = _resolve_health_check_mode( 

526 model_info, 

527 litellm_params, # any-ok: untyped router config dict 

528 ) 

529 litellm_params = _update_litellm_params_for_health_check(model_info, litellm_params) 

530 timeout: Final = model_info.get("health_check_timeout") or HEALTH_CHECK_TIMEOUT_SECONDS 

531 

532 return await run_with_timeout( 

533 litellm.ahealth_check( 

534 litellm_params, 

535 mode=mode, 

536 prompt=DEFAULT_HEALTH_CHECK_PROMPT, 

537 input=["test from litellm"], 

538 ), 

539 timeout, 

540 ) 

541 

542 

543async def _run_health_checks_with_bounded_concurrency(models: list, concurrency_limit: int) -> tuple[list, int]: 

544 """ 

545 Run health checks with at most `concurrency_limit` active tasks. 

546 Preserves result ordering to match `models`. 

547 """ 

548 results: Final[list] = [None] * len(models) 

549 tasks_to_index: Final[dict[asyncio.Task, int]] = {} 

550 model_iter: Final = iter(enumerate(models)) 

551 peak_in_flight = 0 

552 

553 def _schedule_next() -> bool: 

554 nonlocal peak_in_flight 

555 try: 

556 idx, next_model = next(model_iter) 

557 except StopIteration: 

558 return False 

559 task: Final = asyncio.create_task(_run_model_health_check(next_model)) 

560 tasks_to_index[task] = idx 

561 peak_in_flight = max(peak_in_flight, len(tasks_to_index)) 

562 return True 

563 

564 for _ in range(min(concurrency_limit, len(models))): 

565 _schedule_next() 

566 

567 while tasks_to_index: 

568 done, _ = await asyncio.wait( 

569 set(tasks_to_index.keys()), 

570 return_when=asyncio.FIRST_COMPLETED, 

571 ) 

572 for task in done: 

573 idx = tasks_to_index.pop(task) 

574 try: 

575 results[idx] = task.result() 

576 except Exception as e: 

577 results[idx] = e 

578 _schedule_next() 

579 

580 return results, peak_in_flight 

581 

582 

583async def _perform_health_check( 

584 model_list: list, 

585 details: bool | None = True, 

586 max_concurrency: int | None = None, 

587 instrumentation_context: dict | None = None, 

588): 

589 """ 

590 Perform a health check for each model in the list. 

591 

592 max_concurrency: Optional limit on concurrent health check requests. 

593 """ 

594 

595 instrumentation_context = instrumentation_context or {} 

596 instrumentation_enabled: Final = bool(instrumentation_context.get("enabled", False)) 

597 cycle_id: Final = instrumentation_context.get("cycle_id", "unknown") 

598 source: Final = instrumentation_context.get("source", "unknown") 

599 

600 dispatch_mode = "unbounded" 

601 peak_in_flight = 0 

602 if isinstance(max_concurrency, int) and max_concurrency > 0: 602 ↛ 603line 602 didn't jump to line 603 because the condition on line 602 was never true

603 dispatch_mode = "bounded" 

604 results, peak_in_flight = await _run_health_checks_with_bounded_concurrency(model_list, max_concurrency) 

605 else: 

606 tasks: Final = [asyncio.create_task(_run_model_health_check(model)) for model in model_list] 

607 peak_in_flight = len(tasks) 

608 results = await asyncio.gather(*tasks, return_exceptions=True) 

609 

610 if instrumentation_enabled: 610 ↛ 611line 610 didn't jump to line 611 because the condition on line 610 was never true

611 logger.debug( 

612 "health_check_dispatch_summary source=%s cycle_id=%s mode=%s model_count=%d max_concurrency=%s peak_in_flight=%d thread_count=%d rss_mb=%s", 

613 source, 

614 cycle_id, 

615 dispatch_mode, 

616 len(model_list), 

617 max_concurrency, 

618 peak_in_flight, 

619 threading.active_count(), 

620 _rss_mb_for_log(), 

621 ) 

622 

623 healthy_endpoints: Final = [] 

624 unhealthy_endpoints: Final = [] 

625 # Exceptions keyed by model_id; returned separately so callers can use 

626 # them for cooldown integration without risking JSON-serialization errors 

627 # in the /health response. 

628 exceptions_by_model_id: Final[dict] = {} 

629 

630 for is_healthy, model in zip(results, model_list): 

631 litellm_params = model["litellm_params"] 

632 _model_id = (model.get("model_info") or {}).get("id") 

633 

634 if isinstance(is_healthy, dict) and "error" not in is_healthy: 634 ↛ 635line 634 didn't jump to line 635 because the condition on line 634 was never true

635 cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details) 

636 if _model_id: 

637 cleaned["model_id"] = _model_id 

638 healthy_endpoints.append(cleaned) 

639 elif isinstance(is_healthy, dict): 639 ↛ 651line 639 didn't jump to line 651 because the condition on line 639 was always true

640 cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details) 

641 if _model_id: 641 ↛ 649line 641 didn't jump to line 649 because the condition on line 641 was always true

642 cleaned["model_id"] = _model_id 

643 if "exception" in is_healthy: 643 ↛ 649line 643 didn't jump to line 649 because the condition on line 643 was always true

644 exc = is_healthy["exception"] 

645 exceptions_by_model_id[_model_id] = exc 

646 # Store integer status code so shared-cache readers can 

647 # reconstruct the transient-error filter without the exception object. 

648 cleaned["exception_status"] = getattr(exc, "status_code", 500) 

649 unhealthy_endpoints.append(cleaned) 

650 else: 

651 cleaned = _clean_endpoint_data(litellm_params, details) 

652 if _model_id: 

653 cleaned["model_id"] = _model_id 

654 if isinstance(is_healthy, Exception): 

655 exceptions_by_model_id[_model_id] = is_healthy 

656 cleaned["exception_status"] = getattr(is_healthy, "status_code", 500) 

657 unhealthy_endpoints.append(cleaned) 

658 

659 return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id 

660 

661 

662def build_deployment_health_states( 

663 healthy_endpoints: list, 

664 unhealthy_endpoints: list, 

665) -> dict: 

666 """ 

667 Build a dict mapping deployment_id -> DeploymentHealthStateValue from 

668 health check endpoint results. 

669 

670 Each endpoint dict includes a 'model_id' field (added by _perform_health_check) 

671 that maps back to the deployment's model_info.id. 

672 

673 Used by the background health check loop to feed health state into 

674 the router's DeploymentHealthCache for health-check-driven routing. 

675 """ 

676 now: Final = time.time() 

677 states: Final[dict] = {} 

678 

679 for ep in healthy_endpoints: 

680 model_id = ep.get("model_id") 

681 if model_id: 

682 states[model_id] = { 

683 "is_healthy": True, 

684 "timestamp": now, 

685 "reason": "", 

686 } 

687 

688 for ep in unhealthy_endpoints: 

689 model_id = ep.get("model_id") 

690 if model_id: 

691 states[model_id] = { 

692 "is_healthy": False, 

693 "timestamp": now, 

694 "reason": "background_health_check_failed", 

695 } 

696 

697 return states 

698 

699 

700def _deployment_model_string_for_health_check(litellm_params: dict) -> str: 

701 """Deployment model from litellm_params (before Bedrock rewrite). 

702 

703 Used for reasoning vs non-reasoning max_tokens and wildcard detection only. 

704 Does not use ``health_check_model``; that override applies later to the request. 

705 """ 

706 return litellm_params.get("model") or "" 

707 

708 

709def _health_check_deployment_is_wildcard(litellm_params: dict) -> bool: 

710 return "*" in _deployment_model_string_for_health_check(litellm_params) 

711 

712 

713def _resolve_health_check_max_tokens(model_info: dict, litellm_params: dict) -> int | None: 

714 """ 

715 Pick max_tokens for the health check request. 

716 

717 Priority: 

718 1. model_info.health_check_max_tokens (explicit override) 

719 2. For non-wildcard routes: health_check_max_tokens_reasoning / _non_reasoning 

720 from model_info based on litellm.supports_reasoning(litellm_params["model"]) 

721 3. For non-wildcard reasoning routes: BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING 

722 from env (if set) 

723 4. BACKGROUND_HEALTH_CHECK_MAX_TOKENS (global, any route including wildcards) 

724 5. Non-wildcard default: 16 

725 6. Wildcard and nothing from (1)(4): leave unset (caller omits max_tokens) 

726 """ 

727 explicit: Final = model_info.get("health_check_max_tokens", None) 

728 if explicit is not None: 728 ↛ 729line 728 didn't jump to line 729 because the condition on line 728 was never true

729 return int(explicit) 

730 

731 is_wildcard: Final = _health_check_deployment_is_wildcard(litellm_params) 

732 deployment_model: Final = _deployment_model_string_for_health_check(litellm_params) 

733 

734 if not is_wildcard: 734 ↛ 749line 734 didn't jump to line 749 because the condition on line 734 was always true

735 try: 

736 is_reasoning = litellm.supports_reasoning(deployment_model) 

737 except Exception: 

738 is_reasoning = False 

739 tokens_reasoning: Final = model_info.get("health_check_max_tokens_reasoning", None) 

740 tokens_non_reasoning: Final = model_info.get("health_check_max_tokens_non_reasoning", None) 

741 if tokens_reasoning is not None or tokens_non_reasoning is not None: 741 ↛ 742line 741 didn't jump to line 742 because the condition on line 741 was never true

742 if is_reasoning and tokens_reasoning is not None: 

743 return int(tokens_reasoning) 

744 if not is_reasoning and tokens_non_reasoning is not None: 

745 return int(tokens_non_reasoning) 

746 if is_reasoning and BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING is not None: 746 ↛ 747line 746 didn't jump to line 747 because the condition on line 746 was never true

747 return int(BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING) 

748 

749 if BACKGROUND_HEALTH_CHECK_MAX_TOKENS is not None: 749 ↛ 750line 749 didn't jump to line 750 because the condition on line 749 was never true

750 return int(BACKGROUND_HEALTH_CHECK_MAX_TOKENS) 

751 

752 if not is_wildcard: 752 ↛ 755line 752 didn't jump to line 755 because the condition on line 752 was always true

753 return 16 

754 

755 return None 

756 

757 

758def _update_litellm_params_for_health_check(model_info: dict, litellm_params: dict) -> dict: 

759 """ 

760 Update the litellm params for health check. 

761 

762 - merges `model_info.health_check_params` into the probe request, so a deployment whose provider 

763 requires a payload field litellm does not synthesize (e.g. `mediaSource` for Bedrock TwelveLabs 

764 Pegasus) can supply it. The dedicated knobs below are applied afterwards and win on conflict. 

765 - gets a short `messages` param for health check 

766 - adds a bounded `max_tokens` when the deployment is a chat-style mode 

767 (`chat`, `completion`, `responses`) or the operator explicitly opts in 

768 via `model_info.health_check_supports_max_tokens`. Non-chat endpoints 

769 (image, embedding, audio_*, rerank, video, ocr, search, moderation, ...) 

770 reject unknown fields with 400 "Unknown parameter: 'max_tokens'". 

771 - updates the `model` param with the `health_check_model` if it exists Doc: https://docs.litellm.ai/docs/proxy/health#wildcard-routes 

772 - updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models 

773 - for Bedrock models with region routing (bedrock/region/model), strips the litellm routing prefix but preserves the model ID, and pins `custom_llm_provider` to `bedrock` (only when the deployment hasn't already set one, so an explicit `bedrock_converse` survives) so the bare model id still resolves to the provider (e.g. cross-region ids like `us.cohere.embed-v4:0`) 

774 """ 

775 mode: Final = _resolve_health_check_mode( 

776 model_info, 

777 litellm_params, # any-ok: untyped router config dict 

778 ) 

779 _health_check_params: Final = model_info.get("health_check_params", None) 

780 if isinstance(_health_check_params, dict): 780 ↛ 781line 780 didn't jump to line 781 because the condition on line 780 was never true

781 litellm_params.update(_health_check_params) 

782 elif _health_check_params is not None: 782 ↛ 783line 782 didn't jump to line 783 because the condition on line 782 was never true

783 logger.warning( 

784 "health_check_params for model %s is a %s, expected a dict. Ignoring it.", 

785 litellm_params.get("model"), 

786 type(_health_check_params).__name__, 

787 ) 

788 

789 litellm_params["messages"] = _get_random_llm_message() 

790 if _should_inject_health_check_max_tokens( 790 ↛ 799line 790 didn't jump to line 799 because the condition on line 790 was always true

791 model_info, 

792 mode, # any-ok: untyped router config dict 

793 ): 

794 _resolved_max_tokens: Final = _resolve_health_check_max_tokens(model_info, litellm_params) 

795 if _resolved_max_tokens is not None: 795 ↛ 799line 795 didn't jump to line 799 because the condition on line 795 was always true

796 litellm_params["max_tokens"] = _resolved_max_tokens 

797 

798 # Per-model reasoning effort for health checks only (e.g. reasoning_effort=none). 

799 if mode in _HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: 799 ↛ 804line 799 didn't jump to line 804 because the condition on line 799 was always true

800 _hc_reasoning_effort: Final = model_info.get("health_check_reasoning_effort", None) 

801 if _hc_reasoning_effort is not None: 801 ↛ 802line 801 didn't jump to line 802 because the condition on line 801 was never true

802 litellm_params["reasoning_effort"] = _hc_reasoning_effort 

803 

804 _health_check_model: Final = model_info.get("health_check_model", None) 

805 if _health_check_model is not None: 805 ↛ 806line 805 didn't jump to line 806 because the condition on line 805 was never true

806 litellm_params["model"] = _health_check_model 

807 if mode == "audio_speech": 807 ↛ 808line 807 didn't jump to line 808 because the condition on line 807 was never true

808 litellm_params["voice"] = model_info.get("health_check_voice", "alloy") 

809 

810 # Handle Bedrock region routing format: bedrock/region/model 

811 # This is needed because health checks bypass get_llm_provider() for the model param 

812 # Issue #15807: Without this, health checks send "region/model" as the model ID to AWS 

813 # which causes: "bedrock-runtime.../model/us-west-2/mistral.../invoke" (region in model ID) 

814 # 

815 # However, we must preserve cross-region inference profile prefixes like "us.", "eu.", etc. 

816 # Issue: Stripping these breaks AWS requirement for inference profile IDs 

817 # 

818 # Must also preserve route prefixes (converse/, invoke/) and handlers (llama/, deepseek_r1/, etc.) 

819 if litellm_params["model"].startswith("bedrock/"): 819 ↛ 820line 819 didn't jump to line 820 because the condition on line 819 was never true

820 from litellm.llms.bedrock.common_utils import BedrockModelInfo 

821 

822 model = litellm_params["model"] 

823 # Strip only the bedrock/ prefix (preserve routes like converse/, invoke/) 

824 model = model.removeprefix("bedrock/") # len("bedrock/") = 8 

825 

826 # Now check for region routing and strip it if present 

827 # Need to handle formats like: 

828 # - "us-west-2/model" → "model" 

829 # - "converse/us-west-2/model" → "converse/model" 

830 # - "llama/arn:..." → "llama/arn:..." (preserve handler) 

831 # 

832 # Strategy: Check each path segment, remove regions, preserve everything else 

833 parts: Final = model.split("/") 

834 filtered_parts: Final = [] 

835 

836 for part in parts: 

837 # Skip AWS regions, keep everything else 

838 if part not in BedrockModelInfo.all_global_regions: 

839 filtered_parts.append(part) 

840 

841 model = "/".join(filtered_parts) 

842 litellm_params["model"] = model 

843 if not litellm_params.get("custom_llm_provider"): # any-ok: untyped router dict 

844 litellm_params["custom_llm_provider"] = ( # any-ok: untyped router dict 

845 "bedrock" 

846 ) 

847 

848 return litellm_params 

849 

850 

851async def perform_health_check( 

852 model_list: list, 

853 model: str | None = None, 

854 cli_model: str | None = None, 

855 details: bool | None = True, 

856 model_id: str | None = None, 

857 max_concurrency: int | None = None, 

858 instrumentation_context: dict | None = None, 

859 health_check_skip_disabled_background_models: bool = False, 

860 router: "Router | None" = None, 

861 team_id: str | None = None, 

862): 

863 """ 

864 Perform a health check on the system. 

865 

866 When model_id is provided, only the deployment with that id is checked 

867 (so models that share the same name but have different ids are checked separately). 

868 When model (name) is provided, the deployments a request for that name from the 

869 caller (``team_id``) would route to are checked: the caller's team copies published 

870 under that name, else the deployments named that way, else a public name that only 

871 another team's deployment carries, else the deployments whose ``litellm_params.model`` 

872 is that string. 

873 

874 When ``health_check_skip_disabled_background_models`` is True (via 

875 ``general_settings.health_check_skip_disabled_background_models``), deployments 

876 with ``model_info.disable_background_health_check: true`` are omitted from 

877 this run (including targeted ``/health`` queries), consistent with the 

878 background health loop. 

879 

880 Returns: 

881 (bool): True if the health check passes, False otherwise. 

882 """ 

883 instrumentation_context = instrumentation_context or {} 

884 instrumentation_enabled: Final = bool(instrumentation_context.get("enabled", False)) 

885 cycle_id: Final = instrumentation_context.get("cycle_id", "unknown") 

886 source: Final = instrumentation_context.get("source", "unknown") 

887 

888 if not model_list: 888 ↛ 889line 888 didn't jump to line 889 because the condition on line 888 was never true

889 if cli_model: 

890 model_list = [{"model_name": cli_model, "litellm_params": {"model": cli_model}}] 

891 else: 

892 if instrumentation_enabled: 

893 logger.debug( 

894 "health_check_cycle_skipped source=%s cycle_id=%s reason=no_models", 

895 source, 

896 cycle_id, 

897 ) 

898 return [], [], {} 

899 

900 cycle_start_time: Final = time.monotonic() 

901 requested_model_count: Final = len(model_list) 

902 skip_disabled: Final = health_check_skip_disabled_background_models 

903 narrowed: Final = _health_check_eligible(_narrow_to_target(model_list, model, model_id, team_id), skip_disabled) 

904 if not narrowed: 

905 if instrumentation_enabled: 905 ↛ 906line 905 didn't jump to line 906 because the condition on line 905 was never true

906 logger.debug( 

907 "health_check_cycle_skipped source=%s cycle_id=%s reason=no_models_after_filter", 

908 source, 

909 cycle_id, 

910 ) 

911 return [], [], {} 

912 

913 post_filter_model_count: Final = len(narrowed) 

914 requested: Final = filter_deployments_by_id(model_list=narrowed) 

915 deduped_model_count: Final = len(requested) 

916 

917 dependency_probes: Final = ( 

918 _dependency_deployments_to_probe(requested, _health_check_eligible(model_list, skip_disabled), router) 

919 if router is not None 

920 else () 

921 ) 

922 checked: Final = requested + list(dependency_probes) # mutable-ok: _perform_health_check takes a list 

923 

924 if instrumentation_enabled: 924 ↛ 925line 924 didn't jump to line 925 because the condition on line 924 was never true

925 logger.debug( 

926 "health_check_cycle_start source=%s cycle_id=%s requested_model_count=%d post_model_filter_count=%d deduped_model_count=%d max_concurrency=%s thread_count=%d rss_mb=%s", 

927 source, 

928 cycle_id, 

929 requested_model_count, 

930 post_filter_model_count, 

931 deduped_model_count, 

932 max_concurrency, 

933 threading.active_count(), 

934 _rss_mb_for_log(), 

935 ) 

936 

937 try: 

938 ( 

939 probed_healthy, 

940 probed_unhealthy, 

941 exceptions_by_model_id, 

942 ) = await _perform_health_check( 

943 checked, 

944 details, 

945 max_concurrency=max_concurrency, 

946 instrumentation_context=instrumentation_context, 

947 ) 

948 graded_healthy, graded_unhealthy = _finalize_strategy_router_endpoints( 

949 probed_healthy, probed_unhealthy, checked, router, dependency_probes 

950 ) 

951 healthy_endpoints: Final = list(graded_healthy) 

952 unhealthy_endpoints: Final = list(graded_unhealthy) 

953 except Exception: 

954 if instrumentation_enabled: 

955 logger.exception( 

956 "health_check_cycle_failed source=%s cycle_id=%s model_count=%d duration_ms=%.2f thread_count=%d rss_mb=%s", 

957 source, 

958 cycle_id, 

959 deduped_model_count, 

960 (time.monotonic() - cycle_start_time) * 1000, 

961 threading.active_count(), 

962 _rss_mb_for_log(), 

963 ) 

964 raise 

965 

966 if instrumentation_enabled: 966 ↛ 967line 966 didn't jump to line 967 because the condition on line 966 was never true

967 logger.debug( 

968 "health_check_cycle_complete source=%s cycle_id=%s model_count=%d healthy_count=%d unhealthy_count=%d duration_ms=%.2f thread_count=%d rss_mb=%s", 

969 source, 

970 cycle_id, 

971 deduped_model_count, 

972 len(healthy_endpoints), 

973 len(unhealthy_endpoints), 

974 (time.monotonic() - cycle_start_time) * 1000, 

975 threading.active_count(), 

976 _rss_mb_for_log(), 

977 ) 

978 

979 return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id