Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/db/proxy_worker_heartbeat.py: 81%
45 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Live proxy worker census, one row per worker process.
4Every uvicorn worker upserts its own row on a fixed heartbeat, so counting
5rows with a recent heartbeat answers "how many workers share this database?"
6without any coordination. The Admin UI's "no Redis" banner uses that count to
7hide itself for deployments that are provably a single worker, where per-worker
8rate limits, budgets, and router state are already global. All timestamps are
9written and compared with the database's own clock, so pods with skewed clocks
10still agree.
11"""
13from __future__ import annotations
15import socket
16from typing import TYPE_CHECKING, Final
18from pydantic import TypeAdapter
19from typing_extensions import ReadOnly, TypedDict
21from litellm._logging import verbose_proxy_logger
22from litellm._uuid import uuid
23from litellm.proxy.db.routing_prisma_wrapper import RoutingPrismaWrapper
25if TYPE_CHECKING: 25 ↛ 26line 25 didn't jump to line 26 because the condition on line 25 was never true
26 from litellm.proxy.utils import PrismaClient
28PROXY_WORKER_HEARTBEAT_INTERVAL_SECONDS: Final = 60
29PROXY_WORKER_LIVENESS_WINDOW_SECONDS: Final = 3 * PROXY_WORKER_HEARTBEAT_INTERVAL_SECONDS
30STALE_ROW_RETENTION_SECONDS: Final = 3600
32BEAT_SQL: Final = """
33INSERT INTO "LiteLLM_ProxyWorkerHeartbeat" (worker_id, hostname, last_heartbeat_at)
34VALUES ($1, $2, NOW())
35ON CONFLICT (worker_id) DO UPDATE SET last_heartbeat_at = NOW()
36"""
38PRUNE_SQL: Final = """
39DELETE FROM "LiteLLM_ProxyWorkerHeartbeat"
40WHERE last_heartbeat_at < NOW() - make_interval(secs => $1)
41"""
43COUNT_SQL: Final = """
44SELECT COUNT(*)::int AS live_workers FROM "LiteLLM_ProxyWorkerHeartbeat"
45WHERE last_heartbeat_at > NOW() - make_interval(secs => $1)
46"""
48DEREGISTER_SQL: Final = """
49DELETE FROM "LiteLLM_ProxyWorkerHeartbeat" WHERE worker_id = $1
50"""
53class _LiveWorkerCountRow(TypedDict):
54 live_workers: ReadOnly[int]
57_COUNT_ROWS_ADAPTER: Final = TypeAdapter(tuple[_LiveWorkerCountRow, ...])
60class ProxyWorkerHeartbeat:
61 def __init__(self, prisma_client: PrismaClient, worker_id: str | None = None) -> None:
62 self.prisma_client: Final = prisma_client
63 self.worker_id: Final[str] = worker_id or str(uuid.uuid4())
64 self.hostname: Final = socket.gethostname()
66 async def beat(self) -> None:
67 try:
68 await self.prisma_client.db.execute_raw(BEAT_SQL, self.worker_id, self.hostname)
69 await self.prisma_client.db.execute_raw(PRUNE_SQL, STALE_ROW_RETENTION_SECONDS)
70 except Exception as beat_err: # noqa: BLE001 # a missed heartbeat must never take down the worker
71 verbose_proxy_logger.debug("Proxy worker heartbeat write failed: %s", beat_err)
73 async def deregister(self) -> None:
74 try:
75 await self.prisma_client.db.execute_raw(DEREGISTER_SQL, self.worker_id)
76 except Exception as deregister_err: # noqa: BLE001 # best-effort cleanup; the liveness window ages the row out anyway
77 verbose_proxy_logger.debug("Proxy worker heartbeat deregister failed: %s", deregister_err)
80async def count_live_proxy_workers(prisma_client: PrismaClient) -> int | None:
81 """
82 The number of workers with a recent heartbeat, or None when the database
83 cannot answer. Callers must treat None as "unknown", not as zero. Always
84 counts on the primary: a lagging read replica must never undercount.
85 """
86 try:
87 db: Final = prisma_client.db
88 primary_db: Final = db.writer if isinstance(db, RoutingPrismaWrapper) else db
89 rows: Final = await primary_db.query_raw(COUNT_SQL, PROXY_WORKER_LIVENESS_WINDOW_SECONDS)
90 return _COUNT_ROWS_ADAPTER.validate_python(rows)[0]["live_workers"]
91 except Exception as count_err: # noqa: BLE001 # an unknown count must degrade to "warn", never to a 503
92 verbose_proxy_logger.debug("Live proxy worker count unavailable: %s", count_err)
93 return None