Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/shutdown/graceful_shutdown_manager.py: 64%
69 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Application-level graceful shutdown coordination for the LiteLLM proxy.
4Kubernetes terminates a pod by sending ``SIGTERM`` and, after
5``terminationGracePeriodSeconds``, ``SIGKILL``. By default LiteLLM delegates
6the signal to uvicorn and tears down immediately, dropping any in-flight
7requests (streaming, batch inference, long-lived calls).
9A fixed ``preStop`` sleep can not solve this: it has to be sized for the
10*worst-case* request, so it either wastes time on every routine shutdown or is
11too short for a long-running request. This manager instead drains based on the
12*actual* in-flight request counter (already tracked by
13``InFlightRequestsMiddleware``), so a pod terminates as soon as its real
14in-flight work is done — and never waits longer than ``GRACEFUL_SHUTDOWN_TIMEOUT``.
16The state is process-scoped (class-level), matching the per-uvicorn-worker
17granularity of ``InFlightRequestsMiddleware``.
18"""
20import asyncio
21import os
22import time
23from collections.abc import Callable
24from typing import Final
26from litellm._logging import verbose_proxy_logger
27from litellm.proxy.middleware.in_flight_requests_middleware import (
28 get_in_flight_requests,
29)
31# Keep below terminationGracePeriodSeconds so the process exits before SIGKILL.
32DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT: Final = 30.0
33_DRAIN_POLL_INTERVAL: Final = 0.1
34_DRAIN_LOG_INTERVAL: Final = 5.0
37class GracefulShutdownManager:
38 """
39 Process-scoped singleton that tracks whether the worker is draining and
40 blocks until in-flight requests reach zero (or a timeout elapses).
41 """
43 _is_shutting_down: bool = False
44 _shutdown_started_at: float | None = None
45 _drain_performed: bool = False
47 @classmethod
48 def is_shutting_down(cls) -> bool:
49 """Whether this worker has begun graceful shutdown."""
50 return cls._is_shutting_down
52 @classmethod
53 def get_timeout(cls) -> float:
54 """
55 Read GRACEFUL_SHUTDOWN_TIMEOUT (seconds) from the environment on each
56 call so deployments can tune it without code changes. Falls back to the
57 default on an unset or malformed value.
58 """
59 raw: Final = os.getenv("GRACEFUL_SHUTDOWN_TIMEOUT")
60 if raw is None: 60 ↛ 62line 60 didn't jump to line 62 because the condition on line 60 was always true
61 return DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT
62 try:
63 return float(raw)
64 except (TypeError, ValueError):
65 verbose_proxy_logger.warning(
66 "GRACEFUL_SHUTDOWN_TIMEOUT=%r is not a number; using default %ss",
67 raw,
68 DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT,
69 )
70 return DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT
72 @classmethod
73 def start_shutdown(cls) -> None:
74 """
75 Mark the worker as draining. Idempotent — repeated calls (e.g. SIGTERM
76 followed by a preStop hit on /health/drain) do not reset the clock.
77 """
78 if cls._is_shutting_down: 78 ↛ 79line 78 didn't jump to line 79 because the condition on line 78 was never true
79 return
80 cls._is_shutting_down = True
81 cls._shutdown_started_at = time.monotonic()
82 verbose_proxy_logger.info(
83 "graceful_shutdown_started in_flight_requests=%s",
84 get_in_flight_requests(),
85 )
87 @classmethod
88 async def wait_for_drain(
89 cls,
90 timeout: float | None = None,
91 exclude_self: bool = False,
92 count_fn: Callable[[], int] | None = None,
93 poll_interval: float = _DRAIN_POLL_INTERVAL,
94 log_interval: float = _DRAIN_LOG_INTERVAL,
95 ) -> int:
96 """
97 Poll the in-flight request counter until it reaches the drain target or
98 ``timeout`` seconds elapse.
100 Args:
101 timeout: Max seconds to wait. Defaults to ``get_timeout()``.
102 exclude_self: When the caller is itself an in-flight HTTP request
103 (the /health/drain endpoint), set this so the caller's own
104 request is not counted as outstanding work.
105 count_fn: Source of the current in-flight count. Defaults to the
106 live ``InFlightRequestsMiddleware`` counter; injectable for tests.
107 poll_interval: Seconds between counter polls.
108 log_interval: Minimum seconds between ``drain_waiting`` log lines.
110 Returns:
111 Number of requests that drained while waiting (>= 0).
112 """
113 # A preStop /health/drain hook and the lifespan SIGTERM handler both
114 # drain; once one has run, the other must not wait again, otherwise the
115 # effective window is 2x the timeout and terminationGracePeriodSeconds
116 # has to be doubled to avoid a mid-drain SIGKILL.
117 if cls._drain_performed: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true
118 return 0
119 cls._drain_performed = True
121 if timeout is None: 121 ↛ 123line 121 didn't jump to line 123 because the condition on line 121 was always true
122 timeout = cls.get_timeout()
123 if count_fn is None: 123 ↛ 128line 123 didn't jump to line 128 because the condition on line 123 was always true
124 count_fn = get_in_flight_requests
126 # The /health/drain HTTP request flows through InFlightRequestsMiddleware
127 # and so counts itself; treat <=1 as "drained" in that case.
128 target: Final = 1 if exclude_self else 0
130 start: Final = time.monotonic()
131 initial: Final = count_fn()
132 last_log = start
134 if timeout <= 0: 134 ↛ 135line 134 didn't jump to line 135 because the condition on line 134 was never true
135 return max(0, initial - target)
137 while True:
138 current = count_fn()
139 if current <= target: 139 ↛ 148line 139 didn't jump to line 148 because the condition on line 139 was always true
140 drained = max(0, initial - current)
141 verbose_proxy_logger.info(
142 "graceful_shutdown_complete drained_requests=%s elapsed_s=%.2f",
143 drained,
144 time.monotonic() - start,
145 )
146 return drained
148 elapsed = time.monotonic() - start
149 if elapsed >= timeout:
150 verbose_proxy_logger.warning(
151 "graceful_shutdown_timeout in_flight_requests=%s elapsed_s=%.2f "
152 "timeout_s=%s — proceeding with teardown",
153 current,
154 elapsed,
155 timeout,
156 )
157 return max(0, initial - current)
159 now = time.monotonic()
160 if now - last_log >= log_interval:
161 verbose_proxy_logger.info(
162 "drain_waiting in_flight_requests=%s elapsed_s=%.2f",
163 current,
164 elapsed,
165 )
166 last_log = now
168 await asyncio.sleep(poll_interval)
170 @classmethod
171 def reset(cls) -> None:
172 """Reset state. Intended for use in tests."""
173 cls._is_shutting_down = False
174 cls._shutdown_started_at = None
175 cls._drain_performed = False