Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/shutdown/graceful_shutdown_manager.py: 64%

69 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Application-level graceful shutdown coordination for the LiteLLM proxy. 

3 

4Kubernetes terminates a pod by sending ``SIGTERM`` and, after 

5``terminationGracePeriodSeconds``, ``SIGKILL``. By default LiteLLM delegates 

6the signal to uvicorn and tears down immediately, dropping any in-flight 

7requests (streaming, batch inference, long-lived calls). 

8 

9A fixed ``preStop`` sleep can not solve this: it has to be sized for the 

10*worst-case* request, so it either wastes time on every routine shutdown or is 

11too short for a long-running request. This manager instead drains based on the 

12*actual* in-flight request counter (already tracked by 

13``InFlightRequestsMiddleware``), so a pod terminates as soon as its real 

14in-flight work is done — and never waits longer than ``GRACEFUL_SHUTDOWN_TIMEOUT``. 

15 

16The state is process-scoped (class-level), matching the per-uvicorn-worker 

17granularity of ``InFlightRequestsMiddleware``. 

18""" 

19 

20import asyncio 

21import os 

22import time 

23from collections.abc import Callable 

24from typing import Final 

25 

26from litellm._logging import verbose_proxy_logger 

27from litellm.proxy.middleware.in_flight_requests_middleware import ( 

28 get_in_flight_requests, 

29) 

30 

31# Keep below terminationGracePeriodSeconds so the process exits before SIGKILL. 

32DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT: Final = 30.0 

33_DRAIN_POLL_INTERVAL: Final = 0.1 

34_DRAIN_LOG_INTERVAL: Final = 5.0 

35 

36 

37class GracefulShutdownManager: 

38 """ 

39 Process-scoped singleton that tracks whether the worker is draining and 

40 blocks until in-flight requests reach zero (or a timeout elapses). 

41 """ 

42 

43 _is_shutting_down: bool = False 

44 _shutdown_started_at: float | None = None 

45 _drain_performed: bool = False 

46 

47 @classmethod 

48 def is_shutting_down(cls) -> bool: 

49 """Whether this worker has begun graceful shutdown.""" 

50 return cls._is_shutting_down 

51 

52 @classmethod 

53 def get_timeout(cls) -> float: 

54 """ 

55 Read GRACEFUL_SHUTDOWN_TIMEOUT (seconds) from the environment on each 

56 call so deployments can tune it without code changes. Falls back to the 

57 default on an unset or malformed value. 

58 """ 

59 raw: Final = os.getenv("GRACEFUL_SHUTDOWN_TIMEOUT") 

60 if raw is None: 60 ↛ 62line 60 didn't jump to line 62 because the condition on line 60 was always true

61 return DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT 

62 try: 

63 return float(raw) 

64 except (TypeError, ValueError): 

65 verbose_proxy_logger.warning( 

66 "GRACEFUL_SHUTDOWN_TIMEOUT=%r is not a number; using default %ss", 

67 raw, 

68 DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT, 

69 ) 

70 return DEFAULT_GRACEFUL_SHUTDOWN_TIMEOUT 

71 

72 @classmethod 

73 def start_shutdown(cls) -> None: 

74 """ 

75 Mark the worker as draining. Idempotent — repeated calls (e.g. SIGTERM 

76 followed by a preStop hit on /health/drain) do not reset the clock. 

77 """ 

78 if cls._is_shutting_down: 78 ↛ 79line 78 didn't jump to line 79 because the condition on line 78 was never true

79 return 

80 cls._is_shutting_down = True 

81 cls._shutdown_started_at = time.monotonic() 

82 verbose_proxy_logger.info( 

83 "graceful_shutdown_started in_flight_requests=%s", 

84 get_in_flight_requests(), 

85 ) 

86 

87 @classmethod 

88 async def wait_for_drain( 

89 cls, 

90 timeout: float | None = None, 

91 exclude_self: bool = False, 

92 count_fn: Callable[[], int] | None = None, 

93 poll_interval: float = _DRAIN_POLL_INTERVAL, 

94 log_interval: float = _DRAIN_LOG_INTERVAL, 

95 ) -> int: 

96 """ 

97 Poll the in-flight request counter until it reaches the drain target or 

98 ``timeout`` seconds elapse. 

99 

100 Args: 

101 timeout: Max seconds to wait. Defaults to ``get_timeout()``. 

102 exclude_self: When the caller is itself an in-flight HTTP request 

103 (the /health/drain endpoint), set this so the caller's own 

104 request is not counted as outstanding work. 

105 count_fn: Source of the current in-flight count. Defaults to the 

106 live ``InFlightRequestsMiddleware`` counter; injectable for tests. 

107 poll_interval: Seconds between counter polls. 

108 log_interval: Minimum seconds between ``drain_waiting`` log lines. 

109 

110 Returns: 

111 Number of requests that drained while waiting (>= 0). 

112 """ 

113 # A preStop /health/drain hook and the lifespan SIGTERM handler both 

114 # drain; once one has run, the other must not wait again, otherwise the 

115 # effective window is 2x the timeout and terminationGracePeriodSeconds 

116 # has to be doubled to avoid a mid-drain SIGKILL. 

117 if cls._drain_performed: 117 ↛ 118line 117 didn't jump to line 118 because the condition on line 117 was never true

118 return 0 

119 cls._drain_performed = True 

120 

121 if timeout is None: 121 ↛ 123line 121 didn't jump to line 123 because the condition on line 121 was always true

122 timeout = cls.get_timeout() 

123 if count_fn is None: 123 ↛ 128line 123 didn't jump to line 128 because the condition on line 123 was always true

124 count_fn = get_in_flight_requests 

125 

126 # The /health/drain HTTP request flows through InFlightRequestsMiddleware 

127 # and so counts itself; treat <=1 as "drained" in that case. 

128 target: Final = 1 if exclude_self else 0 

129 

130 start: Final = time.monotonic() 

131 initial: Final = count_fn() 

132 last_log = start 

133 

134 if timeout <= 0: 134 ↛ 135line 134 didn't jump to line 135 because the condition on line 134 was never true

135 return max(0, initial - target) 

136 

137 while True: 

138 current = count_fn() 

139 if current <= target: 139 ↛ 148line 139 didn't jump to line 148 because the condition on line 139 was always true

140 drained = max(0, initial - current) 

141 verbose_proxy_logger.info( 

142 "graceful_shutdown_complete drained_requests=%s elapsed_s=%.2f", 

143 drained, 

144 time.monotonic() - start, 

145 ) 

146 return drained 

147 

148 elapsed = time.monotonic() - start 

149 if elapsed >= timeout: 

150 verbose_proxy_logger.warning( 

151 "graceful_shutdown_timeout in_flight_requests=%s elapsed_s=%.2f " 

152 "timeout_s=%s — proceeding with teardown", 

153 current, 

154 elapsed, 

155 timeout, 

156 ) 

157 return max(0, initial - current) 

158 

159 now = time.monotonic() 

160 if now - last_log >= log_interval: 

161 verbose_proxy_logger.info( 

162 "drain_waiting in_flight_requests=%s elapsed_s=%.2f", 

163 current, 

164 elapsed, 

165 ) 

166 last_log = now 

167 

168 await asyncio.sleep(poll_interval) 

169 

170 @classmethod 

171 def reset(cls) -> None: 

172 """Reset state. Intended for use in tests.""" 

173 cls._is_shutting_down = False 

174 cls._shutdown_started_at = None 

175 cls._drain_performed = False