Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/prometheus_metrics_server.py: 0%

99 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1"""Serve Prometheus `/metrics` from its own process so a scrape never runs on an inference worker. 

2 

3Workers write their samples to `PROMETHEUS_MULTIPROC_DIR`; this process reads them back with a 

4``MultiProcessCollector`` and serves the aggregated output on a separate port. The proxy CLI starts 

5it with ``--prometheus_metrics_port``. It can also run as a sidecar sharing the same directory: 

6``python -m litellm.proxy.prometheus_metrics_server --host 0.0.0.0 --port 4001``. 

7""" 

8 

9from __future__ import annotations 

10 

11import argparse 

12import atexit 

13import os 

14import subprocess 

15import sys 

16import threading 

17import time 

18from collections.abc import Sequence 

19from contextlib import closing 

20from types import MappingProxyType 

21from typing import Final 

22 

23import httpx 

24from fastapi import FastAPI 

25from prometheus_client import CollectorRegistry, multiprocess 

26from pydantic import BaseModel, ConfigDict 

27from starlette.types import ASGIApp, Message, Receive, Scope, Send 

28 

29from litellm.integrations.prometheus_metrics_endpoint import make_metrics_asgi_app 

30from litellm.llms.custom_httpx.http_handler import HTTPHandler 

31 

32METRICS_PATH: Final = "/metrics" 

33HEALTH_PATH: Final = "/health" 

34PID_HEADER: Final = "x-litellm-metrics-pid" 

35_PARENT_POLL_INTERVAL_SECONDS: Final = 1.0 

36_STARTUP_TIMEOUT_SECONDS: Final = 30.0 

37_STARTUP_POLL_INTERVAL_SECONDS: Final = 0.1 

38_STARTUP_PROBE_TIMEOUT_SECONDS: Final = 1.0 

39_WILDCARD_TO_LOOPBACK: Final = MappingProxyType({"0.0.0.0": "127.0.0.1", "::": "::1"}) 

40 

41 

42class _CliArgs(BaseModel): 

43 model_config = ConfigDict(frozen=True) 

44 

45 host: str 

46 port: int 

47 multiproc_dir: str | None 

48 

49 

50class MetricsServerStartupError(RuntimeError): 

51 """The metrics process died or never answered on its port before the proxy started serving.""" 

52 

53 

54def _add_pid_header(app: ASGIApp) -> ASGIApp: 

55 async def app_with_pid(scope: Scope, receive: Receive, send: Send) -> None: 

56 async def send_with_pid(message: Message) -> None: 

57 if message["type"] == "http.response.start": 

58 await send( 

59 { 

60 **message, 

61 "headers": [ 

62 *message["headers"], 

63 (PID_HEADER.encode(), str(os.getpid()).encode()), 

64 ], 

65 } 

66 ) 

67 return 

68 await send(message) 

69 

70 await app(scope, receive, send_with_pid) 

71 

72 return app_with_pid 

73 

74 

75def build_metrics_app(multiproc_dir: str) -> FastAPI: 

76 registry: Final = CollectorRegistry() 

77 multiprocess.MultiProcessCollector(registry, path=multiproc_dir) 

78 app: Final = FastAPI(title="LiteLLM Prometheus metrics", docs_url=None, redoc_url=None, openapi_url=None) 

79 app.mount(METRICS_PATH, _add_pid_header(make_metrics_asgi_app(registry))) 

80 

81 @app.get(HEALTH_PATH) 

82 def health() -> dict[str, str]: 

83 return {"status": "healthy", "multiproc_dir": multiproc_dir} 

84 

85 return app 

86 

87 

88def _exit_when_parent_dies(parent_pid: int) -> None: 

89 def watch() -> None: 

90 while os.getppid() == parent_pid: 

91 time.sleep(_PARENT_POLL_INTERVAL_SECONDS) 

92 os._exit(0) 

93 

94 threading.Thread(target=watch, name="litellm-metrics-parent-watchdog", daemon=True).start() 

95 

96 

97def run_metrics_server(host: str, port: int, multiproc_dir: str) -> None: 

98 import uvicorn 

99 

100 _exit_when_parent_dies(os.getppid()) 

101 uvicorn.run(build_metrics_app(multiproc_dir), host=host, port=port, log_level="warning", access_log=False) 

102 

103 

104def metrics_url(host: str, port: int) -> str: 

105 probe_host: Final = _WILDCARD_TO_LOOPBACK.get(host, host) 

106 netloc: Final = f"[{probe_host}]" if ":" in probe_host else probe_host 

107 return f"http://{netloc}:{port}{METRICS_PATH}" 

108 

109 

110def _answered_by(http: HTTPHandler, url: str, pid: int) -> bool: 

111 """True only when the metrics response comes from our child, not from whatever else holds the port.""" 

112 try: 

113 response: Final = http.get(url) # pyright: ignore[reportUnknownMemberType] # HTTPHandler.get exposes untyped optional mappings 

114 return response.status_code == 200 and response.headers.get(PID_HEADER) == str(pid) 

115 except httpx.TransportError: 

116 return False 

117 

118 

119def _wait_until_serving(process: subprocess.Popen[bytes], host: str, port: int) -> None: 

120 url: Final = metrics_url(host, port) 

121 deadline: Final = time.monotonic() + _STARTUP_TIMEOUT_SECONDS 

122 with closing(HTTPHandler(timeout=_STARTUP_PROBE_TIMEOUT_SECONDS)) as http: 

123 while time.monotonic() < deadline: 

124 if (returncode := process.poll()) is not None: 

125 raise MetricsServerStartupError( 

126 f"Prometheus metrics server exited with code {returncode} before serving {host}:{port}; " 

127 "is the port already in use?" 

128 ) 

129 if _answered_by(http, url, process.pid): 

130 return 

131 time.sleep(_STARTUP_POLL_INTERVAL_SECONDS) 

132 process.terminate() 

133 raise MetricsServerStartupError( 

134 f"Prometheus metrics server did not answer {url} within {_STARTUP_TIMEOUT_SECONDS:.0f}s" 

135 ) 

136 

137 

138def start_metrics_server_process(host: str, port: int, multiproc_dir: str) -> subprocess.Popen[bytes]: 

139 """Spawn the metrics server next to the proxy and block until it answers on its port.""" 

140 process: Final = subprocess.Popen( 

141 ( 

142 sys.executable, 

143 "-m", 

144 "litellm.proxy.prometheus_metrics_server", 

145 "--host", 

146 host, 

147 "--port", 

148 str(port), 

149 "--multiproc_dir", 

150 multiproc_dir, 

151 ) 

152 ) 

153 atexit.register(process.terminate) 

154 _wait_until_serving(process, host, port) 

155 return process 

156 

157 

158def main(argv: Sequence[str] | None = None) -> None: 

159 parser: Final = argparse.ArgumentParser( 

160 description="Serve LiteLLM Prometheus metrics from PROMETHEUS_MULTIPROC_DIR" 

161 ) 

162 parser.add_argument("--host", default="0.0.0.0") 

163 parser.add_argument("--port", type=int, required=True) 

164 parser.add_argument("--multiproc_dir", default=os.environ.get("PROMETHEUS_MULTIPROC_DIR")) 

165 args: Final = _CliArgs.model_validate(vars(parser.parse_args(argv))) 

166 if not args.multiproc_dir: 

167 parser.error("--multiproc_dir or PROMETHEUS_MULTIPROC_DIR is required") 

168 run_metrics_server(host=args.host, port=args.port, multiproc_dir=args.multiproc_dir) 

169 

170 

171if __name__ == "__main__": 

172 main()