Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/proxy_cli.py: 46%

624 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1# ruff: noqa: T201 

2import importlib 

3import json 

4import os 

5import random 

6import re 

7import subprocess 

8import sys 

9import urllib.parse as urlparse 

10from collections.abc import Iterable, Mapping, Sequence 

11from pathlib import Path 

12from typing import TYPE_CHECKING, Any, Final 

13 

14import click 

15import httpx 

16from click.core import ParameterSource 

17from dotenv import load_dotenv 

18from pydantic import BaseModel, ConfigDict 

19 

20import litellm 

21from litellm.constants import DEFAULT_NUM_WORKERS_LITELLM_PROXY 

22from litellm.proxy.db.pgbouncer import ( 

23 PgBouncerError, 

24 PgBouncerSettings, 

25 export_pooled_database_url, 

26 start_in_container_pgbouncer, 

27) 

28from litellm.proxy.db.query_engine_reaper import start_query_engine_reaper 

29 

30if TYPE_CHECKING: 30 ↛ 31line 30 didn't jump to line 31 because the condition on line 30 was never true

31 from fastapi import FastAPI 

32else: 

33 FastAPI = Any 

34 

35 

36def _drop_script_dir_from_sys_path() -> None: 

37 """Stop ``litellm/proxy`` modules from shadowing installed packages. 

38 

39 Running this file as a script puts its own directory at ``sys.path[0]``, so 

40 ``import a2a`` resolves to ``litellm/proxy/a2a`` instead of the ``a2a`` SDK 

41 and ``proxy_server`` resolves to a second copy of 

42 ``litellm.proxy.proxy_server``. No-op under the ``litellm`` console script. 

43 """ 

44 script_dir: Final = os.path.dirname(os.path.abspath(__file__)) 

45 if sys.path and os.path.abspath(sys.path[0]) == script_dir: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true

46 sys.path.pop(0) 

47 

48 

49_drop_script_dir_from_sys_path() 

50sys.path.append(os.getcwd()) 

51 

52config_filename: Final = "litellm.secrets" 

53 

54litellm_mode: Final = os.getenv("LITELLM_MODE", "DEV") # "PRODUCTION", "DEV" 

55if litellm_mode == "DEV": 55 ↛ 57line 55 didn't jump to line 57 because the condition on line 55 was always true

56 load_dotenv() 

57from enum import Enum 

58 

59 

60class LiteLLMDatabaseConnectionPool(Enum): 

61 database_connection_pool_limit = 10 

62 database_connection_pool_timeout = 60 

63 

64 

65def _build_db_connection_url_params( 

66 connection_limit: int, 

67 pool_timeout: float | None, 

68 connect_timeout: float | None = None, 

69 socket_timeout: float | None = None, 

70 disable_prepared_statements: bool = False, 

71 extra_params: dict | None = None, 

72) -> dict: 

73 """Build the Prisma DATABASE_URL query params controlling connection pool behavior. 

74 

75 `connect_timeout` / `socket_timeout` map to the Prisma URL params of the same 

76 name (https://www.prisma.io/docs/orm/overview/databases/postgresql) and are 

77 omitted when None so Prisma's defaults apply. `disable_prepared_statements` 

78 sets `pgbouncer=true`, which makes Prisma stop using server-side prepared 

79 statements (pgbouncer transaction-pool compatible; also sidesteps the 

80 "cached plan must not change result type" error during rolling migrations). 

81 `extra_params` is an untyped passthrough — keys it provides win over the 

82 named arguments above, so it can be used to override any default we set here. 

83 """ 

84 params: Final[dict] = { 

85 "connection_limit": connection_limit, 

86 } 

87 if pool_timeout is not None: 87 ↛ 89line 87 didn't jump to line 89 because the condition on line 87 was always true

88 params["pool_timeout"] = pool_timeout 

89 if connect_timeout is not None: 89 ↛ 90line 89 didn't jump to line 90 because the condition on line 89 was never true

90 params["connect_timeout"] = connect_timeout 

91 if socket_timeout is not None: 91 ↛ 92line 91 didn't jump to line 92 because the condition on line 91 was never true

92 params["socket_timeout"] = socket_timeout 

93 if disable_prepared_statements: 93 ↛ 94line 93 didn't jump to line 94 because the condition on line 93 was never true

94 params["pgbouncer"] = "true" 

95 if extra_params: 95 ↛ 96line 95 didn't jump to line 96 because the condition on line 95 was never true

96 params.update(extra_params) 

97 return params 

98 

99 

100class DatabaseTimeoutSettings(BaseModel): 

101 """The `general_settings` keys that bound how long a statement may hold locks. 

102 

103 Validated at the boundary so a mistyped value fails at startup with a clear 

104 pydantic error instead of a TypeError deep inside URL assembly. 

105 """ 

106 

107 model_config = ConfigDict(extra="ignore") 

108 

109 database_statement_timeout: float | None = None 

110 database_lock_timeout: float | None = None 

111 

112 

113def _pg_options_with_timeouts( 

114 existing_options: str, 

115 statement_timeout: float | None, 

116 lock_timeout: float | None, 

117) -> str: 

118 """Return the Postgres ``options`` value carrying the configured timeouts. 

119 

120 Prisma has no URL parameter for `statement_timeout` / `lock_timeout`; they are 

121 server settings delivered through the standard libpq ``options`` parameter as 

122 ``-c <name>=<value>``. Values arrive in seconds (matching the sibling 

123 ``database_*_timeout`` settings) and are emitted as integer milliseconds, 

124 which is the unit Postgres assumes for a unit-less value. 

125 

126 Both settings are bounds on how long a single statement may hold its locks. 

127 Without them a batch that outlives the Prisma client's HTTP read timeout keeps 

128 running server side, holding row locks for as long as the database takes, 

129 while the client has already given up; every later flush queues behind it. 

130 The query engine's own transaction timeout cannot end that wait, because it 

131 cannot interrupt a statement that is already executing. 

132 

133 Any ``options`` the operator already pinned on the URL is preserved, and a 

134 setting they pinned there wins over the configured one. Postgres applies the 

135 last occurrence of a setting, so a pinned one is honoured by not appending 

136 ours at all; it is matched in every spelling the backend accepts 

137 (``-c name=``, ``-cname=``, ``--name=``). The result is applied to 

138 ``DATABASE_URL`` only, never ``DIRECT_URL``: that one serves migrations, 

139 which legitimately run long and must not be cancelled mid-way. 

140 """ 

141 configured: Final = tuple( 

142 f"-c {name}={int(seconds * 1000)}" 

143 for name, seconds in ( 

144 ("statement_timeout", statement_timeout), 

145 ("lock_timeout", lock_timeout), 

146 ) 

147 if seconds is not None and not re.search(rf"(?:-c\s*|--){re.escape(name)}=", existing_options) 

148 ) 

149 return " ".join(part for part in (existing_options, *configured) if part) 

150 

151 

152def _url_query_value(url: str | None, key: str) -> str: 

153 """Return a single query-param value already present on ``url``, else "".""" 

154 if not isinstance(url, str) or url == "": 154 ↛ 155line 154 didn't jump to line 155 because the condition on line 154 was never true

155 return "" 

156 return next(iter(urlparse.parse_qs(urlparse.urlparse(url).query).get(key) or ()), "") 

157 

158 

159def _with_query_value(url: str, key: str, value: str) -> str: 

160 """Return ``url`` with a single query param replaced by ``value``.""" 

161 parsed: Final = urlparse.urlparse(url) 

162 pairs: Final = tuple((k, v) for k, v in urlparse.parse_qsl(parsed.query) if k != key) + ((key, value),) 

163 return urlparse.urlunparse(parsed._replace(query=urlparse.urlencode(pairs))) 

164 

165 

166def append_query_params(url: str | None, params: dict) -> str: 

167 from litellm._logging import verbose_proxy_logger 

168 

169 verbose_proxy_logger.debug("url: %s", url) 

170 verbose_proxy_logger.debug("params: %s", params) 

171 if not isinstance(url, str) or url == "": 171 ↛ 174line 171 didn't jump to line 174 because the condition on line 171 was never true

172 # Preserve previous startup behavior when DATABASE_URL is absent. 

173 # Returning an empty string avoids urlparse type errors in test/dev flows. 

174 verbose_proxy_logger.warning("append_query_params received empty or non-string URL, returning empty string") 

175 return "" 

176 parsed_url: Final = urlparse.urlparse(url) 

177 parsed_query: Final = urlparse.parse_qs(parsed_url.query) 

178 parsed_query.update(params) 

179 encoded_query: Final = urlparse.urlencode(parsed_query, doseq=True) 

180 modified_url: Final = urlparse.urlunparse(parsed_url._replace(query=encoded_query)) 

181 return modified_url 

182 

183 

184def resolve_v2_migration_resolver(*, use_legacy_flag: bool, env_value: str | None) -> bool: 

185 from litellm_proxy_extras.utils import str_to_bool 

186 

187 if use_legacy_flag: 187 ↛ 188line 187 didn't jump to line 188 because the condition on line 187 was never true

188 return False 

189 if env_value is None: 189 ↛ 191line 189 didn't jump to line 191 because the condition on line 189 was always true

190 return True 

191 return bool(str_to_bool(env_value)) 

192 

193 

194def deprecated_v2_flag_passed_on_cli() -> bool: 

195 ctx: Final = click.get_current_context(silent=True) 

196 if ctx is None: 196 ↛ 197line 196 didn't jump to line 197 because the condition on line 196 was never true

197 return False 

198 return ctx.get_parameter_source("use_v2_migration_resolver") is ParameterSource.COMMANDLINE 

199 

200 

201class ProxyInitializationHelpers: 

202 @staticmethod 

203 def _echo_litellm_version(): 

204 pkg_version: Final = importlib.metadata.version("litellm") 

205 click.echo(f"\nLiteLLM: Current Version = {pkg_version}\n") 

206 

207 @staticmethod 

208 def _run_health_check(host, port): 

209 print("\nLiteLLM: Health Testing models in config") 

210 response: Final = httpx.get(url=f"http://{host}:{port}/health") 

211 print(json.dumps(response.json(), indent=4)) 

212 

213 @staticmethod 

214 def _run_config_validation(config: str | None) -> None: 

215 if config is None: 

216 raise click.UsageError("--validate_config requires --config <path>") 

217 import asyncio 

218 

219 from litellm.proxy.proxy_server import ProxyConfig 

220 

221 async def _load() -> int: 

222 _, model_list, _ = await ProxyConfig().load_config(router=None, config_file_path=config) 

223 return len(model_list) 

224 

225 try: 

226 model_count: Final = asyncio.run(_load()) 

227 except Exception as error: 

228 click.echo(f"LiteLLM: config validation failed: {error}", err=True) 

229 raise click.exceptions.Exit(1) from error 

230 click.echo(f"LiteLLM: config OK ({model_count} models)") 

231 

232 @staticmethod 

233 def _run_test_chat_completion( 

234 host: str, 

235 port: int, 

236 model: str, 

237 test: bool | str, 

238 ): 

239 request_model: Final = model or "gpt-3.5-turbo" 

240 click.echo(f"\nLiteLLM: Making a test ChatCompletions request to your proxy. Model={request_model}") 

241 import openai 

242 

243 api_base = f"http://{host}:{port}" 

244 if isinstance(test, str): 

245 api_base = test 

246 else: 

247 raise ValueError("Invalid test value") 

248 client: Final = openai.OpenAI(api_key="My API Key", base_url=api_base) 

249 

250 response: Final = client.chat.completions.create( 

251 model=request_model, 

252 messages=[ 

253 { 

254 "role": "user", 

255 "content": "this is a test request, write a short poem", 

256 } 

257 ], 

258 max_tokens=256, 

259 ) 

260 click.echo(f"\nLiteLLM: response from proxy {response}") 

261 

262 print(f"\n LiteLLM: Making a test ChatCompletions + streaming r equest to proxy. Model={request_model}") 

263 

264 stream_response: Final = client.chat.completions.create( 

265 model=request_model, 

266 messages=[ 

267 { 

268 "role": "user", 

269 "content": "this is a test request, write a short poem", 

270 } 

271 ], 

272 stream=True, 

273 ) 

274 for chunk in stream_response: 

275 click.echo(f"LiteLLM: streaming response from proxy {chunk}") 

276 print("\n making completion request to proxy") 

277 completion_response: Final = client.completions.create( 

278 model=request_model, prompt="this is a test request, write a short poem" 

279 ) 

280 print(completion_response) 

281 

282 @staticmethod 

283 def _get_default_unvicorn_init_args( 

284 host: str, 

285 port: int, 

286 log_config: str | None = None, 

287 keepalive_timeout: int | None = None, 

288 timeout_worker_healthcheck: int | None = None, 

289 ) -> dict: 

290 """ 

291 Get the arguments for `uvicorn` worker 

292 """ 

293 import inspect 

294 

295 import uvicorn 

296 

297 import litellm 

298 from litellm._logging import _get_uvicorn_json_log_config, resolve_log_level 

299 

300 uvicorn_args: Final = { 

301 "app": "litellm.proxy.proxy_server:app", 

302 "host": host, 

303 "port": port, 

304 "server_header": False, 

305 } 

306 if log_config is not None: 306 ↛ 307line 306 didn't jump to line 307 because the condition on line 306 was never true

307 print(f"Using log_config: {log_config}") 

308 uvicorn_args["log_config"] = log_config 

309 elif litellm.json_logs: 309 ↛ 311line 309 didn't jump to line 311 because the condition on line 309 was never true

310 # Use JSON log config for uvicorn to ensure all logs (including exceptions) are JSON 

311 uvicorn_args["log_config"] = _get_uvicorn_json_log_config() 

312 elif litellm_log := os.environ.get("LITELLM_LOG"): 312 ↛ 313line 312 didn't jump to line 313 because the condition on line 312 was never true

313 uvicorn_args["log_level"] = resolve_log_level(litellm_log) 

314 if keepalive_timeout is not None: 314 ↛ 315line 314 didn't jump to line 315 because the condition on line 314 was never true

315 uvicorn_args["timeout_keep_alive"] = keepalive_timeout 

316 if timeout_worker_healthcheck is not None: 316 ↛ 317line 316 didn't jump to line 317 because the condition on line 316 was never true

317 if "timeout_worker_healthcheck" in inspect.signature(uvicorn.Config.__init__).parameters: 

318 uvicorn_args["timeout_worker_healthcheck"] = timeout_worker_healthcheck 

319 else: 

320 print( 

321 f"\033[1;33mLiteLLM Proxy: --timeout_worker_healthcheck " 

322 f"requires uvicorn>=0.37.0, but installed uvicorn=={uvicorn.__version__}. " 

323 f"Ignoring the flag.\033[0m" 

324 ) 

325 return uvicorn_args 

326 

327 @staticmethod 

328 def _apply_uvicorn_max_requests_jitter( 

329 uvicorn_args: dict, 

330 max_requests_before_restart: int | None, 

331 jitter: int, 

332 ) -> None: 

333 """ 

334 Stagger uvicorn worker restarts via limit_max_requests_jitter (uvicorn>=0.41.0). 

335 """ 

336 import inspect 

337 

338 import uvicorn 

339 

340 if max_requests_before_restart is None: 

341 print( 

342 "\033[1;33mLiteLLM Proxy: --max_requests_before_restart_jitter " 

343 "has no effect without --max_requests_before_restart\033[0m\n" 

344 ) 

345 return 

346 if "limit_max_requests_jitter" in inspect.signature(uvicorn.Config.__init__).parameters: 

347 uvicorn_args["limit_max_requests_jitter"] = jitter 

348 else: 

349 print( 

350 f"\033[1;33mLiteLLM Proxy: --max_requests_before_restart_jitter " 

351 f"requires uvicorn>=0.41.0, but installed uvicorn=={uvicorn.__version__}. " 

352 f"Ignoring the flag.\033[0m" 

353 ) 

354 

355 @staticmethod 

356 def _get_reload_options(config_path: str | None) -> dict: 

357 """Build uvicorn reload kwargs so --reload also reacts to .env and YAML edits.""" 

358 cwd: Final = os.path.abspath(os.getcwd()) 

359 reload_dirs: Final = [cwd] 

360 # Must be basenames, not absolute paths: uvicorn's 

361 # resolve_reload_patterns() calls pathlib.Path.glob(), which raises 

362 # NotImplementedError on absolute patterns (uvicorn discussion #2156). 

363 reload_includes: Final = ["*.py", ".env"] 

364 if config_path: 

365 config_abs: Final = os.path.abspath(config_path) 

366 config_dir: Final = os.path.dirname(config_abs) 

367 if config_dir and config_dir != cwd: 

368 reload_dirs.append(config_dir) 

369 reload_includes.append(os.path.basename(config_abs)) 

370 return { 

371 "reload": True, 

372 "reload_dirs": reload_dirs, 

373 "reload_includes": reload_includes, 

374 } 

375 

376 @staticmethod 

377 def _patch_statreload_extra_paths(paths: Iterable[str | None]) -> bool: 

378 """Make uvicorn's StatReload reloader notice non-Python dev files 

379 (the --config YAML and .env). 

380 

381 Uvicorn uses WatchFilesReload when the optional `watchfiles` package 

382 is installed, otherwise StatReload. StatReload hard-codes `*.py` in 

383 `iter_py_files()` and silently ignores `reload_includes`, so the 

384 kwargs from `_get_reload_options` alone don't trigger reloads on those 

385 files. We monkey-patch `iter_py_files` to also yield the given paths. 

386 

387 Idempotent across calls and a no-op for the WatchFilesReload path. 

388 """ 

389 try: 

390 from uvicorn.supervisors.statreload import StatReload 

391 except ImportError: # pragma: no cover - uvicorn is a hard dep 

392 return False 

393 

394 from pathlib import Path 

395 

396 resolved: Final = {Path(p).resolve() for p in paths if p} 

397 if not resolved: 

398 return False 

399 

400 patched_paths = getattr(StatReload, "_litellm_patched_config_paths", None) 

401 if patched_paths is None: 

402 original_iter: Final = StatReload.iter_py_files 

403 patched_paths = set() 

404 

405 def _iter_with_extra(self): 

406 yield from original_iter(self) 

407 for path in StatReload._litellm_patched_config_paths: 

408 if path.exists(): 

409 yield path 

410 

411 StatReload.iter_py_files = _iter_with_extra 

412 StatReload._litellm_patched_config_paths = patched_paths 

413 

414 patched_paths.update(resolved) 

415 return True 

416 

417 @staticmethod 

418 def _configure_dev_reload(uvicorn_args: dict, config_path: str | None) -> None: 

419 """Wire up --reload (dev only): watch *.py, the --config YAML, and .env, 

420 and signal reloaded workers to re-read .env with override so edits to 

421 existing keys actually take effect rather than staying masked by the 

422 value inherited from the reloader process.""" 

423 from litellm._logging import verbose_proxy_logger 

424 

425 uvicorn_args.update(ProxyInitializationHelpers._get_reload_options(config_path)) 

426 os.environ["LITELLM_DEV_ENV_HOT_RELOAD"] = "True" 

427 env_path: Final = os.path.join(os.getcwd(), ".env") 

428 ProxyInitializationHelpers._patch_statreload_extra_paths([config_path, env_path]) 

429 verbose_proxy_logger.warning( 

430 "LiteLLM --reload: worker processes re-read .env with override, so .env " 

431 "values win over shell-exported environment variables. Unset a key in .env " 

432 "to let a shell-exported value take precedence." 

433 ) 

434 

435 @staticmethod 

436 def _init_hypercorn_server( 

437 app: FastAPI, 

438 host: str, 

439 port: int, 

440 ssl_certfile_path: str, 

441 ssl_keyfile_path: str, 

442 ciphers: str | None = None, 

443 ): 

444 """ 

445 Initialize litellm with `hypercorn` 

446 """ 

447 import asyncio 

448 

449 from hypercorn.asyncio import serve 

450 from hypercorn.config import Config 

451 

452 print(f"\033[1;32mLiteLLM Proxy: Starting server on {host}:{port} using Hypercorn\033[0m\n") 

453 config: Final = Config() 

454 config.bind = [f"{host}:{port}"] 

455 

456 if ssl_certfile_path is not None and ssl_keyfile_path is not None: 

457 print( 

458 f"\033[1;32mLiteLLM Proxy: Using SSL with certfile: {ssl_certfile_path} and keyfile: {ssl_keyfile_path}\033[0m\n" 

459 ) 

460 config.certfile = ssl_certfile_path 

461 config.keyfile = ssl_keyfile_path 

462 if ciphers is not None: 

463 config.ciphers = ciphers 

464 

465 # hypercorn serve raises a type warning when passing a fast api app - even though fast API is a valid type 

466 asyncio.run(serve(app, config)) 

467 

468 @staticmethod 

469 def _init_granian_server( 

470 host: str, 

471 port: int, 

472 num_workers: int, 

473 ssl_certfile_path: str | None, 

474 ssl_keyfile_path: str | None, 

475 max_requests_before_restart: int | None, 

476 ciphers: str | None, 

477 granian_runtime_threads: int | None = None, 

478 ) -> None: 

479 """ 

480 Run the proxy with Granian (Rust-backed ASGI server, HTTP/1 + HTTP/2). 

481 

482 Uses a string import path so workers load ``litellm.proxy.proxy_server:app`` 

483 the same way as uvicorn's ``app=`` string target. 

484 """ 

485 from granian import Granian 

486 from granian.constants import Interfaces 

487 

488 print(f"\033[1;32mLiteLLM Proxy: Starting server on {host}:{port} using Granian\033[0m\n") 

489 if max_requests_before_restart is not None: 

490 print( 

491 "\033[1;33mLiteLLM: --max_requests_before_restart is not supported by Granian " 

492 "(Granian uses workers_lifetime in seconds, not a per-request limit).\033[0m\n" 

493 ) 

494 if ciphers is not None: 

495 print("\033[1;33mLiteLLM: --ciphers is not applied when using --run_granian.\033[0m\n") 

496 

497 kwargs: Final[dict[str, Any]] = { 

498 "target": "litellm.proxy.proxy_server:app", 

499 "address": host, 

500 "port": port, 

501 "workers": max(1, num_workers), 

502 "interface": Interfaces.ASGI, 

503 "websockets": True, 

504 } 

505 if granian_runtime_threads is not None: 

506 kwargs["runtime_threads"] = granian_runtime_threads 

507 if ssl_certfile_path is not None and ssl_keyfile_path is not None: 

508 print( 

509 f"\033[1;32mLiteLLM Proxy: Using SSL with certfile: {ssl_certfile_path} and keyfile: {ssl_keyfile_path}\033[0m\n" 

510 ) 

511 kwargs["ssl_cert"] = Path(ssl_certfile_path) 

512 kwargs["ssl_key"] = Path(ssl_keyfile_path) 

513 elif ssl_certfile_path is not None or ssl_keyfile_path is not None: 

514 raise click.ClickException("Both --ssl_certfile_path and --ssl_keyfile_path are required for SSL.") 

515 

516 Granian(**kwargs).serve() 

517 

518 @staticmethod 

519 def _run_gunicorn_server( 

520 host: str, 

521 port: int, 

522 app: FastAPI, 

523 num_workers: int, 

524 ssl_certfile_path: str, 

525 ssl_keyfile_path: str, 

526 max_requests_before_restart: int | None = None, 

527 max_requests_before_restart_jitter: int | None = None, 

528 ): 

529 """ 

530 Run litellm with `gunicorn` 

531 """ 

532 if os.name == "nt": 

533 pass 

534 else: 

535 import gunicorn.app.base 

536 

537 # Gunicorn Application Class 

538 class StandaloneApplication(gunicorn.app.base.BaseApplication): 

539 def __init__(self, app, options=None): 

540 self.options = options or {} # gunicorn options 

541 self.application = app # FastAPI app 

542 super().__init__() 

543 

544 _endpoint_str: Final = f"curl --location 'http://0.0.0.0:{port}/chat/completions' \\" 

545 curl_command: Final = ( 

546 _endpoint_str 

547 + """ 

548 --header 'Content-Type: application/json' \\ 

549 --data ' { 

550 "model": "gpt-3.5-turbo", 

551 "messages": [ 

552 { 

553 "role": "user", 

554 "content": "what llm are you" 

555 } 

556 ] 

557 }' 

558 \n 

559 """ 

560 ) 

561 print() 

562 print( 

563 '\033[1;34mLiteLLM: Test your local proxy with: "litellm --test" This runs an openai.ChatCompletion request to your proxy [In a new terminal tab]\033[0m\n' 

564 ) 

565 print(f"\033[1;34mLiteLLM: Curl Command Test for your local proxy\n {curl_command} \033[0m\n") 

566 print("\033[1;34mDocs: https://docs.litellm.ai/docs/simple_proxy\033[0m\n") 

567 print(f"\033[1;34mSee all Router/Swagger docs on http://0.0.0.0:{port} \033[0m\n") 

568 

569 def load_config(self): 

570 # note: This Loads the gunicorn config - has nothing to do with LiteLLM Proxy config 

571 if self.cfg is not None: 

572 config = { 

573 key: value 

574 for key, value in self.options.items() 

575 if key in self.cfg.settings and value is not None 

576 } 

577 else: 

578 config = {} 

579 for key, value in config.items(): 

580 if self.cfg is not None: 

581 self.cfg.set(key.lower(), value) 

582 

583 def load(self): 

584 # gunicorn app function 

585 return self.application 

586 

587 print(f"\033[1;32mLiteLLM Proxy: Starting server on {host}:{port} with {num_workers} workers\033[0m\n") 

588 gunicorn_options: Final = { 

589 "bind": f"{host}:{port}", 

590 "workers": num_workers, # default is 1 

591 "worker_class": "uvicorn.workers.UvicornWorker", 

592 "preload": True, # Add the preload flag, 

593 "accesslog": "-", # Log to stdout 

594 "timeout": 600, # default to very high number, bedrock/anthropic.claude-v2:1 can take 30+ seconds for the 1st chunk to come in 

595 "access_log_format": '%(h)s %(l)s %(u)s %(t)s "%(r)s" %(s)s %(b)s', 

596 } 

597 

598 # Optional: recycle workers after N requests to mitigate memory growth 

599 if max_requests_before_restart is not None: 

600 gunicorn_options["max_requests"] = max_requests_before_restart 

601 if max_requests_before_restart_jitter is not None: 

602 if max_requests_before_restart is None: 

603 print( 

604 "\033[1;33mLiteLLM Proxy: --max_requests_before_restart_jitter " 

605 "has no effect without --max_requests_before_restart\033[0m\n" 

606 ) 

607 else: 

608 gunicorn_options["max_requests_jitter"] = max_requests_before_restart_jitter 

609 

610 # Clean up prometheus .db files when a worker exits (prevents ghost gauge values) 

611 if os.environ.get("PROMETHEUS_MULTIPROC_DIR"): 

612 from litellm.proxy.prometheus_cleanup import mark_worker_exit 

613 

614 def child_exit(server, worker): 

615 mark_worker_exit(worker.pid) 

616 

617 gunicorn_options["child_exit"] = child_exit 

618 

619 if ssl_certfile_path is not None and ssl_keyfile_path is not None: 

620 print( 

621 f"\033[1;32mLiteLLM Proxy: Using SSL with certfile: {ssl_certfile_path} and keyfile: {ssl_keyfile_path}\033[0m\n" 

622 ) 

623 gunicorn_options["certfile"] = ssl_certfile_path 

624 gunicorn_options["keyfile"] = ssl_keyfile_path 

625 

626 # The master preloads the app and then forks every worker, so native routes are 

627 # forbidden in it: their runtime threads would not survive the fork. 

628 from litellm.rust_bridge.fork_guard import reserve_process_for_forking 

629 

630 reserve_process_for_forking("the gunicorn master") 

631 start_query_engine_reaper() 

632 StandaloneApplication(app=app, options=gunicorn_options).run() # Run gunicorn 

633 

634 @staticmethod 

635 def _run_ollama_serve(): 

636 try: 

637 command: Final = ["ollama", "serve"] 

638 

639 with open(os.devnull, "w") as devnull: 

640 subprocess.Popen(command, stdout=devnull, stderr=devnull) 

641 except Exception as e: 

642 print(f""" 

643 LiteLLM Warning: proxy started with `ollama` model\n`ollama serve` failed with Exception{e}. \nEnsure you run `ollama serve` 

644 """) 

645 

646 @staticmethod 

647 def _is_port_in_use(port): 

648 import socket 

649 

650 with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s: 

651 return s.connect_ex(("localhost", port)) == 0 

652 

653 @staticmethod 

654 def _get_loop_type(): 

655 """Helper function to determine the event loop type based on platform""" 

656 if sys.platform in ("win32", "cygwin", "cli"): 656 ↛ 657line 656 didn't jump to line 657 because the condition on line 656 was never true

657 return None # Let uvicorn choose the default loop on Windows 

658 return "uvloop" 

659 

660 @staticmethod 

661 def _prometheus_callback_configured(litellm_settings: Mapping[str, object] | None) -> bool: 

662 if litellm_settings is None: 

663 return False 

664 configured: Final = tuple( 

665 litellm_settings.get(key) for key in ("callbacks", "success_callback", "failure_callback") 

666 ) 

667 return any( 

668 setting == "prometheus" 

669 if isinstance(setting, str) 

670 else isinstance(setting, Sequence) and "prometheus" in setting 

671 for setting in configured 

672 ) 

673 

674 @staticmethod 

675 def _maybe_setup_prometheus_multiproc_dir( 

676 num_workers: int, 

677 litellm_settings: dict | None, 

678 prometheus_metrics_port: int | None = None, 

679 ) -> str | None: 

680 """ 

681 Auto-create PROMETHEUS_MULTIPROC_DIR when another process needs to read the samples: extra workers 

682 with prometheus configured as a callback in config.yaml, or the separate metrics server (always, since 

683 callbacks may also be enabled from the DB after startup). 

684 """ 

685 import tempfile 

686 

687 if prometheus_metrics_port is None and ( 687 ↛ 692line 687 didn't jump to line 692 because the condition on line 687 was always true

688 num_workers <= 1 or not ProxyInitializationHelpers._prometheus_callback_configured(litellm_settings) 

689 ): 

690 return None 

691 

692 from litellm.proxy.prometheus_cleanup import wipe_directory 

693 

694 configured_dir: Final = os.environ.get("PROMETHEUS_MULTIPROC_DIR") or os.environ.get("prometheus_multiproc_dir") 

695 multiproc_dir: Final = configured_dir or os.path.join(tempfile.gettempdir(), "litellm_prometheus_multiproc") 

696 os.environ["PROMETHEUS_MULTIPROC_DIR"] = multiproc_dir 

697 

698 os.makedirs(multiproc_dir, exist_ok=True) 

699 wipe_directory(multiproc_dir) 

700 action: Final = "Using existing" if configured_dir else "Auto-created" 

701 print(f"LiteLLM: {action} PROMETHEUS_MULTIPROC_DIR={multiproc_dir}") 

702 return multiproc_dir 

703 

704 

705@click.command() 

706@click.argument("cli_args", nargs=-1) 

707@click.option("--host", default="0.0.0.0", help="Host for the server to listen on.", envvar="HOST") 

708@click.option("--port", default=4000, help="Port to bind the server to.", envvar="PORT") 

709@click.option( 

710 "--num_workers", 

711 default=DEFAULT_NUM_WORKERS_LITELLM_PROXY, 

712 help=( 

713 "Number of worker processes for uvicorn / gunicorn, or Granian worker processes " 

714 "(--workers). Default is 1 (from DEFAULT_NUM_WORKERS_LITELLM_PROXY). " 

715 "With --run_granian, use --granian_threads for runtime threads per worker." 

716 ), 

717 envvar="NUM_WORKERS", 

718) 

719@click.option( 

720 "--granian_threads", 

721 default=None, 

722 type=click.IntRange(min=1), 

723 help=( 

724 "Only with --run_granian: runtime threads per worker process " 

725 "(Granian --runtime-threads / GRANIAN_RUNTIME_THREADS). Omit to use Granian's default (1)." 

726 ), 

727 envvar="GRANIAN_RUNTIME_THREADS", 

728) 

729@click.option("--api_base", default=None, help="API base URL.") 

730@click.option( 

731 "--api_version", 

732 default=litellm.AZURE_DEFAULT_API_VERSION, 

733 help="For azure - pass in the api version.", 

734) 

735@click.option("--model", "-m", default=None, help="The model name to pass to litellm expects") 

736@click.option( 

737 "--alias", 

738 default=None, 

739 help='The alias for the model - use this to give a litellm model name (e.g. "huggingface/codellama/CodeLlama-7b-Instruct-hf") a more user-friendly name ("codellama")', 

740) 

741@click.option("--add_key", default=None, help="The model name to pass to litellm expects") 

742@click.option("--headers", default=None, help="headers for the API call") 

743@click.option("--save", is_flag=True, type=bool, help="Save the model-specific config") 

744@click.option( 

745 "--debug", 

746 default=False, 

747 is_flag=True, 

748 type=bool, 

749 help="To debug the input", 

750 envvar="DEBUG", 

751) 

752@click.option( 

753 "--detailed_debug", 

754 default=False, 

755 is_flag=True, 

756 type=bool, 

757 help="To view detailed debug logs", 

758 envvar="DETAILED_DEBUG", 

759) 

760@click.option( 

761 "--use_queue", 

762 default=False, 

763 is_flag=True, 

764 type=bool, 

765 help="To use celery workers for async endpoints", 

766) 

767@click.option("--temperature", default=None, type=float, help="Set temperature for the model") 

768@click.option("--max_tokens", default=None, type=int, help="Set max tokens for the model") 

769@click.option( 

770 "--request_timeout", 

771 default=None, 

772 type=int, 

773 help="Set timeout in seconds for completion calls", 

774) 

775@click.option("--drop_params", is_flag=True, help="Drop any unmapped params") 

776@click.option( 

777 "--add_function_to_prompt", 

778 is_flag=True, 

779 help="If function passed but unsupported, pass it as prompt", 

780) 

781@click.option( 

782 "--config", 

783 "-c", 

784 default=None, 

785 help="Path to the proxy configuration file (e.g. config.yaml). Usage `litellm --config config.yaml`", 

786) 

787@click.option( 

788 "--max_budget", 

789 default=None, 

790 type=float, 

791 help="Set max budget for API calls - works for hosted models like OpenAI, TogetherAI, Anthropic, etc.`", 

792) 

793@click.option( 

794 "--telemetry", 

795 default=None, 

796 type=bool, 

797 hidden=True, 

798 expose_value=False, 

799 help="Deprecated no-op kept so existing start commands still parse", 

800) 

801@click.option( 

802 "--log_config", 

803 default=None, 

804 type=str, 

805 help="Path to the logging configuration file", 

806) 

807@click.option( 

808 "--setup", 

809 is_flag=True, 

810 default=False, 

811 help="Run the interactive setup wizard to configure providers and generate a config file", 

812) 

813@click.option( 

814 "--version", 

815 "-v", 

816 default=False, 

817 is_flag=True, 

818 type=bool, 

819 help="Print LiteLLM version", 

820) 

821@click.option( 

822 "--health", 

823 flag_value=True, 

824 help="Make a chat/completions request to all llms in config.yaml", 

825) 

826@click.option( 

827 "--test", 

828 flag_value=True, 

829 help="proxy chat completions url to make a test request to", 

830) 

831@click.option( 

832 "--test_async", 

833 default=False, 

834 is_flag=True, 

835 help="Calls async endpoints /queue/requests and /queue/response", 

836) 

837@click.option( 

838 "--iam_token_db_auth", 

839 default=False, 

840 is_flag=True, 

841 help="Connects to RDS DB with IAM token", 

842) 

843@click.option( 

844 "--azure_postgresql_auth", 

845 default=False, 

846 is_flag=True, 

847 help="Connects to Azure Database for PostgreSQL with a Microsoft Entra ID token", 

848) 

849@click.option( 

850 "--num_requests", 

851 default=10, 

852 type=int, 

853 help="Number of requests to hit async endpoint with", 

854) 

855@click.option( 

856 "--run_gunicorn", 

857 default=False, 

858 is_flag=True, 

859 help="Starts proxy via gunicorn, instead of uvicorn (better for managing multiple workers)", 

860) 

861@click.option( 

862 "--run_hypercorn", 

863 default=False, 

864 is_flag=True, 

865 help="Starts proxy via hypercorn, instead of uvicorn (supports HTTP/2)", 

866) 

867@click.option( 

868 "--run_granian", 

869 default=False, 

870 is_flag=True, 

871 help=( 

872 "Starts proxy via Granian (Rust ASGI server) instead of uvicorn. " 

873 "Requires Python 3.10+ and the `granian` package." 

874 ), 

875) 

876@click.option( 

877 "--ssl_keyfile_path", 

878 default=None, 

879 type=str, 

880 help="Path to the SSL keyfile. Use this when you want to provide SSL certificate when starting proxy", 

881 envvar="SSL_KEYFILE_PATH", 

882) 

883@click.option( 

884 "--ssl_certfile_path", 

885 default=None, 

886 type=str, 

887 help="Path to the SSL certfile. Use this when you want to provide SSL certificate when starting proxy", 

888 envvar="SSL_CERTFILE_PATH", 

889) 

890@click.option( 

891 "--ciphers", 

892 default=None, 

893 type=str, 

894 help="Ciphers to use for the SSL setup.", 

895) 

896@click.option( 

897 "--use_prisma_db_push", 

898 is_flag=True, 

899 default=False, 

900 help="Use prisma db push instead of prisma migrate for database schema updates", 

901) 

902@click.option("--local", is_flag=True, default=False, help="no-op, kept for backwards compatibility") 

903@click.option( 

904 "--skip_server_startup", 

905 is_flag=True, 

906 default=False, 

907 help="Skip starting the server after setup (useful for migrations only)", 

908) 

909@click.option( 

910 "--validate_config", 

911 is_flag=True, 

912 default=False, 

913 help="Load and validate the config file (including mcp_servers) without starting the server, then exit. Exit code 1 on any config error.", 

914) 

915@click.option( 

916 "--keepalive_timeout", 

917 default=None, 

918 type=int, 

919 help="Set the uvicorn keepalive timeout in seconds (uvicorn timeout_keep_alive parameter)", 

920 envvar="KEEPALIVE_TIMEOUT", 

921) 

922@click.option( 

923 "--timeout_worker_healthcheck", 

924 default=None, 

925 type=int, 

926 help=( 

927 "Set the uvicorn worker health-check timeout in seconds (uvicorn timeout_worker_healthcheck parameter). " 

928 "Requires uvicorn>=0.37.0. Only applies when running uvicorn directly with --num_workers>1; " 

929 "ignored under --run_gunicorn / --run_hypercorn." 

930 ), 

931 envvar="TIMEOUT_WORKER_HEALTHCHECK", 

932) 

933@click.option( 

934 "--max_requests_before_restart", 

935 default=None, 

936 type=int, 

937 help="Restart worker after this many requests (uvicorn: limit_max_requests, gunicorn: max_requests)", 

938 envvar="MAX_REQUESTS_BEFORE_RESTART", 

939) 

940@click.option( 

941 "--max_requests_before_restart_jitter", 

942 default=None, 

943 type=int, 

944 help=( 

945 "Stagger worker restarts by adding a random amount in [0, jitter] to " 

946 "--max_requests_before_restart so workers do not recycle at the same time " 

947 "(uvicorn: limit_max_requests_jitter, requires uvicorn>=0.41.0; gunicorn: max_requests_jitter). " 

948 "Has no effect without --max_requests_before_restart." 

949 ), 

950 envvar="MAX_REQUESTS_BEFORE_RESTART_JITTER", 

951) 

952@click.option( 

953 "--limit_concurrency", 

954 default=None, 

955 type=click.IntRange(min=1), 

956 help=( 

957 "Set uvicorn's concurrency limit. Uvicorn counts both active tasks and " 

958 "accepted connections and returns HTTP 503 after the limit is reached. " 

959 "Idle connections can consume capacity, so use upstream connection/header " 

960 "timeouts and per-client connection limits. Only applies to uvicorn " 

961 "(ignored under --run_gunicorn / --run_hypercorn / --run_granian)." 

962 ), 

963 envvar="LIMIT_CONCURRENCY", 

964) 

965@click.option( 

966 "--enforce_prisma_migration_check/--no-enforce_prisma_migration_check", 

967 default=True, 

968 show_default=True, 

969 help=( 

970 "Exit when database setup fails on startup instead of serving against a database " 

971 "whose schema may be behind the code. Opt out with --no-enforce_prisma_migration_check " 

972 "or ENFORCE_PRISMA_MIGRATION_CHECK=false." 

973 ), 

974 envvar="ENFORCE_PRISMA_MIGRATION_CHECK", 

975) 

976@click.option( 

977 "--use_v2_migration_resolver", 

978 is_flag=True, 

979 default=False, 

980 help=( 

981 "Deprecated and ignored: the v2 migration resolver is now the default, " 

982 "so this flag has no effect. It is still accepted so existing commands " 

983 "keep working. Pass --use_legacy_migration_resolver, or set " 

984 "USE_V2_MIGRATION_RESOLVER=false, to opt back into v1." 

985 ), 

986 envvar="USE_V2_MIGRATION_RESOLVER", 

987) 

988@click.option( 

989 "--use_legacy_migration_resolver", 

990 is_flag=True, 

991 default=False, 

992 help=( 

993 "Fall back to the legacy v1 migration resolver. By default the proxy " 

994 "uses the v2 resolver, which avoids the diff-and-force recovery path " 

995 "that can cause schema thrashing during rolling deploys where two " 

996 "LiteLLM versions contend for the same DB." 

997 ), 

998) 

999@click.option( 

1000 "--reload", 

1001 is_flag=True, 

1002 default=False, 

1003 help="Enable uvicorn hot reload (dev only). Also reloads when the --config YAML file changes. Incompatible with --num_workers>1, --run_gunicorn, and --run_hypercorn.", 

1004) 

1005@click.option( 

1006 "--prometheus_metrics_port", 

1007 default=None, 

1008 type=click.IntRange(min=1, max=65535), 

1009 help=( 

1010 "Serve Prometheus /metrics from a separate process on this port (bound to --host) so scraping and " 

1011 "multi-worker aggregation never run on an inference worker's event loop. Samples appear once the " 

1012 "`prometheus` callback is enabled (config.yaml or DB). /metrics stays mounted on the main port as well; " 

1013 "the separate port has no virtual-key auth, so keep it off public ingress. Startup fails if the metrics " 

1014 "server cannot bind." 

1015 ), 

1016 envvar="PROMETHEUS_METRICS_PORT", 

1017) 

1018def run_server( 

1019 cli_args, 

1020 host, 

1021 port, 

1022 api_base, 

1023 api_version, 

1024 model, 

1025 alias, 

1026 add_key, 

1027 headers, 

1028 save, 

1029 debug, 

1030 detailed_debug, 

1031 temperature, 

1032 max_tokens, 

1033 request_timeout, 

1034 drop_params, 

1035 add_function_to_prompt, 

1036 config, 

1037 max_budget, 

1038 test, 

1039 local, 

1040 num_workers, 

1041 granian_threads, 

1042 test_async, 

1043 iam_token_db_auth, 

1044 azure_postgresql_auth: bool, 

1045 num_requests, 

1046 use_queue, 

1047 health, 

1048 setup, 

1049 version, 

1050 run_gunicorn, 

1051 run_hypercorn, 

1052 run_granian, 

1053 ssl_keyfile_path, 

1054 ssl_certfile_path, 

1055 ciphers, 

1056 log_config, 

1057 use_prisma_db_push: bool, 

1058 skip_server_startup, 

1059 validate_config: bool, 

1060 keepalive_timeout, 

1061 timeout_worker_healthcheck, 

1062 max_requests_before_restart, 

1063 max_requests_before_restart_jitter: int | None, 

1064 limit_concurrency: int | None, 

1065 enforce_prisma_migration_check: bool, 

1066 use_v2_migration_resolver: bool, 

1067 use_legacy_migration_resolver: bool, 

1068 reload: bool, 

1069 prometheus_metrics_port: int | None, 

1070): 

1071 if cli_args: 1071 ↛ 1072line 1071 didn't jump to line 1072 because the condition on line 1071 was never true

1072 if cli_args == ("xai-oauth", "login"): 

1073 from litellm.llms.xai.oauth import XAIOAuthAuthenticator 

1074 

1075 authenticator: Final = XAIOAuthAuthenticator() 

1076 auth_data: Final = authenticator.login() 

1077 click.echo(f"xAI OAuth login successful. Credentials saved to {authenticator.auth_file}.") 

1078 if auth_data.get("expires_at"): 

1079 click.echo(f"Access token expires at {auth_data['expires_at']}.") 

1080 return 

1081 raise click.UsageError(f"Unknown command: {' '.join(cli_args)}") 

1082 

1083 if setup: 1083 ↛ 1084line 1083 didn't jump to line 1084 because the condition on line 1083 was never true

1084 from litellm.setup_wizard import run_setup_wizard 

1085 

1086 run_setup_wizard() 

1087 return 

1088 

1089 args: Final = locals() 

1090 try: 

1091 from litellm.proxy.proxy_server import ( 

1092 KeyManagementSettings, 

1093 ProxyConfig, 

1094 app, 

1095 save_worker_config, 

1096 ) 

1097 except ModuleNotFoundError as e: 

1098 raise ModuleNotFoundError(f"Missing dependency {e}. Run `pip install 'litellm[proxy]'`") from e 

1099 if version is True: 1099 ↛ 1100line 1099 didn't jump to line 1100 because the condition on line 1099 was never true

1100 ProxyInitializationHelpers._echo_litellm_version() 

1101 return 

1102 if validate_config is True: 1102 ↛ 1103line 1102 didn't jump to line 1103 because the condition on line 1102 was never true

1103 ProxyInitializationHelpers._run_config_validation(config) 

1104 return 

1105 if model and "ollama" in model and api_base is None: 1105 ↛ 1106line 1105 didn't jump to line 1106 because the condition on line 1105 was never true

1106 ProxyInitializationHelpers._run_ollama_serve() 

1107 if health is True: 1107 ↛ 1108line 1107 didn't jump to line 1108 because the condition on line 1107 was never true

1108 ProxyInitializationHelpers._run_health_check(host, port) 

1109 return 

1110 if test is True: 1110 ↛ 1111line 1110 didn't jump to line 1111 because the condition on line 1110 was never true

1111 ProxyInitializationHelpers._run_test_chat_completion(host, port, model, test) 

1112 return 

1113 else: 

1114 if headers: 1114 ↛ 1115line 1114 didn't jump to line 1115 because the condition on line 1114 was never true

1115 headers = json.loads(headers) 

1116 save_worker_config( 

1117 model=model, 

1118 alias=alias, 

1119 api_base=api_base, 

1120 api_version=api_version, 

1121 debug=debug, 

1122 detailed_debug=detailed_debug, 

1123 temperature=temperature, 

1124 max_tokens=max_tokens, 

1125 request_timeout=request_timeout, 

1126 max_budget=max_budget, 

1127 drop_params=drop_params, 

1128 add_function_to_prompt=add_function_to_prompt, 

1129 headers=headers, 

1130 save=save, 

1131 config=config, 

1132 use_queue=use_queue, 

1133 ) 

1134 if run_granian: 1134 ↛ 1135line 1134 didn't jump to line 1135 because the condition on line 1134 was never true

1135 try: 

1136 import granian # noqa: F401 

1137 except ImportError as e: 

1138 raise ImportError( 

1139 "granian must be installed to use --run_granian. " 

1140 "Run `pip install granian` or `pip install 'litellm[proxy]'` " 

1141 "(Granian requires Python 3.10+)." 

1142 ) from e 

1143 else: 

1144 try: 

1145 import uvicorn 

1146 except Exception: 

1147 raise ImportError("uvicorn, gunicorn needs to be imported. Run - `pip install 'litellm[proxy]'`") 

1148 

1149 db_connection_pool_limit = 100 

1150 # Starts optional due to config fallback checks; guaranteed non-None before use. 

1151 db_connection_timeout: int | float | None = 60 

1152 db_connect_timeout: int | float | None = None 

1153 db_socket_timeout: int | float | None = None 

1154 db_disable_prepared_statements: bool = False 

1155 db_extra_connection_params: dict | None = None 

1156 db_statement_timeout: float | None = None 

1157 db_lock_timeout: float | None = None 

1158 general_settings = {} 

1159 ### GET DB TOKEN FOR RDS IAM / AZURE ENTRA AUTH ### 

1160 

1161 from litellm.proxy.db.db_url_settings import DatabaseURLSettings 

1162 from litellm.proxy.db.token_auth import ( 

1163 AZURE_POSTGRESQL_AUTH_ENV_VAR, 

1164 IAM_TOKEN_DB_AUTH_ENV_VAR, 

1165 resolve_database_token_auth, 

1166 token_auth_flag_enabled, 

1167 ) 

1168 

1169 wants_rds_iam: Final = iam_token_db_auth or token_auth_flag_enabled( 

1170 os.getenv(IAM_TOKEN_DB_AUTH_ENV_VAR), env_var=IAM_TOKEN_DB_AUTH_ENV_VAR 

1171 ) 

1172 wants_azure_entra: Final = azure_postgresql_auth or token_auth_flag_enabled( 

1173 os.getenv(AZURE_POSTGRESQL_AUTH_ENV_VAR), env_var=AZURE_POSTGRESQL_AUTH_ENV_VAR 

1174 ) 

1175 if wants_rds_iam: 1175 ↛ 1176line 1175 didn't jump to line 1176 because the condition on line 1175 was never true

1176 os.environ[IAM_TOKEN_DB_AUTH_ENV_VAR] = "True" 

1177 if wants_azure_entra: 1177 ↛ 1178line 1177 didn't jump to line 1178 because the condition on line 1177 was never true

1178 os.environ[AZURE_POSTGRESQL_AUTH_ENV_VAR] = "True" 

1179 if wants_rds_iam or wants_azure_entra: 1179 ↛ 1180line 1179 didn't jump to line 1180 because the condition on line 1179 was never true

1180 DatabaseURLSettings.from_env().apply_writer_url_to_env() 

1181 

1182 ### DECRYPT ENV VAR ### 

1183 

1184 from litellm.secret_managers.aws_secret_manager import decrypt_env_var 

1185 

1186 if os.getenv("USE_AWS_KMS", None) is not None and os.getenv("USE_AWS_KMS") == "True": 1186 ↛ 1188line 1186 didn't jump to line 1188 because the condition on line 1186 was never true

1187 ## V2 IMPLEMENTATION OF AWS KMS - USER WANTS TO DECRYPT MULTIPLE KEYS IN THEIR ENV 

1188 new_env_var: Final = decrypt_env_var() 

1189 

1190 for k, v in new_env_var.items(): 

1191 os.environ[k] = v 

1192 

1193 litellm_settings = None 

1194 if config is not None: 1194 ↛ 1283line 1194 didn't jump to line 1283 because the condition on line 1194 was always true

1195 """ 

1196 Allow user to pass in db url via config 

1197 

1198 read from there and save it to os.env['DATABASE_URL'] 

1199 """ 

1200 try: 

1201 import asyncio 

1202 

1203 except Exception: 

1204 raise ImportError("yaml needs to be imported. Run - `pip install 'litellm[proxy]'`") 

1205 

1206 proxy_config: Final = ProxyConfig() 

1207 _config: Final = asyncio.run(proxy_config.get_config(config_file_path=config)) 

1208 

1209 ### LITELLM SETTINGS ### 

1210 litellm_settings = _config.get("litellm_settings", None) 

1211 if ( 1211 ↛ 1216line 1211 didn't jump to line 1216 because the condition on line 1211 was never true

1212 litellm_settings is not None 

1213 and "json_logs" in litellm_settings 

1214 and litellm_settings["json_logs"] is True 

1215 ): 

1216 import litellm 

1217 

1218 litellm.json_logs = True 

1219 

1220 litellm._turn_on_json() 

1221 ### GENERAL SETTINGS ### 

1222 general_settings = _config.get("general_settings", {}) 

1223 if general_settings is None: 1223 ↛ 1224line 1223 didn't jump to line 1224 because the condition on line 1223 was never true

1224 general_settings = {} 

1225 ### LOAD KEY MANAGEMENT SETTINGS FIRST (needed for custom secret manager) ### 

1226 key_management_settings: Final = general_settings.get("key_management_settings", None) 

1227 if key_management_settings is not None: 1227 ↛ 1228line 1227 didn't jump to line 1228 because the condition on line 1227 was never true

1228 import litellm 

1229 

1230 litellm._key_management_settings = KeyManagementSettings(**key_management_settings) 

1231 

1232 if general_settings: 1232 ↛ 1238line 1232 didn't jump to line 1238 because the condition on line 1232 was always true

1233 ### LOAD SECRET MANAGER ### 

1234 key_management_system: Final = general_settings.get("key_management_system", None) 

1235 proxy_config.initialize_secret_manager( 

1236 key_management_system=key_management_system, config_file_path=config 

1237 ) 

1238 database_url = general_settings.get("database_url", None) 

1239 if database_url is None and os.getenv("DATABASE_URL") is None: 1239 ↛ 1241line 1239 didn't jump to line 1241 because the condition on line 1239 was never true

1240 # Use helper function to construct DATABASE_URL from individual variables 

1241 from litellm.proxy.utils import construct_database_url_from_env_vars 

1242 

1243 database_url = construct_database_url_from_env_vars() 

1244 if database_url: 

1245 os.environ["DATABASE_URL"] = database_url 

1246 db_connection_pool_limit = general_settings.get( 

1247 "database_connection_pool_limit", 

1248 LiteLLMDatabaseConnectionPool.database_connection_pool_limit.value, 

1249 ) 

1250 db_connection_timeout = general_settings.get("database_connection_timeout") 

1251 if db_connection_timeout is None: 1251 ↛ 1253line 1251 didn't jump to line 1253 because the condition on line 1251 was always true

1252 db_connection_timeout = general_settings.get("database_connection_pool_timeout") 

1253 if db_connection_timeout is None: 1253 ↛ 1255line 1253 didn't jump to line 1255 because the condition on line 1253 was always true

1254 db_connection_timeout = LiteLLMDatabaseConnectionPool.database_connection_pool_timeout.value 

1255 db_connect_timeout = general_settings.get("database_connect_timeout") 

1256 db_socket_timeout = general_settings.get("database_socket_timeout") 

1257 _disable_prepared_statements: Final = general_settings.get("database_disable_prepared_statements", False) 

1258 if isinstance(_disable_prepared_statements, str): 1258 ↛ 1259line 1258 didn't jump to line 1259 because the condition on line 1258 was never true

1259 from litellm.secret_managers.main import str_to_bool 

1260 

1261 db_disable_prepared_statements = str_to_bool(_disable_prepared_statements) is True 

1262 else: 

1263 db_disable_prepared_statements = bool(_disable_prepared_statements) 

1264 db_extra_connection_params = general_settings.get("database_extra_connection_params") 

1265 db_timeouts: Final = DatabaseTimeoutSettings.model_validate(general_settings) 

1266 db_statement_timeout = db_timeouts.database_statement_timeout 

1267 db_lock_timeout = db_timeouts.database_lock_timeout 

1268 if database_url and database_url.startswith("os.environ/"): 1268 ↛ 1269line 1268 didn't jump to line 1269 because the condition on line 1268 was never true

1269 original_dir: Final = os.getcwd() 

1270 # set the working directory to where this script is 

1271 sys.path.insert( 

1272 0, os.path.abspath("../..") 

1273 ) # Adds the parent directory to the system path - for litellm local dev 

1274 import litellm 

1275 from litellm import get_secret_str 

1276 

1277 database_url = get_secret_str(database_url, default_value=None) 

1278 os.chdir(original_dir) 

1279 if database_url is not None and isinstance(database_url, str): 1279 ↛ 1283line 1279 didn't jump to line 1283 because the condition on line 1279 was always true

1280 os.environ["DATABASE_URL"] = database_url 

1281 

1282 # Handle database URL construction when no config file is used 

1283 if config is None and os.getenv("DATABASE_URL") is None: 1283 ↛ 1285line 1283 didn't jump to line 1285 because the condition on line 1283 was never true

1284 # Use helper function to construct DATABASE_URL from individual variables 

1285 from litellm.proxy.utils import construct_database_url_from_env_vars 

1286 

1287 database_url = construct_database_url_from_env_vars() 

1288 if database_url: 

1289 os.environ["DATABASE_URL"] = database_url 

1290 

1291 # Set default values for connection pool settings when no config is used 

1292 if config is None: 1292 ↛ 1293line 1292 didn't jump to line 1293 because the condition on line 1292 was never true

1293 db_connection_pool_limit = LiteLLMDatabaseConnectionPool.database_connection_pool_limit.value 

1294 db_connection_timeout = LiteLLMDatabaseConnectionPool.database_connection_pool_timeout.value 

1295 

1296 if os.getenv("DATABASE_URL", None) is not None or os.getenv("DIRECT_URL", None) is not None: 1296 ↛ 1449line 1296 didn't jump to line 1449 because the condition on line 1296 was always true

1297 from litellm.proxy.db.db_url_settings import ( 

1298 DISABLE_PREPARED_STATEMENTS_ENV_VAR, 

1299 add_missing_query_params, 

1300 idle_lifetime_params, 

1301 reader_shareable_params, 

1302 translate_libpq_ssl_params, 

1303 unsupported_db_scheme, 

1304 unsupported_db_scheme_message, 

1305 ) 

1306 

1307 for _db_env in ("DATABASE_URL", "DIRECT_URL"): 

1308 _candidate_url = os.getenv(_db_env) 

1309 if _candidate_url is None: 

1310 continue 

1311 _bad_scheme = unsupported_db_scheme(_candidate_url) 

1312 if _bad_scheme is not None: 1312 ↛ 1313line 1312 didn't jump to line 1313 because the condition on line 1312 was never true

1313 print( 

1314 f"\033[1;31mLiteLLM Proxy: {unsupported_db_scheme_message(_db_env, _bad_scheme)}\033[0m", 

1315 file=sys.stderr, 

1316 flush=True, 

1317 ) 

1318 sys.exit(1) 

1319 from litellm.secret_managers.main import get_secret 

1320 

1321 env_disable_prepared_statements: Final = token_auth_flag_enabled( 

1322 os.getenv(DISABLE_PREPARED_STATEMENTS_ENV_VAR), env_var=DISABLE_PREPARED_STATEMENTS_ENV_VAR 

1323 ) 

1324 disable_prepared_statements: Final = db_disable_prepared_statements or env_disable_prepared_statements 

1325 connection_url_params: Final = _build_db_connection_url_params( 

1326 connection_limit=db_connection_pool_limit, 

1327 pool_timeout=db_connection_timeout, 

1328 connect_timeout=db_connect_timeout, 

1329 socket_timeout=db_socket_timeout, 

1330 disable_prepared_statements=disable_prepared_statements, 

1331 extra_params=db_extra_connection_params, 

1332 ) 

1333 lifetime_params: Final = idle_lifetime_params(general_settings.get("database_max_idle_connection_lifetime")) 

1334 if os.getenv("DATABASE_URL", None) is not None: 1334 ↛ 1354line 1334 didn't jump to line 1354 because the condition on line 1334 was always true

1335 database_url = get_secret("DATABASE_URL", default_value=None) 

1336 resolved_url: Final[str | None] = str(database_url) if database_url else None 

1337 pg_options: Final[str] = _pg_options_with_timeouts( 

1338 _url_query_value(resolved_url, "options"), 

1339 db_statement_timeout, 

1340 db_lock_timeout, 

1341 ) 

1342 writer_url: Final = ( 

1343 _with_query_value(resolved_url, "options", pg_options) 

1344 if resolved_url and pg_options 

1345 else resolved_url 

1346 ) 

1347 modified_url = append_query_params( 

1348 writer_url, 

1349 connection_url_params, 

1350 ) 

1351 os.environ["DATABASE_URL"] = translate_libpq_ssl_params( 

1352 add_missing_query_params(modified_url, lifetime_params) 

1353 ) 

1354 if os.getenv("DIRECT_URL", None) is not None: 1354 ↛ 1355line 1354 didn't jump to line 1355 because the condition on line 1354 was never true

1355 database_url = os.getenv("DIRECT_URL") 

1356 modified_url = append_query_params(database_url, connection_url_params) 

1357 os.environ["DIRECT_URL"] = translate_libpq_ssl_params( 

1358 add_missing_query_params(modified_url, lifetime_params) 

1359 ) 

1360 # The reader pool is a real pool against the same configured cap, so it 

1361 # gets the allowlisted pool params. Schema-affecting ones, including any 

1362 # the operator smuggled in through database_extra_connection_params, stay 

1363 # on the writer. Anything pinned on the replica URL wins, unlike the 

1364 # writer where the config is applied on top. 

1365 read_replica_url: Final[str | None] = os.getenv("DATABASE_URL_READ_REPLICA") 

1366 if read_replica_url: 1366 ↛ 1367line 1366 didn't jump to line 1367 because the condition on line 1366 was never true

1367 reader_options: Final[str] = _pg_options_with_timeouts( 

1368 _url_query_value(read_replica_url, "options"), 

1369 db_statement_timeout, 

1370 db_lock_timeout, 

1371 ) 

1372 os.environ["DATABASE_URL_READ_REPLICA"] = translate_libpq_ssl_params( 

1373 add_missing_query_params( 

1374 add_missing_query_params( 

1375 _with_query_value(read_replica_url, "options", reader_options) 

1376 if reader_options 

1377 else read_replica_url, 

1378 reader_shareable_params(connection_url_params), 

1379 ), 

1380 lifetime_params, 

1381 ) 

1382 ) 

1383 from litellm_proxy_extras.prisma_toolchain import prisma_cli_available 

1384 

1385 is_prisma_runnable: Final = prisma_cli_available() 

1386 

1387 if is_prisma_runnable: 1387 ↛ 1445line 1387 didn't jump to line 1445 because the condition on line 1387 was always true

1388 from litellm.proxy.db.check_migration import check_prisma_schema_diff 

1389 from litellm.proxy.db.prisma_client import ( 

1390 PrismaManager, 

1391 should_update_prisma_schema, 

1392 ) 

1393 

1394 if should_update_prisma_schema(general_settings.get("disable_prisma_schema_update")) is False: 1394 ↛ 1395line 1394 didn't jump to line 1395 because the condition on line 1394 was never true

1395 check_prisma_schema_diff(db_url=None) 

1396 else: 

1397 use_v2_resolver: Final = resolve_v2_migration_resolver( 

1398 use_legacy_flag=use_legacy_migration_resolver, 

1399 env_value=os.getenv("USE_V2_MIGRATION_RESOLVER"), 

1400 ) 

1401 if deprecated_v2_flag_passed_on_cli() and use_v2_resolver: 1401 ↛ 1402line 1401 didn't jump to line 1402 because the condition on line 1401 was never true

1402 print( 

1403 "\033[1;33mLiteLLM Proxy: --use_v2_migration_resolver is " 

1404 "deprecated and has no effect, because the v2 migration " 

1405 "resolver is now the default. You can safely remove it. To " 

1406 "opt back into the legacy v1 resolver, pass " 

1407 "--use_legacy_migration_resolver.\033[0m" 

1408 ) 

1409 if not use_v2_resolver: 1409 ↛ 1410line 1409 didn't jump to line 1410 because the condition on line 1409 was never true

1410 print( 

1411 "\033[1;33mLiteLLM Proxy: Using the legacy (v1) migration " 

1412 "resolver. It performs the diff-and-force recovery that can " 

1413 "cause schema thrashing during rolling deploys where two " 

1414 "LiteLLM versions contend for the same DB.\033[0m" 

1415 ) 

1416 try: 

1417 setup_ok: Final = PrismaManager.setup_database( 

1418 use_migrate=not use_prisma_db_push, 

1419 use_v2_resolver=use_v2_resolver, 

1420 ) 

1421 except RuntimeError as e: 

1422 # Raised on unrecoverable migration errors: the v2 

1423 # resolver's non-idempotent failures and permission 

1424 # issues, and any `prisma db push` against a 

1425 # partitioned LiteLLM_SpendLogs. 

1426 print( 

1427 f"\033[1;31mLiteLLM Proxy: Database migration cannot proceed. {e}\033[0m", 

1428 file=sys.stderr, 

1429 flush=True, 

1430 ) 

1431 sys.exit(2) 

1432 if not setup_ok: 1432 ↛ 1433line 1432 didn't jump to line 1433 because the condition on line 1432 was never true

1433 if enforce_prisma_migration_check: 

1434 print( 

1435 "\033[1;31mLiteLLM Proxy: Database setup failed after multiple retries. " 

1436 "The proxy cannot start safely. Please check your database connection and migration status.\033[0m" 

1437 ) 

1438 sys.exit(1) 

1439 else: 

1440 print( 

1441 "\033[1;33mLiteLLM Proxy: Database migration failed but continuing startup because " 

1442 "ENFORCE_PRISMA_MIGRATION_CHECK is disabled. The schema may be behind the code.\033[0m" 

1443 ) 

1444 else: 

1445 print( 

1446 "Unable to connect to DB. DATABASE_URL found in environment, but the prisma CLI is neither on " 

1447 "PATH nor importable as a package." 

1448 ) 

1449 pgbouncer_settings: Final = PgBouncerSettings() 

1450 upstream_database_url: Final = os.getenv("DATABASE_URL") 

1451 if pgbouncer_settings.enabled and upstream_database_url is not None: 1451 ↛ 1452line 1451 didn't jump to line 1452 because the condition on line 1451 was never true

1452 pooled_database_url: Final = start_in_container_pgbouncer( 

1453 pgbouncer_settings, upstream_database_url, token_auth=resolve_database_token_auth() 

1454 ) 

1455 if isinstance(pooled_database_url, PgBouncerError): 

1456 print( 

1457 f"\033[1;31mLiteLLM Proxy: LITELLM_PGBOUNCER_ENABLED is set but the in-container pgbouncer " 

1458 f"could not start: {pooled_database_url.reason}\033[0m", 

1459 file=sys.stderr, 

1460 flush=True, 

1461 ) 

1462 sys.exit(1) 

1463 export_pooled_database_url(pooled_database_url) 

1464 if port == 4000 and ProxyInitializationHelpers._is_port_in_use(port): 1464 ↛ 1465line 1464 didn't jump to line 1465 because the condition on line 1464 was never true

1465 port = random.randint(1024, 49152) 

1466 if prometheus_metrics_port == port: 1466 ↛ 1467line 1466 didn't jump to line 1467 because the condition on line 1466 was never true

1467 raise click.UsageError("--prometheus_metrics_port must differ from --port") 

1468 

1469 import litellm 

1470 

1471 if detailed_debug is True: 1471 ↛ 1472line 1471 didn't jump to line 1472 because the condition on line 1471 was never true

1472 litellm._turn_on_debug() 

1473 

1474 # DO NOT DELETE - enables global variables to work across files 

1475 from litellm.proxy.proxy_server import app 

1476 

1477 os.environ["NUM_WORKERS"] = str(num_workers) 

1478 

1479 # Auto-create PROMETHEUS_MULTIPROC_DIR for multi-worker setups 

1480 prometheus_multiproc_dir: Final = ProxyInitializationHelpers._maybe_setup_prometheus_multiproc_dir( 

1481 num_workers=num_workers, 

1482 litellm_settings=litellm_settings if config else None, 

1483 prometheus_metrics_port=prometheus_metrics_port, 

1484 ) 

1485 

1486 # Skip server startup if requested (after all setup is done) 

1487 if skip_server_startup: 1487 ↛ 1488line 1487 didn't jump to line 1488 because the condition on line 1487 was never true

1488 print("LiteLLM: Setup complete. Skipping server startup as requested.") 

1489 return 

1490 

1491 if prometheus_metrics_port is not None and prometheus_multiproc_dir is not None: 1491 ↛ 1492line 1491 didn't jump to line 1492 because the condition on line 1491 was never true

1492 from litellm.proxy.prometheus_metrics_server import MetricsServerStartupError, start_metrics_server_process 

1493 

1494 try: 

1495 metrics_process: Final = start_metrics_server_process( 

1496 host=host, port=prometheus_metrics_port, multiproc_dir=prometheus_multiproc_dir 

1497 ) 

1498 except MetricsServerStartupError as error: 

1499 raise click.ClickException(str(error)) from error 

1500 print( 

1501 f"\033[1;32mLiteLLM: Serving Prometheus metrics on {host}:{prometheus_metrics_port}/metrics " 

1502 f"(pid {metrics_process.pid})\033[0m" 

1503 ) 

1504 

1505 running_uvicorn: Final = run_gunicorn is False and run_hypercorn is False 

1506 uvicorn_args: Final = ProxyInitializationHelpers._get_default_unvicorn_init_args( 

1507 host=host, 

1508 port=port, 

1509 log_config=log_config, 

1510 keepalive_timeout=keepalive_timeout, 

1511 timeout_worker_healthcheck=(timeout_worker_healthcheck if running_uvicorn else None), 

1512 ) 

1513 # Optional: recycle uvicorn workers after N requests 

1514 if max_requests_before_restart is not None: 1514 ↛ 1515line 1514 didn't jump to line 1515 because the condition on line 1514 was never true

1515 uvicorn_args["limit_max_requests"] = max_requests_before_restart 

1516 if run_gunicorn is False and run_hypercorn is False and run_granian is False: 1516 ↛ 1545line 1516 didn't jump to line 1545 because the condition on line 1516 was always true

1517 if limit_concurrency is not None: 1517 ↛ 1518line 1517 didn't jump to line 1518 because the condition on line 1517 was never true

1518 uvicorn_args["limit_concurrency"] = limit_concurrency 

1519 if max_requests_before_restart_jitter is not None: 1519 ↛ 1520line 1519 didn't jump to line 1520 because the condition on line 1519 was never true

1520 ProxyInitializationHelpers._apply_uvicorn_max_requests_jitter( 

1521 uvicorn_args=uvicorn_args, 

1522 max_requests_before_restart=max_requests_before_restart, 

1523 jitter=max_requests_before_restart_jitter, 

1524 ) 

1525 if ssl_certfile_path is not None and ssl_keyfile_path is not None: 1525 ↛ 1526line 1525 didn't jump to line 1526 because the condition on line 1525 was never true

1526 print( 

1527 f"\033[1;32mLiteLLM Proxy: Using SSL with certfile: {ssl_certfile_path} and keyfile: {ssl_keyfile_path}\033[0m\n" 

1528 ) 

1529 uvicorn_args["ssl_keyfile"] = ssl_keyfile_path 

1530 uvicorn_args["ssl_certfile"] = ssl_certfile_path 

1531 

1532 loop_type: Final = ProxyInitializationHelpers._get_loop_type() 

1533 if loop_type: 1533 ↛ 1536line 1533 didn't jump to line 1536 because the condition on line 1533 was always true

1534 uvicorn_args["loop"] = loop_type 

1535 

1536 if reload: 1536 ↛ 1537line 1536 didn't jump to line 1537 because the condition on line 1536 was never true

1537 ProxyInitializationHelpers._configure_dev_reload(uvicorn_args, config) 

1538 

1539 if num_workers > 1: 1539 ↛ 1540line 1539 didn't jump to line 1540 because the condition on line 1539 was never true

1540 start_query_engine_reaper() 

1541 uvicorn.run( 

1542 **uvicorn_args, 

1543 workers=num_workers, 

1544 ) 

1545 elif run_gunicorn is True: 

1546 ProxyInitializationHelpers._run_gunicorn_server( 

1547 host=host, 

1548 port=port, 

1549 app=app, 

1550 num_workers=num_workers, 

1551 ssl_certfile_path=ssl_certfile_path, 

1552 ssl_keyfile_path=ssl_keyfile_path, 

1553 max_requests_before_restart=max_requests_before_restart, 

1554 max_requests_before_restart_jitter=max_requests_before_restart_jitter, 

1555 ) 

1556 elif run_hypercorn is True: 

1557 ProxyInitializationHelpers._init_hypercorn_server( 

1558 app=app, 

1559 host=host, 

1560 port=port, 

1561 ssl_certfile_path=ssl_certfile_path, 

1562 ssl_keyfile_path=ssl_keyfile_path, 

1563 ciphers=ciphers, 

1564 ) 

1565 elif run_granian is True: 

1566 ProxyInitializationHelpers._init_granian_server( 

1567 host=host, 

1568 port=port, 

1569 num_workers=num_workers, 

1570 ssl_certfile_path=ssl_certfile_path, 

1571 ssl_keyfile_path=ssl_keyfile_path, 

1572 max_requests_before_restart=max_requests_before_restart, 

1573 ciphers=ciphers, 

1574 granian_runtime_threads=granian_threads, 

1575 ) 

1576 

1577 

1578if __name__ == "__main__": 1578 ↛ 1579line 1578 didn't jump to line 1579 because the condition on line 1578 was never true

1579 run_server()