Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/client/cli/commands/agents.py: 0%

391 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1import json 

2import os 

3import re 

4import shutil 

5import subprocess 

6import sys 

7import tempfile 

8from collections.abc import Callable, Mapping, Sequence 

9from dataclasses import dataclass 

10from pathlib import Path 

11from types import MappingProxyType 

12from typing import Final, Literal, TypeAlias 

13 

14import click 

15import requests 

16from pydantic import BaseModel, ConfigDict, TypeAdapter, ValidationError 

17 

18from .auth import CliContextObj, context_secret_vault, get_stored_api_key, login 

19from .claude_settings import ClaudeSettingsError, install_statusline_script 

20from .cmd_quoting import quote_for_cmd 

21from .pi import ( 

22 LITELLM_PROXY_API_KEY_ENV, 

23 PI_PROVIDER_NAME, 

24 ListingFailure, 

25 PiSyncError, 

26 fetch_model_ids, 

27 fetch_model_limits, 

28 models_json_path, 

29 sync_models_json, 

30) 

31 

32ANTHROPIC_BASE_URL_ENV: Final = "ANTHROPIC_BASE_URL" 

33ANTHROPIC_AUTH_TOKEN_ENV: Final = "ANTHROPIC_AUTH_TOKEN" 

34ANTHROPIC_API_KEY_ENV: Final = "ANTHROPIC_API_KEY" 

35ENABLE_TOOL_SEARCH_ENV: Final = "ENABLE_TOOL_SEARCH" 

36ENABLE_TOOL_SEARCH_VALUE: Final = "true" 

37ENABLE_GATEWAY_MODEL_DISCOVERY_ENV: Final = "CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY" 

38ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE: Final = "1" 

39OPENAI_BASE_URL_ENV: Final = "OPENAI_BASE_URL" 

40OPENAI_API_KEY_ENV: Final = "OPENAI_API_KEY" 

41OPENCODE_CONFIG_CONTENT_ENV: Final = "OPENCODE_CONFIG_CONTENT" 

42OPENCODE_PROVIDER_ID: Final = "litellm" 

43OPENCODE_PROVIDER_NAME: Final = "LiteLLM" 

44OPENCODE_PROVIDER_NPM: Final = "@ai-sdk/openai-compatible" 

45 

46_SKIP_VERIFY_FLAG: Final = "--skip-verify" 

47 

48PROFILE_ANTHROPIC: Final = "anthropic" 

49PROFILE_OPENAI: Final = "openai" 

50PROFILE_LITELLM: Final = "litellm" 

51 

52_KNOWN_AGENTS: Final[dict[str, tuple[str, frozenset[str]]]] = { 

53 "claude": ("Claude Code", frozenset({PROFILE_ANTHROPIC})), 

54 "codex": ("Codex", frozenset({PROFILE_OPENAI})), 

55 "opencode": ("OpenCode", frozenset({PROFILE_OPENAI})), 

56 "pi": ("pi", frozenset({PROFILE_LITELLM})), 

57} 

58 

59_INSTALL_DOCS: Final[dict[str, str]] = { 

60 "claude": "https://docs.claude.com/en/docs/claude-code/setup", 

61 "codex": "https://developers.openai.com/codex/cli", 

62 "opencode": "https://opencode.ai/docs", 

63 "pi": "https://pi.dev", 

64} 

65 

66_HIDDEN_AGENTS: Final = frozenset({"pi"}) 

67 

68CODEX_PROXY_PROVIDER: Final = "litellm" 

69CODEX_HOME_ENV: Final = "CODEX_HOME" 

70CODEX_MODEL_CATALOG_FILENAME: Final = "litellm-models.json" 

71_CODEX_BASE_INSTRUCTIONS_PATH: Final = Path(__file__).with_name("codex_base_instructions.md") 

72_CODEX_PREFLIGHT_TIMEOUT_SECONDS: Final = 10.0 

73 

74 

75class AgentRunError(Exception): 

76 """Raised for any user-actionable failure while preparing to run an agent.""" 

77 

78 

79def agent_profile(command: str) -> tuple[str, frozenset[str]]: 

80 """Return the (display name, env profiles) for a wrapped command. 

81 

82 Known agents map to the API family they speak. Anything else gets both 

83 families so it works regardless of which env vars the tool reads. 

84 """ 

85 base: Final = os.path.basename(command) 

86 if base in _KNOWN_AGENTS: 

87 return _KNOWN_AGENTS[base] 

88 return base, frozenset({PROFILE_ANTHROPIC, PROFILE_OPENAI}) 

89 

90 

91def build_agent_env( 

92 base_env: Mapping[str, str], 

93 base_url: str, 

94 api_key: str, 

95 profiles: frozenset[str], 

96) -> dict[str, str]: 

97 """Return a copy of base_env wired to route the agent through the proxy. 

98 

99 Anthropic clients (Claude Code) append /v1/messages to ANTHROPIC_BASE_URL, 

100 so it stays the bare proxy root; OpenAI clients (Codex, OpenCode) expect the 

101 /v1 suffix on OPENAI_BASE_URL. ANTHROPIC_API_KEY is dropped so a stray 

102 Anthropic key cannot win over the bearer token we set. ENABLE_TOOL_SEARCH 

103 defaults to true because Claude Code turns tool search off when 

104 ANTHROPIC_BASE_URL is not a first-party Anthropic host; a value already in 

105 the environment is left alone. CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY 

106 defaults to 1 so Claude Code (v2.1.129+) fills its /model picker from the 

107 proxy's /v1/models; likewise left alone when already set. 

108 pi ignores both base URL variables and instead resolves $LITELLM_PROXY_API_KEY 

109 from its synced models.json provider entry. 

110 """ 

111 env: Final = dict(base_env) 

112 root: Final = base_url.rstrip("/") 

113 if PROFILE_ANTHROPIC in profiles: 

114 env[ANTHROPIC_BASE_URL_ENV] = root 

115 env[ANTHROPIC_AUTH_TOKEN_ENV] = api_key 

116 env.pop(ANTHROPIC_API_KEY_ENV, None) 

117 if ENABLE_TOOL_SEARCH_ENV not in env: 

118 env[ENABLE_TOOL_SEARCH_ENV] = ENABLE_TOOL_SEARCH_VALUE 

119 if ENABLE_GATEWAY_MODEL_DISCOVERY_ENV not in env: 

120 env[ENABLE_GATEWAY_MODEL_DISCOVERY_ENV] = ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE 

121 if PROFILE_OPENAI in profiles: 

122 env[OPENAI_BASE_URL_ENV] = root + "/v1" 

123 env[OPENAI_API_KEY_ENV] = api_key 

124 if PROFILE_LITELLM in profiles: 

125 env[LITELLM_PROXY_API_KEY_ENV] = api_key 

126 return env 

127 

128 

129def codex_proxy_provider(base_url: str) -> Mapping[str, str | bool]: 

130 return MappingProxyType( 

131 { 

132 "name": "LiteLLM proxy", 

133 "base_url": base_url.rstrip("/") + "/v1", 

134 "wire_api": "responses", 

135 "supports_websockets": False, 

136 "requires_openai_auth": False, 

137 } 

138 ) 

139 

140 

141def _codex_proxy_args(base_url: str) -> list[str]: 

142 """Codex `-c` overrides that point it at the proxy. 

143 

144 Codex ignores OPENAI_BASE_URL (it always dials api.openai.com), so the env 

145 profile alone cannot route it. It does honor a custom provider, so define one 

146 inline; supports_websockets=false forces the HTTP/SSE Responses transport 

147 because the proxy does not speak the Responses WebSocket protocol. The key is 

148 read from OPENAI_API_KEY, which build_agent_env already exports. 

149 """ 

150 provider: Final = f"model_providers.{CODEX_PROXY_PROVIDER}" 

151 return [ 

152 "-c", 

153 f'model_provider="{CODEX_PROXY_PROVIDER}"', 

154 *( 

155 argument 

156 for key, value in codex_proxy_provider(base_url).items() 

157 for argument in ("-c", f"{provider}.{key}={json.dumps(value)}") 

158 ), 

159 "-c", 

160 f'{provider}.env_key="{OPENAI_API_KEY_ENV}"', 

161 "-c", 

162 f"{provider}.http_headers={{}}", 

163 ] 

164 

165 

166_PROXY_ARGS: Final[dict[str, Callable[[str], list[str]]]] = { 

167 "codex": _codex_proxy_args, 

168} 

169 

170 

171def prepare_pi( 

172 base_url: str, 

173 api_key: str, 

174 base_env: Mapping[str, str], 

175 *, 

176 get: Callable[..., requests.Response] = requests.get, 

177) -> tuple[str, ...]: 

178 """Sync the proxy's model list into pi's models.json before handoff. 

179 

180 pi has no base-URL env vars, so this file is the only way to point it at the 

181 proxy. Only the litellm provider entry is touched; the synced entry references 

182 the key as $LITELLM_PROXY_API_KEY, which build_agent_env exports. The returned 

183 --model pin is needed because pi ignores a bare --provider when picking the 

184 interactive startup model; a user-supplied --model comes later in argv and wins. 

185 """ 

186 ids: Final = fetch_model_ids(base_url, api_key, get=get) 

187 if isinstance(ids, PiSyncError): 

188 raise AgentRunError( 

189 f"{ids.message} pi would have nothing to run." if ids.kind is ListingFailure.EMPTY else ids.message 

190 ) 

191 limits: Final = fetch_model_limits(base_url, api_key, get=get) 

192 path: Final = models_json_path(base_env) 

193 error: Final = sync_models_json(path, base_url, ids, limits) 

194 if error is not None: 

195 raise AgentRunError(error.message) 

196 click.echo(f"litellm: synced {len(ids)} proxy models into {path}") 

197 return ("--model", f"{PI_PROVIDER_NAME}/{ids[0]}") 

198 

199 

200def _warn(message: str) -> None: 

201 click.echo(message, err=True) 

202 

203 

204_CODEX_STOP_HOOKS_DECLARED: Final = re.compile( 

205 r"^\s*(\[\[\s*\"?hooks\"?\s*\.\s*\"?Stop\"?\s*\]\]|\"?hooks\"?(?:\s*\.\s*\"?Stop\"?)?\s*=|\[\s*\"?hooks\"?\s*\])", 

206 re.MULTILINE, 

207) 

208 

209 

210def codex_config_path(base_env: Mapping[str, str]) -> Path: 

211 return Path(base_env.get("CODEX_HOME") or Path.home() / ".codex") / "config.toml" 

212 

213 

214def codex_declares_stop_hooks(config_path: Path) -> bool: 

215 """A config that cannot be read or decoded declares nothing we can see; Codex reports its own 

216 TOML failure at launch, so the pre-check must not be the thing that stops `lite codex`.""" 

217 try: 

218 return _CODEX_STOP_HOOKS_DECLARED.search(config_path.read_text(encoding="utf-8")) is not None 

219 except (OSError, UnicodeDecodeError): 

220 return False 

221 

222 

223def prepare_codex( 

224 base_url: str, 

225 api_key: str, 

226 base_env: Mapping[str, str], 

227 *, 

228 install: Callable[[], str] = install_statusline_script, 

229 warn: Callable[[str], None] = _warn, 

230) -> tuple[str, ...]: 

231 """A `-c hooks.Stop=` session flag replaces the user's whole Stop list, so their own hooks win over ours.""" 

232 if codex_declares_stop_hooks(codex_config_path(base_env)): 

233 warn("litellm: your Codex config already declares hooks; not adding the routed-model Stop hook") 

234 return () 

235 try: 

236 command: Final = install() 

237 except ClaudeSettingsError as e: 

238 raise AgentRunError(str(e)) from e 

239 return ("-c", f'hooks.Stop=[{{hooks=[{{type="command",command={json.dumps(command)}}}]}}]') 

240 

241 

242_Preparer: TypeAlias = Callable[[str, str, Mapping[str, str]], Sequence[str]] 

243 

244_PREPARERS: Final[Mapping[str, _Preparer]] = MappingProxyType({"pi": prepare_pi, "codex": prepare_codex}) 

245 

246 

247def agent_launch_args(command: str, base_url: str) -> list[str]: 

248 """Extra CLI args an agent needs to actually honor the proxy. 

249 

250 Claude Code and OpenCode respect the exported env vars, so they get nothing 

251 here; Codex needs its provider pointed via config overrides. 

252 """ 

253 builder: Final = _PROXY_ARGS.get(os.path.basename(command)) 

254 return builder(base_url) if builder else [] 

255 

256 

257class ListedModel(BaseModel): 

258 """The fields of a /v1/models entry that an OpenCode or Codex model entry is built from.""" 

259 

260 id: str 

261 mode: str | None = None 

262 max_input_tokens: int | None = None 

263 max_output_tokens: int | None = None 

264 

265 

266class _ModelListing(BaseModel): 

267 data: tuple[ListedModel, ...] 

268 

269 

270_MODEL_LISTING: Final = TypeAdapter(_ModelListing) 

271_CHAT_MODES: Final[frozenset[str]] = frozenset({"chat", "responses"}) 

272_NO_EXTRA_ENV: Final[Mapping[str, str]] = MappingProxyType({}) 

273 

274 

275@dataclass(frozen=True, slots=True) 

276class ModelSyncSkipped: 

277 reason: str 

278 

279 

280@dataclass(frozen=True, slots=True) 

281class ModelSyncArgs: 

282 """CLI args, placed before the user's own, that hand an agent the synced model list.""" 

283 

284 args: tuple[str, ...] 

285 

286 

287ModelSyncResult: TypeAlias = Mapping[str, str] | ModelSyncArgs | ModelSyncSkipped 

288 

289 

290def _chat_models(models: Sequence[ListedModel]) -> tuple[ListedModel, ...]: 

291 return tuple(m for m in models if m.mode is None or m.mode in _CHAT_MODES) 

292 

293 

294def _fetch_model_listing( 

295 base_url: str, 

296 api_key: str, 

297 *, 

298 get: Callable[..., requests.Response], 

299) -> tuple[ListedModel, ...] | ModelSyncSkipped: 

300 url: Final = base_url.rstrip("/") + "/v1/models" 

301 try: 

302 resp: Final = get(url, headers=MappingProxyType({"Authorization": f"Bearer {api_key}"}), timeout=10) 

303 except requests.RequestException as e: 

304 return ModelSyncSkipped(f"could not reach {url}: {e}") 

305 if resp.status_code != 200: 

306 return ModelSyncSkipped(f"{url} returned HTTP {resp.status_code}") 

307 try: 

308 listing: Final = _MODEL_LISTING.validate_json(resp.content) 

309 except ValidationError: 

310 return ModelSyncSkipped(f"{url} returned an unexpected body") 

311 return listing.data 

312 

313 

314class _OpenCodeLimit(BaseModel): 

315 context: int 

316 output: int 

317 

318 

319class _OpenCodeModel(BaseModel): 

320 name: str 

321 limit: _OpenCodeLimit | None = None 

322 

323 

324class _OpenCodeProviderOptions(BaseModel): 

325 baseURL: str 

326 apiKey: str 

327 

328 

329class _OpenCodeProvider(BaseModel): 

330 npm: str 

331 name: str 

332 options: _OpenCodeProviderOptions 

333 models: Mapping[str, _OpenCodeModel] 

334 

335 

336class _OpenCodeConfig(BaseModel): 

337 provider: Mapping[str, _OpenCodeProvider] 

338 

339 

340def _opencode_model_entry(model: ListedModel) -> _OpenCodeModel: 

341 if model.max_input_tokens is None or model.max_output_tokens is None: 

342 return _OpenCodeModel(name=model.id) 

343 return _OpenCodeModel( 

344 name=model.id, limit=_OpenCodeLimit(context=model.max_input_tokens, output=model.max_output_tokens) 

345 ) 

346 

347 

348def opencode_provider_config(base_url: str, models: Sequence[ListedModel]) -> str: 

349 """OPENCODE_CONFIG_CONTENT declaring the proxy as OpenCode provider `litellm`. 

350 

351 One model entry per chat-capable /v1/models row (mode chat, responses, or 

352 unknown), so OpenCode's model picker mirrors what the key can call. The key 

353 is read back through {env:OPENAI_API_KEY}, which build_agent_env exports, so 

354 it never lands in the config text. OpenCode merges this inline config over 

355 the user's own files, leaving unrelated keys and providers untouched. 

356 """ 

357 chat_models: Final = _chat_models(models) 

358 provider: Final = _OpenCodeProvider( 

359 npm=OPENCODE_PROVIDER_NPM, 

360 name=OPENCODE_PROVIDER_NAME, 

361 options=_OpenCodeProviderOptions( 

362 baseURL=base_url.rstrip("/") + "/v1", 

363 apiKey=f"{{env:{OPENAI_API_KEY_ENV}}}", 

364 ), 

365 models=MappingProxyType({m.id: _opencode_model_entry(m) for m in chat_models}), 

366 ) 

367 config: Final = _OpenCodeConfig(provider=MappingProxyType({OPENCODE_PROVIDER_ID: provider})) 

368 return config.model_dump_json(exclude_none=True) 

369 

370 

371def opencode_model_sync_env( 

372 base_env: Mapping[str, str], 

373 base_url: str, 

374 api_key: str, 

375 *, 

376 get: Callable[..., requests.Response] = requests.get, 

377) -> Mapping[str, str] | ModelSyncSkipped: 

378 """Env addition that hands OpenCode the proxy's model list, or why it was skipped. 

379 

380 Fetches /v1/models with the key and packs it into OPENCODE_CONFIG_CONTENT. 

381 An OPENCODE_CONFIG_CONTENT already in the environment is left alone, and a 

382 failed fetch is reported rather than raised: OpenCode still launches on the 

383 plain OPENAI_* env, just without a synced model list. 

384 """ 

385 if OPENCODE_CONFIG_CONTENT_ENV in base_env: 

386 return ModelSyncSkipped(f"{OPENCODE_CONFIG_CONTENT_ENV} is already set") 

387 listing: Final = _fetch_model_listing(base_url, api_key, get=get) 

388 if isinstance(listing, ModelSyncSkipped): 

389 return listing 

390 return MappingProxyType({OPENCODE_CONFIG_CONTENT_ENV: opencode_provider_config(base_url, listing)}) 

391 

392 

393class _CodexTruncationPolicy(BaseModel): 

394 mode: Literal["bytes"] = "bytes" 

395 limit: int = 10_000 

396 

397 

398class _CodexModel(BaseModel): 

399 """One `ModelInfo` entry of a Codex model catalog for a model the installed Codex does not know. 

400 

401 Every field that some Codex release since `model_catalog_json` appeared 

402 (0.105.0) deserializes without a default is spelled out here, so one catalog 

403 parses on all of them; the values match the fallback metadata Codex uses for 

404 a model slug it does not know, so picking such a proxy model behaves the 

405 same as `codex -m` did. 

406 """ 

407 

408 slug: str 

409 display_name: str 

410 description: None = None 

411 supported_reasoning_levels: tuple[()] = () 

412 shell_type: Literal["unified_exec"] = "unified_exec" 

413 visibility: Literal["list"] = "list" 

414 supported_in_api: Literal[True] = True 

415 priority: int 

416 availability_nux: None = None 

417 upgrade: None = None 

418 support_verbosity: Literal[False] = False 

419 supports_reasoning_summaries: Literal[False] = False 

420 supports_parallel_tool_calls: Literal[False] = False 

421 default_verbosity: None = None 

422 apply_patch_tool_type: None = None 

423 truncation_policy: _CodexTruncationPolicy = _CodexTruncationPolicy() 

424 experimental_supported_tools: tuple[()] = () 

425 context_window: int | None 

426 base_instructions: str 

427 

428 

429class _StockCodexUpgrade(BaseModel): 

430 model_config = ConfigDict(extra="allow") 

431 

432 model: str 

433 

434 

435class _StockCodexModel(BaseModel): 

436 """One `ModelInfo` entry as the installed Codex prints it from `codex debug models`. 

437 

438 Only the fields the sync rewrites are named; everything else that release 

439 knows about the model (its reasoning levels, prompt, tool support) rides 

440 along untouched, whatever the release's schema. 

441 """ 

442 

443 model_config = ConfigDict(extra="allow") 

444 

445 slug: str 

446 priority: int 

447 visibility: str 

448 supported_in_api: bool = True 

449 upgrade: _StockCodexUpgrade | None = None 

450 

451 

452class _StockCodexCatalog(BaseModel): 

453 models: tuple[_StockCodexModel, ...] 

454 

455 

456class _CodexCatalog(BaseModel): 

457 models: tuple[_CodexModel | _StockCodexModel, ...] 

458 

459 

460def _codex_catalog_entry( 

461 priority: int, 

462 listed: ListedModel, 

463 stock: _StockCodexModel | None, 

464 served: frozenset[str], 

465 instructions: str, 

466) -> _CodexModel | _StockCodexModel: 

467 if stock is None: 

468 return _CodexModel( 

469 slug=listed.id, 

470 display_name=listed.id, 

471 priority=priority, 

472 context_window=listed.max_input_tokens, 

473 base_instructions=instructions, 

474 ) 

475 upgrade: Final = stock.upgrade if stock.upgrade is not None and stock.upgrade.model in served else None 

476 return stock.model_copy( 

477 update={"priority": priority, "visibility": "list", "supported_in_api": True, "upgrade": upgrade} 

478 ) 

479 

480 

481def codex_model_catalog( 

482 models: Sequence[ListedModel], stock: Sequence[_StockCodexModel], instructions: str 

483) -> str | None: 

484 """The `model_catalog_json` body listing the proxy's chat models, or None if there are none. 

485 

486 Codex refuses an empty catalog, hence None instead of `{"models": []}`. 

487 Passing a catalog replaces Codex's built-in one, so a proxy model the 

488 installed Codex knows keeps that Codex's own entry and the proxy only 

489 decides its place in the picker: the listing orders it, lists it even when 

490 Codex hides it or keeps it off the API, and keeps Codex's upgrade nudge only 

491 when the model it points at is served too. A model Codex does not know gets the fallback 

492 entry, with the same base instructions Codex itself uses so the agent never 

493 runs without a system prompt. 

494 """ 

495 chat_models: Final = _chat_models(models) 

496 if not chat_models: 

497 return None 

498 served: Final = frozenset(m.id for m in chat_models) 

499 known: Final = MappingProxyType({m.slug: m for m in stock}) 

500 catalog: Final = _CodexCatalog( 

501 models=tuple( 

502 _codex_catalog_entry(index, m, known.get(m.id), served, instructions) for index, m in enumerate(chat_models) 

503 ) 

504 ) 

505 return catalog.model_dump_json() 

506 

507 

508def codex_model_catalog_path(env: Mapping[str, str], *, home: Callable[[], Path] = Path.home) -> Path: 

509 override: Final = env.get(CODEX_HOME_ENV) 

510 root: Final = Path(override) if override else home() / ".codex" 

511 return root / CODEX_MODEL_CATALOG_FILENAME 

512 

513 

514def _replace_file(path: Path, text: str) -> None: 

515 with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent, delete=False) as tmp: 

516 _ = tmp.write(text) 

517 try: 

518 os.replace(tmp.name, path) 

519 except OSError: 

520 Path(tmp.name).unlink(missing_ok=True) 

521 raise 

522 

523 

524def _codex_debug_models( 

525 binary: str, 

526 args: Sequence[str], 

527 env: Mapping[str, str], 

528 *, 

529 run: Callable[..., subprocess.CompletedProcess[str]], 

530) -> str | ModelSyncSkipped: 

531 """What `codex debug models` prints with `args` in front, or why the installed Codex could not run it. 

532 

533 The command prints the catalog Codex would launch with, without touching 

534 the network, so it lists the installed Codex's own models and parses a 

535 catalog override the way a launch does. Releases before 0.130.0 have no 

536 such command and are reported the same way. A batch shim goes through 

537 cmd.exe exactly as the launch will. 

538 """ 

539 name: Final = os.path.basename(binary) 

540 command: Final = _windows_command(binary, (binary, *args, "debug", "models")) 

541 try: 

542 completed: Final = run( 

543 command, 

544 env=dict(env), 

545 stdin=subprocess.DEVNULL, 

546 capture_output=True, 

547 encoding="utf-8", 

548 timeout=_CODEX_PREFLIGHT_TIMEOUT_SECONDS, 

549 ) 

550 except (OSError, subprocess.TimeoutExpired) as e: 

551 return ModelSyncSkipped(f"`{name} debug models` failed: {e}") 

552 if completed.returncode == 0: 

553 return completed.stdout 

554 lines: Final = completed.stderr.strip().splitlines() 

555 detail: Final = lines[0] if lines else "no output" 

556 return ModelSyncSkipped(f"`{name} debug models` exited {completed.returncode}: {detail}") 

557 

558 

559def _stock_codex_models( 

560 binary: str, env: Mapping[str, str], *, run: Callable[..., subprocess.CompletedProcess[str]] 

561) -> tuple[_StockCodexModel, ...] | ModelSyncSkipped: 

562 printed: Final = _codex_debug_models(binary, (), env, run=run) 

563 if isinstance(printed, ModelSyncSkipped): 

564 return printed 

565 try: 

566 return _StockCodexCatalog.model_validate_json(printed).models 

567 except ValidationError as e: 

568 name: Final = os.path.basename(binary) 

569 return ModelSyncSkipped(f"`{name} debug models` printed no model catalog: {e.errors()[0]['msg']}") 

570 

571 

572def codex_model_sync_args( 

573 base_env: Mapping[str, str], 

574 base_url: str, 

575 api_key: str, 

576 *, 

577 binary: str = "codex", 

578 get: Callable[..., requests.Response] = requests.get, 

579 run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run, 

580 home: Callable[[], Path] = Path.home, 

581 instructions_path: Path = _CODEX_BASE_INSTRUCTIONS_PATH, 

582) -> ModelSyncArgs | ModelSyncSkipped: 

583 """`-c model_catalog_json=...` pointing Codex at the proxy's model list, or why it was skipped. 

584 

585 Codex has no env or inline equivalent of OPENCODE_CONFIG_CONTENT: the catalog 

586 must be a file, so it is written under $CODEX_HOME (default ~/.codex) and 

587 atomically replaced on every launch. The Codex at `binary` first lists its 

588 own models, so the ones the proxy serves keep that Codex's entries, and then 

589 reads the file back once before it is handed over. The key never lands in 

590 the file. A failed fetch, read, listing, write or read-back is reported 

591 rather than raised: Codex still launches with its built-in catalog and takes 

592 a proxy model by name via -m, and a rejected file stays on disk to be looked 

593 at. 

594 """ 

595 listing: Final = _fetch_model_listing(base_url, api_key, get=get) 

596 if isinstance(listing, ModelSyncSkipped): 

597 return listing 

598 try: 

599 instructions: Final = instructions_path.read_text(encoding="utf-8") 

600 except OSError as e: 

601 return ModelSyncSkipped(f"could not read {instructions_path}: {e}") 

602 path: Final = codex_model_catalog_path(base_env, home=home) 

603 try: 

604 path.parent.mkdir(parents=True, exist_ok=True) 

605 except OSError as e: 

606 return ModelSyncSkipped(f"could not write {path}: {e}") 

607 stock: Final = _stock_codex_models(binary, base_env, run=run) 

608 if isinstance(stock, ModelSyncSkipped): 

609 return stock 

610 catalog: Final = codex_model_catalog(listing, stock, instructions) 

611 if catalog is None: 

612 return ModelSyncSkipped(f"{base_url.rstrip('/')}/v1/models lists no chat models") 

613 try: 

614 _replace_file(path, catalog) 

615 except OSError as e: 

616 return ModelSyncSkipped(f"could not write {path}: {e}") 

617 override: Final = f"model_catalog_json={json.dumps(str(path))}" 

618 read_back: Final = _codex_debug_models(binary, ("-c", override), base_env, run=run) 

619 if isinstance(read_back, ModelSyncSkipped): 

620 return read_back 

621 return ModelSyncArgs(("-c", override)) 

622 

623 

624def agent_model_sync_env( 

625 binary: str, 

626 base_env: Mapping[str, str], 

627 base_url: str, 

628 api_key: str, 

629 skip_verify: bool, 

630 *, 

631 get: Callable[..., requests.Response] = requests.get, 

632 run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run, 

633) -> ModelSyncResult: 

634 """Extra env or args an agent needs to see the proxy's model list. 

635 

636 binary is the resolved path the launch will run (`codex.cmd` on a Windows 

637 npm install). OpenCode takes the list as env, Codex as a `-c` override that 

638 binary has read back first; Claude Code discovers models itself through 

639 CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY. skip_verify means the caller 

640 wants no pre-launch proxy call at all, so the listing is skipped too rather 

641 than hanging on an offline proxy. 

642 """ 

643 agent: Final = os.path.splitext(os.path.basename(binary))[0] 

644 if agent not in ("opencode", "codex"): 

645 return _NO_EXTRA_ENV 

646 if skip_verify: 

647 return ModelSyncSkipped(f"{_SKIP_VERIFY_FLAG} was passed") 

648 if agent == "codex": 

649 return codex_model_sync_args(base_env, base_url, api_key, binary=binary, get=get, run=run) 

650 return opencode_model_sync_env(base_env, base_url, api_key, get=get) 

651 

652 

653def verify_proxy_key( 

654 base_url: str, 

655 api_key: str, 

656 *, 

657 get: Callable[..., requests.Response] = requests.get, 

658) -> None: 

659 """Probe the proxy with the key so bad creds fail here, not inside the agent. 

660 

661 Raises AgentRunError when the proxy is unreachable or rejects the key. Other 

662 non-2xx responses are tolerated; the agent's own call is the real test. 

663 """ 

664 url: Final = base_url.rstrip("/") + "/v1/models" 

665 try: 

666 resp: Final = get(url, headers={"Authorization": f"Bearer {api_key}"}, timeout=10) 

667 except requests.RequestException as e: 

668 raise AgentRunError( 

669 f"Could not reach the LiteLLM proxy at {base_url.rstrip('/')}: {e}. " 

670 "Is it running, and is --base-url (or LITELLM_PROXY_URL) correct?" 

671 ) 

672 if resp.status_code in (401, 403): 

673 raise AgentRunError( 

674 f"LiteLLM rejected your key (HTTP {resp.status_code}). " 

675 "Run `lite login` to refresh it, or pass a valid --api-key." 

676 ) 

677 

678 

679_WINDOWS_SHIM_SUFFIXES: Final[frozenset[str]] = frozenset({".cmd", ".bat"}) 

680_CMD_LINE_BREAKS: Final = ("\r", "\n") 

681 

682 

683def _windows_command(path: str, args: Sequence[str]) -> str | tuple[str, ...]: 

684 """Build what CreateProcess runs, routing batch shims through cmd.exe. 

685 

686 npm installs Claude Code as `claude.cmd`, which PATHEXT lets shutil.which 

687 resolve but CreateProcess refuses to run (WinError 193), so a shim has to go 

688 through the command processor. cmd.exe does not follow the C runtime quoting 

689 that subprocess would apply to an argument list, and it would split on `&` or 

690 `|` in a forwarded argument, so the shim case is emitted as one verbatim 

691 command line with every token quoted. Every switch is load-bearing: `/s` 

692 makes cmd strip only the outer pair, leaving each token quoted and its 

693 metacharacters inert, `/e:on` keeps the command extensions that the percent 

694 guard is built out of, `/v:off` keeps `!` from expanding, and `/d` keeps a 

695 machine's AutoRun commands out of the launch. argv[0] carries the 

696 caller-facing name on POSIX; Windows needs the resolved path there. 

697 

698 Raises AgentRunError for an argument holding a line break, which cmd would 

699 read as the end of the command line and silently drop the rest of. 

700 """ 

701 rest: Final = tuple(args[1:]) 

702 if os.path.splitext(path)[1].lower() not in _WINDOWS_SHIM_SUFFIXES: 

703 return (path, *rest) 

704 if any(brk in token for token in rest for brk in _CMD_LINE_BREAKS): 

705 raise AgentRunError( 

706 f"Cannot pass an argument containing a line break to `{os.path.basename(path)}` on " 

707 "Windows: cmd.exe ends the command line there, so the agent would silently lose it." 

708 ) 

709 inner: Final = " ".join(quote_for_cmd(token) for token in (path, *rest)) 

710 return f'cmd.exe /d /e:on /v:off /s /c "{inner}"' 

711 

712 

713def _spawn_and_wait(command: str | Sequence[str], env: Mapping[str, str]) -> int: 

714 return subprocess.run(command, env=dict(env), check=False).returncode 

715 

716 

717def _replace_process( 

718 path: str, 

719 args: Sequence[str], 

720 env: Mapping[str, str], 

721 *, 

722 execvpe: Callable[..., None] = os.execvpe, 

723) -> None: 

724 execvpe(path, list(args), dict(env)) 

725 

726 

727def _hand_off( 

728 path: str, 

729 args: Sequence[str], 

730 env: Mapping[str, str], 

731 *, 

732 platform: str = sys.platform, 

733 replace: Callable[[str, Sequence[str], Mapping[str, str]], None] = _replace_process, 

734 spawn: Callable[[str | Sequence[str], Mapping[str, str]], int] = _spawn_and_wait, 

735) -> None: 

736 """Replace this process with the agent; on Windows, run it as a child instead. 

737 

738 os.exec* has no process-replacement semantics on Windows: the C runtime 

739 spawns a detached child and terminates the parent, so the shell reclaims the 

740 console and the agent's TUI never gets one. Windows therefore waits on the 

741 child and exits with its status. 

742 """ 

743 if platform.startswith("win"): 

744 raise SystemExit(spawn(_windows_command(path, args), env)) 

745 replace(path, list(args), dict(env)) 

746 

747 

748def _restore_controlling_terminal() -> None: 

749 """Reattach the controlling terminal to stdin before handing off to the agent. 

750 

751 Completing the browser SSO login can leave stdin detached from the terminal, 

752 which makes a TUI agent like Claude Code start in non-interactive mode and 

753 exit immediately. Reopening /dev/tty onto fd 0 gives the agent a live 

754 terminal; when stdin is still a tty (no login happened) this is a no-op. 

755 """ 

756 if sys.stdin.isatty(): 

757 return 

758 try: 

759 fd: Final = os.open("/dev/tty", os.O_RDONLY) 

760 except OSError: 

761 return 

762 try: 

763 os.dup2(fd, 0) 

764 finally: 

765 os.close(fd) 

766 

767 

768def run_agent( 

769 base_url: str, 

770 api_key: str, 

771 command: Sequence[str], 

772 *, 

773 skip_verify: bool = False, 

774 base_env: Mapping[str, str] | None = None, 

775 which: Callable[[str], str | None] = shutil.which, 

776 verify: Callable[[str, str], None] = verify_proxy_key, 

777 sync_models: Callable[[str, Mapping[str, str], str, str, bool], ModelSyncResult] = agent_model_sync_env, 

778 warn: Callable[[str], None] = _warn, 

779 launcher: Callable[[str, Sequence[str], Mapping[str, str]], None] = _hand_off, 

780 reattach_terminal: Callable[[], None] | None = None, 

781 preparers: Mapping[str, _Preparer] = MappingProxyType(_PREPARERS), 

782) -> None: 

783 """Validate, wire the environment, and hand off to the agent. 

784 

785 On success this replaces the current process and never returns. Raises 

786 AgentRunError for missing binaries, an unreachable proxy, a rejected key, or 

787 a failed pre-launch config sync (pi). reattach_terminal, when given, runs 

788 just before handoff to restore stdin. 

789 """ 

790 if not command: 

791 raise AgentRunError("Nothing to run.") 

792 

793 display_name, profiles = agent_profile(command[0]) 

794 binary: Final = which(command[0]) 

795 if binary is None: 

796 docs: Final = _INSTALL_DOCS.get(os.path.basename(command[0])) 

797 hint: Final = f" Install it first: {docs}" if docs else "" 

798 raise AgentRunError(f"Could not find `{command[0]}` on your PATH.{hint}") 

799 

800 if not skip_verify: 

801 verify(base_url, api_key) 

802 

803 env_before_sync: Final = base_env if base_env is not None else os.environ 

804 synced: Final = sync_models(binary, env_before_sync, base_url, api_key, skip_verify) 

805 if isinstance(synced, ModelSyncSkipped): 

806 warn(f"litellm: not syncing {display_name} models from the proxy: {synced.reason}") 

807 

808 prepare: Final = preparers.get(os.path.basename(command[0])) 

809 prepared_args: Final = tuple(prepare(base_url, api_key, env_before_sync)) if prepare is not None else () 

810 

811 env: Final = MappingProxyType( 

812 { 

813 **build_agent_env(env_before_sync, base_url, api_key, profiles), 

814 **(synced if isinstance(synced, Mapping) else _NO_EXTRA_ENV), 

815 } 

816 ) 

817 synced_args: Final = synced.args if isinstance(synced, ModelSyncArgs) else () 

818 extra_args: Final = (*agent_launch_args(command[0], base_url), *synced_args, *prepared_args) 

819 if reattach_terminal is not None: 

820 reattach_terminal() 

821 launcher(binary, [command[0], *extra_args, *command[1:]], env) 

822 

823 

824def _is_interactive() -> bool: 

825 return sys.stdin.isatty() 

826 

827 

828def resolve_api_key(ctx: click.Context) -> str: 

829 ctx_obj: Final[CliContextObj] = ctx.obj 

830 base_url: Final = ctx_obj["base_url"] 

831 api_key = ctx_obj.get("api_key") 

832 if api_key: 

833 return api_key 

834 

835 if not _is_interactive(): 

836 raise click.ClickException( 

837 "No LiteLLM key found. Set LITELLM_PROXY_API_KEY (or pass --api-key) for " 

838 "non-interactive use, or run `lite login` from a terminal." 

839 ) 

840 

841 click.echo("No LiteLLM credentials found; starting login...") 

842 ctx.invoke(login) 

843 api_key = get_stored_api_key(expected_base_url=base_url, vault=context_secret_vault(ctx)) 

844 if not api_key: 

845 raise click.ClickException("Login did not produce an API key; cannot start the agent.") 

846 return api_key 

847 

848 

849_SKIP_VERIFY_HELP: Final = "Skip the pre-launch key check against the proxy." 

850 

851 

852def _launch(ctx: click.Context, binary: str, args: Sequence[str], *, skip_verify: bool) -> None: 

853 ctx_obj: Final[CliContextObj] = ctx.obj 

854 base_url: Final = ctx_obj["base_url"] 

855 started_interactive: Final = _is_interactive() 

856 api_key: Final = resolve_api_key(ctx) 

857 

858 display_name, _profiles = agent_profile(binary) 

859 click.echo(f"litellm: routing {display_name} through proxy at {base_url.rstrip('/')}") 

860 

861 try: 

862 run_agent( 

863 base_url, 

864 api_key, 

865 [binary, *args], 

866 skip_verify=skip_verify, 

867 reattach_terminal=(_restore_controlling_terminal if started_interactive else None), 

868 ) 

869 except AgentRunError as e: 

870 raise click.ClickException(str(e)) 

871 

872 

873def _make_agent_command(binary: str, display_name: str) -> click.Command: 

874 @click.command( 

875 name=binary, 

876 context_settings={"ignore_unknown_options": True}, 

877 short_help=f"Run {display_name} through your LiteLLM proxy", 

878 hidden=binary in _HIDDEN_AGENTS, 

879 ) 

880 @click.option("--skip-verify", is_flag=True, default=False, help=_SKIP_VERIFY_HELP) 

881 @click.argument("args", nargs=-1, type=click.UNPROCESSED) 

882 @click.pass_context 

883 def _command(ctx: click.Context, skip_verify: bool, args: Sequence[str]) -> None: 

884 _launch(ctx, binary, list(args), skip_verify=skip_verify) 

885 

886 _command.help = ( 

887 f"Run {display_name} routed through your LiteLLM proxy.\n\n" 

888 f"Logs in with LiteLLM if needed, verifies your key against the proxy, " 

889 f"exports the env vars {binary} reads, then hands off. Any arguments are " 

890 f"forwarded to `{binary}`." 

891 ) 

892 return _command 

893 

894 

895def agent_commands() -> tuple[click.Command, ...]: 

896 """Build one top-level command per known agent, e.g. `lite claude`.""" 

897 return tuple(_make_agent_command(binary, name) for binary, (name, _profiles) in _KNOWN_AGENTS.items()) 

898 

899 

900__all__ = [ 

901 "AgentRunError", 

902 "ListedModel", 

903 "ModelSyncSkipped", 

904 "agent_commands", 

905 "agent_launch_args", 

906 "agent_model_sync_env", 

907 "agent_profile", 

908 "build_agent_env", 

909 "opencode_model_sync_env", 

910 "opencode_provider_config", 

911 "prepare_pi", 

912 "resolve_api_key", 

913 "run_agent", 

914 "verify_proxy_key", 

915]