Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/client/cli/commands/agents.py: 0%
391 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1import json
2import os
3import re
4import shutil
5import subprocess
6import sys
7import tempfile
8from collections.abc import Callable, Mapping, Sequence
9from dataclasses import dataclass
10from pathlib import Path
11from types import MappingProxyType
12from typing import Final, Literal, TypeAlias
14import click
15import requests
16from pydantic import BaseModel, ConfigDict, TypeAdapter, ValidationError
18from .auth import CliContextObj, context_secret_vault, get_stored_api_key, login
19from .claude_settings import ClaudeSettingsError, install_statusline_script
20from .cmd_quoting import quote_for_cmd
21from .pi import (
22 LITELLM_PROXY_API_KEY_ENV,
23 PI_PROVIDER_NAME,
24 ListingFailure,
25 PiSyncError,
26 fetch_model_ids,
27 fetch_model_limits,
28 models_json_path,
29 sync_models_json,
30)
32ANTHROPIC_BASE_URL_ENV: Final = "ANTHROPIC_BASE_URL"
33ANTHROPIC_AUTH_TOKEN_ENV: Final = "ANTHROPIC_AUTH_TOKEN"
34ANTHROPIC_API_KEY_ENV: Final = "ANTHROPIC_API_KEY"
35ENABLE_TOOL_SEARCH_ENV: Final = "ENABLE_TOOL_SEARCH"
36ENABLE_TOOL_SEARCH_VALUE: Final = "true"
37ENABLE_GATEWAY_MODEL_DISCOVERY_ENV: Final = "CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY"
38ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE: Final = "1"
39OPENAI_BASE_URL_ENV: Final = "OPENAI_BASE_URL"
40OPENAI_API_KEY_ENV: Final = "OPENAI_API_KEY"
41OPENCODE_CONFIG_CONTENT_ENV: Final = "OPENCODE_CONFIG_CONTENT"
42OPENCODE_PROVIDER_ID: Final = "litellm"
43OPENCODE_PROVIDER_NAME: Final = "LiteLLM"
44OPENCODE_PROVIDER_NPM: Final = "@ai-sdk/openai-compatible"
46_SKIP_VERIFY_FLAG: Final = "--skip-verify"
48PROFILE_ANTHROPIC: Final = "anthropic"
49PROFILE_OPENAI: Final = "openai"
50PROFILE_LITELLM: Final = "litellm"
52_KNOWN_AGENTS: Final[dict[str, tuple[str, frozenset[str]]]] = {
53 "claude": ("Claude Code", frozenset({PROFILE_ANTHROPIC})),
54 "codex": ("Codex", frozenset({PROFILE_OPENAI})),
55 "opencode": ("OpenCode", frozenset({PROFILE_OPENAI})),
56 "pi": ("pi", frozenset({PROFILE_LITELLM})),
57}
59_INSTALL_DOCS: Final[dict[str, str]] = {
60 "claude": "https://docs.claude.com/en/docs/claude-code/setup",
61 "codex": "https://developers.openai.com/codex/cli",
62 "opencode": "https://opencode.ai/docs",
63 "pi": "https://pi.dev",
64}
66_HIDDEN_AGENTS: Final = frozenset({"pi"})
68CODEX_PROXY_PROVIDER: Final = "litellm"
69CODEX_HOME_ENV: Final = "CODEX_HOME"
70CODEX_MODEL_CATALOG_FILENAME: Final = "litellm-models.json"
71_CODEX_BASE_INSTRUCTIONS_PATH: Final = Path(__file__).with_name("codex_base_instructions.md")
72_CODEX_PREFLIGHT_TIMEOUT_SECONDS: Final = 10.0
75class AgentRunError(Exception):
76 """Raised for any user-actionable failure while preparing to run an agent."""
79def agent_profile(command: str) -> tuple[str, frozenset[str]]:
80 """Return the (display name, env profiles) for a wrapped command.
82 Known agents map to the API family they speak. Anything else gets both
83 families so it works regardless of which env vars the tool reads.
84 """
85 base: Final = os.path.basename(command)
86 if base in _KNOWN_AGENTS:
87 return _KNOWN_AGENTS[base]
88 return base, frozenset({PROFILE_ANTHROPIC, PROFILE_OPENAI})
91def build_agent_env(
92 base_env: Mapping[str, str],
93 base_url: str,
94 api_key: str,
95 profiles: frozenset[str],
96) -> dict[str, str]:
97 """Return a copy of base_env wired to route the agent through the proxy.
99 Anthropic clients (Claude Code) append /v1/messages to ANTHROPIC_BASE_URL,
100 so it stays the bare proxy root; OpenAI clients (Codex, OpenCode) expect the
101 /v1 suffix on OPENAI_BASE_URL. ANTHROPIC_API_KEY is dropped so a stray
102 Anthropic key cannot win over the bearer token we set. ENABLE_TOOL_SEARCH
103 defaults to true because Claude Code turns tool search off when
104 ANTHROPIC_BASE_URL is not a first-party Anthropic host; a value already in
105 the environment is left alone. CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY
106 defaults to 1 so Claude Code (v2.1.129+) fills its /model picker from the
107 proxy's /v1/models; likewise left alone when already set.
108 pi ignores both base URL variables and instead resolves $LITELLM_PROXY_API_KEY
109 from its synced models.json provider entry.
110 """
111 env: Final = dict(base_env)
112 root: Final = base_url.rstrip("/")
113 if PROFILE_ANTHROPIC in profiles:
114 env[ANTHROPIC_BASE_URL_ENV] = root
115 env[ANTHROPIC_AUTH_TOKEN_ENV] = api_key
116 env.pop(ANTHROPIC_API_KEY_ENV, None)
117 if ENABLE_TOOL_SEARCH_ENV not in env:
118 env[ENABLE_TOOL_SEARCH_ENV] = ENABLE_TOOL_SEARCH_VALUE
119 if ENABLE_GATEWAY_MODEL_DISCOVERY_ENV not in env:
120 env[ENABLE_GATEWAY_MODEL_DISCOVERY_ENV] = ENABLE_GATEWAY_MODEL_DISCOVERY_VALUE
121 if PROFILE_OPENAI in profiles:
122 env[OPENAI_BASE_URL_ENV] = root + "/v1"
123 env[OPENAI_API_KEY_ENV] = api_key
124 if PROFILE_LITELLM in profiles:
125 env[LITELLM_PROXY_API_KEY_ENV] = api_key
126 return env
129def codex_proxy_provider(base_url: str) -> Mapping[str, str | bool]:
130 return MappingProxyType(
131 {
132 "name": "LiteLLM proxy",
133 "base_url": base_url.rstrip("/") + "/v1",
134 "wire_api": "responses",
135 "supports_websockets": False,
136 "requires_openai_auth": False,
137 }
138 )
141def _codex_proxy_args(base_url: str) -> list[str]:
142 """Codex `-c` overrides that point it at the proxy.
144 Codex ignores OPENAI_BASE_URL (it always dials api.openai.com), so the env
145 profile alone cannot route it. It does honor a custom provider, so define one
146 inline; supports_websockets=false forces the HTTP/SSE Responses transport
147 because the proxy does not speak the Responses WebSocket protocol. The key is
148 read from OPENAI_API_KEY, which build_agent_env already exports.
149 """
150 provider: Final = f"model_providers.{CODEX_PROXY_PROVIDER}"
151 return [
152 "-c",
153 f'model_provider="{CODEX_PROXY_PROVIDER}"',
154 *(
155 argument
156 for key, value in codex_proxy_provider(base_url).items()
157 for argument in ("-c", f"{provider}.{key}={json.dumps(value)}")
158 ),
159 "-c",
160 f'{provider}.env_key="{OPENAI_API_KEY_ENV}"',
161 "-c",
162 f"{provider}.http_headers={{}}",
163 ]
166_PROXY_ARGS: Final[dict[str, Callable[[str], list[str]]]] = {
167 "codex": _codex_proxy_args,
168}
171def prepare_pi(
172 base_url: str,
173 api_key: str,
174 base_env: Mapping[str, str],
175 *,
176 get: Callable[..., requests.Response] = requests.get,
177) -> tuple[str, ...]:
178 """Sync the proxy's model list into pi's models.json before handoff.
180 pi has no base-URL env vars, so this file is the only way to point it at the
181 proxy. Only the litellm provider entry is touched; the synced entry references
182 the key as $LITELLM_PROXY_API_KEY, which build_agent_env exports. The returned
183 --model pin is needed because pi ignores a bare --provider when picking the
184 interactive startup model; a user-supplied --model comes later in argv and wins.
185 """
186 ids: Final = fetch_model_ids(base_url, api_key, get=get)
187 if isinstance(ids, PiSyncError):
188 raise AgentRunError(
189 f"{ids.message} pi would have nothing to run." if ids.kind is ListingFailure.EMPTY else ids.message
190 )
191 limits: Final = fetch_model_limits(base_url, api_key, get=get)
192 path: Final = models_json_path(base_env)
193 error: Final = sync_models_json(path, base_url, ids, limits)
194 if error is not None:
195 raise AgentRunError(error.message)
196 click.echo(f"litellm: synced {len(ids)} proxy models into {path}")
197 return ("--model", f"{PI_PROVIDER_NAME}/{ids[0]}")
200def _warn(message: str) -> None:
201 click.echo(message, err=True)
204_CODEX_STOP_HOOKS_DECLARED: Final = re.compile(
205 r"^\s*(\[\[\s*\"?hooks\"?\s*\.\s*\"?Stop\"?\s*\]\]|\"?hooks\"?(?:\s*\.\s*\"?Stop\"?)?\s*=|\[\s*\"?hooks\"?\s*\])",
206 re.MULTILINE,
207)
210def codex_config_path(base_env: Mapping[str, str]) -> Path:
211 return Path(base_env.get("CODEX_HOME") or Path.home() / ".codex") / "config.toml"
214def codex_declares_stop_hooks(config_path: Path) -> bool:
215 """A config that cannot be read or decoded declares nothing we can see; Codex reports its own
216 TOML failure at launch, so the pre-check must not be the thing that stops `lite codex`."""
217 try:
218 return _CODEX_STOP_HOOKS_DECLARED.search(config_path.read_text(encoding="utf-8")) is not None
219 except (OSError, UnicodeDecodeError):
220 return False
223def prepare_codex(
224 base_url: str,
225 api_key: str,
226 base_env: Mapping[str, str],
227 *,
228 install: Callable[[], str] = install_statusline_script,
229 warn: Callable[[str], None] = _warn,
230) -> tuple[str, ...]:
231 """A `-c hooks.Stop=` session flag replaces the user's whole Stop list, so their own hooks win over ours."""
232 if codex_declares_stop_hooks(codex_config_path(base_env)):
233 warn("litellm: your Codex config already declares hooks; not adding the routed-model Stop hook")
234 return ()
235 try:
236 command: Final = install()
237 except ClaudeSettingsError as e:
238 raise AgentRunError(str(e)) from e
239 return ("-c", f'hooks.Stop=[{{hooks=[{{type="command",command={json.dumps(command)}}}]}}]')
242_Preparer: TypeAlias = Callable[[str, str, Mapping[str, str]], Sequence[str]]
244_PREPARERS: Final[Mapping[str, _Preparer]] = MappingProxyType({"pi": prepare_pi, "codex": prepare_codex})
247def agent_launch_args(command: str, base_url: str) -> list[str]:
248 """Extra CLI args an agent needs to actually honor the proxy.
250 Claude Code and OpenCode respect the exported env vars, so they get nothing
251 here; Codex needs its provider pointed via config overrides.
252 """
253 builder: Final = _PROXY_ARGS.get(os.path.basename(command))
254 return builder(base_url) if builder else []
257class ListedModel(BaseModel):
258 """The fields of a /v1/models entry that an OpenCode or Codex model entry is built from."""
260 id: str
261 mode: str | None = None
262 max_input_tokens: int | None = None
263 max_output_tokens: int | None = None
266class _ModelListing(BaseModel):
267 data: tuple[ListedModel, ...]
270_MODEL_LISTING: Final = TypeAdapter(_ModelListing)
271_CHAT_MODES: Final[frozenset[str]] = frozenset({"chat", "responses"})
272_NO_EXTRA_ENV: Final[Mapping[str, str]] = MappingProxyType({})
275@dataclass(frozen=True, slots=True)
276class ModelSyncSkipped:
277 reason: str
280@dataclass(frozen=True, slots=True)
281class ModelSyncArgs:
282 """CLI args, placed before the user's own, that hand an agent the synced model list."""
284 args: tuple[str, ...]
287ModelSyncResult: TypeAlias = Mapping[str, str] | ModelSyncArgs | ModelSyncSkipped
290def _chat_models(models: Sequence[ListedModel]) -> tuple[ListedModel, ...]:
291 return tuple(m for m in models if m.mode is None or m.mode in _CHAT_MODES)
294def _fetch_model_listing(
295 base_url: str,
296 api_key: str,
297 *,
298 get: Callable[..., requests.Response],
299) -> tuple[ListedModel, ...] | ModelSyncSkipped:
300 url: Final = base_url.rstrip("/") + "/v1/models"
301 try:
302 resp: Final = get(url, headers=MappingProxyType({"Authorization": f"Bearer {api_key}"}), timeout=10)
303 except requests.RequestException as e:
304 return ModelSyncSkipped(f"could not reach {url}: {e}")
305 if resp.status_code != 200:
306 return ModelSyncSkipped(f"{url} returned HTTP {resp.status_code}")
307 try:
308 listing: Final = _MODEL_LISTING.validate_json(resp.content)
309 except ValidationError:
310 return ModelSyncSkipped(f"{url} returned an unexpected body")
311 return listing.data
314class _OpenCodeLimit(BaseModel):
315 context: int
316 output: int
319class _OpenCodeModel(BaseModel):
320 name: str
321 limit: _OpenCodeLimit | None = None
324class _OpenCodeProviderOptions(BaseModel):
325 baseURL: str
326 apiKey: str
329class _OpenCodeProvider(BaseModel):
330 npm: str
331 name: str
332 options: _OpenCodeProviderOptions
333 models: Mapping[str, _OpenCodeModel]
336class _OpenCodeConfig(BaseModel):
337 provider: Mapping[str, _OpenCodeProvider]
340def _opencode_model_entry(model: ListedModel) -> _OpenCodeModel:
341 if model.max_input_tokens is None or model.max_output_tokens is None:
342 return _OpenCodeModel(name=model.id)
343 return _OpenCodeModel(
344 name=model.id, limit=_OpenCodeLimit(context=model.max_input_tokens, output=model.max_output_tokens)
345 )
348def opencode_provider_config(base_url: str, models: Sequence[ListedModel]) -> str:
349 """OPENCODE_CONFIG_CONTENT declaring the proxy as OpenCode provider `litellm`.
351 One model entry per chat-capable /v1/models row (mode chat, responses, or
352 unknown), so OpenCode's model picker mirrors what the key can call. The key
353 is read back through {env:OPENAI_API_KEY}, which build_agent_env exports, so
354 it never lands in the config text. OpenCode merges this inline config over
355 the user's own files, leaving unrelated keys and providers untouched.
356 """
357 chat_models: Final = _chat_models(models)
358 provider: Final = _OpenCodeProvider(
359 npm=OPENCODE_PROVIDER_NPM,
360 name=OPENCODE_PROVIDER_NAME,
361 options=_OpenCodeProviderOptions(
362 baseURL=base_url.rstrip("/") + "/v1",
363 apiKey=f"{{env:{OPENAI_API_KEY_ENV}}}",
364 ),
365 models=MappingProxyType({m.id: _opencode_model_entry(m) for m in chat_models}),
366 )
367 config: Final = _OpenCodeConfig(provider=MappingProxyType({OPENCODE_PROVIDER_ID: provider}))
368 return config.model_dump_json(exclude_none=True)
371def opencode_model_sync_env(
372 base_env: Mapping[str, str],
373 base_url: str,
374 api_key: str,
375 *,
376 get: Callable[..., requests.Response] = requests.get,
377) -> Mapping[str, str] | ModelSyncSkipped:
378 """Env addition that hands OpenCode the proxy's model list, or why it was skipped.
380 Fetches /v1/models with the key and packs it into OPENCODE_CONFIG_CONTENT.
381 An OPENCODE_CONFIG_CONTENT already in the environment is left alone, and a
382 failed fetch is reported rather than raised: OpenCode still launches on the
383 plain OPENAI_* env, just without a synced model list.
384 """
385 if OPENCODE_CONFIG_CONTENT_ENV in base_env:
386 return ModelSyncSkipped(f"{OPENCODE_CONFIG_CONTENT_ENV} is already set")
387 listing: Final = _fetch_model_listing(base_url, api_key, get=get)
388 if isinstance(listing, ModelSyncSkipped):
389 return listing
390 return MappingProxyType({OPENCODE_CONFIG_CONTENT_ENV: opencode_provider_config(base_url, listing)})
393class _CodexTruncationPolicy(BaseModel):
394 mode: Literal["bytes"] = "bytes"
395 limit: int = 10_000
398class _CodexModel(BaseModel):
399 """One `ModelInfo` entry of a Codex model catalog for a model the installed Codex does not know.
401 Every field that some Codex release since `model_catalog_json` appeared
402 (0.105.0) deserializes without a default is spelled out here, so one catalog
403 parses on all of them; the values match the fallback metadata Codex uses for
404 a model slug it does not know, so picking such a proxy model behaves the
405 same as `codex -m` did.
406 """
408 slug: str
409 display_name: str
410 description: None = None
411 supported_reasoning_levels: tuple[()] = ()
412 shell_type: Literal["unified_exec"] = "unified_exec"
413 visibility: Literal["list"] = "list"
414 supported_in_api: Literal[True] = True
415 priority: int
416 availability_nux: None = None
417 upgrade: None = None
418 support_verbosity: Literal[False] = False
419 supports_reasoning_summaries: Literal[False] = False
420 supports_parallel_tool_calls: Literal[False] = False
421 default_verbosity: None = None
422 apply_patch_tool_type: None = None
423 truncation_policy: _CodexTruncationPolicy = _CodexTruncationPolicy()
424 experimental_supported_tools: tuple[()] = ()
425 context_window: int | None
426 base_instructions: str
429class _StockCodexUpgrade(BaseModel):
430 model_config = ConfigDict(extra="allow")
432 model: str
435class _StockCodexModel(BaseModel):
436 """One `ModelInfo` entry as the installed Codex prints it from `codex debug models`.
438 Only the fields the sync rewrites are named; everything else that release
439 knows about the model (its reasoning levels, prompt, tool support) rides
440 along untouched, whatever the release's schema.
441 """
443 model_config = ConfigDict(extra="allow")
445 slug: str
446 priority: int
447 visibility: str
448 supported_in_api: bool = True
449 upgrade: _StockCodexUpgrade | None = None
452class _StockCodexCatalog(BaseModel):
453 models: tuple[_StockCodexModel, ...]
456class _CodexCatalog(BaseModel):
457 models: tuple[_CodexModel | _StockCodexModel, ...]
460def _codex_catalog_entry(
461 priority: int,
462 listed: ListedModel,
463 stock: _StockCodexModel | None,
464 served: frozenset[str],
465 instructions: str,
466) -> _CodexModel | _StockCodexModel:
467 if stock is None:
468 return _CodexModel(
469 slug=listed.id,
470 display_name=listed.id,
471 priority=priority,
472 context_window=listed.max_input_tokens,
473 base_instructions=instructions,
474 )
475 upgrade: Final = stock.upgrade if stock.upgrade is not None and stock.upgrade.model in served else None
476 return stock.model_copy(
477 update={"priority": priority, "visibility": "list", "supported_in_api": True, "upgrade": upgrade}
478 )
481def codex_model_catalog(
482 models: Sequence[ListedModel], stock: Sequence[_StockCodexModel], instructions: str
483) -> str | None:
484 """The `model_catalog_json` body listing the proxy's chat models, or None if there are none.
486 Codex refuses an empty catalog, hence None instead of `{"models": []}`.
487 Passing a catalog replaces Codex's built-in one, so a proxy model the
488 installed Codex knows keeps that Codex's own entry and the proxy only
489 decides its place in the picker: the listing orders it, lists it even when
490 Codex hides it or keeps it off the API, and keeps Codex's upgrade nudge only
491 when the model it points at is served too. A model Codex does not know gets the fallback
492 entry, with the same base instructions Codex itself uses so the agent never
493 runs without a system prompt.
494 """
495 chat_models: Final = _chat_models(models)
496 if not chat_models:
497 return None
498 served: Final = frozenset(m.id for m in chat_models)
499 known: Final = MappingProxyType({m.slug: m for m in stock})
500 catalog: Final = _CodexCatalog(
501 models=tuple(
502 _codex_catalog_entry(index, m, known.get(m.id), served, instructions) for index, m in enumerate(chat_models)
503 )
504 )
505 return catalog.model_dump_json()
508def codex_model_catalog_path(env: Mapping[str, str], *, home: Callable[[], Path] = Path.home) -> Path:
509 override: Final = env.get(CODEX_HOME_ENV)
510 root: Final = Path(override) if override else home() / ".codex"
511 return root / CODEX_MODEL_CATALOG_FILENAME
514def _replace_file(path: Path, text: str) -> None:
515 with tempfile.NamedTemporaryFile("w", encoding="utf-8", dir=path.parent, delete=False) as tmp:
516 _ = tmp.write(text)
517 try:
518 os.replace(tmp.name, path)
519 except OSError:
520 Path(tmp.name).unlink(missing_ok=True)
521 raise
524def _codex_debug_models(
525 binary: str,
526 args: Sequence[str],
527 env: Mapping[str, str],
528 *,
529 run: Callable[..., subprocess.CompletedProcess[str]],
530) -> str | ModelSyncSkipped:
531 """What `codex debug models` prints with `args` in front, or why the installed Codex could not run it.
533 The command prints the catalog Codex would launch with, without touching
534 the network, so it lists the installed Codex's own models and parses a
535 catalog override the way a launch does. Releases before 0.130.0 have no
536 such command and are reported the same way. A batch shim goes through
537 cmd.exe exactly as the launch will.
538 """
539 name: Final = os.path.basename(binary)
540 command: Final = _windows_command(binary, (binary, *args, "debug", "models"))
541 try:
542 completed: Final = run(
543 command,
544 env=dict(env),
545 stdin=subprocess.DEVNULL,
546 capture_output=True,
547 encoding="utf-8",
548 timeout=_CODEX_PREFLIGHT_TIMEOUT_SECONDS,
549 )
550 except (OSError, subprocess.TimeoutExpired) as e:
551 return ModelSyncSkipped(f"`{name} debug models` failed: {e}")
552 if completed.returncode == 0:
553 return completed.stdout
554 lines: Final = completed.stderr.strip().splitlines()
555 detail: Final = lines[0] if lines else "no output"
556 return ModelSyncSkipped(f"`{name} debug models` exited {completed.returncode}: {detail}")
559def _stock_codex_models(
560 binary: str, env: Mapping[str, str], *, run: Callable[..., subprocess.CompletedProcess[str]]
561) -> tuple[_StockCodexModel, ...] | ModelSyncSkipped:
562 printed: Final = _codex_debug_models(binary, (), env, run=run)
563 if isinstance(printed, ModelSyncSkipped):
564 return printed
565 try:
566 return _StockCodexCatalog.model_validate_json(printed).models
567 except ValidationError as e:
568 name: Final = os.path.basename(binary)
569 return ModelSyncSkipped(f"`{name} debug models` printed no model catalog: {e.errors()[0]['msg']}")
572def codex_model_sync_args(
573 base_env: Mapping[str, str],
574 base_url: str,
575 api_key: str,
576 *,
577 binary: str = "codex",
578 get: Callable[..., requests.Response] = requests.get,
579 run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run,
580 home: Callable[[], Path] = Path.home,
581 instructions_path: Path = _CODEX_BASE_INSTRUCTIONS_PATH,
582) -> ModelSyncArgs | ModelSyncSkipped:
583 """`-c model_catalog_json=...` pointing Codex at the proxy's model list, or why it was skipped.
585 Codex has no env or inline equivalent of OPENCODE_CONFIG_CONTENT: the catalog
586 must be a file, so it is written under $CODEX_HOME (default ~/.codex) and
587 atomically replaced on every launch. The Codex at `binary` first lists its
588 own models, so the ones the proxy serves keep that Codex's entries, and then
589 reads the file back once before it is handed over. The key never lands in
590 the file. A failed fetch, read, listing, write or read-back is reported
591 rather than raised: Codex still launches with its built-in catalog and takes
592 a proxy model by name via -m, and a rejected file stays on disk to be looked
593 at.
594 """
595 listing: Final = _fetch_model_listing(base_url, api_key, get=get)
596 if isinstance(listing, ModelSyncSkipped):
597 return listing
598 try:
599 instructions: Final = instructions_path.read_text(encoding="utf-8")
600 except OSError as e:
601 return ModelSyncSkipped(f"could not read {instructions_path}: {e}")
602 path: Final = codex_model_catalog_path(base_env, home=home)
603 try:
604 path.parent.mkdir(parents=True, exist_ok=True)
605 except OSError as e:
606 return ModelSyncSkipped(f"could not write {path}: {e}")
607 stock: Final = _stock_codex_models(binary, base_env, run=run)
608 if isinstance(stock, ModelSyncSkipped):
609 return stock
610 catalog: Final = codex_model_catalog(listing, stock, instructions)
611 if catalog is None:
612 return ModelSyncSkipped(f"{base_url.rstrip('/')}/v1/models lists no chat models")
613 try:
614 _replace_file(path, catalog)
615 except OSError as e:
616 return ModelSyncSkipped(f"could not write {path}: {e}")
617 override: Final = f"model_catalog_json={json.dumps(str(path))}"
618 read_back: Final = _codex_debug_models(binary, ("-c", override), base_env, run=run)
619 if isinstance(read_back, ModelSyncSkipped):
620 return read_back
621 return ModelSyncArgs(("-c", override))
624def agent_model_sync_env(
625 binary: str,
626 base_env: Mapping[str, str],
627 base_url: str,
628 api_key: str,
629 skip_verify: bool,
630 *,
631 get: Callable[..., requests.Response] = requests.get,
632 run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run,
633) -> ModelSyncResult:
634 """Extra env or args an agent needs to see the proxy's model list.
636 binary is the resolved path the launch will run (`codex.cmd` on a Windows
637 npm install). OpenCode takes the list as env, Codex as a `-c` override that
638 binary has read back first; Claude Code discovers models itself through
639 CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY. skip_verify means the caller
640 wants no pre-launch proxy call at all, so the listing is skipped too rather
641 than hanging on an offline proxy.
642 """
643 agent: Final = os.path.splitext(os.path.basename(binary))[0]
644 if agent not in ("opencode", "codex"):
645 return _NO_EXTRA_ENV
646 if skip_verify:
647 return ModelSyncSkipped(f"{_SKIP_VERIFY_FLAG} was passed")
648 if agent == "codex":
649 return codex_model_sync_args(base_env, base_url, api_key, binary=binary, get=get, run=run)
650 return opencode_model_sync_env(base_env, base_url, api_key, get=get)
653def verify_proxy_key(
654 base_url: str,
655 api_key: str,
656 *,
657 get: Callable[..., requests.Response] = requests.get,
658) -> None:
659 """Probe the proxy with the key so bad creds fail here, not inside the agent.
661 Raises AgentRunError when the proxy is unreachable or rejects the key. Other
662 non-2xx responses are tolerated; the agent's own call is the real test.
663 """
664 url: Final = base_url.rstrip("/") + "/v1/models"
665 try:
666 resp: Final = get(url, headers={"Authorization": f"Bearer {api_key}"}, timeout=10)
667 except requests.RequestException as e:
668 raise AgentRunError(
669 f"Could not reach the LiteLLM proxy at {base_url.rstrip('/')}: {e}. "
670 "Is it running, and is --base-url (or LITELLM_PROXY_URL) correct?"
671 )
672 if resp.status_code in (401, 403):
673 raise AgentRunError(
674 f"LiteLLM rejected your key (HTTP {resp.status_code}). "
675 "Run `lite login` to refresh it, or pass a valid --api-key."
676 )
679_WINDOWS_SHIM_SUFFIXES: Final[frozenset[str]] = frozenset({".cmd", ".bat"})
680_CMD_LINE_BREAKS: Final = ("\r", "\n")
683def _windows_command(path: str, args: Sequence[str]) -> str | tuple[str, ...]:
684 """Build what CreateProcess runs, routing batch shims through cmd.exe.
686 npm installs Claude Code as `claude.cmd`, which PATHEXT lets shutil.which
687 resolve but CreateProcess refuses to run (WinError 193), so a shim has to go
688 through the command processor. cmd.exe does not follow the C runtime quoting
689 that subprocess would apply to an argument list, and it would split on `&` or
690 `|` in a forwarded argument, so the shim case is emitted as one verbatim
691 command line with every token quoted. Every switch is load-bearing: `/s`
692 makes cmd strip only the outer pair, leaving each token quoted and its
693 metacharacters inert, `/e:on` keeps the command extensions that the percent
694 guard is built out of, `/v:off` keeps `!` from expanding, and `/d` keeps a
695 machine's AutoRun commands out of the launch. argv[0] carries the
696 caller-facing name on POSIX; Windows needs the resolved path there.
698 Raises AgentRunError for an argument holding a line break, which cmd would
699 read as the end of the command line and silently drop the rest of.
700 """
701 rest: Final = tuple(args[1:])
702 if os.path.splitext(path)[1].lower() not in _WINDOWS_SHIM_SUFFIXES:
703 return (path, *rest)
704 if any(brk in token for token in rest for brk in _CMD_LINE_BREAKS):
705 raise AgentRunError(
706 f"Cannot pass an argument containing a line break to `{os.path.basename(path)}` on "
707 "Windows: cmd.exe ends the command line there, so the agent would silently lose it."
708 )
709 inner: Final = " ".join(quote_for_cmd(token) for token in (path, *rest))
710 return f'cmd.exe /d /e:on /v:off /s /c "{inner}"'
713def _spawn_and_wait(command: str | Sequence[str], env: Mapping[str, str]) -> int:
714 return subprocess.run(command, env=dict(env), check=False).returncode
717def _replace_process(
718 path: str,
719 args: Sequence[str],
720 env: Mapping[str, str],
721 *,
722 execvpe: Callable[..., None] = os.execvpe,
723) -> None:
724 execvpe(path, list(args), dict(env))
727def _hand_off(
728 path: str,
729 args: Sequence[str],
730 env: Mapping[str, str],
731 *,
732 platform: str = sys.platform,
733 replace: Callable[[str, Sequence[str], Mapping[str, str]], None] = _replace_process,
734 spawn: Callable[[str | Sequence[str], Mapping[str, str]], int] = _spawn_and_wait,
735) -> None:
736 """Replace this process with the agent; on Windows, run it as a child instead.
738 os.exec* has no process-replacement semantics on Windows: the C runtime
739 spawns a detached child and terminates the parent, so the shell reclaims the
740 console and the agent's TUI never gets one. Windows therefore waits on the
741 child and exits with its status.
742 """
743 if platform.startswith("win"):
744 raise SystemExit(spawn(_windows_command(path, args), env))
745 replace(path, list(args), dict(env))
748def _restore_controlling_terminal() -> None:
749 """Reattach the controlling terminal to stdin before handing off to the agent.
751 Completing the browser SSO login can leave stdin detached from the terminal,
752 which makes a TUI agent like Claude Code start in non-interactive mode and
753 exit immediately. Reopening /dev/tty onto fd 0 gives the agent a live
754 terminal; when stdin is still a tty (no login happened) this is a no-op.
755 """
756 if sys.stdin.isatty():
757 return
758 try:
759 fd: Final = os.open("/dev/tty", os.O_RDONLY)
760 except OSError:
761 return
762 try:
763 os.dup2(fd, 0)
764 finally:
765 os.close(fd)
768def run_agent(
769 base_url: str,
770 api_key: str,
771 command: Sequence[str],
772 *,
773 skip_verify: bool = False,
774 base_env: Mapping[str, str] | None = None,
775 which: Callable[[str], str | None] = shutil.which,
776 verify: Callable[[str, str], None] = verify_proxy_key,
777 sync_models: Callable[[str, Mapping[str, str], str, str, bool], ModelSyncResult] = agent_model_sync_env,
778 warn: Callable[[str], None] = _warn,
779 launcher: Callable[[str, Sequence[str], Mapping[str, str]], None] = _hand_off,
780 reattach_terminal: Callable[[], None] | None = None,
781 preparers: Mapping[str, _Preparer] = MappingProxyType(_PREPARERS),
782) -> None:
783 """Validate, wire the environment, and hand off to the agent.
785 On success this replaces the current process and never returns. Raises
786 AgentRunError for missing binaries, an unreachable proxy, a rejected key, or
787 a failed pre-launch config sync (pi). reattach_terminal, when given, runs
788 just before handoff to restore stdin.
789 """
790 if not command:
791 raise AgentRunError("Nothing to run.")
793 display_name, profiles = agent_profile(command[0])
794 binary: Final = which(command[0])
795 if binary is None:
796 docs: Final = _INSTALL_DOCS.get(os.path.basename(command[0]))
797 hint: Final = f" Install it first: {docs}" if docs else ""
798 raise AgentRunError(f"Could not find `{command[0]}` on your PATH.{hint}")
800 if not skip_verify:
801 verify(base_url, api_key)
803 env_before_sync: Final = base_env if base_env is not None else os.environ
804 synced: Final = sync_models(binary, env_before_sync, base_url, api_key, skip_verify)
805 if isinstance(synced, ModelSyncSkipped):
806 warn(f"litellm: not syncing {display_name} models from the proxy: {synced.reason}")
808 prepare: Final = preparers.get(os.path.basename(command[0]))
809 prepared_args: Final = tuple(prepare(base_url, api_key, env_before_sync)) if prepare is not None else ()
811 env: Final = MappingProxyType(
812 {
813 **build_agent_env(env_before_sync, base_url, api_key, profiles),
814 **(synced if isinstance(synced, Mapping) else _NO_EXTRA_ENV),
815 }
816 )
817 synced_args: Final = synced.args if isinstance(synced, ModelSyncArgs) else ()
818 extra_args: Final = (*agent_launch_args(command[0], base_url), *synced_args, *prepared_args)
819 if reattach_terminal is not None:
820 reattach_terminal()
821 launcher(binary, [command[0], *extra_args, *command[1:]], env)
824def _is_interactive() -> bool:
825 return sys.stdin.isatty()
828def resolve_api_key(ctx: click.Context) -> str:
829 ctx_obj: Final[CliContextObj] = ctx.obj
830 base_url: Final = ctx_obj["base_url"]
831 api_key = ctx_obj.get("api_key")
832 if api_key:
833 return api_key
835 if not _is_interactive():
836 raise click.ClickException(
837 "No LiteLLM key found. Set LITELLM_PROXY_API_KEY (or pass --api-key) for "
838 "non-interactive use, or run `lite login` from a terminal."
839 )
841 click.echo("No LiteLLM credentials found; starting login...")
842 ctx.invoke(login)
843 api_key = get_stored_api_key(expected_base_url=base_url, vault=context_secret_vault(ctx))
844 if not api_key:
845 raise click.ClickException("Login did not produce an API key; cannot start the agent.")
846 return api_key
849_SKIP_VERIFY_HELP: Final = "Skip the pre-launch key check against the proxy."
852def _launch(ctx: click.Context, binary: str, args: Sequence[str], *, skip_verify: bool) -> None:
853 ctx_obj: Final[CliContextObj] = ctx.obj
854 base_url: Final = ctx_obj["base_url"]
855 started_interactive: Final = _is_interactive()
856 api_key: Final = resolve_api_key(ctx)
858 display_name, _profiles = agent_profile(binary)
859 click.echo(f"litellm: routing {display_name} through proxy at {base_url.rstrip('/')}")
861 try:
862 run_agent(
863 base_url,
864 api_key,
865 [binary, *args],
866 skip_verify=skip_verify,
867 reattach_terminal=(_restore_controlling_terminal if started_interactive else None),
868 )
869 except AgentRunError as e:
870 raise click.ClickException(str(e))
873def _make_agent_command(binary: str, display_name: str) -> click.Command:
874 @click.command(
875 name=binary,
876 context_settings={"ignore_unknown_options": True},
877 short_help=f"Run {display_name} through your LiteLLM proxy",
878 hidden=binary in _HIDDEN_AGENTS,
879 )
880 @click.option("--skip-verify", is_flag=True, default=False, help=_SKIP_VERIFY_HELP)
881 @click.argument("args", nargs=-1, type=click.UNPROCESSED)
882 @click.pass_context
883 def _command(ctx: click.Context, skip_verify: bool, args: Sequence[str]) -> None:
884 _launch(ctx, binary, list(args), skip_verify=skip_verify)
886 _command.help = (
887 f"Run {display_name} routed through your LiteLLM proxy.\n\n"
888 f"Logs in with LiteLLM if needed, verifies your key against the proxy, "
889 f"exports the env vars {binary} reads, then hands off. Any arguments are "
890 f"forwarded to `{binary}`."
891 )
892 return _command
895def agent_commands() -> tuple[click.Command, ...]:
896 """Build one top-level command per known agent, e.g. `lite claude`."""
897 return tuple(_make_agent_command(binary, name) for binary, (name, _profiles) in _KNOWN_AGENTS.items())
900__all__ = [
901 "AgentRunError",
902 "ListedModel",
903 "ModelSyncSkipped",
904 "agent_commands",
905 "agent_launch_args",
906 "agent_model_sync_env",
907 "agent_profile",
908 "build_agent_env",
909 "opencode_model_sync_env",
910 "opencode_provider_config",
911 "prepare_pi",
912 "resolve_api_key",
913 "run_agent",
914 "verify_proxy_key",
915]