Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/public_endpoints/public_v1/model_hub.py: 82%
87 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""`GET /public/v1/model_hub`."""
3from collections.abc import Mapping, Sequence
4from dataclasses import dataclass
5from types import MappingProxyType
6from typing import Annotated, Final, Literal, Protocol
8from fastapi import APIRouter, Depends, Request
9from typing_extensions import ReadOnly, TypedDict
11import litellm
12from litellm._logging import verbose_proxy_logger
13from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth
14from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
15from litellm.proxy.list_api.common import PROBLEM_TYPE_BASE, ManagementProblem
16from litellm.proxy.list_api.in_memory import Cells, InMemoryListExecutor
17from litellm.proxy.list_api.list_framework import (
18 FilterSpec,
19 ListSpec,
20 Scope,
21 ScopeAll,
22 SortKey,
23 handle_facet,
24 handle_list,
25)
26from litellm.proxy.utils import PrismaClient
27from litellm.types.proxy.management_endpoints.management_v1 import (
28 FacetListResponse,
29 ListResponse,
30 ProblemDetail,
31)
32from litellm.types.proxy.management_endpoints.model_management_endpoints import (
33 ModelGroupInfoProxy,
34)
36router: Final = APIRouter()
39@dataclass(frozen=True, slots=True)
40class HealthSnapshot:
41 """The health fields a model hub row carries, as the latest health check recorded them."""
43 status: str | None
44 response_time_ms: float | None
45 checked_at: str | None
48class HealthSnapshotLookup(Protocol):
49 """The health half of the list, injected so the page slice decides how much of it runs."""
51 async def latest_for(self, model_groups: Sequence[str]) -> Mapping[str, HealthSnapshot]: ... 51 ↛ exitline 51 didn't return from function 'latest_for' because
54@dataclass(frozen=True, slots=True)
55class PrismaHealthSnapshotLookup:
56 prisma_client: PrismaClient
58 async def latest_for(self, model_groups: Sequence[str]) -> Mapping[str, HealthSnapshot]:
59 checks: Final = await self.prisma_client.get_latest_health_checks_for_models(model_groups)
60 return MappingProxyType(
61 {
62 check.model_name: HealthSnapshot(
63 status=check.status,
64 response_time_ms=check.response_time_ms,
65 checked_at=check.checked_at.isoformat() if check.checked_at else None,
66 )
67 for check in checks
68 }
69 )
72class _HealthFields(TypedDict):
73 health_status: ReadOnly[str | None]
74 health_response_time: ReadOnly[float | None]
75 health_checked_at: ReadOnly[str | None]
78def _with_health(row: ModelGroupInfoProxy, health: HealthSnapshot | None) -> ModelGroupInfoProxy:
79 if health is None:
80 return row
81 update: Final[_HealthFields] = {
82 "health_status": health.status,
83 "health_response_time": health.response_time_ms,
84 "health_checked_at": health.checked_at,
85 }
86 return row.model_copy(update=update)
89@dataclass(frozen=True, slots=True)
90class HealthEnricher:
91 """Resolves health for exactly the rows handed to it, which is the page and never the match set."""
93 lookup: HealthSnapshotLookup
95 async def __call__(self, rows: Sequence[ModelGroupInfoProxy]) -> Sequence[ModelGroupInfoProxy]:
96 health: Final = await self.lookup.latest_for(tuple(row.model_group for row in rows))
97 return tuple(_with_health(row, health.get(row.model_group)) for row in rows)
100FEATURE_PREFIX: Final = "supports_"
103def _features(row: ModelGroupInfoProxy) -> tuple[str, ...]:
104 """A row's capabilities as one repeated field, so selecting two of them matches either.
106 The hub's feature control has always been a multi-select over the `supports_*` flags.
107 One boolean filter per flag would AND them, which is the opposite of what it does.
108 """
109 return tuple(
110 sorted(
111 name.removeprefix(FEATURE_PREFIX)
112 for name, value in row.model_dump().items()
113 if name.startswith(FEATURE_PREFIX) and value is True
114 )
115 )
118def _cells(row: ModelGroupInfoProxy) -> Cells:
119 return MappingProxyType(
120 {
121 "model_group": row.model_group,
122 "mode": row.mode,
123 "providers": tuple(row.providers),
124 "features": _features(row),
125 "max_input_tokens": row.max_input_tokens,
126 "max_output_tokens": row.max_output_tokens,
127 "input_cost_per_token": row.input_cost_per_token,
128 "output_cost_per_token": row.output_cost_per_token,
129 "rpm": row.rpm,
130 "tpm": row.tpm,
131 }
132 )
135def _serialize(row: ModelGroupInfoProxy) -> ModelGroupInfoProxy:
136 """The row shape is the wire shape: the rows served are the router's own model group records."""
137 return row
140def _scope(_caller: UserAPIKeyAuth) -> Scope:
141 """Unconditional, and `/public/v1` is the one surface where that is allowed.
143 Every row here is already a model group the operator published, so a public browse
144 caller seeing all of them is the answer, not a gap in the scoping.
145 """
146 return ScopeAll()
149MODEL_HUB_FILTERS: Final[Mapping[str, FilterSpec]] = MappingProxyType(
150 {
151 "mode": FilterSpec(type=str, ops=frozenset(("eq", "in"))),
152 "providers": FilterSpec(type=str, ops=frozenset(("contains", "in"))),
153 "features": FilterSpec(type=str, ops=frozenset(("in",))),
154 }
155)
157MODEL_HUB_FACETS: Final[Mapping[str, str]] = MappingProxyType(
158 {"providers": "providers", "modes": "mode", "features": "features"}
159)
161MODEL_HUB_LIST_SPEC: Final[ListSpec[ModelGroupInfoProxy, ModelGroupInfoProxy]] = ListSpec(
162 resource="model groups",
163 sortable=frozenset(
164 (
165 "model_group",
166 "mode",
167 "providers",
168 "max_input_tokens",
169 "max_output_tokens",
170 "input_cost_per_token",
171 "output_cost_per_token",
172 "rpm",
173 "tpm",
174 )
175 ),
176 searchable=frozenset(("model_group",)),
177 filters=MODEL_HUB_FILTERS,
178 default_sort=(SortKey(field="model_group", descending=False),),
179 default_page_size=50,
180 max_page_size=100,
181 scope=_scope,
182 serialize=_serialize,
183 tiebreaker="model_group",
184)
187def _published_rows() -> Sequence[ModelGroupInfoProxy]:
188 from litellm.proxy.proxy_server import (
189 _get_model_group_info, # pyright: ignore[reportPrivateUsage] # /public/model_hub imports it the same way
190 llm_router,
191 )
193 if llm_router is None: 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true
194 raise ManagementProblem(
195 ProblemDetail(
196 type=f"{PROBLEM_TYPE_BASE}no-llm-router",
197 title="No models configured",
198 status=400,
199 detail=CommonProxyErrors.no_llm_router.value,
200 )
201 )
202 if litellm.public_model_groups is None: 202 ↛ 203line 202 didn't jump to line 203 because the condition on line 202 was never true
203 return ()
204 return tuple(
205 _get_model_group_info(
206 llm_router=llm_router,
207 all_models_str=litellm.public_model_groups,
208 model_group=None,
209 )
210 )
213def _executor(
214 rows: Sequence[ModelGroupInfoProxy],
215 prisma_client: PrismaClient | None,
216) -> InMemoryListExecutor[ModelGroupInfoProxy]:
217 if prisma_client is None: 217 ↛ 218line 217 didn't jump to line 218 because the condition on line 217 was never true
218 return InMemoryListExecutor(rows=rows, cells=_cells)
219 return InMemoryListExecutor(
220 rows=rows,
221 cells=_cells,
222 enrich_page=HealthEnricher(lookup=PrismaHealthSnapshotLookup(prisma_client=prisma_client)),
223 )
226@router.get(
227 "/model_hub",
228 tags=["public", "model management"], # mutable-ok: fastapi types tags as list[str | Enum]
229 dependencies=(Depends(user_api_key_auth),),
230 response_model=ListResponse[ModelGroupInfoProxy],
231)
232async def public_model_hub_list(
233 request: Request,
234 user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
235) -> ListResponse[ModelGroupInfoProxy]:
236 """
237 The public model groups this proxy publishes, paged, sortable, searchable and
238 filterable, for the public Model Hub page. No authentication.
240 A rejected request answers with the parameters, sort fields and filter operators
241 it would have accepted, so the accepted set stays discoverable from the endpoint
242 itself rather than from a copy of the spec kept here.
244 Example curl:
245 ```
246 curl --location --globoff \
247 'http://0.0.0.0:4000/public/v1/model_hub?sort=-input_cost_per_token&filter[mode][in]=chat&page_size=25'
248 ```
249 """
250 try:
251 from litellm.proxy.proxy_server import prisma_client
253 return await handle_list(
254 spec=MODEL_HUB_LIST_SPEC,
255 executor=_executor(_published_rows(), prisma_client),
256 request=request,
257 caller=user_api_key_dict,
258 )
260 except ManagementProblem:
261 raise
262 except Exception as e: # noqa: BLE001 # a router error answers as a problem document, not the OpenAI error shape
263 verbose_proxy_logger.exception(
264 "litellm.proxy.public_endpoints.public_v1.model_hub.public_model_hub_list(): Exception occured - %s", e
265 )
266 raise ManagementProblem(
267 ProblemDetail(
268 type=f"{PROBLEM_TYPE_BASE}internal-server-error",
269 title="Internal server error",
270 status=500,
271 detail="Failed to list public model groups.",
272 )
273 )
276@router.get(
277 "/model_hub/{facet}",
278 tags=["public", "model management"], # mutable-ok: fastapi types tags as list[str | Enum]
279 dependencies=(Depends(user_api_key_auth),),
280 response_model=FacetListResponse,
281)
282async def public_model_hub_facet(
283 request: Request,
284 facet: Literal["providers", "modes", "features"],
285 user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
286) -> FacetListResponse:
287 """
288 The distinct providers, modes or features across the published model groups, for the
289 Model Hub's filter dropdowns. No authentication.
291 Carries the same filters and search as the list route, so a dropdown offers exactly
292 the values the table can show: asking for providers under `filter[mode][in]=chat`
293 lists only the providers that serve a chat model.
295 Example curl:
296 ```
297 curl --location --globoff \
298 'http://0.0.0.0:4000/public/v1/model_hub/providers?filter[mode][in]=chat&page_size=50'
299 ```
300 """
301 try:
302 return await handle_facet(
303 spec=MODEL_HUB_LIST_SPEC,
304 executor=InMemoryListExecutor(rows=_published_rows(), cells=_cells),
305 request=request,
306 caller=user_api_key_dict,
307 field=MODEL_HUB_FACETS[facet],
308 )
310 except ManagementProblem:
311 raise
312 except Exception as e: # noqa: BLE001 # a router error answers as a problem document, not the OpenAI error shape
313 verbose_proxy_logger.exception(
314 "litellm.proxy.public_endpoints.public_v1.model_hub.public_model_hub_facet(): Exception occured - %s", e
315 )
316 raise ManagementProblem(
317 ProblemDetail(
318 type=f"{PROBLEM_TYPE_BASE}internal-server-error",
319 title="Internal server error",
320 status=500,
321 detail="Failed to list public model group values.",
322 )
323 )