Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/public_endpoints/public_v1/model_hub.py: 82%

87 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1"""`GET /public/v1/model_hub`.""" 

2 

3from collections.abc import Mapping, Sequence 

4from dataclasses import dataclass 

5from types import MappingProxyType 

6from typing import Annotated, Final, Literal, Protocol 

7 

8from fastapi import APIRouter, Depends, Request 

9from typing_extensions import ReadOnly, TypedDict 

10 

11import litellm 

12from litellm._logging import verbose_proxy_logger 

13from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth 

14from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

15from litellm.proxy.list_api.common import PROBLEM_TYPE_BASE, ManagementProblem 

16from litellm.proxy.list_api.in_memory import Cells, InMemoryListExecutor 

17from litellm.proxy.list_api.list_framework import ( 

18 FilterSpec, 

19 ListSpec, 

20 Scope, 

21 ScopeAll, 

22 SortKey, 

23 handle_facet, 

24 handle_list, 

25) 

26from litellm.proxy.utils import PrismaClient 

27from litellm.types.proxy.management_endpoints.management_v1 import ( 

28 FacetListResponse, 

29 ListResponse, 

30 ProblemDetail, 

31) 

32from litellm.types.proxy.management_endpoints.model_management_endpoints import ( 

33 ModelGroupInfoProxy, 

34) 

35 

36router: Final = APIRouter() 

37 

38 

39@dataclass(frozen=True, slots=True) 

40class HealthSnapshot: 

41 """The health fields a model hub row carries, as the latest health check recorded them.""" 

42 

43 status: str | None 

44 response_time_ms: float | None 

45 checked_at: str | None 

46 

47 

48class HealthSnapshotLookup(Protocol): 

49 """The health half of the list, injected so the page slice decides how much of it runs.""" 

50 

51 async def latest_for(self, model_groups: Sequence[str]) -> Mapping[str, HealthSnapshot]: ... 51 ↛ exitline 51 didn't return from function 'latest_for' because

52 

53 

54@dataclass(frozen=True, slots=True) 

55class PrismaHealthSnapshotLookup: 

56 prisma_client: PrismaClient 

57 

58 async def latest_for(self, model_groups: Sequence[str]) -> Mapping[str, HealthSnapshot]: 

59 checks: Final = await self.prisma_client.get_latest_health_checks_for_models(model_groups) 

60 return MappingProxyType( 

61 { 

62 check.model_name: HealthSnapshot( 

63 status=check.status, 

64 response_time_ms=check.response_time_ms, 

65 checked_at=check.checked_at.isoformat() if check.checked_at else None, 

66 ) 

67 for check in checks 

68 } 

69 ) 

70 

71 

72class _HealthFields(TypedDict): 

73 health_status: ReadOnly[str | None] 

74 health_response_time: ReadOnly[float | None] 

75 health_checked_at: ReadOnly[str | None] 

76 

77 

78def _with_health(row: ModelGroupInfoProxy, health: HealthSnapshot | None) -> ModelGroupInfoProxy: 

79 if health is None: 

80 return row 

81 update: Final[_HealthFields] = { 

82 "health_status": health.status, 

83 "health_response_time": health.response_time_ms, 

84 "health_checked_at": health.checked_at, 

85 } 

86 return row.model_copy(update=update) 

87 

88 

89@dataclass(frozen=True, slots=True) 

90class HealthEnricher: 

91 """Resolves health for exactly the rows handed to it, which is the page and never the match set.""" 

92 

93 lookup: HealthSnapshotLookup 

94 

95 async def __call__(self, rows: Sequence[ModelGroupInfoProxy]) -> Sequence[ModelGroupInfoProxy]: 

96 health: Final = await self.lookup.latest_for(tuple(row.model_group for row in rows)) 

97 return tuple(_with_health(row, health.get(row.model_group)) for row in rows) 

98 

99 

100FEATURE_PREFIX: Final = "supports_" 

101 

102 

103def _features(row: ModelGroupInfoProxy) -> tuple[str, ...]: 

104 """A row's capabilities as one repeated field, so selecting two of them matches either. 

105 

106 The hub's feature control has always been a multi-select over the `supports_*` flags. 

107 One boolean filter per flag would AND them, which is the opposite of what it does. 

108 """ 

109 return tuple( 

110 sorted( 

111 name.removeprefix(FEATURE_PREFIX) 

112 for name, value in row.model_dump().items() 

113 if name.startswith(FEATURE_PREFIX) and value is True 

114 ) 

115 ) 

116 

117 

118def _cells(row: ModelGroupInfoProxy) -> Cells: 

119 return MappingProxyType( 

120 { 

121 "model_group": row.model_group, 

122 "mode": row.mode, 

123 "providers": tuple(row.providers), 

124 "features": _features(row), 

125 "max_input_tokens": row.max_input_tokens, 

126 "max_output_tokens": row.max_output_tokens, 

127 "input_cost_per_token": row.input_cost_per_token, 

128 "output_cost_per_token": row.output_cost_per_token, 

129 "rpm": row.rpm, 

130 "tpm": row.tpm, 

131 } 

132 ) 

133 

134 

135def _serialize(row: ModelGroupInfoProxy) -> ModelGroupInfoProxy: 

136 """The row shape is the wire shape: the rows served are the router's own model group records.""" 

137 return row 

138 

139 

140def _scope(_caller: UserAPIKeyAuth) -> Scope: 

141 """Unconditional, and `/public/v1` is the one surface where that is allowed. 

142 

143 Every row here is already a model group the operator published, so a public browse 

144 caller seeing all of them is the answer, not a gap in the scoping. 

145 """ 

146 return ScopeAll() 

147 

148 

149MODEL_HUB_FILTERS: Final[Mapping[str, FilterSpec]] = MappingProxyType( 

150 { 

151 "mode": FilterSpec(type=str, ops=frozenset(("eq", "in"))), 

152 "providers": FilterSpec(type=str, ops=frozenset(("contains", "in"))), 

153 "features": FilterSpec(type=str, ops=frozenset(("in",))), 

154 } 

155) 

156 

157MODEL_HUB_FACETS: Final[Mapping[str, str]] = MappingProxyType( 

158 {"providers": "providers", "modes": "mode", "features": "features"} 

159) 

160 

161MODEL_HUB_LIST_SPEC: Final[ListSpec[ModelGroupInfoProxy, ModelGroupInfoProxy]] = ListSpec( 

162 resource="model groups", 

163 sortable=frozenset( 

164 ( 

165 "model_group", 

166 "mode", 

167 "providers", 

168 "max_input_tokens", 

169 "max_output_tokens", 

170 "input_cost_per_token", 

171 "output_cost_per_token", 

172 "rpm", 

173 "tpm", 

174 ) 

175 ), 

176 searchable=frozenset(("model_group",)), 

177 filters=MODEL_HUB_FILTERS, 

178 default_sort=(SortKey(field="model_group", descending=False),), 

179 default_page_size=50, 

180 max_page_size=100, 

181 scope=_scope, 

182 serialize=_serialize, 

183 tiebreaker="model_group", 

184) 

185 

186 

187def _published_rows() -> Sequence[ModelGroupInfoProxy]: 

188 from litellm.proxy.proxy_server import ( 

189 _get_model_group_info, # pyright: ignore[reportPrivateUsage] # /public/model_hub imports it the same way 

190 llm_router, 

191 ) 

192 

193 if llm_router is None: 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true

194 raise ManagementProblem( 

195 ProblemDetail( 

196 type=f"{PROBLEM_TYPE_BASE}no-llm-router", 

197 title="No models configured", 

198 status=400, 

199 detail=CommonProxyErrors.no_llm_router.value, 

200 ) 

201 ) 

202 if litellm.public_model_groups is None: 202 ↛ 203line 202 didn't jump to line 203 because the condition on line 202 was never true

203 return () 

204 return tuple( 

205 _get_model_group_info( 

206 llm_router=llm_router, 

207 all_models_str=litellm.public_model_groups, 

208 model_group=None, 

209 ) 

210 ) 

211 

212 

213def _executor( 

214 rows: Sequence[ModelGroupInfoProxy], 

215 prisma_client: PrismaClient | None, 

216) -> InMemoryListExecutor[ModelGroupInfoProxy]: 

217 if prisma_client is None: 217 ↛ 218line 217 didn't jump to line 218 because the condition on line 217 was never true

218 return InMemoryListExecutor(rows=rows, cells=_cells) 

219 return InMemoryListExecutor( 

220 rows=rows, 

221 cells=_cells, 

222 enrich_page=HealthEnricher(lookup=PrismaHealthSnapshotLookup(prisma_client=prisma_client)), 

223 ) 

224 

225 

226@router.get( 

227 "/model_hub", 

228 tags=["public", "model management"], # mutable-ok: fastapi types tags as list[str | Enum] 

229 dependencies=(Depends(user_api_key_auth),), 

230 response_model=ListResponse[ModelGroupInfoProxy], 

231) 

232async def public_model_hub_list( 

233 request: Request, 

234 user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], 

235) -> ListResponse[ModelGroupInfoProxy]: 

236 """ 

237 The public model groups this proxy publishes, paged, sortable, searchable and 

238 filterable, for the public Model Hub page. No authentication. 

239 

240 A rejected request answers with the parameters, sort fields and filter operators 

241 it would have accepted, so the accepted set stays discoverable from the endpoint 

242 itself rather than from a copy of the spec kept here. 

243 

244 Example curl: 

245 ``` 

246 curl --location --globoff \ 

247 'http://0.0.0.0:4000/public/v1/model_hub?sort=-input_cost_per_token&filter[mode][in]=chat&page_size=25' 

248 ``` 

249 """ 

250 try: 

251 from litellm.proxy.proxy_server import prisma_client 

252 

253 return await handle_list( 

254 spec=MODEL_HUB_LIST_SPEC, 

255 executor=_executor(_published_rows(), prisma_client), 

256 request=request, 

257 caller=user_api_key_dict, 

258 ) 

259 

260 except ManagementProblem: 

261 raise 

262 except Exception as e: # noqa: BLE001 # a router error answers as a problem document, not the OpenAI error shape 

263 verbose_proxy_logger.exception( 

264 "litellm.proxy.public_endpoints.public_v1.model_hub.public_model_hub_list(): Exception occured - %s", e 

265 ) 

266 raise ManagementProblem( 

267 ProblemDetail( 

268 type=f"{PROBLEM_TYPE_BASE}internal-server-error", 

269 title="Internal server error", 

270 status=500, 

271 detail="Failed to list public model groups.", 

272 ) 

273 ) 

274 

275 

276@router.get( 

277 "/model_hub/{facet}", 

278 tags=["public", "model management"], # mutable-ok: fastapi types tags as list[str | Enum] 

279 dependencies=(Depends(user_api_key_auth),), 

280 response_model=FacetListResponse, 

281) 

282async def public_model_hub_facet( 

283 request: Request, 

284 facet: Literal["providers", "modes", "features"], 

285 user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], 

286) -> FacetListResponse: 

287 """ 

288 The distinct providers, modes or features across the published model groups, for the 

289 Model Hub's filter dropdowns. No authentication. 

290 

291 Carries the same filters and search as the list route, so a dropdown offers exactly 

292 the values the table can show: asking for providers under `filter[mode][in]=chat` 

293 lists only the providers that serve a chat model. 

294 

295 Example curl: 

296 ``` 

297 curl --location --globoff \ 

298 'http://0.0.0.0:4000/public/v1/model_hub/providers?filter[mode][in]=chat&page_size=50' 

299 ``` 

300 """ 

301 try: 

302 return await handle_facet( 

303 spec=MODEL_HUB_LIST_SPEC, 

304 executor=InMemoryListExecutor(rows=_published_rows(), cells=_cells), 

305 request=request, 

306 caller=user_api_key_dict, 

307 field=MODEL_HUB_FACETS[facet], 

308 ) 

309 

310 except ManagementProblem: 

311 raise 

312 except Exception as e: # noqa: BLE001 # a router error answers as a problem document, not the OpenAI error shape 

313 verbose_proxy_logger.exception( 

314 "litellm.proxy.public_endpoints.public_v1.model_hub.public_model_hub_facet(): Exception occured - %s", e 

315 ) 

316 raise ManagementProblem( 

317 ProblemDetail( 

318 type=f"{PROBLEM_TYPE_BASE}internal-server-error", 

319 title="Internal server error", 

320 status=500, 

321 detail="Failed to list public model group values.", 

322 ) 

323 )