Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/anthropic_endpoints/endpoints.py: 48%

120 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Unified /v1/messages endpoint - (Anthropic Spec) 

3""" 

4 

5from typing import Final 

6 

7from fastapi import APIRouter, Depends, HTTPException, Request, Response 

8from fastapi.responses import JSONResponse 

9 

10import litellm 

11from litellm.anthropic_interface.exceptions import ( 

12 AnthropicErrorDetail, 

13 AnthropicErrorResponse, 

14 AnthropicExceptionMapping, 

15) 

16from litellm.integrations.custom_guardrail import ModifyResponseException 

17from litellm.llms.anthropic.experimental_pass_through.context_management import ( 

18 AnthropicContextManagementError, 

19) 

20from litellm.llms.base_llm.guardrail_translation.utils import ( 

21 blocked_response_usage as _blocked_response_usage, 

22) 

23from litellm.proxy._types import * 

24from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

25from litellm.proxy.common_request_processing import ( 

26 ProxyBaseLLMRequestProcessing, 

27 create_response, 

28 log_llm_api_exception, 

29 proxy_exception_from_http_exception, 

30 resolve_litellm_call_id, 

31) 

32from litellm.proxy.common_utils.error_body_call_id import error_body_call_id 

33from litellm.proxy.common_utils.http_parsing_utils import _read_request_body 

34from litellm.proxy.common_utils.openai_error_payload import ( 

35 LITELLM_CALL_ID_HEADER, 

36 error_status_code, 

37 openai_error_param, 

38 openai_error_type, 

39 with_litellm_call_id, 

40) 

41from litellm.types.utils import TokenCountResponse 

42 

43router: Final = APIRouter() 

44 

45 

46def _with_provider_specific_fields(exc: ProxyException, detail: AnthropicErrorDetail) -> AnthropicErrorDetail: 

47 if not exc.provider_specific_fields: 47 ↛ 48line 47 didn't jump to line 48 because the condition on line 47 was never true

48 return detail 

49 with_fields: Final[AnthropicErrorDetail] = {**detail, "provider_specific_fields": exc.provider_specific_fields} 

50 return with_fields 

51 

52 

53def _anthropic_error_detail( 

54 exc: ProxyException, detail: AnthropicErrorDetail, call_id: str | None 

55) -> AnthropicErrorDetail: 

56 if call_id is None: 56 ↛ 58line 56 didn't jump to line 58 because the condition on line 56 was always true

57 return _with_provider_specific_fields(exc, detail) 

58 with_call_id: Final[AnthropicErrorDetail] = { 

59 **_with_provider_specific_fields(exc, detail), 

60 "litellm_call_id": call_id, 

61 } 

62 return with_call_id 

63 

64 

65def _anthropic_error_json_response(exc: ProxyException, request: Request) -> JSONResponse: 

66 from litellm.proxy.proxy_server import ( 

67 _close_dangling_otel_server_span, # pyright: ignore[reportPrivateUsage] # proxy_server keeps the span-close helper private; error JSONResponses returned by the route must stamp the OTel server span like the global ProxyException handler does 

68 general_settings_view, 

69 ) 

70 

71 status_code: Final = int(exc.code) if exc.code is not None and exc.code.isdigit() else 500 

72 _close_dangling_otel_server_span(request, status_code, exc=exc) 

73 envelope: Final = AnthropicExceptionMapping.transform_to_anthropic_error( 

74 status_code=status_code, 

75 raw_message=exc.message, 

76 request_id=request.headers.get("x-request-id"), 

77 ) 

78 body_call_id: Final = error_body_call_id(general_settings_view(), exc.headers.get(LITELLM_CALL_ID_HEADER)) 

79 content: Final[AnthropicErrorResponse] = { 

80 **envelope, 

81 "error": _anthropic_error_detail(exc, envelope["error"], body_call_id), 

82 } 

83 return JSONResponse(status_code=status_code, content=content, headers=exc.headers) 

84 

85 

86def _strip_total_tokens_from_anthropic_response(response: Any) -> None: 

87 """Remove the OpenAI-flavored `usage.total_tokens` field that LiteLLM 

88 injects into Anthropic /v1/messages responses. 

89 

90 The Anthropic /v1/messages spec only defines: 

91 input_tokens, output_tokens, cache_creation_input_tokens, 

92 cache_read_input_tokens, cache_creation.{ephemeral_5m,ephemeral_1h} 

93 The streaming SSE path (message_delta.usage) already does not include 

94 total_tokens; this brings the non-streaming path into the same shape. 

95 

96 Handles both shapes returned by `base_process_llm_request`: 

97 - plain `dict` (most common — `AnthropicMessagesResponse` is a TypedDict 

98 and is `dict` at runtime) 

99 - Pydantic model whose `usage` attribute is dict-shaped (e.g. a 

100 BaseModel that holds raw Anthropic usage as a `dict[str, int]`) 

101 

102 Streaming results (StreamingResponse, AsyncIterator, etc.) and Pydantic 

103 models with strongly-typed Usage sub-models are left untouched — 

104 those paths either have separate serialization handling or impose 

105 type constraints the helper does not try to subvert. 

106 """ 

107 if response is None: 

108 return 

109 if isinstance(response, dict): 

110 usage = response.get("usage") 

111 if isinstance(usage, dict) and "total_tokens" in usage: 

112 usage.pop("total_tokens", None) 

113 return 

114 # Pydantic-model fallback: only mutate if `usage` is a dict. 

115 usage = getattr(response, "usage", None) 

116 if isinstance(usage, dict) and "total_tokens" in usage: 

117 usage.pop("total_tokens", None) 

118 

119 

120@router.post( 

121 "/v1/messages", 

122 tags=["[beta] Anthropic `/v1/messages`"], 

123 dependencies=[Depends(user_api_key_auth)], 

124) 

125async def anthropic_response( 

126 fastapi_response: Response, 

127 request: Request, 

128 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

129): 

130 """ 

131 Use `{PROXY_BASE_URL}/anthropic/v1/messages` instead - [Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion). 

132 

133 This was a BETA endpoint that calls 100+ LLMs in the anthropic format. 

134 """ 

135 from litellm.proxy.proxy_server import ( 

136 general_settings, 

137 llm_router, 

138 proxy_config, 

139 proxy_logging_obj, 

140 user_api_base, 

141 user_max_tokens, 

142 user_model, 

143 user_request_timeout, 

144 user_temperature, 

145 version, 

146 ) 

147 

148 data: Final = await _read_request_body(request=request) 

149 base_llm_response_processor: Final = ProxyBaseLLMRequestProcessing(data=data) 

150 try: 

151 result: Final = await base_llm_response_processor.base_process_llm_request( 

152 request=request, 

153 fastapi_response=fastapi_response, 

154 user_api_key_dict=user_api_key_dict, 

155 route_type="anthropic_messages", 

156 proxy_logging_obj=proxy_logging_obj, 

157 llm_router=llm_router, 

158 general_settings=general_settings, 

159 proxy_config=proxy_config, 

160 select_data_generator=None, 

161 model=None, 

162 user_model=user_model, 

163 user_temperature=user_temperature, 

164 user_request_timeout=user_request_timeout, 

165 user_max_tokens=user_max_tokens, 

166 user_api_base=user_api_base, 

167 version=version, 

168 ) 

169 # Optionally strip the non-Anthropic `usage.total_tokens` field 

170 # LiteLLM adds internally. Anthropic's official /v1/messages spec 

171 # only defines input_tokens / output_tokens / cache_*_input_tokens; 

172 # total_tokens is an OpenAI convention. Default off 

173 # (`litellm.strip_anthropic_total_tokens = False`) to preserve 

174 # backward compatibility for clients that currently read it; set 

175 # to True to align the wire response with the spec (and with the 

176 # streaming SSE path, which already omits total_tokens). 

177 # spend_logs / Prometheus still compute total internally — this 

178 # only affects the wire response. 

179 if litellm.strip_anthropic_total_tokens: 

180 _strip_total_tokens_from_anthropic_response(result) 

181 return result 

182 except ModifyResponseException as e: 

183 # Guardrail flagged content in passthrough mode - return 200 with violation message 

184 _data: Final = e.request_data 

185 await proxy_logging_obj.post_call_failure_hook( 

186 user_api_key_dict=user_api_key_dict, 

187 original_exception=e, 

188 request_data=_data, 

189 ) 

190 

191 # Create Anthropic-formatted response with violation message 

192 import uuid 

193 

194 from litellm.types.utils import AnthropicMessagesResponse 

195 

196 # Report the blocked LLM response's real token usage (carried on the 

197 # exception) instead of discarding it; zero for pre-call blocks. 

198 _usage: Final = _blocked_response_usage(e.original_response) 

199 

200 _anthropic_response: Final = AnthropicMessagesResponse( 

201 id=f"msg_{uuid.uuid4()}", 

202 type="message", 

203 role="assistant", 

204 content=[{"type": "text", "text": e.message}], 

205 model=e.model, 

206 stop_reason="end_turn", 

207 usage=_usage, 

208 ) 

209 

210 if data.get("stream", None) is not None and data["stream"] is True: 

211 # For streaming, use the standard SSE data generator 

212 async def _passthrough_stream_generator(): 

213 yield _anthropic_response 

214 

215 selected_data_generator: Final = ProxyBaseLLMRequestProcessing.async_sse_data_generator( 

216 response=_passthrough_stream_generator(), 

217 user_api_key_dict=user_api_key_dict, 

218 request_data=_data, 

219 proxy_logging_obj=proxy_logging_obj, 

220 ) 

221 

222 return await create_response( 

223 generator=selected_data_generator, 

224 media_type="text/event-stream", 

225 headers={}, 

226 ) 

227 

228 return _anthropic_response 

229 except AnthropicContextManagementError as e: 

230 if e.status_code >= 500: 

231 # Server-side polyfill failures hit the failure hook for spend/alert 

232 # parity with the generic handler; 4xx validation errors do not. 

233 await proxy_logging_obj.post_call_failure_hook( 

234 user_api_key_dict=user_api_key_dict, 

235 original_exception=e, 

236 request_data=base_llm_response_processor.data, 

237 ) 

238 body: Final = AnthropicExceptionMapping.transform_to_anthropic_error( 

239 status_code=e.status_code, 

240 raw_message=e.message, 

241 request_id=request.headers.get("x-request-id"), 

242 ) 

243 return JSONResponse(status_code=e.status_code, content=body) 

244 except Exception as e: 

245 await proxy_logging_obj.post_call_failure_hook( 

246 user_api_key_dict=user_api_key_dict, original_exception=e, request_data=base_llm_response_processor.data 

247 ) 

248 log_llm_api_exception(e, base_llm_response_processor.litellm_call_id) 

249 

250 if isinstance(e, ProxyException): 250 ↛ 251line 250 didn't jump to line 251 because the condition on line 250 was never true

251 return _anthropic_error_json_response( 

252 with_litellm_call_id(e, base_llm_response_processor.litellm_call_id), request 

253 ) 

254 

255 # Extract model_id from request metadata (same as success path) 

256 litellm_metadata: Final = data.get("litellm_metadata", {}) or {} 

257 model_info: Final = litellm_metadata.get("model_info", {}) or {} 

258 model_id: Final = model_info.get("id", "") or "" 

259 

260 # Get headers 

261 headers: Final = ProxyBaseLLMRequestProcessing.get_custom_headers( 

262 user_api_key_dict=user_api_key_dict, 

263 call_id=base_llm_response_processor.litellm_call_id, 

264 model_id=model_id, 

265 version=version, 

266 response_cost=0, 

267 model_region=getattr(user_api_key_dict, "allowed_model_region", ""), 

268 request_data=data, 

269 timeout=getattr(e, "timeout", None), 

270 litellm_logging_obj=None, 

271 ) 

272 

273 if isinstance(e, HTTPException): 273 ↛ 276line 273 didn't jump to line 276 because the condition on line 273 was always true

274 return _anthropic_error_json_response(proxy_exception_from_http_exception(e, headers), request) 

275 

276 error_msg: Final = f"{e}" 

277 return _anthropic_error_json_response( 

278 ProxyException( 

279 message=getattr(e, "message", error_msg), 

280 type=openai_error_type(e, error_status_code(e, 500)), 

281 param=openai_error_param(e), 

282 code=error_status_code(e, 500), 

283 headers=headers, 

284 ), 

285 request, 

286 ) 

287 

288 

289@router.post( 

290 "/v1/messages/count_tokens", 

291 tags=["[beta] Anthropic Messages Token Counting"], 

292 dependencies=[Depends(user_api_key_auth)], 

293) 

294async def count_tokens( 

295 request: Request, 

296 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), # Used for auth 

297): 

298 """ 

299 Count tokens for Anthropic Messages API format. 

300  

301 This endpoint follows the Anthropic Messages API token counting specification. 

302 It accepts the same parameters as the /v1/messages endpoint but returns 

303 token counts instead of generating a response. 

304  

305 Example usage: 

306 ``` 

307 curl -X POST "http://localhost:4000/v1/messages/count_tokens?beta=true" \ 

308 -H "Content-Type: application/json" \ 

309 -H "Authorization: Bearer your-key" \ 

310 -d '{ 

311 "model": "claude-3-sonnet-20240229", 

312 "messages": [{"role": "user", "content": "Hello Claude!"}] 

313 }' 

314 ``` 

315  

316 Returns: {"input_tokens": <number>} 

317 """ 

318 from litellm.proxy.proxy_server import token_counter as internal_token_counter 

319 

320 litellm_call_id: Final = resolve_litellm_call_id(request.headers.get("x-litellm-call-id")) 

321 try: 

322 request_data: Final = await _read_request_body(request=request) 

323 data: Final[dict] = {**request_data} 

324 

325 # Extract required fields 

326 model_name: Final = data.get("model") 

327 messages: Final = data.get("messages", []) 

328 

329 if not model_name: 329 ↛ 332line 329 didn't jump to line 332 because the condition on line 329 was always true

330 raise HTTPException(status_code=400, detail={"error": "model parameter is required"}) 

331 

332 if not messages: 

333 raise HTTPException(status_code=400, detail={"error": "messages parameter is required"}) 

334 

335 # Create TokenCountRequest for the internal endpoint 

336 from litellm.proxy._types import TokenCountRequest 

337 

338 token_request: Final = TokenCountRequest( 

339 model=model_name, 

340 messages=messages, 

341 tools=data.get("tools"), 

342 system=data.get("system"), 

343 ) 

344 

345 # Call the internal token counter function with direct request flag set to False 

346 token_response: Final = await internal_token_counter( 

347 request=token_request, 

348 call_endpoint=True, 

349 ) 

350 _token_response_dict: dict = {} 

351 if isinstance(token_response, TokenCountResponse): 

352 _token_response_dict = token_response.model_dump() 

353 elif isinstance(token_response, dict): 

354 _token_response_dict = token_response 

355 

356 # Convert the internal response to Anthropic API format 

357 return {"input_tokens": _token_response_dict.get("total_tokens", 0)} 

358 

359 except HTTPException: 

360 raise 

361 except ProxyException as e: 

362 status_code: Final = int(e.code) if e.code and e.code.isdigit() else 500 

363 detail: Final = AnthropicExceptionMapping.transform_to_anthropic_error( 

364 status_code=status_code, 

365 raw_message=e.message, 

366 ) 

367 raise HTTPException( 

368 status_code=status_code, 

369 detail=detail, 

370 ) 

371 except Exception as e: 

372 log_llm_api_exception(e, litellm_call_id) 

373 raise HTTPException(status_code=500, detail={"error": f"Internal server error: {e}"}) 

374 

375 

376@router.post( 

377 "/api/event_logging/batch", 

378 tags=["[beta] Anthropic Event Logging"], 

379) 

380async def event_logging_batch( 

381 request: Request, 

382): 

383 """ 

384 Stubbed endpoint for Anthropic event logging batch requests. 

385 

386 This endpoint accepts event logging requests but does nothing with them. 

387 It exists to prevent 404 errors from Claude Code clients that send telemetry. 

388 """ 

389 return {"status": "ok"}