Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/anthropic_endpoints/endpoints.py: 48%
120 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Unified /v1/messages endpoint - (Anthropic Spec)
3"""
5from typing import Final
7from fastapi import APIRouter, Depends, HTTPException, Request, Response
8from fastapi.responses import JSONResponse
10import litellm
11from litellm.anthropic_interface.exceptions import (
12 AnthropicErrorDetail,
13 AnthropicErrorResponse,
14 AnthropicExceptionMapping,
15)
16from litellm.integrations.custom_guardrail import ModifyResponseException
17from litellm.llms.anthropic.experimental_pass_through.context_management import (
18 AnthropicContextManagementError,
19)
20from litellm.llms.base_llm.guardrail_translation.utils import (
21 blocked_response_usage as _blocked_response_usage,
22)
23from litellm.proxy._types import *
24from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
25from litellm.proxy.common_request_processing import (
26 ProxyBaseLLMRequestProcessing,
27 create_response,
28 log_llm_api_exception,
29 proxy_exception_from_http_exception,
30 resolve_litellm_call_id,
31)
32from litellm.proxy.common_utils.error_body_call_id import error_body_call_id
33from litellm.proxy.common_utils.http_parsing_utils import _read_request_body
34from litellm.proxy.common_utils.openai_error_payload import (
35 LITELLM_CALL_ID_HEADER,
36 error_status_code,
37 openai_error_param,
38 openai_error_type,
39 with_litellm_call_id,
40)
41from litellm.types.utils import TokenCountResponse
43router: Final = APIRouter()
46def _with_provider_specific_fields(exc: ProxyException, detail: AnthropicErrorDetail) -> AnthropicErrorDetail:
47 if not exc.provider_specific_fields: 47 ↛ 48line 47 didn't jump to line 48 because the condition on line 47 was never true
48 return detail
49 with_fields: Final[AnthropicErrorDetail] = {**detail, "provider_specific_fields": exc.provider_specific_fields}
50 return with_fields
53def _anthropic_error_detail(
54 exc: ProxyException, detail: AnthropicErrorDetail, call_id: str | None
55) -> AnthropicErrorDetail:
56 if call_id is None: 56 ↛ 58line 56 didn't jump to line 58 because the condition on line 56 was always true
57 return _with_provider_specific_fields(exc, detail)
58 with_call_id: Final[AnthropicErrorDetail] = {
59 **_with_provider_specific_fields(exc, detail),
60 "litellm_call_id": call_id,
61 }
62 return with_call_id
65def _anthropic_error_json_response(exc: ProxyException, request: Request) -> JSONResponse:
66 from litellm.proxy.proxy_server import (
67 _close_dangling_otel_server_span, # pyright: ignore[reportPrivateUsage] # proxy_server keeps the span-close helper private; error JSONResponses returned by the route must stamp the OTel server span like the global ProxyException handler does
68 general_settings_view,
69 )
71 status_code: Final = int(exc.code) if exc.code is not None and exc.code.isdigit() else 500
72 _close_dangling_otel_server_span(request, status_code, exc=exc)
73 envelope: Final = AnthropicExceptionMapping.transform_to_anthropic_error(
74 status_code=status_code,
75 raw_message=exc.message,
76 request_id=request.headers.get("x-request-id"),
77 )
78 body_call_id: Final = error_body_call_id(general_settings_view(), exc.headers.get(LITELLM_CALL_ID_HEADER))
79 content: Final[AnthropicErrorResponse] = {
80 **envelope,
81 "error": _anthropic_error_detail(exc, envelope["error"], body_call_id),
82 }
83 return JSONResponse(status_code=status_code, content=content, headers=exc.headers)
86def _strip_total_tokens_from_anthropic_response(response: Any) -> None:
87 """Remove the OpenAI-flavored `usage.total_tokens` field that LiteLLM
88 injects into Anthropic /v1/messages responses.
90 The Anthropic /v1/messages spec only defines:
91 input_tokens, output_tokens, cache_creation_input_tokens,
92 cache_read_input_tokens, cache_creation.{ephemeral_5m,ephemeral_1h}
93 The streaming SSE path (message_delta.usage) already does not include
94 total_tokens; this brings the non-streaming path into the same shape.
96 Handles both shapes returned by `base_process_llm_request`:
97 - plain `dict` (most common — `AnthropicMessagesResponse` is a TypedDict
98 and is `dict` at runtime)
99 - Pydantic model whose `usage` attribute is dict-shaped (e.g. a
100 BaseModel that holds raw Anthropic usage as a `dict[str, int]`)
102 Streaming results (StreamingResponse, AsyncIterator, etc.) and Pydantic
103 models with strongly-typed Usage sub-models are left untouched —
104 those paths either have separate serialization handling or impose
105 type constraints the helper does not try to subvert.
106 """
107 if response is None:
108 return
109 if isinstance(response, dict):
110 usage = response.get("usage")
111 if isinstance(usage, dict) and "total_tokens" in usage:
112 usage.pop("total_tokens", None)
113 return
114 # Pydantic-model fallback: only mutate if `usage` is a dict.
115 usage = getattr(response, "usage", None)
116 if isinstance(usage, dict) and "total_tokens" in usage:
117 usage.pop("total_tokens", None)
120@router.post(
121 "/v1/messages",
122 tags=["[beta] Anthropic `/v1/messages`"],
123 dependencies=[Depends(user_api_key_auth)],
124)
125async def anthropic_response(
126 fastapi_response: Response,
127 request: Request,
128 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
129):
130 """
131 Use `{PROXY_BASE_URL}/anthropic/v1/messages` instead - [Docs](https://docs.litellm.ai/docs/pass_through/anthropic_completion).
133 This was a BETA endpoint that calls 100+ LLMs in the anthropic format.
134 """
135 from litellm.proxy.proxy_server import (
136 general_settings,
137 llm_router,
138 proxy_config,
139 proxy_logging_obj,
140 user_api_base,
141 user_max_tokens,
142 user_model,
143 user_request_timeout,
144 user_temperature,
145 version,
146 )
148 data: Final = await _read_request_body(request=request)
149 base_llm_response_processor: Final = ProxyBaseLLMRequestProcessing(data=data)
150 try:
151 result: Final = await base_llm_response_processor.base_process_llm_request(
152 request=request,
153 fastapi_response=fastapi_response,
154 user_api_key_dict=user_api_key_dict,
155 route_type="anthropic_messages",
156 proxy_logging_obj=proxy_logging_obj,
157 llm_router=llm_router,
158 general_settings=general_settings,
159 proxy_config=proxy_config,
160 select_data_generator=None,
161 model=None,
162 user_model=user_model,
163 user_temperature=user_temperature,
164 user_request_timeout=user_request_timeout,
165 user_max_tokens=user_max_tokens,
166 user_api_base=user_api_base,
167 version=version,
168 )
169 # Optionally strip the non-Anthropic `usage.total_tokens` field
170 # LiteLLM adds internally. Anthropic's official /v1/messages spec
171 # only defines input_tokens / output_tokens / cache_*_input_tokens;
172 # total_tokens is an OpenAI convention. Default off
173 # (`litellm.strip_anthropic_total_tokens = False`) to preserve
174 # backward compatibility for clients that currently read it; set
175 # to True to align the wire response with the spec (and with the
176 # streaming SSE path, which already omits total_tokens).
177 # spend_logs / Prometheus still compute total internally — this
178 # only affects the wire response.
179 if litellm.strip_anthropic_total_tokens:
180 _strip_total_tokens_from_anthropic_response(result)
181 return result
182 except ModifyResponseException as e:
183 # Guardrail flagged content in passthrough mode - return 200 with violation message
184 _data: Final = e.request_data
185 await proxy_logging_obj.post_call_failure_hook(
186 user_api_key_dict=user_api_key_dict,
187 original_exception=e,
188 request_data=_data,
189 )
191 # Create Anthropic-formatted response with violation message
192 import uuid
194 from litellm.types.utils import AnthropicMessagesResponse
196 # Report the blocked LLM response's real token usage (carried on the
197 # exception) instead of discarding it; zero for pre-call blocks.
198 _usage: Final = _blocked_response_usage(e.original_response)
200 _anthropic_response: Final = AnthropicMessagesResponse(
201 id=f"msg_{uuid.uuid4()}",
202 type="message",
203 role="assistant",
204 content=[{"type": "text", "text": e.message}],
205 model=e.model,
206 stop_reason="end_turn",
207 usage=_usage,
208 )
210 if data.get("stream", None) is not None and data["stream"] is True:
211 # For streaming, use the standard SSE data generator
212 async def _passthrough_stream_generator():
213 yield _anthropic_response
215 selected_data_generator: Final = ProxyBaseLLMRequestProcessing.async_sse_data_generator(
216 response=_passthrough_stream_generator(),
217 user_api_key_dict=user_api_key_dict,
218 request_data=_data,
219 proxy_logging_obj=proxy_logging_obj,
220 )
222 return await create_response(
223 generator=selected_data_generator,
224 media_type="text/event-stream",
225 headers={},
226 )
228 return _anthropic_response
229 except AnthropicContextManagementError as e:
230 if e.status_code >= 500:
231 # Server-side polyfill failures hit the failure hook for spend/alert
232 # parity with the generic handler; 4xx validation errors do not.
233 await proxy_logging_obj.post_call_failure_hook(
234 user_api_key_dict=user_api_key_dict,
235 original_exception=e,
236 request_data=base_llm_response_processor.data,
237 )
238 body: Final = AnthropicExceptionMapping.transform_to_anthropic_error(
239 status_code=e.status_code,
240 raw_message=e.message,
241 request_id=request.headers.get("x-request-id"),
242 )
243 return JSONResponse(status_code=e.status_code, content=body)
244 except Exception as e:
245 await proxy_logging_obj.post_call_failure_hook(
246 user_api_key_dict=user_api_key_dict, original_exception=e, request_data=base_llm_response_processor.data
247 )
248 log_llm_api_exception(e, base_llm_response_processor.litellm_call_id)
250 if isinstance(e, ProxyException): 250 ↛ 251line 250 didn't jump to line 251 because the condition on line 250 was never true
251 return _anthropic_error_json_response(
252 with_litellm_call_id(e, base_llm_response_processor.litellm_call_id), request
253 )
255 # Extract model_id from request metadata (same as success path)
256 litellm_metadata: Final = data.get("litellm_metadata", {}) or {}
257 model_info: Final = litellm_metadata.get("model_info", {}) or {}
258 model_id: Final = model_info.get("id", "") or ""
260 # Get headers
261 headers: Final = ProxyBaseLLMRequestProcessing.get_custom_headers(
262 user_api_key_dict=user_api_key_dict,
263 call_id=base_llm_response_processor.litellm_call_id,
264 model_id=model_id,
265 version=version,
266 response_cost=0,
267 model_region=getattr(user_api_key_dict, "allowed_model_region", ""),
268 request_data=data,
269 timeout=getattr(e, "timeout", None),
270 litellm_logging_obj=None,
271 )
273 if isinstance(e, HTTPException): 273 ↛ 276line 273 didn't jump to line 276 because the condition on line 273 was always true
274 return _anthropic_error_json_response(proxy_exception_from_http_exception(e, headers), request)
276 error_msg: Final = f"{e}"
277 return _anthropic_error_json_response(
278 ProxyException(
279 message=getattr(e, "message", error_msg),
280 type=openai_error_type(e, error_status_code(e, 500)),
281 param=openai_error_param(e),
282 code=error_status_code(e, 500),
283 headers=headers,
284 ),
285 request,
286 )
289@router.post(
290 "/v1/messages/count_tokens",
291 tags=["[beta] Anthropic Messages Token Counting"],
292 dependencies=[Depends(user_api_key_auth)],
293)
294async def count_tokens(
295 request: Request,
296 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), # Used for auth
297):
298 """
299 Count tokens for Anthropic Messages API format.
301 This endpoint follows the Anthropic Messages API token counting specification.
302 It accepts the same parameters as the /v1/messages endpoint but returns
303 token counts instead of generating a response.
305 Example usage:
306 ```
307 curl -X POST "http://localhost:4000/v1/messages/count_tokens?beta=true" \
308 -H "Content-Type: application/json" \
309 -H "Authorization: Bearer your-key" \
310 -d '{
311 "model": "claude-3-sonnet-20240229",
312 "messages": [{"role": "user", "content": "Hello Claude!"}]
313 }'
314 ```
316 Returns: {"input_tokens": <number>}
317 """
318 from litellm.proxy.proxy_server import token_counter as internal_token_counter
320 litellm_call_id: Final = resolve_litellm_call_id(request.headers.get("x-litellm-call-id"))
321 try:
322 request_data: Final = await _read_request_body(request=request)
323 data: Final[dict] = {**request_data}
325 # Extract required fields
326 model_name: Final = data.get("model")
327 messages: Final = data.get("messages", [])
329 if not model_name: 329 ↛ 332line 329 didn't jump to line 332 because the condition on line 329 was always true
330 raise HTTPException(status_code=400, detail={"error": "model parameter is required"})
332 if not messages:
333 raise HTTPException(status_code=400, detail={"error": "messages parameter is required"})
335 # Create TokenCountRequest for the internal endpoint
336 from litellm.proxy._types import TokenCountRequest
338 token_request: Final = TokenCountRequest(
339 model=model_name,
340 messages=messages,
341 tools=data.get("tools"),
342 system=data.get("system"),
343 )
345 # Call the internal token counter function with direct request flag set to False
346 token_response: Final = await internal_token_counter(
347 request=token_request,
348 call_endpoint=True,
349 )
350 _token_response_dict: dict = {}
351 if isinstance(token_response, TokenCountResponse):
352 _token_response_dict = token_response.model_dump()
353 elif isinstance(token_response, dict):
354 _token_response_dict = token_response
356 # Convert the internal response to Anthropic API format
357 return {"input_tokens": _token_response_dict.get("total_tokens", 0)}
359 except HTTPException:
360 raise
361 except ProxyException as e:
362 status_code: Final = int(e.code) if e.code and e.code.isdigit() else 500
363 detail: Final = AnthropicExceptionMapping.transform_to_anthropic_error(
364 status_code=status_code,
365 raw_message=e.message,
366 )
367 raise HTTPException(
368 status_code=status_code,
369 detail=detail,
370 )
371 except Exception as e:
372 log_llm_api_exception(e, litellm_call_id)
373 raise HTTPException(status_code=500, detail={"error": f"Internal server error: {e}"})
376@router.post(
377 "/api/event_logging/batch",
378 tags=["[beta] Anthropic Event Logging"],
379)
380async def event_logging_batch(
381 request: Request,
382):
383 """
384 Stubbed endpoint for Anthropic event logging batch requests.
386 This endpoint accepts event logging requests but does nothing with them.
387 It exists to prevent 404 errors from Claude Code clients that send telemetry.
388 """
389 return {"status": "ok"}