Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/common_utils/proxy_rate_limit_error.py: 17%
42 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2ProxyRateLimitError — a unified rate-limit exception used by litellm's
3proxy-side hooks.
5Background
6----------
7LiteLLM previously surfaced rate-limit conditions through *several* unrelated
8exception types:
10* :class:`litellm.exceptions.RateLimitError` — raised by exception mapping when
11 an upstream LLM provider returns 429.
12* :class:`fastapi.HTTPException` (status 429) — raised directly by proxy hooks
13 such as ``parallel_request_limiter``, ``dynamic_rate_limiter``,
14 ``batch_rate_limiter``, ``max_iterations_limiter``,
15 etc.
16* :class:`litellm.llms.base_llm.chat.transformation.BaseLLMException` (status
17 429) — raised by some provider transports.
19This made it impossible for downstream code (and end users) to express
20"is this a rate limit?" with a single ``except`` clause, and impossible to
21distinguish *where* the rate limit originated (vendor vs. litellm, batch vs.
22chat) without ad-hoc string-matching on the message.
24This module provides a single proxy-side error class that:
261. Is a subclass of :class:`litellm.exceptions.RateLimitError`, so user code
27 that catches ``RateLimitError`` works for *every* rate-limit source.
282. Is also a subclass of :class:`fastapi.HTTPException`, so existing proxy
29 plumbing (``isinstance(e, HTTPException)`` branches in route handlers and
30 FastAPI's own dispatcher) continues to behave the same way and the
31 ``retry-after`` / ``rate_limit_type`` / ``reset_at`` headers are preserved
32 on the wire.
333. Carries a :attr:`category` field (one of
34 :class:`litellm.exceptions.RateLimitErrorCategory`) so callers can switch on
35 the rate limit source.
36"""
38import json
39from collections.abc import Mapping
40from typing import Any, Final
42from fastapi import HTTPException
44from litellm.exceptions import RateLimitError, RateLimitErrorCategory, RateLimitType
47def map_v3_rate_limit_type(
48 v3_value: str | None,
49) -> RateLimitType | None:
50 """
51 Map the v3 rate limiter's internal `status["rate_limit_type"]` strings
52 onto the public :class:`RateLimitType` enum.
54 The v3 limiter uses the literal values ``"requests"``, ``"tokens"``, and
55 ``"max_parallel_requests"``. We collapse the last one onto
56 :attr:`RateLimitType.CONCURRENT_REQUESTS` because that's the public name
57 documented for users and dashboards. Unrecognized values return ``None``
58 so the field stays absent rather than carrying garbage downstream.
59 """
60 if v3_value == "tokens":
61 return RateLimitType.TOKENS
62 if v3_value == "max_parallel_requests":
63 return RateLimitType.CONCURRENT_REQUESTS
64 if v3_value == "requests":
65 return RateLimitType.REQUESTS
66 return None
69def _coerce_message(detail: object) -> str:
70 """Best-effort, JSON-friendly stringification of an HTTPException-style detail."""
71 if detail is None:
72 return ""
73 if isinstance(detail, str):
74 return detail
75 if isinstance(detail, Mapping):
76 for key in ("error", "message"):
77 if isinstance(detail.get(key), str):
78 return detail[key]
79 inner = detail.get(key)
80 if isinstance(inner, Mapping) and isinstance(inner.get("message"), str):
81 return inner["message"]
82 try:
83 return json.dumps(detail)
84 except (TypeError, ValueError):
85 return str(detail)
86 return str(detail)
89# NOTE: mypy emits two `[misc]` errors on the class line below because the
90# bases declare overlapping attributes with related-but-not-identical
91# annotations:
92# * `status_code` is `int` on starlette HTTPException but `Literal[429]` on
93# openai.RateLimitError (every openai status-error subclass narrows it
94# this way and silences pyright with the same convention).
95# * `headers` is `Mapping[str, str] | None` on HTTPException; we narrow it
96# to `Optional[Dict[str, str]]` on RateLimitError because we always carry
97# a stringified dict.
98# Both narrowings are intentional and handled at construction time — every
99# instance always has status_code == 429 and a Dict-typed headers — so we
100# silence the ATTR-overlap check rather than relax the annotations.
101class ProxyRateLimitError(HTTPException, RateLimitError):
102 """
103 A 429 raised by litellm's proxy-side rate limiting hooks.
105 This class deliberately inherits from BOTH
106 :class:`litellm.exceptions.RateLimitError` and :class:`fastapi.HTTPException`
107 so the same instance can flow through:
109 * ``except RateLimitError`` (user / SDK code that wants a category-aware
110 handler), and
111 * ``isinstance(e, HTTPException)`` (FastAPI / proxy_server.py route
112 handlers that need to forward ``status_code``, ``detail`` and
113 ``headers`` back to the client).
115 Downstream code should prefer this class over
116 ``raise HTTPException(status_code=429, ...)`` for litellm-internal rate
117 limits.
119 Parameters
120 ----------
121 detail:
122 The structured error payload. Forwarded as ``HTTPException.detail`` so
123 FastAPI's default exception handler will serialize it verbatim.
124 headers:
125 Optional response headers (e.g. ``retry-after``). Values are stringified
126 to satisfy FastAPI's typing.
127 category:
128 One of :class:`RateLimitErrorCategory`. Defaults to
129 ``LITELLM_RATE_LIMIT`` since this class is only used by litellm's own
130 proxy-side limiters; pass ``LITELLM_BATCH_RATE_LIMIT`` for the batch
131 limiter, etc.
132 model / llm_provider:
133 Optional context, propagated to the inherited ``RateLimitError`` for
134 compatibility with logging / standard payload extraction.
135 """
137 # Prometheus' ``exception_class`` label is pinned to "HTTPException" for
138 # this type: before the unified class existed, proxy-side 429s surfaced as
139 # ``fastapi.HTTPException`` and existing dashboards/alerts key off that exact
140 # value. Distinguishing vendor vs. litellm 429s is now the job of the
141 # ``rate_limit_category`` / ``rate_limit_type`` labels.
142 prometheus_exception_class_name = "HTTPException"
144 def __init__(
145 self,
146 detail: Any,
147 headers: Mapping[str, object] | None = None,
148 category: str | RateLimitErrorCategory = RateLimitErrorCategory.LITELLM_RATE_LIMIT,
149 rate_limit_type: str | RateLimitType | None = None,
150 model: str | None = None,
151 llm_provider: str | None = "litellm_proxy",
152 ):
153 # Normalize None → safe defaults so callers (and the resolver helper
154 # in `rate_limiter_utils`) can pass `None` without producing an
155 # instance whose `.llm_provider` attribute is `None` — that would
156 # break Prometheus' `_get_exception_class_name` (it calls
157 # `.capitalize()` on the provider string).
158 model = model or ""
159 llm_provider = llm_provider or "litellm_proxy"
160 message: Final = _coerce_message(detail)
161 stringified_headers: Final[dict[str, str] | None] = {k: str(v) for k, v in headers.items()} if headers else None
163 # Initialize the FastAPI HTTPException portion first so its attributes
164 # (status_code, detail, headers) are already on the instance before
165 # RateLimitError.__init__ runs and possibly overrides them.
166 HTTPException.__init__(
167 self,
168 status_code=429,
169 detail=detail,
170 headers=stringified_headers,
171 )
173 # Now initialize the litellm RateLimitError portion. We deliberately
174 # pass the structured detail through so RateLimitError preserves it as
175 # its `.detail` attribute too — keeping both sides of the MRO
176 # consistent.
177 RateLimitError.__init__(
178 self,
179 message=message,
180 llm_provider=llm_provider,
181 model=model,
182 category=category,
183 rate_limit_type=rate_limit_type,
184 headers=stringified_headers,
185 detail=detail,
186 )
187 # RateLimitError.__init__ overwrites self.headers with its own copy and
188 # leaves self.status_code at 429 — restore the HTTPException-style
189 # headers value so downstream code that pulls headers off the
190 # instance gets back exactly what the limiter passed in.
191 self.headers = stringified_headers
192 self.detail = detail
193 self.status_code = 429