Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/hooks/rate_limiter_utils.py: 10%
63 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Shared utility functions for rate limiter hooks.
3"""
5from typing import Final
7import litellm
8from litellm._logging import verbose_proxy_logger
9from litellm.constants import PROXY_LLM_PROVIDER_FALLBACK
10from litellm.types.router import ModelGroupInfo
11from litellm.types.utils import PriorityReservationDict
14def resolve_llm_provider_for_rate_limit(
15 model: str | None,
16) -> tuple[str, str]:
17 """
18 Resolve ``(model, llm_provider)`` for a request being rejected by an
19 internal proxy-side rate-limit hook.
21 These hooks fire from ``async_pre_call_hook`` — well before
22 :func:`litellm.get_llm_provider` is invoked anywhere else in the request
23 lifecycle — so the raised 429 would otherwise have an empty
24 ``llm_provider`` field, making the resulting Prometheus
25 ``litellm_proxy_failed_requests_metric`` show up with
26 ``exception_class="RateLimitError"`` and no provider attribution.
28 Resolution order:
30 1. ``litellm.get_llm_provider(model)`` — covers raw provider/model
31 strings the SDK already understands (``"gpt-4o-mini"``,
32 ``"anthropic/claude-3-5-sonnet"``, ``"bedrock/..."`` etc.).
33 2. **Router alias fallback** — nearly every real proxy deployment
34 routes through a router ``model_name`` alias (e.g.
35 ``"tpm-locked"`` → ``litellm_params.model: openai/gpt-4o-mini``).
36 ``get_llm_provider`` doesn't know router aliases, so without this
37 step every alias call ended up labeled ``"litellm_proxy"``,
38 defeating the field's purpose for the most common case.
39 3. Defensive fallback to ``("", "litellm_proxy")`` — used only when
40 ``model`` is missing, malformed, or both lookups fail. We never let
41 a secondary exception escape and mask the rate-limit error we're
42 trying to surface.
43 """
44 if not model:
45 return "", PROXY_LLM_PROVIDER_FALLBACK
46 try:
47 resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
48 model=model,
49 )
50 return (
51 resolved_model or model,
52 custom_llm_provider or PROXY_LLM_PROVIDER_FALLBACK,
53 )
54 except Exception as e:
55 alias_resolution: Final = _resolve_provider_from_router_alias(model)
56 if alias_resolution is not None:
57 return alias_resolution
58 verbose_proxy_logger.debug(
59 "rate_limiter_utils.resolve_llm_provider_for_rate_limit: "
60 "could not resolve provider for model=%s, falling back to %s. err=%s",
61 model,
62 PROXY_LLM_PROVIDER_FALLBACK,
63 str(e),
64 )
65 return model, PROXY_LLM_PROVIDER_FALLBACK
68def _resolve_provider_from_router_alias(
69 model: str,
70) -> tuple[str, str] | None:
71 """
72 Resolve a router ``model_name`` alias to ``(underlying_model, provider)``
73 by scanning the active router's ``model_list``.
75 Returns ``None`` if the router isn't initialized, the alias isn't
76 registered, the deployment has no usable ``litellm_params.model``, or
77 any underlying lookup raises. Callers fall through to the defensive
78 ``litellm_proxy`` fallback in that case — never raising secondary
79 exceptions out of the rate-limit raise path.
80 """
81 try:
82 from litellm.proxy.proxy_server import llm_router
83 except Exception:
84 return None
85 if llm_router is None:
86 return None
87 try:
88 model_list: Final = getattr(llm_router, "model_list", None)
89 if not model_list:
90 return None
91 for deployment in model_list:
92 if not isinstance(deployment, dict):
93 continue
94 if deployment.get("model_name") != model:
95 continue
96 params = deployment.get("litellm_params")
97 if not isinstance(params, dict):
98 continue
99 underlying_model = params.get("model")
100 if not isinstance(underlying_model, str) or not underlying_model:
101 continue
102 try:
103 resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
104 model=underlying_model,
105 )
106 except Exception:
107 continue
108 if not custom_llm_provider:
109 continue
110 # Prefer the underlying provider-qualified model so the failure
111 # callback / Prometheus label points at the actual deployment, not
112 # the alias.
113 return (
114 resolved_model or underlying_model,
115 custom_llm_provider,
116 )
117 return None
118 except Exception:
119 return None
122def convert_priority_to_percent(value: float | PriorityReservationDict, model_info: ModelGroupInfo | None) -> float:
123 """
124 Convert priority reservation value to percentage (0.0-1.0).
126 Supports three formats:
127 1. Plain float/int: 0.9 -> 0.9 (90%)
128 2. Dict with percent: {"type": "percent", "value": 0.9} -> 0.9
129 3. Dict with rpm: {"type": "rpm", "value": 900} -> 900/model_rpm
130 4. Dict with tpm: {"type": "tpm", "value": 900000} -> 900000/model_tpm
132 Args:
133 value: Priority value as float or dict with type/value keys
134 model_info: Model configuration containing rpm/tpm limits
136 Returns:
137 float: Percentage value between 0.0 and 1.0
138 """
139 if isinstance(value, (int, float)):
140 return float(value)
142 if isinstance(value, dict):
143 val_type: Final = value.get("type", "percent")
144 val_num: Final = value.get("value", 1.0)
146 if val_type == "percent":
147 return float(val_num)
148 elif val_type == "rpm" and model_info and model_info.rpm and model_info.rpm > 0:
149 return float(val_num) / model_info.rpm
150 elif val_type == "tpm" and model_info and model_info.tpm and model_info.tpm > 0:
151 return float(val_num) / model_info.tpm
153 # Fallback: treat as percent
154 return float(val_num)