Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/hooks/rate_limiter_utils.py: 10%

63 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Shared utility functions for rate limiter hooks. 

3""" 

4 

5from typing import Final 

6 

7import litellm 

8from litellm._logging import verbose_proxy_logger 

9from litellm.constants import PROXY_LLM_PROVIDER_FALLBACK 

10from litellm.types.router import ModelGroupInfo 

11from litellm.types.utils import PriorityReservationDict 

12 

13 

14def resolve_llm_provider_for_rate_limit( 

15 model: str | None, 

16) -> tuple[str, str]: 

17 """ 

18 Resolve ``(model, llm_provider)`` for a request being rejected by an 

19 internal proxy-side rate-limit hook. 

20 

21 These hooks fire from ``async_pre_call_hook`` — well before 

22 :func:`litellm.get_llm_provider` is invoked anywhere else in the request 

23 lifecycle — so the raised 429 would otherwise have an empty 

24 ``llm_provider`` field, making the resulting Prometheus 

25 ``litellm_proxy_failed_requests_metric`` show up with 

26 ``exception_class="RateLimitError"`` and no provider attribution. 

27 

28 Resolution order: 

29 

30 1. ``litellm.get_llm_provider(model)`` — covers raw provider/model 

31 strings the SDK already understands (``"gpt-4o-mini"``, 

32 ``"anthropic/claude-3-5-sonnet"``, ``"bedrock/..."`` etc.). 

33 2. **Router alias fallback** — nearly every real proxy deployment 

34 routes through a router ``model_name`` alias (e.g. 

35 ``"tpm-locked"`` → ``litellm_params.model: openai/gpt-4o-mini``). 

36 ``get_llm_provider`` doesn't know router aliases, so without this 

37 step every alias call ended up labeled ``"litellm_proxy"``, 

38 defeating the field's purpose for the most common case. 

39 3. Defensive fallback to ``("", "litellm_proxy")`` — used only when 

40 ``model`` is missing, malformed, or both lookups fail. We never let 

41 a secondary exception escape and mask the rate-limit error we're 

42 trying to surface. 

43 """ 

44 if not model: 

45 return "", PROXY_LLM_PROVIDER_FALLBACK 

46 try: 

47 resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider( 

48 model=model, 

49 ) 

50 return ( 

51 resolved_model or model, 

52 custom_llm_provider or PROXY_LLM_PROVIDER_FALLBACK, 

53 ) 

54 except Exception as e: 

55 alias_resolution: Final = _resolve_provider_from_router_alias(model) 

56 if alias_resolution is not None: 

57 return alias_resolution 

58 verbose_proxy_logger.debug( 

59 "rate_limiter_utils.resolve_llm_provider_for_rate_limit: " 

60 "could not resolve provider for model=%s, falling back to %s. err=%s", 

61 model, 

62 PROXY_LLM_PROVIDER_FALLBACK, 

63 str(e), 

64 ) 

65 return model, PROXY_LLM_PROVIDER_FALLBACK 

66 

67 

68def _resolve_provider_from_router_alias( 

69 model: str, 

70) -> tuple[str, str] | None: 

71 """ 

72 Resolve a router ``model_name`` alias to ``(underlying_model, provider)`` 

73 by scanning the active router's ``model_list``. 

74 

75 Returns ``None`` if the router isn't initialized, the alias isn't 

76 registered, the deployment has no usable ``litellm_params.model``, or 

77 any underlying lookup raises. Callers fall through to the defensive 

78 ``litellm_proxy`` fallback in that case — never raising secondary 

79 exceptions out of the rate-limit raise path. 

80 """ 

81 try: 

82 from litellm.proxy.proxy_server import llm_router 

83 except Exception: 

84 return None 

85 if llm_router is None: 

86 return None 

87 try: 

88 model_list: Final = getattr(llm_router, "model_list", None) 

89 if not model_list: 

90 return None 

91 for deployment in model_list: 

92 if not isinstance(deployment, dict): 

93 continue 

94 if deployment.get("model_name") != model: 

95 continue 

96 params = deployment.get("litellm_params") 

97 if not isinstance(params, dict): 

98 continue 

99 underlying_model = params.get("model") 

100 if not isinstance(underlying_model, str) or not underlying_model: 

101 continue 

102 try: 

103 resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider( 

104 model=underlying_model, 

105 ) 

106 except Exception: 

107 continue 

108 if not custom_llm_provider: 

109 continue 

110 # Prefer the underlying provider-qualified model so the failure 

111 # callback / Prometheus label points at the actual deployment, not 

112 # the alias. 

113 return ( 

114 resolved_model or underlying_model, 

115 custom_llm_provider, 

116 ) 

117 return None 

118 except Exception: 

119 return None 

120 

121 

122def convert_priority_to_percent(value: float | PriorityReservationDict, model_info: ModelGroupInfo | None) -> float: 

123 """ 

124 Convert priority reservation value to percentage (0.0-1.0). 

125 

126 Supports three formats: 

127 1. Plain float/int: 0.9 -> 0.9 (90%) 

128 2. Dict with percent: {"type": "percent", "value": 0.9} -> 0.9 

129 3. Dict with rpm: {"type": "rpm", "value": 900} -> 900/model_rpm 

130 4. Dict with tpm: {"type": "tpm", "value": 900000} -> 900000/model_tpm 

131 

132 Args: 

133 value: Priority value as float or dict with type/value keys 

134 model_info: Model configuration containing rpm/tpm limits 

135 

136 Returns: 

137 float: Percentage value between 0.0 and 1.0 

138 """ 

139 if isinstance(value, (int, float)): 

140 return float(value) 

141 

142 if isinstance(value, dict): 

143 val_type: Final = value.get("type", "percent") 

144 val_num: Final = value.get("value", 1.0) 

145 

146 if val_type == "percent": 

147 return float(val_num) 

148 elif val_type == "rpm" and model_info and model_info.rpm and model_info.rpm > 0: 

149 return float(val_num) / model_info.rpm 

150 elif val_type == "tpm" and model_info and model_info.tpm and model_info.tpm > 0: 

151 return float(val_num) / model_info.tpm 

152 

153 # Fallback: treat as percent 

154 return float(val_num)