Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/hooks/batch_redis_get.py: 0%

64 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1# What this does? 

2## Gets a key's redis cache, and store it in memory for 1 minute. 

3## This reduces the number of REDIS GET requests made during high-traffic by the proxy. 

4### [BETA] this is in Beta. And might change. 

5 

6import traceback 

7from typing import Final, Literal 

8 

9from fastapi import HTTPException 

10 

11import litellm 

12from litellm._logging import verbose_proxy_logger 

13from litellm.caching.caching import DualCache, InMemoryCache, RedisCache 

14from litellm.integrations.custom_logger import CustomLogger 

15from litellm.proxy._types import UserAPIKeyAuth 

16 

17 

18class _PROXY_BatchRedisRequests(CustomLogger): 

19 # Class variables or attributes 

20 in_memory_cache: InMemoryCache | None = None 

21 

22 def __init__(self): 

23 if litellm.cache is not None: 

24 litellm.cache.async_get_cache = ( 

25 self.async_get_cache 

26 ) # map the litellm 'get_cache' function to our custom function 

27 

28 def print_verbose(self, print_statement, debug_level: Literal["INFO", "DEBUG"] = "DEBUG"): 

29 if debug_level == "DEBUG" or debug_level == "INFO": 

30 verbose_proxy_logger.debug(print_statement) 

31 if litellm.set_verbose is True: 

32 print(print_statement) # noqa: T201 

33 

34 async def async_pre_call_hook( 

35 self, 

36 user_api_key_dict: UserAPIKeyAuth, 

37 cache: DualCache, 

38 data: dict, 

39 call_type: str, 

40 ): 

41 try: 

42 """ 

43 Get the user key 

44 

45 Check if a key starting with `litellm:<api_key>:<call_type:` exists in-memory 

46 

47 If no, then get relevant cache from redis 

48 """ 

49 api_key: Final = user_api_key_dict.api_key 

50 

51 cache_key_name: Final = f"litellm:{api_key}:{call_type}" 

52 self.in_memory_cache = cache.in_memory_cache 

53 

54 key_value_dict = {} 

55 in_memory_cache_exists = False 

56 for key in cache.in_memory_cache.cache_dict: 

57 if isinstance(key, str) and key.startswith(cache_key_name): 

58 in_memory_cache_exists = True 

59 

60 if in_memory_cache_exists is False and litellm.cache is not None: 

61 """ 

62 - Check if `litellm.Cache` is redis 

63 - Get the relevant values 

64 """ 

65 if litellm.cache.type is not None and isinstance(litellm.cache.cache, RedisCache): 

66 # Initialize an empty list to store the keys 

67 keys = [] 

68 self.print_verbose(f"cache_key_name: {cache_key_name}") 

69 # Use the SCAN iterator to fetch keys matching the pattern 

70 keys = await litellm.cache.cache.async_scan_iter(pattern=cache_key_name, count=100) 

71 # If you need the truly "last" based on time or another criteria, 

72 # ensure your key naming or storage strategy allows this determination 

73 # Here you would sort or filter the keys as needed based on your strategy 

74 self.print_verbose(f"redis keys: {keys}") 

75 if len(keys) > 0: 

76 key_value_dict = await litellm.cache.cache.async_batch_get_cache(key_list=keys) 

77 

78 ## Add to cache 

79 if len(key_value_dict.items()) > 0: 

80 await cache.in_memory_cache.async_set_cache_pipeline(cache_list=list(key_value_dict.items()), ttl=60) 

81 ## Set cache namespace if it's a miss 

82 data["metadata"]["redis_namespace"] = cache_key_name 

83 except HTTPException as e: 

84 raise e 

85 except Exception as e: 

86 verbose_proxy_logger.error( 

87 "litellm.proxy.hooks.batch_redis_get.py::async_pre_call_hook(): Exception occured - %s", e 

88 ) 

89 verbose_proxy_logger.debug(traceback.format_exc()) 

90 

91 async def async_get_cache(self, *args, **kwargs): 

92 """ 

93 - Check if the cache key is in-memory 

94 

95 - Else: 

96 - add missing cache key from REDIS 

97 - update in-memory cache 

98 - return redis cache request 

99 """ 

100 try: # never block execution 

101 cache_key: str | None = None 

102 if "cache_key" in kwargs: 

103 cache_key = kwargs["cache_key"] 

104 elif litellm.cache is not None: 

105 cache_key = litellm.cache.get_cache_key( 

106 *args, **kwargs 

107 ) # returns "<cache_key_name>:<hash>" - we pass redis_namespace in async_pre_call_hook. Done to avoid rewriting the async_set_cache logic 

108 

109 if cache_key is not None and self.in_memory_cache is not None and litellm.cache is not None: 

110 cache_control_args: Final = kwargs.get("cache", {}) 

111 max_age: Final = cache_control_args.get("s-max-age", cache_control_args.get("s-maxage", float("inf"))) 

112 cached_result = self.in_memory_cache.get_cache(cache_key, *args, **kwargs) 

113 if cached_result is None: 

114 cached_result = await litellm.cache.cache.async_get_cache(cache_key, *args, **kwargs) 

115 if cached_result is not None: 

116 await self.in_memory_cache.async_set_cache(cache_key, cached_result, ttl=60) 

117 return litellm.cache._get_cache_logic(cached_result=cached_result, max_age=max_age) 

118 except Exception: 

119 return None