Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/common_utils/debug_utils.py: 30%

355 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1# Start tracing memory allocations 

2import asyncio 

3import gc 

4import json 

5import os 

6import socket 

7import sys 

8import tracemalloc 

9from collections import Counter 

10from collections.abc import Mapping, Sequence 

11from typing import Annotated, Any, Final, NamedTuple, Protocol, TypedDict 

12 

13from fastapi import APIRouter, Depends, HTTPException, Query 

14from typing_extensions import ReadOnly 

15 

16from litellm import get_secret_str 

17from litellm._logging import verbose_proxy_logger 

18from litellm.constants import PYTHON_GC_THRESHOLD 

19from litellm.litellm_core_utils.bug_report import EnvironmentReport 

20from litellm.proxy._types import UserAPIKeyAuth 

21from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

22from litellm.proxy.bug_report_config import build_proxy_environment_report 

23from litellm.proxy.common_utils.resource_ownership import is_proxy_admin 

24 

25router: Final = APIRouter() 

26 

27 

28# Configure garbage collection thresholds from environment variables 

29def configure_gc_thresholds(): 

30 """Configure Python garbage collection thresholds from environment variables.""" 

31 gc_threshold_env: Final = PYTHON_GC_THRESHOLD 

32 if gc_threshold_env: 32 ↛ 33line 32 didn't jump to line 33 because the condition on line 32 was never true

33 try: 

34 # Parse threshold string like "1000,50,50" 

35 thresholds: Final = [int(x.strip()) for x in gc_threshold_env.split(",")] 

36 if len(thresholds) == 3: 

37 gc.set_threshold(*thresholds) 

38 verbose_proxy_logger.info("GC thresholds set to: %s", thresholds) 

39 else: 

40 verbose_proxy_logger.warning( 

41 "GC threshold not set: %s. Expected format: 'gen0,gen1,gen2'", gc_threshold_env 

42 ) 

43 except ValueError as e: 

44 verbose_proxy_logger.warning("Failed to parse GC threshold: %s. Error: %s", gc_threshold_env, e) 

45 

46 # Log current thresholds 

47 current_thresholds: Final = gc.get_threshold() 

48 verbose_proxy_logger.info( 

49 "Current GC thresholds: gen0=%s, gen1=%s, gen2=%s", 

50 current_thresholds[0], 

51 current_thresholds[1], 

52 current_thresholds[2], 

53 ) 

54 

55 

56# Initialize GC configuration 

57configure_gc_thresholds() 

58 

59 

60@router.get( 

61 "/debug/asyncio-tasks", 

62 dependencies=[Depends(user_api_key_auth)], 

63) 

64async def get_active_tasks_stats(): 

65 """ 

66 Returns: 

67 total_active_tasks: int 

68 by_name: { coroutine_name: count } 

69 """ 

70 MAX_TASKS_TO_CHECK: Final = 5000 

71 # Gather all tasks in this event loop (including this endpoint’s own task). 

72 all_tasks: Final = asyncio.all_tasks() 

73 

74 # Filter out tasks that are already done. 

75 active_tasks: Final = [t for t in all_tasks if not t.done()] 

76 

77 # Count how many active tasks exist, grouped by coroutine function name. 

78 counter: Final = Counter() 

79 for idx, task in enumerate(active_tasks): 

80 # reasonable max circuit breaker 

81 if idx >= MAX_TASKS_TO_CHECK: 81 ↛ 82line 81 didn't jump to line 82 because the condition on line 81 was never true

82 break 

83 coro = task.get_coro() 

84 # Derive a human‐readable name from the coroutine: 

85 name = getattr(coro, "__qualname__", None) or getattr(coro, "__name__", None) or repr(coro) 

86 counter[name] += 1 

87 

88 return { 

89 "total_active_tasks": len(active_tasks), 

90 "by_name": dict(counter), 

91 } 

92 

93 

94if os.environ.get("LITELLM_PROFILE", "false").lower() == "true": 94 ↛ 95line 94 didn't jump to line 95 because the condition on line 94 was never true

95 try: 

96 import objgraph 

97 

98 print("growth of objects") # noqa: T201 

99 objgraph.show_growth() 

100 print("\n\nMost common types") # noqa: T201 

101 objgraph.show_most_common_types() 

102 roots: Final = objgraph.get_leaking_objects() 

103 print("\n\nLeaking objects") # noqa: T201 

104 objgraph.show_most_common_types(objects=roots) 

105 except ImportError: 

106 raise ImportError("objgraph not found. Please install objgraph to use this feature.") 

107 

108 tracemalloc.start(10) 

109 

110 @router.get( 

111 "/memory-usage", 

112 dependencies=[Depends(user_api_key_auth)], 

113 include_in_schema=False, 

114 ) 

115 async def memory_usage(): 

116 # Take a snapshot of the current memory usage 

117 snapshot: Final = tracemalloc.take_snapshot() 

118 top_stats: Final = snapshot.statistics("lineno") 

119 verbose_proxy_logger.debug("TOP STATS: %s", top_stats) 

120 

121 # Get the top 50 memory usage lines 

122 top_50: Final = top_stats[:50] 

123 result: Final = [] 

124 for stat in top_50: 

125 result.append(f"{stat.traceback.format(limit=10)}: {stat.size / 1024} KiB") 

126 

127 return {"top_50_memory_usage": result} 

128 

129 

130@router.get("/memory-usage-in-mem-cache", include_in_schema=False) 

131async def memory_usage_in_mem_cache( 

132 _: UserAPIKeyAuth = Depends(user_api_key_auth), 

133): 

134 # returns the size of all in-memory caches on the proxy server 

135 """ 

136 1. user_api_key_cache 

137 2. router_cache 

138 3. proxy_logging_cache 

139 4. internal_usage_cache 

140 """ 

141 from litellm.proxy.proxy_server import ( 

142 llm_router, 

143 proxy_logging_obj, 

144 user_api_key_cache, 

145 ) 

146 

147 if llm_router is None: 

148 num_items_in_llm_router_cache = 0 

149 else: 

150 num_items_in_llm_router_cache = len(llm_router.cache.in_memory_cache.cache_dict) + len( 

151 llm_router.cache.in_memory_cache.ttl_dict 

152 ) 

153 

154 num_items_in_user_api_key_cache: Final = ( 

155 len(user_api_key_cache.in_memory_cache.cache_dict) 

156 + len(user_api_key_cache.in_memory_cache.ttl_dict) 

157 + len(user_api_key_cache.key_object_cache.in_memory_cache.cache_dict) 

158 + len(user_api_key_cache.key_object_cache.in_memory_cache.ttl_dict) 

159 ) 

160 

161 num_items_in_proxy_logging_obj_cache: Final = len( 

162 proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict 

163 ) + len(proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.ttl_dict) 

164 

165 return { 

166 "num_items_in_user_api_key_cache": num_items_in_user_api_key_cache, 

167 "num_items_in_llm_router_cache": num_items_in_llm_router_cache, 

168 "num_items_in_proxy_logging_obj_cache": num_items_in_proxy_logging_obj_cache, 

169 } 

170 

171 

172@router.get("/memory-usage-in-mem-cache-items", include_in_schema=False) 

173async def memory_usage_in_mem_cache_items( 

174 _: UserAPIKeyAuth = Depends(user_api_key_auth), 

175): 

176 # returns the size of all in-memory caches on the proxy server 

177 """ 

178 1. user_api_key_cache 

179 2. router_cache 

180 3. proxy_logging_cache 

181 4. internal_usage_cache 

182 """ 

183 from litellm.proxy.proxy_server import ( 

184 llm_router, 

185 proxy_logging_obj, 

186 user_api_key_cache, 

187 ) 

188 

189 if llm_router is None: 

190 llm_router_in_memory_cache_dict = {} 

191 llm_router_in_memory_ttl_dict = {} 

192 else: 

193 llm_router_in_memory_cache_dict = llm_router.cache.in_memory_cache.cache_dict 

194 llm_router_in_memory_ttl_dict = llm_router.cache.in_memory_cache.ttl_dict 

195 

196 return { 

197 "user_api_key_cache": user_api_key_cache.in_memory_cache.cache_dict, 

198 "user_api_key_ttl": user_api_key_cache.in_memory_cache.ttl_dict, 

199 "user_key_object_cache": user_api_key_cache.key_object_cache.in_memory_cache.cache_dict, 

200 "user_key_object_ttl": user_api_key_cache.key_object_cache.in_memory_cache.ttl_dict, 

201 "llm_router_cache": llm_router_in_memory_cache_dict, 

202 "llm_router_ttl": llm_router_in_memory_ttl_dict, 

203 "proxy_logging_obj_cache": proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict, 

204 "proxy_logging_obj_ttl": proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.ttl_dict, 

205 } 

206 

207 

208class _ProcessMemoryInfo(Protocol): 

209 """The resident and virtual sizes psutil reports for a process.""" 

210 

211 @property 

212 def rss(self) -> int: ... 212 ↛ exitline 212 didn't return from function 'rss' because

213 

214 @property 

215 def vms(self) -> int: ... 215 ↛ exitline 215 didn't return from function 'vms' because

216 

217 

218class _ProcessHandle(Protocol): 

219 """The psutil process handle members this module reads.""" 

220 

221 def memory_info(self) -> _ProcessMemoryInfo: ... 221 ↛ exitline 221 didn't return from function 'memory_info' because

222 

223 def memory_percent(self) -> float: ... 223 ↛ exitline 223 didn't return from function 'memory_percent' because

224 

225 

226class _ProcessMemoryUsage(NamedTuple): 

227 """Memory usage of a single worker process.""" 

228 

229 resident_megabytes: float 

230 virtual_megabytes: float 

231 percent: float 

232 

233 

234def _process_memory_usage(process: _ProcessHandle) -> _ProcessMemoryUsage: 

235 """Read resident/virtual megabytes and system memory share for ``process``.""" 

236 memory_info: Final = process.memory_info() 

237 return _ProcessMemoryUsage( 

238 resident_megabytes=memory_info.rss / (1024 * 1024), 

239 virtual_megabytes=memory_info.vms / (1024 * 1024), 

240 percent=process.memory_percent(), 

241 ) 

242 

243 

244PROC_STATM_PATH: Final = "/proc/self/statm" 

245PROC_MEMINFO_PATH: Final = "/proc/meminfo" 

246PSUTIL_MISSING_ERROR: Final = "Install psutil for memory monitoring: pip install psutil" 

247 

248 

249class _ProcMemoryInfo(NamedTuple): 

250 rss: int 

251 vms: int 

252 

253 

254class _ProcFilesystemProcess: 

255 """Memory of the running process read from the Linux proc filesystem, for images without psutil.""" 

256 

257 def __init__( 

258 self, 

259 statm_path: str = PROC_STATM_PATH, 

260 meminfo_path: str = PROC_MEMINFO_PATH, 

261 page_size: int | None = None, 

262 ) -> None: 

263 self._statm_path: Final = statm_path 

264 self._meminfo_path: Final = meminfo_path 

265 self._page_size: Final = os.sysconf("SC_PAGE_SIZE") if page_size is None else page_size 

266 

267 def memory_info(self) -> _ProcMemoryInfo: 

268 with open(self._statm_path, encoding="ascii") as statm: 

269 size_pages, resident_pages = statm.read().split()[:2] 

270 return _ProcMemoryInfo(rss=int(resident_pages) * self._page_size, vms=int(size_pages) * self._page_size) 

271 

272 def memory_percent(self) -> float: 

273 with open(self._meminfo_path, encoding="ascii") as meminfo: 

274 total_kilobytes: Final = next(int(line.split()[1]) for line in meminfo if line.startswith("MemTotal:")) 

275 return self.memory_info().rss / (total_kilobytes * 1024) * 100 

276 

277 

278def _process_handle() -> _ProcessHandle | None: 

279 try: 

280 import psutil 

281 except ImportError: 

282 return _ProcFilesystemProcess() if os.path.exists(PROC_STATM_PATH) else None 

283 return psutil.Process() 

284 

285 

286def _health_status(memory_percent: float) -> str: 

287 if memory_percent > 80: 

288 return "critical" 

289 if memory_percent > 60: 

290 return "warning" 

291 return "healthy" 

292 

293 

294class _SummaryProcessMemory(TypedDict, total=False): 

295 summary: ReadOnly[str] 

296 ram_usage_mb: ReadOnly[float] 

297 system_memory_percent: ReadOnly[float] 

298 error: ReadOnly[str] 

299 

300 

301def _summary_process_memory(process: _ProcessHandle | None) -> tuple[_SummaryProcessMemory, str]: 

302 if process is None: 

303 missing: Final[_SummaryProcessMemory] = {"error": PSUTIL_MISSING_ERROR} 

304 return missing, "healthy" 

305 try: 

306 usage: Final = _process_memory_usage(process) 

307 except Exception as e: 

308 unreadable: Final[_SummaryProcessMemory] = {"error": str(e)} 

309 return unreadable, "healthy" 

310 memory: Final[_SummaryProcessMemory] = { 

311 "summary": f"{usage.resident_megabytes:.1f} MB ({usage.percent:.1f}% of system memory)", 

312 "ram_usage_mb": round(usage.resident_megabytes, 2), 

313 "system_memory_percent": round(usage.percent, 2), 

314 } 

315 return memory, _health_status(usage.percent) 

316 

317 

318@router.get("/debug/memory/summary", include_in_schema=False) 

319async def get_memory_summary( 

320 _: UserAPIKeyAuth = Depends(user_api_key_auth), 

321) -> dict[str, Any]: 

322 """ 

323 Get simplified memory usage summary for the proxy. 

324 

325 Returns: 

326 - worker_pid: Process ID 

327 - hostname: Host (the pod on Kubernetes) the worker runs on 

328 - status: Overall health based on memory usage 

329 - memory: Process memory usage and RAM info 

330 - caches: Cache item counts and descriptions 

331 - garbage_collector: GC status and pending object counts 

332 

333 Example usage: 

334 curl http://localhost:4000/debug/memory/summary -H "Authorization: Bearer sk-1234" 

335 

336 For detailed analysis, call GET /debug/memory/details 

337 For cache management, use the cache management endpoints 

338 """ 

339 from litellm.proxy.proxy_server import ( 

340 llm_router, 

341 proxy_logging_obj, 

342 user_api_key_cache, 

343 ) 

344 

345 process_memory, health_status = _summary_process_memory(_process_handle()) 

346 

347 # Get cache information 

348 caches: Final[dict[str, object]] = {} 

349 total_cache_items = 0 

350 

351 try: 

352 # User API key cache 

353 user_cache_items: Final = len(user_api_key_cache.in_memory_cache.cache_dict) + len( 

354 user_api_key_cache.key_object_cache.in_memory_cache.cache_dict 

355 ) 

356 total_cache_items += user_cache_items 

357 caches["user_api_keys"] = { 

358 "count": user_cache_items, 

359 "count_readable": f"{user_cache_items:,}", 

360 "what_it_stores": "Validated API keys for faster authentication", 

361 } 

362 

363 # Router cache 

364 if llm_router is not None: 

365 router_cache_items: Final = len(llm_router.cache.in_memory_cache.cache_dict) 

366 total_cache_items += router_cache_items 

367 caches["llm_responses"] = { 

368 "count": router_cache_items, 

369 "count_readable": f"{router_cache_items:,}", 

370 "what_it_stores": "LLM responses for identical requests", 

371 } 

372 

373 # Proxy logging cache 

374 logging_cache_items: Final = len(proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict) 

375 total_cache_items += logging_cache_items 

376 caches["usage_tracking"] = { 

377 "count": logging_cache_items, 

378 "count_readable": f"{logging_cache_items:,}", 

379 "what_it_stores": "Usage metrics before database write", 

380 } 

381 

382 except Exception as e: 

383 caches["error"] = str(e) 

384 

385 # Get garbage collector stats 

386 gc_enabled: Final = gc.isenabled() 

387 objects_pending: Final = gc.get_count()[0] 

388 uncollectable: Final = len(gc.garbage) 

389 

390 gc_info: Final = { 

391 "status": "enabled" if gc_enabled else "disabled", 

392 "objects_awaiting_collection": objects_pending, 

393 } 

394 

395 # Add warning if garbage collection issues detected 

396 if uncollectable > 0: 

397 gc_info["warning"] = f"{uncollectable} uncollectable objects (possible memory leak)" 

398 

399 return { 

400 "worker_pid": os.getpid(), 

401 "hostname": socket.gethostname(), 

402 "status": health_status, 

403 "memory": process_memory, 

404 "caches": { 

405 "total_items": total_cache_items, 

406 "breakdown": caches, 

407 }, 

408 "garbage_collector": gc_info, 

409 } 

410 

411 

412def _get_gc_statistics() -> Mapping[str, object]: 

413 """Get garbage collector statistics.""" 

414 return { 

415 "enabled": gc.isenabled(), 

416 "thresholds": { 

417 "generation_0": gc.get_threshold()[0], 

418 "generation_1": gc.get_threshold()[1], 

419 "generation_2": gc.get_threshold()[2], 

420 "explanation": "Number of allocations before automatic collection for each generation", 

421 }, 

422 "current_counts": { 

423 "generation_0": gc.get_count()[0], 

424 "generation_1": gc.get_count()[1], 

425 "generation_2": gc.get_count()[2], 

426 "explanation": "Current number of allocated objects in each generation", 

427 }, 

428 "collection_history": [ 

429 { 

430 "generation": i, 

431 "total_collections": stat["collections"], 

432 "total_collected": stat["collected"], 

433 "uncollectable": stat["uncollectable"], 

434 } 

435 for i, stat in enumerate(gc.get_stats()) 

436 ], 

437 } 

438 

439 

440class _ObjectTypeCount(TypedDict): 

441 """One row of the tracked-object histogram.""" 

442 

443 type: ReadOnly[str] 

444 count: ReadOnly[int] 

445 count_readable: ReadOnly[str] 

446 

447 

448def _type_name_counts(objects: Sequence[object]) -> Counter[str]: 

449 """Count ``objects`` by the name of their type.""" 

450 return Counter(type(obj).__name__ for obj in objects) 

451 

452 

453def _get_object_type_counts(top_n: int) -> tuple[int, list[_ObjectTypeCount]]: 

454 """Count objects by type and return total count and top N types.""" 

455 type_counts: Final = _type_name_counts(gc.get_objects()) 

456 

457 top_object_types: Final[list[_ObjectTypeCount]] = [ 

458 {"type": obj_type, "count": count, "count_readable": f"{count:,}"} 

459 for obj_type, count in type_counts.most_common(top_n) 

460 ] 

461 

462 return sum(type_counts.values()), top_object_types 

463 

464 

465def _type_names(objects: Sequence[object]) -> Sequence[str]: 

466 """The type name of each object in ``objects``.""" 

467 return [type(obj).__name__ for obj in objects] 

468 

469 

470def _get_uncollectable_objects_info() -> Mapping[str, object]: 

471 """Get information about uncollectable objects (potential memory leaks).""" 

472 uncollectable: Final = gc.garbage 

473 return { 

474 "count": len(uncollectable), 

475 "sample_types": _type_names(uncollectable[:10]), 

476 "warning": ( 

477 "If count > 0, you may have reference cycles preventing garbage collection" 

478 if len(uncollectable) > 0 

479 else None 

480 ), 

481 } 

482 

483 

484def _get_cache_memory_stats( 

485 user_api_key_cache, llm_router, proxy_logging_obj, redis_usage_cache 

486) -> Mapping[str, object]: 

487 """Calculate memory usage for all caches.""" 

488 cache_stats: Final[dict[str, object]] = {} 

489 try: 

490 # User API key cache 

491 key_object_in_memory_cache: Final = user_api_key_cache.key_object_cache.in_memory_cache 

492 user_cache_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.cache_dict) + sys.getsizeof( 

493 key_object_in_memory_cache.cache_dict 

494 ) 

495 user_ttl_size: Final = sys.getsizeof(user_api_key_cache.in_memory_cache.ttl_dict) + sys.getsizeof( 

496 key_object_in_memory_cache.ttl_dict 

497 ) 

498 cache_stats["user_api_key_cache"] = { 

499 "num_items": len(user_api_key_cache.in_memory_cache.cache_dict) 

500 + len(key_object_in_memory_cache.cache_dict), 

501 "cache_dict_size_bytes": user_cache_size, 

502 "ttl_dict_size_bytes": user_ttl_size, 

503 "total_size_mb": round((user_cache_size + user_ttl_size) / (1024 * 1024), 2), 

504 } 

505 

506 # Router cache 

507 if llm_router is not None: 

508 router_cache_size: Final = sys.getsizeof(llm_router.cache.in_memory_cache.cache_dict) 

509 router_ttl_size: Final = sys.getsizeof(llm_router.cache.in_memory_cache.ttl_dict) 

510 cache_stats["llm_router_cache"] = { 

511 "num_items": len(llm_router.cache.in_memory_cache.cache_dict), 

512 "cache_dict_size_bytes": router_cache_size, 

513 "ttl_dict_size_bytes": router_ttl_size, 

514 "total_size_mb": round((router_cache_size + router_ttl_size) / (1024 * 1024), 2), 

515 } 

516 

517 # Proxy logging cache 

518 logging_cache_size = sys.getsizeof(proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict) 

519 logging_ttl_size = sys.getsizeof(proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.ttl_dict) 

520 cache_stats["proxy_logging_cache"] = { 

521 "num_items": len(proxy_logging_obj.internal_usage_cache.dual_cache.in_memory_cache.cache_dict), 

522 "cache_dict_size_bytes": logging_cache_size, 

523 "ttl_dict_size_bytes": logging_ttl_size, 

524 "total_size_mb": round((logging_cache_size + logging_ttl_size) / (1024 * 1024), 2), 

525 } 

526 

527 # Redis cache info 

528 if redis_usage_cache is not None: 

529 cache_stats["redis_usage_cache"] = { 

530 "enabled": True, 

531 "cache_type": type(redis_usage_cache).__name__, 

532 } 

533 # Try to get Redis connection pool info if available 

534 try: 

535 if hasattr(redis_usage_cache, "redis_client") and redis_usage_cache.redis_client: 

536 if hasattr(redis_usage_cache.redis_client, "connection_pool"): 

537 pool_info: Final = redis_usage_cache.redis_client.connection_pool 

538 cache_stats["redis_usage_cache"]["connection_pool"] = { 

539 "max_connections": ( 

540 pool_info.max_connections if hasattr(pool_info, "max_connections") else None 

541 ), 

542 "connection_class": ( 

543 pool_info.connection_class.__name__ if hasattr(pool_info, "connection_class") else None 

544 ), 

545 } 

546 except Exception as e: 

547 verbose_proxy_logger.debug("Error getting Redis pool info: %s", e) 

548 else: 

549 cache_stats["redis_usage_cache"] = {"enabled": False} 

550 

551 except Exception as e: 

552 verbose_proxy_logger.debug("Error calculating cache stats: %s", e) 

553 cache_stats["error"] = str(e) 

554 

555 return cache_stats 

556 

557 

558def _get_router_memory_stats(llm_router) -> Mapping[str, object]: 

559 """Get memory usage statistics for LiteLLM router.""" 

560 litellm_router_memory: dict[str, object] = {} 

561 try: 

562 if llm_router is not None: 

563 # Model list memory size 

564 if hasattr(llm_router, "model_list") and llm_router.model_list: 

565 model_list_size: Final = sys.getsizeof(llm_router.model_list) 

566 litellm_router_memory["model_list"] = { 

567 "num_models": len(llm_router.model_list), 

568 "size_bytes": model_list_size, 

569 "size_mb": round(model_list_size / (1024 * 1024), 4), 

570 } 

571 

572 # Model names set 

573 if hasattr(llm_router, "model_names") and llm_router.model_names: 

574 model_names_size: Final = sys.getsizeof(llm_router.model_names) 

575 litellm_router_memory["model_names_set"] = { 

576 "num_model_groups": len(llm_router.model_names), 

577 "size_bytes": model_names_size, 

578 "size_mb": round(model_names_size / (1024 * 1024), 4), 

579 } 

580 

581 # Deployment names list 

582 if hasattr(llm_router, "deployment_names") and llm_router.deployment_names: 

583 deployment_names_size: Final = sys.getsizeof(llm_router.deployment_names) 

584 litellm_router_memory["deployment_names"] = { 

585 "num_deployments": len(llm_router.deployment_names), 

586 "size_bytes": deployment_names_size, 

587 "size_mb": round(deployment_names_size / (1024 * 1024), 4), 

588 } 

589 

590 # Deployment latency map 

591 if hasattr(llm_router, "deployment_latency_map") and llm_router.deployment_latency_map: 

592 latency_map_size: Final = sys.getsizeof(llm_router.deployment_latency_map) 

593 litellm_router_memory["deployment_latency_map"] = { 

594 "num_tracked_deployments": len(llm_router.deployment_latency_map), 

595 "size_bytes": latency_map_size, 

596 "size_mb": round(latency_map_size / (1024 * 1024), 4), 

597 } 

598 

599 # Fallback configuration 

600 if hasattr(llm_router, "fallbacks") and llm_router.fallbacks: 

601 fallbacks_size: Final = sys.getsizeof(llm_router.fallbacks) 

602 litellm_router_memory["fallbacks"] = { 

603 "num_fallback_configs": len(llm_router.fallbacks), 

604 "size_bytes": fallbacks_size, 

605 "size_mb": round(fallbacks_size / (1024 * 1024), 4), 

606 } 

607 

608 # Total router object size 

609 router_obj_size: Final = sys.getsizeof(llm_router) 

610 litellm_router_memory["router_object"] = { 

611 "size_bytes": router_obj_size, 

612 "size_mb": round(router_obj_size / (1024 * 1024), 4), 

613 } 

614 

615 else: 

616 litellm_router_memory = {"note": "Router not initialized"} 

617 except Exception as e: 

618 verbose_proxy_logger.debug("Error getting router memory info: %s", e) 

619 litellm_router_memory = {"error": str(e)} 

620 

621 return litellm_router_memory 

622 

623 

624def _get_process_memory_info(worker_pid: int, include_process_info: bool) -> Mapping[str, object] | None: 

625 """Get process-level memory information using psutil.""" 

626 if not include_process_info: 

627 return None 

628 

629 try: 

630 import psutil 

631 

632 process: Final = psutil.Process() 

633 usage: Final = _process_memory_usage(process) 

634 ram_usage_mb: Final = round(usage.resident_megabytes, 2) 

635 virtual_memory_mb: Final = round(usage.virtual_megabytes, 2) 

636 memory_percent: Final = round(usage.percent, 2) 

637 

638 return { 

639 "pid": worker_pid, 

640 "summary": f"Worker PID {worker_pid} using {ram_usage_mb:.1f} MB of RAM ({memory_percent:.1f}% of system memory)", 

641 "ram_usage": { 

642 "megabytes": ram_usage_mb, 

643 "description": "Actual physical RAM used by this process", 

644 }, 

645 "virtual_memory": { 

646 "megabytes": virtual_memory_mb, 

647 "description": "Total virtual memory allocated (includes swapped memory)", 

648 }, 

649 "system_memory_percent": { 

650 "percent": memory_percent, 

651 "description": "Percentage of total system RAM being used", 

652 }, 

653 "open_file_handles": { 

654 "count": (process.num_fds() if hasattr(process, "num_fds") else "N/A (Windows)"), 

655 "description": "Number of open file descriptors/handles", 

656 }, 

657 "threads": { 

658 "count": process.num_threads(), 

659 "description": "Number of active threads in this process", 

660 }, 

661 } 

662 except ImportError: 

663 return { 

664 "pid": worker_pid, 

665 "error": "psutil not installed. Install with: pip install psutil", 

666 } 

667 except Exception as e: 

668 verbose_proxy_logger.debug("Error getting process info: %s", e) 

669 return {"pid": worker_pid, "error": str(e)} 

670 

671 

672@router.get("/debug/memory/details", include_in_schema=False) 

673async def get_memory_details( 

674 _: UserAPIKeyAuth = Depends(user_api_key_auth), 

675 top_n: int = Query(20, description="Number of top object types to return"), 

676 include_process_info: bool = Query(True, description="Include process memory info"), 

677) -> dict[str, Any]: 

678 """ 

679 Get detailed memory diagnostics for deep debugging. 

680 

681 Returns: 

682 - worker_pid: Process ID 

683 - process_memory: RAM usage, virtual memory, file handles, threads 

684 - garbage_collector: GC thresholds, counts, collection history 

685 - objects: Total tracked objects and top object types 

686 - uncollectable: Objects that can't be garbage collected (potential leaks) 

687 - cache_memory: Memory usage of user_api_key, router, and logging caches 

688 - router_memory: Memory usage of router components (model_list, deployment_names, etc.) 

689 

690 Query Parameters: 

691 - top_n: Number of top object types to return (default: 20) 

692 - include_process_info: Include process-level memory info using psutil (default: true) 

693 

694 Example usage: 

695 curl "http://localhost:4000/debug/memory/details?top_n=30" -H "Authorization: Bearer sk-1234" 

696 

697 All memory sizes are reported in both bytes and MB. 

698 """ 

699 from litellm.proxy.proxy_server import ( 

700 llm_router, 

701 proxy_logging_obj, 

702 redis_usage_cache, 

703 user_api_key_cache, 

704 ) 

705 

706 worker_pid: Final = os.getpid() 

707 

708 # Collect all diagnostics using helper functions 

709 gc_stats: Final = _get_gc_statistics() 

710 total_objects, top_object_types = _get_object_type_counts(top_n) 

711 uncollectable_info: Final = _get_uncollectable_objects_info() 

712 cache_stats: Final = _get_cache_memory_stats(user_api_key_cache, llm_router, proxy_logging_obj, redis_usage_cache) 

713 litellm_router_memory: Final = _get_router_memory_stats(llm_router) 

714 process_info: Final = _get_process_memory_info(worker_pid, include_process_info) 

715 

716 return { 

717 "worker_pid": worker_pid, 

718 "process_memory": process_info, 

719 "garbage_collector": gc_stats, 

720 "objects": { 

721 "total_tracked": total_objects, 

722 "total_tracked_readable": f"{total_objects:,}", 

723 "top_types": top_object_types, 

724 }, 

725 "uncollectable": uncollectable_info, 

726 "cache_memory": cache_stats, 

727 "router_memory": litellm_router_memory, 

728 } 

729 

730 

731@router.post("/debug/memory/gc/configure", include_in_schema=False) 

732async def configure_gc_thresholds_endpoint( 

733 _: UserAPIKeyAuth = Depends(user_api_key_auth), 

734 generation_0: int = Query(700, description="Generation 0 threshold (default: 700)"), 

735 generation_1: int = Query(10, description="Generation 1 threshold (default: 10)"), 

736 generation_2: int = Query(10, description="Generation 2 threshold (default: 10)"), 

737) -> dict[str, Any]: 

738 """ 

739 Configure Python garbage collection thresholds. 

740 

741 Lower thresholds mean more frequent GC cycles (less memory, more CPU overhead). 

742 Higher thresholds mean less frequent GC cycles (more memory, less CPU overhead). 

743 

744 Returns: 

745 - message: Confirmation message 

746 - previous_thresholds: Old threshold values 

747 - new_thresholds: New threshold values 

748 - objects_awaiting_collection: Current object count in gen-0 

749 - tip: Hint about when next collection will occur 

750 

751 Query Parameters: 

752 - generation_0: Number of allocations before gen-0 collection (default: 700) 

753 - generation_1: Number of gen-0 collections before gen-1 collection (default: 10) 

754 - generation_2: Number of gen-1 collections before gen-2 collection (default: 10) 

755 

756 Example for more aggressive collection: 

757 curl -X POST "http://localhost:4000/debug/memory/gc/configure?generation_0=500" -H "Authorization: Bearer sk-1234" 

758 

759 Example for less aggressive collection: 

760 curl -X POST "http://localhost:4000/debug/memory/gc/configure?generation_0=1000" -H "Authorization: Bearer sk-1234" 

761 

762 Monitor memory usage with GET /debug/memory/summary after changes. 

763 """ 

764 # Get current thresholds for logging 

765 old_thresholds: Final = gc.get_threshold() 

766 

767 # Set new thresholds with error handling 

768 try: 

769 gc.set_threshold(generation_0, generation_1, generation_2) 

770 verbose_proxy_logger.info( 

771 "GC thresholds updated from %s to (%s, %s, %s)", old_thresholds, generation_0, generation_1, generation_2 

772 ) 

773 except Exception as e: 

774 verbose_proxy_logger.error("Failed to set GC thresholds: %s", e) 

775 raise HTTPException(status_code=500, detail=f"Failed to set GC thresholds: {e}") 

776 

777 # Get current object count to show immediate impact 

778 current_count: Final = gc.get_count()[0] 

779 

780 return { 

781 "message": "GC thresholds updated", 

782 "previous_thresholds": f"{old_thresholds[0]}, {old_thresholds[1]}, {old_thresholds[2]}", 

783 "new_thresholds": f"{generation_0}, {generation_1}, {generation_2}", 

784 "objects_awaiting_collection": current_count, 

785 "tip": f"Next collection will run after {generation_0 - current_count} more allocations", 

786 } 

787 

788 

789@router.get("/debug/report", include_in_schema=False) 

790async def get_debug_report( 

791 user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], 

792) -> EnvironmentReport: 

793 """ 

794 The same LiteLLM-owned environment facts the bug report link puts in a GitHub issue: 

795 versions, deployment kind, and config flags whose keys and values LiteLLM defines. 

796 Nothing from the operator's config values, request data, or errors 

797 

798 Example usage: 

799 curl http://localhost:4000/debug/report -H "Authorization: Bearer sk-1234" 

800 """ 

801 if not is_proxy_admin(user_api_key_dict): 

802 raise HTTPException(status_code=403, detail="Only proxy admins can read /debug/report") 

803 return build_proxy_environment_report() 

804 

805 

806@router.get( 

807 "/otel-spans", 

808 dependencies=[Depends(user_api_key_auth)], 

809 include_in_schema=False, 

810) 

811async def get_otel_spans(): 

812 from litellm.proxy.proxy_server import open_telemetry_logger 

813 

814 if open_telemetry_logger is None: 

815 return { 

816 "otel_spans": [], 

817 "spans_grouped_by_parent": {}, 

818 "most_recent_parent": None, 

819 } 

820 

821 otel_exporter: Final = open_telemetry_logger.OTEL_EXPORTER 

822 if hasattr(otel_exporter, "get_finished_spans"): 

823 recorded_spans = otel_exporter.get_finished_spans() 

824 else: 

825 recorded_spans = [] 

826 

827 print("Spans: ", recorded_spans) # noqa: T201 

828 

829 most_recent_parent = None 

830 most_recent_start_time = 1000000 

831 spans_grouped_by_parent: Final = {} 

832 for span in recorded_spans: 

833 if span.parent is not None: 

834 parent_trace_id = span.parent.trace_id 

835 if parent_trace_id not in spans_grouped_by_parent: 

836 spans_grouped_by_parent[parent_trace_id] = [] 

837 spans_grouped_by_parent[parent_trace_id].append(span.name) 

838 

839 # check time of span 

840 if span.start_time > most_recent_start_time: 

841 most_recent_parent = parent_trace_id 

842 most_recent_start_time = span.start_time 

843 

844 # these are otel spans - get the span name 

845 span_names: Final = [span.name for span in recorded_spans] 

846 return { 

847 "otel_spans": span_names, 

848 "spans_grouped_by_parent": spans_grouped_by_parent, 

849 "most_recent_parent": most_recent_parent, 

850 } 

851 

852 

853# Helper functions for debugging 

854def init_verbose_loggers(): 

855 try: 

856 worker_config: Final = get_secret_str("WORKER_CONFIG") 

857 # if not, assume it's a json string 

858 if worker_config is None: 858 ↛ 859line 858 didn't jump to line 859 because the condition on line 858 was never true

859 return 

860 if os.path.isfile(worker_config): 860 ↛ 861line 860 didn't jump to line 861 because the condition on line 860 was never true

861 return 

862 _settings: Final = json.loads(worker_config) 

863 if not isinstance(_settings, dict): 863 ↛ 864line 863 didn't jump to line 864 because the condition on line 863 was never true

864 return 

865 

866 debug: Final = _settings.get("debug", None) 

867 detailed_debug: Final = _settings.get("detailed_debug", None) 

868 if debug is True: # this needs to be first, so users can see Router init debugg 868 ↛ 869line 868 didn't jump to line 869 because the condition on line 868 was never true

869 import logging 

870 

871 from litellm._logging import ( 

872 verbose_logger, 

873 verbose_proxy_logger, 

874 verbose_router_logger, 

875 ) 

876 

877 # this must ALWAYS remain logging.INFO, DO NOT MODIFY THIS 

878 verbose_logger.setLevel(level=logging.INFO) # sets package logs to info 

879 verbose_router_logger.setLevel(level=logging.INFO) # set router logs to info 

880 verbose_proxy_logger.setLevel(level=logging.INFO) # set proxy logs to info 

881 if detailed_debug is True: 881 ↛ 882line 881 didn't jump to line 882 because the condition on line 881 was never true

882 import logging 

883 

884 from litellm._logging import ( 

885 verbose_logger, 

886 verbose_proxy_logger, 

887 verbose_router_logger, 

888 ) 

889 

890 verbose_logger.setLevel(level=logging.DEBUG) # set package log to debug 

891 verbose_router_logger.setLevel(level=logging.DEBUG) # set router logs to debug 

892 verbose_proxy_logger.setLevel(level=logging.DEBUG) # set proxy logs to debug 

893 elif debug is False and detailed_debug is False: 893 ↛ exitline 893 didn't return from function 'init_verbose_loggers' because the condition on line 893 was always true

894 # users can control proxy debugging using env variable = 'LITELLM_LOG' 

895 litellm_log_setting: Final = os.environ.get("LITELLM_LOG", "") 

896 if litellm_log_setting is not None: 896 ↛ exitline 896 didn't return from function 'init_verbose_loggers' because the condition on line 896 was always true

897 if litellm_log_setting.upper() == "INFO": 897 ↛ 898line 897 didn't jump to line 898 because the condition on line 897 was never true

898 import logging 

899 

900 from litellm._logging import ( 

901 verbose_proxy_logger, 

902 verbose_router_logger, 

903 ) 

904 

905 # this must ALWAYS remain logging.INFO, DO NOT MODIFY THIS 

906 

907 verbose_router_logger.setLevel(level=logging.INFO) # set router logs to info 

908 verbose_proxy_logger.setLevel(level=logging.INFO) # set proxy logs to info 

909 elif litellm_log_setting.upper() == "DEBUG": 909 ↛ 910line 909 didn't jump to line 910 because the condition on line 909 was never true

910 import logging 

911 

912 from litellm._logging import ( 

913 verbose_proxy_logger, 

914 verbose_router_logger, 

915 ) 

916 

917 verbose_router_logger.setLevel(level=logging.DEBUG) # set router logs to info 

918 verbose_proxy_logger.setLevel(level=logging.DEBUG) # set proxy logs to debug 

919 except Exception as e: 

920 import logging 

921 

922 logging.warning("Failed to init verbose loggers: %s", e)