Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/db/db_transaction_queue/spend_log_cleanup_metrics.py: 37%

53 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Prometheus metrics for the spend-log retention cleanup job. 

3 

4The job runs in the background on a single elected pod, so its cost is invisible 

5from request-path metrics. These instruments make a run's database footprint 

6observable: how much it deleted, how long each batch took, how much work is 

7still outstanding, and why a run stopped. 

8 

9``prometheus_client`` is an optional dependency, so every recorder degrades to a 

10no-op when it is absent. 

11""" 

12 

13from typing import TYPE_CHECKING, Final, Literal, TypeAlias 

14 

15from litellm._logging import verbose_proxy_logger 

16 

17if TYPE_CHECKING: 17 ↛ 19line 17 didn't jump to line 19 because the condition on line 17 was never true

18 # aliased so the annotations below cannot be mistaken for collections.Counter 

19 from prometheus_client import Counter as PrometheusCounter 

20 from prometheus_client import Gauge as PrometheusGauge 

21 from prometheus_client import Histogram as PrometheusHistogram 

22 

23RunOutcome: TypeAlias = Literal[ 

24 "completed", 

25 "budget_exhausted", 

26 "batch_cap_reached", 

27 "skipped_locked", 

28 "skipped_disabled", 

29 "aborted", 

30] 

31 

32_BATCH_DURATION_BUCKETS: Final = (0.005, 0.025, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0, 30.0, 60.0) 

33_TABLE_LABEL: Final = ("table",) 

34_OUTCOME_LABEL: Final = ("outcome",) 

35 

36 

37class SpendLogCleanupMetrics: 

38 """ 

39 Lazily-registered Prometheus instruments for the retention cleanup job. 

40 

41 Registration is deferred to first use so that importing this module never 

42 touches the Prometheus registry, which keeps it safe to import from the 

43 proxy regardless of whether Prometheus is a configured callback. 

44 """ 

45 

46 _initialized: bool = False 

47 rows_deleted: "PrometheusCounter | None" = None 

48 batch_duration: "PrometheusHistogram | None" = None 

49 rows_remaining: "PrometheusGauge | None" = None 

50 batch_failures: "PrometheusCounter | None" = None 

51 runs: "PrometheusCounter | None" = None 

52 

53 @classmethod 

54 def _ensure_initialized(cls) -> None: 

55 if cls._initialized: 

56 return 

57 cls._initialized = True 

58 try: 

59 # prometheus_client is an optional extra, so it is resolved here rather 

60 # than at module import: this module is reachable from proxy startup 

61 # regardless of whether Prometheus is a configured callback. 

62 from prometheus_client import Counter, Gauge, Histogram 

63 

64 cls.rows_deleted = Counter( 

65 "litellm_spend_log_cleanup_rows_deleted_total", 

66 "Rows deleted by the spend-log retention cleanup job", 

67 labelnames=_TABLE_LABEL, 

68 ) 

69 cls.batch_duration = Histogram( 

70 "litellm_spend_log_cleanup_batch_duration_seconds", 

71 "Wall-clock duration of one retention cleanup delete batch", 

72 labelnames=_TABLE_LABEL, 

73 buckets=_BATCH_DURATION_BUCKETS, 

74 ) 

75 cls.rows_remaining = Gauge( 

76 "litellm_spend_log_cleanup_rows_remaining", 

77 "Expired rows still awaiting deletion, counted only up to " 

78 "SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP so the probe itself cannot scan a " 

79 "large table; a value equal to that cap means at least that many remain", 

80 labelnames=_TABLE_LABEL, 

81 multiprocess_mode="livemax", 

82 ) 

83 cls.batch_failures = Counter( 

84 "litellm_spend_log_cleanup_batch_failures_total", 

85 "Retention cleanup delete batches that raised", 

86 labelnames=_TABLE_LABEL, 

87 ) 

88 cls.runs = Counter( 

89 "litellm_spend_log_cleanup_runs_total", 

90 "Retention cleanup runs, labelled by why the run ended", 

91 labelnames=_OUTCOME_LABEL, 

92 ) 

93 except Exception as e: # noqa: BLE001 - a metrics problem must never fail the cleanup run 

94 # Covers the extra being absent, a duplicate registration (repeated 

95 # imports under a test runner), and registry misconfiguration alike. 

96 verbose_proxy_logger.warning("Could not register spend-log cleanup metrics: %s", e) 

97 

98 @classmethod 

99 def record_batch(cls, table_name: str, rows_deleted: int, duration_seconds: float) -> None: 

100 cls._ensure_initialized() 

101 if cls.rows_deleted is not None: 

102 cls.rows_deleted.labels(table=table_name).inc(rows_deleted) 

103 if cls.batch_duration is not None: 

104 cls.batch_duration.labels(table=table_name).observe(duration_seconds) 

105 

106 @classmethod 

107 def record_batch_failure(cls, table_name: str) -> None: 

108 cls._ensure_initialized() 

109 if cls.batch_failures is not None: 

110 cls.batch_failures.labels(table=table_name).inc() 

111 

112 @classmethod 

113 def set_rows_remaining(cls, table_name: str, remaining: int) -> None: 

114 cls._ensure_initialized() 

115 if cls.rows_remaining is not None: 

116 cls.rows_remaining.labels(table=table_name).set(remaining) 

117 

118 @classmethod 

119 def record_run(cls, outcome: RunOutcome) -> None: 

120 cls._ensure_initialized() 

121 if cls.runs is not None: 

122 cls.runs.labels(outcome=outcome).inc()