Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/db/db_transaction_queue/spend_log_cleanup_metrics.py: 37%
53 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Prometheus metrics for the spend-log retention cleanup job.
4The job runs in the background on a single elected pod, so its cost is invisible
5from request-path metrics. These instruments make a run's database footprint
6observable: how much it deleted, how long each batch took, how much work is
7still outstanding, and why a run stopped.
9``prometheus_client`` is an optional dependency, so every recorder degrades to a
10no-op when it is absent.
11"""
13from typing import TYPE_CHECKING, Final, Literal, TypeAlias
15from litellm._logging import verbose_proxy_logger
17if TYPE_CHECKING: 17 ↛ 19line 17 didn't jump to line 19 because the condition on line 17 was never true
18 # aliased so the annotations below cannot be mistaken for collections.Counter
19 from prometheus_client import Counter as PrometheusCounter
20 from prometheus_client import Gauge as PrometheusGauge
21 from prometheus_client import Histogram as PrometheusHistogram
23RunOutcome: TypeAlias = Literal[
24 "completed",
25 "budget_exhausted",
26 "batch_cap_reached",
27 "skipped_locked",
28 "skipped_disabled",
29 "aborted",
30]
32_BATCH_DURATION_BUCKETS: Final = (0.005, 0.025, 0.1, 0.25, 0.5, 1.0, 2.5, 5.0, 10.0, 30.0, 60.0)
33_TABLE_LABEL: Final = ("table",)
34_OUTCOME_LABEL: Final = ("outcome",)
37class SpendLogCleanupMetrics:
38 """
39 Lazily-registered Prometheus instruments for the retention cleanup job.
41 Registration is deferred to first use so that importing this module never
42 touches the Prometheus registry, which keeps it safe to import from the
43 proxy regardless of whether Prometheus is a configured callback.
44 """
46 _initialized: bool = False
47 rows_deleted: "PrometheusCounter | None" = None
48 batch_duration: "PrometheusHistogram | None" = None
49 rows_remaining: "PrometheusGauge | None" = None
50 batch_failures: "PrometheusCounter | None" = None
51 runs: "PrometheusCounter | None" = None
53 @classmethod
54 def _ensure_initialized(cls) -> None:
55 if cls._initialized:
56 return
57 cls._initialized = True
58 try:
59 # prometheus_client is an optional extra, so it is resolved here rather
60 # than at module import: this module is reachable from proxy startup
61 # regardless of whether Prometheus is a configured callback.
62 from prometheus_client import Counter, Gauge, Histogram
64 cls.rows_deleted = Counter(
65 "litellm_spend_log_cleanup_rows_deleted_total",
66 "Rows deleted by the spend-log retention cleanup job",
67 labelnames=_TABLE_LABEL,
68 )
69 cls.batch_duration = Histogram(
70 "litellm_spend_log_cleanup_batch_duration_seconds",
71 "Wall-clock duration of one retention cleanup delete batch",
72 labelnames=_TABLE_LABEL,
73 buckets=_BATCH_DURATION_BUCKETS,
74 )
75 cls.rows_remaining = Gauge(
76 "litellm_spend_log_cleanup_rows_remaining",
77 "Expired rows still awaiting deletion, counted only up to "
78 "SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP so the probe itself cannot scan a "
79 "large table; a value equal to that cap means at least that many remain",
80 labelnames=_TABLE_LABEL,
81 multiprocess_mode="livemax",
82 )
83 cls.batch_failures = Counter(
84 "litellm_spend_log_cleanup_batch_failures_total",
85 "Retention cleanup delete batches that raised",
86 labelnames=_TABLE_LABEL,
87 )
88 cls.runs = Counter(
89 "litellm_spend_log_cleanup_runs_total",
90 "Retention cleanup runs, labelled by why the run ended",
91 labelnames=_OUTCOME_LABEL,
92 )
93 except Exception as e: # noqa: BLE001 - a metrics problem must never fail the cleanup run
94 # Covers the extra being absent, a duplicate registration (repeated
95 # imports under a test runner), and registry misconfiguration alike.
96 verbose_proxy_logger.warning("Could not register spend-log cleanup metrics: %s", e)
98 @classmethod
99 def record_batch(cls, table_name: str, rows_deleted: int, duration_seconds: float) -> None:
100 cls._ensure_initialized()
101 if cls.rows_deleted is not None:
102 cls.rows_deleted.labels(table=table_name).inc(rows_deleted)
103 if cls.batch_duration is not None:
104 cls.batch_duration.labels(table=table_name).observe(duration_seconds)
106 @classmethod
107 def record_batch_failure(cls, table_name: str) -> None:
108 cls._ensure_initialized()
109 if cls.batch_failures is not None:
110 cls.batch_failures.labels(table=table_name).inc()
112 @classmethod
113 def set_rows_remaining(cls, table_name: str, remaining: int) -> None:
114 cls._ensure_initialized()
115 if cls.rows_remaining is not None:
116 cls.rows_remaining.labels(table=table_name).set(remaining)
118 @classmethod
119 def record_run(cls, outcome: RunOutcome) -> None:
120 cls._ensure_initialized()
121 if cls.runs is not None:
122 cls.runs.labels(outcome=outcome).inc()