Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/spend_tracking/compression_savings.py: 42%
26 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Single chokepoint for reading prompt-compression token savings out of a parsed
3SpendLog ``metadata`` JSON dict. Imported by the daily-spend DB writer and by
4cost-savings read endpoints.
5"""
7from collections.abc import Mapping
8from typing import Final
10HEADROOM_GUARDRAIL_PROVIDER: Final = "headroom"
13def _saved_tokens_or_zero(value: object) -> int:
14 if isinstance(value, bool) or not isinstance(value, (int, float)):
15 return 0
16 if value < 0:
17 return 0
18 return int(value)
21def _tokens_saved_from_stats(stats: object) -> int:
22 if not isinstance(stats, Mapping): 22 ↛ 24line 22 didn't jump to line 24 because the condition on line 22 was always true
23 return 0
24 return _saved_tokens_or_zero(stats.get("tokens_saved"))
27def _headroom_entry_saved_tokens(entry: object) -> int:
28 if not isinstance(entry, Mapping):
29 return 0
30 if entry.get("guardrail_provider") != HEADROOM_GUARDRAIL_PROVIDER:
31 return 0
32 return _tokens_saved_from_stats(entry.get("guardrail_response"))
35def _headroom_saved_tokens(guardrail_information: object) -> int:
36 entries: Final = [guardrail_information] if isinstance(guardrail_information, Mapping) else guardrail_information
37 if not isinstance(entries, list): 37 ↛ 39line 37 didn't jump to line 39 because the condition on line 37 was always true
38 return 0
39 return sum(_headroom_entry_saved_tokens(entry) for entry in entries)
42def extract_compression_saved_tokens(metadata: Mapping[str, object]) -> int:
43 """
44 Return the total prompt tokens saved by compression for one request.
46 Sums two disjoint sources:
48 - the native ``compression_savings`` key, written only by
49 ``CompressionInterceptionLogger`` in its pre-call deployment hook
50 - ``guardrail_information`` entries with ``guardrail_provider ==
51 "headroom"``, written only by the Headroom guardrail
53 Each writer records only its own transform pass and the two run at
54 different stages (guardrail pre-call vs deployment pre-call), so when both
55 fire on one request their measured savings are independent and additive;
56 summing them never double-counts. Malformed or missing values contribute 0.
57 A bare dict ``guardrail_information`` is treated as a single entry, matching
58 the spend-log redactor's normalization of that legacy shape.
59 """
60 return _tokens_saved_from_stats(metadata.get("compression_savings")) + _headroom_saved_tokens(
61 metadata.get("guardrail_information")
62 )