Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/spend_tracking/compression_savings.py: 42%

26 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Single chokepoint for reading prompt-compression token savings out of a parsed 

3SpendLog ``metadata`` JSON dict. Imported by the daily-spend DB writer and by 

4cost-savings read endpoints. 

5""" 

6 

7from collections.abc import Mapping 

8from typing import Final 

9 

10HEADROOM_GUARDRAIL_PROVIDER: Final = "headroom" 

11 

12 

13def _saved_tokens_or_zero(value: object) -> int: 

14 if isinstance(value, bool) or not isinstance(value, (int, float)): 

15 return 0 

16 if value < 0: 

17 return 0 

18 return int(value) 

19 

20 

21def _tokens_saved_from_stats(stats: object) -> int: 

22 if not isinstance(stats, Mapping): 22 ↛ 24line 22 didn't jump to line 24 because the condition on line 22 was always true

23 return 0 

24 return _saved_tokens_or_zero(stats.get("tokens_saved")) 

25 

26 

27def _headroom_entry_saved_tokens(entry: object) -> int: 

28 if not isinstance(entry, Mapping): 

29 return 0 

30 if entry.get("guardrail_provider") != HEADROOM_GUARDRAIL_PROVIDER: 

31 return 0 

32 return _tokens_saved_from_stats(entry.get("guardrail_response")) 

33 

34 

35def _headroom_saved_tokens(guardrail_information: object) -> int: 

36 entries: Final = [guardrail_information] if isinstance(guardrail_information, Mapping) else guardrail_information 

37 if not isinstance(entries, list): 37 ↛ 39line 37 didn't jump to line 39 because the condition on line 37 was always true

38 return 0 

39 return sum(_headroom_entry_saved_tokens(entry) for entry in entries) 

40 

41 

42def extract_compression_saved_tokens(metadata: Mapping[str, object]) -> int: 

43 """ 

44 Return the total prompt tokens saved by compression for one request. 

45 

46 Sums two disjoint sources: 

47 

48 - the native ``compression_savings`` key, written only by 

49 ``CompressionInterceptionLogger`` in its pre-call deployment hook 

50 - ``guardrail_information`` entries with ``guardrail_provider == 

51 "headroom"``, written only by the Headroom guardrail 

52 

53 Each writer records only its own transform pass and the two run at 

54 different stages (guardrail pre-call vs deployment pre-call), so when both 

55 fire on one request their measured savings are independent and additive; 

56 summing them never double-counts. Malformed or missing values contribute 0. 

57 A bare dict ``guardrail_information`` is treated as a single entry, matching 

58 the spend-log redactor's normalization of that legacy shape. 

59 """ 

60 return _tokens_saved_from_stats(metadata.get("compression_savings")) + _headroom_saved_tokens( 

61 metadata.get("guardrail_information") 

62 )