Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/db/spend_log_batching.py: 82%
32 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""Split a spend-log flush into statements the Prisma query engine can afford.
3The query engine is a separate Rust process whose resident memory is a
4high-water mark: it grows with the payload of the largest single statement it
5is asked to execute and glibc never returns that memory to the OS, so a pod's
6memory floor ratchets up to its worst-ever write and stays there for the life
7of the worker. Under ``store_prompts_in_spend_logs`` a single spend-log row
8carries the full prompt and response, so a fixed 1000-row ``create_many``
9hands the engine tens of megabytes in one statement and permanently costs
10hundreds of megabytes of RSS, which is what makes memory-based autoscaling
11read the wrong number.
13Bounding each statement caps that floor, and it takes two budgets because the
14engine charges for both terms. A byte budget is what tracks a prompt-carrying
15row, whose size swings by orders of magnitude, and a row budget is what tracks
16the engine's per-row bookkeeping, which a byte budget cannot see: rows holding
17attribution metadata only stay far under any useful byte budget, so it never
18binds and every statement runs at the caller's row cap. Measured on such a
19flush, the same 100,000 rows cost 151 MB of permanently resident engine RSS at
201000 rows per statement against 25 MB at 100, with no statement anywhere near
21a 2 MB byte budget.
22"""
24import json
25from collections.abc import Iterator, Mapping, Sequence
26from itertools import accumulate
27from typing import Final
29SpendLogRow = Mapping[str, object]
31_STATEMENT_FRAMING_BYTES: Final = len(json.dumps([]))
32_ROW_SEPARATOR_BYTES: Final = len(json.dumps([0, 0])) - len(json.dumps([0])) - len(json.dumps(0))
35def _row_payload_bytes(row: SpendLogRow) -> int:
36 """Bytes this row contributes to the encoded write statement.
38 The whole row is serialized rather than its values summed, so the count
39 includes the field names, separators and braces the row carries on the
40 wire and not only its payload. Those are what make the difference between
41 a measurement and an estimate for a row of many small columns, where the
42 keys outweigh the values.
44 Serializing is also what makes the count a byte count. Character counts
45 under-measure a prompt in a non-Latin script by its bytes-per-character
46 factor, and even an all-ASCII prompt grows when JSON escapes its quotes,
47 backslashes and newlines (about 18% for a realistic stored prompt, and up
48 to double for escape-dense content). ``json.dumps`` escapes non-ASCII to
49 ``\\uXXXX`` and defaults to ASCII output, so its length never under-states
50 the wire size. ``default=str`` covers the datetimes and other scalars a
51 row carries.
53 A row the serializer refuses (a self-reference is the reachable case)
54 counts as zero rather than raising: measuring a row must never be what
55 loses spend data, since raising here would propagate out of the flush and
56 drop every row queued behind it. Such a row is still written, it just does
57 not contribute to the budget.
58 """
59 try:
60 return len(json.dumps(row, default=str))
61 except (TypeError, ValueError):
62 return 0
65def spend_log_row_bytes(row: SpendLogRow) -> int:
66 """Bytes this row costs, measured the same way the write budget measures it."""
67 return _row_payload_bytes(row)
70def spend_log_queue_within_budget(
71 rows: Sequence[SpendLogRow],
72 queued_bytes: int,
73 max_bytes: int,
74) -> tuple[Sequence[SpendLogRow], int]:
75 """Drop the oldest rows until the queue costs at most ``max_bytes``.
77 Returns the rows to keep and what they cost, so a caller tracking the total
78 across calls does not have to re-measure the rows it kept. ``queued_bytes``
79 is that running total for ``rows``; only the rows actually dropped are
80 measured here, which is what keeps an append off an O(queue) path.
82 A queue is bounded by bytes rather than by row count because a row's size
83 swings by orders of magnitude with ``store_prompts_in_spend_logs``, so any
84 row cap generous enough to ride out an outage of counter-only rows is an
85 OOM once prompts are stored.
87 The newest row is kept whatever it costs, for the same reason a statement
88 over budget is still written: the budget is a memory guardrail, not an
89 admission filter, and losing spend data to protect RSS is the worse failure.
90 """
91 if queued_bytes <= max_bytes or len(rows) <= 1: 91 ↛ 93line 91 didn't jump to line 93 because the condition on line 91 was always true
92 return rows, queued_bytes
93 droppable: Final = rows[:-1]
94 remaining_by_drops: Final = (
95 queued_bytes - freed for freed in accumulate(_row_payload_bytes(row) for row in droppable)
96 )
97 fits: Final = next(
98 ((drops, remaining) for drops, remaining in enumerate(remaining_by_drops, start=1) if remaining <= max_bytes),
99 (len(droppable), _row_payload_bytes(rows[-1])),
100 )
101 return rows[fits[0] :], fits[1]
104def spend_log_write_batches(
105 rows: Sequence[SpendLogRow],
106 max_bytes: int,
107 max_rows: int,
108) -> Iterator[Sequence[SpendLogRow]]:
109 """Yield consecutive slices of ``rows`` within both ``max_bytes`` and ``max_rows``.
111 What is measured for the byte budget is the encoded slice, not the sum of
112 its rows: rows become one collection on the wire, so the brackets around
113 them and the separator between each pair count too. Summing rows alone
114 under-states a slice by one separator per row, which is negligible for
115 prompt-carrying rows and is not for a slice of many small ones, where the
116 budget would be exceeded by the row count. The two framing constants are
117 derived from the serializer rather than written down so they cannot drift
118 from it.
120 Both budgets are needed because the engine's cost has two terms. Payload
121 bytes dominate when prompts are stored, and per-row bookkeeping dominates
122 when they are not: a slice of narrow rows costs the engine far more than
123 its bytes suggest, so a byte budget alone never binds on a deployment whose
124 rows carry no prompts and every statement stays at the caller's row cap.
125 Measured on a spend-log flush of rows carrying attribution metadata only,
126 writing the same 100,000 rows at 1000 rows per statement left 151 MB of
127 engine RSS resident against 25 MB at 100, with neither reaching a 2 MB byte
128 budget.
130 Slices preserve input order and together cover every row exactly once. A
131 row larger than ``max_bytes`` on its own is yielded alone rather than
132 dropped: the budget is a memory guardrail, not an admission filter, and
133 losing spend data to protect RSS would be the worse failure.
134 """
135 sizes: Final = tuple(_row_payload_bytes(row) for row in rows)
136 start = 0
137 while start < len(rows):
138 end = start + 1
139 used = _STATEMENT_FRAMING_BYTES + sizes[start]
140 while end < len(rows) and end - start < max_rows and used + _ROW_SEPARATOR_BYTES + sizes[end] <= max_bytes:
141 used += _ROW_SEPARATOR_BYTES + sizes[end]
142 end += 1
143 yield rows[start:end]
144 start = end