Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/db/exception_handler.py: 66%
197 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1import re
2from collections.abc import Awaitable, Callable, Iterator
3from http import HTTPStatus
4from typing import Final, Protocol, TypeVar
6from pydantic import TypeAdapter, ValidationError
8from litellm._logging import verbose_proxy_logger
9from litellm.proxy._types import (
10 DB_CONNECTION_ERROR_TYPES,
11 ProxyErrorTypes,
12 ProxyException,
13)
14from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded
15from litellm.secret_managers.main import str_to_bool
17# Bounds the __cause__/__context__ walk in find_database_service_unavailable_error_in_chain.
18# Real exception chains are a few links deep; the cap also makes the walk cycle-safe.
19_MAX_EXCEPTION_CHAIN_DEPTH: Final = 20
21_TRANSIENT_DB_UNAVAILABLE_MESSAGE: Final = (
22 "Service Unavailable, the authentication database is temporarily unreachable. Please retry shortly."
23)
25_DATABASE_ERROR_META: Final = TypeAdapter(dict[str, object])
26_BATCH_POSTGRES_ERROR_CODE: Final = re.compile(r'PostgresError \{ code: "([0-9A-Z]{5})"')
29def _exception_chain(e: BaseException) -> Iterator[BaseException]:
30 current = e # rebind-ok: advances one link per iteration of the bounded walk
31 for _ in range(_MAX_EXCEPTION_CHAIN_DEPTH): 31 ↛ exitline 31 didn't return from function '_exception_chain' because the loop on line 31 didn't complete
32 yield current
33 following = current.__cause__ or current.__context__
34 if following is None:
35 return
36 current = following
39def _database_service_unavailable_errors(e: BaseException) -> tuple[Exception, ...]:
40 return tuple(
41 link
42 for link in _exception_chain(e)
43 if isinstance(link, Exception) and PrismaDBExceptionHandler.is_database_service_unavailable_error(link)
44 )
47def _batch_postgres_sqlstate(e: Exception) -> str | None:
48 """The SQLSTATE a batched statement failed with: prisma reports those without a
49 ``meta`` payload and only prints the connector error into the message."""
50 match: Final = _BATCH_POSTGRES_ERROR_CODE.search(str(e))
51 return match.group(1) if match is not None else None
54def _exception_types(*candidates: object) -> tuple[type[BaseException], ...]:
55 """Keep only the real exception classes among ``candidates``.
57 The predicates below resolve prisma's error classes at call time, so a test
58 that swaps ``sys.modules["prisma"]`` for a ``MagicMock`` hands them mocks,
59 and ``isinstance`` against a mock raises ``TypeError`` instead of answering
60 False. Dropping the non-types lets the call fall through to the other checks.
61 """
62 return tuple(c for c in candidates if isinstance(c, type) and issubclass(c, BaseException))
65class PrismaDBExceptionHandler:
66 """
67 Class to handle DB Exceptions or Connection Errors
68 """
70 @staticmethod
71 def should_allow_request_on_db_unavailable() -> bool:
72 """
73 Returns True if the request should be allowed to proceed despite the DB connection error
74 """
75 from litellm.proxy.proxy_server import general_settings
77 _allow_requests_on_db_unavailable: bool | str = general_settings.get("allow_requests_on_db_unavailable", False)
78 if isinstance(_allow_requests_on_db_unavailable, bool): 78 ↛ 80line 78 didn't jump to line 80 because the condition on line 78 was always true
79 return _allow_requests_on_db_unavailable
80 if str_to_bool(_allow_requests_on_db_unavailable) is True:
81 return True
82 return False
84 @staticmethod
85 def is_database_connection_error(e: Exception) -> bool:
86 """True only for a database that is temporarily unreachable and is
87 expected to come back on its own.
89 This is the gate for ``allow_requests_on_db_unavailable``, which lets
90 the proxy keep serving, and issue fallback identities, without a
91 verified database. Only a transient outage justifies that. A fault that
92 will never resolve by itself, such as a query engine that is missing or
93 version-skewed, a malformed query the client library built, or a
94 transaction used incorrectly, must surface rather than be absorbed into
95 an indefinite degraded mode.
97 Membership is an allowlist, so an unrecognized failure is treated as
98 permanent. A genuine outage reaches the caller as one of
99 ``DB_CONNECTION_ERROR_TYPES``: the engine is a local HTTP server, and an
100 unreachable database surfaces as a transport error against it rather
101 than as a prisma type.
103 Reporting decisions want the opposite breadth; use
104 ``is_database_infrastructure_error`` for those.
105 """
106 import prisma.engine.errors
108 if isinstance(e, (*DB_CONNECTION_ERROR_TYPES, DBLookupDeadlineExceeded)): 108 ↛ 109line 108 didn't jump to line 109 because the condition on line 108 was never true
109 return True
110 if isinstance(e, _exception_types(prisma.engine.errors.EngineConnectionError)): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true
111 return True
112 return isinstance(e, ProxyException) and e.type == ProxyErrorTypes.no_db_connection
114 @staticmethod
115 def is_database_infrastructure_error(e: Exception) -> bool:
116 """True when the failure came from the database or its query engine
117 rather than from the caller's request.
119 This answers a reporting question, not a serving one: should the caller
120 be told the service is at fault, or that their request was. It stays
121 deliberately broad, because a permanently faulted engine is still a
122 service problem, and reporting one as a credential failure sends an
123 operator looking in the wrong place. Widening it can only change which
124 error a caller sees; it never grants access.
126 Known data-layer PrismaError subclasses (``UniqueViolationError``,
127 ``RecordNotFoundError``, etc.) are excluded — the DB IS reachable and
128 the request itself is what failed.
129 """
130 import prisma
132 data_layer_errors: Final = _exception_types(
133 prisma.errors.DataError,
134 prisma.errors.UniqueViolationError,
135 prisma.errors.ForeignKeyViolationError,
136 prisma.errors.MissingRequiredValueError,
137 prisma.errors.RawQueryError,
138 prisma.errors.TableNotFoundError,
139 prisma.errors.RecordNotFoundError,
140 )
141 if isinstance(e, data_layer_errors):
142 return False
143 if isinstance(e, DB_CONNECTION_ERROR_TYPES): 143 ↛ 144line 143 didn't jump to line 144 because the condition on line 143 was never true
144 return True
145 if isinstance(e, _exception_types(prisma.errors.PrismaError)): 145 ↛ 146line 145 didn't jump to line 146 because the condition on line 145 was never true
146 return True
147 if isinstance(e, ProxyException) and e.type == ProxyErrorTypes.no_db_connection: 147 ↛ 148line 147 didn't jump to line 148 because the condition on line 147 was never true
148 return True
149 return False
151 @staticmethod
152 def is_prisma_data_error(e: Exception) -> bool:
153 """True iff ``e`` is a base prisma ``DataError``: the database processed
154 the statement and refused the data itself (e.g. ``invalid byte sequence
155 for encoding "UTF8": 0x00``), as opposed to a connectivity failure.
157 Matched by exact type, not ``isinstance``: the specific data-layer
158 subclasses (``UniqueViolationError``, ``TableNotFoundError``,
159 ``MissingRequiredValueError`` ...) all derive from ``DataError`` but
160 carry their own semantics, and a systemic one like a missing table must
161 not be mistaken for a single poison row and bisected away. A raw
162 Postgres execution error with no prisma P-code surfaces as the base
163 ``DataError``.
165 prisma also wraps the P1001 "can't reach database server" outage as a
166 base ``DataError``, so a caller that must not treat an outage as a
167 per-row data rejection has to additionally consult
168 ``is_database_service_unavailable_error`` before acting on a True here.
169 """
170 import prisma
172 return type(e) is prisma.errors.DataError
174 @staticmethod
175 def is_database_transport_error(e: Exception) -> bool:
176 """
177 Returns True only for transport/connectivity failures where a reconnect
178 attempt makes sense (e.g. DB is unreachable, connection dropped).
180 Use this for reconnect logic — data-layer errors like UniqueViolationError
181 mean the DB IS reachable, so reconnecting would be pointless.
182 """
183 import prisma
185 if isinstance(e, DB_CONNECTION_ERROR_TYPES): 185 ↛ 186line 185 didn't jump to line 186 because the condition on line 185 was never true
186 return True
187 if isinstance( 187 ↛ 194line 187 didn't jump to line 194 because the condition on line 187 was never true
188 e,
189 _exception_types(
190 prisma.errors.ClientNotConnectedError,
191 prisma.errors.HTTPClientClosedError,
192 ),
193 ):
194 return True
195 if isinstance(e, _exception_types(prisma.errors.PrismaError)):
196 error_message: Final = str(e).lower()
197 connection_keywords: Final = (
198 "can't reach database server",
199 "cannot reach database server",
200 "can't connect",
201 "cannot connect",
202 "connection error",
203 "connection closed",
204 "timed out",
205 "timeout",
206 "connection refused",
207 "network is unreachable",
208 "no route to host",
209 "broken pipe",
210 )
211 if any(keyword in error_message for keyword in connection_keywords): 211 ↛ 212line 211 didn't jump to line 212 because the condition on line 211 was never true
212 return True
213 if isinstance(e, ProxyException) and e.type == ProxyErrorTypes.no_db_connection: 213 ↛ 214line 213 didn't jump to line 214 because the condition on line 213 was never true
214 return True
215 return False
217 @staticmethod
218 def is_prisma_error(e: Exception) -> bool:
219 import prisma
221 return isinstance(e, _exception_types(prisma.errors.PrismaError))
223 @staticmethod
224 def is_deadlock_error(e: Exception) -> bool:
225 """True iff ``e`` is a Postgres deadlock (P2034 / 40P01) surfaced through prisma."""
226 import prisma
228 if not isinstance(e, _exception_types(prisma.errors.PrismaError)):
229 return False
230 if getattr(e, "code", None) == "P2034": 230 ↛ 231line 230 didn't jump to line 231 because the condition on line 230 was never true
231 return True
232 error_message = str(e).lower()
233 return (
234 "deadlock detected" in error_message
235 or "40p01" in error_message
236 or "write conflict or a deadlock" in error_message
237 )
239 @staticmethod
240 def postgres_sqlstate(e: Exception) -> str | None:
241 """The SQLSTATE Postgres attached to a failed statement, as prisma surfaces it, or None."""
242 import prisma
244 if not isinstance(e, _exception_types(prisma.errors.DataError)):
245 return None
246 try:
247 meta: Final = _DATABASE_ERROR_META.validate_python(getattr(e, "meta", None))
248 except ValidationError:
249 return _batch_postgres_sqlstate(e)
250 code: Final = meta.get("code")
251 return code if isinstance(code, str) else _batch_postgres_sqlstate(e)
253 @staticmethod
254 def is_read_only_transaction_error(e: Exception) -> bool:
255 """True iff ``e`` is Postgres SQLSTATE 25006 surfaced through prisma: the
256 pooled session answers reads but rejects writes, so the connection is
257 poisoned until the client is recreated."""
258 import prisma
260 if not isinstance(e, _exception_types(prisma.errors.PrismaError)):
261 return False
262 error_message: Final = str(e).lower()
263 return '"25006"' in error_message or "read-only transaction" in error_message
265 @staticmethod
266 def is_prisma_engine_internal_error(e: Exception) -> bool:
267 """True iff ``e`` is a non-``PrismaError`` exception raised from inside
268 prisma-client-py's query-engine layer.
270 During the instant a DB connection is torn down, the query engine can
271 return a malformed error payload (``user_facing_error.meta`` is
272 ``null``). prisma-client-py's ``handle_response_errors`` then crashes
273 with ``AttributeError: 'NoneType' object has no attribute 'get'``
274 before it can raise the proper P1001 "can't reach database server"
275 error. That AttributeError carries no connection keyword, so it can't
276 be matched by message; identify it by its ``prisma.engine`` origin
277 instead.
279 Recognized ``PrismaError`` subclasses are excluded: connectivity ones
280 are already classified by type/keyword above, and data-layer ones
281 (the DB IS reachable) must stay 401.
282 """
283 import prisma
285 if isinstance(e, _exception_types(prisma.errors.PrismaError)):
286 return False
287 tb = e.__traceback__ if hasattr(e, "__traceback__") else None
288 while tb is not None:
289 if tb.tb_frame.f_globals.get("__name__", "").startswith("prisma.engine"): 289 ↛ 290line 289 didn't jump to line 290 because the condition on line 289 was never true
290 return True
291 tb = tb.tb_next
292 return False
294 @staticmethod
295 def is_database_service_unavailable_error(e: Exception) -> bool:
296 """True iff the exception means the database could not answer at the
297 infrastructure level (connection refused, socket/interface failure,
298 timeout) rather than a genuine auth failure (key not found) or a
299 data-layer error (the DB IS reachable and rejected the data).
301 Auth must answer 401 only for a key the DB confirms is invalid. When
302 the DB itself is unreachable, the request has to surface as 503 so
303 callers retry instead of treating valid keys as invalid during an
304 outage.
306 Note: prisma-client-py mislabels the P1001 "can't reach database
307 server" connectivity failure as a ``DataError`` (a data-layer type),
308 so a type-only check misses real outages. ``is_database_transport_error``
309 keyword-matches the connection message and catches that masquerade,
310 while genuine data errors (no connection keyword) correctly stay 401.
312 The Postgres "cached plan must not change result type" error is matched
313 here, not in ``is_database_transport_error``: it is a transient stale-DB-
314 state condition (not an invalid key), but the connection is healthy so it
315 must not trigger a reconnect.
317 A non-``PrismaError`` raised from inside the prisma query engine (e.g.
318 the ``AttributeError`` from ``handle_response_errors`` when the engine
319 returns a malformed error payload mid-tear-down) is also treated as
320 unavailable; see ``is_prisma_engine_internal_error``.
321 """
322 import asyncio
324 if PrismaDBExceptionHandler.is_database_infrastructure_error(e): 324 ↛ 325line 324 didn't jump to line 325 because the condition on line 324 was never true
325 return True
326 if PrismaDBExceptionHandler.is_database_transport_error(e): 326 ↛ 327line 326 didn't jump to line 327 because the condition on line 326 was never true
327 return True
328 if PrismaDBExceptionHandler.is_prisma_engine_internal_error(e): 328 ↛ 329line 328 didn't jump to line 329 because the condition on line 328 was never true
329 return True
330 if "cached plan must not change result type" in str(e).lower(): 330 ↛ 331line 330 didn't jump to line 331 because the condition on line 330 was never true
331 return True
333 # OSError already covers ConnectionError and (Py3.3+) TimeoutError.
334 # asyncio.TimeoutError is a distinct class before Py3.11.
335 if isinstance(e, (OSError, asyncio.TimeoutError)):
336 return True
338 try:
339 import asyncpg
340 except ImportError:
341 return False
343 return isinstance(
344 e,
345 (
346 asyncpg.exceptions.PostgresConnectionError,
347 asyncpg.exceptions.InterfaceError,
348 ),
349 )
351 @staticmethod
352 def is_permanent_database_fault(e: Exception) -> bool:
353 """True for a service-unavailable failure that will not clear on its
354 own: an engine-layer ``PrismaError`` (missing or version-skewed engine
355 binary, engine error status, misused transaction) that is neither the
356 transient ``EngineConnectionError`` nor a reconnectable transport failure.
358 Picks only the wording of a 503, never whether one is sent;
359 ``is_database_service_unavailable_error`` stays the status gate.
360 """
361 if PrismaDBExceptionHandler.is_database_connection_error(e): 361 ↛ 362line 361 didn't jump to line 362 because the condition on line 361 was never true
362 return False
363 if PrismaDBExceptionHandler.is_database_transport_error(e): 363 ↛ 364line 363 didn't jump to line 364 because the condition on line 363 was never true
364 return False
365 return PrismaDBExceptionHandler.is_database_infrastructure_error(e)
367 @staticmethod
368 def database_unavailable_message(e: Exception) -> str:
369 """The 503 detail for a service-unavailable database failure: retry
370 guidance for a transient outage, a pointer at the deployment for a
371 fault that retrying cannot fix. A permanent fault anywhere in the
372 exception chain wins, since the transport error that surfaced it is
373 not what blocks recovery."""
374 fault: Final = PrismaDBExceptionHandler.find_database_service_unavailable_error_in_chain(e) or e
375 if not PrismaDBExceptionHandler.is_permanent_database_fault(fault): 375 ↛ 377line 375 didn't jump to line 377 because the condition on line 375 was always true
376 return _TRANSIENT_DB_UNAVAILABLE_MESSAGE
377 return (
378 "Service Unavailable, the authentication database query engine reported "
379 f"{type(fault).__name__}, which is not a transient outage and will not clear by retrying. "
380 "The proxy deployment needs attention."
381 )
383 @staticmethod
384 def service_unavailable_proxy_exception(e: Exception) -> ProxyException:
385 return ProxyException(
386 message=PrismaDBExceptionHandler.database_unavailable_message(e),
387 type=ProxyErrorTypes.no_db_connection,
388 param="None",
389 code=HTTPStatus.SERVICE_UNAVAILABLE.value,
390 )
392 @staticmethod
393 def find_database_service_unavailable_error_in_chain(e: BaseException) -> Exception | None:
394 """The exception in the ``__cause__`` / ``__context__`` chain that
395 ``is_database_service_unavailable_error`` accepts, or ``None``. Callers
396 that word a response by the kind of outage need the wrapped database
397 error itself, not just the fact that one is present. A permanent fault
398 outranks a transient one wherever it sits in the chain: a reconnect that
399 dies on a missing engine binary raises the transport error last, but the
400 binary is what keeps the database down."""
401 outages: Final = _database_service_unavailable_errors(e)
402 permanent: Final = next(filter(PrismaDBExceptionHandler.is_permanent_database_fault, outages), None)
403 return permanent if permanent is not None else next(iter(outages), None)
405 @staticmethod
406 def is_database_service_unavailable_error_in_chain(e: BaseException) -> bool:
407 """Like ``is_database_service_unavailable_error`` but also walks the
408 ``__cause__`` / ``__context__`` chain.
410 ``is_database_service_unavailable_error`` classifies a single exception
411 by type, which a caller that catches a raw DB failure and re-raises a
412 domain exception of a different type defeats. A type check on the
413 wrapper misses the outage, so the caller would mistake an
414 infrastructure fault for an auth failure. Walking the chain recovers the
415 real signal, which is the PEP 3134 way to inspect a wrapped cause.
417 The walk is depth-bounded, which also makes it cycle-safe.
418 """
419 return PrismaDBExceptionHandler.find_database_service_unavailable_error_in_chain(e) is not None
421 @staticmethod
422 def handle_db_exception(e: Exception):
423 """
424 Primary handler for `allow_requests_on_db_unavailable` flag. Decides whether to raise a DB Exception or not based on the flag.
426 - If exception is a DB Connection Error, and `allow_requests_on_db_unavailable` is True,
427 - Do not raise an exception, return None
428 - Else, raise the exception
429 """
430 if (
431 PrismaDBExceptionHandler.is_database_connection_error(e)
432 and PrismaDBExceptionHandler.should_allow_request_on_db_unavailable()
433 ):
434 return
435 raise e
438# Default fallback timeouts when neither the caller nor the prisma_client
439# expose `_db_auth_reconnect_timeout_seconds` / `_db_auth_reconnect_lock_timeout_seconds`.
440# Match the auth path's existing defaults so behavior is uniform across read paths.
441_DEFAULT_RECONNECT_TIMEOUT_SECONDS: Final = 2.0
442_DEFAULT_RECONNECT_LOCK_TIMEOUT_SECONDS: Final = 0.1
445def _coerce_timeout(value: object, fallback: float) -> float:
446 """Return `value` if it is a real int/float, else `fallback`. Guards
447 against tests that mock `prisma_client` and leave the timeout slots as
448 MagicMock instances."""
449 if isinstance(value, (int, float)) and not isinstance(value, bool):
450 return float(value)
451 return fallback
454_ReadResultT: Final = TypeVar("_ReadResultT")
457class _DBReconnectClient(Protocol):
458 """The one method `call_with_db_reconnect_retry` needs from a Prisma client."""
460 async def attempt_db_reconnect( 460 ↛ exitline 460 didn't return from function 'attempt_db_reconnect' because
461 self,
462 *,
463 reason: str,
464 timeout_seconds: float | None = None,
465 lock_timeout_seconds: float | None = None,
466 ) -> bool: ...
469async def call_with_db_reconnect_retry(
470 prisma_client: _DBReconnectClient,
471 coro_factory: Callable[[], Awaitable[_ReadResultT]],
472 *,
473 reason: str,
474 retry_safe_error_types: tuple[type[Exception], ...] | None = None,
475 timeout_seconds: float | None = None,
476 lock_timeout_seconds: float | None = None,
477) -> _ReadResultT:
478 """Run a Prisma read coroutine with one transport-reconnect-and-retry.
480 The canonical "self-heal a transient DB transport blip" wrapper used by
481 `PrismaClient.get_generic_data` and other read paths. Mirrors the inline
482 pattern in `auth_checks._fetch_key_object_from_db_with_reconnect` so we
483 have a single implementation rather than three drifting copies.
485 Behavior:
486 1. Await `coro_factory()`. On success, return its value.
487 2. On exception, if it is NOT a transport error (per
488 `is_database_transport_error`), re-raise — data-layer errors like
489 `UniqueViolationError` mean the DB is reachable, reconnect would be
490 pointless. Transport errors outside `retry_safe_error_types` are
491 re-raised too.
492 3. If `prisma_client` does not expose `attempt_db_reconnect`, re-raise.
493 This guards against partial stand-ins / older clients in tests.
494 4. Call `prisma_client.attempt_db_reconnect(reason=...)`. If it returns
495 False (cooldown / lock contention / reconnect failure), re-raise.
496 5. Otherwise await `coro_factory()` a second time and return / propagate
497 its result. At-most-one retry by construction — no infinite loop.
499 `coro_factory` MUST be a zero-arg callable that returns a fresh awaitable
500 on each call. Passing an already-awaited coroutine would fail on retry
501 with `RuntimeError: cannot reuse already awaited coroutine`.
503 `reason` should follow `<subsystem>_<operation>_<table>_failure` so
504 telemetry distinguishes between fan-out callers (e.g.
505 `_update_config_from_db` issues four concurrent reads).
507 Args:
508 prisma_client: The `PrismaClient` (or stand-in) that owns
509 `attempt_db_reconnect` and the `_db_auth_reconnect_*` defaults.
510 coro_factory: Zero-arg callable returning the read awaitable.
511 reason: Telemetry tag forwarded to `attempt_db_reconnect`.
512 retry_safe_error_types: Which transport errors may be replayed, or
513 None for every transport error. A non-idempotent write must narrow
514 this to `DB_RETRY_SAFE_ERROR_TYPES`, where the statements provably
515 never reached the database.
516 timeout_seconds: Optional override for the reconnect cycle timeout.
517 Defaults to `prisma_client._db_auth_reconnect_timeout_seconds`,
518 then to 2.0s.
519 lock_timeout_seconds: Optional override for how long the helper will
520 wait to acquire the reconnect lock. Defaults to
521 `prisma_client._db_auth_reconnect_lock_timeout_seconds`, then to
522 0.1s.
524 Returns:
525 Whatever `coro_factory()` returns (on first or second attempt).
527 Raises:
528 Whatever `coro_factory()` raises if the failure is not a transport
529 error, or if the reconnect attempt does not succeed, or if the retry
530 also fails.
531 """
532 try:
533 return await coro_factory()
534 except Exception as first_exc:
535 if not PrismaDBExceptionHandler.is_database_transport_error(first_exc): 535 ↛ 537line 535 didn't jump to line 537 because the condition on line 535 was always true
536 raise
537 if retry_safe_error_types is not None and not isinstance(first_exc, retry_safe_error_types):
538 raise
539 if not hasattr(prisma_client, "attempt_db_reconnect"):
540 raise
542 resolved_timeout: Final = _coerce_timeout(
543 (
544 timeout_seconds
545 if timeout_seconds is not None
546 else getattr(prisma_client, "_db_auth_reconnect_timeout_seconds", None)
547 ),
548 _DEFAULT_RECONNECT_TIMEOUT_SECONDS,
549 )
550 resolved_lock_timeout: Final = _coerce_timeout(
551 (
552 lock_timeout_seconds
553 if lock_timeout_seconds is not None
554 else getattr(prisma_client, "_db_auth_reconnect_lock_timeout_seconds", None)
555 ),
556 _DEFAULT_RECONNECT_LOCK_TIMEOUT_SECONDS,
557 )
559 verbose_proxy_logger.warning(
560 "DB transport error on read; attempting reconnect-and-retry. reason=%s error=%s",
561 reason,
562 first_exc,
563 )
565 # Preserve the original transport error in telemetry. If
566 # `attempt_db_reconnect` itself raises (e.g. lock cancellation, timer
567 # error, unexpected internal failure), surfacing that exception
568 # instead of `first_exc` would mask the actual DB transport problem
569 # in `failure_handler` / `db_exceptions` alerts. Chain the reconnect
570 # error as the cause for debuggability without losing the original.
571 try:
572 did_reconnect: Final = await prisma_client.attempt_db_reconnect(
573 reason=reason,
574 timeout_seconds=resolved_timeout,
575 lock_timeout_seconds=resolved_lock_timeout,
576 )
577 except Exception as reconnect_exc:
578 verbose_proxy_logger.warning(
579 "DB reconnect attempt raised; preserving original transport error. reason=%s reconnect_error=%s",
580 reason,
581 reconnect_exc,
582 )
583 raise first_exc from reconnect_exc
584 if not did_reconnect:
585 raise
587 # At most one retry. If the retry also raises a transport error, we
588 # propagate — repeated reconnect-loops are the watchdog's job, not
589 # this helper's.
590 return await coro_factory()