Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/model_armor/file_scanning.py: 37%
114 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""Resolve inline file/document attachments in chat messages to Model Armor byte payloads.
3Model Armor scans documents through its ``byteItem`` API (PDF, Office docs, CSV, plaintext).
4This module walks message content blocks (``type: file`` with inline ``file_data`` and
5``type: document`` with an inline base64 ``source``), validates each block into a typed model,
6maps its MIME type to a Model Armor ``byteDataType``, and returns the decoded bytes so the
7guardrail hooks can submit them.
9``plan_file_scans`` classifies each block: blocks with no inline bytes (``file_id`` or remote
10``gs://`` / ``http(s)`` references) and supported documents whose base64 will not decode are
11reported as unscannable so the guardrail hook can fail closed (blocking unless ``fail_on_error``
12is false) rather than letting an unscanned document reach the model.
13"""
15import base64
16import binascii
17import mimetypes
18from collections.abc import Sequence
19from dataclasses import dataclass
20from typing import Annotated, Final, Literal
22from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
24from litellm._logging import verbose_proxy_logger
25from litellm.types.llms.openai import AllMessageValues
27MODEL_ARMOR_MAX_FILE_SIZE_BYTES: Final = 4 * 1024 * 1024
29_REMOTE_URI_SCHEMES: Final = ("gs://", "http://", "https://")
31ModelArmorByteDataType = Literal["PDF", "WORD_DOCUMENT", "EXCEL_DOCUMENT", "POWERPOINT_DOCUMENT", "CSV", "TXT"]
33_MIME_TO_BYTE_DATA_TYPE: Final[tuple[tuple[str, ModelArmorByteDataType], ...]] = (
34 ("application/pdf", "PDF"),
35 # Word family: legacy, OOXML, macro-enabled, and templates all map to WORD_DOCUMENT
36 ("application/msword", "WORD_DOCUMENT"),
37 ("application/vnd.openxmlformats-officedocument.wordprocessingml.document", "WORD_DOCUMENT"),
38 ("application/vnd.openxmlformats-officedocument.wordprocessingml.template", "WORD_DOCUMENT"),
39 ("application/vnd.ms-word.document.macroenabled.12", "WORD_DOCUMENT"),
40 ("application/vnd.ms-word.template.macroenabled.12", "WORD_DOCUMENT"),
41 # Excel family
42 ("application/vnd.ms-excel", "EXCEL_DOCUMENT"),
43 ("application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "EXCEL_DOCUMENT"),
44 ("application/vnd.openxmlformats-officedocument.spreadsheetml.template", "EXCEL_DOCUMENT"),
45 ("application/vnd.ms-excel.sheet.macroenabled.12", "EXCEL_DOCUMENT"),
46 ("application/vnd.ms-excel.template.macroenabled.12", "EXCEL_DOCUMENT"),
47 # PowerPoint family
48 ("application/vnd.ms-powerpoint", "POWERPOINT_DOCUMENT"),
49 ("application/vnd.openxmlformats-officedocument.presentationml.presentation", "POWERPOINT_DOCUMENT"),
50 ("application/vnd.openxmlformats-officedocument.presentationml.template", "POWERPOINT_DOCUMENT"),
51 ("application/vnd.openxmlformats-officedocument.presentationml.slideshow", "POWERPOINT_DOCUMENT"),
52 ("application/vnd.ms-powerpoint.presentation.macroenabled.12", "POWERPOINT_DOCUMENT"),
53 ("application/vnd.ms-powerpoint.template.macroenabled.12", "POWERPOINT_DOCUMENT"),
54 ("application/vnd.ms-powerpoint.slideshow.macroenabled.12", "POWERPOINT_DOCUMENT"),
55 ("text/csv", "CSV"),
56 ("text/plain", "TXT"),
57)
60@dataclass(frozen=True, slots=True)
61class ModelArmorFileAttachment:
62 file_bytes: bytes
63 byte_data_type: ModelArmorByteDataType
66@dataclass(frozen=True, slots=True)
67class FileScanPlan:
68 # Decoded attachments ready to submit to Model Armor.
69 attachments: tuple[ModelArmorFileAttachment, ...]
70 # Document/file blocks the guardrail recognized but could not turn into scannable bytes
71 # (file_id/remote references, or a supported type whose inline base64 failed to decode).
72 unscannable_count: int
75class _FileData(BaseModel):
76 model_config = ConfigDict(extra="ignore")
77 file_data: str | None = None
78 format: str | None = None
79 filename: str | None = None
82class _FileBlock(BaseModel):
83 model_config = ConfigDict(extra="ignore")
84 type: Literal["file"]
85 file: _FileData
88class _DocumentSource(BaseModel):
89 model_config = ConfigDict(extra="ignore")
90 data: str | None = None
91 media_type: str | None = None
94class _DocumentBlock(BaseModel):
95 model_config = ConfigDict(extra="ignore")
96 type: Literal["document"]
97 source: _DocumentSource
100_AttachmentBlock = Annotated[_FileBlock | _DocumentBlock, Field(discriminator="type")]
101_BLOCK_ADAPTER: Final[TypeAdapter[_FileBlock | _DocumentBlock]] = TypeAdapter(_AttachmentBlock)
104def plan_file_scans(messages: Sequence[AllMessageValues]) -> FileScanPlan:
105 """Classify every document/file block into scannable attachments vs unscannable ones.
107 Unscannable covers references with no inline bytes and supported documents whose inline
108 base64 fails to decode; the hook fails closed on these. Inline content of an unsupported
109 type (for example an image) is neither scanned nor counted, it is simply left alone.
110 """
111 classified: Final = tuple(_classify_block(block) for message in messages for block in _content_blocks(message))
112 attachments: Final = tuple(attachment for attachment, _ in classified if attachment is not None)
113 unscannable_count = sum(1 for attachment, is_unscannable in classified if attachment is None and is_unscannable)
114 return FileScanPlan(attachments=attachments, unscannable_count=unscannable_count)
117def _content_blocks(message: AllMessageValues) -> tuple[object, ...]:
118 content: Final = message.get("content")
119 return tuple(content) if isinstance(content, list) else ()
122def _classify_block(block: object) -> tuple[ModelArmorFileAttachment | None, bool]:
123 """Return (attachment, is_unscannable). At most one is meaningful; (None, False) means skip."""
124 parsed: Final = _parse_block(block)
125 if parsed is None:
126 return None, False
127 if _is_reference(parsed):
128 return None, True
130 byte_data_type, data = _block_byte_data_type_and_data(parsed)
131 if data is None:
132 return None, True
133 if byte_data_type is None:
134 # Recognized inline content of a type Model Armor's byte API does not scan (e.g. an image).
135 return None, False
137 decoded: Final = _safe_b64decode(data)
138 if decoded is None:
139 # A supported document whose base64 will not decode cannot be scanned, so fail closed.
140 return None, True
142 return ModelArmorFileAttachment(file_bytes=decoded, byte_data_type=byte_data_type), False
145def _is_reference(block: _FileBlock | _DocumentBlock) -> bool:
146 if isinstance(block, _DocumentBlock):
147 return not block.source.data
148 raw: Final = block.file.file_data
149 return not raw or _is_remote_uri(raw)
152def _parse_block(block: object) -> _FileBlock | _DocumentBlock | None:
153 try:
154 return _BLOCK_ADAPTER.validate_python(block)
155 except ValidationError:
156 return None
159def _block_byte_data_type_and_data(
160 block: _FileBlock | _DocumentBlock,
161) -> tuple[ModelArmorByteDataType | None, str | None]:
162 if isinstance(block, _DocumentBlock):
163 return _mime_to_byte_data_type(block.source.media_type), block.source.data
165 raw: Final = block.file.file_data
166 if not raw:
167 return None, None
168 uri_mime, data = _parse_data_uri(raw)
169 if data is None:
170 data = raw
171 # The data URI header is the least reliable signal: it can be generic (application/octet-stream)
172 # or mislabeled (text/plain for a PDF). Prefer the explicit format and filename, falling back to
173 # the header only when neither resolves, and warn rather than let a conflicting header downgrade a
174 # recognized document to the wrong filter.
175 declared: Final = _first_supported_byte_data_type((block.file.format, _mime_from_filename(block.file.filename)))
176 header: Final = _mime_to_byte_data_type(uri_mime)
177 if declared is None:
178 return header, data
179 if header is not None and header != declared:
180 verbose_proxy_logger.warning(
181 "Model Armor: data URI MIME %s maps to %s but the attachment declares %s; scanning as %s",
182 uri_mime,
183 header,
184 declared,
185 declared,
186 )
187 return declared, data
190def _first_supported_byte_data_type(
191 mimes: tuple[str | None, ...],
192) -> ModelArmorByteDataType | None:
193 return next(
194 (byte_data_type for mime in mimes for byte_data_type in (_mime_to_byte_data_type(mime),) if byte_data_type),
195 None,
196 )
199def _parse_data_uri(raw: str) -> tuple[str | None, str | None]:
200 if not raw.startswith("data:") or ";base64," not in raw:
201 return None, None
202 header, data = raw.split(";base64,", 1)
203 return header[len("data:") :] or None, data
206def _mime_to_byte_data_type(mime: str | None) -> ModelArmorByteDataType | None:
207 if mime is None:
208 return None
209 normalized: Final = mime.split(";")[0].strip().lower()
210 return next(
211 (byte_data_type for candidate, byte_data_type in _MIME_TO_BYTE_DATA_TYPE if candidate == normalized), None
212 )
215def _mime_from_filename(filename: str | None) -> str | None:
216 if filename is None:
217 return None
218 guessed, _ = mimetypes.guess_type(filename)
219 return guessed
222def _safe_b64decode(data: str) -> bytes | None:
223 try:
224 return base64.b64decode(data, validate=True)
225 except (binascii.Error, ValueError):
226 verbose_proxy_logger.warning("Model Armor: skipping attachment with undecodable base64 content")
227 return None
230def _is_remote_uri(raw: str) -> bool:
231 return raw.strip().lower().startswith(_REMOTE_URI_SCHEMES)