Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/model_armor/file_scanning.py: 37%

114 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1"""Resolve inline file/document attachments in chat messages to Model Armor byte payloads. 

2 

3Model Armor scans documents through its ``byteItem`` API (PDF, Office docs, CSV, plaintext). 

4This module walks message content blocks (``type: file`` with inline ``file_data`` and 

5``type: document`` with an inline base64 ``source``), validates each block into a typed model, 

6maps its MIME type to a Model Armor ``byteDataType``, and returns the decoded bytes so the 

7guardrail hooks can submit them. 

8 

9``plan_file_scans`` classifies each block: blocks with no inline bytes (``file_id`` or remote 

10``gs://`` / ``http(s)`` references) and supported documents whose base64 will not decode are 

11reported as unscannable so the guardrail hook can fail closed (blocking unless ``fail_on_error`` 

12is false) rather than letting an unscanned document reach the model. 

13""" 

14 

15import base64 

16import binascii 

17import mimetypes 

18from collections.abc import Sequence 

19from dataclasses import dataclass 

20from typing import Annotated, Final, Literal 

21 

22from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError 

23 

24from litellm._logging import verbose_proxy_logger 

25from litellm.types.llms.openai import AllMessageValues 

26 

27MODEL_ARMOR_MAX_FILE_SIZE_BYTES: Final = 4 * 1024 * 1024 

28 

29_REMOTE_URI_SCHEMES: Final = ("gs://", "http://", "https://") 

30 

31ModelArmorByteDataType = Literal["PDF", "WORD_DOCUMENT", "EXCEL_DOCUMENT", "POWERPOINT_DOCUMENT", "CSV", "TXT"] 

32 

33_MIME_TO_BYTE_DATA_TYPE: Final[tuple[tuple[str, ModelArmorByteDataType], ...]] = ( 

34 ("application/pdf", "PDF"), 

35 # Word family: legacy, OOXML, macro-enabled, and templates all map to WORD_DOCUMENT 

36 ("application/msword", "WORD_DOCUMENT"), 

37 ("application/vnd.openxmlformats-officedocument.wordprocessingml.document", "WORD_DOCUMENT"), 

38 ("application/vnd.openxmlformats-officedocument.wordprocessingml.template", "WORD_DOCUMENT"), 

39 ("application/vnd.ms-word.document.macroenabled.12", "WORD_DOCUMENT"), 

40 ("application/vnd.ms-word.template.macroenabled.12", "WORD_DOCUMENT"), 

41 # Excel family 

42 ("application/vnd.ms-excel", "EXCEL_DOCUMENT"), 

43 ("application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "EXCEL_DOCUMENT"), 

44 ("application/vnd.openxmlformats-officedocument.spreadsheetml.template", "EXCEL_DOCUMENT"), 

45 ("application/vnd.ms-excel.sheet.macroenabled.12", "EXCEL_DOCUMENT"), 

46 ("application/vnd.ms-excel.template.macroenabled.12", "EXCEL_DOCUMENT"), 

47 # PowerPoint family 

48 ("application/vnd.ms-powerpoint", "POWERPOINT_DOCUMENT"), 

49 ("application/vnd.openxmlformats-officedocument.presentationml.presentation", "POWERPOINT_DOCUMENT"), 

50 ("application/vnd.openxmlformats-officedocument.presentationml.template", "POWERPOINT_DOCUMENT"), 

51 ("application/vnd.openxmlformats-officedocument.presentationml.slideshow", "POWERPOINT_DOCUMENT"), 

52 ("application/vnd.ms-powerpoint.presentation.macroenabled.12", "POWERPOINT_DOCUMENT"), 

53 ("application/vnd.ms-powerpoint.template.macroenabled.12", "POWERPOINT_DOCUMENT"), 

54 ("application/vnd.ms-powerpoint.slideshow.macroenabled.12", "POWERPOINT_DOCUMENT"), 

55 ("text/csv", "CSV"), 

56 ("text/plain", "TXT"), 

57) 

58 

59 

60@dataclass(frozen=True, slots=True) 

61class ModelArmorFileAttachment: 

62 file_bytes: bytes 

63 byte_data_type: ModelArmorByteDataType 

64 

65 

66@dataclass(frozen=True, slots=True) 

67class FileScanPlan: 

68 # Decoded attachments ready to submit to Model Armor. 

69 attachments: tuple[ModelArmorFileAttachment, ...] 

70 # Document/file blocks the guardrail recognized but could not turn into scannable bytes 

71 # (file_id/remote references, or a supported type whose inline base64 failed to decode). 

72 unscannable_count: int 

73 

74 

75class _FileData(BaseModel): 

76 model_config = ConfigDict(extra="ignore") 

77 file_data: str | None = None 

78 format: str | None = None 

79 filename: str | None = None 

80 

81 

82class _FileBlock(BaseModel): 

83 model_config = ConfigDict(extra="ignore") 

84 type: Literal["file"] 

85 file: _FileData 

86 

87 

88class _DocumentSource(BaseModel): 

89 model_config = ConfigDict(extra="ignore") 

90 data: str | None = None 

91 media_type: str | None = None 

92 

93 

94class _DocumentBlock(BaseModel): 

95 model_config = ConfigDict(extra="ignore") 

96 type: Literal["document"] 

97 source: _DocumentSource 

98 

99 

100_AttachmentBlock = Annotated[_FileBlock | _DocumentBlock, Field(discriminator="type")] 

101_BLOCK_ADAPTER: Final[TypeAdapter[_FileBlock | _DocumentBlock]] = TypeAdapter(_AttachmentBlock) 

102 

103 

104def plan_file_scans(messages: Sequence[AllMessageValues]) -> FileScanPlan: 

105 """Classify every document/file block into scannable attachments vs unscannable ones. 

106 

107 Unscannable covers references with no inline bytes and supported documents whose inline 

108 base64 fails to decode; the hook fails closed on these. Inline content of an unsupported 

109 type (for example an image) is neither scanned nor counted, it is simply left alone. 

110 """ 

111 classified: Final = tuple(_classify_block(block) for message in messages for block in _content_blocks(message)) 

112 attachments: Final = tuple(attachment for attachment, _ in classified if attachment is not None) 

113 unscannable_count = sum(1 for attachment, is_unscannable in classified if attachment is None and is_unscannable) 

114 return FileScanPlan(attachments=attachments, unscannable_count=unscannable_count) 

115 

116 

117def _content_blocks(message: AllMessageValues) -> tuple[object, ...]: 

118 content: Final = message.get("content") 

119 return tuple(content) if isinstance(content, list) else () 

120 

121 

122def _classify_block(block: object) -> tuple[ModelArmorFileAttachment | None, bool]: 

123 """Return (attachment, is_unscannable). At most one is meaningful; (None, False) means skip.""" 

124 parsed: Final = _parse_block(block) 

125 if parsed is None: 

126 return None, False 

127 if _is_reference(parsed): 

128 return None, True 

129 

130 byte_data_type, data = _block_byte_data_type_and_data(parsed) 

131 if data is None: 

132 return None, True 

133 if byte_data_type is None: 

134 # Recognized inline content of a type Model Armor's byte API does not scan (e.g. an image). 

135 return None, False 

136 

137 decoded: Final = _safe_b64decode(data) 

138 if decoded is None: 

139 # A supported document whose base64 will not decode cannot be scanned, so fail closed. 

140 return None, True 

141 

142 return ModelArmorFileAttachment(file_bytes=decoded, byte_data_type=byte_data_type), False 

143 

144 

145def _is_reference(block: _FileBlock | _DocumentBlock) -> bool: 

146 if isinstance(block, _DocumentBlock): 

147 return not block.source.data 

148 raw: Final = block.file.file_data 

149 return not raw or _is_remote_uri(raw) 

150 

151 

152def _parse_block(block: object) -> _FileBlock | _DocumentBlock | None: 

153 try: 

154 return _BLOCK_ADAPTER.validate_python(block) 

155 except ValidationError: 

156 return None 

157 

158 

159def _block_byte_data_type_and_data( 

160 block: _FileBlock | _DocumentBlock, 

161) -> tuple[ModelArmorByteDataType | None, str | None]: 

162 if isinstance(block, _DocumentBlock): 

163 return _mime_to_byte_data_type(block.source.media_type), block.source.data 

164 

165 raw: Final = block.file.file_data 

166 if not raw: 

167 return None, None 

168 uri_mime, data = _parse_data_uri(raw) 

169 if data is None: 

170 data = raw 

171 # The data URI header is the least reliable signal: it can be generic (application/octet-stream) 

172 # or mislabeled (text/plain for a PDF). Prefer the explicit format and filename, falling back to 

173 # the header only when neither resolves, and warn rather than let a conflicting header downgrade a 

174 # recognized document to the wrong filter. 

175 declared: Final = _first_supported_byte_data_type((block.file.format, _mime_from_filename(block.file.filename))) 

176 header: Final = _mime_to_byte_data_type(uri_mime) 

177 if declared is None: 

178 return header, data 

179 if header is not None and header != declared: 

180 verbose_proxy_logger.warning( 

181 "Model Armor: data URI MIME %s maps to %s but the attachment declares %s; scanning as %s", 

182 uri_mime, 

183 header, 

184 declared, 

185 declared, 

186 ) 

187 return declared, data 

188 

189 

190def _first_supported_byte_data_type( 

191 mimes: tuple[str | None, ...], 

192) -> ModelArmorByteDataType | None: 

193 return next( 

194 (byte_data_type for mime in mimes for byte_data_type in (_mime_to_byte_data_type(mime),) if byte_data_type), 

195 None, 

196 ) 

197 

198 

199def _parse_data_uri(raw: str) -> tuple[str | None, str | None]: 

200 if not raw.startswith("data:") or ";base64," not in raw: 

201 return None, None 

202 header, data = raw.split(";base64,", 1) 

203 return header[len("data:") :] or None, data 

204 

205 

206def _mime_to_byte_data_type(mime: str | None) -> ModelArmorByteDataType | None: 

207 if mime is None: 

208 return None 

209 normalized: Final = mime.split(";")[0].strip().lower() 

210 return next( 

211 (byte_data_type for candidate, byte_data_type in _MIME_TO_BYTE_DATA_TYPE if candidate == normalized), None 

212 ) 

213 

214 

215def _mime_from_filename(filename: str | None) -> str | None: 

216 if filename is None: 

217 return None 

218 guessed, _ = mimetypes.guess_type(filename) 

219 return guessed 

220 

221 

222def _safe_b64decode(data: str) -> bytes | None: 

223 try: 

224 return base64.b64decode(data, validate=True) 

225 except (binascii.Error, ValueError): 

226 verbose_proxy_logger.warning("Model Armor: skipping attachment with undecodable base64 content") 

227 return None 

228 

229 

230def _is_remote_uri(raw: str) -> bool: 

231 return raw.strip().lower().startswith(_REMOTE_URI_SCHEMES)