Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/common_utils/http_parsing_utils.py: 61%

308 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1import json 

2import re 

3from collections.abc import Collection, Mapping 

4from types import MappingProxyType, UnionType 

5from typing import Annotated, Any, Final, Union, get_args, get_origin 

6 

7import orjson 

8from fastapi import Request, UploadFile, status 

9from typing_extensions import NotRequired, ReadOnly, Required 

10 

11from litellm._logging import verbose_proxy_logger 

12from litellm.constants import ( 

13 AZURE_SPEECH_PASS_THROUGH_ROUTE_PREFIX, 

14 CLIENT_REQUESTED_MODEL_SCOPE_KEY, 

15 MAX_REQUEST_BODY_SIZE_TO_REPAIR_MB, 

16) 

17from litellm.proxy._types import ProxyException 

18from litellm.proxy.common_utils.callback_utils import ( 

19 get_metadata_variable_name_from_kwargs, 

20) 

21from litellm.types.router import Deployment 

22 

23_FORM_CONTENT_TYPES: Final[frozenset[str]] = frozenset({"application/x-www-form-urlencoded", "multipart/form-data"}) 

24 

25_ANNOTATION_QUALIFIERS: Final[frozenset[object]] = frozenset({Annotated, NotRequired, ReadOnly, Required}) 

26 

27 

28def _normalize_media_type(content_type: str) -> str: 

29 """Return the bare media type per RFC 7231: strip params, trim, lowercase.""" 

30 if not content_type: 

31 return "" 

32 return content_type.split(";", 1)[0].strip().lower() 

33 

34 

35def _is_form_content_type(content_type: str) -> bool: 

36 """ 

37 True iff Starlette's ``request.form()`` will actually parse this body. 

38 

39 Substring matching ``"form"`` is unsafe: ``request.form()`` returns empty 

40 ``FormData`` for non-canonical types without consuming the body, leaving 

41 the auth-time pre-read and the handler's read seeing different payloads. 

42 """ 

43 return _normalize_media_type(content_type) in _FORM_CONTENT_TYPES 

44 

45 

46def is_json_content_type(content_type: str) -> bool: 

47 """True iff the body should be parsed as JSON.""" 

48 return _normalize_media_type(content_type) == "application/json" 

49 

50 

51def _unqualified(annotation: object) -> object: 

52 """Which qualifiers ``get_type_hints`` already stripped varies by interpreter version, so peel them all.""" 

53 if get_origin(annotation) not in _ANNOTATION_QUALIFIERS: 53 ↛ 55line 53 didn't jump to line 55 because the condition on line 53 was always true

54 return annotation 

55 qualified: Final[tuple[object, ...]] = get_args(annotation) 

56 return _unqualified(qualified[0]) 

57 

58 

59def _union_members(annotation: object) -> tuple[object, ...]: 

60 """The non-``None`` members of a union annotation, or the annotation itself when it is not a union.""" 

61 if get_origin(annotation) not in (Union, UnionType): 

62 return (annotation,) 

63 members: Final[tuple[object, ...]] = get_args(annotation) 

64 return tuple(arg for arg in members if arg is not type(None)) 

65 

66 

67def _numeric_form_type(annotation: object) -> type[int] | type[float] | None: 

68 """The scalar to parse an ``int``/``float``-typed field as, else ``None``.""" 

69 unwrapped: Final = _unqualified(annotation) 

70 candidates: Final = _union_members(unwrapped) 

71 if len(candidates) != 1: 

72 return None 

73 if candidates[0] is int: 

74 return int 

75 if candidates[0] is float: 75 ↛ 76line 75 didn't jump to line 76 because the condition on line 75 was never true

76 return float 

77 return None 

78 

79 

80def numeric_form_fields(annotations: Mapping[str, object]) -> Mapping[str, type[int] | type[float]]: 

81 """ 

82 The numeric fields of a request schema, mapped to the scalar to parse them as. 

83 

84 Only a bare ``int``/``float`` or an optional one qualifies, so container and 

85 literal fields are left alone and ``bool`` is excluded on purpose. 

86 """ 

87 return MappingProxyType( 

88 { 

89 name: scalar 

90 for name, annotation in annotations.items() 

91 if (scalar := _numeric_form_type(annotation)) is not None 

92 } 

93 ) 

94 

95 

96def _numeric_form_value(value: object, scalar: type[int] | type[float]) -> object: 

97 if not isinstance(value, str): 

98 return value 

99 try: 

100 return scalar(value) 

101 except ValueError: 

102 return value 

103 

104 

105def coerce_numeric_form_fields( 

106 parsed_body: Mapping[str, object], 

107 numeric_fields: Mapping[str, type[int] | type[float]], 

108) -> Mapping[str, object]: 

109 """ 

110 Parse the numeric fields of a form-encoded body back into numbers. 

111 

112 ``request.form()`` yields every field as a string, so a provider that puts the 

113 value in a JSON body would send a string where its API requires a number. A 

114 value that will not parse is left as-is for the provider to reject as before. 

115 """ 

116 return { 

117 name: _numeric_form_value(value, numeric_fields[name]) if name in numeric_fields else value 

118 for name, value in parsed_body.items() 

119 } 

120 

121 

122async def _read_request_body(request: Request | None) -> dict: 

123 """ 

124 Safely read the request body and parse it as JSON. 

125 

126 Parameters: 

127 - request: The request object to read the body from 

128 

129 Returns: 

130 - dict: Parsed request data as a dictionary or an empty dictionary if parsing fails 

131 """ 

132 try: 

133 if request is None: 133 ↛ 134line 133 didn't jump to line 134 because the condition on line 133 was never true

134 return {} 

135 

136 # Check if we already read and parsed the body 

137 _cached_request_body: Final[dict | None] = _safe_get_request_parsed_body(request=request) 

138 if _cached_request_body is not None: 

139 return _cached_request_body 

140 

141 _request_headers: Final[dict] = _safe_get_request_headers(request=request) 

142 content_type: Final = _request_headers.get("content-type", "") 

143 

144 if _is_form_content_type(content_type): 

145 try: 

146 form_data: Final = await request.form() 

147 except Exception as e: 

148 # ``request.form()`` raises on malformed multipart (missing 

149 # boundary, malformed chunk encoding, …). Surface as 400 so 

150 # the auth-time pre-read does not silently cache ``{}`` while 

151 # a later raw-body re-read sees the original payload — 

152 # banned-param checks must see the same body the handler 

153 # acts on. 

154 verbose_proxy_logger.error("Invalid form payload: %s", e) 

155 raise ProxyException( 

156 message=f"Invalid form payload: {e}", 

157 type="invalid_request_error", 

158 param="request_body", 

159 code=status.HTTP_400_BAD_REQUEST, 

160 ) 

161 parsed_body = dict(form_data) 

162 if "metadata" in parsed_body and isinstance(parsed_body["metadata"], str): 162 ↛ 163line 162 didn't jump to line 163 because the condition on line 162 was never true

163 parsed_body["metadata"] = json.loads(parsed_body["metadata"]) 

164 else: 

165 # Read the request body 

166 body: Final = await request.body() 

167 

168 # Return empty dict if body is empty or None 

169 if not body: 

170 parsed_body = {} 

171 else: 

172 try: 

173 parsed_body = orjson.loads(body) 

174 except orjson.JSONDecodeError as e: 

175 # The surrogate-repair fallback below runs two full-body re.sub 

176 # passes, which block the event loop on multi-MB malformed bodies. 

177 # Above the configured size, skip the repair and raise the 400 now. 

178 repair_limit_bytes: Final = MAX_REQUEST_BODY_SIZE_TO_REPAIR_MB * 1024 * 1024 

179 if repair_limit_bytes > 0 and len(body) > repair_limit_bytes: 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true

180 verbose_proxy_logger.error("Invalid JSON payload received: %s", e) 

181 raise ProxyException( 

182 message=f"Invalid JSON payload: {e}", 

183 type="invalid_request_error", 

184 param="request_body", 

185 code=status.HTTP_400_BAD_REQUEST, 

186 ) 

187 

188 # First try the standard json module which is more forgiving 

189 # First decode bytes to string if needed 

190 body_str = body.decode("utf-8") if isinstance(body, bytes) else body 

191 

192 # Replace invalid surrogate pairs 

193 # This regex finds incomplete surrogate pairs 

194 body_str = re.sub(r"[\uD800-\uDBFF](?![\uDC00-\uDFFF])", "", body_str) 

195 # This regex finds low surrogates without high surrogates 

196 body_str = re.sub(r"(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]", "", body_str) 

197 

198 try: 

199 parsed_body = json.loads(body_str) 

200 json.dumps(parsed_body, ensure_ascii=False).encode("utf-8") 

201 except (json.JSONDecodeError, UnicodeEncodeError): 

202 # json.loads accepts lone surrogate escapes that no provider can encode 

203 verbose_proxy_logger.error("Invalid JSON payload received: %s", e) 

204 raise ProxyException( 

205 message=f"Invalid JSON payload: {e}", 

206 type="invalid_request_error", 

207 param="request_body", 

208 code=status.HTTP_400_BAD_REQUEST, 

209 ) 

210 

211 # Cache the parsed result 

212 _safe_set_request_parsed_body(request=request, parsed_body=parsed_body) 

213 return parsed_body 

214 

215 except (json.JSONDecodeError, orjson.JSONDecodeError, ProxyException) as e: 

216 # Re-raise ProxyException as-is 

217 verbose_proxy_logger.error("Invalid JSON payload received: %s", e) 

218 raise 

219 except Exception as e: 

220 # Catch unexpected errors to avoid crashes 

221 verbose_proxy_logger.exception("Unexpected error reading request body - %s", e) 

222 return {} 

223 

224 

225def is_opaque_audio_pass_through_request(route: str, content_type: str) -> bool: 

226 """Azure Speech bodies (raw audio, multipart uploads) are forwarded byte for byte, so auth must not consume them.""" 

227 media_type: Final = _normalize_media_type(content_type) 

228 return route.startswith(f"{AZURE_SPEECH_PASS_THROUGH_ROUTE_PREFIX}/") and ( 

229 media_type.startswith("audio/") or media_type == "multipart/form-data" 

230 ) 

231 

232 

233async def read_raw_json_body(request: Request | None) -> bytes | None: 

234 if request is None or _safe_get_request_parsed_body(request=request) is None: 

235 return None 

236 content_type: Final = _safe_get_request_headers(request=request).get("content-type", "") 

237 if _is_form_content_type(content_type): 

238 return None 

239 try: 

240 return await request.body() 

241 except RuntimeError: 

242 return None 

243 

244 

245def _safe_get_request_parsed_body(request: Request | None) -> dict | None: 

246 if request is None: 246 ↛ 247line 246 didn't jump to line 247 because the condition on line 246 was never true

247 return None 

248 if hasattr(request, "scope") and "parsed_body" in request.scope and isinstance(request.scope["parsed_body"], tuple): 

249 accepted_keys, parsed_body = request.scope["parsed_body"] 

250 return {key: parsed_body[key] for key in accepted_keys} 

251 return None 

252 

253 

254def get_client_requested_model(request: Request | None) -> str | None: 

255 if request is None or not hasattr(request, "scope"): 255 ↛ 256line 255 didn't jump to line 256 because the condition on line 255 was never true

256 return None 

257 model: Final = request.scope.get(CLIENT_REQUESTED_MODEL_SCOPE_KEY) 

258 return model if isinstance(model, str) else None 

259 

260 

261def _safe_get_request_query_params(request: Request | None) -> dict: 

262 if request is None: 262 ↛ 263line 262 didn't jump to line 263 because the condition on line 262 was never true

263 return {} 

264 try: 

265 if hasattr(request, "query_params"): 265 ↛ 267line 265 didn't jump to line 267 because the condition on line 265 was always true

266 return dict(request.query_params) 

267 return {} 

268 except Exception as e: 

269 verbose_proxy_logger.debug("Unexpected error reading request query params - %s", e) 

270 return {} 

271 

272 

273def _safe_set_request_parsed_body( 

274 request: Request | None, 

275 parsed_body: dict, 

276) -> None: 

277 try: 

278 if request is None: 278 ↛ 279line 278 didn't jump to line 279 because the condition on line 278 was never true

279 return 

280 request.scope["parsed_body"] = (tuple(parsed_body.keys()), parsed_body) 

281 except Exception as e: 

282 verbose_proxy_logger.debug("Unexpected error setting request parsed body - %s", e) 

283 

284 

285def rewrite_request_model( 

286 request_data: dict[str, object], # mutable-ok: the request body is rewritten in place for every downstream reader 

287 request: Request | None, 

288 model: str, 

289) -> None: 

290 """Point the auth-time payload, the parsed-body cache, ``request.json()`` and ``request.body()`` at ``model``. 

291 The cache and raw body keep only the keys the client sent, not params auth merged into ``request_data``. 

292 """ 

293 request_data["model"] = model 

294 if request is None: 

295 return 

296 cached_body: Final = _safe_get_request_parsed_body(request=request) 

297 body: Final = {**cached_body, "model": model} if cached_body is not None else request_data 

298 _safe_set_request_parsed_body(request=request, parsed_body=body) 

299 request._json = body 

300 request._body = orjson.dumps(body) 

301 

302 

303def _safe_get_request_headers(request: Request | None) -> dict: 

304 """ 

305 [Non-Blocking] Safely get the request headers. 

306 Caches the result on request.state to avoid re-creating dict(request.headers) per call. 

307 

308 Warning: Callers must NOT mutate the returned dict — it is shared across 

309 all callers within the same request via the cache. 

310 """ 

311 if request is None: 311 ↛ 312line 311 didn't jump to line 312 because the condition on line 311 was never true

312 return {} 

313 state: Final = getattr(request, "state", None) 

314 cached: Final[object] = getattr(state, "_cached_headers", None) 

315 if isinstance(cached, dict): 

316 return cached 

317 if cached is not None: 317 ↛ 318line 317 didn't jump to line 318 because the condition on line 317 was never true

318 verbose_proxy_logger.debug("Unexpected cached request headers type - %s", type(cached)) 

319 try: 

320 headers = dict(request.headers) 

321 except Exception as e: 

322 verbose_proxy_logger.debug("Unexpected error reading request headers - %s", e) 

323 headers = {} 

324 try: 

325 if state is not None: 325 ↛ 329line 325 didn't jump to line 329 because the condition on line 325 was always true

326 state._cached_headers = headers 

327 except Exception: 

328 pass # request.state may not be available in all contexts 

329 return headers 

330 

331 

332def check_file_size_under_limit( 

333 request_data: dict, 

334 file: UploadFile, 

335 router_model_names: Collection[str], 

336) -> bool: 

337 """ 

338 Check if any files passed in request are under max_file_size_mb 

339 

340 Returns True -> when file size is under max_file_size_mb limit 

341 Raises ProxyException -> when file size is over max_file_size_mb limit or not a premium_user 

342 """ 

343 from litellm.proxy.proxy_server import ( 

344 CommonProxyErrors, 

345 ProxyException, 

346 llm_router, 

347 premium_user, 

348 ) 

349 

350 file_contents_size: Final = file.size or 0 

351 file_content_size_in_mb: Final = file_contents_size / (1024 * 1024) 

352 if "metadata" not in request_data: 352 ↛ 353line 352 didn't jump to line 353 because the condition on line 352 was never true

353 request_data["metadata"] = {} 

354 request_data["metadata"]["file_size_in_mb"] = file_content_size_in_mb 

355 max_file_size_mb = None 

356 

357 if llm_router is not None and request_data["model"] in router_model_names: 357 ↛ 358line 357 didn't jump to line 358 because the condition on line 357 was never true

358 try: 

359 deployment: Final[Deployment | None] = llm_router.get_deployment_by_model_group_name( 

360 model_group_name=request_data["model"] 

361 ) 

362 if ( 

363 deployment 

364 and deployment.litellm_params is not None 

365 and deployment.litellm_params.max_file_size_mb is not None 

366 ): 

367 max_file_size_mb = deployment.litellm_params.max_file_size_mb 

368 except Exception as e: 

369 verbose_proxy_logger.error("Got error when checking file size: %s", (str(e))) 

370 

371 if max_file_size_mb is not None: 371 ↛ 372line 371 didn't jump to line 372 because the condition on line 371 was never true

372 verbose_proxy_logger.debug( 

373 "Checking file size, file content size=%s, max_file_size_mb=%s", 

374 file_content_size_in_mb, 

375 max_file_size_mb, 

376 ) 

377 if not premium_user: 

378 raise ProxyException( 

379 message=f"Tried setting max_file_size_mb for /audio/transcriptions. {CommonProxyErrors.not_premium_user.value}", 

380 code=status.HTTP_400_BAD_REQUEST, 

381 type="bad_request", 

382 param="file", 

383 ) 

384 if file_content_size_in_mb > max_file_size_mb: 

385 raise ProxyException( 

386 message=f"File size is too large. Please check your file size. Passed file size: {file_content_size_in_mb} MB. Max file size: {max_file_size_mb} MB", 

387 code=status.HTTP_400_BAD_REQUEST, 

388 type="bad_request", 

389 param="file", 

390 ) 

391 

392 return True 

393 

394 

395async def get_form_data(request: Request) -> dict[str, Any]: 

396 """ 

397 Read form data from request 

398 

399 Handles when OpenAI SDKs pass form keys as `timestamp_granularities[]="word"` instead of `timestamp_granularities=["word", "sentence"]` 

400 """ 

401 form: Final = await request.form() 

402 parsed_form_data: Final[dict[str, Any]] = {} 

403 for key, value in form.multi_items(): # not dict(form), which keeps only the last repeat 

404 if key.endswith("[]"): 404 ↛ 405line 404 didn't jump to line 405 because the condition on line 404 was never true

405 clean_key = key[:-2] 

406 parsed_form_data.setdefault(clean_key, []).append(value) 

407 else: 

408 parsed_form_data[key] = value 

409 return parsed_form_data 

410 

411 

412async def convert_upload_files_to_file_data( 

413 form_data: dict[str, Any], 

414) -> dict[str, Any]: 

415 """ 

416 Convert FastAPI UploadFile objects to file data tuples for litellm. 

417 

418 Converts UploadFile objects to tuples of (filename, content, content_type) 

419 which is the format expected by httpx and litellm's HTTP handlers. 

420 

421 Args: 

422 form_data: Dictionary containing form data with potential UploadFile objects 

423 

424 Returns: 

425 Dictionary with UploadFile objects converted to file data tuples 

426 

427 Example: 

428 ```python 

429 form_data = await get_form_data(request) 

430 data = await convert_upload_files_to_file_data(form_data) 

431 # data["files"] is now [(filename, content, content_type), ...] 

432 ``` 

433 """ 

434 data: Final = {} 

435 for key, value in form_data.items(): 435 ↛ 436line 435 didn't jump to line 436 because the loop on line 435 never started

436 if isinstance(value, list): 

437 # Check if it's a list of UploadFile objects 

438 if value and hasattr(value[0], "read"): 

439 files = [] 

440 for f in value: 

441 file_content = await f.read() 

442 # Create tuple: (filename, content, content_type) 

443 files.append((f.filename, file_content, f.content_type)) 

444 data[key] = files 

445 else: 

446 data[key] = value 

447 elif hasattr(value, "read"): 

448 # Single UploadFile object - read and convert to list for consistency 

449 file_content = await value.read() 

450 data[key] = [(value.filename, file_content, value.content_type)] 

451 else: 

452 # Regular form field 

453 data[key] = value 

454 return data 

455 

456 

457async def get_request_body(request: Request) -> dict[str, Any]: 

458 """ 

459 Read the request body and parse it as JSON. 

460 """ 

461 if request.method == "POST": 

462 content_type: Final = request.headers.get("content-type", "") 

463 if is_json_content_type(content_type): 463 ↛ 464line 463 didn't jump to line 464 because the condition on line 463 was never true

464 return await _read_request_body(request) 

465 elif _is_form_content_type(content_type): 465 ↛ 466line 465 didn't jump to line 466 because the condition on line 465 was never true

466 return await get_form_data(request) 

467 else: 

468 raise ValueError(f"Unsupported content type: {content_type}") 

469 return {} 

470 

471 

472def extract_nested_form_metadata( 

473 form_data: Mapping[str, object], prefix: str = "litellm_metadata[" 

474) -> dict[str, object]: 

475 """ 

476 Extract nested metadata from form data with bracket notation. 

477 

478 Handles form data that uses bracket notation to represent nested dictionaries, 

479 such as litellm_metadata[spend_logs_metadata][owner] = "value". 

480 

481 This is commonly encountered when SDKs or clients send form data with nested 

482 structures using bracket notation instead of JSON. 

483 

484 Args: 

485 form_data: Dictionary containing form data (from request.form()) 

486 prefix: The prefix to look for in form keys (default: "litellm_metadata[") 

487 

488 Returns: 

489 Dictionary with nested structure reconstructed from bracket notation 

490 

491 Example: 

492 Input form_data: 

493 { 

494 "litellm_metadata[spend_logs_metadata][owner]": "john", 

495 "litellm_metadata[spend_logs_metadata][team]": "engineering", 

496 "litellm_metadata[tags]": "production", 

497 "other_field": "value" 

498 } 

499 

500 Output: 

501 { 

502 "spend_logs_metadata": { 

503 "owner": "john", 

504 "team": "engineering" 

505 }, 

506 "tags": "production" 

507 } 

508 """ 

509 if not form_data: 

510 return {} 

511 

512 metadata: Final[dict[str, object]] = {} 

513 

514 for key, value in form_data.items(): 

515 # Skip keys that don't start with the prefix 

516 if not isinstance(key, str) or not key.startswith(prefix): 

517 continue 

518 

519 # Skip UploadFile objects - they should not be in metadata 

520 if isinstance(value, UploadFile): 

521 verbose_proxy_logger.warning("Skipping UploadFile in metadata extraction for key: %s", key) 

522 continue 

523 

524 # Extract the nested path from bracket notation 

525 # Example: "litellm_metadata[spend_logs_metadata][owner]" -> ["spend_logs_metadata", "owner"] 

526 try: 

527 # Remove the prefix and strip trailing ']' 

528 path_string = key.replace(prefix, "").rstrip("]") 

529 

530 # Split by "][" to get individual path parts 

531 parts = path_string.split("][") 

532 

533 if not parts or not parts[0]: 

534 verbose_proxy_logger.warning("Invalid metadata key format (empty path): %s", key) 

535 continue 

536 

537 # Navigate/create nested dictionary structure 

538 current = metadata 

539 for part in parts[:-1]: 

540 if not isinstance(current, dict): 

541 verbose_proxy_logger.warning( 

542 "Cannot create nested path - intermediate value is not a dict at: %s", part 

543 ) 

544 break 

545 current = current.setdefault(part, {}) 

546 else: 

547 # Set the final value (only if we didn't break out of the loop) 

548 if isinstance(current, dict): 

549 current[parts[-1]] = value 

550 else: 

551 verbose_proxy_logger.warning("Cannot set value - parent is not a dict for key: %s", key) 

552 

553 except Exception as e: 

554 verbose_proxy_logger.error("Error parsing metadata key '%s': %s", key, e) 

555 continue 

556 

557 return metadata 

558 

559 

560def get_tags_from_request_body(request_body: Mapping[str, object]) -> list[str]: 

561 """ 

562 Extract tags from request body metadata. 

563 

564 Args: 

565 request_body: The request body dictionary 

566 

567 Returns: 

568 List of tag names (strings), empty list if no valid tags found 

569 """ 

570 metadata_variable_name: Final = get_metadata_variable_name_from_kwargs(request_body) 

571 metadata = request_body.get(metadata_variable_name) 

572 # metadata can arrive as a JSON string from multipart/form-data or extra_body; 

573 # coerce defensively so .get() below never raises AttributeError. 

574 if isinstance(metadata, str): 

575 from litellm.litellm_core_utils.safe_json_loads import safe_json_loads 

576 

577 parsed: Final[object] = safe_json_loads(metadata) 

578 metadata = parsed if isinstance(parsed, dict) else {} 

579 elif not isinstance(metadata, dict): 

580 metadata = {} 

581 tags_in_metadata: Final[object] = metadata.get("tags", []) 

582 tags_in_request_body: Final[object] = request_body.get("tags", []) 

583 combined_tags: Final[list[str]] = [] 

584 

585 ###################################### 

586 # Only combine tags if they are lists 

587 ###################################### 

588 if isinstance(tags_in_metadata, list): 588 ↛ 590line 588 didn't jump to line 590 because the condition on line 588 was always true

589 combined_tags.extend(tags_in_metadata) 

590 if isinstance(tags_in_request_body, list): 

591 combined_tags.extend(tags_in_request_body) 

592 ###################################### 

593 return [tag for tag in combined_tags if isinstance(tag, str)] 

594 

595 

596def populate_request_with_path_params(request_data: dict, request: Request) -> dict: 

597 """ 

598 Copy FastAPI path params and query params into the request payload so downstream checks 

599 (e.g. vector store RBAC, organization RBAC) see them the same way as body params. 

600 

601 Since path_params may not be available during dependency injection, 

602 we parse the URL path directly for known patterns. 

603 

604 Args: 

605 request_data: The request data dictionary to populate 

606 request: The FastAPI Request object 

607 

608 Returns: 

609 dict: Updated request_data with path parameters and query parameters added 

610 """ 

611 # Add query parameters to request_data (for GET requests, etc.) 

612 query_params: Final = _safe_get_request_query_params(request) 

613 if query_params: 

614 for key, value in query_params.items(): 

615 # Don't overwrite existing values from request body 

616 request_data.setdefault(key, value) 

617 

618 # Try to get path_params if available (sometimes populated by FastAPI) 

619 path_params: Final = getattr(request, "path_params", None) 

620 if isinstance(path_params, dict) and path_params: 

621 for key, value in path_params.items(): 

622 if key == "vector_store_id": 

623 request_data.setdefault("vector_store_id", value) 

624 existing_ids = request_data.get("vector_store_ids") 

625 if isinstance(existing_ids, list): 625 ↛ 626line 625 didn't jump to line 626 because the condition on line 625 was never true

626 if value not in existing_ids: 

627 existing_ids.append(value) 

628 else: 

629 request_data["vector_store_ids"] = [value] 

630 continue 

631 request_data.setdefault(key, value) 

632 verbose_proxy_logger.debug( 

633 "populate_request_with_path_params: Found path_params, vector_store_ids=%s", 

634 request_data.get("vector_store_ids"), 

635 ) 

636 return request_data 

637 

638 # Fallback: parse the URL path directly to extract vector_store_id 

639 _add_vector_store_id_from_path(request_data=request_data, request=request) 

640 

641 return request_data 

642 

643 

644def _add_vector_store_id_from_path(request_data: dict, request: Request) -> None: 

645 """ 

646 Parse the request path to find /vector_stores/{vector_store_id}/... segments. 

647 

648 When found, ensure both vector_store_id and vector_store_ids are populated. 

649 

650 Args: 

651 request_data: The request data dictionary to populate 

652 request: The FastAPI Request object 

653 """ 

654 # Inline import — auth_utils participates in a proxy import cycle. 

655 from litellm.proxy.auth.auth_utils import get_request_route # noqa: PLC0415 

656 

657 path: Final = get_request_route(request) 

658 vector_store_match: Final = re.search(r"/vector_stores/([^/]+)/", path) 

659 if vector_store_match: 659 ↛ 660line 659 didn't jump to line 660 because the condition on line 659 was never true

660 vector_store_id: Final = vector_store_match.group(1) 

661 verbose_proxy_logger.debug( 

662 "populate_request_with_path_params: Extracted vector_store_id=%s from path=%s", vector_store_id, path 

663 ) 

664 request_data.setdefault("vector_store_id", vector_store_id) 

665 existing_ids: Final = request_data.get("vector_store_ids") 

666 if isinstance(existing_ids, list): 

667 if vector_store_id not in existing_ids: 

668 existing_ids.append(vector_store_id) 

669 else: 

670 request_data["vector_store_ids"] = [vector_store_id] 

671 verbose_proxy_logger.debug( 

672 "populate_request_with_path_params: Updated request_data with vector_store_ids=%s", 

673 request_data.get("vector_store_ids"), 

674 ) 

675 else: 

676 verbose_proxy_logger.debug("populate_request_with_path_params: No vector_store_id present in path=%s", path)