Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/noma/noma.py: 13%

348 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1# +-------------------------------------------------------------+ 

2# 

3# Noma Security Guardrail Integration for LiteLLM 

4# https://noma.security 

5# 

6# +-------------------------------------------------------------+ 

7 

8import asyncio 

9import json 

10import os 

11import warnings 

12from collections.abc import AsyncGenerator, AsyncIterable 

13from datetime import datetime 

14from typing import ( 

15 TYPE_CHECKING, 

16 Final, 

17 Literal, 

18 TypeVar, 

19) 

20from urllib.parse import urljoin 

21 

22from fastapi import HTTPException 

23 

24import litellm 

25from litellm import DualCache, ModelResponse 

26from litellm._logging import verbose_proxy_logger 

27from litellm.completion_extras.litellm_responses_transformation.transformation import ( 

28 LiteLLMResponsesTransformationHandler, 

29) 

30from litellm.integrations.custom_guardrail import CustomGuardrail 

31from litellm.llms.base_llm.base_model_iterator import MockResponseIterator 

32from litellm.llms.custom_httpx.http_handler import ( 

33 get_async_httpx_client, 

34 httpxSpecialProvider, 

35) 

36from litellm.main import stream_chunk_builder 

37from litellm.proxy._types import UserAPIKeyAuth 

38from litellm.types.guardrails import GuardrailEventHooks 

39from litellm.types.utils import ( 

40 CallTypes, 

41 CallTypesLiteral, 

42 GuardrailStatus, 

43 ModelResponseStream, 

44 TextCompletionResponse, 

45) 

46 

47# Constants 

48USER_ROLE: Final = "user" 

49ASSISTANT_ROLE: Final = "assistant" 

50SENSITIVE_DATA_DETECTOR_KEYS: Final[list[str]] = ["sensitiveData", "dataDetector"] 

51 

52# Type aliases 

53MessageRole = Literal["user", "assistant"] 

54LLMResponse = object 

55_LLMResponseT: Final = TypeVar("_LLMResponseT") 

56_LEGACY_NOMA_DEPRECATION_WARNED = False 

57 

58if TYPE_CHECKING: 58 ↛ 59line 58 didn't jump to line 59 because the condition on line 58 was never true

59 from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel 

60 

61 

62class NomaBlockedMessage(HTTPException): 

63 """Exception raised when Noma guardrail blocks a message""" 

64 

65 def __init__(self, classification_response: dict): 

66 super().__init__( 

67 status_code=400, 

68 detail={ 

69 "error": "Request blocked by Noma guardrail", 

70 "details": classification_response, 

71 }, 

72 ) 

73 

74 def _is_result_true(self, result_obj: dict[str, object] | None) -> bool: 

75 """ 

76 Check if a result object has a "result" field that is True. 

77 

78 Args: 

79 result_obj: A dictionary that may contain a "result" field 

80 

81 Returns: 

82 True if the "result" field exists and is True, False otherwise 

83 """ 

84 if not result_obj or not isinstance(result_obj, dict): 

85 return False 

86 

87 return result_obj.get("result") is True 

88 

89 

90class NomaGuardrail(CustomGuardrail): 

91 """ 

92 Noma Security Guardrail for LiteLLM 

93 

94 This guardrail integrates with Noma Security's AI-DR API to provide 

95 content moderation and safety checks for LLM inputs and outputs. 

96 """ 

97 

98 _DEFAULT_API_BASE = "https://api.noma.security/" 

99 _AIDR_ENDPOINT = "/ai-dr/v2/prompt/scan" 

100 

101 @classmethod 

102 def get_supported_event_hooks(cls) -> list[GuardrailEventHooks]: 

103 return [ 

104 GuardrailEventHooks.pre_call, 

105 GuardrailEventHooks.during_call, 

106 GuardrailEventHooks.post_call, 

107 GuardrailEventHooks.pre_mcp_call, 

108 ] 

109 

110 def __init__( 

111 self, 

112 api_key: str | None = None, 

113 api_base: str | None = None, 

114 application_id: str | None = None, 

115 monitor_mode: bool | None = None, 

116 block_failures: bool | None = None, 

117 anonymize_input: bool | None = None, 

118 **kwargs, 

119 ): 

120 global _LEGACY_NOMA_DEPRECATION_WARNED 

121 if not _LEGACY_NOMA_DEPRECATION_WARNED: 

122 warnings.warn( 

123 "Guardrail provider 'noma' is deprecated. " 

124 "Please migrate to 'noma_v2'. " 

125 "The legacy 'noma' API will no longer be supported after March 31, 2026.", 

126 DeprecationWarning, 

127 stacklevel=2, 

128 ) 

129 _LEGACY_NOMA_DEPRECATION_WARNED = True 

130 

131 self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback) 

132 self._responses_transform_handler = LiteLLMResponsesTransformationHandler() 

133 self.api_key = api_key or os.environ.get("NOMA_API_KEY") 

134 self.api_base = api_base or os.environ.get("NOMA_API_BASE", NomaGuardrail._DEFAULT_API_BASE) 

135 self.application_id = application_id or os.environ.get("NOMA_APPLICATION_ID") 

136 self.default_application_id = "litellm" 

137 

138 if monitor_mode is None: 

139 self.monitor_mode = os.environ.get("NOMA_MONITOR_MODE", "false").lower() == "true" 

140 else: 

141 self.monitor_mode = monitor_mode 

142 

143 if block_failures is None: 

144 self.block_failures = os.environ.get("NOMA_BLOCK_FAILURES", "true").lower() == "true" 

145 else: 

146 self.block_failures = block_failures 

147 

148 if anonymize_input is None: 

149 self.anonymize_input = os.environ.get("NOMA_ANONYMIZE_INPUT", "false").lower() == "true" 

150 else: 

151 self.anonymize_input = anonymize_input 

152 

153 kwargs.setdefault("supported_event_hooks", list(self.get_supported_event_hooks())) 

154 super().__init__(**kwargs) 

155 

156 def _create_background_noma_check( 

157 self, 

158 coro, 

159 ) -> None: 

160 """Create a background task for Noma API calls without blocking the main flow""" 

161 try: 

162 asyncio.create_task(coro) 

163 except Exception as e: 

164 verbose_proxy_logger.error("Failed to create background Noma task: %s", e) 

165 

166 async def _process_user_message_check( 

167 self, 

168 request_data: dict, 

169 user_auth: UserAPIKeyAuth, 

170 event_type: GuardrailEventHooks | None = None, 

171 ) -> str | None: 

172 """Shared logic for processing user message checks""" 

173 start_time: Final = datetime.now() 

174 extra_data: Final = self.get_guardrail_dynamic_request_body_params(request_data) 

175 

176 messages: Final = request_data.get("messages") or [] 

177 if not messages: 

178 return None 

179 

180 input_items, instructions = self._responses_transform_handler.convert_chat_completion_messages_to_responses_api( 

181 messages 

182 ) 

183 

184 if instructions: 

185 system_message: Final = { 

186 "type": "message", 

187 "role": "system", 

188 "content": [ 

189 {"type": "input_text", "text": instructions}, 

190 ], 

191 } 

192 input_items.insert(0, system_message) 

193 

194 if not input_items: 

195 return None 

196 

197 payload: Final = {"input": input_items} 

198 response_json: Final = await self._call_noma_api( 

199 payload=payload, 

200 llm_request_id=None, 

201 request_data=request_data, 

202 user_auth=user_auth, 

203 extra_data=extra_data, 

204 ) 

205 

206 end_time: Final = datetime.now() 

207 duration: Final = (end_time - start_time).total_seconds() 

208 

209 # Determine guardrail status based on response 

210 guardrail_status: Final = self._determine_guardrail_status(response_json) 

211 

212 # Always log guardrail information for consistency 

213 self.add_standard_logging_guardrail_information_to_request_data( 

214 guardrail_provider="noma", 

215 guardrail_json_response=response_json, 

216 request_data=request_data, 

217 guardrail_status=guardrail_status, 

218 start_time=start_time.timestamp(), 

219 end_time=end_time.timestamp(), 

220 duration=duration, 

221 event_type=event_type, 

222 ) 

223 

224 if self.monitor_mode: 

225 await self._handle_verdict_background(USER_ROLE, json.dumps(input_items), response_json) 

226 return json.dumps(input_items) 

227 

228 # Check if we should anonymize content 

229 if self._should_anonymize(response_json, USER_ROLE): 

230 anonymized_content: Final = self._extract_anonymized_content(response_json, USER_ROLE) 

231 if anonymized_content: 

232 # Replace the user message content with anonymized version 

233 self._replace_user_message_content(request_data, anonymized_content) 

234 verbose_proxy_logger.debug("Noma guardrail anonymized user message: %s", anonymized_content) 

235 return anonymized_content 

236 

237 await self._check_verdict(USER_ROLE, json.dumps(input_items), response_json) 

238 return json.dumps(input_items) 

239 

240 async def _process_llm_response_check( 

241 self, 

242 request_data: dict, 

243 response: LLMResponse, 

244 user_auth: UserAPIKeyAuth, 

245 event_type: GuardrailEventHooks | None = None, 

246 ) -> str | None: 

247 """Shared logic for processing LLM response checks""" 

248 

249 start_time: Final = datetime.now() 

250 extra_data: Final = self.get_guardrail_dynamic_request_body_params(request_data) 

251 

252 if not isinstance(response, litellm.ModelResponse): 

253 return None 

254 

255 content = None 

256 for choice in response.choices: 

257 if isinstance(choice, litellm.Choices) and choice.message.content: 

258 content = choice.message.content 

259 break 

260 

261 if not content or not isinstance(content, str): 

262 return None 

263 

264 payload: Final = { 

265 "input": [ 

266 { 

267 "type": "message", 

268 "role": "assistant", 

269 "content": [{"type": "input_text", "text": content}], 

270 } 

271 ] 

272 } 

273 

274 response_json: Final = await self._call_noma_api( 

275 payload=payload, 

276 llm_request_id=response.id, 

277 request_data=request_data, 

278 user_auth=user_auth, 

279 extra_data=extra_data, 

280 ) 

281 

282 end_time: Final = datetime.now() 

283 duration: Final = (end_time - start_time).total_seconds() 

284 

285 # Determine guardrail status based on response 

286 guardrail_status: Final = self._determine_guardrail_status(response_json) 

287 

288 # Always log guardrail information for consistency 

289 self.add_standard_logging_guardrail_information_to_request_data( 

290 guardrail_provider="noma", 

291 guardrail_json_response=response_json, 

292 request_data=request_data, 

293 guardrail_status=guardrail_status, 

294 start_time=start_time.timestamp(), 

295 end_time=end_time.timestamp(), 

296 duration=duration, 

297 event_type=event_type, 

298 ) 

299 

300 if self.monitor_mode: 

301 await self._handle_verdict_background(ASSISTANT_ROLE, json.dumps(content), response_json) 

302 return content 

303 

304 # Check if we should anonymize content 

305 if self._should_anonymize(response_json, ASSISTANT_ROLE): 

306 anonymized_content: Final = self._extract_anonymized_content(response_json, ASSISTANT_ROLE) 

307 if anonymized_content: 

308 # Replace the LLM response content with anonymized version 

309 self._replace_llm_response_content(response, anonymized_content) 

310 verbose_proxy_logger.debug("Noma guardrail anonymized LLM response: %s", anonymized_content) 

311 return anonymized_content 

312 

313 await self._check_verdict(ASSISTANT_ROLE, content, response_json) 

314 return content 

315 

316 def _determine_guardrail_status(self, response_json: dict) -> GuardrailStatus: 

317 """ 

318 Determine the guardrail status based on NOMA API response. 

319 

320 Args: 

321 response_json: Response from NOMA API 

322 

323 Returns: 

324 "success": Content allowed through with no violations 

325 "guardrail_intervened": Content blocked due to policy violations 

326 "guardrail_failed_to_respond": Technical error or API failure 

327 """ 

328 try: 

329 # Check if we got a valid response structure 

330 if not isinstance(response_json, dict): 

331 return "guardrail_failed_to_respond" 

332 

333 # Get the aggregatedScanResult from the response 

334 # aggregatedScanResult=True means unsafe (block), False means safe (allow) 

335 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False) 

336 

337 # If aggregatedScanResult is False, content is safe/allowed 

338 if aggregated_scan_result is False: 

339 return "success" 

340 

341 # If aggregatedScanResult is True, content is blocked/flagged 

342 if aggregated_scan_result is True: 

343 return "guardrail_intervened" 

344 

345 # If aggregatedScanResult is missing or invalid, treat as failure 

346 return "guardrail_failed_to_respond" 

347 

348 except Exception as e: 

349 verbose_proxy_logger.error("Error determining NOMA guardrail status: %s", e) 

350 return "guardrail_failed_to_respond" 

351 

352 def _should_only_sensitive_data_failed(self, classification_obj: dict) -> bool: 

353 """ 

354 Check if only sensitive data detectors (PII, PCI, secrets) have result=true in the classification. 

355 

356 Args: 

357 classification_obj: The prompt or response classification object from Noma API 

358 

359 Returns: 

360 True if only sensitiveData detectors have result=true, False otherwise 

361 """ 

362 if not classification_obj: 

363 return False 

364 

365 # Track which detectors have result=true (detected violations) 

366 failed_detectors: Final = [] 

367 sensitive_data_detected = False 

368 

369 for key, value in classification_obj.items(): 

370 if key in SENSITIVE_DATA_DETECTOR_KEYS and isinstance(value, dict): 

371 # Check if any sensitive data detector has result=true 

372 for data_type, data_result in value.items(): 

373 if self._is_result_true(data_result): 

374 sensitive_data_detected = True 

375 # Don't add to failed_detectors as we want to allow these 

376 

377 elif isinstance(value, dict) and "result" in value: 

378 # Check other detectors - these should NOT have result=true 

379 if self._is_result_true(value): 

380 failed_detectors.append(key) 

381 

382 elif isinstance(value, dict): 

383 # Handle nested detectors 

384 for nested_key, nested_value in value.items(): 

385 if self._is_result_true(nested_value): 

386 failed_detectors.append(f"{key}.{nested_key}") 

387 

388 # Return True only if sensitive data was detected AND no other detectors have result=true 

389 return sensitive_data_detected and len(failed_detectors) == 0 

390 

391 def _extract_anonymized_content(self, response_json: dict, message_type: MessageRole) -> str | None: 

392 """ 

393 Extract anonymized content from Noma API response. 

394 

395 Args: 

396 response_json: The full response from Noma API 

397 message_type: Either 'user' or 'assistant' to determine which content to extract 

398 

399 Returns: 

400 The anonymized content string if available, None otherwise 

401 """ 

402 # Extract from new scanResult structure 

403 scan_result: Final = response_json.get("scanResult", []) 

404 if not scan_result: 

405 return None 

406 

407 # Find the scan result matching the message type (role) 

408 for result_item in scan_result: 

409 if result_item.get("role") == message_type: 

410 return result_item.get("results", {}).get("anonymizedContent", {}).get("anonymized", "") 

411 

412 return None 

413 

414 def _should_anonymize(self, response_json: dict, message_type: MessageRole) -> bool: 

415 """ 

416 Determine if content should be anonymized based on Noma API response. 

417 

418 Logic: 

419 - If aggregatedScanResult=False: Content is safe, anonymize if anonymized version exists 

420 - If aggregatedScanResult=True: Check if only sensitiveData detectors have result=True 

421 - If yes: Anonymize 

422 - If no: Block (other violations detected) 

423 

424 Args: 

425 response_json: The full response from Noma API 

426 message_type: Either 'user' or 'assistant' to determine which classification to check 

427 

428 Returns: 

429 True if content should be anonymized, False if it should be blocked 

430 """ 

431 # Only anonymize in blocking mode when anonymize_input is enabled 

432 if self.monitor_mode or not self.anonymize_input: 

433 return False 

434 

435 # aggregatedScanResult=False means safe, True means unsafe 

436 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False) 

437 

438 # If aggregatedScanResult is False, content is safe - anonymize if available 

439 if not aggregated_scan_result: 

440 return True 

441 

442 # If aggregatedScanResult is True (unsafe), check if only sensitive data detectors triggered 

443 scan_result: Final = response_json.get("scanResult", []) 

444 if not scan_result: 

445 return False 

446 

447 if not isinstance(scan_result, list) or len(scan_result) == 0: 

448 return False 

449 

450 for result_item in scan_result: 

451 if result_item.get("role") == message_type: 

452 return self._should_only_sensitive_data_failed(result_item.get("results", {})) 

453 

454 return False 

455 

456 def _is_result_true(self, result_obj: dict[str, object] | None) -> bool: 

457 """ 

458 Check if a result object has a "result" field that is True. 

459 

460 Args: 

461 result_obj: A dictionary that may contain a "result" field 

462 

463 Returns: 

464 True if the "result" field exists and is True, False otherwise 

465 """ 

466 if not result_obj or not isinstance(result_obj, dict): 

467 return False 

468 

469 return result_obj.get("result") is True 

470 

471 def _replace_user_message_content(self, request_data: dict, anonymized_content: str): 

472 """ 

473 Replace the user message content in request data with anonymized version. 

474 

475 Args: 

476 request_data: The original request data 

477 anonymized_content: The anonymized content to replace with 

478 """ 

479 messages: Final = request_data.get("messages", []) 

480 if not messages: 

481 return 

482 

483 # Find and replace the last user message 

484 for i in range(len(messages) - 1, -1, -1): 

485 if messages[i].get("role") == USER_ROLE: 

486 messages[i]["content"] = anonymized_content 

487 break 

488 

489 def _replace_llm_response_content(self, response: LLMResponse, anonymized_content: str): 

490 """ 

491 Replace the LLM response content with anonymized version. 

492 

493 Args: 

494 response: The original LLM response 

495 anonymized_content: The anonymized content to replace with 

496 """ 

497 if not isinstance(response, litellm.ModelResponse): 

498 return 

499 

500 # Replace content in all choices 

501 for choice in response.choices: 

502 if isinstance(choice, litellm.Choices) and choice.message.content: 

503 choice.message.content = anonymized_content 

504 

505 async def _check_user_message_background( 

506 self, 

507 request_data: dict, 

508 user_auth: UserAPIKeyAuth, 

509 ) -> None: 

510 """Check user message in background for monitor mode - non-blocking""" 

511 try: 

512 await self._process_user_message_check(request_data, user_auth) 

513 except Exception as e: 

514 verbose_proxy_logger.error("Noma background user message check failed: %s", e) 

515 

516 async def _check_llm_response_background( 

517 self, 

518 request_data: dict, 

519 response: LLMResponse, 

520 user_auth: UserAPIKeyAuth, 

521 ) -> None: 

522 """Check LLM response in background for monitor mode - non-blocking""" 

523 try: 

524 await self._process_llm_response_check(request_data, response, user_auth) 

525 except Exception as e: 

526 verbose_proxy_logger.error("Noma background response check failed: %s", e) 

527 

528 async def _handle_verdict_background( 

529 self, 

530 type: MessageRole, 

531 message: str, 

532 response_json: dict, 

533 ) -> None: 

534 """Handle aggregatedScanResult from Noma API in background - logging only, never blocks 

535 aggregatedScanResult=True means unsafe, False means safe 

536 """ 

537 try: 

538 # aggregatedScanResult=True means blocked, False means allowed 

539 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False) 

540 

541 if aggregated_scan_result: # True = unsafe 

542 msg = f"Noma guardrail blocked {type} message: {message}" 

543 verbose_proxy_logger.warning(msg) 

544 else: # False = safe 

545 msg = f"Noma guardrail allowed {type} message: {message}" 

546 verbose_proxy_logger.info(msg) 

547 except Exception as e: 

548 verbose_proxy_logger.error("Noma background verdict handling failed: %s", e) 

549 

550 async def async_pre_call_hook( 

551 self, 

552 user_api_key_dict: UserAPIKeyAuth, 

553 cache: DualCache, 

554 data: dict, 

555 call_type: CallTypesLiteral, 

556 ) -> Exception | str | dict | None: 

557 verbose_proxy_logger.debug("Running Noma pre-call hook") 

558 

559 event_type = GuardrailEventHooks.pre_call 

560 if call_type == CallTypes.call_mcp_tool.value: 

561 event_type = GuardrailEventHooks.pre_mcp_call 

562 

563 if self.should_run_guardrail(data=data, event_type=event_type) is False: 

564 return data 

565 

566 # In monitor mode, run Noma check in background and return immediately 

567 if self.monitor_mode: 

568 try: 

569 self._create_background_noma_check(self._check_user_message_background(data, user_api_key_dict)) 

570 except Exception as e: 

571 verbose_proxy_logger.error("Failed to start background Noma pre-call check: %s", e) 

572 return data 

573 

574 try: 

575 return await self._check_user_message(data, user_api_key_dict, GuardrailEventHooks.pre_call) 

576 except NomaBlockedMessage: 

577 # Blocked requests were already logged in _process_user_message_check with "blocked" status 

578 raise 

579 except Exception as e: 

580 # Log technical failures 

581 from datetime import datetime 

582 

583 start_time: Final = datetime.now() 

584 self.add_standard_logging_guardrail_information_to_request_data( 

585 guardrail_provider="noma", 

586 guardrail_json_response=str(e), 

587 request_data=data, 

588 guardrail_status="guardrail_failed_to_respond", 

589 start_time=start_time.timestamp(), 

590 end_time=start_time.timestamp(), 

591 duration=0.0, 

592 event_type=GuardrailEventHooks.pre_call, 

593 ) 

594 

595 verbose_proxy_logger.error("Noma pre-call hook failed: %s", e) 

596 

597 if self.block_failures: 

598 raise 

599 return data 

600 

601 async def async_moderation_hook( 

602 self, 

603 data: dict, 

604 user_api_key_dict: UserAPIKeyAuth, 

605 call_type: CallTypesLiteral, 

606 ) -> Exception | str | dict | None: 

607 event_type: GuardrailEventHooks = GuardrailEventHooks.during_call 

608 if call_type == CallTypes.call_mcp_tool.value: 

609 event_type = GuardrailEventHooks.pre_mcp_call 

610 

611 if self.should_run_guardrail(data=data, event_type=event_type) is not True: 

612 return data 

613 

614 # In monitor mode, run Noma check in background and return immediately 

615 if self.monitor_mode: 

616 try: 

617 self._create_background_noma_check(self._check_user_message_background(data, user_api_key_dict)) 

618 except Exception as e: 

619 verbose_proxy_logger.error("Failed to start background Noma moderation check: %s", e) 

620 return data 

621 

622 try: 

623 return await self._check_user_message(data, user_api_key_dict, GuardrailEventHooks.during_call) 

624 except NomaBlockedMessage: 

625 # Blocked requests were already logged in _process_user_message_check with "blocked" status 

626 raise 

627 except Exception as e: 

628 # Log technical failures 

629 from datetime import datetime 

630 

631 start_time: Final = datetime.now() 

632 self.add_standard_logging_guardrail_information_to_request_data( 

633 guardrail_provider="noma", 

634 guardrail_json_response=str(e), 

635 request_data=data, 

636 guardrail_status="guardrail_failed_to_respond", 

637 start_time=start_time.timestamp(), 

638 end_time=start_time.timestamp(), 

639 duration=0.0, 

640 event_type=GuardrailEventHooks.during_call, 

641 ) 

642 

643 verbose_proxy_logger.error("Noma moderation hook failed: %s", e) 

644 

645 if self.block_failures: 

646 raise 

647 return data 

648 

649 async def async_post_call_success_hook( 

650 self, 

651 data: dict, 

652 user_api_key_dict: UserAPIKeyAuth, 

653 response: LLMResponse, 

654 ): 

655 event_type: Final[GuardrailEventHooks] = GuardrailEventHooks.post_call 

656 if self.should_run_guardrail(data=data, event_type=event_type) is not True: 

657 return response 

658 

659 # In monitor mode, run Noma check in background and return immediately 

660 if self.monitor_mode: 

661 try: 

662 self._create_background_noma_check( 

663 self._check_llm_response_background(data, response, user_api_key_dict) 

664 ) 

665 except Exception as e: 

666 verbose_proxy_logger.error("Failed to start background Noma post-call check: %s", e) 

667 return response 

668 

669 try: 

670 return await self._check_llm_response(data, response, user_api_key_dict, GuardrailEventHooks.post_call) 

671 except NomaBlockedMessage: 

672 # Blocked requests were already logged in _process_llm_response_check with "blocked" status 

673 raise 

674 except Exception as e: 

675 # Log technical failures 

676 from datetime import datetime 

677 

678 start_time: Final = datetime.now() 

679 self.add_standard_logging_guardrail_information_to_request_data( 

680 guardrail_provider="noma", 

681 guardrail_json_response=str(e), 

682 request_data=data, 

683 guardrail_status="guardrail_failed_to_respond", 

684 start_time=start_time.timestamp(), 

685 end_time=start_time.timestamp(), 

686 duration=0.0, 

687 event_type=GuardrailEventHooks.post_call, 

688 ) 

689 

690 verbose_proxy_logger.error("Noma post-call hook failed: %s", e) 

691 if self.block_failures: 

692 raise 

693 return response 

694 

695 async def _check_user_message( 

696 self, 

697 request_data: dict, 

698 user_auth: UserAPIKeyAuth, 

699 event_type: GuardrailEventHooks | None = None, 

700 ) -> Exception | str | dict | None: 

701 """Check user message for policy violations""" 

702 user_message: Final = await self._process_user_message_check(request_data, user_auth, event_type) 

703 if not user_message: 

704 return request_data 

705 

706 return request_data 

707 

708 async def _check_llm_response( 

709 self, 

710 request_data: dict, 

711 response: _LLMResponseT, 

712 user_auth: UserAPIKeyAuth, 

713 event_type: GuardrailEventHooks | None = None, 

714 ) -> _LLMResponseT: 

715 """Check LLM response for policy violations""" 

716 content: Final = await self._process_llm_response_check(request_data, response, user_auth, event_type) 

717 if not content: 

718 return response 

719 

720 return response 

721 

722 async def _call_noma_api( 

723 self, 

724 payload: dict, 

725 llm_request_id: str | None, 

726 request_data: dict, 

727 user_auth: UserAPIKeyAuth, 

728 extra_data: dict, 

729 ) -> dict: 

730 call_id: Final = request_data.get("litellm_call_id") 

731 headers: Final = { 

732 **({"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}), 

733 **({"X-Noma-Request-ID": call_id} if call_id else {}), 

734 } 

735 endpoint: Final = urljoin(self.api_base or "https://api.noma.security/", NomaGuardrail._AIDR_ENDPOINT) 

736 

737 response: Final = await self.async_handler.post( 

738 endpoint, 

739 headers=headers, 

740 json={ 

741 **payload, 

742 "x-noma-context": { 

743 "applicationId": extra_data.get("application_id") 

744 or request_data.get("metadata", {}).get("headers", {}).get("x-noma-application-id") 

745 or self.application_id 

746 or user_auth.key_alias 

747 or self.default_application_id, 

748 "ipAddress": request_data.get("metadata", {}).get("requester_ip_address", None), 

749 "userId": (user_auth.user_email if user_auth.user_email else user_auth.user_id), 

750 "sessionId": call_id, 

751 "requestId": llm_request_id, 

752 }, 

753 }, 

754 ) 

755 response.raise_for_status() 

756 

757 return response.json() 

758 

759 async def _check_verdict( 

760 self, 

761 type: MessageRole, 

762 message: str, 

763 response_json: dict, 

764 ) -> None: 

765 """ 

766 Check the aggregatedScanResult from the Noma API and raise an exception if needed. 

767 aggregatedScanResult=True means unsafe (block), False means safe (allow) 

768 """ 

769 # aggregatedScanResult=True means blocked, False means allowed 

770 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False) 

771 

772 if aggregated_scan_result: # True = unsafe, block it 

773 msg = f"Noma guardrail blocked {type} message: {message}" 

774 

775 if self.monitor_mode: 

776 verbose_proxy_logger.warning(msg) 

777 else: 

778 verbose_proxy_logger.debug(msg) 

779 original_response: Final = response_json.get("scanResult", {}) 

780 # Use the full response as the original response for error details 

781 raise NomaBlockedMessage(original_response) 

782 else: # False = safe, allow it 

783 msg = f"Noma guardrail allowed {type} message: {message}" 

784 if self.monitor_mode: 

785 verbose_proxy_logger.info(msg) 

786 else: 

787 verbose_proxy_logger.debug(msg) 

788 

789 @staticmethod 

790 def get_config_model() -> type["GuardrailConfigModel"] | None: 

791 from litellm.types.proxy.guardrails.guardrail_hooks.noma import ( 

792 NomaGuardrailConfigModel, 

793 ) 

794 

795 return NomaGuardrailConfigModel 

796 

797 async def async_post_call_streaming_iterator_hook( 

798 self, 

799 user_api_key_dict: UserAPIKeyAuth, 

800 response: AsyncIterable[ModelResponseStream], 

801 request_data: dict, 

802 ) -> AsyncGenerator[ModelResponseStream, None]: 

803 """Process streaming response chunks with Noma guardrail.""" 

804 

805 all_chunks: Final[list[ModelResponseStream]] = [] 

806 async for chunk in response: 

807 all_chunks.append(chunk) 

808 

809 if not all_chunks: 

810 return 

811 

812 assembled_model_response: Final[ModelResponse | TextCompletionResponse | None] = stream_chunk_builder( 

813 chunks=all_chunks 

814 ) 

815 

816 if isinstance(assembled_model_response, ModelResponse): 

817 try: 

818 processed_response: Final = await self._check_llm_response( 

819 request_data, 

820 assembled_model_response, 

821 user_api_key_dict, 

822 GuardrailEventHooks.post_call, 

823 ) 

824 except NomaBlockedMessage: 

825 raise 

826 except Exception as e: 

827 if self.block_failures: 

828 raise 

829 verbose_proxy_logger.error("Noma streaming post-call hook failed: %s", e) 

830 for chunk in all_chunks: 

831 yield chunk 

832 return 

833 

834 mock_response: Final = MockResponseIterator(model_response=processed_response) 

835 async for chunk in mock_response: 

836 yield chunk 

837 return 

838 

839 for chunk in all_chunks: 

840 yield chunk