Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/noma/noma.py: 13%
348 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1# +-------------------------------------------------------------+
2#
3# Noma Security Guardrail Integration for LiteLLM
4# https://noma.security
5#
6# +-------------------------------------------------------------+
8import asyncio
9import json
10import os
11import warnings
12from collections.abc import AsyncGenerator, AsyncIterable
13from datetime import datetime
14from typing import (
15 TYPE_CHECKING,
16 Final,
17 Literal,
18 TypeVar,
19)
20from urllib.parse import urljoin
22from fastapi import HTTPException
24import litellm
25from litellm import DualCache, ModelResponse
26from litellm._logging import verbose_proxy_logger
27from litellm.completion_extras.litellm_responses_transformation.transformation import (
28 LiteLLMResponsesTransformationHandler,
29)
30from litellm.integrations.custom_guardrail import CustomGuardrail
31from litellm.llms.base_llm.base_model_iterator import MockResponseIterator
32from litellm.llms.custom_httpx.http_handler import (
33 get_async_httpx_client,
34 httpxSpecialProvider,
35)
36from litellm.main import stream_chunk_builder
37from litellm.proxy._types import UserAPIKeyAuth
38from litellm.types.guardrails import GuardrailEventHooks
39from litellm.types.utils import (
40 CallTypes,
41 CallTypesLiteral,
42 GuardrailStatus,
43 ModelResponseStream,
44 TextCompletionResponse,
45)
47# Constants
48USER_ROLE: Final = "user"
49ASSISTANT_ROLE: Final = "assistant"
50SENSITIVE_DATA_DETECTOR_KEYS: Final[list[str]] = ["sensitiveData", "dataDetector"]
52# Type aliases
53MessageRole = Literal["user", "assistant"]
54LLMResponse = object
55_LLMResponseT: Final = TypeVar("_LLMResponseT")
56_LEGACY_NOMA_DEPRECATION_WARNED = False
58if TYPE_CHECKING: 58 ↛ 59line 58 didn't jump to line 59 because the condition on line 58 was never true
59 from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
62class NomaBlockedMessage(HTTPException):
63 """Exception raised when Noma guardrail blocks a message"""
65 def __init__(self, classification_response: dict):
66 super().__init__(
67 status_code=400,
68 detail={
69 "error": "Request blocked by Noma guardrail",
70 "details": classification_response,
71 },
72 )
74 def _is_result_true(self, result_obj: dict[str, object] | None) -> bool:
75 """
76 Check if a result object has a "result" field that is True.
78 Args:
79 result_obj: A dictionary that may contain a "result" field
81 Returns:
82 True if the "result" field exists and is True, False otherwise
83 """
84 if not result_obj or not isinstance(result_obj, dict):
85 return False
87 return result_obj.get("result") is True
90class NomaGuardrail(CustomGuardrail):
91 """
92 Noma Security Guardrail for LiteLLM
94 This guardrail integrates with Noma Security's AI-DR API to provide
95 content moderation and safety checks for LLM inputs and outputs.
96 """
98 _DEFAULT_API_BASE = "https://api.noma.security/"
99 _AIDR_ENDPOINT = "/ai-dr/v2/prompt/scan"
101 @classmethod
102 def get_supported_event_hooks(cls) -> list[GuardrailEventHooks]:
103 return [
104 GuardrailEventHooks.pre_call,
105 GuardrailEventHooks.during_call,
106 GuardrailEventHooks.post_call,
107 GuardrailEventHooks.pre_mcp_call,
108 ]
110 def __init__(
111 self,
112 api_key: str | None = None,
113 api_base: str | None = None,
114 application_id: str | None = None,
115 monitor_mode: bool | None = None,
116 block_failures: bool | None = None,
117 anonymize_input: bool | None = None,
118 **kwargs,
119 ):
120 global _LEGACY_NOMA_DEPRECATION_WARNED
121 if not _LEGACY_NOMA_DEPRECATION_WARNED:
122 warnings.warn(
123 "Guardrail provider 'noma' is deprecated. "
124 "Please migrate to 'noma_v2'. "
125 "The legacy 'noma' API will no longer be supported after March 31, 2026.",
126 DeprecationWarning,
127 stacklevel=2,
128 )
129 _LEGACY_NOMA_DEPRECATION_WARNED = True
131 self.async_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.GuardrailCallback)
132 self._responses_transform_handler = LiteLLMResponsesTransformationHandler()
133 self.api_key = api_key or os.environ.get("NOMA_API_KEY")
134 self.api_base = api_base or os.environ.get("NOMA_API_BASE", NomaGuardrail._DEFAULT_API_BASE)
135 self.application_id = application_id or os.environ.get("NOMA_APPLICATION_ID")
136 self.default_application_id = "litellm"
138 if monitor_mode is None:
139 self.monitor_mode = os.environ.get("NOMA_MONITOR_MODE", "false").lower() == "true"
140 else:
141 self.monitor_mode = monitor_mode
143 if block_failures is None:
144 self.block_failures = os.environ.get("NOMA_BLOCK_FAILURES", "true").lower() == "true"
145 else:
146 self.block_failures = block_failures
148 if anonymize_input is None:
149 self.anonymize_input = os.environ.get("NOMA_ANONYMIZE_INPUT", "false").lower() == "true"
150 else:
151 self.anonymize_input = anonymize_input
153 kwargs.setdefault("supported_event_hooks", list(self.get_supported_event_hooks()))
154 super().__init__(**kwargs)
156 def _create_background_noma_check(
157 self,
158 coro,
159 ) -> None:
160 """Create a background task for Noma API calls without blocking the main flow"""
161 try:
162 asyncio.create_task(coro)
163 except Exception as e:
164 verbose_proxy_logger.error("Failed to create background Noma task: %s", e)
166 async def _process_user_message_check(
167 self,
168 request_data: dict,
169 user_auth: UserAPIKeyAuth,
170 event_type: GuardrailEventHooks | None = None,
171 ) -> str | None:
172 """Shared logic for processing user message checks"""
173 start_time: Final = datetime.now()
174 extra_data: Final = self.get_guardrail_dynamic_request_body_params(request_data)
176 messages: Final = request_data.get("messages") or []
177 if not messages:
178 return None
180 input_items, instructions = self._responses_transform_handler.convert_chat_completion_messages_to_responses_api(
181 messages
182 )
184 if instructions:
185 system_message: Final = {
186 "type": "message",
187 "role": "system",
188 "content": [
189 {"type": "input_text", "text": instructions},
190 ],
191 }
192 input_items.insert(0, system_message)
194 if not input_items:
195 return None
197 payload: Final = {"input": input_items}
198 response_json: Final = await self._call_noma_api(
199 payload=payload,
200 llm_request_id=None,
201 request_data=request_data,
202 user_auth=user_auth,
203 extra_data=extra_data,
204 )
206 end_time: Final = datetime.now()
207 duration: Final = (end_time - start_time).total_seconds()
209 # Determine guardrail status based on response
210 guardrail_status: Final = self._determine_guardrail_status(response_json)
212 # Always log guardrail information for consistency
213 self.add_standard_logging_guardrail_information_to_request_data(
214 guardrail_provider="noma",
215 guardrail_json_response=response_json,
216 request_data=request_data,
217 guardrail_status=guardrail_status,
218 start_time=start_time.timestamp(),
219 end_time=end_time.timestamp(),
220 duration=duration,
221 event_type=event_type,
222 )
224 if self.monitor_mode:
225 await self._handle_verdict_background(USER_ROLE, json.dumps(input_items), response_json)
226 return json.dumps(input_items)
228 # Check if we should anonymize content
229 if self._should_anonymize(response_json, USER_ROLE):
230 anonymized_content: Final = self._extract_anonymized_content(response_json, USER_ROLE)
231 if anonymized_content:
232 # Replace the user message content with anonymized version
233 self._replace_user_message_content(request_data, anonymized_content)
234 verbose_proxy_logger.debug("Noma guardrail anonymized user message: %s", anonymized_content)
235 return anonymized_content
237 await self._check_verdict(USER_ROLE, json.dumps(input_items), response_json)
238 return json.dumps(input_items)
240 async def _process_llm_response_check(
241 self,
242 request_data: dict,
243 response: LLMResponse,
244 user_auth: UserAPIKeyAuth,
245 event_type: GuardrailEventHooks | None = None,
246 ) -> str | None:
247 """Shared logic for processing LLM response checks"""
249 start_time: Final = datetime.now()
250 extra_data: Final = self.get_guardrail_dynamic_request_body_params(request_data)
252 if not isinstance(response, litellm.ModelResponse):
253 return None
255 content = None
256 for choice in response.choices:
257 if isinstance(choice, litellm.Choices) and choice.message.content:
258 content = choice.message.content
259 break
261 if not content or not isinstance(content, str):
262 return None
264 payload: Final = {
265 "input": [
266 {
267 "type": "message",
268 "role": "assistant",
269 "content": [{"type": "input_text", "text": content}],
270 }
271 ]
272 }
274 response_json: Final = await self._call_noma_api(
275 payload=payload,
276 llm_request_id=response.id,
277 request_data=request_data,
278 user_auth=user_auth,
279 extra_data=extra_data,
280 )
282 end_time: Final = datetime.now()
283 duration: Final = (end_time - start_time).total_seconds()
285 # Determine guardrail status based on response
286 guardrail_status: Final = self._determine_guardrail_status(response_json)
288 # Always log guardrail information for consistency
289 self.add_standard_logging_guardrail_information_to_request_data(
290 guardrail_provider="noma",
291 guardrail_json_response=response_json,
292 request_data=request_data,
293 guardrail_status=guardrail_status,
294 start_time=start_time.timestamp(),
295 end_time=end_time.timestamp(),
296 duration=duration,
297 event_type=event_type,
298 )
300 if self.monitor_mode:
301 await self._handle_verdict_background(ASSISTANT_ROLE, json.dumps(content), response_json)
302 return content
304 # Check if we should anonymize content
305 if self._should_anonymize(response_json, ASSISTANT_ROLE):
306 anonymized_content: Final = self._extract_anonymized_content(response_json, ASSISTANT_ROLE)
307 if anonymized_content:
308 # Replace the LLM response content with anonymized version
309 self._replace_llm_response_content(response, anonymized_content)
310 verbose_proxy_logger.debug("Noma guardrail anonymized LLM response: %s", anonymized_content)
311 return anonymized_content
313 await self._check_verdict(ASSISTANT_ROLE, content, response_json)
314 return content
316 def _determine_guardrail_status(self, response_json: dict) -> GuardrailStatus:
317 """
318 Determine the guardrail status based on NOMA API response.
320 Args:
321 response_json: Response from NOMA API
323 Returns:
324 "success": Content allowed through with no violations
325 "guardrail_intervened": Content blocked due to policy violations
326 "guardrail_failed_to_respond": Technical error or API failure
327 """
328 try:
329 # Check if we got a valid response structure
330 if not isinstance(response_json, dict):
331 return "guardrail_failed_to_respond"
333 # Get the aggregatedScanResult from the response
334 # aggregatedScanResult=True means unsafe (block), False means safe (allow)
335 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False)
337 # If aggregatedScanResult is False, content is safe/allowed
338 if aggregated_scan_result is False:
339 return "success"
341 # If aggregatedScanResult is True, content is blocked/flagged
342 if aggregated_scan_result is True:
343 return "guardrail_intervened"
345 # If aggregatedScanResult is missing or invalid, treat as failure
346 return "guardrail_failed_to_respond"
348 except Exception as e:
349 verbose_proxy_logger.error("Error determining NOMA guardrail status: %s", e)
350 return "guardrail_failed_to_respond"
352 def _should_only_sensitive_data_failed(self, classification_obj: dict) -> bool:
353 """
354 Check if only sensitive data detectors (PII, PCI, secrets) have result=true in the classification.
356 Args:
357 classification_obj: The prompt or response classification object from Noma API
359 Returns:
360 True if only sensitiveData detectors have result=true, False otherwise
361 """
362 if not classification_obj:
363 return False
365 # Track which detectors have result=true (detected violations)
366 failed_detectors: Final = []
367 sensitive_data_detected = False
369 for key, value in classification_obj.items():
370 if key in SENSITIVE_DATA_DETECTOR_KEYS and isinstance(value, dict):
371 # Check if any sensitive data detector has result=true
372 for data_type, data_result in value.items():
373 if self._is_result_true(data_result):
374 sensitive_data_detected = True
375 # Don't add to failed_detectors as we want to allow these
377 elif isinstance(value, dict) and "result" in value:
378 # Check other detectors - these should NOT have result=true
379 if self._is_result_true(value):
380 failed_detectors.append(key)
382 elif isinstance(value, dict):
383 # Handle nested detectors
384 for nested_key, nested_value in value.items():
385 if self._is_result_true(nested_value):
386 failed_detectors.append(f"{key}.{nested_key}")
388 # Return True only if sensitive data was detected AND no other detectors have result=true
389 return sensitive_data_detected and len(failed_detectors) == 0
391 def _extract_anonymized_content(self, response_json: dict, message_type: MessageRole) -> str | None:
392 """
393 Extract anonymized content from Noma API response.
395 Args:
396 response_json: The full response from Noma API
397 message_type: Either 'user' or 'assistant' to determine which content to extract
399 Returns:
400 The anonymized content string if available, None otherwise
401 """
402 # Extract from new scanResult structure
403 scan_result: Final = response_json.get("scanResult", [])
404 if not scan_result:
405 return None
407 # Find the scan result matching the message type (role)
408 for result_item in scan_result:
409 if result_item.get("role") == message_type:
410 return result_item.get("results", {}).get("anonymizedContent", {}).get("anonymized", "")
412 return None
414 def _should_anonymize(self, response_json: dict, message_type: MessageRole) -> bool:
415 """
416 Determine if content should be anonymized based on Noma API response.
418 Logic:
419 - If aggregatedScanResult=False: Content is safe, anonymize if anonymized version exists
420 - If aggregatedScanResult=True: Check if only sensitiveData detectors have result=True
421 - If yes: Anonymize
422 - If no: Block (other violations detected)
424 Args:
425 response_json: The full response from Noma API
426 message_type: Either 'user' or 'assistant' to determine which classification to check
428 Returns:
429 True if content should be anonymized, False if it should be blocked
430 """
431 # Only anonymize in blocking mode when anonymize_input is enabled
432 if self.monitor_mode or not self.anonymize_input:
433 return False
435 # aggregatedScanResult=False means safe, True means unsafe
436 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False)
438 # If aggregatedScanResult is False, content is safe - anonymize if available
439 if not aggregated_scan_result:
440 return True
442 # If aggregatedScanResult is True (unsafe), check if only sensitive data detectors triggered
443 scan_result: Final = response_json.get("scanResult", [])
444 if not scan_result:
445 return False
447 if not isinstance(scan_result, list) or len(scan_result) == 0:
448 return False
450 for result_item in scan_result:
451 if result_item.get("role") == message_type:
452 return self._should_only_sensitive_data_failed(result_item.get("results", {}))
454 return False
456 def _is_result_true(self, result_obj: dict[str, object] | None) -> bool:
457 """
458 Check if a result object has a "result" field that is True.
460 Args:
461 result_obj: A dictionary that may contain a "result" field
463 Returns:
464 True if the "result" field exists and is True, False otherwise
465 """
466 if not result_obj or not isinstance(result_obj, dict):
467 return False
469 return result_obj.get("result") is True
471 def _replace_user_message_content(self, request_data: dict, anonymized_content: str):
472 """
473 Replace the user message content in request data with anonymized version.
475 Args:
476 request_data: The original request data
477 anonymized_content: The anonymized content to replace with
478 """
479 messages: Final = request_data.get("messages", [])
480 if not messages:
481 return
483 # Find and replace the last user message
484 for i in range(len(messages) - 1, -1, -1):
485 if messages[i].get("role") == USER_ROLE:
486 messages[i]["content"] = anonymized_content
487 break
489 def _replace_llm_response_content(self, response: LLMResponse, anonymized_content: str):
490 """
491 Replace the LLM response content with anonymized version.
493 Args:
494 response: The original LLM response
495 anonymized_content: The anonymized content to replace with
496 """
497 if not isinstance(response, litellm.ModelResponse):
498 return
500 # Replace content in all choices
501 for choice in response.choices:
502 if isinstance(choice, litellm.Choices) and choice.message.content:
503 choice.message.content = anonymized_content
505 async def _check_user_message_background(
506 self,
507 request_data: dict,
508 user_auth: UserAPIKeyAuth,
509 ) -> None:
510 """Check user message in background for monitor mode - non-blocking"""
511 try:
512 await self._process_user_message_check(request_data, user_auth)
513 except Exception as e:
514 verbose_proxy_logger.error("Noma background user message check failed: %s", e)
516 async def _check_llm_response_background(
517 self,
518 request_data: dict,
519 response: LLMResponse,
520 user_auth: UserAPIKeyAuth,
521 ) -> None:
522 """Check LLM response in background for monitor mode - non-blocking"""
523 try:
524 await self._process_llm_response_check(request_data, response, user_auth)
525 except Exception as e:
526 verbose_proxy_logger.error("Noma background response check failed: %s", e)
528 async def _handle_verdict_background(
529 self,
530 type: MessageRole,
531 message: str,
532 response_json: dict,
533 ) -> None:
534 """Handle aggregatedScanResult from Noma API in background - logging only, never blocks
535 aggregatedScanResult=True means unsafe, False means safe
536 """
537 try:
538 # aggregatedScanResult=True means blocked, False means allowed
539 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False)
541 if aggregated_scan_result: # True = unsafe
542 msg = f"Noma guardrail blocked {type} message: {message}"
543 verbose_proxy_logger.warning(msg)
544 else: # False = safe
545 msg = f"Noma guardrail allowed {type} message: {message}"
546 verbose_proxy_logger.info(msg)
547 except Exception as e:
548 verbose_proxy_logger.error("Noma background verdict handling failed: %s", e)
550 async def async_pre_call_hook(
551 self,
552 user_api_key_dict: UserAPIKeyAuth,
553 cache: DualCache,
554 data: dict,
555 call_type: CallTypesLiteral,
556 ) -> Exception | str | dict | None:
557 verbose_proxy_logger.debug("Running Noma pre-call hook")
559 event_type = GuardrailEventHooks.pre_call
560 if call_type == CallTypes.call_mcp_tool.value:
561 event_type = GuardrailEventHooks.pre_mcp_call
563 if self.should_run_guardrail(data=data, event_type=event_type) is False:
564 return data
566 # In monitor mode, run Noma check in background and return immediately
567 if self.monitor_mode:
568 try:
569 self._create_background_noma_check(self._check_user_message_background(data, user_api_key_dict))
570 except Exception as e:
571 verbose_proxy_logger.error("Failed to start background Noma pre-call check: %s", e)
572 return data
574 try:
575 return await self._check_user_message(data, user_api_key_dict, GuardrailEventHooks.pre_call)
576 except NomaBlockedMessage:
577 # Blocked requests were already logged in _process_user_message_check with "blocked" status
578 raise
579 except Exception as e:
580 # Log technical failures
581 from datetime import datetime
583 start_time: Final = datetime.now()
584 self.add_standard_logging_guardrail_information_to_request_data(
585 guardrail_provider="noma",
586 guardrail_json_response=str(e),
587 request_data=data,
588 guardrail_status="guardrail_failed_to_respond",
589 start_time=start_time.timestamp(),
590 end_time=start_time.timestamp(),
591 duration=0.0,
592 event_type=GuardrailEventHooks.pre_call,
593 )
595 verbose_proxy_logger.error("Noma pre-call hook failed: %s", e)
597 if self.block_failures:
598 raise
599 return data
601 async def async_moderation_hook(
602 self,
603 data: dict,
604 user_api_key_dict: UserAPIKeyAuth,
605 call_type: CallTypesLiteral,
606 ) -> Exception | str | dict | None:
607 event_type: GuardrailEventHooks = GuardrailEventHooks.during_call
608 if call_type == CallTypes.call_mcp_tool.value:
609 event_type = GuardrailEventHooks.pre_mcp_call
611 if self.should_run_guardrail(data=data, event_type=event_type) is not True:
612 return data
614 # In monitor mode, run Noma check in background and return immediately
615 if self.monitor_mode:
616 try:
617 self._create_background_noma_check(self._check_user_message_background(data, user_api_key_dict))
618 except Exception as e:
619 verbose_proxy_logger.error("Failed to start background Noma moderation check: %s", e)
620 return data
622 try:
623 return await self._check_user_message(data, user_api_key_dict, GuardrailEventHooks.during_call)
624 except NomaBlockedMessage:
625 # Blocked requests were already logged in _process_user_message_check with "blocked" status
626 raise
627 except Exception as e:
628 # Log technical failures
629 from datetime import datetime
631 start_time: Final = datetime.now()
632 self.add_standard_logging_guardrail_information_to_request_data(
633 guardrail_provider="noma",
634 guardrail_json_response=str(e),
635 request_data=data,
636 guardrail_status="guardrail_failed_to_respond",
637 start_time=start_time.timestamp(),
638 end_time=start_time.timestamp(),
639 duration=0.0,
640 event_type=GuardrailEventHooks.during_call,
641 )
643 verbose_proxy_logger.error("Noma moderation hook failed: %s", e)
645 if self.block_failures:
646 raise
647 return data
649 async def async_post_call_success_hook(
650 self,
651 data: dict,
652 user_api_key_dict: UserAPIKeyAuth,
653 response: LLMResponse,
654 ):
655 event_type: Final[GuardrailEventHooks] = GuardrailEventHooks.post_call
656 if self.should_run_guardrail(data=data, event_type=event_type) is not True:
657 return response
659 # In monitor mode, run Noma check in background and return immediately
660 if self.monitor_mode:
661 try:
662 self._create_background_noma_check(
663 self._check_llm_response_background(data, response, user_api_key_dict)
664 )
665 except Exception as e:
666 verbose_proxy_logger.error("Failed to start background Noma post-call check: %s", e)
667 return response
669 try:
670 return await self._check_llm_response(data, response, user_api_key_dict, GuardrailEventHooks.post_call)
671 except NomaBlockedMessage:
672 # Blocked requests were already logged in _process_llm_response_check with "blocked" status
673 raise
674 except Exception as e:
675 # Log technical failures
676 from datetime import datetime
678 start_time: Final = datetime.now()
679 self.add_standard_logging_guardrail_information_to_request_data(
680 guardrail_provider="noma",
681 guardrail_json_response=str(e),
682 request_data=data,
683 guardrail_status="guardrail_failed_to_respond",
684 start_time=start_time.timestamp(),
685 end_time=start_time.timestamp(),
686 duration=0.0,
687 event_type=GuardrailEventHooks.post_call,
688 )
690 verbose_proxy_logger.error("Noma post-call hook failed: %s", e)
691 if self.block_failures:
692 raise
693 return response
695 async def _check_user_message(
696 self,
697 request_data: dict,
698 user_auth: UserAPIKeyAuth,
699 event_type: GuardrailEventHooks | None = None,
700 ) -> Exception | str | dict | None:
701 """Check user message for policy violations"""
702 user_message: Final = await self._process_user_message_check(request_data, user_auth, event_type)
703 if not user_message:
704 return request_data
706 return request_data
708 async def _check_llm_response(
709 self,
710 request_data: dict,
711 response: _LLMResponseT,
712 user_auth: UserAPIKeyAuth,
713 event_type: GuardrailEventHooks | None = None,
714 ) -> _LLMResponseT:
715 """Check LLM response for policy violations"""
716 content: Final = await self._process_llm_response_check(request_data, response, user_auth, event_type)
717 if not content:
718 return response
720 return response
722 async def _call_noma_api(
723 self,
724 payload: dict,
725 llm_request_id: str | None,
726 request_data: dict,
727 user_auth: UserAPIKeyAuth,
728 extra_data: dict,
729 ) -> dict:
730 call_id: Final = request_data.get("litellm_call_id")
731 headers: Final = {
732 **({"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}),
733 **({"X-Noma-Request-ID": call_id} if call_id else {}),
734 }
735 endpoint: Final = urljoin(self.api_base or "https://api.noma.security/", NomaGuardrail._AIDR_ENDPOINT)
737 response: Final = await self.async_handler.post(
738 endpoint,
739 headers=headers,
740 json={
741 **payload,
742 "x-noma-context": {
743 "applicationId": extra_data.get("application_id")
744 or request_data.get("metadata", {}).get("headers", {}).get("x-noma-application-id")
745 or self.application_id
746 or user_auth.key_alias
747 or self.default_application_id,
748 "ipAddress": request_data.get("metadata", {}).get("requester_ip_address", None),
749 "userId": (user_auth.user_email if user_auth.user_email else user_auth.user_id),
750 "sessionId": call_id,
751 "requestId": llm_request_id,
752 },
753 },
754 )
755 response.raise_for_status()
757 return response.json()
759 async def _check_verdict(
760 self,
761 type: MessageRole,
762 message: str,
763 response_json: dict,
764 ) -> None:
765 """
766 Check the aggregatedScanResult from the Noma API and raise an exception if needed.
767 aggregatedScanResult=True means unsafe (block), False means safe (allow)
768 """
769 # aggregatedScanResult=True means blocked, False means allowed
770 aggregated_scan_result: Final = response_json.get("aggregatedScanResult", False)
772 if aggregated_scan_result: # True = unsafe, block it
773 msg = f"Noma guardrail blocked {type} message: {message}"
775 if self.monitor_mode:
776 verbose_proxy_logger.warning(msg)
777 else:
778 verbose_proxy_logger.debug(msg)
779 original_response: Final = response_json.get("scanResult", {})
780 # Use the full response as the original response for error details
781 raise NomaBlockedMessage(original_response)
782 else: # False = safe, allow it
783 msg = f"Noma guardrail allowed {type} message: {message}"
784 if self.monitor_mode:
785 verbose_proxy_logger.info(msg)
786 else:
787 verbose_proxy_logger.debug(msg)
789 @staticmethod
790 def get_config_model() -> type["GuardrailConfigModel"] | None:
791 from litellm.types.proxy.guardrails.guardrail_hooks.noma import (
792 NomaGuardrailConfigModel,
793 )
795 return NomaGuardrailConfigModel
797 async def async_post_call_streaming_iterator_hook(
798 self,
799 user_api_key_dict: UserAPIKeyAuth,
800 response: AsyncIterable[ModelResponseStream],
801 request_data: dict,
802 ) -> AsyncGenerator[ModelResponseStream, None]:
803 """Process streaming response chunks with Noma guardrail."""
805 all_chunks: Final[list[ModelResponseStream]] = []
806 async for chunk in response:
807 all_chunks.append(chunk)
809 if not all_chunks:
810 return
812 assembled_model_response: Final[ModelResponse | TextCompletionResponse | None] = stream_chunk_builder(
813 chunks=all_chunks
814 )
816 if isinstance(assembled_model_response, ModelResponse):
817 try:
818 processed_response: Final = await self._check_llm_response(
819 request_data,
820 assembled_model_response,
821 user_api_key_dict,
822 GuardrailEventHooks.post_call,
823 )
824 except NomaBlockedMessage:
825 raise
826 except Exception as e:
827 if self.block_failures:
828 raise
829 verbose_proxy_logger.error("Noma streaming post-call hook failed: %s", e)
830 for chunk in all_chunks:
831 yield chunk
832 return
834 mock_response: Final = MockResponseIterator(model_response=processed_response)
835 async for chunk in mock_response:
836 yield chunk
837 return
839 for chunk in all_chunks:
840 yield chunk