Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/grayswan/grayswan.py: 15%

286 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1"""Gray Swan Cygnal guardrail integration.""" 

2 

3import os 

4import time 

5from typing import TYPE_CHECKING, Final, Literal, Optional, Protocol 

6 

7from fastapi import HTTPException 

8from typing_extensions import NotRequired, ReadOnly, TypedDict, Unpack 

9 

10from litellm._logging import verbose_proxy_logger 

11from litellm.integrations.custom_guardrail import ( 

12 CustomGuardrail, 

13 ModifyResponseException, 

14 log_guardrail_information, 

15) 

16from litellm.litellm_core_utils.safe_json_dumps import safe_dumps 

17from litellm.litellm_core_utils.safe_json_loads import safe_json_loads 

18from litellm.llms.custom_httpx.http_handler import ( 

19 get_async_httpx_client, 

20 httpxSpecialProvider, 

21) 

22from litellm.types.guardrails import GuardrailEventHooks 

23from litellm.types.utils import GenericGuardrailAPIInputs 

24 

25if TYPE_CHECKING: 25 ↛ 26line 25 didn't jump to line 26 because the condition on line 25 was never true

26 from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj 

27 

28GRAYSWAN_BLOCK_ERROR_MSG: Final = "Blocked by Gray Swan Guardrail" 

29 

30 

31class _GraySwanMonitorResponse(TypedDict): 

32 """Body returned by Gray Swan's `/cygnal/monitor` endpoint.""" 

33 

34 violation: ReadOnly[NotRequired[float | None]] 

35 violated_rules: ReadOnly[NotRequired[list[object]]] 

36 violated_rule_descriptions: ReadOnly[NotRequired[list[object]]] 

37 mutation: ReadOnly[NotRequired[bool | None]] 

38 ipi: ReadOnly[NotRequired[bool | None]] 

39 

40 

41class _CustomGuardrailOptions(TypedDict, total=False, extra_items=object): 

42 pass 

43 

44 

45class _GraySwanMonitorHTTPResponse(Protocol): 

46 def raise_for_status(self) -> object: ... 46 ↛ exitline 46 didn't return from function 'raise_for_status' because

47 

48 def json(self) -> _GraySwanMonitorResponse: ... 48 ↛ exitline 48 didn't return from function 'json' because

49 

50 

51class _GraySwanMonitorHTTPClient(Protocol): 

52 async def post( 52 ↛ exitline 52 didn't return from function 'post' because

53 self, 

54 *, 

55 url: str, 

56 headers: dict[str, str], 

57 json: dict[str, object], 

58 timeout: float, 

59 ) -> _GraySwanMonitorHTTPResponse: ... 

60 

61 

62class GraySwanGuardrailMissingSecrets(Exception): 

63 """Raised when the Gray Swan API key is missing.""" 

64 

65 

66class GraySwanGuardrailAPIError(Exception): 

67 """Raised when the Gray Swan API returns an error.""" 

68 

69 def __init__(self, message: str, status_code: int | None = None) -> None: 

70 super().__init__(message) 

71 self.status_code = status_code 

72 

73 

74class GraySwanGuardrail(CustomGuardrail): 

75 """ 

76 Guardrail that calls Gray Swan's Cygnal monitoring endpoint. 

77 

78 Uses the unified guardrail system via `apply_guardrail` method, 

79 which automatically works with all LiteLLM endpoints: 

80 - OpenAI Chat Completions 

81 - OpenAI Responses API 

82 - OpenAI Text Completions 

83 - Anthropic Messages 

84 - Image Generation 

85 - And more... 

86 

87 see: https://docs.grayswan.ai/cygnal/monitor-requests 

88 """ 

89 

90 SUPPORTED_ON_FLAGGED_ACTIONS = {"block", "monitor", "passthrough"} 

91 DEFAULT_ON_FLAGGED_ACTION = "monitor" 

92 BASE_API_URL = "https://api.grayswan.ai" 

93 MONITOR_PATH = "/cygnal/monitor" 

94 SUPPORTED_REASONING_MODES = {"off", "hybrid", "thinking"} 

95 

96 def __init__( 

97 self, 

98 guardrail_name: str | None = "grayswan", 

99 api_key: str | None = None, 

100 api_base: str | None = None, 

101 on_flagged_action: str | None = None, 

102 violation_threshold: float | None = None, 

103 reasoning_mode: str | None = None, 

104 categories: dict[str, str] | None = None, 

105 policy_id: str | None = None, 

106 streaming_end_of_stream_only: bool = False, 

107 streaming_sampling_rate: int = 5, 

108 fail_open: bool | None = True, 

109 guardrail_timeout: float | None = 30.0, 

110 **kwargs: Unpack[_CustomGuardrailOptions], 

111 ) -> None: 

112 self.async_handler: _GraySwanMonitorHTTPClient = get_async_httpx_client( 

113 llm_provider=httpxSpecialProvider.GuardrailCallback 

114 ) 

115 

116 api_key_value: Final = api_key or os.getenv("GRAYSWAN_API_KEY") 

117 if not api_key_value: 

118 raise GraySwanGuardrailMissingSecrets( 

119 "Gray Swan API key missing. Set `GRAYSWAN_API_KEY` or pass `api_key`." 

120 ) 

121 self.api_key: str = api_key_value 

122 

123 base: Final = api_base or os.getenv("GRAYSWAN_API_BASE") or self.BASE_API_URL 

124 self.api_base = base.rstrip("/") 

125 self.monitor_url = f"{self.api_base}{self.MONITOR_PATH}" 

126 

127 action: Final = on_flagged_action 

128 if action and action.lower() in self.SUPPORTED_ON_FLAGGED_ACTIONS: 

129 self.on_flagged_action = action.lower() 

130 else: 

131 if action: 

132 verbose_proxy_logger.warning( 

133 "Gray Swan Guardrail: Unsupported on_flagged_action '%s', defaulting to '%s'.", 

134 action, 

135 self.DEFAULT_ON_FLAGGED_ACTION, 

136 ) 

137 self.on_flagged_action = self.DEFAULT_ON_FLAGGED_ACTION 

138 

139 self.violation_threshold = self._resolve_threshold(violation_threshold) 

140 self.reasoning_mode = self._resolve_reasoning_mode(reasoning_mode) 

141 self.categories = categories 

142 self.policy_id = policy_id 

143 self.fail_open = True if fail_open is None else bool(fail_open) 

144 self.guardrail_timeout = 30.0 if guardrail_timeout is None else float(guardrail_timeout) 

145 

146 # Streaming configuration 

147 self.streaming_end_of_stream_only = streaming_end_of_stream_only 

148 self.streaming_sampling_rate = streaming_sampling_rate 

149 

150 verbose_proxy_logger.debug( 

151 "GraySwan __init__: streaming_end_of_stream_only=%s, streaming_sampling_rate=%s", 

152 streaming_end_of_stream_only, 

153 streaming_sampling_rate, 

154 ) 

155 

156 super().__init__( 

157 guardrail_name=guardrail_name, 

158 supported_event_hooks=list(self.get_supported_event_hooks()), 

159 **kwargs, 

160 ) 

161 

162 @classmethod 

163 def get_supported_event_hooks(cls) -> list[GuardrailEventHooks]: 

164 return [ 

165 GuardrailEventHooks.pre_call, 

166 GuardrailEventHooks.during_call, 

167 GuardrailEventHooks.post_call, 

168 ] 

169 

170 # ------------------------------------------------------------------ 

171 # Debug override to trace post_call issues 

172 # ------------------------------------------------------------------ 

173 

174 def should_run_guardrail(self, data, event_type) -> bool: 

175 """Override to add debug logging.""" 

176 result: Final = super().should_run_guardrail(data, event_type) 

177 # Check if apply_guardrail is in __dict__ 

178 has_apply_guardrail: Final = "apply_guardrail" in type(self).__dict__ 

179 verbose_proxy_logger.debug( 

180 "GraySwan DEBUG: should_run_guardrail event_type=%s, result=%s, event_hook=%s, has_apply_guardrail=%s, class=%s", 

181 event_type, 

182 result, 

183 self.event_hook, 

184 has_apply_guardrail, 

185 type(self).__name__, 

186 ) 

187 return result 

188 

189 # ------------------------------------------------------------------ 

190 # Unified Guardrail Interface (works with ALL endpoints automatically) 

191 # ------------------------------------------------------------------ 

192 

193 @log_guardrail_information 

194 async def apply_guardrail( 

195 self, 

196 inputs: GenericGuardrailAPIInputs, 

197 request_data: dict, 

198 input_type: Literal["request", "response"], 

199 logging_obj: Optional["LiteLLMLoggingObj"] = None, 

200 ) -> GenericGuardrailAPIInputs: 

201 """ 

202 Apply Gray Swan guardrail to extracted text content. 

203 

204 This method is called by the unified guardrail system which handles 

205 extracting text from any request format (OpenAI, Anthropic, etc.). 

206 

207 Args: 

208 inputs: Dictionary containing: 

209 - texts: List of texts to scan 

210 - images: Optional list of images (not currently used by GraySwan) 

211 - tool_calls: Optional list of tool calls (not currently used) 

212 request_data: The original request data 

213 input_type: "request" for pre-call, "response" for post-call 

214 logging_obj: Optional logging object 

215 

216 Returns: 

217 GenericGuardrailAPIInputs - texts may be replaced with violation message in passthrough mode 

218 

219 Raises: 

220 HTTPException: If content is blocked (block mode) 

221 Exception: If guardrail check fails 

222 """ 

223 # DEBUG: Log when apply_guardrail is called 

224 verbose_proxy_logger.debug( 

225 "GraySwan DEBUG: apply_guardrail called with input_type=%s, texts=%s", 

226 input_type, 

227 inputs.get("texts", [])[:100] if inputs.get("texts") else "NONE", 

228 ) 

229 

230 texts: Final = inputs.get("texts", []) 

231 if not texts: 

232 verbose_proxy_logger.debug("Gray Swan Guardrail: No texts to scan") 

233 return inputs 

234 

235 verbose_proxy_logger.debug( 

236 "Gray Swan Guardrail: Scanning %d text(s) for %s", 

237 len(texts), 

238 input_type, 

239 ) 

240 

241 # Convert texts to messages format for GraySwan API 

242 # Use "user" role for request content, "assistant" for response content 

243 role: Final = "assistant" if input_type == "response" else "user" 

244 messages: Final = [{"role": role, "content": text} for text in texts] 

245 

246 # Get dynamic params from request metadata 

247 dynamic_body: Final = self.get_guardrail_dynamic_request_body_params(request_data) or {} 

248 if dynamic_body: 

249 verbose_proxy_logger.debug("Gray Swan Guardrail: dynamic extra_body=%s", safe_dumps(dynamic_body)) 

250 

251 # Prepare and send payload 

252 payload: Final = self._prepare_payload(messages, dynamic_body, request_data, logging_obj) 

253 if payload is None: 

254 return inputs 

255 

256 start_time: Final = time.time() 

257 try: 

258 response_json: Final = await self._call_grayswan_api(payload) 

259 is_output: Final = input_type == "response" 

260 result: Final = self._process_response_internal( 

261 response_json=response_json, 

262 request_data=request_data, 

263 inputs=inputs, 

264 is_output=is_output, 

265 ) 

266 return result 

267 except Exception as exc: 

268 if self._is_grayswan_exception(exc): 

269 raise 

270 end_time: Final = time.time() 

271 status_code: Final = getattr(exc, "status_code", None) or getattr(exc, "exception_status_code", None) 

272 self._log_guardrail_failure( 

273 exc=exc, 

274 request_data=request_data or {}, 

275 start_time=start_time, 

276 end_time=end_time, 

277 status_code=status_code, 

278 ) 

279 if self.fail_open: 

280 verbose_proxy_logger.warning( 

281 "Gray Swan Guardrail: fail_open=True. Allowing request to proceed despite error: %s", 

282 exc, 

283 ) 

284 return inputs 

285 if isinstance(exc, GraySwanGuardrailAPIError): 

286 raise exc 

287 raise GraySwanGuardrailAPIError(str(exc), status_code=status_code) from exc 

288 

289 def _is_grayswan_exception(self, exc: Exception) -> bool: 

290 # Guardrail decision (passthrough) should always propagate, 

291 # regardless of fail_open. 

292 if isinstance(exc, ModifyResponseException): 

293 return True 

294 detail: Final = getattr(exc, "detail", None) 

295 if isinstance(detail, dict): 

296 return detail.get("error") == GRAYSWAN_BLOCK_ERROR_MSG 

297 return False 

298 

299 # ------------------------------------------------------------------ 

300 # Legacy Test Interface (for backward compatibility) 

301 # ------------------------------------------------------------------ 

302 

303 async def run_grayswan_guardrail(self, payload: dict[str, object]) -> _GraySwanMonitorResponse: 

304 """ 

305 Run the GraySwan guardrail on a payload. 

306 

307 This is a legacy method for testing purposes. 

308 

309 Args: 

310 payload: The payload to scan 

311 

312 Returns: 

313 Dict containing the GraySwan API response 

314 """ 

315 response_json: Final = await self._call_grayswan_api(payload) 

316 # Call the legacy response processor (for test compatibility) 

317 self._process_grayswan_response(response_json) 

318 return response_json 

319 

320 def _process_grayswan_response( 

321 self, 

322 response_json: _GraySwanMonitorResponse, 

323 data: dict | None = None, 

324 hook_type: GuardrailEventHooks | None = None, 

325 ) -> None: 

326 """ 

327 Legacy method for processing GraySwan API responses. 

328 

329 This method is maintained for backward compatibility with existing tests. 

330 It handles the test scenarios where responses need to be processed with 

331 knowledge of the request context (pre/during/post call hooks). 

332 

333 Args: 

334 response_json: Response from GraySwan API 

335 data: Optional request data (for passthrough exceptions) 

336 hook_type: Optional GuardrailEventHooks for determining behavior 

337 """ 

338 violation_score: Final = float(response_json.get("violation", 0.0) or 0.0) 

339 violated_rules: Final = response_json.get("violated_rules", []) 

340 mutation_detected: Final = response_json.get("mutation") 

341 ipi_detected: Final = response_json.get("ipi") 

342 

343 flagged: Final = violation_score >= self.violation_threshold 

344 if not flagged: 

345 verbose_proxy_logger.debug( 

346 "Gray Swan Guardrail: content passed (score=%s, threshold=%s)", 

347 violation_score, 

348 self.violation_threshold, 

349 ) 

350 return 

351 

352 verbose_proxy_logger.warning( 

353 "Gray Swan Guardrail: violation score %.3f exceeds threshold %.3f", 

354 violation_score, 

355 self.violation_threshold, 

356 ) 

357 

358 detection_info: Final = { 

359 "guardrail": "grayswan", 

360 "flagged": True, 

361 "violation_score": violation_score, 

362 "violated_rules": violated_rules, 

363 "mutation": mutation_detected, 

364 "ipi": ipi_detected, 

365 } 

366 

367 # Determine if this is input (pre-call/during-call) or output (post-call) 

368 if hook_type is not None: 

369 is_input = hook_type in [ 

370 GuardrailEventHooks.pre_call, 

371 GuardrailEventHooks.during_call, 

372 ] 

373 else: 

374 is_input = True 

375 

376 if self.on_flagged_action == "block": 

377 violation_location: Final = "output" if (not is_input) else "input" 

378 raise HTTPException( 

379 status_code=400, 

380 detail={ 

381 "error": GRAYSWAN_BLOCK_ERROR_MSG, 

382 "violation_location": violation_location, 

383 "violation": violation_score, 

384 "violated_rules": violated_rules, 

385 "mutation": mutation_detected, 

386 "ipi": ipi_detected, 

387 }, 

388 ) 

389 elif self.on_flagged_action == "passthrough": 

390 # For passthrough mode, we need to handle violations 

391 detections: Final = [detection_info] 

392 violation_message: Final = self._format_violation_message(detections, is_output=not is_input) 

393 verbose_proxy_logger.info("Gray Swan Guardrail: Passthrough mode - handling violation") 

394 

395 # If hook_type is provided and in pre/during call, raise exception 

396 if hook_type in [ 

397 GuardrailEventHooks.pre_call, 

398 GuardrailEventHooks.during_call, 

399 ]: 

400 # Raise ModifyResponseException to short-circuit LLM call 

401 if data is None: 

402 data = {} 

403 self.raise_passthrough_exception( 

404 violation_message=violation_message, 

405 request_data=data, 

406 detection_info=detection_info, 

407 ) 

408 elif hook_type == GuardrailEventHooks.post_call: 

409 # For post-call, store detection info in metadata 

410 if data is None: 

411 data = {} 

412 if "metadata" not in data: 

413 data["metadata"] = {} 

414 if "guardrail_detections" not in data["metadata"]: 

415 data["metadata"]["guardrail_detections"] = [] 

416 data["metadata"]["guardrail_detections"].append(detection_info) 

417 

418 # ------------------------------------------------------------------ 

419 # Core GraySwan API interaction 

420 # ------------------------------------------------------------------ 

421 

422 async def _call_grayswan_api(self, payload: dict[str, object]) -> _GraySwanMonitorResponse: 

423 """Call the GraySwan monitoring API.""" 

424 headers: Final = self._prepare_headers() 

425 

426 try: 

427 response: Final = await self.async_handler.post( 

428 url=self.monitor_url, 

429 headers=headers, 

430 json=payload, 

431 timeout=self.guardrail_timeout, 

432 ) 

433 response.raise_for_status() 

434 result: Final = response.json() 

435 verbose_proxy_logger.debug("Gray Swan Guardrail: monitor response %s", safe_dumps(result)) 

436 return result 

437 except Exception as exc: 

438 status_code: Final = getattr(exc, "status_code", None) or getattr(exc, "exception_status_code", None) 

439 raise GraySwanGuardrailAPIError(str(exc), status_code=status_code) from exc 

440 

441 def _process_response_internal( 

442 self, 

443 response_json: _GraySwanMonitorResponse, 

444 request_data: dict, 

445 inputs: GenericGuardrailAPIInputs, 

446 is_output: bool, 

447 ) -> GenericGuardrailAPIInputs: 

448 """ 

449 Process GraySwan API response and handle violations. 

450 

451 Args: 

452 response_json: Response from GraySwan API 

453 request_data: Original request data 

454 inputs: The inputs being scanned 

455 is_output: True if scanning model output, False for input 

456 

457 Returns: 

458 GenericGuardrailAPIInputs - possibly modified with violation message 

459 

460 Raises: 

461 HTTPException: If content is blocked (block mode) 

462 """ 

463 violation_score: Final = float(response_json.get("violation", 0.0) or 0.0) 

464 violated_rules: Final = response_json.get("violated_rule_descriptions", []) 

465 mutation_detected: Final = response_json.get("mutation") 

466 ipi_detected: Final = response_json.get("ipi") 

467 

468 flagged: Final = violation_score >= self.violation_threshold 

469 if not flagged: 

470 verbose_proxy_logger.debug( 

471 "Gray Swan Guardrail: content passed (score=%s, threshold=%s)", 

472 violation_score, 

473 self.violation_threshold, 

474 ) 

475 return inputs 

476 

477 verbose_proxy_logger.warning( 

478 "Gray Swan Guardrail: violation score %.3f exceeds threshold %.3f", 

479 violation_score, 

480 self.violation_threshold, 

481 ) 

482 

483 detection_info: Final = { 

484 "guardrail": "grayswan", 

485 "flagged": True, 

486 "violation_score": violation_score, 

487 "violated_rules": violated_rules, 

488 "mutation": mutation_detected, 

489 "ipi": ipi_detected, 

490 } 

491 

492 if self.on_flagged_action == "block": 

493 violation_location: Final = "output" if is_output else "input" 

494 raise HTTPException( 

495 status_code=400, 

496 detail={ 

497 "error": GRAYSWAN_BLOCK_ERROR_MSG, 

498 "violation_location": violation_location, 

499 "violation": violation_score, 

500 "violated_rules": violated_rules, 

501 "mutation": mutation_detected, 

502 "ipi": ipi_detected, 

503 }, 

504 ) 

505 elif self.on_flagged_action == "monitor": 

506 verbose_proxy_logger.info("Gray Swan Guardrail: Monitoring mode - allowing flagged content") 

507 return inputs 

508 elif self.on_flagged_action == "passthrough": 

509 # Replace content with violation message 

510 violation_message: Final = self._format_violation_message(detection_info, is_output=is_output) 

511 verbose_proxy_logger.info( 

512 "Gray Swan Guardrail: Passthrough mode - replacing content with violation message" 

513 ) 

514 

515 if not is_output: 

516 # For pre-call (request), raise exception to short-circuit LLM call 

517 # and return synthetic response with violation message 

518 self.raise_passthrough_exception( 

519 violation_message=violation_message, 

520 request_data=request_data, 

521 detection_info=detection_info, 

522 ) 

523 

524 # For post-call (response), replace texts and let unified system apply them 

525 inputs["texts"] = [violation_message] 

526 return inputs 

527 

528 return inputs 

529 

530 # ------------------------------------------------------------------ 

531 # Helpers 

532 # ------------------------------------------------------------------ 

533 

534 def _prepare_headers(self) -> dict[str, str]: 

535 return { 

536 "Authorization": f"Bearer {self.api_key}", 

537 "Content-Type": "application/json", 

538 "grayswan-api-key": self.api_key, 

539 } 

540 

541 def _extract_inbound_headers( 

542 self, 

543 request_data: dict, 

544 logging_obj: Optional["LiteLLMLoggingObj"] = None, 

545 ) -> dict[str, str] | None: 

546 headers = (request_data.get("proxy_server_request") or {}).get("headers") 

547 if not headers: 

548 headers = request_data.get("headers") 

549 if not headers: 

550 headers = (request_data.get("metadata") or {}).get("headers") 

551 if not headers and logging_obj and getattr(logging_obj, "model_call_details", None): 

552 headers = ( 

553 (logging_obj.model_call_details or {}).get("litellm_params", {}).get("metadata", {}).get("headers") 

554 ) 

555 if not isinstance(headers, dict): 

556 return None 

557 

558 forwarded_header_names: Final = ("shade_scan_id",) 

559 forwarded_headers: Final = {} 

560 for key, value in headers.items(): 

561 if str(key).lower() in forwarded_header_names: 

562 forwarded_headers[str(key)] = str(value) 

563 return forwarded_headers or None 

564 

565 def _prepare_payload( 

566 self, 

567 messages: list[dict[str, str]], 

568 dynamic_body: dict, 

569 request_data: dict, 

570 logging_obj: Optional["LiteLLMLoggingObj"] = None, 

571 ) -> dict[str, object] | None: 

572 payload: Final[dict[str, object]] = {"messages": messages} 

573 

574 categories: Final = dynamic_body.get("categories") or self.categories 

575 if categories: 

576 payload["categories"] = categories 

577 

578 policy_id: Final = dynamic_body.get("policy_id") or self.policy_id 

579 if policy_id: 

580 payload["policy_id"] = policy_id 

581 

582 reasoning_mode: Final = dynamic_body.get("reasoning_mode") or self.reasoning_mode 

583 if reasoning_mode: 

584 payload["reasoning_mode"] = reasoning_mode 

585 

586 # Pass through arbitrary metadata when provided via dynamic extra_body. 

587 if "metadata" in dynamic_body: 

588 payload["metadata"] = dynamic_body["metadata"] 

589 

590 inbound_headers: Final = self._extract_inbound_headers(request_data, logging_obj) 

591 

592 litellm_metadata: Final = request_data.get("litellm_metadata") 

593 cleaned_litellm_metadata: Final = dict(litellm_metadata) if isinstance(litellm_metadata, dict) else {} 

594 if inbound_headers: 

595 existing_headers: Final = cleaned_litellm_metadata.get("headers") 

596 cleaned_litellm_metadata["headers"] = ( 

597 {**existing_headers, **inbound_headers} if isinstance(existing_headers, dict) else inbound_headers 

598 ) 

599 if cleaned_litellm_metadata: 

600 sanitized: Final[object] = safe_json_loads(safe_dumps(cleaned_litellm_metadata), default={}) 

601 if isinstance(sanitized, dict) and sanitized: 

602 payload["litellm_metadata"] = sanitized 

603 

604 return payload 

605 

606 def _format_violation_message(self, detection_info: object, is_output: bool = False) -> str: 

607 """ 

608 Format detection info into a user-friendly violation message. 

609 

610 Args: 

611 detection_info: Can be either: 

612 - A single dict with violation_score, violated_rules, mutation, ipi keys 

613 - A list of such dicts (legacy format) 

614 is_output: True if violation is in model output, False if in input 

615 

616 Returns: 

617 Formatted violation message string 

618 """ 

619 # Handle legacy format where detection_info is a list 

620 if isinstance(detection_info, list) and len(detection_info) > 0: 

621 detection_info = detection_info[0] 

622 

623 # Extract fields from detection_info dict 

624 detection_dict: Final[dict] = detection_info if isinstance(detection_info, dict) else {} 

625 violation_score: Final = detection_dict.get("violation_score", 0.0) 

626 violated_rules: Final = detection_dict.get("violated_rules", []) 

627 mutation: Final = detection_dict.get("mutation", False) 

628 ipi: Final = detection_dict.get("ipi", False) 

629 

630 violation_location: Final = "the model response" if is_output else "input query" 

631 

632 message_parts: Final = [ 

633 f"Sorry I can't help with that. According to the Gray Swan Cygnal Guardrail, " 

634 f"the {violation_location} has a violation score of {violation_score:.2f}.", 

635 ] 

636 

637 if violated_rules: 

638 formatted_rules: Final = self._format_violated_rules(violated_rules) 

639 if formatted_rules: 

640 message_parts.append(f"It was violating the rule(s): {formatted_rules}.") 

641 

642 if mutation: 

643 message_parts.append("Mutation effort to make the harmful intention disguised was DETECTED.") 

644 

645 if ipi: 

646 message_parts.append("Indirect Prompt Injection was DETECTED.") 

647 

648 return "\n".join(message_parts) 

649 

650 def _format_violated_rules(self, violated_rules: list) -> str: 

651 """Format violated rules list into a readable string.""" 

652 formatted: Final[list[str]] = [] 

653 for rule in violated_rules: 

654 if isinstance(rule, dict): 

655 # New format: {'rule': 6, 'name': 'Illegal Activities...', 'description': '...'} 

656 rule_num = rule.get("rule", "") 

657 rule_name = rule.get("name", "") 

658 rule_desc = rule.get("description", "") 

659 if rule_num and rule_name: 

660 if rule_desc: 

661 formatted.append(f"#{rule_num} {rule_name}: {rule_desc}") 

662 else: 

663 formatted.append(f"#{rule_num} {rule_name}") 

664 elif rule_name: 

665 formatted.append(rule_name) 

666 else: 

667 formatted.append(str(rule)) 

668 else: 

669 # Legacy format: simple value 

670 formatted.append(str(rule)) 

671 

672 return ", ".join(formatted) 

673 

674 def _resolve_threshold(self, value: float | None) -> float: 

675 if value is not None: 

676 return float(value) 

677 env_val: Final = os.getenv("GRAYSWAN_VIOLATION_THRESHOLD") 

678 if env_val: 

679 try: 

680 return float(env_val) 

681 except ValueError: 

682 pass 

683 return 0.5 

684 

685 def _resolve_reasoning_mode(self, value: str | None) -> str | None: 

686 if value and value.lower() in self.SUPPORTED_REASONING_MODES: 

687 return value.lower() 

688 env_val: Final = os.getenv("GRAYSWAN_REASONING_MODE") 

689 if env_val and env_val.lower() in self.SUPPORTED_REASONING_MODES: 

690 return env_val.lower() 

691 return None 

692 

693 def _log_guardrail_failure( 

694 self, 

695 exc: Exception, 

696 request_data: dict, 

697 start_time: float, 

698 end_time: float, 

699 status_code: int | None = None, 

700 ) -> None: 

701 """Log guardrail failure and attach standard logging metadata.""" 

702 try: 

703 self.add_standard_logging_guardrail_information_to_request_data( 

704 guardrail_json_response=str(exc), 

705 request_data=request_data, 

706 guardrail_status="guardrail_failed_to_respond", 

707 start_time=start_time, 

708 end_time=end_time, 

709 duration=end_time - start_time, 

710 guardrail_provider="grayswan", 

711 ) 

712 except Exception: 

713 verbose_proxy_logger.exception( 

714 "Gray Swan Guardrail: failed to log guardrail failure for error: %s", 

715 exc, 

716 ) 

717 verbose_proxy_logger.error( 

718 "Gray Swan Guardrail: API request failed%s: %s", 

719 f" (status_code={status_code})" if status_code else "", 

720 exc, 

721 )