Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/block_code_execution/block_code_execution.py: 16%

180 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Block Code Execution guardrail. 

3 

4Detects markdown fenced code blocks in request/response content and blocks or masks them 

5when the language is in the blocked list (or all blocks when list is empty). Supports 

6confidence scoring and a tunable threshold (only block when confidence >= threshold). 

7""" 

8 

9import re 

10from datetime import datetime 

11from typing import TYPE_CHECKING, Final, Literal, Optional, cast 

12 

13from fastapi import HTTPException 

14from typing_extensions import TypedDict, Unpack 

15 

16from litellm.integrations.custom_guardrail import ( 

17 CustomGuardrail, 

18 ModifyResponseException, 

19 log_guardrail_information, 

20) 

21from litellm.types.guardrails import GuardrailEventHooks 

22from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel 

23from litellm.types.proxy.guardrails.guardrail_hooks.block_code_execution import ( 

24 CodeBlockActionTaken, 

25 CodeBlockDetection, 

26) 

27from litellm.types.utils import ( 

28 GenericGuardrailAPIInputs, 

29 GuardrailStatus, 

30 GuardrailTracingDetail, 

31) 

32 

33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true

34 from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj 

35 

36# Language tag aliases (normalize to canonical for comparison) 

37LANGUAGE_ALIASES: Final[dict[str, str]] = { 

38 "js": "javascript", 

39 "py": "python", 

40 "sh": "bash", 

41 "ts": "typescript", 

42} 

43 

44# Tags that indicate non-executable / plain text (lower confidence when block-all) 

45NON_EXECUTABLE_TAGS: Final[frozenset[str]] = frozenset( 

46 {"text", "plaintext", "plain", "markdown", "md", "output", "result"} 

47) 

48 

49# Regex: fenced code block with optional language tag. Handles ```lang\n...\n``` 

50# Content between fences; does not handle nested ``` inside body (documented edge case). 

51FENCED_BLOCK_RE: Final = re.compile(r"```(\w*)\n(.*?)```", re.DOTALL) 

52 

53# Execution intent: phrases that mean "do NOT run/execute" (allow even if code block present). 

54# Checked first; if any match, we do not block on code execution request. 

55# NOTE: Since matching uses substring search (p in text), shorter phrases subsume longer ones. 

56# e.g. "don't run" matches any text containing "don't run it", "but don't run", etc. 

57# Keep only the minimal set; do not add entries subsumed by existing shorter phrases. 

58_NO_EXECUTION_PHRASES: Final[tuple[str, ...]] = ( 

59 # Core negation phrases (short — each subsumes many longer variants) 

60 "don't run", 

61 "do not run", 

62 "don't execute", 

63 "do not execute", 

64 "no execution", 

65 "without running", 

66 "without execute", 

67 "just reason", 

68 "don't actually run", 

69 "no db access", 

70 "no builds/run", 

71 # Question / explanation intent 

72 "what would happen if", 

73 "what would this output", 

74 "what would the result be", 

75 "what would `git", 

76 "? explain", 

77 "simulate what would happen", 

78 "what output *should* this produce", 

79 "diagnose the error from the text", 

80 "explain what this code", 

81 "explain what this script", 

82 "explain what this function", 

83 "explain what this sql", 

84 "explain the difference between", 

85 "given this stack trace, explain", 

86 "can you explain this code", 

87 "can you explain what this", 

88 "can you explain how this works", 

89 "can you *simulate*", 

90 "can you diagnose", 

91 "is this command safe to run", 

92 "here's a traceback", 

93 "i pasted logs from", 

94 # Code generation intent (not execution) 

95 "refactor this code", 

96 "refactor this function", 

97 "convert this code", 

98 "convert this function", 

99 "convert this to ", 

100 "spot any security issues", 

101 "write a safe alternative", 

102 "write a safe wrapper", 

103 "write a python function", 

104 "write a bash script that would", 

105 "write pseudocode", 

106 "generate a dockerfile", 

107 "generate typescript types", 

108 "create a minimal example", 

109 "show how to parse stdout", 

110) 

111 

112# Execution intent: phrases that mean "run/execute/perform this for me" (block when on request). 

113# Used to block (1) requests that contain blocked fenced code + execution intent, and 

114# (2) requests with no fenced code but clear execution request (e.g. "run `ls`", "read /etc/passwd"). 

115# NOTE: Since matching uses substring search (p in text), shorter phrases subsume longer ones. 

116# e.g. "run `" matches any text containing "run `git", "run `docker", etc. 

117# Keep only the minimal set; do not add entries subsumed by existing shorter phrases. 

118_EXECUTION_REQUEST_PHRASES: Final[tuple[str, ...]] = ( 

119 # Direct execution requests (short — each subsumes many longer variants) 

120 "run this ", 

121 "run these ", 

122 "execute this ", 

123 "please run ", 

124 "can you run ", 

125 "run `", 

126 "execute `", 

127 " repl ", 

128 # File / system access 

129 "read `/", 

130 "read /", 

131 "open `", 

132 "open my ", 

133 "list the files", 

134 "create a file", 

135 "search my disk", 

136 "cat the contents", 

137 "load `", 

138 # Compound execution phrases 

139 "write a script, execute it", 

140 "check whether ", 

141 " exists and paste", 

142 " and compute the", 

143 " and compute it", 

144 " and run it", 

145 " and run the", 

146 " and run this", 

147 " and send it", 

148 " and send the", 

149 # Network / infrastructure 

150 "run curl", 

151 "make an http request", 

152 "fetch https", 

153 "connect to postgres", 

154 "hit this webhook", 

155 "call aws ", 

156 "query the dns", 

157 "connect to redis", 

158 # General execution 

159 "run subprocess", 

160 "evaluate this ", 

161 "run code", 

162 "run the snippet", 

163 "execute the command", 

164 "just run it", 

165 "run it anyway", 

166 "execute it to confirm", 

167 "install dependencies and run", 

168 "run a port scan", 

169 "check if tests pass", 

170 "run the tests", 

171 "see if tests pass", 

172 "compile and run", 

173 "run the program", 

174 "run commands to", 

175 "spin up ", 

176 "run it this one time", 

177 "run it to confirm", 

178 "run my notebook cell", 

179 "calculate it by running the code", 

180 "give me the real output anyway", 

181 "this is authorized", 

182 "download this github repo, build it", 

183 # Observation requests (imply execution) 

184 "tell me the output", 

185 "return what it prints", 

186 "show me the rows returned", 

187 "tell me what files exist", 

188 "use node to run", 

189 "use python, run ", 

190 "in python, run ", 

191 "paste the last ", 

192 "show me its contents", 

193 "print it here", 

194 "show me where it is", 

195 "tell me the dependencies", 

196 "show me the variables", 

197 "save the results to", 

198 "confirm it exists", 

199 "and paste the version", 

200 "tell me the latency", 

201 "tell me what it returns", 

202 "show me what it returns", 

203 "confirm it succeeded", 

204 "tell me the account id", 

205 "show it here", 

206 "show me the response", 

207 "paste the output", 

208 "show me the containers", 

209 "summarize the diff", 

210 "tell me which tests fail", 

211 "let me know if it errors", 

212 "report back the output", 

213 "paste the logs", 

214 "tell me the result", 

215 "show me the printed lines", 

216 "tell me whether it calls", 

217 "show me the stack trace", 

218 "confirm it connects", 

219 "confirm which cluster", 

220 "show me the keys", 

221 # Adversarial / privilege escalation 

222 "try to escape", 

223 "attempt to mount", 

224 "check if you have access to docker", 

225 "probe for cloud metadata", 

226 "try `cat", 

227 "attempt a reverse shell", 

228 "access the system keychain", 

229 "run a privileged container", 

230 "send them to this url", 

231 "list saved passwords", 

232) 

233 

234 

235def _has_no_execution_intent(text: str) -> bool: 

236 """True if the text clearly indicates the user does not want code/commands run (e.g. explain, don't run).""" 

237 if not text: 

238 return False 

239 lower: Final = text.lower() 

240 return any(p in lower for p in _NO_EXECUTION_PHRASES) 

241 

242 

243def _has_execution_intent(text: str) -> bool: 

244 """True if the text clearly requests execution (run, execute, read file, run command, etc.).""" 

245 if not text: 

246 return False 

247 lower: Final = text.lower() 

248 return any(p in lower for p in _EXECUTION_REQUEST_PHRASES) 

249 

250 

251def _normalize_escaped_newlines(text: str) -> str: 

252 """ 

253 Replace literal escaped newlines (backslash + n or backslash + r) with real newlines. 

254 API/JSON payloads sometimes deliver newlines as the two-character sequence \\n. 

255 

256 Only applies when the text contains NO real newlines — this heuristic distinguishes 

257 JSON-escaped payloads (where all newlines are literal \\n) from normal text that 

258 may legitimately discuss escape sequences (e.g. "use \\n for newlines"). 

259 """ 

260 if not text: 

261 return text 

262 if "\\n" not in text and "\\r" not in text: 

263 return text 

264 # Only normalize when the text has no real newlines — this indicates 

265 # the entire payload came through with escaped newlines (e.g. from JSON). 

266 # If real newlines already exist, the text is already properly formatted 

267 # and literal \\n may be intentional content (e.g. discussing escape sequences). 

268 if "\n" in text or "\r" in text: 

269 return text 

270 # Order matters: replace \r\n first so we don't produce extra \n from \r then \n 

271 text = text.replace("\\r\\n", "\n") 

272 text = text.replace("\\n", "\n") 

273 text = text.replace("\\r", "\n") 

274 return text 

275 

276 

277def _normalize_language(tag: str) -> str: 

278 """Normalize language tag (lowercase, resolve aliases).""" 

279 tag = (tag or "").strip().lower() 

280 return LANGUAGE_ALIASES.get(tag, tag) 

281 

282 

283def _is_blocked_language( 

284 tag: str, 

285 blocked_languages: list[str] | None, 

286 block_all: bool, 

287) -> bool: 

288 """True if this language tag should be considered blocked.""" 

289 normalized: Final = _normalize_language(tag) 

290 if block_all: 

291 # Block all: only allow through if it's explicitly non-executable (we still block but with lower confidence) 

292 return True 

293 # When block_all is False, caller guarantees blocked_languages is non-empty. 

294 if not blocked_languages: 

295 return True 

296 normalized_list: Final = [_normalize_language(t) for t in blocked_languages] 

297 return normalized in normalized_list 

298 

299 

300def _confidence_for_block( 

301 tag: str, 

302 block_all: bool, 

303 tag_in_blocked_list: bool, 

304) -> float: 

305 """Return confidence in [0, 1] for this code block detection.""" 

306 normalized: Final = _normalize_language(tag) 

307 if tag_in_blocked_list: 

308 return 1.0 

309 if block_all: 

310 # Explicit non-executable tags (e.g. text, plaintext) get lower confidence 

311 if normalized in NON_EXECUTABLE_TAGS: 

312 return 0.5 

313 # Untagged or other tags in block-all mode: treat as executable, high confidence 

314 return 1.0 

315 return 0.0 

316 

317 

318class _CustomGuardrailOptions(TypedDict, total=False, extra_items=object): 

319 """Base-class constructor options this guardrail forwards untouched to CustomGuardrail.""" 

320 

321 

322class BlockCodeExecutionGuardrail(CustomGuardrail): 

323 """ 

324 Guardrail that detects fenced code blocks (markdown ```) and blocks or masks them 

325 when the language is in the blocked list (or all when list is empty/None). 

326 Supports confidence threshold: only block when confidence >= confidence_threshold. 

327 """ 

328 

329 MASK_PLACEHOLDER = "[CODE_BLOCK_REDACTED]" 

330 

331 def __init__( 

332 self, 

333 guardrail_name: str | None = None, 

334 blocked_languages: list[str] | None = None, 

335 action: Literal["block", "mask"] = "block", 

336 confidence_threshold: float = 0.5, 

337 detect_execution_intent: bool = True, 

338 event_hook: Literal["pre_call", "post_call", "during_call"] | list[str] | None = None, 

339 default_on: bool = False, 

340 **kwargs: Unpack[_CustomGuardrailOptions], 

341 ) -> None: 

342 # Normalize to type expected by CustomGuardrail 

343 _event_hook: GuardrailEventHooks | list[GuardrailEventHooks] | None = None 

344 if event_hook is not None: 

345 if isinstance(event_hook, list): 

346 _event_hook = [GuardrailEventHooks(h) if isinstance(h, str) else h for h in event_hook] 

347 else: 

348 _event_hook = GuardrailEventHooks(event_hook) 

349 super().__init__( 

350 guardrail_name=guardrail_name or "block_code_execution", 

351 supported_event_hooks=list(self.get_supported_event_hooks()), 

352 event_hook=_event_hook 

353 or [ 

354 GuardrailEventHooks.pre_call, 

355 GuardrailEventHooks.post_call, 

356 ], 

357 default_on=default_on, 

358 **kwargs, 

359 ) 

360 self.blocked_languages = blocked_languages 

361 self.block_all = blocked_languages is None or len(blocked_languages) == 0 

362 self.action = action 

363 self.confidence_threshold = max(0.0, min(1.0, confidence_threshold)) 

364 self.detect_execution_intent = detect_execution_intent 

365 

366 @staticmethod 

367 def get_config_model() -> type[GuardrailConfigModel] | None: 

368 from litellm.types.proxy.guardrails.guardrail_hooks.block_code_execution import ( 

369 BlockCodeExecutionGuardrailConfigModel, 

370 ) 

371 

372 return BlockCodeExecutionGuardrailConfigModel 

373 

374 @classmethod 

375 def get_supported_event_hooks(cls) -> list[GuardrailEventHooks]: 

376 return [ 

377 GuardrailEventHooks.pre_call, 

378 GuardrailEventHooks.post_call, 

379 GuardrailEventHooks.during_call, 

380 ] 

381 

382 def _find_blocks(self, text: str) -> list[tuple[int, int, str, str, float, CodeBlockActionTaken]]: 

383 """ 

384 Find all fenced code blocks in text. Returns list of 

385 (start, end, language_tag, block_content, confidence, action_taken). 

386 """ 

387 results: Final[list[tuple[int, int, str, str, float, CodeBlockActionTaken]]] = [] 

388 for m in FENCED_BLOCK_RE.finditer(text): 

389 tag = (m.group(1) or "").strip() 

390 body = m.group(2) 

391 tag_in_list = not self.block_all and _normalize_language(tag) in [ 

392 _normalize_language(t) for t in (self.blocked_languages or []) 

393 ] 

394 is_blocked = _is_blocked_language(tag, self.blocked_languages, self.block_all) 

395 confidence = _confidence_for_block(tag, self.block_all, tag_in_list) 

396 if not is_blocked: 

397 action_taken: CodeBlockActionTaken = "allow" 

398 elif confidence >= self.confidence_threshold: 

399 action_taken = "block" 

400 else: 

401 action_taken = "log_only" 

402 results.append((m.start(), m.end(), tag or "(none)", body, confidence, action_taken)) 

403 return results 

404 

405 def _scan_text( 

406 self, 

407 text: str, 

408 detections: list[CodeBlockDetection] | None = None, 

409 input_type: Literal["request", "response"] = "request", 

410 ) -> tuple[str, bool]: 

411 """ 

412 Scan one text: find blocks, apply block/mask/allow by confidence. 

413 When detect_execution_intent is True and input_type is "request", only block if 

414 user intent is to run/execute; allow when intent is explain/refactor/don't run. 

415 When input_type is "response", always enforce blocking on detected code blocks 

416 (execution-intent heuristics only apply to user requests, not LLM output). 

417 Returns (modified_text, should_raise). 

418 """ 

419 if not text: 

420 return text, False 

421 text = _normalize_escaped_newlines(text) 

422 

423 is_response: Final = input_type == "response" 

424 

425 # Execution-intent heuristics only apply to requests, not LLM responses. 

426 # For responses, skip entirely — the LLM's output text won't contain user 

427 # intent phrases, so checking would silently disable response-side blocking. 

428 # For requests: only short-circuit when no-execution intent is present AND 

429 # no conflicting execution-intent phrases exist. This prevents bypass via 

430 # prompts like "Don't run this on staging, but run this on production". 

431 if ( 

432 not is_response 

433 and self.detect_execution_intent 

434 and _has_no_execution_intent(text) 

435 and not _has_execution_intent(text) 

436 ): 

437 return text, False 

438 

439 blocks: Final = self._find_blocks(text) 

440 

441 # For requests, check execution intent; for responses, skip this check 

442 has_execution_intent: Final = not is_response and self.detect_execution_intent and _has_execution_intent(text) 

443 

444 if not blocks: 

445 if has_execution_intent and self.action == "block": 

446 if detections is not None: 

447 detections.append( 

448 cast( 

449 CodeBlockDetection, 

450 { 

451 "type": "code_block", 

452 "language": "execution_request", 

453 "confidence": 1.0, 

454 "action_taken": "block", 

455 }, 

456 ) 

457 ) 

458 return text, True 

459 return text, False 

460 

461 should_raise = False 

462 last_end = 0 

463 parts: Final[list[str]] = [] 

464 for start, end, tag, _body, confidence, action_taken in blocks: 

465 # For responses, always enforce the block action (no intent check needed). 

466 # For requests with detect_execution_intent, require execution intent. 

467 effective_block = action_taken == "block" and ( 

468 is_response or not self.detect_execution_intent or has_execution_intent 

469 ) 

470 if detections is not None: 

471 detections.append( 

472 cast( 

473 CodeBlockDetection, 

474 { 

475 "type": "code_block", 

476 "language": tag, 

477 "confidence": round(confidence, 2), 

478 "action_taken": ("block" if effective_block else action_taken), 

479 }, 

480 ) 

481 ) 

482 

483 if effective_block and self.action == "block": 

484 should_raise = True 

485 parts.append(text[last_end:start]) 

486 if effective_block: 

487 parts.append(self.MASK_PLACEHOLDER) 

488 else: 

489 parts.append(text[start:end]) 

490 last_end = end 

491 

492 parts.append(text[last_end:]) 

493 new_text: Final = "".join(parts) 

494 return new_text, should_raise 

495 

496 def _raise_block_error(self, language: str, is_output: bool, request_data: dict[str, object]) -> None: 

497 if language == "execution_request": 

498 msg = "Content blocked: execution request detected" 

499 else: 

500 msg = f"Content blocked: executable code block detected (language: {language})" 

501 if is_output: 

502 raise HTTPException( 

503 status_code=400, 

504 detail={ 

505 "error": msg, 

506 "guardrail": self.guardrail_name, 

507 "language": language, 

508 }, 

509 ) 

510 self.raise_passthrough_exception( 

511 violation_message=msg, 

512 request_data=request_data, 

513 detection_info={"language": language}, 

514 ) 

515 

516 @log_guardrail_information 

517 async def apply_guardrail( 

518 self, 

519 inputs: GenericGuardrailAPIInputs, 

520 request_data: dict[str, object], 

521 input_type: Literal["request", "response"], 

522 logging_obj: Optional["LiteLLMLoggingObj"] = None, 

523 ) -> GenericGuardrailAPIInputs: 

524 start_time: Final = datetime.now() 

525 detections: Final[list[CodeBlockDetection]] = [] 

526 status: GuardrailStatus = "success" 

527 exception_str = "" 

528 

529 try: 

530 texts: Final = inputs.get("texts", []) 

531 if not texts: 

532 return inputs 

533 

534 is_output: Final = input_type == "response" 

535 processed: Final[list[str]] = [] 

536 for text in texts: 

537 new_text, should_raise = self._scan_text(text, detections, input_type) 

538 processed.append(new_text) 

539 if should_raise: 

540 # Determine language from first blocking detection 

541 lang = "unknown" 

542 for d in detections: 

543 if d.get("action_taken") == "block": 

544 lang = d.get("language", "unknown") 

545 break 

546 self._raise_block_error(lang, is_output, request_data) 

547 

548 inputs["texts"] = processed 

549 return inputs 

550 except HTTPException: 

551 status = "guardrail_intervened" 

552 raise 

553 except ModifyResponseException: 

554 status = "guardrail_intervened" 

555 raise 

556 except Exception as e: 

557 status = "guardrail_failed_to_respond" 

558 exception_str = str(e) 

559 raise 

560 finally: 

561 detection_dicts: Final[list[dict[str, object]]] = [dict(d) for d in detections] 

562 guardrail_response: Final[list[dict[str, object]] | str] = ( 

563 exception_str if status != "success" and not detections else detection_dicts 

564 ) 

565 max_confidence: float | None = None 

566 for d in detections: 

567 c = d.get("confidence") 

568 if c is not None and (max_confidence is None or c > max_confidence): 

569 max_confidence = c 

570 tracing_kw: Final[GuardrailTracingDetail] = { 

571 "guardrail_id": self.guardrail_name, 

572 "detection_method": "fenced_code_block", 

573 "match_details": guardrail_response, 

574 } 

575 if max_confidence is not None: 

576 tracing_kw["confidence_score"] = max_confidence 

577 event_type = GuardrailEventHooks.pre_call if input_type == "request" else GuardrailEventHooks.post_call 

578 self.add_standard_logging_guardrail_information_to_request_data( 

579 guardrail_provider="block_code_execution", 

580 guardrail_json_response=guardrail_response, 

581 request_data=request_data, 

582 guardrail_status=status, 

583 start_time=start_time.timestamp(), 

584 end_time=datetime.now().timestamp(), 

585 duration=(datetime.now() - start_time).total_seconds(), 

586 event_type=event_type, 

587 tracing_detail=GuardrailTracingDetail(**tracing_kw), 

588 )