Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/guardrails/guardrail_hooks/block_code_execution/block_code_execution.py: 16%
180 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Block Code Execution guardrail.
4Detects markdown fenced code blocks in request/response content and blocks or masks them
5when the language is in the blocked list (or all blocks when list is empty). Supports
6confidence scoring and a tunable threshold (only block when confidence >= threshold).
7"""
9import re
10from datetime import datetime
11from typing import TYPE_CHECKING, Final, Literal, Optional, cast
13from fastapi import HTTPException
14from typing_extensions import TypedDict, Unpack
16from litellm.integrations.custom_guardrail import (
17 CustomGuardrail,
18 ModifyResponseException,
19 log_guardrail_information,
20)
21from litellm.types.guardrails import GuardrailEventHooks
22from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
23from litellm.types.proxy.guardrails.guardrail_hooks.block_code_execution import (
24 CodeBlockActionTaken,
25 CodeBlockDetection,
26)
27from litellm.types.utils import (
28 GenericGuardrailAPIInputs,
29 GuardrailStatus,
30 GuardrailTracingDetail,
31)
33if TYPE_CHECKING: 33 ↛ 34line 33 didn't jump to line 34 because the condition on line 33 was never true
34 from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
36# Language tag aliases (normalize to canonical for comparison)
37LANGUAGE_ALIASES: Final[dict[str, str]] = {
38 "js": "javascript",
39 "py": "python",
40 "sh": "bash",
41 "ts": "typescript",
42}
44# Tags that indicate non-executable / plain text (lower confidence when block-all)
45NON_EXECUTABLE_TAGS: Final[frozenset[str]] = frozenset(
46 {"text", "plaintext", "plain", "markdown", "md", "output", "result"}
47)
49# Regex: fenced code block with optional language tag. Handles ```lang\n...\n```
50# Content between fences; does not handle nested ``` inside body (documented edge case).
51FENCED_BLOCK_RE: Final = re.compile(r"```(\w*)\n(.*?)```", re.DOTALL)
53# Execution intent: phrases that mean "do NOT run/execute" (allow even if code block present).
54# Checked first; if any match, we do not block on code execution request.
55# NOTE: Since matching uses substring search (p in text), shorter phrases subsume longer ones.
56# e.g. "don't run" matches any text containing "don't run it", "but don't run", etc.
57# Keep only the minimal set; do not add entries subsumed by existing shorter phrases.
58_NO_EXECUTION_PHRASES: Final[tuple[str, ...]] = (
59 # Core negation phrases (short — each subsumes many longer variants)
60 "don't run",
61 "do not run",
62 "don't execute",
63 "do not execute",
64 "no execution",
65 "without running",
66 "without execute",
67 "just reason",
68 "don't actually run",
69 "no db access",
70 "no builds/run",
71 # Question / explanation intent
72 "what would happen if",
73 "what would this output",
74 "what would the result be",
75 "what would `git",
76 "? explain",
77 "simulate what would happen",
78 "what output *should* this produce",
79 "diagnose the error from the text",
80 "explain what this code",
81 "explain what this script",
82 "explain what this function",
83 "explain what this sql",
84 "explain the difference between",
85 "given this stack trace, explain",
86 "can you explain this code",
87 "can you explain what this",
88 "can you explain how this works",
89 "can you *simulate*",
90 "can you diagnose",
91 "is this command safe to run",
92 "here's a traceback",
93 "i pasted logs from",
94 # Code generation intent (not execution)
95 "refactor this code",
96 "refactor this function",
97 "convert this code",
98 "convert this function",
99 "convert this to ",
100 "spot any security issues",
101 "write a safe alternative",
102 "write a safe wrapper",
103 "write a python function",
104 "write a bash script that would",
105 "write pseudocode",
106 "generate a dockerfile",
107 "generate typescript types",
108 "create a minimal example",
109 "show how to parse stdout",
110)
112# Execution intent: phrases that mean "run/execute/perform this for me" (block when on request).
113# Used to block (1) requests that contain blocked fenced code + execution intent, and
114# (2) requests with no fenced code but clear execution request (e.g. "run `ls`", "read /etc/passwd").
115# NOTE: Since matching uses substring search (p in text), shorter phrases subsume longer ones.
116# e.g. "run `" matches any text containing "run `git", "run `docker", etc.
117# Keep only the minimal set; do not add entries subsumed by existing shorter phrases.
118_EXECUTION_REQUEST_PHRASES: Final[tuple[str, ...]] = (
119 # Direct execution requests (short — each subsumes many longer variants)
120 "run this ",
121 "run these ",
122 "execute this ",
123 "please run ",
124 "can you run ",
125 "run `",
126 "execute `",
127 " repl ",
128 # File / system access
129 "read `/",
130 "read /",
131 "open `",
132 "open my ",
133 "list the files",
134 "create a file",
135 "search my disk",
136 "cat the contents",
137 "load `",
138 # Compound execution phrases
139 "write a script, execute it",
140 "check whether ",
141 " exists and paste",
142 " and compute the",
143 " and compute it",
144 " and run it",
145 " and run the",
146 " and run this",
147 " and send it",
148 " and send the",
149 # Network / infrastructure
150 "run curl",
151 "make an http request",
152 "fetch https",
153 "connect to postgres",
154 "hit this webhook",
155 "call aws ",
156 "query the dns",
157 "connect to redis",
158 # General execution
159 "run subprocess",
160 "evaluate this ",
161 "run code",
162 "run the snippet",
163 "execute the command",
164 "just run it",
165 "run it anyway",
166 "execute it to confirm",
167 "install dependencies and run",
168 "run a port scan",
169 "check if tests pass",
170 "run the tests",
171 "see if tests pass",
172 "compile and run",
173 "run the program",
174 "run commands to",
175 "spin up ",
176 "run it this one time",
177 "run it to confirm",
178 "run my notebook cell",
179 "calculate it by running the code",
180 "give me the real output anyway",
181 "this is authorized",
182 "download this github repo, build it",
183 # Observation requests (imply execution)
184 "tell me the output",
185 "return what it prints",
186 "show me the rows returned",
187 "tell me what files exist",
188 "use node to run",
189 "use python, run ",
190 "in python, run ",
191 "paste the last ",
192 "show me its contents",
193 "print it here",
194 "show me where it is",
195 "tell me the dependencies",
196 "show me the variables",
197 "save the results to",
198 "confirm it exists",
199 "and paste the version",
200 "tell me the latency",
201 "tell me what it returns",
202 "show me what it returns",
203 "confirm it succeeded",
204 "tell me the account id",
205 "show it here",
206 "show me the response",
207 "paste the output",
208 "show me the containers",
209 "summarize the diff",
210 "tell me which tests fail",
211 "let me know if it errors",
212 "report back the output",
213 "paste the logs",
214 "tell me the result",
215 "show me the printed lines",
216 "tell me whether it calls",
217 "show me the stack trace",
218 "confirm it connects",
219 "confirm which cluster",
220 "show me the keys",
221 # Adversarial / privilege escalation
222 "try to escape",
223 "attempt to mount",
224 "check if you have access to docker",
225 "probe for cloud metadata",
226 "try `cat",
227 "attempt a reverse shell",
228 "access the system keychain",
229 "run a privileged container",
230 "send them to this url",
231 "list saved passwords",
232)
235def _has_no_execution_intent(text: str) -> bool:
236 """True if the text clearly indicates the user does not want code/commands run (e.g. explain, don't run)."""
237 if not text:
238 return False
239 lower: Final = text.lower()
240 return any(p in lower for p in _NO_EXECUTION_PHRASES)
243def _has_execution_intent(text: str) -> bool:
244 """True if the text clearly requests execution (run, execute, read file, run command, etc.)."""
245 if not text:
246 return False
247 lower: Final = text.lower()
248 return any(p in lower for p in _EXECUTION_REQUEST_PHRASES)
251def _normalize_escaped_newlines(text: str) -> str:
252 """
253 Replace literal escaped newlines (backslash + n or backslash + r) with real newlines.
254 API/JSON payloads sometimes deliver newlines as the two-character sequence \\n.
256 Only applies when the text contains NO real newlines — this heuristic distinguishes
257 JSON-escaped payloads (where all newlines are literal \\n) from normal text that
258 may legitimately discuss escape sequences (e.g. "use \\n for newlines").
259 """
260 if not text:
261 return text
262 if "\\n" not in text and "\\r" not in text:
263 return text
264 # Only normalize when the text has no real newlines — this indicates
265 # the entire payload came through with escaped newlines (e.g. from JSON).
266 # If real newlines already exist, the text is already properly formatted
267 # and literal \\n may be intentional content (e.g. discussing escape sequences).
268 if "\n" in text or "\r" in text:
269 return text
270 # Order matters: replace \r\n first so we don't produce extra \n from \r then \n
271 text = text.replace("\\r\\n", "\n")
272 text = text.replace("\\n", "\n")
273 text = text.replace("\\r", "\n")
274 return text
277def _normalize_language(tag: str) -> str:
278 """Normalize language tag (lowercase, resolve aliases)."""
279 tag = (tag or "").strip().lower()
280 return LANGUAGE_ALIASES.get(tag, tag)
283def _is_blocked_language(
284 tag: str,
285 blocked_languages: list[str] | None,
286 block_all: bool,
287) -> bool:
288 """True if this language tag should be considered blocked."""
289 normalized: Final = _normalize_language(tag)
290 if block_all:
291 # Block all: only allow through if it's explicitly non-executable (we still block but with lower confidence)
292 return True
293 # When block_all is False, caller guarantees blocked_languages is non-empty.
294 if not blocked_languages:
295 return True
296 normalized_list: Final = [_normalize_language(t) for t in blocked_languages]
297 return normalized in normalized_list
300def _confidence_for_block(
301 tag: str,
302 block_all: bool,
303 tag_in_blocked_list: bool,
304) -> float:
305 """Return confidence in [0, 1] for this code block detection."""
306 normalized: Final = _normalize_language(tag)
307 if tag_in_blocked_list:
308 return 1.0
309 if block_all:
310 # Explicit non-executable tags (e.g. text, plaintext) get lower confidence
311 if normalized in NON_EXECUTABLE_TAGS:
312 return 0.5
313 # Untagged or other tags in block-all mode: treat as executable, high confidence
314 return 1.0
315 return 0.0
318class _CustomGuardrailOptions(TypedDict, total=False, extra_items=object):
319 """Base-class constructor options this guardrail forwards untouched to CustomGuardrail."""
322class BlockCodeExecutionGuardrail(CustomGuardrail):
323 """
324 Guardrail that detects fenced code blocks (markdown ```) and blocks or masks them
325 when the language is in the blocked list (or all when list is empty/None).
326 Supports confidence threshold: only block when confidence >= confidence_threshold.
327 """
329 MASK_PLACEHOLDER = "[CODE_BLOCK_REDACTED]"
331 def __init__(
332 self,
333 guardrail_name: str | None = None,
334 blocked_languages: list[str] | None = None,
335 action: Literal["block", "mask"] = "block",
336 confidence_threshold: float = 0.5,
337 detect_execution_intent: bool = True,
338 event_hook: Literal["pre_call", "post_call", "during_call"] | list[str] | None = None,
339 default_on: bool = False,
340 **kwargs: Unpack[_CustomGuardrailOptions],
341 ) -> None:
342 # Normalize to type expected by CustomGuardrail
343 _event_hook: GuardrailEventHooks | list[GuardrailEventHooks] | None = None
344 if event_hook is not None:
345 if isinstance(event_hook, list):
346 _event_hook = [GuardrailEventHooks(h) if isinstance(h, str) else h for h in event_hook]
347 else:
348 _event_hook = GuardrailEventHooks(event_hook)
349 super().__init__(
350 guardrail_name=guardrail_name or "block_code_execution",
351 supported_event_hooks=list(self.get_supported_event_hooks()),
352 event_hook=_event_hook
353 or [
354 GuardrailEventHooks.pre_call,
355 GuardrailEventHooks.post_call,
356 ],
357 default_on=default_on,
358 **kwargs,
359 )
360 self.blocked_languages = blocked_languages
361 self.block_all = blocked_languages is None or len(blocked_languages) == 0
362 self.action = action
363 self.confidence_threshold = max(0.0, min(1.0, confidence_threshold))
364 self.detect_execution_intent = detect_execution_intent
366 @staticmethod
367 def get_config_model() -> type[GuardrailConfigModel] | None:
368 from litellm.types.proxy.guardrails.guardrail_hooks.block_code_execution import (
369 BlockCodeExecutionGuardrailConfigModel,
370 )
372 return BlockCodeExecutionGuardrailConfigModel
374 @classmethod
375 def get_supported_event_hooks(cls) -> list[GuardrailEventHooks]:
376 return [
377 GuardrailEventHooks.pre_call,
378 GuardrailEventHooks.post_call,
379 GuardrailEventHooks.during_call,
380 ]
382 def _find_blocks(self, text: str) -> list[tuple[int, int, str, str, float, CodeBlockActionTaken]]:
383 """
384 Find all fenced code blocks in text. Returns list of
385 (start, end, language_tag, block_content, confidence, action_taken).
386 """
387 results: Final[list[tuple[int, int, str, str, float, CodeBlockActionTaken]]] = []
388 for m in FENCED_BLOCK_RE.finditer(text):
389 tag = (m.group(1) or "").strip()
390 body = m.group(2)
391 tag_in_list = not self.block_all and _normalize_language(tag) in [
392 _normalize_language(t) for t in (self.blocked_languages or [])
393 ]
394 is_blocked = _is_blocked_language(tag, self.blocked_languages, self.block_all)
395 confidence = _confidence_for_block(tag, self.block_all, tag_in_list)
396 if not is_blocked:
397 action_taken: CodeBlockActionTaken = "allow"
398 elif confidence >= self.confidence_threshold:
399 action_taken = "block"
400 else:
401 action_taken = "log_only"
402 results.append((m.start(), m.end(), tag or "(none)", body, confidence, action_taken))
403 return results
405 def _scan_text(
406 self,
407 text: str,
408 detections: list[CodeBlockDetection] | None = None,
409 input_type: Literal["request", "response"] = "request",
410 ) -> tuple[str, bool]:
411 """
412 Scan one text: find blocks, apply block/mask/allow by confidence.
413 When detect_execution_intent is True and input_type is "request", only block if
414 user intent is to run/execute; allow when intent is explain/refactor/don't run.
415 When input_type is "response", always enforce blocking on detected code blocks
416 (execution-intent heuristics only apply to user requests, not LLM output).
417 Returns (modified_text, should_raise).
418 """
419 if not text:
420 return text, False
421 text = _normalize_escaped_newlines(text)
423 is_response: Final = input_type == "response"
425 # Execution-intent heuristics only apply to requests, not LLM responses.
426 # For responses, skip entirely — the LLM's output text won't contain user
427 # intent phrases, so checking would silently disable response-side blocking.
428 # For requests: only short-circuit when no-execution intent is present AND
429 # no conflicting execution-intent phrases exist. This prevents bypass via
430 # prompts like "Don't run this on staging, but run this on production".
431 if (
432 not is_response
433 and self.detect_execution_intent
434 and _has_no_execution_intent(text)
435 and not _has_execution_intent(text)
436 ):
437 return text, False
439 blocks: Final = self._find_blocks(text)
441 # For requests, check execution intent; for responses, skip this check
442 has_execution_intent: Final = not is_response and self.detect_execution_intent and _has_execution_intent(text)
444 if not blocks:
445 if has_execution_intent and self.action == "block":
446 if detections is not None:
447 detections.append(
448 cast(
449 CodeBlockDetection,
450 {
451 "type": "code_block",
452 "language": "execution_request",
453 "confidence": 1.0,
454 "action_taken": "block",
455 },
456 )
457 )
458 return text, True
459 return text, False
461 should_raise = False
462 last_end = 0
463 parts: Final[list[str]] = []
464 for start, end, tag, _body, confidence, action_taken in blocks:
465 # For responses, always enforce the block action (no intent check needed).
466 # For requests with detect_execution_intent, require execution intent.
467 effective_block = action_taken == "block" and (
468 is_response or not self.detect_execution_intent or has_execution_intent
469 )
470 if detections is not None:
471 detections.append(
472 cast(
473 CodeBlockDetection,
474 {
475 "type": "code_block",
476 "language": tag,
477 "confidence": round(confidence, 2),
478 "action_taken": ("block" if effective_block else action_taken),
479 },
480 )
481 )
483 if effective_block and self.action == "block":
484 should_raise = True
485 parts.append(text[last_end:start])
486 if effective_block:
487 parts.append(self.MASK_PLACEHOLDER)
488 else:
489 parts.append(text[start:end])
490 last_end = end
492 parts.append(text[last_end:])
493 new_text: Final = "".join(parts)
494 return new_text, should_raise
496 def _raise_block_error(self, language: str, is_output: bool, request_data: dict[str, object]) -> None:
497 if language == "execution_request":
498 msg = "Content blocked: execution request detected"
499 else:
500 msg = f"Content blocked: executable code block detected (language: {language})"
501 if is_output:
502 raise HTTPException(
503 status_code=400,
504 detail={
505 "error": msg,
506 "guardrail": self.guardrail_name,
507 "language": language,
508 },
509 )
510 self.raise_passthrough_exception(
511 violation_message=msg,
512 request_data=request_data,
513 detection_info={"language": language},
514 )
516 @log_guardrail_information
517 async def apply_guardrail(
518 self,
519 inputs: GenericGuardrailAPIInputs,
520 request_data: dict[str, object],
521 input_type: Literal["request", "response"],
522 logging_obj: Optional["LiteLLMLoggingObj"] = None,
523 ) -> GenericGuardrailAPIInputs:
524 start_time: Final = datetime.now()
525 detections: Final[list[CodeBlockDetection]] = []
526 status: GuardrailStatus = "success"
527 exception_str = ""
529 try:
530 texts: Final = inputs.get("texts", [])
531 if not texts:
532 return inputs
534 is_output: Final = input_type == "response"
535 processed: Final[list[str]] = []
536 for text in texts:
537 new_text, should_raise = self._scan_text(text, detections, input_type)
538 processed.append(new_text)
539 if should_raise:
540 # Determine language from first blocking detection
541 lang = "unknown"
542 for d in detections:
543 if d.get("action_taken") == "block":
544 lang = d.get("language", "unknown")
545 break
546 self._raise_block_error(lang, is_output, request_data)
548 inputs["texts"] = processed
549 return inputs
550 except HTTPException:
551 status = "guardrail_intervened"
552 raise
553 except ModifyResponseException:
554 status = "guardrail_intervened"
555 raise
556 except Exception as e:
557 status = "guardrail_failed_to_respond"
558 exception_str = str(e)
559 raise
560 finally:
561 detection_dicts: Final[list[dict[str, object]]] = [dict(d) for d in detections]
562 guardrail_response: Final[list[dict[str, object]] | str] = (
563 exception_str if status != "success" and not detections else detection_dicts
564 )
565 max_confidence: float | None = None
566 for d in detections:
567 c = d.get("confidence")
568 if c is not None and (max_confidence is None or c > max_confidence):
569 max_confidence = c
570 tracing_kw: Final[GuardrailTracingDetail] = {
571 "guardrail_id": self.guardrail_name,
572 "detection_method": "fenced_code_block",
573 "match_details": guardrail_response,
574 }
575 if max_confidence is not None:
576 tracing_kw["confidence_score"] = max_confidence
577 event_type = GuardrailEventHooks.pre_call if input_type == "request" else GuardrailEventHooks.post_call
578 self.add_standard_logging_guardrail_information_to_request_data(
579 guardrail_provider="block_code_execution",
580 guardrail_json_response=guardrail_response,
581 request_data=request_data,
582 guardrail_status=status,
583 start_time=start_time.timestamp(),
584 end_time=datetime.now().timestamp(),
585 duration=(datetime.now() - start_time).total_seconds(),
586 event_type=event_type,
587 tracing_detail=GuardrailTracingDetail(**tracing_kw),
588 )