Coverage for open_webui/tools/knowledge_fs.py: 6%
681 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1"""
2Knowledge Base Filesystem Interface.
4Provides a filesystem-like command interface (ls, cat, grep, find, etc.)
5for AI models to interact with knowledge bases using commands they already know.
7Re-exported through builtin.py for consistent imports.
8"""
10import asyncio
11import contextvars
12import logging
13import re
14import shlex
15import time
16from collections.abc import Callable
17from contextlib import contextmanager
18from typing import Optional
20import re2
21from fastapi import Request
23from open_webui.env import (
24 KB_EXEC_MAX_GREP_FILES,
25 KB_EXEC_MAX_OUTPUT_CHARS,
26 KNOWLEDGE_GREP_MAX_MATCHES,
27)
29log = logging.getLogger(__name__)
31DEFAULT_HEAD_LINES = 10
32DEFAULT_TAIL_LINES = 10
34# Total matching time allowed per tool call, checked between RE2's linear-time searches.
35MATCH_BUDGET_SECONDS = 2.0
36MAX_SEARCH_PATTERN_LENGTH = 4_096
39class MatchBudgetExceeded(Exception):
40 """A tool call spent its whole matching budget, so the caller reports it."""
43class MatchBudget:
44 """Matching time remaining, counted only inside search() so awaits do not consume it."""
46 def __init__(self):
47 self.remaining = MATCH_BUDGET_SECONDS
50# Scoped to the running task, so one budget covers every matcher a command builds without
51# threading it through each handler.
52_active_budget: contextvars.ContextVar[MatchBudget | None] = contextvars.ContextVar('kb_match_budget', default=None)
55@contextmanager
56def match_budget():
57 """Bound the matching time of one tool call rather than of each search it runs."""
58 token = _active_budget.set(MatchBudget())
59 try:
60 yield
61 finally:
62 _active_budget.reset(token)
65# =============================================================================
66# SHARED REGEX UTILITIES — also used by builtin.py grep_knowledge_files
67# =============================================================================
70def is_regex_pattern(pattern: str) -> bool:
71 r"""Detect if a pattern looks like regex (|, .*, .+, \d, \w, \s, [...])."""
72 return (
73 '|' in pattern
74 or '.*' in pattern
75 or '.+' in pattern
76 or '.?' in pattern
77 or r'\d' in pattern
78 or r'\w' in pattern
79 or r'\s' in pattern
80 or bool(re.search(r'\[.+\]', pattern))
81 )
84def normalize_regex(pattern: str) -> str:
85 r"""Normalize POSIX BRE patterns to Python regex (\| → |)."""
86 # Two passes: an escaped backslash in front of a pipe leaves a second escape behind.
87 return pattern.replace(r'\|', '|').replace(r'\|', '|')
90def build_matcher(pattern: str, case_insensitive: bool = False, use_regex: bool = False) -> tuple:
91 """Build a matcher function. Returns (match_fn, error_str_or_None)."""
92 if len(pattern) > MAX_SEARCH_PATTERN_LENGTH:
93 return None, f'Search patterns over {MAX_SEARCH_PATTERN_LENGTH} characters are not supported'
95 if not use_regex and is_regex_pattern(pattern):
96 use_regex = True
98 if use_regex:
99 normalized = normalize_regex(pattern)
100 try:
101 options = re2.Options()
102 options.case_sensitive = not case_insensitive
103 options.max_mem = 1 << 20 # Bound compiled programs and the engine's matching cache to 1 MiB.
104 options.log_errors = False
105 compiled = re2.compile(normalized, options=options)
106 except re2.error as e:
107 return None, f'Invalid or unsupported regex (RE2 syntax): {e}'
109 budget = _active_budget.get() or MatchBudget()
111 def matches(line: str) -> bool:
112 started = time.monotonic()
113 try:
114 if budget.remaining <= 0:
115 raise TimeoutError
116 matched = bool(compiled.search(line))
117 if time.monotonic() - started >= budget.remaining:
118 raise TimeoutError
119 return matched
120 except TimeoutError:
121 raise MatchBudgetExceeded(f'Search exceeded {MATCH_BUDGET_SECONDS:g}s, narrow the pattern') from None
122 finally:
123 budget.remaining -= time.monotonic() - started
125 return matches, None
126 else:
127 sp = pattern.lower() if case_insensitive else pattern
128 return (lambda line: sp in (line.lower() if case_insensitive else line)), None
131# =============================================================================
132# COMMAND PARSING
133# =============================================================================
136def _parse_pipeline(command: str) -> list[list[str]]:
137 """Split command on pipes, then tokenize each segment."""
138 # Split on | but not inside quotes
139 segments = []
140 current = []
141 in_single = False
142 in_double = False
143 buf = []
145 for ch in command:
146 if ch == "'" and not in_double:
147 in_single = not in_single
148 buf.append(ch)
149 elif ch == '"' and not in_single:
150 in_double = not in_double
151 buf.append(ch)
152 elif ch == '|' and not in_single and not in_double:
153 segments.append(''.join(buf).strip())
154 buf = []
155 else:
156 buf.append(ch)
158 remaining = ''.join(buf).strip()
159 if remaining:
160 segments.append(remaining)
162 result = []
163 for seg in segments:
164 if not seg:
165 continue
166 try:
167 tokens = shlex.split(seg)
168 except ValueError:
169 # Fallback for malformed quotes
170 tokens = seg.split()
171 if tokens:
172 result.append(tokens)
174 return result
177def _extract_flags(tokens: list[str]) -> tuple[set[str], list[str]]:
178 """Extract single-char flags (e.g. -i, -l, -c, -n, -la) from tokens.
180 Returns (flags_set, remaining_args).
181 """
182 flags = set()
183 args = []
184 for token in tokens:
185 if token.startswith('-') and len(token) > 1 and not token[1:].isdigit():
186 # Could be -ilc (combined) or -20 (number, skip)
187 for ch in token[1:]:
188 flags.add(ch)
189 else:
190 args.append(token)
191 return flags, args
194def _extract_numeric_flag(tokens: list[str]) -> tuple[Optional[int], list[str]]:
195 """Extract a numeric flag like -20 from tokens. Returns (number, remaining)."""
196 num = None
197 remaining = []
198 for token in tokens:
199 if num is None and re.match(r'^-\d+$', token):
200 num = int(token[1:])
201 else:
202 remaining.append(token)
203 return num, remaining
206# =============================================================================
207# DIRECTORY TREE & PATH RESOLUTION
208# =============================================================================
211async def _build_directory_tree(knowledge_id: str) -> dict:
212 """Build an in-memory directory tree for a KB. Returns {dirs, files, path_to_dir_id, dir_id_to_path}."""
213 from open_webui.models.knowledge import Knowledges
215 all_dirs = await Knowledges.get_all_directories(knowledge_id)
216 files_with_dirs = await Knowledges.get_files_with_directory_ids(knowledge_id)
218 # Build dir_id -> dir info map
219 dir_map = {}
220 for d in all_dirs:
221 dir_map[d.id] = {'name': d.name, 'parent_id': d.parent_id, 'id': d.id}
223 # Compute full path for each directory
224 dir_id_to_path = {}
226 def _get_dir_path(dir_id):
227 if dir_id in dir_id_to_path:
228 return dir_id_to_path[dir_id]
229 d = dir_map.get(dir_id)
230 if not d:
231 return ''
232 if d['parent_id'] and d['parent_id'] in dir_map:
233 parent_path = _get_dir_path(d['parent_id'])
234 path = f'{parent_path}/{d["name"]}' if parent_path else d['name']
235 else:
236 path = d['name']
237 dir_id_to_path[dir_id] = path
238 return path
240 for d_id in dir_map:
241 _get_dir_path(d_id)
243 path_to_dir_id = {v: k for k, v in dir_id_to_path.items()}
245 # Build file list with paths
246 files = []
247 for file_model, directory_id in files_with_dirs:
248 if directory_id and directory_id in dir_id_to_path:
249 file_path = f'{dir_id_to_path[directory_id]}/{file_model.filename}'
250 else:
251 file_path = file_model.filename
252 files.append(
253 {
254 'id': file_model.id,
255 'filename': file_model.filename,
256 'path': file_path,
257 'directory_id': directory_id,
258 'size': file_model.meta.get('size') if file_model.meta else None,
259 'type': file_model.meta.get('content_type') if file_model.meta else None,
260 'updated_at': file_model.updated_at,
261 }
262 )
264 return {
265 'dirs': dir_map,
266 'files': files,
267 'path_to_dir_id': path_to_dir_id,
268 'dir_id_to_path': dir_id_to_path,
269 }
272def _resolve_path(path: str, tree: dict) -> str | None:
273 """Resolve a directory path string to a dir_id. Returns None if not found."""
274 path = path.strip('/')
275 return tree['path_to_dir_id'].get(path)
278def _get_files_in_dir(tree: dict, dir_id: str | None) -> list[dict]:
279 """Get files directly in a directory (None = root)."""
280 return [f for f in tree['files'] if f['directory_id'] == dir_id]
283def _get_subdirs(tree: dict, parent_id: str | None) -> list[dict]:
284 """Get immediate child directories."""
285 return sorted([d for d in tree['dirs'].values() if d['parent_id'] == parent_id], key=lambda d: d['name'])
288def _get_files_under_dir(tree: dict, dir_id: str) -> list[dict]:
289 """Get all files recursively under a directory."""
290 # Collect this dir + all descendant dir IDs
291 target_ids = {dir_id}
292 changed = True
293 while changed:
294 changed = False
295 for d in tree['dirs'].values():
296 if d['parent_id'] in target_ids and d['id'] not in target_ids:
297 target_ids.add(d['id'])
298 changed = True
299 return [f for f in tree['files'] if f['directory_id'] in target_ids]
302# =============================================================================
303# FILE RESOLUTION & ACCESS CONTROL
304# =============================================================================
307async def _get_accessible_kb_ids(
308 user: dict, model_knowledge: list[dict] | None, knowledge_id: str | None = None
309) -> list[tuple[str, str, str]]:
310 """Get list of (kb_id, kb_name, kb_description) the user can access."""
311 from open_webui.models.access_grants import AccessGrants
312 from open_webui.models.groups import Groups
313 from open_webui.models.knowledge import Knowledges
315 user_id = user.get('id')
316 user_role = user.get('role', 'user')
317 user_group_ids = [g.id for g in await Groups.get_groups_by_member_id(user_id)]
319 async def _has_access(kb):
320 return (
321 user_role == 'admin'
322 or kb.user_id == user_id
323 or await AccessGrants.has_access(
324 user_id=user_id,
325 resource_type='knowledge',
326 resource_id=kb.id,
327 permission='read',
328 user_group_ids=set(user_group_ids),
329 )
330 )
332 result = []
334 if model_knowledge:
335 attached_kb_ids = set()
336 for item in model_knowledge:
337 if item.get('type') == 'collection':
338 attached_kb_ids.add(item.get('id'))
339 if knowledge_id:
340 if knowledge_id not in attached_kb_ids:
341 return []
342 attached_kb_ids = {knowledge_id}
343 for kb_id in attached_kb_ids:
344 kb = await Knowledges.get_knowledge_by_id(kb_id)
345 if kb and await _has_access(kb):
346 result.append((kb.id, kb.name, kb.description or ''))
347 elif knowledge_id:
348 kb = await Knowledges.get_knowledge_by_id(knowledge_id)
349 if kb and await _has_access(kb):
350 result.append((kb.id, kb.name, kb.description or ''))
351 else:
352 search = await Knowledges.search_knowledge_bases(
353 user_id,
354 filter={'query': '', 'user_id': user_id, 'group_ids': user_group_ids},
355 skip=0,
356 limit=50,
357 )
358 for kb in search.items:
359 result.append((kb.id, kb.name, kb.description or ''))
361 return result
364async def _get_accessible_files(
365 user: dict, model_knowledge: list[dict] | None, knowledge_id: str | None = None
366) -> list[dict]:
367 """Get all files the user can access, with KB metadata and directory_id (no path computation)."""
368 from open_webui.models.files import Files
369 from open_webui.models.knowledge import Knowledges
371 kb_ids = await _get_accessible_kb_ids(user, model_knowledge, knowledge_id)
372 files = []
374 for kb_id, kb_name, _ in kb_ids:
375 kb_files = await Knowledges.get_files_with_directory_ids(kb_id)
376 for file_model, dir_id in kb_files:
377 files.append(
378 {
379 'id': file_model.id,
380 'filename': file_model.filename,
381 'directory_id': dir_id,
382 'size': file_model.meta.get('size') if file_model.meta else None,
383 'type': file_model.meta.get('content_type') if file_model.meta else None,
384 'updated_at': file_model.updated_at,
385 'knowledge_id': kb_id,
386 'knowledge_name': kb_name,
387 }
388 )
390 # Also handle directly attached files (not in any KB)
391 if model_knowledge:
392 attached_file_ids = set()
393 for item in model_knowledge:
394 if item.get('type') == 'file':
395 attached_file_ids.add(item.get('id'))
396 for fid in attached_file_ids:
397 f = await Files.get_file_by_id(fid)
398 if f:
399 files.append(
400 {
401 'id': f.id,
402 'filename': f.filename,
403 'directory_id': None,
404 'size': f.meta.get('size') if f.meta else None,
405 'type': f.meta.get('content_type') if f.meta else None,
406 'updated_at': f.updated_at,
407 'knowledge_id': None,
408 'knowledge_name': None,
409 }
410 )
412 return files
415async def _resolve_dir_path(path: str, knowledge_id: str) -> str | None:
416 """Walk a directory path one level at a time. Returns dir_id or None."""
417 from open_webui.models.knowledge import Knowledges
419 parts = path.strip('/').split('/')
420 current_parent = None
422 for part in parts:
423 dirs = await Knowledges.get_directories(knowledge_id, parent_id=current_parent)
424 match = next((d for d in dirs if d.name == part), None)
425 if not match:
426 return None
427 current_parent = match.id
429 return current_parent
432async def _get_descendant_dir_ids(dir_id: str, knowledge_id: str) -> set[str]:
433 """Collect all descendant directory IDs recursively."""
434 from open_webui.models.knowledge import Knowledges
436 result = {dir_id}
437 queue = [dir_id]
438 while queue:
439 parent = queue.pop()
440 children = await Knowledges.get_directories(knowledge_id, parent_id=parent)
441 for child in children:
442 if child.id not in result:
443 result.add(child.id)
444 queue.append(child.id)
445 return result
448async def _resolve_file(ref: str, user: dict, model_knowledge: list[dict] | None) -> dict | None:
449 """Resolve a file reference (ID, path, or filename) to a file info dict with content."""
450 from open_webui.models.files import Files
452 # Get accessible file IDs (lightweight — no path computation)
453 accessible = await _get_accessible_files(user, model_knowledge)
454 accessible_ids = {fi['id'] for fi in accessible}
456 # Try direct ID lookup first — but verify access
457 f = await Files.get_file_by_id(ref)
458 if f and f.data:
459 if f.id not in accessible_ids:
460 return None
461 return {
462 'id': f.id,
463 'filename': f.filename,
464 'content': f.data.get('content', ''),
465 'meta': f.meta,
466 'updated_at': f.updated_at,
467 'created_at': f.created_at,
468 }
470 # Try path match (e.g. "docs/api/auth.md") — lazy dir walk
471 ref_clean = ref.strip('/')
472 if '/' in ref_clean:
473 dir_path, filename = ref_clean.rsplit('/', 1)
474 # Try resolving in each accessible KB
475 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')}
476 for kb_id in kb_ids:
477 dir_id = await _resolve_dir_path(dir_path, kb_id)
478 if dir_id is None:
479 continue
480 # Find file with that name in that directory
481 matches = [fi for fi in accessible if fi['filename'] == filename and fi['directory_id'] == dir_id]
482 if len(matches) == 1:
483 f = await Files.get_file_by_id(matches[0]['id'])
484 if f and f.data:
485 return {
486 'id': f.id,
487 'filename': f.filename,
488 'content': f.data.get('content', ''),
489 'meta': f.meta,
490 'updated_at': f.updated_at,
491 'created_at': f.created_at,
492 'knowledge_id': matches[0].get('knowledge_id'),
493 'knowledge_name': matches[0].get('knowledge_name'),
494 }
496 # Try filename match within accessible files
497 matches = [fi for fi in accessible if fi['filename'] == ref]
499 if len(matches) == 1:
500 f = await Files.get_file_by_id(matches[0]['id'])
501 if f and f.data:
502 return {
503 'id': f.id,
504 'filename': f.filename,
505 'content': f.data.get('content', ''),
506 'meta': f.meta,
507 'updated_at': f.updated_at,
508 'created_at': f.created_at,
509 'knowledge_id': matches[0].get('knowledge_id'),
510 'knowledge_name': matches[0].get('knowledge_name'),
511 }
512 elif len(matches) > 1:
513 return {
514 'error': f'Ambiguous filename "{ref}". Use full path to disambiguate:\n'
515 + '\n'.join(f' {m["id"]} {m["filename"]} ({m.get("knowledge_name", "direct")})' for m in matches)
516 }
518 return None
521async def _get_file_content(file_id: str) -> str | None:
522 """Get file content by ID."""
523 from open_webui.models.files import Files
525 f = await Files.get_file_by_id(file_id)
526 if f and f.data:
527 return f.data.get('content', '')
528 return None
531# =============================================================================
532# COMMAND HANDLERS
533# =============================================================================
536async def _kb_ls(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str:
537 """List files and directories. Supports: ls, ls <path>, ls -a (flat)."""
538 from open_webui.models.knowledge import Knowledges
540 flat_mode = 'a' in flags
541 path_arg = args[0] if args else None
543 kb_ids = await _get_accessible_kb_ids(user, model_knowledge, knowledge_id=None)
544 direct_files = (
545 [f for f in await _get_accessible_files(user, model_knowledge) if not f.get('knowledge_id')]
546 if model_knowledge
547 else []
548 )
550 # If path_arg looks like a KB ID, scope to that KB
551 target_kb_id = None
552 dir_path = None
553 if path_arg:
554 for kb_id, kb_name, _ in kb_ids:
555 if kb_id == path_arg:
556 target_kb_id = kb_id
557 break
558 if not target_kb_id:
559 dir_path = path_arg.strip('/')
561 if target_kb_id:
562 kb_ids = [(kid, kn, kd) for kid, kn, kd in kb_ids if kid == target_kb_id]
564 if not kb_ids and not direct_files:
565 return 'No knowledge bases found.'
567 lines = []
568 for kb_id, kb_name, kb_desc in kb_ids:
569 header = f'Knowledge Base: {kb_name} ({kb_id})'
570 if kb_desc:
571 header += f'\n {kb_desc}'
572 lines.append(header)
574 if flat_mode:
575 # Flat mode: build full tree (legitimate use)
576 tree = await _build_directory_tree(kb_id)
577 for f in _sort_files(tree['files'], flags):
578 lines.append(f' {f["id"]} {f["path"]} {_fmt_size(f)} {_fmt_date(f)}')
579 lines.append('')
580 continue
582 # Resolve target directory (lazy walk)
583 target_dir_id = None
584 if dir_path:
585 target_dir_id = await _resolve_dir_path(dir_path, kb_id)
586 if target_dir_id is None:
587 lines.append(f' Directory not found: {dir_path}')
588 lines.append('')
589 continue
590 lines.append(f' Path: {dir_path}/')
592 # Show subdirectories (targeted query — only this level)
593 subdirs = await Knowledges.get_directories(kb_id, parent_id=target_dir_id)
594 for d in subdirs:
595 lines.append(f' 📁 {d.name}/')
597 # Show files at this level (filter from accessible files)
598 accessible = await _get_accessible_files(user, model_knowledge, knowledge_id=kb_id)
599 dir_files = [f for f in accessible if f['directory_id'] == target_dir_id]
600 for f in _sort_files(dir_files, flags):
601 lines.append(f' {f["id"]} {f["filename"]} {_fmt_size(f)} {_fmt_date(f)}')
603 if not subdirs and not dir_files:
604 lines.append(' (empty)')
605 lines.append('')
607 if direct_files and not target_kb_id and not dir_path:
608 lines.append('Attached Files:')
609 for f in _sort_files(direct_files, flags):
610 lines.append(f' {f["id"]} {f["filename"]} {_fmt_size(f)} {_fmt_date(f)}')
611 lines.append('')
613 return '\n'.join(lines).rstrip()
616def _sort_files(files: list[dict], flags: set[str]) -> list[dict]:
617 """ls-style ordering: by name or path, -t newest first, -S largest first, -r reverses."""
618 if 't' in flags:
619 files = sorted(files, key=lambda f: f.get('updated_at') or 0, reverse=True)
620 elif 'S' in flags:
621 files = sorted(files, key=lambda f: f.get('size') or 0, reverse=True)
622 else:
623 files = sorted(files, key=lambda f: f.get('path') or f['filename'])
624 return files[::-1] if 'r' in flags else files
627def _fmt_size(f: dict) -> str:
628 return f'{f["size"]:,} bytes' if f.get('size') else ''
631def _fmt_date(f: dict) -> str:
632 if f.get('updated_at'):
633 from datetime import datetime, timezone
635 dt = datetime.fromtimestamp(f['updated_at'], tz=timezone.utc)
636 return dt.strftime('%Y-%m-%d')
637 return ''
640async def _kb_cat(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str:
641 """Read file content. Use -n for line numbers."""
642 if not args:
643 return 'Usage: cat [-n] <file_id or filename>'
645 resolved = await _resolve_file(args[0], user, model_knowledge)
646 if not resolved:
647 return f'File not found: {args[0]}'
648 if 'error' in resolved:
649 return resolved['error']
651 content = resolved['content']
652 if 'n' in flags:
653 lines = content.split('\n')
654 content = '\n'.join(f'{i}: {line}' for i, line in enumerate(lines, 1))
656 return content
659async def _kb_head(
660 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None
661) -> str:
662 """First N lines of a file or piped input."""
663 n, args = _extract_numeric_flag(args)
664 if n is None:
665 n = DEFAULT_HEAD_LINES
667 if piped_input is not None:
668 lines = piped_input.split('\n')
669 return '\n'.join(lines[:n])
671 if not args:
672 return 'Usage: head [-N] <file>'
674 resolved = await _resolve_file(args[0], user, model_knowledge)
675 if not resolved:
676 return f'File not found: {args[0]}'
677 if 'error' in resolved:
678 return resolved['error']
680 lines = resolved['content'].split('\n')
681 total = len(lines)
682 result = '\n'.join(lines[:n])
683 if total > n:
684 result += f'\n[showing {n} of {total} lines]'
685 return result
688async def _kb_tail(
689 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None
690) -> str:
691 """Last N lines of a file or piped input."""
692 n, args = _extract_numeric_flag(args)
693 if n is None:
694 n = DEFAULT_TAIL_LINES
696 if piped_input is not None:
697 lines = piped_input.split('\n')
698 return '\n'.join(lines[-n:])
700 if not args:
701 return 'Usage: tail [-N] <file>'
703 resolved = await _resolve_file(args[0], user, model_knowledge)
704 if not resolved:
705 return f'File not found: {args[0]}'
706 if 'error' in resolved:
707 return resolved['error']
709 lines = resolved['content'].split('\n')
710 total = len(lines)
711 result = '\n'.join(lines[-n:])
712 if total > n:
713 result += f'\n[showing last {n} of {total} lines]'
714 return result
717def _match_lines(content: str, matches: Callable[[str], bool]) -> list[tuple[int, str]]:
718 return [(i, line) for i, line in enumerate(content.split('\n'), 1) if matches(line)]
721async def _kb_grep(
722 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None
723) -> str:
724 """Text search across files or piped input. Supports -E for regex."""
725 if not args:
726 return 'Usage: grep [-E] [-i] [-l] [-c] "pattern" [file] [*.ext]'
728 pattern = args[0]
729 file_ref = None
730 ext_filter = None
731 dir_scope = None
733 for arg in args[1:]:
734 if '*' in arg or arg.startswith('.'):
735 ext_filter = arg.lstrip('*').lstrip('.')
736 elif arg.endswith('/'):
737 dir_scope = arg.strip('/')
738 else:
739 file_ref = arg
741 case_insensitive = 'i' in flags
742 filenames_only = 'l' in flags
743 count_only = 'c' in flags
744 use_regex = 'E' in flags
746 _matches, err = await asyncio.to_thread(build_matcher, pattern, case_insensitive, use_regex)
747 if err:
748 return err
750 # Grep on piped input
751 if piped_input is not None:
752 found = await asyncio.to_thread(_match_lines, piped_input, _matches)
753 matched = [f'{i}: {line}' for i, line in found]
754 if count_only:
755 return str(len(matched))
756 if filenames_only:
757 return '(standard input)' if matched else f'No matches for "{pattern}"'
758 return '\n'.join(matched) if matched else f'No matches for "{pattern}"'
760 # Single file grep
761 if file_ref and not dir_scope:
762 resolved = await _resolve_file(file_ref, user, model_knowledge)
763 if not resolved:
764 # Maybe it's a directory path without trailing /
765 dir_scope = file_ref
766 elif 'error' in resolved:
767 return resolved['error']
768 else:
769 found = await asyncio.to_thread(_match_lines, resolved['content'], _matches)
770 matched = [f'{i}: {line}' for i, line in found]
772 if count_only:
773 return f'{resolved["id"]} {resolved["filename"]}: {len(matched)}'
774 if filenames_only:
775 return f'{resolved["id"]} {resolved["filename"]}' if matched else f'No matches for "{pattern}"'
777 if not matched:
778 return f'No matches for "{pattern}" in {resolved["filename"]}'
779 return '\n'.join(matched)
781 # Cross-file grep (optionally scoped to directory)
782 accessible = await _get_accessible_files(user, model_knowledge)
784 if dir_scope:
785 # Resolve directory and collect all descendant IDs
786 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')}
787 target_dir_ids = set()
788 for kb_id in kb_ids:
789 dir_id = await _resolve_dir_path(dir_scope, kb_id)
790 if dir_id:
791 desc = await _get_descendant_dir_ids(dir_id, kb_id)
792 target_dir_ids.update(desc)
793 if not target_dir_ids:
794 return f'No files found under "{dir_scope}/"'
795 accessible = [f for f in accessible if f.get('directory_id') in target_dir_ids]
796 if not accessible:
797 return f'No files found under "{dir_scope}/"'
799 if ext_filter:
800 accessible = [f for f in accessible if f['filename'].endswith(f'.{ext_filter}')]
802 if len(accessible) > KB_EXEC_MAX_GREP_FILES:
803 return f'Too many files ({len(accessible)}). Scope your search: grep "{pattern}" docs/ or grep "{pattern}" *.py'
805 from open_webui.models.files import Files
807 results = []
808 file_match_counts = []
809 files_with_matches = []
810 total_matches = 0
812 for file_info in accessible:
813 f = await Files.get_file_by_id(file_info['id'])
814 if not f or not f.data:
815 continue
817 content = f.data.get('content', '')
818 if not content:
819 continue
821 file_matches = await asyncio.to_thread(_match_lines, content, _matches)
823 if file_matches:
824 files_with_matches.append(file_info)
825 file_match_counts.append((file_info, len(file_matches)))
826 total_matches += len(file_matches)
828 if not count_only and not filenames_only:
829 for line_num, line_text in file_matches:
830 if len(results) < KNOWLEDGE_GREP_MAX_MATCHES:
831 results.append(f'{file_info["id"]} {file_info["filename"]}:{line_num}: {line_text.rstrip()}')
833 if count_only:
834 if not file_match_counts:
835 return f'No matches for "{pattern}"'
836 lines = [f'{fi["id"]} {fi["filename"]}: {cnt}' for fi, cnt in file_match_counts]
837 lines.append(f'Total: {total_matches} matches in {len(file_match_counts)} files')
838 return '\n'.join(lines)
840 if filenames_only:
841 if not files_with_matches:
842 return f'No matches for "{pattern}"'
843 return '\n'.join(f'{fi["id"]} {fi["filename"]}' for fi in files_with_matches)
845 if not results:
846 return f'No matches for "{pattern}" across {len(accessible)} files'
848 output = '\n'.join(results)
849 if total_matches > KNOWLEDGE_GREP_MAX_MATCHES:
850 output += f'\n[showing {KNOWLEDGE_GREP_MAX_MATCHES} of {total_matches} matches]'
851 return output
854async def _kb_find(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str:
855 """Find files by name/glob pattern, optionally scoped to a directory."""
856 if not args:
857 return 'Usage: find "*.md" or find docs/ "*.md"'
859 import fnmatch
861 # If two args and first looks like a dir scope
862 dir_scope = None
863 if len(args) >= 2 and ('/' in args[0] or not ('*' in args[0] or '?' in args[0])):
864 dir_scope = args[0].strip('/')
865 pattern = args[1]
866 else:
867 pattern = args[0]
869 accessible = await _get_accessible_files(user, model_knowledge)
871 if dir_scope:
872 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')}
873 target_dir_ids = set()
874 for kb_id in kb_ids:
875 dir_id = await _resolve_dir_path(dir_scope, kb_id)
876 if dir_id:
877 desc = await _get_descendant_dir_ids(dir_id, kb_id)
878 target_dir_ids.update(desc)
879 accessible = [f for f in accessible if f.get('directory_id') in target_dir_ids]
881 matched = [f for f in accessible if fnmatch.fnmatch(f['filename'], pattern)]
883 if not matched:
884 scope_str = f' under "{dir_scope}/"' if dir_scope else ''
885 return f'No files matching "{pattern}"{scope_str}'
887 lines = []
888 for f in _sort_files(matched, flags):
889 kb_info = f' ({f["knowledge_name"]})' if f.get('knowledge_name') else ''
890 lines.append(f'{f["id"]} {f["filename"]}{kb_info}')
891 return '\n'.join(lines)
894async def _kb_wc(
895 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None
896) -> str:
897 """Word, line, character counts."""
898 if piped_input is not None:
899 lines = piped_input.count('\n') + (1 if piped_input and not piped_input.endswith('\n') else 0)
900 words = len(piped_input.split())
901 chars = len(piped_input)
902 if 'l' in flags:
903 return str(lines)
904 return f' {lines} {words} {chars}'
906 if not args:
907 return 'Usage: wc [-l] <file>'
909 resolved = await _resolve_file(args[0], user, model_knowledge)
910 if not resolved:
911 return f'File not found: {args[0]}'
912 if 'error' in resolved:
913 return resolved['error']
915 content = resolved['content']
916 lines = content.count('\n') + (1 if content and not content.endswith('\n') else 0)
917 words = len(content.split())
918 chars = len(content)
920 if 'l' in flags:
921 return f' {lines} {resolved["filename"]}'
922 return f' {lines} {words} {chars} {resolved["filename"]}'
925async def _kb_stat(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str:
926 """File metadata."""
927 if not args:
928 return 'Usage: stat <file>'
930 resolved = await _resolve_file(args[0], user, model_knowledge)
931 if not resolved:
932 return f'File not found: {args[0]}'
933 if 'error' in resolved:
934 return resolved['error']
936 content = resolved['content']
937 lines = content.count('\n') + (1 if content and not content.endswith('\n') else 0)
938 words = len(content.split())
939 chars = len(content)
941 meta = resolved.get('meta') or {}
942 size = meta.get('size', chars)
943 content_type = meta.get('content_type', 'unknown')
945 out = [
946 f' File: {resolved["filename"]}',
947 f' ID: {resolved["id"]}',
948 f' Size: {size:,} bytes',
949 f' Type: {content_type}',
950 f' Lines: {lines:,}',
951 f' Words: {words:,}',
952 f' Chars: {chars:,}',
953 ]
955 if resolved.get('created_at'):
956 from datetime import datetime, timezone
958 dt = datetime.fromtimestamp(resolved['created_at'], tz=timezone.utc)
959 out.append(f' Created: {dt.strftime("%Y-%m-%d %H:%M:%S UTC")}')
960 if resolved.get('updated_at'):
961 from datetime import datetime, timezone
963 dt = datetime.fromtimestamp(resolved['updated_at'], tz=timezone.utc)
964 out.append(f' Updated: {dt.strftime("%Y-%m-%d %H:%M:%S UTC")}')
965 if resolved.get('knowledge_name'):
966 out.append(f' KB: {resolved["knowledge_name"]} ({resolved.get("knowledge_id", "")})')
968 return '\n'.join(out)
971async def _kb_sed(
972 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None
973) -> str:
974 """Extract line range from a file. Usage: sed -n 'M,Np' <file>"""
975 if piped_input is not None:
976 # sed on piped input: parse range from args
977 start, end = 1, None
978 if 'n' in flags and args:
979 m = re.match(r'^(\d+),(\d+)p?$', args[0])
980 if m:
981 start, end = int(m.group(1)), int(m.group(2))
982 args = args[1:]
983 lines = piped_input.split('\n')
984 selected = lines[max(0, start - 1) : (end or len(lines))]
985 return '\n'.join(selected)
987 # Parse: sed -n '40,60p' <file>
988 range_str = None
989 file_ref = None
991 for arg in args:
992 m = re.match(r"^'?(\d+),(\d+)p?'?$", arg)
993 if m:
994 range_str = arg
995 else:
996 file_ref = arg
998 if not range_str or not file_ref:
999 return "Usage: sed -n '40,60p' <file>"
1001 m = re.match(r"^'?(\d+),(\d+)p?'?$", range_str)
1002 start, end = int(m.group(1)), int(m.group(2))
1004 if start > end:
1005 return f'Invalid range: start ({start}) > end ({end})'
1007 resolved = await _resolve_file(file_ref, user, model_knowledge)
1008 if not resolved:
1009 return f'File not found: {file_ref}'
1010 if 'error' in resolved:
1011 return resolved['error']
1013 lines = resolved['content'].split('\n')
1014 total = len(lines)
1015 selected = lines[max(0, start - 1) : end]
1016 result = '\n'.join(selected)
1017 result += f'\n[lines {start}-{min(end, total)} of {total}]'
1018 return result
1021# =============================================================================
1022# PIPE EXECUTOR
1023# =============================================================================
1026async def _kb_tree(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str:
1027 """Show directory tree structure."""
1028 kb_ids = await _get_accessible_kb_ids(user, model_knowledge)
1029 direct_files = (
1030 [f for f in await _get_accessible_files(user, model_knowledge) if not f.get('knowledge_id')]
1031 if model_knowledge
1032 else []
1033 )
1034 if not kb_ids and not direct_files:
1035 return 'No knowledge bases found.'
1037 dir_scope = args[0].strip('/') if args else None
1038 output = []
1040 for kb_id, kb_name, kb_desc in kb_ids:
1041 tree = await _build_directory_tree(kb_id)
1042 header = f'Knowledge Base: {kb_name} ({kb_id})'
1043 if kb_desc:
1044 header += f'\n {kb_desc}'
1045 output.append(header)
1047 # Find root to start from
1048 root_dir_id = None
1049 if dir_scope:
1050 root_dir_id = _resolve_path(dir_scope, tree)
1051 if root_dir_id is None:
1052 output.append(f' Directory not found: {dir_scope}')
1053 output.append('')
1054 continue
1055 output.append(f' {dir_scope}/')
1057 def _render_tree(parent_id, prefix=' '):
1058 items = []
1059 subdirs = _get_subdirs(tree, parent_id)
1060 files = _sort_files(_get_files_in_dir(tree, parent_id), flags)
1061 entries = [('dir', d) for d in subdirs] + [('file', f) for f in files]
1063 for idx, (etype, entry) in enumerate(entries):
1064 is_last = idx == len(entries) - 1
1065 connector = '└── ' if is_last else '├── '
1066 child_prefix = prefix + (' ' if is_last else '│ ')
1068 if etype == 'dir':
1069 items.append(f'{prefix}{connector}📁 {entry["name"]}/')
1070 items.extend(_render_tree(entry['id'], child_prefix))
1071 else:
1072 items.append(f'{prefix}{connector}{entry["filename"]}')
1073 return items
1075 output.extend(_render_tree(root_dir_id))
1077 # Summary
1078 total_dirs = len(tree['dirs'])
1079 total_files = len(tree['files'])
1080 output.append(f'\n {total_dirs} directories, {total_files} files')
1081 output.append('')
1083 if direct_files and not dir_scope:
1084 output.append('Attached Files:')
1085 for idx, f in enumerate(_sort_files(direct_files, flags)):
1086 connector = '└── ' if idx == len(direct_files) - 1 else '├── '
1087 output.append(f' {connector}{f["filename"]}')
1088 output.append(f'\n 0 directories, {len(direct_files)} files')
1089 output.append('')
1091 return '\n'.join(output).rstrip()
1094COMMAND_MAP = {
1095 'ls': _kb_ls,
1096 'cat': _kb_cat,
1097 'head': _kb_head,
1098 'tail': _kb_tail,
1099 'grep': _kb_grep,
1100 'find': _kb_find,
1101 'wc': _kb_wc,
1102 'stat': _kb_stat,
1103 'sed': _kb_sed,
1104 'tree': _kb_tree,
1105}
1108async def _execute_pipeline(
1109 segments: list[list[str]],
1110 user: dict,
1111 model_knowledge: list[dict] | None,
1112) -> str:
1113 """Execute a pipeline of commands, passing text between them."""
1114 piped_input = None
1116 for tokens in segments:
1117 cmd_name = tokens[0].lower()
1118 rest = tokens[1:]
1120 handler = COMMAND_MAP.get(cmd_name)
1121 if not handler:
1122 return f'Unknown command: {cmd_name}. Available: {", ".join(sorted(COMMAND_MAP.keys()))}'
1124 flags, args = _extract_flags(rest)
1126 # Commands that accept piped input
1127 if piped_input is not None and cmd_name in ('head', 'tail', 'grep', 'wc', 'sed'):
1128 piped_input = await handler(args, flags, user, model_knowledge, piped_input=piped_input)
1129 else:
1130 piped_input = await handler(args, flags, user, model_knowledge)
1132 return piped_input or ''
1135# =============================================================================
1136# ENTRY POINT
1137# =============================================================================
1140async def kb_exec(
1141 command: str,
1142 __request__: Request = None,
1143 __user__: dict = None,
1144 __model_knowledge__: Optional[list[dict]] = None,
1145) -> str:
1146 """
1147 Run a filesystem command against the knowledge base.
1149 Commands:
1150 ls — list root files and directories
1151 ls docs/ — list contents of a directory
1152 ls -a — flat list of all files with full paths
1153 ls -t — newest modified first
1154 ls -S — largest first
1155 ls -r — reverse file order
1156 tree — recursive directory tree view
1157 tree docs/ — subtree from a directory
1158 cat -n <file> — read file with line numbers
1159 head -20 <file> — first 20 lines
1160 tail -10 <file> — last 10 lines
1161 sed -n '40,60p' <file> — view lines 40-60
1162 grep "text" <file> — exact text search (auto-detects regex)
1163 grep -i "text" — case-insensitive
1164 grep -l "text" — filenames-only
1165 grep -c "text" — match counts
1166 grep "text" docs/ — search within a directory
1167 grep "text" *.py — filter by extension
1168 find "*.md" — find files by glob
1169 find docs/ "*.md" — find within a directory
1170 find -t "*.md", tree -t — same sort flags as ls
1171 wc <file> — line/word/char counts
1172 stat <file> — file metadata
1174 Pipes: grep "auth" | head -5
1175 Files: reference by path (docs/api/auth.md), filename, or file ID
1176 Regex: RE2 syntax; no lookarounds/backreferences. Shorthand character classes are ASCII-only.
1178 :param command: A filesystem command string
1179 :return: Command output as text
1180 """
1181 if not __user__:
1182 return 'Error: User context not available'
1184 if not command or not command.strip():
1185 return 'Usage: kb_exec("<command>"). Run kb_exec("ls") to start.'
1187 try:
1188 segments = _parse_pipeline(command.strip())
1189 if not segments:
1190 return 'Could not parse command. Run kb_exec("ls") to start.'
1192 # One budget for the whole command: a per-search budget would multiply by segment count.
1193 with match_budget():
1194 output = await _execute_pipeline(segments, __user__, __model_knowledge__)
1195 if len(output) > KB_EXEC_MAX_OUTPUT_CHARS:
1196 output = output[:KB_EXEC_MAX_OUTPUT_CHARS] + (
1197 f'\n[output truncated at {KB_EXEC_MAX_OUTPUT_CHARS:,} chars'
1198 ' — narrow the command with a path, glob, head/tail/sed or grep]'
1199 )
1200 return output
1201 except Exception as e:
1202 log.exception(f'kb_exec error: {e}')
1203 return f'Error: {e}'