Coverage for open_webui/tools/knowledge_fs.py: 6%

681 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1""" 

2Knowledge Base Filesystem Interface. 

3 

4Provides a filesystem-like command interface (ls, cat, grep, find, etc.) 

5for AI models to interact with knowledge bases using commands they already know. 

6 

7Re-exported through builtin.py for consistent imports. 

8""" 

9 

10import asyncio 

11import contextvars 

12import logging 

13import re 

14import shlex 

15import time 

16from collections.abc import Callable 

17from contextlib import contextmanager 

18from typing import Optional 

19 

20import re2 

21from fastapi import Request 

22 

23from open_webui.env import ( 

24 KB_EXEC_MAX_GREP_FILES, 

25 KB_EXEC_MAX_OUTPUT_CHARS, 

26 KNOWLEDGE_GREP_MAX_MATCHES, 

27) 

28 

29log = logging.getLogger(__name__) 

30 

31DEFAULT_HEAD_LINES = 10 

32DEFAULT_TAIL_LINES = 10 

33 

34# Total matching time allowed per tool call, checked between RE2's linear-time searches. 

35MATCH_BUDGET_SECONDS = 2.0 

36MAX_SEARCH_PATTERN_LENGTH = 4_096 

37 

38 

39class MatchBudgetExceeded(Exception): 

40 """A tool call spent its whole matching budget, so the caller reports it.""" 

41 

42 

43class MatchBudget: 

44 """Matching time remaining, counted only inside search() so awaits do not consume it.""" 

45 

46 def __init__(self): 

47 self.remaining = MATCH_BUDGET_SECONDS 

48 

49 

50# Scoped to the running task, so one budget covers every matcher a command builds without 

51# threading it through each handler. 

52_active_budget: contextvars.ContextVar[MatchBudget | None] = contextvars.ContextVar('kb_match_budget', default=None) 

53 

54 

55@contextmanager 

56def match_budget(): 

57 """Bound the matching time of one tool call rather than of each search it runs.""" 

58 token = _active_budget.set(MatchBudget()) 

59 try: 

60 yield 

61 finally: 

62 _active_budget.reset(token) 

63 

64 

65# ============================================================================= 

66# SHARED REGEX UTILITIES — also used by builtin.py grep_knowledge_files 

67# ============================================================================= 

68 

69 

70def is_regex_pattern(pattern: str) -> bool: 

71 r"""Detect if a pattern looks like regex (|, .*, .+, \d, \w, \s, [...]).""" 

72 return ( 

73 '|' in pattern 

74 or '.*' in pattern 

75 or '.+' in pattern 

76 or '.?' in pattern 

77 or r'\d' in pattern 

78 or r'\w' in pattern 

79 or r'\s' in pattern 

80 or bool(re.search(r'\[.+\]', pattern)) 

81 ) 

82 

83 

84def normalize_regex(pattern: str) -> str: 

85 r"""Normalize POSIX BRE patterns to Python regex (\| → |).""" 

86 # Two passes: an escaped backslash in front of a pipe leaves a second escape behind. 

87 return pattern.replace(r'\|', '|').replace(r'\|', '|') 

88 

89 

90def build_matcher(pattern: str, case_insensitive: bool = False, use_regex: bool = False) -> tuple: 

91 """Build a matcher function. Returns (match_fn, error_str_or_None).""" 

92 if len(pattern) > MAX_SEARCH_PATTERN_LENGTH: 

93 return None, f'Search patterns over {MAX_SEARCH_PATTERN_LENGTH} characters are not supported' 

94 

95 if not use_regex and is_regex_pattern(pattern): 

96 use_regex = True 

97 

98 if use_regex: 

99 normalized = normalize_regex(pattern) 

100 try: 

101 options = re2.Options() 

102 options.case_sensitive = not case_insensitive 

103 options.max_mem = 1 << 20 # Bound compiled programs and the engine's matching cache to 1 MiB. 

104 options.log_errors = False 

105 compiled = re2.compile(normalized, options=options) 

106 except re2.error as e: 

107 return None, f'Invalid or unsupported regex (RE2 syntax): {e}' 

108 

109 budget = _active_budget.get() or MatchBudget() 

110 

111 def matches(line: str) -> bool: 

112 started = time.monotonic() 

113 try: 

114 if budget.remaining <= 0: 

115 raise TimeoutError 

116 matched = bool(compiled.search(line)) 

117 if time.monotonic() - started >= budget.remaining: 

118 raise TimeoutError 

119 return matched 

120 except TimeoutError: 

121 raise MatchBudgetExceeded(f'Search exceeded {MATCH_BUDGET_SECONDS:g}s, narrow the pattern') from None 

122 finally: 

123 budget.remaining -= time.monotonic() - started 

124 

125 return matches, None 

126 else: 

127 sp = pattern.lower() if case_insensitive else pattern 

128 return (lambda line: sp in (line.lower() if case_insensitive else line)), None 

129 

130 

131# ============================================================================= 

132# COMMAND PARSING 

133# ============================================================================= 

134 

135 

136def _parse_pipeline(command: str) -> list[list[str]]: 

137 """Split command on pipes, then tokenize each segment.""" 

138 # Split on | but not inside quotes 

139 segments = [] 

140 current = [] 

141 in_single = False 

142 in_double = False 

143 buf = [] 

144 

145 for ch in command: 

146 if ch == "'" and not in_double: 

147 in_single = not in_single 

148 buf.append(ch) 

149 elif ch == '"' and not in_single: 

150 in_double = not in_double 

151 buf.append(ch) 

152 elif ch == '|' and not in_single and not in_double: 

153 segments.append(''.join(buf).strip()) 

154 buf = [] 

155 else: 

156 buf.append(ch) 

157 

158 remaining = ''.join(buf).strip() 

159 if remaining: 

160 segments.append(remaining) 

161 

162 result = [] 

163 for seg in segments: 

164 if not seg: 

165 continue 

166 try: 

167 tokens = shlex.split(seg) 

168 except ValueError: 

169 # Fallback for malformed quotes 

170 tokens = seg.split() 

171 if tokens: 

172 result.append(tokens) 

173 

174 return result 

175 

176 

177def _extract_flags(tokens: list[str]) -> tuple[set[str], list[str]]: 

178 """Extract single-char flags (e.g. -i, -l, -c, -n, -la) from tokens. 

179 

180 Returns (flags_set, remaining_args). 

181 """ 

182 flags = set() 

183 args = [] 

184 for token in tokens: 

185 if token.startswith('-') and len(token) > 1 and not token[1:].isdigit(): 

186 # Could be -ilc (combined) or -20 (number, skip) 

187 for ch in token[1:]: 

188 flags.add(ch) 

189 else: 

190 args.append(token) 

191 return flags, args 

192 

193 

194def _extract_numeric_flag(tokens: list[str]) -> tuple[Optional[int], list[str]]: 

195 """Extract a numeric flag like -20 from tokens. Returns (number, remaining).""" 

196 num = None 

197 remaining = [] 

198 for token in tokens: 

199 if num is None and re.match(r'^-\d+$', token): 

200 num = int(token[1:]) 

201 else: 

202 remaining.append(token) 

203 return num, remaining 

204 

205 

206# ============================================================================= 

207# DIRECTORY TREE & PATH RESOLUTION 

208# ============================================================================= 

209 

210 

211async def _build_directory_tree(knowledge_id: str) -> dict: 

212 """Build an in-memory directory tree for a KB. Returns {dirs, files, path_to_dir_id, dir_id_to_path}.""" 

213 from open_webui.models.knowledge import Knowledges 

214 

215 all_dirs = await Knowledges.get_all_directories(knowledge_id) 

216 files_with_dirs = await Knowledges.get_files_with_directory_ids(knowledge_id) 

217 

218 # Build dir_id -> dir info map 

219 dir_map = {} 

220 for d in all_dirs: 

221 dir_map[d.id] = {'name': d.name, 'parent_id': d.parent_id, 'id': d.id} 

222 

223 # Compute full path for each directory 

224 dir_id_to_path = {} 

225 

226 def _get_dir_path(dir_id): 

227 if dir_id in dir_id_to_path: 

228 return dir_id_to_path[dir_id] 

229 d = dir_map.get(dir_id) 

230 if not d: 

231 return '' 

232 if d['parent_id'] and d['parent_id'] in dir_map: 

233 parent_path = _get_dir_path(d['parent_id']) 

234 path = f'{parent_path}/{d["name"]}' if parent_path else d['name'] 

235 else: 

236 path = d['name'] 

237 dir_id_to_path[dir_id] = path 

238 return path 

239 

240 for d_id in dir_map: 

241 _get_dir_path(d_id) 

242 

243 path_to_dir_id = {v: k for k, v in dir_id_to_path.items()} 

244 

245 # Build file list with paths 

246 files = [] 

247 for file_model, directory_id in files_with_dirs: 

248 if directory_id and directory_id in dir_id_to_path: 

249 file_path = f'{dir_id_to_path[directory_id]}/{file_model.filename}' 

250 else: 

251 file_path = file_model.filename 

252 files.append( 

253 { 

254 'id': file_model.id, 

255 'filename': file_model.filename, 

256 'path': file_path, 

257 'directory_id': directory_id, 

258 'size': file_model.meta.get('size') if file_model.meta else None, 

259 'type': file_model.meta.get('content_type') if file_model.meta else None, 

260 'updated_at': file_model.updated_at, 

261 } 

262 ) 

263 

264 return { 

265 'dirs': dir_map, 

266 'files': files, 

267 'path_to_dir_id': path_to_dir_id, 

268 'dir_id_to_path': dir_id_to_path, 

269 } 

270 

271 

272def _resolve_path(path: str, tree: dict) -> str | None: 

273 """Resolve a directory path string to a dir_id. Returns None if not found.""" 

274 path = path.strip('/') 

275 return tree['path_to_dir_id'].get(path) 

276 

277 

278def _get_files_in_dir(tree: dict, dir_id: str | None) -> list[dict]: 

279 """Get files directly in a directory (None = root).""" 

280 return [f for f in tree['files'] if f['directory_id'] == dir_id] 

281 

282 

283def _get_subdirs(tree: dict, parent_id: str | None) -> list[dict]: 

284 """Get immediate child directories.""" 

285 return sorted([d for d in tree['dirs'].values() if d['parent_id'] == parent_id], key=lambda d: d['name']) 

286 

287 

288def _get_files_under_dir(tree: dict, dir_id: str) -> list[dict]: 

289 """Get all files recursively under a directory.""" 

290 # Collect this dir + all descendant dir IDs 

291 target_ids = {dir_id} 

292 changed = True 

293 while changed: 

294 changed = False 

295 for d in tree['dirs'].values(): 

296 if d['parent_id'] in target_ids and d['id'] not in target_ids: 

297 target_ids.add(d['id']) 

298 changed = True 

299 return [f for f in tree['files'] if f['directory_id'] in target_ids] 

300 

301 

302# ============================================================================= 

303# FILE RESOLUTION & ACCESS CONTROL 

304# ============================================================================= 

305 

306 

307async def _get_accessible_kb_ids( 

308 user: dict, model_knowledge: list[dict] | None, knowledge_id: str | None = None 

309) -> list[tuple[str, str, str]]: 

310 """Get list of (kb_id, kb_name, kb_description) the user can access.""" 

311 from open_webui.models.access_grants import AccessGrants 

312 from open_webui.models.groups import Groups 

313 from open_webui.models.knowledge import Knowledges 

314 

315 user_id = user.get('id') 

316 user_role = user.get('role', 'user') 

317 user_group_ids = [g.id for g in await Groups.get_groups_by_member_id(user_id)] 

318 

319 async def _has_access(kb): 

320 return ( 

321 user_role == 'admin' 

322 or kb.user_id == user_id 

323 or await AccessGrants.has_access( 

324 user_id=user_id, 

325 resource_type='knowledge', 

326 resource_id=kb.id, 

327 permission='read', 

328 user_group_ids=set(user_group_ids), 

329 ) 

330 ) 

331 

332 result = [] 

333 

334 if model_knowledge: 

335 attached_kb_ids = set() 

336 for item in model_knowledge: 

337 if item.get('type') == 'collection': 

338 attached_kb_ids.add(item.get('id')) 

339 if knowledge_id: 

340 if knowledge_id not in attached_kb_ids: 

341 return [] 

342 attached_kb_ids = {knowledge_id} 

343 for kb_id in attached_kb_ids: 

344 kb = await Knowledges.get_knowledge_by_id(kb_id) 

345 if kb and await _has_access(kb): 

346 result.append((kb.id, kb.name, kb.description or '')) 

347 elif knowledge_id: 

348 kb = await Knowledges.get_knowledge_by_id(knowledge_id) 

349 if kb and await _has_access(kb): 

350 result.append((kb.id, kb.name, kb.description or '')) 

351 else: 

352 search = await Knowledges.search_knowledge_bases( 

353 user_id, 

354 filter={'query': '', 'user_id': user_id, 'group_ids': user_group_ids}, 

355 skip=0, 

356 limit=50, 

357 ) 

358 for kb in search.items: 

359 result.append((kb.id, kb.name, kb.description or '')) 

360 

361 return result 

362 

363 

364async def _get_accessible_files( 

365 user: dict, model_knowledge: list[dict] | None, knowledge_id: str | None = None 

366) -> list[dict]: 

367 """Get all files the user can access, with KB metadata and directory_id (no path computation).""" 

368 from open_webui.models.files import Files 

369 from open_webui.models.knowledge import Knowledges 

370 

371 kb_ids = await _get_accessible_kb_ids(user, model_knowledge, knowledge_id) 

372 files = [] 

373 

374 for kb_id, kb_name, _ in kb_ids: 

375 kb_files = await Knowledges.get_files_with_directory_ids(kb_id) 

376 for file_model, dir_id in kb_files: 

377 files.append( 

378 { 

379 'id': file_model.id, 

380 'filename': file_model.filename, 

381 'directory_id': dir_id, 

382 'size': file_model.meta.get('size') if file_model.meta else None, 

383 'type': file_model.meta.get('content_type') if file_model.meta else None, 

384 'updated_at': file_model.updated_at, 

385 'knowledge_id': kb_id, 

386 'knowledge_name': kb_name, 

387 } 

388 ) 

389 

390 # Also handle directly attached files (not in any KB) 

391 if model_knowledge: 

392 attached_file_ids = set() 

393 for item in model_knowledge: 

394 if item.get('type') == 'file': 

395 attached_file_ids.add(item.get('id')) 

396 for fid in attached_file_ids: 

397 f = await Files.get_file_by_id(fid) 

398 if f: 

399 files.append( 

400 { 

401 'id': f.id, 

402 'filename': f.filename, 

403 'directory_id': None, 

404 'size': f.meta.get('size') if f.meta else None, 

405 'type': f.meta.get('content_type') if f.meta else None, 

406 'updated_at': f.updated_at, 

407 'knowledge_id': None, 

408 'knowledge_name': None, 

409 } 

410 ) 

411 

412 return files 

413 

414 

415async def _resolve_dir_path(path: str, knowledge_id: str) -> str | None: 

416 """Walk a directory path one level at a time. Returns dir_id or None.""" 

417 from open_webui.models.knowledge import Knowledges 

418 

419 parts = path.strip('/').split('/') 

420 current_parent = None 

421 

422 for part in parts: 

423 dirs = await Knowledges.get_directories(knowledge_id, parent_id=current_parent) 

424 match = next((d for d in dirs if d.name == part), None) 

425 if not match: 

426 return None 

427 current_parent = match.id 

428 

429 return current_parent 

430 

431 

432async def _get_descendant_dir_ids(dir_id: str, knowledge_id: str) -> set[str]: 

433 """Collect all descendant directory IDs recursively.""" 

434 from open_webui.models.knowledge import Knowledges 

435 

436 result = {dir_id} 

437 queue = [dir_id] 

438 while queue: 

439 parent = queue.pop() 

440 children = await Knowledges.get_directories(knowledge_id, parent_id=parent) 

441 for child in children: 

442 if child.id not in result: 

443 result.add(child.id) 

444 queue.append(child.id) 

445 return result 

446 

447 

448async def _resolve_file(ref: str, user: dict, model_knowledge: list[dict] | None) -> dict | None: 

449 """Resolve a file reference (ID, path, or filename) to a file info dict with content.""" 

450 from open_webui.models.files import Files 

451 

452 # Get accessible file IDs (lightweight — no path computation) 

453 accessible = await _get_accessible_files(user, model_knowledge) 

454 accessible_ids = {fi['id'] for fi in accessible} 

455 

456 # Try direct ID lookup first — but verify access 

457 f = await Files.get_file_by_id(ref) 

458 if f and f.data: 

459 if f.id not in accessible_ids: 

460 return None 

461 return { 

462 'id': f.id, 

463 'filename': f.filename, 

464 'content': f.data.get('content', ''), 

465 'meta': f.meta, 

466 'updated_at': f.updated_at, 

467 'created_at': f.created_at, 

468 } 

469 

470 # Try path match (e.g. "docs/api/auth.md") — lazy dir walk 

471 ref_clean = ref.strip('/') 

472 if '/' in ref_clean: 

473 dir_path, filename = ref_clean.rsplit('/', 1) 

474 # Try resolving in each accessible KB 

475 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')} 

476 for kb_id in kb_ids: 

477 dir_id = await _resolve_dir_path(dir_path, kb_id) 

478 if dir_id is None: 

479 continue 

480 # Find file with that name in that directory 

481 matches = [fi for fi in accessible if fi['filename'] == filename and fi['directory_id'] == dir_id] 

482 if len(matches) == 1: 

483 f = await Files.get_file_by_id(matches[0]['id']) 

484 if f and f.data: 

485 return { 

486 'id': f.id, 

487 'filename': f.filename, 

488 'content': f.data.get('content', ''), 

489 'meta': f.meta, 

490 'updated_at': f.updated_at, 

491 'created_at': f.created_at, 

492 'knowledge_id': matches[0].get('knowledge_id'), 

493 'knowledge_name': matches[0].get('knowledge_name'), 

494 } 

495 

496 # Try filename match within accessible files 

497 matches = [fi for fi in accessible if fi['filename'] == ref] 

498 

499 if len(matches) == 1: 

500 f = await Files.get_file_by_id(matches[0]['id']) 

501 if f and f.data: 

502 return { 

503 'id': f.id, 

504 'filename': f.filename, 

505 'content': f.data.get('content', ''), 

506 'meta': f.meta, 

507 'updated_at': f.updated_at, 

508 'created_at': f.created_at, 

509 'knowledge_id': matches[0].get('knowledge_id'), 

510 'knowledge_name': matches[0].get('knowledge_name'), 

511 } 

512 elif len(matches) > 1: 

513 return { 

514 'error': f'Ambiguous filename "{ref}". Use full path to disambiguate:\n' 

515 + '\n'.join(f' {m["id"]} {m["filename"]} ({m.get("knowledge_name", "direct")})' for m in matches) 

516 } 

517 

518 return None 

519 

520 

521async def _get_file_content(file_id: str) -> str | None: 

522 """Get file content by ID.""" 

523 from open_webui.models.files import Files 

524 

525 f = await Files.get_file_by_id(file_id) 

526 if f and f.data: 

527 return f.data.get('content', '') 

528 return None 

529 

530 

531# ============================================================================= 

532# COMMAND HANDLERS 

533# ============================================================================= 

534 

535 

536async def _kb_ls(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str: 

537 """List files and directories. Supports: ls, ls <path>, ls -a (flat).""" 

538 from open_webui.models.knowledge import Knowledges 

539 

540 flat_mode = 'a' in flags 

541 path_arg = args[0] if args else None 

542 

543 kb_ids = await _get_accessible_kb_ids(user, model_knowledge, knowledge_id=None) 

544 direct_files = ( 

545 [f for f in await _get_accessible_files(user, model_knowledge) if not f.get('knowledge_id')] 

546 if model_knowledge 

547 else [] 

548 ) 

549 

550 # If path_arg looks like a KB ID, scope to that KB 

551 target_kb_id = None 

552 dir_path = None 

553 if path_arg: 

554 for kb_id, kb_name, _ in kb_ids: 

555 if kb_id == path_arg: 

556 target_kb_id = kb_id 

557 break 

558 if not target_kb_id: 

559 dir_path = path_arg.strip('/') 

560 

561 if target_kb_id: 

562 kb_ids = [(kid, kn, kd) for kid, kn, kd in kb_ids if kid == target_kb_id] 

563 

564 if not kb_ids and not direct_files: 

565 return 'No knowledge bases found.' 

566 

567 lines = [] 

568 for kb_id, kb_name, kb_desc in kb_ids: 

569 header = f'Knowledge Base: {kb_name} ({kb_id})' 

570 if kb_desc: 

571 header += f'\n {kb_desc}' 

572 lines.append(header) 

573 

574 if flat_mode: 

575 # Flat mode: build full tree (legitimate use) 

576 tree = await _build_directory_tree(kb_id) 

577 for f in _sort_files(tree['files'], flags): 

578 lines.append(f' {f["id"]} {f["path"]} {_fmt_size(f)} {_fmt_date(f)}') 

579 lines.append('') 

580 continue 

581 

582 # Resolve target directory (lazy walk) 

583 target_dir_id = None 

584 if dir_path: 

585 target_dir_id = await _resolve_dir_path(dir_path, kb_id) 

586 if target_dir_id is None: 

587 lines.append(f' Directory not found: {dir_path}') 

588 lines.append('') 

589 continue 

590 lines.append(f' Path: {dir_path}/') 

591 

592 # Show subdirectories (targeted query — only this level) 

593 subdirs = await Knowledges.get_directories(kb_id, parent_id=target_dir_id) 

594 for d in subdirs: 

595 lines.append(f' 📁 {d.name}/') 

596 

597 # Show files at this level (filter from accessible files) 

598 accessible = await _get_accessible_files(user, model_knowledge, knowledge_id=kb_id) 

599 dir_files = [f for f in accessible if f['directory_id'] == target_dir_id] 

600 for f in _sort_files(dir_files, flags): 

601 lines.append(f' {f["id"]} {f["filename"]} {_fmt_size(f)} {_fmt_date(f)}') 

602 

603 if not subdirs and not dir_files: 

604 lines.append(' (empty)') 

605 lines.append('') 

606 

607 if direct_files and not target_kb_id and not dir_path: 

608 lines.append('Attached Files:') 

609 for f in _sort_files(direct_files, flags): 

610 lines.append(f' {f["id"]} {f["filename"]} {_fmt_size(f)} {_fmt_date(f)}') 

611 lines.append('') 

612 

613 return '\n'.join(lines).rstrip() 

614 

615 

616def _sort_files(files: list[dict], flags: set[str]) -> list[dict]: 

617 """ls-style ordering: by name or path, -t newest first, -S largest first, -r reverses.""" 

618 if 't' in flags: 

619 files = sorted(files, key=lambda f: f.get('updated_at') or 0, reverse=True) 

620 elif 'S' in flags: 

621 files = sorted(files, key=lambda f: f.get('size') or 0, reverse=True) 

622 else: 

623 files = sorted(files, key=lambda f: f.get('path') or f['filename']) 

624 return files[::-1] if 'r' in flags else files 

625 

626 

627def _fmt_size(f: dict) -> str: 

628 return f'{f["size"]:,} bytes' if f.get('size') else '' 

629 

630 

631def _fmt_date(f: dict) -> str: 

632 if f.get('updated_at'): 

633 from datetime import datetime, timezone 

634 

635 dt = datetime.fromtimestamp(f['updated_at'], tz=timezone.utc) 

636 return dt.strftime('%Y-%m-%d') 

637 return '' 

638 

639 

640async def _kb_cat(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str: 

641 """Read file content. Use -n for line numbers.""" 

642 if not args: 

643 return 'Usage: cat [-n] <file_id or filename>' 

644 

645 resolved = await _resolve_file(args[0], user, model_knowledge) 

646 if not resolved: 

647 return f'File not found: {args[0]}' 

648 if 'error' in resolved: 

649 return resolved['error'] 

650 

651 content = resolved['content'] 

652 if 'n' in flags: 

653 lines = content.split('\n') 

654 content = '\n'.join(f'{i}: {line}' for i, line in enumerate(lines, 1)) 

655 

656 return content 

657 

658 

659async def _kb_head( 

660 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None 

661) -> str: 

662 """First N lines of a file or piped input.""" 

663 n, args = _extract_numeric_flag(args) 

664 if n is None: 

665 n = DEFAULT_HEAD_LINES 

666 

667 if piped_input is not None: 

668 lines = piped_input.split('\n') 

669 return '\n'.join(lines[:n]) 

670 

671 if not args: 

672 return 'Usage: head [-N] <file>' 

673 

674 resolved = await _resolve_file(args[0], user, model_knowledge) 

675 if not resolved: 

676 return f'File not found: {args[0]}' 

677 if 'error' in resolved: 

678 return resolved['error'] 

679 

680 lines = resolved['content'].split('\n') 

681 total = len(lines) 

682 result = '\n'.join(lines[:n]) 

683 if total > n: 

684 result += f'\n[showing {n} of {total} lines]' 

685 return result 

686 

687 

688async def _kb_tail( 

689 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None 

690) -> str: 

691 """Last N lines of a file or piped input.""" 

692 n, args = _extract_numeric_flag(args) 

693 if n is None: 

694 n = DEFAULT_TAIL_LINES 

695 

696 if piped_input is not None: 

697 lines = piped_input.split('\n') 

698 return '\n'.join(lines[-n:]) 

699 

700 if not args: 

701 return 'Usage: tail [-N] <file>' 

702 

703 resolved = await _resolve_file(args[0], user, model_knowledge) 

704 if not resolved: 

705 return f'File not found: {args[0]}' 

706 if 'error' in resolved: 

707 return resolved['error'] 

708 

709 lines = resolved['content'].split('\n') 

710 total = len(lines) 

711 result = '\n'.join(lines[-n:]) 

712 if total > n: 

713 result += f'\n[showing last {n} of {total} lines]' 

714 return result 

715 

716 

717def _match_lines(content: str, matches: Callable[[str], bool]) -> list[tuple[int, str]]: 

718 return [(i, line) for i, line in enumerate(content.split('\n'), 1) if matches(line)] 

719 

720 

721async def _kb_grep( 

722 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None 

723) -> str: 

724 """Text search across files or piped input. Supports -E for regex.""" 

725 if not args: 

726 return 'Usage: grep [-E] [-i] [-l] [-c] "pattern" [file] [*.ext]' 

727 

728 pattern = args[0] 

729 file_ref = None 

730 ext_filter = None 

731 dir_scope = None 

732 

733 for arg in args[1:]: 

734 if '*' in arg or arg.startswith('.'): 

735 ext_filter = arg.lstrip('*').lstrip('.') 

736 elif arg.endswith('/'): 

737 dir_scope = arg.strip('/') 

738 else: 

739 file_ref = arg 

740 

741 case_insensitive = 'i' in flags 

742 filenames_only = 'l' in flags 

743 count_only = 'c' in flags 

744 use_regex = 'E' in flags 

745 

746 _matches, err = await asyncio.to_thread(build_matcher, pattern, case_insensitive, use_regex) 

747 if err: 

748 return err 

749 

750 # Grep on piped input 

751 if piped_input is not None: 

752 found = await asyncio.to_thread(_match_lines, piped_input, _matches) 

753 matched = [f'{i}: {line}' for i, line in found] 

754 if count_only: 

755 return str(len(matched)) 

756 if filenames_only: 

757 return '(standard input)' if matched else f'No matches for "{pattern}"' 

758 return '\n'.join(matched) if matched else f'No matches for "{pattern}"' 

759 

760 # Single file grep 

761 if file_ref and not dir_scope: 

762 resolved = await _resolve_file(file_ref, user, model_knowledge) 

763 if not resolved: 

764 # Maybe it's a directory path without trailing / 

765 dir_scope = file_ref 

766 elif 'error' in resolved: 

767 return resolved['error'] 

768 else: 

769 found = await asyncio.to_thread(_match_lines, resolved['content'], _matches) 

770 matched = [f'{i}: {line}' for i, line in found] 

771 

772 if count_only: 

773 return f'{resolved["id"]} {resolved["filename"]}: {len(matched)}' 

774 if filenames_only: 

775 return f'{resolved["id"]} {resolved["filename"]}' if matched else f'No matches for "{pattern}"' 

776 

777 if not matched: 

778 return f'No matches for "{pattern}" in {resolved["filename"]}' 

779 return '\n'.join(matched) 

780 

781 # Cross-file grep (optionally scoped to directory) 

782 accessible = await _get_accessible_files(user, model_knowledge) 

783 

784 if dir_scope: 

785 # Resolve directory and collect all descendant IDs 

786 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')} 

787 target_dir_ids = set() 

788 for kb_id in kb_ids: 

789 dir_id = await _resolve_dir_path(dir_scope, kb_id) 

790 if dir_id: 

791 desc = await _get_descendant_dir_ids(dir_id, kb_id) 

792 target_dir_ids.update(desc) 

793 if not target_dir_ids: 

794 return f'No files found under "{dir_scope}/"' 

795 accessible = [f for f in accessible if f.get('directory_id') in target_dir_ids] 

796 if not accessible: 

797 return f'No files found under "{dir_scope}/"' 

798 

799 if ext_filter: 

800 accessible = [f for f in accessible if f['filename'].endswith(f'.{ext_filter}')] 

801 

802 if len(accessible) > KB_EXEC_MAX_GREP_FILES: 

803 return f'Too many files ({len(accessible)}). Scope your search: grep "{pattern}" docs/ or grep "{pattern}" *.py' 

804 

805 from open_webui.models.files import Files 

806 

807 results = [] 

808 file_match_counts = [] 

809 files_with_matches = [] 

810 total_matches = 0 

811 

812 for file_info in accessible: 

813 f = await Files.get_file_by_id(file_info['id']) 

814 if not f or not f.data: 

815 continue 

816 

817 content = f.data.get('content', '') 

818 if not content: 

819 continue 

820 

821 file_matches = await asyncio.to_thread(_match_lines, content, _matches) 

822 

823 if file_matches: 

824 files_with_matches.append(file_info) 

825 file_match_counts.append((file_info, len(file_matches))) 

826 total_matches += len(file_matches) 

827 

828 if not count_only and not filenames_only: 

829 for line_num, line_text in file_matches: 

830 if len(results) < KNOWLEDGE_GREP_MAX_MATCHES: 

831 results.append(f'{file_info["id"]} {file_info["filename"]}:{line_num}: {line_text.rstrip()}') 

832 

833 if count_only: 

834 if not file_match_counts: 

835 return f'No matches for "{pattern}"' 

836 lines = [f'{fi["id"]} {fi["filename"]}: {cnt}' for fi, cnt in file_match_counts] 

837 lines.append(f'Total: {total_matches} matches in {len(file_match_counts)} files') 

838 return '\n'.join(lines) 

839 

840 if filenames_only: 

841 if not files_with_matches: 

842 return f'No matches for "{pattern}"' 

843 return '\n'.join(f'{fi["id"]} {fi["filename"]}' for fi in files_with_matches) 

844 

845 if not results: 

846 return f'No matches for "{pattern}" across {len(accessible)} files' 

847 

848 output = '\n'.join(results) 

849 if total_matches > KNOWLEDGE_GREP_MAX_MATCHES: 

850 output += f'\n[showing {KNOWLEDGE_GREP_MAX_MATCHES} of {total_matches} matches]' 

851 return output 

852 

853 

854async def _kb_find(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str: 

855 """Find files by name/glob pattern, optionally scoped to a directory.""" 

856 if not args: 

857 return 'Usage: find "*.md" or find docs/ "*.md"' 

858 

859 import fnmatch 

860 

861 # If two args and first looks like a dir scope 

862 dir_scope = None 

863 if len(args) >= 2 and ('/' in args[0] or not ('*' in args[0] or '?' in args[0])): 

864 dir_scope = args[0].strip('/') 

865 pattern = args[1] 

866 else: 

867 pattern = args[0] 

868 

869 accessible = await _get_accessible_files(user, model_knowledge) 

870 

871 if dir_scope: 

872 kb_ids = {fi['knowledge_id'] for fi in accessible if fi.get('knowledge_id')} 

873 target_dir_ids = set() 

874 for kb_id in kb_ids: 

875 dir_id = await _resolve_dir_path(dir_scope, kb_id) 

876 if dir_id: 

877 desc = await _get_descendant_dir_ids(dir_id, kb_id) 

878 target_dir_ids.update(desc) 

879 accessible = [f for f in accessible if f.get('directory_id') in target_dir_ids] 

880 

881 matched = [f for f in accessible if fnmatch.fnmatch(f['filename'], pattern)] 

882 

883 if not matched: 

884 scope_str = f' under "{dir_scope}/"' if dir_scope else '' 

885 return f'No files matching "{pattern}"{scope_str}' 

886 

887 lines = [] 

888 for f in _sort_files(matched, flags): 

889 kb_info = f' ({f["knowledge_name"]})' if f.get('knowledge_name') else '' 

890 lines.append(f'{f["id"]} {f["filename"]}{kb_info}') 

891 return '\n'.join(lines) 

892 

893 

894async def _kb_wc( 

895 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None 

896) -> str: 

897 """Word, line, character counts.""" 

898 if piped_input is not None: 

899 lines = piped_input.count('\n') + (1 if piped_input and not piped_input.endswith('\n') else 0) 

900 words = len(piped_input.split()) 

901 chars = len(piped_input) 

902 if 'l' in flags: 

903 return str(lines) 

904 return f' {lines} {words} {chars}' 

905 

906 if not args: 

907 return 'Usage: wc [-l] <file>' 

908 

909 resolved = await _resolve_file(args[0], user, model_knowledge) 

910 if not resolved: 

911 return f'File not found: {args[0]}' 

912 if 'error' in resolved: 

913 return resolved['error'] 

914 

915 content = resolved['content'] 

916 lines = content.count('\n') + (1 if content and not content.endswith('\n') else 0) 

917 words = len(content.split()) 

918 chars = len(content) 

919 

920 if 'l' in flags: 

921 return f' {lines} {resolved["filename"]}' 

922 return f' {lines} {words} {chars} {resolved["filename"]}' 

923 

924 

925async def _kb_stat(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str: 

926 """File metadata.""" 

927 if not args: 

928 return 'Usage: stat <file>' 

929 

930 resolved = await _resolve_file(args[0], user, model_knowledge) 

931 if not resolved: 

932 return f'File not found: {args[0]}' 

933 if 'error' in resolved: 

934 return resolved['error'] 

935 

936 content = resolved['content'] 

937 lines = content.count('\n') + (1 if content and not content.endswith('\n') else 0) 

938 words = len(content.split()) 

939 chars = len(content) 

940 

941 meta = resolved.get('meta') or {} 

942 size = meta.get('size', chars) 

943 content_type = meta.get('content_type', 'unknown') 

944 

945 out = [ 

946 f' File: {resolved["filename"]}', 

947 f' ID: {resolved["id"]}', 

948 f' Size: {size:,} bytes', 

949 f' Type: {content_type}', 

950 f' Lines: {lines:,}', 

951 f' Words: {words:,}', 

952 f' Chars: {chars:,}', 

953 ] 

954 

955 if resolved.get('created_at'): 

956 from datetime import datetime, timezone 

957 

958 dt = datetime.fromtimestamp(resolved['created_at'], tz=timezone.utc) 

959 out.append(f' Created: {dt.strftime("%Y-%m-%d %H:%M:%S UTC")}') 

960 if resolved.get('updated_at'): 

961 from datetime import datetime, timezone 

962 

963 dt = datetime.fromtimestamp(resolved['updated_at'], tz=timezone.utc) 

964 out.append(f' Updated: {dt.strftime("%Y-%m-%d %H:%M:%S UTC")}') 

965 if resolved.get('knowledge_name'): 

966 out.append(f' KB: {resolved["knowledge_name"]} ({resolved.get("knowledge_id", "")})') 

967 

968 return '\n'.join(out) 

969 

970 

971async def _kb_sed( 

972 args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None, piped_input: str | None = None 

973) -> str: 

974 """Extract line range from a file. Usage: sed -n 'M,Np' <file>""" 

975 if piped_input is not None: 

976 # sed on piped input: parse range from args 

977 start, end = 1, None 

978 if 'n' in flags and args: 

979 m = re.match(r'^(\d+),(\d+)p?$', args[0]) 

980 if m: 

981 start, end = int(m.group(1)), int(m.group(2)) 

982 args = args[1:] 

983 lines = piped_input.split('\n') 

984 selected = lines[max(0, start - 1) : (end or len(lines))] 

985 return '\n'.join(selected) 

986 

987 # Parse: sed -n '40,60p' <file> 

988 range_str = None 

989 file_ref = None 

990 

991 for arg in args: 

992 m = re.match(r"^'?(\d+),(\d+)p?'?$", arg) 

993 if m: 

994 range_str = arg 

995 else: 

996 file_ref = arg 

997 

998 if not range_str or not file_ref: 

999 return "Usage: sed -n '40,60p' <file>" 

1000 

1001 m = re.match(r"^'?(\d+),(\d+)p?'?$", range_str) 

1002 start, end = int(m.group(1)), int(m.group(2)) 

1003 

1004 if start > end: 

1005 return f'Invalid range: start ({start}) > end ({end})' 

1006 

1007 resolved = await _resolve_file(file_ref, user, model_knowledge) 

1008 if not resolved: 

1009 return f'File not found: {file_ref}' 

1010 if 'error' in resolved: 

1011 return resolved['error'] 

1012 

1013 lines = resolved['content'].split('\n') 

1014 total = len(lines) 

1015 selected = lines[max(0, start - 1) : end] 

1016 result = '\n'.join(selected) 

1017 result += f'\n[lines {start}-{min(end, total)} of {total}]' 

1018 return result 

1019 

1020 

1021# ============================================================================= 

1022# PIPE EXECUTOR 

1023# ============================================================================= 

1024 

1025 

1026async def _kb_tree(args: list[str], flags: set[str], user: dict, model_knowledge: list[dict] | None) -> str: 

1027 """Show directory tree structure.""" 

1028 kb_ids = await _get_accessible_kb_ids(user, model_knowledge) 

1029 direct_files = ( 

1030 [f for f in await _get_accessible_files(user, model_knowledge) if not f.get('knowledge_id')] 

1031 if model_knowledge 

1032 else [] 

1033 ) 

1034 if not kb_ids and not direct_files: 

1035 return 'No knowledge bases found.' 

1036 

1037 dir_scope = args[0].strip('/') if args else None 

1038 output = [] 

1039 

1040 for kb_id, kb_name, kb_desc in kb_ids: 

1041 tree = await _build_directory_tree(kb_id) 

1042 header = f'Knowledge Base: {kb_name} ({kb_id})' 

1043 if kb_desc: 

1044 header += f'\n {kb_desc}' 

1045 output.append(header) 

1046 

1047 # Find root to start from 

1048 root_dir_id = None 

1049 if dir_scope: 

1050 root_dir_id = _resolve_path(dir_scope, tree) 

1051 if root_dir_id is None: 

1052 output.append(f' Directory not found: {dir_scope}') 

1053 output.append('') 

1054 continue 

1055 output.append(f' {dir_scope}/') 

1056 

1057 def _render_tree(parent_id, prefix=' '): 

1058 items = [] 

1059 subdirs = _get_subdirs(tree, parent_id) 

1060 files = _sort_files(_get_files_in_dir(tree, parent_id), flags) 

1061 entries = [('dir', d) for d in subdirs] + [('file', f) for f in files] 

1062 

1063 for idx, (etype, entry) in enumerate(entries): 

1064 is_last = idx == len(entries) - 1 

1065 connector = '└── ' if is_last else '├── ' 

1066 child_prefix = prefix + (' ' if is_last else '│ ') 

1067 

1068 if etype == 'dir': 

1069 items.append(f'{prefix}{connector}📁 {entry["name"]}/') 

1070 items.extend(_render_tree(entry['id'], child_prefix)) 

1071 else: 

1072 items.append(f'{prefix}{connector}{entry["filename"]}') 

1073 return items 

1074 

1075 output.extend(_render_tree(root_dir_id)) 

1076 

1077 # Summary 

1078 total_dirs = len(tree['dirs']) 

1079 total_files = len(tree['files']) 

1080 output.append(f'\n {total_dirs} directories, {total_files} files') 

1081 output.append('') 

1082 

1083 if direct_files and not dir_scope: 

1084 output.append('Attached Files:') 

1085 for idx, f in enumerate(_sort_files(direct_files, flags)): 

1086 connector = '└── ' if idx == len(direct_files) - 1 else '├── ' 

1087 output.append(f' {connector}{f["filename"]}') 

1088 output.append(f'\n 0 directories, {len(direct_files)} files') 

1089 output.append('') 

1090 

1091 return '\n'.join(output).rstrip() 

1092 

1093 

1094COMMAND_MAP = { 

1095 'ls': _kb_ls, 

1096 'cat': _kb_cat, 

1097 'head': _kb_head, 

1098 'tail': _kb_tail, 

1099 'grep': _kb_grep, 

1100 'find': _kb_find, 

1101 'wc': _kb_wc, 

1102 'stat': _kb_stat, 

1103 'sed': _kb_sed, 

1104 'tree': _kb_tree, 

1105} 

1106 

1107 

1108async def _execute_pipeline( 

1109 segments: list[list[str]], 

1110 user: dict, 

1111 model_knowledge: list[dict] | None, 

1112) -> str: 

1113 """Execute a pipeline of commands, passing text between them.""" 

1114 piped_input = None 

1115 

1116 for tokens in segments: 

1117 cmd_name = tokens[0].lower() 

1118 rest = tokens[1:] 

1119 

1120 handler = COMMAND_MAP.get(cmd_name) 

1121 if not handler: 

1122 return f'Unknown command: {cmd_name}. Available: {", ".join(sorted(COMMAND_MAP.keys()))}' 

1123 

1124 flags, args = _extract_flags(rest) 

1125 

1126 # Commands that accept piped input 

1127 if piped_input is not None and cmd_name in ('head', 'tail', 'grep', 'wc', 'sed'): 

1128 piped_input = await handler(args, flags, user, model_knowledge, piped_input=piped_input) 

1129 else: 

1130 piped_input = await handler(args, flags, user, model_knowledge) 

1131 

1132 return piped_input or '' 

1133 

1134 

1135# ============================================================================= 

1136# ENTRY POINT 

1137# ============================================================================= 

1138 

1139 

1140async def kb_exec( 

1141 command: str, 

1142 __request__: Request = None, 

1143 __user__: dict = None, 

1144 __model_knowledge__: Optional[list[dict]] = None, 

1145) -> str: 

1146 """ 

1147 Run a filesystem command against the knowledge base. 

1148 

1149 Commands: 

1150 ls — list root files and directories 

1151 ls docs/ — list contents of a directory 

1152 ls -a — flat list of all files with full paths 

1153 ls -t — newest modified first 

1154 ls -S — largest first 

1155 ls -r — reverse file order 

1156 tree — recursive directory tree view 

1157 tree docs/ — subtree from a directory 

1158 cat -n <file> — read file with line numbers 

1159 head -20 <file> — first 20 lines 

1160 tail -10 <file> — last 10 lines 

1161 sed -n '40,60p' <file> — view lines 40-60 

1162 grep "text" <file> — exact text search (auto-detects regex) 

1163 grep -i "text" — case-insensitive 

1164 grep -l "text" — filenames-only 

1165 grep -c "text" — match counts 

1166 grep "text" docs/ — search within a directory 

1167 grep "text" *.py — filter by extension 

1168 find "*.md" — find files by glob 

1169 find docs/ "*.md" — find within a directory 

1170 find -t "*.md", tree -t — same sort flags as ls 

1171 wc <file> — line/word/char counts 

1172 stat <file> — file metadata 

1173 

1174 Pipes: grep "auth" | head -5 

1175 Files: reference by path (docs/api/auth.md), filename, or file ID 

1176 Regex: RE2 syntax; no lookarounds/backreferences. Shorthand character classes are ASCII-only. 

1177 

1178 :param command: A filesystem command string 

1179 :return: Command output as text 

1180 """ 

1181 if not __user__: 

1182 return 'Error: User context not available' 

1183 

1184 if not command or not command.strip(): 

1185 return 'Usage: kb_exec("<command>"). Run kb_exec("ls") to start.' 

1186 

1187 try: 

1188 segments = _parse_pipeline(command.strip()) 

1189 if not segments: 

1190 return 'Could not parse command. Run kb_exec("ls") to start.' 

1191 

1192 # One budget for the whole command: a per-search budget would multiply by segment count. 

1193 with match_budget(): 

1194 output = await _execute_pipeline(segments, __user__, __model_knowledge__) 

1195 if len(output) > KB_EXEC_MAX_OUTPUT_CHARS: 

1196 output = output[:KB_EXEC_MAX_OUTPUT_CHARS] + ( 

1197 f'\n[output truncated at {KB_EXEC_MAX_OUTPUT_CHARS:,} chars' 

1198 ' — narrow the command with a path, glob, head/tail/sed or grep]' 

1199 ) 

1200 return output 

1201 except Exception as e: 

1202 log.exception(f'kb_exec error: {e}') 

1203 return f'Error: {e}'