"""Read-only, bounded access to the explicitly configured project workspace.""" from __future__ import annotations import os import re from pathlib import Path ALLOWED_SUFFIXES = { ".py", ".dart", ".md", ".txt", ".json", ".yaml", ".yml", ".toml", ".html", ".css", ".js", ".ts", ".tsx", ".jsx", ".sh", ".ps1", } IGNORED_PARTS = { ".git", ".venv", ".dart_tool", ".video-lab", "build", "node_modules", "__pycache__", ".idea", ".vscode", } MAX_FILE_BYTES = 256 * 1024 MAX_SCAN_FILES = 500 def configured_root() -> Path | None: value = os.getenv("SOVEREIGNAI_WORKSPACE") if not value: return None root = Path(value).expanduser().resolve() return root if root.is_dir() else None def relative_file(root: Path, relative_path: str) -> Path: candidate = (root / relative_path).resolve(strict=True) try: candidate.relative_to(root) except ValueError as exc: raise ValueError("المسار المطلوب خارج مساحة العمل.") from exc if not candidate.is_file() or candidate.suffix.lower() not in ALLOWED_SUFFIXES: raise ValueError("هذا النوع من الملفات غير مسموح بقراءته.") if any( part in IGNORED_PARTS or part.startswith(".") for part in candidate.relative_to(root).parts ): raise ValueError("قراءة الملفات المخفية أو المستثناة غير مسموحة.") if candidate.stat().st_size > MAX_FILE_BYTES: raise ValueError("الملف أكبر من الحد المسموح للقراءة (256 كيلوبايت).") return candidate def list_text_files(root: Path) -> list[Path]: files: list[Path] = [] for current, directories, filenames in os.walk(root, followlinks=False): directories[:] = [ name for name in directories if name not in IGNORED_PARTS and not name.startswith(".") ] for filename in filenames: path = Path(current) / filename if path.suffix.lower() not in ALLOWED_SUFFIXES: continue try: relative_file(root, path.relative_to(root).as_posix()) except (OSError, ValueError): continue files.append(path) if len(files) >= MAX_SCAN_FILES: return files return files def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]: terms = { term.casefold() for term in re.findall(r"[\w\u0600-\u06ff]{3,}", task) if term.casefold() not in { "the", "and", "for", "with", "this", "that", "من", "على", "في", "عن", "كيف", "شو", "ما", "ماذا", "هذا", "هذه", "التي", "الذي", "اشرح", "دور", "ملف", "ملفات", "اذكر", "المستخدمة", "المستخدم", "المشروع", "التفسير", } } mentioned_paths = { match.replace("\\", "/").casefold() for match in re.findall( r"(?:[\w.-]+[\\/])+[\w.-]+\.(?:py|dart|md|txt|json|ya?ml|toml|html|css|js|ts|tsx|jsx|sh|ps1)", task, flags=re.IGNORECASE, ) } ranked: list[tuple[int, str, str]] = [] for path in list_text_files(root): try: text = path.read_text(encoding="utf-8", errors="replace") except OSError: continue if not text.strip(): continue lowered = text.casefold() words = re.findall(r"[\w\u0600-\u06ff]+", lowered) score = sum(words.count(term) for term in terms) relative = path.relative_to(root).as_posix() if relative.casefold() in mentioned_paths: score += 100_000 if score: ranked.append((score, relative, text[:1800])) ranked.sort(key=lambda item: (-item[0], item[1])) return [(relative, content) for _, relative, content in ranked[:limit]]