104 lines
3.9 KiB
Python
104 lines
3.9 KiB
Python
"""Read-only, bounded access to the explicitly configured project workspace."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import re
|
|
from pathlib import Path
|
|
|
|
ALLOWED_SUFFIXES = {
|
|
".py", ".dart", ".md", ".txt", ".json", ".yaml", ".yml", ".toml",
|
|
".html", ".css", ".js", ".ts", ".tsx", ".jsx", ".sh", ".ps1",
|
|
}
|
|
IGNORED_PARTS = {
|
|
".git", ".venv", ".dart_tool", ".video-lab", "build", "node_modules",
|
|
"__pycache__", ".idea", ".vscode",
|
|
}
|
|
MAX_FILE_BYTES = 256 * 1024
|
|
MAX_SCAN_FILES = 500
|
|
|
|
|
|
def configured_root() -> Path | None:
|
|
value = os.getenv("SOVEREIGNAI_WORKSPACE")
|
|
if not value:
|
|
return None
|
|
root = Path(value).expanduser().resolve()
|
|
return root if root.is_dir() else None
|
|
|
|
|
|
def relative_file(root: Path, relative_path: str) -> Path:
|
|
candidate = (root / relative_path).resolve(strict=True)
|
|
try:
|
|
candidate.relative_to(root)
|
|
except ValueError as exc:
|
|
raise ValueError("المسار المطلوب خارج مساحة العمل.") from exc
|
|
if not candidate.is_file() or candidate.suffix.lower() not in ALLOWED_SUFFIXES:
|
|
raise ValueError("هذا النوع من الملفات غير مسموح بقراءته.")
|
|
if any(
|
|
part in IGNORED_PARTS or part.startswith(".")
|
|
for part in candidate.relative_to(root).parts
|
|
):
|
|
raise ValueError("قراءة الملفات المخفية أو المستثناة غير مسموحة.")
|
|
if candidate.stat().st_size > MAX_FILE_BYTES:
|
|
raise ValueError("الملف أكبر من الحد المسموح للقراءة (256 كيلوبايت).")
|
|
return candidate
|
|
|
|
|
|
def list_text_files(root: Path) -> list[Path]:
|
|
files: list[Path] = []
|
|
for current, directories, filenames in os.walk(root, followlinks=False):
|
|
directories[:] = [
|
|
name for name in directories
|
|
if name not in IGNORED_PARTS and not name.startswith(".")
|
|
]
|
|
for filename in filenames:
|
|
path = Path(current) / filename
|
|
if path.suffix.lower() not in ALLOWED_SUFFIXES:
|
|
continue
|
|
try:
|
|
relative_file(root, path.relative_to(root).as_posix())
|
|
except (OSError, ValueError):
|
|
continue
|
|
files.append(path)
|
|
if len(files) >= MAX_SCAN_FILES:
|
|
return files
|
|
return files
|
|
|
|
|
|
def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
|
|
terms = {
|
|
term.casefold()
|
|
for term in re.findall(r"[\w\u0600-\u06ff]{3,}", task)
|
|
if term.casefold() not in {
|
|
"the", "and", "for", "with", "this", "that", "من", "على", "في", "عن",
|
|
"كيف", "شو", "ما", "ماذا", "هذا", "هذه", "التي", "الذي", "اشرح", "دور",
|
|
"ملف", "ملفات", "اذكر", "المستخدمة", "المستخدم", "المشروع", "التفسير",
|
|
}
|
|
}
|
|
mentioned_paths = {
|
|
match.replace("\\", "/").casefold()
|
|
for match in re.findall(
|
|
r"(?:[\w.-]+[\\/])+[\w.-]+\.(?:py|dart|md|txt|json|ya?ml|toml|html|css|js|ts|tsx|jsx|sh|ps1)",
|
|
task,
|
|
flags=re.IGNORECASE,
|
|
)
|
|
}
|
|
ranked: list[tuple[int, str, str]] = []
|
|
for path in list_text_files(root):
|
|
try:
|
|
text = path.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
if not text.strip():
|
|
continue
|
|
lowered = text.casefold()
|
|
words = re.findall(r"[\w\u0600-\u06ff]+", lowered)
|
|
score = sum(words.count(term) for term in terms)
|
|
relative = path.relative_to(root).as_posix()
|
|
if relative.casefold() in mentioned_paths:
|
|
score += 100_000
|
|
if score:
|
|
ranked.append((score, relative, text[:1800]))
|
|
ranked.sort(key=lambda item: (-item[0], item[1]))
|
|
return [(relative, content) for _, relative, content in ranked[:limit]]
|