Files
sovereign_ai/SovereignAI-Starter/app/workspace.py
T

104 lines
3.9 KiB
Python

"""Read-only, bounded access to the explicitly configured project workspace."""
from __future__ import annotations
import os
import re
from pathlib import Path
ALLOWED_SUFFIXES = {
".py", ".dart", ".md", ".txt", ".json", ".yaml", ".yml", ".toml",
".html", ".css", ".js", ".ts", ".tsx", ".jsx", ".sh", ".ps1",
}
IGNORED_PARTS = {
".git", ".venv", ".dart_tool", ".video-lab", "build", "node_modules",
"__pycache__", ".idea", ".vscode",
}
MAX_FILE_BYTES = 256 * 1024
MAX_SCAN_FILES = 500
def configured_root() -> Path | None:
value = os.getenv("SOVEREIGNAI_WORKSPACE")
if not value:
return None
root = Path(value).expanduser().resolve()
return root if root.is_dir() else None
def relative_file(root: Path, relative_path: str) -> Path:
candidate = (root / relative_path).resolve(strict=True)
try:
candidate.relative_to(root)
except ValueError as exc:
raise ValueError("المسار المطلوب خارج مساحة العمل.") from exc
if not candidate.is_file() or candidate.suffix.lower() not in ALLOWED_SUFFIXES:
raise ValueError("هذا النوع من الملفات غير مسموح بقراءته.")
if any(
part in IGNORED_PARTS or part.startswith(".")
for part in candidate.relative_to(root).parts
):
raise ValueError("قراءة الملفات المخفية أو المستثناة غير مسموحة.")
if candidate.stat().st_size > MAX_FILE_BYTES:
raise ValueError("الملف أكبر من الحد المسموح للقراءة (256 كيلوبايت).")
return candidate
def list_text_files(root: Path) -> list[Path]:
files: list[Path] = []
for current, directories, filenames in os.walk(root, followlinks=False):
directories[:] = [
name for name in directories
if name not in IGNORED_PARTS and not name.startswith(".")
]
for filename in filenames:
path = Path(current) / filename
if path.suffix.lower() not in ALLOWED_SUFFIXES:
continue
try:
relative_file(root, path.relative_to(root).as_posix())
except (OSError, ValueError):
continue
files.append(path)
if len(files) >= MAX_SCAN_FILES:
return files
return files
def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
terms = {
term.casefold()
for term in re.findall(r"[\w\u0600-\u06ff]{3,}", task)
if term.casefold() not in {
"the", "and", "for", "with", "this", "that", "من", "على", "في", "عن",
"كيف", "شو", "ما", "ماذا", "هذا", "هذه", "التي", "الذي", "اشرح", "دور",
"ملف", "ملفات", "اذكر", "المستخدمة", "المستخدم", "المشروع", "التفسير",
}
}
mentioned_paths = {
match.replace("\\", "/").casefold()
for match in re.findall(
r"(?:[\w.-]+[\\/])+[\w.-]+\.(?:py|dart|md|txt|json|ya?ml|toml|html|css|js|ts|tsx|jsx|sh|ps1)",
task,
flags=re.IGNORECASE,
)
}
ranked: list[tuple[int, str, str]] = []
for path in list_text_files(root):
try:
text = path.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
if not text.strip():
continue
lowered = text.casefold()
words = re.findall(r"[\w\u0600-\u06ff]+", lowered)
score = sum(words.count(term) for term in terms)
relative = path.relative_to(root).as_posix()
if relative.casefold() in mentioned_paths:
score += 100_000
if score:
ranked.append((score, relative, text[:1800]))
ranked.sort(key=lambda item: (-item[0], item[1]))
return [(relative, content) for _, relative, content in ranked[:limit]]