Files
sovereign_ai/SovereignAI-Starter/app/workspace.py
T

357 lines
14 KiB
Python

"""Read-only, bounded access to the explicitly configured project workspace."""
from __future__ import annotations
import os
import re
import difflib
import hashlib
import tempfile
import time
from pathlib import Path
from pathlib import PurePosixPath
from uuid import uuid4
ALLOWED_SUFFIXES = {
".py", ".dart", ".md", ".txt", ".json", ".yaml", ".yml", ".toml",
".html", ".css", ".js", ".ts", ".tsx", ".jsx", ".sh", ".ps1",
}
IGNORED_PARTS = {
".git", ".venv", ".dart_tool", ".video-lab", "build", "node_modules",
"__pycache__", ".idea", ".vscode",
}
MAX_FILE_BYTES = 256 * 1024
MAX_SCAN_FILES = 500
PROPOSAL_TTL_SECONDS = 10 * 60
MAX_PENDING_PROPOSALS = 32
_pending_changes: dict[str, dict[str, object]] = {}
def configured_root() -> Path | None:
value = os.getenv("SOVEREIGNAI_WORKSPACE")
if not value:
return None
root = Path(value).expanduser().resolve()
return root if root.is_dir() else None
def selected_root(value: str | None) -> Path | None:
"""Resolve an explicitly selected directory, falling back to server config."""
if value is None or not value.strip():
return configured_root()
try:
root = Path(value).expanduser().resolve(strict=True)
except (OSError, RuntimeError) as exc:
raise ValueError("مجلد مساحة العمل المحدد غير موجود أو غير متاح.") from exc
if not root.is_dir():
raise ValueError("يجب اختيار مجلد صالح لمساحة العمل.")
if root == Path(root.anchor):
raise ValueError("اختر مجلد مشروع محددًا، وليس جذر القرص.")
if root.name.startswith(".") or root.name in IGNORED_PARTS:
raise ValueError("لا يمكن استخدام مجلد مخفي أو مستثنى كمساحة عمل.")
return root
def relative_file(root: Path, relative_path: str) -> Path:
candidate = (root / relative_path).resolve(strict=True)
try:
candidate.relative_to(root)
except ValueError as exc:
raise ValueError("المسار المطلوب خارج مساحة العمل.") from exc
if not candidate.is_file() or candidate.suffix.lower() not in ALLOWED_SUFFIXES:
raise ValueError("هذا النوع من الملفات غير مسموح بقراءته.")
if any(
part in IGNORED_PARTS or part.startswith(".")
for part in candidate.relative_to(root).parts
):
raise ValueError("قراءة الملفات المخفية أو المستثناة غير مسموحة.")
if candidate.stat().st_size > MAX_FILE_BYTES:
raise ValueError("الملف أكبر من الحد المسموح للقراءة (256 كيلوبايت).")
return candidate
def relative_knowledge_file(root: Path, relative_path: str) -> Path:
"""Resolve a bounded UTF-8 document or PDF selected for local knowledge use."""
candidate = (root / relative_path).resolve(strict=True)
try:
candidate.relative_to(root)
except ValueError as exc:
raise ValueError("المسار المطلوب خارج مساحة العمل.") from exc
if not candidate.is_file() or candidate.suffix.lower() not in (ALLOWED_SUFFIXES | {".pdf"}):
raise ValueError("هذا النوع من الملفات غير مسموح بفهرسته.")
if any(
part in IGNORED_PARTS or part.startswith(".")
for part in candidate.relative_to(root).parts
):
raise ValueError("قراءة الملفات المخفية أو المستثناة غير مسموحة.")
if candidate.stat().st_size > MAX_FILE_BYTES:
raise ValueError("الملف أكبر من الحد المسموح للقراءة (256 كيلوبايت).")
return candidate
def _write_target(root: Path, relative_path: str, operation: str) -> tuple[Path, bytes]:
if (
len(relative_path) > 1024
or "\x00" in relative_path
or any(char in relative_path for char in '<>:"|?*')
):
raise ValueError("مسار الملف غير صالح.")
normalized = relative_path.replace("\\", "/")
relative = PurePosixPath(normalized)
if (
relative.is_absolute()
or not relative.parts
or any(part in {"", ".", ".."} for part in relative.parts)
or any(part.startswith(".") or part in IGNORED_PARTS for part in relative.parts)
):
raise ValueError("يسمح بالكتابة داخل مسارات نسبية غير مخفية في مساحة العمل فقط.")
reserved_names = {"CON", "PRN", "AUX", "NUL"} | {
f"{prefix}{number}"
for prefix in ("COM", "LPT")
for number in range(1, 10)
}
if any(
part.endswith((".", " ")) or part.split(".", 1)[0].upper() in reserved_names
for part in relative.parts
):
raise ValueError("اسم الملف غير صالح على Windows.")
if relative.suffix.lower() not in ALLOWED_SUFFIXES:
raise ValueError("امتداد الملف غير مسموح للوكيل.")
root = root.resolve(strict=True)
target = root.joinpath(*relative.parts)
current = root
for part in relative.parts[:-1]:
current = current / part
if current.is_symlink():
raise ValueError("لا يسمح بالكتابة عبر مجلدات الروابط الرمزية.")
if not target.parent.is_dir():
raise ValueError("يجب أن يكون المجلد الأب موجودًا؛ لا ينشئ الوكيل مجلدات تلقائيًا.")
resolved_parent = target.parent.resolve(strict=True)
try:
resolved_parent.relative_to(root)
except ValueError as exc:
raise ValueError("مجلد الملف خارج مساحة العمل المحددة.") from exc
if target.is_symlink():
raise ValueError("لا يسمح باستبدال ملف رابط رمزي.")
exists = target.exists()
if operation == "create" and exists:
raise ValueError("الملف موجود بالفعل؛ اطلب تحديثه بدل إنشائه.")
if operation == "update" and not exists:
raise ValueError("الملف المراد تحديثه غير موجود.")
if operation not in {"create", "update"}:
raise ValueError("نوع التغيير غير مسموح.")
if not exists:
return target, b""
safe_target = relative_file(root, relative.as_posix())
raw = safe_target.read_bytes()
return safe_target, raw
def create_change_preview(
root: Path,
relative_path: str,
operation: str,
content: str,
*,
user_id: str | None = None,
) -> dict[str, object]:
"""Build and retain a short-lived diff; this function never writes the file."""
raw_content = content.encode("utf-8")
if len(raw_content) > MAX_FILE_BYTES:
raise ValueError("المحتوى المقترح يتجاوز حد 256 كيلوبايت.")
target, original = _write_target(root, relative_path, operation)
try:
old_text = original.decode("utf-8")
except UnicodeDecodeError as exc:
raise ValueError("لا يمكن تحديث ملف غير محفوظ بترميز UTF-8.") from exc
newline = "\r\n" if b"\r\n" in original else "\n"
proposed_text = content.replace("\r\n", "\n").replace("\n", newline)
proposed_bytes = proposed_text.encode("utf-8")
if operation == "update" and original == proposed_bytes:
raise ValueError("المحتوى المقترح مطابق للملف الحالي ولا يحتاج إلى تعديل.")
relative = relative_path.replace("\\", "/")
before = old_text.splitlines(keepends=True)
after = proposed_text.splitlines(keepends=True)
diff = "".join(
difflib.unified_diff(
before,
after,
fromfile=f"a/{relative}" if operation == "update" else "/dev/null",
tofile=f"b/{relative}",
lineterm="",
)
)
expired = [
key for key, item in _pending_changes.items()
if float(item["expires_at"]) <= time.time()
]
for key in expired:
_pending_changes.pop(key, None)
if len(_pending_changes) >= MAX_PENDING_PROPOSALS:
raise ValueError("هناك عدد كبير من معاينات التغيير المعلقة؛ ألغِ بعضها قبل إنشاء معاينة أخرى.")
token = str(uuid4())
_pending_changes[token] = {
"root": str(root.resolve(strict=True)),
"path": relative,
"operation": operation,
"content": proposed_bytes,
"expected_hash": hashlib.sha256(original).hexdigest(),
"expires_at": time.time() + PROPOSAL_TTL_SECONDS,
"user_id": user_id,
}
return {
"token": token,
"path": relative,
"operation": operation,
"diff": diff,
"expires_in_seconds": PROPOSAL_TTL_SECONDS,
}
def apply_change_preview(
token: str, *, user_id: str | None = None
) -> dict[str, str]:
"""Apply a reviewed proposal once, only if its target is still unchanged."""
proposal = _pending_changes.get(token)
if proposal is None or float(proposal["expires_at"]) <= time.time():
_pending_changes.pop(token, None)
raise ValueError("انتهت صلاحية معاينة التغيير أو استُخدمت مسبقًا؛ أنشئ معاينة جديدة.")
if proposal.get("user_id") != user_id:
raise ValueError("معاينة التعديل لا تخص جلسة المستخدم الحالية.")
_pending_changes.pop(token, None)
root = Path(str(proposal["root"])).resolve(strict=True)
relative_path = str(proposal["path"])
operation = str(proposal["operation"])
target, current = _write_target(root, relative_path, operation)
current_hash = hashlib.sha256(current).hexdigest()
if current_hash != proposal["expected_hash"]:
raise ValueError("تغير الملف منذ عرض المعاينة؛ أنشئ diff جديدًا قبل التطبيق.")
content = proposal["content"]
if not isinstance(content, bytes):
raise ValueError("بيانات المعاينة غير صالحة.")
temporary_path: str | None = None
try:
with tempfile.NamedTemporaryFile(
mode="wb",
prefix=".sovereignai-review-",
suffix=".tmp",
dir=target.parent,
delete=False,
) as temporary_file:
temporary_path = temporary_file.name
temporary_file.write(content)
temporary_file.flush()
os.fsync(temporary_file.fileno())
if operation == "update":
os.chmod(temporary_path, target.stat().st_mode)
# Recheck to avoid clobbering edits made while the temporary file was written.
_, latest = _write_target(root, relative_path, operation)
if hashlib.sha256(latest).hexdigest() != proposal["expected_hash"]:
raise ValueError("تغير الملف أثناء التطبيق؛ لم تُحفظ المعاينة.")
os.replace(temporary_path, target)
temporary_path = None
finally:
if temporary_path is not None:
try:
os.unlink(temporary_path)
except OSError:
pass
return {"path": relative_path, "operation": operation, "status": "applied"}
def list_text_files(root: Path) -> list[Path]:
files: list[Path] = []
for current, directories, filenames in os.walk(root, followlinks=False):
directories[:] = [
name for name in directories
if name not in IGNORED_PARTS and not name.startswith(".")
]
for filename in filenames:
path = Path(current) / filename
if path.suffix.lower() not in ALLOWED_SUFFIXES:
continue
try:
relative_file(root, path.relative_to(root).as_posix())
except (OSError, ValueError):
continue
files.append(path)
if len(files) >= MAX_SCAN_FILES:
return files
return files
def list_knowledge_files(root: Path) -> list[Path]:
files: list[Path] = []
for current, directories, filenames in os.walk(root, followlinks=False):
directories[:] = [
name for name in directories
if name not in IGNORED_PARTS and not name.startswith(".")
]
for filename in filenames:
path = Path(current) / filename
if path.suffix.lower() not in (ALLOWED_SUFFIXES | {".pdf"}):
continue
try:
relative_knowledge_file(root, path.relative_to(root).as_posix())
except (OSError, ValueError):
continue
files.append(path)
if len(files) >= MAX_SCAN_FILES:
return files
return files
def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
terms = {
term.casefold()
for term in re.findall(r"[\w\u0600-\u06ff]{3,}", task)
if term.casefold() not in {
"the", "and", "for", "with", "this", "that", "من", "على", "في", "عن",
"كيف", "شو", "ما", "ماذا", "هذا", "هذه", "التي", "الذي", "اشرح", "دور",
"ملف", "ملفات", "اذكر", "المستخدمة", "المستخدم", "المشروع", "التفسير",
}
}
mentioned_paths = {
match.replace("\\", "/").casefold()
for match in re.findall(
r"(?:[\w.-]+[\\/])+[\w.-]+\.(?:py|dart|md|txt|json|ya?ml|toml|html|css|js|ts|tsx|jsx|sh|ps1)",
task,
flags=re.IGNORECASE,
)
}
ranked: list[tuple[int, str, str]] = []
for path in list_text_files(root):
try:
text = path.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
if not text.strip():
continue
lowered = text.casefold()
words = re.findall(r"[\w\u0600-\u06ff]+", lowered)
score = sum(words.count(term) for term in terms)
relative = path.relative_to(root).as_posix()
if relative.casefold() in mentioned_paths:
score += 100_000
if score:
# If the user named a source file explicitly, include a larger bounded
# excerpt so the agent can explain the code rather than just locate it.
excerpt_limit = 12_000 if relative.casefold() in mentioned_paths else 1_800
ranked.append((score, relative, text[:excerpt_limit]))
ranked.sort(key=lambda item: (-item[0], item[1]))
explicit_matches = [
item for item in ranked if item[1].casefold() in mentioned_paths
]
if explicit_matches:
ranked = explicit_matches
return [(relative, content) for _, relative, content in ranked[:limit]]