Files
sovereign_ai/SovereignAI-Starter/tests/test_agent_skills.py
T

367 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import asyncio
import json
import os
import tempfile
import unittest
from pathlib import Path
from unittest.mock import AsyncMock, patch
from fastapi import HTTPException
_TEST_DATA_DIR = None
if "SOVEREIGNAI_DATA_DIR" not in os.environ:
_TEST_DATA_DIR = tempfile.TemporaryDirectory(prefix="sovereignai-skills-tests-")
os.environ["SOVEREIGNAI_DATA_DIR"] = _TEST_DATA_DIR.name
from app.main import AgentRequest, _execute_agent, app, safe_arithmetic
from tests.api_client import authenticated_client
class AgentSkillTests(unittest.TestCase):
@classmethod
def setUpClass(cls) -> None:
cls.client = authenticated_client(app)
cls.workspace = str(Path(__file__).resolve().parents[1])
cls.workspace_environment = patch.dict(
os.environ, {"SOVEREIGNAI_ALLOWED_WORKSPACES": cls.workspace}
)
cls.workspace_environment.start()
cls.addClassCleanup(cls.workspace_environment.stop)
def test_skill_catalog_discloses_scope_and_permissions(self) -> None:
response = self.client.get("/v1/agent/skills")
self.assertEqual(response.status_code, 200)
skills = {item["id"]: item for item in response.json()["skills"]}
self.assertEqual(set(skills), {"code_explain", "code_review", "test_plan"})
self.assertNotIn("propose_file_change", skills["code_explain"]["allowed_tools"])
self.assertIn("propose_file_change", skills["code_review"]["allowed_tools"])
self.assertEqual(response.json()["default"], None)
def test_safe_arithmetic_accepts_common_unicode_operator_symbols(self) -> None:
self.assertEqual(safe_arithmetic("137 × 29"), 3973.0)
self.assertEqual(safe_arithmetic("12 ÷ 3 − 1"), 3.0)
def test_explicit_calculator_request_uses_bounded_local_calculator(self) -> None:
with patch("app.main.get_completion", new=AsyncMock()) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="استخدم الحاسبة المتاحة لحساب 137 × 29، ثم أجب بالناتج فقط.",
model="qwen2.5:1.5b-instruct-q4_K_M",
)
)
)
model.assert_not_awaited()
self.assertEqual(result["tool"], "calculator")
self.assertEqual(result["result"], 3973.0)
def test_explain_skill_is_sent_to_model_and_hides_file_write_tool(self) -> None:
completion = {"choices": [{"message": {"content": "شرح مختصر."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح بنية المشروع باختصار",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
payload = model.await_args.args[0]
tool_names = [tool["function"]["name"] for tool in payload["tools"]]
system = payload["messages"][0]["content"]
self.assertIn("شرح الكود", system)
self.assertIn("ميّز بين ما قرأته وما استنتجته", system)
self.assertEqual(tool_names, ["calculator", "search_workspace", "search_knowledge"])
self.assertEqual(result["skill"], "code_explain")
def test_review_skill_exposes_preview_tool_but_never_applies_it(self) -> None:
completion = {"choices": [{"message": {"content": "سأعرض النتائج."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="راجع الملفات دون تعديل",
workspace_path=self.workspace,
skill_id="code_review",
)
)
)
payload = model.await_args.args[0]
tool_names = [tool["function"]["name"] for tool in payload["tools"]]
self.assertIn("propose_file_change", tool_names)
self.assertIn("لا تطبق الكتابة", payload["messages"][0]["content"])
self.assertEqual(result["skill"], "code_review")
self.assertNotIn("proposal", result)
def test_model_cannot_call_a_tool_after_file_proposal_ends_tool_access(self) -> None:
completions = [
{"choices": [{"message": {"tool_calls": [{
"id": "proposal-1",
"function": {
"name": "propose_file_change",
"arguments": '{"path":"new.py","operation":"create","content":"print(1)"}',
},
}]}}]},
{"choices": [{"message": {"tool_calls": [{
"id": "late-calc",
"function": {"name": "calculator", "arguments": '{"expression":"1 + 1"}'},
}]}}]},
]
proposal = {
"path": "new.py",
"operation": "create",
"expires_in_seconds": 600,
}
with (
patch("app.main.workspace.create_change_preview", return_value=proposal) as create_preview,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(
task="أنشئ معاينة ملف new.py",
workspace_path=self.workspace,
skill_id="code_review",
)
)
)
self.assertEqual(error.exception.status_code, 422)
self.assertEqual(create_preview.call_count, 1)
self.assertEqual(model.await_count, 2)
def test_model_tool_names_must_be_strings_from_the_offered_tool_set(self) -> None:
malformed = {
"choices": [{"message": {"tool_calls": [{
"id": "bad-name",
"function": {"name": ["calculator"], "arguments": "{}"},
}]}}]
}
with patch("app.main.get_completion", new=AsyncMock(return_value=malformed)):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(task="احسب 1+1", skill_id="code_explain")
)
)
self.assertEqual(error.exception.status_code, 422)
def test_preselected_file_is_read_once_without_redundant_search_call(self) -> None:
completion = {"choices": [{"message": {"content": "المهارات مسجلة في قاموس محلي."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح الملف المحدد",
workspace_path=self.workspace,
workspace_files=["app/skills.py"],
skill_id="code_explain",
)
)
)
self.assertEqual(model.await_count, 1)
self.assertEqual(result["files"], ["app/skills.py"])
self.assertIn("Curated, local agent skills", model.await_args.args[0]["messages"][1]["content"])
def test_explicit_knowledge_search_is_prefetched_before_model_answer(self) -> None:
completion = {"choices": [{"message": {"content": "المرحلة 5 تضيف الفهرسة المحلية."}}]}
match = {"path": "ROADMAP.md", "chunk": 2, "text": "SQLite FTS5 local index", "excerpt": "SQLite FTS5"}
with (
patch("app.main.knowledge.search", return_value=[match]) as retrieve,
patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task="ابحث في فهرس المعرفة عن المرحلة 5",
workspace_path=self.workspace,
skill_id="test_plan",
)
)
)
retrieve.assert_called_once()
model_payload = model.await_args.args[0]
self.assertEqual(model_payload["max_tokens"], 384)
self.assertIn("SQLite FTS5 local index", model_payload["messages"][1]["content"])
self.assertNotIn(
"search_knowledge",
[tool["function"]["name"] for tool in model_payload["tools"]],
)
self.assertEqual(result["tool"], "search_knowledge")
self.assertEqual(result["files"], ["ROADMAP.md"])
def test_test_plan_skill_only_advertises_workspace_search(self) -> None:
completion = {"choices": [{"message": {"content": "ثلاث حالات اختبار مقترحة."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
asyncio.run(
_execute_agent(
AgentRequest(
task="أنشئ خطة اختبار",
workspace_path=self.workspace,
skill_id="test_plan",
)
)
)
tools = model.await_args.args[0]["tools"]
self.assertEqual(
[tool["function"]["name"] for tool in tools],
["search_workspace", "search_knowledge"],
)
def test_server_rejects_tool_call_outside_active_skill_permissions(self) -> None:
completion = {
"choices": [
{
"message": {
"tool_calls": [
{
"id": "call-forbidden",
"function": {
"name": "propose_file_change",
"arguments": '{"path":"new.py","operation":"create","content":"print(1)"}',
},
}
]
}
}
]
}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح المشروع",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
self.assertEqual(error.exception.status_code, 422)
def test_agent_can_chain_read_only_search_and_calculation(self) -> None:
def tool_call(call_id: str, name: str, arguments: dict[str, str]) -> dict:
return {
"id": call_id,
"type": "function",
"function": {"name": name, "arguments": json.dumps(arguments)},
}
completions = [
{"choices": [{"message": {"tool_calls": [
tool_call("search-1", "search_workspace", {"query": "secret key config"})
]}}]},
{"choices": [{"message": {"tool_calls": [
tool_call("calc-1", "calculator", {"expression": "19 * 23"})
]}}]},
{"choices": [{"message": {"content": "وجدت الإعداد، والحساب يساوي 437."}}]},
]
with (
patch("app.main.workspace.retrieve", return_value=[("app/config.py", "key comes from env")]) as search,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task="ابحث عن مصدر المفتاح واحسب 19 في 23",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
search.assert_called_once()
self.assertEqual(model.await_count, 3)
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "calculator"],
)
self.assertEqual(result["files"], ["app/config.py"])
self.assertIn("437", result["result"])
second_messages = model.await_args_list[1].args[0]["messages"]
self.assertEqual(second_messages[-1]["role"], "tool")
self.assertIn("key comes from env", second_messages[-1]["content"])
third_messages = model.await_args_list[2].args[0]["messages"]
self.assertEqual([message["role"] for message in third_messages[-2:]], ["assistant", "tool"])
def test_explicit_workspace_search_prefetches_before_followup_tool(self) -> None:
task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS، ثم احسب القيمة مضروبة في 7."
completions = [
{"choices": [{"message": {"tool_calls": [{
"id": "calc-after-search",
"function": {"name": "calculator", "arguments": '{"expression":"3 * 7"}'},
}]}}]},
{"choices": [{"message": {"content": "القيمة 3، والناتج 21 من app/main.py."}}]},
]
with (
patch("app.main.workspace.retrieve", return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")]) as search,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task=task,
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
search.assert_called_once_with(task, Path(self.workspace).resolve())
initial_payload = model.await_args_list[0].args[0]
self.assertIn(
"search_workspace",
[tool["function"]["name"] for tool in initial_payload["tools"]],
)
self.assertIn("MAX_AGENT_TOOL_CALLS = 3", initial_payload["messages"][1]["content"])
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "calculator"],
)
self.assertEqual(result["files"], ["app/main.py"])
self.assertIn("21", result["result"])
def test_agent_caps_sequential_tools_and_forces_final_model_turn(self) -> None:
tool_calls = [
{"choices": [{"message": {"tool_calls": [{
"id": f"calc-{index}",
"function": {"name": "calculator", "arguments": '{"expression":"2 + 3"}'},
}]}}]}
for index in range(3)
]
tool_calls.append({"choices": [{"message": {"content": "انتهيت بعد ثلاث خطوات."}}]})
with patch("app.main.get_completion", new=AsyncMock(side_effect=tool_calls)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(task="استخدم الحاسبة ثلاث مرات", skill_id="code_explain")
)
)
self.assertEqual(model.await_count, 4)
self.assertEqual(len(result["steps"]), 3)
self.assertNotIn("tools", model.await_args_list[3].args[0])
self.assertEqual(result["result"], "انتهيت بعد ثلاث خطوات.")
def test_unknown_skill_is_rejected_by_request_contract(self) -> None:
response = self.client.post(
"/v1/agent/run",
json={"task": "سؤال", "skill_id": "execute_shell"},
)
self.assertEqual(response.status_code, 422)
if __name__ == "__main__":
unittest.main()