Files
sovereign_ai/SovereignAI-Starter/tests/test_agent_skills.py
T

562 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import asyncio
import json
import os
import tempfile
import unittest
from pathlib import Path
from unittest.mock import AsyncMock, patch
from fastapi import HTTPException
_TEST_DATA_DIR = None
if "SOVEREIGNAI_DATA_DIR" not in os.environ:
_TEST_DATA_DIR = tempfile.TemporaryDirectory(prefix="sovereignai-skills-tests-")
os.environ["SOVEREIGNAI_DATA_DIR"] = _TEST_DATA_DIR.name
from app.main import (
AgentRequest,
_execute_agent,
_knowledge_context_for_model,
_is_knowledge_answer_insufficient,
app,
safe_arithmetic,
select_workspace_file_excerpt,
)
from tests.api_client import authenticated_client
class AgentSkillTests(unittest.TestCase):
@classmethod
def setUpClass(cls) -> None:
cls.client = authenticated_client(app)
cls.workspace = str(Path(__file__).resolve().parents[1])
cls.workspace_environment = patch.dict(
os.environ, {"SOVEREIGNAI_ALLOWED_WORKSPACES": cls.workspace}
)
cls.workspace_environment.start()
cls.addClassCleanup(cls.workspace_environment.stop)
def test_skill_catalog_discloses_scope_and_permissions(self) -> None:
response = self.client.get("/v1/agent/skills")
self.assertEqual(response.status_code, 200)
skills = {item["id"]: item for item in response.json()["skills"]}
self.assertEqual(set(skills), {"code_explain", "code_review", "test_plan"})
self.assertNotIn("propose_file_change", skills["code_explain"]["allowed_tools"])
self.assertIn("propose_file_change", skills["code_review"]["allowed_tools"])
self.assertEqual(response.json()["default"], None)
def test_safe_arithmetic_accepts_common_unicode_operator_symbols(self) -> None:
self.assertEqual(safe_arithmetic("137 × 29"), 3973.0)
self.assertEqual(safe_arithmetic("12 ÷ 3 − 1"), 3.0)
def test_knowledge_context_is_bounded_and_drops_redundant_fields(self) -> None:
matches = [
{
"path": "file-a.py" if index < 4 else "file-b.py",
"chunk": index,
"text": "x" * 1200,
"excerpt": "duplicate excerpt",
"similarity": 0.99,
}
for index in range(6)
]
context = _knowledge_context_for_model(matches)
self.assertEqual(len(context), 4)
self.assertTrue(all(len(item["text"]) <= 1000 for item in context))
self.assertTrue(all(set(item) == {"path", "chunk", "text"} for item in context))
self.assertEqual(
[item["path"] for item in context],
["file-a.py", "file-b.py", "file-a.py", "file-a.py"],
)
def test_explicit_calculator_request_uses_bounded_local_calculator(self) -> None:
with patch("app.main.get_completion", new=AsyncMock()) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="استخدم الحاسبة المتاحة لحساب 137 × 29، ثم أجب بالناتج فقط.",
model="qwen2.5:1.5b-instruct-q4_K_M",
)
)
)
model.assert_not_awaited()
self.assertEqual(result["tool"], "calculator")
self.assertEqual(result["result"], 3973.0)
def test_explain_skill_is_sent_to_model_and_hides_file_write_tool(self) -> None:
completion = {"choices": [{"message": {"content": "شرح مختصر."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح بنية المشروع باختصار",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
payload = model.await_args.args[0]
tool_names = [tool["function"]["name"] for tool in payload["tools"]]
system = payload["messages"][0]["content"]
self.assertIn("شرح الكود", system)
self.assertIn("ميّز بين ما قرأته وما استنتجته", system)
self.assertEqual(tool_names, ["calculator", "search_workspace", "search_knowledge"])
self.assertEqual(result["skill"], "code_explain")
def test_review_skill_exposes_preview_tool_but_never_applies_it(self) -> None:
completion = {"choices": [{"message": {"content": "سأعرض النتائج."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="راجع الملفات دون تعديل",
workspace_path=self.workspace,
skill_id="code_review",
)
)
)
payload = model.await_args.args[0]
tool_names = [tool["function"]["name"] for tool in payload["tools"]]
self.assertIn("propose_file_change", tool_names)
self.assertIn("لا تطبق الكتابة", payload["messages"][0]["content"])
self.assertEqual(result["skill"], "code_review")
self.assertNotIn("proposal", result)
def test_model_cannot_call_a_tool_after_file_proposal_ends_tool_access(self) -> None:
completions = [
{"choices": [{"message": {"tool_calls": [{
"id": "proposal-1",
"function": {
"name": "propose_file_change",
"arguments": '{"path":"new.py","operation":"create","content":"print(1)"}',
},
}]}}]},
{"choices": [{"message": {"tool_calls": [{
"id": "late-calc",
"function": {"name": "calculator", "arguments": '{"expression":"1 + 1"}'},
}]}}]},
]
proposal = {
"path": "new.py",
"operation": "create",
"expires_in_seconds": 600,
}
with (
patch("app.main.workspace.create_change_preview", return_value=proposal) as create_preview,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(
task="أنشئ معاينة ملف new.py",
workspace_path=self.workspace,
skill_id="code_review",
)
)
)
self.assertEqual(error.exception.status_code, 422)
self.assertEqual(create_preview.call_count, 1)
self.assertEqual(model.await_count, 2)
def test_model_tool_names_must_be_strings_from_the_offered_tool_set(self) -> None:
malformed = {
"choices": [{"message": {"tool_calls": [{
"id": "bad-name",
"function": {"name": ["calculator"], "arguments": "{}"},
}]}}]
}
with patch("app.main.get_completion", new=AsyncMock(return_value=malformed)):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(task="احسب 1+1", skill_id="code_explain")
)
)
self.assertEqual(error.exception.status_code, 422)
def test_selected_file_excerpt_prefers_passage_matching_question(self) -> None:
lines = [
*(f"Unrelated setup notes {index} {'x' * 100}" for index in range(80)),
"النموذج الافتراضي هو gemma4:e2b.",
]
text = "\n".join(lines)
excerpt = select_workspace_file_excerpt("ما النموذج الافتراضي؟", text)
self.assertIn("gemma4:e2b", excerpt)
self.assertIn("Unrelated setup notes 0", excerpt)
self.assertLessEqual(len(excerpt), 4000)
def test_preselected_file_is_read_once_without_redundant_search_call(self) -> None:
completion = {"choices": [{"message": {"content": "المهارات مسجلة في قاموس محلي."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح الملف المحدد",
workspace_path=self.workspace,
workspace_files=["app/skills.py"],
skill_id="code_explain",
)
)
)
self.assertEqual(model.await_count, 1)
self.assertEqual(result["files"], ["app/skills.py"])
payload = model.await_args.args[0]
self.assertEqual(payload["temperature"], 0.0)
self.assertEqual(payload["tools"], [])
self.assertNotIn("tool_choice", payload)
self.assertIn("أنت مساعد يجيب عن أسئلة الملفات", payload["messages"][0]["content"])
user_content = payload["messages"][1]["content"]
self.assertIn("Curated, local agent skills", user_content)
self.assertLess(user_content.index("اشرح الملف المحدد"), user_content.index("Curated, local agent skills"))
def test_explicit_knowledge_search_is_prefetched_before_model_answer(self) -> None:
completion = {"choices": [{"message": {"content": "المرحلة 5 تضيف الفهرسة المحلية."}}]}
match = {"path": "ROADMAP.md", "chunk": 2, "text": "SQLite FTS5 local index", "excerpt": "SQLite FTS5"}
with (
patch("app.main.knowledge.search", return_value=[match]) as retrieve,
patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task="ابحث في فهرس المعرفة عن المرحلة 5",
workspace_path=self.workspace,
skill_id="test_plan",
)
)
)
retrieve.assert_called_once()
model_payload = model.await_args.args[0]
self.assertEqual(model_payload["max_tokens"], 384)
self.assertIn("SQLite FTS5 local index", model_payload["messages"][1]["content"])
self.assertNotIn(
"search_knowledge",
[tool["function"]["name"] for tool in model_payload["tools"]],
)
self.assertEqual(model_payload["tools"], [])
self.assertNotIn("tool_choice", model_payload)
self.assertEqual(result["tool"], "search_knowledge")
self.assertEqual(result["files"], ["ROADMAP.md"])
def test_knowledge_refusal_falls_back_to_bounded_cited_evidence(self) -> None:
match = {
"path": "migration.md",
"chunk": 0,
"text": "The SQLite migration adds selected_version to old messages and preserves saved assistant answer versions.",
}
refusal = "لم يتم العثور على مقاطع تجيب مباشرة على السؤال في فهرس المعرفة المحلي."
live_refusal = 'لا يوجد مقطع في فهرس المعرفة المحلي يجيب مباشرة على السؤال "How are old conversation answers migrated?".'
with (
patch(
"app.main._search_local_knowledge",
new=AsyncMock(return_value=([match], "keyword")),
),
patch(
"app.main.get_completion",
new=AsyncMock(return_value={"choices": [{"message": {"content": refusal}}]}),
),
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task=(
"ابحث في فهرس المعرفة المحلي عن المقاطع التي تجيب عن السؤال، "
"ثم أجب استنادًا إلى المقاطع فقط. السؤال: "
"How are old conversation answers migrated?"
),
workspace_path=self.workspace,
)
)
)
citation_only = "المقطع `routing.md`, المقطع 0."
self.assertTrue(_is_knowledge_answer_insufficient(refusal))
self.assertTrue(_is_knowledge_answer_insufficient(live_refusal))
self.assertTrue(_is_knowledge_answer_insufficient(citation_only))
self.assertFalse(_is_knowledge_answer_insufficient("تضيف الهجرة selected_version."))
self.assertIn("لم يصغ النموذج جوابًا كافيًا رغم وجود نتائج", result["result"])
self.assertIn("migration.md · المقطع 0", result["result"])
self.assertIn("> The SQLite migration adds selected_version", result["result"])
def test_test_plan_skill_only_advertises_workspace_search(self) -> None:
completion = {"choices": [{"message": {"content": "ثلاث حالات اختبار مقترحة."}}]}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model:
asyncio.run(
_execute_agent(
AgentRequest(
task="أنشئ خطة اختبار",
workspace_path=self.workspace,
skill_id="test_plan",
)
)
)
tools = model.await_args.args[0]["tools"]
self.assertEqual(
[tool["function"]["name"] for tool in tools],
["search_workspace", "search_knowledge"],
)
def test_server_rejects_tool_call_outside_active_skill_permissions(self) -> None:
completion = {
"choices": [
{
"message": {
"tool_calls": [
{
"id": "call-forbidden",
"function": {
"name": "propose_file_change",
"arguments": '{"path":"new.py","operation":"create","content":"print(1)"}',
},
}
]
}
}
]
}
with patch("app.main.get_completion", new=AsyncMock(return_value=completion)):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(
task="اشرح المشروع",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
self.assertEqual(error.exception.status_code, 422)
def test_agent_can_chain_read_only_search_and_calculation(self) -> None:
def tool_call(call_id: str, name: str, arguments: dict[str, str]) -> dict:
return {
"id": call_id,
"type": "function",
"function": {"name": name, "arguments": json.dumps(arguments)},
}
completions = [
{"choices": [{"message": {"tool_calls": [
tool_call("search-1", "search_workspace", {"query": "secret key config"})
]}}]},
{"choices": [{"message": {"tool_calls": [
tool_call("calc-1", "calculator", {"expression": "19 * 23"})
]}}]},
{"choices": [{"message": {"content": "وجدت الإعداد، والحساب يساوي 437."}}]},
]
with (
patch("app.main.workspace.retrieve", return_value=[("app/config.py", "key comes from env")]) as search,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task="ابحث عن مصدر المفتاح واحسب 19 في 23",
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
search.assert_called_once()
self.assertEqual(model.await_count, 3)
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "calculator"],
)
self.assertEqual(result["files"], ["app/config.py"])
self.assertIn("437", result["result"])
second_messages = model.await_args_list[1].args[0]["messages"]
self.assertEqual(second_messages[-1]["role"], "tool")
self.assertIn("key comes from env", second_messages[-1]["content"])
third_messages = model.await_args_list[2].args[0]["messages"]
self.assertEqual([message["role"] for message in third_messages[-2:]], ["assistant", "tool"])
def test_agent_reuses_duplicate_read_only_tool_result_and_finishes(self) -> None:
repeated_call = {
"id": "search-repeat",
"type": "function",
"function": {
"name": "search_workspace",
"arguments": '{"query":"same query"}',
},
}
completions = [
{"choices": [{"message": {"tool_calls": [repeated_call]}}]},
{"choices": [{"message": {"tool_calls": [{**repeated_call, "id": "search-repeat-2"}]}}]},
{"choices": [{"message": {"content": "وجدت المعلومة في الملف."}}]},
]
with (
patch(
"app.main.workspace.retrieve",
return_value=[("app/answer.py", "المعلومة المطلوبة")],
) as search,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(task="ابحث عن المعلومة", workspace_path=self.workspace)
)
)
search.assert_called_once_with("same query", Path(self.workspace).resolve())
self.assertEqual(model.await_count, 3)
self.assertNotIn("tools", model.await_args_list[2].args[0])
self.assertEqual(result["files"], ["app/answer.py"])
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "search_workspace"],
)
self.assertEqual(result["result"], "وجدت المعلومة في الملف.")
def test_explicit_workspace_search_prefetches_before_followup_tool(self) -> None:
task = "ابحث في ملفات المشروع عن كلمة privacy، ثم احسب 3 × 7."
completions = [
{"choices": [{"message": {"tool_calls": [{
"id": "calc-after-search",
"function": {"name": "calculator", "arguments": '{"expression":"3 * 7"}'},
}]}}]},
{"choices": [{"message": {"content": "القيمة 3، والناتج 21 من app/main.py."}}]},
]
with (
patch("app.main.workspace.retrieve", return_value=[("README.md", "Local workspace stays private.")]) as search,
patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task=task,
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
search.assert_called_once_with(task, Path(self.workspace).resolve())
initial_payload = model.await_args_list[0].args[0]
self.assertNotIn(
"search_workspace",
[tool["function"]["name"] for tool in initial_payload["tools"]],
)
self.assertNotIn(
"search_knowledge",
[tool["function"]["name"] for tool in initial_payload["tools"]],
)
self.assertEqual(
[tool["function"]["name"] for tool in initial_payload["tools"]],
["calculator"],
)
system_message = initial_payload["messages"][0]["content"]
self.assertIn("بحث الخادم في مساحة العمل المسموحة مسبقًا", system_message)
self.assertNotIn("استخدم search_workspace", system_message)
self.assertIn("Local workspace stays private.", initial_payload["messages"][1]["content"])
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "calculator"],
)
self.assertEqual(result["files"], ["README.md"])
self.assertIn("21", result["result"])
def test_explicit_search_multiplies_one_retrieved_numeric_constant_locally(self) -> None:
task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS، ثم احسبها مضروبة في 7."
with (
patch(
"app.main.workspace.retrieve",
return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")],
) as search,
patch("app.main.get_completion", new=AsyncMock()) as model,
):
result = asyncio.run(
_execute_agent(
AgentRequest(
task=task,
workspace_path=self.workspace,
skill_id="code_explain",
)
)
)
search.assert_called_once_with(task, Path(self.workspace).resolve())
model.assert_not_awaited()
self.assertEqual(
[step["tool"] for step in result["steps"]],
["search_workspace", "calculator"],
)
self.assertEqual(result["files"], ["app/main.py"])
self.assertIn("3 × 7 = 21", result["result"])
def test_explicit_workspace_search_rejects_repeating_prefetched_search(self) -> None:
task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS."
completion = {
"choices": [{"message": {"tool_calls": [{
"id": "duplicate-prefetch",
"function": {
"name": "search_workspace",
"arguments": '{"query":"MAX_AGENT_TOOL_CALLS"}',
},
}]}}]
}
with (
patch("app.main.workspace.retrieve", return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")]) as search,
patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model,
):
with self.assertRaises(HTTPException) as error:
asyncio.run(
_execute_agent(
AgentRequest(task=task, workspace_path=self.workspace)
)
)
self.assertEqual(error.exception.status_code, 422)
search.assert_called_once_with(task, Path(self.workspace).resolve())
offered = model.await_args.args[0]["tools"]
self.assertNotIn("search_workspace", [item["function"]["name"] for item in offered])
def test_agent_caps_sequential_tools_and_forces_final_model_turn(self) -> None:
tool_calls = [
{"choices": [{"message": {"tool_calls": [{
"id": f"calc-{index}",
"function": {
"name": "calculator",
"arguments": json.dumps({"expression": f"2 + {index + 3}"}),
},
}]}}]}
for index in range(3)
]
tool_calls.append({"choices": [{"message": {"content": "انتهيت بعد ثلاث خطوات."}}]})
with patch("app.main.get_completion", new=AsyncMock(side_effect=tool_calls)) as model:
result = asyncio.run(
_execute_agent(
AgentRequest(task="استخدم الحاسبة ثلاث مرات", skill_id="code_explain")
)
)
self.assertEqual(model.await_count, 4)
self.assertEqual(len(result["steps"]), 3)
self.assertNotIn("tools", model.await_args_list[3].args[0])
self.assertEqual(result["result"], "انتهيت بعد ثلاث خطوات.")
def test_unknown_skill_is_rejected_by_request_contract(self) -> None:
response = self.client.post(
"/v1/agent/run",
json={"task": "سؤال", "skill_id": "execute_shell"},
)
self.assertEqual(response.status_code, 422)
if __name__ == "__main__":
unittest.main()