import asyncio import json import os import tempfile import unittest from pathlib import Path from unittest.mock import AsyncMock, patch from fastapi import HTTPException _TEST_DATA_DIR = None if "SOVEREIGNAI_DATA_DIR" not in os.environ: _TEST_DATA_DIR = tempfile.TemporaryDirectory(prefix="sovereignai-skills-tests-") os.environ["SOVEREIGNAI_DATA_DIR"] = _TEST_DATA_DIR.name from app.main import AgentRequest, _execute_agent, app, safe_arithmetic from tests.api_client import authenticated_client class AgentSkillTests(unittest.TestCase): @classmethod def setUpClass(cls) -> None: cls.client = authenticated_client(app) cls.workspace = str(Path(__file__).resolve().parents[1]) cls.workspace_environment = patch.dict( os.environ, {"SOVEREIGNAI_ALLOWED_WORKSPACES": cls.workspace} ) cls.workspace_environment.start() cls.addClassCleanup(cls.workspace_environment.stop) def test_skill_catalog_discloses_scope_and_permissions(self) -> None: response = self.client.get("/v1/agent/skills") self.assertEqual(response.status_code, 200) skills = {item["id"]: item for item in response.json()["skills"]} self.assertEqual(set(skills), {"code_explain", "code_review", "test_plan"}) self.assertNotIn("propose_file_change", skills["code_explain"]["allowed_tools"]) self.assertIn("propose_file_change", skills["code_review"]["allowed_tools"]) self.assertEqual(response.json()["default"], None) def test_safe_arithmetic_accepts_common_unicode_operator_symbols(self) -> None: self.assertEqual(safe_arithmetic("137 × 29"), 3973.0) self.assertEqual(safe_arithmetic("12 ÷ 3 − 1"), 3.0) def test_explicit_calculator_request_uses_bounded_local_calculator(self) -> None: with patch("app.main.get_completion", new=AsyncMock()) as model: result = asyncio.run( _execute_agent( AgentRequest( task="استخدم الحاسبة المتاحة لحساب 137 × 29، ثم أجب بالناتج فقط.", model="qwen2.5:1.5b-instruct-q4_K_M", ) ) ) model.assert_not_awaited() self.assertEqual(result["tool"], "calculator") self.assertEqual(result["result"], 3973.0) def test_explain_skill_is_sent_to_model_and_hides_file_write_tool(self) -> None: completion = {"choices": [{"message": {"content": "شرح مختصر."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="اشرح بنية المشروع باختصار", workspace_path=self.workspace, skill_id="code_explain", ) ) ) payload = model.await_args.args[0] tool_names = [tool["function"]["name"] for tool in payload["tools"]] system = payload["messages"][0]["content"] self.assertIn("شرح الكود", system) self.assertIn("ميّز بين ما قرأته وما استنتجته", system) self.assertEqual(tool_names, ["calculator", "search_workspace", "search_knowledge"]) self.assertEqual(result["skill"], "code_explain") def test_review_skill_exposes_preview_tool_but_never_applies_it(self) -> None: completion = {"choices": [{"message": {"content": "سأعرض النتائج."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="راجع الملفات دون تعديل", workspace_path=self.workspace, skill_id="code_review", ) ) ) payload = model.await_args.args[0] tool_names = [tool["function"]["name"] for tool in payload["tools"]] self.assertIn("propose_file_change", tool_names) self.assertIn("لا تطبق الكتابة", payload["messages"][0]["content"]) self.assertEqual(result["skill"], "code_review") self.assertNotIn("proposal", result) def test_model_cannot_call_a_tool_after_file_proposal_ends_tool_access(self) -> None: completions = [ {"choices": [{"message": {"tool_calls": [{ "id": "proposal-1", "function": { "name": "propose_file_change", "arguments": '{"path":"new.py","operation":"create","content":"print(1)"}', }, }]}}]}, {"choices": [{"message": {"tool_calls": [{ "id": "late-calc", "function": {"name": "calculator", "arguments": '{"expression":"1 + 1"}'}, }]}}]}, ] proposal = { "path": "new.py", "operation": "create", "expires_in_seconds": 600, } with ( patch("app.main.workspace.create_change_preview", return_value=proposal) as create_preview, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest( task="أنشئ معاينة ملف new.py", workspace_path=self.workspace, skill_id="code_review", ) ) ) self.assertEqual(error.exception.status_code, 422) self.assertEqual(create_preview.call_count, 1) self.assertEqual(model.await_count, 2) def test_model_tool_names_must_be_strings_from_the_offered_tool_set(self) -> None: malformed = { "choices": [{"message": {"tool_calls": [{ "id": "bad-name", "function": {"name": ["calculator"], "arguments": "{}"}, }]}}] } with patch("app.main.get_completion", new=AsyncMock(return_value=malformed)): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest(task="احسب 1+1", skill_id="code_explain") ) ) self.assertEqual(error.exception.status_code, 422) def test_preselected_file_is_read_once_without_redundant_search_call(self) -> None: completion = {"choices": [{"message": {"content": "المهارات مسجلة في قاموس محلي."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="اشرح الملف المحدد", workspace_path=self.workspace, workspace_files=["app/skills.py"], skill_id="code_explain", ) ) ) self.assertEqual(model.await_count, 1) self.assertEqual(result["files"], ["app/skills.py"]) self.assertIn("Curated, local agent skills", model.await_args.args[0]["messages"][1]["content"]) def test_explicit_knowledge_search_is_prefetched_before_model_answer(self) -> None: completion = {"choices": [{"message": {"content": "المرحلة 5 تضيف الفهرسة المحلية."}}]} match = {"path": "ROADMAP.md", "chunk": 2, "text": "SQLite FTS5 local index", "excerpt": "SQLite FTS5"} with ( patch("app.main.knowledge.search", return_value=[match]) as retrieve, patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task="ابحث في فهرس المعرفة عن المرحلة 5", workspace_path=self.workspace, skill_id="test_plan", ) ) ) retrieve.assert_called_once() model_payload = model.await_args.args[0] self.assertEqual(model_payload["max_tokens"], 384) self.assertIn("SQLite FTS5 local index", model_payload["messages"][1]["content"]) self.assertNotIn( "search_knowledge", [tool["function"]["name"] for tool in model_payload["tools"]], ) self.assertEqual(result["tool"], "search_knowledge") self.assertEqual(result["files"], ["ROADMAP.md"]) def test_test_plan_skill_only_advertises_workspace_search(self) -> None: completion = {"choices": [{"message": {"content": "ثلاث حالات اختبار مقترحة."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: asyncio.run( _execute_agent( AgentRequest( task="أنشئ خطة اختبار", workspace_path=self.workspace, skill_id="test_plan", ) ) ) tools = model.await_args.args[0]["tools"] self.assertEqual( [tool["function"]["name"] for tool in tools], ["search_workspace", "search_knowledge"], ) def test_server_rejects_tool_call_outside_active_skill_permissions(self) -> None: completion = { "choices": [ { "message": { "tool_calls": [ { "id": "call-forbidden", "function": { "name": "propose_file_change", "arguments": '{"path":"new.py","operation":"create","content":"print(1)"}', }, } ] } } ] } with patch("app.main.get_completion", new=AsyncMock(return_value=completion)): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest( task="اشرح المشروع", workspace_path=self.workspace, skill_id="code_explain", ) ) ) self.assertEqual(error.exception.status_code, 422) def test_agent_can_chain_read_only_search_and_calculation(self) -> None: def tool_call(call_id: str, name: str, arguments: dict[str, str]) -> dict: return { "id": call_id, "type": "function", "function": {"name": name, "arguments": json.dumps(arguments)}, } completions = [ {"choices": [{"message": {"tool_calls": [ tool_call("search-1", "search_workspace", {"query": "secret key config"}) ]}}]}, {"choices": [{"message": {"tool_calls": [ tool_call("calc-1", "calculator", {"expression": "19 * 23"}) ]}}]}, {"choices": [{"message": {"content": "وجدت الإعداد، والحساب يساوي 437."}}]}, ] with ( patch("app.main.workspace.retrieve", return_value=[("app/config.py", "key comes from env")]) as search, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task="ابحث عن مصدر المفتاح واحسب 19 في 23", workspace_path=self.workspace, skill_id="code_explain", ) ) ) search.assert_called_once() self.assertEqual(model.await_count, 3) self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "calculator"], ) self.assertEqual(result["files"], ["app/config.py"]) self.assertIn("437", result["result"]) second_messages = model.await_args_list[1].args[0]["messages"] self.assertEqual(second_messages[-1]["role"], "tool") self.assertIn("key comes from env", second_messages[-1]["content"]) third_messages = model.await_args_list[2].args[0]["messages"] self.assertEqual([message["role"] for message in third_messages[-2:]], ["assistant", "tool"]) def test_explicit_workspace_search_prefetches_before_followup_tool(self) -> None: task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS، ثم احسب القيمة مضروبة في 7." completions = [ {"choices": [{"message": {"tool_calls": [{ "id": "calc-after-search", "function": {"name": "calculator", "arguments": '{"expression":"3 * 7"}'}, }]}}]}, {"choices": [{"message": {"content": "القيمة 3، والناتج 21 من app/main.py."}}]}, ] with ( patch("app.main.workspace.retrieve", return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")]) as search, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task=task, workspace_path=self.workspace, skill_id="code_explain", ) ) ) search.assert_called_once_with(task, Path(self.workspace).resolve()) initial_payload = model.await_args_list[0].args[0] self.assertIn( "search_workspace", [tool["function"]["name"] for tool in initial_payload["tools"]], ) self.assertIn("MAX_AGENT_TOOL_CALLS = 3", initial_payload["messages"][1]["content"]) self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "calculator"], ) self.assertEqual(result["files"], ["app/main.py"]) self.assertIn("21", result["result"]) def test_agent_caps_sequential_tools_and_forces_final_model_turn(self) -> None: tool_calls = [ {"choices": [{"message": {"tool_calls": [{ "id": f"calc-{index}", "function": {"name": "calculator", "arguments": '{"expression":"2 + 3"}'}, }]}}]} for index in range(3) ] tool_calls.append({"choices": [{"message": {"content": "انتهيت بعد ثلاث خطوات."}}]}) with patch("app.main.get_completion", new=AsyncMock(side_effect=tool_calls)) as model: result = asyncio.run( _execute_agent( AgentRequest(task="استخدم الحاسبة ثلاث مرات", skill_id="code_explain") ) ) self.assertEqual(model.await_count, 4) self.assertEqual(len(result["steps"]), 3) self.assertNotIn("tools", model.await_args_list[3].args[0]) self.assertEqual(result["result"], "انتهيت بعد ثلاث خطوات.") def test_unknown_skill_is_rejected_by_request_contract(self) -> None: response = self.client.post( "/v1/agent/run", json={"task": "سؤال", "skill_id": "execute_shell"}, ) self.assertEqual(response.status_code, 422) if __name__ == "__main__": unittest.main()