import asyncio import json import os import tempfile import unittest from pathlib import Path from unittest.mock import AsyncMock, patch from fastapi import HTTPException _TEST_DATA_DIR = None if "SOVEREIGNAI_DATA_DIR" not in os.environ: _TEST_DATA_DIR = tempfile.TemporaryDirectory(prefix="sovereignai-skills-tests-") os.environ["SOVEREIGNAI_DATA_DIR"] = _TEST_DATA_DIR.name from app.main import ( AgentRequest, _execute_agent, _knowledge_context_for_model, _is_knowledge_answer_insufficient, app, safe_arithmetic, select_workspace_file_excerpt, ) from tests.api_client import authenticated_client class AgentSkillTests(unittest.TestCase): @classmethod def setUpClass(cls) -> None: cls.client = authenticated_client(app) cls.workspace = str(Path(__file__).resolve().parents[1]) cls.workspace_environment = patch.dict( os.environ, {"SOVEREIGNAI_ALLOWED_WORKSPACES": cls.workspace} ) cls.workspace_environment.start() cls.addClassCleanup(cls.workspace_environment.stop) def test_skill_catalog_discloses_scope_and_permissions(self) -> None: response = self.client.get("/v1/agent/skills") self.assertEqual(response.status_code, 200) skills = {item["id"]: item for item in response.json()["skills"]} self.assertEqual(set(skills), {"code_explain", "code_review", "test_plan"}) self.assertNotIn("propose_file_change", skills["code_explain"]["allowed_tools"]) self.assertIn("propose_file_change", skills["code_review"]["allowed_tools"]) self.assertEqual(response.json()["default"], None) def test_safe_arithmetic_accepts_common_unicode_operator_symbols(self) -> None: self.assertEqual(safe_arithmetic("137 × 29"), 3973.0) self.assertEqual(safe_arithmetic("12 ÷ 3 − 1"), 3.0) def test_knowledge_context_is_bounded_and_drops_redundant_fields(self) -> None: matches = [ { "path": "file-a.py" if index < 4 else "file-b.py", "chunk": index, "text": "x" * 1200, "excerpt": "duplicate excerpt", "similarity": 0.99, } for index in range(6) ] context = _knowledge_context_for_model(matches) self.assertEqual(len(context), 4) self.assertTrue(all(len(item["text"]) <= 1000 for item in context)) self.assertTrue(all(set(item) == {"path", "chunk", "text"} for item in context)) self.assertEqual( [item["path"] for item in context], ["file-a.py", "file-b.py", "file-a.py", "file-a.py"], ) def test_explicit_calculator_request_uses_bounded_local_calculator(self) -> None: with patch("app.main.get_completion", new=AsyncMock()) as model: result = asyncio.run( _execute_agent( AgentRequest( task="استخدم الحاسبة المتاحة لحساب 137 × 29، ثم أجب بالناتج فقط.", model="qwen2.5:1.5b-instruct-q4_K_M", ) ) ) model.assert_not_awaited() self.assertEqual(result["tool"], "calculator") self.assertEqual(result["result"], 3973.0) def test_explain_skill_is_sent_to_model_and_hides_file_write_tool(self) -> None: completion = {"choices": [{"message": {"content": "شرح مختصر."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="اشرح بنية المشروع باختصار", workspace_path=self.workspace, skill_id="code_explain", ) ) ) payload = model.await_args.args[0] tool_names = [tool["function"]["name"] for tool in payload["tools"]] system = payload["messages"][0]["content"] self.assertIn("شرح الكود", system) self.assertIn("ميّز بين ما قرأته وما استنتجته", system) self.assertEqual(tool_names, ["calculator", "search_workspace", "search_knowledge"]) self.assertEqual(result["skill"], "code_explain") def test_review_skill_exposes_preview_tool_but_never_applies_it(self) -> None: completion = {"choices": [{"message": {"content": "سأعرض النتائج."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="راجع الملفات دون تعديل", workspace_path=self.workspace, skill_id="code_review", ) ) ) payload = model.await_args.args[0] tool_names = [tool["function"]["name"] for tool in payload["tools"]] self.assertIn("propose_file_change", tool_names) self.assertIn("لا تطبق الكتابة", payload["messages"][0]["content"]) self.assertEqual(result["skill"], "code_review") self.assertNotIn("proposal", result) def test_model_cannot_call_a_tool_after_file_proposal_ends_tool_access(self) -> None: completions = [ {"choices": [{"message": {"tool_calls": [{ "id": "proposal-1", "function": { "name": "propose_file_change", "arguments": '{"path":"new.py","operation":"create","content":"print(1)"}', }, }]}}]}, {"choices": [{"message": {"tool_calls": [{ "id": "late-calc", "function": {"name": "calculator", "arguments": '{"expression":"1 + 1"}'}, }]}}]}, ] proposal = { "path": "new.py", "operation": "create", "expires_in_seconds": 600, } with ( patch("app.main.workspace.create_change_preview", return_value=proposal) as create_preview, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest( task="أنشئ معاينة ملف new.py", workspace_path=self.workspace, skill_id="code_review", ) ) ) self.assertEqual(error.exception.status_code, 422) self.assertEqual(create_preview.call_count, 1) self.assertEqual(model.await_count, 2) def test_model_tool_names_must_be_strings_from_the_offered_tool_set(self) -> None: malformed = { "choices": [{"message": {"tool_calls": [{ "id": "bad-name", "function": {"name": ["calculator"], "arguments": "{}"}, }]}}] } with patch("app.main.get_completion", new=AsyncMock(return_value=malformed)): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest(task="احسب 1+1", skill_id="code_explain") ) ) self.assertEqual(error.exception.status_code, 422) def test_selected_file_excerpt_prefers_passage_matching_question(self) -> None: lines = [ *(f"Unrelated setup notes {index} {'x' * 100}" for index in range(80)), "النموذج الافتراضي هو gemma4:e2b.", ] text = "\n".join(lines) excerpt = select_workspace_file_excerpt("ما النموذج الافتراضي؟", text) self.assertIn("gemma4:e2b", excerpt) self.assertIn("Unrelated setup notes 0", excerpt) self.assertLessEqual(len(excerpt), 4000) def test_preselected_file_is_read_once_without_redundant_search_call(self) -> None: completion = {"choices": [{"message": {"content": "المهارات مسجلة في قاموس محلي."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: result = asyncio.run( _execute_agent( AgentRequest( task="اشرح الملف المحدد", workspace_path=self.workspace, workspace_files=["app/skills.py"], skill_id="code_explain", ) ) ) self.assertEqual(model.await_count, 1) self.assertEqual(result["files"], ["app/skills.py"]) payload = model.await_args.args[0] self.assertEqual(payload["temperature"], 0.0) self.assertEqual(payload["tools"], []) self.assertNotIn("tool_choice", payload) self.assertIn("أنت مساعد يجيب عن أسئلة الملفات", payload["messages"][0]["content"]) user_content = payload["messages"][1]["content"] self.assertIn("Curated, local agent skills", user_content) self.assertLess(user_content.index("اشرح الملف المحدد"), user_content.index("Curated, local agent skills")) def test_explicit_knowledge_search_is_prefetched_before_model_answer(self) -> None: completion = {"choices": [{"message": {"content": "المرحلة 5 تضيف الفهرسة المحلية."}}]} match = {"path": "ROADMAP.md", "chunk": 2, "text": "SQLite FTS5 local index", "excerpt": "SQLite FTS5"} with ( patch("app.main.knowledge.search", return_value=[match]) as retrieve, patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task="ابحث في فهرس المعرفة عن المرحلة 5", workspace_path=self.workspace, skill_id="test_plan", ) ) ) retrieve.assert_called_once() model_payload = model.await_args.args[0] self.assertEqual(model_payload["max_tokens"], 384) self.assertIn("SQLite FTS5 local index", model_payload["messages"][1]["content"]) self.assertNotIn( "search_knowledge", [tool["function"]["name"] for tool in model_payload["tools"]], ) self.assertEqual(model_payload["tools"], []) self.assertNotIn("tool_choice", model_payload) self.assertEqual(result["tool"], "search_knowledge") self.assertEqual(result["files"], ["ROADMAP.md"]) def test_knowledge_refusal_falls_back_to_bounded_cited_evidence(self) -> None: match = { "path": "migration.md", "chunk": 0, "text": "The SQLite migration adds selected_version to old messages and preserves saved assistant answer versions.", } refusal = "لم يتم العثور على مقاطع تجيب مباشرة على السؤال في فهرس المعرفة المحلي." live_refusal = 'لا يوجد مقطع في فهرس المعرفة المحلي يجيب مباشرة على السؤال "How are old conversation answers migrated?".' with ( patch( "app.main._search_local_knowledge", new=AsyncMock(return_value=([match], "keyword")), ), patch( "app.main.get_completion", new=AsyncMock(return_value={"choices": [{"message": {"content": refusal}}]}), ), ): result = asyncio.run( _execute_agent( AgentRequest( task=( "ابحث في فهرس المعرفة المحلي عن المقاطع التي تجيب عن السؤال، " "ثم أجب استنادًا إلى المقاطع فقط. السؤال: " "How are old conversation answers migrated?" ), workspace_path=self.workspace, ) ) ) citation_only = "المقطع `routing.md`, المقطع 0." self.assertTrue(_is_knowledge_answer_insufficient(refusal)) self.assertTrue(_is_knowledge_answer_insufficient(live_refusal)) self.assertTrue(_is_knowledge_answer_insufficient(citation_only)) self.assertFalse(_is_knowledge_answer_insufficient("تضيف الهجرة selected_version.")) self.assertIn("لم يصغ النموذج جوابًا كافيًا رغم وجود نتائج", result["result"]) self.assertIn("migration.md · المقطع 0", result["result"]) self.assertIn("> The SQLite migration adds selected_version", result["result"]) def test_test_plan_skill_only_advertises_workspace_search(self) -> None: completion = {"choices": [{"message": {"content": "ثلاث حالات اختبار مقترحة."}}]} with patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model: asyncio.run( _execute_agent( AgentRequest( task="أنشئ خطة اختبار", workspace_path=self.workspace, skill_id="test_plan", ) ) ) tools = model.await_args.args[0]["tools"] self.assertEqual( [tool["function"]["name"] for tool in tools], ["search_workspace", "search_knowledge"], ) def test_server_rejects_tool_call_outside_active_skill_permissions(self) -> None: completion = { "choices": [ { "message": { "tool_calls": [ { "id": "call-forbidden", "function": { "name": "propose_file_change", "arguments": '{"path":"new.py","operation":"create","content":"print(1)"}', }, } ] } } ] } with patch("app.main.get_completion", new=AsyncMock(return_value=completion)): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest( task="اشرح المشروع", workspace_path=self.workspace, skill_id="code_explain", ) ) ) self.assertEqual(error.exception.status_code, 422) def test_agent_can_chain_read_only_search_and_calculation(self) -> None: def tool_call(call_id: str, name: str, arguments: dict[str, str]) -> dict: return { "id": call_id, "type": "function", "function": {"name": name, "arguments": json.dumps(arguments)}, } completions = [ {"choices": [{"message": {"tool_calls": [ tool_call("search-1", "search_workspace", {"query": "secret key config"}) ]}}]}, {"choices": [{"message": {"tool_calls": [ tool_call("calc-1", "calculator", {"expression": "19 * 23"}) ]}}]}, {"choices": [{"message": {"content": "وجدت الإعداد، والحساب يساوي 437."}}]}, ] with ( patch("app.main.workspace.retrieve", return_value=[("app/config.py", "key comes from env")]) as search, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task="ابحث عن مصدر المفتاح واحسب 19 في 23", workspace_path=self.workspace, skill_id="code_explain", ) ) ) search.assert_called_once() self.assertEqual(model.await_count, 3) self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "calculator"], ) self.assertEqual(result["files"], ["app/config.py"]) self.assertIn("437", result["result"]) second_messages = model.await_args_list[1].args[0]["messages"] self.assertEqual(second_messages[-1]["role"], "tool") self.assertIn("key comes from env", second_messages[-1]["content"]) third_messages = model.await_args_list[2].args[0]["messages"] self.assertEqual([message["role"] for message in third_messages[-2:]], ["assistant", "tool"]) def test_agent_reuses_duplicate_read_only_tool_result_and_finishes(self) -> None: repeated_call = { "id": "search-repeat", "type": "function", "function": { "name": "search_workspace", "arguments": '{"query":"same query"}', }, } completions = [ {"choices": [{"message": {"tool_calls": [repeated_call]}}]}, {"choices": [{"message": {"tool_calls": [{**repeated_call, "id": "search-repeat-2"}]}}]}, {"choices": [{"message": {"content": "وجدت المعلومة في الملف."}}]}, ] with ( patch( "app.main.workspace.retrieve", return_value=[("app/answer.py", "المعلومة المطلوبة")], ) as search, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): result = asyncio.run( _execute_agent( AgentRequest(task="ابحث عن المعلومة", workspace_path=self.workspace) ) ) search.assert_called_once_with("same query", Path(self.workspace).resolve()) self.assertEqual(model.await_count, 3) self.assertNotIn("tools", model.await_args_list[2].args[0]) self.assertEqual(result["files"], ["app/answer.py"]) self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "search_workspace"], ) self.assertEqual(result["result"], "وجدت المعلومة في الملف.") def test_explicit_workspace_search_prefetches_before_followup_tool(self) -> None: task = "ابحث في ملفات المشروع عن كلمة privacy، ثم احسب 3 × 7." completions = [ {"choices": [{"message": {"tool_calls": [{ "id": "calc-after-search", "function": {"name": "calculator", "arguments": '{"expression":"3 * 7"}'}, }]}}]}, {"choices": [{"message": {"content": "القيمة 3، والناتج 21 من app/main.py."}}]}, ] with ( patch("app.main.workspace.retrieve", return_value=[("README.md", "Local workspace stays private.")]) as search, patch("app.main.get_completion", new=AsyncMock(side_effect=completions)) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task=task, workspace_path=self.workspace, skill_id="code_explain", ) ) ) search.assert_called_once_with(task, Path(self.workspace).resolve()) initial_payload = model.await_args_list[0].args[0] self.assertNotIn( "search_workspace", [tool["function"]["name"] for tool in initial_payload["tools"]], ) self.assertNotIn( "search_knowledge", [tool["function"]["name"] for tool in initial_payload["tools"]], ) self.assertEqual( [tool["function"]["name"] for tool in initial_payload["tools"]], ["calculator"], ) system_message = initial_payload["messages"][0]["content"] self.assertIn("بحث الخادم في مساحة العمل المسموحة مسبقًا", system_message) self.assertNotIn("استخدم search_workspace", system_message) self.assertIn("Local workspace stays private.", initial_payload["messages"][1]["content"]) self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "calculator"], ) self.assertEqual(result["files"], ["README.md"]) self.assertIn("21", result["result"]) def test_explicit_search_multiplies_one_retrieved_numeric_constant_locally(self) -> None: task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS، ثم احسبها مضروبة في 7." with ( patch( "app.main.workspace.retrieve", return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")], ) as search, patch("app.main.get_completion", new=AsyncMock()) as model, ): result = asyncio.run( _execute_agent( AgentRequest( task=task, workspace_path=self.workspace, skill_id="code_explain", ) ) ) search.assert_called_once_with(task, Path(self.workspace).resolve()) model.assert_not_awaited() self.assertEqual( [step["tool"] for step in result["steps"]], ["search_workspace", "calculator"], ) self.assertEqual(result["files"], ["app/main.py"]) self.assertIn("3 × 7 = 21", result["result"]) def test_explicit_workspace_search_rejects_repeating_prefetched_search(self) -> None: task = "ابحث في ملفات المشروع عن قيمة MAX_AGENT_TOOL_CALLS." completion = { "choices": [{"message": {"tool_calls": [{ "id": "duplicate-prefetch", "function": { "name": "search_workspace", "arguments": '{"query":"MAX_AGENT_TOOL_CALLS"}', }, }]}}] } with ( patch("app.main.workspace.retrieve", return_value=[("app/main.py", "MAX_AGENT_TOOL_CALLS = 3")]) as search, patch("app.main.get_completion", new=AsyncMock(return_value=completion)) as model, ): with self.assertRaises(HTTPException) as error: asyncio.run( _execute_agent( AgentRequest(task=task, workspace_path=self.workspace) ) ) self.assertEqual(error.exception.status_code, 422) search.assert_called_once_with(task, Path(self.workspace).resolve()) offered = model.await_args.args[0]["tools"] self.assertNotIn("search_workspace", [item["function"]["name"] for item in offered]) def test_agent_caps_sequential_tools_and_forces_final_model_turn(self) -> None: tool_calls = [ {"choices": [{"message": {"tool_calls": [{ "id": f"calc-{index}", "function": { "name": "calculator", "arguments": json.dumps({"expression": f"2 + {index + 3}"}), }, }]}}]} for index in range(3) ] tool_calls.append({"choices": [{"message": {"content": "انتهيت بعد ثلاث خطوات."}}]}) with patch("app.main.get_completion", new=AsyncMock(side_effect=tool_calls)) as model: result = asyncio.run( _execute_agent( AgentRequest(task="استخدم الحاسبة ثلاث مرات", skill_id="code_explain") ) ) self.assertEqual(model.await_count, 4) self.assertEqual(len(result["steps"]), 3) self.assertNotIn("tools", model.await_args_list[3].args[0]) self.assertEqual(result["result"], "انتهيت بعد ثلاث خطوات.") def test_unknown_skill_is_rejected_by_request_contract(self) -> None: response = self.client.post( "/v1/agent/run", json={"task": "سؤال", "skill_id": "execute_shell"}, ) self.assertEqual(response.status_code, 422) if __name__ == "__main__": unittest.main()