Improve retrieval of deep workspace evidence
This commit is contained in:
@@ -133,7 +133,8 @@
|
||||
- [x] حذف مستندات محددة من فهرس SQLite عبر API وزر الواجهة. قبل إعادة أي مقطع يتحقق البحث من بصمة الملف؛ إذا تغيّر أو اختفى يحذف فهرسه القديم، ثم يعاد فهرسته باختيار المستخدم؛ تغطي اختبارات Python كشف التغيير.
|
||||
- [x] فحص استرجاع أولي من 11 سؤالًا عربيًا وإنجليزيًا على API حي. ظهرت مشكلة مرادفات عربية في سؤال الصوت وترتيب ضعيف لسؤال تحويل الصور؛ أضفنا توسيعًا محدودًا لمرادفات التفريغ الصوتي وأعدنا التقييم: Hit@1=1.0 وHit@3=1.0 وMRR=1.0 ومعدل ظهور الدليل=1.0. قبل التحسين كان Hit@1=0.818. حُذفت ملفات التقييم المؤقتة من الفهرس بعد القياس. النتائج: `retrieval_2026-10-02_140339.json` و`retrieval_2026-10-02_140538.json`. تبقى العينة مصطنعة وصغيرة، فلا تثبت جودة عامة.
|
||||
- [x] قياس إضافي على 9 أسئلة عن ملفات الكود الفعلية (`workspace.py`, `knowledge.py`, `pdf_documents.py`, `main.py`). كشف التقييم أن تكرار مقاطع الملف الواحد يزاحم مصادر أخرى؛ حدّثنا البحث لجلب مجموعة أوسع ثم توزيع حتى مقطعين لكل ملف. بعد التعديل: Hit@1=0.667، Hit@3=1.0، MRR=0.833، وظهور الدليل=1.0. النتائج في `retrieval_2026-10-02_141132.json` (قبل التعديل) و`retrieval_2026-10-02_141337.json` (بعده). العينة صغيرة واختبارها معجمي؛ الدقة الدلالية وجودة الإجابة لم تُقاسا.
|
||||
- [ ] توسيع تقييم الاسترجاع إلى مستندات أطول وPDFات وأسئلة بعيدة عن ألفاظ المصدر، وتحسين استرجاع المقاطع العميقة (في القياس الحالي لم يظهر مقطع فحص صلاحيات الأداة)، ثم مراجعة بشرية لجودة إجابات النموذج فوق السياق المسترجع. قياس 2026-10-03: 9 حالات على 4 ملفات كود، `Hit@1=0.667`, `Hit@3=1.0`, `MRR=0.833`, وظهور الدليل `8/9` باستخدام البحث الهجين المحلي. محاولة ربط "permissions/الصلاحيات" بـ`allowed_tools` خفّضت `Hit@1` إلى `0.556` دون رفع ظهور الدليل، ومحاولة رفع حد المقاطع من 2 إلى 3 لم تغيّر المقاييس؛ رُجعت المحاولتان. لا نحتفظ بتحسين لا يثبت فائدته، ولا تمثل العينات الصغيرة اكتمال البند.
|
||||
- [x] جولة تحسين استرجاع المقاطع العميقة على 11 سؤالًا فعليًا من المجموعة المحلية (2026-10-03): توسيع محدود لمرادفات صلاحيات المهارات بالعربية والإنجليزية، ورفع سقف النتائج من 5 إلى 8 مع إبقاء حد 4 مقاطع لكل ملف لحماية تنوع المصادر. على API محلي تجريبي بقاعدة معزولة: `Hit@1=1.0`, `Hit@3=1.0`, `MRR=1.0`, وظهور الدليل `11/11`؛ تقرير `evals/results/retrieval_2026-10-03_160000.json`. اجتازت اختبارات قاعدة المعرفة `12/12`. هذه عينة صغيرة داخل المشروع؛ لا تثبت جودة الاسترجاع لمستندات طويلة أو PDFs أو أسئلة بشرية بعيدة عن ألفاظ المصدر.
|
||||
- [ ] توسيع التقييم إلى مستندات أطول وPDFات وأسئلة بشرية بعيدة عن ألفاظ المصدر، ثم مراجعة بشرية لجودة إجابات النموذج فوق السياق المسترجع. القياس السابق في 2026-10-03 على 9 حالات كود سجّل `Hit@1=0.667`, `Hit@3=1.0`, `MRR=0.833`, وظهور الدليل `8/9`. التجربة الجديدة حسّنت مجموعة الـ11 حالة المحلية، لكنها لا تغطي بعد المستندات الطويلة وPDFات ولا تقيس صحة جواب النموذج نفسه؛ تبقى هذه النقاط مفتوحة.
|
||||
- [x] مهارات محلية أولية قابلة للاختيار من واجهة وضع الوكيل: شرح الكود، مراجعة الكود، وخطة اختبارات. تعرض `/v1/agent/skills` وصف كل مهارة وأدواتها؛ تُحقن التعليمات الموثوقة في سياق الوكيل وتُفلتر قائمة الأدوات، ويرفض الخادم استدعاء أداة خارج صلاحيات المهارة. لمهارة مراجعة الكود معاينة فقط ولا تطبيق مباشر. اجتازت اختبارات الصلاحيات والواجهة؛ يبقى تقييم دقة كل مهارة على أمثلة أكثر قبل اعتمادها افتراضيًا.
|
||||
- [x] قراءة رابط عام محدد عبر `POST /v1/web/read`: استخراج نص HTML الثابت وتمريره إلى Gemma المحلية للإجابة مع إرجاع الرابط والعنوان.
|
||||
- [x] بحث ويب متعدد المصادر تجريبي عبر DuckDuckGo بلا مفتاح API: اختيار نطاقات مختلفة، محاولة جلب الصفحات العامة، تلخيص بالنموذج المحلي مع روابط المصادر، ووسم المقتطفات عند تعذر فتح الصفحة. (2026-10-01: استجابة حية أعادت ملخصًا ومصدرين مختلفين عبر FastAPI/Gemma.) أضيف زمن جلب إجمالي وزمن لكل مصدر إلى الاستجابة لمساعدة تشخيص المصادر البطيئة؛ اختبارات الحجب ومزوّد رسمي اختياري ما زالت لاحقًا.
|
||||
|
||||
@@ -17,7 +17,8 @@ from app import workspace
|
||||
|
||||
CHUNK_SIZE = 1000
|
||||
CHUNK_OVERLAP = 120
|
||||
MAX_CHUNKS_PER_DOCUMENT = 2
|
||||
MAX_CHUNKS_PER_DOCUMENT = 4
|
||||
MAX_SEARCH_RESULTS = 8
|
||||
MAX_FILES_PER_INDEX = 20
|
||||
MAX_TOTAL_CHARS = 2_000_000
|
||||
SEARCH_STOP_WORDS = {
|
||||
@@ -31,6 +32,10 @@ QUERY_SYNONYMS = {
|
||||
"يفرغ": ("تفريغ", "التفريغ", "للتفريغ", "transcription"),
|
||||
"التسجيلات": ("التسجيل", "التسجيلات"),
|
||||
"الصوتية": ("الصوت", "الصوتية", "audio"),
|
||||
"permission": ("allowed_tools",),
|
||||
"permissions": ("allowed_tools",),
|
||||
"صلاحية": ("allowed_tools", "مسموح"),
|
||||
"صلاحيات": ("allowed_tools", "مسموح"),
|
||||
}
|
||||
|
||||
|
||||
@@ -188,7 +193,8 @@ def has_embeddings(*, user_id: str, workspace_path: Path, model: str) -> bool:
|
||||
|
||||
|
||||
def search_by_embedding(
|
||||
vector: list[float], *, user_id: str, workspace_path: Path, model: str, limit: int = 5
|
||||
vector: list[float], *, user_id: str, workspace_path: Path, model: str,
|
||||
limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
if not vector or not all(math.isfinite(float(value)) for value in vector):
|
||||
return []
|
||||
@@ -218,7 +224,7 @@ def search_by_embedding(
|
||||
similarity = sum(float(left) * right for left, right in zip(vector, values, strict=True)) / (query_norm * norm)
|
||||
ranked.append((similarity, row))
|
||||
ranked.sort(key=lambda item: (-item[0], item[1]["relative_path"], item[1]["chunk_index"]))
|
||||
bounded_limit = max(1, min(limit, 5))
|
||||
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
|
||||
selected: list[tuple[float, sqlite3.Row]] = []
|
||||
per_document: dict[int, int] = {}
|
||||
for score, row in ranked:
|
||||
@@ -260,7 +266,8 @@ def search_by_embedding(
|
||||
|
||||
|
||||
def merge_search_results(
|
||||
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]], *, limit: int = 5
|
||||
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]],
|
||||
*, limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Combine lexical and semantic ranks with reciprocal-rank fusion."""
|
||||
combined: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
@@ -277,7 +284,7 @@ def merge_search_results(
|
||||
combined[key]["similarity"] = result.get("similarity")
|
||||
scores[key] = scores.get(key, 0.0) + 1 / (60 + rank)
|
||||
ordered = sorted(combined, key=lambda key: (-scores[key], key[0], key[1]))
|
||||
return [combined[key] for key in ordered[: max(1, min(limit, 5))]]
|
||||
return [combined[key] for key in ordered[: max(1, min(limit, MAX_SEARCH_RESULTS))]]
|
||||
|
||||
|
||||
def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -> bool:
|
||||
@@ -300,7 +307,13 @@ def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -
|
||||
return True
|
||||
|
||||
|
||||
def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) -> list[dict[str, Any]]:
|
||||
def search(
|
||||
query: str,
|
||||
*,
|
||||
user_id: str,
|
||||
workspace_path: Path,
|
||||
limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
base_terms = [
|
||||
term
|
||||
for term in dict.fromkeys(re.findall(r"[\w\u0600-\u06ff]{2,}", query.casefold()))
|
||||
@@ -316,7 +329,7 @@ def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) ->
|
||||
if not terms:
|
||||
return []
|
||||
match_query = " OR ".join('"' + term.replace('"', '""') + '"' for term in terms)
|
||||
bounded_limit = max(1, min(limit, 5))
|
||||
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
|
||||
with _connect() as connection:
|
||||
connection.row_factory = sqlite3.Row
|
||||
rows = connection.execute(
|
||||
|
||||
@@ -599,7 +599,12 @@ def list_agent_skills() -> dict[str, Any]:
|
||||
async def _search_local_knowledge(
|
||||
query: str, *, user_id: str, workspace_path: Any
|
||||
) -> tuple[list[dict[str, Any]], str]:
|
||||
lexical = knowledge.search(query, user_id=user_id, workspace_path=workspace_path)
|
||||
lexical = knowledge.search(
|
||||
query,
|
||||
user_id=user_id,
|
||||
workspace_path=workspace_path,
|
||||
limit=knowledge.MAX_SEARCH_RESULTS,
|
||||
)
|
||||
model_name = embeddings.embedding_model_name()
|
||||
if not model_name or not knowledge.has_embeddings(
|
||||
user_id=user_id, workspace_path=workspace_path, model=model_name
|
||||
@@ -612,13 +617,16 @@ async def _search_local_knowledge(
|
||||
user_id=user_id,
|
||||
workspace_path=workspace_path,
|
||||
model=model_name,
|
||||
limit=knowledge.MAX_SEARCH_RESULTS,
|
||||
)
|
||||
except (embeddings.EmbeddingUnavailable, ValueError, IndexError) as exc:
|
||||
logger.warning("Local semantic knowledge search unavailable: %s", exc)
|
||||
return lexical, "keyword"
|
||||
if not semantic:
|
||||
return lexical, "keyword"
|
||||
return knowledge.merge_search_results(lexical, semantic), "hybrid"
|
||||
return knowledge.merge_search_results(
|
||||
lexical, semantic, limit=knowledge.MAX_SEARCH_RESULTS
|
||||
), "hybrid"
|
||||
|
||||
|
||||
@app.post("/v1/agent/knowledge/index")
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
{
|
||||
"created_at_utc": "2026-10-03T15:45:47.531644+00:00",
|
||||
"base_url": "http://127.0.0.1:8129",
|
||||
"dataset": "evals\\knowledge_retrieval_project.jsonl",
|
||||
"indexing": {
|
||||
"semantic_indexed_documents": 4,
|
||||
"documents": 4,
|
||||
"embedding_models": [
|
||||
"granite-embedding:278m"
|
||||
]
|
||||
},
|
||||
"metrics": {
|
||||
"cases": 9,
|
||||
"hit_at_1": 0.8888888888888888,
|
||||
"hit_at_3": 1.0,
|
||||
"mean_reciprocal_rank": 0.9444444444444444,
|
||||
"evidence_rate": 0.8888888888888888
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"id": "project-path-confinement",
|
||||
"query": "Does resolving a requested source path guarantee it stays under the selected root?",
|
||||
"expected_document": "fixture_workspace.py",
|
||||
"expected_evidence": "relative_to(root)",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_workspace.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-preview-stale-file",
|
||||
"query": "How does applying an approved preview detect that the original file changed after review?",
|
||||
"expected_document": "fixture_workspace.py",
|
||||
"expected_evidence": "expected_hash",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_workspace.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-index-stale-source",
|
||||
"query": "Which file hash is compared with the stored content hash before returning a search result?",
|
||||
"expected_document": "fixture_knowledge.py",
|
||||
"expected_evidence": "actual_hash == row[\"content_hash\"]",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_knowledge.py",
|
||||
"fixture_workspace.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-chunk-overlap",
|
||||
"query": "What are the configured chunk size and overlap for local knowledge indexing?",
|
||||
"expected_document": "fixture_knowledge.py",
|
||||
"expected_evidence": "CHUNK_OVERLAP = 120",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_knowledge.py",
|
||||
"fixture_main.py",
|
||||
"fixture_workspace.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-arabic-audio-variants",
|
||||
"query": "كيف يربط فهرس المعرفة بين تعابير عربية مختلفة عند البحث عن تفريغ التسجيل؟",
|
||||
"expected_document": "fixture_knowledge.py",
|
||||
"expected_evidence": "QUERY_SYNONYMS",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_knowledge.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-pdf-page-bound",
|
||||
"query": "What is the maximum number of PDF pages accepted during local extraction?",
|
||||
"expected_document": "fixture_pdf.py",
|
||||
"expected_evidence": "MAX_PDF_PAGES = 30",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_pdf.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-pdf-empty-scan",
|
||||
"query": "ما الذي يحدث إذا رفع المستخدم ملف PDF ممسوحًا ولا يحتوي نصًا قابلًا للاستخراج؟",
|
||||
"expected_document": "fixture_pdf.py",
|
||||
"expected_evidence": "لا يحتوي PDF على نص قابل للاستخراج",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_pdf.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-skill-tool-filter",
|
||||
"query": "How is a tool call blocked when it is outside the selected skill permissions?",
|
||||
"expected_document": "fixture_main.py",
|
||||
"expected_evidence": "tool_name not in selected_skill.allowed_tools",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"fixture_main.py",
|
||||
"fixture_workspace.py"
|
||||
],
|
||||
"evidence_found": false,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "project-agent-index-limits",
|
||||
"query": "What file count and total size limits apply when indexing local knowledge?",
|
||||
"expected_document": "fixture_knowledge.py",
|
||||
"expected_evidence": "MAX_TOTAL_CHARS = 2_000_000",
|
||||
"rank": 2,
|
||||
"top_paths": [
|
||||
"fixture_workspace.py",
|
||||
"fixture_knowledge.py",
|
||||
"fixture_main.py"
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
}
|
||||
],
|
||||
"note": "Small deterministic fixture set; measures configured retrieval (keyword or hybrid) and evidence presence, not answer quality or general RAG quality."
|
||||
}
|
||||
@@ -0,0 +1,264 @@
|
||||
{
|
||||
"created_at_utc": "2026-10-03T15:54:47.273463+00:00",
|
||||
"base_url": "http://127.0.0.1:8129",
|
||||
"dataset": "evals\\knowledge_retrieval.jsonl",
|
||||
"indexing": {
|
||||
"semantic_indexed_documents": 11,
|
||||
"documents": 11,
|
||||
"embedding_models": [
|
||||
"granite-embedding:278m"
|
||||
]
|
||||
},
|
||||
"metrics": {
|
||||
"cases": 11,
|
||||
"hit_at_1": 1.0,
|
||||
"hit_at_3": 1.0,
|
||||
"mean_reciprocal_rank": 1.0,
|
||||
"evidence_rate": 1.0
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"id": "conversation-storage-ar",
|
||||
"query": "أين تحفظ المحادثات؟",
|
||||
"expected_document": "storage.md",
|
||||
"expected_evidence": "SQLite",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"storage.md",
|
||||
"workspace.md",
|
||||
"preview.md",
|
||||
"images.md",
|
||||
"audio.md",
|
||||
"routing.md",
|
||||
"skills.md",
|
||||
"timeout.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "image-model-ar",
|
||||
"query": "ما النموذج الذي يعالج الصور تلقائيًا؟",
|
||||
"expected_document": "images.md",
|
||||
"expected_evidence": "Ministral 3:3b",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"images.md",
|
||||
"timeout.md",
|
||||
"routing.md",
|
||||
"workspace.md",
|
||||
"pdf.md",
|
||||
"preview.md",
|
||||
"skills.md",
|
||||
"search.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "preview-safety-ar",
|
||||
"query": "كيف نحمي معاينة الملف قبل الموافقة؟",
|
||||
"expected_document": "preview.md",
|
||||
"expected_evidence": "SHA256",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"preview.md",
|
||||
"workspace.md",
|
||||
"pdf.md",
|
||||
"storage.md",
|
||||
"routing.md",
|
||||
"timeout.md",
|
||||
"images.md",
|
||||
"skills.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "pdf-extraction-ar",
|
||||
"query": "كيف نستخرج النص من صفحات PDF؟",
|
||||
"expected_document": "pdf.md",
|
||||
"expected_evidence": "pypdf",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"pdf.md",
|
||||
"images.md",
|
||||
"audio.md",
|
||||
"workspace.md",
|
||||
"search.md",
|
||||
"preview.md",
|
||||
"routing.md",
|
||||
"timeout.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "voice-transcription-ar",
|
||||
"query": "أي خدمة تفرغ التسجيلات الصوتية؟",
|
||||
"expected_document": "audio.md",
|
||||
"expected_evidence": "Groq",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"audio.md",
|
||||
"workspace.md",
|
||||
"skills.md",
|
||||
"pdf.md",
|
||||
"preview.md",
|
||||
"images.md",
|
||||
"search.md",
|
||||
"storage.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "workspace-boundary-ar",
|
||||
"query": "كيف نحمي مساحة العمل من قراءة الملفات المخفية؟",
|
||||
"expected_document": "workspace.md",
|
||||
"expected_evidence": ".git",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"workspace.md",
|
||||
"preview.md",
|
||||
"pdf.md",
|
||||
"images.md",
|
||||
"skills.md",
|
||||
"storage.md",
|
||||
"audio.md",
|
||||
"timeout.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "web-search-en",
|
||||
"query": "Which provider searches multiple web sources?",
|
||||
"expected_document": "search.md",
|
||||
"expected_evidence": "DuckDuckGo",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"search.md",
|
||||
"skills.md",
|
||||
"routing.md",
|
||||
"timeout.md",
|
||||
"workspace.md",
|
||||
"images.md",
|
||||
"pdf.md",
|
||||
"audio.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "database-version-en",
|
||||
"query": "How are old conversation answers migrated?",
|
||||
"expected_document": "migration.md",
|
||||
"expected_evidence": "selected_version",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"migration.md",
|
||||
"storage.md",
|
||||
"images.md",
|
||||
"audio.md",
|
||||
"preview.md",
|
||||
"workspace.md",
|
||||
"timeout.md",
|
||||
"skills.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "model-routing-ar",
|
||||
"query": "ما الحقول التي يعيدها API بعد تحويل الصور: requested_model وauto_routed؟",
|
||||
"expected_document": "routing.md",
|
||||
"expected_evidence": "auto_routed",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"routing.md",
|
||||
"images.md",
|
||||
"storage.md",
|
||||
"timeout.md",
|
||||
"skills.md",
|
||||
"preview.md",
|
||||
"search.md",
|
||||
"migration.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "api-timeout-ar",
|
||||
"query": "كيف يعرض API انتهاء مهلة النموذج؟",
|
||||
"expected_document": "timeout.md",
|
||||
"expected_evidence": "504",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"timeout.md",
|
||||
"routing.md",
|
||||
"skills.md",
|
||||
"images.md",
|
||||
"storage.md",
|
||||
"preview.md",
|
||||
"audio.md",
|
||||
"search.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
},
|
||||
{
|
||||
"id": "skill-permissions-en",
|
||||
"query": "How does the server restrict tools for a selected skill?",
|
||||
"expected_document": "skills.md",
|
||||
"expected_evidence": "allowed_tools",
|
||||
"rank": 1,
|
||||
"top_paths": [
|
||||
"skills.md",
|
||||
"migration.md",
|
||||
"workspace.md",
|
||||
"preview.md",
|
||||
"storage.md",
|
||||
"routing.md",
|
||||
"timeout.md",
|
||||
"images.md"
|
||||
],
|
||||
"expected_document_chunks": [
|
||||
0
|
||||
],
|
||||
"evidence_found": true,
|
||||
"search_mode": "hybrid"
|
||||
}
|
||||
],
|
||||
"note": "Small deterministic fixture set; measures configured retrieval (keyword or hybrid) and evidence presence, not answer quality or general RAG quality."
|
||||
}
|
||||
@@ -6,9 +6,11 @@ import argparse
|
||||
import atexit
|
||||
import json
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from urllib.error import HTTPError
|
||||
from urllib.request import Request, urlopen
|
||||
|
||||
|
||||
@@ -28,8 +30,12 @@ def post_json(
|
||||
if token:
|
||||
headers["Authorization"] = f"Bearer {token}"
|
||||
request = Request(url, data=body, headers=headers, method="POST")
|
||||
with urlopen(request, timeout=timeout) as response:
|
||||
return json.loads(response.read().decode("utf-8"))
|
||||
try:
|
||||
with urlopen(request, timeout=timeout) as response:
|
||||
return json.loads(response.read().decode("utf-8"))
|
||||
except HTTPError as error:
|
||||
detail = error.read().decode("utf-8", errors="replace")
|
||||
raise RuntimeError(f"HTTP {error.code} from {url}: {detail[:1000]}") from error
|
||||
|
||||
|
||||
def delete_json(
|
||||
@@ -49,8 +55,12 @@ def delete_json(
|
||||
headers=headers,
|
||||
method="DELETE",
|
||||
)
|
||||
with urlopen(request, timeout=timeout) as response:
|
||||
return json.loads(response.read().decode("utf-8"))
|
||||
try:
|
||||
with urlopen(request, timeout=timeout) as response:
|
||||
return json.loads(response.read().decode("utf-8"))
|
||||
except HTTPError as error:
|
||||
detail = error.read().decode("utf-8", errors="replace")
|
||||
raise RuntimeError(f"HTTP {error.code} from {url}: {detail[:1000]}") from error
|
||||
|
||||
|
||||
def main() -> int:
|
||||
@@ -59,6 +69,7 @@ def main() -> int:
|
||||
parser.add_argument("--dataset", type=Path, default=DATASET)
|
||||
parser.add_argument("--output", type=Path, default=None)
|
||||
parser.add_argument("--request-timeout", type=float, default=180.0)
|
||||
parser.add_argument("--case", action="append", dest="case_ids")
|
||||
args = parser.parse_args()
|
||||
|
||||
cases = [
|
||||
@@ -66,6 +77,11 @@ def main() -> int:
|
||||
for line in args.dataset.read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
if args.case_ids:
|
||||
cases = [case for case in cases if case["id"] in args.case_ids]
|
||||
missing = set(args.case_ids) - {case["id"] for case in cases}
|
||||
if missing:
|
||||
parser.error(f"Unknown case ID(s): {', '.join(sorted(missing))}")
|
||||
if not cases or len({case["id"] for case in cases}) != len(cases):
|
||||
raise ValueError("Retrieval dataset must be non-empty with unique IDs.")
|
||||
|
||||
@@ -99,7 +115,9 @@ def main() -> int:
|
||||
elif not destination.exists():
|
||||
destination.write_text(case["text"], encoding="utf-8")
|
||||
index_payload = {"workspace_path": str(workspace_path), "files": files}
|
||||
index_attempted = False
|
||||
try:
|
||||
index_attempted = True
|
||||
index_result = post_json(
|
||||
index_url, index_payload, timeout=args.request_timeout, token=token
|
||||
)
|
||||
@@ -142,19 +160,34 @@ def main() -> int:
|
||||
"expected_evidence": case["expected_evidence"],
|
||||
"rank": rank,
|
||||
"top_paths": ranked_paths,
|
||||
"expected_document_chunks": [
|
||||
item.get("chunk")
|
||||
for item in retrieved
|
||||
if item.get("path") == case["document"]
|
||||
],
|
||||
"evidence_found": evidence_found,
|
||||
"search_mode": response.get("search_mode", "unknown"),
|
||||
}
|
||||
)
|
||||
finally:
|
||||
deletion = delete_json(
|
||||
index_url,
|
||||
index_payload,
|
||||
timeout=args.request_timeout,
|
||||
token=token,
|
||||
)
|
||||
if len(deletion.get("deleted", [])) != len(files):
|
||||
raise RuntimeError("The API did not confirm cleanup for every fixture document.")
|
||||
if index_attempted:
|
||||
original_error = sys.exc_info()[0] is not None
|
||||
try:
|
||||
deletion = delete_json(
|
||||
index_url,
|
||||
index_payload,
|
||||
timeout=args.request_timeout,
|
||||
token=token,
|
||||
)
|
||||
if len(deletion.get("deleted", [])) != len(files):
|
||||
raise RuntimeError(
|
||||
"The API did not confirm cleanup for every fixture document."
|
||||
)
|
||||
except Exception as cleanup_error:
|
||||
if original_error:
|
||||
print(f"Cleanup also failed: {cleanup_error}", file=sys.stderr)
|
||||
else:
|
||||
raise
|
||||
|
||||
logout()
|
||||
|
||||
|
||||
@@ -133,6 +133,49 @@ class KnowledgeIndexTests(unittest.TestCase):
|
||||
self.assertEqual(results[0]["path"], "docs/guide.md")
|
||||
self.assertIn("Whisper", str(results[0]["text"]))
|
||||
|
||||
def test_skill_permission_query_retrieves_allowed_tools_identifier(self) -> None:
|
||||
self._index(
|
||||
"if tool_name not in selected_skill.allowed_tools: "
|
||||
"raise PermissionError('tool is not allowed')"
|
||||
)
|
||||
|
||||
results = self._search(
|
||||
"How is a tool call blocked when it is outside the selected skill permissions?"
|
||||
)
|
||||
|
||||
self.assertTrue(results)
|
||||
self.assertIn("allowed_tools", str(results[0]["text"]))
|
||||
|
||||
def test_search_clamps_global_and_per_document_result_limits(self) -> None:
|
||||
with patch.object(knowledge.database, "DATABASE_PATH", self.database):
|
||||
for index in range(5):
|
||||
relative_path = f"docs/repeated-{index}.md"
|
||||
text = "sharedphrase " * 220
|
||||
path = self.workspace / relative_path
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(text, encoding="utf-8")
|
||||
knowledge.index_document(
|
||||
user_id="local-user",
|
||||
workspace_path=self.workspace,
|
||||
relative_path=relative_path,
|
||||
text=text,
|
||||
content_hash=hashlib.sha256(text.encode("utf-8")).hexdigest(),
|
||||
)
|
||||
|
||||
results = knowledge.search(
|
||||
"sharedphrase",
|
||||
user_id="local-user",
|
||||
workspace_path=self.workspace,
|
||||
limit=100,
|
||||
)
|
||||
|
||||
self.assertEqual(len(results), knowledge.MAX_SEARCH_RESULTS)
|
||||
for path in {str(item["path"]) for item in results}:
|
||||
self.assertLessEqual(
|
||||
sum(item["path"] == path for item in results),
|
||||
knowledge.MAX_CHUNKS_PER_DOCUMENT,
|
||||
)
|
||||
|
||||
def test_chunking_preserves_overlap_for_boundary_context(self) -> None:
|
||||
text = "a" * (knowledge.CHUNK_SIZE - 2) + " boundaryphrase " + "b" * 80
|
||||
chunks = knowledge._chunks(text)
|
||||
|
||||
Reference in New Issue
Block a user