fix: keep agent streams alive during long tasks

This commit is contained in:
Hamza Ayed
2026-10-04 13:41:22 +03:00
parent ad221e24aa
commit d1010b79d7
11 changed files with 334 additions and 37 deletions
@@ -0,0 +1,37 @@
{
"created_at_utc": "2026-10-04T09:42:15.933573+00:00",
"base_url": "http://127.0.0.1:8000",
"dataset": "evals\\knowledge_retrieval_extended.jsonl",
"indexing": {
"semantic_indexed_documents": 1,
"documents": 1,
"embedding_models": [
"granite-embedding:278m"
]
},
"metrics": {
"cases": 1,
"hit_at_1": 1.0,
"hit_at_3": 1.0,
"mean_reciprocal_rank": 1.0,
"evidence_rate": 1.0
},
"results": [
{
"id": "pdf-archive-page-limit-en",
"query": "What is the maximum number of pages accepted when indexing one PDF?",
"expected_document": "archive_retention_policy.pdf",
"expected_evidence": "30 pages",
"rank": 1,
"top_paths": [
"archive_retention_policy.pdf"
],
"expected_document_chunks": [
0
],
"evidence_found": true,
"search_mode": "hybrid"
}
],
"note": "Small deterministic fixture set. Retrieval metrics cover source ranking and evidence presence, not general RAG quality. Optional agent answers are retained for human review and are not auto-scored."
}
@@ -0,0 +1,37 @@
{
"created_at_utc": "2026-10-04T09:45:22.724212+00:00",
"base_url": "http://127.0.0.1:8000",
"dataset": "evals\\knowledge_retrieval_extended.jsonl",
"indexing": {
"semantic_indexed_documents": 1,
"documents": 1,
"embedding_models": [
"granite-embedding:278m"
]
},
"metrics": {
"cases": 1,
"hit_at_1": 1.0,
"hit_at_3": 1.0,
"mean_reciprocal_rank": 1.0,
"evidence_rate": 1.0
},
"results": [
{
"id": "scanned-pdf-ocr-deadline-en",
"query": "What calendar date ends the archive restoration approval window?",
"expected_document": "scanned_approval_notice.pdf",
"expected_evidence": "17 October 2026",
"rank": 1,
"top_paths": [
"scanned_approval_notice.pdf"
],
"expected_document_chunks": [
0
],
"evidence_found": true,
"search_mode": "hybrid"
}
],
"note": "Small deterministic fixture set. Retrieval metrics cover source ranking and evidence presence, not general RAG quality. Optional agent answers are retained for human review and are not auto-scored."
}
@@ -0,0 +1,37 @@
{
"created_at_utc": "2026-10-04T09:19:36.348536+00:00",
"base_url": "http://127.0.0.1:8000",
"dataset": "evals\\knowledge_retrieval_extended.jsonl",
"indexing": {
"semantic_indexed_documents": 1,
"documents": 1,
"embedding_models": [
"granite-embedding:278m"
]
},
"metrics": {
"cases": 1,
"hit_at_1": 1.0,
"hit_at_3": 1.0,
"mean_reciprocal_rank": 1.0,
"evidence_rate": 1.0
},
"results": [
{
"id": "conversation-storage-ar",
"query": "أين تحفظ المحادثات؟",
"expected_document": "storage.md",
"expected_evidence": "SQLite",
"rank": 1,
"top_paths": [
"storage.md"
],
"expected_document_chunks": [
0
],
"evidence_found": true,
"search_mode": "hybrid"
}
],
"note": "Small deterministic fixture set. Retrieval metrics cover source ranking and evidence presence, not general RAG quality. Optional agent answers are retained for human review and are not auto-scored."
}