Improve retrieval of deep workspace evidence

This commit is contained in:
Hamza Ayed
2026-10-03 18:56:36 +03:00
parent 652546d462
commit cfca1ebff0
7 changed files with 526 additions and 22 deletions
+20 -7
View File
@@ -17,7 +17,8 @@ from app import workspace
CHUNK_SIZE = 1000
CHUNK_OVERLAP = 120
MAX_CHUNKS_PER_DOCUMENT = 2
MAX_CHUNKS_PER_DOCUMENT = 4
MAX_SEARCH_RESULTS = 8
MAX_FILES_PER_INDEX = 20
MAX_TOTAL_CHARS = 2_000_000
SEARCH_STOP_WORDS = {
@@ -31,6 +32,10 @@ QUERY_SYNONYMS = {
"يفرغ": ("تفريغ", "التفريغ", "للتفريغ", "transcription"),
"التسجيلات": ("التسجيل", "التسجيلات"),
"الصوتية": ("الصوت", "الصوتية", "audio"),
"permission": ("allowed_tools",),
"permissions": ("allowed_tools",),
"صلاحية": ("allowed_tools", "مسموح"),
"صلاحيات": ("allowed_tools", "مسموح"),
}
@@ -188,7 +193,8 @@ def has_embeddings(*, user_id: str, workspace_path: Path, model: str) -> bool:
def search_by_embedding(
vector: list[float], *, user_id: str, workspace_path: Path, model: str, limit: int = 5
vector: list[float], *, user_id: str, workspace_path: Path, model: str,
limit: int = MAX_SEARCH_RESULTS,
) -> list[dict[str, Any]]:
if not vector or not all(math.isfinite(float(value)) for value in vector):
return []
@@ -218,7 +224,7 @@ def search_by_embedding(
similarity = sum(float(left) * right for left, right in zip(vector, values, strict=True)) / (query_norm * norm)
ranked.append((similarity, row))
ranked.sort(key=lambda item: (-item[0], item[1]["relative_path"], item[1]["chunk_index"]))
bounded_limit = max(1, min(limit, 5))
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
selected: list[tuple[float, sqlite3.Row]] = []
per_document: dict[int, int] = {}
for score, row in ranked:
@@ -260,7 +266,8 @@ def search_by_embedding(
def merge_search_results(
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]], *, limit: int = 5
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]],
*, limit: int = MAX_SEARCH_RESULTS,
) -> list[dict[str, Any]]:
"""Combine lexical and semantic ranks with reciprocal-rank fusion."""
combined: dict[tuple[str, int], dict[str, Any]] = {}
@@ -277,7 +284,7 @@ def merge_search_results(
combined[key]["similarity"] = result.get("similarity")
scores[key] = scores.get(key, 0.0) + 1 / (60 + rank)
ordered = sorted(combined, key=lambda key: (-scores[key], key[0], key[1]))
return [combined[key] for key in ordered[: max(1, min(limit, 5))]]
return [combined[key] for key in ordered[: max(1, min(limit, MAX_SEARCH_RESULTS))]]
def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -> bool:
@@ -300,7 +307,13 @@ def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -
return True
def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) -> list[dict[str, Any]]:
def search(
query: str,
*,
user_id: str,
workspace_path: Path,
limit: int = MAX_SEARCH_RESULTS,
) -> list[dict[str, Any]]:
base_terms = [
term
for term in dict.fromkeys(re.findall(r"[\w\u0600-\u06ff]{2,}", query.casefold()))
@@ -316,7 +329,7 @@ def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) ->
if not terms:
return []
match_query = " OR ".join('"' + term.replace('"', '""') + '"' for term in terms)
bounded_limit = max(1, min(limit, 5))
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
with _connect() as connection:
connection.row_factory = sqlite3.Row
rows = connection.execute(
+10 -2
View File
@@ -599,7 +599,12 @@ def list_agent_skills() -> dict[str, Any]:
async def _search_local_knowledge(
query: str, *, user_id: str, workspace_path: Any
) -> tuple[list[dict[str, Any]], str]:
lexical = knowledge.search(query, user_id=user_id, workspace_path=workspace_path)
lexical = knowledge.search(
query,
user_id=user_id,
workspace_path=workspace_path,
limit=knowledge.MAX_SEARCH_RESULTS,
)
model_name = embeddings.embedding_model_name()
if not model_name or not knowledge.has_embeddings(
user_id=user_id, workspace_path=workspace_path, model=model_name
@@ -612,13 +617,16 @@ async def _search_local_knowledge(
user_id=user_id,
workspace_path=workspace_path,
model=model_name,
limit=knowledge.MAX_SEARCH_RESULTS,
)
except (embeddings.EmbeddingUnavailable, ValueError, IndexError) as exc:
logger.warning("Local semantic knowledge search unavailable: %s", exc)
return lexical, "keyword"
if not semantic:
return lexical, "keyword"
return knowledge.merge_search_results(lexical, semantic), "hybrid"
return knowledge.merge_search_results(
lexical, semantic, limit=knowledge.MAX_SEARCH_RESULTS
), "hybrid"
@app.post("/v1/agent/knowledge/index")