Improve retrieval of deep workspace evidence
This commit is contained in:
@@ -17,7 +17,8 @@ from app import workspace
|
||||
|
||||
CHUNK_SIZE = 1000
|
||||
CHUNK_OVERLAP = 120
|
||||
MAX_CHUNKS_PER_DOCUMENT = 2
|
||||
MAX_CHUNKS_PER_DOCUMENT = 4
|
||||
MAX_SEARCH_RESULTS = 8
|
||||
MAX_FILES_PER_INDEX = 20
|
||||
MAX_TOTAL_CHARS = 2_000_000
|
||||
SEARCH_STOP_WORDS = {
|
||||
@@ -31,6 +32,10 @@ QUERY_SYNONYMS = {
|
||||
"يفرغ": ("تفريغ", "التفريغ", "للتفريغ", "transcription"),
|
||||
"التسجيلات": ("التسجيل", "التسجيلات"),
|
||||
"الصوتية": ("الصوت", "الصوتية", "audio"),
|
||||
"permission": ("allowed_tools",),
|
||||
"permissions": ("allowed_tools",),
|
||||
"صلاحية": ("allowed_tools", "مسموح"),
|
||||
"صلاحيات": ("allowed_tools", "مسموح"),
|
||||
}
|
||||
|
||||
|
||||
@@ -188,7 +193,8 @@ def has_embeddings(*, user_id: str, workspace_path: Path, model: str) -> bool:
|
||||
|
||||
|
||||
def search_by_embedding(
|
||||
vector: list[float], *, user_id: str, workspace_path: Path, model: str, limit: int = 5
|
||||
vector: list[float], *, user_id: str, workspace_path: Path, model: str,
|
||||
limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
if not vector or not all(math.isfinite(float(value)) for value in vector):
|
||||
return []
|
||||
@@ -218,7 +224,7 @@ def search_by_embedding(
|
||||
similarity = sum(float(left) * right for left, right in zip(vector, values, strict=True)) / (query_norm * norm)
|
||||
ranked.append((similarity, row))
|
||||
ranked.sort(key=lambda item: (-item[0], item[1]["relative_path"], item[1]["chunk_index"]))
|
||||
bounded_limit = max(1, min(limit, 5))
|
||||
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
|
||||
selected: list[tuple[float, sqlite3.Row]] = []
|
||||
per_document: dict[int, int] = {}
|
||||
for score, row in ranked:
|
||||
@@ -260,7 +266,8 @@ def search_by_embedding(
|
||||
|
||||
|
||||
def merge_search_results(
|
||||
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]], *, limit: int = 5
|
||||
lexical: list[dict[str, Any]], semantic: list[dict[str, Any]],
|
||||
*, limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Combine lexical and semantic ranks with reciprocal-rank fusion."""
|
||||
combined: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
@@ -277,7 +284,7 @@ def merge_search_results(
|
||||
combined[key]["similarity"] = result.get("similarity")
|
||||
scores[key] = scores.get(key, 0.0) + 1 / (60 + rank)
|
||||
ordered = sorted(combined, key=lambda key: (-scores[key], key[0], key[1]))
|
||||
return [combined[key] for key in ordered[: max(1, min(limit, 5))]]
|
||||
return [combined[key] for key in ordered[: max(1, min(limit, MAX_SEARCH_RESULTS))]]
|
||||
|
||||
|
||||
def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -> bool:
|
||||
@@ -300,7 +307,13 @@ def delete_document(*, user_id: str, workspace_path: Path, relative_path: str) -
|
||||
return True
|
||||
|
||||
|
||||
def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) -> list[dict[str, Any]]:
|
||||
def search(
|
||||
query: str,
|
||||
*,
|
||||
user_id: str,
|
||||
workspace_path: Path,
|
||||
limit: int = MAX_SEARCH_RESULTS,
|
||||
) -> list[dict[str, Any]]:
|
||||
base_terms = [
|
||||
term
|
||||
for term in dict.fromkeys(re.findall(r"[\w\u0600-\u06ff]{2,}", query.casefold()))
|
||||
@@ -316,7 +329,7 @@ def search(query: str, *, user_id: str, workspace_path: Path, limit: int = 5) ->
|
||||
if not terms:
|
||||
return []
|
||||
match_query = " OR ".join('"' + term.replace('"', '""') + '"' for term in terms)
|
||||
bounded_limit = max(1, min(limit, 5))
|
||||
bounded_limit = max(1, min(limit, MAX_SEARCH_RESULTS))
|
||||
with _connect() as connection:
|
||||
connection.row_factory = sqlite3.Row
|
||||
rows = connection.execute(
|
||||
|
||||
@@ -599,7 +599,12 @@ def list_agent_skills() -> dict[str, Any]:
|
||||
async def _search_local_knowledge(
|
||||
query: str, *, user_id: str, workspace_path: Any
|
||||
) -> tuple[list[dict[str, Any]], str]:
|
||||
lexical = knowledge.search(query, user_id=user_id, workspace_path=workspace_path)
|
||||
lexical = knowledge.search(
|
||||
query,
|
||||
user_id=user_id,
|
||||
workspace_path=workspace_path,
|
||||
limit=knowledge.MAX_SEARCH_RESULTS,
|
||||
)
|
||||
model_name = embeddings.embedding_model_name()
|
||||
if not model_name or not knowledge.has_embeddings(
|
||||
user_id=user_id, workspace_path=workspace_path, model=model_name
|
||||
@@ -612,13 +617,16 @@ async def _search_local_knowledge(
|
||||
user_id=user_id,
|
||||
workspace_path=workspace_path,
|
||||
model=model_name,
|
||||
limit=knowledge.MAX_SEARCH_RESULTS,
|
||||
)
|
||||
except (embeddings.EmbeddingUnavailable, ValueError, IndexError) as exc:
|
||||
logger.warning("Local semantic knowledge search unavailable: %s", exc)
|
||||
return lexical, "keyword"
|
||||
if not semantic:
|
||||
return lexical, "keyword"
|
||||
return knowledge.merge_search_results(lexical, semantic), "hybrid"
|
||||
return knowledge.merge_search_results(
|
||||
lexical, semantic, limit=knowledge.MAX_SEARCH_RESULTS
|
||||
), "hybrid"
|
||||
|
||||
|
||||
@app.post("/v1/agent/knowledge/index")
|
||||
|
||||
Reference in New Issue
Block a user