fix multi-query agent workspace searches

This commit is contained in:
Hamza Ayed
2026-10-04 17:42:57 +03:00
parent b3dc6154e5
commit a62a831e90
6 changed files with 253 additions and 17 deletions
+82 -4
View File
@@ -476,6 +476,52 @@ def list_knowledge_files(root: Path) -> list[Path]:
return files
def _matching_excerpt(text: str, terms: set[str], max_chars: int) -> str:
lines = text.splitlines()
if not lines:
return text[:max_chars]
scored_lines: list[tuple[int, int]] = []
for index, line in enumerate(lines):
words = set(re.findall(r"[\w\u0600-\u06ff]+", line.casefold()))
matched = terms & words
if matched:
score = sum(2 if len(term) >= 8 else 1 for term in matched)
scored_lines.append((score, index))
if not scored_lines:
return text[:max_chars]
windows: list[tuple[int, int, int]] = []
for score, index in scored_lines:
start = max(0, index - 1)
end = min(len(lines), index + 2)
windows.append((score, start, end))
windows.sort(key=lambda item: (-item[0], item[1]))
selected: set[int] = set()
for _score, start, end in windows:
if any(line_number in selected for line_number in range(start, end)):
continue
candidate = set(range(start, end))
rendered = [f"{line_number + 1}: {lines[line_number]}" for line_number in sorted(selected | candidate)]
if len("\n".join(rendered)) > max_chars:
continue
selected.update(candidate)
if not selected:
_score, index = scored_lines[0]
line = lines[index]
matching_terms = [term for term in terms if term in set(re.findall(r"[\w\u0600-\u06ff]+", line.casefold()))]
positions = [line.casefold().find(term) for term in matching_terms if term]
if len(line) > max_chars and positions:
start = max(0, positions[0] - max_chars // 3)
line = ("…" if start else "") + line[start : start + max_chars - 32] + ("…" if start + max_chars - 32 < len(lines[index]) else "")
return f"{index + 1}: {line}"[:max_chars]
return "\n…\n".join(
f"{line_number + 1}: {lines[line_number]}"
for line_number in sorted(selected)
)[:max_chars]
def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
terms = {
term.casefold()
@@ -484,8 +530,26 @@ def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
"the", "and", "for", "with", "this", "that", "من", "على", "في", "عن",
"كيف", "شو", "ما", "ماذا", "هذا", "هذه", "التي", "الذي", "اشرح", "دور",
"ملف", "ملفات", "اذكر", "المستخدمة", "المستخدم", "المشروع", "التفسير",
"استخدم", "أداة", "البحث", "مساحة", "العمل", "مرات", "منفصلة", "للعثور",
"بعد", "كل", "ثم", "افحص", "النتيجة", "لخص", "اشرح", "اذكر", "علاقة",
}
}
term_expansions = {
"مهلة": {"timeout", "timeouts"},
"عميل": {"client"},
"مسار": {"path", "uri", "route"},
"بث": {"stream"},
"تدفق": {"stream"},
"واجهة": {"api", "endpoint"},
}
terms.update(
alias
for term in tuple(terms)
for alias in term_expansions.get(term, set())
)
code_search = bool(
re.search(r"[A-Za-z][A-Za-z0-9]*_[A-Za-z0-9_]+|run/stream|\b(?:api|http|flutter|timeout)\b", task, re.IGNORECASE)
)
mentioned_paths = {
match.replace("\\", "/").casefold()
for match in re.findall(
@@ -508,11 +572,25 @@ def retrieve(task: str, root: Path, limit: int = 3) -> list[tuple[str, str]]:
relative = path.relative_to(root).as_posix()
if relative.casefold() in mentioned_paths:
score += 100_000
if code_search:
if path.suffix.casefold() in {".py", ".dart", ".js", ".ts", ".tsx", ".jsx"}:
score += 30
if "flutter" in terms and path.suffix.casefold() == ".dart":
score += 60
if path.suffix.casefold() in {".md", ".txt"}:
score = max(1, score - 30)
if any(part.casefold() in {"test", "tests"} for part in path.parts) and not any(
term in terms for term in {"test", "tests", "اختبار", "اختبارات"}
):
score = max(1, score - 30)
if score:
# If the user named a source file explicitly, include a larger bounded
# excerpt so the agent can explain the code rather than just locate it.
excerpt_limit = 12_000 if relative.casefold() in mentioned_paths else 1_800
ranked.append((score, relative, text[:excerpt_limit]))
# Explicit paths need enough context for explanation; search results
# should show the matching lines instead of an unrelated file prefix.
if relative.casefold() in mentioned_paths:
excerpt = text[:12_000]
else:
excerpt = _matching_excerpt(text, terms, 1_800)
ranked.append((score, relative, excerpt))
ranked.sort(key=lambda item: (-item[0], item[1]))
explicit_matches = [
item for item in ranked if item[1].casefold() in mentioned_paths