Expand retrieval evaluation to long docs and PDF
This commit is contained in:
@@ -70,6 +70,11 @@ def main() -> int:
|
||||
parser.add_argument("--output", type=Path, default=None)
|
||||
parser.add_argument("--request-timeout", type=float, default=180.0)
|
||||
parser.add_argument("--case", action="append", dest="case_ids")
|
||||
parser.add_argument(
|
||||
"--agent-answers",
|
||||
action="store_true",
|
||||
help="Also ask the local agent to answer each selected case from indexed knowledge.",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
cases = [
|
||||
@@ -152,8 +157,7 @@ def main() -> int:
|
||||
for item in retrieved
|
||||
if item.get("path") == case["document"]
|
||||
)
|
||||
results.append(
|
||||
{
|
||||
result = {
|
||||
"id": case["id"],
|
||||
"query": case["query"],
|
||||
"expected_document": case["document"],
|
||||
@@ -168,7 +172,28 @@ def main() -> int:
|
||||
"evidence_found": evidence_found,
|
||||
"search_mode": response.get("search_mode", "unknown"),
|
||||
}
|
||||
)
|
||||
if args.agent_answers:
|
||||
agent_response = post_json(
|
||||
base_url + "/v1/agent/run",
|
||||
{
|
||||
"workspace_path": str(workspace_path),
|
||||
"task": (
|
||||
"Search the indexed knowledge base and answer the user's question "
|
||||
"using only retrieved evidence. State when the evidence is insufficient. "
|
||||
f"Question: {case['query']}"
|
||||
),
|
||||
},
|
||||
timeout=args.request_timeout,
|
||||
token=token,
|
||||
)
|
||||
result["agent_answer"] = agent_response.get(
|
||||
"result", agent_response.get("answer", "")
|
||||
)
|
||||
result["agent_tool"] = agent_response.get("tool")
|
||||
result["agent_files"] = agent_response.get("files", [])
|
||||
result["agent_steps"] = agent_response.get("steps", [])
|
||||
result["human_rating"] = None
|
||||
results.append(result)
|
||||
finally:
|
||||
if index_attempted:
|
||||
original_error = sys.exc_info()[0] is not None
|
||||
@@ -208,7 +233,10 @@ def main() -> int:
|
||||
"indexing": index_summary,
|
||||
"metrics": metrics,
|
||||
"results": results,
|
||||
"note": "Small deterministic fixture set; measures configured retrieval (keyword or hybrid) and evidence presence, not answer quality or general RAG quality.",
|
||||
"note": (
|
||||
"Small deterministic fixture set. Retrieval metrics cover source ranking and evidence presence, "
|
||||
"not general RAG quality. Optional agent answers are retained for human review and are not auto-scored."
|
||||
),
|
||||
}
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
Reference in New Issue
Block a user