Expand retrieval evaluation to long docs and PDF

This commit is contained in:
Hamza Ayed
2026-10-03 19:06:54 +03:00
parent cfca1ebff0
commit 332f43aaf0
7 changed files with 530 additions and 6 deletions
@@ -0,0 +1,43 @@
%PDF-1.4
%âãÏÓ
1 0 obj
<< /Type /Catalog /Pages 2 0 R >>
endobj
2 0 obj
<< /Type /Pages /Kids [3 0 R] /Count 1 >>
endobj
3 0 obj
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>
endobj
4 0 obj
<< /Length 306 >>
stream
BT
/F1 14 Tf
72 720 Td
(Archive retention policy) Tj
0 -28 Td
(A single PDF may contain at most 30 pages for local indexing.) Tj
0 -28 Td
(Digital text is extracted locally and each excerpt keeps its page number.) Tj
0 -28 Td
(Scanned pages are handled separately by the configured local OCR engine.) Tj
ET
endstream
endobj
5 0 obj
<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>
endobj
xref
0 6
0000000000 65535 f
0000000015 00000 n
0000000064 00000 n
0000000121 00000 n
0000000247 00000 n
0000000604 00000 n
trailer
<< /Size 6 /Root 1 0 R >>
startxref
674
%%EOF