Complete local hybrid search and improve agent reliability
This commit is contained in:
@@ -0,0 +1,203 @@
|
||||
"""Safe extraction of embedded text from a bounded, user-selected PDF."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import math
|
||||
import threading
|
||||
from io import BytesIO
|
||||
from typing import Any
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
from pypdf import PdfReader, apply_configuration
|
||||
from pypdf.errors import LimitReachedError, PdfReadError
|
||||
from PIL import Image
|
||||
|
||||
|
||||
MAX_PDF_PAGES = 30
|
||||
MAX_SCANNED_PAGES = 3
|
||||
MAX_RENDERED_PAGE_PIXELS = 2_000_000
|
||||
MAX_RENDERED_TOTAL_BYTES = 8_000_000
|
||||
PDF_RENDER_DPI = 144
|
||||
_PDFIUM_LOCK = threading.Lock()
|
||||
|
||||
|
||||
class PdfDocumentError(ValueError):
|
||||
def __init__(self, status_code: int, message: str) -> None:
|
||||
super().__init__(message)
|
||||
self.status_code = status_code
|
||||
|
||||
|
||||
def extract_pdf_text(
|
||||
raw: bytes,
|
||||
name: str,
|
||||
*,
|
||||
max_pages: int = MAX_PDF_PAGES,
|
||||
max_chars: int = 24_000,
|
||||
allow_empty: bool = False,
|
||||
) -> str:
|
||||
parsed = extract_pdf_pages_text(raw, name, max_pages=max_pages, max_chars=max_chars)
|
||||
sections = [
|
||||
f"--- صفحة {page['page']} ---\n{page['text']}"
|
||||
for page in parsed["pages"]
|
||||
if page["text"]
|
||||
]
|
||||
if parsed["truncated"]:
|
||||
sections.append(f"[اقتُصر النص المستخرج على {max_chars} محرف.]")
|
||||
if not sections and not allow_empty:
|
||||
raise PdfDocumentError(
|
||||
415,
|
||||
f"لا يحتوي PDF على نص قابل للاستخراج؛ قد تكون صفحاته صورًا ممسوحة: {name}.",
|
||||
)
|
||||
return "\n\n".join(sections)
|
||||
|
||||
|
||||
def extract_pdf_pages_text(
|
||||
raw: bytes,
|
||||
name: str,
|
||||
*,
|
||||
max_pages: int = MAX_PDF_PAGES,
|
||||
max_chars: int = 24_000,
|
||||
) -> dict[str, Any]:
|
||||
"""Extract bounded text per page while preserving which pages are image-only."""
|
||||
if not raw.startswith(b"%PDF-"):
|
||||
raise PdfDocumentError(415, f"ترويسة ملف PDF غير صالحة: {name}.")
|
||||
try:
|
||||
with apply_configuration(
|
||||
maximum_declared_stream_length=8_000_000,
|
||||
array_based_stream_maximum_output_length=8_000_000,
|
||||
zlib_maximum_output_length=8_000_000,
|
||||
lzw_maximum_output_length=8_000_000,
|
||||
run_length_maximum_output_length=8_000_000,
|
||||
jbig2_maximum_output_length=8_000_000,
|
||||
image_maximum_buffer_size=8_000_000,
|
||||
page_tree_maximum_entries=1_000,
|
||||
page_tree_maximum_depth=30,
|
||||
):
|
||||
reader = PdfReader(BytesIO(raw), strict=True)
|
||||
if reader.is_encrypted:
|
||||
raise PdfDocumentError(415, f"ملف PDF محمي بكلمة مرور وغير مدعوم: {name}.")
|
||||
if len(reader.pages) > max_pages:
|
||||
raise PdfDocumentError(413, f"الحد الأقصى {max_pages} صفحة لكل PDF: {name}.")
|
||||
pages: list[dict[str, Any]] = []
|
||||
extracted_length = 0
|
||||
truncated = False
|
||||
for page_number, page in enumerate(reader.pages, 1):
|
||||
text = (page.extract_text() or "").strip()
|
||||
if not text:
|
||||
pages.append({"page": page_number, "text": "", "has_text": False})
|
||||
continue
|
||||
section = f"--- صفحة {page_number} ---\n{text}"
|
||||
remaining = max_chars - extracted_length
|
||||
if remaining <= 0:
|
||||
truncated = True
|
||||
pages.append({"page": page_number, "text": "", "has_text": True})
|
||||
continue
|
||||
if len(section) > remaining:
|
||||
section = section[:remaining]
|
||||
truncated = True
|
||||
page_text = section.split("\n", 1)[1] if "\n" in section else ""
|
||||
pages.append({"page": page_number, "text": page_text, "has_text": True})
|
||||
extracted_length += len(section)
|
||||
if extracted_length >= max_chars:
|
||||
truncated = truncated or page_number < len(reader.pages)
|
||||
if truncated:
|
||||
for later_page_number in range(page_number + 1, len(reader.pages) + 1):
|
||||
pages.append(
|
||||
{"page": later_page_number, "text": "", "has_text": True}
|
||||
)
|
||||
break
|
||||
except PdfDocumentError:
|
||||
raise
|
||||
except (PdfReadError, LimitReachedError, ValueError, KeyError) as exc:
|
||||
raise PdfDocumentError(415, f"تعذر قراءة بنية PDF: {name}.") from exc
|
||||
return {
|
||||
"pages": pages,
|
||||
"total_pages": len(pages),
|
||||
"truncated": truncated,
|
||||
}
|
||||
|
||||
|
||||
def render_scanned_pdf_pages(
|
||||
raw: bytes,
|
||||
name: str,
|
||||
*,
|
||||
max_pages: int = MAX_SCANNED_PAGES,
|
||||
page_numbers: list[int] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Render selected PDF pages to bounded JPEG data for local vision/OCR."""
|
||||
try:
|
||||
# pypdfium2/PDFium is not thread-safe; serialize all native PDFium work.
|
||||
with _PDFIUM_LOCK:
|
||||
document = pdfium.PdfDocument(raw)
|
||||
try:
|
||||
total_pages = len(document)
|
||||
if total_pages > MAX_PDF_PAGES:
|
||||
raise PdfDocumentError(413, f"الحد الأقصى {MAX_PDF_PAGES} صفحة لكل PDF: {name}.")
|
||||
page_limit = max(0, min(max_pages, MAX_SCANNED_PAGES))
|
||||
if page_numbers is None:
|
||||
selected_pages = list(range(1, min(total_pages, page_limit) + 1))
|
||||
truncated = total_pages > len(selected_pages)
|
||||
else:
|
||||
if len(page_numbers) > page_limit:
|
||||
raise PdfDocumentError(
|
||||
413,
|
||||
f"الحد الأقصى {MAX_SCANNED_PAGES} صفحة ممسوحة للتحليل في الطلب الواحد.",
|
||||
)
|
||||
if (
|
||||
len(set(page_numbers)) != len(page_numbers)
|
||||
or any(number < 1 or number > total_pages for number in page_numbers)
|
||||
):
|
||||
raise PdfDocumentError(422, f"أرقام صفحات PDF المحددة غير صالحة: {name}.")
|
||||
selected_pages = page_numbers
|
||||
truncated = False
|
||||
pages: list[dict[str, Any]] = []
|
||||
total_rendered_bytes = 0
|
||||
for page_number in selected_pages:
|
||||
index = page_number - 1
|
||||
page = document[index]
|
||||
bitmap = None
|
||||
try:
|
||||
width, height = page.get_size()
|
||||
if width <= 0 or height <= 0:
|
||||
continue
|
||||
target_scale = PDF_RENDER_DPI / 72
|
||||
pixel_scale = math.sqrt(MAX_RENDERED_PAGE_PIXELS / (width * height))
|
||||
scale = min(target_scale, pixel_scale)
|
||||
bitmap = page.render(scale=scale, rev_byteorder=True, limit_image_cache=True)
|
||||
image = bitmap.to_pil().convert("RGB")
|
||||
output = BytesIO()
|
||||
try:
|
||||
image.save(output, format="JPEG", quality=82, optimize=True)
|
||||
finally:
|
||||
image.close()
|
||||
jpeg = output.getvalue()
|
||||
if len(jpeg) > 4_000_000:
|
||||
raise PdfDocumentError(413, f"حجم الصفحة الممسوحة كبير بعد التحويل: {name}.")
|
||||
total_rendered_bytes += len(jpeg)
|
||||
if total_rendered_bytes > MAX_RENDERED_TOTAL_BYTES:
|
||||
raise PdfDocumentError(413, "تجاوز مجموع الصور المحولة من PDF حد 8 ميغابايت.")
|
||||
pages.append(
|
||||
{
|
||||
"page": page_number,
|
||||
"mime_type": "image/jpeg",
|
||||
"data": base64.b64encode(jpeg).decode("ascii"),
|
||||
}
|
||||
)
|
||||
finally:
|
||||
if bitmap is not None:
|
||||
bitmap.close()
|
||||
page.close()
|
||||
if not pages:
|
||||
raise PdfDocumentError(415, f"لا توجد صفحات PDF قابلة للتحويل إلى صورة: {name}.")
|
||||
return {
|
||||
"pages": pages,
|
||||
"total_pages": total_pages,
|
||||
"truncated": truncated,
|
||||
}
|
||||
finally:
|
||||
document.close()
|
||||
except PdfDocumentError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise PdfDocumentError(415, f"تعذر تحويل صفحات PDF الممسوح إلى صور: {name}.") from exc
|
||||
Reference in New Issue
Block a user