204 lines
8.6 KiB
Python
204 lines
8.6 KiB
Python
"""Safe extraction of embedded text from a bounded, user-selected PDF."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import math
|
|
import threading
|
|
from io import BytesIO
|
|
from typing import Any
|
|
|
|
import pypdfium2 as pdfium
|
|
from pypdf import PdfReader, apply_configuration
|
|
from pypdf.errors import LimitReachedError, PdfReadError
|
|
from PIL import Image
|
|
|
|
|
|
MAX_PDF_PAGES = 30
|
|
MAX_SCANNED_PAGES = 3
|
|
MAX_RENDERED_PAGE_PIXELS = 2_000_000
|
|
MAX_RENDERED_TOTAL_BYTES = 8_000_000
|
|
PDF_RENDER_DPI = 144
|
|
_PDFIUM_LOCK = threading.Lock()
|
|
|
|
|
|
class PdfDocumentError(ValueError):
|
|
def __init__(self, status_code: int, message: str) -> None:
|
|
super().__init__(message)
|
|
self.status_code = status_code
|
|
|
|
|
|
def extract_pdf_text(
|
|
raw: bytes,
|
|
name: str,
|
|
*,
|
|
max_pages: int = MAX_PDF_PAGES,
|
|
max_chars: int = 24_000,
|
|
allow_empty: bool = False,
|
|
) -> str:
|
|
parsed = extract_pdf_pages_text(raw, name, max_pages=max_pages, max_chars=max_chars)
|
|
sections = [
|
|
f"--- صفحة {page['page']} ---\n{page['text']}"
|
|
for page in parsed["pages"]
|
|
if page["text"]
|
|
]
|
|
if parsed["truncated"]:
|
|
sections.append(f"[اقتُصر النص المستخرج على {max_chars} محرف.]")
|
|
if not sections and not allow_empty:
|
|
raise PdfDocumentError(
|
|
415,
|
|
f"لا يحتوي PDF على نص قابل للاستخراج؛ قد تكون صفحاته صورًا ممسوحة: {name}.",
|
|
)
|
|
return "\n\n".join(sections)
|
|
|
|
|
|
def extract_pdf_pages_text(
|
|
raw: bytes,
|
|
name: str,
|
|
*,
|
|
max_pages: int = MAX_PDF_PAGES,
|
|
max_chars: int = 24_000,
|
|
) -> dict[str, Any]:
|
|
"""Extract bounded text per page while preserving which pages are image-only."""
|
|
if not raw.startswith(b"%PDF-"):
|
|
raise PdfDocumentError(415, f"ترويسة ملف PDF غير صالحة: {name}.")
|
|
try:
|
|
with apply_configuration(
|
|
maximum_declared_stream_length=8_000_000,
|
|
array_based_stream_maximum_output_length=8_000_000,
|
|
zlib_maximum_output_length=8_000_000,
|
|
lzw_maximum_output_length=8_000_000,
|
|
run_length_maximum_output_length=8_000_000,
|
|
jbig2_maximum_output_length=8_000_000,
|
|
image_maximum_buffer_size=8_000_000,
|
|
page_tree_maximum_entries=1_000,
|
|
page_tree_maximum_depth=30,
|
|
):
|
|
reader = PdfReader(BytesIO(raw), strict=True)
|
|
if reader.is_encrypted:
|
|
raise PdfDocumentError(415, f"ملف PDF محمي بكلمة مرور وغير مدعوم: {name}.")
|
|
if len(reader.pages) > max_pages:
|
|
raise PdfDocumentError(413, f"الحد الأقصى {max_pages} صفحة لكل PDF: {name}.")
|
|
pages: list[dict[str, Any]] = []
|
|
extracted_length = 0
|
|
truncated = False
|
|
for page_number, page in enumerate(reader.pages, 1):
|
|
text = (page.extract_text() or "").strip()
|
|
if not text:
|
|
pages.append({"page": page_number, "text": "", "has_text": False})
|
|
continue
|
|
section = f"--- صفحة {page_number} ---\n{text}"
|
|
remaining = max_chars - extracted_length
|
|
if remaining <= 0:
|
|
truncated = True
|
|
pages.append({"page": page_number, "text": "", "has_text": True})
|
|
continue
|
|
if len(section) > remaining:
|
|
section = section[:remaining]
|
|
truncated = True
|
|
page_text = section.split("\n", 1)[1] if "\n" in section else ""
|
|
pages.append({"page": page_number, "text": page_text, "has_text": True})
|
|
extracted_length += len(section)
|
|
if extracted_length >= max_chars:
|
|
truncated = truncated or page_number < len(reader.pages)
|
|
if truncated:
|
|
for later_page_number in range(page_number + 1, len(reader.pages) + 1):
|
|
pages.append(
|
|
{"page": later_page_number, "text": "", "has_text": True}
|
|
)
|
|
break
|
|
except PdfDocumentError:
|
|
raise
|
|
except (PdfReadError, LimitReachedError, ValueError, KeyError) as exc:
|
|
raise PdfDocumentError(415, f"تعذر قراءة بنية PDF: {name}.") from exc
|
|
return {
|
|
"pages": pages,
|
|
"total_pages": len(pages),
|
|
"truncated": truncated,
|
|
}
|
|
|
|
|
|
def render_scanned_pdf_pages(
|
|
raw: bytes,
|
|
name: str,
|
|
*,
|
|
max_pages: int = MAX_SCANNED_PAGES,
|
|
page_numbers: list[int] | None = None,
|
|
) -> dict[str, Any]:
|
|
"""Render selected PDF pages to bounded JPEG data for local vision/OCR."""
|
|
try:
|
|
# pypdfium2/PDFium is not thread-safe; serialize all native PDFium work.
|
|
with _PDFIUM_LOCK:
|
|
document = pdfium.PdfDocument(raw)
|
|
try:
|
|
total_pages = len(document)
|
|
if total_pages > MAX_PDF_PAGES:
|
|
raise PdfDocumentError(413, f"الحد الأقصى {MAX_PDF_PAGES} صفحة لكل PDF: {name}.")
|
|
page_limit = max(0, min(max_pages, MAX_SCANNED_PAGES))
|
|
if page_numbers is None:
|
|
selected_pages = list(range(1, min(total_pages, page_limit) + 1))
|
|
truncated = total_pages > len(selected_pages)
|
|
else:
|
|
if len(page_numbers) > page_limit:
|
|
raise PdfDocumentError(
|
|
413,
|
|
f"الحد الأقصى {MAX_SCANNED_PAGES} صفحة ممسوحة للتحليل في الطلب الواحد.",
|
|
)
|
|
if (
|
|
len(set(page_numbers)) != len(page_numbers)
|
|
or any(number < 1 or number > total_pages for number in page_numbers)
|
|
):
|
|
raise PdfDocumentError(422, f"أرقام صفحات PDF المحددة غير صالحة: {name}.")
|
|
selected_pages = page_numbers
|
|
truncated = False
|
|
pages: list[dict[str, Any]] = []
|
|
total_rendered_bytes = 0
|
|
for page_number in selected_pages:
|
|
index = page_number - 1
|
|
page = document[index]
|
|
bitmap = None
|
|
try:
|
|
width, height = page.get_size()
|
|
if width <= 0 or height <= 0:
|
|
continue
|
|
target_scale = PDF_RENDER_DPI / 72
|
|
pixel_scale = math.sqrt(MAX_RENDERED_PAGE_PIXELS / (width * height))
|
|
scale = min(target_scale, pixel_scale)
|
|
bitmap = page.render(scale=scale, rev_byteorder=True, limit_image_cache=True)
|
|
image = bitmap.to_pil().convert("RGB")
|
|
output = BytesIO()
|
|
try:
|
|
image.save(output, format="JPEG", quality=82, optimize=True)
|
|
finally:
|
|
image.close()
|
|
jpeg = output.getvalue()
|
|
if len(jpeg) > 4_000_000:
|
|
raise PdfDocumentError(413, f"حجم الصفحة الممسوحة كبير بعد التحويل: {name}.")
|
|
total_rendered_bytes += len(jpeg)
|
|
if total_rendered_bytes > MAX_RENDERED_TOTAL_BYTES:
|
|
raise PdfDocumentError(413, "تجاوز مجموع الصور المحولة من PDF حد 8 ميغابايت.")
|
|
pages.append(
|
|
{
|
|
"page": page_number,
|
|
"mime_type": "image/jpeg",
|
|
"data": base64.b64encode(jpeg).decode("ascii"),
|
|
}
|
|
)
|
|
finally:
|
|
if bitmap is not None:
|
|
bitmap.close()
|
|
page.close()
|
|
if not pages:
|
|
raise PdfDocumentError(415, f"لا توجد صفحات PDF قابلة للتحويل إلى صورة: {name}.")
|
|
return {
|
|
"pages": pages,
|
|
"total_pages": total_pages,
|
|
"truncated": truncated,
|
|
}
|
|
finally:
|
|
document.close()
|
|
except PdfDocumentError:
|
|
raise
|
|
except Exception as exc:
|
|
raise PdfDocumentError(415, f"تعذر تحويل صفحات PDF الممسوح إلى صور: {name}.") from exc
|