168 lines
6.2 KiB
Python
168 lines
6.2 KiB
Python
"""Optional, local-only Arabic/English OCR for images and scanned documents."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import re
|
|
import threading
|
|
import time
|
|
from io import BytesIO
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from PIL import Image, UnidentifiedImageError
|
|
|
|
|
|
MAX_OCR_PIXELS = 2_000_000
|
|
MAX_OCR_RESULTS = 120
|
|
MAX_OCR_TEXT_CHARS = 12_000
|
|
_OCR_LOCK = threading.Lock()
|
|
_READER: Any | None = None
|
|
_INITIALIZATION_FAILED = False
|
|
_ARABIC_TEXT = re.compile(r"[\u0600-\u06ff]")
|
|
|
|
|
|
class LocalOCRError(RuntimeError):
|
|
"""OCR is not installed, its weights are unavailable, or the image is invalid."""
|
|
|
|
|
|
def _sort_reading_order(detections: list[Any]) -> list[Any]:
|
|
"""Group detected text into rows, then sort Arabic rows right-to-left."""
|
|
located = []
|
|
heights = []
|
|
for detection in detections:
|
|
try:
|
|
box, text, confidence = detection
|
|
points = [(float(point[0]), float(point[1])) for point in box]
|
|
xs = [point[0] for point in points]
|
|
ys = [point[1] for point in points]
|
|
center_x = (min(xs) + max(xs)) / 2
|
|
center_y = (min(ys) + max(ys)) / 2
|
|
box_height = max(1.0, max(ys) - min(ys))
|
|
heights.append(box_height)
|
|
located.append((center_y, center_x, box, str(text), float(confidence)))
|
|
except (TypeError, ValueError, IndexError, KeyError):
|
|
continue
|
|
if not located:
|
|
return []
|
|
|
|
heights.sort()
|
|
band = max(12.0, heights[len(heights) // 2] * 0.7)
|
|
rows: list[list[tuple[float, float, Any, str, float]]] = []
|
|
for item in sorted(located, key=lambda value: value[0]):
|
|
if not rows:
|
|
rows.append([item])
|
|
continue
|
|
current_y = sum(part[0] for part in rows[-1]) / len(rows[-1])
|
|
if item[0] - current_y <= band:
|
|
rows[-1].append(item)
|
|
else:
|
|
rows.append([item])
|
|
|
|
ordered = []
|
|
for row in rows:
|
|
rtl = any(_ARABIC_TEXT.search(item[3]) for item in row)
|
|
ordered.extend(sorted(row, key=lambda item: item[1], reverse=rtl))
|
|
return [(box, text, confidence) for _, _, box, text, confidence in ordered]
|
|
|
|
|
|
def _get_reader() -> Any:
|
|
global _READER, _INITIALIZATION_FAILED
|
|
if _INITIALIZATION_FAILED:
|
|
raise LocalOCRError("محرك OCR المحلي غير متاح.")
|
|
if _READER is not None:
|
|
return _READER
|
|
|
|
with _OCR_LOCK:
|
|
if _INITIALIZATION_FAILED:
|
|
raise LocalOCRError("محرك OCR المحلي غير متاح.")
|
|
if _READER is not None:
|
|
return _READER
|
|
try:
|
|
import easyocr
|
|
|
|
configured_dir = os.getenv("LOCAL_OCR_MODEL_DIR", "").strip()
|
|
if configured_dir:
|
|
model_dir = Path(configured_dir).expanduser()
|
|
else:
|
|
local_app_data = os.getenv("LOCALAPPDATA")
|
|
base_dir = Path(local_app_data) if local_app_data else Path.home() / ".local" / "share"
|
|
model_dir = base_dir / "SovereignAI" / "models" / "easyocr"
|
|
model_dir.mkdir(parents=True, exist_ok=True)
|
|
user_network_dir = model_dir / "user_network"
|
|
user_network_dir.mkdir(parents=True, exist_ok=True)
|
|
allow_download = os.getenv("LOCAL_OCR_ALLOW_DOWNLOAD", "true").strip().lower() in {
|
|
"1", "true", "yes", "on"
|
|
}
|
|
_READER = easyocr.Reader(
|
|
["ar", "en"],
|
|
gpu=False,
|
|
model_storage_directory=str(model_dir),
|
|
user_network_directory=str(user_network_dir),
|
|
download_enabled=allow_download,
|
|
verbose=False,
|
|
)
|
|
return _READER
|
|
except Exception as exc:
|
|
_INITIALIZATION_FAILED = True
|
|
raise LocalOCRError("تعذر تهيئة OCR المحلي؛ تحقق من تثبيت المتطلبات ووجود أوزان النموذج.") from exc
|
|
|
|
|
|
def recognize_image_text(raw: bytes, *, label: str = "الصورة") -> dict[str, Any]:
|
|
"""Read bounded Arabic/English text in memory without writing the upload to disk."""
|
|
if not raw:
|
|
raise LocalOCRError("ملف الصورة فارغ.")
|
|
try:
|
|
with Image.open(BytesIO(raw)) as source:
|
|
if source.width <= 0 or source.height <= 0:
|
|
raise LocalOCRError("أبعاد الصورة غير صالحة.")
|
|
if source.width * source.height > MAX_OCR_PIXELS:
|
|
raise LocalOCRError(f"تجاوزت {label} حد OCR البالغ {MAX_OCR_PIXELS} بكسل.")
|
|
source.load()
|
|
rgb_image = source.convert("RGB")
|
|
except LocalOCRError:
|
|
raise
|
|
except (UnidentifiedImageError, OSError, ValueError) as exc:
|
|
raise LocalOCRError("تعذر فك الصورة لإجراء OCR المحلي.") from exc
|
|
|
|
try:
|
|
import numpy as np
|
|
|
|
reader = _get_reader()
|
|
started = time.perf_counter()
|
|
# EasyOCR/PyTorch inference is serialized: one image at a time on this CPU.
|
|
with _OCR_LOCK:
|
|
detections = reader.readtext(
|
|
np.asarray(rgb_image),
|
|
detail=1,
|
|
paragraph=False,
|
|
batch_size=1,
|
|
workers=0,
|
|
)
|
|
detections = _sort_reading_order(detections)
|
|
lines = [
|
|
{
|
|
"text": str(text)[:1000],
|
|
"confidence": round(float(confidence), 3),
|
|
"box": [[round(float(point[0]), 1), round(float(point[1]), 1)] for point in box],
|
|
}
|
|
for box, text, confidence in detections[:MAX_OCR_RESULTS]
|
|
if str(text).strip()
|
|
]
|
|
combined_text = "\n".join(item["text"] for item in lines)[:MAX_OCR_TEXT_CHARS]
|
|
return {
|
|
"engine": "easyocr-local-ar-en",
|
|
"text": combined_text,
|
|
"lines": lines,
|
|
"average_confidence": round(
|
|
sum(item["confidence"] for item in lines) / len(lines), 3
|
|
) if lines else None,
|
|
"elapsed_seconds": round(time.perf_counter() - started, 2),
|
|
}
|
|
except LocalOCRError:
|
|
raise
|
|
except Exception as exc:
|
|
raise LocalOCRError("فشل OCR المحلي أثناء تحليل الصورة.") from exc
|
|
finally:
|
|
rgb_image.close()
|