diff --git a/backend/app/Controllers/CurriculumController.php b/backend/app/Controllers/CurriculumController.php index 8f6e893..8219a62 100644 --- a/backend/app/Controllers/CurriculumController.php +++ b/backend/app/Controllers/CurriculumController.php @@ -10,7 +10,7 @@ use App\Services\CurriculumExtractorService; class CurriculumController { /** - * Upload and Deep-Extract Real Ministry PDF Document + * Upload and Queue Real Ministry PDF Document for Deep Extraction */ public function uploadPdf(Request $request, Response $response): void { @@ -25,52 +25,63 @@ class CurriculumController $file = $_FILES['pdf_file']; $origName = $file['name']; $tmpPath = $file['tmp_name']; + $taskId = uniqid('task_'); - // Save original uploaded PDF in storage/uploads/curriculum - $uploadsDir = __DIR__ . '/../../storage/uploads/curriculum'; - if (!is_dir($uploadsDir)) { - mkdir($uploadsDir, 0777, true); + // Setup processing directory + $processingDir = __DIR__ . '/../../storage/curriculum/processing'; + if (!is_dir($processingDir)) { + mkdir($processingDir, 0777, true); } - $savedPdfPath = $uploadsDir . '/' . time() . '_' . preg_replace('/[^A-Za-z0-9_\-\.]/u', '_', $origName); + + $savedPdfPath = $processingDir . '/' . $taskId . '.pdf'; move_uploaded_file($tmpPath, $savedPdfPath); - // Perform Deep Multi-Engine Extraction & Structuring - $parsedStructure = CurriculumExtractorService::extractAndStructure($origName, $savedPdfPath); + // Initialize Task State + $taskState = [ + 'task_id' => $taskId, + 'status' => 'queued', + 'progress' => 0, + 'message' => 'تم استلام الملف وجاري تحويله للمعالجة المعمقة...', + 'orig_name' => $origName + ]; + file_put_contents($processingDir . '/' . $taskId . '.json', json_encode($taskState, JSON_UNESCAPED_UNICODE)); - // Merge into live tree and save to disk - $updatedTree = CurriculumService::mergeExtractedCurriculum($parsedStructure); - - $firstLessonFile = ''; - $firstLessonMd = ''; - $firstLessonTitle = ''; - $firstLessonOutcomes = []; - $breadcrumb = ''; - - if (!empty($parsedStructure['units'][0]['lessons'][0])) { - $firstLes = $parsedStructure['units'][0]['lessons'][0]; - $firstLessonFile = "{$parsedStructure['grade_key']}/{$parsedStructure['subject_key']}/{$parsedStructure['semester_key']}/{$parsedStructure['units'][0]['unit_key']}/{$firstLes['lesson_id']}.md"; - $firstLessonMd = CurriculumService::getLessonMarkdown($firstLessonFile); - $firstLessonTitle = $firstLes['title']; - $firstLessonOutcomes = $firstLes['outcomes'] ?? []; - $breadcrumb = "{$parsedStructure['grade_name']} ⟵ {$parsedStructure['subject_name']} ⟵ {$parsedStructure['units'][0]['unit_name']}"; - } - - $serverStorageDir = realpath(__DIR__ . '/../../storage/curriculum'); + // Launch Background Worker + $scriptPath = realpath(__DIR__ . '/../../scripts/curriculum_worker.php'); + $logPath = $processingDir . '/' . $taskId . '.log'; + $cmd = "php " . escapeshellarg($scriptPath) . " " . escapeshellarg($taskId) . " > " . escapeshellarg($logPath) . " 2>&1 &"; + exec($cmd); $response->json([ - 'status' => 'success', - 'message' => "تم تفريغ وفهرسة المنهاج [{$origName}] بنجاح في قاعدة المعرفة!", - 'extracted_data' => $parsedStructure, - 'tree' => $updatedTree, - 'active_file' => $firstLessonFile, - 'active_md' => $firstLessonMd, - 'active_title' => $firstLessonTitle, - 'active_outcomes' => $firstLessonOutcomes, - 'active_breadcrumb' => $breadcrumb, - 'server_storage_dir' => $serverStorageDir + 'status' => 'processing', + 'task_id' => $taskId, + 'message' => 'تم استلام الكتاب، وجاري المعالجة المعمقة بواسطة الذكاء الاصطناعي. يرجى الانتظار...' ]); } + /** + * Poll Upload Status + */ + public function getUploadStatus(Request $request, Response $response): void + { + $taskId = $request->getQueryParams()['task_id'] ?? ''; + if (empty($taskId)) { + $response->status(400)->json(['status' => 'error', 'message' => 'رقم المهمة مفقود']); + return; + } + + $processingDir = __DIR__ . '/../../storage/curriculum/processing'; + $stateFile = $processingDir . '/' . $taskId . '.json'; + + if (!file_exists($stateFile)) { + $response->status(404)->json(['status' => 'error', 'message' => 'المهمة غير موجودة']); + return; + } + + $state = json_decode(file_get_contents($stateFile), true); + $response->json($state); + } + /** * Get Complete Live Tree */ diff --git a/backend/scripts/curriculum_worker.php b/backend/scripts/curriculum_worker.php new file mode 100644 index 0000000..8869f44 --- /dev/null +++ b/backend/scripts/curriculum_worker.php @@ -0,0 +1,222 @@ +/dev/null"; +$rawText = @shell_exec($cmd); + +if (empty($rawText) || mb_strlen(trim($rawText)) < 100) { + updateState($stateFile, 'error', 0, 'فشل استخراج النصوص من الـ PDF. قد يكون الملف عبارة عن صور فقط (Scanned).'); + exit; +} + +updateState($stateFile, 'analyzing', 40, 'جاري تحليل بنية الوحدات والدروس من النصوص المستخرجة...'); + +Env::load(__DIR__ . '/../.env'); +$geminiKey = Env::get('GEMINI_API_KEY') ?: getenv('GEMINI_API_KEY'); + +$parsedStructure = null; + +if (!empty($geminiKey)) { + updateState($stateFile, 'analyzing', 50, 'يتم الآن تحليل المنهج عبر الذكاء الاصطناعي (Gemini)...'); + + // We send chunks or a large chunk to Gemini + $prompt = "أنت خبير مناهج تعليمية. قم بتحليل هذا النص المستخرج من كتاب دراسي بعنوان '{$origName}' واستخراج الفهرس والوحدات والدروس بدقة. +النص: +" . mb_substr($rawText, 0, 15000) . " + +المطلوب إرجاع JSON صالح فقط بالصيغة التالية (يجب أن يحتوي على النص الحقيقي للدرس وليس نصاً وهمياً): +{ + \"grade_name\": \"اسم الصف\", + \"grade_key\": \"grade_X\", + \"subject_name\": \"اسم المادة\", + \"subject_key\": \"subject_key\", + \"semester_name\": \"الفصل الدراسي الأول\", + \"semester_key\": \"semester_1\", + \"units\": [ + { + \"unit_key\": \"unit_1\", + \"unit_name\": \"اسم الوحدة الحقيقي\", + \"lessons\": [ + { + \"lesson_id\": \"lesson_1\", + \"title\": \"اسم الدرس الحقيقي\", + \"outcomes\": [\"الهدف 1\"], + \"markdown_content\": \"# عنوان الدرس\\n\\nالنص الحقيقي الكامل المستخرج للدرس...\" + } + ] + } + ] +}"; + + $url = "https://generativelanguage.googleapis.com/v1beta/models/gemini-2.0-flash:generateContent?key=" . $geminiKey; + $payload = [ + 'contents' => [['parts' => [['text' => $prompt]]]], + 'generationConfig' => ['responseMimeType' => 'application/json', 'temperature' => 0.2] + ]; + + $ch = curl_init($url); + curl_setopt_array($ch, [ + CURLOPT_POST => true, + CURLOPT_POSTFIELDS => json_encode($payload), + CURLOPT_HTTPHEADER => ['Content-Type: application/json'], + CURLOPT_RETURNTRANSFER => true, + CURLOPT_TIMEOUT => 60 + ]); + $res = curl_exec($ch); + $code = curl_getinfo($ch, CURLINFO_HTTP_CODE); + curl_close($ch); + + if ($code === 200 && !empty($res)) { + $json = json_decode($res, true); + $text = $json['candidates'][0]['content']['parts'][0]['text'] ?? ''; + $parsedStructure = json_decode($text, true); + } +} + +// Fallback to Heuristic Regex Parser if Gemini failed or no key +if (empty($parsedStructure) || empty($parsedStructure['units'])) { + updateState($stateFile, 'analyzing', 60, 'يتم تحليل البنية باستخدام محرك التحليل الديناميكي العميق للنصوص...'); + + // Build an authentic structure based on ACTUAL text chunks + $cleanTitle = trim(preg_replace('/\.[^.]+$/u', '', $origName)); + + // Simple heuristic: split text by "Unit" or "الوحدة" or "Module" + $chunks = preg_split('/(Unit\s+\d+|Module\s+\d+|الوحدة\s+(?:الأولى|الثانية|الثالثة|الرابعة|الخامسة|\d+))/iu', $rawText, -1, PREG_SPLIT_DELIM_CAPTURE); + + $units = []; + $unitCounter = 1; + + // First chunk is preamble + for ($i = 1; $i < count($chunks); $i += 2) { + $unitTitleRaw = trim($chunks[$i]); + $unitContentRaw = trim($chunks[$i+1] ?? ''); + + $lines = explode("\n", $unitContentRaw); + $unitNameAddition = trim($lines[0] ?? ''); + $unitName = $unitTitleRaw . ($unitNameAddition ? ': ' . mb_substr($unitNameAddition, 0, 40) : ''); + + $lessonChunks = preg_split('/(Lesson\s+\d+|الدرس\s+(?:الأول|الثاني|الثالث|الرابع|\d+))/iu', $unitContentRaw, -1, PREG_SPLIT_DELIM_CAPTURE); + + $lessons = []; + $lessonCounter = 1; + + if (count($lessonChunks) > 1) { + for ($j = 1; $j < count($lessonChunks); $j += 2) { + $lessonTitleRaw = trim($lessonChunks[$j]); + $lessonContentRaw = trim($lessonChunks[$j+1] ?? ''); + $lLines = explode("\n", $lessonContentRaw); + $lNameAddition = trim($lLines[0] ?? ''); + + $lessons[] = [ + 'lesson_id' => 'lesson_' . $lessonCounter, + 'title' => $lessonTitleRaw . ' ' . mb_substr($lNameAddition, 0, 40), + 'outcomes' => ['تم استخراج النص تلقائياً من الكتاب'], + 'markdown_content' => "# " . $lessonTitleRaw . "\n\n" . mb_substr($lessonContentRaw, 0, 5000) + ]; + $lessonCounter++; + } + } else { + // No explicit lessons found, make chunks of 1500 chars + $textLen = mb_strlen($unitContentRaw); + $chunkSize = 2500; + for ($c = 0; $c < $textLen; $c += $chunkSize) { + $lessons[] = [ + 'lesson_id' => 'part_' . $lessonCounter, + 'title' => 'Part ' . $lessonCounter, + 'outcomes' => ['نص مستخرج آلياً'], + 'markdown_content' => "# Part " . $lessonCounter . "\n\n" . mb_substr($unitContentRaw, $c, $chunkSize) + ]; + $lessonCounter++; + } + } + + $units[] = [ + 'unit_key' => 'unit_' . $unitCounter, + 'unit_name' => $unitName, + 'lessons' => $lessons + ]; + $unitCounter++; + } + + // If no units matched the regex at all, just split the whole book into 4 parts + if (empty($units)) { + $textLen = mb_strlen($rawText); + $chunkSize = 3000; + $lessons = []; + for ($c = 0, $idx=1; $c < $textLen && $idx <= 10; $c += $chunkSize, $idx++) { + $lessons[] = [ + 'lesson_id' => 'section_' . $idx, + 'title' => 'Section ' . $idx, + 'outcomes' => ['نص حقيقي مستخرج آلياً'], + 'markdown_content' => "# Section " . $idx . "\n\n" . mb_substr($rawText, $c, $chunkSize) + ]; + } + $units[] = [ + 'unit_key' => 'unit_1', + 'unit_name' => 'Textbook Content', + 'lessons' => $lessons + ]; + } + + $parsedStructure = [ + 'grade_name' => 'الصف الأساسي / مستخرج', + 'grade_key' => 'grade_extracted', + 'subject_name' => $cleanTitle, + 'subject_key' => 'subject_' . substr(md5($cleanTitle), 0, 8), + 'semester_name' => 'الفصل المستخرج', + 'semester_key' => 'semester_1', + 'units' => $units + ]; +} + +updateState($stateFile, 'generating', 80, 'جاري بناء هيكل المنهاج وملفات المارك داون على السيرفر...'); + +// Save to disk +$updatedTree = CurriculumService::mergeExtractedCurriculum($parsedStructure); + +$firstLessonFile = ''; +if (!empty($parsedStructure['units'][0]['lessons'][0])) { + $firstLes = $parsedStructure['units'][0]['lessons'][0]; + $firstLessonFile = "{$parsedStructure['grade_key']}/{$parsedStructure['subject_key']}/{$parsedStructure['semester_key']}/{$parsedStructure['units'][0]['unit_key']}/{$firstLes['lesson_id']}.md"; +} + +updateState($stateFile, 'completed', 100, 'تم الاستخراج والفهرسة بنجاح!', [ + 'active_file' => $firstLessonFile, + 'tree' => $updatedTree, + 'extracted_data' => $parsedStructure +]); + diff --git a/books/أنظمة المعادلات الرياضيات للصف العاشر أ. إياد النجار.pdf b/books/أنظمة المعادلات الرياضيات للصف العاشر أ. إياد النجار.pdf new file mode 100644 index 0000000..ecf1729 Binary files /dev/null and b/books/أنظمة المعادلات الرياضيات للصف العاشر أ. إياد النجار.pdf differ diff --git a/books/كتاب التمارين لمادة الرياضيات الصف العاشر الفصل الأول.pdf b/books/كتاب التمارين لمادة الرياضيات الصف العاشر الفصل الأول.pdf new file mode 100644 index 0000000..0225664 Binary files /dev/null and b/books/كتاب التمارين لمادة الرياضيات الصف العاشر الفصل الأول.pdf differ diff --git a/books/كتاب الطالب لمادة الرياضيات الصف العاشر الفصل الأول.pdf b/books/كتاب الطالب لمادة الرياضيات الصف العاشر الفصل الأول.pdf new file mode 100644 index 0000000..00e7005 Binary files /dev/null and b/books/كتاب الطالب لمادة الرياضيات الصف العاشر الفصل الأول.pdf differ diff --git a/books/كتاب الطالب لمادة اللغة الإنجليزية الصف العاشر الفصل الأول.pdf b/books/كتاب الطالب لمادة اللغة الإنجليزية الصف العاشر الفصل الأول.pdf new file mode 100644 index 0000000..f10c3a6 Binary files /dev/null and b/books/كتاب الطالب لمادة اللغة الإنجليزية الصف العاشر الفصل الأول.pdf differ diff --git a/books/ملخص الوحدة الثانية الاشتقاق الرياضيات الصف العاشر أ. رعد الخمايسة.pdf b/books/ملخص الوحدة الثانية الاشتقاق الرياضيات الصف العاشر أ. رعد الخمايسة.pdf new file mode 100644 index 0000000..78c66c6 Binary files /dev/null and b/books/ملخص الوحدة الثانية الاشتقاق الرياضيات الصف العاشر أ. رعد الخمايسة.pdf differ diff --git a/replace_ui.py b/replace_ui.py new file mode 100644 index 0000000..c5319ef --- /dev/null +++ b/replace_ui.py @@ -0,0 +1,95 @@ +import re + +with open("backend/app/Views/CurriculumStudio.php", "r") as f: + content = f.read() + +new_script = """ async function handleRealPdfUpload(input) { + if (!input.files || !input.files[0]) return; + const file = input.files[0]; + + const btn = document.getElementById('btn_upload_pdf'); + const desc = document.getElementById('upload_status_desc'); + btn.disabled = true; + btn.innerHTML = '⏳ جاري الرفع...'; + desc.textContent = `جاري رفع ملف [${file.name}] وبدء عملية استخراج المناهج...`; + + const formData = new FormData(); + formData.append('pdf_file', file); + + try { + const res = await fetch('/api/curriculum/upload-pdf', { + method: 'POST', + body: formData + }); + const data = await res.json(); + + if (res.ok && data.status === 'processing') { + // Start Polling + pollTaskStatus(data.task_id, file.name); + } else if (res.ok && data.status === 'success') { + // Fallback if it returned sync + processSuccessData(data, file.name); + } else { + alert('⚠️ ' + (data.message || 'فشلت معالجة الملف')); + resetUploadBtn(); + } + } catch (err) { + console.error('PDF Upload error:', err); + alert('حدث خطأ في الاتصال بالخادم أثناء رفع الملف.'); + resetUploadBtn(); + } + } + + async function pollTaskStatus(taskId, fileName) { + const btn = document.getElementById('btn_upload_pdf'); + const desc = document.getElementById('upload_status_desc'); + + try { + const res = await fetch(`/api/curriculum/upload-status?task_id=${taskId}`); + if (!res.ok) throw new Error("Network response was not ok"); + const state = await res.json(); + + if (state.status === 'completed') { + desc.textContent = `تم الانتهاء بنجاح! تم استخراج المنهج بالكامل لـ [${fileName}]`; + btn.innerHTML = '✅ اكتمل التفريغ'; + setTimeout(() => { resetUploadBtn(); }, 3000); + + if (state.tree) { + window.CURRICULUM_TREE = state.tree; + currentFilePath = state.active_file || ''; + renderCurriculumTree(state.tree); + if (state.active_file) { + selectLesson(state.active_file, 'الدرس المستخرج', 'المسار التعليمي', []); + fetchLessonContent(state.active_file); + } + } else { + // Reload tree completely + loadCurriculumTree(); + } + } else if (state.status === 'error') { + alert('⚠️ فشل في معالجة الملف: ' + state.message); + resetUploadBtn(); + } else { + // Still processing + btn.innerHTML = `⏳ ${state.progress}% | ${state.status}`; + desc.textContent = state.message || "جاري المعالجة بالذكاء الاصطناعي... قد يستغرق دقيقة أو أكثر"; + setTimeout(() => pollTaskStatus(taskId, fileName), 2500); // poll every 2.5 seconds + } + } catch(e) { + setTimeout(() => pollTaskStatus(taskId, fileName), 3000); + } + } + + function resetUploadBtn() { + const btn = document.getElementById('btn_upload_pdf'); + const desc = document.getElementById('upload_status_desc'); + btn.disabled = false; + btn.innerHTML = '📄 رفع واستخراج المنهاج (PDF)'; + document.getElementById('curriculum_pdf_input').value = ''; + } +""" + +content = re.sub(r'async function handleRealPdfUpload\(input\) \{.*?(?=async function selectLesson|function renderCurriculumTree)', new_script + '\n ', content, flags=re.DOTALL) + +with open("backend/app/Views/CurriculumStudio.php", "w") as f: + f.write(content)