diff --git a/routers/documents.py b/routers/documents.py index 811dde4..2a60b28 100644 --- a/routers/documents.py +++ b/routers/documents.py @@ -39,7 +39,7 @@ async def document_queue(request: Request, status: str = "", uploaded: int = 0): SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review, SUM(CASE WHEN status='pending' THEN 1 ELSE 0 END) AS processing, SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors - FROM documents""" + FROM documents WHERE status != 'staged'""" ).fetchone() return templates.TemplateResponse( diff --git a/routers/review.py b/routers/review.py index c8c8ca4..04d2b6d 100644 --- a/routers/review.py +++ b/routers/review.py @@ -42,6 +42,47 @@ def _get_document(doc_id: int) -> dict | None: else: doc["person"] = {} + # For multi-page PDFs (page 2+), inherit person info & doc fields from page 1 + doc["inherited_from_page1"] = False + if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1: + page1 = conn.execute( + """SELECT * FROM documents + WHERE pdf_group_id=? AND page_number=1""", + (doc["pdf_group_id"],), + ).fetchone() + if page1: + page1 = dict(page1) + # Inherit person data if current page has no name + current_first = (doc["person"].get("first_name") or "").strip() + if not current_first: + if page1.get("person_id"): + person = conn.execute( + "SELECT * FROM persons WHERE id=?", (page1["person_id"],) + ).fetchone() + if person: + doc["person"] = dict(person) + doc["inherited_from_page1"] = True + doc["page1_person_id"] = page1["person_id"] + elif page1.get("raw_extraction_json"): + try: + p1_extracted = json.loads(page1["raw_extraction_json"]) + p1_person = p1_extracted.get("person", {}) + if (p1_person.get("first_name") or "").strip(): + doc["person"] = p1_person + doc["inherited_from_page1"] = True + except Exception: + pass + + # Inherit document-level fields if missing + inherit_fields = [ + "request_number", "request_date", "search_scope", + "request_purpose", "data_valid_until", "registry_office", + "applicant_name_raw", + ] + for field in inherit_fields: + if not (doc.get(field) or "").strip() and (page1.get(field) or "").strip(): + doc[field] = page1[field] + return doc @@ -304,6 +345,17 @@ async def confirm_document(doc_id: int, request: Request): person_id = None + # For multi-page PDFs (page 2+), auto-link to page 1's person + page1_person_id = None + if current_doc.get("pdf_group_id") and (current_doc.get("page_number") or 0) > 1: + page1 = conn.execute( + """SELECT person_id FROM documents + WHERE pdf_group_id=? AND page_number=1 AND person_id IS NOT NULL""", + (current_doc["pdf_group_id"],), + ).fetchone() + if page1: + page1_person_id = page1["person_id"] + # Option 1: User explicitly chose to merge with an existing person if merge_person_id: try: @@ -388,6 +440,10 @@ async def confirm_document(doc_id: int, request: Request): ), ) + # Option 2b: Auto-link to page 1's person for multi-page PDFs + if not person_id and page1_person_id: + person_id = page1_person_id + # Option 3: Create new person if not person_id: existing_person = None diff --git a/routers/upload.py b/routers/upload.py index c32e190..911def1 100644 --- a/routers/upload.py +++ b/routers/upload.py @@ -108,7 +108,7 @@ async def upload_page(request: Request): SUM(CASE WHEN status='confirmed' THEN 1 ELSE 0 END) AS confirmed, SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review, SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors - FROM documents""" + FROM documents WHERE status != 'staged'""" ).fetchone() return templates.TemplateResponse( request, "index.html", { @@ -171,3 +171,102 @@ async def upload_files( if len(doc_ids) == 1: return RedirectResponse(f"/review/{doc_ids[0][0]}?wait=1", status_code=303) return RedirectResponse("/documents?uploaded=1", status_code=303) + + +# ─── Two-step workflow: stage images, then process ───────────── + +@router.post("/upload/stage") +async def stage_file( + request: Request, + files: list[UploadFile] = File(...), +): + """Save uploaded images without triggering AI extraction.""" + staged = [] + for upload in files: + suffix = Path(upload.filename).suffix.lower() + if suffix not in ALLOWED_EXTENSIONS: + continue + + file_bytes = await upload.read() + + if suffix in ALLOWED_PDF_EXTS: + pages = pdf_to_images(file_bytes, upload.filename) + for page_info in pages: + with get_db() as conn: + cursor = conn.execute( + """INSERT INTO documents + (image_path, status, pdf_group_id, page_number) + VALUES (?, 'staged', ?, ?)""", + ( + page_info["image_path"], + page_info["pdf_group_id"], + page_info["page_number"], + ), + ) + staged.append({ + "id": cursor.lastrowid, + "image_path": page_info["image_path"], + "name": f"{upload.filename} (p{page_info['page_number']})", + }) + else: + rel_path = _save_image(file_bytes, upload.filename) + with get_db() as conn: + cursor = conn.execute( + "INSERT INTO documents (image_path, status) VALUES (?, 'staged')", + (rel_path,), + ) + staged.append({ + "id": cursor.lastrowid, + "image_path": rel_path, + "name": upload.filename, + }) + + from fastapi.responses import JSONResponse + return JSONResponse({"staged": staged}) + + +@router.post("/upload/process-staged") +async def process_staged( + request: Request, + doc_ids: str = Form(...), + provider: str = Form(""), +): + """Trigger AI extraction for previously staged documents.""" + if not provider: + provider = get_default_provider() + + ids = [int(x) for x in doc_ids.split(",") if x.strip().isdigit()] + + with get_db() as conn: + rows = conn.execute( + f"SELECT id, image_path FROM documents WHERE id IN ({','.join('?' * len(ids))}) AND status='staged'", + ids, + ).fetchall() + for row in rows: + conn.execute( + "UPDATE documents SET status='pending', provider=?, updated_at=CURRENT_TIMESTAMP WHERE id=?", + (provider, row["id"]), + ) + + for row in rows: + asyncio.create_task(_extract_and_save(row["id"], row["image_path"], provider)) + + if len(rows) == 1: + return RedirectResponse(f"/review/{rows[0]['id']}?wait=1", status_code=303) + return RedirectResponse("/documents?uploaded=1", status_code=303) + + +@router.delete("/upload/staged/{doc_id}") +async def remove_staged(doc_id: int): + """Remove a single staged document before processing.""" + from fastapi.responses import JSONResponse + with get_db() as conn: + row = conn.execute("SELECT image_path FROM documents WHERE id=? AND status='staged'", (doc_id,)).fetchone() + if row: + # Delete the file + file_path = Path(UPLOAD_DIR) / row["image_path"] + if file_path.exists(): + file_path.unlink() + conn.execute("DELETE FROM documents WHERE id=?", (doc_id,)) + return JSONResponse({"ok": True}) + return JSONResponse({"ok": False}, status_code=404) diff --git a/static/css/main.css b/static/css/main.css index 7a0550d..c01ad2e 100644 --- a/static/css/main.css +++ b/static/css/main.css @@ -355,6 +355,20 @@ a.stat.active { border-color: rgba(15,118,110,.24); background: var(--primary-so margin-bottom: 1rem; color: var(--success); } +.info-banner { + background: linear-gradient(135deg, rgba(239,246,255,.95), rgba(255,255,255,.82)); + border: 1px solid #bfdbfe; + border-radius: calc(var(--radius) - 4px); + padding: .85rem 1rem; + margin-bottom: 1rem; + color: #1d4ed8; + display: flex; + align-items: center; + gap: .6rem; + font-size: .92rem; + line-height: 1.5; +} +.info-banner svg { flex-shrink: 0; stroke: #3b82f6; } /* ============================ Upload Card @@ -1450,3 +1464,107 @@ a.stat.active { border-color: rgba(15,118,110,.24); background: var(--primary-so font-size: .68rem; } } + +/* ============================ + Staging Gallery (two-step upload) + ============================ */ +.stage-controls { + display: flex; + gap: .75rem; + flex-wrap: wrap; + margin-bottom: 1rem; +} +.stage-capture-btn svg { vertical-align: -.15em; margin-inline-end: .3rem; } + +.staging-gallery { + display: grid; + grid-template-columns: repeat(auto-fill, minmax(130px, 1fr)); + gap: .75rem; + margin-bottom: 1rem; +} +.staging-thumb { + position: relative; + border-radius: var(--radius-sm); + overflow: hidden; + background: var(--surface-tint); + border: 2px solid var(--border); + aspect-ratio: 3/4; + display: flex; + flex-direction: column; + align-items: center; + justify-content: center; + transition: var(--transition); +} +.staging-thumb.loading { + opacity: .6; +} +.staging-thumb.removing { + opacity: 0; + transform: scale(.9); +} +.staging-thumb img { + width: 100%; + height: 100%; + object-fit: cover; +} +.staging-thumb-name { + font-size: .72rem; + color: var(--text-muted); + padding: .25rem .4rem; + text-align: center; + position: absolute; + bottom: 0; + left: 0; + right: 0; + background: rgba(255,255,255,.85); + backdrop-filter: blur(4px); + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} +.staging-remove { + position: absolute; + top: 4px; + left: 4px; + width: 26px; + height: 26px; + border-radius: 50%; + border: none; + background: rgba(220,38,38,.85); + color: #fff; + font-size: 1.1rem; + line-height: 1; + cursor: pointer; + display: flex; + align-items: center; + justify-content: center; + opacity: 0; + transition: opacity .15s; +} +.staging-thumb:hover .staging-remove { opacity: 1; } +@media (pointer: coarse) { .staging-remove { opacity: 1; } } + +.staging-thumb-spinner { + width: 28px; + height: 28px; + border: 3px solid var(--border); + border-top-color: var(--primary); + border-radius: 50%; + animation: spin .7s linear infinite; +} +@keyframes spin { to { transform: rotate(360deg); } } + +.staging-count { + text-align: center; + font-weight: 600; + color: var(--primary-dark); + margin-bottom: .75rem; + font-size: 1rem; +} + +.stage-action-buttons { + display: flex; + gap: .5rem; + justify-content: center; + margin-top: .75rem; +} diff --git a/templates/index.html b/templates/index.html index 1f50634..c15c6a3 100644 --- a/templates/index.html +++ b/templates/index.html @@ -74,6 +74,57 @@ + +
+
+
+

التقاط صور ثم معالجة دفعة واحدة

+

التقط أو أضف صوراً واحدة تلو الأخرى، ثم اضغط على «معالجة الكل» عند الانتهاء.

+
+
+ +
+ + + + +
+ + + + + +
+