feat: two-step staging upload + multi-page PDF person inheritance

- Add stage/process-staged/remove-staged endpoints for batch upload workflow
- Add staging gallery UI with thumbnail previews and per-image removal
- Inherit person info & doc fields from page 1 for multi-page PDFs
- Auto-link page 2+ to same person on confirmation
- Show info banner on review page for multi-page documents
- Exclude staged docs from stats counters
This commit is contained in:
Georges Haddad
2026-04-11 17:09:48 +03:00
parent 1b6d5543a6
commit 3b3bcf2cf2
6 changed files with 446 additions and 2 deletions
+1 -1
View File
@@ -39,7 +39,7 @@ async def document_queue(request: Request, status: str = "", uploaded: int = 0):
SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review,
SUM(CASE WHEN status='pending' THEN 1 ELSE 0 END) AS processing,
SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors
FROM documents"""
FROM documents WHERE status != 'staged'"""
).fetchone()
return templates.TemplateResponse(
+56
View File
@@ -42,6 +42,47 @@ def _get_document(doc_id: int) -> dict | None:
else:
doc["person"] = {}
# For multi-page PDFs (page 2+), inherit person info & doc fields from page 1
doc["inherited_from_page1"] = False
if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1:
page1 = conn.execute(
"""SELECT * FROM documents
WHERE pdf_group_id=? AND page_number=1""",
(doc["pdf_group_id"],),
).fetchone()
if page1:
page1 = dict(page1)
# Inherit person data if current page has no name
current_first = (doc["person"].get("first_name") or "").strip()
if not current_first:
if page1.get("person_id"):
person = conn.execute(
"SELECT * FROM persons WHERE id=?", (page1["person_id"],)
).fetchone()
if person:
doc["person"] = dict(person)
doc["inherited_from_page1"] = True
doc["page1_person_id"] = page1["person_id"]
elif page1.get("raw_extraction_json"):
try:
p1_extracted = json.loads(page1["raw_extraction_json"])
p1_person = p1_extracted.get("person", {})
if (p1_person.get("first_name") or "").strip():
doc["person"] = p1_person
doc["inherited_from_page1"] = True
except Exception:
pass
# Inherit document-level fields if missing
inherit_fields = [
"request_number", "request_date", "search_scope",
"request_purpose", "data_valid_until", "registry_office",
"applicant_name_raw",
]
for field in inherit_fields:
if not (doc.get(field) or "").strip() and (page1.get(field) or "").strip():
doc[field] = page1[field]
return doc
@@ -304,6 +345,17 @@ async def confirm_document(doc_id: int, request: Request):
person_id = None
# For multi-page PDFs (page 2+), auto-link to page 1's person
page1_person_id = None
if current_doc.get("pdf_group_id") and (current_doc.get("page_number") or 0) > 1:
page1 = conn.execute(
"""SELECT person_id FROM documents
WHERE pdf_group_id=? AND page_number=1 AND person_id IS NOT NULL""",
(current_doc["pdf_group_id"],),
).fetchone()
if page1:
page1_person_id = page1["person_id"]
# Option 1: User explicitly chose to merge with an existing person
if merge_person_id:
try:
@@ -388,6 +440,10 @@ async def confirm_document(doc_id: int, request: Request):
),
)
# Option 2b: Auto-link to page 1's person for multi-page PDFs
if not person_id and page1_person_id:
person_id = page1_person_id
# Option 3: Create new person
if not person_id:
existing_person = None
+100 -1
View File
@@ -108,7 +108,7 @@ async def upload_page(request: Request):
SUM(CASE WHEN status='confirmed' THEN 1 ELSE 0 END) AS confirmed,
SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review,
SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors
FROM documents"""
FROM documents WHERE status != 'staged'"""
).fetchone()
return templates.TemplateResponse(
request, "index.html", {
@@ -171,3 +171,102 @@ async def upload_files(
if len(doc_ids) == 1:
return RedirectResponse(f"/review/{doc_ids[0][0]}?wait=1", status_code=303)
return RedirectResponse("/documents?uploaded=1", status_code=303)
# ─── Two-step workflow: stage images, then process ─────────────
@router.post("/upload/stage")
async def stage_file(
request: Request,
files: list[UploadFile] = File(...),
):
"""Save uploaded images without triggering AI extraction."""
staged = []
for upload in files:
suffix = Path(upload.filename).suffix.lower()
if suffix not in ALLOWED_EXTENSIONS:
continue
file_bytes = await upload.read()
if suffix in ALLOWED_PDF_EXTS:
pages = pdf_to_images(file_bytes, upload.filename)
for page_info in pages:
with get_db() as conn:
cursor = conn.execute(
"""INSERT INTO documents
(image_path, status, pdf_group_id, page_number)
VALUES (?, 'staged', ?, ?)""",
(
page_info["image_path"],
page_info["pdf_group_id"],
page_info["page_number"],
),
)
staged.append({
"id": cursor.lastrowid,
"image_path": page_info["image_path"],
"name": f"{upload.filename} (p{page_info['page_number']})",
})
else:
rel_path = _save_image(file_bytes, upload.filename)
with get_db() as conn:
cursor = conn.execute(
"INSERT INTO documents (image_path, status) VALUES (?, 'staged')",
(rel_path,),
)
staged.append({
"id": cursor.lastrowid,
"image_path": rel_path,
"name": upload.filename,
})
from fastapi.responses import JSONResponse
return JSONResponse({"staged": staged})
@router.post("/upload/process-staged")
async def process_staged(
request: Request,
doc_ids: str = Form(...),
provider: str = Form(""),
):
"""Trigger AI extraction for previously staged documents."""
if not provider:
provider = get_default_provider()
ids = [int(x) for x in doc_ids.split(",") if x.strip().isdigit()]
with get_db() as conn:
rows = conn.execute(
f"SELECT id, image_path FROM documents WHERE id IN ({','.join('?' * len(ids))}) AND status='staged'",
ids,
).fetchall()
for row in rows:
conn.execute(
"UPDATE documents SET status='pending', provider=?, updated_at=CURRENT_TIMESTAMP WHERE id=?",
(provider, row["id"]),
)
for row in rows:
asyncio.create_task(_extract_and_save(row["id"], row["image_path"], provider))
if len(rows) == 1:
return RedirectResponse(f"/review/{rows[0]['id']}?wait=1", status_code=303)
return RedirectResponse("/documents?uploaded=1", status_code=303)
@router.delete("/upload/staged/{doc_id}")
async def remove_staged(doc_id: int):
"""Remove a single staged document before processing."""
from fastapi.responses import JSONResponse
with get_db() as conn:
row = conn.execute("SELECT image_path FROM documents WHERE id=? AND status='staged'", (doc_id,)).fetchone()
if row:
# Delete the file
file_path = Path(UPLOAD_DIR) / row["image_path"]
if file_path.exists():
file_path.unlink()
conn.execute("DELETE FROM documents WHERE id=?", (doc_id,))
return JSONResponse({"ok": True})
return JSONResponse({"ok": False}, status_code=404)