feat: two-step staging upload + multi-page PDF person inheritance
- Add stage/process-staged/remove-staged endpoints for batch upload workflow - Add staging gallery UI with thumbnail previews and per-image removal - Inherit person info & doc fields from page 1 for multi-page PDFs - Auto-link page 2+ to same person on confirmation - Show info banner on review page for multi-page documents - Exclude staged docs from stats counters
This commit is contained in:
@@ -39,7 +39,7 @@ async def document_queue(request: Request, status: str = "", uploaded: int = 0):
|
||||
SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review,
|
||||
SUM(CASE WHEN status='pending' THEN 1 ELSE 0 END) AS processing,
|
||||
SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors
|
||||
FROM documents"""
|
||||
FROM documents WHERE status != 'staged'"""
|
||||
).fetchone()
|
||||
|
||||
return templates.TemplateResponse(
|
||||
|
||||
@@ -42,6 +42,47 @@ def _get_document(doc_id: int) -> dict | None:
|
||||
else:
|
||||
doc["person"] = {}
|
||||
|
||||
# For multi-page PDFs (page 2+), inherit person info & doc fields from page 1
|
||||
doc["inherited_from_page1"] = False
|
||||
if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1:
|
||||
page1 = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE pdf_group_id=? AND page_number=1""",
|
||||
(doc["pdf_group_id"],),
|
||||
).fetchone()
|
||||
if page1:
|
||||
page1 = dict(page1)
|
||||
# Inherit person data if current page has no name
|
||||
current_first = (doc["person"].get("first_name") or "").strip()
|
||||
if not current_first:
|
||||
if page1.get("person_id"):
|
||||
person = conn.execute(
|
||||
"SELECT * FROM persons WHERE id=?", (page1["person_id"],)
|
||||
).fetchone()
|
||||
if person:
|
||||
doc["person"] = dict(person)
|
||||
doc["inherited_from_page1"] = True
|
||||
doc["page1_person_id"] = page1["person_id"]
|
||||
elif page1.get("raw_extraction_json"):
|
||||
try:
|
||||
p1_extracted = json.loads(page1["raw_extraction_json"])
|
||||
p1_person = p1_extracted.get("person", {})
|
||||
if (p1_person.get("first_name") or "").strip():
|
||||
doc["person"] = p1_person
|
||||
doc["inherited_from_page1"] = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Inherit document-level fields if missing
|
||||
inherit_fields = [
|
||||
"request_number", "request_date", "search_scope",
|
||||
"request_purpose", "data_valid_until", "registry_office",
|
||||
"applicant_name_raw",
|
||||
]
|
||||
for field in inherit_fields:
|
||||
if not (doc.get(field) or "").strip() and (page1.get(field) or "").strip():
|
||||
doc[field] = page1[field]
|
||||
|
||||
return doc
|
||||
|
||||
|
||||
@@ -304,6 +345,17 @@ async def confirm_document(doc_id: int, request: Request):
|
||||
|
||||
person_id = None
|
||||
|
||||
# For multi-page PDFs (page 2+), auto-link to page 1's person
|
||||
page1_person_id = None
|
||||
if current_doc.get("pdf_group_id") and (current_doc.get("page_number") or 0) > 1:
|
||||
page1 = conn.execute(
|
||||
"""SELECT person_id FROM documents
|
||||
WHERE pdf_group_id=? AND page_number=1 AND person_id IS NOT NULL""",
|
||||
(current_doc["pdf_group_id"],),
|
||||
).fetchone()
|
||||
if page1:
|
||||
page1_person_id = page1["person_id"]
|
||||
|
||||
# Option 1: User explicitly chose to merge with an existing person
|
||||
if merge_person_id:
|
||||
try:
|
||||
@@ -388,6 +440,10 @@ async def confirm_document(doc_id: int, request: Request):
|
||||
),
|
||||
)
|
||||
|
||||
# Option 2b: Auto-link to page 1's person for multi-page PDFs
|
||||
if not person_id and page1_person_id:
|
||||
person_id = page1_person_id
|
||||
|
||||
# Option 3: Create new person
|
||||
if not person_id:
|
||||
existing_person = None
|
||||
|
||||
+100
-1
@@ -108,7 +108,7 @@ async def upload_page(request: Request):
|
||||
SUM(CASE WHEN status='confirmed' THEN 1 ELSE 0 END) AS confirmed,
|
||||
SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review,
|
||||
SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors
|
||||
FROM documents"""
|
||||
FROM documents WHERE status != 'staged'"""
|
||||
).fetchone()
|
||||
return templates.TemplateResponse(
|
||||
request, "index.html", {
|
||||
@@ -171,3 +171,102 @@ async def upload_files(
|
||||
if len(doc_ids) == 1:
|
||||
return RedirectResponse(f"/review/{doc_ids[0][0]}?wait=1", status_code=303)
|
||||
return RedirectResponse("/documents?uploaded=1", status_code=303)
|
||||
|
||||
|
||||
# ─── Two-step workflow: stage images, then process ─────────────
|
||||
|
||||
@router.post("/upload/stage")
|
||||
async def stage_file(
|
||||
request: Request,
|
||||
files: list[UploadFile] = File(...),
|
||||
):
|
||||
"""Save uploaded images without triggering AI extraction."""
|
||||
staged = []
|
||||
for upload in files:
|
||||
suffix = Path(upload.filename).suffix.lower()
|
||||
if suffix not in ALLOWED_EXTENSIONS:
|
||||
continue
|
||||
|
||||
file_bytes = await upload.read()
|
||||
|
||||
if suffix in ALLOWED_PDF_EXTS:
|
||||
pages = pdf_to_images(file_bytes, upload.filename)
|
||||
for page_info in pages:
|
||||
with get_db() as conn:
|
||||
cursor = conn.execute(
|
||||
"""INSERT INTO documents
|
||||
(image_path, status, pdf_group_id, page_number)
|
||||
VALUES (?, 'staged', ?, ?)""",
|
||||
(
|
||||
page_info["image_path"],
|
||||
page_info["pdf_group_id"],
|
||||
page_info["page_number"],
|
||||
),
|
||||
)
|
||||
staged.append({
|
||||
"id": cursor.lastrowid,
|
||||
"image_path": page_info["image_path"],
|
||||
"name": f"{upload.filename} (p{page_info['page_number']})",
|
||||
})
|
||||
else:
|
||||
rel_path = _save_image(file_bytes, upload.filename)
|
||||
with get_db() as conn:
|
||||
cursor = conn.execute(
|
||||
"INSERT INTO documents (image_path, status) VALUES (?, 'staged')",
|
||||
(rel_path,),
|
||||
)
|
||||
staged.append({
|
||||
"id": cursor.lastrowid,
|
||||
"image_path": rel_path,
|
||||
"name": upload.filename,
|
||||
})
|
||||
|
||||
from fastapi.responses import JSONResponse
|
||||
return JSONResponse({"staged": staged})
|
||||
|
||||
|
||||
@router.post("/upload/process-staged")
|
||||
async def process_staged(
|
||||
request: Request,
|
||||
doc_ids: str = Form(...),
|
||||
provider: str = Form(""),
|
||||
):
|
||||
"""Trigger AI extraction for previously staged documents."""
|
||||
if not provider:
|
||||
provider = get_default_provider()
|
||||
|
||||
ids = [int(x) for x in doc_ids.split(",") if x.strip().isdigit()]
|
||||
|
||||
with get_db() as conn:
|
||||
rows = conn.execute(
|
||||
f"SELECT id, image_path FROM documents WHERE id IN ({','.join('?' * len(ids))}) AND status='staged'",
|
||||
ids,
|
||||
).fetchall()
|
||||
for row in rows:
|
||||
conn.execute(
|
||||
"UPDATE documents SET status='pending', provider=?, updated_at=CURRENT_TIMESTAMP WHERE id=?",
|
||||
(provider, row["id"]),
|
||||
)
|
||||
|
||||
for row in rows:
|
||||
asyncio.create_task(_extract_and_save(row["id"], row["image_path"], provider))
|
||||
|
||||
if len(rows) == 1:
|
||||
return RedirectResponse(f"/review/{rows[0]['id']}?wait=1", status_code=303)
|
||||
return RedirectResponse("/documents?uploaded=1", status_code=303)
|
||||
|
||||
|
||||
@router.delete("/upload/staged/{doc_id}")
|
||||
async def remove_staged(doc_id: int):
|
||||
"""Remove a single staged document before processing."""
|
||||
from fastapi.responses import JSONResponse
|
||||
with get_db() as conn:
|
||||
row = conn.execute("SELECT image_path FROM documents WHERE id=? AND status='staged'", (doc_id,)).fetchone()
|
||||
if row:
|
||||
# Delete the file
|
||||
file_path = Path(UPLOAD_DIR) / row["image_path"]
|
||||
if file_path.exists():
|
||||
file_path.unlink()
|
||||
conn.execute("DELETE FROM documents WHERE id=?", (doc_id,))
|
||||
return JSONResponse({"ok": True})
|
||||
return JSONResponse({"ok": False}, status_code=404)
|
||||
|
||||
Reference in New Issue
Block a user