421 lines
15 KiB
Python
421 lines
15 KiB
Python
import os
|
|
from pathlib import Path
|
|
|
|
from fastapi import APIRouter, Request
|
|
from fastapi.responses import FileResponse, JSONResponse
|
|
from fastapi.templating import Jinja2Templates
|
|
|
|
from config import UPLOAD_DIR
|
|
from database.connection import get_db
|
|
from services.template_utils import render_template
|
|
from services.thumbnails import get_or_create_thumbnail
|
|
|
|
router = APIRouter()
|
|
templates = Jinja2Templates(directory="templates")
|
|
|
|
|
|
@router.get("/thumbs/{rel_path:path}")
|
|
async def serve_thumbnail(rel_path: str, size: int = 320):
|
|
"""Serve a cached JPEG thumbnail for an uploaded image.
|
|
Falls back to the original file if thumbnail generation fails."""
|
|
# Clamp size to sensible values to prevent DoS
|
|
if size not in (160, 240, 320, 480, 640):
|
|
size = 320
|
|
|
|
thumb = get_or_create_thumbnail(rel_path, size=size)
|
|
if thumb is None:
|
|
# Fallback: serve original if it exists
|
|
base = Path(UPLOAD_DIR).resolve()
|
|
full = (base / rel_path).resolve()
|
|
try:
|
|
full.relative_to(base)
|
|
except ValueError:
|
|
return JSONResponse({"error": "not found"}, status_code=404)
|
|
if not full.is_file():
|
|
return JSONResponse({"error": "not found"}, status_code=404)
|
|
return FileResponse(
|
|
full,
|
|
headers={"Cache-Control": "public, max-age=604800"},
|
|
)
|
|
|
|
return FileResponse(
|
|
thumb,
|
|
media_type="image/jpeg",
|
|
headers={"Cache-Control": "public, max-age=604800, immutable"},
|
|
)
|
|
|
|
_DOC_LIST_COLUMNS = (
|
|
"d.id, d.person_id, d.image_path, d.request_number, d.status, "
|
|
"d.duplicate_of, d.extraction_error, d.created_at"
|
|
)
|
|
|
|
PAGE_SIZE = 50
|
|
|
|
|
|
@router.get("/documents")
|
|
async def document_queue(
|
|
request: Request,
|
|
status: str = "",
|
|
uploaded: int = 0,
|
|
duplicates: int = 0,
|
|
page: int = 1,
|
|
):
|
|
if page < 1:
|
|
page = 1
|
|
offset = (page - 1) * PAGE_SIZE
|
|
|
|
with get_db() as conn:
|
|
if status:
|
|
rows = conn.execute(
|
|
f"""SELECT {_DOC_LIST_COLUMNS}, p.first_name, p.family_name
|
|
FROM documents d
|
|
LEFT JOIN persons p ON p.id = d.person_id
|
|
WHERE d.status=?
|
|
ORDER BY d.id DESC
|
|
LIMIT ? OFFSET ?""",
|
|
(status, PAGE_SIZE, offset),
|
|
).fetchall()
|
|
total_filtered = conn.execute(
|
|
"SELECT COUNT(*) AS n FROM documents WHERE status=?", (status,)
|
|
).fetchone()["n"]
|
|
else:
|
|
rows = conn.execute(
|
|
f"""SELECT {_DOC_LIST_COLUMNS}, p.first_name, p.family_name
|
|
FROM documents d
|
|
LEFT JOIN persons p ON p.id = d.person_id
|
|
ORDER BY d.id DESC
|
|
LIMIT ? OFFSET ?""",
|
|
(PAGE_SIZE, offset),
|
|
).fetchall()
|
|
total_filtered = conn.execute(
|
|
"SELECT COUNT(*) AS n FROM documents"
|
|
).fetchone()["n"]
|
|
|
|
stats = conn.execute(
|
|
"""SELECT
|
|
COUNT(*) AS total,
|
|
SUM(CASE WHEN status='confirmed' THEN 1 ELSE 0 END) AS confirmed,
|
|
SUM(CASE WHEN status='extracted' THEN 1 ELSE 0 END) AS pending_review,
|
|
SUM(CASE WHEN status='pending' THEN 1 ELSE 0 END) AS processing,
|
|
SUM(CASE WHEN status='error' THEN 1 ELSE 0 END) AS errors,
|
|
SUM(CASE WHEN duplicate_of IS NOT NULL THEN 1 ELSE 0 END) AS duplicates
|
|
FROM documents WHERE status != 'staged'"""
|
|
).fetchone()
|
|
|
|
total_pages = max(1, (total_filtered + PAGE_SIZE - 1) // PAGE_SIZE)
|
|
|
|
return render_template(
|
|
templates,
|
|
request,
|
|
"documents.html",
|
|
{
|
|
"documents": [dict(r) for r in rows],
|
|
"stats": dict(stats) if stats else {},
|
|
"current_status": status,
|
|
"uploaded": uploaded,
|
|
"duplicates_skipped": duplicates,
|
|
"page": page,
|
|
"total_pages": total_pages,
|
|
"page_size": PAGE_SIZE,
|
|
"total_filtered": total_filtered,
|
|
},
|
|
)
|
|
|
|
|
|
@router.delete("/documents/{doc_id}")
|
|
async def delete_document(doc_id: int):
|
|
"""Delete a document and its properties. Also removes the image file from disk."""
|
|
with get_db() as conn:
|
|
doc = conn.execute(
|
|
"SELECT image_path, person_id FROM documents WHERE id=?", (doc_id,)
|
|
).fetchone()
|
|
if not doc:
|
|
return JSONResponse({"error": "not found"}, status_code=404)
|
|
|
|
# If other documents were flagged as duplicates of this one, detach them
|
|
# (promote the earliest orphan to be the new keeper so they remain grouped).
|
|
orphans = conn.execute(
|
|
"SELECT id FROM documents WHERE duplicate_of=? ORDER BY id",
|
|
(doc_id,),
|
|
).fetchall()
|
|
if orphans:
|
|
new_keeper = orphans[0]["id"]
|
|
conn.execute(
|
|
"UPDATE documents SET duplicate_of=NULL WHERE id=?",
|
|
(new_keeper,),
|
|
)
|
|
conn.execute(
|
|
"UPDATE documents SET duplicate_of=? WHERE duplicate_of=? AND id != ?",
|
|
(new_keeper, doc_id, new_keeper),
|
|
)
|
|
|
|
# Delete properties for this document
|
|
conn.execute("DELETE FROM properties WHERE document_id=?", (doc_id,))
|
|
|
|
# Delete the document record
|
|
conn.execute("DELETE FROM documents WHERE id=?", (doc_id,))
|
|
|
|
# If the person has no remaining documents, remove them too
|
|
if doc["person_id"]:
|
|
remaining = conn.execute(
|
|
"SELECT COUNT(*) AS n FROM documents WHERE person_id=?",
|
|
(doc["person_id"],),
|
|
).fetchone()["n"]
|
|
if remaining == 0:
|
|
conn.execute("DELETE FROM properties WHERE person_id=?", (doc["person_id"],))
|
|
conn.execute("DELETE FROM persons WHERE id=?", (doc["person_id"],))
|
|
|
|
# Remove the image file from disk
|
|
image_file = Path(UPLOAD_DIR) / doc["image_path"]
|
|
try:
|
|
image_file.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
|
|
return JSONResponse({"ok": True})
|
|
|
|
|
|
@router.post("/documents/{doc_id}/retry")
|
|
async def retry_document(doc_id: int):
|
|
"""Re-run AI extraction on an errored or extracted document."""
|
|
import asyncio
|
|
from services.extractor import get_default_provider
|
|
from routers.upload import _extract_and_save
|
|
|
|
with get_db() as conn:
|
|
doc = conn.execute(
|
|
"SELECT id, image_path, provider, status FROM documents WHERE id=?",
|
|
(doc_id,),
|
|
).fetchone()
|
|
if not doc:
|
|
return JSONResponse({"error": "not found"}, status_code=404)
|
|
|
|
# Clear previous extraction data and requeue
|
|
conn.execute("DELETE FROM properties WHERE document_id=?", (doc_id,))
|
|
conn.execute(
|
|
"""UPDATE documents SET status='pending',
|
|
extraction_error=NULL,
|
|
raw_extraction_json=NULL,
|
|
updated_at=CURRENT_TIMESTAMP
|
|
WHERE id=?""",
|
|
(doc_id,),
|
|
)
|
|
|
|
provider = doc["provider"] or get_default_provider()
|
|
asyncio.create_task(_extract_and_save(doc_id, doc["image_path"], provider))
|
|
return JSONResponse({"ok": True, "status": "pending"})
|
|
|
|
|
|
@router.post("/documents/retry-errors")
|
|
async def retry_all_errors():
|
|
"""Re-run extraction for every document currently in error state."""
|
|
import asyncio
|
|
from services.extractor import get_default_provider
|
|
from routers.upload import _extract_and_save
|
|
|
|
default_provider = get_default_provider()
|
|
with get_db() as conn:
|
|
rows = conn.execute(
|
|
"SELECT id, image_path, provider FROM documents WHERE status='error'"
|
|
).fetchall()
|
|
ids = [r["id"] for r in rows]
|
|
if ids:
|
|
conn.execute("DELETE FROM properties WHERE document_id IN (" + ",".join("?" * len(ids)) + ")", ids)
|
|
conn.execute(
|
|
"UPDATE documents SET status='pending', extraction_error=NULL, raw_extraction_json=NULL, updated_at=CURRENT_TIMESTAMP WHERE id IN (" + ",".join("?" * len(ids)) + ")",
|
|
ids,
|
|
)
|
|
|
|
for row in rows:
|
|
provider = row["provider"] or default_provider
|
|
asyncio.create_task(_extract_and_save(row["id"], row["image_path"], provider))
|
|
|
|
return JSONResponse({"ok": True, "retried": len(rows)})
|
|
|
|
|
|
@router.post("/documents/scan-duplicates")
|
|
async def scan_duplicates():
|
|
"""
|
|
Backfill image_hash for existing documents and flag duplicates.
|
|
A doc is marked duplicate_of the earliest document (lowest id) that shares
|
|
either the exact image hash OR the same request_number + search_scope + page_info.
|
|
"""
|
|
import hashlib
|
|
|
|
def _hash_file(path: Path) -> str | None:
|
|
try:
|
|
h = hashlib.sha256()
|
|
with open(path, "rb") as f:
|
|
for chunk in iter(lambda: f.read(1 << 20), b""):
|
|
h.update(chunk)
|
|
return h.hexdigest()
|
|
except OSError:
|
|
return None
|
|
|
|
hashed = 0
|
|
flagged_by_hash = 0
|
|
flagged_by_request = 0
|
|
|
|
with get_db() as conn:
|
|
# 1. Hash any document missing image_hash
|
|
rows = conn.execute(
|
|
"SELECT id, image_path FROM documents WHERE image_hash IS NULL OR image_hash=''"
|
|
).fetchall()
|
|
for row in rows:
|
|
full_path = Path(UPLOAD_DIR) / row["image_path"]
|
|
digest = _hash_file(full_path)
|
|
if digest:
|
|
conn.execute(
|
|
"UPDATE documents SET image_hash=? WHERE id=?",
|
|
(digest, row["id"]),
|
|
)
|
|
hashed += 1
|
|
|
|
# 2. Flag duplicates by image_hash (keep earliest)
|
|
hash_groups = conn.execute(
|
|
"""SELECT image_hash, MIN(id) AS keeper, COUNT(*) AS n
|
|
FROM documents
|
|
WHERE image_hash IS NOT NULL AND image_hash != ''
|
|
AND status != 'staged'
|
|
GROUP BY image_hash
|
|
HAVING n > 1"""
|
|
).fetchall()
|
|
for g in hash_groups:
|
|
result = conn.execute(
|
|
"""UPDATE documents SET duplicate_of=?
|
|
WHERE image_hash=? AND id != ? AND status != 'staged'
|
|
AND COALESCE(duplicate_dismissed, 0) = 0""",
|
|
(g["keeper"], g["image_hash"], g["keeper"]),
|
|
)
|
|
flagged_by_hash += result.rowcount or 0
|
|
|
|
# 3. Flag duplicates by request_number + search_scope + page_info (keep earliest)
|
|
logical_groups = conn.execute(
|
|
"""SELECT TRIM(request_number) AS rn,
|
|
COALESCE(TRIM(search_scope),'') AS sc,
|
|
COALESCE(TRIM(page_info),'') AS pi,
|
|
MIN(id) AS keeper,
|
|
COUNT(*) AS n
|
|
FROM documents
|
|
WHERE status IN ('extracted','confirmed')
|
|
AND request_number IS NOT NULL AND TRIM(request_number) != ''
|
|
GROUP BY rn, sc, pi
|
|
HAVING n > 1"""
|
|
).fetchall()
|
|
for g in logical_groups:
|
|
result = conn.execute(
|
|
"""UPDATE documents SET duplicate_of=?
|
|
WHERE TRIM(request_number)=?
|
|
AND COALESCE(TRIM(search_scope),'')=?
|
|
AND COALESCE(TRIM(page_info),'')=?
|
|
AND id != ?
|
|
AND duplicate_of IS NULL
|
|
AND COALESCE(duplicate_dismissed, 0) = 0
|
|
AND status IN ('extracted','confirmed')""",
|
|
(g["keeper"], g["rn"], g["sc"], g["pi"], g["keeper"]),
|
|
)
|
|
flagged_by_request += result.rowcount or 0
|
|
|
|
total_dupes = conn.execute(
|
|
"SELECT COUNT(*) AS n FROM documents WHERE duplicate_of IS NOT NULL"
|
|
).fetchone()["n"]
|
|
|
|
return JSONResponse({
|
|
"ok": True,
|
|
"hashed": hashed,
|
|
"flagged_by_hash": flagged_by_hash,
|
|
"flagged_by_request": flagged_by_request,
|
|
"total_duplicates": total_dupes,
|
|
})
|
|
|
|
|
|
@router.get("/documents/duplicates")
|
|
async def duplicates_view(request: Request):
|
|
"""Group duplicates with their originals so the user can compare side-by-side."""
|
|
with get_db() as conn:
|
|
# Fetch all docs that have been flagged as duplicates + their keepers
|
|
dup_rows = conn.execute(
|
|
"""SELECT d.*, p.first_name, p.family_name
|
|
FROM documents d
|
|
LEFT JOIN persons p ON p.id = d.person_id
|
|
WHERE d.duplicate_of IS NOT NULL
|
|
ORDER BY d.duplicate_of, d.id"""
|
|
).fetchall()
|
|
|
|
keeper_ids = sorted({r["duplicate_of"] for r in dup_rows if r["duplicate_of"]})
|
|
keepers_by_id = {}
|
|
if keeper_ids:
|
|
placeholders = ",".join("?" * len(keeper_ids))
|
|
keeper_rows = conn.execute(
|
|
f"""SELECT d.*, p.first_name, p.family_name
|
|
FROM documents d
|
|
LEFT JOIN persons p ON p.id = d.person_id
|
|
WHERE d.id IN ({placeholders})""",
|
|
keeper_ids,
|
|
).fetchall()
|
|
keepers_by_id = {r["id"]: dict(r) for r in keeper_rows}
|
|
|
|
groups = []
|
|
seen_keepers = []
|
|
dups_by_keeper: dict[int, list[dict]] = {}
|
|
for r in dup_rows:
|
|
k = r["duplicate_of"]
|
|
dups_by_keeper.setdefault(k, []).append(dict(r))
|
|
if k not in seen_keepers:
|
|
seen_keepers.append(k)
|
|
|
|
for k in seen_keepers:
|
|
keeper = keepers_by_id.get(k)
|
|
if not keeper:
|
|
continue
|
|
groups.append({"keeper": keeper, "duplicates": dups_by_keeper.get(k, [])})
|
|
|
|
return render_template(
|
|
templates,
|
|
request,
|
|
"duplicates.html",
|
|
{"groups": groups, "total_duplicates": len(dup_rows)},
|
|
)
|
|
|
|
|
|
@router.post("/documents/delete-duplicates")
|
|
async def delete_all_duplicates():
|
|
"""Delete every document flagged as duplicate_of another doc."""
|
|
deleted = 0
|
|
with get_db() as conn:
|
|
rows = conn.execute(
|
|
"SELECT id, image_path FROM documents WHERE duplicate_of IS NOT NULL"
|
|
).fetchall()
|
|
for row in rows:
|
|
conn.execute("DELETE FROM properties WHERE document_id=?", (row["id"],))
|
|
conn.execute("DELETE FROM documents WHERE id=?", (row["id"],))
|
|
# Remove file from disk
|
|
try:
|
|
(Path(UPLOAD_DIR) / row["image_path"]).unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
deleted += 1
|
|
return JSONResponse({"ok": True, "deleted": deleted})
|
|
|
|
|
|
@router.post("/documents/{doc_id}/unflag-duplicate")
|
|
async def unflag_duplicate(doc_id: int):
|
|
"""Mark a flagged-duplicate document as NOT a duplicate (clear duplicate_of
|
|
and remember the decision so future scans don't re-flag it)."""
|
|
with get_db() as conn:
|
|
row = conn.execute(
|
|
"SELECT id FROM documents WHERE id=?",
|
|
(doc_id,),
|
|
).fetchone()
|
|
if not row:
|
|
return JSONResponse({"error": "not found"}, status_code=404)
|
|
conn.execute(
|
|
"""UPDATE documents
|
|
SET duplicate_of=NULL,
|
|
duplicate_dismissed=1,
|
|
updated_at=CURRENT_TIMESTAMP
|
|
WHERE id=?""",
|
|
(doc_id,),
|
|
)
|
|
return JSONResponse({"ok": True})
|