perf: paginate docs list, lazy thumbnails, static cache headers

This commit is contained in:
Krikorios
2026-04-19 00:05:09 +03:00
parent 5b1e770948
commit da900cab2d
4 changed files with 45 additions and 17 deletions
+12 -5
View File
@@ -281,7 +281,8 @@ async def scan_duplicates():
for g in hash_groups:
result = conn.execute(
"""UPDATE documents SET duplicate_of=?
WHERE image_hash=? AND id != ? AND status != 'staged'""",
WHERE image_hash=? AND id != ? AND status != 'staged'
AND COALESCE(duplicate_dismissed, 0) = 0""",
(g["keeper"], g["image_hash"], g["keeper"]),
)
flagged_by_hash += result.rowcount or 0
@@ -307,6 +308,7 @@ async def scan_duplicates():
AND COALESCE(TRIM(page_info),'')=?
AND id != ?
AND duplicate_of IS NULL
AND COALESCE(duplicate_dismissed, 0) = 0
AND status IN ('extracted','confirmed')""",
(g["keeper"], g["rn"], g["sc"], g["pi"], g["keeper"]),
)
@@ -395,16 +397,21 @@ async def delete_all_duplicates():
@router.post("/documents/{doc_id}/unflag-duplicate")
async def unflag_duplicate(doc_id: int):
"""Mark a flagged-duplicate document as NOT a duplicate (clear duplicate_of)."""
"""Mark a flagged-duplicate document as NOT a duplicate (clear duplicate_of
and remember the decision so future scans don't re-flag it)."""
with get_db() as conn:
row = conn.execute(
"SELECT id FROM documents WHERE id=? AND duplicate_of IS NOT NULL",
"SELECT id FROM documents WHERE id=?",
(doc_id,),
).fetchone()
if not row:
return JSONResponse({"error": "not flagged"}, status_code=404)
return JSONResponse({"error": "not found"}, status_code=404)
conn.execute(
"UPDATE documents SET duplicate_of=NULL, updated_at=CURRENT_TIMESTAMP WHERE id=?",
"""UPDATE documents
SET duplicate_of=NULL,
duplicate_dismissed=1,
updated_at=CURRENT_TIMESTAMP
WHERE id=?""",
(doc_id,),
)
return JSONResponse({"ok": True})
+14 -5
View File
@@ -35,11 +35,13 @@ def _hash_file(path: Path) -> str:
def _find_duplicate(conn, image_hash: str) -> dict | None:
"""Return an existing non-staged document sharing the same image hash."""
"""Return an existing non-staged document sharing the same image hash.
Skips rows the user has explicitly dismissed as 'not a duplicate'."""
row = conn.execute(
"""SELECT id, status, image_path, person_id, request_number
FROM documents
WHERE image_hash=? AND status != 'staged'
AND COALESCE(duplicate_dismissed, 0) = 0
ORDER BY id LIMIT 1""",
(image_hash,),
).fetchone()
@@ -126,14 +128,21 @@ async def _extract_and_save(doc_id: int, image_path: str, provider: str = ""):
AND COALESCE(search_scope,'')=?
AND COALESCE(page_info,'')=?
AND status IN ('extracted','confirmed')
AND COALESCE(duplicate_dismissed, 0) = 0
ORDER BY id LIMIT 1""",
(doc_id, req_num, scope, page_info),
).fetchone()
if existing:
conn.execute(
"UPDATE documents SET duplicate_of=? WHERE id=?",
(existing["id"], doc_id),
)
# Only auto-flag if THIS document hasn't itself been dismissed.
self_row = conn.execute(
"SELECT COALESCE(duplicate_dismissed, 0) AS d FROM documents WHERE id=?",
(doc_id,),
).fetchone()
if not (self_row and self_row["d"]):
conn.execute(
"UPDATE documents SET duplicate_of=? WHERE id=?",
(existing["id"], doc_id),
)
except Exception as e:
with get_db() as conn: