diff --git a/routers/review.py b/routers/review.py index c883e87..aceea5a 100644 --- a/routers/review.py +++ b/routers/review.py @@ -8,7 +8,12 @@ from fastapi.templating import Jinja2Templates from config import UPLOAD_DIR from database.connection import get_db -from services.extractor import extract_document +from services.extractor import ( + extract_document, + get_available_providers, + get_default_provider, + verify_page_correlation, +) from services.search_service import normalize_arabic, _normalize_scope router = APIRouter() @@ -36,6 +41,60 @@ def _is_subsequent_page(page_info: str) -> bool: return False +def _find_page1_candidate(conn, doc: dict) -> tuple[bool, dict | None]: + """Return whether the doc looks like a later page and the best page-1 candidate.""" + is_subsequent_page = False + sibling_doc = None + + if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1: + is_subsequent_page = True + sibling_doc = conn.execute( + """SELECT * FROM documents + WHERE pdf_group_id=? AND page_number=1""", + (doc["pdf_group_id"],), + ).fetchone() + + if not sibling_doc: + page_info_str = doc.get("page_info") or "" + req_num = (doc.get("request_number") or "").strip() + doc_scope = (doc.get("search_scope") or "").strip() + if req_num and _is_subsequent_page(page_info_str): + is_subsequent_page = True + if doc_scope: + sibling_doc = conn.execute( + """SELECT * FROM documents + WHERE request_number=? AND search_scope=? AND id != ? + ORDER BY CASE WHEN page_number=1 THEN 0 ELSE 1 END, + CASE WHEN person_id IS NOT NULL THEN 0 ELSE 1 END, + id + LIMIT 1""", + (req_num, doc_scope, doc["id"]), + ).fetchone() + + return is_subsequent_page, dict(sibling_doc) if sibling_doc else None + + +def _build_correlation_context(doc: dict) -> dict: + person = doc.get("person") or {} + return { + "document_id": doc.get("id"), + "request_number": doc.get("request_number"), + "request_date": doc.get("request_date"), + "page_info": doc.get("page_info"), + "page_number": doc.get("page_number"), + "search_scope": doc.get("search_scope"), + "applicant_name_raw": doc.get("applicant_name_raw"), + "person": { + "first_name": person.get("first_name"), + "father_name": person.get("father_name"), + "mother_name": person.get("mother_name"), + "family_name": person.get("family_name"), + "registry_number": person.get("registry_number"), + "registry_place": person.get("registry_place"), + }, + } + + def _get_document(doc_id: int) -> dict | None: with get_db() as conn: row = conn.execute("SELECT * FROM documents WHERE id=?", (doc_id,)).fetchone() @@ -66,58 +125,24 @@ def _get_document(doc_id: int) -> dict | None: # Works for both PDF splits (pdf_group_id) and individually uploaded images (via request_number + page_info). doc["inherited_from_page1"] = False doc["page1_person_id"] = None + doc["page1_doc_id"] = None current_first = (doc["person"].get("first_name") or "").strip() - is_subsequent_page = False - sibling_doc = None + is_subsequent_page, sibling_doc = _find_page1_candidate(conn, doc) - # Method 1: PDF group link (reliable, unique grouping) - if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1: - is_subsequent_page = True - sibling_doc = conn.execute( - """SELECT * FROM documents - WHERE pdf_group_id=? AND page_number=1""", - (doc["pdf_group_id"],), - ).fetchone() - - # Method 2: Same request_number + search_scope + page_info indicates page 2+ - # MUST match on BOTH request_number AND search_scope to avoid cross-person contamination - if not sibling_doc and not current_first: - page_info_str = doc.get("page_info") or "" - req_num = (doc.get("request_number") or "").strip() - doc_scope = (doc.get("search_scope") or "").strip() - if req_num and _is_subsequent_page(page_info_str): - is_subsequent_page = True - if doc_scope: - # Match by request_number + search_scope (safe) - sibling_doc = conn.execute( - """SELECT * FROM documents - WHERE request_number=? AND search_scope=? AND id != ? - AND person_id IS NOT NULL - ORDER BY id LIMIT 1""", - (req_num, doc_scope, doc["id"]), - ).fetchone() - if not sibling_doc: - sibling_doc = conn.execute( - """SELECT * FROM documents - WHERE request_number=? AND search_scope=? AND id != ? - AND raw_extraction_json IS NOT NULL - ORDER BY id LIMIT 1""", - (req_num, doc_scope, doc["id"]), - ).fetchone() - # If no search_scope on this doc, DON'T guess — leave it for manual entry - - if sibling_doc and not current_first: - sibling_doc = dict(sibling_doc) + if sibling_doc: + doc["page1_doc_id"] = sibling_doc.get("id") if sibling_doc.get("person_id"): + doc["page1_person_id"] = sibling_doc["person_id"] + + if sibling_doc.get("person_id") and not current_first: person = conn.execute( "SELECT * FROM persons WHERE id=?", (sibling_doc["person_id"],) ).fetchone() if person: doc["person"] = dict(person) doc["inherited_from_page1"] = True - doc["page1_person_id"] = sibling_doc["person_id"] - elif sibling_doc.get("raw_extraction_json"): + elif sibling_doc.get("raw_extraction_json") and not current_first: try: sib_extracted = json.loads(sibling_doc["raw_extraction_json"]) sib_person = sib_extracted.get("person", {}) @@ -282,7 +307,64 @@ async def review_document(request: Request, doc_id: int, wait: int = 0): ) return templates.TemplateResponse( - request, "review.html", {"doc": doc, "upload_dir": "/uploads"} + request, + "review.html", + { + "doc": doc, + "upload_dir": "/uploads", + "providers": get_available_providers(), + "current_provider": doc.get("provider") or get_default_provider(), + "ai_verification_available": any( + provider["id"] in {"claude", "gemini"} for provider in get_available_providers() + ), + }, + ) + + +@router.post("/review/{doc_id}/verify-correlation") +async def review_verify_correlation(doc_id: int): + with get_db() as conn: + row = conn.execute("SELECT * FROM documents WHERE id=?", (doc_id,)).fetchone() + if not row: + return JSONResponse({"error": "Document not found"}, status_code=404) + + current_doc = _get_document(doc_id) + if not current_doc: + return JSONResponse({"error": "Document not found"}, status_code=404) + + is_subsequent_page, sibling_doc = _find_page1_candidate(conn, dict(row)) + if not is_subsequent_page: + return JSONResponse( + {"error": "This document is not identified as page 2 or later."}, + status_code=400, + ) + if not sibling_doc: + return JSONResponse( + {"error": "No page 1 candidate was found for AI verification."}, + status_code=404, + ) + + candidate_doc = _get_document(sibling_doc["id"]) + if not candidate_doc: + return JSONResponse({"error": "Candidate page not found"}, status_code=404) + + try: + result = await verify_page_correlation( + current_doc["image_path"], + candidate_doc["image_path"], + _build_correlation_context(current_doc), + _build_correlation_context(candidate_doc), + ) + except Exception as exc: + return JSONResponse({"error": str(exc)}, status_code=400) + + return JSONResponse( + { + "ok": True, + "current_doc_id": current_doc["id"], + "candidate_doc_id": candidate_doc["id"], + **result, + } ) @@ -353,44 +435,9 @@ async def confirm_document(doc_id: int, request: Request): # For multi-page documents (page 2+), resolve sibling document's person # Works via pdf_group_id OR request_number+search_scope + page_info page1_person_id = None - is_subsequent = ( - current_doc.get("pdf_group_id") - and (current_doc.get("page_number") or 0) > 1 - ) - - # Method 1: PDF group link - if is_subsequent: - page1_doc = conn.execute( - """SELECT d.person_id, p.first_name, p.father_name, p.family_name, - p.registry_number - FROM documents d - LEFT JOIN persons p ON p.id = d.person_id - WHERE d.pdf_group_id=? AND d.page_number=1 AND d.person_id IS NOT NULL""", - (current_doc["pdf_group_id"],), - ).fetchone() - if page1_doc and page1_doc["person_id"]: - page1_person_id = page1_doc["person_id"] - - # Method 2: Same request_number + search_scope + page_info indicates page 2+ - if not page1_person_id and not first_name: - req_num = (current_doc.get("request_number") or "").strip() - page_info_val = current_doc.get("page_info") or "" - doc_scope = (current_doc.get("search_scope") or "").strip() - if req_num and _is_subsequent_page(page_info_val): - is_subsequent = True - if doc_scope: - sibling = conn.execute( - """SELECT d.person_id, p.first_name, p.father_name, p.family_name, - p.registry_number - FROM documents d - LEFT JOIN persons p ON p.id = d.person_id - WHERE d.request_number=? AND d.search_scope=? AND d.id != ? - AND d.person_id IS NOT NULL - ORDER BY d.id LIMIT 1""", - (req_num, doc_scope, doc_id), - ).fetchone() - if sibling and sibling["person_id"]: - page1_person_id = sibling["person_id"] + is_subsequent, sibling = _find_page1_candidate(conn, current_doc) + if sibling and sibling.get("person_id"): + page1_person_id = sibling["person_id"] # Inherit person fields from sibling if not provided if page1_person_id and not first_name: @@ -658,7 +705,7 @@ async def confirm_document(doc_id: int, request: Request): @router.post("/extract/{doc_id}") async def retrigger_extraction(doc_id: int, provider: str = ""): - """Re-run extraction for a document (retry after error).""" + """Re-run extraction for a document and reset the extracted state first.""" with get_db() as conn: doc = conn.execute( "SELECT image_path, provider FROM documents WHERE id=?", (doc_id,) @@ -666,14 +713,32 @@ async def retrigger_extraction(doc_id: int, provider: str = ""): if not doc: return JSONResponse({"error": "not found"}, status_code=404) + use_provider = provider or doc["provider"] or get_default_provider() + with get_db() as conn: conn.execute( - "UPDATE documents SET status='pending', extraction_error=NULL WHERE id=?", - (doc_id,), + """UPDATE documents + SET status='pending', + person_id=NULL, + request_number=NULL, + request_date=NULL, + applicant_name_raw=NULL, + request_purpose=NULL, + data_valid_until=NULL, + registry_office=NULL, + owns_properties=NULL, + declared_property_count=NULL, + page_info=NULL, + search_scope=NULL, + raw_extraction_json=NULL, + extraction_error=NULL, + provider=?, + updated_at=CURRENT_TIMESTAMP + WHERE id=?""", + (use_provider, doc_id), ) conn.execute("DELETE FROM properties WHERE document_id=?", (doc_id,)) from routers.upload import _extract_and_save - use_provider = provider or doc["provider"] or "" asyncio.create_task(_extract_and_save(doc_id, doc["image_path"], use_provider)) return JSONResponse({"ok": True, "message": "Extraction started"}) diff --git a/services/extractor.py b/services/extractor.py index 4b0107f..daff2be 100644 --- a/services/extractor.py +++ b/services/extractor.py @@ -28,18 +28,19 @@ CRITICAL INSTRUCTIONS: - معلومات محولة لغاية (data_valid_until) found near the bottom. - أمانة السجل (registry_office) found at the bottom right. - عدد العقارات: (declared_property_count) located right under the properties table. -7. The properties table has 8 columns in order from RIGHT to LEFT as they appear on the page: +7. If the property owner is a company, religious entity, or organization (e.g., شركة, وقف, مطرانية, جمعية) rather than a natural person, extract its full name into the `first_name` field and leave `father_name`, `mother_name`, `family_name`, etc. as null. +8. The properties table has 8 columns in order from RIGHT to LEFT as they appear on the page: col1(rightmost)=اسم الفريق, col2=رقم العقار, col3=القسم, col4=البلوك, col5=المنطقة العقارية, col6=القضاء, col7=عدد الأسهم, col8(leftmost)=نوع الملكية Map these to JSON keys: party_name, property_number, section, block, real_estate_district, qaza, num_shares, ownership_type -8. Return ONLY valid JSON matching the schema. No markdown, no explanation.""" +9. Return ONLY valid JSON matching the schema. No markdown, no explanation.""" USER_PROMPT = ( "Extract all data from this Lebanese real estate property card. " "Return a single JSON object with these keys: " "request_number, request_date, applicant_name_raw, request_purpose, data_valid_until, " "registry_office, page_info, search_scope, owns_properties, declared_property_count, " - "person (object with: first_name, father_name, mother_name, family_name, " + "person (object with: first_name (or company/entity name), father_name, mother_name, family_name, " "family_origin, nationality, birth_date, registry_number, registry_place), " "properties (array of objects with: party_name, property_number, section, block, " "real_estate_district, qaza, num_shares, ownership_type), " @@ -47,6 +48,149 @@ USER_PROMPT = ( "Use null for missing fields. Include every property row from the table (or empty array if none)." ) +CORRELATION_SYSTEM_PROMPT = """You are verifying whether two scanned Lebanese real estate document pages belong to the same multi-page request and the same person. + +Rules: +1. Compare both page images and the extracted metadata together. +2. Strong signals: matching request number, matching search scope, matching page numbering in the same sequence, matching applicant/person names, matching footer/header identifiers, and obvious continuation of the same document layout. +3. Weak OCR differences are common in Arabic letters such as س and ن, ب and ت, or Arabic-Indic digits. Do not reject a match solely because of one likely OCR confusion. +4. Reject when there are clear contradictions in request number, search scope, person identity, or unrelated page numbering. +5. Return JSON only. +""" + + +def _get_ai_verification_provider(preferred_provider: str = "") -> str: + """Return a provider capable of vision reasoning for verification.""" + providers = [p["id"] for p in get_available_providers() if p["id"] in {"claude", "gemini"}] + if preferred_provider in providers: + return preferred_provider + if DEFAULT_PROVIDER in providers: + return DEFAULT_PROVIDER + if providers: + return providers[0] + raise ValueError("No AI verification provider configured") + + +def _build_correlation_user_prompt(current_context: dict, candidate_context: dict) -> str: + return ( + "Determine whether PAGE_A and PAGE_B belong to the same multi-page request/document for the same person. " + "Treat minor OCR mistakes as possible noise. Return one JSON object with keys: " + "same_document (boolean), confidence ('high'|'medium'|'low'), verdict_ar (short Arabic sentence), " + "reasons_ar (array of short Arabic bullet strings), mismatch_flags (array of short Arabic strings), " + "recommended_action ('auto-link'|'manual-review').\n\n" + f"PAGE_A_CONTEXT={json.dumps(current_context, ensure_ascii=False)}\n" + f"PAGE_B_CONTEXT={json.dumps(candidate_context, ensure_ascii=False)}" + ) + + +async def verify_page_correlation( + current_image_path: str, + candidate_image_path: str, + current_context: dict, + candidate_context: dict, + provider: str = "", +) -> dict: + """Ask a vision model if two pages belong to the same multi-page request.""" + provider = _get_ai_verification_provider(provider) + + if provider == "claude": + import anthropic + + client = anthropic.AsyncAnthropic(api_key=ANTHROPIC_API_KEY) + current_full_path = _resolve_path(current_image_path) + candidate_full_path = _resolve_path(candidate_image_path) + current_img_data, current_media_type = _encode_image(current_full_path) + candidate_img_data, candidate_media_type = _encode_image(candidate_full_path) + + message = await client.messages.create( + model=CLAUDE_MODEL, + max_tokens=1200, + system=[{"type": "text", "text": CORRELATION_SYSTEM_PROMPT}], + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": _build_correlation_user_prompt(current_context, candidate_context)}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": current_media_type, + "data": current_img_data, + }, + }, + { + "type": "image", + "source": { + "type": "base64", + "media_type": candidate_media_type, + "data": candidate_img_data, + }, + }, + ], + } + ], + ) + raw_text = _strip_code_fences(message.content[0].text) + result = json.loads(raw_text) + elif provider == "gemini": + import asyncio + from google import genai + from google.genai import types + + client = genai.Client(api_key=GEMINI_API_KEY) + current_full_path = _resolve_path(current_image_path) + candidate_full_path = _resolve_path(candidate_image_path) + + with open(current_full_path, "rb") as f: + current_bytes = f.read() + with open(candidate_full_path, "rb") as f: + candidate_bytes = f.read() + + mime_map = { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".png": "image/png", + ".webp": "image/webp", + } + current_mime = mime_map.get(Path(current_full_path).suffix.lower(), "image/jpeg") + candidate_mime = mime_map.get(Path(candidate_full_path).suffix.lower(), "image/jpeg") + + def _call(): + response = client.models.generate_content( + model=GEMINI_MODEL, + contents=[ + types.Content( + parts=[ + types.Part(text=CORRELATION_SYSTEM_PROMPT + "\n\n" + _build_correlation_user_prompt(current_context, candidate_context)), + types.Part(inline_data=types.Blob(mime_type=current_mime, data=current_bytes)), + types.Part(inline_data=types.Blob(mime_type=candidate_mime, data=candidate_bytes)), + ] + ) + ], + config=types.GenerateContentConfig( + temperature=0.1, + max_output_tokens=1600, + ), + ) + return response.text + + raw_text = await asyncio.get_event_loop().run_in_executor(None, _call) + raw_text = _strip_code_fences(raw_text) + result = json.loads(raw_text) + else: + raise ValueError(f"Provider {provider} does not support AI correlation verification") + + return { + "provider": provider, + "same_document": bool(result.get("same_document")), + "confidence": result.get("confidence") or "medium", + "verdict_ar": result.get("verdict_ar") or "تعذر توليد خلاصة واضحة.", + "reasons_ar": result.get("reasons_ar") or [], + "mismatch_flags": result.get("mismatch_flags") or [], + "recommended_action": result.get("recommended_action") or "manual-review", + } + def _resolve_path(image_path: str) -> str: if Path(image_path).is_absolute(): return image_path diff --git a/static/css/review.css b/static/css/review.css index aff368c..3ad1729 100644 --- a/static/css/review.css +++ b/static/css/review.css @@ -75,6 +75,87 @@ line-height: 1.35; } +.page-group-banner { + align-items: flex-start; +} + +.page-group-copy { + display: flex; + flex: 1; + gap: .9rem; + align-items: flex-start; + justify-content: space-between; + flex-wrap: wrap; +} + +.page-group-actions { + margin-inline-start: auto; +} + +.rescan-banner { + justify-content: space-between; + flex-wrap: wrap; +} + +.rescan-copy { + display: inline-block; + margin-inline-start: .35rem; +} + +.rescan-controls { + display: flex; + align-items: center; + gap: .65rem; + margin-inline-start: auto; +} + +.rescan-controls select { + min-width: 220px; + padding: .62rem .8rem; + border: 1px solid rgba(59, 130, 246, .28); + border-radius: 12px; + background: rgba(255,255,255,.96); + color: var(--text); +} + +.sr-only { + position: absolute; + width: 1px; + height: 1px; + padding: 0; + margin: -1px; + overflow: hidden; + clip: rect(0, 0, 0, 0); + white-space: nowrap; + border: 0; +} + +.correlation-result { + margin-bottom: 1rem; +} + +.correlation-heading { + display: flex; + justify-content: space-between; + gap: .75rem; + align-items: baseline; + flex-wrap: wrap; +} + +.correlation-meta { + font-size: .82rem; + opacity: .9; +} + +.correlation-list { + margin: .6rem 0 0; + padding-inline-start: 1.25rem; +} + +.correlation-warnings { + margin-top: .4rem; +} + /* ============================ Review — Responsive ============================ */ @@ -95,6 +176,18 @@ .review-nav .btn { flex: 1 1 calc(50% - .5rem); } + .page-group-actions { + width: 100%; + margin-inline-start: 0; + } + .rescan-controls { + width: 100%; + margin-inline-start: 0; + } + .rescan-controls select, + .rescan-controls .btn { + flex: 1 1 0; + } } @media (max-width: 640px) { @@ -111,6 +204,17 @@ .review-nav .btn { flex-basis: 100%; } + .page-group-actions .btn { + width: 100%; + } + .rescan-controls { + flex-direction: column; + align-items: stretch; + } + .rescan-controls select { + min-width: 0; + width: 100%; + } .form-section { padding: 1rem; } diff --git a/templates/review.html b/templates/review.html index 647c5f2..62df0fe 100644 --- a/templates/review.html +++ b/templates/review.html @@ -24,6 +24,28 @@ +{% if doc.status in ['extracted', 'confirmed', 'error'] %} +
+{% endif %} + {% if doc.status == 'error' %}