Fix extraction for companies and religious entities
This commit is contained in:
+150
-85
@@ -8,7 +8,12 @@ from fastapi.templating import Jinja2Templates
|
||||
|
||||
from config import UPLOAD_DIR
|
||||
from database.connection import get_db
|
||||
from services.extractor import extract_document
|
||||
from services.extractor import (
|
||||
extract_document,
|
||||
get_available_providers,
|
||||
get_default_provider,
|
||||
verify_page_correlation,
|
||||
)
|
||||
from services.search_service import normalize_arabic, _normalize_scope
|
||||
|
||||
router = APIRouter()
|
||||
@@ -36,6 +41,60 @@ def _is_subsequent_page(page_info: str) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _find_page1_candidate(conn, doc: dict) -> tuple[bool, dict | None]:
|
||||
"""Return whether the doc looks like a later page and the best page-1 candidate."""
|
||||
is_subsequent_page = False
|
||||
sibling_doc = None
|
||||
|
||||
if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1:
|
||||
is_subsequent_page = True
|
||||
sibling_doc = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE pdf_group_id=? AND page_number=1""",
|
||||
(doc["pdf_group_id"],),
|
||||
).fetchone()
|
||||
|
||||
if not sibling_doc:
|
||||
page_info_str = doc.get("page_info") or ""
|
||||
req_num = (doc.get("request_number") or "").strip()
|
||||
doc_scope = (doc.get("search_scope") or "").strip()
|
||||
if req_num and _is_subsequent_page(page_info_str):
|
||||
is_subsequent_page = True
|
||||
if doc_scope:
|
||||
sibling_doc = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE request_number=? AND search_scope=? AND id != ?
|
||||
ORDER BY CASE WHEN page_number=1 THEN 0 ELSE 1 END,
|
||||
CASE WHEN person_id IS NOT NULL THEN 0 ELSE 1 END,
|
||||
id
|
||||
LIMIT 1""",
|
||||
(req_num, doc_scope, doc["id"]),
|
||||
).fetchone()
|
||||
|
||||
return is_subsequent_page, dict(sibling_doc) if sibling_doc else None
|
||||
|
||||
|
||||
def _build_correlation_context(doc: dict) -> dict:
|
||||
person = doc.get("person") or {}
|
||||
return {
|
||||
"document_id": doc.get("id"),
|
||||
"request_number": doc.get("request_number"),
|
||||
"request_date": doc.get("request_date"),
|
||||
"page_info": doc.get("page_info"),
|
||||
"page_number": doc.get("page_number"),
|
||||
"search_scope": doc.get("search_scope"),
|
||||
"applicant_name_raw": doc.get("applicant_name_raw"),
|
||||
"person": {
|
||||
"first_name": person.get("first_name"),
|
||||
"father_name": person.get("father_name"),
|
||||
"mother_name": person.get("mother_name"),
|
||||
"family_name": person.get("family_name"),
|
||||
"registry_number": person.get("registry_number"),
|
||||
"registry_place": person.get("registry_place"),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _get_document(doc_id: int) -> dict | None:
|
||||
with get_db() as conn:
|
||||
row = conn.execute("SELECT * FROM documents WHERE id=?", (doc_id,)).fetchone()
|
||||
@@ -66,58 +125,24 @@ def _get_document(doc_id: int) -> dict | None:
|
||||
# Works for both PDF splits (pdf_group_id) and individually uploaded images (via request_number + page_info).
|
||||
doc["inherited_from_page1"] = False
|
||||
doc["page1_person_id"] = None
|
||||
doc["page1_doc_id"] = None
|
||||
|
||||
current_first = (doc["person"].get("first_name") or "").strip()
|
||||
is_subsequent_page = False
|
||||
sibling_doc = None
|
||||
is_subsequent_page, sibling_doc = _find_page1_candidate(conn, doc)
|
||||
|
||||
# Method 1: PDF group link (reliable, unique grouping)
|
||||
if doc.get("pdf_group_id") and (doc.get("page_number") or 0) > 1:
|
||||
is_subsequent_page = True
|
||||
sibling_doc = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE pdf_group_id=? AND page_number=1""",
|
||||
(doc["pdf_group_id"],),
|
||||
).fetchone()
|
||||
|
||||
# Method 2: Same request_number + search_scope + page_info indicates page 2+
|
||||
# MUST match on BOTH request_number AND search_scope to avoid cross-person contamination
|
||||
if not sibling_doc and not current_first:
|
||||
page_info_str = doc.get("page_info") or ""
|
||||
req_num = (doc.get("request_number") or "").strip()
|
||||
doc_scope = (doc.get("search_scope") or "").strip()
|
||||
if req_num and _is_subsequent_page(page_info_str):
|
||||
is_subsequent_page = True
|
||||
if doc_scope:
|
||||
# Match by request_number + search_scope (safe)
|
||||
sibling_doc = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE request_number=? AND search_scope=? AND id != ?
|
||||
AND person_id IS NOT NULL
|
||||
ORDER BY id LIMIT 1""",
|
||||
(req_num, doc_scope, doc["id"]),
|
||||
).fetchone()
|
||||
if not sibling_doc:
|
||||
sibling_doc = conn.execute(
|
||||
"""SELECT * FROM documents
|
||||
WHERE request_number=? AND search_scope=? AND id != ?
|
||||
AND raw_extraction_json IS NOT NULL
|
||||
ORDER BY id LIMIT 1""",
|
||||
(req_num, doc_scope, doc["id"]),
|
||||
).fetchone()
|
||||
# If no search_scope on this doc, DON'T guess — leave it for manual entry
|
||||
|
||||
if sibling_doc and not current_first:
|
||||
sibling_doc = dict(sibling_doc)
|
||||
if sibling_doc:
|
||||
doc["page1_doc_id"] = sibling_doc.get("id")
|
||||
if sibling_doc.get("person_id"):
|
||||
doc["page1_person_id"] = sibling_doc["person_id"]
|
||||
|
||||
if sibling_doc.get("person_id") and not current_first:
|
||||
person = conn.execute(
|
||||
"SELECT * FROM persons WHERE id=?", (sibling_doc["person_id"],)
|
||||
).fetchone()
|
||||
if person:
|
||||
doc["person"] = dict(person)
|
||||
doc["inherited_from_page1"] = True
|
||||
doc["page1_person_id"] = sibling_doc["person_id"]
|
||||
elif sibling_doc.get("raw_extraction_json"):
|
||||
elif sibling_doc.get("raw_extraction_json") and not current_first:
|
||||
try:
|
||||
sib_extracted = json.loads(sibling_doc["raw_extraction_json"])
|
||||
sib_person = sib_extracted.get("person", {})
|
||||
@@ -282,7 +307,64 @@ async def review_document(request: Request, doc_id: int, wait: int = 0):
|
||||
)
|
||||
|
||||
return templates.TemplateResponse(
|
||||
request, "review.html", {"doc": doc, "upload_dir": "/uploads"}
|
||||
request,
|
||||
"review.html",
|
||||
{
|
||||
"doc": doc,
|
||||
"upload_dir": "/uploads",
|
||||
"providers": get_available_providers(),
|
||||
"current_provider": doc.get("provider") or get_default_provider(),
|
||||
"ai_verification_available": any(
|
||||
provider["id"] in {"claude", "gemini"} for provider in get_available_providers()
|
||||
),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@router.post("/review/{doc_id}/verify-correlation")
|
||||
async def review_verify_correlation(doc_id: int):
|
||||
with get_db() as conn:
|
||||
row = conn.execute("SELECT * FROM documents WHERE id=?", (doc_id,)).fetchone()
|
||||
if not row:
|
||||
return JSONResponse({"error": "Document not found"}, status_code=404)
|
||||
|
||||
current_doc = _get_document(doc_id)
|
||||
if not current_doc:
|
||||
return JSONResponse({"error": "Document not found"}, status_code=404)
|
||||
|
||||
is_subsequent_page, sibling_doc = _find_page1_candidate(conn, dict(row))
|
||||
if not is_subsequent_page:
|
||||
return JSONResponse(
|
||||
{"error": "This document is not identified as page 2 or later."},
|
||||
status_code=400,
|
||||
)
|
||||
if not sibling_doc:
|
||||
return JSONResponse(
|
||||
{"error": "No page 1 candidate was found for AI verification."},
|
||||
status_code=404,
|
||||
)
|
||||
|
||||
candidate_doc = _get_document(sibling_doc["id"])
|
||||
if not candidate_doc:
|
||||
return JSONResponse({"error": "Candidate page not found"}, status_code=404)
|
||||
|
||||
try:
|
||||
result = await verify_page_correlation(
|
||||
current_doc["image_path"],
|
||||
candidate_doc["image_path"],
|
||||
_build_correlation_context(current_doc),
|
||||
_build_correlation_context(candidate_doc),
|
||||
)
|
||||
except Exception as exc:
|
||||
return JSONResponse({"error": str(exc)}, status_code=400)
|
||||
|
||||
return JSONResponse(
|
||||
{
|
||||
"ok": True,
|
||||
"current_doc_id": current_doc["id"],
|
||||
"candidate_doc_id": candidate_doc["id"],
|
||||
**result,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@@ -353,43 +435,8 @@ async def confirm_document(doc_id: int, request: Request):
|
||||
# For multi-page documents (page 2+), resolve sibling document's person
|
||||
# Works via pdf_group_id OR request_number+search_scope + page_info
|
||||
page1_person_id = None
|
||||
is_subsequent = (
|
||||
current_doc.get("pdf_group_id")
|
||||
and (current_doc.get("page_number") or 0) > 1
|
||||
)
|
||||
|
||||
# Method 1: PDF group link
|
||||
if is_subsequent:
|
||||
page1_doc = conn.execute(
|
||||
"""SELECT d.person_id, p.first_name, p.father_name, p.family_name,
|
||||
p.registry_number
|
||||
FROM documents d
|
||||
LEFT JOIN persons p ON p.id = d.person_id
|
||||
WHERE d.pdf_group_id=? AND d.page_number=1 AND d.person_id IS NOT NULL""",
|
||||
(current_doc["pdf_group_id"],),
|
||||
).fetchone()
|
||||
if page1_doc and page1_doc["person_id"]:
|
||||
page1_person_id = page1_doc["person_id"]
|
||||
|
||||
# Method 2: Same request_number + search_scope + page_info indicates page 2+
|
||||
if not page1_person_id and not first_name:
|
||||
req_num = (current_doc.get("request_number") or "").strip()
|
||||
page_info_val = current_doc.get("page_info") or ""
|
||||
doc_scope = (current_doc.get("search_scope") or "").strip()
|
||||
if req_num and _is_subsequent_page(page_info_val):
|
||||
is_subsequent = True
|
||||
if doc_scope:
|
||||
sibling = conn.execute(
|
||||
"""SELECT d.person_id, p.first_name, p.father_name, p.family_name,
|
||||
p.registry_number
|
||||
FROM documents d
|
||||
LEFT JOIN persons p ON p.id = d.person_id
|
||||
WHERE d.request_number=? AND d.search_scope=? AND d.id != ?
|
||||
AND d.person_id IS NOT NULL
|
||||
ORDER BY d.id LIMIT 1""",
|
||||
(req_num, doc_scope, doc_id),
|
||||
).fetchone()
|
||||
if sibling and sibling["person_id"]:
|
||||
is_subsequent, sibling = _find_page1_candidate(conn, current_doc)
|
||||
if sibling and sibling.get("person_id"):
|
||||
page1_person_id = sibling["person_id"]
|
||||
|
||||
# Inherit person fields from sibling if not provided
|
||||
@@ -658,7 +705,7 @@ async def confirm_document(doc_id: int, request: Request):
|
||||
|
||||
@router.post("/extract/{doc_id}")
|
||||
async def retrigger_extraction(doc_id: int, provider: str = ""):
|
||||
"""Re-run extraction for a document (retry after error)."""
|
||||
"""Re-run extraction for a document and reset the extracted state first."""
|
||||
with get_db() as conn:
|
||||
doc = conn.execute(
|
||||
"SELECT image_path, provider FROM documents WHERE id=?", (doc_id,)
|
||||
@@ -666,14 +713,32 @@ async def retrigger_extraction(doc_id: int, provider: str = ""):
|
||||
if not doc:
|
||||
return JSONResponse({"error": "not found"}, status_code=404)
|
||||
|
||||
use_provider = provider or doc["provider"] or get_default_provider()
|
||||
|
||||
with get_db() as conn:
|
||||
conn.execute(
|
||||
"UPDATE documents SET status='pending', extraction_error=NULL WHERE id=?",
|
||||
(doc_id,),
|
||||
"""UPDATE documents
|
||||
SET status='pending',
|
||||
person_id=NULL,
|
||||
request_number=NULL,
|
||||
request_date=NULL,
|
||||
applicant_name_raw=NULL,
|
||||
request_purpose=NULL,
|
||||
data_valid_until=NULL,
|
||||
registry_office=NULL,
|
||||
owns_properties=NULL,
|
||||
declared_property_count=NULL,
|
||||
page_info=NULL,
|
||||
search_scope=NULL,
|
||||
raw_extraction_json=NULL,
|
||||
extraction_error=NULL,
|
||||
provider=?,
|
||||
updated_at=CURRENT_TIMESTAMP
|
||||
WHERE id=?""",
|
||||
(use_provider, doc_id),
|
||||
)
|
||||
conn.execute("DELETE FROM properties WHERE document_id=?", (doc_id,))
|
||||
|
||||
from routers.upload import _extract_and_save
|
||||
use_provider = provider or doc["provider"] or ""
|
||||
asyncio.create_task(_extract_and_save(doc_id, doc["image_path"], use_provider))
|
||||
return JSONResponse({"ok": True, "message": "Extraction started"})
|
||||
|
||||
+147
-3
@@ -28,18 +28,19 @@ CRITICAL INSTRUCTIONS:
|
||||
- معلومات محولة لغاية (data_valid_until) found near the bottom.
|
||||
- أمانة السجل (registry_office) found at the bottom right.
|
||||
- عدد العقارات: (declared_property_count) located right under the properties table.
|
||||
7. The properties table has 8 columns in order from RIGHT to LEFT as they appear on the page:
|
||||
7. If the property owner is a company, religious entity, or organization (e.g., شركة, وقف, مطرانية, جمعية) rather than a natural person, extract its full name into the `first_name` field and leave `father_name`, `mother_name`, `family_name`, etc. as null.
|
||||
8. The properties table has 8 columns in order from RIGHT to LEFT as they appear on the page:
|
||||
col1(rightmost)=اسم الفريق, col2=رقم العقار, col3=القسم, col4=البلوك,
|
||||
col5=المنطقة العقارية, col6=القضاء, col7=عدد الأسهم, col8(leftmost)=نوع الملكية
|
||||
Map these to JSON keys: party_name, property_number, section, block, real_estate_district, qaza, num_shares, ownership_type
|
||||
8. Return ONLY valid JSON matching the schema. No markdown, no explanation."""
|
||||
9. Return ONLY valid JSON matching the schema. No markdown, no explanation."""
|
||||
|
||||
USER_PROMPT = (
|
||||
"Extract all data from this Lebanese real estate property card. "
|
||||
"Return a single JSON object with these keys: "
|
||||
"request_number, request_date, applicant_name_raw, request_purpose, data_valid_until, "
|
||||
"registry_office, page_info, search_scope, owns_properties, declared_property_count, "
|
||||
"person (object with: first_name, father_name, mother_name, family_name, "
|
||||
"person (object with: first_name (or company/entity name), father_name, mother_name, family_name, "
|
||||
"family_origin, nationality, birth_date, registry_number, registry_place), "
|
||||
"properties (array of objects with: party_name, property_number, section, block, "
|
||||
"real_estate_district, qaza, num_shares, ownership_type), "
|
||||
@@ -47,6 +48,149 @@ USER_PROMPT = (
|
||||
"Use null for missing fields. Include every property row from the table (or empty array if none)."
|
||||
)
|
||||
|
||||
CORRELATION_SYSTEM_PROMPT = """You are verifying whether two scanned Lebanese real estate document pages belong to the same multi-page request and the same person.
|
||||
|
||||
Rules:
|
||||
1. Compare both page images and the extracted metadata together.
|
||||
2. Strong signals: matching request number, matching search scope, matching page numbering in the same sequence, matching applicant/person names, matching footer/header identifiers, and obvious continuation of the same document layout.
|
||||
3. Weak OCR differences are common in Arabic letters such as س and ن, ب and ت, or Arabic-Indic digits. Do not reject a match solely because of one likely OCR confusion.
|
||||
4. Reject when there are clear contradictions in request number, search scope, person identity, or unrelated page numbering.
|
||||
5. Return JSON only.
|
||||
"""
|
||||
|
||||
|
||||
def _get_ai_verification_provider(preferred_provider: str = "") -> str:
|
||||
"""Return a provider capable of vision reasoning for verification."""
|
||||
providers = [p["id"] for p in get_available_providers() if p["id"] in {"claude", "gemini"}]
|
||||
if preferred_provider in providers:
|
||||
return preferred_provider
|
||||
if DEFAULT_PROVIDER in providers:
|
||||
return DEFAULT_PROVIDER
|
||||
if providers:
|
||||
return providers[0]
|
||||
raise ValueError("No AI verification provider configured")
|
||||
|
||||
|
||||
def _build_correlation_user_prompt(current_context: dict, candidate_context: dict) -> str:
|
||||
return (
|
||||
"Determine whether PAGE_A and PAGE_B belong to the same multi-page request/document for the same person. "
|
||||
"Treat minor OCR mistakes as possible noise. Return one JSON object with keys: "
|
||||
"same_document (boolean), confidence ('high'|'medium'|'low'), verdict_ar (short Arabic sentence), "
|
||||
"reasons_ar (array of short Arabic bullet strings), mismatch_flags (array of short Arabic strings), "
|
||||
"recommended_action ('auto-link'|'manual-review').\n\n"
|
||||
f"PAGE_A_CONTEXT={json.dumps(current_context, ensure_ascii=False)}\n"
|
||||
f"PAGE_B_CONTEXT={json.dumps(candidate_context, ensure_ascii=False)}"
|
||||
)
|
||||
|
||||
|
||||
async def verify_page_correlation(
|
||||
current_image_path: str,
|
||||
candidate_image_path: str,
|
||||
current_context: dict,
|
||||
candidate_context: dict,
|
||||
provider: str = "",
|
||||
) -> dict:
|
||||
"""Ask a vision model if two pages belong to the same multi-page request."""
|
||||
provider = _get_ai_verification_provider(provider)
|
||||
|
||||
if provider == "claude":
|
||||
import anthropic
|
||||
|
||||
client = anthropic.AsyncAnthropic(api_key=ANTHROPIC_API_KEY)
|
||||
current_full_path = _resolve_path(current_image_path)
|
||||
candidate_full_path = _resolve_path(candidate_image_path)
|
||||
current_img_data, current_media_type = _encode_image(current_full_path)
|
||||
candidate_img_data, candidate_media_type = _encode_image(candidate_full_path)
|
||||
|
||||
message = await client.messages.create(
|
||||
model=CLAUDE_MODEL,
|
||||
max_tokens=1200,
|
||||
system=[{"type": "text", "text": CORRELATION_SYSTEM_PROMPT}],
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": _build_correlation_user_prompt(current_context, candidate_context)},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": current_media_type,
|
||||
"data": current_img_data,
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": candidate_media_type,
|
||||
"data": candidate_img_data,
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
],
|
||||
)
|
||||
raw_text = _strip_code_fences(message.content[0].text)
|
||||
result = json.loads(raw_text)
|
||||
elif provider == "gemini":
|
||||
import asyncio
|
||||
from google import genai
|
||||
from google.genai import types
|
||||
|
||||
client = genai.Client(api_key=GEMINI_API_KEY)
|
||||
current_full_path = _resolve_path(current_image_path)
|
||||
candidate_full_path = _resolve_path(candidate_image_path)
|
||||
|
||||
with open(current_full_path, "rb") as f:
|
||||
current_bytes = f.read()
|
||||
with open(candidate_full_path, "rb") as f:
|
||||
candidate_bytes = f.read()
|
||||
|
||||
mime_map = {
|
||||
".jpg": "image/jpeg",
|
||||
".jpeg": "image/jpeg",
|
||||
".png": "image/png",
|
||||
".webp": "image/webp",
|
||||
}
|
||||
current_mime = mime_map.get(Path(current_full_path).suffix.lower(), "image/jpeg")
|
||||
candidate_mime = mime_map.get(Path(candidate_full_path).suffix.lower(), "image/jpeg")
|
||||
|
||||
def _call():
|
||||
response = client.models.generate_content(
|
||||
model=GEMINI_MODEL,
|
||||
contents=[
|
||||
types.Content(
|
||||
parts=[
|
||||
types.Part(text=CORRELATION_SYSTEM_PROMPT + "\n\n" + _build_correlation_user_prompt(current_context, candidate_context)),
|
||||
types.Part(inline_data=types.Blob(mime_type=current_mime, data=current_bytes)),
|
||||
types.Part(inline_data=types.Blob(mime_type=candidate_mime, data=candidate_bytes)),
|
||||
]
|
||||
)
|
||||
],
|
||||
config=types.GenerateContentConfig(
|
||||
temperature=0.1,
|
||||
max_output_tokens=1600,
|
||||
),
|
||||
)
|
||||
return response.text
|
||||
|
||||
raw_text = await asyncio.get_event_loop().run_in_executor(None, _call)
|
||||
raw_text = _strip_code_fences(raw_text)
|
||||
result = json.loads(raw_text)
|
||||
else:
|
||||
raise ValueError(f"Provider {provider} does not support AI correlation verification")
|
||||
|
||||
return {
|
||||
"provider": provider,
|
||||
"same_document": bool(result.get("same_document")),
|
||||
"confidence": result.get("confidence") or "medium",
|
||||
"verdict_ar": result.get("verdict_ar") or "تعذر توليد خلاصة واضحة.",
|
||||
"reasons_ar": result.get("reasons_ar") or [],
|
||||
"mismatch_flags": result.get("mismatch_flags") or [],
|
||||
"recommended_action": result.get("recommended_action") or "manual-review",
|
||||
}
|
||||
|
||||
def _resolve_path(image_path: str) -> str:
|
||||
if Path(image_path).is_absolute():
|
||||
return image_path
|
||||
|
||||
@@ -75,6 +75,87 @@
|
||||
line-height: 1.35;
|
||||
}
|
||||
|
||||
.page-group-banner {
|
||||
align-items: flex-start;
|
||||
}
|
||||
|
||||
.page-group-copy {
|
||||
display: flex;
|
||||
flex: 1;
|
||||
gap: .9rem;
|
||||
align-items: flex-start;
|
||||
justify-content: space-between;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
.page-group-actions {
|
||||
margin-inline-start: auto;
|
||||
}
|
||||
|
||||
.rescan-banner {
|
||||
justify-content: space-between;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
.rescan-copy {
|
||||
display: inline-block;
|
||||
margin-inline-start: .35rem;
|
||||
}
|
||||
|
||||
.rescan-controls {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: .65rem;
|
||||
margin-inline-start: auto;
|
||||
}
|
||||
|
||||
.rescan-controls select {
|
||||
min-width: 220px;
|
||||
padding: .62rem .8rem;
|
||||
border: 1px solid rgba(59, 130, 246, .28);
|
||||
border-radius: 12px;
|
||||
background: rgba(255,255,255,.96);
|
||||
color: var(--text);
|
||||
}
|
||||
|
||||
.sr-only {
|
||||
position: absolute;
|
||||
width: 1px;
|
||||
height: 1px;
|
||||
padding: 0;
|
||||
margin: -1px;
|
||||
overflow: hidden;
|
||||
clip: rect(0, 0, 0, 0);
|
||||
white-space: nowrap;
|
||||
border: 0;
|
||||
}
|
||||
|
||||
.correlation-result {
|
||||
margin-bottom: 1rem;
|
||||
}
|
||||
|
||||
.correlation-heading {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
gap: .75rem;
|
||||
align-items: baseline;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
.correlation-meta {
|
||||
font-size: .82rem;
|
||||
opacity: .9;
|
||||
}
|
||||
|
||||
.correlation-list {
|
||||
margin: .6rem 0 0;
|
||||
padding-inline-start: 1.25rem;
|
||||
}
|
||||
|
||||
.correlation-warnings {
|
||||
margin-top: .4rem;
|
||||
}
|
||||
|
||||
/* ============================
|
||||
Review — Responsive
|
||||
============================ */
|
||||
@@ -95,6 +176,18 @@
|
||||
.review-nav .btn {
|
||||
flex: 1 1 calc(50% - .5rem);
|
||||
}
|
||||
.page-group-actions {
|
||||
width: 100%;
|
||||
margin-inline-start: 0;
|
||||
}
|
||||
.rescan-controls {
|
||||
width: 100%;
|
||||
margin-inline-start: 0;
|
||||
}
|
||||
.rescan-controls select,
|
||||
.rescan-controls .btn {
|
||||
flex: 1 1 0;
|
||||
}
|
||||
}
|
||||
|
||||
@media (max-width: 640px) {
|
||||
@@ -111,6 +204,17 @@
|
||||
.review-nav .btn {
|
||||
flex-basis: 100%;
|
||||
}
|
||||
.page-group-actions .btn {
|
||||
width: 100%;
|
||||
}
|
||||
.rescan-controls {
|
||||
flex-direction: column;
|
||||
align-items: stretch;
|
||||
}
|
||||
.rescan-controls select {
|
||||
min-width: 0;
|
||||
width: 100%;
|
||||
}
|
||||
.form-section {
|
||||
padding: 1rem;
|
||||
}
|
||||
|
||||
+98
-3
@@ -24,6 +24,28 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{% if doc.status in ['extracted', 'confirmed', 'error'] %}
|
||||
<div class="info-banner rescan-banner">
|
||||
<div>
|
||||
<strong>إعادة المسح</strong>
|
||||
<span class="rescan-copy">إذا خلط الذكاء الاصطناعي بين أحرف مثل س و ن، يمكنك إعادة القراءة بنفس المحرك أو بمحرك آخر.</span>
|
||||
</div>
|
||||
<div class="rescan-controls">
|
||||
<label for="providerSelect" class="sr-only">محرك الاستخراج</label>
|
||||
<select id="providerSelect" title="محرك الاستخراج">
|
||||
{% for provider in providers %}
|
||||
<option value="{{ provider.id }}" {{ 'selected' if provider.id == current_provider else '' }}>
|
||||
{{ provider.name }}
|
||||
</option>
|
||||
{% endfor %}
|
||||
</select>
|
||||
<button type="button" class="btn btn-warning btn-sm" id="rescanBtn" onclick="retriggerExtraction({{ doc.id }})">
|
||||
إعادة المسح
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
{% endif %}
|
||||
|
||||
{% if doc.status == 'error' %}
|
||||
<div class="error-banner">
|
||||
⚠ خطأ في الاستخراج: {{ doc.extraction_error }}
|
||||
@@ -57,6 +79,8 @@
|
||||
{% if doc.get('is_subsequent_page') %}
|
||||
<div class="info-banner page-group-banner">
|
||||
<svg width="18" height="18" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><circle cx="12" cy="12" r="10"/><line x1="12" y1="16" x2="12" y2="12"/><line x1="12" y1="8" x2="12.01" y2="8"/></svg>
|
||||
<div class="page-group-copy">
|
||||
<div>
|
||||
هذه صفحة تكميلية ({{ doc.get('page_info') or doc.get('page_number') or '' }}) من مستند متعدد الصفحات.
|
||||
{% if doc.get('inherited_from_page1') %}
|
||||
بيانات الشخص مستوردة تلقائياً من الصفحة الأولى.
|
||||
@@ -64,6 +88,16 @@
|
||||
يرجى التأكد من بيانات الشخص يدوياً — لم يتم العثور على صفحة أولى مؤكدة بعد.
|
||||
{% endif %}
|
||||
</div>
|
||||
{% if doc.get('page1_doc_id') and ai_verification_available %}
|
||||
<div class="page-group-actions">
|
||||
<button type="button" class="btn btn-secondary btn-sm" id="verifyCorrelationBtn" onclick="verifyCorrelation({{ doc.id }})">
|
||||
تحقق بالذكاء الاصطناعي
|
||||
</button>
|
||||
</div>
|
||||
{% endif %}
|
||||
</div>
|
||||
</div>
|
||||
<div id="correlationResult" class="hidden"></div>
|
||||
{% endif %}
|
||||
|
||||
<!-- Person Info -->
|
||||
@@ -418,9 +452,70 @@ async function confirmDocument() {
|
||||
}
|
||||
|
||||
async function retriggerExtraction(docId) {
|
||||
if (!confirm('إعادة استخراج البيانات بالذكاء الاصطناعي؟')) return;
|
||||
await fetch(`/extract/${docId}`, { method: 'POST' });
|
||||
location.reload();
|
||||
const provider = document.getElementById('providerSelect')?.value || '';
|
||||
const btn = document.getElementById('rescanBtn');
|
||||
const providerLabel = document.getElementById('providerSelect')?.selectedOptions?.[0]?.text || 'المحرك الحالي';
|
||||
if (!confirm(`إعادة استخراج البيانات باستخدام ${providerLabel}؟ سيتم استبدال النتائج الحالية.`)) return;
|
||||
|
||||
if (btn) {
|
||||
btn.disabled = true;
|
||||
btn.textContent = 'جارٍ إعادة المسح...';
|
||||
}
|
||||
|
||||
try {
|
||||
const params = new URLSearchParams();
|
||||
if (provider) params.set('provider', provider);
|
||||
const res = await fetch(`/extract/${docId}?${params.toString()}`, { method: 'POST' });
|
||||
const data = await res.json();
|
||||
if (!res.ok || !data.ok) {
|
||||
throw new Error(data.error || data.detail || 'تعذر بدء إعادة المسح.');
|
||||
}
|
||||
window.location.href = `/review/${docId}?wait=1`;
|
||||
} catch (e) {
|
||||
alert('حدث خطأ أثناء إعادة المسح: ' + e.message);
|
||||
if (btn) {
|
||||
btn.disabled = false;
|
||||
btn.textContent = 'إعادة المسح';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function verifyCorrelation(docId) {
|
||||
const btn = document.getElementById('verifyCorrelationBtn');
|
||||
const resultEl = document.getElementById('correlationResult');
|
||||
if (!btn || !resultEl) return;
|
||||
|
||||
btn.disabled = true;
|
||||
btn.textContent = 'جارٍ التحقق...';
|
||||
resultEl.className = 'info-banner correlation-result';
|
||||
resultEl.innerHTML = 'جارٍ فحص الترابط بين الصفحة الحالية والصفحة الأولى...';
|
||||
|
||||
try {
|
||||
const res = await fetch(`/review/${docId}/verify-correlation`, { method: 'POST' });
|
||||
const data = await res.json();
|
||||
if (!res.ok || !data.ok) {
|
||||
throw new Error(data.error || 'تعذر التحقق من الترابط.');
|
||||
}
|
||||
|
||||
const statusClass = data.same_document ? 'success-banner' : 'error-banner';
|
||||
const reasons = (data.reasons_ar || []).map(item => `<li>${item}</li>`).join('');
|
||||
const mismatches = (data.mismatch_flags || []).map(item => `<li>${item}</li>`).join('');
|
||||
resultEl.className = `${statusClass} correlation-result`;
|
||||
resultEl.innerHTML = `
|
||||
<div class="correlation-heading">
|
||||
<strong>${data.verdict_ar}</strong>
|
||||
<span class="correlation-meta">الثقة: ${data.confidence} | المحرك: ${data.provider}</span>
|
||||
</div>
|
||||
${reasons ? `<ul class="correlation-list">${reasons}</ul>` : ''}
|
||||
${mismatches ? `<ul class="correlation-list correlation-warnings">${mismatches}</ul>` : ''}
|
||||
`;
|
||||
} catch (e) {
|
||||
resultEl.className = 'error-banner correlation-result';
|
||||
resultEl.textContent = 'فشل التحقق: ' + e.message;
|
||||
} finally {
|
||||
btn.disabled = false;
|
||||
btn.textContent = 'تحقق بالذكاء الاصطناعي';
|
||||
}
|
||||
}
|
||||
|
||||
async function deleteDoc(docId) {
|
||||
|
||||
Reference in New Issue
Block a user