"""Legal Aid PDF parser endpoint. Parses a monthly Legal Aid (סיוע משפטי) payment-report PDF using pdfplumber and returns a JSON payload in the shape that GreenInvoiceBilling's LegalAidPaymentReportService.ingest() expects. This is the same parsing logic used by the (currently unimported) n8n workflow `workflows/legal-aid-report-ingest.json` in the GreenInvoiceBilling repo — extracted here so the EspoCRM extension can do manual upload + preview without depending on n8n being live. """ from __future__ import annotations import io import logging import os import re from typing import Any from fastapi import APIRouter, File, HTTPException, Request, UploadFile logger = logging.getLogger("shira.legal_aid") router = APIRouter(prefix="/api/legal-aid", tags=["legal-aid"]) def _verify_admin(request: Request) -> None: expected = os.environ.get("ADMIN_API_KEY") or os.environ.get("API_KEY", "") if not expected: raise HTTPException(status_code=503, detail="Admin API key not configured") provided = request.headers.get("X-Admin-Key") or request.headers.get("X-Api-Key") or "" if provided != expected: raise HTTPException(status_code=401, detail="Unauthorized") # District name → enum code used in LegalAidPaymentReport.issuerOffice _OFFICE_HINTS: list[tuple[str, str]] = [ ("ירושלים", "Jerusalem"), ("תל אביב", "TelAviv"), ("חיפה", "Haifa"), ("באר שבע", "BeerSheva"), ("נצרת", "Nazareth"), ("מרכז", "Center"), ] @router.post("/parse-pdf-report") async def parse_pdf_report( request: Request, file: UploadFile = File(..., description="Legal Aid monthly payment report PDF"), subject: str | None = None, ) -> dict[str, Any]: """Parse a Legal Aid monthly payment-report PDF. Auth: X-Api-Key (same as other admin endpoints). Body: multipart with `file` (the PDF bytes). Optional `subject` query param: original email subject — when supplied, used to extract the [סימוכין NNN] sysref. Manual uploads typically omit this. Returns the same JSON shape that LegalAidPaymentReportService expects: {subject, sysref, issuerOffice, reportDate, reportMonth, totalExclVat, totalVat, totalInclVat, lines:[...], travelReimbursementLines:[]} """ _verify_admin(request) pdf_bytes = await file.read() if not pdf_bytes: raise HTTPException(status_code=400, detail="Empty file") if not (pdf_bytes[:4] == b"%PDF"): raise HTTPException(status_code=400, detail="File is not a PDF") try: import pdfplumber except ImportError as e: raise HTTPException( status_code=503, detail=f"pdfplumber not installed in shira-hermes: {e}", ) sysref = None if subject: m = re.search(r"\[סימוכין\s+(\d+)\]", subject) if m: sysref = m.group(1) issuer_office: str | None = None report_month: str | None = None report_date: str | None = None lines: list[dict[str, Any]] = [] total_excl = total_vat = total_incl = 0.0 parse_warnings: list[str] = [] try: with pdfplumber.open(io.BytesIO(pdf_bytes)) as pdf: for page in pdf.pages: text = page.extract_text() or "" if issuer_office is None: for hint, code in _OFFICE_HINTS: if hint in text: issuer_office = code break if report_month is None: m = re.search(r"עבור חודש:\s*(\d{1,2})/(\d{4})", text) if m: report_month = f"{m.group(2)}-{int(m.group(1)):02d}" if report_date is None: m = re.search(r"(\d{1,2})/(\d{1,2})/(\d{4})", text) if m: report_date = ( f"{m.group(3)}-{int(m.group(2)):02d}-{int(m.group(1)):02d}" ) for table in page.extract_tables() or []: if not table or len(table) < 2: continue headers = [(h or "").strip() for h in table[0]] # Skip the empty travel-reimbursement table on page 2 etc. if not any("חשבונית" in h for h in headers): continue for row in table[1:]: if not row or all((c or "").strip() == "" for c in row): continue record = _row_to_record(headers, row) if record is None: continue total_excl += record.get("amountExclVat") or 0.0 total_vat += record.get("vat") or 0.0 total_incl += record.get("amountInclVat") or 0.0 lines.append(record) except Exception as e: logger.exception("Legal Aid PDF parse failed") raise HTTPException(status_code=422, detail=f"PDF parse error: {e}") out: dict[str, Any] = { "subject": subject, "sysref": sysref, "issuerOffice": issuer_office, "reportDate": report_date, "reportMonth": report_month, "totalExclVat": round(total_excl, 2), "totalVat": round(total_vat, 2), "totalInclVat": round(total_incl, 2), "lines": lines, "travelReimbursementLines": [], } if parse_warnings: out["parseWarnings"] = parse_warnings return out def _row_to_record(headers: list[str], row: list[str | None]) -> dict[str, Any] | None: """Translate one PDF table row into the structured record the ingest service expects. Returns None if the row has nothing to extract. """ cells = [(c or "").strip() for c in row] def col(needle: str) -> str: for i, h in enumerate(headers): if h and needle in h: if i < len(cells): return cells[i].strip() return "" invoice_number = col("חשבונית") request_number = col("בקשה") case_number = col("תיק") client_name = col("לקוח") or col("שם") activity_category_raw = col("פעולה") activity_subtype = col("תת") activity_date_raw = col("תאריך פעילות") or col("תאריך פעיל") amount_excl_raw = col("סכום") vat_raw = col("מע\"מ") or col('מע"מ') amount_incl_raw = col("סה\"כ") or col('סה"כ') if not (invoice_number or request_number or case_number): return None activity_category = activity_category_raw.split()[0] if activity_category_raw else "" activity_date = _normalize_date(activity_date_raw) return { "invoiceNumber": invoice_number, "requestNumber": request_number, "legalAidCaseNumber": case_number, "clientName": client_name, "activityCategory": activity_category, "activitySubTypeHebrew": activity_subtype, "activityDate": activity_date, "amountExclVat": _to_float(amount_excl_raw), "vat": _to_float(vat_raw), "amountInclVat": _to_float(amount_incl_raw), "raw": dict(zip(headers, cells)), } def _normalize_date(s: str) -> str | None: if not s: return None m = re.search(r"(\d{1,2})[./](\d{1,2})[./](\d{4})", s) if not m: return None return f"{m.group(3)}-{int(m.group(2)):02d}-{int(m.group(1)):02d}" def _to_float(s: str) -> float: if not s: return 0.0 cleaned = s.replace(",", "").replace("₪", "").replace("₪", "").strip() try: return float(cleaned) except ValueError: return 0.0