From 3fa0775ffe3774137e13d0591a578869100424c7 Mon Sep 17 00:00:00 2001 From: Chaim Date: Sun, 3 May 2026 10:56:25 +0000 Subject: [PATCH] feat(legal-aid): add /api/legal-aid/parse-pdf-report endpoint Ports the pdfplumber-based parser from the (currently inactive) n8n workflow legal-aid-report-ingest.json so EspoCRM's GreenInvoiceBilling extension can do manual upload + dry-run preview of monthly Legal Aid payment reports without depending on n8n being live. Auth: X-Api-Key (matches admin_kb pattern). Response shape: same JSON the existing LegalAidPaymentReportService ingest pipeline expects, so the EspoCRM controller can hand it straight back to ::ingest($payload, $pdf, dryRun=true). Co-Authored-By: Claude Opus 4.7 (1M context) --- api/main.py | 2 + api/routes/legal_aid.py | 224 ++++++++++++++++++++++++++++++++++++++++ pyproject.toml | 1 + 3 files changed, 227 insertions(+) create mode 100644 api/routes/legal_aid.py diff --git a/api/main.py b/api/main.py index 0d1b197..0249df8 100644 --- a/api/main.py +++ b/api/main.py @@ -13,6 +13,7 @@ from api import __version__ from api.routes.admin_kb import router as admin_kb_router from api.routes.health import router as health_router from api.routes.kb_public import router as kb_public_router +from api.routes.legal_aid import router as legal_aid_router from api.routes.smart_assistant import router as smart_assistant_router from api.services.skills import seed_skills_from_image @@ -40,6 +41,7 @@ app.include_router(health_router) app.include_router(smart_assistant_router) app.include_router(admin_kb_router) app.include_router(kb_public_router) +app.include_router(legal_aid_router) @app.on_event("startup") diff --git a/api/routes/legal_aid.py b/api/routes/legal_aid.py new file mode 100644 index 0000000..09b8724 --- /dev/null +++ b/api/routes/legal_aid.py @@ -0,0 +1,224 @@ +"""Legal Aid PDF parser endpoint. + +Parses a monthly Legal Aid (סיוע משפטי) payment-report PDF using pdfplumber +and returns a JSON payload in the shape that GreenInvoiceBilling's +LegalAidPaymentReportService.ingest() expects. + +This is the same parsing logic used by the (currently unimported) n8n workflow +`workflows/legal-aid-report-ingest.json` in the GreenInvoiceBilling repo — +extracted here so the EspoCRM extension can do manual upload + preview without +depending on n8n being live. +""" + +from __future__ import annotations + +import io +import logging +import os +import re +from typing import Any + +from fastapi import APIRouter, File, HTTPException, Request, UploadFile + +logger = logging.getLogger("shira.legal_aid") + +router = APIRouter(prefix="/api/legal-aid", tags=["legal-aid"]) + + +def _verify_admin(request: Request) -> None: + expected = os.environ.get("ADMIN_API_KEY") or os.environ.get("API_KEY", "") + if not expected: + raise HTTPException(status_code=503, detail="Admin API key not configured") + provided = request.headers.get("X-Admin-Key") or request.headers.get("X-Api-Key") or "" + if provided != expected: + raise HTTPException(status_code=401, detail="Unauthorized") + + +# District name → enum code used in LegalAidPaymentReport.issuerOffice +_OFFICE_HINTS: list[tuple[str, str]] = [ + ("ירושלים", "Jerusalem"), + ("תל אביב", "TelAviv"), + ("חיפה", "Haifa"), + ("באר שבע", "BeerSheva"), + ("נצרת", "Nazareth"), + ("מרכז", "Center"), +] + + +@router.post("/parse-pdf-report") +async def parse_pdf_report( + request: Request, + file: UploadFile = File(..., description="Legal Aid monthly payment report PDF"), + subject: str | None = None, +) -> dict[str, Any]: + """Parse a Legal Aid monthly payment-report PDF. + + Auth: X-Api-Key (same as other admin endpoints). + Body: multipart with `file` (the PDF bytes). + Optional `subject` query param: original email subject — when supplied, used + to extract the [סימוכין NNN] sysref. Manual uploads typically omit this. + + Returns the same JSON shape that LegalAidPaymentReportService expects: + {subject, sysref, issuerOffice, reportDate, reportMonth, + totalExclVat, totalVat, totalInclVat, lines:[...], travelReimbursementLines:[]} + """ + _verify_admin(request) + + pdf_bytes = await file.read() + if not pdf_bytes: + raise HTTPException(status_code=400, detail="Empty file") + + if not (pdf_bytes[:4] == b"%PDF"): + raise HTTPException(status_code=400, detail="File is not a PDF") + + try: + import pdfplumber + except ImportError as e: + raise HTTPException( + status_code=503, + detail=f"pdfplumber not installed in shira-hermes: {e}", + ) + + sysref = None + if subject: + m = re.search(r"\[סימוכין\s+(\d+)\]", subject) + if m: + sysref = m.group(1) + + issuer_office: str | None = None + report_month: str | None = None + report_date: str | None = None + lines: list[dict[str, Any]] = [] + total_excl = total_vat = total_incl = 0.0 + parse_warnings: list[str] = [] + + try: + with pdfplumber.open(io.BytesIO(pdf_bytes)) as pdf: + for page in pdf.pages: + text = page.extract_text() or "" + + if issuer_office is None: + for hint, code in _OFFICE_HINTS: + if hint in text: + issuer_office = code + break + + if report_month is None: + m = re.search(r"עבור חודש:\s*(\d{1,2})/(\d{4})", text) + if m: + report_month = f"{m.group(2)}-{int(m.group(1)):02d}" + + if report_date is None: + m = re.search(r"(\d{1,2})/(\d{1,2})/(\d{4})", text) + if m: + report_date = ( + f"{m.group(3)}-{int(m.group(2)):02d}-{int(m.group(1)):02d}" + ) + + for table in page.extract_tables() or []: + if not table or len(table) < 2: + continue + + headers = [(h or "").strip() for h in table[0]] + + # Skip the empty travel-reimbursement table on page 2 etc. + if not any("חשבונית" in h for h in headers): + continue + + for row in table[1:]: + if not row or all((c or "").strip() == "" for c in row): + continue + + record = _row_to_record(headers, row) + if record is None: + continue + + total_excl += record.get("amountExclVat") or 0.0 + total_vat += record.get("vat") or 0.0 + total_incl += record.get("amountInclVat") or 0.0 + lines.append(record) + except Exception as e: + logger.exception("Legal Aid PDF parse failed") + raise HTTPException(status_code=422, detail=f"PDF parse error: {e}") + + out: dict[str, Any] = { + "subject": subject, + "sysref": sysref, + "issuerOffice": issuer_office, + "reportDate": report_date, + "reportMonth": report_month, + "totalExclVat": round(total_excl, 2), + "totalVat": round(total_vat, 2), + "totalInclVat": round(total_incl, 2), + "lines": lines, + "travelReimbursementLines": [], + } + + if parse_warnings: + out["parseWarnings"] = parse_warnings + + return out + + +def _row_to_record(headers: list[str], row: list[str | None]) -> dict[str, Any] | None: + """Translate one PDF table row into the structured record the ingest + service expects. Returns None if the row has nothing to extract. + """ + cells = [(c or "").strip() for c in row] + + def col(needle: str) -> str: + for i, h in enumerate(headers): + if h and needle in h: + if i < len(cells): + return cells[i].strip() + return "" + + invoice_number = col("חשבונית") + request_number = col("בקשה") + case_number = col("תיק") + client_name = col("לקוח") or col("שם") + activity_category_raw = col("פעולה") + activity_subtype = col("תת") + activity_date_raw = col("תאריך פעילות") or col("תאריך פעיל") + amount_excl_raw = col("סכום") + vat_raw = col("מע\"מ") or col('מע"מ') + amount_incl_raw = col("סה\"כ") or col('סה"כ') + + if not (invoice_number or request_number or case_number): + return None + + activity_category = activity_category_raw.split()[0] if activity_category_raw else "" + activity_date = _normalize_date(activity_date_raw) + + return { + "invoiceNumber": invoice_number, + "requestNumber": request_number, + "legalAidCaseNumber": case_number, + "clientName": client_name, + "activityCategory": activity_category, + "activitySubTypeHebrew": activity_subtype, + "activityDate": activity_date, + "amountExclVat": _to_float(amount_excl_raw), + "vat": _to_float(vat_raw), + "amountInclVat": _to_float(amount_incl_raw), + "raw": dict(zip(headers, cells)), + } + + +def _normalize_date(s: str) -> str | None: + if not s: + return None + m = re.search(r"(\d{1,2})[./](\d{1,2})[./](\d{4})", s) + if not m: + return None + return f"{m.group(3)}-{int(m.group(2)):02d}-{int(m.group(1)):02d}" + + +def _to_float(s: str) -> float: + if not s: + return 0.0 + cleaned = s.replace(",", "").replace("₪", "").replace("₪", "").strip() + try: + return float(cleaned) + except ValueError: + return 0.0 diff --git a/pyproject.toml b/pyproject.toml index 0692290..30a7147 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,6 +11,7 @@ dependencies = [ "openai>=1.50.0", "pyyaml>=6.0", "pymupdf>=1.24.0", + "pdfplumber>=0.11.0", "asyncpg>=0.30.0", "python-docx>=1.1.2", "boto3>=1.35.0",