This repository has been archived on 2026-07-19. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
shira-hermes/api/services/ocr.py
T
chaim 51c76c1cf9 feat: Claude Vision OCR fallback for unreadable documents
When the CRM's readDocument returns success:false (scanned PDFs,
image-only DOCX, corrupted files), shira-hermes now:

1. Fetches the raw file bytes via new getDocumentBytes endpoint
2. Sends them to Claude Vision through ai-gateway:
   - Images (PNG/JPG/TIFF/etc): direct vision call
   - PDFs: render each page at 180 DPI via PyMuPDF, vision call per page
   - DOCX: extract images from word/media/, vision call per image

New tool: read_document_ocr (explicit OCR request)
Existing tool read_document auto-falls-back to OCR on extraction failure.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-19 11:37:51 +00:00

132 lines
5.1 KiB
Python

"""OCR fallback — extract text from images and scanned PDFs via Claude Vision."""
from __future__ import annotations
import base64
import io
import logging
import os
from openai import AsyncOpenAI
logger = logging.getLogger("shira.ocr")
OCR_PROMPT = (
"Extract ALL visible text from this image. Return ONLY the raw text content — no commentary, "
"no formatting notes, no headers like 'Here is the text:'. Preserve the original language "
"(Hebrew, English, Arabic, etc.) and line breaks. If the image contains a form or table, "
"preserve the structure with tabs or spaces."
)
MAX_PDF_PAGES = 20
PDF_RENDER_DPI = 180
def _build_vision_messages(images_b64: list[tuple[str, str]], prompt: str = OCR_PROMPT) -> list[dict]:
"""Build OpenAI-format messages with vision content. images_b64 is list of (mime, base64)."""
content = [{"type": "text", "text": prompt}]
for mime, b64 in images_b64:
content.append({
"type": "image_url",
"image_url": {"url": f"data:{mime};base64,{b64}"},
})
return [{"role": "user", "content": content}]
async def _call_vision(images_b64: list[tuple[str, str]], prompt: str = OCR_PROMPT) -> str:
"""Send images to Claude Vision via ai-gateway and return extracted text."""
client = AsyncOpenAI(
base_url=os.environ.get("AI_GATEWAY_URL", "http://localhost:3000").rstrip("/") + "/v1",
api_key=os.environ.get("AI_GATEWAY_API_KEY", ""),
)
response = await client.chat.completions.create(
model=os.environ.get("CLAUDE_MODEL", "sonnet"),
messages=_build_vision_messages(images_b64, prompt),
max_tokens=4096,
)
return (response.choices[0].message.content or "").strip()
async def ocr_image(base64_data: str, mime_type: str) -> str:
"""OCR a single image via Claude Vision."""
logger.info("[ocr] image mime=%s size=%d", mime_type, len(base64_data))
return await _call_vision([(mime_type, base64_data)])
async def ocr_pdf(base64_data: str) -> str:
"""Render PDF pages as images and OCR each via Claude Vision. Returns combined text."""
try:
import fitz # PyMuPDF
except ImportError:
return "❌ PyMuPDF לא מותקן — לא ניתן לסרוק PDF כתמונה."
pdf_bytes = base64.b64decode(base64_data)
doc = fitz.open(stream=pdf_bytes, filetype="pdf")
page_count = min(len(doc), MAX_PDF_PAGES)
logger.info("[ocr] pdf pages=%d (limit=%d)", len(doc), MAX_PDF_PAGES)
pages_text = []
for i in range(page_count):
page = doc[i]
pix = page.get_pixmap(dpi=PDF_RENDER_DPI)
img_bytes = pix.tobytes("png")
img_b64 = base64.b64encode(img_bytes).decode("ascii")
try:
text = await _call_vision([("image/png", img_b64)])
pages_text.append(f"=== עמוד {i + 1} ===\n{text}")
except Exception as e:
logger.error("[ocr] page %d failed: %s", i + 1, e)
pages_text.append(f"=== עמוד {i + 1} ===\n[שגיאה: {e}]")
doc.close()
if len(doc) > MAX_PDF_PAGES:
pages_text.append(f"\n[הוגבל ל-{MAX_PDF_PAGES} עמודים מתוך {len(doc)}]")
return "\n\n".join(pages_text)
async def ocr_docx_images(base64_data: str) -> str:
"""Extract images from a DOCX archive and OCR each via Claude Vision."""
import zipfile
docx_bytes = base64.b64decode(base64_data)
image_extracts = []
try:
with zipfile.ZipFile(io.BytesIO(docx_bytes)) as z:
media = [n for n in z.namelist() if n.startswith("word/media/")]
logger.info("[ocr] docx images found=%d", len(media))
for i, name in enumerate(media[:MAX_PDF_PAGES]):
ext = name.rsplit(".", 1)[-1].lower()
mime = {"png": "image/png", "jpg": "image/jpeg", "jpeg": "image/jpeg", "gif": "image/gif"}.get(ext, "image/png")
img_bytes = z.read(name)
img_b64 = base64.b64encode(img_bytes).decode("ascii")
try:
text = await _call_vision([(mime, img_b64)])
image_extracts.append(f"=== תמונה {i + 1} ({name}) ===\n{text}")
except Exception as e:
logger.error("[ocr] docx image %s failed: %s", name, e)
except zipfile.BadZipFile:
return "❌ הקובץ אינו DOCX תקין."
if not image_extracts:
return "❌ לא נמצאו תמונות ב-DOCX לחילוץ טקסט."
return "\n\n".join(image_extracts)
async def extract_text_from_bytes(base64_data: str, mime_type: str, file_name: str = "") -> str:
"""Route to the right OCR method based on mime type. Returns extracted text."""
mt = (mime_type or "").lower()
if mt.startswith("image/"):
return await ocr_image(base64_data, mt)
if mt == "application/pdf":
return await ocr_pdf(base64_data)
if mt in (
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"application/msword",
) or file_name.lower().endswith((".docx", ".doc")):
return await ocr_docx_images(base64_data)
return f"❌ סוג קובץ לא נתמך ל-OCR: {mime_type}"