From 40a3c976e38196388710718fd64413aeb50133e1 Mon Sep 17 00:00:00 2001 From: Chaim Date: Fri, 24 Apr 2026 10:07:37 +0000 Subject: [PATCH] fix(kb/ingest): block-based PDF extraction preserves paragraph structure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit pymupdf's `get_text("text")` reads glyph runs in raw order, which on multi-column Hebrew PDFs (ספר הליקויים, ספר המבחנים) interleaves column headers, page numbers, form labels, and body paragraphs into jumbled lines — words split to single characters on separate lines, paragraphs lost. Switch to `get_text("blocks")` and sort each page's blocks by (y-axis ascending, x-axis descending) so each paragraph stays intact and the reading order is top-to-bottom, right-to-left as expected. Side-effect: chunk content becomes readable in the browse view. Refs Task Master #2 Co-Authored-By: Claude Opus 4.7 (1M context) --- api/services/kb/ingest.py | 26 ++++++++++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/api/services/kb/ingest.py b/api/services/kb/ingest.py index 18f3ce4..288b555 100644 --- a/api/services/kb/ingest.py +++ b/api/services/kb/ingest.py @@ -38,11 +38,33 @@ def _coerce_date(value) -> _dt.date | None: def _parse_pdf(data: bytes) -> str: + """Extract text from a PDF, preferring block-based parsing. + + Plain `get_text("text")` reads runs in raw glyph order which interleaves + column labels, page numbers, and body copy for multi-column Hebrew PDFs + (e.g. ספר הליקויים). Using `get_text("blocks")` and sorting by vertical + position then horizontal position keeps each paragraph intact and + preserves reading order top-to-bottom. + """ parts: list[str] = [] with fitz.open(stream=data, filetype="pdf") as doc: for page in doc: - parts.append(page.get_text("text")) - return "\n\n".join(parts).strip() + blocks = page.get_text("blocks") or [] + # Filter out empty / image-only blocks and sort by (y, x). For RTL + # languages, sorting by descending x within the same y keeps the + # reading order correct (right-to-left); pymupdf's default block + # order already respects this but only per-column — sorting + # explicitly makes it robust across oddly-placed floaters. + textual = [ + (b[1], -b[0], b[4]) + for b in blocks + if isinstance(b, tuple) and len(b) >= 5 and isinstance(b[4], str) and b[4].strip() + ] + textual.sort() # by y ascending, then x descending (via negation) + # Join paragraphs with a blank line so the chunker's header regexes + # still see line boundaries. + parts.append("\n\n".join(b[2].rstrip() for b in textual)) + return "\n\n".join(p for p in parts if p).strip() def _parse_docx(data: bytes) -> str: