fix(kb/ingest): block-based PDF extraction preserves paragraph structure
pymupdf's `get_text("text")` reads glyph runs in raw order, which on
multi-column Hebrew PDFs (ספר הליקויים, ספר המבחנים) interleaves column
headers, page numbers, form labels, and body paragraphs into jumbled
lines — words split to single characters on separate lines, paragraphs
lost. Switch to `get_text("blocks")` and sort each page's blocks by
(y-axis ascending, x-axis descending) so each paragraph stays intact
and the reading order is top-to-bottom, right-to-left as expected.
Side-effect: chunk content becomes readable in the browse view.
Refs Task Master #2
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -38,11 +38,33 @@ def _coerce_date(value) -> _dt.date | None:
|
||||
|
||||
|
||||
def _parse_pdf(data: bytes) -> str:
|
||||
"""Extract text from a PDF, preferring block-based parsing.
|
||||
|
||||
Plain `get_text("text")` reads runs in raw glyph order which interleaves
|
||||
column labels, page numbers, and body copy for multi-column Hebrew PDFs
|
||||
(e.g. ספר הליקויים). Using `get_text("blocks")` and sorting by vertical
|
||||
position then horizontal position keeps each paragraph intact and
|
||||
preserves reading order top-to-bottom.
|
||||
"""
|
||||
parts: list[str] = []
|
||||
with fitz.open(stream=data, filetype="pdf") as doc:
|
||||
for page in doc:
|
||||
parts.append(page.get_text("text"))
|
||||
return "\n\n".join(parts).strip()
|
||||
blocks = page.get_text("blocks") or []
|
||||
# Filter out empty / image-only blocks and sort by (y, x). For RTL
|
||||
# languages, sorting by descending x within the same y keeps the
|
||||
# reading order correct (right-to-left); pymupdf's default block
|
||||
# order already respects this but only per-column — sorting
|
||||
# explicitly makes it robust across oddly-placed floaters.
|
||||
textual = [
|
||||
(b[1], -b[0], b[4])
|
||||
for b in blocks
|
||||
if isinstance(b, tuple) and len(b) >= 5 and isinstance(b[4], str) and b[4].strip()
|
||||
]
|
||||
textual.sort() # by y ascending, then x descending (via negation)
|
||||
# Join paragraphs with a blank line so the chunker's header regexes
|
||||
# still see line boundaries.
|
||||
parts.append("\n\n".join(b[2].rstrip() for b in textual))
|
||||
return "\n\n".join(p for p in parts if p).strip()
|
||||
|
||||
|
||||
def _parse_docx(data: bytes) -> str:
|
||||
|
||||
Reference in New Issue
Block a user