This repository has been archived on 2026-07-19. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
shira-hermes/api/services/kb/chunker.py
T
chaim 16eeeff7b6 feat(kb): add 'caselaw' (פסיקה) as a fourth kind
Court rulings are a distinct content type — long discussion / reasoning,
no enumerated statutory hierarchy, case identifier instead of statute id —
so they get their own kind alongside law/regulation/circular rather than
being shoehorned into one of the existing three.

Migration 003 alters the CHECK constraints on kb_source.kind and
kb_ingest_job.kind to include 'caselaw'. The chunker uses the
circulars path for caselaw too (paragraph + size split, no statutory
section regex); the case identifier is captured at upload time in
kb_source.identifier.

Updates Literal types in ingest_source, admin_kb upload, and the
kb_public SearchRequest. _KINDS in s3.py grows to four so failed/
processed/inbox listings include caselaw subdirs.

Refs Task Master #18 (espocrm-extensions/KnowledgeBase)

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-25 16:57:46 +00:00

414 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Text chunking — strategies per source kind.
All strategies return list[dict] with keys:
chunk_index, heading_path, section_ref, content, token_count, page_number
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Literal
# Page marker sentinel embedded by the PDF parser (api/services/kb/ingest.py
# _page_marker). Format: \x00KB_PAGE:<N>\x00 where N is 1-indexed.
# _build_page_map returns a list of (offset, page_number) for each marker so
# the chunker can attach the right page to any chunk by position — _strip_page_markers
# then removes the sentinels from chunk.content.
_PAGE_MARKER_RE = re.compile(r"\x00KB_PAGE:(\d+)\x00")
def _build_page_map(text: str) -> list[tuple[int, int]]:
"""Return sorted [(offset, page_number), …] of every marker in `text`."""
return [(m.start(), int(m.group(1))) for m in _PAGE_MARKER_RE.finditer(text)]
def _page_for_offset(page_map: list[tuple[int, int]], offset: int) -> int | None:
"""Last-seen page number at or before `offset`. None if nothing earlier."""
if not page_map:
return None
lo, hi = 0, len(page_map) - 1
best: int | None = None
while lo <= hi:
mid = (lo + hi) // 2
mark_off, mark_page = page_map[mid]
if mark_off <= offset:
best = mark_page
lo = mid + 1
else:
hi = mid - 1
return best
# Approx tokens per char for Hebrew/mixed text (voyage-multilingual-2 tokenizer).
# Voyage tends to split Hebrew into 1.5-2 tokens per word; 0.55 tokens/char
# is a conservative estimate that matches what the API actually reports.
_TOKENS_PER_CHAR = 0.55
# Hard safety ceiling per chunk. Voyage accepts up to 32K tokens per item,
# and a batch is capped at 120K tokens. Keep chunks small so batches fit.
_MAX_TOKENS_PER_CHUNK = 1500
# Hierarchy markers in Israeli statutes. Each regex captures the full heading line
# and is ordered from largest container to smallest. The section-level match
# (_RE_SECTION) also captures the section number and optional sub-letter.
#
# Hebrew ordinal numbers: "א", "ב", … "ט׳", "י״א", "י״ב", … "ט״ז", etc.
# Pattern: one or two Hebrew letters with optional gershayim/geresh between them,
# and optionally followed by a trailing geresh.
_HEB_ORD = r"[א-ת](?:[\"״][א-ת])?[׳'\"]?"
_RE_PART = re.compile(rf"(?m)^\s*חלק\s+({_HEB_ORD})\s*[:\-–—]?\s*(.*?)\s*$")
_RE_CHAPTER = re.compile(rf"(?m)^\s*פרק\s+({_HEB_ORD})\s*[:\-–—]?\s*(.*?)\s*$")
_RE_SUBCHAPTER = re.compile(rf"(?m)^\s*סימן\s+({_HEB_ORD})\s*[:\-–—]?\s*(.*?)\s*$")
# Section header: "סעיף 12.", "12א.", "תקנה 5.", at start of a line, followed by
# optional sub-letter "(ב)" and then a dot / colon / dash, then either EOL or
# real Hebrew/English text (NOT a row of dots / a phone number / more digits).
#
# The key change vs the original: the tail `[\.:\-–—]` must be followed by
# at least one letter within the same short lookahead. This filters out:
# - phone numbers like "02-6709701"
# - table-of-contents rows like "3 . . . . . . . . . . . . . . ."
# - page-number-only lines like "23."
# while still catching real section headers like "סעיף 12. הגדרות" or
# "195. חישוב קצבת נכות".
_RE_SECTION = re.compile(
r"(?m)^\s*(?:סעיף\s+|תקנה\s+|§\s*)" # prefix is now REQUIRED (one of סעיף / תקנה / §)
r"(\d+[א-ת]?)\s*(?:\(([א-ת\d]+)\))?"
r"\s*[\.:\-–—]\s*"
r"(?=[^\d\.\s])" # next non-space char must not be digit/dot — i.e. real text follows
)
_STATUTE_MARKERS = [
("part", _RE_PART),
("chapter", _RE_CHAPTER),
("subchapter", _RE_SUBCHAPTER),
("section", _RE_SECTION),
]
# Header in circulars — markdown-ish headings, or numbered "נושא: ..." / "מס' N. ..."
_HEADER_RE = re.compile(
r"(?m)^\s*(?:#{1,4}\s+|מס['׳]\s*\d+\s*[\.\:]\s*|[A-Zא-ת][^\n]{0,80}\:\s*$)",
)
@dataclass
class Chunk:
chunk_index: int
heading_path: str
section_ref: str
content: str
token_count: int
page_number: int | None = None
def _count_tokens(text: str) -> int:
return max(1, int(len(text) * _TOKENS_PER_CHAR))
def _strip_page_markers(text: str) -> tuple[str, int | None]:
"""Remove page markers from `text` and return the FIRST page number seen.
Only useful as a fallback when the chunker didn't attach a page via
offset math. Markers that show up mid-content (e.g. a section that
spans pages) would resolve to the spanning page rather than the
chunk's starting page — so prefer c.page_number (the caller's
offset-derived value) and fall back to this only when that is None.
"""
first_page: int | None = None
m = _PAGE_MARKER_RE.search(text)
if m:
try:
first_page = int(m.group(1))
except ValueError:
first_page = None
return _PAGE_MARKER_RE.sub("", text), first_page
def _as_dicts(chunks: list[Chunk]) -> list[dict]:
result: list[dict] = []
for c in chunks:
clean, page_in_body = _strip_page_markers(c.content)
# _split_oversized may slice a piece in the middle of a marker
# sentinel, leaving a stray \x00 that PostgreSQL's UTF-8 check
# rejects. Drop any surviving null bytes.
clean = clean.replace("\x00", "")
# Prefer the page the caller computed from the chunk's START offset
# (the "where does this quote begin" answer). Only fall back to any
# marker inside the body if the caller didn't set one.
page = c.page_number if c.page_number is not None else page_in_body
result.append({
"chunk_index": c.chunk_index,
"heading_path": c.heading_path or None,
"section_ref": c.section_ref or None,
"content": clean.strip(),
"token_count": c.token_count,
"page_number": page,
})
return result
def _split_oversized(chunks: list[Chunk]) -> list[Chunk]:
"""Split any chunk whose token_count exceeds the ceiling into fixed-size pieces.
Preserves heading_path/section_ref on each piece and re-indexes.
Page numbers are re-derived per piece by scanning the parent chunk's
content for markers: the parent's own page_number is treated as a
virtual marker at offset 0, and any real \\x00KB_PAGE:N\\x00 markers
preserved in content advance the active page from that point onward.
"""
max_chars = int(_MAX_TOKENS_PER_CHUNK / _TOKENS_PER_CHAR)
result: list[Chunk] = []
for c in chunks:
if c.token_count <= _MAX_TOKENS_PER_CHUNK:
c.chunk_index = len(result)
result.append(c)
continue
text = c.content
local_map = _build_page_map(text)
# Virtual marker at offset 0 so pieces before the first real marker
# inherit the parent's known page.
if c.page_number is not None and (not local_map or local_map[0][0] > 0):
local_map = [(0, c.page_number)] + local_map
for start in range(0, len(text), max_chars):
piece = text[start : start + max_chars]
result.append(
Chunk(
chunk_index=len(result),
heading_path=c.heading_path,
section_ref=c.section_ref,
content=piece,
token_count=_count_tokens(piece),
page_number=_page_for_offset(local_map, start) or c.page_number,
)
)
return result
def _collect_markers(text: str) -> list[dict]:
"""Collect all hierarchy markers from the text, sorted by position.
Each marker carries: {level, start, end, number, title, heading}.
level ∈ {'chapter', 'subchapter', 'section'}.
"""
markers: list[dict] = []
for level, rx in _STATUTE_MARKERS:
for m in rx.finditer(text):
g1 = m.group(1) if m.groups() else ""
if level == "section":
number = g1
sub = m.group(2) if m.lastindex and m.lastindex >= 2 else None
ref = f"ס' {number}" + (f"({sub})" if sub else "")
heading = ref
title = ""
else:
prefix = {"part": "חלק", "chapter": "פרק", "subchapter": "סימן"}[level]
title = m.group(2).strip() if m.lastindex and m.lastindex >= 2 else ""
number = g1
ref = f"{prefix} {number}"
heading = f"{ref}{('' + title) if title else ''}"
markers.append({
"level": level,
"start": m.start(),
"end": m.end(),
"number": number,
"ref": ref,
"heading": heading,
"title": title,
})
markers.sort(key=lambda x: x["start"])
return markers
def chunk_statute(text: str) -> list[Chunk]:
"""Law / regulation: one chunk per section, with hierarchical heading_path.
heading_path format: "חלק ט' — ביטוח נכות > פרק ב' > סימן א' > ס' 195(א)".
section_ref stays the leaf ("ס' 195(א)") for precise citation.
Returns Chunk objects with page markers still embedded in content.
chunk() calls _split_oversized next, which uses those markers to
compute per-piece page numbers, and _as_dicts finally strips them.
"""
text = text.strip("\n ")
if not text:
return []
page_map = _build_page_map(text)
markers = _collect_markers(text)
section_markers = [m for m in markers if m["level"] == "section"]
if not section_markers:
return [Chunk(
0, "", "", text, _count_tokens(text),
page_number=_page_for_offset(page_map, 0),
)]
# Walk markers in order, tracking the active part/chapter/subchapter.
active_part: dict | None = None
active_chapter: dict | None = None
active_subchapter: dict | None = None
chunks: list[Chunk] = []
for i, m in enumerate(markers):
if m["level"] == "part":
active_part = m
active_chapter = None
active_subchapter = None
continue
if m["level"] == "chapter":
active_chapter = m
active_subchapter = None
continue
if m["level"] == "subchapter":
active_subchapter = m
continue
# section marker
start = m["start"]
end = markers[i + 1]["start"] if i + 1 < len(markers) else len(text)
content = text[start:end].strip()
if not content:
continue
parts: list[str] = []
if active_part:
parts.append(active_part["heading"])
if active_chapter:
parts.append(active_chapter["heading"])
if active_subchapter:
parts.append(active_subchapter["heading"])
parts.append(m["ref"])
heading_path = " > ".join(parts)
chunks.append(
Chunk(
chunk_index=len(chunks),
heading_path=heading_path,
section_ref=m["ref"],
content=content,
token_count=_count_tokens(content),
page_number=_page_for_offset(page_map, start),
)
)
return chunks
def chunk_circular(
text: str,
target_tokens: int = 600,
overlap_tokens: int = 80,
) -> list[Chunk]:
"""Circular: header-aware split, then pack paragraphs up to target size with overlap.
Returns Chunk objects with page markers still embedded in content.
Each sub-chunk's page_number is derived from the absolute offset of
the first paragraph it contains — so chunks that start mid-document
reflect the correct page, not the segment's opening page.
"""
text = text.strip("\n ")
if not text:
return []
page_map = _build_page_map(text)
target_chars = int(target_tokens / _TOKENS_PER_CHAR)
overlap_chars = int(overlap_tokens / _TOKENS_PER_CHAR)
headers = list(_HEADER_RE.finditer(text))
# (heading, body_raw_with_markers, body_start_offset_in_text)
segments: list[tuple[str, str, int]] = []
if not headers:
segments.append(("", text, 0))
else:
if headers[0].start() > 0:
segments.append(("", text[: headers[0].start()], 0))
for i, h in enumerate(headers):
end = headers[i + 1].start() if i + 1 < len(headers) else len(text)
heading = h.group(0).strip().rstrip(":").lstrip("#").strip()
body_raw = text[h.end() : end]
if body_raw.strip():
segments.append((heading, body_raw, h.end()))
# Paragraph regex: one or more non-empty lines with no blank line between them.
_PARA_RE = re.compile(r"[^\n]+(?:\n(?!\n)[^\n]+)*")
chunks: list[Chunk] = []
for heading, body_raw, body_start in segments:
body = body_raw.strip()
seg_page = _page_for_offset(page_map, body_start)
if len(body) <= target_chars:
chunks.append(
Chunk(
chunk_index=len(chunks),
heading_path=heading,
section_ref=heading,
content=body,
token_count=_count_tokens(body),
page_number=seg_page,
)
)
continue
# Collect paragraphs with their absolute offsets in the original text.
paragraphs: list[tuple[str, int]] = []
for pm in _PARA_RE.finditer(body_raw):
p_text = pm.group(0).strip()
if p_text:
paragraphs.append((p_text, body_start + pm.start()))
buf = ""
buf_abs_start: int | None = None
for p_text, p_abs in paragraphs:
if buf and len(buf) + len(p_text) + 2 > target_chars:
chunks.append(
Chunk(
chunk_index=len(chunks),
heading_path=heading,
section_ref=heading,
content=buf.strip(),
token_count=_count_tokens(buf),
page_number=(
_page_for_offset(page_map, buf_abs_start)
if buf_abs_start is not None else seg_page
),
)
)
# Overlap: keep last N chars of prior buf, then start fresh at p.
# The new buf's "start page" is the new paragraph's page — the
# overlap tail is treated as prefix context, not page anchor.
buf = (buf[-overlap_chars:] + "\n\n" + p_text) if overlap_chars > 0 else p_text
buf_abs_start = p_abs
else:
if not buf:
buf_abs_start = p_abs
buf = f"{buf}\n\n{p_text}" if buf else p_text
if buf.strip():
chunks.append(
Chunk(
chunk_index=len(chunks),
heading_path=heading,
section_ref=heading,
content=buf.strip(),
token_count=_count_tokens(buf),
page_number=(
_page_for_offset(page_map, buf_abs_start)
if buf_abs_start is not None else seg_page
),
)
)
return chunks
def chunk(text: str, kind: Literal["law", "regulation", "circular", "caselaw"]) -> list[dict]:
if kind in ("law", "regulation"):
chunks = chunk_statute(text)
else:
# circulars and caselaw share the same fallback: no enumerated
# statutory hierarchy, so we split by paragraph + size. The
# caselaw section_ref ends up empty (the chunker can't infer a
# case identifier from body text reliably) — the user supplies
# the case identifier via the upload form's `identifier` field
# and it's stored on the parent kb_source row instead.
chunks = chunk_circular(text)
# _split_oversized uses page markers still embedded in content to derive
# per-piece page numbers; _as_dicts then strips markers and emits the
# final serializable shape.
return _as_dicts(_split_oversized(chunks))