@@ -1,23 +1,33 @@
""" Text extraction from PDF, DOCX, DOC, and RTF files.
Primary PDF extraction: PyMuPDF direct text (for born-digital PDFs).
Fallback: Google Cloud Vision OCR (for scanned documents).
Fallback: Mistral OCR (for scanned documents or broken OCR layers ).
Routing logic (document-level, not per-page):
1. PyMuPDF extracts text from every page.
2. Pages are quality-checked via _text_quality_ok().
3. If ALL pages pass → use PyMuPDF output (free, ~50ms, no API call).
4. If ANY page fails → call Mistral OCR once for the entire PDF.
Mistral returns per-page Markdown; page_offsets are computed from it.
DOC files: converted to DOCX via LibreOffice before extraction.
Post-processing: Hebrew abbreviation quote fixer.
Post-processing: Hebrew abbreviation quote fixer (PyMuPDF path only;
Mistral handles gershayim natively).
"""
from __future__ import annotations
import asyncio
import base64
import io
import logging
import re
import subprocess
import tempfile
from pathlib import Path
from typing import TYPE_CHECKING
import fitz # PyMuPDF
import httpx
from PIL import Image
from docx import Document as DocxDocument
from striprtf . striprtf import rtf_to_text
@@ -25,36 +35,65 @@ from striprtf.striprtf import rtf_to_text
from legal_mcp import config
from legal_mcp . services import storage
if TYPE_CHECKING :
from google . cloud import vision
logger = logging . getLogger ( __name__ )
# ── Google Cloud Vision client (imported lazily — saves ~550ms at MCP startup) ──
# ── Mistral OCR ───────────────────────────────────────────────── ──
_vision_client : " vision.ImageAnnotatorClient | None " = None
_MISTRAL_OCR_URL = " https://api.mistral.ai/v1/ocr "
_MISTRAL_OCR_MODEL = " mistral-ocr-latest "
def _get_vision_client ( ) - > " vision.ImageAnnotatorClient " :
global _vision_client
if _vision_client is None :
from google . cloud import vision
_vision_client = vision . ImageAnnotatorClient (
client_options = { " api_key " : config . GOOGLE_CLOUD_VISION_API_KEY }
async def _call_mistral_ocr ( path : Path ) - > list [ str ] :
""" Call Mistral OCR API on a PDF. Returns per-page Markdown text list.
The Mistral response contains ``pages[i].markdown`` for each page.
If the response has fewer pages than the PDF, trailing pages are
padded with empty strings by the caller.
"""
if not config . MISTRAL_API_KEY :
raise RuntimeError (
" MISTRAL_API_KEY not configured — cannot OCR scanned PDF. "
" Set the env var in Coolify. "
)
return _vision_client
pdf_b64 = base64 . b64encode ( path . read_bytes ( ) ) . decode ( )
async with httpx . AsyncClient ( timeout = 300.0 ) as client :
resp = await client . post (
_MISTRAL_OCR_URL ,
headers = {
" Authorization " : f " Bearer { config . MISTRAL_API_KEY } " ,
" Content-Type " : " application/json " ,
} ,
json = {
" model " : _MISTRAL_OCR_MODEL ,
" document " : {
" type " : " document_url " ,
" document_url " : f " data:application/pdf;base64, { pdf_b64 } " ,
} ,
" include_image_base64 " : False ,
} ,
)
if resp . status_code != 200 :
raise RuntimeError (
f " Mistral OCR returned { resp . status_code } : { resp . text [ : 400 ] } "
)
pages = resp . json ( ) . get ( " pages " , [ ] )
return [ p . get ( " markdown " , " " ) for p in pages ]
# ── Hebrew text quality detection ────────────────────────────────
_HEBREW_RE = re . compile ( r ' [\ u0590- \ u05FF ] ' )
_HEBREW_RE = re . compile ( r ' [- ] ' )
_WORD_RE = re . compile ( r ' \ S+ ' )
def _text_quality_ok ( text : str ) - > bool :
""" Check if extracted text is real content vs broken OCR layer .
""" Check if PyMuPDF- extracted text is genuine Hebrew legal content .
Returns True if text appears to be genuine Hebrew leg al content.
Returns True if text appears to be re al content.
Broken OCR layers from scanned PDFs often have:
- Very short words / single-character fragments
- Each word on its own line (high words-per-line ratio)
@@ -64,26 +103,19 @@ def _text_quality_ok(text: str) -> bool:
if len ( words ) < 10 :
return False
# Average word length — real Hebrew words avg 4-6 chars.
avg_len = sum ( len ( w ) for w in words ) / len ( words )
if avg_len < 2.5 :
return False
# Percentage of single-character "words"
single_char_pct = sum ( 1 for w in words if len ( w ) == 1 ) / len ( words )
if single_char_pct > 0.4 :
return False
# Words per line — broken OCR puts each word on its own line.
# Real text has 5-15 words per line; broken OCR has ~1-2.
lines = [ l for l in text . split ( " \n " ) if l . strip ( ) ]
if lines :
words_per_line = len ( words ) / len ( lines )
if words_per_line < 3.0 :
return False
if lines and len ( words ) / len ( lines ) < 3.0 :
return False
# Hebrew character ratio among letter characters
letters = re . findall ( r ' [a-zA-Z \ u0590- \ u05FF] ' , text )
letters = re . findall ( r ' [a-zA-Z-] ' , text )
if letters :
hebrew_pct = sum ( 1 for c in letters if _HEBREW_RE . match ( c ) ) / len ( letters )
if hebrew_pct < 0.5 :
@@ -92,7 +124,7 @@ def _text_quality_ok(text: str) -> bool:
return True
# ── Hebrew abbreviation quote fixer ──────────────────── ──────────
# ── Hebrew abbreviation quote fixer (PyMuPDF path only) ──────────
_HEBREW_ABBREV_FIXES : dict [ str , str ] = {
' עוהייד ' : ' עוה " ד ' ,
@@ -111,50 +143,133 @@ _HEBREW_ABBREV_FIXES: dict[str, str] = {
' יחייד ' : ' יח " ד ' ,
' בייכ ' : ' ב " כ ' ,
# Patterns where double-yod (י י ) substitutes for gershayim (״) in born-digital PDFs
' בליימ ' : ' בל " מ ' , # בקשה להארכת מועד — appears in RTL legal docs
' תמייא ' : ' תמ " א ' , # תכנית מתאר ארצית
' בליימ ' : ' בל " מ ' ,
' תמייא ' : ' תמ " א ' ,
}
_ABBREV_PATTERN = re . compile (
' | ' . join ( re . escape ( k ) for k in sorted ( _HEBREW_ABBREV_FIXES , key = len , reverse = True ) )
)
# Matches Hebrew law year abbreviations where gershayim was encoded as double-yod.
# e.g. תשכייה → תשכ"ה, תשנייב → תשנ"ב
_HEBREW_YEAR_RE = re . compile ( r ' (תש[א-ת]+)י י ([א-ת]) ' )
def _fix_hebrew_quotes ( text : str ) - > str :
""" Fix known Hebrew abbreviation quote replacements.
Applied to both Google Vision OCR output and direct PyMuPDF extraction —
some born-digital PDFs encode gershayim (״) as double-yod (י י ), producing
the same corruption patterns as OCR.
"""
""" Fix gershayim encoded as double-yod in born-digital PDFs. """
text = _ABBREV_PATTERN . sub ( lambda m : _HEBREW_ABBREV_FIXES [ m . group ( ) ] , text )
text = _HEBREW_YEAR_RE . sub ( r ' \ 1 " \ 2 ' , text )
return text
# ── Extraction ─ ──────────────────────────────────────────────────
# ── Page joining ──────────────────────────────────────────────────
# Separator used when joining per-page text. Constant so chunker /
# retrofit can reproduce the join when computing page offsets.
PAGE_SEPARATOR = " \n \n "
def _join_pages ( pages_text : list [ str ] ) - > tuple [ str , list [ int ] ] :
""" Join per-page text with PAGE_SEPARATOR while recording start offsets. """
offsets : list [ int ] = [ ]
parts : list [ str ] = [ ]
cursor = 0
for i , pg in enumerate ( pages_text ) :
offsets . append ( cursor )
parts . append ( pg )
cursor + = len ( pg )
if i < len ( pages_text ) - 1 :
parts . append ( PAGE_SEPARATOR )
cursor + = len ( PAGE_SEPARATOR )
return " " . join ( parts ) , offsets
# ── PDF extraction ────────────────────────────────────────────────
async def _extract_pdf ( path : Path ) - > tuple [ str , int , list [ int ] ] :
""" Extract text from PDF using document-level routing.
Stage 1 — PyMuPDF pre-screen (free, ~50ms, no API call):
Run on every page. Collect per-page text; flag pages where
PyMuPDF returns < 50 chars or _text_quality_ok() fails
(scanned, blank, or broken embedded OCR layer).
Stage 2 — Mistral OCR (triggered when any page fails Stage 1):
Send the entire PDF to Mistral once. Use its per-page Markdown
for ALL pages — consistent source, no mixed plain/Markdown formats.
Mistral handles gershayim natively; no quote-fix applied.
Page offsets are always computed so the chunker can attribute each
chunk to its source page number (multimodal hybrid retrieval).
"""
doc = fitz . open ( str ( path ) )
page_count = len ( doc )
# Stage 1: PyMuPDF pre-screen
pymupdf_pages : list [ str ] = [ ]
failed : list [ int ] = [ ]
for i in range ( page_count ) :
text = doc [ i ] . get_text ( ) . strip ( )
if len ( text ) > 50 and _text_quality_ok ( text ) :
pymupdf_pages . append ( _fix_hebrew_quotes ( text ) )
else :
pymupdf_pages . append ( " " )
failed . append ( i )
doc . close ( )
if not failed :
logger . debug (
" PDF %s : all %d pages digital — PyMuPDF only " , path . name , page_count
)
joined , offsets = _join_pages ( pymupdf_pages )
return joined , page_count , offsets
# Stage 2: Mistral OCR for entire document
logger . info (
" PDF %s : %d / %d pages failed quality check → Mistral OCR " ,
path . name , len ( failed ) , page_count ,
)
mistral_pages = await _call_mistral_ocr ( path )
# Pad if Mistral returns fewer pages than PyMuPDF counted
while len ( mistral_pages ) < page_count :
mistral_pages . append ( " " )
joined , offsets = _join_pages ( mistral_pages [ : page_count ] )
return joined , page_count , offsets
def page_at_offset ( offset : int , page_offsets : list [ int ] ) - > int :
""" Return the 1-based page number containing a given char offset.
page_offsets[i] is the start of page (i+1) in the joined text.
"""
if not page_offsets :
return 1
page = 1
for i , start in enumerate ( page_offsets ) :
if start < = offset :
page = i + 1
else :
break
return page
# ── Public entry point ────────────────────────────────────────────
async def extract_text ( file_path : str ) - > tuple [ str , int , list [ int ] | None ] :
""" Extract text from a document file.
Returns:
``(text, page_count, page_offsets)`` where:
- ``text``: concatena ted extracted text
- ``text``: extrac ted t ext. Plain text for PyMuPDF path;
Markdown for Mistral path (tables, ``##`` headers preserved).
- ``page_count``: number of pages (0 for non-PDF)
- ``page_offsets``: ``page_offsets[i]`` = char start offset of
page (i+1) inside ``text``. ``None`` for non-PDFs (where the
notion of pages doesn ' t apply). Used by the chunker to assign
a ``page_number`` to each chunk.
- ``page_offsets``: char start of each page inside ``text``,
or ``None`` for non-PDF formats
"""
path = Path ( file_path )
suffix = path . suffix . lower ( )
@@ -173,95 +288,11 @@ async def extract_text(file_path: str) -> tuple[str, int, list[int] | None]:
raise ValueError ( f " Unsupported file type: { suffix } " )
def _join_pages ( pages_text : list [ str ] ) - > tuple [ str , list [ int ] ] :
""" Join per-page text with PAGE_SEPARATOR while recording the start
offset of each page in the joined output. """
offsets : list [ int ] = [ ]
parts : list [ str ] = [ ]
cursor = 0
for i , pg in enumerate ( pages_text ) :
offsets . append ( cursor )
parts . append ( pg )
cursor + = len ( pg )
if i < len ( pages_text ) - 1 :
parts . append ( PAGE_SEPARATOR )
cursor + = len ( PAGE_SEPARATOR )
return " " . join ( parts ) , offsets
async def _extract_pdf ( path : Path ) - > tuple [ str , int , list [ int ] ] :
""" Extract text from PDF.
Try direct text first, fall back to Google Cloud Vision for scanned
or broken-OCR pages.
"""
doc = fitz . open ( str ( path ) )
page_count = len ( doc )
pages_text : list [ str ] = [ ]
for page_num in range ( page_count ) :
page = doc [ page_num ]
text = page . get_text ( ) . strip ( )
if len ( text ) > 50 and _text_quality_ok ( text ) :
pages_text . append ( _fix_hebrew_quotes ( text ) )
logger . debug ( " Page %d : direct extraction ( %d chars, quality OK) " , page_num + 1 , len ( text ) )
else :
reason = " insufficient text " if len ( text ) < = 50 else " low quality OCR layer "
logger . info ( " Page %d : Google Vision OCR ( %s ) " , page_num + 1 , reason )
pix = page . get_pixmap ( dpi = 300 )
img_bytes = pix . tobytes ( " png " )
ocr_text = await asyncio . to_thread (
_ocr_with_google_vision , img_bytes , page_num + 1
)
pages_text . append ( ocr_text )
doc . close ( )
joined , offsets = _join_pages ( pages_text )
return joined , page_count , offsets
def page_at_offset ( offset : int , page_offsets : list [ int ] ) - > int :
""" Look up the page number containing a given char offset.
page_offsets[i] is the start of page (i+1) in the joined text;
a chunk starting at ``offset`` belongs to the highest-indexed page
whose start is ``<= offset``. Returns 1-based page number.
"""
if not page_offsets :
return 1
# Linear scan is fine — page_offsets is short (≤ ~200 for our PDFs).
page = 1
for i , start in enumerate ( page_offsets ) :
if start < = offset :
page = i + 1
else :
break
return page
def _ocr_with_google_vision ( image_bytes : bytes , page_num : int ) - > str :
""" OCR a single page image using Google Cloud Vision API. """
from google . cloud import vision # lazy: keeps MCP startup fast
client = _get_vision_client ( )
image = vision . Image ( content = image_bytes )
response = client . document_text_detection (
image = image ,
image_context = vision . ImageContext ( language_hints = [ " he " ] ) ,
)
if response . error . message :
raise RuntimeError (
f " Google Vision error on page { page_num } : { response . error . message } "
)
text = response . full_text_annotation . text if response . full_text_annotation else " "
return _fix_hebrew_quotes ( text )
# ── Non-PDF formats ───────────────────────────────────────────────
def _extract_doc ( path : Path ) - > str :
""" Extract text from legacy .doc file by converting to .docx via LibreOffice . """
""" Extract text from legacy .doc via LibreOffice → DOCX conversion . """
with tempfile . TemporaryDirectory ( ) as tmp_dir :
# Isolate the LibreOffice user profile per call: headless soffice
# locks a single shared profile, so concurrent .doc conversions would
@@ -296,13 +327,13 @@ def _extract_rtf(path: Path) -> str:
# ── Multimodal page rendering (V9) ───────────────────────────────
# Unchanged — multimodal embedding always uses PyMuPDF-rendered images
# regardless of whether text extraction used PyMuPDF or Mistral.
def _pixmap_to_pil ( pix : fitz . Pixmap ) - > Image . Image :
""" Convert a PyMuPDF pixmap to PIL.Image (RGB) without going through
PNG bytes. Faster than tobytes( ' png ' ) → Image.open(). """
""" Convert a PyMuPDF pixmap to PIL.Image (RGB). """
if pix . alpha :
# Drop alpha channel — voyage multimodal expects RGB.
pix = fitz . Pixmap ( pix , 0 )
return Image . frombytes ( " RGB " , ( pix . width , pix . height ) , pix . samples )
@@ -314,12 +345,9 @@ def render_pages_for_multimodal(
thumbnail_dir : Path | None = None ,
) - > list [ tuple [ Image . Image , Path | None ] ] :
""" Render each PDF page as PIL.Image at ``embed_dpi`` for the
multimodal embedder, and optionally save a smaller JPEG thumbnail
at ``thumb_dpi`` to ``thumbnail_dir`` for UI preview.
multimodal embedder, and optionally save JPEG thumbnails.
Returns ``[(pil_image, thumb_path_or_None), ...]`` in page order.
The full-DPI image stays in memory only — only the thumbnail is
persisted to disk.
"""
src = Path ( pdf_path )
if not src . is_file ( ) :
@@ -338,17 +366,12 @@ def render_pages_for_multimodal(
thumb_path : Path | None = None
if thumbnail_dir is not None and thumb_dpi :
thumb_path = thumbnail_dir / f " p { page_num : 03d } .jpg "
# Downsample the same render rather than re-rendering
# with PyMuPDF — far faster.
ratio = thumb_dpi / embed_dpi
thumb_size = (
max ( 1 , int ( img . width * ratio ) ) ,
max ( 1 , int ( img . height * ratio ) ) ,
)
thumb = img . resize ( thumb_size , Image . Resampling . LANCZOS )
# Persist the thumbnail (a DERIVED, regenerable artifact)
# through the storage layer (INV-STG1). Under the filesystem
# backend it lands at thumb_path exactly as before.
_tbuf = io . BytesIO ( )
thumb . save ( _tbuf , " JPEG " , quality = 75 , optimize = True )
try :
@@ -366,44 +389,28 @@ def render_pages_for_multimodal(
return out
# ── Nevo preamble stripping ──────────────────────────────────────
# ── Nevo preamble stripping ───────────────────────────────────────
_NEVO_MARKERS = ( " ספרות: " , " חקיקה שאוזכרה: " , " מיני-רציו: " , " פסקי דין שאוזכרו: " ,
" כתבי עת: " , " הועתק מנבו " )
# Markers for where the actual decision body begins (everything before is Nevo
# preamble: bibliography + מיני-רציו). Two families:
# - ועדת ערר / district openings (בפנינו / הערר שבנדון / ...)
# - COURT-RULING openings (#86.1): a פסק-דין header or the authoring judge's
# line. Without these, Nevo court judgments — exactly the ones carrying a
# מיני-רציו — slipped through unstripped (e.g. בג"ץ 1764/05).
#
# #86.2 hardening — two over-strip bugs found while backfilling:
# 1. ``פסק-דין`` headers are often markdown-wrapped (``**פסק דין**``); the old
# ``^פסק[- ]דין`` required the keyword to be the very first char of the line
# and allowed only one separator, so it missed the header and fell through
# to a citation 32K deep (עמ"נ 50567-07-21). We now tolerate leading
# markdown/whitespace and 0-3 separators.
# 2. Bare ``השופט``/``הנשיא`` matched *citations* ("השופט מ' חשין, פסקה 23"),
# stripping real decision body. The authoring-judge line ends with a COLON
# ("השופט י ' עמית:"); citations use a comma. We now require the colon.
_DECISION_START = re . compile (
r " ^[ \ t>*_#] { 0,6}(?: "
r " בפנינו|לפנינו|לפניי|הערר שבנדון|ועדת הערר לתכנון|רקע עובדתי|עסקינן| "
r " פסק[ \ t \ -] { 0,3}די(?:ן |נו)| " # פסק-דין / פסק דין / **פסק דין** header (final-nun ן vs דינו)
r " (?:כב(?:וד)?[ ' ׳ \" ]? \ s*)?(?:ה?שופט[ת]?|ה?נשיא[ה]?|המשנה לנשיא) \ s+[^ \ n,] { 1,40}: " # author line → colon
r " פסק[ \ t \ -] { 0,3}די(?:ן |נו)| "
r " (?:כב(?:וד)?[ ' ׳ \" ]? \ s*)?(?:ה?שופט[ת]?|ה?נשיא[ה]?|המשנה לנשיא) \ s+[^ \ n,] { 1,40}: "
r " ) " ,
re . MULTILINE ,
)
def strip_nevo_preamble ( text : str ) - > str :
""" Remove Nevo database preamble (bibliography, legislation, mini-ratio) from decision text .
""" Remove Nevo database preamble (bibliography, legislation, mini-ratio).
Returns the original text unchanged if no preamble is detected.
Works on both plain text (PyMuPDF) and Markdown (Mistral) since
_DECISION_START already tolerates leading ``[ \t >*_#] { 0,6}``.
"""
# Window wide enough to catch the Nevo markers even when a long court/parties
# header precedes them (court rulings push חקיקה שאוזכרה:/מיני-רציו: down).
head = text [ : 1500 ]
if not any ( marker in head for marker in _NEVO_MARKERS ) :
return text
@@ -419,17 +426,7 @@ _RATIO_MARKER = "מיני-רציו:"
def extract_nevo_ratio ( text : str ) - > str :
""" Return the Nevo מיני-רציו block (editorial holdings summary), or ' ' .
The mini-ratio is Nevo ' s own headnote — a concise, professionally-written
list of the holdings. We capture it *before* :func:`strip_nevo_preamble`
discards it, to serve as a free gold-set for benchmarking how well our
halacha extractor covers the real holdings (#86.3).
The block runs from the ``מיני-רציו:`` marker to whichever comes first:
the decision body (``_DECISION_START``) or the next preamble marker
(bibliography / legislation). Returns ' ' when there is no mini-ratio.
"""
""" Return the Nevo מיני-רציו block (editorial holdings summary), or ' ' . """
if not text :
return " "
start = text . find ( _RATIO_MARKER )
@@ -437,9 +434,6 @@ def extract_nevo_ratio(text: str) -> str:
return " "
body = text [ start + len ( _RATIO_MARKER ) : ]
# End at the earliest of: decision body start, or a following preamble
# marker (ספרות: / חקיקה שאוזכרה: / ...). Both are measured relative to
# the ratio body so we never run past it into the judgment itself.
end = len ( body )
dm = _DECISION_START . search ( body )
if dm :