feat(ocr): replace Google Vision with Mistral OCR as PDF fallback
All checks were successful
G12 Leak-Guard / leak-guard (pull_request) Successful in 4s
Lint — undefined names / undefined-names (pull_request) Successful in 10s

Switches the scanned-PDF fallback from Google Cloud Vision to
Mistral OCR (mistral-ocr-latest) for better Hebrew accuracy and
robustness against broken embedded OCR layers (e.g. case 1044-03-26
which returned English garbage through Vision).

Routing strategy (document-level, not per-page):
- PyMuPDF extracts all pages; pages that pass _text_quality_ok()
  use PyMuPDF output directly (free, ~50ms).
- If ANY page fails quality → Mistral OCR called once for the whole
  PDF, returning per-page Markdown for all pages (consistent format,
  no plain-text/Markdown mix within a document).

Markdown output preserved: Mistral returns ## headers and |tables|;
chunker updated to recognise ATX Markdown headers (##/###) as section
boundaries in _split_into_sections().

Config: GOOGLE_CLOUD_VISION_API_KEY → MISTRAL_API_KEY; allowlist
updated vision.googleapis.com → api.mistral.ai.
MISTRAL_API_KEY added to Coolify container env.

Invariants: G1 (single OCR fallback path, not parallel), G2 (no
duplicate extractor route), INV-AH (Mistral handles gershayim
natively; quote-fix only on PyMuPDF path).

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-06-27 10:11:45 +00:00
parent f2a264a7da
commit 9ae7304d44
4 changed files with 183 additions and 185 deletions

View File

@@ -275,8 +275,8 @@ HALACHA_CANONICAL_SYNTH_MODEL = os.environ.get("HALACHA_CANONICAL_SYNTH_MODEL",
HALACHA_CANONICAL_SYNTH_EFFORT = os.environ.get("HALACHA_CANONICAL_SYNTH_EFFORT", "high")
HALACHA_CANONICAL_SYNTH_DRIFT_FLOOR = float(os.environ.get("HALACHA_CANONICAL_SYNTH_DRIFT_FLOOR", "0.80"))
# Google Cloud Vision (OCR for scanned PDFs)
GOOGLE_CLOUD_VISION_API_KEY = os.environ.get("GOOGLE_CLOUD_VISION_API_KEY", "")
# Mistral OCR (fallback for scanned PDFs — replaces Google Cloud Vision)
MISTRAL_API_KEY = os.environ.get("MISTRAL_API_KEY", "")
# Data directory
DATA_DIR = Path(os.environ.get("DATA_DIR", str(Path.home() / "legal-ai" / "data")))
@@ -361,7 +361,7 @@ PARENT_DOC_CHILD_OVERLAP_TOKENS = int(
# External service allowlist — case materials may ONLY be sent to these domains
ALLOWED_EXTERNAL_SERVICES = {
"api.voyageai.com", # Voyage AI (embeddings)
"vision.googleapis.com", # Google Cloud Vision (OCR)
"api.mistral.ai", # Mistral OCR (scanned PDFs)
}
# Audit

View File

@@ -144,9 +144,12 @@ def _split_into_sections(text: str) -> list[tuple[str, str]]:
markers: list[tuple[int, str]] = []
for pattern, section_type in SECTION_PATTERNS:
# ^ + MULTILINE: line start only. Optional leading spaces/tabs and an
# ^ + MULTILINE: line start only. Optional leading spaces/tabs, an
# optional Markdown ATX header prefix (``## ``/``### ``), and an
# optional ordinal prefix ("5.", "5)", "ג.") before the keyword.
anchored = rf"^[ \t]*(?:\d+[.)]\s*|[א-ת][.)]\s*)?(?:{pattern})"
# The Markdown prefix handles Mistral OCR output where section
# titles are rendered as ``## נימוקי הערר`` etc.
anchored = rf"^[ \t]*(?:#{1,3}\s+)?(?:\d+[.)]\s*|[א-ת][.)]\s*)?(?:{pattern})"
for match in re.finditer(anchored, text, re.MULTILINE):
markers.append((match.start(), section_type))

View File

@@ -1,23 +1,33 @@
"""Text extraction from PDF, DOCX, DOC, and RTF files.
Primary PDF extraction: PyMuPDF direct text (for born-digital PDFs).
Fallback: Google Cloud Vision OCR (for scanned documents).
Fallback: Mistral OCR (for scanned documents or broken OCR layers).
Routing logic (document-level, not per-page):
1. PyMuPDF extracts text from every page.
2. Pages are quality-checked via _text_quality_ok().
3. If ALL pages pass → use PyMuPDF output (free, ~50ms, no API call).
4. If ANY page fails → call Mistral OCR once for the entire PDF.
Mistral returns per-page Markdown; page_offsets are computed from it.
DOC files: converted to DOCX via LibreOffice before extraction.
Post-processing: Hebrew abbreviation quote fixer.
Post-processing: Hebrew abbreviation quote fixer (PyMuPDF path only;
Mistral handles gershayim natively).
"""
from __future__ import annotations
import asyncio
import base64
import io
import logging
import re
import subprocess
import tempfile
from pathlib import Path
from typing import TYPE_CHECKING
import fitz # PyMuPDF
import httpx
from PIL import Image
from docx import Document as DocxDocument
from striprtf.striprtf import rtf_to_text
@@ -25,36 +35,65 @@ from striprtf.striprtf import rtf_to_text
from legal_mcp import config
from legal_mcp.services import storage
if TYPE_CHECKING:
from google.cloud import vision
logger = logging.getLogger(__name__)
# ── Google Cloud Vision client (imported lazily — saves ~550ms at MCP startup) ──
# ── Mistral OCR ───────────────────────────────────────────────────
_vision_client: "vision.ImageAnnotatorClient | None" = None
_MISTRAL_OCR_URL = "https://api.mistral.ai/v1/ocr"
_MISTRAL_OCR_MODEL = "mistral-ocr-latest"
def _get_vision_client() -> "vision.ImageAnnotatorClient":
global _vision_client
if _vision_client is None:
from google.cloud import vision
_vision_client = vision.ImageAnnotatorClient(
client_options={"api_key": config.GOOGLE_CLOUD_VISION_API_KEY}
async def _call_mistral_ocr(path: Path) -> list[str]:
"""Call Mistral OCR API on a PDF. Returns per-page Markdown text list.
The Mistral response contains ``pages[i].markdown`` for each page.
If the response has fewer pages than the PDF, trailing pages are
padded with empty strings by the caller.
"""
if not config.MISTRAL_API_KEY:
raise RuntimeError(
"MISTRAL_API_KEY not configured — cannot OCR scanned PDF. "
"Set the env var in Coolify."
)
return _vision_client
pdf_b64 = base64.b64encode(path.read_bytes()).decode()
async with httpx.AsyncClient(timeout=300.0) as client:
resp = await client.post(
_MISTRAL_OCR_URL,
headers={
"Authorization": f"Bearer {config.MISTRAL_API_KEY}",
"Content-Type": "application/json",
},
json={
"model": _MISTRAL_OCR_MODEL,
"document": {
"type": "document_url",
"document_url": f"data:application/pdf;base64,{pdf_b64}",
},
"include_image_base64": False,
},
)
if resp.status_code != 200:
raise RuntimeError(
f"Mistral OCR returned {resp.status_code}: {resp.text[:400]}"
)
pages = resp.json().get("pages", [])
return [p.get("markdown", "") for p in pages]
# ── Hebrew text quality detection ────────────────────────────────
_HEBREW_RE = re.compile(r'[\u0590-\u05FF]')
_HEBREW_RE = re.compile(r'[֐-׿]')
_WORD_RE = re.compile(r'\S+')
def _text_quality_ok(text: str) -> bool:
"""Check if extracted text is real content vs broken OCR layer.
"""Check if PyMuPDF-extracted text is genuine Hebrew legal content.
Returns True if text appears to be genuine Hebrew legal content.
Returns True if text appears to be real content.
Broken OCR layers from scanned PDFs often have:
- Very short words / single-character fragments
- Each word on its own line (high words-per-line ratio)
@@ -64,26 +103,19 @@ def _text_quality_ok(text: str) -> bool:
if len(words) < 10:
return False
# Average word length — real Hebrew words avg 4-6 chars.
avg_len = sum(len(w) for w in words) / len(words)
if avg_len < 2.5:
return False
# Percentage of single-character "words"
single_char_pct = sum(1 for w in words if len(w) == 1) / len(words)
if single_char_pct > 0.4:
return False
# Words per line — broken OCR puts each word on its own line.
# Real text has 5-15 words per line; broken OCR has ~1-2.
lines = [l for l in text.split("\n") if l.strip()]
if lines:
words_per_line = len(words) / len(lines)
if words_per_line < 3.0:
return False
if lines and len(words) / len(lines) < 3.0:
return False
# Hebrew character ratio among letter characters
letters = re.findall(r'[a-zA-Z\u0590-\u05FF]', text)
letters = re.findall(r'[a-zA-Z֐-׿]', text)
if letters:
hebrew_pct = sum(1 for c in letters if _HEBREW_RE.match(c)) / len(letters)
if hebrew_pct < 0.5:
@@ -92,7 +124,7 @@ def _text_quality_ok(text: str) -> bool:
return True
# ── Hebrew abbreviation quote fixer ──────────────────────────────
# ── Hebrew abbreviation quote fixer (PyMuPDF path only) ──────────
_HEBREW_ABBREV_FIXES: dict[str, str] = {
'עוהייד': 'עוה"ד',
@@ -111,50 +143,133 @@ _HEBREW_ABBREV_FIXES: dict[str, str] = {
'יחייד': 'יח"ד',
'בייכ': 'ב"כ',
# Patterns where double-yod (יי) substitutes for gershayim (״) in born-digital PDFs
'בליימ': 'בל"מ', # בקשה להארכת מועד — appears in RTL legal docs
'תמייא': 'תמ"א', # תכנית מתאר ארצית
'בליימ': 'בל"מ',
'תמייא': 'תמ"א',
}
_ABBREV_PATTERN = re.compile(
'|'.join(re.escape(k) for k in sorted(_HEBREW_ABBREV_FIXES, key=len, reverse=True))
)
# Matches Hebrew law year abbreviations where gershayim was encoded as double-yod.
# e.g. תשכייה → תשכ"ה, תשנייב → תשנ"ב
_HEBREW_YEAR_RE = re.compile(r'(תש[א-ת]+)יי([א-ת])')
def _fix_hebrew_quotes(text: str) -> str:
"""Fix known Hebrew abbreviation quote replacements.
Applied to both Google Vision OCR output and direct PyMuPDF extraction —
some born-digital PDFs encode gershayim (״) as double-yod (יי), producing
the same corruption patterns as OCR.
"""
"""Fix gershayim encoded as double-yod in born-digital PDFs."""
text = _ABBREV_PATTERN.sub(lambda m: _HEBREW_ABBREV_FIXES[m.group()], text)
text = _HEBREW_YEAR_RE.sub(r'\1"\2', text)
return text
# ── Extraction ───────────────────────────────────────────────────
# ── Page joining ──────────────────────────────────────────────────
# Separator used when joining per-page text. Constant so chunker /
# retrofit can reproduce the join when computing page offsets.
PAGE_SEPARATOR = "\n\n"
def _join_pages(pages_text: list[str]) -> tuple[str, list[int]]:
"""Join per-page text with PAGE_SEPARATOR while recording start offsets."""
offsets: list[int] = []
parts: list[str] = []
cursor = 0
for i, pg in enumerate(pages_text):
offsets.append(cursor)
parts.append(pg)
cursor += len(pg)
if i < len(pages_text) - 1:
parts.append(PAGE_SEPARATOR)
cursor += len(PAGE_SEPARATOR)
return "".join(parts), offsets
# ── PDF extraction ────────────────────────────────────────────────
async def _extract_pdf(path: Path) -> tuple[str, int, list[int]]:
"""Extract text from PDF using document-level routing.
Stage 1 — PyMuPDF pre-screen (free, ~50ms, no API call):
Run on every page. Collect per-page text; flag pages where
PyMuPDF returns < 50 chars or _text_quality_ok() fails
(scanned, blank, or broken embedded OCR layer).
Stage 2 — Mistral OCR (triggered when any page fails Stage 1):
Send the entire PDF to Mistral once. Use its per-page Markdown
for ALL pages — consistent source, no mixed plain/Markdown formats.
Mistral handles gershayim natively; no quote-fix applied.
Page offsets are always computed so the chunker can attribute each
chunk to its source page number (multimodal hybrid retrieval).
"""
doc = fitz.open(str(path))
page_count = len(doc)
# Stage 1: PyMuPDF pre-screen
pymupdf_pages: list[str] = []
failed: list[int] = []
for i in range(page_count):
text = doc[i].get_text().strip()
if len(text) > 50 and _text_quality_ok(text):
pymupdf_pages.append(_fix_hebrew_quotes(text))
else:
pymupdf_pages.append("")
failed.append(i)
doc.close()
if not failed:
logger.debug(
"PDF %s: all %d pages digital — PyMuPDF only", path.name, page_count
)
joined, offsets = _join_pages(pymupdf_pages)
return joined, page_count, offsets
# Stage 2: Mistral OCR for entire document
logger.info(
"PDF %s: %d/%d pages failed quality check → Mistral OCR",
path.name, len(failed), page_count,
)
mistral_pages = await _call_mistral_ocr(path)
# Pad if Mistral returns fewer pages than PyMuPDF counted
while len(mistral_pages) < page_count:
mistral_pages.append("")
joined, offsets = _join_pages(mistral_pages[:page_count])
return joined, page_count, offsets
def page_at_offset(offset: int, page_offsets: list[int]) -> int:
"""Return the 1-based page number containing a given char offset.
page_offsets[i] is the start of page (i+1) in the joined text.
"""
if not page_offsets:
return 1
page = 1
for i, start in enumerate(page_offsets):
if start <= offset:
page = i + 1
else:
break
return page
# ── Public entry point ────────────────────────────────────────────
async def extract_text(file_path: str) -> tuple[str, int, list[int] | None]:
"""Extract text from a document file.
Returns:
``(text, page_count, page_offsets)`` where:
- ``text``: concatenated extracted text
- ``text``: extracted text. Plain text for PyMuPDF path;
Markdown for Mistral path (tables, ``##`` headers preserved).
- ``page_count``: number of pages (0 for non-PDF)
- ``page_offsets``: ``page_offsets[i]`` = char start offset of
page (i+1) inside ``text``. ``None`` for non-PDFs (where the
notion of pages doesn't apply). Used by the chunker to assign
a ``page_number`` to each chunk.
- ``page_offsets``: char start of each page inside ``text``,
or ``None`` for non-PDF formats
"""
path = Path(file_path)
suffix = path.suffix.lower()
@@ -173,95 +288,11 @@ async def extract_text(file_path: str) -> tuple[str, int, list[int] | None]:
raise ValueError(f"Unsupported file type: {suffix}")
def _join_pages(pages_text: list[str]) -> tuple[str, list[int]]:
"""Join per-page text with PAGE_SEPARATOR while recording the start
offset of each page in the joined output."""
offsets: list[int] = []
parts: list[str] = []
cursor = 0
for i, pg in enumerate(pages_text):
offsets.append(cursor)
parts.append(pg)
cursor += len(pg)
if i < len(pages_text) - 1:
parts.append(PAGE_SEPARATOR)
cursor += len(PAGE_SEPARATOR)
return "".join(parts), offsets
async def _extract_pdf(path: Path) -> tuple[str, int, list[int]]:
"""Extract text from PDF.
Try direct text first, fall back to Google Cloud Vision for scanned
or broken-OCR pages.
"""
doc = fitz.open(str(path))
page_count = len(doc)
pages_text: list[str] = []
for page_num in range(page_count):
page = doc[page_num]
text = page.get_text().strip()
if len(text) > 50 and _text_quality_ok(text):
pages_text.append(_fix_hebrew_quotes(text))
logger.debug("Page %d: direct extraction (%d chars, quality OK)", page_num + 1, len(text))
else:
reason = "insufficient text" if len(text) <= 50 else "low quality OCR layer"
logger.info("Page %d: Google Vision OCR (%s)", page_num + 1, reason)
pix = page.get_pixmap(dpi=300)
img_bytes = pix.tobytes("png")
ocr_text = await asyncio.to_thread(
_ocr_with_google_vision, img_bytes, page_num + 1
)
pages_text.append(ocr_text)
doc.close()
joined, offsets = _join_pages(pages_text)
return joined, page_count, offsets
def page_at_offset(offset: int, page_offsets: list[int]) -> int:
"""Look up the page number containing a given char offset.
page_offsets[i] is the start of page (i+1) in the joined text;
a chunk starting at ``offset`` belongs to the highest-indexed page
whose start is ``<= offset``. Returns 1-based page number.
"""
if not page_offsets:
return 1
# Linear scan is fine — page_offsets is short (≤ ~200 for our PDFs).
page = 1
for i, start in enumerate(page_offsets):
if start <= offset:
page = i + 1
else:
break
return page
def _ocr_with_google_vision(image_bytes: bytes, page_num: int) -> str:
"""OCR a single page image using Google Cloud Vision API."""
from google.cloud import vision # lazy: keeps MCP startup fast
client = _get_vision_client()
image = vision.Image(content=image_bytes)
response = client.document_text_detection(
image=image,
image_context=vision.ImageContext(language_hints=["he"]),
)
if response.error.message:
raise RuntimeError(
f"Google Vision error on page {page_num}: {response.error.message}"
)
text = response.full_text_annotation.text if response.full_text_annotation else ""
return _fix_hebrew_quotes(text)
# ── Non-PDF formats ───────────────────────────────────────────────
def _extract_doc(path: Path) -> str:
"""Extract text from legacy .doc file by converting to .docx via LibreOffice."""
"""Extract text from legacy .doc via LibreOffice → DOCX conversion."""
with tempfile.TemporaryDirectory() as tmp_dir:
# Isolate the LibreOffice user profile per call: headless soffice
# locks a single shared profile, so concurrent .doc conversions would
@@ -296,13 +327,13 @@ def _extract_rtf(path: Path) -> str:
# ── Multimodal page rendering (V9) ───────────────────────────────
# Unchanged — multimodal embedding always uses PyMuPDF-rendered images
# regardless of whether text extraction used PyMuPDF or Mistral.
def _pixmap_to_pil(pix: fitz.Pixmap) -> Image.Image:
"""Convert a PyMuPDF pixmap to PIL.Image (RGB) without going through
PNG bytes. Faster than tobytes('png') → Image.open()."""
"""Convert a PyMuPDF pixmap to PIL.Image (RGB)."""
if pix.alpha:
# Drop alpha channel — voyage multimodal expects RGB.
pix = fitz.Pixmap(pix, 0)
return Image.frombytes("RGB", (pix.width, pix.height), pix.samples)
@@ -314,12 +345,9 @@ def render_pages_for_multimodal(
thumbnail_dir: Path | None = None,
) -> list[tuple[Image.Image, Path | None]]:
"""Render each PDF page as PIL.Image at ``embed_dpi`` for the
multimodal embedder, and optionally save a smaller JPEG thumbnail
at ``thumb_dpi`` to ``thumbnail_dir`` for UI preview.
multimodal embedder, and optionally save JPEG thumbnails.
Returns ``[(pil_image, thumb_path_or_None), ...]`` in page order.
The full-DPI image stays in memory only — only the thumbnail is
persisted to disk.
"""
src = Path(pdf_path)
if not src.is_file():
@@ -338,17 +366,12 @@ def render_pages_for_multimodal(
thumb_path: Path | None = None
if thumbnail_dir is not None and thumb_dpi:
thumb_path = thumbnail_dir / f"p{page_num:03d}.jpg"
# Downsample the same render rather than re-rendering
# with PyMuPDF — far faster.
ratio = thumb_dpi / embed_dpi
thumb_size = (
max(1, int(img.width * ratio)),
max(1, int(img.height * ratio)),
)
thumb = img.resize(thumb_size, Image.Resampling.LANCZOS)
# Persist the thumbnail (a DERIVED, regenerable artifact)
# through the storage layer (INV-STG1). Under the filesystem
# backend it lands at thumb_path exactly as before.
_tbuf = io.BytesIO()
thumb.save(_tbuf, "JPEG", quality=75, optimize=True)
try:
@@ -366,44 +389,28 @@ def render_pages_for_multimodal(
return out
# ── Nevo preamble stripping ──────────────────────────────────────
# ── Nevo preamble stripping ──────────────────────────────────────
_NEVO_MARKERS = ("ספרות:", "חקיקה שאוזכרה:", "מיני-רציו:", "פסקי דין שאוזכרו:",
"כתבי עת:", "הועתק מנבו")
# Markers for where the actual decision body begins (everything before is Nevo
# preamble: bibliography + מיני-רציו). Two families:
# - ועדת ערר / district openings (בפנינו / הערר שבנדון / ...)
# - COURT-RULING openings (#86.1): a פסק-דין header or the authoring judge's
# line. Without these, Nevo court judgments — exactly the ones carrying a
# מיני-רציו — slipped through unstripped (e.g. בג"ץ 1764/05).
#
# #86.2 hardening — two over-strip bugs found while backfilling:
# 1. ``פסק-דין`` headers are often markdown-wrapped (``**פסק דין**``); the old
# ``^פסק[- ]דין`` required the keyword to be the very first char of the line
# and allowed only one separator, so it missed the header and fell through
# to a citation 32K deep (עמ"נ 50567-07-21). We now tolerate leading
# markdown/whitespace and 0-3 separators.
# 2. Bare ``השופט``/``הנשיא`` matched *citations* ("השופט מ' חשין, פסקה 23"),
# stripping real decision body. The authoring-judge line ends with a COLON
# ("השופט י' עמית:"); citations use a comma. We now require the colon.
_DECISION_START = re.compile(
r"^[ \t>*_#]{0,6}(?:"
r"בפנינו|לפנינו|לפניי|הערר שבנדון|ועדת הערר לתכנון|רקע עובדתי|עסקינן|"
r"פסק[ \t\-]{0,3}די(?:ן|נו)|" # פסק-דין / פסק דין / **פסק דין** header (final-nun ן vs דינו)
r"(?:כב(?:וד)?['׳\"]?\s*)?(?:ה?שופט[ת]?|ה?נשיא[ה]?|המשנה לנשיא)\s+[^\n,]{1,40}:" # author line → colon
r"פסק[ \t\-]{0,3}די(?:ן|נו)|"
r"(?:כב(?:וד)?['׳\"]?\s*)?(?:ה?שופט[ת]?|ה?נשיא[ה]?|המשנה לנשיא)\s+[^\n,]{1,40}:"
r")",
re.MULTILINE,
)
def strip_nevo_preamble(text: str) -> str:
"""Remove Nevo database preamble (bibliography, legislation, mini-ratio) from decision text.
"""Remove Nevo database preamble (bibliography, legislation, mini-ratio).
Returns the original text unchanged if no preamble is detected.
Works on both plain text (PyMuPDF) and Markdown (Mistral) since
_DECISION_START already tolerates leading ``[ \t>*_#]{0,6}``.
"""
# Window wide enough to catch the Nevo markers even when a long court/parties
# header precedes them (court rulings push חקיקה שאוזכרה:/מיני-רציו: down).
head = text[:1500]
if not any(marker in head for marker in _NEVO_MARKERS):
return text
@@ -419,17 +426,7 @@ _RATIO_MARKER = "מיני-רציו:"
def extract_nevo_ratio(text: str) -> str:
"""Return the Nevo מיני-רציו block (editorial holdings summary), or ''.
The mini-ratio is Nevo's own headnote — a concise, professionally-written
list of the holdings. We capture it *before* :func:`strip_nevo_preamble`
discards it, to serve as a free gold-set for benchmarking how well our
halacha extractor covers the real holdings (#86.3).
The block runs from the ``מיני-רציו:`` marker to whichever comes first:
the decision body (``_DECISION_START``) or the next preamble marker
(bibliography / legislation). Returns '' when there is no mini-ratio.
"""
"""Return the Nevo מיני-רציו block (editorial holdings summary), or ''."""
if not text:
return ""
start = text.find(_RATIO_MARKER)
@@ -437,9 +434,6 @@ def extract_nevo_ratio(text: str) -> str:
return ""
body = text[start + len(_RATIO_MARKER):]
# End at the earliest of: decision body start, or a following preamble
# marker (ספרות: / חקיקה שאוזכרה: / ...). Both are measured relative to
# the ratio body so we never run past it into the judgment itself.
end = len(body)
dm = _DECISION_START.search(body)
if dm: