Merge pull request 'fix(extraction): מספר-תיק שהומצא נכתב לשדה-הזהות — עיגון בטקסט לפני כתיבה (#232 מלכודת 3)' (#432) from worktree-docket-grounding into main
All checks were successful
All checks were successful
This commit was merged in pull request #432.
This commit is contained in:
@@ -22,6 +22,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import unicodedata
|
||||
from datetime import date as date_type
|
||||
from uuid import UUID
|
||||
|
||||
@@ -72,7 +73,7 @@ METADATA_EXTRACTION_PROMPT = """אתה מסייע משפטי בכיר. קרא א
|
||||
"source_type": "אחד מ-2: 'court_ruling' (פסק דין של בית משפט — עליון/מנהלי) / 'appeals_committee' (החלטה של ועדת ערר). אם לא ברור — מחרוזת ריקה.",
|
||||
"proceeding_type": "אחד מ-2 (רק להחלטות ועדת ערר): 'ערר' (הליך ערר עיקרי על החלטת ועדה מקומית) / 'בל\\\"מ' (בקשה להארכת מועד להגשת ערר). זהה דרך כותרת המסמך: 'ערר (ועדות ערר ...) NNNN/YY' → 'ערר'; 'בל\\\"מ NNNN/YY' או נושא 'בקשה להארכת מועד להגשת ערר' → 'בל\\\"מ'. בפסיקת בית משפט (לא ועדת ערר) — מחרוזת ריקה.",
|
||||
"court": "שם הערכאה כפי שהוא מופיע בכותרת (למשל 'בית המשפט העליון', 'בית המשפט המחוזי בירושלים בשבתו כבית משפט לעניינים מנהליים', 'ועדת הערר לתכנון ובניה פיצויים והיטלי השבחה — מחוז ירושלים'). מחרוזת ריקה אם לא ניתן לזהות.",
|
||||
"case_number_clean": "מספר הערר/תיק כפי שמופיע בכותרת — רק הספרות והאלכסון, למשל '1062/24' או '8031/21'. ללא המילה 'ערר', ללא שם הצדדים, ללא סוגריים. אם יש כמה עררים מאוחדים — הרשום הראשון. מחרוזת ריקה אם לא ניתן לזהות.",
|
||||
"case_number_clean": "מספר הערר/תיק **בדיוק כפי שמופיע בכותרת** — רק ספרות ומפרידים. שתי צורות קיימות ושתיהן חוקיות: דו-חלקית ('1062/24', '8031/21') ותלת-חלקית של ועדות ערר ('1094-09-19', '85074-09-24' — סידורי-חודש-שנה). **העתק את הספרות מהמסמך; אל תשלים, אל תנחש ואל תתקן ספרה.** ללא המילה 'ערר', ללא שם הצדדים, ללא סוגריים. אם יש כמה עררים מאוחדים — הרשום הראשון. **אם המספר אינו מופיע בטקסט — מחרוזת ריקה** (הקוד דוחה ממילא מספר שאינו מעוגן בטקסט).",
|
||||
"chair_name": "שם יו\\\"ר ההרכב של **ההחלטה הזו** — רלוונטי **רק להחלטות ועדת ערר**, לא לפסקי בית משפט. כמעט תמיד מופיע — בשני מקומות: (א) בכותרת/רובריקה בראש המסמך, ליד 'בפני:' / 'בהרכב:' / רשימת חברי הוועדה; (ב) בבלוק-החתימה בסוף ההחלטה, אחרי 'ההחלטה ניתנה' — שם מופיעים זה-לצד-זה מזכיר/ת הוועדה והיו\\\"ר (למשל בשתי עמודות: בצד אחד 'פלוני, עו\\\"ד / מזכיר ועדת הערר' ובצד השני 'אלמוני, עו\\\"ד / יו\\\"ר ועדת הערר'). **קח את השם שמעליו/לצדו כתוב 'יו\\\"ר' — לא את המזכיר/ה.** השאר שם פרטי+משפחה בלבד, בלי תוארים ('עו\\\"ד', 'אדריכל', 'עו\\\"ד דפנה תמיר'→'דפנה תמיר'). **אזהרה קריטית:** אל תיקח שם יו\\\"ר של פסק/החלטה אחרים ש**מצוטטים** בגוף ההחלטה (למשל 'כפי שנקבע ברשותה של יו\\\"ר פלונית בערר אחר...') — אלה תקדימים מצוטטים, לא היו\\\"ר של ההחלטה הנוכחית. אם זה פסק דין של בית משפט — מחרוזת ריקה.",
|
||||
"district": "מחוז ועדת הערר — רלוונטי **רק להחלטות ועדת ערר**. ערכים מותרים: 'ירושלים', 'תל אביב', 'מרכז', 'חיפה', 'צפון', 'דרום', 'ארצית'. זהה מהכותרת ('ועדת הערר לתכנון ובניה — מחוז ירושלים' → 'ירושלים'; 'ועדות ערר - תכנון ובנייה תל אביב-יפו' → 'תל אביב'). אם זה פסק דין של בית משפט — מחרוזת ריקה.",
|
||||
"parties": "שמות הצדדים בשורה אחת בצורה 'עורר נ\\' משיב' — בדיוק כפי שמופיעים בכותרת/רובריקה. בלי הדגשה, בלי מספר-תיק, בלי תוארים מיותרים. למשל 'ישיבת חברת אהבת שלום נ\\' תאיה' או 'ראם חיים נ\\' הוועדה המקומית לתכנון ובניה ירושלים'. אם הצדדים אינם מופיעים בטקסט (למשל החלטה שמתחילה בגוף בלי רובריקה) — מחרוזת ריקה. **אל תמציא שמות.**",
|
||||
@@ -242,6 +243,37 @@ def _is_clean_docket(s: str) -> bool:
|
||||
return bool(_DOCKET_RE.fullmatch((s or "").strip()))
|
||||
|
||||
|
||||
def _strip_invisibles(s: str) -> str:
|
||||
"""Drop Unicode format chars (category Cf) — RLM/LRM/ZWJ and friends.
|
||||
|
||||
Hebrew legal PDFs carry bidi marks *inside* docket numbers, so a plain
|
||||
substring test against the raw text misses a docket that is plainly there.
|
||||
"""
|
||||
return "".join(ch for ch in (s or "") if unicodedata.category(ch) != "Cf")
|
||||
|
||||
|
||||
def _docket_grounded(docket: str, *sources: str) -> bool:
|
||||
"""True when every digit group of ``docket`` appears, in order, in a source.
|
||||
|
||||
INV-AH (quote-or-retract) applied to the identity field. ``_is_clean_docket``
|
||||
only checks the *shape*, so a model that misreads one digit produces a
|
||||
perfectly well-formed but wrong docket — which is exactly how ערר 1094-09-19
|
||||
(פדילה) was stored as ``1094-09-14`` while all five case documents said
|
||||
...-19 (#232, trap 3). Shape validation cannot catch that; grounding can.
|
||||
|
||||
Tolerant of the separator (``-`` vs ``/``), of whitespace around it, and of
|
||||
bidi marks, so a real docket in the source still matches. Sources are the
|
||||
document text and the value being replaced — never the LLM's own output,
|
||||
which would make the check circular.
|
||||
"""
|
||||
parts = re.split(r"[-/]", (docket or "").strip())
|
||||
if not parts or not all(p.isdigit() for p in parts):
|
||||
return False
|
||||
pattern = r"\s*[-/]\s*".join(re.escape(p) for p in parts)
|
||||
haystack = _strip_invisibles("\n".join(s or "" for s in sources))
|
||||
return re.search(pattern, haystack) is not None
|
||||
|
||||
|
||||
def _source_type_for_level(level: str) -> str:
|
||||
"""Derive source_type from precedent_level — the library section is driven by
|
||||
source_type, so the two MUST agree (an LLM slip pairing
|
||||
@@ -400,6 +432,23 @@ async def apply_to_record(
|
||||
"already owned by another non-internal row (likely duplicate)",
|
||||
cur_cn, cn_clean,
|
||||
)
|
||||
elif not _docket_grounded(
|
||||
cn_clean,
|
||||
record.get("full_text") or "",
|
||||
cur_cn,
|
||||
record.get("citation_formatted") or "",
|
||||
):
|
||||
# The docket is well-formed but appears nowhere in the decision text
|
||||
# or in the value it would replace — i.e. the model produced digits
|
||||
# it cannot point at. case_number is the identity field; a wrong one
|
||||
# silently detaches the row from every reference to the real case.
|
||||
# Refuse the write and say so (§6) rather than trust the shape.
|
||||
logger.warning(
|
||||
"metadata_extractor: case_number normalization %r→%r REFUSED — the "
|
||||
"docket does not appear in the decision text or in the current "
|
||||
"value (ungrounded extraction, INV-AH). Keeping %r.",
|
||||
cur_cn, cn_clean, cur_cn,
|
||||
)
|
||||
else:
|
||||
fields_to_update["case_number"] = cn_clean
|
||||
|
||||
|
||||
65
mcp-server/tests/test_docket_grounding.py
Normal file
65
mcp-server/tests/test_docket_grounding.py
Normal file
@@ -0,0 +1,65 @@
|
||||
"""#232 trap 3 — a well-formed docket is not necessarily the right docket.
|
||||
|
||||
ערר (מרכז) 1094-09-19 (פדילה) was stored as ``1094-09-14``: shape-valid, so
|
||||
``_is_clean_docket`` waved it through, but wrong — and case_number is the
|
||||
identity field, so the row detached from every reference to the real case.
|
||||
Grounding the digits in the source text is what shape validation cannot do.
|
||||
"""
|
||||
|
||||
from legal_mcp.services.precedent_metadata_extractor import (
|
||||
_docket_grounded,
|
||||
_is_clean_docket,
|
||||
_strip_invisibles,
|
||||
)
|
||||
|
||||
|
||||
HEADER = "ערר (ועדות ערר - תכנון ובנייה מרכז) 1094-09-19 פדילה אברהים נ' הוועדה המקומית"
|
||||
|
||||
|
||||
def test_the_regression_shape_valid_but_wrong_digit():
|
||||
"""Both pass the shape check; only the real one is grounded."""
|
||||
assert _is_clean_docket("1094-09-14")
|
||||
assert _is_clean_docket("1094-09-19")
|
||||
assert _docket_grounded("1094-09-19", HEADER)
|
||||
assert not _docket_grounded("1094-09-14", HEADER)
|
||||
|
||||
|
||||
def test_separator_and_spacing_are_tolerated():
|
||||
"""A real docket must still match when the source writes it differently."""
|
||||
assert _docket_grounded("1094-09-19", "בערר 1094/09/19 נקבע")
|
||||
assert _docket_grounded("4768/22", "עת\"מ 4768-22 פלוני")
|
||||
assert _docket_grounded("1132-09-24", "תיק 1132 - 09 - 24")
|
||||
|
||||
|
||||
def test_bidi_marks_inside_the_number_do_not_defeat_grounding():
|
||||
"""Hebrew legal PDFs embed RLM/LRM between digits and separators."""
|
||||
noisy = "ערר (מרכז) 1094-09-19 פדילה"
|
||||
assert _strip_invisibles(noisy).count("") == 0
|
||||
assert _docket_grounded("1094-09-19", noisy)
|
||||
|
||||
|
||||
def test_grounding_accepts_the_value_being_replaced():
|
||||
"""Normalising an uploader's citation string into a clean docket is the
|
||||
whole point of the rewrite — the digits come from there, not the text."""
|
||||
citation = "ערר (ועדות ערר - תכנון ובנייה מרכז) 1094-09-19 פדילה נ' טירה (נבו 4.12.2019)"
|
||||
assert _docket_grounded("1094-09-19", "", citation)
|
||||
assert not _docket_grounded("1094-09-14", "", citation)
|
||||
|
||||
|
||||
def test_two_and_three_part_dockets_both_ground():
|
||||
assert _docket_grounded("8031/21", "בהיטל השבחה 8031/21 נדון")
|
||||
assert _docket_grounded("85074-09-24", "בל\"מ 85074-09-24")
|
||||
|
||||
|
||||
def test_non_numeric_or_empty_never_grounds():
|
||||
assert not _docket_grounded("", HEADER)
|
||||
assert not _docket_grounded("ערר 1094", HEADER)
|
||||
assert not _docket_grounded("abc-de", HEADER)
|
||||
|
||||
|
||||
def test_absent_from_every_source_is_refused():
|
||||
"""The פדילה failure mode: text has no docket at all, so anything the
|
||||
model offers is ungrounded and must not reach the identity field."""
|
||||
body = "בפני: יו\"ר הוועדה: רונית אלפר, עו\"ד\nהעוררים: 1. פדילה אברהים"
|
||||
assert not _docket_grounded("1094-09-14", body)
|
||||
assert not _docket_grounded("1094-09-19", body)
|
||||
Reference in New Issue
Block a user