From 63387c0d7d8dee73da94dc587bce8de7acdd84d9 Mon Sep 17 00:00:00 2001 From: Chaim Date: Wed, 5 Aug 2026 10:45:22 +0000 Subject: [PATCH] =?UTF-8?q?fix(extraction):=20=D7=9E=D7=A1=D7=A4=D7=A8-?= =?UTF-8?q?=D7=AA=D7=99=D7=A7=20=D7=A9=D7=94=D7=95=D7=9E=D7=A6=D7=90=20?= =?UTF-8?q?=D7=A0=D7=9B=D7=AA=D7=91=20=D7=9C=D7=A9=D7=93=D7=94-=D7=94?= =?UTF-8?q?=D7=96=D7=94=D7=95=D7=AA=20=E2=80=94=20=D7=A2=D7=99=D7=92=D7=95?= =?UTF-8?q?=D7=9F=20=D7=91=D7=98=D7=A7=D7=A1=D7=98=20=D7=9C=D7=A4=D7=A0?= =?UTF-8?q?=D7=99=20=D7=9B=D7=AA=D7=99=D7=91=D7=94=20(#232=20=D7=9E=D7=9C?= =?UTF-8?q?=D7=9B=D7=95=D7=93=D7=AA=203)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ערר (מרכז) 1094-09-19 (פדילה) נשמר כ-`1094-09-14` בעוד כל חמשת מסמכי התיק גורסים ...-19. `_is_clean_docket` בדק **צורה בלבד**, ושתי הצורות תקינות — כך שספרה שגויה אחת עברה את השער וניתקה את השורה מכל הפניה לתיק האמיתי. `case_number` הוא שדה-זהות; זה לא שדה-תצוגה שאפשר לתקן בקריאה. שורש נוסף שהתגלה תוך כדי: ב-1094-09-19 המספר **אינו מופיע ב-full_text כלל** (הטקסט מתחיל ב"בפני:"), כלומר המודל הפיק ספרות שאין להן עיגון במקור — בדיוק מה ש-INV-AH בא למנוע. מה שונה - `_docket_grounded()` — כל קבוצת-ספרות של ה-docket חייבת להופיע, בסדר, בטקסט ההחלטה או בערך שהיא מחליפה. סובלני למפריד (`-` מול `/`), לרווחים סביבו, ולתווי-כיווניות (RLM/LRM) שנדחסים בתוך המספר ב-PDF עברי. המקורות לעולם אינם פלט-המודל עצמו — אחרת הבדיקה מעגלית. - סירוב לכתוב מלווה `logger.warning` מפורש (§6) במקום להסתמך על הצורה. - הפרומפט תוקן: הדוגמאות היו דו-חלקיות בלבד ('1062/24'), מה שהטה נגד docket תלת-חלקי של ועדות ערר. נוספו דוגמאות תלת-חלקיות והוראה מפורשת לא להשלים/לנחש/לתקן ספרה, ולהחזיר ריק כשהמספר אינו בטקסט. אימות מול הקורפוס החי (386 שורות): מתוך 50 השורות שהנרמול חל עליהן בפועל — **0 נחסמות**. הגארד חוסם רק ספרות שאין להן עיגון באף מקור. invariants: INV-AH (quote-or-retract על שדה-זהות) · G1 (נרמול במקור) · §6 (סירוב מדווח, לא נבלע) טסטים: 7 חדשים (tests/test_docket_grounding.py), הראשון שבהם משחזר בדיוק את הרגרסיה — שתי הצורות עוברות את בדיקת-הצורה, רק הנכונה מעוגנת. 521 עוברים. --- .../services/precedent_metadata_extractor.py | 51 ++++++++++++++- mcp-server/tests/test_docket_grounding.py | 65 +++++++++++++++++++ 2 files changed, 115 insertions(+), 1 deletion(-) create mode 100644 mcp-server/tests/test_docket_grounding.py diff --git a/mcp-server/src/legal_mcp/services/precedent_metadata_extractor.py b/mcp-server/src/legal_mcp/services/precedent_metadata_extractor.py index 2bdf42f..17f06ce 100644 --- a/mcp-server/src/legal_mcp/services/precedent_metadata_extractor.py +++ b/mcp-server/src/legal_mcp/services/precedent_metadata_extractor.py @@ -22,6 +22,7 @@ from __future__ import annotations import logging import re +import unicodedata from datetime import date as date_type from uuid import UUID @@ -72,7 +73,7 @@ METADATA_EXTRACTION_PROMPT = """אתה מסייע משפטי בכיר. קרא א "source_type": "אחד מ-2: 'court_ruling' (פסק דין של בית משפט — עליון/מנהלי) / 'appeals_committee' (החלטה של ועדת ערר). אם לא ברור — מחרוזת ריקה.", "proceeding_type": "אחד מ-2 (רק להחלטות ועדת ערר): 'ערר' (הליך ערר עיקרי על החלטת ועדה מקומית) / 'בל\\\"מ' (בקשה להארכת מועד להגשת ערר). זהה דרך כותרת המסמך: 'ערר (ועדות ערר ...) NNNN/YY' → 'ערר'; 'בל\\\"מ NNNN/YY' או נושא 'בקשה להארכת מועד להגשת ערר' → 'בל\\\"מ'. בפסיקת בית משפט (לא ועדת ערר) — מחרוזת ריקה.", "court": "שם הערכאה כפי שהוא מופיע בכותרת (למשל 'בית המשפט העליון', 'בית המשפט המחוזי בירושלים בשבתו כבית משפט לעניינים מנהליים', 'ועדת הערר לתכנון ובניה פיצויים והיטלי השבחה — מחוז ירושלים'). מחרוזת ריקה אם לא ניתן לזהות.", - "case_number_clean": "מספר הערר/תיק כפי שמופיע בכותרת — רק הספרות והאלכסון, למשל '1062/24' או '8031/21'. ללא המילה 'ערר', ללא שם הצדדים, ללא סוגריים. אם יש כמה עררים מאוחדים — הרשום הראשון. מחרוזת ריקה אם לא ניתן לזהות.", + "case_number_clean": "מספר הערר/תיק **בדיוק כפי שמופיע בכותרת** — רק ספרות ומפרידים. שתי צורות קיימות ושתיהן חוקיות: דו-חלקית ('1062/24', '8031/21') ותלת-חלקית של ועדות ערר ('1094-09-19', '85074-09-24' — סידורי-חודש-שנה). **העתק את הספרות מהמסמך; אל תשלים, אל תנחש ואל תתקן ספרה.** ללא המילה 'ערר', ללא שם הצדדים, ללא סוגריים. אם יש כמה עררים מאוחדים — הרשום הראשון. **אם המספר אינו מופיע בטקסט — מחרוזת ריקה** (הקוד דוחה ממילא מספר שאינו מעוגן בטקסט).", "chair_name": "שם יו\\\"ר ההרכב של **ההחלטה הזו** — רלוונטי **רק להחלטות ועדת ערר**, לא לפסקי בית משפט. כמעט תמיד מופיע — בשני מקומות: (א) בכותרת/רובריקה בראש המסמך, ליד 'בפני:' / 'בהרכב:' / רשימת חברי הוועדה; (ב) בבלוק-החתימה בסוף ההחלטה, אחרי 'ההחלטה ניתנה' — שם מופיעים זה-לצד-זה מזכיר/ת הוועדה והיו\\\"ר (למשל בשתי עמודות: בצד אחד 'פלוני, עו\\\"ד / מזכיר ועדת הערר' ובצד השני 'אלמוני, עו\\\"ד / יו\\\"ר ועדת הערר'). **קח את השם שמעליו/לצדו כתוב 'יו\\\"ר' — לא את המזכיר/ה.** השאר שם פרטי+משפחה בלבד, בלי תוארים ('עו\\\"ד', 'אדריכל', 'עו\\\"ד דפנה תמיר'→'דפנה תמיר'). **אזהרה קריטית:** אל תיקח שם יו\\\"ר של פסק/החלטה אחרים ש**מצוטטים** בגוף ההחלטה (למשל 'כפי שנקבע ברשותה של יו\\\"ר פלונית בערר אחר...') — אלה תקדימים מצוטטים, לא היו\\\"ר של ההחלטה הנוכחית. אם זה פסק דין של בית משפט — מחרוזת ריקה.", "district": "מחוז ועדת הערר — רלוונטי **רק להחלטות ועדת ערר**. ערכים מותרים: 'ירושלים', 'תל אביב', 'מרכז', 'חיפה', 'צפון', 'דרום', 'ארצית'. זהה מהכותרת ('ועדת הערר לתכנון ובניה — מחוז ירושלים' → 'ירושלים'; 'ועדות ערר - תכנון ובנייה תל אביב-יפו' → 'תל אביב'). אם זה פסק דין של בית משפט — מחרוזת ריקה.", "parties": "שמות הצדדים בשורה אחת בצורה 'עורר נ\\' משיב' — בדיוק כפי שמופיעים בכותרת/רובריקה. בלי הדגשה, בלי מספר-תיק, בלי תוארים מיותרים. למשל 'ישיבת חברת אהבת שלום נ\\' תאיה' או 'ראם חיים נ\\' הוועדה המקומית לתכנון ובניה ירושלים'. אם הצדדים אינם מופיעים בטקסט (למשל החלטה שמתחילה בגוף בלי רובריקה) — מחרוזת ריקה. **אל תמציא שמות.**", @@ -242,6 +243,37 @@ def _is_clean_docket(s: str) -> bool: return bool(_DOCKET_RE.fullmatch((s or "").strip())) +def _strip_invisibles(s: str) -> str: + """Drop Unicode format chars (category Cf) — RLM/LRM/ZWJ and friends. + + Hebrew legal PDFs carry bidi marks *inside* docket numbers, so a plain + substring test against the raw text misses a docket that is plainly there. + """ + return "".join(ch for ch in (s or "") if unicodedata.category(ch) != "Cf") + + +def _docket_grounded(docket: str, *sources: str) -> bool: + """True when every digit group of ``docket`` appears, in order, in a source. + + INV-AH (quote-or-retract) applied to the identity field. ``_is_clean_docket`` + only checks the *shape*, so a model that misreads one digit produces a + perfectly well-formed but wrong docket — which is exactly how ערר 1094-09-19 + (פדילה) was stored as ``1094-09-14`` while all five case documents said + ...-19 (#232, trap 3). Shape validation cannot catch that; grounding can. + + Tolerant of the separator (``-`` vs ``/``), of whitespace around it, and of + bidi marks, so a real docket in the source still matches. Sources are the + document text and the value being replaced — never the LLM's own output, + which would make the check circular. + """ + parts = re.split(r"[-/]", (docket or "").strip()) + if not parts or not all(p.isdigit() for p in parts): + return False + pattern = r"\s*[-/]\s*".join(re.escape(p) for p in parts) + haystack = _strip_invisibles("\n".join(s or "" for s in sources)) + return re.search(pattern, haystack) is not None + + def _source_type_for_level(level: str) -> str: """Derive source_type from precedent_level — the library section is driven by source_type, so the two MUST agree (an LLM slip pairing @@ -400,6 +432,23 @@ async def apply_to_record( "already owned by another non-internal row (likely duplicate)", cur_cn, cn_clean, ) + elif not _docket_grounded( + cn_clean, + record.get("full_text") or "", + cur_cn, + record.get("citation_formatted") or "", + ): + # The docket is well-formed but appears nowhere in the decision text + # or in the value it would replace — i.e. the model produced digits + # it cannot point at. case_number is the identity field; a wrong one + # silently detaches the row from every reference to the real case. + # Refuse the write and say so (§6) rather than trust the shape. + logger.warning( + "metadata_extractor: case_number normalization %r→%r REFUSED — the " + "docket does not appear in the decision text or in the current " + "value (ungrounded extraction, INV-AH). Keeping %r.", + cur_cn, cn_clean, cur_cn, + ) else: fields_to_update["case_number"] = cn_clean diff --git a/mcp-server/tests/test_docket_grounding.py b/mcp-server/tests/test_docket_grounding.py new file mode 100644 index 0000000..bf30fae --- /dev/null +++ b/mcp-server/tests/test_docket_grounding.py @@ -0,0 +1,65 @@ +"""#232 trap 3 — a well-formed docket is not necessarily the right docket. + +ערר (מרכז) 1094-09-19 (פדילה) was stored as ``1094-09-14``: shape-valid, so +``_is_clean_docket`` waved it through, but wrong — and case_number is the +identity field, so the row detached from every reference to the real case. +Grounding the digits in the source text is what shape validation cannot do. +""" + +from legal_mcp.services.precedent_metadata_extractor import ( + _docket_grounded, + _is_clean_docket, + _strip_invisibles, +) + + +HEADER = "ערר (ועדות ערר - תכנון ובנייה מרכז) 1094-09-19 פדילה אברהים נ' הוועדה המקומית" + + +def test_the_regression_shape_valid_but_wrong_digit(): + """Both pass the shape check; only the real one is grounded.""" + assert _is_clean_docket("1094-09-14") + assert _is_clean_docket("1094-09-19") + assert _docket_grounded("1094-09-19", HEADER) + assert not _docket_grounded("1094-09-14", HEADER) + + +def test_separator_and_spacing_are_tolerated(): + """A real docket must still match when the source writes it differently.""" + assert _docket_grounded("1094-09-19", "בערר 1094/09/19 נקבע") + assert _docket_grounded("4768/22", "עת\"מ 4768-22 פלוני") + assert _docket_grounded("1132-09-24", "תיק 1132 - 09 - 24") + + +def test_bidi_marks_inside_the_number_do_not_defeat_grounding(): + """Hebrew legal PDFs embed RLM/LRM between digits and separators.""" + noisy = "ערר ‏(‏מרכז‏)‏ 1094‏-‏09‏-‏19 פדילה" + assert _strip_invisibles(noisy).count("‏") == 0 + assert _docket_grounded("1094-09-19", noisy) + + +def test_grounding_accepts_the_value_being_replaced(): + """Normalising an uploader's citation string into a clean docket is the + whole point of the rewrite — the digits come from there, not the text.""" + citation = "ערר (ועדות ערר - תכנון ובנייה מרכז) 1094-09-19 פדילה נ' טירה (נבו 4.12.2019)" + assert _docket_grounded("1094-09-19", "", citation) + assert not _docket_grounded("1094-09-14", "", citation) + + +def test_two_and_three_part_dockets_both_ground(): + assert _docket_grounded("8031/21", "בהיטל השבחה 8031/21 נדון") + assert _docket_grounded("85074-09-24", "בל\"מ 85074-09-24") + + +def test_non_numeric_or_empty_never_grounds(): + assert not _docket_grounded("", HEADER) + assert not _docket_grounded("ערר 1094", HEADER) + assert not _docket_grounded("abc-de", HEADER) + + +def test_absent_from_every_source_is_refused(): + """The פדילה failure mode: text has no docket at all, so anything the + model offers is ungrounded and must not reach the identity field.""" + body = "בפני: יו\"ר הוועדה: רונית אלפר, עו\"ד\nהעוררים: 1. פדילה אברהים" + assert not _docket_grounded("1094-09-14", body) + assert not _docket_grounded("1094-09-19", body)