fix(extraction): סינון cited_only מתור/מוני החילוץ (#140)

31 שורות case_law עם source_kind='cited_only' (ציטוט-בלבד, ללא full_text/chunks) נושאות halacha_extraction_status='pending' רק כברירת-מחדל ומזהמות את מונה ה-pending ובמתזמר/בדף-התפעול — אין להן מה לחלץ. תיקון (G1 — תיקון-במקור, G2 — מסנן יחיד משותף): - db.EXTRACTION_ELIGIBLE_PREDICATE — מקור-אמת יחיד ל"שורה ברת-חילוץ" (source_kind <> 'cited_only' AND יש precedent_chunks). מוחל ב-list_pending_extraction_requests; #139 יעשה בו שימוש-חוזר ל-reconcile (אותו כלל, לא כפול). - מוני-snapshot מסננים cited_only: halacha_drain_supervisor.db_snapshot, web/app.py meta+hal_ext (GROUP BY status). - reconcile_metadata_status.py מורחב לכסות גם את תור-ההלכות: cited_only→'skipped' (אותו terminal-state כמו צד-המטא, תור-תאום, G2). בוצע על ה-DB החי: 31 הועברו ל-'skipped' (metadata כבר היה מיושב — אידמפוטנטי). התפלגות-אחרי: halacha pending=9 (עבודה אמיתית), skipped=31, completed=309. בדיקות: test_extraction_queue_eligibility (predicate + list_pending מחיל אותו, שני ה-kinds). כל 345 בדיקות mcp עוברות. guards נקיים. Invariants: G1 (terminal-state אמיתי במקור), G2 (predicate יחיד, ללא תור מקביל), INV-DM1 (stub לא-searchable אינו מועמד-חילוץ), G12 (leak-guard נקי). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-15 04:03:21 +00:00
parent 7043de0ac2
commit c348903e4b
6 changed files with 114 additions and 9 deletions
--- a/mcp-server/src/legal_mcp/services/db.py
+++ b/mcp-server/src/legal_mcp/services/db.py
@@ -6746,6 +6746,19 @@ async def request_halacha_extraction(case_law_id: UUID) -> bool:
    return result == "UPDATE 1"


+# Single source of truth for "this row can actually be extracted" (#140, G2):
+# a real precedent whose text was chunked into precedent_chunks. ``cited_only``
+# stubs are citation-only (no full_text, no chunks) and can NEVER yield an
+# extraction, so they must never enter the work queue or get proactively
+# re-queued. Shared by the queue reader (list_pending_extraction_requests) and
+# the orphan-reconcile job (#139) so the eligibility rule lives in ONE place.
+# Assumes the surrounding query exposes the table as ``case_law``.
+EXTRACTION_ELIGIBLE_PREDICATE = (
+    "case_law.source_kind <> 'cited_only' "
+    "AND EXISTS (SELECT 1 FROM precedent_chunks pc WHERE pc.case_law_id = case_law.id)"
+)
+
+
 async def list_pending_extraction_requests(
    kind: str = "metadata",  # 'metadata' | 'halacha'
    limit: int = 20,
@@ -6764,11 +6777,14 @@ async def list_pending_extraction_requests(
    # internal_committee rows could be stamped (we opened that gate in
    # request_metadata_extraction / request_halacha_extraction) but stayed
    # invisible to the worker forever.
+    # Exclude ineligible rows (cited_only / chunkless) so a stub can never sit in
+    # the work queue — same predicate the reconcile job uses (#140, G2).
    rows = await pool.fetch(
        f"""SELECT id, case_number, case_name, court, date,
                  practice_area, is_binding, {col} AS requested_at
            FROM case_law
            WHERE {col} IS NOT NULL
+              AND {EXTRACTION_ELIGIBLE_PREDICATE}
            ORDER BY {col} ASC
            LIMIT $1""",
        limit,
--- a/mcp-server/tests/test_extraction_queue_eligibility.py
+++ b/mcp-server/tests/test_extraction_queue_eligibility.py
@@ -0,0 +1,66 @@
+"""Regression test for #140 — cited_only stubs must never enter the extraction
+work queue.
+
+``list_pending_extraction_requests`` must apply ``EXTRACTION_ELIGIBLE_PREDICATE``
+so a citation-only stub (no full_text, no precedent_chunks) is excluded even if
+it carries a stamped ``*_extraction_requested_at`` and a default 'pending'
+status. The predicate is the single shared eligibility rule (#139 reuses it).
+
+Runs OFFLINE — a fake pool captures the SQL and asserts the predicate is wired
+into the WHERE clause (same style as test_halacha_reextract_preserves_approved).
+"""
+
+from __future__ import annotations
+
+import asyncio
+
+import pytest
+
+from legal_mcp.services import db
+
+
+class _FakePool:
+    def __init__(self) -> None:
+        self.fetched: list[str] = []
+
+    async def fetch(self, sql: str, *args):  # noqa: ANN002
+        self.fetched.append(sql)
+        return []
+
+
+@pytest.fixture()
+def fake_pool(monkeypatch: pytest.MonkeyPatch) -> _FakePool:
+    pool = _FakePool()
+
+    async def _get_pool() -> _FakePool:
+        return pool
+
+    monkeypatch.setattr(db, "get_pool", _get_pool)
+    return pool
+
+
+def _norm(sql: str) -> str:
+    return " ".join(sql.split())
+
+
+def test_predicate_excludes_cited_only_and_requires_chunks() -> None:
+    pred = _norm(db.EXTRACTION_ELIGIBLE_PREDICATE)
+    assert "source_kind <> 'cited_only'" in pred
+    assert "precedent_chunks" in pred and "EXISTS" in pred.upper()
+
+
+@pytest.mark.parametrize("kind", ["metadata", "halacha"])
+def test_list_pending_applies_eligibility_predicate(fake_pool: _FakePool, kind: str) -> None:
+    loop = asyncio.new_event_loop()
+    try:
+        loop.run_until_complete(db.list_pending_extraction_requests(kind=kind))
+    finally:
+        loop.close()
+
+    assert fake_pool.fetched, "expected a queue query"
+    sql = _norm(fake_pool.fetched[0])
+    # The eligibility predicate must be ANDed into the queue WHERE clause.
+    assert _norm(db.EXTRACTION_ELIGIBLE_PREDICATE) in sql, sql
+    # ...alongside the requested_at gate, for the correct kind.
+    col = "metadata_extraction_requested_at" if kind == "metadata" else "halacha_extraction_requested_at"
+    assert f"{col} IS NOT NULL" in sql, sql