feat(documents): מסמך-עיקרי — is_primary נגזר + רשימה קנונית (#200) #354

Merged
chaim merged 1 commits from worktree-agent-a59ef17c17ecd1d31 into main 2026-06-30 12:03:01 +00:00
3 changed files with 87 additions and 2 deletions

View File

@@ -19,7 +19,7 @@
| ישות | תפקיד | מזהה-קנוני | שדות-מפתח (מאומתים `db.py`) |
|------|--------|-------------|------------------------------|
| `cases` | תיק ערר חי (1xxx/8xxx/9xxx) | `case_number` + `proceeding_type` | `title`, `status`, `practice_area`, `appeal_subtype`, `proceeding_type`, `chair_name` (`db.py:74-91,182-189,747,912`) |
| `documents` | מסמך-מקור משויך לתיק | `id` (UUID); FK→`cases` | `doc_type`, `title`, `file_path`, `extracted_text`, `extraction_status`, `page_count` (`db.py:93-104`) |
| `documents` | מסמך-מקור משויך לתיק | `id` (UUID); FK→`cases` | `doc_type`, `is_primary` (**נגזר** מ-`doc_type`, ראה [§2ג](#2ג-מסמך-עיקרי-primary-document)), `title`, `file_path`, `extracted_text`, `extraction_status`, `page_count` (`db.py:93-104`) |
| `document_chunks` | chunk של מסמך-תיק + embedding | `id`; FK→`documents`/`cases` | `chunk_index`, `content`, `section_type`, `embedding vector(1024)`, `page_number` (`db.py:106-116`) |
| `case_law` | קורפוס פסיקה — חיצוני **וגם** החלטות-ועדה | ראה [§2 + INV-DM2](#inv-dm2-מזהה-קנוני-יחיד-לכל-ישות) | `case_name`, `court`, `practice_area`, `source_kind`, `proceeding_type`, `source_type`, `headnote`, `summary`, `subject_tags`, `extraction_status`, `halacha_extraction_status` (`db.py:366-378,522-526,599-611,883,907`) |
| `precedent_chunks` | chunk של פסק-דין מואנדקס (`source_kind='external_upload'`/`internal_committee`) | `id`; FK→`case_law` | `chunk_index`, `content`, `section_type`, `page_number`, `embedding vector(1024)`, `content_tsv` (`db.py:624-634,776`) |
@@ -77,7 +77,25 @@ proceeding_type)`. לכן המזהה הקנוני הוא **(`case_number` מנו
- `decision_blocks` → usable: `block_id`∈12-הבלוקים; "מוכן": `status=final` ו-`content` לא-ריק.
- `chair_feedback` → usable: `feedback_text`+`category` מהמילון; "פתוח" עד `resolved=true`.
### 2ג. ישויות-נגזרות (אחסון-ניתוחים)
### 2ג. מסמך-עיקרי (primary document)
WS2 (#200) מבחין בין **מסמך-עיקרי** (מסמך-מהות שהיו"ר עוקבת אחריו לניתוח) לבין מסמך-משני.
**הרשימה הקנונית של doc_types עיקריים** (מאושרת-יו"ר): `appeal` (כתב-ערר) · `response`
(תשובה/תגובה) · `objection` (התנגדות) · `protocol` (פרוטוקול-דיון) · `appraisal` (שומה) ·
`decision` (החלטת-ועדה). **כל doc_type אחר משני** (`plan`/`permit`/`court_decision`/
`exhibit`/`reference`).
**`is_primary` נגזר, לא נכתב** ([G1](00-constitution.md#inv-g1-מזהה-קנוני-מנורמל-בכתיבה)/[G2](00-constitution.md#inv-g2-מקור-אמת-יחיד--אין-מסלולים-מקבילים-מתפצלים)/
[INV-DM7](#inv-dm7-סיווג-הלכה--סמכות-נגזרת--תפקיד-כלל-מסווג-שני-צירים-לא-enum-אחד)): מקור-האמת היחיד הוא
`doc_type` + הרשימה הקנונית `PRIMARY_DOC_TYPES` (`db.py`). העמודה
`documents.is_primary BOOLEAN GENERATED ALWAYS AS (doc_type = ANY(PRIMARY_DOC_TYPES)) STORED`
(SCHEMA_V47) מחושבת ע"י Postgres — **אין מסלול-כתיבה מקביל ולא יכולה לסטות** מ-`doc_type`,
בדיוק כמו ה-tsvectors ב-[INV-DM3](#inv-dm3-שינוי-תוכן--re-index). `_row_to_doc` גוזר גם בקריאה
(לרשומות טרום-מיגרציה) ומוסיף `doc_category ∈ {primary, secondary}` לפלט. נחשף דרך
`document_list` / `case_get` / API-המסמכים. אינדקס חלקי `idx_documents_primary(case_id) WHERE
is_primary` משרת את שאילתת "מסמכים-עיקריים שטרם-נותחו" (#201).
### 2ד. ישויות-נגזרות (אחסון-ניתוחים)
מעבר לישויות-המקור, המערכת **שומרת ניתוחים נגזרים** — תוצרי-חילוץ של LLM/קוד. אלו כפופים לכללי
ה-provenance של [X8](X8-field-provenance.md) ולשערי [G10](00-constitution.md#inv-g10-המערכת-מסייעת--שערים-אנושיים-הם-invariant):

View File

@@ -18,6 +18,43 @@ from legal_mcp.services import court_citation, halacha_quality, principles
logger = logging.getLogger(__name__)
# ── Primary-document classification (WS2 / task #200) ────────────────
# Canonical, single-source list of "primary" (מסמך עיקרי) doc_types — the
# substantive case documents the chair tracks for analysis (appeal, the
# replies/objections, the hearing protocol, the appraisal, and the
# committee decision). Every OTHER doc_type (plan, permit, court_decision,
# exhibit, reference…) is secondary. Chair-approved set (plan WS2 §1).
#
# G1/G2/INV-DM7: `is_primary` is DERIVED from `doc_type`, never an
# independently-written column. There is ONE source of truth: this tuple.
# The DB column `documents.is_primary` is a GENERATED ALWAYS … STORED column
# (V47) computed by Postgres from `doc_type`, so it can never drift; the
# Python helper below mirrors the same list for read-time derivation and for
# building the generated-column expression. No parallel write path.
PRIMARY_DOC_TYPES: tuple[str, ...] = (
"appeal", # כתב-ערר
"response", # תשובה / תגובה
"objection", # התנגדות
"protocol", # פרוטוקול-דיון
"appraisal", # שומה
"decision", # החלטת-ועדה
)
def is_primary_doc_type(doc_type: str | None) -> bool:
"""Whether a doc_type is a 'primary' (מסמך עיקרי) document. Single source
of truth = PRIMARY_DOC_TYPES; mirrors the V47 generated column."""
return doc_type in PRIMARY_DOC_TYPES
def _primary_doc_types_sql_array() -> str:
"""SQL ARRAY[...] literal of PRIMARY_DOC_TYPES for the generated column.
Values are fixed identifiers from this module (no user input) — safe to
inline; kept derived from the tuple so the list has ONE definition."""
quoted = ", ".join("'" + t.replace("'", "''") + "'" for t in PRIMARY_DOC_TYPES)
return f"ARRAY[{quoted}]::text[]"
_pool: asyncpg.Pool | None = None
_schema_ready: bool = False
_init_lock: asyncio.Lock = asyncio.Lock()
@@ -1782,6 +1819,22 @@ CREATE INDEX IF NOT EXISTS idx_decision_lessons_vec
ON decision_lessons USING ivfflat (embedding vector_cosine_ops) WITH (lists = 30);
"""
# V47 (WS2 / task #200): "primary document" (מסמך עיקרי) concept.
# `documents.is_primary` is a GENERATED ALWAYS … STORED column derived purely
# from `doc_type` against the canonical PRIMARY_DOC_TYPES list (built from the
# single-source tuple via _primary_doc_types_sql_array). Because Postgres
# computes it, there is NO parallel write path and it can never drift from
# doc_type (G1/G2/INV-DM7 — same drift-free pattern as the GENERATED tsvectors,
# INV-DM3). Idempotent: ADD COLUMN IF NOT EXISTS; the generated expression is
# fixed, so re-running is a no-op. The partial index serves the
# "primary docs not yet analysed" queries that #201 builds on.
SCHEMA_V47_SQL = f"""
ALTER TABLE documents ADD COLUMN IF NOT EXISTS is_primary BOOLEAN
GENERATED ALWAYS AS (doc_type = ANY({_primary_doc_types_sql_array()})) STORED;
CREATE INDEX IF NOT EXISTS idx_documents_primary
ON documents(case_id) WHERE is_primary;
"""
# Stable, arbitrary key for the session-level advisory lock that serialises
# schema DDL across processes. Every short-lived process (cron drains, services)
@@ -1851,6 +1904,7 @@ async def _apply_schema_ddl(conn: asyncpg.Connection) -> None:
await conn.execute(SCHEMA_V44_SQL)
await conn.execute(SCHEMA_V45_SQL)
await conn.execute(SCHEMA_V46_SQL)
await conn.execute(SCHEMA_V47_SQL)
async def init_schema() -> None:
@@ -2223,6 +2277,14 @@ def _row_to_doc(row: asyncpg.Record) -> dict:
d["case_id"] = str(d["case_id"])
if isinstance(d.get("metadata"), str):
d["metadata"] = json.loads(d["metadata"])
# Primary/secondary classification (WS2 / #200). `is_primary` is the
# generated DB column (V47); derive it at read-time too so the field is
# always present even on rows fetched before the migration ran, and expose
# a human-facing `doc_category`. Single source of truth: PRIMARY_DOC_TYPES.
is_primary = bool(d["is_primary"]) if d.get("is_primary") is not None \
else is_primary_doc_type(d.get("doc_type"))
d["is_primary"] = is_primary
d["doc_category"] = "primary" if is_primary else "secondary"
return d

View File

@@ -264,6 +264,11 @@ async def document_get_text(case_number: str, doc_title: str = "") -> str:
async def document_list(case_number: str) -> str:
"""רשימת מסמכים בתיק.
כל מסמך כולל `doc_type`, וכן את הסיווג הנגזר `is_primary` (bool) ו-
`doc_category` ("primary"/"secondary") — מסמך-עיקרי (ערר/תשובה/התנגדות/
פרוטוקול/שומה/החלטת-ועדה) מול משני. נגזר מ-`doc_type` (db.PRIMARY_DOC_TYPES),
אינו נכתב ידנית.
Args:
case_number: מספר תיק הערר
"""