"""Chair-position round-trip in analysis-and-research.md. Regression cover for the reader/writer asymmetry that made "עמדת ועדת הערר" appear to save on threshold claims and then vanish on refresh: the writer matched the label anywhere on a line, the reader only at the start of one, and the analyst agent writes threshold claims as a bullet list ("- **עמדת ועדת הערר:**") while it writes issues flush-left. The invariant these tests pin down: whatever update_chair_position writes, parse() must read back — for every label form that appears in the corpus. """ from __future__ import annotations from pathlib import Path import pytest from legal_mcp.services import research_md BULLETED = """# ניתוח משפטי — ערר 1069-04-26 תאריך: 2026-08-02 ## 5. טענות סף ### ס-1 — אי-מילוי תנאי ההפקדה - **טענה (claim):** התכנית אושרה מבלי שמולאו תנאי ההפקדה. - **שאלה משפטית:** האם הסטת התנאים פוגמת בחוקיות? - **עמדת ועדת הערר:** [ימולא ע"י יו"ר הוועדה] ### ס-2 — זכות עמידה - **טענה (claim):** לעוררים אין אינטרס מוגן. - **עמדת ועדת הערר:** [ימולא ע"י יו"ר הוועדה] ## 6. סוגיות להכרעה ### סוגיה 1: סמכות לפי 62א(א)(11) **ניתוח:** - הכלל החל: ... **עמדת ועדת הערר:** [ימולא ע"י יו"ר הוועדה] --- """ @pytest.fixture() def analysis_file(tmp_path: Path) -> Path: path = tmp_path / "analysis-and-research.md" path.write_text(BULLETED, encoding="utf-8") return path def _positions(path: Path) -> dict[str, str]: parsed = research_md.parse(path) items = parsed["threshold_claims"] + parsed["issues"] return {item["id"]: item["chair_position"] for item in items} # ── the reported bug ──────────────────────────────────────────── def test_bulleted_threshold_position_survives_a_reload(analysis_file: Path) -> None: """The bug as chaim hit it: saved on a threshold claim, gone on refresh.""" research_md.update_chair_position(analysis_file, "threshold_1", "עמדתנו היא שהטענה נדחית.") assert _positions(analysis_file)["threshold_1"] == "עמדתנו היא שהטענה נדחית." def test_flush_left_issue_position_still_round_trips(analysis_file: Path) -> None: """The surface that already worked must keep working.""" research_md.update_chair_position(analysis_file, "issue_1", "יש לקבל את הערר בנקודה זו.") assert _positions(analysis_file)["issue_1"] == "יש לקבל את הערר בנקודה זו." def test_bulleted_threshold_fields_are_parsed_at_all(analysis_file: Path) -> None: """Bulleted labels were invisible to the reader — the whole claim looked empty.""" threshold = research_md.parse(analysis_file)["threshold_claims"] labels = [f["label"] for f in threshold[0]["fields"]] assert "טענה (claim)" in labels assert "שאלה משפטית" in labels # the chair field is surfaced separately, never as a regular field assert not any(research_md._is_chair_label(label) for label in labels) # ── writer/reader agreement on content boundaries ─────────────── def test_position_containing_markdown_is_not_truncated(analysis_file: Path) -> None: """The old writer stopped the field at the first '*' and swallowed the rest.""" text = 'העמדה כוללת **הדגשה** וגם קו --- באמצע\nושורה שנייה.' research_md.update_chair_position(analysis_file, "issue_1", text) assert _positions(analysis_file)["issue_1"] == text def test_update_does_not_touch_sibling_subsections(analysis_file: Path) -> None: research_md.update_chair_position(analysis_file, "threshold_1", "ראשונה") research_md.update_chair_position(analysis_file, "threshold_2", "שנייה") positions = _positions(analysis_file) assert positions["threshold_1"] == "ראשונה" assert positions["threshold_2"] == "שנייה" assert positions["issue_1"] == "" def test_closing_rule_and_bullet_marker_are_preserved(analysis_file: Path) -> None: """An update must not restructure the file the analyst produced.""" research_md.update_chair_position(analysis_file, "threshold_1", "עמדה") research_md.update_chair_position(analysis_file, "issue_1", "עמדה") content = analysis_file.read_text(encoding="utf-8") assert "- **עמדת ועדת הערר:**" in content assert content.rstrip().endswith("---") def test_clearing_a_position_restores_the_placeholder(analysis_file: Path) -> None: research_md.update_chair_position(analysis_file, "threshold_1", "עמדה") research_md.update_chair_position(analysis_file, "threshold_1", "") assert _positions(analysis_file)["threshold_1"] == "" assert research_md.CHAIR_POSITION_PLACEHOLDERS[0] in analysis_file.read_text( encoding="utf-8" ) def test_repeated_saves_do_not_accumulate_blocks(analysis_file: Path) -> None: for text in ("ראשון", "שני", "שלישי"): research_md.update_chair_position(analysis_file, "threshold_1", text) content = analysis_file.read_text(encoding="utf-8") assert content.count(f"**{research_md.CHAIR_POSITION_LABEL}:**") == 3 assert _positions(analysis_file)["threshold_1"] == "שלישי" # ── decorated and duplicated labels ───────────────────────────── def test_decorated_label_is_updated_not_duplicated(tmp_path: Path) -> None: """'עמדת ועדת הערר (הכוונת יו"ר 24.6)' appears in the corpus (8125-09-24).""" path = tmp_path / "a.md" path.write_text( "## 6. סוגיות להכרעה\n\n" "### סוגיה 1: כותרת\n\n" '**עמדת ועדת הערר (הכוונת יו"ר 24.6):** [ימולא ע"י יו"ר הוועדה]\n', encoding="utf-8", ) research_md.update_chair_position(path, "issue_1", "עמדה מעודכנת") content = path.read_text(encoding="utf-8") assert content.count(research_md.CHAIR_POSITION_LABEL) == 1 assert _positions(path)["issue_1"] == "עמדה מעודכנת" def test_duplicate_labels_read_back_the_one_that_was_written(tmp_path: Path) -> None: """An H3 covering several issues (1033-02-25) must not read a sibling's text.""" path = tmp_path / "a.md" path.write_text( "## 6. סוגיות להכרעה\n\n" "### סוגיות 4–6: מקובצות\n\n" "**עמדת ועדת הערר:** ראשונה\n\n" "**עמדת ועדת הערר:** אחרונה\n", encoding="utf-8", ) research_md.update_chair_position(path, "issue_1", "העמדה הנכונה") assert _positions(path)["issue_1"] == "העמדה הנכונה" # ── read-after-write guard ────────────────────────────────────── def test_update_returns_the_persisted_value(analysis_file: Path) -> None: result = research_md.update_chair_position(analysis_file, "threshold_1", " עמדה ") assert result["saved"] is True assert result["position"] == "עמדה" def test_a_write_the_parser_cannot_read_back_is_reported_as_failure( analysis_file: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """Regression guard: the bug hid behind a success response for months.""" monkeypatch.setattr( research_md, "_stored_chair_position", lambda *_args: "משהו אחר" ) with pytest.raises(RuntimeError, match="לא נקראה בחזרה"): research_md.update_chair_position(analysis_file, "threshold_1", "עמדה") # ── downstream consumers see the recovered positions ──────────── def test_chair_directions_count_bulleted_positions(analysis_file: Path) -> None: """extract_chair_directions feeds legal-writer; it under-reported before.""" research_md.update_chair_position(analysis_file, "threshold_1", "עמדה") directions = research_md.extract_chair_directions(analysis_file) assert directions["filled_count"] == 1 assert directions["status"] == "partial" filled = [t for t in directions["threshold_claims"] if t["direction"]] assert [t["id"] for t in filled] == ["threshold_1"] # ── the real corpus ───────────────────────────────────────────── CORPUS = sorted( Path("/home/chaim/legal-ai/data/cases").glob( "*/documents/research/analysis-and-research.md" ) ) @pytest.mark.skipif(not CORPUS, reason="case corpus not present on this host") @pytest.mark.parametrize("source", CORPUS, ids=lambda p: p.parts[-4]) def test_every_corpus_subsection_round_trips(source: Path, tmp_path: Path) -> None: sentinel = "עמדת-בדיקה — **הדגשה** ו---קו\nושורה שנייה." before = _positions(source) for section_id in before: working = tmp_path / f"{section_id}.md" working.write_text(source.read_text(encoding="utf-8"), encoding="utf-8") research_md.update_chair_position(working, section_id, sentinel) after = _positions(working) assert after[section_id] == sentinel, f"{source.parts[-4]} {section_id}" untouched = {k: v for k, v in after.items() if k != section_id} assert untouched == {k: v for k, v in before.items() if k != section_id}