"""Unit tests for the extraction adapters (core/extraction/adapters/*). Real files are generated in-memory per format (pymupdf/python-docx/PIL) so no binary fixtures live in the repo. The OCR adapter's engine-dependent happy path is covered when tesseract is available; error-contract tests always run. """ from __future__ import annotations from pathlib import Path import pytest from contract_check.core.extraction import ( ExtractedDocument, ExtractionFailedError, MammothDocxExtractor, PyMuPDFExtractor, RtfExtractor, TesseractOcrExtractor, TxtExtractor, ) # ── shared fixture builders ────────────────────────────────────────────────── PARAGRAPH = ( "Стороны обязуются выполнять условия договора. " "Сторона А обязуется передать товар в срок. " "Сторона Б обязуется оплатить товар в течение десяти банковских дней. " "Ответственность сторон ограничена суммой договора. " "Споры подлежат рассмотрению в арбитражном суде города Москвы. " ) def build_pdf_bytes(*, with_table: bool = False) -> bytes: import pymupdf doc = pymupdf.open() page = doc.new_page() if with_table: page.insert_htmlbox( page.rect, f'
{PARAGRAPH}
', ) x0, y0, cw, ch = 50, 200, 150, 30 data = [["Услуга", "Цена"], ["Консультация", "5000"], ["Аудит", "15000"]] for r in range(len(data) + 1): page.draw_line(pymupdf.Point(x0, y0 + r * ch), pymupdf.Point(x0 + 2 * cw, y0 + r * ch)) for c in range(3): page.draw_line(pymupdf.Point(x0 + c * cw, y0), pymupdf.Point(x0 + c * cw, y0 + 3 * ch)) for r, row in enumerate(data): for c, val in enumerate(row): rect = pymupdf.Rect( x0 + c * cw + 3, y0 + r * ch + 3, x0 + (c + 1) * cw - 3, y0 + (r + 1) * ch - 3 ) page.insert_htmlbox( rect, f'{val}' ) else: page.insert_htmlbox( page.rect, f'{PARAGRAPH}
' ) data = doc.tobytes() doc.close() return data def build_docx_bytes() -> bytes: import io from docx import Document doc = Document() doc.add_heading("Договор поставки", level=1) doc.add_paragraph(PARAGRAPH) doc.add_paragraph(PARAGRAPH) buf = io.BytesIO() doc.save(buf) return buf.getvalue() # ── PDF (pymupdf) ──────────────────────────────────────────────────────────── def test_pdf_text_only() -> None: result = PyMuPDFExtractor().extract(build_pdf_bytes()) assert "арбитражном суде" in result.markdown assert result.is_structured is False assert result.metadata["format"] == "pdf" assert result.metadata["has_tables"] is False def test_pdf_table_becomes_markdown_pipes() -> None: result = PyMuPDFExtractor().extract(build_pdf_bytes(with_table=True)) assert result.is_structured is True assert result.metadata["has_tables"] is True # Header row + separator + data rows as pipes; no cell-text duplication. assert "| Услуга | Цена |" in result.markdown assert "| Консультация | 5000 |" in result.markdown assert result.markdown.count("Консультация") == 1 def test_pdf_too_short_raises() -> None: import pymupdf doc = pymupdf.open() doc.new_page() data = doc.tobytes() doc.close() with pytest.raises(ExtractionFailedError, match="слишком мало"): PyMuPDFExtractor().extract(data) def test_pdf_garbage_raises() -> None: with pytest.raises(ExtractionFailedError): PyMuPDFExtractor().extract(b"definitely not a pdf") # ── DOCX (mammoth) ─────────────────────────────────────────────────────────── def test_docx_heading_preserved() -> None: result = MammothDocxExtractor().extract(build_docx_bytes()) assert result.markdown.startswith("# Договор поставки") assert result.is_structured is True assert "арбитражном суде" in result.markdown def test_docx_garbage_raises() -> None: with pytest.raises(ExtractionFailedError): MammothDocxExtractor().extract(b"not a zip") # ── RTF (striprtf) ─────────────────────────────────────────────────────────── def test_rtf_extracted_as_plain_text() -> None: body = " ".join(f"Clause {i}: the parties agree to the terms herein." for i in range(30)) data = r"{\rtf1\ansi\deff0 " + body.replace("\n", r"\par ") + "}" result = RtfExtractor().extract(data.encode("ascii")) assert "the parties agree" in result.markdown assert result.is_structured is False def test_rtf_missing_header_raises() -> None: with pytest.raises(ExtractionFailedError, match="RTF"): RtfExtractor().extract(b"just some text that is long enough " * 10) # ── TXT (chardet) ──────────────────────────────────────────────────────────── def test_txt_windows1251_detected() -> None: data = (PARAGRAPH * 10).encode("windows-1251") result = TxtExtractor().extract(data) assert "арбитражном суде" in result.markdown # decoded, not mojibake assert result.metadata["encoding"] == "windows-1251" assert result.is_structured is False def test_txt_utf8() -> None: result = TxtExtractor().extract((PARAGRAPH * 5).encode("utf-8")) assert "Стороны" in result.markdown assert result.metadata["encoding"] in ("utf-8", "utf-8-sig") def test_txt_too_short_raises() -> None: with pytest.raises(ExtractionFailedError): TxtExtractor().extract("коротко".encode()) # ── OCR (tesseract) ───────────────────────────────────────────────────────── def _tesseract_available() -> bool: import os import shutil if shutil.which("tesseract") is None: return False # The default adapter uses rus+eng; skip if traineddata is missing. tessdata = os.environ.get("TESSDATA_PREFIX", "/usr/share/tessdata") return os.path.exists(os.path.join(tessdata, "eng.traineddata")) def build_png_bytes(tmp_path: Path) -> bytes: from PIL import Image, ImageDraw, ImageFont text = ( "Contract. The parties agree to the terms herein. " "Party A shall deliver the goods within ten business days. " "Party B shall pay the invoice within thirty days. " "Liability is limited to the contract value. " "Disputes shall be resolved in arbitration in Moscow." ) font_path = "/usr/share/fonts/noto/NotoSans-Regular.ttf" try: font = ImageFont.truetype(font_path, 24) except OSError: font = ImageFont.load_default() img = Image.new("RGB", (1200, 400), color="white") draw = ImageDraw.Draw(img) draw.text((20, 20), text, fill="black", font=font) path = tmp_path / "contract.png" img.save(path, format="PNG") return path.read_bytes() @pytest.mark.skipif(not _tesseract_available(), reason="tesseract not installed") def test_ocr_image_extracts_text(tmp_path: Path) -> None: data = build_png_bytes(tmp_path) result = TesseractOcrExtractor().extract(data) assert "Contract" in result.markdown assert result.metadata["format"] == "image" assert result.metadata["pages"] == 1 def test_ocr_garbage_pdf_raises() -> None: from contract_check.core.analysis.ocr import OCRError with pytest.raises((OCRError, ExtractionFailedError)): TesseractOcrExtractor().extract(b"%PDF-1.4 garbage") def test_ocr_garbage_image_raises() -> None: from contract_check.core.analysis.ocr import OCRError with pytest.raises((OCRError, ExtractionFailedError)): TesseractOcrExtractor().extract(b"\x89PNG\r\n\x1a\n not really a png") # ── ExtractedDocument DTO ──────────────────────────────────────────────────── def test_extracted_document_defaults() -> None: doc = ExtractedDocument(markdown="x") assert doc.is_structured is False assert doc.metadata == {}