diff --git a/packages/markitdown/README.md b/packages/markitdown/README.md index 2ac98708d6..d29e7d415a 100644 --- a/packages/markitdown/README.md +++ b/packages/markitdown/README.md @@ -53,3 +53,14 @@ trademarks or logos is subject to and must follow [Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general). Use of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship. Any use of third-party trademarks or logos are subject to those third-party's policies. + +### Optional PDF text recovery + +Install `markitdown[pdf,pdf-recovery]` to enable local PyMuPDF recovery for +plain-text pages truncated around inline images. Recovery also works with +compressed page streams and empty primary text. A replacement must extend the +primary word sequence; pages already recognized as tables/forms retain their +Markdown. The normal extraction path is retained when no page can be recovered +or the optional backend fails. This is conservative text recovery, not OCR; +it does not recover every PDF layout or text missing inside a table/form. +The `pdf-recovery` extra is separate from `all` and `pdf`. diff --git a/packages/markitdown/pyproject.toml b/packages/markitdown/pyproject.toml index 43144043da..d9d393687b 100644 --- a/packages/markitdown/pyproject.toml +++ b/packages/markitdown/pyproject.toml @@ -56,6 +56,7 @@ docx = ["mammoth~=1.11.0", "lxml"] xlsx = ["pandas", "openpyxl"] xls = ["pandas", "xlrd"] pdf = ["pdfminer.six>=20251230", "pdfplumber>=0.11.9"] +pdf-recovery = ["pymupdf>=1.24.3"] outlook = ["olefile"] audio-transcription = ["pydub", "SpeechRecognition"] youtube-transcription = ["youtube-transcript-api~=1.2.3"] diff --git a/packages/markitdown/src/markitdown/converters/_pdf_converter.py b/packages/markitdown/src/markitdown/converters/_pdf_converter.py index ffbcbd990c..1632dd367e 100644 --- a/packages/markitdown/src/markitdown/converters/_pdf_converter.py +++ b/packages/markitdown/src/markitdown/converters/_pdf_converter.py @@ -492,6 +492,45 @@ def _extract_tables_from_words(page: Any) -> list[list[list[str]]]: return [table_rows] +def _recover_inline_image_pages( + pdf_bytes: io.BytesIO, plain_pages: dict[int, str] +) -> dict[int, str]: + """Recover truncated plain pages when the optional PyMuPDF extra is installed.""" + try: + import pymupdf + except ImportError: + return {} + + recovered: dict[int, str] = {} + try: + with pymupdf.open(stream=pdf_bytes.getvalue(), filetype="pdf") as pdf: + for page_index, primary in plain_pages.items(): + try: + page = pdf[page_index] + # Inspect decoded page images, including compressed content + # streams. Inline images have no indirect object reference. + if not any( + image["xref"] == 0 for image in page.get_image_info(xrefs=True) + ): + continue + candidate = page.get_text("text", sort=True).strip() + primary_words = primary.split() + candidate_words = candidate.split() + # Only accept a strict continuation of the primary text; + # length alone can select unrelated or reordered content. + if ( + len(candidate_words) > len(primary_words) + and candidate_words[: len(primary_words)] == primary_words + ): + recovered[page_index] = candidate + except Exception: + # Recovery must not discard an otherwise usable page. + continue + except Exception: + pass + return recovered + + class PdfConverter(DocumentConverter): """ Converts PDFs to Markdown. @@ -547,7 +586,7 @@ def convert( # keep memory usage constant regardless of page count. markdown_chunks: list[str] = [] form_page_count = 0 - plain_page_indices: list[int] = [] + plain_pages: dict[int, str] = {} with pdfplumber.open(pdf_bytes) as pdf: for page_idx, page in enumerate(pdf.pages): @@ -555,23 +594,26 @@ def convert( if page_content is not None: form_page_count += 1 - if page_content.strip(): - markdown_chunks.append(page_content) + markdown_chunks.append(page_content.strip()) else: - plain_page_indices.append(page_idx) - text = page.extract_text() - if text and text.strip(): - markdown_chunks.append(text.strip()) + text = (page.extract_text() or "").strip() + plain_pages[page_idx] = text + markdown_chunks.append(text) page.close() # Free cached page data immediately # If no pages had form-style content, use pdfminer for # the whole document (better text spacing for prose). - if form_page_count == 0: + recovered = _recover_inline_image_pages(pdf_bytes, plain_pages) + for page_idx, text in recovered.items(): + markdown_chunks[page_idx] = text + if form_page_count == 0 and not recovered: pdf_bytes.seek(0) markdown = pdfminer.high_level.extract_text(pdf_bytes) else: - markdown = "\n\n".join(markdown_chunks).strip() + markdown = "\n\n".join( + chunk for chunk in markdown_chunks if chunk + ).strip() except Exception: # Fallback if pdfplumber fails diff --git a/packages/markitdown/tests/test_files/inline-truncated-False.pdf b/packages/markitdown/tests/test_files/inline-truncated-False.pdf new file mode 100644 index 0000000000..199d37b37a Binary files /dev/null and b/packages/markitdown/tests/test_files/inline-truncated-False.pdf differ diff --git a/packages/markitdown/tests/test_files/inline-truncated-True.pdf b/packages/markitdown/tests/test_files/inline-truncated-True.pdf new file mode 100644 index 0000000000..dccc2e0660 Binary files /dev/null and b/packages/markitdown/tests/test_files/inline-truncated-True.pdf differ diff --git a/packages/markitdown/tests/test_files/inline-truncated-mixed.pdf b/packages/markitdown/tests/test_files/inline-truncated-mixed.pdf new file mode 100644 index 0000000000..7e8c8199ac Binary files /dev/null and b/packages/markitdown/tests/test_files/inline-truncated-mixed.pdf differ diff --git a/packages/markitdown/tests/test_files/inline-truncated.txt b/packages/markitdown/tests/test_files/inline-truncated.txt new file mode 100644 index 0000000000..a468f5993a --- /dev/null +++ b/packages/markitdown/tests/test_files/inline-truncated.txt @@ -0,0 +1,5 @@ +Synthetic PDFs generated with PyMuPDF 1.28.2. No real documents or personal data. +An inline 16x16 one-bit image uses ASCII85+Flate with a bare ~ terminator. +PyMuPDF tolerates this malformed terminator; pdfminer consumes following text. +False/True denote uncompressed/Flate-compressed page content streams. +The mixed fixture adds long prose and a four-row three-column table. diff --git a/packages/markitdown/tests/test_pdf_inline_recovery.py b/packages/markitdown/tests/test_pdf_inline_recovery.py new file mode 100644 index 0000000000..ad3d8534ee --- /dev/null +++ b/packages/markitdown/tests/test_pdf_inline_recovery.py @@ -0,0 +1,75 @@ +import io +import sys +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock + +import pytest + +from markitdown import MarkItDown +from markitdown.converters import _pdf_converter as module + +FILES = Path(__file__).parent / "test_files" + + +@pytest.mark.parametrize("compressed", [False, True]) +def test_real_inline_image_recovers_following_text(compressed, monkeypatch): + pytest.importorskip("pymupdf") + path = FILES / f"inline-truncated-{compressed}.pdf" + recovered = MarkItDown().convert(str(path)).markdown + assert "AFTER_IMAGE: line 11" in recovered + monkeypatch.setitem(sys.modules, "pymupdf", None) + primary = MarkItDown().convert(str(path)).markdown + assert "BEFORE_IMAGE" in primary + assert "AFTER_IMAGE" not in primary + + +def test_mixed_document_preserves_table_and_long_neighbor(monkeypatch): + pytest.importorskip("pymupdf") + path = FILES / "inline-truncated-mixed.pdf" + recovered = MarkItDown().convert(str(path)).markdown + monkeypatch.setitem(sys.modules, "pymupdf", None) + primary = MarkItDown().convert(str(path)).markdown + assert len(primary) > 2048 + assert "AFTER_IMAGE: line 11" in recovered + table = primary[primary.index("| Item") :] + assert table in recovered + assert "Healthy prose line 44" in recovered + + +@pytest.mark.parametrize( + "primary,candidate,expected", + [ + ("", "recovered text", True), + ("prefix text", "prefix text more words", True), + ("prefix text", "unrelated much longer candidate text", False), + ("same text", "same text", False), + ], +) +def test_only_strict_text_continuations_are_selected( + monkeypatch, primary, candidate, expected +): + page = MagicMock() + page.get_image_info.return_value = [{"xref": 0}] + page.get_text.return_value = candidate + document = MagicMock() + document.__enter__.return_value = document + document.__getitem__.return_value = page + monkeypatch.setitem( + sys.modules, "pymupdf", SimpleNamespace(open=lambda **kw: document) + ) + result = module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: primary}) + assert result == ({0: candidate} if expected else {}) + document.__exit__.assert_called_once() + + +def test_missing_backend_is_optional(monkeypatch): + monkeypatch.setitem(sys.modules, "pymupdf", None) + assert module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: ""}) == {} + + +def test_backend_failure_preserves_primary(monkeypatch): + backend = MagicMock() + backend.open.side_effect = RuntimeError("broken backend") + monkeypatch.setitem(sys.modules, "pymupdf", backend) + assert module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: "primary"}) == {}