Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions packages/markitdown/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -53,3 +53,14 @@ trademarks or logos is subject to and must follow
[Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general).
Use of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship.
Any use of third-party trademarks or logos are subject to those third-party's policies.

### Optional PDF text recovery

Install `markitdown[pdf,pdf-recovery]` to enable local PyMuPDF recovery for
plain-text pages truncated around inline images. Recovery also works with
compressed page streams and empty primary text. A replacement must extend the
primary word sequence; pages already recognized as tables/forms retain their
Markdown. The normal extraction path is retained when no page can be recovered
or the optional backend fails. This is conservative text recovery, not OCR;
it does not recover every PDF layout or text missing inside a table/form.
The `pdf-recovery` extra is separate from `all` and `pdf`.
1 change: 1 addition & 0 deletions packages/markitdown/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ docx = ["mammoth~=1.11.0", "lxml"]
xlsx = ["pandas", "openpyxl"]
xls = ["pandas", "xlrd"]
pdf = ["pdfminer.six>=20251230", "pdfplumber>=0.11.9"]
pdf-recovery = ["pymupdf>=1.24.3"]
outlook = ["olefile"]
audio-transcription = ["pydub", "SpeechRecognition"]
youtube-transcription = ["youtube-transcript-api~=1.2.3"]
Expand Down
60 changes: 51 additions & 9 deletions packages/markitdown/src/markitdown/converters/_pdf_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -492,6 +492,45 @@ def _extract_tables_from_words(page: Any) -> list[list[list[str]]]:
return [table_rows]


def _recover_inline_image_pages(
pdf_bytes: io.BytesIO, plain_pages: dict[int, str]
) -> dict[int, str]:
"""Recover truncated plain pages when the optional PyMuPDF extra is installed."""
try:
import pymupdf
except ImportError:
return {}

recovered: dict[int, str] = {}
try:
with pymupdf.open(stream=pdf_bytes.getvalue(), filetype="pdf") as pdf:
for page_index, primary in plain_pages.items():
try:
page = pdf[page_index]
# Inspect decoded page images, including compressed content
# streams. Inline images have no indirect object reference.
if not any(
image["xref"] == 0 for image in page.get_image_info(xrefs=True)
):
continue
candidate = page.get_text("text", sort=True).strip()
primary_words = primary.split()
candidate_words = candidate.split()
# Only accept a strict continuation of the primary text;
# length alone can select unrelated or reordered content.
if (
len(candidate_words) > len(primary_words)
and candidate_words[: len(primary_words)] == primary_words
):
recovered[page_index] = candidate
except Exception:
# Recovery must not discard an otherwise usable page.
continue
except Exception:
pass
return recovered


class PdfConverter(DocumentConverter):
"""
Converts PDFs to Markdown.
Expand Down Expand Up @@ -547,31 +586,34 @@ def convert(
# keep memory usage constant regardless of page count.
markdown_chunks: list[str] = []
form_page_count = 0
plain_page_indices: list[int] = []
plain_pages: dict[int, str] = {}

with pdfplumber.open(pdf_bytes) as pdf:
for page_idx, page in enumerate(pdf.pages):
page_content = _extract_form_content_from_words(page)

if page_content is not None:
form_page_count += 1
if page_content.strip():
markdown_chunks.append(page_content)
markdown_chunks.append(page_content.strip())
else:
plain_page_indices.append(page_idx)
text = page.extract_text()
if text and text.strip():
markdown_chunks.append(text.strip())
text = (page.extract_text() or "").strip()
plain_pages[page_idx] = text
markdown_chunks.append(text)

page.close() # Free cached page data immediately

# If no pages had form-style content, use pdfminer for
# the whole document (better text spacing for prose).
if form_page_count == 0:
recovered = _recover_inline_image_pages(pdf_bytes, plain_pages)
for page_idx, text in recovered.items():
markdown_chunks[page_idx] = text
if form_page_count == 0 and not recovered:
pdf_bytes.seek(0)
markdown = pdfminer.high_level.extract_text(pdf_bytes)
else:
markdown = "\n\n".join(markdown_chunks).strip()
markdown = "\n\n".join(
chunk for chunk in markdown_chunks if chunk
).strip()

except Exception:
# Fallback if pdfplumber fails
Expand Down
Binary file not shown.
Binary file not shown.
Binary file not shown.
5 changes: 5 additions & 0 deletions packages/markitdown/tests/test_files/inline-truncated.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
Synthetic PDFs generated with PyMuPDF 1.28.2. No real documents or personal data.
An inline 16x16 one-bit image uses ASCII85+Flate with a bare ~ terminator.
PyMuPDF tolerates this malformed terminator; pdfminer consumes following text.
False/True denote uncompressed/Flate-compressed page content streams.
The mixed fixture adds long prose and a four-row three-column table.
75 changes: 75 additions & 0 deletions packages/markitdown/tests/test_pdf_inline_recovery.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,75 @@
import io
import sys
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import MagicMock

import pytest

from markitdown import MarkItDown
from markitdown.converters import _pdf_converter as module

FILES = Path(__file__).parent / "test_files"


@pytest.mark.parametrize("compressed", [False, True])
def test_real_inline_image_recovers_following_text(compressed, monkeypatch):
pytest.importorskip("pymupdf")
path = FILES / f"inline-truncated-{compressed}.pdf"
recovered = MarkItDown().convert(str(path)).markdown
assert "AFTER_IMAGE: line 11" in recovered
monkeypatch.setitem(sys.modules, "pymupdf", None)
primary = MarkItDown().convert(str(path)).markdown
assert "BEFORE_IMAGE" in primary
assert "AFTER_IMAGE" not in primary


def test_mixed_document_preserves_table_and_long_neighbor(monkeypatch):
pytest.importorskip("pymupdf")
path = FILES / "inline-truncated-mixed.pdf"
recovered = MarkItDown().convert(str(path)).markdown
monkeypatch.setitem(sys.modules, "pymupdf", None)
primary = MarkItDown().convert(str(path)).markdown
assert len(primary) > 2048
assert "AFTER_IMAGE: line 11" in recovered
table = primary[primary.index("| Item") :]
assert table in recovered
assert "Healthy prose line 44" in recovered


@pytest.mark.parametrize(
"primary,candidate,expected",
[
("", "recovered text", True),
("prefix text", "prefix text more words", True),
("prefix text", "unrelated much longer candidate text", False),
("same text", "same text", False),
],
)
def test_only_strict_text_continuations_are_selected(
monkeypatch, primary, candidate, expected
):
page = MagicMock()
page.get_image_info.return_value = [{"xref": 0}]
page.get_text.return_value = candidate
document = MagicMock()
document.__enter__.return_value = document
document.__getitem__.return_value = page
monkeypatch.setitem(
sys.modules, "pymupdf", SimpleNamespace(open=lambda **kw: document)
)
result = module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: primary})
assert result == ({0: candidate} if expected else {})
document.__exit__.assert_called_once()


def test_missing_backend_is_optional(monkeypatch):
monkeypatch.setitem(sys.modules, "pymupdf", None)
assert module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: ""}) == {}


def test_backend_failure_preserves_primary(monkeypatch):
backend = MagicMock()
backend.open.side_effect = RuntimeError("broken backend")
monkeypatch.setitem(sys.modules, "pymupdf", backend)
assert module._recover_inline_image_pages(io.BytesIO(b"pdf"), {0: "primary"}) == {}