Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -271,6 +271,18 @@ result = md.convert("test.xlsx")
print(result.markdown)
```

For PDFs containing rotated text, the built-in PDF converter accepts
`pdf_char_dir_rotated="btt"` (bottom-to-top) or `"ttb"` (top-to-bottom):

```python
result = md.convert("rotated.pdf", pdf_char_dir_rotated="btt")
```

An explicit direction uses pdfplumber output for both text and table pages.
Omitting the option or passing `None` retains the default extraction behavior.
Other values raise a conversion error. The option controls character order;
table column detection and multi-line header layout follow the existing rules.

Document Intelligence conversion in Python:

```python
Expand Down
35 changes: 27 additions & 8 deletions packages/markitdown/src/markitdown/converters/_pdf_converter.py
Original file line number Diff line number Diff line change
Expand Up @@ -117,7 +117,9 @@ def fmt_row(row: list[str]) -> str:
return "\n".join(md)


def _extract_form_content_from_words(page: Any) -> str | None:
def _extract_form_content_from_words(
page: Any, *, char_dir_rotated: str | None = None
) -> str | None:
"""
Extract form-style content from a PDF page by analyzing word positions.
This handles borderless forms/tables where words are aligned in columns.
Expand All @@ -129,7 +131,14 @@ def _extract_form_content_from_words(page: Any) -> str | None:
Returns None if the page doesn't appear to be a form-style document,
indicating that pdfminer should be used instead for better text spacing.
"""
words = page.extract_words(keep_blank_chars=True, x_tolerance=3, y_tolerance=3)
extract_words_kwargs: dict[str, Any] = {
"keep_blank_chars": True,
"x_tolerance": 3,
"y_tolerance": 3,
}
if char_dir_rotated is not None:
extract_words_kwargs["char_dir_rotated"] = char_dir_rotated
words = page.extract_words(**extract_words_kwargs)
if not words:
return None

Expand Down Expand Up @@ -536,6 +545,13 @@ def convert(

assert isinstance(file_stream, io.IOBase)

char_dir_rotated = kwargs.get("pdf_char_dir_rotated")
if char_dir_rotated not in (None, "ttb", "btt"):
raise ValueError("pdf_char_dir_rotated must be 'ttb', 'btt', or None")
extract_text_kwargs: dict[str, Any] = {}
if char_dir_rotated is not None:
extract_text_kwargs["char_dir_rotated"] = char_dir_rotated

# Read file stream into BytesIO for compatibility with pdfplumber
pdf_bytes = io.BytesIO(file_stream.read())

Expand All @@ -551,35 +567,38 @@ def convert(

with pdfplumber.open(pdf_bytes) as pdf:
for page_idx, page in enumerate(pdf.pages):
page_content = _extract_form_content_from_words(page)
page_content = _extract_form_content_from_words(
page, **extract_text_kwargs
)

if page_content is not None:
form_page_count += 1
if page_content.strip():
markdown_chunks.append(page_content)
else:
plain_page_indices.append(page_idx)
text = page.extract_text()
text = page.extract_text(**extract_text_kwargs)
if text and text.strip():
markdown_chunks.append(text.strip())

page.close() # Free cached page data immediately

# If no pages had form-style content, use pdfminer for
# the whole document (better text spacing for prose).
if form_page_count == 0:
# 显式指定旋转方向时,保留 pdfplumber 的提取结果。
if form_page_count == 0 and char_dir_rotated is None:
pdf_bytes.seek(0)
markdown = pdfminer.high_level.extract_text(pdf_bytes)
else:
markdown = "\n\n".join(markdown_chunks).strip()

except Exception:
if char_dir_rotated is not None:
raise
# Fallback if pdfplumber fails
pdf_bytes.seek(0)
markdown = pdfminer.high_level.extract_text(pdf_bytes)

# Fallback if still empty
if not markdown:
if not markdown and char_dir_rotated is None:
pdf_bytes.seek(0)
markdown = pdfminer.high_level.extract_text(pdf_bytes)

Expand Down
6 changes: 6 additions & 0 deletions packages/markitdown/tests/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -59,3 +59,9 @@ parsing, fallback, and cleanup without changing either source fixture. Rebuild
them with `python tests/test_files/generate_pdf_cleanup_fixtures.py` in an
environment with PyMuPDF installed; PyMuPDF is needed only to rebuild the files,
not to run the core tests.

The `rotated_plain_{btt,ttb}.pdf`, `rotated_mixed.pdf`, and `rotated_empty.pdf`
fixtures exercise directional text extraction on plain, mixed, and empty pages.
Rebuild them with
`python tests/test_files/generate_rotated_pdf_fixtures.py` in an environment
with ReportLab installed. ReportLab is needed only to rebuild the fixtures.
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
from pathlib import Path

from reportlab.pdfgen import canvas


FIXTURES = Path(__file__).resolve().parent


def draw_rotated_text(pdf: canvas.Canvas, direction: str) -> None:
pdf.drawString(50, 750, "Rotation example")
pdf.saveState()
pdf.translate(250, 400)
pdf.rotate(90 if direction == "btt" else -90)
pdf.drawString(0, 0, "Projected Population Kharif Rice")
pdf.restoreState()


def main() -> None:
for direction in ("btt", "ttb"):
pdf = canvas.Canvas(
str(FIXTURES / f"rotated_plain_{direction}.pdf"), invariant=True
)
draw_rotated_text(pdf, direction)
pdf.save()

pdf = canvas.Canvas(str(FIXTURES / "rotated_mixed.pdf"), invariant=True)
draw_rotated_text(pdf, "btt")
pdf.showPage()
pdf.drawString(50, 780, "Inventory")
for y, row in zip(
(750, 730, 710),
(
("Column A", "Column B", "Column C"),
("Alpha", "100", "kg"),
("Beta", "200", "lb"),
),
):
for x, value in zip((50, 250, 450), row):
pdf.drawString(x, y, value)
pdf.showPage()
draw_rotated_text(pdf, "btt")
pdf.save()

pdf = canvas.Canvas(str(FIXTURES / "rotated_empty.pdf"), invariant=True)
pdf.showPage()
pdf.save()


if __name__ == "__main__":
main()
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
78 changes: 77 additions & 1 deletion packages/markitdown/tests/test_pdf.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,10 @@

import pytest

from markitdown import MarkItDown
from markitdown import FileConversionException, MarkItDown, StreamInfo
from markitdown.converters._pdf_converter import (
PARTIAL_NUMBERING_PATTERN,
PdfConverter,
_merge_partial_numbering_lines,
)

Expand Down Expand Up @@ -1043,6 +1044,81 @@ def test_borderless_table_data_integrity(self, markitdown):
assert "Electronics" in table_text, "Second table should contain Electronics"
assert "Hardware" in table_text, "Second table should contain Hardware"

def test_rotated_text_direction_can_be_configured(self, markitdown):
"""Test that bottom-to-top rotated text can use pdfplumber's direction option."""
pdf_path = os.path.join(TEST_FILES_DIR, "rotated_table.pdf")

default_result = markitdown.convert(pdf_path)
assert "noitalupoP" in default_result.text_content

result = markitdown.convert(pdf_path, pdf_char_dir_rotated="btt")
assert "Projected Population" in result.text_content
assert "noitalupoP" not in result.text_content
table = (
"| Column A | Column B | Column C |\n"
"| -------- | -------- | -------- |\n"
"| Alpha | 100 | kg |\n"
"| Beta | 200 | lb |"
)
assert table in default_result.text_content
assert table in result.text_content

@pytest.mark.parametrize("direction", ["btt", "ttb"])
def test_rotated_plain_text_direction(self, markitdown, direction):
pdf_path = os.path.join(TEST_FILES_DIR, f"rotated_plain_{direction}.pdf")
with open(pdf_path, "rb") as stream:
result = markitdown.convert_stream(
stream,
stream_info=StreamInfo(extension=".pdf"),
pdf_char_dir_rotated=direction,
)
assert result.text_content == (
"Rotation example\nProjected\nPopulation\nKharif\nRice"
)

def test_rotated_mixed_pages_preserve_order(self, markitdown):
pdf_path = os.path.join(TEST_FILES_DIR, "rotated_mixed.pdf")
result = markitdown.convert(pdf_path, pdf_char_dir_rotated="btt")
plain_page = "Rotation example\nProjected\nPopulation\nKharif\nRice"
table_page = (
"Inventory\n"
"| Column A | Column B | Column C |\n"
"| -------- | -------- | -------- |\n"
"| Alpha | 100 | kg |\n"
"| Beta | 200 | lb |"
)
assert result.text_content == "\n\n".join((plain_page, table_page, plain_page))

@pytest.mark.parametrize("direction", ["", "foo", "ltr", "rtl", True, 1, [], {}])
def test_invalid_rotated_direction_raises(self, markitdown, direction):
pdf_path = os.path.join(TEST_FILES_DIR, "rotated_table.pdf")
with open(pdf_path, "rb") as stream:
with pytest.raises(ValueError, match="pdf_char_dir_rotated"):
PdfConverter().convert(
stream,
StreamInfo(extension=".pdf"),
pdf_char_dir_rotated=direction,
)
assert stream.tell() == 0
with pytest.raises(FileConversionException, match="pdf_char_dir_rotated"):
markitdown.convert(pdf_path, pdf_char_dir_rotated=direction)

@pytest.mark.parametrize("kind", ["plain", "form", "mixed"])
def test_unset_rotated_direction_preserves_default(self, markitdown, kind):
pdf_path = os.path.join(TEST_FILES_DIR, f"pdf_cleanup_{kind}.pdf")
default = markitdown.convert(pdf_path).text_content
assert default.strip()
assert (
markitdown.convert(pdf_path, pdf_char_dir_rotated=None).text_content
== default
)

@pytest.mark.parametrize("direction", ["btt", "ttb"])
def test_rotated_direction_with_empty_page(self, markitdown, direction):
pdf_path = os.path.join(TEST_FILES_DIR, "rotated_empty.pdf")
result = markitdown.convert(pdf_path, pdf_char_dir_rotated=direction)
assert result.text_content == ""


# MasterFormat numbering

Expand Down