diff --git a/README.md b/README.md index fa33f6b343..9cd1f61343 100644 --- a/README.md +++ b/README.md @@ -271,6 +271,18 @@ result = md.convert("test.xlsx") print(result.markdown) ``` +For PDFs containing rotated text, the built-in PDF converter accepts +`pdf_char_dir_rotated="btt"` (bottom-to-top) or `"ttb"` (top-to-bottom): + +```python +result = md.convert("rotated.pdf", pdf_char_dir_rotated="btt") +``` + +An explicit direction uses pdfplumber output for both text and table pages. +Omitting the option or passing `None` retains the default extraction behavior. +Other values raise a conversion error. The option controls character order; +table column detection and multi-line header layout follow the existing rules. + Document Intelligence conversion in Python: ```python diff --git a/packages/markitdown/src/markitdown/converters/_pdf_converter.py b/packages/markitdown/src/markitdown/converters/_pdf_converter.py index ffbcbd990c..3ca0738a0a 100644 --- a/packages/markitdown/src/markitdown/converters/_pdf_converter.py +++ b/packages/markitdown/src/markitdown/converters/_pdf_converter.py @@ -117,7 +117,9 @@ def fmt_row(row: list[str]) -> str: return "\n".join(md) -def _extract_form_content_from_words(page: Any) -> str | None: +def _extract_form_content_from_words( + page: Any, *, char_dir_rotated: str | None = None +) -> str | None: """ Extract form-style content from a PDF page by analyzing word positions. This handles borderless forms/tables where words are aligned in columns. @@ -129,7 +131,14 @@ def _extract_form_content_from_words(page: Any) -> str | None: Returns None if the page doesn't appear to be a form-style document, indicating that pdfminer should be used instead for better text spacing. """ - words = page.extract_words(keep_blank_chars=True, x_tolerance=3, y_tolerance=3) + extract_words_kwargs: dict[str, Any] = { + "keep_blank_chars": True, + "x_tolerance": 3, + "y_tolerance": 3, + } + if char_dir_rotated is not None: + extract_words_kwargs["char_dir_rotated"] = char_dir_rotated + words = page.extract_words(**extract_words_kwargs) if not words: return None @@ -536,6 +545,13 @@ def convert( assert isinstance(file_stream, io.IOBase) + char_dir_rotated = kwargs.get("pdf_char_dir_rotated") + if char_dir_rotated not in (None, "ttb", "btt"): + raise ValueError("pdf_char_dir_rotated must be 'ttb', 'btt', or None") + extract_text_kwargs: dict[str, Any] = {} + if char_dir_rotated is not None: + extract_text_kwargs["char_dir_rotated"] = char_dir_rotated + # Read file stream into BytesIO for compatibility with pdfplumber pdf_bytes = io.BytesIO(file_stream.read()) @@ -551,7 +567,9 @@ def convert( with pdfplumber.open(pdf_bytes) as pdf: for page_idx, page in enumerate(pdf.pages): - page_content = _extract_form_content_from_words(page) + page_content = _extract_form_content_from_words( + page, **extract_text_kwargs + ) if page_content is not None: form_page_count += 1 @@ -559,27 +577,28 @@ def convert( markdown_chunks.append(page_content) else: plain_page_indices.append(page_idx) - text = page.extract_text() + text = page.extract_text(**extract_text_kwargs) if text and text.strip(): markdown_chunks.append(text.strip()) page.close() # Free cached page data immediately - # If no pages had form-style content, use pdfminer for - # the whole document (better text spacing for prose). - if form_page_count == 0: + # 显式指定旋转方向时,保留 pdfplumber 的提取结果。 + if form_page_count == 0 and char_dir_rotated is None: pdf_bytes.seek(0) markdown = pdfminer.high_level.extract_text(pdf_bytes) else: markdown = "\n\n".join(markdown_chunks).strip() except Exception: + if char_dir_rotated is not None: + raise # Fallback if pdfplumber fails pdf_bytes.seek(0) markdown = pdfminer.high_level.extract_text(pdf_bytes) # Fallback if still empty - if not markdown: + if not markdown and char_dir_rotated is None: pdf_bytes.seek(0) markdown = pdfminer.high_level.extract_text(pdf_bytes) diff --git a/packages/markitdown/tests/README.md b/packages/markitdown/tests/README.md index 86fd8f4661..8c0b1b5cdb 100644 --- a/packages/markitdown/tests/README.md +++ b/packages/markitdown/tests/README.md @@ -59,3 +59,9 @@ parsing, fallback, and cleanup without changing either source fixture. Rebuild them with `python tests/test_files/generate_pdf_cleanup_fixtures.py` in an environment with PyMuPDF installed; PyMuPDF is needed only to rebuild the files, not to run the core tests. + +The `rotated_plain_{btt,ttb}.pdf`, `rotated_mixed.pdf`, and `rotated_empty.pdf` +fixtures exercise directional text extraction on plain, mixed, and empty pages. +Rebuild them with +`python tests/test_files/generate_rotated_pdf_fixtures.py` in an environment +with ReportLab installed. ReportLab is needed only to rebuild the fixtures. diff --git a/packages/markitdown/tests/test_files/generate_rotated_pdf_fixtures.py b/packages/markitdown/tests/test_files/generate_rotated_pdf_fixtures.py new file mode 100644 index 0000000000..daa9eb0182 --- /dev/null +++ b/packages/markitdown/tests/test_files/generate_rotated_pdf_fixtures.py @@ -0,0 +1,50 @@ +from pathlib import Path + +from reportlab.pdfgen import canvas + + +FIXTURES = Path(__file__).resolve().parent + + +def draw_rotated_text(pdf: canvas.Canvas, direction: str) -> None: + pdf.drawString(50, 750, "Rotation example") + pdf.saveState() + pdf.translate(250, 400) + pdf.rotate(90 if direction == "btt" else -90) + pdf.drawString(0, 0, "Projected Population Kharif Rice") + pdf.restoreState() + + +def main() -> None: + for direction in ("btt", "ttb"): + pdf = canvas.Canvas( + str(FIXTURES / f"rotated_plain_{direction}.pdf"), invariant=True + ) + draw_rotated_text(pdf, direction) + pdf.save() + + pdf = canvas.Canvas(str(FIXTURES / "rotated_mixed.pdf"), invariant=True) + draw_rotated_text(pdf, "btt") + pdf.showPage() + pdf.drawString(50, 780, "Inventory") + for y, row in zip( + (750, 730, 710), + ( + ("Column A", "Column B", "Column C"), + ("Alpha", "100", "kg"), + ("Beta", "200", "lb"), + ), + ): + for x, value in zip((50, 250, 450), row): + pdf.drawString(x, y, value) + pdf.showPage() + draw_rotated_text(pdf, "btt") + pdf.save() + + pdf = canvas.Canvas(str(FIXTURES / "rotated_empty.pdf"), invariant=True) + pdf.showPage() + pdf.save() + + +if __name__ == "__main__": + main() diff --git a/packages/markitdown/tests/test_files/rotated_empty.pdf b/packages/markitdown/tests/test_files/rotated_empty.pdf new file mode 100644 index 0000000000..15bcd39e4c Binary files /dev/null and b/packages/markitdown/tests/test_files/rotated_empty.pdf differ diff --git a/packages/markitdown/tests/test_files/rotated_mixed.pdf b/packages/markitdown/tests/test_files/rotated_mixed.pdf new file mode 100644 index 0000000000..9a37bd8bde Binary files /dev/null and b/packages/markitdown/tests/test_files/rotated_mixed.pdf differ diff --git a/packages/markitdown/tests/test_files/rotated_plain_btt.pdf b/packages/markitdown/tests/test_files/rotated_plain_btt.pdf new file mode 100644 index 0000000000..84a1833347 Binary files /dev/null and b/packages/markitdown/tests/test_files/rotated_plain_btt.pdf differ diff --git a/packages/markitdown/tests/test_files/rotated_plain_ttb.pdf b/packages/markitdown/tests/test_files/rotated_plain_ttb.pdf new file mode 100644 index 0000000000..600ef139c3 Binary files /dev/null and b/packages/markitdown/tests/test_files/rotated_plain_ttb.pdf differ diff --git a/packages/markitdown/tests/test_files/rotated_table.pdf b/packages/markitdown/tests/test_files/rotated_table.pdf new file mode 100644 index 0000000000..5f3ed67d3e Binary files /dev/null and b/packages/markitdown/tests/test_files/rotated_table.pdf differ diff --git a/packages/markitdown/tests/test_pdf.py b/packages/markitdown/tests/test_pdf.py index c4f8c64a9c..64f03210ea 100644 --- a/packages/markitdown/tests/test_pdf.py +++ b/packages/markitdown/tests/test_pdf.py @@ -6,9 +6,10 @@ import pytest -from markitdown import MarkItDown +from markitdown import FileConversionException, MarkItDown, StreamInfo from markitdown.converters._pdf_converter import ( PARTIAL_NUMBERING_PATTERN, + PdfConverter, _merge_partial_numbering_lines, ) @@ -1043,6 +1044,81 @@ def test_borderless_table_data_integrity(self, markitdown): assert "Electronics" in table_text, "Second table should contain Electronics" assert "Hardware" in table_text, "Second table should contain Hardware" + def test_rotated_text_direction_can_be_configured(self, markitdown): + """Test that bottom-to-top rotated text can use pdfplumber's direction option.""" + pdf_path = os.path.join(TEST_FILES_DIR, "rotated_table.pdf") + + default_result = markitdown.convert(pdf_path) + assert "noitalupoP" in default_result.text_content + + result = markitdown.convert(pdf_path, pdf_char_dir_rotated="btt") + assert "Projected Population" in result.text_content + assert "noitalupoP" not in result.text_content + table = ( + "| Column A | Column B | Column C |\n" + "| -------- | -------- | -------- |\n" + "| Alpha | 100 | kg |\n" + "| Beta | 200 | lb |" + ) + assert table in default_result.text_content + assert table in result.text_content + + @pytest.mark.parametrize("direction", ["btt", "ttb"]) + def test_rotated_plain_text_direction(self, markitdown, direction): + pdf_path = os.path.join(TEST_FILES_DIR, f"rotated_plain_{direction}.pdf") + with open(pdf_path, "rb") as stream: + result = markitdown.convert_stream( + stream, + stream_info=StreamInfo(extension=".pdf"), + pdf_char_dir_rotated=direction, + ) + assert result.text_content == ( + "Rotation example\nProjected\nPopulation\nKharif\nRice" + ) + + def test_rotated_mixed_pages_preserve_order(self, markitdown): + pdf_path = os.path.join(TEST_FILES_DIR, "rotated_mixed.pdf") + result = markitdown.convert(pdf_path, pdf_char_dir_rotated="btt") + plain_page = "Rotation example\nProjected\nPopulation\nKharif\nRice" + table_page = ( + "Inventory\n" + "| Column A | Column B | Column C |\n" + "| -------- | -------- | -------- |\n" + "| Alpha | 100 | kg |\n" + "| Beta | 200 | lb |" + ) + assert result.text_content == "\n\n".join((plain_page, table_page, plain_page)) + + @pytest.mark.parametrize("direction", ["", "foo", "ltr", "rtl", True, 1, [], {}]) + def test_invalid_rotated_direction_raises(self, markitdown, direction): + pdf_path = os.path.join(TEST_FILES_DIR, "rotated_table.pdf") + with open(pdf_path, "rb") as stream: + with pytest.raises(ValueError, match="pdf_char_dir_rotated"): + PdfConverter().convert( + stream, + StreamInfo(extension=".pdf"), + pdf_char_dir_rotated=direction, + ) + assert stream.tell() == 0 + with pytest.raises(FileConversionException, match="pdf_char_dir_rotated"): + markitdown.convert(pdf_path, pdf_char_dir_rotated=direction) + + @pytest.mark.parametrize("kind", ["plain", "form", "mixed"]) + def test_unset_rotated_direction_preserves_default(self, markitdown, kind): + pdf_path = os.path.join(TEST_FILES_DIR, f"pdf_cleanup_{kind}.pdf") + default = markitdown.convert(pdf_path).text_content + assert default.strip() + assert ( + markitdown.convert(pdf_path, pdf_char_dir_rotated=None).text_content + == default + ) + + @pytest.mark.parametrize("direction", ["btt", "ttb"]) + def test_rotated_direction_with_empty_page(self, markitdown, direction): + pdf_path = os.path.join(TEST_FILES_DIR, "rotated_empty.pdf") + result = markitdown.convert(pdf_path, pdf_char_dir_rotated=direction) + assert result.text_content == "" + # MasterFormat numbering