From 5f88506fc218c973a961c05d177f03fd214fd105 Mon Sep 17 00:00:00 2001 From: RRiiiccckkk Date: Mon, 28 Sep 2026 13:58:13 +0800 Subject: [PATCH] fix(pdf): expose rotated text direction Signed-off-by: RRiiiccckkk --- README.md | 9 +++++ .../markitdown/converters/_pdf_converter.py | 37 +++++++++++++++--- .../tests/test_files/rotated_table.pdf | Bin 0 -> 1467 bytes packages/markitdown/tests/test_pdf_tables.py | 12 ++++++ 4 files changed, 52 insertions(+), 6 deletions(-) create mode 100644 packages/markitdown/tests/test_files/rotated_table.pdf diff --git a/README.md b/README.md index fa33f6b343..633fd0ba64 100644 --- a/README.md +++ b/README.md @@ -110,6 +110,15 @@ At the moment, the following optional dependencies are available: * `[audio-transcription]` Installs dependencies for audio transcription of wav and mp3 files * `[youtube-transcription]` Installs dependencies for fetching YouTube video transcription +For PDFs containing bottom-to-top rotated text, pass the direction expected by +pdfplumber when converting the document: + +```python +from markitdown import MarkItDown + +result = MarkItDown().convert("rotated.pdf", pdf_char_dir_rotated="btt") +``` + ### Plugins MarkItDown also supports 3rd-party plugins. Plugins are disabled by default. To list installed plugins: diff --git a/packages/markitdown/src/markitdown/converters/_pdf_converter.py b/packages/markitdown/src/markitdown/converters/_pdf_converter.py index ffbcbd990c..cd538b92f5 100644 --- a/packages/markitdown/src/markitdown/converters/_pdf_converter.py +++ b/packages/markitdown/src/markitdown/converters/_pdf_converter.py @@ -117,7 +117,9 @@ def fmt_row(row: list[str]) -> str: return "\n".join(md) -def _extract_form_content_from_words(page: Any) -> str | None: +def _extract_form_content_from_words( + page: Any, *, char_dir_rotated: str | None = None +) -> str | None: """ Extract form-style content from a PDF page by analyzing word positions. This handles borderless forms/tables where words are aligned in columns. @@ -129,7 +131,14 @@ def _extract_form_content_from_words(page: Any) -> str | None: Returns None if the page doesn't appear to be a form-style document, indicating that pdfminer should be used instead for better text spacing. """ - words = page.extract_words(keep_blank_chars=True, x_tolerance=3, y_tolerance=3) + extract_words_kwargs: dict[str, Any] = { + "keep_blank_chars": True, + "x_tolerance": 3, + "y_tolerance": 3, + } + if char_dir_rotated is not None: + extract_words_kwargs["char_dir_rotated"] = char_dir_rotated + words = page.extract_words(**extract_words_kwargs) if not words: return None @@ -395,7 +404,9 @@ def extract_cells(info: dict) -> list[str]: return "\n".join(result_lines) -def _extract_tables_from_words(page: Any) -> list[list[list[str]]]: +def _extract_tables_from_words( + page: Any, *, char_dir_rotated: str | None = None +) -> list[list[list[str]]]: """ Extract tables from a PDF page by analyzing word positions. This handles borderless tables where words are aligned in columns. @@ -403,7 +414,14 @@ def _extract_tables_from_words(page: Any) -> list[list[list[str]]]: This function is designed for structured tabular data (like invoices), not for multi-column text layouts in scientific documents. """ - words = page.extract_words(keep_blank_chars=True, x_tolerance=3, y_tolerance=3) + extract_words_kwargs: dict[str, Any] = { + "keep_blank_chars": True, + "x_tolerance": 3, + "y_tolerance": 3, + } + if char_dir_rotated is not None: + extract_words_kwargs["char_dir_rotated"] = char_dir_rotated + words = page.extract_words(**extract_words_kwargs) if not words: return [] @@ -536,6 +554,11 @@ def convert( assert isinstance(file_stream, io.IOBase) + char_dir_rotated = kwargs.get("pdf_char_dir_rotated") + extract_text_kwargs: dict[str, Any] = {} + if char_dir_rotated is not None: + extract_text_kwargs["char_dir_rotated"] = char_dir_rotated + # Read file stream into BytesIO for compatibility with pdfplumber pdf_bytes = io.BytesIO(file_stream.read()) @@ -551,7 +574,9 @@ def convert( with pdfplumber.open(pdf_bytes) as pdf: for page_idx, page in enumerate(pdf.pages): - page_content = _extract_form_content_from_words(page) + page_content = _extract_form_content_from_words( + page, char_dir_rotated=char_dir_rotated + ) if page_content is not None: form_page_count += 1 @@ -559,7 +584,7 @@ def convert( markdown_chunks.append(page_content) else: plain_page_indices.append(page_idx) - text = page.extract_text() + text = page.extract_text(**extract_text_kwargs) if text and text.strip(): markdown_chunks.append(text.strip()) diff --git a/packages/markitdown/tests/test_files/rotated_table.pdf b/packages/markitdown/tests/test_files/rotated_table.pdf new file mode 100644 index 0000000000000000000000000000000000000000..5f3ed67d3e7a3fa361759340bdee13e879a71bde GIT binary patch literal 1467 zcmah}TX)(x5PtWsm?ldDOTibg!71SouHkSA1=4IvpcS?NQINnEZPJJR1G{hgzQ1)x z1R9@Y2Q&1mM+%%_=&^{Vxhna*gjfBydKU+B|~2xOZ~p+Pwfh@?KC1^B|-ttpof z7aPh$u?sw!&`dBuOqW`rpy_oJEEM}%v8Yk*E47l3$7K@IxS&x$TtQ&6D#U^kVI8x)54M_Da+f>@EOGLfq)c!Z&bkW=x1-l0AtW$^$b zlnt<5J8k7`2l81nozJ^@7=zv;fy&N#^61m23`5vwrovXKB33UD-hfJk(iJWGkUn5a z%{7_R@?otU)+UTGK+yX_B8}*-4+FyYkV>z?F_j)VM^ee6>5(;c)Db@BdlS;G;J8t$ z@EGV57ZoDOO3Wh|b_C;cCl?}sIN^dnuEj1)fL;kGkxcNaiY#6|w+zcMTq|#8Y$IzW zjr^HmoE`06I=ATKcM zsfPu(u%B7YDrlpE zHM6NZGrX)z49~H%l-Ptg#LZ+K;uG5{{@P-!#Ui6~{{N#hl>(n(^S&6On%?9y0jg#D h8cqu#L5%?157-ZijP9$61aeQYX}FFSi`A~{+JE9qpQ!)< literal 0 HcmV?d00001 diff --git a/packages/markitdown/tests/test_pdf_tables.py b/packages/markitdown/tests/test_pdf_tables.py index d26de0cb98..7972ceb231 100644 --- a/packages/markitdown/tests/test_pdf_tables.py +++ b/packages/markitdown/tests/test_pdf_tables.py @@ -1196,3 +1196,15 @@ def test_borderless_table_data_integrity(self, markitdown): table_text = str(second_table) assert "Electronics" in table_text, "Second table should contain Electronics" assert "Hardware" in table_text, "Second table should contain Hardware" + + def test_rotated_text_direction_can_be_configured(self, markitdown): + """Test that bottom-to-top rotated text can use pdfplumber's direction option.""" + pdf_path = os.path.join(TEST_FILES_DIR, "rotated_table.pdf") + + default_result = markitdown.convert(pdf_path) + assert "noitalupoP" in default_result.text_content + + result = markitdown.convert(pdf_path, pdf_char_dir_rotated="btt") + assert "Projected" in result.text_content + assert "Population" in result.text_content + assert "noitalupoP" not in result.text_content