From b0550a3e79ebcc3ed5d79555ab7bf9b5d1272bb0 Mon Sep 17 00:00:00 2001 From: xThreeh Date: Thu, 8 Oct 2026 08:18:05 -0300 Subject: [PATCH] Keep short non-ASCII UTF-8 text from being guessed as UTF-16 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit charset_normalizer can read a short UTF-8 sample as UTF-16-BE, so a .txt file holding "ok ✓" converts to "潫⃢鲓". When the sample has non-ASCII bytes that decode as UTF-8, the stream is UTF-8; ASCII-only samples and other encodings still use the detector's answer. --- packages/markitdown/src/markitdown/_markitdown.py | 9 +++++++++ packages/markitdown/tests/test_module_misc.py | 12 ++++++++++++ 2 files changed, 21 insertions(+) diff --git a/packages/markitdown/src/markitdown/_markitdown.py b/packages/markitdown/src/markitdown/_markitdown.py index 8e2fe6a37c..59bf0d1e61 100644 --- a/packages/markitdown/src/markitdown/_markitdown.py +++ b/packages/markitdown/src/markitdown/_markitdown.py @@ -775,6 +775,15 @@ def _get_stream_info_guesses( if charset_result is not None: charset = self._normalize_charset(charset_result.encoding) + # The detector can read a short non-ASCII UTF-8 sample as + # UTF-16 and garble it. Bytes that decode as UTF-8 are UTF-8. + if not stream_page.isascii(): + try: + stream_page.decode("utf-8") + charset = "utf-8" + except UnicodeDecodeError: + pass + # Normalize the first extension listed guessed_extension = None if len(result.prediction.output.extensions) > 0: diff --git a/packages/markitdown/tests/test_module_misc.py b/packages/markitdown/tests/test_module_misc.py index 4d6c9be59b..d86493b316 100644 --- a/packages/markitdown/tests/test_module_misc.py +++ b/packages/markitdown/tests/test_module_misc.py @@ -748,6 +748,18 @@ def test_split_utf8_json_preserves_content( assert result.markdown == data.decode("utf-8") +@pytest.mark.parametrize("text", ["ok ✓", "AAAㇰ"]) +def test_short_utf8_text_is_not_guessed_as_utf16( + markitdown: MarkItDown, text: str +) -> None: + # charset_normalizer reads these short UTF-8 samples as UTF-16-BE. + stream = io.BytesIO(text.encode("utf-8")) + + result = markitdown.convert_stream(stream, stream_info=StreamInfo(extension=".txt")) + + assert result.markdown == text + + @pytest.mark.parametrize( "sample,tail", [