diff --git a/packages/markitdown/src/markitdown/_markitdown.py b/packages/markitdown/src/markitdown/_markitdown.py index 8e2fe6a37c..59bf0d1e61 100644 --- a/packages/markitdown/src/markitdown/_markitdown.py +++ b/packages/markitdown/src/markitdown/_markitdown.py @@ -775,6 +775,15 @@ def _get_stream_info_guesses( if charset_result is not None: charset = self._normalize_charset(charset_result.encoding) + # The detector can read a short non-ASCII UTF-8 sample as + # UTF-16 and garble it. Bytes that decode as UTF-8 are UTF-8. + if not stream_page.isascii(): + try: + stream_page.decode("utf-8") + charset = "utf-8" + except UnicodeDecodeError: + pass + # Normalize the first extension listed guessed_extension = None if len(result.prediction.output.extensions) > 0: diff --git a/packages/markitdown/tests/test_module_misc.py b/packages/markitdown/tests/test_module_misc.py index 4d6c9be59b..d86493b316 100644 --- a/packages/markitdown/tests/test_module_misc.py +++ b/packages/markitdown/tests/test_module_misc.py @@ -748,6 +748,18 @@ def test_split_utf8_json_preserves_content( assert result.markdown == data.decode("utf-8") +@pytest.mark.parametrize("text", ["ok ✓", "AAAㇰ"]) +def test_short_utf8_text_is_not_guessed_as_utf16( + markitdown: MarkItDown, text: str +) -> None: + # charset_normalizer reads these short UTF-8 samples as UTF-16-BE. + stream = io.BytesIO(text.encode("utf-8")) + + result = markitdown.convert_stream(stream, stream_info=StreamInfo(extension=".txt")) + + assert result.markdown == text + + @pytest.mark.parametrize( "sample,tail", [