diff --git a/README.md b/README.md index fa33f6b343..8c1007e1d1 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ MarkItDown currently supports the conversion from: - Excel - Images (EXIF metadata and OCR) - Audio (EXIF metadata and speech transcription) -- HTML +- HTML (and MHTML web archives) - Text-based formats (CSV, JSON, XML) - ZIP files (iterates over contents) - YouTube URLs diff --git a/packages/markitdown/src/markitdown/_markitdown.py b/packages/markitdown/src/markitdown/_markitdown.py index 8e2fe6a37c..54a28e5f0a 100644 --- a/packages/markitdown/src/markitdown/_markitdown.py +++ b/packages/markitdown/src/markitdown/_markitdown.py @@ -42,6 +42,7 @@ DocumentIntelligenceConverter, ContentUnderstandingConverter, CsvConverter, + MhtmlConverter, ) from ._base_converter import DocumentConverter, DocumentConverterResult @@ -250,6 +251,7 @@ def enable_builtins(self, **kwargs) -> None: self.register_converter(OutlookMsgConverter()) self.register_converter(EpubConverter()) self.register_converter(CsvConverter()) + self.register_converter(MhtmlConverter()) # Register Document Intelligence converter at the top of the stack if endpoint is provided docintel_endpoint = kwargs.get("docintel_endpoint") diff --git a/packages/markitdown/src/markitdown/converters/__init__.py b/packages/markitdown/src/markitdown/converters/__init__.py index 77f8b1acd1..05de95ddc9 100644 --- a/packages/markitdown/src/markitdown/converters/__init__.py +++ b/packages/markitdown/src/markitdown/converters/__init__.py @@ -27,6 +27,7 @@ ) from ._epub_converter import EpubConverter from ._csv_converter import CsvConverter +from ._mhtml_converter import MhtmlConverter __all__ = [ "PlainTextConverter", @@ -51,4 +52,5 @@ "ContentUnderstandingFileType", "EpubConverter", "CsvConverter", + "MhtmlConverter", ] diff --git a/packages/markitdown/src/markitdown/converters/_mhtml_converter.py b/packages/markitdown/src/markitdown/converters/_mhtml_converter.py new file mode 100644 index 0000000000..1d3f196dc7 --- /dev/null +++ b/packages/markitdown/src/markitdown/converters/_mhtml_converter.py @@ -0,0 +1,92 @@ +import io +from email import policy +from email.parser import BytesParser +from email.message import EmailMessage +from typing import Any, BinaryIO, Optional + +from .._base_converter import DocumentConverter, DocumentConverterResult +from .._stream_info import StreamInfo +from ._html_converter import HtmlConverter + +ACCEPTED_MIME_TYPE_PREFIXES = [ + "application/x-mimearchive", + "multipart/related", +] + +ACCEPTED_FILE_EXTENSIONS = [".mhtml", ".mht"] + +HTML_TYPES = ["text/html", "application/xhtml+xml"] + + +class MhtmlConverter(DocumentConverter): + """ + Converts MHTML (.mhtml / .mht) web page archives to Markdown. + + The main HTML document is pulled out of the MIME container and handed to + HtmlConverter. Embedded resources such as images and stylesheets are ignored. + """ + + def __init__(self): + super().__init__() + self._html_converter = HtmlConverter() + + def accepts( + self, + file_stream: BinaryIO, + stream_info: StreamInfo, + **kwargs: Any, # Options to pass to the converter + ) -> bool: + mimetype = (stream_info.mimetype or "").lower() + extension = (stream_info.extension or "").lower() + + if extension in ACCEPTED_FILE_EXTENSIONS: + return True + + for prefix in ACCEPTED_MIME_TYPE_PREFIXES: + if mimetype.startswith(prefix): + return True + + return False + + def convert( + self, + file_stream: BinaryIO, + stream_info: StreamInfo, + **kwargs: Any, # Options to pass to the converter + ) -> DocumentConverterResult: + message = BytesParser(policy=policy.default).parse(file_stream) + + html_part = self._find_html_part(message) + if html_part is None: + raise ValueError("No text/html part found in the MHTML file.") + + # Undo the transfer encoding. HtmlConverter decodes the bytes, using the + # MIME charset when there is one and UTF-8 otherwise. + payload = html_part.get_payload(decode=True) + + result = self._html_converter.convert( + io.BytesIO(payload), + StreamInfo( + mimetype="text/html", + extension=".html", + charset=html_part.get_content_charset(), + ), + **kwargs, + ) + if result.title is None and message["Subject"]: + result.title = str(message["Subject"]) + return result + + @staticmethod + def _find_html_part(message: EmailMessage) -> Optional[EmailMessage]: + """Return the page itself: the part named by the `start` parameter if + there is one, otherwise the first HTML part.""" + html_parts = [ + part for part in message.walk() if part.get_content_type() in HTML_TYPES + ] + start = message.get_param("start") + if isinstance(start, str): + for part in html_parts: + if part.get("Content-ID", "").strip() == start.strip(): + return part + return html_parts[0] if html_parts else None diff --git a/packages/markitdown/tests/test_mhtml.py b/packages/markitdown/tests/test_mhtml.py new file mode 100644 index 0000000000..44948db604 --- /dev/null +++ b/packages/markitdown/tests/test_mhtml.py @@ -0,0 +1,161 @@ +"""MHTML conversion: HTML extraction, transfer encodings, charsets, and errors.""" + +import base64 +import io + +import pytest + +from markitdown import MarkItDown, StreamInfo +from markitdown.converters import MhtmlConverter + +HEADERS = ( + "From: \n" + "Subject: Saved Subject\n" + "MIME-Version: 1.0\n" + 'Content-Type: multipart/related; type="text/html"; boundary="B"\n' +) + +PNG_PART = ( + "--B\n" + "Content-Type: image/png\n" + "Content-Transfer-Encoding: base64\n" + "Content-Location: https://example.com/pic.png\n" + "\n" + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==\n" +) + + +def _mhtml(html_headers: str, html_body: str, extra_parts: str = "") -> bytes: + text = f"{HEADERS}\n--B\n{html_headers}\n{html_body}\n{extra_parts}--B--\n" + return text.encode("utf-8") + + +@pytest.fixture(scope="module") +def converter() -> MarkItDown: + return MarkItDown(enable_plugins=False) + + +def test_mhtml_quoted_printable_html_is_converted(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: text/html; charset=utf-8\n" + "Content-Transfer-Encoding: quoted-printable\n", + "Page

Caf=C3=A9

" + '

See this.

', + PNG_PART, + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.markdown == "# Café\n\nSee [this](https://example.com/a)." + assert result.title == "Page" + + +def test_mhtml_base64_non_utf8_charset_is_decoded(converter: MarkItDown) -> None: + html = "

Grüße

".encode("iso-8859-1") + data = _mhtml( + "Content-Type: text/html; charset=iso-8859-1\n" + "Content-Transfer-Encoding: base64\n", + base64.encodebytes(html).decode("ascii"), + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mht") + ) + + assert result.markdown == "Grüße" + + +def test_mhtml_part_without_charset_is_read_as_utf8(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: text/html\nContent-Transfer-Encoding: quoted-printable\n", + "

Caf=C3=A9

", + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.markdown == "Café" + + +def test_mhtml_ignores_embedded_resources_and_scripts(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n", + "

Body

", + PNG_PART, + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.markdown == "Body" + + +def test_mhtml_title_falls_back_to_subject(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n", + "

No title element

", + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.title == "Saved Subject" + + +def test_mhtml_detected_from_content_without_extension(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n", + "

Detected

", + ) + + result = converter.convert_stream(io.BytesIO(data)) + + assert result.markdown == "# Detected" + + +def test_mhtml_without_html_part_raises() -> None: + data = _mhtml( + "Content-Type: text/plain; charset=utf-8\nContent-Transfer-Encoding: 7bit\n", + "just text", + ) + + with pytest.raises(ValueError, match="No text/html part"): + MhtmlConverter().convert(io.BytesIO(data), StreamInfo(extension=".mhtml")) + + +def test_mhtml_xhtml_root_part_is_converted(converter: MarkItDown) -> None: + data = _mhtml( + "Content-Type: application/xhtml+xml; charset=utf-8\n" + "Content-Transfer-Encoding: 7bit\n", + '

XHTML

', + ) + + result = converter.convert_stream( + io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.markdown == "# XHTML" + + +def test_mhtml_start_parameter_selects_root_part(converter: MarkItDown) -> None: + text = ( + "MIME-Version: 1.0\n" + 'Content-Type: multipart/related; type="text/html"; boundary="B";' + ' start=""\n\n' + "--B\nContent-Type: text/html; charset=utf-8\nContent-ID: \n\n" + "

IFRAME

\n" + "--B\nContent-Type: text/html; charset=utf-8\nContent-ID: \n\n" + "

MAIN

\n" + "--B--\n" + ) + + result = converter.convert_stream( + io.BytesIO(text.encode()), stream_info=StreamInfo(extension=".mhtml") + ) + + assert result.markdown == "# MAIN"