Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ MarkItDown currently supports the conversion from:
- Excel
- Images (EXIF metadata and OCR)
- Audio (EXIF metadata and speech transcription)
- HTML
- HTML (and MHTML web archives)
- Text-based formats (CSV, JSON, XML)
- ZIP files (iterates over contents)
- YouTube URLs
Expand Down
2 changes: 2 additions & 0 deletions packages/markitdown/src/markitdown/_markitdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@
DocumentIntelligenceConverter,
ContentUnderstandingConverter,
CsvConverter,
MhtmlConverter,
)

from ._base_converter import DocumentConverter, DocumentConverterResult
Expand Down Expand Up @@ -250,6 +251,7 @@ def enable_builtins(self, **kwargs) -> None:
self.register_converter(OutlookMsgConverter())
self.register_converter(EpubConverter())
self.register_converter(CsvConverter())
self.register_converter(MhtmlConverter())

# Register Document Intelligence converter at the top of the stack if endpoint is provided
docintel_endpoint = kwargs.get("docintel_endpoint")
Expand Down
2 changes: 2 additions & 0 deletions packages/markitdown/src/markitdown/converters/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@
)
from ._epub_converter import EpubConverter
from ._csv_converter import CsvConverter
from ._mhtml_converter import MhtmlConverter

__all__ = [
"PlainTextConverter",
Expand All @@ -51,4 +52,5 @@
"ContentUnderstandingFileType",
"EpubConverter",
"CsvConverter",
"MhtmlConverter",
]
92 changes: 92 additions & 0 deletions packages/markitdown/src/markitdown/converters/_mhtml_converter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
import io
from email import policy
from email.parser import BytesParser
from email.message import EmailMessage
from typing import Any, BinaryIO, Optional

from .._base_converter import DocumentConverter, DocumentConverterResult
from .._stream_info import StreamInfo
from ._html_converter import HtmlConverter

ACCEPTED_MIME_TYPE_PREFIXES = [
"application/x-mimearchive",
"multipart/related",
]

ACCEPTED_FILE_EXTENSIONS = [".mhtml", ".mht"]

HTML_TYPES = ["text/html", "application/xhtml+xml"]


class MhtmlConverter(DocumentConverter):
"""
Converts MHTML (.mhtml / .mht) web page archives to Markdown.

The main HTML document is pulled out of the MIME container and handed to
HtmlConverter. Embedded resources such as images and stylesheets are ignored.
"""

def __init__(self):
super().__init__()
self._html_converter = HtmlConverter()

def accepts(
self,
file_stream: BinaryIO,
stream_info: StreamInfo,
**kwargs: Any, # Options to pass to the converter
) -> bool:
mimetype = (stream_info.mimetype or "").lower()
extension = (stream_info.extension or "").lower()

if extension in ACCEPTED_FILE_EXTENSIONS:
return True

for prefix in ACCEPTED_MIME_TYPE_PREFIXES:
if mimetype.startswith(prefix):
return True

return False

def convert(
self,
file_stream: BinaryIO,
stream_info: StreamInfo,
**kwargs: Any, # Options to pass to the converter
) -> DocumentConverterResult:
message = BytesParser(policy=policy.default).parse(file_stream)

html_part = self._find_html_part(message)
if html_part is None:
raise ValueError("No text/html part found in the MHTML file.")

# Undo the transfer encoding. HtmlConverter decodes the bytes, using the
# MIME charset when there is one and UTF-8 otherwise.
payload = html_part.get_payload(decode=True)

result = self._html_converter.convert(
io.BytesIO(payload),
StreamInfo(
mimetype="text/html",
extension=".html",
charset=html_part.get_content_charset(),
),
**kwargs,
)
if result.title is None and message["Subject"]:
result.title = str(message["Subject"])
return result

@staticmethod
def _find_html_part(message: EmailMessage) -> Optional[EmailMessage]:
"""Return the page itself: the part named by the `start` parameter if
there is one, otherwise the first HTML part."""
html_parts = [
part for part in message.walk() if part.get_content_type() in HTML_TYPES
]
start = message.get_param("start")
if isinstance(start, str):
for part in html_parts:
if part.get("Content-ID", "").strip() == start.strip():
return part
return html_parts[0] if html_parts else None
161 changes: 161 additions & 0 deletions packages/markitdown/tests/test_mhtml.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,161 @@
"""MHTML conversion: HTML extraction, transfer encodings, charsets, and errors."""

import base64
import io

import pytest

from markitdown import MarkItDown, StreamInfo
from markitdown.converters import MhtmlConverter

HEADERS = (
"From: <Saved by Blink>\n"
"Subject: Saved Subject\n"
"MIME-Version: 1.0\n"
'Content-Type: multipart/related; type="text/html"; boundary="B"\n'
)

PNG_PART = (
"--B\n"
"Content-Type: image/png\n"
"Content-Transfer-Encoding: base64\n"
"Content-Location: https://example.com/pic.png\n"
"\n"
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==\n"
)


def _mhtml(html_headers: str, html_body: str, extra_parts: str = "") -> bytes:
text = f"{HEADERS}\n--B\n{html_headers}\n{html_body}\n{extra_parts}--B--\n"
return text.encode("utf-8")


@pytest.fixture(scope="module")
def converter() -> MarkItDown:
return MarkItDown(enable_plugins=False)


def test_mhtml_quoted_printable_html_is_converted(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: text/html; charset=utf-8\n"
"Content-Transfer-Encoding: quoted-printable\n",
"<html><head><title>Page</title></head><body><h1>Caf=C3=A9</h1>"
'<p>See <a href=3D"https://example.com/a">this</a>.</p></body></html>',
PNG_PART,
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml")
)

assert result.markdown == "# Café\n\nSee [this](https://example.com/a)."
assert result.title == "Page"


def test_mhtml_base64_non_utf8_charset_is_decoded(converter: MarkItDown) -> None:
html = "<html><body><p>Grüße</p></body></html>".encode("iso-8859-1")
data = _mhtml(
"Content-Type: text/html; charset=iso-8859-1\n"
"Content-Transfer-Encoding: base64\n",
base64.encodebytes(html).decode("ascii"),
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mht")
)

assert result.markdown == "Grüße"


def test_mhtml_part_without_charset_is_read_as_utf8(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: text/html\nContent-Transfer-Encoding: quoted-printable\n",
"<html><body><p>Caf=C3=A9</p></body></html>",
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml")
)

assert result.markdown == "Café"


def test_mhtml_ignores_embedded_resources_and_scripts(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n",
"<html><body><script>var x = 1;</script><p>Body</p></body></html>",
PNG_PART,
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml")
)

assert result.markdown == "Body"


def test_mhtml_title_falls_back_to_subject(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n",
"<html><body><p>No title element</p></body></html>",
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml")
)

assert result.title == "Saved Subject"


def test_mhtml_detected_from_content_without_extension(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: text/html; charset=utf-8\nContent-Transfer-Encoding: 7bit\n",
"<html><body><h1>Detected</h1></body></html>",
)

result = converter.convert_stream(io.BytesIO(data))

assert result.markdown == "# Detected"


def test_mhtml_without_html_part_raises() -> None:
data = _mhtml(
"Content-Type: text/plain; charset=utf-8\nContent-Transfer-Encoding: 7bit\n",
"just text",
)

with pytest.raises(ValueError, match="No text/html part"):
MhtmlConverter().convert(io.BytesIO(data), StreamInfo(extension=".mhtml"))


def test_mhtml_xhtml_root_part_is_converted(converter: MarkItDown) -> None:
data = _mhtml(
"Content-Type: application/xhtml+xml; charset=utf-8\n"
"Content-Transfer-Encoding: 7bit\n",
'<html xmlns="http://www.w3.org/1999/xhtml"><body><h1>XHTML</h1></body></html>',
)

result = converter.convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".mhtml")
)

assert result.markdown == "# XHTML"


def test_mhtml_start_parameter_selects_root_part(converter: MarkItDown) -> None:
text = (
"MIME-Version: 1.0\n"
'Content-Type: multipart/related; type="text/html"; boundary="B";'
' start="<root@x>"\n\n'
"--B\nContent-Type: text/html; charset=utf-8\nContent-ID: <frame@x>\n\n"
"<html><body><p>IFRAME</p></body></html>\n"
"--B\nContent-Type: text/html; charset=utf-8\nContent-ID: <root@x>\n\n"
"<html><body><h1>MAIN</h1></body></html>\n"
"--B--\n"
)

result = converter.convert_stream(
io.BytesIO(text.encode()), stream_info=StreamInfo(extension=".mhtml")
)

assert result.markdown == "# MAIN"