diff --git a/packages/markitdown/src/markitdown/_markitdown.py b/packages/markitdown/src/markitdown/_markitdown.py index 8e2fe6a37c..7c40f80343 100644 --- a/packages/markitdown/src/markitdown/_markitdown.py +++ b/packages/markitdown/src/markitdown/_markitdown.py @@ -42,6 +42,7 @@ DocumentIntelligenceConverter, ContentUnderstandingConverter, CsvConverter, + SrtConverter, ) from ._base_converter import DocumentConverter, DocumentConverterResult @@ -250,6 +251,7 @@ def enable_builtins(self, **kwargs) -> None: self.register_converter(OutlookMsgConverter()) self.register_converter(EpubConverter()) self.register_converter(CsvConverter()) + self.register_converter(SrtConverter()) # Register Document Intelligence converter at the top of the stack if endpoint is provided docintel_endpoint = kwargs.get("docintel_endpoint") diff --git a/packages/markitdown/src/markitdown/converters/__init__.py b/packages/markitdown/src/markitdown/converters/__init__.py index 77f8b1acd1..7b0d4c40a4 100644 --- a/packages/markitdown/src/markitdown/converters/__init__.py +++ b/packages/markitdown/src/markitdown/converters/__init__.py @@ -27,6 +27,7 @@ ) from ._epub_converter import EpubConverter from ._csv_converter import CsvConverter +from ._srt_converter import SrtConverter __all__ = [ "PlainTextConverter", @@ -51,4 +52,5 @@ "ContentUnderstandingFileType", "EpubConverter", "CsvConverter", + "SrtConverter", ] diff --git a/packages/markitdown/src/markitdown/converters/_srt_converter.py b/packages/markitdown/src/markitdown/converters/_srt_converter.py new file mode 100644 index 0000000000..4bac19f2e5 --- /dev/null +++ b/packages/markitdown/src/markitdown/converters/_srt_converter.py @@ -0,0 +1,101 @@ +import re +from typing import Any, BinaryIO + +from charset_normalizer import from_bytes + +from .._base_converter import DocumentConverter, DocumentConverterResult +from .._stream_info import StreamInfo + +ACCEPTED_MIME_TYPE_PREFIXES = [ + "text/srt", + "text/x-srt", + "application/x-subrip", +] +ACCEPTED_FILE_EXTENSIONS = [".srt"] + +# "00:01:05,250 --> 00:01:07,000", capturing the start time without milliseconds. +_TIMING_RE = re.compile( + r"^(\d{1,3}):(\d{2}):(\d{2})(?:[,.]\d{1,3})?\s*-->\s*\d{1,3}:\d{2}:\d{2}(?:[,.]\d{1,3})?" +) +# Styling markup: , , , and {\an8}-style overrides. Other +# angle brackets, like "" or "x]{0,200})?>|\{\\[^}]{0,200}\}", re.I) + + +class SrtConverter(DocumentConverter): + """ + Converts SubRip (.srt) subtitle files to a Markdown transcript, one line per + cue, each prefixed with its start time. + """ + + def accepts( + self, + file_stream: BinaryIO, + stream_info: StreamInfo, + **kwargs: Any, # Options to pass to the converter + ) -> bool: + mimetype = (stream_info.mimetype or "").lower() + extension = (stream_info.extension or "").lower() + + if extension in ACCEPTED_FILE_EXTENSIONS: + return True + + for prefix in ACCEPTED_MIME_TYPE_PREFIXES: + if mimetype.startswith(prefix): + return True + + return False + + def convert( + self, + file_stream: BinaryIO, + stream_info: StreamInfo, + **kwargs: Any, # Options to pass to the converter + ) -> DocumentConverterResult: + data = file_stream.read() + content = None + if stream_info.charset: + # The charset hint comes from a sample of the file, so it can be + # wrong for later bytes (e.g. "ascii" for a file that turns non-ASCII + # after the sample). Fall through to a full-file guess in that case. + try: + content = data.decode(stream_info.charset) + except (UnicodeDecodeError, LookupError): + pass + if content is None: + detected = from_bytes(data).best() + content = ( + str(detected) + if detected is not None + else data.decode("utf-8", errors="ignore") + ) + + content = content.lstrip("\ufeff").replace("\r\n", "\n").replace("\r", "\n") + + # A cue starts at each timing line and runs to the next one, so a cue + # with a blank line inside it, or a missing separator, still parses. + lines = content.split("\n") + timings = [ + (i, m) + for i, line in enumerate(lines) + if (m := _TIMING_RE.match(line.strip())) + ] + if not timings and content.strip(): + raise ValueError("No SRT cues found.") + + cues = [] + for n, (i, timing) in enumerate(timings): + end = timings[n + 1][0] if n + 1 < len(timings) else len(lines) + body = [line.strip() for line in lines[i + 1 : end]] + while body and not body[-1]: + body.pop() + # The last line before the next timing line is its cue number. + if n + 1 < len(timings) and body and body[-1].isdigit(): + body.pop() + + text = " ".join(_MARKUP_RE.sub("", " ".join(body)).split()) + if text: + hours, minutes, seconds = timing.groups() + cues.append(f"[{hours.zfill(2)}:{minutes}:{seconds}] {text}") + + return DocumentConverterResult(markdown="\n\n".join(cues)) diff --git a/packages/markitdown/tests/test_srt.py b/packages/markitdown/tests/test_srt.py new file mode 100644 index 0000000000..281b4d0ebb --- /dev/null +++ b/packages/markitdown/tests/test_srt.py @@ -0,0 +1,116 @@ +"""SRT subtitle conversion: cue text, timestamps, markup, and line endings.""" + +import io + +import pytest + +from markitdown import MarkItDown, StreamInfo + + +@pytest.fixture(scope="module") +def converter() -> MarkItDown: + return MarkItDown(enable_plugins=False) + + +def _convert(converter: MarkItDown, content: str, **info) -> str: + return converter.convert_stream( + io.BytesIO(content.encode("utf-8")), + stream_info=StreamInfo(extension=".srt", **info), + ).markdown + + +@pytest.mark.parametrize("newline", ["\n", "\r\n", "\r"], ids=["LF", "CRLF", "CR"]) +def test_srt_cues_become_timestamped_lines(converter: MarkItDown, newline: str) -> None: + content = newline.join( + [ + "1", + "00:00:01,000 --> 00:00:03,500", + "Hello there.", + "", + "2", + "01:02:05,250 --> 01:02:07,000", + "Goodbye.", + "", + ] + ) + + assert _convert(converter, content) == ( + "[00:00:01] Hello there.\n\n[01:02:05] Goodbye." + ) + + +def test_srt_multiline_cue_is_joined_and_markup_removed( + converter: MarkItDown, +) -> None: + content = ( + "1\n00:00:01,000 --> 00:00:02,000\n" + 'Hello there.\n{\\an8}Second line\n' + ) + + assert _convert(converter, content) == "[00:00:01] Hello there. Second line" + + +def test_srt_bom_and_non_ascii_are_preserved(converter: MarkItDown) -> None: + content = "1\n00:00:01,000 --> 00:00:02,000\nCafé 你好\n" + + assert _convert(converter, content, charset="utf-8") == "[00:00:01] Café 你好" + + +def test_srt_cue_without_number_and_without_text(converter: MarkItDown) -> None: + content = ( + "00:00:01,000 --> 00:00:02,000\nNo number\n\n" + "2\n00:00:03,000 --> 00:00:04,000\n\n\n" + "3\n00:00:05.5 --> 00:00:06.5\nDot milliseconds\n" + ) + + assert _convert(converter, content) == ( + "[00:00:01] No number\n\n[00:00:05] Dot milliseconds" + ) + + +def test_srt_detected_without_extension(converter: MarkItDown) -> None: + content = "1\n00:00:01,000 --> 00:00:02,000\nDetected\n" + + result = converter.convert_stream(io.BytesIO(content.encode())) + + assert result.markdown == "[00:00:01] Detected" + + +def test_srt_late_non_ascii_after_charset_sample(converter: MarkItDown) -> None: + cue = "1\n00:00:01,000 --> 00:00:02,000\nplain ascii text\n\n" + content = cue * 3000 + "2\n00:00:03,000 --> 00:00:04,000\nCafé 你好\n" + assert len(content.encode()) > 65536 + + result = _convert(converter, content, charset="ascii") + + assert result.endswith("[00:00:03] Café 你好") + + +def test_srt_missing_separator_and_blank_line_inside_cue( + converter: MarkItDown, +) -> None: + content = ( + "1\n00:00:01,000 --> 00:00:02,000\nHello\n" + "2\n00:00:03,000 --> 00:00:04,000\nLine one\n\nLine two\n\n" + "3\n00:00:05,000 --> 00:00:06,000\nEnd\n" + ) + + assert _convert(converter, content) == ( + "[00:00:01] Hello\n\n[00:00:03] Line one Line two\n\n[00:00:05] End" + ) + + +def test_srt_keeps_angle_brackets_that_are_not_styling(converter: MarkItDown) -> None: + content = "1\n00:00:01,000 --> 00:00:02,000\n if xw bold\n" + + assert _convert(converter, content) == "[00:00:01] if xw bold" + + +def test_srt_hours_are_zero_padded(converter: MarkItDown) -> None: + content = "1\n0:00:01,000 --> 0:00:02,000\nHi\n" + + assert _convert(converter, content) == "[00:00:01] Hi" + + +def test_srt_without_cues_falls_back_to_plain_text(converter: MarkItDown) -> None: + assert _convert(converter, "just some plain text\n") == "just some plain text\n"