Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions packages/markitdown/src/markitdown/_markitdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@
DocumentIntelligenceConverter,
ContentUnderstandingConverter,
CsvConverter,
SrtConverter,
)

from ._base_converter import DocumentConverter, DocumentConverterResult
Expand Down Expand Up @@ -250,6 +251,7 @@ def enable_builtins(self, **kwargs) -> None:
self.register_converter(OutlookMsgConverter())
self.register_converter(EpubConverter())
self.register_converter(CsvConverter())
self.register_converter(SrtConverter())

# Register Document Intelligence converter at the top of the stack if endpoint is provided
docintel_endpoint = kwargs.get("docintel_endpoint")
Expand Down
2 changes: 2 additions & 0 deletions packages/markitdown/src/markitdown/converters/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@
)
from ._epub_converter import EpubConverter
from ._csv_converter import CsvConverter
from ._srt_converter import SrtConverter

__all__ = [
"PlainTextConverter",
Expand All @@ -51,4 +52,5 @@
"ContentUnderstandingFileType",
"EpubConverter",
"CsvConverter",
"SrtConverter",
]
101 changes: 101 additions & 0 deletions packages/markitdown/src/markitdown/converters/_srt_converter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
import re
from typing import Any, BinaryIO

from charset_normalizer import from_bytes

from .._base_converter import DocumentConverter, DocumentConverterResult
from .._stream_info import StreamInfo

ACCEPTED_MIME_TYPE_PREFIXES = [
"text/srt",
"text/x-srt",
"application/x-subrip",
]
ACCEPTED_FILE_EXTENSIONS = [".srt"]

# "00:01:05,250 --> 00:01:07,000", capturing the start time without milliseconds.
_TIMING_RE = re.compile(
r"^(\d{1,3}):(\d{2}):(\d{2})(?:[,.]\d{1,3})?\s*-->\s*\d{1,3}:\d{2}:\d{2}(?:[,.]\d{1,3})?"
)
# Styling markup: <i>, <b>, <u>, <font ...> and {\an8}-style overrides. Other
# angle brackets, like "<Bob>" or "x<y", are real text and are kept.
_MARKUP_RE = re.compile(r"</?(?:i|b|u|font)(?:\s[^>]{0,200})?>|\{\\[^}]{0,200}\}", re.I)


class SrtConverter(DocumentConverter):
"""
Converts SubRip (.srt) subtitle files to a Markdown transcript, one line per
cue, each prefixed with its start time.
"""

def accepts(
self,
file_stream: BinaryIO,
stream_info: StreamInfo,
**kwargs: Any, # Options to pass to the converter
) -> bool:
mimetype = (stream_info.mimetype or "").lower()
extension = (stream_info.extension or "").lower()

if extension in ACCEPTED_FILE_EXTENSIONS:
return True

for prefix in ACCEPTED_MIME_TYPE_PREFIXES:
if mimetype.startswith(prefix):
return True

return False

def convert(
self,
file_stream: BinaryIO,
stream_info: StreamInfo,
**kwargs: Any, # Options to pass to the converter
) -> DocumentConverterResult:
data = file_stream.read()
content = None
if stream_info.charset:
# The charset hint comes from a sample of the file, so it can be
# wrong for later bytes (e.g. "ascii" for a file that turns non-ASCII
# after the sample). Fall through to a full-file guess in that case.
try:
content = data.decode(stream_info.charset)
except (UnicodeDecodeError, LookupError):
pass
if content is None:
detected = from_bytes(data).best()
content = (
str(detected)
if detected is not None
else data.decode("utf-8", errors="ignore")
)

content = content.lstrip("\ufeff").replace("\r\n", "\n").replace("\r", "\n")

# A cue starts at each timing line and runs to the next one, so a cue
# with a blank line inside it, or a missing separator, still parses.
lines = content.split("\n")
timings = [
(i, m)
for i, line in enumerate(lines)
if (m := _TIMING_RE.match(line.strip()))
]
if not timings and content.strip():
raise ValueError("No SRT cues found.")

cues = []
for n, (i, timing) in enumerate(timings):
end = timings[n + 1][0] if n + 1 < len(timings) else len(lines)
body = [line.strip() for line in lines[i + 1 : end]]
while body and not body[-1]:
body.pop()
# The last line before the next timing line is its cue number.
if n + 1 < len(timings) and body and body[-1].isdigit():
body.pop()

text = " ".join(_MARKUP_RE.sub("", " ".join(body)).split())
if text:
hours, minutes, seconds = timing.groups()
cues.append(f"[{hours.zfill(2)}:{minutes}:{seconds}] {text}")

return DocumentConverterResult(markdown="\n\n".join(cues))
116 changes: 116 additions & 0 deletions packages/markitdown/tests/test_srt.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
"""SRT subtitle conversion: cue text, timestamps, markup, and line endings."""

import io

import pytest

from markitdown import MarkItDown, StreamInfo


@pytest.fixture(scope="module")
def converter() -> MarkItDown:
return MarkItDown(enable_plugins=False)


def _convert(converter: MarkItDown, content: str, **info) -> str:
return converter.convert_stream(
io.BytesIO(content.encode("utf-8")),
stream_info=StreamInfo(extension=".srt", **info),
).markdown


@pytest.mark.parametrize("newline", ["\n", "\r\n", "\r"], ids=["LF", "CRLF", "CR"])
def test_srt_cues_become_timestamped_lines(converter: MarkItDown, newline: str) -> None:
content = newline.join(
[
"1",
"00:00:01,000 --> 00:00:03,500",
"Hello there.",
"",
"2",
"01:02:05,250 --> 01:02:07,000",
"Goodbye.",
"",
]
)

assert _convert(converter, content) == (
"[00:00:01] Hello there.\n\n[01:02:05] Goodbye."
)


def test_srt_multiline_cue_is_joined_and_markup_removed(
converter: MarkItDown,
) -> None:
content = (
"1\n00:00:01,000 --> 00:00:02,000\n"
'<i>Hello</i> <font color="red">there</font>.\n{\\an8}Second line\n'
)

assert _convert(converter, content) == "[00:00:01] Hello there. Second line"


def test_srt_bom_and_non_ascii_are_preserved(converter: MarkItDown) -> None:
content = "1\n00:00:01,000 --> 00:00:02,000\nCafé 你好\n"

assert _convert(converter, content, charset="utf-8") == "[00:00:01] Café 你好"


def test_srt_cue_without_number_and_without_text(converter: MarkItDown) -> None:
content = (
"00:00:01,000 --> 00:00:02,000\nNo number\n\n"
"2\n00:00:03,000 --> 00:00:04,000\n\n\n"
"3\n00:00:05.5 --> 00:00:06.5\nDot milliseconds\n"
)

assert _convert(converter, content) == (
"[00:00:01] No number\n\n[00:00:05] Dot milliseconds"
)


def test_srt_detected_without_extension(converter: MarkItDown) -> None:
content = "1\n00:00:01,000 --> 00:00:02,000\nDetected\n"

result = converter.convert_stream(io.BytesIO(content.encode()))

assert result.markdown == "[00:00:01] Detected"


def test_srt_late_non_ascii_after_charset_sample(converter: MarkItDown) -> None:
cue = "1\n00:00:01,000 --> 00:00:02,000\nplain ascii text\n\n"
content = cue * 3000 + "2\n00:00:03,000 --> 00:00:04,000\nCafé 你好\n"
assert len(content.encode()) > 65536

result = _convert(converter, content, charset="ascii")

assert result.endswith("[00:00:03] Café 你好")


def test_srt_missing_separator_and_blank_line_inside_cue(
converter: MarkItDown,
) -> None:
content = (
"1\n00:00:01,000 --> 00:00:02,000\nHello\n"
"2\n00:00:03,000 --> 00:00:04,000\nLine one\n\nLine two\n\n"
"3\n00:00:05,000 --> 00:00:06,000\nEnd\n"
)

assert _convert(converter, content) == (
"[00:00:01] Hello\n\n[00:00:03] Line one Line two\n\n[00:00:05] End"
)


def test_srt_keeps_angle_brackets_that_are_not_styling(converter: MarkItDown) -> None:
content = "1\n00:00:01,000 --> 00:00:02,000\n<Bob> if x<y and z>w <B>bold</B>\n"

assert _convert(converter, content) == "[00:00:01] <Bob> if x<y and z>w bold"


def test_srt_hours_are_zero_padded(converter: MarkItDown) -> None:
content = "1\n0:00:01,000 --> 0:00:02,000\nHi\n"

assert _convert(converter, content) == "[00:00:01] Hi"


def test_srt_without_cues_falls_back_to_plain_text(converter: MarkItDown) -> None:
assert _convert(converter, "just some plain text\n") == "just some plain text\n"