aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/tests/test_stream_subtitle_tracks.py
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-09-17 13:39:23 +0200
committerChristophe Besson <cbesson@gmail.com>2026-09-17 13:39:23 +0200
commitad4ca3229002997934ccb5b2eaeb553c13b8888f (patch)
treea680f9e5d3ba43a84236b84f73e0b71eaf916db4 /packages/meshbay-node/tests/test_stream_subtitle_tracks.py
parent3e6d514663a5df1be3b2f0286c5f67f669d9c1d6 (diff)
downloadmeshbay-ad4ca3229002997934ccb5b2eaeb553c13b8888f.tar.gz
feat: embedded subtitles in the video player (MNP 3.3)
MSE decodes no in-band text track, so a subtitle cannot ride inside the fragmented MP4 the player is fed. The node extracts one track whole, converts it to WebVTT and caches it under its own hash; the client pulls that blob through the ordinary file_req/chunk path and hangs a <track> on the video element — the same indirection as a TMDB poster or an audio transcode, which is what makes a film's subtitles extracted once in the life of the file rather than once per viewing. Whole-file also makes the cues absolute, so a seek and an audio-language change both leave the track untouched. **The ordinal counts every subtitle stream, including the ones never listed.** Only text codecs are offered: a bitmap track (PGS, VOBSUB — about a fifth of a real library) has no path to WebVTT without OCR, and one extracted anyway yields a header with no cues, which is a menu entry that shows nothing and reports no error. Numbering the survivors of that filter would give a PGS/SRT/SRT file the ordinals 0 and 1 for its text tracks and `-map 0:s:0` would then extract the PGS — the same trap `AudioTrack.ordinal` exists for, one level deeper. A fixture whose first subtitle stream cannot be decoded pins it, and the handler checks membership of the probed list, never a range. Additive and MINOR: the selector is drawn from `subtitle_tracks` in the node's own `stream_init` and from no version number, so `subtitle_req` is never sent to a peer that would not answer it. The floor stays at 3.0. Also here: a failed extraction never touches playback, a superseded reply cannot install its blob over a newer choice, and `_languageName` is shared with the audio labels — lifted by both label harnesses, since a lift that names one function stops covering the rule the moment logic moves out of it. Tests: 9 node (tracks told apart by the words in the extracted cues, not by tags), 10 client. Full suite green: 1545 node/common, 1252 hub. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01UGY17EPph5LsLzePPXhUVc
Diffstat (limited to 'packages/meshbay-node/tests/test_stream_subtitle_tracks.py')
-rw-r--r--packages/meshbay-node/tests/test_stream_subtitle_tracks.py327
1 files changed, 327 insertions, 0 deletions
diff --git a/packages/meshbay-node/tests/test_stream_subtitle_tracks.py b/packages/meshbay-node/tests/test_stream_subtitle_tracks.py
new file mode 100644
index 0000000..cafb2e3
--- /dev/null
+++ b/packages/meshbay-node/tests/test_stream_subtitle_tracks.py
@@ -0,0 +1,327 @@
+"""
+The viewer picks a subtitle track, and only the ones that can be shown.
+
+MSE decodes no in-band text track, so a subtitle cannot ride inside the
+fragmented MP4 the player is fed: it is extracted whole, converted to WebVTT,
+cached under its own hash and pulled through the ordinary chunk path. Whole,
+because that makes the cue timestamps absolute — a seek re-extracts nothing
+and the `<track>` survives every restart of the MediaSource underneath it.
+
+Two things here are about *not* offering something. Roughly a fifth of the
+subtitle streams in a real library are bitmap (PGS, VOBSUB) and have no path
+to WebVTT without OCR; a bitmap track extracted anyway yields a WebVTT with a
+header and no cues, which is a subtitle track that appears in the menu and
+does nothing. So they are not listed — and, because they still occupy a
+position in `-map 0:s:<n>`, the ordinal of the tracks that *are* listed is not
+their position in the list. That is the whole trap, and it is the same one
+`AudioTrack.ordinal` exists for, one level deeper.
+
+**The fixture's unusable stream is TTML, not bitmap, and that is deliberate.**
+ffmpeg refuses to encode text to bitmap, so a PGS stream cannot be synthesised
+here at all; TTML is a stream this ffmpeg has no decoder for, which is the
+same branch — `codec_name not in TEXT_SUBTITLE_CODECS` — reached by exactly
+the same route. The real bitmap codec names are asserted against the allow-list
+directly, where no fixture is needed.
+
+Tracks are told apart by **the words in the extracted cues**, never by their
+language tags: a tag only proves the node copied a string it was handed.
+"""
+
+import shutil
+import subprocess
+from pathlib import Path
+
+import pytest
+from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
+
+from meshbay_common.crypto import generate_gek
+from meshbay_common.webcrypto import chunk_key_aes, decrypt_chunk_aes
+from meshbay_node.indexer.group_index import GroupIndex
+from meshbay_node.media_probe import TEXT_SUBTITLE_CODECS
+from meshbay_node.transport.webrtc_server import WebRTCPeerSession, _probe_video
+
+from conftest import needs_subprocess, one_root
+
+_HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe")
+# `asyncio` is per-test rather than on the module: one test here needs no
+# event loop, and a module-wide mark on a synchronous function is a warning
+# that reads as a broken test every time the suite runs.
+pytestmark = [
+ pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed"),
+ needs_subprocess,
+]
+
+# Ordinal 0 is the unusable one and is never listed; 1 and 2 are the text
+# tracks. The words differ per track because that is what the assertions read.
+_CUE_WORD = {1: "francaise", 2: "English"}
+
+_SRT_FR = """1
+00:00:01,000 --> 00:00:03,000
+Ceci est la piste francaise.
+"""
+
+_SRT_EN = """1
+00:00:01,000 --> 00:00:03,000
+This is the English track.
+"""
+
+
+def _make_subtitled_clip(path: Path) -> None:
+ """~6 s of video, then three subtitle streams: TTML, then two text ones.
+
+ The video and audio are muxed first, so the subtitle streams sit at
+ container indices 2, 3 and 4 while their subtitle *ordinals* are 0, 1 and
+ 2 — and the first ordinal belongs to a stream that is never listed, so the
+ listed tracks are 1 and 2 and never 0 and 1.
+ """
+ tmp = path.parent
+ fr, en = tmp / "fr.srt", tmp / "en.srt"
+ fr.write_text(_SRT_FR, encoding="utf-8")
+ en.write_text(_SRT_EN, encoding="utf-8")
+ base = tmp / "base.mp4"
+ subprocess.run(
+ ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
+ "-f", "lavfi", "-i", "testsrc=size=320x240:rate=10:duration=6",
+ "-f", "lavfi", "-i", "sine=duration=6",
+ "-c:v", "libx264", "-preset", "ultrafast", "-c:a", "aac",
+ "-shortest", str(base)],
+ check=True, capture_output=True)
+ subprocess.run(
+ ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
+ "-i", str(base), "-i", str(fr), "-i", str(en),
+ "-map", "0:v", "-map", "0:a", "-map", "1", "-map", "1", "-map", "2",
+ "-c:v", "copy", "-c:a", "copy",
+ "-c:s:0", "ttml", "-c:s:1", "mov_text", "-c:s:2", "mov_text",
+ "-metadata:s:s:0", "language=fre",
+ "-metadata:s:s:1", "language=fre",
+ "-metadata:s:s:2", "language=eng",
+ str(path)],
+ check=True, capture_output=True)
+
+
+class _FakeMediaCache:
+ """The three methods `_do_subtitle_request` uses, and a count of the puts.
+
+ A double rather than the real cache because what is under test is the
+ handler's use of it — that it looks before extracting, and extracts once.
+ """
+
+ def __init__(self):
+ self.blobs: dict[str, bytes] = {}
+ self.by_file_id: dict[str, str] = {}
+ self.puts = 0
+
+ async def get_thumb_hash_by_file_id(self, file_id: str) -> str | None:
+ return self.by_file_id.get(file_id)
+
+ async def get_thumb(self, thumb_hash: str) -> bytes | None:
+ return self.blobs.get(thumb_hash)
+
+ async def put_thumb(self, thumb_hash: str, file_id: str, blob: bytes) -> None:
+ self.puts += 1
+ self.blobs[thumb_hash] = blob
+ self.by_file_id[file_id] = thumb_hash
+
+
+def _session(video_path: Path, gek: bytes):
+ import blake3
+ file_bytes = video_path.read_bytes()
+ file_id = blake3.blake3(file_bytes).hexdigest()
+
+ sk_node = Ed25519PrivateKey.generate()
+ index = GroupIndex(group_id="g" * 32, sk_node=sk_node, gek=gek)
+ from meshbay_common.protocol import IndexEntry
+ index.add_entry(IndexEntry(
+ id=file_id, name=video_path.name, path=video_path.parent.name,
+ size=len(file_bytes), type="video", added_at=0))
+
+ session = WebRTCPeerSession.__new__(WebRTCPeerSession)
+ session._ctx = {
+ "roots": one_root(video_path.parent),
+ "index": index,
+ "gek": gek,
+ "sk_node": sk_node,
+ "max_concurrent_streams": 4,
+ "media_cache": _FakeMediaCache(),
+ }
+ session._group_id = None
+ session._user_id = "tester"
+ session._stream_stopped = False
+ session._stream_keepalives = 0
+ session.sent = []
+ session._send = session.sent.append
+ session._audit = lambda *a, **k: None
+ return session, file_id
+
+
+async def _ask_for(session, file_id: str, track: int) -> dict:
+ before = len(session.sent)
+ await session._do_subtitle_request({"file_id": file_id, "track": track})
+ replies = session.sent[before:]
+ assert len(replies) == 1, f"expected one reply, got {replies}"
+ return replies[0]
+
+
+@pytest.mark.asyncio
+async def test_probe_lists_only_text_tracks_and_numbers_them_by_stream(tmp_path):
+ """The trap this feature is one wrong line away from.
+
+ Numbering the survivors of the filter would give the two text tracks the
+ ordinals 0 and 1, and `-map 0:s:0` would then extract the stream that
+ cannot be decoded — which produces an empty WebVTT, not an error.
+ """
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+
+ probe = await _probe_video(str(clip))
+
+ assert [tr.ordinal for tr in probe.subtitle_tracks] == [1, 2], (
+ "the listed tracks must keep their position among all subtitle "
+ "streams, not be renumbered from zero")
+ assert [tr.language for tr in probe.subtitle_tracks] == ["fre", "eng"]
+ assert all(tr.codec_name == "mov_text" for tr in probe.subtitle_tracks)
+
+ # The fixture really does carry a subtitle stream that is not listed, and
+ # really does put the subtitles at container indices of their own — or the
+ # assertion above distinguishes nothing.
+ raw = subprocess.run(
+ ["ffprobe", "-v", "error", "-select_streams", "s",
+ "-show_entries", "stream=index,codec_name", "-of", "csv=p=0", str(clip)],
+ check=True, capture_output=True, text=True)
+ rows = [line.split(",") for line in raw.stdout.split()]
+ assert [int(r[0]) for r in rows] == [2, 3, 4]
+ assert [r[1] for r in rows] == ["ttml", "mov_text", "mov_text"]
+
+
+def test_bitmap_codecs_are_not_offered():
+ """The 20 % no amount of ffmpeg turns into text.
+
+ Asserted against the allow-list rather than a fixture because ffmpeg
+ cannot encode text to bitmap, so a PGS or VOBSUB stream cannot be built
+ here — while the names ffprobe reports for them are fixed and are what the
+ filter is actually matched against.
+ """
+ for codec in ("hdmv_pgs_subtitle", "dvd_subtitle", "dvb_subtitle", "xsub"):
+ assert codec not in TEXT_SUBTITLE_CODECS
+ # And the two that make up four-fifths of a real library are.
+ assert "subrip" in TEXT_SUBTITLE_CODECS
+ assert "ass" in TEXT_SUBTITLE_CODECS
+
+
+@pytest.mark.asyncio
+async def test_stream_init_announces_the_tracks(tmp_path):
+ """How a client discovers this node can do subtitles at all.
+
+ From the answer, never from a version number: a node too old to enumerate
+ sends no list, the client draws no selector and never asks.
+ """
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ await session._stream_video_inner(
+ {"file_id": file_id, "start": 0, "credits": 0})
+
+ init = next(m for m in session.sent if m.get("type") == "stream_init")
+ assert [tr["i"] for tr in init["subtitle_tracks"]] == [1, 2]
+ assert [tr["lang"] for tr in init["subtitle_tracks"]] == ["fre", "eng"]
+
+
+@pytest.mark.parametrize("track", [1, 2])
+@pytest.mark.asyncio
+async def test_the_requested_track_is_the_one_extracted(tmp_path, track):
+ """Read out of the cues, not out of the reply's language tag."""
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ reply = await _ask_for(session, file_id, track)
+
+ assert reply["type"] == "subtitle_resp"
+ assert reply["track"] == track
+ assert reply["mime"] == "text/vtt"
+ vtt = session._ctx["media_cache"].blobs[reply["hash"]].decode("utf-8")
+ assert vtt.startswith("WEBVTT")
+ assert _CUE_WORD[track] in vtt
+ other = _CUE_WORD[1 if track == 2 else 2]
+ assert other not in vtt, (
+ f"track {track} carries the other track's words, so the ordinal was "
+ "mapped to the wrong stream")
+
+
+@pytest.mark.asyncio
+async def test_a_track_that_cannot_be_decoded_is_refused_not_served_empty(tmp_path):
+ """Ordinal 0 exists in the container and is not in the list.
+
+ A viewer cannot ask for it through the interface, which draws its menu
+ from the list — but the ordinal travels on the wire, and a reply carrying
+ a WebVTT with no cues in it would be a track that appears and shows
+ nothing, with no error anywhere to lead back here.
+ """
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ reply = await _ask_for(session, file_id, 0)
+
+ assert reply["type"] == "error"
+ assert session._ctx["media_cache"].puts == 0, (
+ "nothing may be cached for a track that could not be extracted")
+
+
+@pytest.mark.asyncio
+async def test_an_ordinal_past_the_end_is_refused(tmp_path):
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ reply = await _ask_for(session, file_id, 9)
+
+ assert reply["type"] == "error"
+
+
+@pytest.mark.asyncio
+async def test_a_second_request_is_served_from_the_cache(tmp_path):
+ """The reason the extraction is whole-file rather than per-seek.
+
+ A film's subtitles are extracted once in the life of the file: the second
+ viewing, the second seek and the second sitting all answer from the cache,
+ and ffmpeg runs exactly once.
+ """
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ first = await _ask_for(session, file_id, 1)
+ second = await _ask_for(session, file_id, 1)
+
+ assert first["hash"] == second["hash"]
+ assert session._ctx["media_cache"].puts == 1, (
+ "the second request re-extracted instead of reading the cache")
+
+
+@pytest.mark.asyncio
+async def test_the_result_is_fetched_through_the_ordinary_chunk_path(tmp_path):
+ """The reply names a cache hash, not a new transfer mechanism.
+
+ Same indirection as an audio transcode or a TMDB poster — and it has to
+ actually resolve, or the client is handed a hash it cannot pull.
+ """
+ clip = tmp_path / "clip.mp4"
+ _make_subtitled_clip(clip)
+ gek = generate_gek()
+ session, file_id = _session(clip, gek)
+
+ reply = await _ask_for(session, file_id, 2)
+ chunk = await session._try_serve_thumbnail(reply["hash"], 0, gek)
+
+ assert chunk is not None, "the hash in the reply resolves to nothing"
+ key = chunk_key_aes(gek, bytes.fromhex(reply["hash"]), 0)
+ plain = decrypt_chunk_aes(key, chunk["nonce"], chunk["ct"])
+ assert plain.decode("utf-8").startswith("WEBVTT")
+ assert _CUE_WORD[2] in plain.decode("utf-8")