aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/tests/test_stream_subtitle_tracks.py
blob: cafb2e32d83fb6ad21748e479a16bd84eb91ee7f (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
"""
The viewer picks a subtitle track, and only the ones that can be shown.

MSE decodes no in-band text track, so a subtitle cannot ride inside the
fragmented MP4 the player is fed: it is extracted whole, converted to WebVTT,
cached under its own hash and pulled through the ordinary chunk path. Whole,
because that makes the cue timestamps absolute — a seek re-extracts nothing
and the `<track>` survives every restart of the MediaSource underneath it.

Two things here are about *not* offering something. Roughly a fifth of the
subtitle streams in a real library are bitmap (PGS, VOBSUB) and have no path
to WebVTT without OCR; a bitmap track extracted anyway yields a WebVTT with a
header and no cues, which is a subtitle track that appears in the menu and
does nothing. So they are not listed — and, because they still occupy a
position in `-map 0:s:<n>`, the ordinal of the tracks that *are* listed is not
their position in the list. That is the whole trap, and it is the same one
`AudioTrack.ordinal` exists for, one level deeper.

**The fixture's unusable stream is TTML, not bitmap, and that is deliberate.**
ffmpeg refuses to encode text to bitmap, so a PGS stream cannot be synthesised
here at all; TTML is a stream this ffmpeg has no decoder for, which is the
same branch — `codec_name not in TEXT_SUBTITLE_CODECS` — reached by exactly
the same route. The real bitmap codec names are asserted against the allow-list
directly, where no fixture is needed.

Tracks are told apart by **the words in the extracted cues**, never by their
language tags: a tag only proves the node copied a string it was handed.
"""

import shutil
import subprocess
from pathlib import Path

import pytest
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey

from meshbay_common.crypto import generate_gek
from meshbay_common.webcrypto import chunk_key_aes, decrypt_chunk_aes
from meshbay_node.indexer.group_index import GroupIndex
from meshbay_node.media_probe import TEXT_SUBTITLE_CODECS
from meshbay_node.transport.webrtc_server import WebRTCPeerSession, _probe_video

from conftest import needs_subprocess, one_root

_HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe")
# `asyncio` is per-test rather than on the module: one test here needs no
# event loop, and a module-wide mark on a synchronous function is a warning
# that reads as a broken test every time the suite runs.
pytestmark = [
    pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed"),
    needs_subprocess,
]

# Ordinal 0 is the unusable one and is never listed; 1 and 2 are the text
# tracks. The words differ per track because that is what the assertions read.
_CUE_WORD = {1: "francaise", 2: "English"}

_SRT_FR = """1
00:00:01,000 --> 00:00:03,000
Ceci est la piste francaise.
"""

_SRT_EN = """1
00:00:01,000 --> 00:00:03,000
This is the English track.
"""


def _make_subtitled_clip(path: Path) -> None:
    """~6 s of video, then three subtitle streams: TTML, then two text ones.

    The video and audio are muxed first, so the subtitle streams sit at
    container indices 2, 3 and 4 while their subtitle *ordinals* are 0, 1 and
    2 — and the first ordinal belongs to a stream that is never listed, so the
    listed tracks are 1 and 2 and never 0 and 1.
    """
    tmp = path.parent
    fr, en = tmp / "fr.srt", tmp / "en.srt"
    fr.write_text(_SRT_FR, encoding="utf-8")
    en.write_text(_SRT_EN, encoding="utf-8")
    base = tmp / "base.mp4"
    subprocess.run(
        ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
         "-f", "lavfi", "-i", "testsrc=size=320x240:rate=10:duration=6",
         "-f", "lavfi", "-i", "sine=duration=6",
         "-c:v", "libx264", "-preset", "ultrafast", "-c:a", "aac",
         "-shortest", str(base)],
        check=True, capture_output=True)
    subprocess.run(
        ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
         "-i", str(base), "-i", str(fr), "-i", str(en),
         "-map", "0:v", "-map", "0:a", "-map", "1", "-map", "1", "-map", "2",
         "-c:v", "copy", "-c:a", "copy",
         "-c:s:0", "ttml", "-c:s:1", "mov_text", "-c:s:2", "mov_text",
         "-metadata:s:s:0", "language=fre",
         "-metadata:s:s:1", "language=fre",
         "-metadata:s:s:2", "language=eng",
         str(path)],
        check=True, capture_output=True)


class _FakeMediaCache:
    """The three methods `_do_subtitle_request` uses, and a count of the puts.

    A double rather than the real cache because what is under test is the
    handler's use of it — that it looks before extracting, and extracts once.
    """

    def __init__(self):
        self.blobs: dict[str, bytes] = {}
        self.by_file_id: dict[str, str] = {}
        self.puts = 0

    async def get_thumb_hash_by_file_id(self, file_id: str) -> str | None:
        return self.by_file_id.get(file_id)

    async def get_thumb(self, thumb_hash: str) -> bytes | None:
        return self.blobs.get(thumb_hash)

    async def put_thumb(self, thumb_hash: str, file_id: str, blob: bytes) -> None:
        self.puts += 1
        self.blobs[thumb_hash] = blob
        self.by_file_id[file_id] = thumb_hash


def _session(video_path: Path, gek: bytes):
    import blake3
    file_bytes = video_path.read_bytes()
    file_id = blake3.blake3(file_bytes).hexdigest()

    sk_node = Ed25519PrivateKey.generate()
    index = GroupIndex(group_id="g" * 32, sk_node=sk_node, gek=gek)
    from meshbay_common.protocol import IndexEntry
    index.add_entry(IndexEntry(
        id=file_id, name=video_path.name, path=video_path.parent.name,
        size=len(file_bytes), type="video", added_at=0))

    session = WebRTCPeerSession.__new__(WebRTCPeerSession)
    session._ctx = {
        "roots": one_root(video_path.parent),
        "index": index,
        "gek": gek,
        "sk_node": sk_node,
        "max_concurrent_streams": 4,
        "media_cache": _FakeMediaCache(),
    }
    session._group_id = None
    session._user_id = "tester"
    session._stream_stopped = False
    session._stream_keepalives = 0
    session.sent = []
    session._send = session.sent.append
    session._audit = lambda *a, **k: None
    return session, file_id


async def _ask_for(session, file_id: str, track: int) -> dict:
    before = len(session.sent)
    await session._do_subtitle_request({"file_id": file_id, "track": track})
    replies = session.sent[before:]
    assert len(replies) == 1, f"expected one reply, got {replies}"
    return replies[0]


@pytest.mark.asyncio
async def test_probe_lists_only_text_tracks_and_numbers_them_by_stream(tmp_path):
    """The trap this feature is one wrong line away from.

    Numbering the survivors of the filter would give the two text tracks the
    ordinals 0 and 1, and `-map 0:s:0` would then extract the stream that
    cannot be decoded — which produces an empty WebVTT, not an error.
    """
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)

    probe = await _probe_video(str(clip))

    assert [tr.ordinal for tr in probe.subtitle_tracks] == [1, 2], (
        "the listed tracks must keep their position among all subtitle "
        "streams, not be renumbered from zero")
    assert [tr.language for tr in probe.subtitle_tracks] == ["fre", "eng"]
    assert all(tr.codec_name == "mov_text" for tr in probe.subtitle_tracks)

    # The fixture really does carry a subtitle stream that is not listed, and
    # really does put the subtitles at container indices of their own — or the
    # assertion above distinguishes nothing.
    raw = subprocess.run(
        ["ffprobe", "-v", "error", "-select_streams", "s",
         "-show_entries", "stream=index,codec_name", "-of", "csv=p=0", str(clip)],
        check=True, capture_output=True, text=True)
    rows = [line.split(",") for line in raw.stdout.split()]
    assert [int(r[0]) for r in rows] == [2, 3, 4]
    assert [r[1] for r in rows] == ["ttml", "mov_text", "mov_text"]


def test_bitmap_codecs_are_not_offered():
    """The 20 % no amount of ffmpeg turns into text.

    Asserted against the allow-list rather than a fixture because ffmpeg
    cannot encode text to bitmap, so a PGS or VOBSUB stream cannot be built
    here — while the names ffprobe reports for them are fixed and are what the
    filter is actually matched against.
    """
    for codec in ("hdmv_pgs_subtitle", "dvd_subtitle", "dvb_subtitle", "xsub"):
        assert codec not in TEXT_SUBTITLE_CODECS
    # And the two that make up four-fifths of a real library are.
    assert "subrip" in TEXT_SUBTITLE_CODECS
    assert "ass" in TEXT_SUBTITLE_CODECS


@pytest.mark.asyncio
async def test_stream_init_announces_the_tracks(tmp_path):
    """How a client discovers this node can do subtitles at all.

    From the answer, never from a version number: a node too old to enumerate
    sends no list, the client draws no selector and never asks.
    """
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    await session._stream_video_inner(
        {"file_id": file_id, "start": 0, "credits": 0})

    init = next(m for m in session.sent if m.get("type") == "stream_init")
    assert [tr["i"] for tr in init["subtitle_tracks"]] == [1, 2]
    assert [tr["lang"] for tr in init["subtitle_tracks"]] == ["fre", "eng"]


@pytest.mark.parametrize("track", [1, 2])
@pytest.mark.asyncio
async def test_the_requested_track_is_the_one_extracted(tmp_path, track):
    """Read out of the cues, not out of the reply's language tag."""
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    reply = await _ask_for(session, file_id, track)

    assert reply["type"] == "subtitle_resp"
    assert reply["track"] == track
    assert reply["mime"] == "text/vtt"
    vtt = session._ctx["media_cache"].blobs[reply["hash"]].decode("utf-8")
    assert vtt.startswith("WEBVTT")
    assert _CUE_WORD[track] in vtt
    other = _CUE_WORD[1 if track == 2 else 2]
    assert other not in vtt, (
        f"track {track} carries the other track's words, so the ordinal was "
        "mapped to the wrong stream")


@pytest.mark.asyncio
async def test_a_track_that_cannot_be_decoded_is_refused_not_served_empty(tmp_path):
    """Ordinal 0 exists in the container and is not in the list.

    A viewer cannot ask for it through the interface, which draws its menu
    from the list — but the ordinal travels on the wire, and a reply carrying
    a WebVTT with no cues in it would be a track that appears and shows
    nothing, with no error anywhere to lead back here.
    """
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    reply = await _ask_for(session, file_id, 0)

    assert reply["type"] == "error"
    assert session._ctx["media_cache"].puts == 0, (
        "nothing may be cached for a track that could not be extracted")


@pytest.mark.asyncio
async def test_an_ordinal_past_the_end_is_refused(tmp_path):
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    reply = await _ask_for(session, file_id, 9)

    assert reply["type"] == "error"


@pytest.mark.asyncio
async def test_a_second_request_is_served_from_the_cache(tmp_path):
    """The reason the extraction is whole-file rather than per-seek.

    A film's subtitles are extracted once in the life of the file: the second
    viewing, the second seek and the second sitting all answer from the cache,
    and ffmpeg runs exactly once.
    """
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    first = await _ask_for(session, file_id, 1)
    second = await _ask_for(session, file_id, 1)

    assert first["hash"] == second["hash"]
    assert session._ctx["media_cache"].puts == 1, (
        "the second request re-extracted instead of reading the cache")


@pytest.mark.asyncio
async def test_the_result_is_fetched_through_the_ordinary_chunk_path(tmp_path):
    """The reply names a cache hash, not a new transfer mechanism.

    Same indirection as an audio transcode or a TMDB poster — and it has to
    actually resolve, or the client is handed a hash it cannot pull.
    """
    clip = tmp_path / "clip.mp4"
    _make_subtitled_clip(clip)
    gek = generate_gek()
    session, file_id = _session(clip, gek)

    reply = await _ask_for(session, file_id, 2)
    chunk = await session._try_serve_thumbnail(reply["hash"], 0, gek)

    assert chunk is not None, "the hash in the reply resolves to nothing"
    key = chunk_key_aes(gek, bytes.fromhex(reply["hash"]), 0)
    plain = decrypt_chunk_aes(key, chunk["nonce"], chunk["ct"])
    assert plain.decode("utf-8").startswith("WEBVTT")
    assert _CUE_WORD[2] in plain.decode("utf-8")