1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
|
"""
Index-time enrichment for the Videos group app: technical probe (ffprobe),
filename parsing (title_parse), and thumbnail generation (ffmpeg) for a
newly-added video IndexEntry.
Runs through its own small bounded worker pool — separate from the streaming
transcode pool (docs/mediacenter.md §5.2, mirroring webrtc_server.py's
`_transcode_semaphore`) — so indexing a large library never blocks on this,
and enrichment never competes with an active viewer for CPU. The scan itself
already put the entry in the index with hash/size/type only; this fills in
the rest asynchronously and hands the result back via a callback.
"""
import asyncio
import logging
from pathlib import Path
from typing import Awaitable, Callable
import blake3
from meshbay_common.protocol import IndexEntry
from meshbay_node.indexer import title_parse
from meshbay_node.indexer.indexer import MEDIA_EXTENSIONS
from meshbay_node.media_cache import MediaCache
from meshbay_node.media_probe import probe_video
log = logging.getLogger(__name__)
DEFAULT_MAX_CONCURRENT = 2
PROBE_TIMEOUT_SECS = 30
THUMB_TIMEOUT_SECS = 30
THUMB_WIDTH = 320
# "Show/SeasonFolder/episode.mkv" is the expected shape, with a little slack
# for an extra wrapper folder — not an attempt to find the exact group root.
MAX_ANCESTOR_DEPTH = 4
# Bounds the "borrow a title from a sibling episode filename" scan (§3.4) so
# a folder with thousands of files costs a fixed, small amount of work.
MAX_SIBLINGS_CHECKED = 20
# Bounds the whole-show scan _synthetic_episode_number uses to rank a
# season's files across more than one folder (§3.4c) — a show's total file
# count, not just one folder's, so this needs more headroom than
# MAX_SIBLINGS_CHECKED.
MAX_SEASON_FILES_CHECKED = 500
def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, Path] | None:
"""
§3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered
season, or Specials/Bonus/Extras -> season 0) — season from the first
(innermost) match, but the show's own name from *above every
consecutive season-like ancestor*, not just the first one. A
per-season Bonus folder (`Show/Season N/Bonus/file.ext`) is nested two
levels inside the show, both of them season-like on their own
("Bonus" and "Season N") — stopping at the first would hand back
"Season N" as the show's name instead of "Show".
Trusted over any per-file guessit title once found: a bare episode
numbering convention with no show name embedded at all
(`001 Episode's Own Title.mkv`, no SxxExx, no show prefix) is
completely ordinary and makes guessit invent a title from whatever
text follows the number — that text is the individual episode's own
name, never the show's, and every episode in the folder produces a
different one. None of that ambiguity exists for the folder structure
itself: the show's own root folder names it once, not per file.
None if no ancestor looks like a season folder at all — a flat
library has nothing to borrow a show name from this way, and
_title_from_siblings' per-file logic is what applies instead.
"""
folder = file_path.parent
season: int | None = None
depth = 0
while folder is not None and folder.parent != folder and depth < MAX_ANCESTOR_DEPTH:
this_season = title_parse.season_from_folder_name(folder.name)
if this_season is None:
if season is None:
folder = folder.parent
depth += 1
continue
return season, folder
if season is None:
season = this_season
folder = folder.parent
depth += 1
return None
def _title_from_siblings(file_path: Path) -> str | None:
"""
§3.4: an episode filename with no show name in it borrows the title from
a representative sibling in the same folder, never from the folder name
alone (an acronym-named show folder is a real, observed case).
Requires the sibling to carry its own episode number too, not just a
title — a folder where every file is a one-off-named Special (§3.4b)
has plenty of `display_title`s (guessit reads *a* title off nearly
anything) but none of them name the show; requiring a real episode
number alongside is what tells apart a genuinely representative sibling
from another Special just like this one.
"""
try:
names = sorted(p.name for p in file_path.parent.iterdir() if p.is_file())
except OSError:
return None
checked = 0
for name in names:
if name == file_path.name:
continue
if Path(name).suffix.lower() not in MEDIA_EXTENSIONS["video"]:
continue
checked += 1
if checked > MAX_SIBLINGS_CHECKED:
break
parsed = title_parse.parse_episode_filename(name)
if parsed.display_title and parsed.episode is not None:
return parsed.display_title
return None
def _synthetic_episode_number(file_path: Path, show_root: Path, season: int) -> int:
"""
§3.4b/§3.4c: a Specials/Bonus folder's files often carry no episode
number at all — each is just named after its own one-off title. The
frontend (video-app.js's buildSeasons) sorts within a season by this
number but only needs it to provide a stable order, not to mean
anything beyond that.
Ranked across the *whole show*, not just this file's own folder: season
0 routinely spans more than one folder under the show's root — a
Bonus folder nested inside every numbered season
(`Show/Season N/Bonus/...`) is exactly the shape that produces this —
and ranking within just this file's own folder hands out the same
"episode 1" again in every other one, found live as three unrelated
Specials all showing up as "S0E01". Deterministic across re-scans as
long as the show's contents don't change. 1-based so it reads as
"episode 1", not "episode 0", in a UI that already uses season 0 for
"Specials" itself. Bounded (MAX_SEASON_FILES_CHECKED) so a huge show
costs a fixed amount of work rather than scaling with its size.
"""
try:
candidates = sorted(
(p for p in show_root.rglob("*")
if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]),
key=lambda p: (str(p.parent), p.name),
)
except OSError:
return 1
rank = 0
for i, candidate in enumerate(candidates):
if i >= MAX_SEASON_FILES_CHECKED:
break
ancestor = _season_and_show_from_ancestors(candidate)
if ancestor is None or ancestor[0] != season:
continue
rank += 1
if candidate == file_path:
return rank
return 1
async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None:
"""One ffmpeg frame grab at ~10% of duration (or 5s if unknown), scaled down."""
seek = max(0.0, (duration or 50.0) * 0.1)
from meshbay_node.platform import ffmpeg_cmd
proc = await asyncio.create_subprocess_exec(
ffmpeg_cmd(), "-v", "error", "-ss", str(seek), "-i", str(file_path),
"-frames:v", "1", "-vf", f"scale={THUMB_WIDTH}:-1",
"-f", "image2", "-c:v", "mjpeg", "pipe:1",
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE,
)
try:
stdout, _ = await asyncio.wait_for(proc.communicate(), THUMB_TIMEOUT_SECS)
except asyncio.TimeoutError:
proc.kill()
await proc.wait()
return None
return stdout or None
class Enricher:
"""Owns the node's bounded index-time enrichment pool."""
def __init__(self, media_cache: MediaCache, max_concurrent: int = DEFAULT_MAX_CONCURRENT):
self._media_cache = media_cache
self._sem = asyncio.Semaphore(max_concurrent)
self._tasks: set[asyncio.Task] = set()
def spawn(
self, entry: IndexEntry, file_path: Path,
on_done: Callable[[str, dict], Awaitable[None]],
) -> asyncio.Task:
"""
Fire-and-forget one file's enrichment. `on_done(file_id, fields)` is
awaited with the index fields to merge in once ready — never blocks
the caller (a scan or watchdog event). The reference this method
returns is what keeps the task alive; callers should hold it the
same way `WebRTCPeerSession._spawn` holds streaming tasks.
"""
task = asyncio.ensure_future(self._run(entry, file_path, on_done))
self._tasks.add(task)
def _cleanup(t: asyncio.Task) -> None:
self._tasks.discard(t)
if not t.cancelled() and t.exception():
log.error("Enrichment failed for %s: %s", entry.id[:12], t.exception(),
exc_info=t.exception())
task.add_done_callback(_cleanup)
return task
async def _run(
self, entry: IndexEntry, file_path: Path,
on_done: Callable[[str, dict], Awaitable[None]],
) -> None:
async with self._sem:
fields: dict = {}
# entry.id is the file's own content hash — a probe/thumbnail
# cache hit here means this exact content was already handled
# (this run, an earlier one, even a previous daemon process).
# Neither survives a restart on its own (the in-memory
# GroupIndex entry is rebuilt from scratch every time), but
# media_cache.db does — nothing was checking it before spawning
# ffprobe/ffmpeg again on every file, every restart.
#
# Deliberately *not* cached this way: display_title/season/
# episode. Those come from guessit against entry.name, which is
# exactly what a rename needs re-derived —
# _reenrich_renamed_video_entries exists for precisely that —
# and reusing a stale parse under a new name would silently
# defeat it. ffprobe's own output has no such concern: the same
# bytes probe the same regardless of what the file is called.
cached_meta = await self._media_cache.get_video_meta(entry.id)
duration: float | None = None
if cached_meta is not None:
fields["duration"] = cached_meta["duration"]
fields["width"] = cached_meta["width"]
fields["height"] = cached_meta["height"]
duration = cached_meta["duration"]
else:
try:
probe = await asyncio.wait_for(
probe_video(str(file_path)), timeout=PROBE_TIMEOUT_SECS)
duration = probe.duration
fields["duration"] = int(duration) if duration else None
fields["width"] = probe.width
fields["height"] = probe.height
except Exception as e:
log.warning("Probe failed for %s: %s", file_path, e)
await self._media_cache.put_video_meta(
entry.id, fields.get("duration"), fields.get("width"), fields.get("height"))
ep = title_parse.parse_episode_filename(entry.name)
ancestor = await asyncio.to_thread(_season_and_show_from_ancestors, file_path)
if ancestor is not None:
# A season-like ancestor folder exists — the show's own
# root folder names it, trusted over any per-file guessit
# title. This is what actually groups every episode under
# one show: a bare numbering convention with no show name
# in the filename at all (`001 Episode's Own Title.mkv`,
# ordinary enough on its own) makes guessit invent a title
# from whatever text follows the number, which is that
# episode's own name, never the show's — and differs for
# every episode, so nothing would ever group together.
# Found live: a real show's episodes and its Specials
# folder alike, both named this way, showed up as
# individual "movies" each searched against — and matched
# to — an unrelated real film sharing that one-off title,
# instead of anything grouped under the show at all.
season, show_folder = ancestor
# Not ep.episode: guessit reads a bare 3-digit leading
# number (routine once a season-like ancestor is already
# doing the real season/episode work) as a concatenated
# season+episode guess rather than a plain episode number
# — confirmed live, "100 Title.mkv" parsed as episode 0,
# not 100, with nothing in guessit's own output telling
# that apart from a real 2-digit episode. The direct regex
# reads the whole leading number as it is.
episode = title_parse.leading_episode_number(entry.name)
if episode is None:
episode = ep.episode
if episode is None:
episode = await asyncio.to_thread(
_synthetic_episode_number, file_path, show_folder, season)
fields["display_title"] = show_folder.name
fields["season"] = season
fields["episode"] = episode
elif (ep.season is not None and ep.episode is not None
and (title_parse.has_episode_marker(entry.name)
or title_parse.year_in(entry.name) is None)):
# No season-like ancestor at all (a flat library) but the
# filename itself carries season+episode (§3.4) — *and* it
# is a real marker, not guessit reading a bare number as
# SxxExx. A movie whose "1080p" tag was truncated to "108",
# or "1280" left in the name, otherwise parses to S01E08 /
# S12E80 and gets shelved as a nonexistent series
# (found live 2026-08-29). A genuine flat-dumped episode
# has an explicit SxxExx/1x08/"Episode N" marker; a movie
# has a "(2019)"-style year and no such marker.
title = ep.display_title or await asyncio.to_thread(
_title_from_siblings, file_path)
fields["display_title"] = title or title_parse.naive_title(entry.name)
fields["season"] = ep.season
fields["episode"] = ep.episode
else:
mv = title_parse.parse_movie_filename(entry.name)
fields["display_title"] = mv.display_title or mv.naive_title
cached_thumb_hash = await self._media_cache.get_thumb_hash_by_file_id(entry.id)
if cached_thumb_hash:
fields["thumb_hash"] = cached_thumb_hash
else:
try:
thumb = await _make_thumbnail(file_path, duration)
except Exception as e:
log.warning("Thumbnail generation failed for %s: %s", file_path, e)
thumb = None
if thumb:
thumb_hash = blake3.blake3(thumb).hexdigest()
await self._media_cache.put_thumb(thumb_hash, entry.id, thumb)
fields["thumb_hash"] = thumb_hash
await on_done(entry.id, fields)
|