summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node
diff options
context:
space:
mode:
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node')
-rw-r--r--packages/meshbay-node/src/meshbay_node/linkpreview.py251
-rw-r--r--packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py79
2 files changed, 329 insertions, 1 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/linkpreview.py b/packages/meshbay-node/src/meshbay_node/linkpreview.py
new file mode 100644
index 0000000..64067d1
--- /dev/null
+++ b/packages/meshbay-node/src/meshbay_node/linkpreview.py
@@ -0,0 +1,251 @@
+"""
+Link unfurling for chat — fetch a URL a member pasted and pull an
+OpenGraph-style card out of it (title, description, site name, image).
+
+Who does the fetch matters. It is **the node**, not the browser and not the
+hub:
+
+ * the browser cannot — a strict `img-src`/`connect-src` and CORS block it,
+ and a direct fetch would leak every reader's IP to the linked host on
+ every render;
+ * the hub must not — it never touches group content (draft-v6 §2.5);
+ * the node already fetches third-party metadata for the Videos and Music
+ apps (`_fetch_and_cache_poster`), over the same authorised path.
+
+Because the node makes an outbound request to an address a *member* chose,
+this is an SSRF surface. `safe_url()` is the gate: http(s) only, no
+credentials, and the resolved address must be globally routable — no
+loopback, private, link-local, multicast or reserved range. Redirects are
+followed by hand so every hop is re-checked. Residual: a DNS name that
+resolves clean here and to something internal microseconds later at connect
+time (rebinding) — narrow, and closed properly by pinning the checked IP,
+which is a follow-up.
+
+Nothing is stored durably: the caller keeps an in-memory TTL cache and the
+OG image rides the existing `media_cache` thumb store (same as a poster).
+"""
+
+from __future__ import annotations
+
+import ipaddress
+import logging
+import socket
+from html.parser import HTMLParser
+from io import BytesIO
+from urllib.parse import urljoin, urlsplit
+
+import httpx
+
+log = logging.getLogger(__name__)
+
+_TIMEOUT = 5.0
+_MAX_REDIRECTS = 3
+_MAX_HTML_BYTES = 512 * 1024
+_MAX_IMAGE_BYTES = 2 * 1024 * 1024
+_IMAGE_MAX_DIM = 600
+_UA = "MeshBayBot/1.0 (+https://meshbay.org; link preview)"
+
+
+class UnsafeURL(ValueError):
+ """The URL points somewhere the node must not fetch from."""
+
+
+def _addr_is_public(ip: str) -> bool:
+ try:
+ addr = ipaddress.ip_address(ip)
+ except ValueError:
+ return False
+ if isinstance(addr, ipaddress.IPv6Address) and addr.ipv4_mapped is not None:
+ addr = addr.ipv4_mapped
+ return not (
+ addr.is_private or addr.is_loopback or addr.is_link_local
+ or addr.is_multicast or addr.is_reserved or addr.is_unspecified
+ )
+
+
+def safe_url(url: str) -> str:
+ """Return the URL unchanged if it is safe to fetch, else raise UnsafeURL."""
+ if not isinstance(url, str) or len(url) > 2048:
+ raise UnsafeURL("missing or oversized")
+ parts = urlsplit(url)
+ if parts.scheme not in ("http", "https"):
+ raise UnsafeURL(f"scheme {parts.scheme!r}")
+ if parts.username or parts.password:
+ raise UnsafeURL("credentials in URL")
+ host = parts.hostname
+ if not host:
+ raise UnsafeURL("no host")
+ # An IP literal is checked directly; a name is resolved and every answer
+ # must be public — a hostname with one public and one 127.0.0.1 record
+ # would otherwise be a way in.
+ try:
+ infos = socket.getaddrinfo(host, parts.port or (443 if parts.scheme == "https" else 80),
+ proto=socket.IPPROTO_TCP)
+ except socket.gaierror as e:
+ raise UnsafeURL(f"cannot resolve: {e}")
+ resolved = {info[4][0] for info in infos}
+ if not resolved:
+ raise UnsafeURL("resolves to nothing")
+ bad = [ip for ip in resolved if not _addr_is_public(ip)]
+ if bad:
+ raise UnsafeURL(f"non-public address {bad[0]}")
+ return url
+
+
+class _HeadParser(HTMLParser):
+ """Collects <title> text and name/property→content from <meta> in <head>.
+
+ Stops caring once <body> starts: everything a card needs is in the head,
+ and a 512 KB page of body is not worth walking.
+ """
+
+ def __init__(self) -> None:
+ super().__init__(convert_charrefs=True)
+ self.metas: dict[str, str] = {}
+ self.title: str | None = None
+ self._in_title = False
+ self.done = False
+
+ def handle_starttag(self, tag, attrs):
+ if tag == "body":
+ self.done = True
+ elif tag == "title":
+ self._in_title = True
+ elif tag == "meta":
+ a = {k.lower(): (v or "") for k, v in attrs}
+ key = (a.get("property") or a.get("name") or "").lower().strip()
+ if key and "content" in a and key not in self.metas:
+ self.metas[key] = a["content"].strip()
+
+ def handle_endtag(self, tag):
+ if tag == "title":
+ self._in_title = False
+
+ def handle_data(self, data):
+ if self._in_title and self.title is None:
+ text = data.strip()
+ if text:
+ self.title = text
+
+
+def _first(metas: dict[str, str], *keys: str) -> str | None:
+ for k in keys:
+ v = metas.get(k)
+ if v:
+ return v
+ return None
+
+
+async def _get(client: httpx.AsyncClient, url: str) -> httpx.Response:
+ """One GET with manual, re-validated redirects."""
+ current = safe_url(url)
+ for _ in range(_MAX_REDIRECTS + 1):
+ resp = await client.get(current, headers={"User-Agent": _UA},
+ follow_redirects=False)
+ if resp.is_redirect and "location" in resp.headers:
+ current = safe_url(urljoin(current, resp.headers["location"]))
+ continue
+ return resp
+ raise UnsafeURL("too many redirects")
+
+
+async def fetch_preview(url: str, *, client: httpx.AsyncClient | None = None) -> dict | None:
+ """
+ Return {url, title, description, site_name, image_url} for a URL, or None
+ if it cannot be unfurled (unreachable, not HTML, nothing worth showing).
+ Never raises for an ordinary failure — the caller treats "no preview" as
+ the common case, exactly like a TMDB miss.
+ """
+ own = client is None
+ if own:
+ client = httpx.AsyncClient(timeout=_TIMEOUT, max_redirects=0)
+ try:
+ safe_url(url)
+ resp = await _get(client, url)
+ ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower()
+ if resp.status_code != 200 or ctype not in ("text/html", "application/xhtml+xml"):
+ return None
+
+ body = b""
+ async for chunk in resp.aiter_bytes():
+ body += chunk
+ if len(body) >= _MAX_HTML_BYTES:
+ break
+ final_url = str(resp.url)
+
+ parser = _HeadParser()
+ try:
+ parser.feed(body.decode(resp.encoding or "utf-8", errors="replace"))
+ except Exception:
+ pass
+ m = parser.metas
+
+ title = _first(m, "og:title", "twitter:title") or parser.title
+ description = _first(m, "og:description", "twitter:description", "description")
+ site_name = _first(m, "og:site_name") or urlsplit(final_url).hostname
+ image = _first(m, "og:image", "og:image:url", "og:image:secure_url",
+ "twitter:image", "twitter:image:src")
+ if image:
+ image = urljoin(final_url, image)
+ try:
+ safe_url(image)
+ except UnsafeURL:
+ image = None
+
+ if not title and not description:
+ return None
+
+ return {
+ "url": url,
+ "title": (title or "")[:300] or None,
+ "description": (description or "")[:600] or None,
+ "site_name": (site_name or "")[:120] or None,
+ "image_url": image,
+ }
+ except (httpx.HTTPError, UnsafeURL) as e:
+ log.debug("link preview for %s: %s", url[:80], e)
+ return None
+ finally:
+ if own:
+ await client.aclose()
+
+
+async def fetch_image(url: str, *, client: httpx.AsyncClient | None = None) -> bytes | None:
+ """Fetch and re-encode an OG image to a small JPEG. None on any failure."""
+ own = client is None
+ if own:
+ client = httpx.AsyncClient(timeout=_TIMEOUT, max_redirects=0)
+ try:
+ safe_url(url)
+ resp = await _get(client, url)
+ ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower()
+ if resp.status_code != 200 or not ctype.startswith("image/"):
+ return None
+ raw = b""
+ async for chunk in resp.aiter_bytes():
+ raw += chunk
+ if len(raw) > _MAX_IMAGE_BYTES:
+ return None
+ return _downscale(raw)
+ except (httpx.HTTPError, UnsafeURL) as e:
+ log.debug("link preview image %s: %s", url[:80], e)
+ return None
+ finally:
+ if own:
+ await client.aclose()
+
+
+def _downscale(raw: bytes) -> bytes | None:
+ try:
+ from PIL import Image
+ except ImportError:
+ return None
+ try:
+ with Image.open(BytesIO(raw)) as im:
+ im = im.convert("RGB")
+ im.thumbnail((_IMAGE_MAX_DIM, _IMAGE_MAX_DIM))
+ out = BytesIO()
+ im.save(out, format="JPEG", quality=80)
+ return out.getvalue()
+ except Exception:
+ return None
diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
index 252e202..4958b92 100644
--- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
+++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
@@ -97,7 +97,7 @@ from meshbay_common.webcrypto import chunk_key_aes, encrypt_chunk_aes
from meshbay_common.protocol import MNP, index_entry_wire
from meshbay_node.indexer import GroupIndex
from meshbay_node.indexer.indexer import DirectoryIndexer
-from meshbay_node import ops
+from meshbay_node import linkpreview, ops
# Re-imported under its original name: every call site and existing test in
# this module still refers to it as `_probe_video`. The implementation lives
# in media_probe.py so the indexer package (imported just above) can call it
@@ -115,6 +115,34 @@ log = logging.getLogger(__name__)
CHUNK_SIZE = 1024 * 1024
MAX_MSG = 64 * 1024 * 1024
+# Chat link-preview results, kept in memory only (draft-v6 §2.7: the node
+# produces enrichment on demand and keeps nothing durable — the asking device
+# caches). Bounded and time-limited so a busy group cannot grow it without end
+# and a page that changed its card is picked up within the hour.
+_LINK_PREVIEW_TTL = 3600
+_LINK_PREVIEW_MAX = 256
+_link_preview_cache: dict[str, tuple[float, dict]] = {}
+
+
+def _link_preview_cache_get(url: str) -> dict | None:
+ hit = _link_preview_cache.get(url)
+ if hit is None:
+ return None
+ ts, value = hit
+ if time.time() - ts > _LINK_PREVIEW_TTL:
+ _link_preview_cache.pop(url, None)
+ return None
+ return value
+
+
+def _link_preview_cache_put(url: str, value: dict) -> None:
+ if not url:
+ return
+ if len(_link_preview_cache) >= _LINK_PREVIEW_MAX:
+ oldest = min(_link_preview_cache, key=lambda k: _link_preview_cache[k][0])
+ _link_preview_cache.pop(oldest, None)
+ _link_preview_cache[url] = (time.time(), value)
+
# Upload limits (finding C5a). Uploads used to land directly in the shared root under
# a name the client chose, overwriting whatever was already there — which both violated
# node sovereignty and defeated the delete authorization (overwrite a file, become its
@@ -396,6 +424,8 @@ class WebRTCPeerSession:
self._do_chat_message(msg)
elif mtype == MNP.CHAT_HISTORY:
self._do_chat_history(msg)
+ elif mtype == MNP.LINK_PREVIEW_REQ:
+ self._spawn(self._do_link_preview_request(msg))
elif mtype == MNP.PING:
self._do_ping(msg)
elif mtype == MNP.FILE_UPLOAD:
@@ -3409,6 +3439,53 @@ class WebRTCPeerSession:
],
})
+ async def _do_link_preview_request(self, msg: dict) -> None:
+ """
+ Unfurl a URL a member pasted into chat (draft-v6 §2.7 enrichment rule:
+ the client asks, the node produces on demand, the asking device
+ caches — nothing durable here).
+
+ `linkpreview.safe_url` is the SSRF gate: the URL a *member* chose
+ decides an outbound request from the operator's machine, so http(s)
+ only and the resolved address must be globally routable. Failure of
+ any kind — blocked, unreachable, not HTML, nothing worth showing —
+ comes back as `ok: false`, the way a TMDB miss does; the client then
+ just shows the bare link.
+ """
+ url = msg.get("url")
+ key = url if isinstance(url, str) else ""
+ cached = _link_preview_cache_get(key)
+ if cached is not None:
+ self._send({**cached, "type": MNP.LINK_PREVIEW_RESP, "v": MNP_VERSION})
+ return
+
+ resp: dict = {"type": MNP.LINK_PREVIEW_RESP, "v": MNP_VERSION,
+ "url": key, "ok": False}
+ try:
+ meta = await linkpreview.fetch_preview(url)
+ if meta is not None:
+ resp.update(ok=True, title=meta["title"],
+ description=meta["description"],
+ site_name=meta["site_name"])
+ image_url = meta.get("image_url")
+ media_cache = self._ctx.get("media_cache")
+ if image_url and media_cache is not None:
+ synthetic_id = f"linkpreview:{image_url}"
+ thumb_hash = await media_cache.get_thumb_hash_by_file_id(synthetic_id)
+ if thumb_hash is None:
+ jpeg = await linkpreview.fetch_image(image_url)
+ if jpeg:
+ thumb_hash = blake3.blake3(jpeg).hexdigest()
+ await media_cache.put_thumb(thumb_hash, synthetic_id, jpeg)
+ if thumb_hash:
+ resp["image_thumb_hash"] = thumb_hash
+ except Exception as e:
+ log.debug("link_preview_req %s: %s", key[:80], e)
+
+ _link_preview_cache_put(key, {k: v for k, v in resp.items()
+ if k not in ("type", "v")})
+ self._send(resp)
+
def _do_file_upload(self, msg: dict) -> None:
ctx = self._group_ctx()
filename = msg.get("filename", "")