""" Link unfurling for chat — fetch a URL a member pasted and pull an OpenGraph-style card out of it (title, description, site name, image). Who does the fetch matters. It is **the node**, not the browser and not the hub: * the browser cannot — a strict `img-src`/`connect-src` and CORS block it, and a direct fetch would leak every reader's IP to the linked host on every render; * the hub must not — it never touches group content (draft-v6 §2.5); * the node already fetches third-party metadata for the Videos and Music apps (`_fetch_and_cache_poster`), over the same authorised path. Because the node makes an outbound request to an address a *member* chose, this is an SSRF surface. `safe_url()` is the gate: http(s) only, no credentials, and the resolved address must be globally routable — no loopback, private, link-local, multicast or reserved range. Redirects are followed by hand so every hop is re-checked. Residual: a DNS name that resolves clean here and to something internal microseconds later at connect time (rebinding) — narrow, and closed properly by pinning the checked IP, which is a follow-up. Nothing is stored durably: the caller keeps an in-memory TTL cache and the OG image rides the existing `media_cache` thumb store (same as a poster). """ from __future__ import annotations import ipaddress import logging import socket from html.parser import HTMLParser from io import BytesIO from urllib.parse import urljoin, urlsplit import httpx log = logging.getLogger(__name__) _TIMEOUT = 5.0 _MAX_REDIRECTS = 3 _MAX_HTML_BYTES = 512 * 1024 _MAX_IMAGE_BYTES = 2 * 1024 * 1024 _IMAGE_MAX_DIM = 600 _UA = "MeshBayBot/1.0 (+https://meshbay.org; link preview)" class UnsafeURL(ValueError): """The URL points somewhere the node must not fetch from.""" def _addr_is_public(ip: str) -> bool: try: addr = ipaddress.ip_address(ip) except ValueError: return False if isinstance(addr, ipaddress.IPv6Address) and addr.ipv4_mapped is not None: addr = addr.ipv4_mapped return not ( addr.is_private or addr.is_loopback or addr.is_link_local or addr.is_multicast or addr.is_reserved or addr.is_unspecified ) def safe_url(url: str) -> str: """Return the URL unchanged if it is safe to fetch, else raise UnsafeURL.""" if not isinstance(url, str) or len(url) > 2048: raise UnsafeURL("missing or oversized") parts = urlsplit(url) if parts.scheme not in ("http", "https"): raise UnsafeURL(f"scheme {parts.scheme!r}") if parts.username or parts.password: raise UnsafeURL("credentials in URL") host = parts.hostname if not host: raise UnsafeURL("no host") # An IP literal is checked directly; a name is resolved and every answer # must be public — a hostname with one public and one 127.0.0.1 record # would otherwise be a way in. try: infos = socket.getaddrinfo(host, parts.port or (443 if parts.scheme == "https" else 80), proto=socket.IPPROTO_TCP) except socket.gaierror as e: raise UnsafeURL(f"cannot resolve: {e}") resolved = {info[4][0] for info in infos} if not resolved: raise UnsafeURL("resolves to nothing") bad = [ip for ip in resolved if not _addr_is_public(ip)] if bad: raise UnsafeURL(f"non-public address {bad[0]}") return url class _HeadParser(HTMLParser): """Collects text and name/property→content from <meta> in <head>. Stops caring once <body> starts: everything a card needs is in the head, and a 512 KB page of body is not worth walking. """ def __init__(self) -> None: super().__init__(convert_charrefs=True) self.metas: dict[str, str] = {} self.title: str | None = None self._in_title = False self.done = False def handle_starttag(self, tag, attrs): if tag == "body": self.done = True elif tag == "title": self._in_title = True elif tag == "meta": a = {k.lower(): (v or "") for k, v in attrs} key = (a.get("property") or a.get("name") or "").lower().strip() if key and "content" in a and key not in self.metas: self.metas[key] = a["content"].strip() def handle_endtag(self, tag): if tag == "title": self._in_title = False def handle_data(self, data): if self._in_title and self.title is None: text = data.strip() if text: self.title = text def _first(metas: dict[str, str], *keys: str) -> str | None: for k in keys: v = metas.get(k) if v: return v return None async def _get(client: httpx.AsyncClient, url: str) -> httpx.Response: """One GET with manual, re-validated redirects.""" current = safe_url(url) for _ in range(_MAX_REDIRECTS + 1): resp = await client.get(current, headers={"User-Agent": _UA}, follow_redirects=False) if resp.is_redirect and "location" in resp.headers: current = safe_url(urljoin(current, resp.headers["location"])) continue return resp raise UnsafeURL("too many redirects") async def fetch_preview(url: str, *, client: httpx.AsyncClient | None = None) -> dict | None: """ Return {url, title, description, site_name, image_url} for a URL, or None if it cannot be unfurled (unreachable, not HTML, nothing worth showing). Never raises for an ordinary failure — the caller treats "no preview" as the common case, exactly like a TMDB miss. """ own = client is None if own: client = httpx.AsyncClient(timeout=_TIMEOUT, max_redirects=0) try: safe_url(url) resp = await _get(client, url) ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower() if resp.status_code != 200 or ctype not in ("text/html", "application/xhtml+xml"): return None body = b"" async for chunk in resp.aiter_bytes(): body += chunk if len(body) >= _MAX_HTML_BYTES: break final_url = str(resp.url) parser = _HeadParser() try: parser.feed(body.decode(resp.encoding or "utf-8", errors="replace")) except Exception: pass m = parser.metas title = _first(m, "og:title", "twitter:title") or parser.title description = _first(m, "og:description", "twitter:description", "description") site_name = _first(m, "og:site_name") or urlsplit(final_url).hostname image = _first(m, "og:image", "og:image:url", "og:image:secure_url", "twitter:image", "twitter:image:src") if image: image = urljoin(final_url, image) try: safe_url(image) except UnsafeURL: image = None if not title and not description: return None return { "url": url, "title": (title or "")[:300] or None, "description": (description or "")[:600] or None, "site_name": (site_name or "")[:120] or None, "image_url": image, } except (httpx.HTTPError, UnsafeURL) as e: log.debug("link preview for %s: %s", url[:80], e) return None finally: if own: await client.aclose() async def fetch_image(url: str, *, client: httpx.AsyncClient | None = None) -> bytes | None: """Fetch and re-encode an OG image to a small JPEG. None on any failure.""" own = client is None if own: client = httpx.AsyncClient(timeout=_TIMEOUT, max_redirects=0) try: safe_url(url) resp = await _get(client, url) ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower() if resp.status_code != 200 or not ctype.startswith("image/"): return None raw = b"" async for chunk in resp.aiter_bytes(): raw += chunk if len(raw) > _MAX_IMAGE_BYTES: return None return _downscale(raw) except (httpx.HTTPError, UnsafeURL) as e: log.debug("link preview image %s: %s", url[:80], e) return None finally: if own: await client.aclose() def _downscale(raw: bytes) -> bytes | None: try: from PIL import Image except ImportError: return None try: with Image.open(BytesIO(raw)) as im: im = im.convert("RGB") im.thumbnail((_IMAGE_MAX_DIM, _IMAGE_MAX_DIM)) out = BytesIO() im.save(out, format="JPEG", quality=80) return out.getvalue() except Exception: return None