""" Link unfurling for chat — fetch a URL a member pasted and pull an OpenGraph-style card out of it (title, description, site name, image). Who does the fetch matters. It is **the node**, not the browser and not the hub: * the browser cannot — a strict `img-src`/`connect-src` and CORS block it, and a direct fetch would leak every reader's IP to the linked host on every render; * the hub must not — it never touches group content (draft-v6 §2.5); * the node already fetches third-party metadata for the Videos and Music apps (`_fetch_and_cache_poster`), over the same authorised path. Because the node makes an outbound request to an address a *member* chose, this is an SSRF surface. `check_url()` is the gate: http(s) only, no credentials, the port restricted to the web set, and every resolved address must be globally routable — no loopback, private, link-local, multicast or reserved range. Redirects are followed by hand so every hop is re-checked. **The connection goes to the address that was checked** (`_PinnedBackend`): the name is resolved once, off the event loop, every answer is checked, and the socket is opened to that IP literal — TLS still verifies the certificate for the name. Resolving to check and letting the HTTP client resolve again to connect would let a name answer clean the first time and with a LAN address the second (DNS rebinding), and the request would be sent before anything could look. How many previews a member can trigger is rate-limited by the caller (`_do_link_preview_request`). Nothing is stored durably: the caller keeps an in-memory TTL cache and the OG image rides the existing `media_cache` thumb store (same as a poster). """ from __future__ import annotations import asyncio import ipaddress import logging import socket from html.parser import HTMLParser from io import BytesIO from urllib.parse import urljoin, urlsplit import httpcore import httpx log = logging.getLogger(__name__) _TIMEOUT = 5.0 # The whole fetch, redirects and body included. `_TIMEOUT` is per network # operation, so a server that keeps sending slowly never trips it on its own. _TOTAL_DEADLINE = 15.0 _MAX_REDIRECTS = 3 _MAX_HTML_BYTES = 512 * 1024 _MAX_IMAGE_BYTES = 2 * 1024 * 1024 _MAX_IMAGE_PIXELS = 40_000_000 # ~40 MP; an OG card image is a fraction of this _IMAGE_MAX_DIM = 600 _UA = "MeshBayBot/1.0 (+https://meshbay.org; link preview)" # Ports a real OpenGraph-bearing page is served on. Everything else — SSH, mail, # databases, caches, search, admin panels — is refused, so a member cannot aim # the node at an arbitrary service even on a public host. _ALLOWED_PORTS = frozenset({80, 443, 8080, 8443}) class UnsafeURL(ValueError): """The URL points somewhere the node must not fetch from.""" def _addr_is_public(ip: str) -> bool: try: addr = ipaddress.ip_address(ip) except ValueError: return False if isinstance(addr, ipaddress.IPv6Address) and addr.ipv4_mapped is not None: addr = addr.ipv4_mapped return not ( addr.is_private or addr.is_loopback or addr.is_link_local or addr.is_multicast or addr.is_reserved or addr.is_unspecified ) def safe_url(url: str) -> str: """ The URL unchanged if its shape is safe to fetch, else raise UnsafeURL. Shape only — scheme, credentials, port, and the address when it is a literal. A name is checked by `check_url` and, again, at connect time; resolving here would block the event loop on a member's choice of name. """ if not isinstance(url, str) or len(url) > 2048: raise UnsafeURL("missing or oversized") parts = urlsplit(url) if parts.scheme not in ("http", "https"): raise UnsafeURL(f"scheme {parts.scheme!r}") if parts.username or parts.password: raise UnsafeURL("credentials in URL") host = parts.hostname if not host: raise UnsafeURL("no host") try: port = parts.port except ValueError: raise UnsafeURL("bad port") if port is not None and port not in _ALLOWED_PORTS: raise UnsafeURL(f"port {port}") if _is_literal(host) and not _addr_is_public(host): raise UnsafeURL(f"non-public address {host}") return url def _is_literal(host: str) -> bool: try: ipaddress.ip_address(host) return True except ValueError: return False _RESOLVE_TIMEOUT = 5.0 async def resolve_public(host: str, port: int) -> str: """ One public address for `host`, resolved off the event loop, or UnsafeURL. Every answer must be public: a name with one public and one 127.0.0.1 record would otherwise be a way in. """ if _is_literal(host): if not _addr_is_public(host): raise UnsafeURL(f"non-public address {host}") return host loop = asyncio.get_running_loop() try: infos = await asyncio.wait_for( loop.getaddrinfo(host, port, proto=socket.IPPROTO_TCP), _RESOLVE_TIMEOUT) except (socket.gaierror, TimeoutError) as e: raise UnsafeURL(f"cannot resolve: {e}") resolved = list(dict.fromkeys(info[4][0] for info in infos)) if not resolved: raise UnsafeURL("resolves to nothing") bad = [ip for ip in resolved if not _addr_is_public(ip)] if bad: raise UnsafeURL(f"non-public address {bad[0]}") return resolved[0] async def check_url(url: str) -> str: """`safe_url`, and the name's addresses checked too. The URL unchanged.""" safe_url(url) parts = urlsplit(url) await resolve_public(parts.hostname or "", parts.port or (443 if parts.scheme == "https" else 80)) return url class _PinnedBackend(httpcore.AsyncNetworkBackend): """ Opens every connection to an address `resolve_public` checked. httpcore hands the backend the request's host; the TLS layer above still uses that name for SNI and certificate verification, so pinning the socket changes where it connects and nothing about whom it trusts. """ def __init__(self) -> None: self._inner = httpcore.AnyIOBackend() async def connect_tcp(self, host, port, timeout=None, local_address=None, socket_options=None): ip = await resolve_public(host, port) return await self._inner.connect_tcp(ip, port, timeout=timeout, local_address=local_address, socket_options=socket_options) async def connect_unix_socket(self, path, timeout=None, socket_options=None): raise UnsafeURL("no unix sockets") async def sleep(self, seconds: float) -> None: await self._inner.sleep(seconds) class _PinnedTransport(httpx.AsyncHTTPTransport): """httpx's transport over a pool that connects through `_PinnedBackend`. No proxy from the environment (`trust_env=False`): a proxy would resolve the name itself, and the pin would bind nothing. """ def __init__(self) -> None: super().__init__(trust_env=False, retries=0) self._pool = httpcore.AsyncConnectionPool( ssl_context=httpx.create_ssl_context(trust_env=False), network_backend=_PinnedBackend(), max_connections=10) def _new_client() -> httpx.AsyncClient: return httpx.AsyncClient(transport=_PinnedTransport(), timeout=_TIMEOUT, max_redirects=0, trust_env=False) class _HeadParser(HTMLParser): """Collects text and name/property→content from <meta> in <head>. Stops caring once <body> starts: everything a card needs is in the head, and a 512 KB page of body is not worth walking. """ def __init__(self) -> None: super().__init__(convert_charrefs=True) self.metas: dict[str, str] = {} self.title: str | None = None self._in_title = False self.done = False def handle_starttag(self, tag, attrs): if tag == "body": self.done = True elif tag == "title": self._in_title = True elif tag == "meta": a = {k.lower(): (v or "") for k, v in attrs} key = (a.get("property") or a.get("name") or "").lower().strip() if key and "content" in a and key not in self.metas: self.metas[key] = a["content"].strip() def handle_endtag(self, tag): if tag == "title": self._in_title = False def handle_data(self, data): if self._in_title and self.title is None: text = data.strip() if text: self.title = text def _first(metas: dict[str, str], *keys: str) -> str | None: for k in keys: v = metas.get(k) if v: return v return None async def _get(client: httpx.AsyncClient, url: str) -> httpx.Response: """ One GET with manual, re-validated redirects, **body not read**. The caller reads it through `_read_capped` and must close it. `client.get` is not usable here: it reads and decodes the whole body before returning, so the size caps applied afterwards bounded nothing — a page, an image or a compressed stream of any size was held in memory first. The caps are the only thing between a URL a member pasted and the node's memory. """ current = await check_url(url) for _ in range(_MAX_REDIRECTS + 1): request = client.build_request("GET", current, headers={"User-Agent": _UA}) resp = await client.send(request, stream=True, follow_redirects=False) if resp.is_redirect and "location" in resp.headers: location = resp.headers["location"] await resp.aclose() current = await check_url(urljoin(current, location)) continue return resp raise UnsafeURL("too many redirects") async def _read_capped(resp: httpx.Response, cap: int) -> tuple[bytes, bool]: """ At most `cap` bytes of the *decoded* body, and whether there was more. Counted after decoding, so a small compressed response that inflates to gigabytes stops at the cap like any other. A declared length over the cap is not read at all. """ try: declared = int(resp.headers.get("content-length", "")) except ValueError: declared = -1 if resp.headers.get("content-encoding", "identity") == "identity" \ and declared > cap: return b"", True body = bytearray() async for chunk in resp.aiter_bytes(): body += chunk if len(body) > cap: return bytes(body[:cap]), True return bytes(body), False async def fetch_preview(url: str, *, client: httpx.AsyncClient | None = None) -> dict | None: """ Return {url, title, description, site_name, image_url} for a URL, or None if it cannot be unfurled (unreachable, not HTML, nothing worth showing). Never raises for an ordinary failure — the caller treats "no preview" as the common case, exactly like a TMDB miss. """ own = client is None if own: client = _new_client() try: return await asyncio.wait_for(_preview(client, url), _TOTAL_DEADLINE) except (httpx.HTTPError, UnsafeURL, TimeoutError) as e: log.debug("link preview for %s: %s", url[:80], e) return None finally: if own: await client.aclose() async def _preview(client: httpx.AsyncClient, url: str) -> dict | None: resp = await _get(client, url) try: ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower() if resp.status_code != 200 or ctype not in ("text/html", "application/xhtml+xml"): return None # A page longer than this is cut, not refused: what a card needs is in # the head. body, _truncated = await _read_capped(resp, _MAX_HTML_BYTES) final_url = str(resp.url) parser = _HeadParser() try: parser.feed(body.decode(resp.encoding or "utf-8", errors="replace")) except Exception: pass m = parser.metas title = _first(m, "og:title", "twitter:title") or parser.title description = _first(m, "og:description", "twitter:description", "description") site_name = _first(m, "og:site_name") or urlsplit(final_url).hostname image = _first(m, "og:image", "og:image:url", "og:image:secure_url", "twitter:image", "twitter:image:src") if image: image = urljoin(final_url, image) try: await check_url(image) except UnsafeURL: image = None if not title and not description: return None return { "url": url, "title": (title or "")[:300] or None, "description": (description or "")[:600] or None, "site_name": (site_name or "")[:120] or None, "image_url": image, } finally: await resp.aclose() async def fetch_image(url: str, *, client: httpx.AsyncClient | None = None) -> bytes | None: """Fetch and re-encode an OG image to a small JPEG. None on any failure.""" own = client is None if own: client = _new_client() try: raw = await asyncio.wait_for(_image_bytes(client, url), _TOTAL_DEADLINE) except (httpx.HTTPError, UnsafeURL, TimeoutError) as e: log.debug("link preview image %s: %s", url[:80], e) return None finally: if own: await client.aclose() if raw is None: return None # Decoding an image is CPU work on untrusted bytes; not on the event loop. return await asyncio.to_thread(_downscale, raw) async def _image_bytes(client: httpx.AsyncClient, url: str) -> bytes | None: resp = await _get(client, url) try: ctype = resp.headers.get("content-type", "").split(";")[0].strip().lower() if resp.status_code != 200 or not ctype.startswith("image/"): return None raw, too_big = await _read_capped(resp, _MAX_IMAGE_BYTES) return None if too_big else raw finally: await resp.aclose() def _downscale(raw: bytes) -> bytes | None: try: from PIL import Image except ImportError: return None try: with Image.open(BytesIO(raw)) as im: # The header is parsed but the pixels are not decoded yet — refuse a # decompression bomb before convert()/thumbnail() allocate for it. if im.width * im.height > _MAX_IMAGE_PIXELS: return None im = im.convert("RGB") im.thumbnail((_IMAGE_MAX_DIM, _IMAGE_MAX_DIM)) out = BytesIO() im.save(out, format="JPEG", quality=80) return out.getvalue() except Exception: return None