"""SSRF-hardened link unfurling — fetch a URL server-side and extract an OG/meta preview. Dependency-free (stdlib only), matching the project's hand-rolled ethos. The server fetching arbitrary user-supplied URLs is a classic SSRF surface, so the defenses are deliberate and layered: - http/https only (no file://, gopher://, …). - Resolve the host and require EVERY resolved address to be public — reject private / loopback / link-local / reserved / multicast / unspecified ranges. - Connect to the exact vetted IP (with SNI = the hostname), so a name that re-resolves to an internal address between check and connect (DNS rebinding) can't slip through. - At most 3 redirects, each hop re-validated the same way. - 5s timeout, 512 KB body cap, text/html only. No AI — this just parses OpenGraph/Twitter/`` meta tags. """ from __future__ import annotations import asyncio import http.client import ipaddress import re import socket import ssl from html import unescape from urllib.parse import urljoin, urlparse MAX_REDIRECTS = 3 TIMEOUT_S = 5.0 MAX_BYTES = 512 * 1024 USER_AGENT = "ThoughtSync-LinkPreview/1.0" _REDIRECT_CODES = {301, 302, 303, 307, 308} class UnfurlError(Exception): """A URL could not be safely unfurled (bad scheme, blocked address, fetch error).""" def is_public_ip(ip: ipaddress.IPv4Address | ipaddress.IPv6Address) -> bool: """True only for globally-routable addresses — everything internal is rejected.""" return not ( ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_multicast or ip.is_reserved or ip.is_unspecified ) def validate_url(raw: str) -> tuple[str, str, int, str]: """Parse a safe http(s) URL → (scheme, host, port, path+query). Raise otherwise.""" parsed = urlparse((raw or "").strip()) if parsed.scheme not in ("http", "https"): raise UnfurlError("only http and https links can be previewed") host = parsed.hostname if not host: raise UnfurlError("that link has no host") port = parsed.port or (443 if parsed.scheme == "https" else 80) path = parsed.path or "/" if parsed.query: path = f"{path}?{parsed.query}" return parsed.scheme, host, port, path def _resolve_public(host: str, port: int) -> str: """Resolve host; require ALL resolved addresses to be public. Return one vetted IP.""" try: infos = socket.getaddrinfo(host, port, proto=socket.IPPROTO_TCP) except socket.gaierror as e: raise UnfurlError("could not resolve that host") from e chosen: str | None = None for info in infos: addr = info[4][0] try: ip = ipaddress.ip_address(addr.split("%")[0]) # strip any IPv6 zone id except ValueError as e: raise UnfurlError("could not resolve that host") from e if not is_public_ip(ip): raise UnfurlError("that address isn't allowed") if chosen is None: chosen = addr if chosen is None: raise UnfurlError("could not resolve that host") return chosen def _fetch_once(scheme: str, host: str, port: int, path: str) -> tuple[int, str | None, bytes]: """One blocking GET to the VETTED public IP for `host`. Returns (status, location, body). Reads at most MAX_BYTES of a text/html body.""" ip = _resolve_public(host, port) sock: socket.socket = socket.create_connection((ip, port), timeout=TIMEOUT_S) try: if scheme == "https": ctx = ssl.create_default_context() sock = ctx.wrap_socket(sock, server_hostname=host) # SNI + cert check vs host sock.settimeout(TIMEOUT_S) # ensure reads can't hang after the TLS wrap conn = http.client.HTTPConnection(host, port, timeout=TIMEOUT_S) conn.sock = sock # use our pre-vetted (and, for https, wrapped) socket conn.request( "GET", path, headers={ "User-Agent": USER_AGENT, "Accept": "text/html,application/xhtml+xml", "Accept-Encoding": "identity", "Connection": "close", }, ) resp = conn.getresponse() headers = {k.lower(): v for k, v in resp.getheaders()} if resp.status in _REDIRECT_CODES: return resp.status, headers.get("location"), b"" ctype = headers.get("content-type", "").split(";")[0].strip().lower() if ctype and ctype not in ("text/html", "application/xhtml+xml"): raise UnfurlError("that link isn't a web page") return resp.status, None, resp.read(MAX_BYTES) finally: try: sock.close() except OSError: pass def _fetch(url: str) -> tuple[str, bytes]: """Follow up to MAX_REDIRECTS, re-validating each hop. Returns (final_url, html).""" current = url for _ in range(MAX_REDIRECTS + 1): scheme, host, port, path = validate_url(current) try: status, location, body = _fetch_once(scheme, host, port, path) except UnfurlError: raise except (OSError, ssl.SSLError, http.client.HTTPException) as e: raise UnfurlError("could not fetch that link") from e if status in _REDIRECT_CODES and location: current = urljoin(current, location) continue if status >= 400: raise UnfurlError(f"the site returned an error ({status})") return current, body raise UnfurlError("too many redirects") def _meta_content(html: str, key: str) -> str | None: """The content of a <meta property|name="key" content="…"> tag (either attr order).""" k = re.escape(key) for pat in ( rf'<meta[^>]+(?:property|name)=["\']{k}["\'][^>]*content=["\']([^"\']*)["\']', rf'<meta[^>]+content=["\']([^"\']*)["\'][^>]*(?:property|name)=["\']{k}["\']', ): m = re.search(pat, html, re.IGNORECASE | re.DOTALL) if m: val = unescape(m.group(1)).strip() if val: return val return None def extract_preview(final_url: str, body: bytes) -> dict: """Pull a link preview (title/description/image/site) from HTML. OpenGraph first, then Twitter cards, then <title> / bare <meta name=description>.""" html = body.decode("utf-8", errors="replace") title = _meta_content(html, "og:title") or _meta_content(html, "twitter:title") if not title: tm = re.search(r"<title[^>]*>(.*?)", html, re.IGNORECASE | re.DOTALL) if tm: title = unescape(re.sub(r"\s+", " ", tm.group(1)).strip()) or None description = ( _meta_content(html, "og:description") or _meta_content(html, "twitter:description") or _meta_content(html, "description") ) image = _meta_content(html, "og:image") or _meta_content(html, "twitter:image") if image: image = urljoin(final_url, image) # resolve a relative image path if urlparse(image).scheme not in ("http", "https"): image = None site_name = _meta_content(html, "og:site_name") host = urlparse(final_url).hostname or "" return { "url": final_url, "title": (title or host)[:300], "description": description[:500] if description else None, "image_url": image[:1000] if image else None, "site_name": (site_name or host)[:100] or None, } async def unfurl(url: str) -> dict: """Fetch `url` server-side (SSRF-guarded) and return a link preview. The blocking socket IO runs in a worker thread so it never stalls the event loop. Raises UnfurlError on any failure.""" final_url, body = await asyncio.to_thread(_fetch, url) return extract_preview(final_url, body)