Files
inkwell/src/thoughtsync/unfurl.py
T
bvandeusenandClaude Opus 4.8 69bf04e948
CI & Build / Python lint (push) Successful in 3s
CI & Build / Python tests (push) Successful in 12s
CI & Build / TypeScript typecheck (push) Successful in 34s
CI & Build / Build & push image (push) Successful in 33s
M6 1901: URL capture with link-preview unfurl (SSRF-hardened)
Paste a link → fetch its OpenGraph/meta preview (title, description, image,
site) and show a rich card. User-triggered + persisted (never auto-fetches;
cached so it never re-fetches). Opt-in via a new admin setting
enable_url_unfurl (default on, rule 26).

Security (the whole point of this task): a new dependency-free unfurl.py
does the fetch with layered SSRF defenses — http/https only; resolve the
host and reject EVERY non-public address (private/loopback/link-local/
reserved/multicast/unspecified — blocks 169.254.169.254 etc.); connect to
the vetted IP with SNI so DNS-rebinding can't slip through; ≤3 redirects
each re-validated; 5s timeout; 512 KB cap; text/html only; blocking IO in a
worker thread. No server-side image fetch — the og:image URL is loaded by
the browser.

- note_link_previews table (migration 0020), one per (note, url); serialized
  inline on notes (+ rides the sync pull feed read-only).
- POST /api/notes/<id>/unfurl {url} (owner-scoped, setting-gated, 502 on
  fetch failure); DELETE /api/notes/<id>/previews/<id>.
- enable_url_unfurl exposed in public config so the UI hides the affordance
  when disabled.

Frontend: LinkPreview.vue card; editor detects URLs in the body and offers a
"Preview <domain>" chip per un-previewed link (ensureDraft first), renders
preview cards with remove; card shows previews read-only. New link icon;
notes-store unfurl()/deletePreview().

Tests (DB-free): is_public_ip range blocking, validate_url scheme/parts,
extract_preview (OG + <title> fallback + relative-image resolve), endpoint
auth-guards.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01FRgehjoz7Yv8LkUfADxACm
2026-07-23 08:05:24 -04:00

195 lines
7.6 KiB
Python

"""SSRF-hardened link unfurling — fetch a URL server-side and extract an OG/meta
preview. Dependency-free (stdlib only), matching the project's hand-rolled ethos.
The server fetching arbitrary user-supplied URLs is a classic SSRF surface, so the
defenses are deliberate and layered:
- http/https only (no file://, gopher://, …).
- Resolve the host and require EVERY resolved address to be public — reject
private / loopback / link-local / reserved / multicast / unspecified ranges.
- Connect to the exact vetted IP (with SNI = the hostname), so a name that
re-resolves to an internal address between check and connect (DNS rebinding)
can't slip through.
- At most 3 redirects, each hop re-validated the same way.
- 5s timeout, 512 KB body cap, text/html only.
No AI — this just parses OpenGraph/Twitter/`<title>` meta tags.
"""
from __future__ import annotations
import asyncio
import http.client
import ipaddress
import re
import socket
import ssl
from html import unescape
from urllib.parse import urljoin, urlparse
MAX_REDIRECTS = 3
TIMEOUT_S = 5.0
MAX_BYTES = 512 * 1024
USER_AGENT = "ThoughtSync-LinkPreview/1.0"
_REDIRECT_CODES = {301, 302, 303, 307, 308}
class UnfurlError(Exception):
"""A URL could not be safely unfurled (bad scheme, blocked address, fetch error)."""
def is_public_ip(ip: ipaddress.IPv4Address | ipaddress.IPv6Address) -> bool:
"""True only for globally-routable addresses — everything internal is rejected."""
return not (
ip.is_private
or ip.is_loopback
or ip.is_link_local
or ip.is_multicast
or ip.is_reserved
or ip.is_unspecified
)
def validate_url(raw: str) -> tuple[str, str, int, str]:
"""Parse a safe http(s) URL → (scheme, host, port, path+query). Raise otherwise."""
parsed = urlparse((raw or "").strip())
if parsed.scheme not in ("http", "https"):
raise UnfurlError("only http and https links can be previewed")
host = parsed.hostname
if not host:
raise UnfurlError("that link has no host")
port = parsed.port or (443 if parsed.scheme == "https" else 80)
path = parsed.path or "/"
if parsed.query:
path = f"{path}?{parsed.query}"
return parsed.scheme, host, port, path
def _resolve_public(host: str, port: int) -> str:
"""Resolve host; require ALL resolved addresses to be public. Return one vetted IP."""
try:
infos = socket.getaddrinfo(host, port, proto=socket.IPPROTO_TCP)
except socket.gaierror as e:
raise UnfurlError("could not resolve that host") from e
chosen: str | None = None
for info in infos:
addr = info[4][0]
try:
ip = ipaddress.ip_address(addr.split("%")[0]) # strip any IPv6 zone id
except ValueError as e:
raise UnfurlError("could not resolve that host") from e
if not is_public_ip(ip):
raise UnfurlError("that address isn't allowed")
if chosen is None:
chosen = addr
if chosen is None:
raise UnfurlError("could not resolve that host")
return chosen
def _fetch_once(scheme: str, host: str, port: int, path: str) -> tuple[int, str | None, bytes]:
"""One blocking GET to the VETTED public IP for `host`. Returns (status, location,
body). Reads at most MAX_BYTES of a text/html body."""
ip = _resolve_public(host, port)
sock: socket.socket = socket.create_connection((ip, port), timeout=TIMEOUT_S)
try:
if scheme == "https":
ctx = ssl.create_default_context()
sock = ctx.wrap_socket(sock, server_hostname=host) # SNI + cert check vs host
sock.settimeout(TIMEOUT_S) # ensure reads can't hang after the TLS wrap
conn = http.client.HTTPConnection(host, port, timeout=TIMEOUT_S)
conn.sock = sock # use our pre-vetted (and, for https, wrapped) socket
conn.request(
"GET",
path,
headers={
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml",
"Accept-Encoding": "identity",
"Connection": "close",
},
)
resp = conn.getresponse()
headers = {k.lower(): v for k, v in resp.getheaders()}
if resp.status in _REDIRECT_CODES:
return resp.status, headers.get("location"), b""
ctype = headers.get("content-type", "").split(";")[0].strip().lower()
if ctype and ctype not in ("text/html", "application/xhtml+xml"):
raise UnfurlError("that link isn't a web page")
return resp.status, None, resp.read(MAX_BYTES)
finally:
try:
sock.close()
except OSError:
pass
def _fetch(url: str) -> tuple[str, bytes]:
"""Follow up to MAX_REDIRECTS, re-validating each hop. Returns (final_url, html)."""
current = url
for _ in range(MAX_REDIRECTS + 1):
scheme, host, port, path = validate_url(current)
try:
status, location, body = _fetch_once(scheme, host, port, path)
except UnfurlError:
raise
except (OSError, ssl.SSLError, http.client.HTTPException) as e:
raise UnfurlError("could not fetch that link") from e
if status in _REDIRECT_CODES and location:
current = urljoin(current, location)
continue
if status >= 400:
raise UnfurlError(f"the site returned an error ({status})")
return current, body
raise UnfurlError("too many redirects")
def _meta_content(html: str, key: str) -> str | None:
"""The content of a <meta property|name="key" content="…"> tag (either attr order)."""
k = re.escape(key)
for pat in (
rf'<meta[^>]+(?:property|name)=["\']{k}["\'][^>]*content=["\']([^"\']*)["\']',
rf'<meta[^>]+content=["\']([^"\']*)["\'][^>]*(?:property|name)=["\']{k}["\']',
):
m = re.search(pat, html, re.IGNORECASE | re.DOTALL)
if m:
val = unescape(m.group(1)).strip()
if val:
return val
return None
def extract_preview(final_url: str, body: bytes) -> dict:
"""Pull a link preview (title/description/image/site) from HTML. OpenGraph first,
then Twitter cards, then <title> / bare <meta name=description>."""
html = body.decode("utf-8", errors="replace")
title = _meta_content(html, "og:title") or _meta_content(html, "twitter:title")
if not title:
tm = re.search(r"<title[^>]*>(.*?)</title>", html, re.IGNORECASE | re.DOTALL)
if tm:
title = unescape(re.sub(r"\s+", " ", tm.group(1)).strip()) or None
description = (
_meta_content(html, "og:description")
or _meta_content(html, "twitter:description")
or _meta_content(html, "description")
)
image = _meta_content(html, "og:image") or _meta_content(html, "twitter:image")
if image:
image = urljoin(final_url, image) # resolve a relative image path
if urlparse(image).scheme not in ("http", "https"):
image = None
site_name = _meta_content(html, "og:site_name")
host = urlparse(final_url).hostname or ""
return {
"url": final_url,
"title": (title or host)[:300],
"description": description[:500] if description else None,
"image_url": image[:1000] if image else None,
"site_name": (site_name or host)[:100] or None,
}
async def unfurl(url: str) -> dict:
"""Fetch `url` server-side (SSRF-guarded) and return a link preview. The blocking
socket IO runs in a worker thread so it never stalls the event loop. Raises
UnfurlError on any failure."""
final_url, body = await asyncio.to_thread(_fetch, url)
return extract_preview(final_url, body)