Files
siftlode/backend/app/downloads/og.py
T
peter ae14aa1c6a fix(share): declare og:image dimensions so the FB card includes the image on first scrape
The FB Sharing Debugger confirmed the real cause of the missing desktop-Messenger preview: the
scraper fetches og:image ASYNCHRONOUSLY on a URL's first scrape ("The provided 'og:image'
properties are not yet available… specify the dimensions using 'og:image:width'/'og:image:height'"),
so a freshly shared /watch link unfurled title-only. Read the poster JPEG's real dimensions from its
SOF marker (Pillow isn't a dep; the poster aspect differs from the video's, so the stored video
width/height won't do) and emit og:image:width/height. Only runs on a crawler hit. Unit-tested.
2026-07-28 01:54:14 +02:00

150 lines
6.9 KiB
Python

"""Server-rendered Open Graph tags for the public /watch/{token} page.
The watch page is a client-side SPA, so a social crawler (facebookexternalhit, Twitterbot, …) that
fetches /watch/{token} would otherwise only see the generic index.html — a blank/greyed link card.
This injects per-video OG/Twitter tags into the served HTML so a shared link unfurls with the
video's title, channel and thumbnail. Real browsers ignore the extra tags and hydrate the SPA as
usual. Password-protected / expired / invalid links fall back to the generic card (no metadata leak).
"""
import html
import re
from pathlib import Path
from sqlalchemy import select
from sqlalchemy.orm import Session
from app.config import settings
from app.downloads import links as linksmod
from app.downloads import storage
from app.models import DownloadJob, DownloadLink, MediaAsset
_OG_BLOCK = re.compile(r"<!--OG:START-->.*?<!--OG:END-->", re.DOTALL)
def _tag(prop: str, content: str, attr: str = "property") -> str:
return f'<meta {attr}="{prop}" content="{html.escape(content, quote=True)}" />'
def _jpeg_size(path: Path) -> tuple[int, int] | None:
"""Read (width, height) from a JPEG's SOF marker without an image library (Pillow isn't a dep).
Facebook's scraper fetches og:image ASYNCHRONOUSLY on a URL's first scrape, so a card built then
has NO image unless og:image:width/height are declared up front — which is exactly why a freshly
shared /watch link unfurled title-only on desktop Messenger. We can't declare dimensions we don't
know, and the poster's aspect differs from the video's, so read them straight from the JPEG here
(cheap — only on a crawler hit). Returns None for a non-JPEG/truncated file, in which case the
dimensions are simply omitted (the prior behaviour)."""
try:
with path.open("rb") as f:
if f.read(2) != b"\xff\xd8": # SOI — not a JPEG
return None
while True:
b = f.read(1)
if not b:
return None
if b != b"\xff":
continue
marker = f.read(1)
while marker == b"\xff": # skip fill bytes
marker = f.read(1)
if not marker:
return None
m = marker[0]
if m == 0x01 or 0xD0 <= m <= 0xD9: # standalone markers, no length payload
continue
seg = f.read(2)
if len(seg) < 2:
return None
length = (seg[0] << 8) + seg[1]
# SOF0..SOF15 carry the frame dimensions (skip DHT/DAC/RST, which share the range).
if 0xC0 <= m <= 0xCF and m not in (0xC4, 0xC8, 0xCC):
data = f.read(5)
if len(data) < 5:
return None
height = (data[1] << 8) + data[2]
width = (data[3] << 8) + data[4]
return width, height
f.seek(length - 2, 1) # skip this segment's payload
except OSError:
return None
def _watch_tags(token: str, db: Session) -> str | None:
"""Build the OG/Twitter meta block for a watch token, or None to keep the generic card."""
link = db.execute(
select(DownloadLink).where(DownloadLink.token == token)
).scalar_one_or_none()
if link is None or linksmod.is_expired(link):
return None
url = f"{settings.app_base}/watch/{token}"
if link.password_hash:
# Password-gated: don't leak the title/thumbnail in the unfurl.
return "\n ".join(
[
"<title>Siftlode</title>",
_tag("og:title", "Password-protected video"),
_tag("og:description", "Shared via Siftlode"),
_tag("og:type", "video.other"),
_tag("og:url", url),
_tag("og:site_name", "Siftlode"),
]
)
job = db.get(DownloadJob, link.job_id)
asset = db.get(MediaAsset, job.asset_id) if job and job.asset_id else None
if job is None or asset is None or asset.status != "ready":
return None
title = (job.display_name or asset.title or "Video").strip()
channel = (job.display_uploader or asset.uploader or "").strip()
# Prefer our self-hosted poster for the link-preview image: a remote thumbnail (Facebook's
# signed CDN URL especially) expires and isn't reliably fetchable cross-origin by the crawler.
# (This block only runs for non-password links, so the poster needs no grant.) Fall back to the
# remote thumbnail only if we somehow have no poster.
image = (
f"{settings.app_base}/api/public/watch/{token}/poster.jpg"
if asset.poster_path
else asset.thumbnail_url
)
tags = [
f"<title>{html.escape(title)}</title>",
_tag("og:title", title),
_tag("og:description", channel or "Shared via Siftlode"),
_tag("og:type", "video.other"),
_tag("og:url", url),
_tag("og:site_name", "Siftlode"),
_tag("twitter:title", title, attr="name"),
]
if channel:
tags.append(_tag("twitter:description", channel, attr="name"))
if image:
# A real thumbnail → a large image card; without one, a plain (text) card is served.
tags.append(_tag("og:image", image))
# Facebook's scraper is pickier than Apple's on-device unfurl: spell out that the image is a
# public HTTPS JPEG, and declare its dimensions so the card includes the image on the very
# first (async) scrape instead of unfurling title-only — see _jpeg_size.
if image.startswith("https://"):
tags.append(_tag("og:image:secure_url", image))
if image.endswith("poster.jpg") and asset.poster_path:
tags.append(_tag("og:image:type", "image/jpeg"))
dims = _jpeg_size(storage.safe_abs_path(settings.download_root, asset.poster_path))
if dims:
tags.append(_tag("og:image:width", str(dims[0])))
tags.append(_tag("og:image:height", str(dims[1])))
tags.append(_tag("twitter:card", "summary_large_image", attr="name"))
tags.append(_tag("twitter:image", image, attr="name"))
else:
tags.append(_tag("twitter:card", "summary", attr="name"))
return "\n ".join(tags)
def render_watch_html(token: str, index_html: Path, db: Session) -> str | None:
"""Return index.html with the OG block replaced for this watch token, or None to serve the
generic page unchanged (invalid/expired/not-ready link, or a template without the markers)."""
tags = _watch_tags(token, db)
if tags is None:
return None
text = index_html.read_text(encoding="utf-8")
if "<!--OG:START-->" not in text:
return None
# A function replacement avoids re.sub interpreting backslashes/group refs in `tags`.
return _OG_BLOCK.sub(lambda _m: tags, text, count=1)