v3: generated placeholder thumbnails for pages with no preview image, and a fix for hotlink-protected ones
All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 47s
All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 47s
Thumbnails were failing in two different ways that both ended as an empty box. Cloudflare hotlink protection answers a cross-site Referer with 403, so images we had extracted correctly (dles.aukspot.com's og:image among them) never rendered — every onError handler set display:none and swallowed it. The new Thumbnail component loads with referrerPolicy="no-referrer", retries once through /api/proxy-image for hosts that reject an empty referrer too, and only then falls back to a placeholder. Separately, the extraction cascade ended at the page's icon and then a guessed /favicon.ico, so thumbnailUrl was almost never empty — just a 16x16 icon cover-cropped into a 128x72 box. It now stops at real artwork, with faviconUrl and accentColor (theme-color / msapplication-TileColor / mask-icon) as their own fields. An absent thumbnailUrl finally means "no artwork", which is what makes the placeholder possible: the site's own color mixed into the theme surface, with its favicon centered on it, or its initial. Contrast holds for any third-party color by construction rather than by luminance math, so nyt only has to set --thumb-tint-strength to 0% to stay monochrome and geocities only has to raise it. Missing accents fall back to a stable hostname-derived hue, so rows saved before this get a tint with no backfill. Migration 0011 reclassifies favicon-shaped thumbnailUrls on existing dumps. The journal mosaic keeps its pull-quote and text fallbacks — the placeholder appears there only to repair a broken image. Also fixed: refresh silently overwrote good metadata with a failure stub, the refresh button swallowed every error, refresh never broadcast the update, extractBestIcon ranked SVG icons below 16x16 PNGs, and shared links with no artwork carried no og:image at all. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -184,6 +184,117 @@ export function extractPageTitle(html: string): string | undefined {
|
||||
return match ? decodeHtmlEntities(match[1].trim()) : undefined;
|
||||
}
|
||||
|
||||
// ── Brand color helpers ───────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Find the page's `theme-color`.
|
||||
*
|
||||
* Sites routinely ship the tag twice, scoped by `media`, and often list the
|
||||
* dark-scheme one first — so unlike `extractMetaName` this can't just take the
|
||||
* first match. Prefer the unscoped tag, then the light-scheme one, then any.
|
||||
*/
|
||||
export function extractThemeColor(html: string): string | undefined {
|
||||
const tagRe = /<meta\b[^>]*>/gi;
|
||||
const contentRe = /\bcontent=(["'])([\s\S]*?)\1/i;
|
||||
const mediaRe = /\bmedia=(["'])([\s\S]*?)\1/i;
|
||||
let unscoped: string | undefined;
|
||||
let light: string | undefined;
|
||||
let any: string | undefined;
|
||||
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = tagRe.exec(html)) !== null) {
|
||||
const tag = m[0];
|
||||
if (!/\bname=["']theme-color["']/i.test(tag)) continue;
|
||||
const content = contentRe.exec(tag)?.[2];
|
||||
if (!content) continue;
|
||||
const media = mediaRe.exec(tag)?.[2];
|
||||
if (!media) unscoped ??= content;
|
||||
else if (/light/i.test(media)) light ??= content;
|
||||
any ??= content;
|
||||
}
|
||||
return unscoped ?? light ?? any;
|
||||
}
|
||||
|
||||
/** Extract the `color` of `<link rel="mask-icon">` (Safari pinned tabs). */
|
||||
export function extractMaskIconColor(html: string): string | undefined {
|
||||
const linkRe = /<link[^>]+>/gi;
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = linkRe.exec(html)) !== null) {
|
||||
const tag = m[0];
|
||||
if (!/\brel=["'][^"']*mask-icon[^"']*["']/i.test(tag)) continue;
|
||||
const color = /\bcolor=(["'])([\s\S]*?)\1/i.exec(tag)?.[2];
|
||||
if (color) return color;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// The handful of named colors that actually show up in `theme-color`. The full
|
||||
// 148-name table isn't worth carrying for the long tail.
|
||||
const NAMED_COLORS: Record<string, string> = {
|
||||
black: "#000000",
|
||||
white: "#ffffff",
|
||||
red: "#ff0000",
|
||||
green: "#008000",
|
||||
blue: "#0000ff",
|
||||
yellow: "#ffff00",
|
||||
orange: "#ffa500",
|
||||
purple: "#800080",
|
||||
gray: "#808080",
|
||||
grey: "#808080",
|
||||
silver: "#c0c0c0",
|
||||
maroon: "#800000",
|
||||
navy: "#000080",
|
||||
teal: "#008080",
|
||||
olive: "#808000",
|
||||
lime: "#00ff00",
|
||||
aqua: "#00ffff",
|
||||
cyan: "#00ffff",
|
||||
fuchsia: "#ff00ff",
|
||||
magenta: "#ff00ff",
|
||||
};
|
||||
|
||||
function clampByte(n: number): number {
|
||||
return Math.max(0, Math.min(255, Math.round(n)));
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a CSS color to canonical lowercase `#rrggbb`, or `undefined` when
|
||||
* it isn't one of the forms sites actually use. Alpha is dropped — the color is
|
||||
* only ever used as a tint over the app's own surface.
|
||||
*/
|
||||
export function normalizeCssColor(raw: string | undefined): string | undefined {
|
||||
if (!raw) return undefined;
|
||||
const value = raw.trim().toLowerCase();
|
||||
|
||||
const named = NAMED_COLORS[value];
|
||||
if (named) return named;
|
||||
|
||||
const hex = /^#([0-9a-f]{3,8})$/.exec(value)?.[1];
|
||||
if (hex) {
|
||||
if (hex.length === 3 || hex.length === 4) {
|
||||
const [r, g, b] = [...hex.slice(0, 3)];
|
||||
return `#${r}${r}${g}${g}${b}${b}`;
|
||||
}
|
||||
if (hex.length === 6 || hex.length === 8) return `#${hex.slice(0, 6)}`;
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const rgb = /^rgba?\(([^)]+)\)$/.exec(value)?.[1];
|
||||
if (rgb) {
|
||||
const parts = rgb.split(/[\s,/]+/).filter(Boolean).slice(0, 3);
|
||||
if (parts.length !== 3) return undefined;
|
||||
const bytes = parts.map((p) => {
|
||||
const n = parseFloat(p);
|
||||
if (!Number.isFinite(n)) return NaN;
|
||||
return clampByte(p.endsWith("%") ? (n / 100) * 255 : n);
|
||||
});
|
||||
if (bytes.some(Number.isNaN)) return undefined;
|
||||
return `#${bytes.map((b) => b.toString(16).padStart(2, "0")).join("")}`;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// ── JSON-LD helpers (file-private) ────────────────────────────────────────────
|
||||
|
||||
type JsonLdResult = {
|
||||
@@ -281,6 +392,10 @@ export function extractLargeImage(
|
||||
* Collect all `<link rel="icon">` / `<link rel="apple-touch-icon">` tags, rank
|
||||
* them by declared size (largest wins), and return the best resolved URL.
|
||||
* Falls back to the first match when no `sizes` attribute is present.
|
||||
*
|
||||
* SVG icons are ranked above everything: they carry no `sizes` attribute, so
|
||||
* they'd otherwise score 0 and lose to a 16×16 PNG, yet they scale cleanly to
|
||||
* whatever size the placeholder renders them at.
|
||||
*/
|
||||
export function extractBestIcon(
|
||||
html: string,
|
||||
@@ -302,7 +417,13 @@ export function extractBestIcon(
|
||||
if (!href) continue;
|
||||
const sizesStr = sizesRe.exec(tag)?.[1] ?? "";
|
||||
const sm = sizesStr.match(/(\d+)x(\d+)/i);
|
||||
const area = sm ? parseInt(sm[1]) * parseInt(sm[2]) : 0;
|
||||
const isSvg = /\.svg(\?|$)/i.test(href) ||
|
||||
/\btype=["']image\/svg\+xml["']/i.test(tag);
|
||||
const area = isSvg
|
||||
? Number.MAX_SAFE_INTEGER
|
||||
: sm
|
||||
? parseInt(sm[1]) * parseInt(sm[2])
|
||||
: 0;
|
||||
try {
|
||||
candidates.push({ href: new URL(href, baseUrl).toString(), area });
|
||||
} catch {
|
||||
@@ -431,24 +552,46 @@ export function isValidHttpUrl(raw: string): boolean {
|
||||
}
|
||||
}
|
||||
|
||||
export async function fetchRichContent(
|
||||
export interface FetchRichContentResult {
|
||||
/** False when the page couldn't be reached at all, as opposed to reached and
|
||||
* found to carry no metadata. Callers that already hold good metadata must
|
||||
* not overwrite it with a failure stub. */
|
||||
ok: boolean;
|
||||
content?: RichContent;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch metadata for `url`, reporting whether the fetch itself succeeded.
|
||||
*
|
||||
* On failure it still yields a minimal stub so a *new* dump has something
|
||||
* displayable, but `ok: false` lets callers with existing metadata keep it.
|
||||
*/
|
||||
export async function tryFetchRichContent(
|
||||
url: string,
|
||||
): Promise<RichContent | undefined> {
|
||||
): Promise<FetchRichContentResult> {
|
||||
try {
|
||||
const provider = providers.find((p) => p.matches(url))!;
|
||||
return await provider.fetch(url);
|
||||
return { ok: true, content: await provider.fetch(url) };
|
||||
} catch (err) {
|
||||
console.error(`[rich-content] Failed to fetch metadata for ${url}:`, err);
|
||||
// Return a minimal stub so the caller always gets something displayable
|
||||
// (e.g. when the site has a bad TLS cert or the fetch times out).
|
||||
try {
|
||||
return {
|
||||
type: "generic",
|
||||
url,
|
||||
siteName: new URL(url).hostname.replace(/^www\./, ""),
|
||||
ok: false,
|
||||
content: {
|
||||
type: "generic",
|
||||
url,
|
||||
siteName: new URL(url).hostname.replace(/^www\./, ""),
|
||||
},
|
||||
};
|
||||
} catch {
|
||||
return undefined;
|
||||
return { ok: false };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Metadata for `url`, or a minimal stub when it can't be fetched. */
|
||||
export async function fetchRichContent(
|
||||
url: string,
|
||||
): Promise<RichContent | undefined> {
|
||||
return (await tryFetchRichContent(url)).content;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user