All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 47s
Thumbnails were failing in two different ways that both ended as an empty box. Cloudflare hotlink protection answers a cross-site Referer with 403, so images we had extracted correctly (dles.aukspot.com's og:image among them) never rendered — every onError handler set display:none and swallowed it. The new Thumbnail component loads with referrerPolicy="no-referrer", retries once through /api/proxy-image for hosts that reject an empty referrer too, and only then falls back to a placeholder. Separately, the extraction cascade ended at the page's icon and then a guessed /favicon.ico, so thumbnailUrl was almost never empty — just a 16x16 icon cover-cropped into a 128x72 box. It now stops at real artwork, with faviconUrl and accentColor (theme-color / msapplication-TileColor / mask-icon) as their own fields. An absent thumbnailUrl finally means "no artwork", which is what makes the placeholder possible: the site's own color mixed into the theme surface, with its favicon centered on it, or its initial. Contrast holds for any third-party color by construction rather than by luminance math, so nyt only has to set --thumb-tint-strength to 0% to stay monochrome and geocities only has to raise it. Missing accents fall back to a stable hostname-derived hue, so rows saved before this get a tint with no backfill. Migration 0011 reclassifies favicon-shaped thumbnailUrls on existing dumps. The journal mosaic keeps its pull-quote and text fallbacks — the placeholder appears there only to repair a broken image. Also fixed: refresh silently overwrote good metadata with a failure stub, the refresh button swallowed every error, refresh never broadcast the update, extractBestIcon ranked SVG icons below 16x16 PNGs, and shared links with no artwork carried no og:image at all. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
100 lines
3.4 KiB
TypeScript
100 lines
3.4 KiB
TypeScript
import type { RichContent } from "../../model/interfaces.ts";
|
||
import type { RichContentProvider } from "../rich-content-service.ts";
|
||
import {
|
||
extractBestIcon,
|
||
extractFirstContentImage,
|
||
extractJsonLd,
|
||
extractLargeImage,
|
||
extractMaskIconColor,
|
||
extractMetaName,
|
||
extractOgTag,
|
||
extractPageTitle,
|
||
extractThemeColor,
|
||
fetchWithTimeout,
|
||
normalizeCssColor,
|
||
} from "../rich-content-service.ts";
|
||
|
||
export const genericProvider: RichContentProvider = {
|
||
name: "generic",
|
||
|
||
matches(_url: string): boolean {
|
||
return true; // fallback — always matches
|
||
},
|
||
|
||
async fetch(url: string): Promise<RichContent> {
|
||
const res = await fetchWithTimeout(url);
|
||
const contentType = res.headers.get("content-type") ?? "";
|
||
|
||
if (!contentType.startsWith("text/html")) {
|
||
return { type: "generic", url };
|
||
}
|
||
|
||
const html = await res.text();
|
||
const ld = extractJsonLd(html);
|
||
|
||
// If og:url is present but points to a different page (e.g. the homepage),
|
||
// the og: block is a site-level fallback, not page-specific metadata.
|
||
// In that case skip og:title and og:image so page-level signals win.
|
||
const ogUrl = extractOgTag(html, "url");
|
||
const useOg = !ogUrl || (() => {
|
||
try {
|
||
const ogPath = new URL(ogUrl).pathname.replace(/\/+$/, "") || "/";
|
||
const pagePath = new URL(url).pathname.replace(/\/+$/, "") || "/";
|
||
return ogPath === pagePath;
|
||
} catch {
|
||
return true;
|
||
}
|
||
})();
|
||
|
||
// Title: og:title (page-matched) → twitter:title → JSON-LD → <title>
|
||
const title = (useOg ? extractOgTag(html, "title") : undefined) ??
|
||
extractMetaName(html, "twitter:title") ??
|
||
ld.title ??
|
||
extractPageTitle(html);
|
||
|
||
// Site name: og:site_name → hostname
|
||
const siteName = extractOgTag(html, "site_name") ??
|
||
new URL(url).hostname.replace(/^www\./, "");
|
||
|
||
// Description: og:description → twitter:description → JSON-LD → <meta name="description">
|
||
const description = extractOgTag(html, "description") ??
|
||
extractMetaName(html, "twitter:description") ??
|
||
ld.description ??
|
||
extractMetaName(html, "description");
|
||
|
||
// Image: og:image (page-matched) → twitter:image → JSON-LD → large <img> →
|
||
// first content <img>. The chain deliberately stops there: a favicon is not
|
||
// artwork, and pretending otherwise means every art-less page gets a 16×16
|
||
// icon cover-cropped into a 128×72 box. No match here means "no artwork",
|
||
// and the frontend draws a generated placeholder instead.
|
||
const thumbnailUrl = (useOg ? extractOgTag(html, "image") : undefined) ??
|
||
extractMetaName(html, "twitter:image") ??
|
||
ld.thumbnailUrl ??
|
||
extractLargeImage(html, url) ??
|
||
extractFirstContentImage(html, url);
|
||
|
||
// Icon and brand color are independent facts about the page, so they're
|
||
// collected whether or not there's artwork. `/favicon.ico` is a guess; when
|
||
// it 404s the placeholder falls back to the site's initial.
|
||
const faviconUrl = extractBestIcon(html, url) ??
|
||
`${new URL(url).origin}/favicon.ico`;
|
||
|
||
const accentColor = normalizeCssColor(
|
||
extractThemeColor(html) ??
|
||
extractMetaName(html, "msapplication-TileColor") ??
|
||
extractMaskIconColor(html),
|
||
);
|
||
|
||
return {
|
||
type: "generic",
|
||
url,
|
||
title,
|
||
description,
|
||
thumbnailUrl,
|
||
faviconUrl,
|
||
accentColor,
|
||
siteName,
|
||
};
|
||
},
|
||
};
|