import type { RichContent } from "../../model/interfaces.ts"; import type { RichContentProvider } from "../rich-content-service.ts"; import { extractBestIcon, extractFirstContentImage, extractJsonLd, extractLargeImage, extractMaskIconColor, extractMetaName, extractOgTag, extractPageTitle, extractThemeColor, fetchWithTimeout, normalizeCssColor, } from "../rich-content-service.ts"; export const genericProvider: RichContentProvider = { name: "generic", matches(_url: string): boolean { return true; // fallback — always matches }, async fetch(url: string): Promise { const res = await fetchWithTimeout(url); const contentType = res.headers.get("content-type") ?? ""; if (!contentType.startsWith("text/html")) { return { type: "generic", url }; } const html = await res.text(); const ld = extractJsonLd(html); // If og:url is present but points to a different page (e.g. the homepage), // the og: block is a site-level fallback, not page-specific metadata. // In that case skip og:title and og:image so page-level signals win. const ogUrl = extractOgTag(html, "url"); const useOg = !ogUrl || (() => { try { const ogPath = new URL(ogUrl).pathname.replace(/\/+$/, "") || "/"; const pagePath = new URL(url).pathname.replace(/\/+$/, "") || "/"; return ogPath === pagePath; } catch { return true; } })(); // Title: og:title (page-matched) → twitter:title → JSON-LD → const title = (useOg ? extractOgTag(html, "title") : undefined) ?? extractMetaName(html, "twitter:title") ?? ld.title ?? extractPageTitle(html); // Site name: og:site_name → hostname const siteName = extractOgTag(html, "site_name") ?? new URL(url).hostname.replace(/^www\./, ""); // Description: og:description → twitter:description → JSON-LD → <meta name="description"> const description = extractOgTag(html, "description") ?? extractMetaName(html, "twitter:description") ?? ld.description ?? extractMetaName(html, "description"); // Image: og:image (page-matched) → twitter:image → JSON-LD → large <img> → // first content <img>. The chain deliberately stops there: a favicon is not // artwork, and pretending otherwise means every art-less page gets a 16×16 // icon cover-cropped into a 128×72 box. No match here means "no artwork", // and the frontend draws a generated placeholder instead. const thumbnailUrl = (useOg ? extractOgTag(html, "image") : undefined) ?? extractMetaName(html, "twitter:image") ?? ld.thumbnailUrl ?? extractLargeImage(html, url) ?? extractFirstContentImage(html, url); // Icon and brand color are independent facts about the page, so they're // collected whether or not there's artwork. `/favicon.ico` is a guess; when // it 404s the placeholder falls back to the site's initial. const faviconUrl = extractBestIcon(html, url) ?? `${new URL(url).origin}/favicon.ico`; const accentColor = normalizeCssColor( extractThemeColor(html) ?? extractMetaName(html, "msapplication-TileColor") ?? extractMaskIconColor(html), ); return { type: "generic", url, title, description, thumbnailUrl, faviconUrl, accentColor, siteName, }; }, };