v3: reworked posting — three panels, drafts, duplicate detection, upload progress
All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 42s

Posting a dump was a single form where the important choices were the easiest
to miss. It is now three panels: link or file, why & where, playlists.

Composition:
- No more URL/File toggle. An empty panel offers both ways in at once and the
  kind follows what you actually did; a file dropped anywhere in the modal is
  accepted, not just on the zone.
- Categories and visibility get their own panel instead of a disclosure that
  read as optional, and the primary button stays "Next" until they've been
  seen. Visibility carries a real label now.
- The draft (link, title, why, categories, visibility) is mirrored to
  localStorage on every change and restored on reopen, so Escape or a stray
  backdrop click costs nothing. Only an attached file can't be restored, so
  that is the one case that asks before closing.
- URL dumps can carry a poster-supplied title instead of being stuck with
  whatever the page scraped, editable right under the preview.
- Multipart uploads go through XHR so there is a real progress bar and a
  percentage on the button, rather than 50 MB of silence.

Duplicates:
- New dumps.url_canonical column (+ index, backfilled by 0013) holding a lossy
  key that ignores scheme, www., trailing slashes, tracking parameters and
  YouTube share shapes. GET /api/dumps/by-url reads it, and the create form
  warns "already dumped by X" while it fetches the preview. Never blocking.

Fixes:
- /api/preview now reports whether the page was actually reached: a failed
  fetch still yields a hostname-only stub, so a dead link and a page without
  metadata used to render identically.
- The Web Share Target never worked. The manifest posts to "/", but the index
  redirect dropped the query string, so every Android share landed on the feed
  with nothing pre-filled.
- File dumps no longer take the extension into their title.
- The link field no longer autofocuses on touch, where it raised a keyboard
  over the modal.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01TiAPtJZeCYYk8rKehtLUQU
This commit is contained in:
khannurien
2026-09-08 14:40:37 +00:00
parent d76154d15d
commit fb8364e24d
21 changed files with 1688 additions and 304 deletions

View File

@@ -11,6 +11,7 @@ import { up as up0009ChatReply } from "./migrations/0009_chat_reply.ts";
import { up as up0010YoutubeEmbedStart } from "./migrations/0010_youtube_embed_start.ts";
import { up as up0011SplitFaviconThumbnail } from "./migrations/0011_split_favicon_thumbnail.ts";
import { up as up0012FixFaviconReclassification } from "./migrations/0012_fix_favicon_reclassification.ts";
import { up as up0013DumpUrlCanonical } from "./migrations/0013_dump_url_canonical.ts";
interface Migration {
name: string;
@@ -36,6 +37,7 @@ const MIGRATIONS: Migration[] = [
name: "0012_fix_favicon_reclassification",
up: up0012FixFaviconReclassification,
},
{ name: "0013_dump_url_canonical", up: up0013DumpUrlCanonical },
];
export function runMigrations(db: DatabaseSync): void {

View File

@@ -0,0 +1,119 @@
import type { DatabaseSync } from "node:sqlite";
// Adds `dumps.url_canonical` — the lookup key behind the "already dumped?"
// check on the create form — and backfills it for every existing URL dump.
//
// Purely local: it re-derives the key from the URL already stored on each row
// and makes no network calls.
//
// The helpers below are a deliberate frozen copy of `api/lib/canonical-url.ts`
// as it stood when this migration shipped, not an import of it. A migration
// runs exactly once per database, so importing the live version would mean two
// databases migrating at different times end up with keys computed under
// different rules. Improving the shared canonicalizer therefore calls for a new
// backfill migration rather than an edit here.
//
// Idempotent: the column is only added when missing and only rows whose key is
// still NULL are touched, so a fresh database built from schema.sql is a no-op.
const TRACKING_PARAMS = new Set([
"fbclid",
"gclid",
"dclid",
"msclkid",
"twclid",
"yclid",
"mc_cid",
"mc_eid",
"igshid",
"igsh",
"si",
"spm",
"ref_src",
"ref_url",
"_ga",
"_gl",
"__twitter_impression",
]);
function isTrackingParam(key: string): boolean {
const k = key.toLowerCase();
return k.startsWith("utm_") || TRACKING_PARAMS.has(k);
}
const YOUTUBE_HOSTS = new Set([
"youtube.com",
"m.youtube.com",
"music.youtube.com",
"youtube-nocookie.com",
]);
function youtubeVideoId(
host: string,
pathname: string,
params: URLSearchParams,
): string | null {
if (host === "youtu.be") return pathname.split("/")[1] || null;
if (!YOUTUBE_HOSTS.has(host)) return null;
if (pathname === "/watch") return params.get("v");
if (/^\/(embed|shorts|live)\//.test(pathname)) {
return pathname.split("/")[2] || null;
}
return null;
}
function canonicalizeUrl(raw: string): string | null {
let u: URL;
try {
u = new URL(raw);
} catch {
return null;
}
if (u.protocol !== "http:" && u.protocol !== "https:") return null;
const host = u.hostname.toLowerCase().replace(/^www\./, "");
if (!host) return null;
const videoId = youtubeVideoId(host, u.pathname, u.searchParams);
if (videoId) return `https://youtube.com/watch?v=${videoId}`;
const listId = YOUTUBE_HOSTS.has(host) && u.pathname === "/playlist"
? u.searchParams.get("list")
: null;
if (listId) return `https://youtube.com/playlist?list=${listId}`;
const path = u.pathname.replace(/\/+$/, "");
const params = [...u.searchParams.entries()]
.filter(([key]) => !isTrackingParam(key))
.sort(([a, av], [b, bv]) => a.localeCompare(b) || av.localeCompare(bv));
const query = new URLSearchParams(params).toString();
const hash = /^#!?\//.test(u.hash) ? u.hash : "";
return `https://${host}${path}${query ? `?${query}` : ""}${hash}`;
}
export function up(db: DatabaseSync): void {
const columns = db.prepare(`PRAGMA table_info(dumps);`).all() as {
name: string;
}[];
if (!columns.some((c) => c.name === "url_canonical")) {
db.exec(`ALTER TABLE dumps ADD COLUMN url_canonical TEXT;`);
}
db.exec(
`CREATE INDEX IF NOT EXISTS idx_dumps_url_canonical ON dumps(url_canonical);`,
);
const rows = db.prepare(
`SELECT id, url FROM dumps WHERE url IS NOT NULL AND url_canonical IS NULL;`,
).all() as { id: string; url: string }[];
const update = db.prepare(
`UPDATE dumps SET url_canonical = ? WHERE id = ?;`,
);
for (const row of rows) {
const canonical = canonicalizeUrl(row.url);
if (canonical) update.run(canonical, row.id);
}
}

View File

@@ -7,6 +7,8 @@ CREATE TABLE dumps (
created_at TEXT NOT NULL,
updated_at TEXT,
url TEXT,
-- Lossy lookup key for "has this been dumped already?" — see api/lib/canonical-url.ts
url_canonical TEXT,
slug TEXT,
rich_content TEXT,
file_name TEXT,
@@ -117,6 +119,7 @@ CREATE TABLE dump_backlinks (
CREATE INDEX idx_dumps_user ON dumps(user_id);
CREATE INDEX idx_dumps_url ON dumps(url);
CREATE INDEX idx_dumps_url_canonical ON dumps(url_canonical);
CREATE INDEX idx_votes_user ON votes(user_id);
CREATE INDEX idx_playlists_user ON playlists(user_id);
CREATE INDEX idx_playlist_dumps_order ON playlist_dumps(playlist_id, position);

104
api/lib/canonical-url.ts Normal file
View File

@@ -0,0 +1,104 @@
/**
* Canonical form of a URL, used *only* to recognise that two links point at the
* same thing — the duplicate check the create form runs while it fetches a
* preview. It is a lookup key, never something we display or fetch: `dumps.url`
* keeps the exact string the poster submitted.
*
* Because it is only ever compared against other canonical forms, it may be
* lossy in ways a real URL never could be — it forces `https`, drops `www.`,
* and throws away share/tracking parameters and timestamps, so
* `http://www.example.com/a/?utm_source=x` and `https://example.com/a` collapse
* to one key.
*
* Stored in `dumps.url_canonical`. Migration 0013 carries a frozen copy of this
* logic to backfill existing rows; changing the rules here therefore needs a
* new backfill migration, or old rows keep keys computed under the old rules.
*/
const TRACKING_PARAMS = new Set([
"fbclid",
"gclid",
"dclid",
"msclkid",
"twclid",
"yclid",
"mc_cid",
"mc_eid",
"igshid",
"igsh",
"si",
"spm",
"ref_src",
"ref_url",
"_ga",
"_gl",
"__twitter_impression",
]);
function isTrackingParam(key: string): boolean {
const k = key.toLowerCase();
return k.startsWith("utm_") || TRACKING_PARAMS.has(k);
}
// Hosts already stripped of a leading "www.".
const YOUTUBE_HOSTS = new Set([
"youtube.com",
"m.youtube.com",
"music.youtube.com",
"youtube-nocookie.com",
]);
/**
* The video a YouTube URL points at, in any of the shapes people paste
* (`youtu.be/ID`, `/watch?v=ID`, `/embed/ID`, `/shorts/ID`, `/live/ID`).
* Timestamps are deliberately ignored: the same video linked at 2:30 is still
* the same video for duplicate purposes.
*/
function youtubeVideoId(
host: string,
pathname: string,
params: URLSearchParams,
): string | null {
if (host === "youtu.be") return pathname.split("/")[1] || null;
if (!YOUTUBE_HOSTS.has(host)) return null;
if (pathname === "/watch") return params.get("v");
if (/^\/(embed|shorts|live)\//.test(pathname)) {
return pathname.split("/")[2] || null;
}
return null;
}
export function canonicalizeUrl(raw: string): string | null {
let u: URL;
try {
u = new URL(raw);
} catch {
return null;
}
if (u.protocol !== "http:" && u.protocol !== "https:") return null;
const host = u.hostname.toLowerCase().replace(/^www\./, "");
if (!host) return null;
const videoId = youtubeVideoId(host, u.pathname, u.searchParams);
if (videoId) return `https://youtube.com/watch?v=${videoId}`;
const listId = YOUTUBE_HOSTS.has(host) && u.pathname === "/playlist"
? u.searchParams.get("list")
: null;
if (listId) return `https://youtube.com/playlist?list=${listId}`;
// A single trailing slash is never meaningful; "/" itself becomes "".
const path = u.pathname.replace(/\/+$/, "");
const params = [...u.searchParams.entries()]
.filter(([key]) => !isTrackingParam(key))
.sort(([a, av], [b, bv]) => a.localeCompare(b) || av.localeCompare(bv));
const query = new URLSearchParams(params).toString();
// A bare "#section" anchor points into the same page, so it is dropped —
// but "#/route" and "#!/route" address distinct pages of a hash-routed app.
const hash = /^#!?\//.test(u.hash) ? u.hash : "";
return `https://${host}${path}${query ? `?${query}` : ""}${hash}`;
}

View File

@@ -479,11 +479,27 @@ function isStringArray(value: unknown): value is string[] {
export interface CreateUrlDumpRequest {
url: string;
/** Overrides the title scraped from the page — the poster edited the preview. */
title?: string;
comment?: string;
isPrivate?: boolean;
categoryIds?: string[];
}
/**
* An existing dump pointing at the same URL, as shown by the create form's
* duplicate hint. Display-only projection — see `findDumpsByUrl`.
*/
export interface DumpUrlMatch {
id: string;
slug?: string;
title: string;
username: string;
createdAt: Date;
voteCount: number;
commentCount: number;
}
export function isCreateUrlDumpRequest(
obj: unknown,
): obj is CreateUrlDumpRequest {
@@ -499,6 +515,13 @@ export function isCreateUrlDumpRequest(
typeof o.comment === "string" &&
(o.comment as string).length > VALIDATION.DUMP_COMMENT_MAX
) return false;
if ("title" in o && typeof o.title !== "string" && o.title !== null) {
return false;
}
if (
typeof o.title === "string" &&
(o.title as string).length > VALIDATION.DUMP_TITLE_MAX
) return false;
if ("isPrivate" in o && typeof o.isPrivate !== "boolean") return false;
if ("categoryIds" in o && !isStringArray(o.categoryIds)) return false;
return true;

View File

@@ -5,6 +5,7 @@ import {
APIException,
type APIResponse,
type Dump,
type DumpUrlMatch,
isCreateUrlDumpRequest,
isUpdateDumpRequest,
type PaginatedData,
@@ -21,6 +22,7 @@ import {
createFileDump,
createUrlDump,
deleteDump,
findDumpsByUrl,
getDump,
listDumps,
refreshDumpMetadata,
@@ -100,6 +102,17 @@ router.post(
},
);
// Registered ahead of "/:dumpId" so the literal path wins over the parameter.
router.get("/by-url", async (ctx) => {
const requestingUserId = await parseOptionalAuth(ctx) ?? undefined;
const url = ctx.request.url.searchParams.get("url") ?? "";
const responseBody: APIResponse<DumpUrlMatch[]> = {
success: true,
data: findDumpsByUrl(url, requestingUserId),
};
ctx.response.body = responseBody;
});
router.get("/:dumpId", async (ctx) => {
const requestingUserId = await parseOptionalAuth(ctx) ?? undefined;
const dump = getDump(ctx.params.dumpId, requestingUserId);

View File

@@ -1,8 +1,8 @@
import { Router } from "@oak/oak";
import {
fetchRichContent,
fetchWithTimeout,
isValidHttpUrl,
tryFetchRichContent,
} from "../services/rich-content-service.ts";
import { APIErrorCode } from "../model/interfaces.ts";
@@ -18,8 +18,15 @@ previewRouter.get("/api/preview", async (ctx) => {
};
return;
}
const data = await fetchRichContent(url);
ctx.response.body = { success: true, data: data ?? null };
// `reached` is reported separately because a failed fetch still yields a
// usable stub (hostname only). Without the flag the create form cannot tell
// "this page has no preview" from "this link is dead", and shows the same
// bare card for both.
const { ok, content } = await tryFetchRichContent(url);
ctx.response.body = {
success: true,
data: { reached: ok, richContent: content ?? null },
};
});
/**

View File

@@ -3,6 +3,7 @@ import {
APIException,
type CreateUrlDumpRequest,
type Dump,
type DumpUrlMatch,
type UpdateDumpRequest,
} from "../model/interfaces.ts";
import {
@@ -28,6 +29,7 @@ import {
notifyUserFollowersNewDump,
} from "./notification-service.ts";
import { makeSlug, UUID_RE } from "../lib/slugify.ts";
import { canonicalizeUrl } from "../lib/canonical-url.ts";
import {
DUMP_ALLOWED_MIME_PREFIXES,
DUMP_ALLOWED_MIME_TYPES,
@@ -63,13 +65,16 @@ export async function createUrlDump(
const dumpId = crypto.randomUUID();
const createdAt = new Date();
const richContent = await fetchRichContent(request.url);
const title = richContent?.title ?? titleFromUrl(request.url);
// A title typed on the create form wins over the one scraped from the page:
// the poster has seen the preview and is correcting it.
const title = request.title?.trim() || richContent?.title ||
titleFromUrl(request.url);
const isPrivate = request.isPrivate ?? false;
const slug = makeSlug(title, dumpId);
db.prepare(
`INSERT INTO dumps (id, kind, title, slug, comment, user_id, created_at, url, rich_content, is_private)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?);`,
`INSERT INTO dumps (id, kind, title, slug, comment, user_id, created_at, url, url_canonical, rich_content, is_private)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?);`,
).run(
dumpId,
"url",
@@ -79,6 +84,7 @@ export async function createUrlDump(
userId,
createdAt.toISOString(),
request.url,
canonicalizeUrl(request.url),
richContent ? JSON.stringify(richContent) : null,
isPrivate ? 1 : 0,
);
@@ -294,6 +300,54 @@ export function listDumps(
return { items, total: totalRow?.count ?? 0 };
}
/**
* Dumps that already point at the same thing as `url`, newest first.
*
* Matched on the canonical key rather than the raw URL, so a link reposted with
* different tracking parameters, scheme, `www.`, or (for YouTube) in a
* different share shape still counts as the same dump. Private dumps are
* visible only to their owner, exactly as everywhere else.
*
* Returns a small display-only projection: this feeds a hint on the create
* form, not a feed.
*/
export function findDumpsByUrl(
url: string,
requestingUserId?: string,
limit = 3,
): DumpUrlMatch[] {
const canonical = canonicalizeUrl(url);
if (!canonical) return [];
const rows = db.prepare(
`SELECT d.id, d.slug, d.title, d.created_at, d.vote_count, u.username,
(SELECT COUNT(*) FROM comments WHERE dump_id = d.id AND deleted = 0) as comment_count
FROM dumps d
JOIN users u ON u.id = d.user_id
WHERE d.url_canonical = ? AND (d.is_private = 0 OR d.user_id = ?)
ORDER BY d.created_at DESC
LIMIT ?;`,
).all(canonical, requestingUserId ?? null, limit) as {
id: string;
slug: string | null;
title: string;
created_at: string;
vote_count: number;
username: string;
comment_count: number;
}[];
return rows.map((row) => ({
id: row.id,
slug: row.slug ?? undefined,
title: row.title,
username: row.username,
createdAt: new Date(row.created_at),
voteCount: row.vote_count,
commentCount: row.comment_count,
}));
}
export async function updateDump(
dumpId: string,
request: UpdateDumpRequest,
@@ -387,12 +441,13 @@ export async function updateDump(
const row = dumpApiToRow(updatedDump);
const result = db.prepare(
`UPDATE dumps SET title = ?, slug = ?, comment = ?, url = ?, rich_content = ?, is_private = ?, updated_at = ? WHERE id = ?;`,
`UPDATE dumps SET title = ?, slug = ?, comment = ?, url = ?, url_canonical = ?, rich_content = ?, is_private = ?, updated_at = ? WHERE id = ?;`,
).run(
row.title,
row.slug,
row.comment,
row.url,
row.url ? canonicalizeUrl(row.url) : null,
row.rich_content,
row.is_private,
now.toISOString(),