v3: reworked posting — three panels, drafts, duplicate detection, upload progress
All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 42s
All checks were successful
Build and Publish Docker Image / build-and-push (push) Successful in 42s
Posting a dump was a single form where the important choices were the easiest to miss. It is now three panels: link or file, why & where, playlists. Composition: - No more URL/File toggle. An empty panel offers both ways in at once and the kind follows what you actually did; a file dropped anywhere in the modal is accepted, not just on the zone. - Categories and visibility get their own panel instead of a disclosure that read as optional, and the primary button stays "Next" until they've been seen. Visibility carries a real label now. - The draft (link, title, why, categories, visibility) is mirrored to localStorage on every change and restored on reopen, so Escape or a stray backdrop click costs nothing. Only an attached file can't be restored, so that is the one case that asks before closing. - URL dumps can carry a poster-supplied title instead of being stuck with whatever the page scraped, editable right under the preview. - Multipart uploads go through XHR so there is a real progress bar and a percentage on the button, rather than 50 MB of silence. Duplicates: - New dumps.url_canonical column (+ index, backfilled by 0013) holding a lossy key that ignores scheme, www., trailing slashes, tracking parameters and YouTube share shapes. GET /api/dumps/by-url reads it, and the create form warns "already dumped by X" while it fetches the preview. Never blocking. Fixes: - /api/preview now reports whether the page was actually reached: a failed fetch still yields a hostname-only stub, so a dead link and a page without metadata used to render identically. - The Web Share Target never worked. The manifest posts to "/", but the index redirect dropped the query string, so every Android share landed on the feed with nothing pre-filled. - File dumps no longer take the extension into their title. - The link field no longer autofocuses on touch, where it raised a keyboard over the modal. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TiAPtJZeCYYk8rKehtLUQU
This commit is contained in:
@@ -11,6 +11,7 @@ import { up as up0009ChatReply } from "./migrations/0009_chat_reply.ts";
|
||||
import { up as up0010YoutubeEmbedStart } from "./migrations/0010_youtube_embed_start.ts";
|
||||
import { up as up0011SplitFaviconThumbnail } from "./migrations/0011_split_favicon_thumbnail.ts";
|
||||
import { up as up0012FixFaviconReclassification } from "./migrations/0012_fix_favicon_reclassification.ts";
|
||||
import { up as up0013DumpUrlCanonical } from "./migrations/0013_dump_url_canonical.ts";
|
||||
|
||||
interface Migration {
|
||||
name: string;
|
||||
@@ -36,6 +37,7 @@ const MIGRATIONS: Migration[] = [
|
||||
name: "0012_fix_favicon_reclassification",
|
||||
up: up0012FixFaviconReclassification,
|
||||
},
|
||||
{ name: "0013_dump_url_canonical", up: up0013DumpUrlCanonical },
|
||||
];
|
||||
|
||||
export function runMigrations(db: DatabaseSync): void {
|
||||
|
||||
119
api/db/migrations/0013_dump_url_canonical.ts
Normal file
119
api/db/migrations/0013_dump_url_canonical.ts
Normal file
@@ -0,0 +1,119 @@
|
||||
import type { DatabaseSync } from "node:sqlite";
|
||||
|
||||
// Adds `dumps.url_canonical` — the lookup key behind the "already dumped?"
|
||||
// check on the create form — and backfills it for every existing URL dump.
|
||||
//
|
||||
// Purely local: it re-derives the key from the URL already stored on each row
|
||||
// and makes no network calls.
|
||||
//
|
||||
// The helpers below are a deliberate frozen copy of `api/lib/canonical-url.ts`
|
||||
// as it stood when this migration shipped, not an import of it. A migration
|
||||
// runs exactly once per database, so importing the live version would mean two
|
||||
// databases migrating at different times end up with keys computed under
|
||||
// different rules. Improving the shared canonicalizer therefore calls for a new
|
||||
// backfill migration rather than an edit here.
|
||||
//
|
||||
// Idempotent: the column is only added when missing and only rows whose key is
|
||||
// still NULL are touched, so a fresh database built from schema.sql is a no-op.
|
||||
|
||||
const TRACKING_PARAMS = new Set([
|
||||
"fbclid",
|
||||
"gclid",
|
||||
"dclid",
|
||||
"msclkid",
|
||||
"twclid",
|
||||
"yclid",
|
||||
"mc_cid",
|
||||
"mc_eid",
|
||||
"igshid",
|
||||
"igsh",
|
||||
"si",
|
||||
"spm",
|
||||
"ref_src",
|
||||
"ref_url",
|
||||
"_ga",
|
||||
"_gl",
|
||||
"__twitter_impression",
|
||||
]);
|
||||
|
||||
function isTrackingParam(key: string): boolean {
|
||||
const k = key.toLowerCase();
|
||||
return k.startsWith("utm_") || TRACKING_PARAMS.has(k);
|
||||
}
|
||||
|
||||
const YOUTUBE_HOSTS = new Set([
|
||||
"youtube.com",
|
||||
"m.youtube.com",
|
||||
"music.youtube.com",
|
||||
"youtube-nocookie.com",
|
||||
]);
|
||||
|
||||
function youtubeVideoId(
|
||||
host: string,
|
||||
pathname: string,
|
||||
params: URLSearchParams,
|
||||
): string | null {
|
||||
if (host === "youtu.be") return pathname.split("/")[1] || null;
|
||||
if (!YOUTUBE_HOSTS.has(host)) return null;
|
||||
if (pathname === "/watch") return params.get("v");
|
||||
if (/^\/(embed|shorts|live)\//.test(pathname)) {
|
||||
return pathname.split("/")[2] || null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function canonicalizeUrl(raw: string): string | null {
|
||||
let u: URL;
|
||||
try {
|
||||
u = new URL(raw);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (u.protocol !== "http:" && u.protocol !== "https:") return null;
|
||||
|
||||
const host = u.hostname.toLowerCase().replace(/^www\./, "");
|
||||
if (!host) return null;
|
||||
|
||||
const videoId = youtubeVideoId(host, u.pathname, u.searchParams);
|
||||
if (videoId) return `https://youtube.com/watch?v=${videoId}`;
|
||||
|
||||
const listId = YOUTUBE_HOSTS.has(host) && u.pathname === "/playlist"
|
||||
? u.searchParams.get("list")
|
||||
: null;
|
||||
if (listId) return `https://youtube.com/playlist?list=${listId}`;
|
||||
|
||||
const path = u.pathname.replace(/\/+$/, "");
|
||||
|
||||
const params = [...u.searchParams.entries()]
|
||||
.filter(([key]) => !isTrackingParam(key))
|
||||
.sort(([a, av], [b, bv]) => a.localeCompare(b) || av.localeCompare(bv));
|
||||
const query = new URLSearchParams(params).toString();
|
||||
|
||||
const hash = /^#!?\//.test(u.hash) ? u.hash : "";
|
||||
|
||||
return `https://${host}${path}${query ? `?${query}` : ""}${hash}`;
|
||||
}
|
||||
|
||||
export function up(db: DatabaseSync): void {
|
||||
const columns = db.prepare(`PRAGMA table_info(dumps);`).all() as {
|
||||
name: string;
|
||||
}[];
|
||||
if (!columns.some((c) => c.name === "url_canonical")) {
|
||||
db.exec(`ALTER TABLE dumps ADD COLUMN url_canonical TEXT;`);
|
||||
}
|
||||
db.exec(
|
||||
`CREATE INDEX IF NOT EXISTS idx_dumps_url_canonical ON dumps(url_canonical);`,
|
||||
);
|
||||
|
||||
const rows = db.prepare(
|
||||
`SELECT id, url FROM dumps WHERE url IS NOT NULL AND url_canonical IS NULL;`,
|
||||
).all() as { id: string; url: string }[];
|
||||
|
||||
const update = db.prepare(
|
||||
`UPDATE dumps SET url_canonical = ? WHERE id = ?;`,
|
||||
);
|
||||
for (const row of rows) {
|
||||
const canonical = canonicalizeUrl(row.url);
|
||||
if (canonical) update.run(canonical, row.id);
|
||||
}
|
||||
}
|
||||
@@ -7,6 +7,8 @@ CREATE TABLE dumps (
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT,
|
||||
url TEXT,
|
||||
-- Lossy lookup key for "has this been dumped already?" — see api/lib/canonical-url.ts
|
||||
url_canonical TEXT,
|
||||
slug TEXT,
|
||||
rich_content TEXT,
|
||||
file_name TEXT,
|
||||
@@ -117,6 +119,7 @@ CREATE TABLE dump_backlinks (
|
||||
|
||||
CREATE INDEX idx_dumps_user ON dumps(user_id);
|
||||
CREATE INDEX idx_dumps_url ON dumps(url);
|
||||
CREATE INDEX idx_dumps_url_canonical ON dumps(url_canonical);
|
||||
CREATE INDEX idx_votes_user ON votes(user_id);
|
||||
CREATE INDEX idx_playlists_user ON playlists(user_id);
|
||||
CREATE INDEX idx_playlist_dumps_order ON playlist_dumps(playlist_id, position);
|
||||
|
||||
104
api/lib/canonical-url.ts
Normal file
104
api/lib/canonical-url.ts
Normal file
@@ -0,0 +1,104 @@
|
||||
/**
|
||||
* Canonical form of a URL, used *only* to recognise that two links point at the
|
||||
* same thing — the duplicate check the create form runs while it fetches a
|
||||
* preview. It is a lookup key, never something we display or fetch: `dumps.url`
|
||||
* keeps the exact string the poster submitted.
|
||||
*
|
||||
* Because it is only ever compared against other canonical forms, it may be
|
||||
* lossy in ways a real URL never could be — it forces `https`, drops `www.`,
|
||||
* and throws away share/tracking parameters and timestamps, so
|
||||
* `http://www.example.com/a/?utm_source=x` and `https://example.com/a` collapse
|
||||
* to one key.
|
||||
*
|
||||
* Stored in `dumps.url_canonical`. Migration 0013 carries a frozen copy of this
|
||||
* logic to backfill existing rows; changing the rules here therefore needs a
|
||||
* new backfill migration, or old rows keep keys computed under the old rules.
|
||||
*/
|
||||
|
||||
const TRACKING_PARAMS = new Set([
|
||||
"fbclid",
|
||||
"gclid",
|
||||
"dclid",
|
||||
"msclkid",
|
||||
"twclid",
|
||||
"yclid",
|
||||
"mc_cid",
|
||||
"mc_eid",
|
||||
"igshid",
|
||||
"igsh",
|
||||
"si",
|
||||
"spm",
|
||||
"ref_src",
|
||||
"ref_url",
|
||||
"_ga",
|
||||
"_gl",
|
||||
"__twitter_impression",
|
||||
]);
|
||||
|
||||
function isTrackingParam(key: string): boolean {
|
||||
const k = key.toLowerCase();
|
||||
return k.startsWith("utm_") || TRACKING_PARAMS.has(k);
|
||||
}
|
||||
|
||||
// Hosts already stripped of a leading "www.".
|
||||
const YOUTUBE_HOSTS = new Set([
|
||||
"youtube.com",
|
||||
"m.youtube.com",
|
||||
"music.youtube.com",
|
||||
"youtube-nocookie.com",
|
||||
]);
|
||||
|
||||
/**
|
||||
* The video a YouTube URL points at, in any of the shapes people paste
|
||||
* (`youtu.be/ID`, `/watch?v=ID`, `/embed/ID`, `/shorts/ID`, `/live/ID`).
|
||||
* Timestamps are deliberately ignored: the same video linked at 2:30 is still
|
||||
* the same video for duplicate purposes.
|
||||
*/
|
||||
function youtubeVideoId(
|
||||
host: string,
|
||||
pathname: string,
|
||||
params: URLSearchParams,
|
||||
): string | null {
|
||||
if (host === "youtu.be") return pathname.split("/")[1] || null;
|
||||
if (!YOUTUBE_HOSTS.has(host)) return null;
|
||||
if (pathname === "/watch") return params.get("v");
|
||||
if (/^\/(embed|shorts|live)\//.test(pathname)) {
|
||||
return pathname.split("/")[2] || null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
export function canonicalizeUrl(raw: string): string | null {
|
||||
let u: URL;
|
||||
try {
|
||||
u = new URL(raw);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (u.protocol !== "http:" && u.protocol !== "https:") return null;
|
||||
|
||||
const host = u.hostname.toLowerCase().replace(/^www\./, "");
|
||||
if (!host) return null;
|
||||
|
||||
const videoId = youtubeVideoId(host, u.pathname, u.searchParams);
|
||||
if (videoId) return `https://youtube.com/watch?v=${videoId}`;
|
||||
|
||||
const listId = YOUTUBE_HOSTS.has(host) && u.pathname === "/playlist"
|
||||
? u.searchParams.get("list")
|
||||
: null;
|
||||
if (listId) return `https://youtube.com/playlist?list=${listId}`;
|
||||
|
||||
// A single trailing slash is never meaningful; "/" itself becomes "".
|
||||
const path = u.pathname.replace(/\/+$/, "");
|
||||
|
||||
const params = [...u.searchParams.entries()]
|
||||
.filter(([key]) => !isTrackingParam(key))
|
||||
.sort(([a, av], [b, bv]) => a.localeCompare(b) || av.localeCompare(bv));
|
||||
const query = new URLSearchParams(params).toString();
|
||||
|
||||
// A bare "#section" anchor points into the same page, so it is dropped —
|
||||
// but "#/route" and "#!/route" address distinct pages of a hash-routed app.
|
||||
const hash = /^#!?\//.test(u.hash) ? u.hash : "";
|
||||
|
||||
return `https://${host}${path}${query ? `?${query}` : ""}${hash}`;
|
||||
}
|
||||
@@ -479,11 +479,27 @@ function isStringArray(value: unknown): value is string[] {
|
||||
|
||||
export interface CreateUrlDumpRequest {
|
||||
url: string;
|
||||
/** Overrides the title scraped from the page — the poster edited the preview. */
|
||||
title?: string;
|
||||
comment?: string;
|
||||
isPrivate?: boolean;
|
||||
categoryIds?: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* An existing dump pointing at the same URL, as shown by the create form's
|
||||
* duplicate hint. Display-only projection — see `findDumpsByUrl`.
|
||||
*/
|
||||
export interface DumpUrlMatch {
|
||||
id: string;
|
||||
slug?: string;
|
||||
title: string;
|
||||
username: string;
|
||||
createdAt: Date;
|
||||
voteCount: number;
|
||||
commentCount: number;
|
||||
}
|
||||
|
||||
export function isCreateUrlDumpRequest(
|
||||
obj: unknown,
|
||||
): obj is CreateUrlDumpRequest {
|
||||
@@ -499,6 +515,13 @@ export function isCreateUrlDumpRequest(
|
||||
typeof o.comment === "string" &&
|
||||
(o.comment as string).length > VALIDATION.DUMP_COMMENT_MAX
|
||||
) return false;
|
||||
if ("title" in o && typeof o.title !== "string" && o.title !== null) {
|
||||
return false;
|
||||
}
|
||||
if (
|
||||
typeof o.title === "string" &&
|
||||
(o.title as string).length > VALIDATION.DUMP_TITLE_MAX
|
||||
) return false;
|
||||
if ("isPrivate" in o && typeof o.isPrivate !== "boolean") return false;
|
||||
if ("categoryIds" in o && !isStringArray(o.categoryIds)) return false;
|
||||
return true;
|
||||
|
||||
@@ -5,6 +5,7 @@ import {
|
||||
APIException,
|
||||
type APIResponse,
|
||||
type Dump,
|
||||
type DumpUrlMatch,
|
||||
isCreateUrlDumpRequest,
|
||||
isUpdateDumpRequest,
|
||||
type PaginatedData,
|
||||
@@ -21,6 +22,7 @@ import {
|
||||
createFileDump,
|
||||
createUrlDump,
|
||||
deleteDump,
|
||||
findDumpsByUrl,
|
||||
getDump,
|
||||
listDumps,
|
||||
refreshDumpMetadata,
|
||||
@@ -100,6 +102,17 @@ router.post(
|
||||
},
|
||||
);
|
||||
|
||||
// Registered ahead of "/:dumpId" so the literal path wins over the parameter.
|
||||
router.get("/by-url", async (ctx) => {
|
||||
const requestingUserId = await parseOptionalAuth(ctx) ?? undefined;
|
||||
const url = ctx.request.url.searchParams.get("url") ?? "";
|
||||
const responseBody: APIResponse<DumpUrlMatch[]> = {
|
||||
success: true,
|
||||
data: findDumpsByUrl(url, requestingUserId),
|
||||
};
|
||||
ctx.response.body = responseBody;
|
||||
});
|
||||
|
||||
router.get("/:dumpId", async (ctx) => {
|
||||
const requestingUserId = await parseOptionalAuth(ctx) ?? undefined;
|
||||
const dump = getDump(ctx.params.dumpId, requestingUserId);
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { Router } from "@oak/oak";
|
||||
import {
|
||||
fetchRichContent,
|
||||
fetchWithTimeout,
|
||||
isValidHttpUrl,
|
||||
tryFetchRichContent,
|
||||
} from "../services/rich-content-service.ts";
|
||||
import { APIErrorCode } from "../model/interfaces.ts";
|
||||
|
||||
@@ -18,8 +18,15 @@ previewRouter.get("/api/preview", async (ctx) => {
|
||||
};
|
||||
return;
|
||||
}
|
||||
const data = await fetchRichContent(url);
|
||||
ctx.response.body = { success: true, data: data ?? null };
|
||||
// `reached` is reported separately because a failed fetch still yields a
|
||||
// usable stub (hostname only). Without the flag the create form cannot tell
|
||||
// "this page has no preview" from "this link is dead", and shows the same
|
||||
// bare card for both.
|
||||
const { ok, content } = await tryFetchRichContent(url);
|
||||
ctx.response.body = {
|
||||
success: true,
|
||||
data: { reached: ok, richContent: content ?? null },
|
||||
};
|
||||
});
|
||||
|
||||
/**
|
||||
|
||||
@@ -3,6 +3,7 @@ import {
|
||||
APIException,
|
||||
type CreateUrlDumpRequest,
|
||||
type Dump,
|
||||
type DumpUrlMatch,
|
||||
type UpdateDumpRequest,
|
||||
} from "../model/interfaces.ts";
|
||||
import {
|
||||
@@ -28,6 +29,7 @@ import {
|
||||
notifyUserFollowersNewDump,
|
||||
} from "./notification-service.ts";
|
||||
import { makeSlug, UUID_RE } from "../lib/slugify.ts";
|
||||
import { canonicalizeUrl } from "../lib/canonical-url.ts";
|
||||
import {
|
||||
DUMP_ALLOWED_MIME_PREFIXES,
|
||||
DUMP_ALLOWED_MIME_TYPES,
|
||||
@@ -63,13 +65,16 @@ export async function createUrlDump(
|
||||
const dumpId = crypto.randomUUID();
|
||||
const createdAt = new Date();
|
||||
const richContent = await fetchRichContent(request.url);
|
||||
const title = richContent?.title ?? titleFromUrl(request.url);
|
||||
// A title typed on the create form wins over the one scraped from the page:
|
||||
// the poster has seen the preview and is correcting it.
|
||||
const title = request.title?.trim() || richContent?.title ||
|
||||
titleFromUrl(request.url);
|
||||
const isPrivate = request.isPrivate ?? false;
|
||||
const slug = makeSlug(title, dumpId);
|
||||
|
||||
db.prepare(
|
||||
`INSERT INTO dumps (id, kind, title, slug, comment, user_id, created_at, url, rich_content, is_private)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?);`,
|
||||
`INSERT INTO dumps (id, kind, title, slug, comment, user_id, created_at, url, url_canonical, rich_content, is_private)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?);`,
|
||||
).run(
|
||||
dumpId,
|
||||
"url",
|
||||
@@ -79,6 +84,7 @@ export async function createUrlDump(
|
||||
userId,
|
||||
createdAt.toISOString(),
|
||||
request.url,
|
||||
canonicalizeUrl(request.url),
|
||||
richContent ? JSON.stringify(richContent) : null,
|
||||
isPrivate ? 1 : 0,
|
||||
);
|
||||
@@ -294,6 +300,54 @@ export function listDumps(
|
||||
return { items, total: totalRow?.count ?? 0 };
|
||||
}
|
||||
|
||||
/**
|
||||
* Dumps that already point at the same thing as `url`, newest first.
|
||||
*
|
||||
* Matched on the canonical key rather than the raw URL, so a link reposted with
|
||||
* different tracking parameters, scheme, `www.`, or (for YouTube) in a
|
||||
* different share shape still counts as the same dump. Private dumps are
|
||||
* visible only to their owner, exactly as everywhere else.
|
||||
*
|
||||
* Returns a small display-only projection: this feeds a hint on the create
|
||||
* form, not a feed.
|
||||
*/
|
||||
export function findDumpsByUrl(
|
||||
url: string,
|
||||
requestingUserId?: string,
|
||||
limit = 3,
|
||||
): DumpUrlMatch[] {
|
||||
const canonical = canonicalizeUrl(url);
|
||||
if (!canonical) return [];
|
||||
|
||||
const rows = db.prepare(
|
||||
`SELECT d.id, d.slug, d.title, d.created_at, d.vote_count, u.username,
|
||||
(SELECT COUNT(*) FROM comments WHERE dump_id = d.id AND deleted = 0) as comment_count
|
||||
FROM dumps d
|
||||
JOIN users u ON u.id = d.user_id
|
||||
WHERE d.url_canonical = ? AND (d.is_private = 0 OR d.user_id = ?)
|
||||
ORDER BY d.created_at DESC
|
||||
LIMIT ?;`,
|
||||
).all(canonical, requestingUserId ?? null, limit) as {
|
||||
id: string;
|
||||
slug: string | null;
|
||||
title: string;
|
||||
created_at: string;
|
||||
vote_count: number;
|
||||
username: string;
|
||||
comment_count: number;
|
||||
}[];
|
||||
|
||||
return rows.map((row) => ({
|
||||
id: row.id,
|
||||
slug: row.slug ?? undefined,
|
||||
title: row.title,
|
||||
username: row.username,
|
||||
createdAt: new Date(row.created_at),
|
||||
voteCount: row.vote_count,
|
||||
commentCount: row.comment_count,
|
||||
}));
|
||||
}
|
||||
|
||||
export async function updateDump(
|
||||
dumpId: string,
|
||||
request: UpdateDumpRequest,
|
||||
@@ -387,12 +441,13 @@ export async function updateDump(
|
||||
|
||||
const row = dumpApiToRow(updatedDump);
|
||||
const result = db.prepare(
|
||||
`UPDATE dumps SET title = ?, slug = ?, comment = ?, url = ?, rich_content = ?, is_private = ?, updated_at = ? WHERE id = ?;`,
|
||||
`UPDATE dumps SET title = ?, slug = ?, comment = ?, url = ?, url_canonical = ?, rich_content = ?, is_private = ?, updated_at = ? WHERE id = ?;`,
|
||||
).run(
|
||||
row.title,
|
||||
row.slug,
|
||||
row.comment,
|
||||
row.url,
|
||||
row.url ? canonicalizeUrl(row.url) : null,
|
||||
row.rich_content,
|
||||
row.is_private,
|
||||
now.toISOString(),
|
||||
|
||||
Reference in New Issue
Block a user