mirror of
https://github.com/tiennm99/blog.git
synced 2026-10-11 03:13:10 +00:00
Translates the seven-subcommand engine to Node ESM, one module per former Go file, invoked as `node scripts/newsletter <command>` from the repo root. The Go implementation stays in place for now so parity can be measured against it. Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual string surgery in caption extraction and covers RSS and sitemap XML through xmlMode without a second parser. fetch-via-defuddle gains a local extraction stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a third IP is the whole point when this machine's IP is the blocked one. Both stages now emit the same YAML frontmatter plus body, and an extraction that looks like a bot challenge counts as a local failure so the proxy still runs. Behaviour is preserved where it is load-bearing rather than where it is merely idiomatic: query strings are rebuilt by string surgery so surviving parameters keep their original order and encoding, duplicate detection stays an index loop with byte-offset boundary checks so a prefix of a stored URL is not a false match, empty optional fields are omitted rather than emitted as "", a deep-crawl miss reports cutoff as null rather than dropping the key, bullet filtering counts code points, tag counts keep their six-column alignment, and a malformed percent sequence falls back to the raw substring. The printer lives in its own module so a command module never imports the dispatcher: index.js runs main() at module scope, so that cycle would execute the CLI as a side effect of any import. Missing dependencies report one actionable line instead of a module-resolution stack trace, and a reader that closes early exits quietly instead of raising EPIPE.
128 lines
4.0 KiB
JavaScript
128 lines
4.0 KiB
JavaScript
// Meta URL router for the mt-add-url skill — the single entry per URL.
|
|
// Usage: node scripts/newsletter add-url "<url>"
|
|
// Outputs: JSON { original_url, clean_url, http_status, accessible,
|
|
// duplicate, route, title?, author? }
|
|
//
|
|
// route ∈ youtube | image | video | document | article
|
|
|
|
import { printJson } from "./json-out.js";
|
|
import {
|
|
checkAccessibility,
|
|
checkDuplicate,
|
|
classifyType,
|
|
cleanUrl,
|
|
contentDir,
|
|
fetchTextOk,
|
|
isSubstackImage,
|
|
} from "./url-utils.js";
|
|
|
|
const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]);
|
|
|
|
/**
|
|
* detectYouTube extracts a video id from the supported URL shapes:
|
|
* youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID.
|
|
* Playlists/channels are intentionally NOT YouTube routes (fall through to type).
|
|
* @param {string} target
|
|
* @returns {{isYouTube: boolean, videoId: string}}
|
|
*/
|
|
export function detectYouTube(target) {
|
|
let u;
|
|
try {
|
|
u = new URL(target);
|
|
} catch {
|
|
return { isYouTube: false, videoId: "" };
|
|
}
|
|
const host = u.host.toLowerCase();
|
|
if (host === "youtu.be") {
|
|
const id = u.pathname.replace(/^\//, "").split("/")[0];
|
|
return { isYouTube: id !== "", videoId: id };
|
|
}
|
|
if (YT_HOSTS.has(host)) {
|
|
if (u.pathname === "/watch") {
|
|
const id = u.searchParams.get("v") ?? "";
|
|
return { isYouTube: id !== "", videoId: id };
|
|
}
|
|
if (u.pathname.startsWith("/shorts/")) {
|
|
const parts = u.pathname.split("/");
|
|
if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] };
|
|
}
|
|
}
|
|
return { isYouTube: false, videoId: "" };
|
|
}
|
|
|
|
/**
|
|
* canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes.
|
|
* @param {string} videoId
|
|
* @returns {string}
|
|
*/
|
|
function canonicalWatchUrl(videoId) {
|
|
return "https://www.youtube.com/watch?v=" + videoId;
|
|
}
|
|
|
|
/**
|
|
* fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key).
|
|
* Best-effort: any failure returns empty strings so the route stays `youtube`
|
|
* and the skill can fall back.
|
|
* @param {string} watchUrl
|
|
* @returns {Promise<{title: string, author: string}>}
|
|
*/
|
|
async function fetchYouTubeMeta(watchUrl) {
|
|
const endpoint =
|
|
"https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json";
|
|
const body = await fetchTextOk(endpoint, 10_000);
|
|
if (body === "") return { title: "", author: "" };
|
|
try {
|
|
const data = JSON.parse(body);
|
|
return { title: data.title ?? "", author: data.author_name ?? "" };
|
|
} catch {
|
|
return { title: "", author: "" };
|
|
}
|
|
}
|
|
|
|
/**
|
|
* @param {string[]} args
|
|
* @returns {Promise<void>}
|
|
*/
|
|
export async function runAddUrl(args) {
|
|
if (args.length < 1 || args[0] === "") {
|
|
process.stderr.write("Usage: node scripts/newsletter add-url <url>\n");
|
|
process.exit(1);
|
|
}
|
|
const target = args[0];
|
|
|
|
const cleaned = cleanUrl(target);
|
|
const { isYouTube, videoId } = detectYouTube(cleaned);
|
|
|
|
// For YouTube, dedup/store against the canonical watch URL so youtu.be and
|
|
// shorts links collapse onto the same identity-param key as watch URLs.
|
|
const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned;
|
|
|
|
// Route order: YouTube → Substack image (by host, not extension, so f_auto /
|
|
// .avif / .heic / extensionless CDN URLs still route to the image handler) →
|
|
// file-extension classification.
|
|
let route;
|
|
if (isYouTube) route = "youtube";
|
|
else if (isSubstackImage(cleaned)) route = "image";
|
|
else route = classifyType(cleaned);
|
|
|
|
const httpStatus = await checkAccessibility(cleaned);
|
|
|
|
// title/author are omitted entirely when empty, not emitted as "": the
|
|
// handlers treat a present key as "metadata was resolved".
|
|
/** @type {Record<string, unknown>} */
|
|
const out = {
|
|
original_url: target,
|
|
clean_url: effectiveUrl,
|
|
http_status: httpStatus,
|
|
accessible: httpStatus === "200",
|
|
duplicate: checkDuplicate(effectiveUrl, contentDir()),
|
|
route,
|
|
};
|
|
if (route === "youtube") {
|
|
const { title, author } = await fetchYouTubeMeta(effectiveUrl);
|
|
if (title !== "") out.title = title;
|
|
if (author !== "") out.author = author;
|
|
}
|
|
printJson(out);
|
|
}
|