Files
tiennm99 64c7b76b4e feat(newsletter): port the engine to JavaScript
Translates the seven-subcommand engine to Node ESM, one module per former Go
file, invoked as `node scripts/newsletter <command>` from the repo root. The Go
implementation stays in place for now so parity can be measured against it.

Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual
string surgery in caption extraction and covers RSS and sitemap XML through
xmlMode without a second parser. fetch-via-defuddle gains a local extraction
stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a
third IP is the whole point when this machine's IP is the blocked one. Both
stages now emit the same YAML frontmatter plus body, and an extraction that
looks like a bot challenge counts as a local failure so the proxy still runs.

Behaviour is preserved where it is load-bearing rather than where it is merely
idiomatic: query strings are rebuilt by string surgery so surviving parameters
keep their original order and encoding, duplicate detection stays an index loop
with byte-offset boundary checks so a prefix of a stored URL is not a false
match, empty optional fields are omitted rather than emitted as "", a deep-crawl
miss reports cutoff as null rather than dropping the key, bullet filtering counts
code points, tag counts keep their six-column alignment, and a malformed percent
sequence falls back to the raw substring.

The printer lives in its own module so a command module never imports the
dispatcher: index.js runs main() at module scope, so that cycle would execute
the CLI as a side effect of any import. Missing dependencies report one
actionable line instead of a module-resolution stack trace, and a reader that
closes early exits quietly instead of raising EPIPE.
2026-09-18 16:23:58 +07:00

128 lines
4.0 KiB
JavaScript

// Meta URL router for the mt-add-url skill — the single entry per URL.
// Usage: node scripts/newsletter add-url "<url>"
// Outputs: JSON { original_url, clean_url, http_status, accessible,
// duplicate, route, title?, author? }
//
// route ∈ youtube | image | video | document | article
import { printJson } from "./json-out.js";
import {
checkAccessibility,
checkDuplicate,
classifyType,
cleanUrl,
contentDir,
fetchTextOk,
isSubstackImage,
} from "./url-utils.js";
const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]);
/**
* detectYouTube extracts a video id from the supported URL shapes:
* youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID.
* Playlists/channels are intentionally NOT YouTube routes (fall through to type).
* @param {string} target
* @returns {{isYouTube: boolean, videoId: string}}
*/
export function detectYouTube(target) {
let u;
try {
u = new URL(target);
} catch {
return { isYouTube: false, videoId: "" };
}
const host = u.host.toLowerCase();
if (host === "youtu.be") {
const id = u.pathname.replace(/^\//, "").split("/")[0];
return { isYouTube: id !== "", videoId: id };
}
if (YT_HOSTS.has(host)) {
if (u.pathname === "/watch") {
const id = u.searchParams.get("v") ?? "";
return { isYouTube: id !== "", videoId: id };
}
if (u.pathname.startsWith("/shorts/")) {
const parts = u.pathname.split("/");
if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] };
}
}
return { isYouTube: false, videoId: "" };
}
/**
* canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes.
* @param {string} videoId
* @returns {string}
*/
function canonicalWatchUrl(videoId) {
return "https://www.youtube.com/watch?v=" + videoId;
}
/**
* fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key).
* Best-effort: any failure returns empty strings so the route stays `youtube`
* and the skill can fall back.
* @param {string} watchUrl
* @returns {Promise<{title: string, author: string}>}
*/
async function fetchYouTubeMeta(watchUrl) {
const endpoint =
"https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json";
const body = await fetchTextOk(endpoint, 10_000);
if (body === "") return { title: "", author: "" };
try {
const data = JSON.parse(body);
return { title: data.title ?? "", author: data.author_name ?? "" };
} catch {
return { title: "", author: "" };
}
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runAddUrl(args) {
if (args.length < 1 || args[0] === "") {
process.stderr.write("Usage: node scripts/newsletter add-url <url>\n");
process.exit(1);
}
const target = args[0];
const cleaned = cleanUrl(target);
const { isYouTube, videoId } = detectYouTube(cleaned);
// For YouTube, dedup/store against the canonical watch URL so youtu.be and
// shorts links collapse onto the same identity-param key as watch URLs.
const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned;
// Route order: YouTube → Substack image (by host, not extension, so f_auto /
// .avif / .heic / extensionless CDN URLs still route to the image handler) →
// file-extension classification.
let route;
if (isYouTube) route = "youtube";
else if (isSubstackImage(cleaned)) route = "image";
else route = classifyType(cleaned);
const httpStatus = await checkAccessibility(cleaned);
// title/author are omitted entirely when empty, not emitted as "": the
// handlers treat a present key as "metadata was resolved".
/** @type {Record<string, unknown>} */
const out = {
original_url: target,
clean_url: effectiveUrl,
http_status: httpStatus,
accessible: httpStatus === "200",
duplicate: checkDuplicate(effectiveUrl, contentDir()),
route,
};
if (route === "youtube") {
const { title, author } = await fetchYouTubeMeta(effectiveUrl);
if (title !== "") out.title = title;
if (author !== "") out.author = author;
}
printJson(out);
}