mirror of
https://github.com/tiennm99/blog.git
synced 2026-10-11 03:13:10 +00:00
feat(newsletter): port the engine to JavaScript
Translates the seven-subcommand engine to Node ESM, one module per former Go file, invoked as `node scripts/newsletter <command>` from the repo root. The Go implementation stays in place for now so parity can be measured against it. Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual string surgery in caption extraction and covers RSS and sitemap XML through xmlMode without a second parser. fetch-via-defuddle gains a local extraction stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a third IP is the whole point when this machine's IP is the blocked one. Both stages now emit the same YAML frontmatter plus body, and an extraction that looks like a bot challenge counts as a local failure so the proxy still runs. Behaviour is preserved where it is load-bearing rather than where it is merely idiomatic: query strings are rebuilt by string surgery so surviving parameters keep their original order and encoding, duplicate detection stays an index loop with byte-offset boundary checks so a prefix of a stored URL is not a false match, empty optional fields are omitted rather than emitted as "", a deep-crawl miss reports cutoff as null rather than dropping the key, bullet filtering counts code points, tag counts keep their six-column alignment, and a malformed percent sequence falls back to the raw substring. The printer lives in its own module so a command module never imports the dispatcher: index.js runs main() at module scope, so that cycle would execute the CLI as a side effect of any import. Missing dependencies report one actionable line instead of a module-resolution stack trace, and a reader that closes early exits quietly instead of raising EPIPE.
This commit is contained in:
1 parent
fdbbc5ba42
commit
64c7b76b4e
14 files changed
+1995
-2
No files matched your search
@@ -0,0 +1,127 @@
|
||||
// Meta URL router for the mt-add-url skill — the single entry per URL.
|
||||
// Usage: node scripts/newsletter add-url "<url>"
|
||||
// Outputs: JSON { original_url, clean_url, http_status, accessible,
|
||||
// duplicate, route, title?, author? }
|
||||
//
|
||||
// route ∈ youtube | image | video | document | article
|
||||
|
||||
import { printJson } from "./json-out.js";
|
||||
import {
|
||||
checkAccessibility,
|
||||
checkDuplicate,
|
||||
classifyType,
|
||||
cleanUrl,
|
||||
contentDir,
|
||||
fetchTextOk,
|
||||
isSubstackImage,
|
||||
} from "./url-utils.js";
|
||||
|
||||
const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]);
|
||||
|
||||
/**
|
||||
* detectYouTube extracts a video id from the supported URL shapes:
|
||||
* youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID.
|
||||
* Playlists/channels are intentionally NOT YouTube routes (fall through to type).
|
||||
* @param {string} target
|
||||
* @returns {{isYouTube: boolean, videoId: string}}
|
||||
*/
|
||||
export function detectYouTube(target) {
|
||||
let u;
|
||||
try {
|
||||
u = new URL(target);
|
||||
} catch {
|
||||
return { isYouTube: false, videoId: "" };
|
||||
}
|
||||
const host = u.host.toLowerCase();
|
||||
if (host === "youtu.be") {
|
||||
const id = u.pathname.replace(/^\//, "").split("/")[0];
|
||||
return { isYouTube: id !== "", videoId: id };
|
||||
}
|
||||
if (YT_HOSTS.has(host)) {
|
||||
if (u.pathname === "/watch") {
|
||||
const id = u.searchParams.get("v") ?? "";
|
||||
return { isYouTube: id !== "", videoId: id };
|
||||
}
|
||||
if (u.pathname.startsWith("/shorts/")) {
|
||||
const parts = u.pathname.split("/");
|
||||
if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] };
|
||||
}
|
||||
}
|
||||
return { isYouTube: false, videoId: "" };
|
||||
}
|
||||
|
||||
/**
|
||||
* canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes.
|
||||
* @param {string} videoId
|
||||
* @returns {string}
|
||||
*/
|
||||
function canonicalWatchUrl(videoId) {
|
||||
return "https://www.youtube.com/watch?v=" + videoId;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key).
|
||||
* Best-effort: any failure returns empty strings so the route stays `youtube`
|
||||
* and the skill can fall back.
|
||||
* @param {string} watchUrl
|
||||
* @returns {Promise<{title: string, author: string}>}
|
||||
*/
|
||||
async function fetchYouTubeMeta(watchUrl) {
|
||||
const endpoint =
|
||||
"https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json";
|
||||
const body = await fetchTextOk(endpoint, 10_000);
|
||||
if (body === "") return { title: "", author: "" };
|
||||
try {
|
||||
const data = JSON.parse(body);
|
||||
return { title: data.title ?? "", author: data.author_name ?? "" };
|
||||
} catch {
|
||||
return { title: "", author: "" };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runAddUrl(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter add-url <url>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const target = args[0];
|
||||
|
||||
const cleaned = cleanUrl(target);
|
||||
const { isYouTube, videoId } = detectYouTube(cleaned);
|
||||
|
||||
// For YouTube, dedup/store against the canonical watch URL so youtu.be and
|
||||
// shorts links collapse onto the same identity-param key as watch URLs.
|
||||
const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned;
|
||||
|
||||
// Route order: YouTube → Substack image (by host, not extension, so f_auto /
|
||||
// .avif / .heic / extensionless CDN URLs still route to the image handler) →
|
||||
// file-extension classification.
|
||||
let route;
|
||||
if (isYouTube) route = "youtube";
|
||||
else if (isSubstackImage(cleaned)) route = "image";
|
||||
else route = classifyType(cleaned);
|
||||
|
||||
const httpStatus = await checkAccessibility(cleaned);
|
||||
|
||||
// title/author are omitted entirely when empty, not emitted as "": the
|
||||
// handlers treat a present key as "metadata was resolved".
|
||||
/** @type {Record<string, unknown>} */
|
||||
const out = {
|
||||
original_url: target,
|
||||
clean_url: effectiveUrl,
|
||||
http_status: httpStatus,
|
||||
accessible: httpStatus === "200",
|
||||
duplicate: checkDuplicate(effectiveUrl, contentDir()),
|
||||
route,
|
||||
};
|
||||
if (route === "youtube") {
|
||||
const { title, author } = await fetchYouTubeMeta(effectiveUrl);
|
||||
if (title !== "") out.title = title;
|
||||
if (author !== "") out.author = author;
|
||||
}
|
||||
printJson(out);
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// Detect whether an image URL is Substack-hosted and extract its S3 image UUID.
|
||||
// Usage: node scripts/newsletter detect-image-source "<image-url>"
|
||||
// Output: JSON { original_url, clean_url, isSubstack, uuid?, innerUrl? }
|
||||
//
|
||||
// Substack images are usually served via a CDN wrapper:
|
||||
//
|
||||
// https://substackcdn.com/image/fetch/$s_!x!,.../https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F<uuid>_WxH.png
|
||||
//
|
||||
// The publication is NOT encoded in the URL — only the image identity (uuid) is.
|
||||
|
||||
import { printJson } from "./json-out.js";
|
||||
import { cleanUrl, isSubstackImage, substackImageUuid } from "./url-utils.js";
|
||||
|
||||
/**
|
||||
* extractInnerUrl pulls the inner S3 URL out of a substackcdn /image/fetch/
|
||||
* wrapper (if present). A malformed percent sequence falls back to the raw
|
||||
* substring rather than throwing.
|
||||
* @param {string} target
|
||||
* @returns {string}
|
||||
*/
|
||||
export function extractInnerUrl(target) {
|
||||
const marker = target.indexOf("/https%3A%2F%2F");
|
||||
if (marker !== -1) {
|
||||
const raw = target.slice(marker + 1);
|
||||
try {
|
||||
return decodeURIComponent(raw);
|
||||
} catch {
|
||||
return raw;
|
||||
}
|
||||
}
|
||||
// Some forms embed a plain (already-decoded) inner https URL.
|
||||
if (target.length > 8) {
|
||||
const plain = target.slice(8).indexOf("/https://");
|
||||
if (plain !== -1) return target.slice(8 + plain + 1);
|
||||
}
|
||||
return target;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runDetectImageSource(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter detect-image-source <image-url>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const target = args[0];
|
||||
const isSubstack = isSubstackImage(target);
|
||||
|
||||
// Empty optional fields are omitted, not emitted as "": the skills branch on
|
||||
// the key being present.
|
||||
/** @type {Record<string, unknown>} */
|
||||
const out = {
|
||||
original_url: target,
|
||||
clean_url: cleanUrl(target),
|
||||
isSubstack,
|
||||
};
|
||||
if (isSubstack) {
|
||||
const uuid = substackImageUuid(target);
|
||||
const innerUrl = extractInnerUrl(target);
|
||||
if (uuid !== "") out.uuid = uuid;
|
||||
if (innerUrl !== "") out.innerUrl = innerUrl;
|
||||
}
|
||||
printJson(out);
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
// mt-fetch-url fallback fetcher.
|
||||
// Usage: node scripts/newsletter fetch-via-defuddle <target_url>
|
||||
// Exit codes: 0 = content returned, 1 = every tier failed, 2 = bad arguments.
|
||||
//
|
||||
// Two tiers, tried in order:
|
||||
// 1. local defuddle — extracts on this machine, so the chain no longer depends
|
||||
// on a single third-party service being reachable.
|
||||
// 2. the defuddle.md proxy — fetches from a third IP, which is the point when
|
||||
// this machine's IP is the one being blocked.
|
||||
//
|
||||
// Both tiers emit YAML frontmatter followed by the markdown body, so callers
|
||||
// parse one shape regardless of which tier answered.
|
||||
|
||||
const BROWSER_UA =
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36";
|
||||
|
||||
// A bot wall answers 200 with a real body, so a non-empty extraction is not by
|
||||
// itself a success. Recognising the wall is what lets the proxy tier — which
|
||||
// fetches from a different IP — still get its turn.
|
||||
const CHALLENGE_MARKERS = [
|
||||
/just a moment/i,
|
||||
/checking your browser/i,
|
||||
/attention required/i,
|
||||
/cloudflare/i,
|
||||
/enable javascript (and cookies )?to continue/i,
|
||||
/verify (that )?you('re| are) (a )?human/i,
|
||||
/are you a robot/i,
|
||||
/access denied/i,
|
||||
/captcha/i,
|
||||
];
|
||||
|
||||
/**
|
||||
* looksLikeChallenge reports whether an extraction is a bot wall rather than the
|
||||
* page that was asked for.
|
||||
* @param {string} title
|
||||
* @param {string} body
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function looksLikeChallenge(title, body) {
|
||||
// Only the opening of the body: an article may legitimately discuss Cloudflare.
|
||||
const sample = title + "\n" + body.slice(0, 400);
|
||||
return CHALLENGE_MARKERS.some((re) => re.test(sample));
|
||||
}
|
||||
|
||||
/**
|
||||
* yamlFrontmatter renders the metadata block, omitting fields with no value.
|
||||
* @param {Record<string, string>} fields
|
||||
* @returns {string}
|
||||
*/
|
||||
function yamlFrontmatter(fields) {
|
||||
const lines = Object.entries(fields)
|
||||
.filter(([, v]) => typeof v === "string" && v !== "")
|
||||
.map(([k, v]) => `${k}: ${JSON.stringify(v)}`);
|
||||
return lines.length === 0 ? "" : "---\n" + lines.join("\n") + "\n---\n\n";
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchLocally extracts article content with defuddle running in-process, and
|
||||
* renders it in the same frontmatter-plus-body shape the proxy returns.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>} the document, or "" when extraction yields nothing usable
|
||||
*/
|
||||
async function fetchLocally(target) {
|
||||
const { Defuddle } = await import("defuddle/node");
|
||||
const { parseHTML } = await import("linkedom");
|
||||
const res = await fetch(target, {
|
||||
headers: { "user-agent": BROWSER_UA },
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
});
|
||||
if (!res.ok) throw new Error(`upstream returned ${res.status}`);
|
||||
const html = await res.text();
|
||||
const { document } = parseHTML(html);
|
||||
const result = await Defuddle(document, target, { markdown: true });
|
||||
const body = String(result?.content ?? "");
|
||||
if (body.trim() === "") return "";
|
||||
const title = String(result?.title ?? "");
|
||||
if (looksLikeChallenge(title, body)) {
|
||||
throw new Error("extraction looks like a bot challenge, not the page");
|
||||
}
|
||||
const frontmatter = yamlFrontmatter({
|
||||
title,
|
||||
author: String(result?.author ?? ""),
|
||||
description: String(result?.description ?? ""),
|
||||
site: String(result?.site ?? ""),
|
||||
published: String(result?.published ?? ""),
|
||||
source: target,
|
||||
});
|
||||
return frontmatter + body;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchViaProxy fetches through defuddle.md, which resolves the page from its
|
||||
* own IP.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>} the response body
|
||||
*/
|
||||
async function fetchViaProxy(target) {
|
||||
const res = await fetch("https://defuddle.md/" + target, {
|
||||
headers: { "user-agent": "mt-fetch-url/1.0" },
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
});
|
||||
const body = await res.text();
|
||||
if (!res.ok || body.trim() === "") {
|
||||
throw new Error(`defuddle returned ${res.status} / empty body`);
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFetchViaDefuddle(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter fetch-via-defuddle <target_url>\n");
|
||||
process.exit(2);
|
||||
}
|
||||
const target = args[0];
|
||||
|
||||
try {
|
||||
const doc = await fetchLocally(target);
|
||||
if (doc !== "") {
|
||||
process.stdout.write(doc.endsWith("\n") ? doc : doc + "\n");
|
||||
return;
|
||||
}
|
||||
process.stderr.write(`mt-fetch-url: local defuddle extracted nothing for ${target}\n`);
|
||||
} catch (err) {
|
||||
process.stderr.write(
|
||||
`mt-fetch-url: local defuddle failed for ${target}: ${String(err?.message ?? err)}\n`,
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
process.stdout.write(await fetchViaProxy(target));
|
||||
} catch (err) {
|
||||
process.stderr.write(
|
||||
`mt-fetch-url: defuddle.md failed for ${target}: ${String(err?.message ?? err)}\n`,
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
// Find the most recent newsletter number and return the next one.
|
||||
// Usage: node scripts/newsletter find-newsletter-number
|
||||
// Outputs: the next newsletter number.
|
||||
|
||||
import { readdirSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { contentDir } from "./url-utils.js";
|
||||
|
||||
/** Matches the "Newsletter #N" heading a post is numbered by. @type {RegExp} */
|
||||
export const NEWSLETTER_NUM_RE = /Newsletter\s*#(\d+)/;
|
||||
const YEAR_DIR_RE = /^\d{4}$/;
|
||||
const TWO_DIGIT_DIR_RE = /^\d{2}$/;
|
||||
|
||||
/**
|
||||
* listDirsDesc returns dir's subdirectory names matching re, sorted descending.
|
||||
* @param {string} dir
|
||||
* @param {RegExp} re
|
||||
* @returns {string[]}
|
||||
*/
|
||||
function listDirsDesc(dir, re) {
|
||||
let entries;
|
||||
try {
|
||||
entries = readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
return entries
|
||||
.filter((e) => re.test(e.name))
|
||||
.map((e) => e.name)
|
||||
.sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
|
||||
}
|
||||
|
||||
/**
|
||||
* extractNewsletterNumber reads the first "Newsletter #N" in a post.
|
||||
* @param {string} path
|
||||
* @returns {number}
|
||||
*/
|
||||
function extractNewsletterNumber(path) {
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
const m = NEWSLETTER_NUM_RE.exec(content);
|
||||
if (m === null) return 0;
|
||||
const n = Number.parseInt(m[1], 10);
|
||||
return Number.isNaN(n) ? 0 : n;
|
||||
}
|
||||
|
||||
/**
|
||||
* findMostRecentNewsletter scans year/month/day directories newest-first for
|
||||
* the highest newsletter number.
|
||||
* @returns {number}
|
||||
*/
|
||||
export function findMostRecentNewsletter() {
|
||||
let maxNumber = 0;
|
||||
for (const year of listDirsDesc(contentDir(), YEAR_DIR_RE)) {
|
||||
const yearDir = join(contentDir(), year);
|
||||
for (const month of listDirsDesc(yearDir, TWO_DIGIT_DIR_RE)) {
|
||||
const monthDir = join(yearDir, month);
|
||||
for (const day of listDirsDesc(monthDir, TWO_DIGIT_DIR_RE)) {
|
||||
const n = extractNewsletterNumber(join(monthDir, day, "index.md"));
|
||||
if (n > maxNumber) maxNumber = n;
|
||||
}
|
||||
}
|
||||
// Early exit: a newsletter found in this year — no need to go further back.
|
||||
if (maxNumber > 0) break;
|
||||
}
|
||||
return maxNumber;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} _args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFindNewsletterNumber(_args) {
|
||||
process.stdout.write(String(findMostRecentNewsletter() + 1) + "\n");
|
||||
}
|
||||
@@ -0,0 +1,231 @@
|
||||
// Find which Substack post embeds a given image UUID, and extract a label.
|
||||
// Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]
|
||||
// Output on hit: JSON { found:true, source, publication, postTitle, postUrl, caption, candidates }
|
||||
// Output on miss: JSON { found:false } (RSS) or { found:false, source:"sitemap", scanned, budget, cutoff } (--deep)
|
||||
//
|
||||
// Strategy: RSS feed first (fast, ~recent weeks). With --deep, fall back to a
|
||||
// heavier sitemap crawl up to ~3 months back — opt-in because it fetches many
|
||||
// posts. A Substack CDN URL does not encode its publication, so we search each
|
||||
// publication listed in config/substack-publications.json, read at runtime so
|
||||
// editing the JSON takes effect immediately.
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { parseArgs } from "node:util";
|
||||
import { printJson } from "./json-out.js";
|
||||
import { fetchTextOk } from "./url-utils.js";
|
||||
import {
|
||||
captionForUuid,
|
||||
extractCandidates,
|
||||
itemLink,
|
||||
itemTitle,
|
||||
loadXml,
|
||||
postTitleFromHtml,
|
||||
} from "./html-text.js";
|
||||
|
||||
/** Total post fetches allowed across ALL publications during a --deep crawl. */
|
||||
const DEEP_FETCH_BUDGET = 40;
|
||||
|
||||
/**
|
||||
* loadPublications reads the publication list, falling back to the default on
|
||||
* any read or parse failure.
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function loadPublications() {
|
||||
try {
|
||||
const raw = readFileSync(new URL("./config/substack-publications.json", import.meta.url), "utf8");
|
||||
const pubs = JSON.parse(raw);
|
||||
if (Array.isArray(pubs) && pubs.length > 0) return pubs;
|
||||
} catch {
|
||||
/* fall through */
|
||||
}
|
||||
return ["blog.bytebytego.com"];
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchPage: body text with a browser-ish UA, or "" on any error.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
function fetchPage(target) {
|
||||
return fetchTextOk(target, 10_000, "Mozilla/5.0");
|
||||
}
|
||||
|
||||
const RFC3339_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$/i;
|
||||
const DATE_ONLY_RE = /^\d{4}-\d{2}-\d{2}$/;
|
||||
|
||||
/**
|
||||
* parseLastmod accepts only the two layouts the Go engine accepted — RFC3339
|
||||
* and a bare date — so a loosely formatted stamp is skipped rather than
|
||||
* silently reinterpreted in local time.
|
||||
* @param {string} s
|
||||
* @returns {Date|null}
|
||||
*/
|
||||
export function parseLastmod(s) {
|
||||
if (!RFC3339_RE.test(s) && !DATE_ONLY_RE.test(s)) return null;
|
||||
const t = new Date(s);
|
||||
return Number.isNaN(t.getTime()) ? null : t;
|
||||
}
|
||||
|
||||
/**
|
||||
* cutoffDate returns the ~3-months-back boundary for the deep crawl.
|
||||
* @returns {Date}
|
||||
*/
|
||||
function cutoffDate() {
|
||||
const d = new Date();
|
||||
d.setUTCMonth(d.getUTCMonth() - 3);
|
||||
return d;
|
||||
}
|
||||
|
||||
/**
|
||||
* @typedef {{found: true, source: string, publication: string, postTitle: string,
|
||||
* postUrl: string, caption: string, candidates: string[]}} PostHit
|
||||
*/
|
||||
|
||||
/**
|
||||
* searchSitemap is the deep fallback: crawl the sitemap back ~3 months, fetch
|
||||
* posts most-recent-first (up to maxFetch from the shared budget), and look for
|
||||
* the UUID. Heavier than RSS — only used on RSS miss.
|
||||
* @param {string} publication
|
||||
* @param {string} id
|
||||
* @param {number} maxFetch
|
||||
* @returns {Promise<{hit: PostHit|null, scanned: number, cutoff: string}>} cutoff is "" when the sitemap itself could not be fetched
|
||||
*/
|
||||
async function searchSitemap(publication, id, maxFetch) {
|
||||
const xml = await fetchPage("https://" + publication + "/sitemap.xml");
|
||||
if (xml === "") return { hit: null, scanned: 0, cutoff: "" };
|
||||
|
||||
const cutoffTime = cutoffDate();
|
||||
const cutoff = cutoffTime.toISOString().slice(0, 10);
|
||||
|
||||
const $ = loadXml(xml);
|
||||
/** @type {{url: string, when: Date}[]} */
|
||||
const candidates = [];
|
||||
$("url").each((_i, el) => {
|
||||
const loc = $(el).children("loc").first().text();
|
||||
const lastmod = $(el).children("lastmod").first().text();
|
||||
if (loc === "" || lastmod === "") return;
|
||||
if (!loc.includes("/p/")) return; // posts only
|
||||
const when = parseLastmod(lastmod);
|
||||
if (when === null || when < cutoffTime) return;
|
||||
candidates.push({ url: loc, when });
|
||||
});
|
||||
candidates.sort((a, b) => b.when.getTime() - a.when.getTime());
|
||||
|
||||
let scanned = 0;
|
||||
for (const c of candidates.slice(0, maxFetch)) {
|
||||
scanned++;
|
||||
const html = await fetchPage(c.url);
|
||||
if (html === "" || !html.includes(id)) continue;
|
||||
return {
|
||||
hit: {
|
||||
found: true,
|
||||
source: "sitemap",
|
||||
publication,
|
||||
postTitle: postTitleFromHtml(html),
|
||||
postUrl: c.url,
|
||||
caption: captionForUuid(html, id),
|
||||
candidates: extractCandidates(html),
|
||||
},
|
||||
scanned,
|
||||
cutoff,
|
||||
};
|
||||
}
|
||||
return { hit: null, scanned, cutoff };
|
||||
}
|
||||
|
||||
/**
|
||||
* searchRss looks for the uuid in a publication's recent feed items.
|
||||
* @param {string} publication
|
||||
* @param {string} id
|
||||
* @returns {Promise<PostHit|null>}
|
||||
*/
|
||||
async function searchRss(publication, id) {
|
||||
const xml = await fetchPage("https://" + publication + "/feed");
|
||||
if (xml === "") return null;
|
||||
const $ = loadXml(xml);
|
||||
const items = $("item").toArray();
|
||||
for (const el of items) {
|
||||
const item = $.html(el);
|
||||
if (!item.includes(id)) continue;
|
||||
return {
|
||||
found: true,
|
||||
source: "rss",
|
||||
publication,
|
||||
postTitle: itemTitle(item),
|
||||
postUrl: itemLink(item),
|
||||
caption: captionForUuid(item, id),
|
||||
candidates: extractCandidates(item),
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFindSubstackPost(args) {
|
||||
let values;
|
||||
try {
|
||||
({ values } = parseArgs({
|
||||
args,
|
||||
options: { uuid: { type: "string" }, deep: { type: "boolean" } },
|
||||
allowPositionals: false,
|
||||
}));
|
||||
} catch (err) {
|
||||
process.stderr.write(String(err?.message ?? err) + "\n");
|
||||
process.stderr.write(
|
||||
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
|
||||
);
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const uuid = values.uuid ?? "";
|
||||
if (uuid === "") {
|
||||
process.stderr.write(
|
||||
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const publications = loadPublications();
|
||||
for (const pub of publications) {
|
||||
const hit = await searchRss(pub, uuid);
|
||||
if (hit !== null) {
|
||||
printJson(hit);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// Deep fallback: sitemap crawl up to ~3 months back, sharing one global fetch
|
||||
// budget across all publications so coverage can't blow up as the
|
||||
// publications list grows.
|
||||
if (values.deep === true) {
|
||||
let totalScanned = 0;
|
||||
/** @type {string|null} */
|
||||
let lastCutoff = null;
|
||||
for (const pub of publications) {
|
||||
const remaining = DEEP_FETCH_BUDGET - totalScanned;
|
||||
if (remaining <= 0) break;
|
||||
const { hit, scanned, cutoff } = await searchSitemap(pub, uuid, remaining);
|
||||
totalScanned += scanned;
|
||||
if (cutoff !== "") lastCutoff = cutoff;
|
||||
if (hit !== null) {
|
||||
printJson(hit);
|
||||
return;
|
||||
}
|
||||
}
|
||||
// cutoff is null (not omitted) when no sitemap could be fetched — the skill
|
||||
// distinguishes "crawled and missed" from "could not crawl".
|
||||
printJson({
|
||||
found: false,
|
||||
source: "sitemap",
|
||||
scanned: totalScanned,
|
||||
budget: DEEP_FETCH_BUDGET,
|
||||
cutoff: lastCutoff,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
printJson({ found: false });
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
// HTML / RSS text-extraction helpers for find-substack-post, ported from
|
||||
// html_text.go. Pure string functions — no network, no fs. A real parser
|
||||
// (cheerio) replaces the hand-rolled regexes the Go version used.
|
||||
|
||||
import * as cheerio from "cheerio";
|
||||
|
||||
/**
|
||||
* loadXml parses an XML fragment (RSS item, sitemap) with CDATA recognition.
|
||||
* @param {string} src
|
||||
* @returns {cheerio.CheerioAPI}
|
||||
*/
|
||||
export function loadXml(src) {
|
||||
return cheerio.load(src, { xmlMode: true }, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* loadHtml parses an HTML blob. CDATA markers are stripped first: RSS carries
|
||||
* post HTML inside CDATA, and the HTML parser would otherwise swallow it as a
|
||||
* bogus comment.
|
||||
* @param {string} src
|
||||
* @returns {cheerio.CheerioAPI}
|
||||
*/
|
||||
export function loadHtml(src) {
|
||||
return cheerio.load(src.split("<![CDATA[").join("").split("]]>").join(""));
|
||||
}
|
||||
|
||||
/**
|
||||
* rawInner returns an element's serialized inner content with any CDATA wrapper
|
||||
* removed — the same substring the Go regexes captured, so entity handling can
|
||||
* stay identical instead of being applied twice.
|
||||
* @param {cheerio.CheerioAPI} $
|
||||
* @param {cheerio.Cheerio<any>} el
|
||||
* @returns {string}
|
||||
*/
|
||||
function rawInner($, el) {
|
||||
const html = $.html(el);
|
||||
const start = html.indexOf(">") + 1;
|
||||
const end = html.lastIndexOf("</");
|
||||
if (start <= 0 || end < start) return "";
|
||||
let inner = html.slice(start, end);
|
||||
if (inner.startsWith("<![CDATA[")) inner = inner.slice(9);
|
||||
if (inner.endsWith("]]>")) inner = inner.slice(0, -3);
|
||||
return inner;
|
||||
}
|
||||
|
||||
/**
|
||||
* decodeEntities decodes HTML entities (named, numeric, hex) and trims.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
export function decodeEntities(s) {
|
||||
if (s === "") return "";
|
||||
return cheerio.load(s, null, false).text().trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* collapseWhitespace squeezes runs of whitespace to one space and trims.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
function collapseWhitespace(s) {
|
||||
return s.replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* stripTags removes markup, decodes entities, and collapses whitespace.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
export function stripTags(s) {
|
||||
if (s === "") return "";
|
||||
return collapseWhitespace(cheerio.load(s, null, false).text());
|
||||
}
|
||||
|
||||
/**
|
||||
* itemTitle pulls the first <title> (CDATA or plain) from an RSS <item> chunk.
|
||||
* @param {string} item
|
||||
* @returns {string}
|
||||
*/
|
||||
export function itemTitle(item) {
|
||||
const $ = loadXml(item);
|
||||
const el = $("title").first();
|
||||
if (el.length === 0) return "";
|
||||
return decodeEntities(rawInner($, el));
|
||||
}
|
||||
|
||||
/**
|
||||
* itemLink pulls the first <link> (CDATA or plain) from an RSS <item> chunk.
|
||||
* Entities are left as stored — the link goes straight into a post.
|
||||
* @param {string} item
|
||||
* @returns {string}
|
||||
*/
|
||||
export function itemLink(item) {
|
||||
const $ = loadXml(item);
|
||||
const el = $("link").first();
|
||||
if (el.length === 0) return "";
|
||||
return rawInner($, el).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* extractCandidates pulls candidate topic titles from a post's TOC bullet list.
|
||||
* ByteByteGo does not attach captions to images — the topic titles live only in
|
||||
* the "in this issue" bullets, and image→title cannot be mapped automatically
|
||||
* (sponsor/video items interleave), so these are surfaced for the user to pick
|
||||
* from. Light filtering keeps the list short: dedupe, drop sub-point
|
||||
* explanations and over-long lines.
|
||||
* @param {string} htmlSrc
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function extractCandidates(htmlSrc) {
|
||||
const $ = loadHtml(htmlSrc);
|
||||
/** @type {Set<string>} */
|
||||
const seen = new Set();
|
||||
/** @type {string[]} */
|
||||
const out = [];
|
||||
$("li").each((_i, el) => {
|
||||
const text = collapseWhitespace($(el).text());
|
||||
if (text === "") return;
|
||||
// Titles are short; long lines are sub-point explanations. Count code
|
||||
// points, not UTF-16 units, so an emoji does not count double.
|
||||
const n = [...text].length;
|
||||
if (n < 6 || n > 70) return;
|
||||
const key = text.toLowerCase();
|
||||
if (seen.has(key)) return; // content is duplicated in the page
|
||||
seen.add(key);
|
||||
out.push(text);
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* captionForUuid returns the <figcaption> text of the <figure> containing the
|
||||
* UUID. Cover images live in <enclosure> (no figure) → "".
|
||||
* @param {string} src
|
||||
* @param {string} id
|
||||
* @returns {string}
|
||||
*/
|
||||
export function captionForUuid(src, id) {
|
||||
if (id === "") return "";
|
||||
const $ = loadHtml(src);
|
||||
let target = null;
|
||||
$("*").each((_i, el) => {
|
||||
if (target !== null) return false;
|
||||
for (const v of Object.values(el.attribs ?? {})) {
|
||||
if (typeof v === "string" && v.includes(id)) {
|
||||
target = el;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (const child of el.children ?? []) {
|
||||
if (child.type === "text" && String(child.data).includes(id)) {
|
||||
target = el;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
});
|
||||
if (target === null) return "";
|
||||
const figure = $(target).closest("figure");
|
||||
if (figure.length === 0) return "";
|
||||
const caption = figure.find("figcaption").first();
|
||||
if (caption.length === 0) return "";
|
||||
return collapseWhitespace(caption.text());
|
||||
}
|
||||
|
||||
/**
|
||||
* postTitleFromHtml extracts a post title from server-rendered post HTML
|
||||
* (og:title preferred, then <h1>, then <title>).
|
||||
* @param {string} htmlSrc
|
||||
* @returns {string}
|
||||
*/
|
||||
export function postTitleFromHtml(htmlSrc) {
|
||||
const $ = loadHtml(htmlSrc);
|
||||
const og = $('meta[property="og:title"]').first().attr("content");
|
||||
if (og !== undefined && og !== "") return og.trim();
|
||||
const h1 = $("h1").first();
|
||||
if (h1.length > 0) return collapseWhitespace(h1.text());
|
||||
const title = $("title").first();
|
||||
if (title.length > 0) return title.text().trim();
|
||||
return "";
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
// Newsletter engine for the mt-* skills — one entry point, one subcommand per
|
||||
// task. Invoked from the repo root:
|
||||
//
|
||||
// node scripts/newsletter <command> [args]
|
||||
//
|
||||
// Repo-relative paths (content/post) resolve from the working directory, so the
|
||||
// repo-root invocation contract from AGENTS.md still applies.
|
||||
|
||||
const USAGE = `Usage: node scripts/newsletter <command> [args]
|
||||
|
||||
Commands:
|
||||
add-url <url> classify + dedup a URL, emit JSON route
|
||||
find-newsletter-number print the next newsletter number
|
||||
list-existing-tags tag frequencies, most-used first (top 40)
|
||||
detect-image-source <url> detect Substack image + uuid
|
||||
find-substack-post --uuid <uuid> [--deep] find the post embedding an image uuid
|
||||
fetch-via-defuddle <url> fallback fetch (local defuddle, then defuddle.md)
|
||||
post-stats <path/to/index.md> count the post's articles/images/videos/documents
|
||||
`;
|
||||
|
||||
// A reader that closes early (`… | head -3`) makes the next write fail with
|
||||
// EPIPE, which Node surfaces as an unhandled error event. Stop quietly instead
|
||||
// of printing a stack trace over the user's terminal.
|
||||
process.stdout.on("error", (err) => {
|
||||
if (err.code === "EPIPE") process.exit(0);
|
||||
throw err;
|
||||
});
|
||||
|
||||
/** @returns {void} */
|
||||
function usage() {
|
||||
process.stderr.write(USAGE);
|
||||
}
|
||||
|
||||
/**
|
||||
* Command modules are imported lazily so a missing node_modules reports the one
|
||||
* actionable fix instead of a module-resolution stack trace.
|
||||
* @param {string} spec
|
||||
* @returns {Promise<Record<string, any>>}
|
||||
*/
|
||||
async function loadCommand(spec) {
|
||||
try {
|
||||
return await import(spec);
|
||||
} catch (err) {
|
||||
if (err !== null && typeof err === "object" && err.code === "ERR_MODULE_NOT_FOUND") {
|
||||
process.stderr.write(
|
||||
"newsletter engine: dependencies missing — run 'npm ci' from the repo root\n",
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
/** @type {Record<string, {module: string, fn: string}>} */
|
||||
const COMMANDS = {
|
||||
"add-url": { module: "./add-url.js", fn: "runAddUrl" },
|
||||
"find-newsletter-number": { module: "./find-newsletter-number.js", fn: "runFindNewsletterNumber" },
|
||||
"list-existing-tags": { module: "./list-existing-tags.js", fn: "runListExistingTags" },
|
||||
"detect-image-source": { module: "./detect-image-source.js", fn: "runDetectImageSource" },
|
||||
"find-substack-post": { module: "./find-substack-post.js", fn: "runFindSubstackPost" },
|
||||
"fetch-via-defuddle": { module: "./fetch-via-defuddle.js", fn: "runFetchViaDefuddle" },
|
||||
"post-stats": { module: "./post-stats.js", fn: "runPostStats" },
|
||||
};
|
||||
|
||||
/** @returns {Promise<void>} */
|
||||
async function main() {
|
||||
const argv = process.argv.slice(2);
|
||||
if (argv.length < 1) {
|
||||
usage();
|
||||
process.exit(1);
|
||||
}
|
||||
const [name, ...args] = argv;
|
||||
const entry = COMMANDS[name];
|
||||
if (entry === undefined) {
|
||||
process.stderr.write(`unknown command: ${name}\n`);
|
||||
usage();
|
||||
process.exit(1);
|
||||
}
|
||||
const mod = await loadCommand(entry.module);
|
||||
await mod[entry.fn](args);
|
||||
}
|
||||
|
||||
try {
|
||||
await main();
|
||||
} catch (err) {
|
||||
process.stderr.write("newsletter engine: " + String(err?.stack ?? err) + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// The shared JSON printer. It lives in its own module rather than in index.js so
|
||||
// that importing a command module never pulls in the dispatcher — an import
|
||||
// cycle there would run the CLI as a side effect of any import.
|
||||
|
||||
/**
|
||||
* printJson mirrors console.log(JSON.stringify(v, null, 2)): 2-space indent, no
|
||||
* HTML escaping (URLs with & must stay readable), trailing newline.
|
||||
* @param {unknown} v
|
||||
* @returns {void}
|
||||
*/
|
||||
export function printJson(v) {
|
||||
process.stdout.write(JSON.stringify(v, null, 2) + "\n");
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
// List existing tags in the repo ranked by frequency.
|
||||
// Usage: node scripts/newsletter list-existing-tags
|
||||
// Outputs: tag count and name, sorted most-used first (top 40).
|
||||
//
|
||||
// NOTE: not currently used by the mt-add-tags skill. Tag normalization is
|
||||
// disabled until existing posts have standardized tags; to enable, uncomment
|
||||
// step 4a in that skill's SKILL.md.
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { basename } from "node:path";
|
||||
import { collectMarkdown, contentDir } from "./url-utils.js";
|
||||
|
||||
const TAGS_LINE_RE = /^tags:\s*\[([^\]]*)\]/m;
|
||||
const QUOTED_TAG_RE = /"([^"]+)"/g;
|
||||
|
||||
/**
|
||||
* extractTags pulls quoted tag strings from an index.md frontmatter tags array.
|
||||
* @param {string} path
|
||||
* @returns {string[]}
|
||||
*/
|
||||
function extractTags(path) {
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
const m = TAGS_LINE_RE.exec(content);
|
||||
if (m === null) return [];
|
||||
return [...m[1].matchAll(QUOTED_TAG_RE)].map((q) => q[1]);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} _args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runListExistingTags(_args) {
|
||||
// Array + index map keeps first-seen order for equal counts, so the stable
|
||||
// sort below ranks ties deterministically (walk order is lexical).
|
||||
/** @type {{tag: string, count: number}[]} */
|
||||
const counts = [];
|
||||
/** @type {Map<string, number>} */
|
||||
const index = new Map();
|
||||
|
||||
for (const path of collectMarkdown(contentDir())) {
|
||||
if (basename(path) !== "index.md") continue;
|
||||
for (const tag of extractTags(path)) {
|
||||
const at = index.get(tag);
|
||||
if (at !== undefined) counts[at].count++;
|
||||
else {
|
||||
index.set(tag, counts.length);
|
||||
counts.push({ tag, count: 1 });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
counts.sort((a, b) => b.count - a.count);
|
||||
// Plain text, not JSON: the count is right-aligned in a 6-character field and
|
||||
// the skill reads that layout.
|
||||
for (const { tag, count } of counts.slice(0, 40)) {
|
||||
process.stdout.write(String(count).padStart(6) + " " + tag + "\n");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// Count the entries already present in a newsletter post, so a handler can
|
||||
// report a running tally after each insertion.
|
||||
// Usage: node scripts/newsletter post-stats <path/to/index.md>
|
||||
// Outputs: JSON { post, newsletter, articles, images, videos, documents, total }
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { printJson } from "./json-out.js";
|
||||
import { NEWSLETTER_NUM_RE } from "./find-newsletter-number.js";
|
||||
|
||||
// Entry shapes, per the Bonus format in the shared post mechanics:
|
||||
//
|
||||
// articles "## [Title](url)" (level-2 heading, main content)
|
||||
// images "" (under **Images:**)
|
||||
// videos "[Title](url)" (under **Videos:**)
|
||||
// documents "[PDF: title](url)" (under **Documents:**)
|
||||
const ARTICLE_HEADING_RE = /^##\s+\[/;
|
||||
const BONUS_HEADING_RE = /^###\s+Bonus\b/;
|
||||
const SUBSECTION_RE = /^\*\*(Images|Videos|Documents):\*\*/;
|
||||
const IMAGE_ENTRY_RE = /^!\[/;
|
||||
const LINK_ENTRY_RE = /^\[/;
|
||||
|
||||
/**
|
||||
* countPostEntries walks the post once. Article headings are counted anywhere
|
||||
* outside Bonus; asset entries are attributed to whichever subsection is open.
|
||||
* @param {string} content
|
||||
* @returns {{articles: number, images: number, videos: number, documents: number, total: number}}
|
||||
*/
|
||||
export function countPostEntries(content) {
|
||||
let articles = 0;
|
||||
let images = 0;
|
||||
let videos = 0;
|
||||
let documents = 0;
|
||||
let inBonus = false;
|
||||
let subsection = "";
|
||||
|
||||
for (const raw of content.split("\n")) {
|
||||
const line = raw.trim();
|
||||
if (BONUS_HEADING_RE.test(line)) {
|
||||
inBonus = true;
|
||||
subsection = "";
|
||||
continue;
|
||||
}
|
||||
const sub = SUBSECTION_RE.exec(line);
|
||||
if (sub !== null) {
|
||||
subsection = sub[1];
|
||||
continue;
|
||||
}
|
||||
if (ARTICLE_HEADING_RE.test(line)) {
|
||||
articles++;
|
||||
continue;
|
||||
}
|
||||
if (!inBonus) continue;
|
||||
if (subsection === "Images") {
|
||||
if (IMAGE_ENTRY_RE.test(line)) images++;
|
||||
} else if (subsection === "Videos") {
|
||||
// A direct video file entry looks the same as a YouTube entry;
|
||||
// both belong to the Videos tally.
|
||||
if (LINK_ENTRY_RE.test(line)) videos++;
|
||||
} else if (subsection === "Documents") {
|
||||
if (LINK_ENTRY_RE.test(line)) documents++;
|
||||
}
|
||||
}
|
||||
return { articles, images, videos, documents, total: articles + images + videos + documents };
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runPostStats(args) {
|
||||
if (args.length < 1) {
|
||||
process.stderr.write("usage: post-stats <path/to/index.md>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const path = args[0];
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch (err) {
|
||||
process.stderr.write("read post: " + String(err?.message ?? err) + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const counted = countPostEntries(content);
|
||||
const m = NEWSLETTER_NUM_RE.exec(content);
|
||||
printJson({
|
||||
post: path,
|
||||
newsletter: m === null ? 0 : Number.parseInt(m[1], 10),
|
||||
articles: counted.articles,
|
||||
images: counted.images,
|
||||
videos: counted.videos,
|
||||
documents: counted.documents,
|
||||
total: counted.total,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,339 @@
|
||||
// Shared URL helpers, ported from url_utils.go. Owned by the add-url router;
|
||||
// reused by the other subcommands.
|
||||
|
||||
import { readdirSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
|
||||
/** Exact-match tracking params; any key starting with utm_ is also dropped. */
|
||||
const EXACT_TRACKING = new Set([
|
||||
"fbclid", "gclid", "msclkid", "mc_eid",
|
||||
"aid", "ref", "ref_src", "ref_url", "source", "s",
|
||||
"ck_subscriber_id", "igshid", "yclid", "vero_id",
|
||||
]);
|
||||
|
||||
/**
|
||||
* contentDir is repo-root relative (invocation contract: run from repo root).
|
||||
* @returns {string}
|
||||
*/
|
||||
export function contentDir() {
|
||||
return join("content", "post");
|
||||
}
|
||||
|
||||
/**
|
||||
* cleanUrl removes common tracking parameters. Surviving query pairs are kept
|
||||
* verbatim (no re-encoding) and in their original order — URLSearchParams would
|
||||
* normalize percent-encoding (%7E → ~, + → %20), which must not leak into
|
||||
* stored clean_url values. Unparseable / non-absolute input is returned
|
||||
* untouched rather than corrupted.
|
||||
* @param {string} raw
|
||||
* @returns {string}
|
||||
*/
|
||||
export function cleanUrl(raw) {
|
||||
let u;
|
||||
try {
|
||||
u = new URL(raw);
|
||||
} catch {
|
||||
return raw;
|
||||
}
|
||||
if (!u.protocol || !u.host) return raw;
|
||||
// WHATWG URL serializes an empty path as "/" for special schemes; force it
|
||||
// for the rest so cleaned URLs keep one shape.
|
||||
if (u.pathname === "") {
|
||||
try {
|
||||
u.pathname = "/";
|
||||
} catch {
|
||||
/* opaque path — leave as parsed */
|
||||
}
|
||||
}
|
||||
|
||||
let kept = "";
|
||||
const query = u.search.slice(1);
|
||||
if (query !== "") {
|
||||
const parts = [];
|
||||
for (const pair of query.split("&")) {
|
||||
if (pair === "") continue;
|
||||
const eq = pair.indexOf("=");
|
||||
const key = (eq === -1 ? pair : pair.slice(0, eq)).toLowerCase();
|
||||
if (key.startsWith("utm_") || EXACT_TRACKING.has(key)) continue;
|
||||
parts.push(pair);
|
||||
}
|
||||
kept = parts.join("&");
|
||||
}
|
||||
|
||||
const hash = u.hash;
|
||||
u.search = "";
|
||||
u.hash = "";
|
||||
return u.toString() + (kept === "" ? "" : "?" + kept) + hash;
|
||||
}
|
||||
|
||||
// --- Substack image helpers (shared by add-url routing and detect-image-source) ---
|
||||
|
||||
const SUBSTACK_IMAGE_HOSTS = new Set([
|
||||
"substackcdn.com",
|
||||
"substack-post-media.s3.amazonaws.com",
|
||||
]);
|
||||
|
||||
const UUID_PATTERN = "[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}";
|
||||
const IMAGE_UUID_RE = new RegExp(`images(?:%2F|/)(${UUID_PATTERN})`, "i");
|
||||
const UUID_EXACT_RE = new RegExp(`^${UUID_PATTERN}$`, "i");
|
||||
|
||||
/**
|
||||
* isSubstackImage reports a Substack-hosted image (CDN wrapper or raw S3),
|
||||
* regardless of file extension.
|
||||
* @param {string} target
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function isSubstackImage(target) {
|
||||
let host = "";
|
||||
try {
|
||||
host = new URL(target).host.toLowerCase();
|
||||
} catch {
|
||||
/* not an absolute URL — fall through to the substring check */
|
||||
}
|
||||
return SUBSTACK_IMAGE_HOSTS.has(host) || target.toLowerCase().includes("substack-post-media");
|
||||
}
|
||||
|
||||
/**
|
||||
* substackImageUuid extracts the stable image identity: the S3 image UUID under
|
||||
* public/images/<uuid>, with raw (/) or percent-encoded (%2F) separators.
|
||||
* @param {string} target
|
||||
* @returns {string} the lowercased uuid, or "" when absent
|
||||
*/
|
||||
export function substackImageUuid(target) {
|
||||
const m = IMAGE_UUID_RE.exec(target);
|
||||
return m === null ? "" : m[1].toLowerCase();
|
||||
}
|
||||
|
||||
/**
|
||||
* isUuid reports whether a bare identity is a Substack image uuid.
|
||||
* @param {string} s
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function isUuid(s) {
|
||||
return UUID_EXACT_RE.test(s);
|
||||
}
|
||||
|
||||
/**
|
||||
* Some sites carry the resource identity in a query param, not the path
|
||||
* (e.g. YouTube /watch?v=ID). Preserve the identity param for those hosts so
|
||||
* dedup does not collapse every video onto the same bare URL.
|
||||
*/
|
||||
const IDENTITY_PARAMS = new Map([
|
||||
["youtube.com", "v"],
|
||||
["www.youtube.com", "v"],
|
||||
["m.youtube.com", "v"],
|
||||
]);
|
||||
|
||||
/**
|
||||
* trimTrailingSlash removes at most one trailing "/".
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
function trimTrailingSlash(s) {
|
||||
return s.endsWith("/") ? s.slice(0, -1) : s;
|
||||
}
|
||||
|
||||
/**
|
||||
* bareUrl reduces a URL to a stable identity for duplicate detection:
|
||||
* - Substack image → its S3 UUID (transform/size variants share one identity)
|
||||
* - YouTube → scheme+host+path + the v= video id
|
||||
* - everything else → scheme + host + path
|
||||
* @param {string} target
|
||||
* @returns {string}
|
||||
*/
|
||||
export function bareUrl(target) {
|
||||
if (isSubstackImage(target)) {
|
||||
const uuid = substackImageUuid(target);
|
||||
if (uuid !== "") return uuid;
|
||||
}
|
||||
let u = null;
|
||||
try {
|
||||
u = new URL(target);
|
||||
} catch {
|
||||
/* unparseable — fall back to crude string surgery below */
|
||||
}
|
||||
if (u === null || !u.protocol || !u.host) {
|
||||
return trimTrailingSlash(target.split("?")[0]);
|
||||
}
|
||||
const host = u.host.toLowerCase();
|
||||
let bare = trimTrailingSlash(u.protocol + "//" + host + u.pathname);
|
||||
const idParam = IDENTITY_PARAMS.get(host);
|
||||
if (idParam !== undefined) {
|
||||
const v = u.searchParams.get(idParam);
|
||||
if (v !== null && v !== "") bare += "?" + idParam + "=" + v;
|
||||
}
|
||||
return bare;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchTextOk GETs target and returns the body on a 2xx response, "" on any
|
||||
* error, non-2xx status, or timeout. Redirects are followed.
|
||||
* @param {string} target
|
||||
* @param {number} timeoutMs
|
||||
* @param {string} [userAgent]
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
export async function fetchTextOk(target, timeoutMs, userAgent = "") {
|
||||
try {
|
||||
const headers = {};
|
||||
if (userAgent !== "") headers["user-agent"] = userAgent;
|
||||
const res = await fetch(target, {
|
||||
headers,
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(timeoutMs),
|
||||
});
|
||||
if (res.status < 200 || res.status >= 300) return "";
|
||||
return await res.text();
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* checkAccessibility HEADs the URL and returns the final HTTP status code as a
|
||||
* string, or "000" on network error / timeout.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
export async function checkAccessibility(target) {
|
||||
try {
|
||||
const res = await fetch(target, {
|
||||
method: "HEAD",
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(10_000),
|
||||
});
|
||||
return String(res.status);
|
||||
} catch {
|
||||
return "000";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* collectMarkdown recursively collects *.md files under dir (the content tree is
|
||||
* small). Entries are visited in lexical order so callers that depend on
|
||||
* first-seen ordering stay deterministic. Missing or unreadable directories
|
||||
* yield nothing.
|
||||
* @param {string} dir
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function collectMarkdown(dir) {
|
||||
/** @type {string[]} */
|
||||
const acc = [];
|
||||
walkMarkdown(dir, acc);
|
||||
return acc;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} dir
|
||||
* @param {string[]} acc
|
||||
* @returns {void}
|
||||
*/
|
||||
function walkMarkdown(dir, acc) {
|
||||
let entries;
|
||||
try {
|
||||
entries = readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return; // skip unreadable entries
|
||||
}
|
||||
entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
|
||||
for (const e of entries) {
|
||||
const p = join(dir, e.name);
|
||||
if (e.isDirectory()) walkMarkdown(p, acc);
|
||||
else if (e.name.toLowerCase().endsWith(".md")) acc.push(p);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* uuidBoundaryOk: the char after a UUID match must not extend the hex id, so
|
||||
* <uuid>.png (cover image), <uuid>_WxH and <uuid>) all match.
|
||||
* @param {string} text
|
||||
* @param {number} end
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function uuidBoundaryOk(text, end) {
|
||||
if (end >= text.length) return true;
|
||||
const c = text[end];
|
||||
return !((c >= "0" && c <= "9") || (c >= "a" && c <= "f") || (c >= "A" && c <= "F"));
|
||||
}
|
||||
|
||||
const URL_DELIMITERS = `)]"'?#<>_&,`;
|
||||
const URL_WHITESPACE = " \t\n\r\f\v";
|
||||
|
||||
/**
|
||||
* urlBoundaryOk: an optional trailing slash (bareUrl strips it, stored URLs may
|
||||
* keep it), then a path/punctuation delimiter, whitespace, or end of text — so
|
||||
* /p/foo does not match a stored /p/foo-bar. The '>' delimiter covers URLs
|
||||
* stored in markdown autolink form <https://…>, which older posts use.
|
||||
* @param {string} text
|
||||
* @param {number} end
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function urlBoundaryOk(text, end) {
|
||||
let at = end;
|
||||
if (at < text.length && text[at] === "/") at++;
|
||||
if (at >= text.length) return true;
|
||||
const c = text[at];
|
||||
if (URL_WHITESPACE.includes(c)) return true;
|
||||
return URL_DELIMITERS.includes(c);
|
||||
}
|
||||
|
||||
/**
|
||||
* hasBoundaryMatch scans every occurrence of needle and applies the boundary
|
||||
* check in code. Kept as an index loop rather than one lookahead regex: the
|
||||
* byte-offset checks (optional trailing slash, delimiter set) are what stop a
|
||||
* needle that is merely a PREFIX of a stored string from matching.
|
||||
* @param {string} text
|
||||
* @param {string} needle
|
||||
* @param {boolean} isUuidNeedle
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function hasBoundaryMatch(text, needle, isUuidNeedle) {
|
||||
for (let from = 0; ; ) {
|
||||
const i = text.indexOf(needle, from);
|
||||
if (i === -1) return false;
|
||||
const end = i + needle.length;
|
||||
if (isUuidNeedle ? uuidBoundaryOk(text, end) : urlBoundaryOk(text, end)) return true;
|
||||
from = i + 1;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* checkDuplicate reports whether a URL identity already exists in the stored
|
||||
* markdown, boundary-aware so a needle that is merely a PREFIX of a stored
|
||||
* longer string is NOT a false duplicate.
|
||||
* @param {string} target
|
||||
* @param {string} dir
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function checkDuplicate(target, dir) {
|
||||
const needle = bareUrl(target);
|
||||
if (needle === "") return false;
|
||||
const uuidNeedle = isUuid(needle);
|
||||
for (const file of collectMarkdown(dir)) {
|
||||
let text;
|
||||
try {
|
||||
text = readFileSync(file, "utf8");
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (text.includes(needle) && hasBoundaryMatch(text, needle, uuidNeedle)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const IMAGE_EXT_RE = /\.(png|jpg|jpeg|gif|webp|svg|avif|heic|heif|bmp|tiff?)(\?.*)?$/;
|
||||
const VIDEO_EXT_RE = /\.(mp4|webm|mov|avi|mkv)(\?.*)?$/;
|
||||
const DOCUMENT_EXT_RE = /\.(pdf|docx?|xlsx?|pptx?)(\?.*)?$/;
|
||||
|
||||
/**
|
||||
* classifyType classifies a URL by file extension.
|
||||
* @param {string} target
|
||||
* @returns {"image"|"video"|"document"|"article"}
|
||||
*/
|
||||
export function classifyType(target) {
|
||||
const lower = target.toLowerCase();
|
||||
if (IMAGE_EXT_RE.test(lower)) return "image";
|
||||
if (VIDEO_EXT_RE.test(lower)) return "video";
|
||||
if (DOCUMENT_EXT_RE.test(lower)) return "document";
|
||||
return "article";
|
||||
}
|
||||
Reference in new issue
Block a user