Files
blog/scripts/newsletter/find-newsletter-number.js
T
tiennm99 64c7b76b4e feat(newsletter): port the engine to JavaScript
Translates the seven-subcommand engine to Node ESM, one module per former Go
file, invoked as `node scripts/newsletter <command>` from the repo root. The Go
implementation stays in place for now so parity can be measured against it.

Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual
string surgery in caption extraction and covers RSS and sitemap XML through
xmlMode without a second parser. fetch-via-defuddle gains a local extraction
stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a
third IP is the whole point when this machine's IP is the blocked one. Both
stages now emit the same YAML frontmatter plus body, and an extraction that
looks like a bot challenge counts as a local failure so the proxy still runs.

Behaviour is preserved where it is load-bearing rather than where it is merely
idiomatic: query strings are rebuilt by string surgery so surviving parameters
keep their original order and encoding, duplicate detection stays an index loop
with byte-offset boundary checks so a prefix of a stored URL is not a false
match, empty optional fields are omitted rather than emitted as "", a deep-crawl
miss reports cutoff as null rather than dropping the key, bullet filtering counts
code points, tag counts keep their six-column alignment, and a malformed percent
sequence falls back to the raw substring.

The printer lives in its own module so a command module never imports the
dispatcher: index.js runs main() at module scope, so that cycle would execute
the CLI as a side effect of any import. Missing dependencies report one
actionable line instead of a module-resolution stack trace, and a reader that
closes early exits quietly instead of raising EPIPE.
2026-09-18 16:23:58 +07:00

80 lines
2.2 KiB
JavaScript

// Find the most recent newsletter number and return the next one.
// Usage: node scripts/newsletter find-newsletter-number
// Outputs: the next newsletter number.
import { readdirSync, readFileSync } from "node:fs";
import { join } from "node:path";
import { contentDir } from "./url-utils.js";
/** Matches the "Newsletter #N" heading a post is numbered by. @type {RegExp} */
export const NEWSLETTER_NUM_RE = /Newsletter\s*#(\d+)/;
const YEAR_DIR_RE = /^\d{4}$/;
const TWO_DIGIT_DIR_RE = /^\d{2}$/;
/**
* listDirsDesc returns dir's subdirectory names matching re, sorted descending.
* @param {string} dir
* @param {RegExp} re
* @returns {string[]}
*/
function listDirsDesc(dir, re) {
let entries;
try {
entries = readdirSync(dir, { withFileTypes: true });
} catch {
return [];
}
return entries
.filter((e) => re.test(e.name))
.map((e) => e.name)
.sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
}
/**
* extractNewsletterNumber reads the first "Newsletter #N" in a post.
* @param {string} path
* @returns {number}
*/
function extractNewsletterNumber(path) {
let content;
try {
content = readFileSync(path, "utf8");
} catch {
return 0;
}
const m = NEWSLETTER_NUM_RE.exec(content);
if (m === null) return 0;
const n = Number.parseInt(m[1], 10);
return Number.isNaN(n) ? 0 : n;
}
/**
* findMostRecentNewsletter scans year/month/day directories newest-first for
* the highest newsletter number.
* @returns {number}
*/
export function findMostRecentNewsletter() {
let maxNumber = 0;
for (const year of listDirsDesc(contentDir(), YEAR_DIR_RE)) {
const yearDir = join(contentDir(), year);
for (const month of listDirsDesc(yearDir, TWO_DIGIT_DIR_RE)) {
const monthDir = join(yearDir, month);
for (const day of listDirsDesc(monthDir, TWO_DIGIT_DIR_RE)) {
const n = extractNewsletterNumber(join(monthDir, day, "index.md"));
if (n > maxNumber) maxNumber = n;
}
}
// Early exit: a newsletter found in this year — no need to go further back.
if (maxNumber > 0) break;
}
return maxNumber;
}
/**
* @param {string[]} _args
* @returns {Promise<void>}
*/
export async function runFindNewsletterNumber(_args) {
process.stdout.write(String(findMostRecentNewsletter() + 1) + "\n");
}