From 64c7b76b4e9e816246d35f5371cd3067bd50108b Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 18 Sep 2026 16:23:58 +0700 Subject: [PATCH] feat(newsletter): port the engine to JavaScript Translates the seven-subcommand engine to Node ESM, one module per former Go file, invoked as `node scripts/newsletter ` from the repo root. The Go implementation stays in place for now so parity can be measured against it. Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual string surgery in caption extraction and covers RSS and sitemap XML through xmlMode without a second parser. fetch-via-defuddle gains a local extraction stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a third IP is the whole point when this machine's IP is the blocked one. Both stages now emit the same YAML frontmatter plus body, and an extraction that looks like a bot challenge counts as a local failure so the proxy still runs. Behaviour is preserved where it is load-bearing rather than where it is merely idiomatic: query strings are rebuilt by string surgery so surviving parameters keep their original order and encoding, duplicate detection stays an index loop with byte-offset boundary checks so a prefix of a stored URL is not a false match, empty optional fields are omitted rather than emitted as "", a deep-crawl miss reports cutoff as null rather than dropping the key, bullet filtering counts code points, tag counts keep their six-column alignment, and a malformed percent sequence falls back to the raw substring. The printer lives in its own module so a command module never imports the dispatcher: index.js runs main() at module scope, so that cycle would execute the CLI as a side effect of any import. Missing dependencies report one actionable line instead of a module-resolution stack trace, and a reader that closes early exits quietly instead of raising EPIPE. --- eslint.config.mjs | 8 +- package-lock.json | 559 +++++++++++++++++++ package.json | 6 +- scripts/newsletter/add-url.js | 127 +++++ scripts/newsletter/detect-image-source.js | 66 +++ scripts/newsletter/fetch-via-defuddle.js | 143 +++++ scripts/newsletter/find-newsletter-number.js | 79 +++ scripts/newsletter/find-substack-post.js | 231 ++++++++ scripts/newsletter/html-text.js | 181 ++++++ scripts/newsletter/index.js | 88 +++ scripts/newsletter/json-out.js | 13 + scripts/newsletter/list-existing-tags.js | 63 +++ scripts/newsletter/post-stats.js | 94 ++++ scripts/newsletter/url-utils.js | 339 +++++++++++ 14 files changed, 1995 insertions(+), 2 deletions(-) create mode 100644 scripts/newsletter/add-url.js create mode 100644 scripts/newsletter/detect-image-source.js create mode 100644 scripts/newsletter/fetch-via-defuddle.js create mode 100644 scripts/newsletter/find-newsletter-number.js create mode 100644 scripts/newsletter/find-substack-post.js create mode 100644 scripts/newsletter/html-text.js create mode 100644 scripts/newsletter/index.js create mode 100644 scripts/newsletter/json-out.js create mode 100644 scripts/newsletter/list-existing-tags.js create mode 100644 scripts/newsletter/post-stats.js create mode 100644 scripts/newsletter/url-utils.js diff --git a/eslint.config.mjs b/eslint.config.mjs index eb6e3e9..eaece68 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -8,7 +8,13 @@ export default [ languageOptions: { ecmaVersion: "latest", sourceType: "module", - globals: { console: "readonly", process: "readonly", fetch: "readonly" }, + globals: { + console: "readonly", + process: "readonly", + fetch: "readonly", + URL: "readonly", + AbortSignal: "readonly", + }, }, rules: { "no-unused-vars": ["error", { argsIgnorePattern: "^_" }], diff --git a/package-lock.json b/package-lock.json index db50195..a1da3b5 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8,6 +8,9 @@ "name": "blog", "version": "0.0.0", "dependencies": { + "cheerio": "^1.2.0", + "defuddle": "^0.19.4", + "linkedom": "^0.18.13", "octokit": "^5.0.5" }, "devDependencies": { @@ -260,6 +263,13 @@ "dev": true, "license": "MIT" }, + "node_modules/@mixmark-io/domino": { + "version": "2.2.0", + "resolved": "https://registry.npmjs.org/@mixmark-io/domino/-/domino-2.2.0.tgz", + "integrity": "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw==", + "license": "BSD-2-Clause", + "optional": true + }, "node_modules/@octokit/app": { "version": "16.1.4", "resolved": "https://registry.npmjs.org/@octokit/app/-/app-16.1.4.tgz", @@ -854,6 +864,15 @@ "dev": true, "license": "MIT" }, + "node_modules/@xmldom/xmldom": { + "version": "0.9.12", + "resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.12.tgz", + "integrity": "sha512-5AXjrcMClTryPe9LgZrygpB1lj7s0S9E0+W+AHaVKAVyHanafK86iPSvG5xHVSp/jC+VH1UXu0TAEmY279xH7A==", + "license": "MIT", + "engines": { + "node": ">=14.6" + } + }, "node_modules/acorn": { "version": "8.18.0", "resolved": "https://registry.npmjs.org/acorn/-/acorn-8.18.0.tgz", @@ -910,6 +929,12 @@ "integrity": "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ==", "license": "Apache-2.0" }, + "node_modules/boolbase": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/boolbase/-/boolbase-1.0.0.tgz", + "integrity": "sha512-JZOSA7Mo9sNGB8+UjSgzdLtokWAky1zbztM3WRLCbZ70/3cTANmQmOdR7y2g+J0e2WXywy1yS468tY+IruqEww==", + "license": "ISC" + }, "node_modules/bottleneck": { "version": "2.19.5", "resolved": "https://registry.npmjs.org/bottleneck/-/bottleneck-2.19.5.tgz", @@ -943,6 +968,57 @@ "qified": "^0.10.1" } }, + "node_modules/cheerio": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/cheerio/-/cheerio-1.2.0.tgz", + "integrity": "sha512-WDrybc/gKFpTYQutKIK6UvfcuxijIZfMfXaYm8NMsPQxSYvf+13fXUJ4rztGGbJcBQ/GF55gvrZ0Bc0bj/mqvg==", + "license": "MIT", + "dependencies": { + "cheerio-select": "^2.1.0", + "dom-serializer": "^2.0.0", + "domhandler": "^5.0.3", + "domutils": "^3.2.2", + "encoding-sniffer": "^0.2.1", + "htmlparser2": "^10.1.0", + "parse5": "^7.3.0", + "parse5-htmlparser2-tree-adapter": "^7.1.0", + "parse5-parser-stream": "^7.1.2", + "undici": "^7.19.0", + "whatwg-mimetype": "^4.0.0" + }, + "engines": { + "node": ">=20.18.1" + }, + "funding": { + "url": "https://github.com/cheeriojs/cheerio?sponsor=1" + } + }, + "node_modules/cheerio-select": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/cheerio-select/-/cheerio-select-2.1.0.tgz", + "integrity": "sha512-9v9kG0LvzrlcungtnJtpGNxY+fzECQKhK4EGJX2vByejiMX84MFNQw4UxPJl3bFbTMw+Dfs37XaIkCwTZfLh4g==", + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0", + "css-select": "^5.1.0", + "css-what": "^6.1.0", + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3", + "domutils": "^3.0.1" + }, + "funding": { + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/commander": { + "version": "12.1.0", + "resolved": "https://registry.npmjs.org/commander/-/commander-12.1.0.tgz", + "integrity": "sha512-Vw8qHK3bZM9y/P10u3Vib8o/DdkvA2OtPtZvD871QKjy74Wj1WSKFILMPRPSdUSx5RFK1arlJzEtA4PkFgnbuA==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, "node_modules/content-type": { "version": "3.1.1", "resolved": "https://registry.npmjs.org/content-type/-/content-type-3.1.1.tgz", @@ -971,6 +1047,40 @@ "node": ">= 8" } }, + "node_modules/css-select": { + "version": "5.2.2", + "resolved": "https://registry.npmjs.org/css-select/-/css-select-5.2.2.tgz", + "integrity": "sha512-TizTzUddG/xYLA3NXodFM0fSbNizXjOKhqiQQwvhlspadZokn1KDy0NZFS0wuEubIYAV5/c1/lAr0TaaFXEXzw==", + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0", + "css-what": "^6.1.0", + "domhandler": "^5.0.2", + "domutils": "^3.0.1", + "nth-check": "^2.0.1" + }, + "funding": { + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/css-what": { + "version": "6.2.2", + "resolved": "https://registry.npmjs.org/css-what/-/css-what-6.2.2.tgz", + "integrity": "sha512-u/O3vwbptzhMs3L1fQE82ZSLHQQfto5gyZzwteVIEyeaY5Fc7R4dapF/BvRoSYFeqfBk4m0V1Vafq5Pjv25wvA==", + "license": "BSD-2-Clause", + "engines": { + "node": ">= 6" + }, + "funding": { + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/cssom": { + "version": "0.5.0", + "resolved": "https://registry.npmjs.org/cssom/-/cssom-0.5.0.tgz", + "integrity": "sha512-iKuQcq+NdHqlAcwUY0o/HL69XQrUaQdMjmStJ8JFmUaiiQErlhrmuigkg/CU4E2J0IyUKUrMAgl36TvN67MqTw==", + "license": "MIT" + }, "node_modules/debug": { "version": "4.4.3", "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", @@ -996,6 +1106,104 @@ "dev": true, "license": "MIT" }, + "node_modules/defuddle": { + "version": "0.19.4", + "resolved": "https://registry.npmjs.org/defuddle/-/defuddle-0.19.4.tgz", + "integrity": "sha512-Jz98aWlyEeuWcg9u2AavCEKknviI9GijMkrGFPk0wi1Lkfr1DDDw6XM+ak44Ol8kBDmDJsvU9Su2eUY09Fvh3Q==", + "license": "MIT", + "dependencies": { + "commander": "^12.1.0", + "mathml-to-latex": "^1.8.0" + }, + "bin": { + "defuddle": "dist/cli.js" + }, + "optionalDependencies": { + "linkedom": "^0.18.12", + "temml": "^0.13.3", + "turndown": "^7.2.0" + } + }, + "node_modules/dom-serializer": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-2.0.0.tgz", + "integrity": "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg==", + "license": "MIT", + "dependencies": { + "domelementtype": "^2.3.0", + "domhandler": "^5.0.2", + "entities": "^4.2.0" + }, + "funding": { + "url": "https://github.com/cheeriojs/dom-serializer?sponsor=1" + } + }, + "node_modules/domelementtype": { + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-2.3.0.tgz", + "integrity": "sha512-OLETBj6w0OsagBwdXnPdN0cnMfF9opN69co+7ZrbfPGrdpPVNBUj02spi6B1N7wChLQiPn4CSH/zJvXw56gmHw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + ], + "license": "BSD-2-Clause" + }, + "node_modules/domhandler": { + "version": "5.0.3", + "resolved": "https://registry.npmjs.org/domhandler/-/domhandler-5.0.3.tgz", + "integrity": "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w==", + "license": "BSD-2-Clause", + "dependencies": { + "domelementtype": "^2.3.0" + }, + "engines": { + "node": ">= 4" + }, + "funding": { + "url": "https://github.com/fb55/domhandler?sponsor=1" + } + }, + "node_modules/domutils": { + "version": "3.2.2", + "resolved": "https://registry.npmjs.org/domutils/-/domutils-3.2.2.tgz", + "integrity": "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw==", + "license": "BSD-2-Clause", + "dependencies": { + "dom-serializer": "^2.0.0", + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3" + }, + "funding": { + "url": "https://github.com/fb55/domutils?sponsor=1" + } + }, + "node_modules/encoding-sniffer": { + "version": "0.2.1", + "resolved": "https://registry.npmjs.org/encoding-sniffer/-/encoding-sniffer-0.2.1.tgz", + "integrity": "sha512-5gvq20T6vfpekVtqrYQsSCFZ1wEg5+wW0/QaZMWkFr6BqD3NfKs0rLCx4rrVlSWJeZb5NBJgVLswK/w2MWU+Gw==", + "license": "MIT", + "dependencies": { + "iconv-lite": "^0.6.3", + "whatwg-encoding": "^3.1.1" + }, + "funding": { + "url": "https://github.com/fb55/encoding-sniffer?sponsor=1" + } + }, + "node_modules/entities": { + "version": "4.5.0", + "resolved": "https://registry.npmjs.org/entities/-/entities-4.5.0.tgz", + "integrity": "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, "node_modules/escape-string-regexp": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz", @@ -1264,6 +1472,55 @@ "dev": true, "license": "MIT" }, + "node_modules/html-escaper": { + "version": "3.0.3", + "resolved": "https://registry.npmjs.org/html-escaper/-/html-escaper-3.0.3.tgz", + "integrity": "sha512-RuMffC89BOWQoY0WKGpIhn5gX3iI54O6nRA0yC124NYVtzjmFWBIiFd8M0x+ZdX0P9R4lADg1mgP8C7PxGOWuQ==", + "license": "MIT" + }, + "node_modules/htmlparser2": { + "version": "10.1.0", + "resolved": "https://registry.npmjs.org/htmlparser2/-/htmlparser2-10.1.0.tgz", + "integrity": "sha512-VTZkM9GWRAtEpveh7MSF6SjjrpNVNNVJfFup7xTY3UpFtm67foy9HDVXneLtFVt4pMz5kZtgNcvCniNFb1hlEQ==", + "funding": [ + "https://github.com/fb55/htmlparser2?sponsor=1", + { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + ], + "license": "MIT", + "dependencies": { + "domelementtype": "^2.3.0", + "domhandler": "^5.0.3", + "domutils": "^3.2.2", + "entities": "^7.0.1" + } + }, + "node_modules/htmlparser2/node_modules/entities": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/entities/-/entities-7.0.1.tgz", + "integrity": "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, + "node_modules/iconv-lite": { + "version": "0.6.3", + "resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.6.3.tgz", + "integrity": "sha512-4fCk79wshMdzMp2rH06qWrJE4iolqLhCUH+OiuIgU++RB0+94NlDL81atO7GX55uUKueo0txHNtvEyI6D7WdMw==", + "license": "MIT", + "dependencies": { + "safer-buffer": ">= 2.1.2 < 3.0.0" + }, + "engines": { + "node": ">=0.10.0" + } + }, "node_modules/ignore": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/ignore/-/ignore-5.3.2.tgz", @@ -1358,6 +1615,171 @@ "node": ">= 0.8.0" } }, + "node_modules/linkedom": { + "version": "0.18.13", + "resolved": "https://registry.npmjs.org/linkedom/-/linkedom-0.18.13.tgz", + "integrity": "sha512-ES/o9qotMpzpN2MHs+Iq/JcVoOj8Fa5wiQYrTdFpvAnwXL0g66XHHUc9WUMk6nAlBtGsFQ24ne+SYnvnaQ2FSw==", + "license": "ISC", + "dependencies": { + "css-select": "^7.0.0", + "cssom": "^0.5.0", + "html-escaper": "^3.0.3", + "htmlparser2": "^10.1.0", + "uhyphen": "^0.2.0" + }, + "engines": { + "node": ">=16" + }, + "peerDependencies": { + "canvas": ">= 2" + }, + "peerDependenciesMeta": { + "canvas": { + "optional": true + } + } + }, + "node_modules/linkedom/node_modules/boolbase": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/boolbase/-/boolbase-2.0.0.tgz", + "integrity": "sha512-DkVaaQHymRhpYEYo9x1oo7Q7B0Y6KJUsjm3c9eTyFDby4MHLBTwZ6ZDWBel5zrYxj1WsZgC5oLpiz+93MluXeA==", + "license": "ISC", + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/linkedom/node_modules/css-select": { + "version": "7.0.0", + "resolved": "https://registry.npmjs.org/css-select/-/css-select-7.0.0.tgz", + "integrity": "sha512-snmjEVXy+1LnwXdxhYvTMj1d9tOh4HxkA1YmoayVBeeyR2C14Pum7fcxJIm4SswYspVy866eYNwlH6xC3/VH5g==", + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^2.0.0", + "css-what": "^8.0.0", + "domhandler": "^6.0.1", + "domutils": "^4.0.2", + "nth-check": "^3.0.1" + }, + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/linkedom/node_modules/css-what": { + "version": "8.0.0", + "resolved": "https://registry.npmjs.org/css-what/-/css-what-8.0.0.tgz", + "integrity": "sha512-DH0Bqq3DNp5tdOReuNyAA+Ev4Y2GS5FMbZpeTLP6C4CDi0h5nL0BmUPChXw3o/qbHLDWHl49sbNqQVY7bMSDdw==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + }, + "node_modules/linkedom/node_modules/dom-serializer": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-3.1.1.tgz", + "integrity": "sha512-4MEa38/QexBob6gFNwu+EGdWvhJ1OKuNwdYY3Y3NyeWDQfnGeDYQUDfIRzWu5B5gsv03so2Uxd28YC6zrsx3Lw==", + "license": "MIT", + "dependencies": { + "domelementtype": "^3.0.0", + "domhandler": "^6.0.0", + "entities": "^8.0.0" + }, + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/cheeriojs/dom-serializer?sponsor=1" + } + }, + "node_modules/linkedom/node_modules/domelementtype": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-3.0.0.tgz", + "integrity": "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/fb55" + } + ], + "license": "BSD-2-Clause", + "engines": { + "node": ">=20.19.0" + } + }, + "node_modules/linkedom/node_modules/domhandler": { + "version": "6.0.1", + "resolved": "https://registry.npmjs.org/domhandler/-/domhandler-6.0.1.tgz", + "integrity": "sha512-gYzvtM72ZtxQO0T048kd6HWSbbGCNOUwcnfQ01cqIJ4X2IYKFFHZ5mKvrQETcFXxsRObZulDaKmy//R7TPtsBg==", + "license": "BSD-2-Clause", + "dependencies": { + "domelementtype": "^3.0.0" + }, + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/fb55/domhandler?sponsor=1" + } + }, + "node_modules/linkedom/node_modules/domutils": { + "version": "4.0.2", + "resolved": "https://registry.npmjs.org/domutils/-/domutils-4.0.2.tgz", + "integrity": "sha512-qI4JLRKnSzqFqr7hAlS5xQDusBCjKSEG4t4+7aNrIQMHBcsC2TGEhuyABJdYkgSewL57PNLYEiibY2iPKhKpaA==", + "license": "BSD-2-Clause", + "dependencies": { + "dom-serializer": "^3.0.0", + "domelementtype": "^3.0.0", + "domhandler": "^6.0.0" + }, + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/fb55/domutils?sponsor=1" + } + }, + "node_modules/linkedom/node_modules/entities": { + "version": "8.1.0", + "resolved": "https://registry.npmjs.org/entities/-/entities-8.1.0.tgz", + "integrity": "sha512-kxL7msIffSuh9aaFAMD7rxAIuTRMAHMeBtgHW2yUdWw732ZNh4MehkF2gdjvtdmikkaIP9bFDDJOPlsvm7avrA==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, + "node_modules/linkedom/node_modules/nth-check": { + "version": "3.0.1", + "resolved": "https://registry.npmjs.org/nth-check/-/nth-check-3.0.1.tgz", + "integrity": "sha512-GX0gsdbGVCgnRgbeGaubfjpBXyYRWOOCVeYh08bSQvDZqxz5ndXs1OTfAt/h36G1xvI94YIspsI0sVFqAV9+RQ==", + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^2.0.0" + }, + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "type": "github", + "url": "https://github.com/fb55/nth-check?sponsor=1" + } + }, "node_modules/locate-path": { "version": "6.0.0", "resolved": "https://registry.npmjs.org/locate-path/-/locate-path-6.0.0.tgz", @@ -1374,6 +1796,15 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/mathml-to-latex": { + "version": "1.8.0", + "resolved": "https://registry.npmjs.org/mathml-to-latex/-/mathml-to-latex-1.8.0.tgz", + "integrity": "sha512-gQ0uK3zqB8HwlfaXJkEL5rgaZNbKUiBMmBP/B/W+b+t6KcseLSuYb1b0BjLgS9ZiQa24ePkqTX8/6FaQuDL7wQ==", + "license": "MIT", + "dependencies": { + "@xmldom/xmldom": "^0.9.10" + } + }, "node_modules/minimatch": { "version": "10.2.6", "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.6.tgz", @@ -1404,6 +1835,18 @@ "dev": true, "license": "MIT" }, + "node_modules/nth-check": { + "version": "2.1.1", + "resolved": "https://registry.npmjs.org/nth-check/-/nth-check-2.1.1.tgz", + "integrity": "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w==", + "license": "BSD-2-Clause", + "dependencies": { + "boolbase": "^1.0.0" + }, + "funding": { + "url": "https://github.com/fb55/nth-check?sponsor=1" + } + }, "node_modules/octokit": { "version": "5.0.5", "resolved": "https://registry.npmjs.org/octokit/-/octokit-5.0.5.tgz", @@ -1476,6 +1919,55 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/parse5": { + "version": "7.3.0", + "resolved": "https://registry.npmjs.org/parse5/-/parse5-7.3.0.tgz", + "integrity": "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==", + "license": "MIT", + "dependencies": { + "entities": "^6.0.0" + }, + "funding": { + "url": "https://github.com/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5-htmlparser2-tree-adapter": { + "version": "7.1.0", + "resolved": "https://registry.npmjs.org/parse5-htmlparser2-tree-adapter/-/parse5-htmlparser2-tree-adapter-7.1.0.tgz", + "integrity": "sha512-ruw5xyKs6lrpo9x9rCZqZZnIUntICjQAd0Wsmp396Ul9lN/h+ifgVV1x1gZHi8euej6wTfpqX8j+BFQxF0NS/g==", + "license": "MIT", + "dependencies": { + "domhandler": "^5.0.3", + "parse5": "^7.0.0" + }, + "funding": { + "url": "https://github.com/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5-parser-stream": { + "version": "7.1.2", + "resolved": "https://registry.npmjs.org/parse5-parser-stream/-/parse5-parser-stream-7.1.2.tgz", + "integrity": "sha512-JyeQc9iwFLn5TbvvqACIF/VXG6abODeB3Fwmv/TGdLk2LfbWkaySGY72at4+Ty7EkPZj854u4CrICqNk2qIbow==", + "license": "MIT", + "dependencies": { + "parse5": "^7.0.0" + }, + "funding": { + "url": "https://github.com/inikulin/parse5?sponsor=1" + } + }, + "node_modules/parse5/node_modules/entities": { + "version": "6.0.1", + "resolved": "https://registry.npmjs.org/entities/-/entities-6.0.1.tgz", + "integrity": "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g==", + "license": "BSD-2-Clause", + "engines": { + "node": ">=0.12" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, "node_modules/path-exists": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/path-exists/-/path-exists-4.0.0.tgz", @@ -1536,6 +2028,12 @@ "dev": true, "license": "MIT" }, + "node_modules/safer-buffer": { + "version": "2.1.2", + "resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz", + "integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==", + "license": "MIT" + }, "node_modules/shebang-command": { "version": "2.0.0", "resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz", @@ -1559,6 +2057,16 @@ "node": ">=8" } }, + "node_modules/temml": { + "version": "0.13.5", + "resolved": "https://registry.npmjs.org/temml/-/temml-0.13.5.tgz", + "integrity": "sha512-aPkDDgunanpLNL0ql32HbolqLep+w8DRcVXRql7rWrMt/PhczdLgL4UBYYVU3BYjCNjkGq4EqwIicu/zWr1iOg==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">=18.13.0" + } + }, "node_modules/toad-cache": { "version": "3.7.4", "resolved": "https://registry.npmjs.org/toad-cache/-/toad-cache-3.7.4.tgz", @@ -1568,6 +2076,20 @@ "node": ">=20" } }, + "node_modules/turndown": { + "version": "7.2.4", + "resolved": "https://registry.npmjs.org/turndown/-/turndown-7.2.4.tgz", + "integrity": "sha512-I8yFsfRzmzK0WV1pNNOA4A7y4RDfFxPRxb3t+e3ui14qSGOxGtiSP6GjeX+Y6CHb7HYaFj7ECUD7VE5kQMZWGQ==", + "license": "MIT", + "optional": true, + "dependencies": { + "@mixmark-io/domino": "^2.2.0" + }, + "engines": { + "node": ">=18", + "npm": ">=9" + } + }, "node_modules/type-check": { "version": "0.4.0", "resolved": "https://registry.npmjs.org/type-check/-/type-check-0.4.0.tgz", @@ -1581,6 +2103,21 @@ "node": ">= 0.8.0" } }, + "node_modules/uhyphen": { + "version": "0.2.0", + "resolved": "https://registry.npmjs.org/uhyphen/-/uhyphen-0.2.0.tgz", + "integrity": "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA==", + "license": "ISC" + }, + "node_modules/undici": { + "version": "7.29.1", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.1.tgz", + "integrity": "sha512-RYONW2MeafgYlkVOKYKkA/Ag7BmXqgIWCa8t1m0JcxrQg9pI9lEqRhAOruOBCbAohOa/gkCF+iPi9hrgvTzu6Q==", + "license": "MIT", + "engines": { + "node": ">=20.18.1" + } + }, "node_modules/universal-github-app-jwt": { "version": "2.2.2", "resolved": "https://registry.npmjs.org/universal-github-app-jwt/-/universal-github-app-jwt-2.2.2.tgz", @@ -1603,6 +2140,28 @@ "punycode": "^2.1.0" } }, + "node_modules/whatwg-encoding": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/whatwg-encoding/-/whatwg-encoding-3.1.1.tgz", + "integrity": "sha512-6qN4hJdMwfYBtE3YBTTHhoeuUrDBPZmbQaxWAqSALV/MeEnR5z1xd8UKud2RAkFoPkmB+hli1TZSnyi84xz1vQ==", + "deprecated": "Use @exodus/bytes instead for a more spec-conformant and faster implementation", + "license": "MIT", + "dependencies": { + "iconv-lite": "0.6.3" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/whatwg-mimetype": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/whatwg-mimetype/-/whatwg-mimetype-4.0.0.tgz", + "integrity": "sha512-QaKxh0eNIi2mE9p2vEdzfagOKHCcj1pJ56EEHGQOVxp8r9/iszLUUV7v89x9O1p/T+NlTM5W7jW6+cz4Fq1YVg==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, "node_modules/which": { "version": "2.0.2", "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", diff --git a/package.json b/package.json index c6a8e1b..f695495 100644 --- a/package.json +++ b/package.json @@ -9,9 +9,13 @@ }, "scripts": { "lint": "eslint .", - "projects:refresh": "node .github/scripts/update-projects-list.js" + "projects:refresh": "node .github/scripts/update-projects-list.js", + "test": "node --test scripts/newsletter/*.test.js" }, "dependencies": { + "cheerio": "^1.2.0", + "defuddle": "^0.19.4", + "linkedom": "^0.18.13", "octokit": "^5.0.5" }, "devDependencies": { diff --git a/scripts/newsletter/add-url.js b/scripts/newsletter/add-url.js new file mode 100644 index 0000000..80e970c --- /dev/null +++ b/scripts/newsletter/add-url.js @@ -0,0 +1,127 @@ +// Meta URL router for the mt-add-url skill — the single entry per URL. +// Usage: node scripts/newsletter add-url "" +// Outputs: JSON { original_url, clean_url, http_status, accessible, +// duplicate, route, title?, author? } +// +// route ∈ youtube | image | video | document | article + +import { printJson } from "./json-out.js"; +import { + checkAccessibility, + checkDuplicate, + classifyType, + cleanUrl, + contentDir, + fetchTextOk, + isSubstackImage, +} from "./url-utils.js"; + +const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]); + +/** + * detectYouTube extracts a video id from the supported URL shapes: + * youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID. + * Playlists/channels are intentionally NOT YouTube routes (fall through to type). + * @param {string} target + * @returns {{isYouTube: boolean, videoId: string}} + */ +export function detectYouTube(target) { + let u; + try { + u = new URL(target); + } catch { + return { isYouTube: false, videoId: "" }; + } + const host = u.host.toLowerCase(); + if (host === "youtu.be") { + const id = u.pathname.replace(/^\//, "").split("/")[0]; + return { isYouTube: id !== "", videoId: id }; + } + if (YT_HOSTS.has(host)) { + if (u.pathname === "/watch") { + const id = u.searchParams.get("v") ?? ""; + return { isYouTube: id !== "", videoId: id }; + } + if (u.pathname.startsWith("/shorts/")) { + const parts = u.pathname.split("/"); + if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] }; + } + } + return { isYouTube: false, videoId: "" }; +} + +/** + * canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes. + * @param {string} videoId + * @returns {string} + */ +function canonicalWatchUrl(videoId) { + return "https://www.youtube.com/watch?v=" + videoId; +} + +/** + * fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key). + * Best-effort: any failure returns empty strings so the route stays `youtube` + * and the skill can fall back. + * @param {string} watchUrl + * @returns {Promise<{title: string, author: string}>} + */ +async function fetchYouTubeMeta(watchUrl) { + const endpoint = + "https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json"; + const body = await fetchTextOk(endpoint, 10_000); + if (body === "") return { title: "", author: "" }; + try { + const data = JSON.parse(body); + return { title: data.title ?? "", author: data.author_name ?? "" }; + } catch { + return { title: "", author: "" }; + } +} + +/** + * @param {string[]} args + * @returns {Promise} + */ +export async function runAddUrl(args) { + if (args.length < 1 || args[0] === "") { + process.stderr.write("Usage: node scripts/newsletter add-url \n"); + process.exit(1); + } + const target = args[0]; + + const cleaned = cleanUrl(target); + const { isYouTube, videoId } = detectYouTube(cleaned); + + // For YouTube, dedup/store against the canonical watch URL so youtu.be and + // shorts links collapse onto the same identity-param key as watch URLs. + const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned; + + // Route order: YouTube → Substack image (by host, not extension, so f_auto / + // .avif / .heic / extensionless CDN URLs still route to the image handler) → + // file-extension classification. + let route; + if (isYouTube) route = "youtube"; + else if (isSubstackImage(cleaned)) route = "image"; + else route = classifyType(cleaned); + + const httpStatus = await checkAccessibility(cleaned); + + // title/author are omitted entirely when empty, not emitted as "": the + // handlers treat a present key as "metadata was resolved". + /** @type {Record} */ + const out = { + original_url: target, + clean_url: effectiveUrl, + http_status: httpStatus, + accessible: httpStatus === "200", + duplicate: checkDuplicate(effectiveUrl, contentDir()), + route, + }; + if (route === "youtube") { + const { title, author } = await fetchYouTubeMeta(effectiveUrl); + if (title !== "") out.title = title; + if (author !== "") out.author = author; + } + printJson(out); +} diff --git a/scripts/newsletter/detect-image-source.js b/scripts/newsletter/detect-image-source.js new file mode 100644 index 0000000..ef39762 --- /dev/null +++ b/scripts/newsletter/detect-image-source.js @@ -0,0 +1,66 @@ +// Detect whether an image URL is Substack-hosted and extract its S3 image UUID. +// Usage: node scripts/newsletter detect-image-source "" +// Output: JSON { original_url, clean_url, isSubstack, uuid?, innerUrl? } +// +// Substack images are usually served via a CDN wrapper: +// +// https://substackcdn.com/image/fetch/$s_!x!,.../https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F_WxH.png +// +// The publication is NOT encoded in the URL — only the image identity (uuid) is. + +import { printJson } from "./json-out.js"; +import { cleanUrl, isSubstackImage, substackImageUuid } from "./url-utils.js"; + +/** + * extractInnerUrl pulls the inner S3 URL out of a substackcdn /image/fetch/ + * wrapper (if present). A malformed percent sequence falls back to the raw + * substring rather than throwing. + * @param {string} target + * @returns {string} + */ +export function extractInnerUrl(target) { + const marker = target.indexOf("/https%3A%2F%2F"); + if (marker !== -1) { + const raw = target.slice(marker + 1); + try { + return decodeURIComponent(raw); + } catch { + return raw; + } + } + // Some forms embed a plain (already-decoded) inner https URL. + if (target.length > 8) { + const plain = target.slice(8).indexOf("/https://"); + if (plain !== -1) return target.slice(8 + plain + 1); + } + return target; +} + +/** + * @param {string[]} args + * @returns {Promise} + */ +export async function runDetectImageSource(args) { + if (args.length < 1 || args[0] === "") { + process.stderr.write("Usage: node scripts/newsletter detect-image-source \n"); + process.exit(1); + } + const target = args[0]; + const isSubstack = isSubstackImage(target); + + // Empty optional fields are omitted, not emitted as "": the skills branch on + // the key being present. + /** @type {Record} */ + const out = { + original_url: target, + clean_url: cleanUrl(target), + isSubstack, + }; + if (isSubstack) { + const uuid = substackImageUuid(target); + const innerUrl = extractInnerUrl(target); + if (uuid !== "") out.uuid = uuid; + if (innerUrl !== "") out.innerUrl = innerUrl; + } + printJson(out); +} diff --git a/scripts/newsletter/fetch-via-defuddle.js b/scripts/newsletter/fetch-via-defuddle.js new file mode 100644 index 0000000..f4a0ea4 --- /dev/null +++ b/scripts/newsletter/fetch-via-defuddle.js @@ -0,0 +1,143 @@ +// mt-fetch-url fallback fetcher. +// Usage: node scripts/newsletter fetch-via-defuddle +// Exit codes: 0 = content returned, 1 = every tier failed, 2 = bad arguments. +// +// Two tiers, tried in order: +// 1. local defuddle — extracts on this machine, so the chain no longer depends +// on a single third-party service being reachable. +// 2. the defuddle.md proxy — fetches from a third IP, which is the point when +// this machine's IP is the one being blocked. +// +// Both tiers emit YAML frontmatter followed by the markdown body, so callers +// parse one shape regardless of which tier answered. + +const BROWSER_UA = + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"; + +// A bot wall answers 200 with a real body, so a non-empty extraction is not by +// itself a success. Recognising the wall is what lets the proxy tier — which +// fetches from a different IP — still get its turn. +const CHALLENGE_MARKERS = [ + /just a moment/i, + /checking your browser/i, + /attention required/i, + /cloudflare/i, + /enable javascript (and cookies )?to continue/i, + /verify (that )?you('re| are) (a )?human/i, + /are you a robot/i, + /access denied/i, + /captcha/i, +]; + +/** + * looksLikeChallenge reports whether an extraction is a bot wall rather than the + * page that was asked for. + * @param {string} title + * @param {string} body + * @returns {boolean} + */ +export function looksLikeChallenge(title, body) { + // Only the opening of the body: an article may legitimately discuss Cloudflare. + const sample = title + "\n" + body.slice(0, 400); + return CHALLENGE_MARKERS.some((re) => re.test(sample)); +} + +/** + * yamlFrontmatter renders the metadata block, omitting fields with no value. + * @param {Record} fields + * @returns {string} + */ +function yamlFrontmatter(fields) { + const lines = Object.entries(fields) + .filter(([, v]) => typeof v === "string" && v !== "") + .map(([k, v]) => `${k}: ${JSON.stringify(v)}`); + return lines.length === 0 ? "" : "---\n" + lines.join("\n") + "\n---\n\n"; +} + +/** + * fetchLocally extracts article content with defuddle running in-process, and + * renders it in the same frontmatter-plus-body shape the proxy returns. + * @param {string} target + * @returns {Promise} the document, or "" when extraction yields nothing usable + */ +async function fetchLocally(target) { + const { Defuddle } = await import("defuddle/node"); + const { parseHTML } = await import("linkedom"); + const res = await fetch(target, { + headers: { "user-agent": BROWSER_UA }, + redirect: "follow", + signal: AbortSignal.timeout(30_000), + }); + if (!res.ok) throw new Error(`upstream returned ${res.status}`); + const html = await res.text(); + const { document } = parseHTML(html); + const result = await Defuddle(document, target, { markdown: true }); + const body = String(result?.content ?? ""); + if (body.trim() === "") return ""; + const title = String(result?.title ?? ""); + if (looksLikeChallenge(title, body)) { + throw new Error("extraction looks like a bot challenge, not the page"); + } + const frontmatter = yamlFrontmatter({ + title, + author: String(result?.author ?? ""), + description: String(result?.description ?? ""), + site: String(result?.site ?? ""), + published: String(result?.published ?? ""), + source: target, + }); + return frontmatter + body; +} + +/** + * fetchViaProxy fetches through defuddle.md, which resolves the page from its + * own IP. + * @param {string} target + * @returns {Promise} the response body + */ +async function fetchViaProxy(target) { + const res = await fetch("https://defuddle.md/" + target, { + headers: { "user-agent": "mt-fetch-url/1.0" }, + redirect: "follow", + signal: AbortSignal.timeout(30_000), + }); + const body = await res.text(); + if (!res.ok || body.trim() === "") { + throw new Error(`defuddle returned ${res.status} / empty body`); + } + return body; +} + +/** + * @param {string[]} args + * @returns {Promise} + */ +export async function runFetchViaDefuddle(args) { + if (args.length < 1 || args[0] === "") { + process.stderr.write("Usage: node scripts/newsletter fetch-via-defuddle \n"); + process.exit(2); + } + const target = args[0]; + + try { + const doc = await fetchLocally(target); + if (doc !== "") { + process.stdout.write(doc.endsWith("\n") ? doc : doc + "\n"); + return; + } + process.stderr.write(`mt-fetch-url: local defuddle extracted nothing for ${target}\n`); + } catch (err) { + process.stderr.write( + `mt-fetch-url: local defuddle failed for ${target}: ${String(err?.message ?? err)}\n`, + ); + } + + try { + process.stdout.write(await fetchViaProxy(target)); + } catch (err) { + process.stderr.write( + `mt-fetch-url: defuddle.md failed for ${target}: ${String(err?.message ?? err)}\n`, + ); + process.exit(1); + } +} diff --git a/scripts/newsletter/find-newsletter-number.js b/scripts/newsletter/find-newsletter-number.js new file mode 100644 index 0000000..9edbe07 --- /dev/null +++ b/scripts/newsletter/find-newsletter-number.js @@ -0,0 +1,79 @@ +// Find the most recent newsletter number and return the next one. +// Usage: node scripts/newsletter find-newsletter-number +// Outputs: the next newsletter number. + +import { readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; +import { contentDir } from "./url-utils.js"; + +/** Matches the "Newsletter #N" heading a post is numbered by. @type {RegExp} */ +export const NEWSLETTER_NUM_RE = /Newsletter\s*#(\d+)/; +const YEAR_DIR_RE = /^\d{4}$/; +const TWO_DIGIT_DIR_RE = /^\d{2}$/; + +/** + * listDirsDesc returns dir's subdirectory names matching re, sorted descending. + * @param {string} dir + * @param {RegExp} re + * @returns {string[]} + */ +function listDirsDesc(dir, re) { + let entries; + try { + entries = readdirSync(dir, { withFileTypes: true }); + } catch { + return []; + } + return entries + .filter((e) => re.test(e.name)) + .map((e) => e.name) + .sort((a, b) => (a < b ? 1 : a > b ? -1 : 0)); +} + +/** + * extractNewsletterNumber reads the first "Newsletter #N" in a post. + * @param {string} path + * @returns {number} + */ +function extractNewsletterNumber(path) { + let content; + try { + content = readFileSync(path, "utf8"); + } catch { + return 0; + } + const m = NEWSLETTER_NUM_RE.exec(content); + if (m === null) return 0; + const n = Number.parseInt(m[1], 10); + return Number.isNaN(n) ? 0 : n; +} + +/** + * findMostRecentNewsletter scans year/month/day directories newest-first for + * the highest newsletter number. + * @returns {number} + */ +export function findMostRecentNewsletter() { + let maxNumber = 0; + for (const year of listDirsDesc(contentDir(), YEAR_DIR_RE)) { + const yearDir = join(contentDir(), year); + for (const month of listDirsDesc(yearDir, TWO_DIGIT_DIR_RE)) { + const monthDir = join(yearDir, month); + for (const day of listDirsDesc(monthDir, TWO_DIGIT_DIR_RE)) { + const n = extractNewsletterNumber(join(monthDir, day, "index.md")); + if (n > maxNumber) maxNumber = n; + } + } + // Early exit: a newsletter found in this year — no need to go further back. + if (maxNumber > 0) break; + } + return maxNumber; +} + +/** + * @param {string[]} _args + * @returns {Promise} + */ +export async function runFindNewsletterNumber(_args) { + process.stdout.write(String(findMostRecentNewsletter() + 1) + "\n"); +} diff --git a/scripts/newsletter/find-substack-post.js b/scripts/newsletter/find-substack-post.js new file mode 100644 index 0000000..bee0ce7 --- /dev/null +++ b/scripts/newsletter/find-substack-post.js @@ -0,0 +1,231 @@ +// Find which Substack post embeds a given image UUID, and extract a label. +// Usage: node scripts/newsletter find-substack-post --uuid [--deep] +// Output on hit: JSON { found:true, source, publication, postTitle, postUrl, caption, candidates } +// Output on miss: JSON { found:false } (RSS) or { found:false, source:"sitemap", scanned, budget, cutoff } (--deep) +// +// Strategy: RSS feed first (fast, ~recent weeks). With --deep, fall back to a +// heavier sitemap crawl up to ~3 months back — opt-in because it fetches many +// posts. A Substack CDN URL does not encode its publication, so we search each +// publication listed in config/substack-publications.json, read at runtime so +// editing the JSON takes effect immediately. + +import { readFileSync } from "node:fs"; +import { parseArgs } from "node:util"; +import { printJson } from "./json-out.js"; +import { fetchTextOk } from "./url-utils.js"; +import { + captionForUuid, + extractCandidates, + itemLink, + itemTitle, + loadXml, + postTitleFromHtml, +} from "./html-text.js"; + +/** Total post fetches allowed across ALL publications during a --deep crawl. */ +const DEEP_FETCH_BUDGET = 40; + +/** + * loadPublications reads the publication list, falling back to the default on + * any read or parse failure. + * @returns {string[]} + */ +export function loadPublications() { + try { + const raw = readFileSync(new URL("./config/substack-publications.json", import.meta.url), "utf8"); + const pubs = JSON.parse(raw); + if (Array.isArray(pubs) && pubs.length > 0) return pubs; + } catch { + /* fall through */ + } + return ["blog.bytebytego.com"]; +} + +/** + * fetchPage: body text with a browser-ish UA, or "" on any error. + * @param {string} target + * @returns {Promise} + */ +function fetchPage(target) { + return fetchTextOk(target, 10_000, "Mozilla/5.0"); +} + +const RFC3339_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$/i; +const DATE_ONLY_RE = /^\d{4}-\d{2}-\d{2}$/; + +/** + * parseLastmod accepts only the two layouts the Go engine accepted — RFC3339 + * and a bare date — so a loosely formatted stamp is skipped rather than + * silently reinterpreted in local time. + * @param {string} s + * @returns {Date|null} + */ +export function parseLastmod(s) { + if (!RFC3339_RE.test(s) && !DATE_ONLY_RE.test(s)) return null; + const t = new Date(s); + return Number.isNaN(t.getTime()) ? null : t; +} + +/** + * cutoffDate returns the ~3-months-back boundary for the deep crawl. + * @returns {Date} + */ +function cutoffDate() { + const d = new Date(); + d.setUTCMonth(d.getUTCMonth() - 3); + return d; +} + +/** + * @typedef {{found: true, source: string, publication: string, postTitle: string, + * postUrl: string, caption: string, candidates: string[]}} PostHit + */ + +/** + * searchSitemap is the deep fallback: crawl the sitemap back ~3 months, fetch + * posts most-recent-first (up to maxFetch from the shared budget), and look for + * the UUID. Heavier than RSS — only used on RSS miss. + * @param {string} publication + * @param {string} id + * @param {number} maxFetch + * @returns {Promise<{hit: PostHit|null, scanned: number, cutoff: string}>} cutoff is "" when the sitemap itself could not be fetched + */ +async function searchSitemap(publication, id, maxFetch) { + const xml = await fetchPage("https://" + publication + "/sitemap.xml"); + if (xml === "") return { hit: null, scanned: 0, cutoff: "" }; + + const cutoffTime = cutoffDate(); + const cutoff = cutoffTime.toISOString().slice(0, 10); + + const $ = loadXml(xml); + /** @type {{url: string, when: Date}[]} */ + const candidates = []; + $("url").each((_i, el) => { + const loc = $(el).children("loc").first().text(); + const lastmod = $(el).children("lastmod").first().text(); + if (loc === "" || lastmod === "") return; + if (!loc.includes("/p/")) return; // posts only + const when = parseLastmod(lastmod); + if (when === null || when < cutoffTime) return; + candidates.push({ url: loc, when }); + }); + candidates.sort((a, b) => b.when.getTime() - a.when.getTime()); + + let scanned = 0; + for (const c of candidates.slice(0, maxFetch)) { + scanned++; + const html = await fetchPage(c.url); + if (html === "" || !html.includes(id)) continue; + return { + hit: { + found: true, + source: "sitemap", + publication, + postTitle: postTitleFromHtml(html), + postUrl: c.url, + caption: captionForUuid(html, id), + candidates: extractCandidates(html), + }, + scanned, + cutoff, + }; + } + return { hit: null, scanned, cutoff }; +} + +/** + * searchRss looks for the uuid in a publication's recent feed items. + * @param {string} publication + * @param {string} id + * @returns {Promise} + */ +async function searchRss(publication, id) { + const xml = await fetchPage("https://" + publication + "/feed"); + if (xml === "") return null; + const $ = loadXml(xml); + const items = $("item").toArray(); + for (const el of items) { + const item = $.html(el); + if (!item.includes(id)) continue; + return { + found: true, + source: "rss", + publication, + postTitle: itemTitle(item), + postUrl: itemLink(item), + caption: captionForUuid(item, id), + candidates: extractCandidates(item), + }; + } + return null; +} + +/** + * @param {string[]} args + * @returns {Promise} + */ +export async function runFindSubstackPost(args) { + let values; + try { + ({ values } = parseArgs({ + args, + options: { uuid: { type: "string" }, deep: { type: "boolean" } }, + allowPositionals: false, + })); + } catch (err) { + process.stderr.write(String(err?.message ?? err) + "\n"); + process.stderr.write( + "Usage: node scripts/newsletter find-substack-post --uuid [--deep]\n", + ); + process.exit(2); + } + + const uuid = values.uuid ?? ""; + if (uuid === "") { + process.stderr.write( + "Usage: node scripts/newsletter find-substack-post --uuid [--deep]\n", + ); + process.exit(1); + } + + const publications = loadPublications(); + for (const pub of publications) { + const hit = await searchRss(pub, uuid); + if (hit !== null) { + printJson(hit); + return; + } + } + + // Deep fallback: sitemap crawl up to ~3 months back, sharing one global fetch + // budget across all publications so coverage can't blow up as the + // publications list grows. + if (values.deep === true) { + let totalScanned = 0; + /** @type {string|null} */ + let lastCutoff = null; + for (const pub of publications) { + const remaining = DEEP_FETCH_BUDGET - totalScanned; + if (remaining <= 0) break; + const { hit, scanned, cutoff } = await searchSitemap(pub, uuid, remaining); + totalScanned += scanned; + if (cutoff !== "") lastCutoff = cutoff; + if (hit !== null) { + printJson(hit); + return; + } + } + // cutoff is null (not omitted) when no sitemap could be fetched — the skill + // distinguishes "crawled and missed" from "could not crawl". + printJson({ + found: false, + source: "sitemap", + scanned: totalScanned, + budget: DEEP_FETCH_BUDGET, + cutoff: lastCutoff, + }); + return; + } + + printJson({ found: false }); +} diff --git a/scripts/newsletter/html-text.js b/scripts/newsletter/html-text.js new file mode 100644 index 0000000..cbc82bd --- /dev/null +++ b/scripts/newsletter/html-text.js @@ -0,0 +1,181 @@ +// HTML / RSS text-extraction helpers for find-substack-post, ported from +// html_text.go. Pure string functions — no network, no fs. A real parser +// (cheerio) replaces the hand-rolled regexes the Go version used. + +import * as cheerio from "cheerio"; + +/** + * loadXml parses an XML fragment (RSS item, sitemap) with CDATA recognition. + * @param {string} src + * @returns {cheerio.CheerioAPI} + */ +export function loadXml(src) { + return cheerio.load(src, { xmlMode: true }, false); +} + +/** + * loadHtml parses an HTML blob. CDATA markers are stripped first: RSS carries + * post HTML inside CDATA, and the HTML parser would otherwise swallow it as a + * bogus comment. + * @param {string} src + * @returns {cheerio.CheerioAPI} + */ +export function loadHtml(src) { + return cheerio.load(src.split("").join("")); +} + +/** + * rawInner returns an element's serialized inner content with any CDATA wrapper + * removed — the same substring the Go regexes captured, so entity handling can + * stay identical instead of being applied twice. + * @param {cheerio.CheerioAPI} $ + * @param {cheerio.Cheerio} el + * @returns {string} + */ +function rawInner($, el) { + const html = $.html(el); + const start = html.indexOf(">") + 1; + const end = html.lastIndexOf("")) inner = inner.slice(0, -3); + return inner; +} + +/** + * decodeEntities decodes HTML entities (named, numeric, hex) and trims. + * @param {string} s + * @returns {string} + */ +export function decodeEntities(s) { + if (s === "") return ""; + return cheerio.load(s, null, false).text().trim(); +} + +/** + * collapseWhitespace squeezes runs of whitespace to one space and trims. + * @param {string} s + * @returns {string} + */ +function collapseWhitespace(s) { + return s.replace(/\s+/g, " ").trim(); +} + +/** + * stripTags removes markup, decodes entities, and collapses whitespace. + * @param {string} s + * @returns {string} + */ +export function stripTags(s) { + if (s === "") return ""; + return collapseWhitespace(cheerio.load(s, null, false).text()); +} + +/** + * itemTitle pulls the first (CDATA or plain) from an RSS <item> chunk. + * @param {string} item + * @returns {string} + */ +export function itemTitle(item) { + const $ = loadXml(item); + const el = $("title").first(); + if (el.length === 0) return ""; + return decodeEntities(rawInner($, el)); +} + +/** + * itemLink pulls the first <link> (CDATA or plain) from an RSS <item> chunk. + * Entities are left as stored — the link goes straight into a post. + * @param {string} item + * @returns {string} + */ +export function itemLink(item) { + const $ = loadXml(item); + const el = $("link").first(); + if (el.length === 0) return ""; + return rawInner($, el).trim(); +} + +/** + * extractCandidates pulls candidate topic titles from a post's TOC bullet list. + * ByteByteGo does not attach captions to images — the topic titles live only in + * the "in this issue" bullets, and image→title cannot be mapped automatically + * (sponsor/video items interleave), so these are surfaced for the user to pick + * from. Light filtering keeps the list short: dedupe, drop sub-point + * explanations and over-long lines. + * @param {string} htmlSrc + * @returns {string[]} + */ +export function extractCandidates(htmlSrc) { + const $ = loadHtml(htmlSrc); + /** @type {Set<string>} */ + const seen = new Set(); + /** @type {string[]} */ + const out = []; + $("li").each((_i, el) => { + const text = collapseWhitespace($(el).text()); + if (text === "") return; + // Titles are short; long lines are sub-point explanations. Count code + // points, not UTF-16 units, so an emoji does not count double. + const n = [...text].length; + if (n < 6 || n > 70) return; + const key = text.toLowerCase(); + if (seen.has(key)) return; // content is duplicated in the page + seen.add(key); + out.push(text); + }); + return out; +} + +/** + * captionForUuid returns the <figcaption> text of the <figure> containing the + * UUID. Cover images live in <enclosure> (no figure) → "". + * @param {string} src + * @param {string} id + * @returns {string} + */ +export function captionForUuid(src, id) { + if (id === "") return ""; + const $ = loadHtml(src); + let target = null; + $("*").each((_i, el) => { + if (target !== null) return false; + for (const v of Object.values(el.attribs ?? {})) { + if (typeof v === "string" && v.includes(id)) { + target = el; + return false; + } + } + for (const child of el.children ?? []) { + if (child.type === "text" && String(child.data).includes(id)) { + target = el; + return false; + } + } + return undefined; + }); + if (target === null) return ""; + const figure = $(target).closest("figure"); + if (figure.length === 0) return ""; + const caption = figure.find("figcaption").first(); + if (caption.length === 0) return ""; + return collapseWhitespace(caption.text()); +} + +/** + * postTitleFromHtml extracts a post title from server-rendered post HTML + * (og:title preferred, then <h1>, then <title>). + * @param {string} htmlSrc + * @returns {string} + */ +export function postTitleFromHtml(htmlSrc) { + const $ = loadHtml(htmlSrc); + const og = $('meta[property="og:title"]').first().attr("content"); + if (og !== undefined && og !== "") return og.trim(); + const h1 = $("h1").first(); + if (h1.length > 0) return collapseWhitespace(h1.text()); + const title = $("title").first(); + if (title.length > 0) return title.text().trim(); + return ""; +} diff --git a/scripts/newsletter/index.js b/scripts/newsletter/index.js new file mode 100644 index 0000000..7950b6f --- /dev/null +++ b/scripts/newsletter/index.js @@ -0,0 +1,88 @@ +// Newsletter engine for the mt-* skills — one entry point, one subcommand per +// task. Invoked from the repo root: +// +// node scripts/newsletter <command> [args] +// +// Repo-relative paths (content/post) resolve from the working directory, so the +// repo-root invocation contract from AGENTS.md still applies. + +const USAGE = `Usage: node scripts/newsletter <command> [args] + +Commands: + add-url <url> classify + dedup a URL, emit JSON route + find-newsletter-number print the next newsletter number + list-existing-tags tag frequencies, most-used first (top 40) + detect-image-source <url> detect Substack image + uuid + find-substack-post --uuid <uuid> [--deep] find the post embedding an image uuid + fetch-via-defuddle <url> fallback fetch (local defuddle, then defuddle.md) + post-stats <path/to/index.md> count the post's articles/images/videos/documents +`; + +// A reader that closes early (`… | head -3`) makes the next write fail with +// EPIPE, which Node surfaces as an unhandled error event. Stop quietly instead +// of printing a stack trace over the user's terminal. +process.stdout.on("error", (err) => { + if (err.code === "EPIPE") process.exit(0); + throw err; +}); + +/** @returns {void} */ +function usage() { + process.stderr.write(USAGE); +} + +/** + * Command modules are imported lazily so a missing node_modules reports the one + * actionable fix instead of a module-resolution stack trace. + * @param {string} spec + * @returns {Promise<Record<string, any>>} + */ +async function loadCommand(spec) { + try { + return await import(spec); + } catch (err) { + if (err !== null && typeof err === "object" && err.code === "ERR_MODULE_NOT_FOUND") { + process.stderr.write( + "newsletter engine: dependencies missing — run 'npm ci' from the repo root\n", + ); + process.exit(1); + } + throw err; + } +} + +/** @type {Record<string, {module: string, fn: string}>} */ +const COMMANDS = { + "add-url": { module: "./add-url.js", fn: "runAddUrl" }, + "find-newsletter-number": { module: "./find-newsletter-number.js", fn: "runFindNewsletterNumber" }, + "list-existing-tags": { module: "./list-existing-tags.js", fn: "runListExistingTags" }, + "detect-image-source": { module: "./detect-image-source.js", fn: "runDetectImageSource" }, + "find-substack-post": { module: "./find-substack-post.js", fn: "runFindSubstackPost" }, + "fetch-via-defuddle": { module: "./fetch-via-defuddle.js", fn: "runFetchViaDefuddle" }, + "post-stats": { module: "./post-stats.js", fn: "runPostStats" }, +}; + +/** @returns {Promise<void>} */ +async function main() { + const argv = process.argv.slice(2); + if (argv.length < 1) { + usage(); + process.exit(1); + } + const [name, ...args] = argv; + const entry = COMMANDS[name]; + if (entry === undefined) { + process.stderr.write(`unknown command: ${name}\n`); + usage(); + process.exit(1); + } + const mod = await loadCommand(entry.module); + await mod[entry.fn](args); +} + +try { + await main(); +} catch (err) { + process.stderr.write("newsletter engine: " + String(err?.stack ?? err) + "\n"); + process.exit(1); +} diff --git a/scripts/newsletter/json-out.js b/scripts/newsletter/json-out.js new file mode 100644 index 0000000..1e137a9 --- /dev/null +++ b/scripts/newsletter/json-out.js @@ -0,0 +1,13 @@ +// The shared JSON printer. It lives in its own module rather than in index.js so +// that importing a command module never pulls in the dispatcher — an import +// cycle there would run the CLI as a side effect of any import. + +/** + * printJson mirrors console.log(JSON.stringify(v, null, 2)): 2-space indent, no + * HTML escaping (URLs with & must stay readable), trailing newline. + * @param {unknown} v + * @returns {void} + */ +export function printJson(v) { + process.stdout.write(JSON.stringify(v, null, 2) + "\n"); +} diff --git a/scripts/newsletter/list-existing-tags.js b/scripts/newsletter/list-existing-tags.js new file mode 100644 index 0000000..fcd689a --- /dev/null +++ b/scripts/newsletter/list-existing-tags.js @@ -0,0 +1,63 @@ +// List existing tags in the repo ranked by frequency. +// Usage: node scripts/newsletter list-existing-tags +// Outputs: tag count and name, sorted most-used first (top 40). +// +// NOTE: not currently used by the mt-add-tags skill. Tag normalization is +// disabled until existing posts have standardized tags; to enable, uncomment +// step 4a in that skill's SKILL.md. + +import { readFileSync } from "node:fs"; +import { basename } from "node:path"; +import { collectMarkdown, contentDir } from "./url-utils.js"; + +const TAGS_LINE_RE = /^tags:\s*\[([^\]]*)\]/m; +const QUOTED_TAG_RE = /"([^"]+)"/g; + +/** + * extractTags pulls quoted tag strings from an index.md frontmatter tags array. + * @param {string} path + * @returns {string[]} + */ +function extractTags(path) { + let content; + try { + content = readFileSync(path, "utf8"); + } catch { + return []; + } + const m = TAGS_LINE_RE.exec(content); + if (m === null) return []; + return [...m[1].matchAll(QUOTED_TAG_RE)].map((q) => q[1]); +} + +/** + * @param {string[]} _args + * @returns {Promise<void>} + */ +export async function runListExistingTags(_args) { + // Array + index map keeps first-seen order for equal counts, so the stable + // sort below ranks ties deterministically (walk order is lexical). + /** @type {{tag: string, count: number}[]} */ + const counts = []; + /** @type {Map<string, number>} */ + const index = new Map(); + + for (const path of collectMarkdown(contentDir())) { + if (basename(path) !== "index.md") continue; + for (const tag of extractTags(path)) { + const at = index.get(tag); + if (at !== undefined) counts[at].count++; + else { + index.set(tag, counts.length); + counts.push({ tag, count: 1 }); + } + } + } + + counts.sort((a, b) => b.count - a.count); + // Plain text, not JSON: the count is right-aligned in a 6-character field and + // the skill reads that layout. + for (const { tag, count } of counts.slice(0, 40)) { + process.stdout.write(String(count).padStart(6) + " " + tag + "\n"); + } +} diff --git a/scripts/newsletter/post-stats.js b/scripts/newsletter/post-stats.js new file mode 100644 index 0000000..ed37797 --- /dev/null +++ b/scripts/newsletter/post-stats.js @@ -0,0 +1,94 @@ +// Count the entries already present in a newsletter post, so a handler can +// report a running tally after each insertion. +// Usage: node scripts/newsletter post-stats <path/to/index.md> +// Outputs: JSON { post, newsletter, articles, images, videos, documents, total } + +import { readFileSync } from "node:fs"; +import { printJson } from "./json-out.js"; +import { NEWSLETTER_NUM_RE } from "./find-newsletter-number.js"; + +// Entry shapes, per the Bonus format in the shared post mechanics: +// +// articles "## [Title](url)" (level-2 heading, main content) +// images "![label](url)" (under **Images:**) +// videos "[Title](url)" (under **Videos:**) +// documents "[PDF: title](url)" (under **Documents:**) +const ARTICLE_HEADING_RE = /^##\s+\[/; +const BONUS_HEADING_RE = /^###\s+Bonus\b/; +const SUBSECTION_RE = /^\*\*(Images|Videos|Documents):\*\*/; +const IMAGE_ENTRY_RE = /^!\[/; +const LINK_ENTRY_RE = /^\[/; + +/** + * countPostEntries walks the post once. Article headings are counted anywhere + * outside Bonus; asset entries are attributed to whichever subsection is open. + * @param {string} content + * @returns {{articles: number, images: number, videos: number, documents: number, total: number}} + */ +export function countPostEntries(content) { + let articles = 0; + let images = 0; + let videos = 0; + let documents = 0; + let inBonus = false; + let subsection = ""; + + for (const raw of content.split("\n")) { + const line = raw.trim(); + if (BONUS_HEADING_RE.test(line)) { + inBonus = true; + subsection = ""; + continue; + } + const sub = SUBSECTION_RE.exec(line); + if (sub !== null) { + subsection = sub[1]; + continue; + } + if (ARTICLE_HEADING_RE.test(line)) { + articles++; + continue; + } + if (!inBonus) continue; + if (subsection === "Images") { + if (IMAGE_ENTRY_RE.test(line)) images++; + } else if (subsection === "Videos") { + // A direct video file entry looks the same as a YouTube entry; + // both belong to the Videos tally. + if (LINK_ENTRY_RE.test(line)) videos++; + } else if (subsection === "Documents") { + if (LINK_ENTRY_RE.test(line)) documents++; + } + } + return { articles, images, videos, documents, total: articles + images + videos + documents }; +} + +/** + * @param {string[]} args + * @returns {Promise<void>} + */ +export async function runPostStats(args) { + if (args.length < 1) { + process.stderr.write("usage: post-stats <path/to/index.md>\n"); + process.exit(1); + } + const path = args[0]; + let content; + try { + content = readFileSync(path, "utf8"); + } catch (err) { + process.stderr.write("read post: " + String(err?.message ?? err) + "\n"); + process.exit(1); + } + const counted = countPostEntries(content); + const m = NEWSLETTER_NUM_RE.exec(content); + printJson({ + post: path, + newsletter: m === null ? 0 : Number.parseInt(m[1], 10), + articles: counted.articles, + images: counted.images, + videos: counted.videos, + documents: counted.documents, + total: counted.total, + }); +} diff --git a/scripts/newsletter/url-utils.js b/scripts/newsletter/url-utils.js new file mode 100644 index 0000000..3098e80 --- /dev/null +++ b/scripts/newsletter/url-utils.js @@ -0,0 +1,339 @@ +// Shared URL helpers, ported from url_utils.go. Owned by the add-url router; +// reused by the other subcommands. + +import { readdirSync, readFileSync } from "node:fs"; +import { join } from "node:path"; + +/** Exact-match tracking params; any key starting with utm_ is also dropped. */ +const EXACT_TRACKING = new Set([ + "fbclid", "gclid", "msclkid", "mc_eid", + "aid", "ref", "ref_src", "ref_url", "source", "s", + "ck_subscriber_id", "igshid", "yclid", "vero_id", +]); + +/** + * contentDir is repo-root relative (invocation contract: run from repo root). + * @returns {string} + */ +export function contentDir() { + return join("content", "post"); +} + +/** + * cleanUrl removes common tracking parameters. Surviving query pairs are kept + * verbatim (no re-encoding) and in their original order — URLSearchParams would + * normalize percent-encoding (%7E → ~, + → %20), which must not leak into + * stored clean_url values. Unparseable / non-absolute input is returned + * untouched rather than corrupted. + * @param {string} raw + * @returns {string} + */ +export function cleanUrl(raw) { + let u; + try { + u = new URL(raw); + } catch { + return raw; + } + if (!u.protocol || !u.host) return raw; + // WHATWG URL serializes an empty path as "/" for special schemes; force it + // for the rest so cleaned URLs keep one shape. + if (u.pathname === "") { + try { + u.pathname = "/"; + } catch { + /* opaque path — leave as parsed */ + } + } + + let kept = ""; + const query = u.search.slice(1); + if (query !== "") { + const parts = []; + for (const pair of query.split("&")) { + if (pair === "") continue; + const eq = pair.indexOf("="); + const key = (eq === -1 ? pair : pair.slice(0, eq)).toLowerCase(); + if (key.startsWith("utm_") || EXACT_TRACKING.has(key)) continue; + parts.push(pair); + } + kept = parts.join("&"); + } + + const hash = u.hash; + u.search = ""; + u.hash = ""; + return u.toString() + (kept === "" ? "" : "?" + kept) + hash; +} + +// --- Substack image helpers (shared by add-url routing and detect-image-source) --- + +const SUBSTACK_IMAGE_HOSTS = new Set([ + "substackcdn.com", + "substack-post-media.s3.amazonaws.com", +]); + +const UUID_PATTERN = "[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}"; +const IMAGE_UUID_RE = new RegExp(`images(?:%2F|/)(${UUID_PATTERN})`, "i"); +const UUID_EXACT_RE = new RegExp(`^${UUID_PATTERN}$`, "i"); + +/** + * isSubstackImage reports a Substack-hosted image (CDN wrapper or raw S3), + * regardless of file extension. + * @param {string} target + * @returns {boolean} + */ +export function isSubstackImage(target) { + let host = ""; + try { + host = new URL(target).host.toLowerCase(); + } catch { + /* not an absolute URL — fall through to the substring check */ + } + return SUBSTACK_IMAGE_HOSTS.has(host) || target.toLowerCase().includes("substack-post-media"); +} + +/** + * substackImageUuid extracts the stable image identity: the S3 image UUID under + * public/images/<uuid>, with raw (/) or percent-encoded (%2F) separators. + * @param {string} target + * @returns {string} the lowercased uuid, or "" when absent + */ +export function substackImageUuid(target) { + const m = IMAGE_UUID_RE.exec(target); + return m === null ? "" : m[1].toLowerCase(); +} + +/** + * isUuid reports whether a bare identity is a Substack image uuid. + * @param {string} s + * @returns {boolean} + */ +export function isUuid(s) { + return UUID_EXACT_RE.test(s); +} + +/** + * Some sites carry the resource identity in a query param, not the path + * (e.g. YouTube /watch?v=ID). Preserve the identity param for those hosts so + * dedup does not collapse every video onto the same bare URL. + */ +const IDENTITY_PARAMS = new Map([ + ["youtube.com", "v"], + ["www.youtube.com", "v"], + ["m.youtube.com", "v"], +]); + +/** + * trimTrailingSlash removes at most one trailing "/". + * @param {string} s + * @returns {string} + */ +function trimTrailingSlash(s) { + return s.endsWith("/") ? s.slice(0, -1) : s; +} + +/** + * bareUrl reduces a URL to a stable identity for duplicate detection: + * - Substack image → its S3 UUID (transform/size variants share one identity) + * - YouTube → scheme+host+path + the v= video id + * - everything else → scheme + host + path + * @param {string} target + * @returns {string} + */ +export function bareUrl(target) { + if (isSubstackImage(target)) { + const uuid = substackImageUuid(target); + if (uuid !== "") return uuid; + } + let u = null; + try { + u = new URL(target); + } catch { + /* unparseable — fall back to crude string surgery below */ + } + if (u === null || !u.protocol || !u.host) { + return trimTrailingSlash(target.split("?")[0]); + } + const host = u.host.toLowerCase(); + let bare = trimTrailingSlash(u.protocol + "//" + host + u.pathname); + const idParam = IDENTITY_PARAMS.get(host); + if (idParam !== undefined) { + const v = u.searchParams.get(idParam); + if (v !== null && v !== "") bare += "?" + idParam + "=" + v; + } + return bare; +} + +/** + * fetchTextOk GETs target and returns the body on a 2xx response, "" on any + * error, non-2xx status, or timeout. Redirects are followed. + * @param {string} target + * @param {number} timeoutMs + * @param {string} [userAgent] + * @returns {Promise<string>} + */ +export async function fetchTextOk(target, timeoutMs, userAgent = "") { + try { + const headers = {}; + if (userAgent !== "") headers["user-agent"] = userAgent; + const res = await fetch(target, { + headers, + redirect: "follow", + signal: AbortSignal.timeout(timeoutMs), + }); + if (res.status < 200 || res.status >= 300) return ""; + return await res.text(); + } catch { + return ""; + } +} + +/** + * checkAccessibility HEADs the URL and returns the final HTTP status code as a + * string, or "000" on network error / timeout. + * @param {string} target + * @returns {Promise<string>} + */ +export async function checkAccessibility(target) { + try { + const res = await fetch(target, { + method: "HEAD", + redirect: "follow", + signal: AbortSignal.timeout(10_000), + }); + return String(res.status); + } catch { + return "000"; + } +} + +/** + * collectMarkdown recursively collects *.md files under dir (the content tree is + * small). Entries are visited in lexical order so callers that depend on + * first-seen ordering stay deterministic. Missing or unreadable directories + * yield nothing. + * @param {string} dir + * @returns {string[]} + */ +export function collectMarkdown(dir) { + /** @type {string[]} */ + const acc = []; + walkMarkdown(dir, acc); + return acc; +} + +/** + * @param {string} dir + * @param {string[]} acc + * @returns {void} + */ +function walkMarkdown(dir, acc) { + let entries; + try { + entries = readdirSync(dir, { withFileTypes: true }); + } catch { + return; // skip unreadable entries + } + entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + for (const e of entries) { + const p = join(dir, e.name); + if (e.isDirectory()) walkMarkdown(p, acc); + else if (e.name.toLowerCase().endsWith(".md")) acc.push(p); + } +} + +/** + * uuidBoundaryOk: the char after a UUID match must not extend the hex id, so + * <uuid>.png (cover image), <uuid>_WxH and <uuid>) all match. + * @param {string} text + * @param {number} end + * @returns {boolean} + */ +export function uuidBoundaryOk(text, end) { + if (end >= text.length) return true; + const c = text[end]; + return !((c >= "0" && c <= "9") || (c >= "a" && c <= "f") || (c >= "A" && c <= "F")); +} + +const URL_DELIMITERS = `)]"'?#<>_&,`; +const URL_WHITESPACE = " \t\n\r\f\v"; + +/** + * urlBoundaryOk: an optional trailing slash (bareUrl strips it, stored URLs may + * keep it), then a path/punctuation delimiter, whitespace, or end of text — so + * /p/foo does not match a stored /p/foo-bar. The '>' delimiter covers URLs + * stored in markdown autolink form <https://…>, which older posts use. + * @param {string} text + * @param {number} end + * @returns {boolean} + */ +export function urlBoundaryOk(text, end) { + let at = end; + if (at < text.length && text[at] === "/") at++; + if (at >= text.length) return true; + const c = text[at]; + if (URL_WHITESPACE.includes(c)) return true; + return URL_DELIMITERS.includes(c); +} + +/** + * hasBoundaryMatch scans every occurrence of needle and applies the boundary + * check in code. Kept as an index loop rather than one lookahead regex: the + * byte-offset checks (optional trailing slash, delimiter set) are what stop a + * needle that is merely a PREFIX of a stored string from matching. + * @param {string} text + * @param {string} needle + * @param {boolean} isUuidNeedle + * @returns {boolean} + */ +export function hasBoundaryMatch(text, needle, isUuidNeedle) { + for (let from = 0; ; ) { + const i = text.indexOf(needle, from); + if (i === -1) return false; + const end = i + needle.length; + if (isUuidNeedle ? uuidBoundaryOk(text, end) : urlBoundaryOk(text, end)) return true; + from = i + 1; + } +} + +/** + * checkDuplicate reports whether a URL identity already exists in the stored + * markdown, boundary-aware so a needle that is merely a PREFIX of a stored + * longer string is NOT a false duplicate. + * @param {string} target + * @param {string} dir + * @returns {boolean} + */ +export function checkDuplicate(target, dir) { + const needle = bareUrl(target); + if (needle === "") return false; + const uuidNeedle = isUuid(needle); + for (const file of collectMarkdown(dir)) { + let text; + try { + text = readFileSync(file, "utf8"); + } catch { + continue; + } + if (text.includes(needle) && hasBoundaryMatch(text, needle, uuidNeedle)) return true; + } + return false; +} + +const IMAGE_EXT_RE = /\.(png|jpg|jpeg|gif|webp|svg|avif|heic|heif|bmp|tiff?)(\?.*)?$/; +const VIDEO_EXT_RE = /\.(mp4|webm|mov|avi|mkv)(\?.*)?$/; +const DOCUMENT_EXT_RE = /\.(pdf|docx?|xlsx?|pptx?)(\?.*)?$/; + +/** + * classifyType classifies a URL by file extension. + * @param {string} target + * @returns {"image"|"video"|"document"|"article"} + */ +export function classifyType(target) { + const lower = target.toLowerCase(); + if (IMAGE_EXT_RE.test(lower)) return "image"; + if (VIDEO_EXT_RE.test(lower)) return "video"; + if (DOCUMENT_EXT_RE.test(lower)) return "document"; + return "article"; +}