feat(newsletter): port the engine to JavaScript

Translates the seven-subcommand engine to Node ESM, one module per former Go
file, invoked as `node scripts/newsletter <command>` from the repo root. The Go
implementation stays in place for now so parity can be measured against it.

Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual
string surgery in caption extraction and covers RSS and sitemap XML through
xmlMode without a second parser. fetch-via-defuddle gains a local extraction
stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a
third IP is the whole point when this machine's IP is the blocked one. Both
stages now emit the same YAML frontmatter plus body, and an extraction that
looks like a bot challenge counts as a local failure so the proxy still runs.

Behaviour is preserved where it is load-bearing rather than where it is merely
idiomatic: query strings are rebuilt by string surgery so surviving parameters
keep their original order and encoding, duplicate detection stays an index loop
with byte-offset boundary checks so a prefix of a stored URL is not a false
match, empty optional fields are omitted rather than emitted as "", a deep-crawl
miss reports cutoff as null rather than dropping the key, bullet filtering counts
code points, tag counts keep their six-column alignment, and a malformed percent
sequence falls back to the raw substring.

The printer lives in its own module so a command module never imports the
dispatcher: index.js runs main() at module scope, so that cycle would execute
the CLI as a side effect of any import. Missing dependencies report one
actionable line instead of a module-resolution stack trace, and a reader that
closes early exits quietly instead of raising EPIPE.
This commit is contained in:
tiennm99 committed 2026-09-18 16:23:58 +07:00
1 parent fdbbc5ba42
commit 64c7b76b4e
14 files changed
+1995 -2

No files matched your search

+7 -1
View File
@@ -8,7 +8,13 @@ export default [
languageOptions: {
ecmaVersion: "latest",
sourceType: "module",
globals: { console: "readonly", process: "readonly", fetch: "readonly" },
globals: {
console: "readonly",
process: "readonly",
fetch: "readonly",
URL: "readonly",
AbortSignal: "readonly",
},
},
rules: {
"no-unused-vars": ["error", { argsIgnorePattern: "^_" }],
+559
View File
@@ -8,6 +8,9 @@
"name": "blog",
"version": "0.0.0",
"dependencies": {
"cheerio": "^1.2.0",
"defuddle": "^0.19.4",
"linkedom": "^0.18.13",
"octokit": "^5.0.5"
},
"devDependencies": {
@@ -260,6 +263,13 @@
"dev": true,
"license": "MIT"
},
"node_modules/@mixmark-io/domino": {
"version": "2.2.0",
"resolved": "https://registry.npmjs.org/@mixmark-io/domino/-/domino-2.2.0.tgz",
"integrity": "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw==",
"license": "BSD-2-Clause",
"optional": true
},
"node_modules/@octokit/app": {
"version": "16.1.4",
"resolved": "https://registry.npmjs.org/@octokit/app/-/app-16.1.4.tgz",
@@ -854,6 +864,15 @@
"dev": true,
"license": "MIT"
},
"node_modules/@xmldom/xmldom": {
"version": "0.9.12",
"resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.12.tgz",
"integrity": "sha512-5AXjrcMClTryPe9LgZrygpB1lj7s0S9E0+W+AHaVKAVyHanafK86iPSvG5xHVSp/jC+VH1UXu0TAEmY279xH7A==",
"license": "MIT",
"engines": {
"node": ">=14.6"
}
},
"node_modules/acorn": {
"version": "8.18.0",
"resolved": "https://registry.npmjs.org/acorn/-/acorn-8.18.0.tgz",
@@ -910,6 +929,12 @@
"integrity": "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ==",
"license": "Apache-2.0"
},
"node_modules/boolbase": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/boolbase/-/boolbase-1.0.0.tgz",
"integrity": "sha512-JZOSA7Mo9sNGB8+UjSgzdLtokWAky1zbztM3WRLCbZ70/3cTANmQmOdR7y2g+J0e2WXywy1yS468tY+IruqEww==",
"license": "ISC"
},
"node_modules/bottleneck": {
"version": "2.19.5",
"resolved": "https://registry.npmjs.org/bottleneck/-/bottleneck-2.19.5.tgz",
@@ -943,6 +968,57 @@
"qified": "^0.10.1"
}
},
"node_modules/cheerio": {
"version": "1.2.0",
"resolved": "https://registry.npmjs.org/cheerio/-/cheerio-1.2.0.tgz",
"integrity": "sha512-WDrybc/gKFpTYQutKIK6UvfcuxijIZfMfXaYm8NMsPQxSYvf+13fXUJ4rztGGbJcBQ/GF55gvrZ0Bc0bj/mqvg==",
"license": "MIT",
"dependencies": {
"cheerio-select": "^2.1.0",
"dom-serializer": "^2.0.0",
"domhandler": "^5.0.3",
"domutils": "^3.2.2",
"encoding-sniffer": "^0.2.1",
"htmlparser2": "^10.1.0",
"parse5": "^7.3.0",
"parse5-htmlparser2-tree-adapter": "^7.1.0",
"parse5-parser-stream": "^7.1.2",
"undici": "^7.19.0",
"whatwg-mimetype": "^4.0.0"
},
"engines": {
"node": ">=20.18.1"
},
"funding": {
"url": "https://github.com/cheeriojs/cheerio?sponsor=1"
}
},
"node_modules/cheerio-select": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/cheerio-select/-/cheerio-select-2.1.0.tgz",
"integrity": "sha512-9v9kG0LvzrlcungtnJtpGNxY+fzECQKhK4EGJX2vByejiMX84MFNQw4UxPJl3bFbTMw+Dfs37XaIkCwTZfLh4g==",
"license": "BSD-2-Clause",
"dependencies": {
"boolbase": "^1.0.0",
"css-select": "^5.1.0",
"css-what": "^6.1.0",
"domelementtype": "^2.3.0",
"domhandler": "^5.0.3",
"domutils": "^3.0.1"
},
"funding": {
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/commander": {
"version": "12.1.0",
"resolved": "https://registry.npmjs.org/commander/-/commander-12.1.0.tgz",
"integrity": "sha512-Vw8qHK3bZM9y/P10u3Vib8o/DdkvA2OtPtZvD871QKjy74Wj1WSKFILMPRPSdUSx5RFK1arlJzEtA4PkFgnbuA==",
"license": "MIT",
"engines": {
"node": ">=18"
}
},
"node_modules/content-type": {
"version": "3.1.1",
"resolved": "https://registry.npmjs.org/content-type/-/content-type-3.1.1.tgz",
@@ -971,6 +1047,40 @@
"node": ">= 8"
}
},
"node_modules/css-select": {
"version": "5.2.2",
"resolved": "https://registry.npmjs.org/css-select/-/css-select-5.2.2.tgz",
"integrity": "sha512-TizTzUddG/xYLA3NXodFM0fSbNizXjOKhqiQQwvhlspadZokn1KDy0NZFS0wuEubIYAV5/c1/lAr0TaaFXEXzw==",
"license": "BSD-2-Clause",
"dependencies": {
"boolbase": "^1.0.0",
"css-what": "^6.1.0",
"domhandler": "^5.0.2",
"domutils": "^3.0.1",
"nth-check": "^2.0.1"
},
"funding": {
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/css-what": {
"version": "6.2.2",
"resolved": "https://registry.npmjs.org/css-what/-/css-what-6.2.2.tgz",
"integrity": "sha512-u/O3vwbptzhMs3L1fQE82ZSLHQQfto5gyZzwteVIEyeaY5Fc7R4dapF/BvRoSYFeqfBk4m0V1Vafq5Pjv25wvA==",
"license": "BSD-2-Clause",
"engines": {
"node": ">= 6"
},
"funding": {
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/cssom": {
"version": "0.5.0",
"resolved": "https://registry.npmjs.org/cssom/-/cssom-0.5.0.tgz",
"integrity": "sha512-iKuQcq+NdHqlAcwUY0o/HL69XQrUaQdMjmStJ8JFmUaiiQErlhrmuigkg/CU4E2J0IyUKUrMAgl36TvN67MqTw==",
"license": "MIT"
},
"node_modules/debug": {
"version": "4.4.3",
"resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz",
@@ -996,6 +1106,104 @@
"dev": true,
"license": "MIT"
},
"node_modules/defuddle": {
"version": "0.19.4",
"resolved": "https://registry.npmjs.org/defuddle/-/defuddle-0.19.4.tgz",
"integrity": "sha512-Jz98aWlyEeuWcg9u2AavCEKknviI9GijMkrGFPk0wi1Lkfr1DDDw6XM+ak44Ol8kBDmDJsvU9Su2eUY09Fvh3Q==",
"license": "MIT",
"dependencies": {
"commander": "^12.1.0",
"mathml-to-latex": "^1.8.0"
},
"bin": {
"defuddle": "dist/cli.js"
},
"optionalDependencies": {
"linkedom": "^0.18.12",
"temml": "^0.13.3",
"turndown": "^7.2.0"
}
},
"node_modules/dom-serializer": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-2.0.0.tgz",
"integrity": "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg==",
"license": "MIT",
"dependencies": {
"domelementtype": "^2.3.0",
"domhandler": "^5.0.2",
"entities": "^4.2.0"
},
"funding": {
"url": "https://github.com/cheeriojs/dom-serializer?sponsor=1"
}
},
"node_modules/domelementtype": {
"version": "2.3.0",
"resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-2.3.0.tgz",
"integrity": "sha512-OLETBj6w0OsagBwdXnPdN0cnMfF9opN69co+7ZrbfPGrdpPVNBUj02spi6B1N7wChLQiPn4CSH/zJvXw56gmHw==",
"funding": [
{
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
],
"license": "BSD-2-Clause"
},
"node_modules/domhandler": {
"version": "5.0.3",
"resolved": "https://registry.npmjs.org/domhandler/-/domhandler-5.0.3.tgz",
"integrity": "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w==",
"license": "BSD-2-Clause",
"dependencies": {
"domelementtype": "^2.3.0"
},
"engines": {
"node": ">= 4"
},
"funding": {
"url": "https://github.com/fb55/domhandler?sponsor=1"
}
},
"node_modules/domutils": {
"version": "3.2.2",
"resolved": "https://registry.npmjs.org/domutils/-/domutils-3.2.2.tgz",
"integrity": "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw==",
"license": "BSD-2-Clause",
"dependencies": {
"dom-serializer": "^2.0.0",
"domelementtype": "^2.3.0",
"domhandler": "^5.0.3"
},
"funding": {
"url": "https://github.com/fb55/domutils?sponsor=1"
}
},
"node_modules/encoding-sniffer": {
"version": "0.2.1",
"resolved": "https://registry.npmjs.org/encoding-sniffer/-/encoding-sniffer-0.2.1.tgz",
"integrity": "sha512-5gvq20T6vfpekVtqrYQsSCFZ1wEg5+wW0/QaZMWkFr6BqD3NfKs0rLCx4rrVlSWJeZb5NBJgVLswK/w2MWU+Gw==",
"license": "MIT",
"dependencies": {
"iconv-lite": "^0.6.3",
"whatwg-encoding": "^3.1.1"
},
"funding": {
"url": "https://github.com/fb55/encoding-sniffer?sponsor=1"
}
},
"node_modules/entities": {
"version": "4.5.0",
"resolved": "https://registry.npmjs.org/entities/-/entities-4.5.0.tgz",
"integrity": "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw==",
"license": "BSD-2-Clause",
"engines": {
"node": ">=0.12"
},
"funding": {
"url": "https://github.com/fb55/entities?sponsor=1"
}
},
"node_modules/escape-string-regexp": {
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz",
@@ -1264,6 +1472,55 @@
"dev": true,
"license": "MIT"
},
"node_modules/html-escaper": {
"version": "3.0.3",
"resolved": "https://registry.npmjs.org/html-escaper/-/html-escaper-3.0.3.tgz",
"integrity": "sha512-RuMffC89BOWQoY0WKGpIhn5gX3iI54O6nRA0yC124NYVtzjmFWBIiFd8M0x+ZdX0P9R4lADg1mgP8C7PxGOWuQ==",
"license": "MIT"
},
"node_modules/htmlparser2": {
"version": "10.1.0",
"resolved": "https://registry.npmjs.org/htmlparser2/-/htmlparser2-10.1.0.tgz",
"integrity": "sha512-VTZkM9GWRAtEpveh7MSF6SjjrpNVNNVJfFup7xTY3UpFtm67foy9HDVXneLtFVt4pMz5kZtgNcvCniNFb1hlEQ==",
"funding": [
"https://github.com/fb55/htmlparser2?sponsor=1",
{
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
],
"license": "MIT",
"dependencies": {
"domelementtype": "^2.3.0",
"domhandler": "^5.0.3",
"domutils": "^3.2.2",
"entities": "^7.0.1"
}
},
"node_modules/htmlparser2/node_modules/entities": {
"version": "7.0.1",
"resolved": "https://registry.npmjs.org/entities/-/entities-7.0.1.tgz",
"integrity": "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA==",
"license": "BSD-2-Clause",
"engines": {
"node": ">=0.12"
},
"funding": {
"url": "https://github.com/fb55/entities?sponsor=1"
}
},
"node_modules/iconv-lite": {
"version": "0.6.3",
"resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.6.3.tgz",
"integrity": "sha512-4fCk79wshMdzMp2rH06qWrJE4iolqLhCUH+OiuIgU++RB0+94NlDL81atO7GX55uUKueo0txHNtvEyI6D7WdMw==",
"license": "MIT",
"dependencies": {
"safer-buffer": ">= 2.1.2 < 3.0.0"
},
"engines": {
"node": ">=0.10.0"
}
},
"node_modules/ignore": {
"version": "5.3.2",
"resolved": "https://registry.npmjs.org/ignore/-/ignore-5.3.2.tgz",
@@ -1358,6 +1615,171 @@
"node": ">= 0.8.0"
}
},
"node_modules/linkedom": {
"version": "0.18.13",
"resolved": "https://registry.npmjs.org/linkedom/-/linkedom-0.18.13.tgz",
"integrity": "sha512-ES/o9qotMpzpN2MHs+Iq/JcVoOj8Fa5wiQYrTdFpvAnwXL0g66XHHUc9WUMk6nAlBtGsFQ24ne+SYnvnaQ2FSw==",
"license": "ISC",
"dependencies": {
"css-select": "^7.0.0",
"cssom": "^0.5.0",
"html-escaper": "^3.0.3",
"htmlparser2": "^10.1.0",
"uhyphen": "^0.2.0"
},
"engines": {
"node": ">=16"
},
"peerDependencies": {
"canvas": ">= 2"
},
"peerDependenciesMeta": {
"canvas": {
"optional": true
}
}
},
"node_modules/linkedom/node_modules/boolbase": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/boolbase/-/boolbase-2.0.0.tgz",
"integrity": "sha512-DkVaaQHymRhpYEYo9x1oo7Q7B0Y6KJUsjm3c9eTyFDby4MHLBTwZ6ZDWBel5zrYxj1WsZgC5oLpiz+93MluXeA==",
"license": "ISC",
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/linkedom/node_modules/css-select": {
"version": "7.0.0",
"resolved": "https://registry.npmjs.org/css-select/-/css-select-7.0.0.tgz",
"integrity": "sha512-snmjEVXy+1LnwXdxhYvTMj1d9tOh4HxkA1YmoayVBeeyR2C14Pum7fcxJIm4SswYspVy866eYNwlH6xC3/VH5g==",
"license": "BSD-2-Clause",
"dependencies": {
"boolbase": "^2.0.0",
"css-what": "^8.0.0",
"domhandler": "^6.0.1",
"domutils": "^4.0.2",
"nth-check": "^3.0.1"
},
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/linkedom/node_modules/css-what": {
"version": "8.0.0",
"resolved": "https://registry.npmjs.org/css-what/-/css-what-8.0.0.tgz",
"integrity": "sha512-DH0Bqq3DNp5tdOReuNyAA+Ev4Y2GS5FMbZpeTLP6C4CDi0h5nL0BmUPChXw3o/qbHLDWHl49sbNqQVY7bMSDdw==",
"license": "BSD-2-Clause",
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
},
"node_modules/linkedom/node_modules/dom-serializer": {
"version": "3.1.1",
"resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-3.1.1.tgz",
"integrity": "sha512-4MEa38/QexBob6gFNwu+EGdWvhJ1OKuNwdYY3Y3NyeWDQfnGeDYQUDfIRzWu5B5gsv03so2Uxd28YC6zrsx3Lw==",
"license": "MIT",
"dependencies": {
"domelementtype": "^3.0.0",
"domhandler": "^6.0.0",
"entities": "^8.0.0"
},
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/cheeriojs/dom-serializer?sponsor=1"
}
},
"node_modules/linkedom/node_modules/domelementtype": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-3.0.0.tgz",
"integrity": "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg==",
"funding": [
{
"type": "github",
"url": "https://github.com/sponsors/fb55"
}
],
"license": "BSD-2-Clause",
"engines": {
"node": ">=20.19.0"
}
},
"node_modules/linkedom/node_modules/domhandler": {
"version": "6.0.1",
"resolved": "https://registry.npmjs.org/domhandler/-/domhandler-6.0.1.tgz",
"integrity": "sha512-gYzvtM72ZtxQO0T048kd6HWSbbGCNOUwcnfQ01cqIJ4X2IYKFFHZ5mKvrQETcFXxsRObZulDaKmy//R7TPtsBg==",
"license": "BSD-2-Clause",
"dependencies": {
"domelementtype": "^3.0.0"
},
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/fb55/domhandler?sponsor=1"
}
},
"node_modules/linkedom/node_modules/domutils": {
"version": "4.0.2",
"resolved": "https://registry.npmjs.org/domutils/-/domutils-4.0.2.tgz",
"integrity": "sha512-qI4JLRKnSzqFqr7hAlS5xQDusBCjKSEG4t4+7aNrIQMHBcsC2TGEhuyABJdYkgSewL57PNLYEiibY2iPKhKpaA==",
"license": "BSD-2-Clause",
"dependencies": {
"dom-serializer": "^3.0.0",
"domelementtype": "^3.0.0",
"domhandler": "^6.0.0"
},
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/fb55/domutils?sponsor=1"
}
},
"node_modules/linkedom/node_modules/entities": {
"version": "8.1.0",
"resolved": "https://registry.npmjs.org/entities/-/entities-8.1.0.tgz",
"integrity": "sha512-kxL7msIffSuh9aaFAMD7rxAIuTRMAHMeBtgHW2yUdWw732ZNh4MehkF2gdjvtdmikkaIP9bFDDJOPlsvm7avrA==",
"license": "BSD-2-Clause",
"engines": {
"node": ">=20.19.0"
},
"funding": {
"url": "https://github.com/fb55/entities?sponsor=1"
}
},
"node_modules/linkedom/node_modules/nth-check": {
"version": "3.0.1",
"resolved": "https://registry.npmjs.org/nth-check/-/nth-check-3.0.1.tgz",
"integrity": "sha512-GX0gsdbGVCgnRgbeGaubfjpBXyYRWOOCVeYh08bSQvDZqxz5ndXs1OTfAt/h36G1xvI94YIspsI0sVFqAV9+RQ==",
"license": "BSD-2-Clause",
"dependencies": {
"boolbase": "^2.0.0"
},
"engines": {
"node": ">=20.19.0"
},
"funding": {
"type": "github",
"url": "https://github.com/fb55/nth-check?sponsor=1"
}
},
"node_modules/locate-path": {
"version": "6.0.0",
"resolved": "https://registry.npmjs.org/locate-path/-/locate-path-6.0.0.tgz",
@@ -1374,6 +1796,15 @@
"url": "https://github.com/sponsors/sindresorhus"
}
},
"node_modules/mathml-to-latex": {
"version": "1.8.0",
"resolved": "https://registry.npmjs.org/mathml-to-latex/-/mathml-to-latex-1.8.0.tgz",
"integrity": "sha512-gQ0uK3zqB8HwlfaXJkEL5rgaZNbKUiBMmBP/B/W+b+t6KcseLSuYb1b0BjLgS9ZiQa24ePkqTX8/6FaQuDL7wQ==",
"license": "MIT",
"dependencies": {
"@xmldom/xmldom": "^0.9.10"
}
},
"node_modules/minimatch": {
"version": "10.2.6",
"resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.6.tgz",
@@ -1404,6 +1835,18 @@
"dev": true,
"license": "MIT"
},
"node_modules/nth-check": {
"version": "2.1.1",
"resolved": "https://registry.npmjs.org/nth-check/-/nth-check-2.1.1.tgz",
"integrity": "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w==",
"license": "BSD-2-Clause",
"dependencies": {
"boolbase": "^1.0.0"
},
"funding": {
"url": "https://github.com/fb55/nth-check?sponsor=1"
}
},
"node_modules/octokit": {
"version": "5.0.5",
"resolved": "https://registry.npmjs.org/octokit/-/octokit-5.0.5.tgz",
@@ -1476,6 +1919,55 @@
"url": "https://github.com/sponsors/sindresorhus"
}
},
"node_modules/parse5": {
"version": "7.3.0",
"resolved": "https://registry.npmjs.org/parse5/-/parse5-7.3.0.tgz",
"integrity": "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==",
"license": "MIT",
"dependencies": {
"entities": "^6.0.0"
},
"funding": {
"url": "https://github.com/inikulin/parse5?sponsor=1"
}
},
"node_modules/parse5-htmlparser2-tree-adapter": {
"version": "7.1.0",
"resolved": "https://registry.npmjs.org/parse5-htmlparser2-tree-adapter/-/parse5-htmlparser2-tree-adapter-7.1.0.tgz",
"integrity": "sha512-ruw5xyKs6lrpo9x9rCZqZZnIUntICjQAd0Wsmp396Ul9lN/h+ifgVV1x1gZHi8euej6wTfpqX8j+BFQxF0NS/g==",
"license": "MIT",
"dependencies": {
"domhandler": "^5.0.3",
"parse5": "^7.0.0"
},
"funding": {
"url": "https://github.com/inikulin/parse5?sponsor=1"
}
},
"node_modules/parse5-parser-stream": {
"version": "7.1.2",
"resolved": "https://registry.npmjs.org/parse5-parser-stream/-/parse5-parser-stream-7.1.2.tgz",
"integrity": "sha512-JyeQc9iwFLn5TbvvqACIF/VXG6abODeB3Fwmv/TGdLk2LfbWkaySGY72at4+Ty7EkPZj854u4CrICqNk2qIbow==",
"license": "MIT",
"dependencies": {
"parse5": "^7.0.0"
},
"funding": {
"url": "https://github.com/inikulin/parse5?sponsor=1"
}
},
"node_modules/parse5/node_modules/entities": {
"version": "6.0.1",
"resolved": "https://registry.npmjs.org/entities/-/entities-6.0.1.tgz",
"integrity": "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g==",
"license": "BSD-2-Clause",
"engines": {
"node": ">=0.12"
},
"funding": {
"url": "https://github.com/fb55/entities?sponsor=1"
}
},
"node_modules/path-exists": {
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/path-exists/-/path-exists-4.0.0.tgz",
@@ -1536,6 +2028,12 @@
"dev": true,
"license": "MIT"
},
"node_modules/safer-buffer": {
"version": "2.1.2",
"resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz",
"integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==",
"license": "MIT"
},
"node_modules/shebang-command": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz",
@@ -1559,6 +2057,16 @@
"node": ">=8"
}
},
"node_modules/temml": {
"version": "0.13.5",
"resolved": "https://registry.npmjs.org/temml/-/temml-0.13.5.tgz",
"integrity": "sha512-aPkDDgunanpLNL0ql32HbolqLep+w8DRcVXRql7rWrMt/PhczdLgL4UBYYVU3BYjCNjkGq4EqwIicu/zWr1iOg==",
"license": "MIT",
"optional": true,
"engines": {
"node": ">=18.13.0"
}
},
"node_modules/toad-cache": {
"version": "3.7.4",
"resolved": "https://registry.npmjs.org/toad-cache/-/toad-cache-3.7.4.tgz",
@@ -1568,6 +2076,20 @@
"node": ">=20"
}
},
"node_modules/turndown": {
"version": "7.2.4",
"resolved": "https://registry.npmjs.org/turndown/-/turndown-7.2.4.tgz",
"integrity": "sha512-I8yFsfRzmzK0WV1pNNOA4A7y4RDfFxPRxb3t+e3ui14qSGOxGtiSP6GjeX+Y6CHb7HYaFj7ECUD7VE5kQMZWGQ==",
"license": "MIT",
"optional": true,
"dependencies": {
"@mixmark-io/domino": "^2.2.0"
},
"engines": {
"node": ">=18",
"npm": ">=9"
}
},
"node_modules/type-check": {
"version": "0.4.0",
"resolved": "https://registry.npmjs.org/type-check/-/type-check-0.4.0.tgz",
@@ -1581,6 +2103,21 @@
"node": ">= 0.8.0"
}
},
"node_modules/uhyphen": {
"version": "0.2.0",
"resolved": "https://registry.npmjs.org/uhyphen/-/uhyphen-0.2.0.tgz",
"integrity": "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA==",
"license": "ISC"
},
"node_modules/undici": {
"version": "7.29.1",
"resolved": "https://registry.npmjs.org/undici/-/undici-7.29.1.tgz",
"integrity": "sha512-RYONW2MeafgYlkVOKYKkA/Ag7BmXqgIWCa8t1m0JcxrQg9pI9lEqRhAOruOBCbAohOa/gkCF+iPi9hrgvTzu6Q==",
"license": "MIT",
"engines": {
"node": ">=20.18.1"
}
},
"node_modules/universal-github-app-jwt": {
"version": "2.2.2",
"resolved": "https://registry.npmjs.org/universal-github-app-jwt/-/universal-github-app-jwt-2.2.2.tgz",
@@ -1603,6 +2140,28 @@
"punycode": "^2.1.0"
}
},
"node_modules/whatwg-encoding": {
"version": "3.1.1",
"resolved": "https://registry.npmjs.org/whatwg-encoding/-/whatwg-encoding-3.1.1.tgz",
"integrity": "sha512-6qN4hJdMwfYBtE3YBTTHhoeuUrDBPZmbQaxWAqSALV/MeEnR5z1xd8UKud2RAkFoPkmB+hli1TZSnyi84xz1vQ==",
"deprecated": "Use @exodus/bytes instead for a more spec-conformant and faster implementation",
"license": "MIT",
"dependencies": {
"iconv-lite": "0.6.3"
},
"engines": {
"node": ">=18"
}
},
"node_modules/whatwg-mimetype": {
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/whatwg-mimetype/-/whatwg-mimetype-4.0.0.tgz",
"integrity": "sha512-QaKxh0eNIi2mE9p2vEdzfagOKHCcj1pJ56EEHGQOVxp8r9/iszLUUV7v89x9O1p/T+NlTM5W7jW6+cz4Fq1YVg==",
"license": "MIT",
"engines": {
"node": ">=18"
}
},
"node_modules/which": {
"version": "2.0.2",
"resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz",
+5 -1
View File
@@ -9,9 +9,13 @@
},
"scripts": {
"lint": "eslint .",
"projects:refresh": "node .github/scripts/update-projects-list.js"
"projects:refresh": "node .github/scripts/update-projects-list.js",
"test": "node --test scripts/newsletter/*.test.js"
},
"dependencies": {
"cheerio": "^1.2.0",
"defuddle": "^0.19.4",
"linkedom": "^0.18.13",
"octokit": "^5.0.5"
},
"devDependencies": {
+127
View File
@@ -0,0 +1,127 @@
// Meta URL router for the mt-add-url skill — the single entry per URL.
// Usage: node scripts/newsletter add-url "<url>"
// Outputs: JSON { original_url, clean_url, http_status, accessible,
// duplicate, route, title?, author? }
//
// route ∈ youtube | image | video | document | article
import { printJson } from "./json-out.js";
import {
checkAccessibility,
checkDuplicate,
classifyType,
cleanUrl,
contentDir,
fetchTextOk,
isSubstackImage,
} from "./url-utils.js";
const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]);
/**
* detectYouTube extracts a video id from the supported URL shapes:
* youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID.
* Playlists/channels are intentionally NOT YouTube routes (fall through to type).
* @param {string} target
* @returns {{isYouTube: boolean, videoId: string}}
*/
export function detectYouTube(target) {
let u;
try {
u = new URL(target);
} catch {
return { isYouTube: false, videoId: "" };
}
const host = u.host.toLowerCase();
if (host === "youtu.be") {
const id = u.pathname.replace(/^\//, "").split("/")[0];
return { isYouTube: id !== "", videoId: id };
}
if (YT_HOSTS.has(host)) {
if (u.pathname === "/watch") {
const id = u.searchParams.get("v") ?? "";
return { isYouTube: id !== "", videoId: id };
}
if (u.pathname.startsWith("/shorts/")) {
const parts = u.pathname.split("/");
if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] };
}
}
return { isYouTube: false, videoId: "" };
}
/**
* canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes.
* @param {string} videoId
* @returns {string}
*/
function canonicalWatchUrl(videoId) {
return "https://www.youtube.com/watch?v=" + videoId;
}
/**
* fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key).
* Best-effort: any failure returns empty strings so the route stays `youtube`
* and the skill can fall back.
* @param {string} watchUrl
* @returns {Promise<{title: string, author: string}>}
*/
async function fetchYouTubeMeta(watchUrl) {
const endpoint =
"https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json";
const body = await fetchTextOk(endpoint, 10_000);
if (body === "") return { title: "", author: "" };
try {
const data = JSON.parse(body);
return { title: data.title ?? "", author: data.author_name ?? "" };
} catch {
return { title: "", author: "" };
}
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runAddUrl(args) {
if (args.length < 1 || args[0] === "") {
process.stderr.write("Usage: node scripts/newsletter add-url <url>\n");
process.exit(1);
}
const target = args[0];
const cleaned = cleanUrl(target);
const { isYouTube, videoId } = detectYouTube(cleaned);
// For YouTube, dedup/store against the canonical watch URL so youtu.be and
// shorts links collapse onto the same identity-param key as watch URLs.
const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned;
// Route order: YouTube → Substack image (by host, not extension, so f_auto /
// .avif / .heic / extensionless CDN URLs still route to the image handler) →
// file-extension classification.
let route;
if (isYouTube) route = "youtube";
else if (isSubstackImage(cleaned)) route = "image";
else route = classifyType(cleaned);
const httpStatus = await checkAccessibility(cleaned);
// title/author are omitted entirely when empty, not emitted as "": the
// handlers treat a present key as "metadata was resolved".
/** @type {Record<string, unknown>} */
const out = {
original_url: target,
clean_url: effectiveUrl,
http_status: httpStatus,
accessible: httpStatus === "200",
duplicate: checkDuplicate(effectiveUrl, contentDir()),
route,
};
if (route === "youtube") {
const { title, author } = await fetchYouTubeMeta(effectiveUrl);
if (title !== "") out.title = title;
if (author !== "") out.author = author;
}
printJson(out);
}
+66
View File
@@ -0,0 +1,66 @@
// Detect whether an image URL is Substack-hosted and extract its S3 image UUID.
// Usage: node scripts/newsletter detect-image-source "<image-url>"
// Output: JSON { original_url, clean_url, isSubstack, uuid?, innerUrl? }
//
// Substack images are usually served via a CDN wrapper:
//
// https://substackcdn.com/image/fetch/$s_!x!,.../https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F<uuid>_WxH.png
//
// The publication is NOT encoded in the URL — only the image identity (uuid) is.
import { printJson } from "./json-out.js";
import { cleanUrl, isSubstackImage, substackImageUuid } from "./url-utils.js";
/**
* extractInnerUrl pulls the inner S3 URL out of a substackcdn /image/fetch/
* wrapper (if present). A malformed percent sequence falls back to the raw
* substring rather than throwing.
* @param {string} target
* @returns {string}
*/
export function extractInnerUrl(target) {
const marker = target.indexOf("/https%3A%2F%2F");
if (marker !== -1) {
const raw = target.slice(marker + 1);
try {
return decodeURIComponent(raw);
} catch {
return raw;
}
}
// Some forms embed a plain (already-decoded) inner https URL.
if (target.length > 8) {
const plain = target.slice(8).indexOf("/https://");
if (plain !== -1) return target.slice(8 + plain + 1);
}
return target;
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runDetectImageSource(args) {
if (args.length < 1 || args[0] === "") {
process.stderr.write("Usage: node scripts/newsletter detect-image-source <image-url>\n");
process.exit(1);
}
const target = args[0];
const isSubstack = isSubstackImage(target);
// Empty optional fields are omitted, not emitted as "": the skills branch on
// the key being present.
/** @type {Record<string, unknown>} */
const out = {
original_url: target,
clean_url: cleanUrl(target),
isSubstack,
};
if (isSubstack) {
const uuid = substackImageUuid(target);
const innerUrl = extractInnerUrl(target);
if (uuid !== "") out.uuid = uuid;
if (innerUrl !== "") out.innerUrl = innerUrl;
}
printJson(out);
}
+143
View File
@@ -0,0 +1,143 @@
// mt-fetch-url fallback fetcher.
// Usage: node scripts/newsletter fetch-via-defuddle <target_url>
// Exit codes: 0 = content returned, 1 = every tier failed, 2 = bad arguments.
//
// Two tiers, tried in order:
// 1. local defuddle — extracts on this machine, so the chain no longer depends
// on a single third-party service being reachable.
// 2. the defuddle.md proxy — fetches from a third IP, which is the point when
// this machine's IP is the one being blocked.
//
// Both tiers emit YAML frontmatter followed by the markdown body, so callers
// parse one shape regardless of which tier answered.
const BROWSER_UA =
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36";
// A bot wall answers 200 with a real body, so a non-empty extraction is not by
// itself a success. Recognising the wall is what lets the proxy tier — which
// fetches from a different IP — still get its turn.
const CHALLENGE_MARKERS = [
/just a moment/i,
/checking your browser/i,
/attention required/i,
/cloudflare/i,
/enable javascript (and cookies )?to continue/i,
/verify (that )?you('re| are) (a )?human/i,
/are you a robot/i,
/access denied/i,
/captcha/i,
];
/**
* looksLikeChallenge reports whether an extraction is a bot wall rather than the
* page that was asked for.
* @param {string} title
* @param {string} body
* @returns {boolean}
*/
export function looksLikeChallenge(title, body) {
// Only the opening of the body: an article may legitimately discuss Cloudflare.
const sample = title + "\n" + body.slice(0, 400);
return CHALLENGE_MARKERS.some((re) => re.test(sample));
}
/**
* yamlFrontmatter renders the metadata block, omitting fields with no value.
* @param {Record<string, string>} fields
* @returns {string}
*/
function yamlFrontmatter(fields) {
const lines = Object.entries(fields)
.filter(([, v]) => typeof v === "string" && v !== "")
.map(([k, v]) => `${k}: ${JSON.stringify(v)}`);
return lines.length === 0 ? "" : "---\n" + lines.join("\n") + "\n---\n\n";
}
/**
* fetchLocally extracts article content with defuddle running in-process, and
* renders it in the same frontmatter-plus-body shape the proxy returns.
* @param {string} target
* @returns {Promise<string>} the document, or "" when extraction yields nothing usable
*/
async function fetchLocally(target) {
const { Defuddle } = await import("defuddle/node");
const { parseHTML } = await import("linkedom");
const res = await fetch(target, {
headers: { "user-agent": BROWSER_UA },
redirect: "follow",
signal: AbortSignal.timeout(30_000),
});
if (!res.ok) throw new Error(`upstream returned ${res.status}`);
const html = await res.text();
const { document } = parseHTML(html);
const result = await Defuddle(document, target, { markdown: true });
const body = String(result?.content ?? "");
if (body.trim() === "") return "";
const title = String(result?.title ?? "");
if (looksLikeChallenge(title, body)) {
throw new Error("extraction looks like a bot challenge, not the page");
}
const frontmatter = yamlFrontmatter({
title,
author: String(result?.author ?? ""),
description: String(result?.description ?? ""),
site: String(result?.site ?? ""),
published: String(result?.published ?? ""),
source: target,
});
return frontmatter + body;
}
/**
* fetchViaProxy fetches through defuddle.md, which resolves the page from its
* own IP.
* @param {string} target
* @returns {Promise<string>} the response body
*/
async function fetchViaProxy(target) {
const res = await fetch("https://defuddle.md/" + target, {
headers: { "user-agent": "mt-fetch-url/1.0" },
redirect: "follow",
signal: AbortSignal.timeout(30_000),
});
const body = await res.text();
if (!res.ok || body.trim() === "") {
throw new Error(`defuddle returned ${res.status} / empty body`);
}
return body;
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runFetchViaDefuddle(args) {
if (args.length < 1 || args[0] === "") {
process.stderr.write("Usage: node scripts/newsletter fetch-via-defuddle <target_url>\n");
process.exit(2);
}
const target = args[0];
try {
const doc = await fetchLocally(target);
if (doc !== "") {
process.stdout.write(doc.endsWith("\n") ? doc : doc + "\n");
return;
}
process.stderr.write(`mt-fetch-url: local defuddle extracted nothing for ${target}\n`);
} catch (err) {
process.stderr.write(
`mt-fetch-url: local defuddle failed for ${target}: ${String(err?.message ?? err)}\n`,
);
}
try {
process.stdout.write(await fetchViaProxy(target));
} catch (err) {
process.stderr.write(
`mt-fetch-url: defuddle.md failed for ${target}: ${String(err?.message ?? err)}\n`,
);
process.exit(1);
}
}
@@ -0,0 +1,79 @@
// Find the most recent newsletter number and return the next one.
// Usage: node scripts/newsletter find-newsletter-number
// Outputs: the next newsletter number.
import { readdirSync, readFileSync } from "node:fs";
import { join } from "node:path";
import { contentDir } from "./url-utils.js";
/** Matches the "Newsletter #N" heading a post is numbered by. @type {RegExp} */
export const NEWSLETTER_NUM_RE = /Newsletter\s*#(\d+)/;
const YEAR_DIR_RE = /^\d{4}$/;
const TWO_DIGIT_DIR_RE = /^\d{2}$/;
/**
* listDirsDesc returns dir's subdirectory names matching re, sorted descending.
* @param {string} dir
* @param {RegExp} re
* @returns {string[]}
*/
function listDirsDesc(dir, re) {
let entries;
try {
entries = readdirSync(dir, { withFileTypes: true });
} catch {
return [];
}
return entries
.filter((e) => re.test(e.name))
.map((e) => e.name)
.sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
}
/**
* extractNewsletterNumber reads the first "Newsletter #N" in a post.
* @param {string} path
* @returns {number}
*/
function extractNewsletterNumber(path) {
let content;
try {
content = readFileSync(path, "utf8");
} catch {
return 0;
}
const m = NEWSLETTER_NUM_RE.exec(content);
if (m === null) return 0;
const n = Number.parseInt(m[1], 10);
return Number.isNaN(n) ? 0 : n;
}
/**
* findMostRecentNewsletter scans year/month/day directories newest-first for
* the highest newsletter number.
* @returns {number}
*/
export function findMostRecentNewsletter() {
let maxNumber = 0;
for (const year of listDirsDesc(contentDir(), YEAR_DIR_RE)) {
const yearDir = join(contentDir(), year);
for (const month of listDirsDesc(yearDir, TWO_DIGIT_DIR_RE)) {
const monthDir = join(yearDir, month);
for (const day of listDirsDesc(monthDir, TWO_DIGIT_DIR_RE)) {
const n = extractNewsletterNumber(join(monthDir, day, "index.md"));
if (n > maxNumber) maxNumber = n;
}
}
// Early exit: a newsletter found in this year — no need to go further back.
if (maxNumber > 0) break;
}
return maxNumber;
}
/**
* @param {string[]} _args
* @returns {Promise<void>}
*/
export async function runFindNewsletterNumber(_args) {
process.stdout.write(String(findMostRecentNewsletter() + 1) + "\n");
}
+231
View File
@@ -0,0 +1,231 @@
// Find which Substack post embeds a given image UUID, and extract a label.
// Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]
// Output on hit: JSON { found:true, source, publication, postTitle, postUrl, caption, candidates }
// Output on miss: JSON { found:false } (RSS) or { found:false, source:"sitemap", scanned, budget, cutoff } (--deep)
//
// Strategy: RSS feed first (fast, ~recent weeks). With --deep, fall back to a
// heavier sitemap crawl up to ~3 months back — opt-in because it fetches many
// posts. A Substack CDN URL does not encode its publication, so we search each
// publication listed in config/substack-publications.json, read at runtime so
// editing the JSON takes effect immediately.
import { readFileSync } from "node:fs";
import { parseArgs } from "node:util";
import { printJson } from "./json-out.js";
import { fetchTextOk } from "./url-utils.js";
import {
captionForUuid,
extractCandidates,
itemLink,
itemTitle,
loadXml,
postTitleFromHtml,
} from "./html-text.js";
/** Total post fetches allowed across ALL publications during a --deep crawl. */
const DEEP_FETCH_BUDGET = 40;
/**
* loadPublications reads the publication list, falling back to the default on
* any read or parse failure.
* @returns {string[]}
*/
export function loadPublications() {
try {
const raw = readFileSync(new URL("./config/substack-publications.json", import.meta.url), "utf8");
const pubs = JSON.parse(raw);
if (Array.isArray(pubs) && pubs.length > 0) return pubs;
} catch {
/* fall through */
}
return ["blog.bytebytego.com"];
}
/**
* fetchPage: body text with a browser-ish UA, or "" on any error.
* @param {string} target
* @returns {Promise<string>}
*/
function fetchPage(target) {
return fetchTextOk(target, 10_000, "Mozilla/5.0");
}
const RFC3339_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$/i;
const DATE_ONLY_RE = /^\d{4}-\d{2}-\d{2}$/;
/**
* parseLastmod accepts only the two layouts the Go engine accepted — RFC3339
* and a bare date — so a loosely formatted stamp is skipped rather than
* silently reinterpreted in local time.
* @param {string} s
* @returns {Date|null}
*/
export function parseLastmod(s) {
if (!RFC3339_RE.test(s) && !DATE_ONLY_RE.test(s)) return null;
const t = new Date(s);
return Number.isNaN(t.getTime()) ? null : t;
}
/**
* cutoffDate returns the ~3-months-back boundary for the deep crawl.
* @returns {Date}
*/
function cutoffDate() {
const d = new Date();
d.setUTCMonth(d.getUTCMonth() - 3);
return d;
}
/**
* @typedef {{found: true, source: string, publication: string, postTitle: string,
* postUrl: string, caption: string, candidates: string[]}} PostHit
*/
/**
* searchSitemap is the deep fallback: crawl the sitemap back ~3 months, fetch
* posts most-recent-first (up to maxFetch from the shared budget), and look for
* the UUID. Heavier than RSS — only used on RSS miss.
* @param {string} publication
* @param {string} id
* @param {number} maxFetch
* @returns {Promise<{hit: PostHit|null, scanned: number, cutoff: string}>} cutoff is "" when the sitemap itself could not be fetched
*/
async function searchSitemap(publication, id, maxFetch) {
const xml = await fetchPage("https://" + publication + "/sitemap.xml");
if (xml === "") return { hit: null, scanned: 0, cutoff: "" };
const cutoffTime = cutoffDate();
const cutoff = cutoffTime.toISOString().slice(0, 10);
const $ = loadXml(xml);
/** @type {{url: string, when: Date}[]} */
const candidates = [];
$("url").each((_i, el) => {
const loc = $(el).children("loc").first().text();
const lastmod = $(el).children("lastmod").first().text();
if (loc === "" || lastmod === "") return;
if (!loc.includes("/p/")) return; // posts only
const when = parseLastmod(lastmod);
if (when === null || when < cutoffTime) return;
candidates.push({ url: loc, when });
});
candidates.sort((a, b) => b.when.getTime() - a.when.getTime());
let scanned = 0;
for (const c of candidates.slice(0, maxFetch)) {
scanned++;
const html = await fetchPage(c.url);
if (html === "" || !html.includes(id)) continue;
return {
hit: {
found: true,
source: "sitemap",
publication,
postTitle: postTitleFromHtml(html),
postUrl: c.url,
caption: captionForUuid(html, id),
candidates: extractCandidates(html),
},
scanned,
cutoff,
};
}
return { hit: null, scanned, cutoff };
}
/**
* searchRss looks for the uuid in a publication's recent feed items.
* @param {string} publication
* @param {string} id
* @returns {Promise<PostHit|null>}
*/
async function searchRss(publication, id) {
const xml = await fetchPage("https://" + publication + "/feed");
if (xml === "") return null;
const $ = loadXml(xml);
const items = $("item").toArray();
for (const el of items) {
const item = $.html(el);
if (!item.includes(id)) continue;
return {
found: true,
source: "rss",
publication,
postTitle: itemTitle(item),
postUrl: itemLink(item),
caption: captionForUuid(item, id),
candidates: extractCandidates(item),
};
}
return null;
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runFindSubstackPost(args) {
let values;
try {
({ values } = parseArgs({
args,
options: { uuid: { type: "string" }, deep: { type: "boolean" } },
allowPositionals: false,
}));
} catch (err) {
process.stderr.write(String(err?.message ?? err) + "\n");
process.stderr.write(
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
);
process.exit(2);
}
const uuid = values.uuid ?? "";
if (uuid === "") {
process.stderr.write(
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
);
process.exit(1);
}
const publications = loadPublications();
for (const pub of publications) {
const hit = await searchRss(pub, uuid);
if (hit !== null) {
printJson(hit);
return;
}
}
// Deep fallback: sitemap crawl up to ~3 months back, sharing one global fetch
// budget across all publications so coverage can't blow up as the
// publications list grows.
if (values.deep === true) {
let totalScanned = 0;
/** @type {string|null} */
let lastCutoff = null;
for (const pub of publications) {
const remaining = DEEP_FETCH_BUDGET - totalScanned;
if (remaining <= 0) break;
const { hit, scanned, cutoff } = await searchSitemap(pub, uuid, remaining);
totalScanned += scanned;
if (cutoff !== "") lastCutoff = cutoff;
if (hit !== null) {
printJson(hit);
return;
}
}
// cutoff is null (not omitted) when no sitemap could be fetched — the skill
// distinguishes "crawled and missed" from "could not crawl".
printJson({
found: false,
source: "sitemap",
scanned: totalScanned,
budget: DEEP_FETCH_BUDGET,
cutoff: lastCutoff,
});
return;
}
printJson({ found: false });
}
+181
View File
@@ -0,0 +1,181 @@
// HTML / RSS text-extraction helpers for find-substack-post, ported from
// html_text.go. Pure string functions — no network, no fs. A real parser
// (cheerio) replaces the hand-rolled regexes the Go version used.
import * as cheerio from "cheerio";
/**
* loadXml parses an XML fragment (RSS item, sitemap) with CDATA recognition.
* @param {string} src
* @returns {cheerio.CheerioAPI}
*/
export function loadXml(src) {
return cheerio.load(src, { xmlMode: true }, false);
}
/**
* loadHtml parses an HTML blob. CDATA markers are stripped first: RSS carries
* post HTML inside CDATA, and the HTML parser would otherwise swallow it as a
* bogus comment.
* @param {string} src
* @returns {cheerio.CheerioAPI}
*/
export function loadHtml(src) {
return cheerio.load(src.split("<![CDATA[").join("").split("]]>").join(""));
}
/**
* rawInner returns an element's serialized inner content with any CDATA wrapper
* removed — the same substring the Go regexes captured, so entity handling can
* stay identical instead of being applied twice.
* @param {cheerio.CheerioAPI} $
* @param {cheerio.Cheerio<any>} el
* @returns {string}
*/
function rawInner($, el) {
const html = $.html(el);
const start = html.indexOf(">") + 1;
const end = html.lastIndexOf("</");
if (start <= 0 || end < start) return "";
let inner = html.slice(start, end);
if (inner.startsWith("<![CDATA[")) inner = inner.slice(9);
if (inner.endsWith("]]>")) inner = inner.slice(0, -3);
return inner;
}
/**
* decodeEntities decodes HTML entities (named, numeric, hex) and trims.
* @param {string} s
* @returns {string}
*/
export function decodeEntities(s) {
if (s === "") return "";
return cheerio.load(s, null, false).text().trim();
}
/**
* collapseWhitespace squeezes runs of whitespace to one space and trims.
* @param {string} s
* @returns {string}
*/
function collapseWhitespace(s) {
return s.replace(/\s+/g, " ").trim();
}
/**
* stripTags removes markup, decodes entities, and collapses whitespace.
* @param {string} s
* @returns {string}
*/
export function stripTags(s) {
if (s === "") return "";
return collapseWhitespace(cheerio.load(s, null, false).text());
}
/**
* itemTitle pulls the first <title> (CDATA or plain) from an RSS <item> chunk.
* @param {string} item
* @returns {string}
*/
export function itemTitle(item) {
const $ = loadXml(item);
const el = $("title").first();
if (el.length === 0) return "";
return decodeEntities(rawInner($, el));
}
/**
* itemLink pulls the first <link> (CDATA or plain) from an RSS <item> chunk.
* Entities are left as stored — the link goes straight into a post.
* @param {string} item
* @returns {string}
*/
export function itemLink(item) {
const $ = loadXml(item);
const el = $("link").first();
if (el.length === 0) return "";
return rawInner($, el).trim();
}
/**
* extractCandidates pulls candidate topic titles from a post's TOC bullet list.
* ByteByteGo does not attach captions to images — the topic titles live only in
* the "in this issue" bullets, and image→title cannot be mapped automatically
* (sponsor/video items interleave), so these are surfaced for the user to pick
* from. Light filtering keeps the list short: dedupe, drop sub-point
* explanations and over-long lines.
* @param {string} htmlSrc
* @returns {string[]}
*/
export function extractCandidates(htmlSrc) {
const $ = loadHtml(htmlSrc);
/** @type {Set<string>} */
const seen = new Set();
/** @type {string[]} */
const out = [];
$("li").each((_i, el) => {
const text = collapseWhitespace($(el).text());
if (text === "") return;
// Titles are short; long lines are sub-point explanations. Count code
// points, not UTF-16 units, so an emoji does not count double.
const n = [...text].length;
if (n < 6 || n > 70) return;
const key = text.toLowerCase();
if (seen.has(key)) return; // content is duplicated in the page
seen.add(key);
out.push(text);
});
return out;
}
/**
* captionForUuid returns the <figcaption> text of the <figure> containing the
* UUID. Cover images live in <enclosure> (no figure) → "".
* @param {string} src
* @param {string} id
* @returns {string}
*/
export function captionForUuid(src, id) {
if (id === "") return "";
const $ = loadHtml(src);
let target = null;
$("*").each((_i, el) => {
if (target !== null) return false;
for (const v of Object.values(el.attribs ?? {})) {
if (typeof v === "string" && v.includes(id)) {
target = el;
return false;
}
}
for (const child of el.children ?? []) {
if (child.type === "text" && String(child.data).includes(id)) {
target = el;
return false;
}
}
return undefined;
});
if (target === null) return "";
const figure = $(target).closest("figure");
if (figure.length === 0) return "";
const caption = figure.find("figcaption").first();
if (caption.length === 0) return "";
return collapseWhitespace(caption.text());
}
/**
* postTitleFromHtml extracts a post title from server-rendered post HTML
* (og:title preferred, then <h1>, then <title>).
* @param {string} htmlSrc
* @returns {string}
*/
export function postTitleFromHtml(htmlSrc) {
const $ = loadHtml(htmlSrc);
const og = $('meta[property="og:title"]').first().attr("content");
if (og !== undefined && og !== "") return og.trim();
const h1 = $("h1").first();
if (h1.length > 0) return collapseWhitespace(h1.text());
const title = $("title").first();
if (title.length > 0) return title.text().trim();
return "";
}
+88
View File
@@ -0,0 +1,88 @@
// Newsletter engine for the mt-* skills — one entry point, one subcommand per
// task. Invoked from the repo root:
//
// node scripts/newsletter <command> [args]
//
// Repo-relative paths (content/post) resolve from the working directory, so the
// repo-root invocation contract from AGENTS.md still applies.
const USAGE = `Usage: node scripts/newsletter <command> [args]
Commands:
add-url <url> classify + dedup a URL, emit JSON route
find-newsletter-number print the next newsletter number
list-existing-tags tag frequencies, most-used first (top 40)
detect-image-source <url> detect Substack image + uuid
find-substack-post --uuid <uuid> [--deep] find the post embedding an image uuid
fetch-via-defuddle <url> fallback fetch (local defuddle, then defuddle.md)
post-stats <path/to/index.md> count the post's articles/images/videos/documents
`;
// A reader that closes early (`… | head -3`) makes the next write fail with
// EPIPE, which Node surfaces as an unhandled error event. Stop quietly instead
// of printing a stack trace over the user's terminal.
process.stdout.on("error", (err) => {
if (err.code === "EPIPE") process.exit(0);
throw err;
});
/** @returns {void} */
function usage() {
process.stderr.write(USAGE);
}
/**
* Command modules are imported lazily so a missing node_modules reports the one
* actionable fix instead of a module-resolution stack trace.
* @param {string} spec
* @returns {Promise<Record<string, any>>}
*/
async function loadCommand(spec) {
try {
return await import(spec);
} catch (err) {
if (err !== null && typeof err === "object" && err.code === "ERR_MODULE_NOT_FOUND") {
process.stderr.write(
"newsletter engine: dependencies missing — run 'npm ci' from the repo root\n",
);
process.exit(1);
}
throw err;
}
}
/** @type {Record<string, {module: string, fn: string}>} */
const COMMANDS = {
"add-url": { module: "./add-url.js", fn: "runAddUrl" },
"find-newsletter-number": { module: "./find-newsletter-number.js", fn: "runFindNewsletterNumber" },
"list-existing-tags": { module: "./list-existing-tags.js", fn: "runListExistingTags" },
"detect-image-source": { module: "./detect-image-source.js", fn: "runDetectImageSource" },
"find-substack-post": { module: "./find-substack-post.js", fn: "runFindSubstackPost" },
"fetch-via-defuddle": { module: "./fetch-via-defuddle.js", fn: "runFetchViaDefuddle" },
"post-stats": { module: "./post-stats.js", fn: "runPostStats" },
};
/** @returns {Promise<void>} */
async function main() {
const argv = process.argv.slice(2);
if (argv.length < 1) {
usage();
process.exit(1);
}
const [name, ...args] = argv;
const entry = COMMANDS[name];
if (entry === undefined) {
process.stderr.write(`unknown command: ${name}\n`);
usage();
process.exit(1);
}
const mod = await loadCommand(entry.module);
await mod[entry.fn](args);
}
try {
await main();
} catch (err) {
process.stderr.write("newsletter engine: " + String(err?.stack ?? err) + "\n");
process.exit(1);
}
+13
View File
@@ -0,0 +1,13 @@
// The shared JSON printer. It lives in its own module rather than in index.js so
// that importing a command module never pulls in the dispatcher — an import
// cycle there would run the CLI as a side effect of any import.
/**
* printJson mirrors console.log(JSON.stringify(v, null, 2)): 2-space indent, no
* HTML escaping (URLs with & must stay readable), trailing newline.
* @param {unknown} v
* @returns {void}
*/
export function printJson(v) {
process.stdout.write(JSON.stringify(v, null, 2) + "\n");
}
+63
View File
@@ -0,0 +1,63 @@
// List existing tags in the repo ranked by frequency.
// Usage: node scripts/newsletter list-existing-tags
// Outputs: tag count and name, sorted most-used first (top 40).
//
// NOTE: not currently used by the mt-add-tags skill. Tag normalization is
// disabled until existing posts have standardized tags; to enable, uncomment
// step 4a in that skill's SKILL.md.
import { readFileSync } from "node:fs";
import { basename } from "node:path";
import { collectMarkdown, contentDir } from "./url-utils.js";
const TAGS_LINE_RE = /^tags:\s*\[([^\]]*)\]/m;
const QUOTED_TAG_RE = /"([^"]+)"/g;
/**
* extractTags pulls quoted tag strings from an index.md frontmatter tags array.
* @param {string} path
* @returns {string[]}
*/
function extractTags(path) {
let content;
try {
content = readFileSync(path, "utf8");
} catch {
return [];
}
const m = TAGS_LINE_RE.exec(content);
if (m === null) return [];
return [...m[1].matchAll(QUOTED_TAG_RE)].map((q) => q[1]);
}
/**
* @param {string[]} _args
* @returns {Promise<void>}
*/
export async function runListExistingTags(_args) {
// Array + index map keeps first-seen order for equal counts, so the stable
// sort below ranks ties deterministically (walk order is lexical).
/** @type {{tag: string, count: number}[]} */
const counts = [];
/** @type {Map<string, number>} */
const index = new Map();
for (const path of collectMarkdown(contentDir())) {
if (basename(path) !== "index.md") continue;
for (const tag of extractTags(path)) {
const at = index.get(tag);
if (at !== undefined) counts[at].count++;
else {
index.set(tag, counts.length);
counts.push({ tag, count: 1 });
}
}
}
counts.sort((a, b) => b.count - a.count);
// Plain text, not JSON: the count is right-aligned in a 6-character field and
// the skill reads that layout.
for (const { tag, count } of counts.slice(0, 40)) {
process.stdout.write(String(count).padStart(6) + " " + tag + "\n");
}
}
+94
View File
@@ -0,0 +1,94 @@
// Count the entries already present in a newsletter post, so a handler can
// report a running tally after each insertion.
// Usage: node scripts/newsletter post-stats <path/to/index.md>
// Outputs: JSON { post, newsletter, articles, images, videos, documents, total }
import { readFileSync } from "node:fs";
import { printJson } from "./json-out.js";
import { NEWSLETTER_NUM_RE } from "./find-newsletter-number.js";
// Entry shapes, per the Bonus format in the shared post mechanics:
//
// articles "## [Title](url)" (level-2 heading, main content)
// images "![label](url)" (under **Images:**)
// videos "[Title](url)" (under **Videos:**)
// documents "[PDF: title](url)" (under **Documents:**)
const ARTICLE_HEADING_RE = /^##\s+\[/;
const BONUS_HEADING_RE = /^###\s+Bonus\b/;
const SUBSECTION_RE = /^\*\*(Images|Videos|Documents):\*\*/;
const IMAGE_ENTRY_RE = /^!\[/;
const LINK_ENTRY_RE = /^\[/;
/**
* countPostEntries walks the post once. Article headings are counted anywhere
* outside Bonus; asset entries are attributed to whichever subsection is open.
* @param {string} content
* @returns {{articles: number, images: number, videos: number, documents: number, total: number}}
*/
export function countPostEntries(content) {
let articles = 0;
let images = 0;
let videos = 0;
let documents = 0;
let inBonus = false;
let subsection = "";
for (const raw of content.split("\n")) {
const line = raw.trim();
if (BONUS_HEADING_RE.test(line)) {
inBonus = true;
subsection = "";
continue;
}
const sub = SUBSECTION_RE.exec(line);
if (sub !== null) {
subsection = sub[1];
continue;
}
if (ARTICLE_HEADING_RE.test(line)) {
articles++;
continue;
}
if (!inBonus) continue;
if (subsection === "Images") {
if (IMAGE_ENTRY_RE.test(line)) images++;
} else if (subsection === "Videos") {
// A direct video file entry looks the same as a YouTube entry;
// both belong to the Videos tally.
if (LINK_ENTRY_RE.test(line)) videos++;
} else if (subsection === "Documents") {
if (LINK_ENTRY_RE.test(line)) documents++;
}
}
return { articles, images, videos, documents, total: articles + images + videos + documents };
}
/**
* @param {string[]} args
* @returns {Promise<void>}
*/
export async function runPostStats(args) {
if (args.length < 1) {
process.stderr.write("usage: post-stats <path/to/index.md>\n");
process.exit(1);
}
const path = args[0];
let content;
try {
content = readFileSync(path, "utf8");
} catch (err) {
process.stderr.write("read post: " + String(err?.message ?? err) + "\n");
process.exit(1);
}
const counted = countPostEntries(content);
const m = NEWSLETTER_NUM_RE.exec(content);
printJson({
post: path,
newsletter: m === null ? 0 : Number.parseInt(m[1], 10),
articles: counted.articles,
images: counted.images,
videos: counted.videos,
documents: counted.documents,
total: counted.total,
});
}
+339
View File
@@ -0,0 +1,339 @@
// Shared URL helpers, ported from url_utils.go. Owned by the add-url router;
// reused by the other subcommands.
import { readdirSync, readFileSync } from "node:fs";
import { join } from "node:path";
/** Exact-match tracking params; any key starting with utm_ is also dropped. */
const EXACT_TRACKING = new Set([
"fbclid", "gclid", "msclkid", "mc_eid",
"aid", "ref", "ref_src", "ref_url", "source", "s",
"ck_subscriber_id", "igshid", "yclid", "vero_id",
]);
/**
* contentDir is repo-root relative (invocation contract: run from repo root).
* @returns {string}
*/
export function contentDir() {
return join("content", "post");
}
/**
* cleanUrl removes common tracking parameters. Surviving query pairs are kept
* verbatim (no re-encoding) and in their original order — URLSearchParams would
* normalize percent-encoding (%7E → ~, + → %20), which must not leak into
* stored clean_url values. Unparseable / non-absolute input is returned
* untouched rather than corrupted.
* @param {string} raw
* @returns {string}
*/
export function cleanUrl(raw) {
let u;
try {
u = new URL(raw);
} catch {
return raw;
}
if (!u.protocol || !u.host) return raw;
// WHATWG URL serializes an empty path as "/" for special schemes; force it
// for the rest so cleaned URLs keep one shape.
if (u.pathname === "") {
try {
u.pathname = "/";
} catch {
/* opaque path — leave as parsed */
}
}
let kept = "";
const query = u.search.slice(1);
if (query !== "") {
const parts = [];
for (const pair of query.split("&")) {
if (pair === "") continue;
const eq = pair.indexOf("=");
const key = (eq === -1 ? pair : pair.slice(0, eq)).toLowerCase();
if (key.startsWith("utm_") || EXACT_TRACKING.has(key)) continue;
parts.push(pair);
}
kept = parts.join("&");
}
const hash = u.hash;
u.search = "";
u.hash = "";
return u.toString() + (kept === "" ? "" : "?" + kept) + hash;
}
// --- Substack image helpers (shared by add-url routing and detect-image-source) ---
const SUBSTACK_IMAGE_HOSTS = new Set([
"substackcdn.com",
"substack-post-media.s3.amazonaws.com",
]);
const UUID_PATTERN = "[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}";
const IMAGE_UUID_RE = new RegExp(`images(?:%2F|/)(${UUID_PATTERN})`, "i");
const UUID_EXACT_RE = new RegExp(`^${UUID_PATTERN}$`, "i");
/**
* isSubstackImage reports a Substack-hosted image (CDN wrapper or raw S3),
* regardless of file extension.
* @param {string} target
* @returns {boolean}
*/
export function isSubstackImage(target) {
let host = "";
try {
host = new URL(target).host.toLowerCase();
} catch {
/* not an absolute URL — fall through to the substring check */
}
return SUBSTACK_IMAGE_HOSTS.has(host) || target.toLowerCase().includes("substack-post-media");
}
/**
* substackImageUuid extracts the stable image identity: the S3 image UUID under
* public/images/<uuid>, with raw (/) or percent-encoded (%2F) separators.
* @param {string} target
* @returns {string} the lowercased uuid, or "" when absent
*/
export function substackImageUuid(target) {
const m = IMAGE_UUID_RE.exec(target);
return m === null ? "" : m[1].toLowerCase();
}
/**
* isUuid reports whether a bare identity is a Substack image uuid.
* @param {string} s
* @returns {boolean}
*/
export function isUuid(s) {
return UUID_EXACT_RE.test(s);
}
/**
* Some sites carry the resource identity in a query param, not the path
* (e.g. YouTube /watch?v=ID). Preserve the identity param for those hosts so
* dedup does not collapse every video onto the same bare URL.
*/
const IDENTITY_PARAMS = new Map([
["youtube.com", "v"],
["www.youtube.com", "v"],
["m.youtube.com", "v"],
]);
/**
* trimTrailingSlash removes at most one trailing "/".
* @param {string} s
* @returns {string}
*/
function trimTrailingSlash(s) {
return s.endsWith("/") ? s.slice(0, -1) : s;
}
/**
* bareUrl reduces a URL to a stable identity for duplicate detection:
* - Substack image → its S3 UUID (transform/size variants share one identity)
* - YouTube → scheme+host+path + the v= video id
* - everything else → scheme + host + path
* @param {string} target
* @returns {string}
*/
export function bareUrl(target) {
if (isSubstackImage(target)) {
const uuid = substackImageUuid(target);
if (uuid !== "") return uuid;
}
let u = null;
try {
u = new URL(target);
} catch {
/* unparseable — fall back to crude string surgery below */
}
if (u === null || !u.protocol || !u.host) {
return trimTrailingSlash(target.split("?")[0]);
}
const host = u.host.toLowerCase();
let bare = trimTrailingSlash(u.protocol + "//" + host + u.pathname);
const idParam = IDENTITY_PARAMS.get(host);
if (idParam !== undefined) {
const v = u.searchParams.get(idParam);
if (v !== null && v !== "") bare += "?" + idParam + "=" + v;
}
return bare;
}
/**
* fetchTextOk GETs target and returns the body on a 2xx response, "" on any
* error, non-2xx status, or timeout. Redirects are followed.
* @param {string} target
* @param {number} timeoutMs
* @param {string} [userAgent]
* @returns {Promise<string>}
*/
export async function fetchTextOk(target, timeoutMs, userAgent = "") {
try {
const headers = {};
if (userAgent !== "") headers["user-agent"] = userAgent;
const res = await fetch(target, {
headers,
redirect: "follow",
signal: AbortSignal.timeout(timeoutMs),
});
if (res.status < 200 || res.status >= 300) return "";
return await res.text();
} catch {
return "";
}
}
/**
* checkAccessibility HEADs the URL and returns the final HTTP status code as a
* string, or "000" on network error / timeout.
* @param {string} target
* @returns {Promise<string>}
*/
export async function checkAccessibility(target) {
try {
const res = await fetch(target, {
method: "HEAD",
redirect: "follow",
signal: AbortSignal.timeout(10_000),
});
return String(res.status);
} catch {
return "000";
}
}
/**
* collectMarkdown recursively collects *.md files under dir (the content tree is
* small). Entries are visited in lexical order so callers that depend on
* first-seen ordering stay deterministic. Missing or unreadable directories
* yield nothing.
* @param {string} dir
* @returns {string[]}
*/
export function collectMarkdown(dir) {
/** @type {string[]} */
const acc = [];
walkMarkdown(dir, acc);
return acc;
}
/**
* @param {string} dir
* @param {string[]} acc
* @returns {void}
*/
function walkMarkdown(dir, acc) {
let entries;
try {
entries = readdirSync(dir, { withFileTypes: true });
} catch {
return; // skip unreadable entries
}
entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
for (const e of entries) {
const p = join(dir, e.name);
if (e.isDirectory()) walkMarkdown(p, acc);
else if (e.name.toLowerCase().endsWith(".md")) acc.push(p);
}
}
/**
* uuidBoundaryOk: the char after a UUID match must not extend the hex id, so
* <uuid>.png (cover image), <uuid>_WxH and <uuid>) all match.
* @param {string} text
* @param {number} end
* @returns {boolean}
*/
export function uuidBoundaryOk(text, end) {
if (end >= text.length) return true;
const c = text[end];
return !((c >= "0" && c <= "9") || (c >= "a" && c <= "f") || (c >= "A" && c <= "F"));
}
const URL_DELIMITERS = `)]"'?#<>_&,`;
const URL_WHITESPACE = " \t\n\r\f\v";
/**
* urlBoundaryOk: an optional trailing slash (bareUrl strips it, stored URLs may
* keep it), then a path/punctuation delimiter, whitespace, or end of text — so
* /p/foo does not match a stored /p/foo-bar. The '>' delimiter covers URLs
* stored in markdown autolink form <https://…>, which older posts use.
* @param {string} text
* @param {number} end
* @returns {boolean}
*/
export function urlBoundaryOk(text, end) {
let at = end;
if (at < text.length && text[at] === "/") at++;
if (at >= text.length) return true;
const c = text[at];
if (URL_WHITESPACE.includes(c)) return true;
return URL_DELIMITERS.includes(c);
}
/**
* hasBoundaryMatch scans every occurrence of needle and applies the boundary
* check in code. Kept as an index loop rather than one lookahead regex: the
* byte-offset checks (optional trailing slash, delimiter set) are what stop a
* needle that is merely a PREFIX of a stored string from matching.
* @param {string} text
* @param {string} needle
* @param {boolean} isUuidNeedle
* @returns {boolean}
*/
export function hasBoundaryMatch(text, needle, isUuidNeedle) {
for (let from = 0; ; ) {
const i = text.indexOf(needle, from);
if (i === -1) return false;
const end = i + needle.length;
if (isUuidNeedle ? uuidBoundaryOk(text, end) : urlBoundaryOk(text, end)) return true;
from = i + 1;
}
}
/**
* checkDuplicate reports whether a URL identity already exists in the stored
* markdown, boundary-aware so a needle that is merely a PREFIX of a stored
* longer string is NOT a false duplicate.
* @param {string} target
* @param {string} dir
* @returns {boolean}
*/
export function checkDuplicate(target, dir) {
const needle = bareUrl(target);
if (needle === "") return false;
const uuidNeedle = isUuid(needle);
for (const file of collectMarkdown(dir)) {
let text;
try {
text = readFileSync(file, "utf8");
} catch {
continue;
}
if (text.includes(needle) && hasBoundaryMatch(text, needle, uuidNeedle)) return true;
}
return false;
}
const IMAGE_EXT_RE = /\.(png|jpg|jpeg|gif|webp|svg|avif|heic|heif|bmp|tiff?)(\?.*)?$/;
const VIDEO_EXT_RE = /\.(mp4|webm|mov|avi|mkv)(\?.*)?$/;
const DOCUMENT_EXT_RE = /\.(pdf|docx?|xlsx?|pptx?)(\?.*)?$/;
/**
* classifyType classifies a URL by file extension.
* @param {string} target
* @returns {"image"|"video"|"document"|"article"}
*/
export function classifyType(target) {
const lower = target.toLowerCase();
if (IMAGE_EXT_RE.test(lower)) return "image";
if (VIDEO_EXT_RE.test(lower)) return "video";
if (DOCUMENT_EXT_RE.test(lower)) return "document";
return "article";
}