mirror of
https://github.com/tiennm99/blog.git
synced 2026-10-11 03:13:10 +00:00
feat(newsletter): port the engine to JavaScript
Translates the seven-subcommand engine to Node ESM, one module per former Go file, invoked as `node scripts/newsletter <command>` from the repo root. The Go implementation stays in place for now so parity can be measured against it. Hand-rolled HTML and XML regexes give way to cheerio, which removes the manual string surgery in caption extraction and covers RSS and sitemap XML through xmlMode without a second parser. fetch-via-defuddle gains a local extraction stage ahead of the defuddle.md proxy; the proxy stays, because fetching from a third IP is the whole point when this machine's IP is the blocked one. Both stages now emit the same YAML frontmatter plus body, and an extraction that looks like a bot challenge counts as a local failure so the proxy still runs. Behaviour is preserved where it is load-bearing rather than where it is merely idiomatic: query strings are rebuilt by string surgery so surviving parameters keep their original order and encoding, duplicate detection stays an index loop with byte-offset boundary checks so a prefix of a stored URL is not a false match, empty optional fields are omitted rather than emitted as "", a deep-crawl miss reports cutoff as null rather than dropping the key, bullet filtering counts code points, tag counts keep their six-column alignment, and a malformed percent sequence falls back to the raw substring. The printer lives in its own module so a command module never imports the dispatcher: index.js runs main() at module scope, so that cycle would execute the CLI as a side effect of any import. Missing dependencies report one actionable line instead of a module-resolution stack trace, and a reader that closes early exits quietly instead of raising EPIPE.
This commit is contained in:
1 parent
fdbbc5ba42
commit
64c7b76b4e
14 files changed
+1995
-2
No files matched your search
+7
-1
@@ -8,7 +8,13 @@ export default [
|
||||
languageOptions: {
|
||||
ecmaVersion: "latest",
|
||||
sourceType: "module",
|
||||
globals: { console: "readonly", process: "readonly", fetch: "readonly" },
|
||||
globals: {
|
||||
console: "readonly",
|
||||
process: "readonly",
|
||||
fetch: "readonly",
|
||||
URL: "readonly",
|
||||
AbortSignal: "readonly",
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
"no-unused-vars": ["error", { argsIgnorePattern: "^_" }],
|
||||
|
||||
Generated
+559
@@ -8,6 +8,9 @@
|
||||
"name": "blog",
|
||||
"version": "0.0.0",
|
||||
"dependencies": {
|
||||
"cheerio": "^1.2.0",
|
||||
"defuddle": "^0.19.4",
|
||||
"linkedom": "^0.18.13",
|
||||
"octokit": "^5.0.5"
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -260,6 +263,13 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@mixmark-io/domino": {
|
||||
"version": "2.2.0",
|
||||
"resolved": "https://registry.npmjs.org/@mixmark-io/domino/-/domino-2.2.0.tgz",
|
||||
"integrity": "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw==",
|
||||
"license": "BSD-2-Clause",
|
||||
"optional": true
|
||||
},
|
||||
"node_modules/@octokit/app": {
|
||||
"version": "16.1.4",
|
||||
"resolved": "https://registry.npmjs.org/@octokit/app/-/app-16.1.4.tgz",
|
||||
@@ -854,6 +864,15 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@xmldom/xmldom": {
|
||||
"version": "0.9.12",
|
||||
"resolved": "https://registry.npmjs.org/@xmldom/xmldom/-/xmldom-0.9.12.tgz",
|
||||
"integrity": "sha512-5AXjrcMClTryPe9LgZrygpB1lj7s0S9E0+W+AHaVKAVyHanafK86iPSvG5xHVSp/jC+VH1UXu0TAEmY279xH7A==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=14.6"
|
||||
}
|
||||
},
|
||||
"node_modules/acorn": {
|
||||
"version": "8.18.0",
|
||||
"resolved": "https://registry.npmjs.org/acorn/-/acorn-8.18.0.tgz",
|
||||
@@ -910,6 +929,12 @@
|
||||
"integrity": "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ==",
|
||||
"license": "Apache-2.0"
|
||||
},
|
||||
"node_modules/boolbase": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/boolbase/-/boolbase-1.0.0.tgz",
|
||||
"integrity": "sha512-JZOSA7Mo9sNGB8+UjSgzdLtokWAky1zbztM3WRLCbZ70/3cTANmQmOdR7y2g+J0e2WXywy1yS468tY+IruqEww==",
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/bottleneck": {
|
||||
"version": "2.19.5",
|
||||
"resolved": "https://registry.npmjs.org/bottleneck/-/bottleneck-2.19.5.tgz",
|
||||
@@ -943,6 +968,57 @@
|
||||
"qified": "^0.10.1"
|
||||
}
|
||||
},
|
||||
"node_modules/cheerio": {
|
||||
"version": "1.2.0",
|
||||
"resolved": "https://registry.npmjs.org/cheerio/-/cheerio-1.2.0.tgz",
|
||||
"integrity": "sha512-WDrybc/gKFpTYQutKIK6UvfcuxijIZfMfXaYm8NMsPQxSYvf+13fXUJ4rztGGbJcBQ/GF55gvrZ0Bc0bj/mqvg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"cheerio-select": "^2.1.0",
|
||||
"dom-serializer": "^2.0.0",
|
||||
"domhandler": "^5.0.3",
|
||||
"domutils": "^3.2.2",
|
||||
"encoding-sniffer": "^0.2.1",
|
||||
"htmlparser2": "^10.1.0",
|
||||
"parse5": "^7.3.0",
|
||||
"parse5-htmlparser2-tree-adapter": "^7.1.0",
|
||||
"parse5-parser-stream": "^7.1.2",
|
||||
"undici": "^7.19.0",
|
||||
"whatwg-mimetype": "^4.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.18.1"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/cheeriojs/cheerio?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/cheerio-select": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/cheerio-select/-/cheerio-select-2.1.0.tgz",
|
||||
"integrity": "sha512-9v9kG0LvzrlcungtnJtpGNxY+fzECQKhK4EGJX2vByejiMX84MFNQw4UxPJl3bFbTMw+Dfs37XaIkCwTZfLh4g==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"boolbase": "^1.0.0",
|
||||
"css-select": "^5.1.0",
|
||||
"css-what": "^6.1.0",
|
||||
"domelementtype": "^2.3.0",
|
||||
"domhandler": "^5.0.3",
|
||||
"domutils": "^3.0.1"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/commander": {
|
||||
"version": "12.1.0",
|
||||
"resolved": "https://registry.npmjs.org/commander/-/commander-12.1.0.tgz",
|
||||
"integrity": "sha512-Vw8qHK3bZM9y/P10u3Vib8o/DdkvA2OtPtZvD871QKjy74Wj1WSKFILMPRPSdUSx5RFK1arlJzEtA4PkFgnbuA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/content-type": {
|
||||
"version": "3.1.1",
|
||||
"resolved": "https://registry.npmjs.org/content-type/-/content-type-3.1.1.tgz",
|
||||
@@ -971,6 +1047,40 @@
|
||||
"node": ">= 8"
|
||||
}
|
||||
},
|
||||
"node_modules/css-select": {
|
||||
"version": "5.2.2",
|
||||
"resolved": "https://registry.npmjs.org/css-select/-/css-select-5.2.2.tgz",
|
||||
"integrity": "sha512-TizTzUddG/xYLA3NXodFM0fSbNizXjOKhqiQQwvhlspadZokn1KDy0NZFS0wuEubIYAV5/c1/lAr0TaaFXEXzw==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"boolbase": "^1.0.0",
|
||||
"css-what": "^6.1.0",
|
||||
"domhandler": "^5.0.2",
|
||||
"domutils": "^3.0.1",
|
||||
"nth-check": "^2.0.1"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/css-what": {
|
||||
"version": "6.2.2",
|
||||
"resolved": "https://registry.npmjs.org/css-what/-/css-what-6.2.2.tgz",
|
||||
"integrity": "sha512-u/O3vwbptzhMs3L1fQE82ZSLHQQfto5gyZzwteVIEyeaY5Fc7R4dapF/BvRoSYFeqfBk4m0V1Vafq5Pjv25wvA==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">= 6"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/cssom": {
|
||||
"version": "0.5.0",
|
||||
"resolved": "https://registry.npmjs.org/cssom/-/cssom-0.5.0.tgz",
|
||||
"integrity": "sha512-iKuQcq+NdHqlAcwUY0o/HL69XQrUaQdMjmStJ8JFmUaiiQErlhrmuigkg/CU4E2J0IyUKUrMAgl36TvN67MqTw==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/debug": {
|
||||
"version": "4.4.3",
|
||||
"resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz",
|
||||
@@ -996,6 +1106,104 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/defuddle": {
|
||||
"version": "0.19.4",
|
||||
"resolved": "https://registry.npmjs.org/defuddle/-/defuddle-0.19.4.tgz",
|
||||
"integrity": "sha512-Jz98aWlyEeuWcg9u2AavCEKknviI9GijMkrGFPk0wi1Lkfr1DDDw6XM+ak44Ol8kBDmDJsvU9Su2eUY09Fvh3Q==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"commander": "^12.1.0",
|
||||
"mathml-to-latex": "^1.8.0"
|
||||
},
|
||||
"bin": {
|
||||
"defuddle": "dist/cli.js"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"linkedom": "^0.18.12",
|
||||
"temml": "^0.13.3",
|
||||
"turndown": "^7.2.0"
|
||||
}
|
||||
},
|
||||
"node_modules/dom-serializer": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-2.0.0.tgz",
|
||||
"integrity": "sha512-wIkAryiqt/nV5EQKqQpo3SToSOV9J0DnbJqwK7Wv/Trc92zIAYZ4FlMu+JPFW1DfGFt81ZTCGgDEabffXeLyJg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"domelementtype": "^2.3.0",
|
||||
"domhandler": "^5.0.2",
|
||||
"entities": "^4.2.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/cheeriojs/dom-serializer?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/domelementtype": {
|
||||
"version": "2.3.0",
|
||||
"resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-2.3.0.tgz",
|
||||
"integrity": "sha512-OLETBj6w0OsagBwdXnPdN0cnMfF9opN69co+7ZrbfPGrdpPVNBUj02spi6B1N7wChLQiPn4CSH/zJvXw56gmHw==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
],
|
||||
"license": "BSD-2-Clause"
|
||||
},
|
||||
"node_modules/domhandler": {
|
||||
"version": "5.0.3",
|
||||
"resolved": "https://registry.npmjs.org/domhandler/-/domhandler-5.0.3.tgz",
|
||||
"integrity": "sha512-cgwlv/1iFQiFnU96XXgROh8xTeetsnJiDsTc7TYCLFd9+/WNkIqPTxiM/8pSd8VIrhXGTf1Ny1q1hquVqDJB5w==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"domelementtype": "^2.3.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">= 4"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/domhandler?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/domutils": {
|
||||
"version": "3.2.2",
|
||||
"resolved": "https://registry.npmjs.org/domutils/-/domutils-3.2.2.tgz",
|
||||
"integrity": "sha512-6kZKyUajlDuqlHKVX1w7gyslj9MPIXzIFiz/rGu35uC1wMi+kMhQwGhl4lt9unC9Vb9INnY9Z3/ZA3+FhASLaw==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"dom-serializer": "^2.0.0",
|
||||
"domelementtype": "^2.3.0",
|
||||
"domhandler": "^5.0.3"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/domutils?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/encoding-sniffer": {
|
||||
"version": "0.2.1",
|
||||
"resolved": "https://registry.npmjs.org/encoding-sniffer/-/encoding-sniffer-0.2.1.tgz",
|
||||
"integrity": "sha512-5gvq20T6vfpekVtqrYQsSCFZ1wEg5+wW0/QaZMWkFr6BqD3NfKs0rLCx4rrVlSWJeZb5NBJgVLswK/w2MWU+Gw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"iconv-lite": "^0.6.3",
|
||||
"whatwg-encoding": "^3.1.1"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/encoding-sniffer?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/entities": {
|
||||
"version": "4.5.0",
|
||||
"resolved": "https://registry.npmjs.org/entities/-/entities-4.5.0.tgz",
|
||||
"integrity": "sha512-V0hjH4dGPh9Ao5p0MoRY6BVqtwCjhz6vI5LT8AJ55H+4g9/4vbHx1I54fS0XuclLhDHArPQCiMjDxjaL8fPxhw==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=0.12"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/entities?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/escape-string-regexp": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz",
|
||||
@@ -1264,6 +1472,55 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/html-escaper": {
|
||||
"version": "3.0.3",
|
||||
"resolved": "https://registry.npmjs.org/html-escaper/-/html-escaper-3.0.3.tgz",
|
||||
"integrity": "sha512-RuMffC89BOWQoY0WKGpIhn5gX3iI54O6nRA0yC124NYVtzjmFWBIiFd8M0x+ZdX0P9R4lADg1mgP8C7PxGOWuQ==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/htmlparser2": {
|
||||
"version": "10.1.0",
|
||||
"resolved": "https://registry.npmjs.org/htmlparser2/-/htmlparser2-10.1.0.tgz",
|
||||
"integrity": "sha512-VTZkM9GWRAtEpveh7MSF6SjjrpNVNNVJfFup7xTY3UpFtm67foy9HDVXneLtFVt4pMz5kZtgNcvCniNFb1hlEQ==",
|
||||
"funding": [
|
||||
"https://github.com/fb55/htmlparser2?sponsor=1",
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"domelementtype": "^2.3.0",
|
||||
"domhandler": "^5.0.3",
|
||||
"domutils": "^3.2.2",
|
||||
"entities": "^7.0.1"
|
||||
}
|
||||
},
|
||||
"node_modules/htmlparser2/node_modules/entities": {
|
||||
"version": "7.0.1",
|
||||
"resolved": "https://registry.npmjs.org/entities/-/entities-7.0.1.tgz",
|
||||
"integrity": "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=0.12"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/entities?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/iconv-lite": {
|
||||
"version": "0.6.3",
|
||||
"resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.6.3.tgz",
|
||||
"integrity": "sha512-4fCk79wshMdzMp2rH06qWrJE4iolqLhCUH+OiuIgU++RB0+94NlDL81atO7GX55uUKueo0txHNtvEyI6D7WdMw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"safer-buffer": ">= 2.1.2 < 3.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=0.10.0"
|
||||
}
|
||||
},
|
||||
"node_modules/ignore": {
|
||||
"version": "5.3.2",
|
||||
"resolved": "https://registry.npmjs.org/ignore/-/ignore-5.3.2.tgz",
|
||||
@@ -1358,6 +1615,171 @@
|
||||
"node": ">= 0.8.0"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom": {
|
||||
"version": "0.18.13",
|
||||
"resolved": "https://registry.npmjs.org/linkedom/-/linkedom-0.18.13.tgz",
|
||||
"integrity": "sha512-ES/o9qotMpzpN2MHs+Iq/JcVoOj8Fa5wiQYrTdFpvAnwXL0g66XHHUc9WUMk6nAlBtGsFQ24ne+SYnvnaQ2FSw==",
|
||||
"license": "ISC",
|
||||
"dependencies": {
|
||||
"css-select": "^7.0.0",
|
||||
"cssom": "^0.5.0",
|
||||
"html-escaper": "^3.0.3",
|
||||
"htmlparser2": "^10.1.0",
|
||||
"uhyphen": "^0.2.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=16"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"canvas": ">= 2"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"canvas": {
|
||||
"optional": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/boolbase": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/boolbase/-/boolbase-2.0.0.tgz",
|
||||
"integrity": "sha512-DkVaaQHymRhpYEYo9x1oo7Q7B0Y6KJUsjm3c9eTyFDby4MHLBTwZ6ZDWBel5zrYxj1WsZgC5oLpiz+93MluXeA==",
|
||||
"license": "ISC",
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/css-select": {
|
||||
"version": "7.0.0",
|
||||
"resolved": "https://registry.npmjs.org/css-select/-/css-select-7.0.0.tgz",
|
||||
"integrity": "sha512-snmjEVXy+1LnwXdxhYvTMj1d9tOh4HxkA1YmoayVBeeyR2C14Pum7fcxJIm4SswYspVy866eYNwlH6xC3/VH5g==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"boolbase": "^2.0.0",
|
||||
"css-what": "^8.0.0",
|
||||
"domhandler": "^6.0.1",
|
||||
"domutils": "^4.0.2",
|
||||
"nth-check": "^3.0.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/css-what": {
|
||||
"version": "8.0.0",
|
||||
"resolved": "https://registry.npmjs.org/css-what/-/css-what-8.0.0.tgz",
|
||||
"integrity": "sha512-DH0Bqq3DNp5tdOReuNyAA+Ev4Y2GS5FMbZpeTLP6C4CDi0h5nL0BmUPChXw3o/qbHLDWHl49sbNqQVY7bMSDdw==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/dom-serializer": {
|
||||
"version": "3.1.1",
|
||||
"resolved": "https://registry.npmjs.org/dom-serializer/-/dom-serializer-3.1.1.tgz",
|
||||
"integrity": "sha512-4MEa38/QexBob6gFNwu+EGdWvhJ1OKuNwdYY3Y3NyeWDQfnGeDYQUDfIRzWu5B5gsv03so2Uxd28YC6zrsx3Lw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"domelementtype": "^3.0.0",
|
||||
"domhandler": "^6.0.0",
|
||||
"entities": "^8.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/cheeriojs/dom-serializer?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/domelementtype": {
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/domelementtype/-/domelementtype-3.0.0.tgz",
|
||||
"integrity": "sha512-umCQid3jKbDmVjx8jGaW7uUykm4DEUeyV21hPxNMo2nV955DhUThwqyOIDtreepP31hl84X7G5U9ZfsWvIB3Pg==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/fb55"
|
||||
}
|
||||
],
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/domhandler": {
|
||||
"version": "6.0.1",
|
||||
"resolved": "https://registry.npmjs.org/domhandler/-/domhandler-6.0.1.tgz",
|
||||
"integrity": "sha512-gYzvtM72ZtxQO0T048kd6HWSbbGCNOUwcnfQ01cqIJ4X2IYKFFHZ5mKvrQETcFXxsRObZulDaKmy//R7TPtsBg==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"domelementtype": "^3.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/fb55/domhandler?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/domutils": {
|
||||
"version": "4.0.2",
|
||||
"resolved": "https://registry.npmjs.org/domutils/-/domutils-4.0.2.tgz",
|
||||
"integrity": "sha512-qI4JLRKnSzqFqr7hAlS5xQDusBCjKSEG4t4+7aNrIQMHBcsC2TGEhuyABJdYkgSewL57PNLYEiibY2iPKhKpaA==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"dom-serializer": "^3.0.0",
|
||||
"domelementtype": "^3.0.0",
|
||||
"domhandler": "^6.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/fb55/domutils?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/entities": {
|
||||
"version": "8.1.0",
|
||||
"resolved": "https://registry.npmjs.org/entities/-/entities-8.1.0.tgz",
|
||||
"integrity": "sha512-kxL7msIffSuh9aaFAMD7rxAIuTRMAHMeBtgHW2yUdWw732ZNh4MehkF2gdjvtdmikkaIP9bFDDJOPlsvm7avrA==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/entities?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/linkedom/node_modules/nth-check": {
|
||||
"version": "3.0.1",
|
||||
"resolved": "https://registry.npmjs.org/nth-check/-/nth-check-3.0.1.tgz",
|
||||
"integrity": "sha512-GX0gsdbGVCgnRgbeGaubfjpBXyYRWOOCVeYh08bSQvDZqxz5ndXs1OTfAt/h36G1xvI94YIspsI0sVFqAV9+RQ==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"boolbase": "^2.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.19.0"
|
||||
},
|
||||
"funding": {
|
||||
"type": "github",
|
||||
"url": "https://github.com/fb55/nth-check?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/locate-path": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/locate-path/-/locate-path-6.0.0.tgz",
|
||||
@@ -1374,6 +1796,15 @@
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/mathml-to-latex": {
|
||||
"version": "1.8.0",
|
||||
"resolved": "https://registry.npmjs.org/mathml-to-latex/-/mathml-to-latex-1.8.0.tgz",
|
||||
"integrity": "sha512-gQ0uK3zqB8HwlfaXJkEL5rgaZNbKUiBMmBP/B/W+b+t6KcseLSuYb1b0BjLgS9ZiQa24ePkqTX8/6FaQuDL7wQ==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@xmldom/xmldom": "^0.9.10"
|
||||
}
|
||||
},
|
||||
"node_modules/minimatch": {
|
||||
"version": "10.2.6",
|
||||
"resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.6.tgz",
|
||||
@@ -1404,6 +1835,18 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/nth-check": {
|
||||
"version": "2.1.1",
|
||||
"resolved": "https://registry.npmjs.org/nth-check/-/nth-check-2.1.1.tgz",
|
||||
"integrity": "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w==",
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"boolbase": "^1.0.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/nth-check?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/octokit": {
|
||||
"version": "5.0.5",
|
||||
"resolved": "https://registry.npmjs.org/octokit/-/octokit-5.0.5.tgz",
|
||||
@@ -1476,6 +1919,55 @@
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/parse5": {
|
||||
"version": "7.3.0",
|
||||
"resolved": "https://registry.npmjs.org/parse5/-/parse5-7.3.0.tgz",
|
||||
"integrity": "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"entities": "^6.0.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/inikulin/parse5?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/parse5-htmlparser2-tree-adapter": {
|
||||
"version": "7.1.0",
|
||||
"resolved": "https://registry.npmjs.org/parse5-htmlparser2-tree-adapter/-/parse5-htmlparser2-tree-adapter-7.1.0.tgz",
|
||||
"integrity": "sha512-ruw5xyKs6lrpo9x9rCZqZZnIUntICjQAd0Wsmp396Ul9lN/h+ifgVV1x1gZHi8euej6wTfpqX8j+BFQxF0NS/g==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"domhandler": "^5.0.3",
|
||||
"parse5": "^7.0.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/inikulin/parse5?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/parse5-parser-stream": {
|
||||
"version": "7.1.2",
|
||||
"resolved": "https://registry.npmjs.org/parse5-parser-stream/-/parse5-parser-stream-7.1.2.tgz",
|
||||
"integrity": "sha512-JyeQc9iwFLn5TbvvqACIF/VXG6abODeB3Fwmv/TGdLk2LfbWkaySGY72at4+Ty7EkPZj854u4CrICqNk2qIbow==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"parse5": "^7.0.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/inikulin/parse5?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/parse5/node_modules/entities": {
|
||||
"version": "6.0.1",
|
||||
"resolved": "https://registry.npmjs.org/entities/-/entities-6.0.1.tgz",
|
||||
"integrity": "sha512-aN97NXWF6AWBTahfVOIrB/NShkzi5H7F9r1s9mD3cDj4Ko5f2qhhVoYMibXF7GlLveb/D2ioWay8lxI97Ven3g==",
|
||||
"license": "BSD-2-Clause",
|
||||
"engines": {
|
||||
"node": ">=0.12"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/fb55/entities?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/path-exists": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/path-exists/-/path-exists-4.0.0.tgz",
|
||||
@@ -1536,6 +2028,12 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/safer-buffer": {
|
||||
"version": "2.1.2",
|
||||
"resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz",
|
||||
"integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/shebang-command": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz",
|
||||
@@ -1559,6 +2057,16 @@
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/temml": {
|
||||
"version": "0.13.5",
|
||||
"resolved": "https://registry.npmjs.org/temml/-/temml-0.13.5.tgz",
|
||||
"integrity": "sha512-aPkDDgunanpLNL0ql32HbolqLep+w8DRcVXRql7rWrMt/PhczdLgL4UBYYVU3BYjCNjkGq4EqwIicu/zWr1iOg==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": ">=18.13.0"
|
||||
}
|
||||
},
|
||||
"node_modules/toad-cache": {
|
||||
"version": "3.7.4",
|
||||
"resolved": "https://registry.npmjs.org/toad-cache/-/toad-cache-3.7.4.tgz",
|
||||
@@ -1568,6 +2076,20 @@
|
||||
"node": ">=20"
|
||||
}
|
||||
},
|
||||
"node_modules/turndown": {
|
||||
"version": "7.2.4",
|
||||
"resolved": "https://registry.npmjs.org/turndown/-/turndown-7.2.4.tgz",
|
||||
"integrity": "sha512-I8yFsfRzmzK0WV1pNNOA4A7y4RDfFxPRxb3t+e3ui14qSGOxGtiSP6GjeX+Y6CHb7HYaFj7ECUD7VE5kQMZWGQ==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"@mixmark-io/domino": "^2.2.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18",
|
||||
"npm": ">=9"
|
||||
}
|
||||
},
|
||||
"node_modules/type-check": {
|
||||
"version": "0.4.0",
|
||||
"resolved": "https://registry.npmjs.org/type-check/-/type-check-0.4.0.tgz",
|
||||
@@ -1581,6 +2103,21 @@
|
||||
"node": ">= 0.8.0"
|
||||
}
|
||||
},
|
||||
"node_modules/uhyphen": {
|
||||
"version": "0.2.0",
|
||||
"resolved": "https://registry.npmjs.org/uhyphen/-/uhyphen-0.2.0.tgz",
|
||||
"integrity": "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA==",
|
||||
"license": "ISC"
|
||||
},
|
||||
"node_modules/undici": {
|
||||
"version": "7.29.1",
|
||||
"resolved": "https://registry.npmjs.org/undici/-/undici-7.29.1.tgz",
|
||||
"integrity": "sha512-RYONW2MeafgYlkVOKYKkA/Ag7BmXqgIWCa8t1m0JcxrQg9pI9lEqRhAOruOBCbAohOa/gkCF+iPi9hrgvTzu6Q==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=20.18.1"
|
||||
}
|
||||
},
|
||||
"node_modules/universal-github-app-jwt": {
|
||||
"version": "2.2.2",
|
||||
"resolved": "https://registry.npmjs.org/universal-github-app-jwt/-/universal-github-app-jwt-2.2.2.tgz",
|
||||
@@ -1603,6 +2140,28 @@
|
||||
"punycode": "^2.1.0"
|
||||
}
|
||||
},
|
||||
"node_modules/whatwg-encoding": {
|
||||
"version": "3.1.1",
|
||||
"resolved": "https://registry.npmjs.org/whatwg-encoding/-/whatwg-encoding-3.1.1.tgz",
|
||||
"integrity": "sha512-6qN4hJdMwfYBtE3YBTTHhoeuUrDBPZmbQaxWAqSALV/MeEnR5z1xd8UKud2RAkFoPkmB+hli1TZSnyi84xz1vQ==",
|
||||
"deprecated": "Use @exodus/bytes instead for a more spec-conformant and faster implementation",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"iconv-lite": "0.6.3"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/whatwg-mimetype": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/whatwg-mimetype/-/whatwg-mimetype-4.0.0.tgz",
|
||||
"integrity": "sha512-QaKxh0eNIi2mE9p2vEdzfagOKHCcj1pJ56EEHGQOVxp8r9/iszLUUV7v89x9O1p/T+NlTM5W7jW6+cz4Fq1YVg==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
}
|
||||
},
|
||||
"node_modules/which": {
|
||||
"version": "2.0.2",
|
||||
"resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz",
|
||||
|
||||
+5
-1
@@ -9,9 +9,13 @@
|
||||
},
|
||||
"scripts": {
|
||||
"lint": "eslint .",
|
||||
"projects:refresh": "node .github/scripts/update-projects-list.js"
|
||||
"projects:refresh": "node .github/scripts/update-projects-list.js",
|
||||
"test": "node --test scripts/newsletter/*.test.js"
|
||||
},
|
||||
"dependencies": {
|
||||
"cheerio": "^1.2.0",
|
||||
"defuddle": "^0.19.4",
|
||||
"linkedom": "^0.18.13",
|
||||
"octokit": "^5.0.5"
|
||||
},
|
||||
"devDependencies": {
|
||||
|
||||
@@ -0,0 +1,127 @@
|
||||
// Meta URL router for the mt-add-url skill — the single entry per URL.
|
||||
// Usage: node scripts/newsletter add-url "<url>"
|
||||
// Outputs: JSON { original_url, clean_url, http_status, accessible,
|
||||
// duplicate, route, title?, author? }
|
||||
//
|
||||
// route ∈ youtube | image | video | document | article
|
||||
|
||||
import { printJson } from "./json-out.js";
|
||||
import {
|
||||
checkAccessibility,
|
||||
checkDuplicate,
|
||||
classifyType,
|
||||
cleanUrl,
|
||||
contentDir,
|
||||
fetchTextOk,
|
||||
isSubstackImage,
|
||||
} from "./url-utils.js";
|
||||
|
||||
const YT_HOSTS = new Set(["youtube.com", "www.youtube.com", "m.youtube.com"]);
|
||||
|
||||
/**
|
||||
* detectYouTube extracts a video id from the supported URL shapes:
|
||||
* youtube.com/watch?v=ID, youtu.be/ID, youtube.com/shorts/ID.
|
||||
* Playlists/channels are intentionally NOT YouTube routes (fall through to type).
|
||||
* @param {string} target
|
||||
* @returns {{isYouTube: boolean, videoId: string}}
|
||||
*/
|
||||
export function detectYouTube(target) {
|
||||
let u;
|
||||
try {
|
||||
u = new URL(target);
|
||||
} catch {
|
||||
return { isYouTube: false, videoId: "" };
|
||||
}
|
||||
const host = u.host.toLowerCase();
|
||||
if (host === "youtu.be") {
|
||||
const id = u.pathname.replace(/^\//, "").split("/")[0];
|
||||
return { isYouTube: id !== "", videoId: id };
|
||||
}
|
||||
if (YT_HOSTS.has(host)) {
|
||||
if (u.pathname === "/watch") {
|
||||
const id = u.searchParams.get("v") ?? "";
|
||||
return { isYouTube: id !== "", videoId: id };
|
||||
}
|
||||
if (u.pathname.startsWith("/shorts/")) {
|
||||
const parts = u.pathname.split("/");
|
||||
if (parts.length > 2 && parts[2] !== "") return { isYouTube: true, videoId: parts[2] };
|
||||
}
|
||||
}
|
||||
return { isYouTube: false, videoId: "" };
|
||||
}
|
||||
|
||||
/**
|
||||
* canonicalWatchUrl — oEmbed accepts watch URLs reliably for all shapes.
|
||||
* @param {string} videoId
|
||||
* @returns {string}
|
||||
*/
|
||||
function canonicalWatchUrl(videoId) {
|
||||
return "https://www.youtube.com/watch?v=" + videoId;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchYouTubeMeta fetches title/author via YouTube oEmbed (no API key).
|
||||
* Best-effort: any failure returns empty strings so the route stays `youtube`
|
||||
* and the skill can fall back.
|
||||
* @param {string} watchUrl
|
||||
* @returns {Promise<{title: string, author: string}>}
|
||||
*/
|
||||
async function fetchYouTubeMeta(watchUrl) {
|
||||
const endpoint =
|
||||
"https://www.youtube.com/oembed?url=" + encodeURIComponent(watchUrl) + "&format=json";
|
||||
const body = await fetchTextOk(endpoint, 10_000);
|
||||
if (body === "") return { title: "", author: "" };
|
||||
try {
|
||||
const data = JSON.parse(body);
|
||||
return { title: data.title ?? "", author: data.author_name ?? "" };
|
||||
} catch {
|
||||
return { title: "", author: "" };
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runAddUrl(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter add-url <url>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const target = args[0];
|
||||
|
||||
const cleaned = cleanUrl(target);
|
||||
const { isYouTube, videoId } = detectYouTube(cleaned);
|
||||
|
||||
// For YouTube, dedup/store against the canonical watch URL so youtu.be and
|
||||
// shorts links collapse onto the same identity-param key as watch URLs.
|
||||
const effectiveUrl = isYouTube ? canonicalWatchUrl(videoId) : cleaned;
|
||||
|
||||
// Route order: YouTube → Substack image (by host, not extension, so f_auto /
|
||||
// .avif / .heic / extensionless CDN URLs still route to the image handler) →
|
||||
// file-extension classification.
|
||||
let route;
|
||||
if (isYouTube) route = "youtube";
|
||||
else if (isSubstackImage(cleaned)) route = "image";
|
||||
else route = classifyType(cleaned);
|
||||
|
||||
const httpStatus = await checkAccessibility(cleaned);
|
||||
|
||||
// title/author are omitted entirely when empty, not emitted as "": the
|
||||
// handlers treat a present key as "metadata was resolved".
|
||||
/** @type {Record<string, unknown>} */
|
||||
const out = {
|
||||
original_url: target,
|
||||
clean_url: effectiveUrl,
|
||||
http_status: httpStatus,
|
||||
accessible: httpStatus === "200",
|
||||
duplicate: checkDuplicate(effectiveUrl, contentDir()),
|
||||
route,
|
||||
};
|
||||
if (route === "youtube") {
|
||||
const { title, author } = await fetchYouTubeMeta(effectiveUrl);
|
||||
if (title !== "") out.title = title;
|
||||
if (author !== "") out.author = author;
|
||||
}
|
||||
printJson(out);
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
// Detect whether an image URL is Substack-hosted and extract its S3 image UUID.
|
||||
// Usage: node scripts/newsletter detect-image-source "<image-url>"
|
||||
// Output: JSON { original_url, clean_url, isSubstack, uuid?, innerUrl? }
|
||||
//
|
||||
// Substack images are usually served via a CDN wrapper:
|
||||
//
|
||||
// https://substackcdn.com/image/fetch/$s_!x!,.../https%3A%2F%2Fsubstack-post-media.s3.amazonaws.com%2Fpublic%2Fimages%2F<uuid>_WxH.png
|
||||
//
|
||||
// The publication is NOT encoded in the URL — only the image identity (uuid) is.
|
||||
|
||||
import { printJson } from "./json-out.js";
|
||||
import { cleanUrl, isSubstackImage, substackImageUuid } from "./url-utils.js";
|
||||
|
||||
/**
|
||||
* extractInnerUrl pulls the inner S3 URL out of a substackcdn /image/fetch/
|
||||
* wrapper (if present). A malformed percent sequence falls back to the raw
|
||||
* substring rather than throwing.
|
||||
* @param {string} target
|
||||
* @returns {string}
|
||||
*/
|
||||
export function extractInnerUrl(target) {
|
||||
const marker = target.indexOf("/https%3A%2F%2F");
|
||||
if (marker !== -1) {
|
||||
const raw = target.slice(marker + 1);
|
||||
try {
|
||||
return decodeURIComponent(raw);
|
||||
} catch {
|
||||
return raw;
|
||||
}
|
||||
}
|
||||
// Some forms embed a plain (already-decoded) inner https URL.
|
||||
if (target.length > 8) {
|
||||
const plain = target.slice(8).indexOf("/https://");
|
||||
if (plain !== -1) return target.slice(8 + plain + 1);
|
||||
}
|
||||
return target;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runDetectImageSource(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter detect-image-source <image-url>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const target = args[0];
|
||||
const isSubstack = isSubstackImage(target);
|
||||
|
||||
// Empty optional fields are omitted, not emitted as "": the skills branch on
|
||||
// the key being present.
|
||||
/** @type {Record<string, unknown>} */
|
||||
const out = {
|
||||
original_url: target,
|
||||
clean_url: cleanUrl(target),
|
||||
isSubstack,
|
||||
};
|
||||
if (isSubstack) {
|
||||
const uuid = substackImageUuid(target);
|
||||
const innerUrl = extractInnerUrl(target);
|
||||
if (uuid !== "") out.uuid = uuid;
|
||||
if (innerUrl !== "") out.innerUrl = innerUrl;
|
||||
}
|
||||
printJson(out);
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
// mt-fetch-url fallback fetcher.
|
||||
// Usage: node scripts/newsletter fetch-via-defuddle <target_url>
|
||||
// Exit codes: 0 = content returned, 1 = every tier failed, 2 = bad arguments.
|
||||
//
|
||||
// Two tiers, tried in order:
|
||||
// 1. local defuddle — extracts on this machine, so the chain no longer depends
|
||||
// on a single third-party service being reachable.
|
||||
// 2. the defuddle.md proxy — fetches from a third IP, which is the point when
|
||||
// this machine's IP is the one being blocked.
|
||||
//
|
||||
// Both tiers emit YAML frontmatter followed by the markdown body, so callers
|
||||
// parse one shape regardless of which tier answered.
|
||||
|
||||
const BROWSER_UA =
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36";
|
||||
|
||||
// A bot wall answers 200 with a real body, so a non-empty extraction is not by
|
||||
// itself a success. Recognising the wall is what lets the proxy tier — which
|
||||
// fetches from a different IP — still get its turn.
|
||||
const CHALLENGE_MARKERS = [
|
||||
/just a moment/i,
|
||||
/checking your browser/i,
|
||||
/attention required/i,
|
||||
/cloudflare/i,
|
||||
/enable javascript (and cookies )?to continue/i,
|
||||
/verify (that )?you('re| are) (a )?human/i,
|
||||
/are you a robot/i,
|
||||
/access denied/i,
|
||||
/captcha/i,
|
||||
];
|
||||
|
||||
/**
|
||||
* looksLikeChallenge reports whether an extraction is a bot wall rather than the
|
||||
* page that was asked for.
|
||||
* @param {string} title
|
||||
* @param {string} body
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function looksLikeChallenge(title, body) {
|
||||
// Only the opening of the body: an article may legitimately discuss Cloudflare.
|
||||
const sample = title + "\n" + body.slice(0, 400);
|
||||
return CHALLENGE_MARKERS.some((re) => re.test(sample));
|
||||
}
|
||||
|
||||
/**
|
||||
* yamlFrontmatter renders the metadata block, omitting fields with no value.
|
||||
* @param {Record<string, string>} fields
|
||||
* @returns {string}
|
||||
*/
|
||||
function yamlFrontmatter(fields) {
|
||||
const lines = Object.entries(fields)
|
||||
.filter(([, v]) => typeof v === "string" && v !== "")
|
||||
.map(([k, v]) => `${k}: ${JSON.stringify(v)}`);
|
||||
return lines.length === 0 ? "" : "---\n" + lines.join("\n") + "\n---\n\n";
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchLocally extracts article content with defuddle running in-process, and
|
||||
* renders it in the same frontmatter-plus-body shape the proxy returns.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>} the document, or "" when extraction yields nothing usable
|
||||
*/
|
||||
async function fetchLocally(target) {
|
||||
const { Defuddle } = await import("defuddle/node");
|
||||
const { parseHTML } = await import("linkedom");
|
||||
const res = await fetch(target, {
|
||||
headers: { "user-agent": BROWSER_UA },
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
});
|
||||
if (!res.ok) throw new Error(`upstream returned ${res.status}`);
|
||||
const html = await res.text();
|
||||
const { document } = parseHTML(html);
|
||||
const result = await Defuddle(document, target, { markdown: true });
|
||||
const body = String(result?.content ?? "");
|
||||
if (body.trim() === "") return "";
|
||||
const title = String(result?.title ?? "");
|
||||
if (looksLikeChallenge(title, body)) {
|
||||
throw new Error("extraction looks like a bot challenge, not the page");
|
||||
}
|
||||
const frontmatter = yamlFrontmatter({
|
||||
title,
|
||||
author: String(result?.author ?? ""),
|
||||
description: String(result?.description ?? ""),
|
||||
site: String(result?.site ?? ""),
|
||||
published: String(result?.published ?? ""),
|
||||
source: target,
|
||||
});
|
||||
return frontmatter + body;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchViaProxy fetches through defuddle.md, which resolves the page from its
|
||||
* own IP.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>} the response body
|
||||
*/
|
||||
async function fetchViaProxy(target) {
|
||||
const res = await fetch("https://defuddle.md/" + target, {
|
||||
headers: { "user-agent": "mt-fetch-url/1.0" },
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
});
|
||||
const body = await res.text();
|
||||
if (!res.ok || body.trim() === "") {
|
||||
throw new Error(`defuddle returned ${res.status} / empty body`);
|
||||
}
|
||||
return body;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFetchViaDefuddle(args) {
|
||||
if (args.length < 1 || args[0] === "") {
|
||||
process.stderr.write("Usage: node scripts/newsletter fetch-via-defuddle <target_url>\n");
|
||||
process.exit(2);
|
||||
}
|
||||
const target = args[0];
|
||||
|
||||
try {
|
||||
const doc = await fetchLocally(target);
|
||||
if (doc !== "") {
|
||||
process.stdout.write(doc.endsWith("\n") ? doc : doc + "\n");
|
||||
return;
|
||||
}
|
||||
process.stderr.write(`mt-fetch-url: local defuddle extracted nothing for ${target}\n`);
|
||||
} catch (err) {
|
||||
process.stderr.write(
|
||||
`mt-fetch-url: local defuddle failed for ${target}: ${String(err?.message ?? err)}\n`,
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
process.stdout.write(await fetchViaProxy(target));
|
||||
} catch (err) {
|
||||
process.stderr.write(
|
||||
`mt-fetch-url: defuddle.md failed for ${target}: ${String(err?.message ?? err)}\n`,
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
// Find the most recent newsletter number and return the next one.
|
||||
// Usage: node scripts/newsletter find-newsletter-number
|
||||
// Outputs: the next newsletter number.
|
||||
|
||||
import { readdirSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { contentDir } from "./url-utils.js";
|
||||
|
||||
/** Matches the "Newsletter #N" heading a post is numbered by. @type {RegExp} */
|
||||
export const NEWSLETTER_NUM_RE = /Newsletter\s*#(\d+)/;
|
||||
const YEAR_DIR_RE = /^\d{4}$/;
|
||||
const TWO_DIGIT_DIR_RE = /^\d{2}$/;
|
||||
|
||||
/**
|
||||
* listDirsDesc returns dir's subdirectory names matching re, sorted descending.
|
||||
* @param {string} dir
|
||||
* @param {RegExp} re
|
||||
* @returns {string[]}
|
||||
*/
|
||||
function listDirsDesc(dir, re) {
|
||||
let entries;
|
||||
try {
|
||||
entries = readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
return entries
|
||||
.filter((e) => re.test(e.name))
|
||||
.map((e) => e.name)
|
||||
.sort((a, b) => (a < b ? 1 : a > b ? -1 : 0));
|
||||
}
|
||||
|
||||
/**
|
||||
* extractNewsletterNumber reads the first "Newsletter #N" in a post.
|
||||
* @param {string} path
|
||||
* @returns {number}
|
||||
*/
|
||||
function extractNewsletterNumber(path) {
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
const m = NEWSLETTER_NUM_RE.exec(content);
|
||||
if (m === null) return 0;
|
||||
const n = Number.parseInt(m[1], 10);
|
||||
return Number.isNaN(n) ? 0 : n;
|
||||
}
|
||||
|
||||
/**
|
||||
* findMostRecentNewsletter scans year/month/day directories newest-first for
|
||||
* the highest newsletter number.
|
||||
* @returns {number}
|
||||
*/
|
||||
export function findMostRecentNewsletter() {
|
||||
let maxNumber = 0;
|
||||
for (const year of listDirsDesc(contentDir(), YEAR_DIR_RE)) {
|
||||
const yearDir = join(contentDir(), year);
|
||||
for (const month of listDirsDesc(yearDir, TWO_DIGIT_DIR_RE)) {
|
||||
const monthDir = join(yearDir, month);
|
||||
for (const day of listDirsDesc(monthDir, TWO_DIGIT_DIR_RE)) {
|
||||
const n = extractNewsletterNumber(join(monthDir, day, "index.md"));
|
||||
if (n > maxNumber) maxNumber = n;
|
||||
}
|
||||
}
|
||||
// Early exit: a newsletter found in this year — no need to go further back.
|
||||
if (maxNumber > 0) break;
|
||||
}
|
||||
return maxNumber;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} _args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFindNewsletterNumber(_args) {
|
||||
process.stdout.write(String(findMostRecentNewsletter() + 1) + "\n");
|
||||
}
|
||||
@@ -0,0 +1,231 @@
|
||||
// Find which Substack post embeds a given image UUID, and extract a label.
|
||||
// Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]
|
||||
// Output on hit: JSON { found:true, source, publication, postTitle, postUrl, caption, candidates }
|
||||
// Output on miss: JSON { found:false } (RSS) or { found:false, source:"sitemap", scanned, budget, cutoff } (--deep)
|
||||
//
|
||||
// Strategy: RSS feed first (fast, ~recent weeks). With --deep, fall back to a
|
||||
// heavier sitemap crawl up to ~3 months back — opt-in because it fetches many
|
||||
// posts. A Substack CDN URL does not encode its publication, so we search each
|
||||
// publication listed in config/substack-publications.json, read at runtime so
|
||||
// editing the JSON takes effect immediately.
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { parseArgs } from "node:util";
|
||||
import { printJson } from "./json-out.js";
|
||||
import { fetchTextOk } from "./url-utils.js";
|
||||
import {
|
||||
captionForUuid,
|
||||
extractCandidates,
|
||||
itemLink,
|
||||
itemTitle,
|
||||
loadXml,
|
||||
postTitleFromHtml,
|
||||
} from "./html-text.js";
|
||||
|
||||
/** Total post fetches allowed across ALL publications during a --deep crawl. */
|
||||
const DEEP_FETCH_BUDGET = 40;
|
||||
|
||||
/**
|
||||
* loadPublications reads the publication list, falling back to the default on
|
||||
* any read or parse failure.
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function loadPublications() {
|
||||
try {
|
||||
const raw = readFileSync(new URL("./config/substack-publications.json", import.meta.url), "utf8");
|
||||
const pubs = JSON.parse(raw);
|
||||
if (Array.isArray(pubs) && pubs.length > 0) return pubs;
|
||||
} catch {
|
||||
/* fall through */
|
||||
}
|
||||
return ["blog.bytebytego.com"];
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchPage: body text with a browser-ish UA, or "" on any error.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
function fetchPage(target) {
|
||||
return fetchTextOk(target, 10_000, "Mozilla/5.0");
|
||||
}
|
||||
|
||||
const RFC3339_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?(Z|[+-]\d{2}:\d{2})$/i;
|
||||
const DATE_ONLY_RE = /^\d{4}-\d{2}-\d{2}$/;
|
||||
|
||||
/**
|
||||
* parseLastmod accepts only the two layouts the Go engine accepted — RFC3339
|
||||
* and a bare date — so a loosely formatted stamp is skipped rather than
|
||||
* silently reinterpreted in local time.
|
||||
* @param {string} s
|
||||
* @returns {Date|null}
|
||||
*/
|
||||
export function parseLastmod(s) {
|
||||
if (!RFC3339_RE.test(s) && !DATE_ONLY_RE.test(s)) return null;
|
||||
const t = new Date(s);
|
||||
return Number.isNaN(t.getTime()) ? null : t;
|
||||
}
|
||||
|
||||
/**
|
||||
* cutoffDate returns the ~3-months-back boundary for the deep crawl.
|
||||
* @returns {Date}
|
||||
*/
|
||||
function cutoffDate() {
|
||||
const d = new Date();
|
||||
d.setUTCMonth(d.getUTCMonth() - 3);
|
||||
return d;
|
||||
}
|
||||
|
||||
/**
|
||||
* @typedef {{found: true, source: string, publication: string, postTitle: string,
|
||||
* postUrl: string, caption: string, candidates: string[]}} PostHit
|
||||
*/
|
||||
|
||||
/**
|
||||
* searchSitemap is the deep fallback: crawl the sitemap back ~3 months, fetch
|
||||
* posts most-recent-first (up to maxFetch from the shared budget), and look for
|
||||
* the UUID. Heavier than RSS — only used on RSS miss.
|
||||
* @param {string} publication
|
||||
* @param {string} id
|
||||
* @param {number} maxFetch
|
||||
* @returns {Promise<{hit: PostHit|null, scanned: number, cutoff: string}>} cutoff is "" when the sitemap itself could not be fetched
|
||||
*/
|
||||
async function searchSitemap(publication, id, maxFetch) {
|
||||
const xml = await fetchPage("https://" + publication + "/sitemap.xml");
|
||||
if (xml === "") return { hit: null, scanned: 0, cutoff: "" };
|
||||
|
||||
const cutoffTime = cutoffDate();
|
||||
const cutoff = cutoffTime.toISOString().slice(0, 10);
|
||||
|
||||
const $ = loadXml(xml);
|
||||
/** @type {{url: string, when: Date}[]} */
|
||||
const candidates = [];
|
||||
$("url").each((_i, el) => {
|
||||
const loc = $(el).children("loc").first().text();
|
||||
const lastmod = $(el).children("lastmod").first().text();
|
||||
if (loc === "" || lastmod === "") return;
|
||||
if (!loc.includes("/p/")) return; // posts only
|
||||
const when = parseLastmod(lastmod);
|
||||
if (when === null || when < cutoffTime) return;
|
||||
candidates.push({ url: loc, when });
|
||||
});
|
||||
candidates.sort((a, b) => b.when.getTime() - a.when.getTime());
|
||||
|
||||
let scanned = 0;
|
||||
for (const c of candidates.slice(0, maxFetch)) {
|
||||
scanned++;
|
||||
const html = await fetchPage(c.url);
|
||||
if (html === "" || !html.includes(id)) continue;
|
||||
return {
|
||||
hit: {
|
||||
found: true,
|
||||
source: "sitemap",
|
||||
publication,
|
||||
postTitle: postTitleFromHtml(html),
|
||||
postUrl: c.url,
|
||||
caption: captionForUuid(html, id),
|
||||
candidates: extractCandidates(html),
|
||||
},
|
||||
scanned,
|
||||
cutoff,
|
||||
};
|
||||
}
|
||||
return { hit: null, scanned, cutoff };
|
||||
}
|
||||
|
||||
/**
|
||||
* searchRss looks for the uuid in a publication's recent feed items.
|
||||
* @param {string} publication
|
||||
* @param {string} id
|
||||
* @returns {Promise<PostHit|null>}
|
||||
*/
|
||||
async function searchRss(publication, id) {
|
||||
const xml = await fetchPage("https://" + publication + "/feed");
|
||||
if (xml === "") return null;
|
||||
const $ = loadXml(xml);
|
||||
const items = $("item").toArray();
|
||||
for (const el of items) {
|
||||
const item = $.html(el);
|
||||
if (!item.includes(id)) continue;
|
||||
return {
|
||||
found: true,
|
||||
source: "rss",
|
||||
publication,
|
||||
postTitle: itemTitle(item),
|
||||
postUrl: itemLink(item),
|
||||
caption: captionForUuid(item, id),
|
||||
candidates: extractCandidates(item),
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runFindSubstackPost(args) {
|
||||
let values;
|
||||
try {
|
||||
({ values } = parseArgs({
|
||||
args,
|
||||
options: { uuid: { type: "string" }, deep: { type: "boolean" } },
|
||||
allowPositionals: false,
|
||||
}));
|
||||
} catch (err) {
|
||||
process.stderr.write(String(err?.message ?? err) + "\n");
|
||||
process.stderr.write(
|
||||
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
|
||||
);
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
const uuid = values.uuid ?? "";
|
||||
if (uuid === "") {
|
||||
process.stderr.write(
|
||||
"Usage: node scripts/newsletter find-substack-post --uuid <uuid> [--deep]\n",
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const publications = loadPublications();
|
||||
for (const pub of publications) {
|
||||
const hit = await searchRss(pub, uuid);
|
||||
if (hit !== null) {
|
||||
printJson(hit);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// Deep fallback: sitemap crawl up to ~3 months back, sharing one global fetch
|
||||
// budget across all publications so coverage can't blow up as the
|
||||
// publications list grows.
|
||||
if (values.deep === true) {
|
||||
let totalScanned = 0;
|
||||
/** @type {string|null} */
|
||||
let lastCutoff = null;
|
||||
for (const pub of publications) {
|
||||
const remaining = DEEP_FETCH_BUDGET - totalScanned;
|
||||
if (remaining <= 0) break;
|
||||
const { hit, scanned, cutoff } = await searchSitemap(pub, uuid, remaining);
|
||||
totalScanned += scanned;
|
||||
if (cutoff !== "") lastCutoff = cutoff;
|
||||
if (hit !== null) {
|
||||
printJson(hit);
|
||||
return;
|
||||
}
|
||||
}
|
||||
// cutoff is null (not omitted) when no sitemap could be fetched — the skill
|
||||
// distinguishes "crawled and missed" from "could not crawl".
|
||||
printJson({
|
||||
found: false,
|
||||
source: "sitemap",
|
||||
scanned: totalScanned,
|
||||
budget: DEEP_FETCH_BUDGET,
|
||||
cutoff: lastCutoff,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
printJson({ found: false });
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
// HTML / RSS text-extraction helpers for find-substack-post, ported from
|
||||
// html_text.go. Pure string functions — no network, no fs. A real parser
|
||||
// (cheerio) replaces the hand-rolled regexes the Go version used.
|
||||
|
||||
import * as cheerio from "cheerio";
|
||||
|
||||
/**
|
||||
* loadXml parses an XML fragment (RSS item, sitemap) with CDATA recognition.
|
||||
* @param {string} src
|
||||
* @returns {cheerio.CheerioAPI}
|
||||
*/
|
||||
export function loadXml(src) {
|
||||
return cheerio.load(src, { xmlMode: true }, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* loadHtml parses an HTML blob. CDATA markers are stripped first: RSS carries
|
||||
* post HTML inside CDATA, and the HTML parser would otherwise swallow it as a
|
||||
* bogus comment.
|
||||
* @param {string} src
|
||||
* @returns {cheerio.CheerioAPI}
|
||||
*/
|
||||
export function loadHtml(src) {
|
||||
return cheerio.load(src.split("<![CDATA[").join("").split("]]>").join(""));
|
||||
}
|
||||
|
||||
/**
|
||||
* rawInner returns an element's serialized inner content with any CDATA wrapper
|
||||
* removed — the same substring the Go regexes captured, so entity handling can
|
||||
* stay identical instead of being applied twice.
|
||||
* @param {cheerio.CheerioAPI} $
|
||||
* @param {cheerio.Cheerio<any>} el
|
||||
* @returns {string}
|
||||
*/
|
||||
function rawInner($, el) {
|
||||
const html = $.html(el);
|
||||
const start = html.indexOf(">") + 1;
|
||||
const end = html.lastIndexOf("</");
|
||||
if (start <= 0 || end < start) return "";
|
||||
let inner = html.slice(start, end);
|
||||
if (inner.startsWith("<![CDATA[")) inner = inner.slice(9);
|
||||
if (inner.endsWith("]]>")) inner = inner.slice(0, -3);
|
||||
return inner;
|
||||
}
|
||||
|
||||
/**
|
||||
* decodeEntities decodes HTML entities (named, numeric, hex) and trims.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
export function decodeEntities(s) {
|
||||
if (s === "") return "";
|
||||
return cheerio.load(s, null, false).text().trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* collapseWhitespace squeezes runs of whitespace to one space and trims.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
function collapseWhitespace(s) {
|
||||
return s.replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* stripTags removes markup, decodes entities, and collapses whitespace.
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
export function stripTags(s) {
|
||||
if (s === "") return "";
|
||||
return collapseWhitespace(cheerio.load(s, null, false).text());
|
||||
}
|
||||
|
||||
/**
|
||||
* itemTitle pulls the first <title> (CDATA or plain) from an RSS <item> chunk.
|
||||
* @param {string} item
|
||||
* @returns {string}
|
||||
*/
|
||||
export function itemTitle(item) {
|
||||
const $ = loadXml(item);
|
||||
const el = $("title").first();
|
||||
if (el.length === 0) return "";
|
||||
return decodeEntities(rawInner($, el));
|
||||
}
|
||||
|
||||
/**
|
||||
* itemLink pulls the first <link> (CDATA or plain) from an RSS <item> chunk.
|
||||
* Entities are left as stored — the link goes straight into a post.
|
||||
* @param {string} item
|
||||
* @returns {string}
|
||||
*/
|
||||
export function itemLink(item) {
|
||||
const $ = loadXml(item);
|
||||
const el = $("link").first();
|
||||
if (el.length === 0) return "";
|
||||
return rawInner($, el).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* extractCandidates pulls candidate topic titles from a post's TOC bullet list.
|
||||
* ByteByteGo does not attach captions to images — the topic titles live only in
|
||||
* the "in this issue" bullets, and image→title cannot be mapped automatically
|
||||
* (sponsor/video items interleave), so these are surfaced for the user to pick
|
||||
* from. Light filtering keeps the list short: dedupe, drop sub-point
|
||||
* explanations and over-long lines.
|
||||
* @param {string} htmlSrc
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function extractCandidates(htmlSrc) {
|
||||
const $ = loadHtml(htmlSrc);
|
||||
/** @type {Set<string>} */
|
||||
const seen = new Set();
|
||||
/** @type {string[]} */
|
||||
const out = [];
|
||||
$("li").each((_i, el) => {
|
||||
const text = collapseWhitespace($(el).text());
|
||||
if (text === "") return;
|
||||
// Titles are short; long lines are sub-point explanations. Count code
|
||||
// points, not UTF-16 units, so an emoji does not count double.
|
||||
const n = [...text].length;
|
||||
if (n < 6 || n > 70) return;
|
||||
const key = text.toLowerCase();
|
||||
if (seen.has(key)) return; // content is duplicated in the page
|
||||
seen.add(key);
|
||||
out.push(text);
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* captionForUuid returns the <figcaption> text of the <figure> containing the
|
||||
* UUID. Cover images live in <enclosure> (no figure) → "".
|
||||
* @param {string} src
|
||||
* @param {string} id
|
||||
* @returns {string}
|
||||
*/
|
||||
export function captionForUuid(src, id) {
|
||||
if (id === "") return "";
|
||||
const $ = loadHtml(src);
|
||||
let target = null;
|
||||
$("*").each((_i, el) => {
|
||||
if (target !== null) return false;
|
||||
for (const v of Object.values(el.attribs ?? {})) {
|
||||
if (typeof v === "string" && v.includes(id)) {
|
||||
target = el;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
for (const child of el.children ?? []) {
|
||||
if (child.type === "text" && String(child.data).includes(id)) {
|
||||
target = el;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
});
|
||||
if (target === null) return "";
|
||||
const figure = $(target).closest("figure");
|
||||
if (figure.length === 0) return "";
|
||||
const caption = figure.find("figcaption").first();
|
||||
if (caption.length === 0) return "";
|
||||
return collapseWhitespace(caption.text());
|
||||
}
|
||||
|
||||
/**
|
||||
* postTitleFromHtml extracts a post title from server-rendered post HTML
|
||||
* (og:title preferred, then <h1>, then <title>).
|
||||
* @param {string} htmlSrc
|
||||
* @returns {string}
|
||||
*/
|
||||
export function postTitleFromHtml(htmlSrc) {
|
||||
const $ = loadHtml(htmlSrc);
|
||||
const og = $('meta[property="og:title"]').first().attr("content");
|
||||
if (og !== undefined && og !== "") return og.trim();
|
||||
const h1 = $("h1").first();
|
||||
if (h1.length > 0) return collapseWhitespace(h1.text());
|
||||
const title = $("title").first();
|
||||
if (title.length > 0) return title.text().trim();
|
||||
return "";
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
// Newsletter engine for the mt-* skills — one entry point, one subcommand per
|
||||
// task. Invoked from the repo root:
|
||||
//
|
||||
// node scripts/newsletter <command> [args]
|
||||
//
|
||||
// Repo-relative paths (content/post) resolve from the working directory, so the
|
||||
// repo-root invocation contract from AGENTS.md still applies.
|
||||
|
||||
const USAGE = `Usage: node scripts/newsletter <command> [args]
|
||||
|
||||
Commands:
|
||||
add-url <url> classify + dedup a URL, emit JSON route
|
||||
find-newsletter-number print the next newsletter number
|
||||
list-existing-tags tag frequencies, most-used first (top 40)
|
||||
detect-image-source <url> detect Substack image + uuid
|
||||
find-substack-post --uuid <uuid> [--deep] find the post embedding an image uuid
|
||||
fetch-via-defuddle <url> fallback fetch (local defuddle, then defuddle.md)
|
||||
post-stats <path/to/index.md> count the post's articles/images/videos/documents
|
||||
`;
|
||||
|
||||
// A reader that closes early (`… | head -3`) makes the next write fail with
|
||||
// EPIPE, which Node surfaces as an unhandled error event. Stop quietly instead
|
||||
// of printing a stack trace over the user's terminal.
|
||||
process.stdout.on("error", (err) => {
|
||||
if (err.code === "EPIPE") process.exit(0);
|
||||
throw err;
|
||||
});
|
||||
|
||||
/** @returns {void} */
|
||||
function usage() {
|
||||
process.stderr.write(USAGE);
|
||||
}
|
||||
|
||||
/**
|
||||
* Command modules are imported lazily so a missing node_modules reports the one
|
||||
* actionable fix instead of a module-resolution stack trace.
|
||||
* @param {string} spec
|
||||
* @returns {Promise<Record<string, any>>}
|
||||
*/
|
||||
async function loadCommand(spec) {
|
||||
try {
|
||||
return await import(spec);
|
||||
} catch (err) {
|
||||
if (err !== null && typeof err === "object" && err.code === "ERR_MODULE_NOT_FOUND") {
|
||||
process.stderr.write(
|
||||
"newsletter engine: dependencies missing — run 'npm ci' from the repo root\n",
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
/** @type {Record<string, {module: string, fn: string}>} */
|
||||
const COMMANDS = {
|
||||
"add-url": { module: "./add-url.js", fn: "runAddUrl" },
|
||||
"find-newsletter-number": { module: "./find-newsletter-number.js", fn: "runFindNewsletterNumber" },
|
||||
"list-existing-tags": { module: "./list-existing-tags.js", fn: "runListExistingTags" },
|
||||
"detect-image-source": { module: "./detect-image-source.js", fn: "runDetectImageSource" },
|
||||
"find-substack-post": { module: "./find-substack-post.js", fn: "runFindSubstackPost" },
|
||||
"fetch-via-defuddle": { module: "./fetch-via-defuddle.js", fn: "runFetchViaDefuddle" },
|
||||
"post-stats": { module: "./post-stats.js", fn: "runPostStats" },
|
||||
};
|
||||
|
||||
/** @returns {Promise<void>} */
|
||||
async function main() {
|
||||
const argv = process.argv.slice(2);
|
||||
if (argv.length < 1) {
|
||||
usage();
|
||||
process.exit(1);
|
||||
}
|
||||
const [name, ...args] = argv;
|
||||
const entry = COMMANDS[name];
|
||||
if (entry === undefined) {
|
||||
process.stderr.write(`unknown command: ${name}\n`);
|
||||
usage();
|
||||
process.exit(1);
|
||||
}
|
||||
const mod = await loadCommand(entry.module);
|
||||
await mod[entry.fn](args);
|
||||
}
|
||||
|
||||
try {
|
||||
await main();
|
||||
} catch (err) {
|
||||
process.stderr.write("newsletter engine: " + String(err?.stack ?? err) + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
// The shared JSON printer. It lives in its own module rather than in index.js so
|
||||
// that importing a command module never pulls in the dispatcher — an import
|
||||
// cycle there would run the CLI as a side effect of any import.
|
||||
|
||||
/**
|
||||
* printJson mirrors console.log(JSON.stringify(v, null, 2)): 2-space indent, no
|
||||
* HTML escaping (URLs with & must stay readable), trailing newline.
|
||||
* @param {unknown} v
|
||||
* @returns {void}
|
||||
*/
|
||||
export function printJson(v) {
|
||||
process.stdout.write(JSON.stringify(v, null, 2) + "\n");
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
// List existing tags in the repo ranked by frequency.
|
||||
// Usage: node scripts/newsletter list-existing-tags
|
||||
// Outputs: tag count and name, sorted most-used first (top 40).
|
||||
//
|
||||
// NOTE: not currently used by the mt-add-tags skill. Tag normalization is
|
||||
// disabled until existing posts have standardized tags; to enable, uncomment
|
||||
// step 4a in that skill's SKILL.md.
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { basename } from "node:path";
|
||||
import { collectMarkdown, contentDir } from "./url-utils.js";
|
||||
|
||||
const TAGS_LINE_RE = /^tags:\s*\[([^\]]*)\]/m;
|
||||
const QUOTED_TAG_RE = /"([^"]+)"/g;
|
||||
|
||||
/**
|
||||
* extractTags pulls quoted tag strings from an index.md frontmatter tags array.
|
||||
* @param {string} path
|
||||
* @returns {string[]}
|
||||
*/
|
||||
function extractTags(path) {
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
const m = TAGS_LINE_RE.exec(content);
|
||||
if (m === null) return [];
|
||||
return [...m[1].matchAll(QUOTED_TAG_RE)].map((q) => q[1]);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} _args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runListExistingTags(_args) {
|
||||
// Array + index map keeps first-seen order for equal counts, so the stable
|
||||
// sort below ranks ties deterministically (walk order is lexical).
|
||||
/** @type {{tag: string, count: number}[]} */
|
||||
const counts = [];
|
||||
/** @type {Map<string, number>} */
|
||||
const index = new Map();
|
||||
|
||||
for (const path of collectMarkdown(contentDir())) {
|
||||
if (basename(path) !== "index.md") continue;
|
||||
for (const tag of extractTags(path)) {
|
||||
const at = index.get(tag);
|
||||
if (at !== undefined) counts[at].count++;
|
||||
else {
|
||||
index.set(tag, counts.length);
|
||||
counts.push({ tag, count: 1 });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
counts.sort((a, b) => b.count - a.count);
|
||||
// Plain text, not JSON: the count is right-aligned in a 6-character field and
|
||||
// the skill reads that layout.
|
||||
for (const { tag, count } of counts.slice(0, 40)) {
|
||||
process.stdout.write(String(count).padStart(6) + " " + tag + "\n");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
// Count the entries already present in a newsletter post, so a handler can
|
||||
// report a running tally after each insertion.
|
||||
// Usage: node scripts/newsletter post-stats <path/to/index.md>
|
||||
// Outputs: JSON { post, newsletter, articles, images, videos, documents, total }
|
||||
|
||||
import { readFileSync } from "node:fs";
|
||||
import { printJson } from "./json-out.js";
|
||||
import { NEWSLETTER_NUM_RE } from "./find-newsletter-number.js";
|
||||
|
||||
// Entry shapes, per the Bonus format in the shared post mechanics:
|
||||
//
|
||||
// articles "## [Title](url)" (level-2 heading, main content)
|
||||
// images "" (under **Images:**)
|
||||
// videos "[Title](url)" (under **Videos:**)
|
||||
// documents "[PDF: title](url)" (under **Documents:**)
|
||||
const ARTICLE_HEADING_RE = /^##\s+\[/;
|
||||
const BONUS_HEADING_RE = /^###\s+Bonus\b/;
|
||||
const SUBSECTION_RE = /^\*\*(Images|Videos|Documents):\*\*/;
|
||||
const IMAGE_ENTRY_RE = /^!\[/;
|
||||
const LINK_ENTRY_RE = /^\[/;
|
||||
|
||||
/**
|
||||
* countPostEntries walks the post once. Article headings are counted anywhere
|
||||
* outside Bonus; asset entries are attributed to whichever subsection is open.
|
||||
* @param {string} content
|
||||
* @returns {{articles: number, images: number, videos: number, documents: number, total: number}}
|
||||
*/
|
||||
export function countPostEntries(content) {
|
||||
let articles = 0;
|
||||
let images = 0;
|
||||
let videos = 0;
|
||||
let documents = 0;
|
||||
let inBonus = false;
|
||||
let subsection = "";
|
||||
|
||||
for (const raw of content.split("\n")) {
|
||||
const line = raw.trim();
|
||||
if (BONUS_HEADING_RE.test(line)) {
|
||||
inBonus = true;
|
||||
subsection = "";
|
||||
continue;
|
||||
}
|
||||
const sub = SUBSECTION_RE.exec(line);
|
||||
if (sub !== null) {
|
||||
subsection = sub[1];
|
||||
continue;
|
||||
}
|
||||
if (ARTICLE_HEADING_RE.test(line)) {
|
||||
articles++;
|
||||
continue;
|
||||
}
|
||||
if (!inBonus) continue;
|
||||
if (subsection === "Images") {
|
||||
if (IMAGE_ENTRY_RE.test(line)) images++;
|
||||
} else if (subsection === "Videos") {
|
||||
// A direct video file entry looks the same as a YouTube entry;
|
||||
// both belong to the Videos tally.
|
||||
if (LINK_ENTRY_RE.test(line)) videos++;
|
||||
} else if (subsection === "Documents") {
|
||||
if (LINK_ENTRY_RE.test(line)) documents++;
|
||||
}
|
||||
}
|
||||
return { articles, images, videos, documents, total: articles + images + videos + documents };
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} args
|
||||
* @returns {Promise<void>}
|
||||
*/
|
||||
export async function runPostStats(args) {
|
||||
if (args.length < 1) {
|
||||
process.stderr.write("usage: post-stats <path/to/index.md>\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const path = args[0];
|
||||
let content;
|
||||
try {
|
||||
content = readFileSync(path, "utf8");
|
||||
} catch (err) {
|
||||
process.stderr.write("read post: " + String(err?.message ?? err) + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
const counted = countPostEntries(content);
|
||||
const m = NEWSLETTER_NUM_RE.exec(content);
|
||||
printJson({
|
||||
post: path,
|
||||
newsletter: m === null ? 0 : Number.parseInt(m[1], 10),
|
||||
articles: counted.articles,
|
||||
images: counted.images,
|
||||
videos: counted.videos,
|
||||
documents: counted.documents,
|
||||
total: counted.total,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,339 @@
|
||||
// Shared URL helpers, ported from url_utils.go. Owned by the add-url router;
|
||||
// reused by the other subcommands.
|
||||
|
||||
import { readdirSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
|
||||
/** Exact-match tracking params; any key starting with utm_ is also dropped. */
|
||||
const EXACT_TRACKING = new Set([
|
||||
"fbclid", "gclid", "msclkid", "mc_eid",
|
||||
"aid", "ref", "ref_src", "ref_url", "source", "s",
|
||||
"ck_subscriber_id", "igshid", "yclid", "vero_id",
|
||||
]);
|
||||
|
||||
/**
|
||||
* contentDir is repo-root relative (invocation contract: run from repo root).
|
||||
* @returns {string}
|
||||
*/
|
||||
export function contentDir() {
|
||||
return join("content", "post");
|
||||
}
|
||||
|
||||
/**
|
||||
* cleanUrl removes common tracking parameters. Surviving query pairs are kept
|
||||
* verbatim (no re-encoding) and in their original order — URLSearchParams would
|
||||
* normalize percent-encoding (%7E → ~, + → %20), which must not leak into
|
||||
* stored clean_url values. Unparseable / non-absolute input is returned
|
||||
* untouched rather than corrupted.
|
||||
* @param {string} raw
|
||||
* @returns {string}
|
||||
*/
|
||||
export function cleanUrl(raw) {
|
||||
let u;
|
||||
try {
|
||||
u = new URL(raw);
|
||||
} catch {
|
||||
return raw;
|
||||
}
|
||||
if (!u.protocol || !u.host) return raw;
|
||||
// WHATWG URL serializes an empty path as "/" for special schemes; force it
|
||||
// for the rest so cleaned URLs keep one shape.
|
||||
if (u.pathname === "") {
|
||||
try {
|
||||
u.pathname = "/";
|
||||
} catch {
|
||||
/* opaque path — leave as parsed */
|
||||
}
|
||||
}
|
||||
|
||||
let kept = "";
|
||||
const query = u.search.slice(1);
|
||||
if (query !== "") {
|
||||
const parts = [];
|
||||
for (const pair of query.split("&")) {
|
||||
if (pair === "") continue;
|
||||
const eq = pair.indexOf("=");
|
||||
const key = (eq === -1 ? pair : pair.slice(0, eq)).toLowerCase();
|
||||
if (key.startsWith("utm_") || EXACT_TRACKING.has(key)) continue;
|
||||
parts.push(pair);
|
||||
}
|
||||
kept = parts.join("&");
|
||||
}
|
||||
|
||||
const hash = u.hash;
|
||||
u.search = "";
|
||||
u.hash = "";
|
||||
return u.toString() + (kept === "" ? "" : "?" + kept) + hash;
|
||||
}
|
||||
|
||||
// --- Substack image helpers (shared by add-url routing and detect-image-source) ---
|
||||
|
||||
const SUBSTACK_IMAGE_HOSTS = new Set([
|
||||
"substackcdn.com",
|
||||
"substack-post-media.s3.amazonaws.com",
|
||||
]);
|
||||
|
||||
const UUID_PATTERN = "[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}";
|
||||
const IMAGE_UUID_RE = new RegExp(`images(?:%2F|/)(${UUID_PATTERN})`, "i");
|
||||
const UUID_EXACT_RE = new RegExp(`^${UUID_PATTERN}$`, "i");
|
||||
|
||||
/**
|
||||
* isSubstackImage reports a Substack-hosted image (CDN wrapper or raw S3),
|
||||
* regardless of file extension.
|
||||
* @param {string} target
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function isSubstackImage(target) {
|
||||
let host = "";
|
||||
try {
|
||||
host = new URL(target).host.toLowerCase();
|
||||
} catch {
|
||||
/* not an absolute URL — fall through to the substring check */
|
||||
}
|
||||
return SUBSTACK_IMAGE_HOSTS.has(host) || target.toLowerCase().includes("substack-post-media");
|
||||
}
|
||||
|
||||
/**
|
||||
* substackImageUuid extracts the stable image identity: the S3 image UUID under
|
||||
* public/images/<uuid>, with raw (/) or percent-encoded (%2F) separators.
|
||||
* @param {string} target
|
||||
* @returns {string} the lowercased uuid, or "" when absent
|
||||
*/
|
||||
export function substackImageUuid(target) {
|
||||
const m = IMAGE_UUID_RE.exec(target);
|
||||
return m === null ? "" : m[1].toLowerCase();
|
||||
}
|
||||
|
||||
/**
|
||||
* isUuid reports whether a bare identity is a Substack image uuid.
|
||||
* @param {string} s
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function isUuid(s) {
|
||||
return UUID_EXACT_RE.test(s);
|
||||
}
|
||||
|
||||
/**
|
||||
* Some sites carry the resource identity in a query param, not the path
|
||||
* (e.g. YouTube /watch?v=ID). Preserve the identity param for those hosts so
|
||||
* dedup does not collapse every video onto the same bare URL.
|
||||
*/
|
||||
const IDENTITY_PARAMS = new Map([
|
||||
["youtube.com", "v"],
|
||||
["www.youtube.com", "v"],
|
||||
["m.youtube.com", "v"],
|
||||
]);
|
||||
|
||||
/**
|
||||
* trimTrailingSlash removes at most one trailing "/".
|
||||
* @param {string} s
|
||||
* @returns {string}
|
||||
*/
|
||||
function trimTrailingSlash(s) {
|
||||
return s.endsWith("/") ? s.slice(0, -1) : s;
|
||||
}
|
||||
|
||||
/**
|
||||
* bareUrl reduces a URL to a stable identity for duplicate detection:
|
||||
* - Substack image → its S3 UUID (transform/size variants share one identity)
|
||||
* - YouTube → scheme+host+path + the v= video id
|
||||
* - everything else → scheme + host + path
|
||||
* @param {string} target
|
||||
* @returns {string}
|
||||
*/
|
||||
export function bareUrl(target) {
|
||||
if (isSubstackImage(target)) {
|
||||
const uuid = substackImageUuid(target);
|
||||
if (uuid !== "") return uuid;
|
||||
}
|
||||
let u = null;
|
||||
try {
|
||||
u = new URL(target);
|
||||
} catch {
|
||||
/* unparseable — fall back to crude string surgery below */
|
||||
}
|
||||
if (u === null || !u.protocol || !u.host) {
|
||||
return trimTrailingSlash(target.split("?")[0]);
|
||||
}
|
||||
const host = u.host.toLowerCase();
|
||||
let bare = trimTrailingSlash(u.protocol + "//" + host + u.pathname);
|
||||
const idParam = IDENTITY_PARAMS.get(host);
|
||||
if (idParam !== undefined) {
|
||||
const v = u.searchParams.get(idParam);
|
||||
if (v !== null && v !== "") bare += "?" + idParam + "=" + v;
|
||||
}
|
||||
return bare;
|
||||
}
|
||||
|
||||
/**
|
||||
* fetchTextOk GETs target and returns the body on a 2xx response, "" on any
|
||||
* error, non-2xx status, or timeout. Redirects are followed.
|
||||
* @param {string} target
|
||||
* @param {number} timeoutMs
|
||||
* @param {string} [userAgent]
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
export async function fetchTextOk(target, timeoutMs, userAgent = "") {
|
||||
try {
|
||||
const headers = {};
|
||||
if (userAgent !== "") headers["user-agent"] = userAgent;
|
||||
const res = await fetch(target, {
|
||||
headers,
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(timeoutMs),
|
||||
});
|
||||
if (res.status < 200 || res.status >= 300) return "";
|
||||
return await res.text();
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* checkAccessibility HEADs the URL and returns the final HTTP status code as a
|
||||
* string, or "000" on network error / timeout.
|
||||
* @param {string} target
|
||||
* @returns {Promise<string>}
|
||||
*/
|
||||
export async function checkAccessibility(target) {
|
||||
try {
|
||||
const res = await fetch(target, {
|
||||
method: "HEAD",
|
||||
redirect: "follow",
|
||||
signal: AbortSignal.timeout(10_000),
|
||||
});
|
||||
return String(res.status);
|
||||
} catch {
|
||||
return "000";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* collectMarkdown recursively collects *.md files under dir (the content tree is
|
||||
* small). Entries are visited in lexical order so callers that depend on
|
||||
* first-seen ordering stay deterministic. Missing or unreadable directories
|
||||
* yield nothing.
|
||||
* @param {string} dir
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function collectMarkdown(dir) {
|
||||
/** @type {string[]} */
|
||||
const acc = [];
|
||||
walkMarkdown(dir, acc);
|
||||
return acc;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string} dir
|
||||
* @param {string[]} acc
|
||||
* @returns {void}
|
||||
*/
|
||||
function walkMarkdown(dir, acc) {
|
||||
let entries;
|
||||
try {
|
||||
entries = readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return; // skip unreadable entries
|
||||
}
|
||||
entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
|
||||
for (const e of entries) {
|
||||
const p = join(dir, e.name);
|
||||
if (e.isDirectory()) walkMarkdown(p, acc);
|
||||
else if (e.name.toLowerCase().endsWith(".md")) acc.push(p);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* uuidBoundaryOk: the char after a UUID match must not extend the hex id, so
|
||||
* <uuid>.png (cover image), <uuid>_WxH and <uuid>) all match.
|
||||
* @param {string} text
|
||||
* @param {number} end
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function uuidBoundaryOk(text, end) {
|
||||
if (end >= text.length) return true;
|
||||
const c = text[end];
|
||||
return !((c >= "0" && c <= "9") || (c >= "a" && c <= "f") || (c >= "A" && c <= "F"));
|
||||
}
|
||||
|
||||
const URL_DELIMITERS = `)]"'?#<>_&,`;
|
||||
const URL_WHITESPACE = " \t\n\r\f\v";
|
||||
|
||||
/**
|
||||
* urlBoundaryOk: an optional trailing slash (bareUrl strips it, stored URLs may
|
||||
* keep it), then a path/punctuation delimiter, whitespace, or end of text — so
|
||||
* /p/foo does not match a stored /p/foo-bar. The '>' delimiter covers URLs
|
||||
* stored in markdown autolink form <https://…>, which older posts use.
|
||||
* @param {string} text
|
||||
* @param {number} end
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function urlBoundaryOk(text, end) {
|
||||
let at = end;
|
||||
if (at < text.length && text[at] === "/") at++;
|
||||
if (at >= text.length) return true;
|
||||
const c = text[at];
|
||||
if (URL_WHITESPACE.includes(c)) return true;
|
||||
return URL_DELIMITERS.includes(c);
|
||||
}
|
||||
|
||||
/**
|
||||
* hasBoundaryMatch scans every occurrence of needle and applies the boundary
|
||||
* check in code. Kept as an index loop rather than one lookahead regex: the
|
||||
* byte-offset checks (optional trailing slash, delimiter set) are what stop a
|
||||
* needle that is merely a PREFIX of a stored string from matching.
|
||||
* @param {string} text
|
||||
* @param {string} needle
|
||||
* @param {boolean} isUuidNeedle
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function hasBoundaryMatch(text, needle, isUuidNeedle) {
|
||||
for (let from = 0; ; ) {
|
||||
const i = text.indexOf(needle, from);
|
||||
if (i === -1) return false;
|
||||
const end = i + needle.length;
|
||||
if (isUuidNeedle ? uuidBoundaryOk(text, end) : urlBoundaryOk(text, end)) return true;
|
||||
from = i + 1;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* checkDuplicate reports whether a URL identity already exists in the stored
|
||||
* markdown, boundary-aware so a needle that is merely a PREFIX of a stored
|
||||
* longer string is NOT a false duplicate.
|
||||
* @param {string} target
|
||||
* @param {string} dir
|
||||
* @returns {boolean}
|
||||
*/
|
||||
export function checkDuplicate(target, dir) {
|
||||
const needle = bareUrl(target);
|
||||
if (needle === "") return false;
|
||||
const uuidNeedle = isUuid(needle);
|
||||
for (const file of collectMarkdown(dir)) {
|
||||
let text;
|
||||
try {
|
||||
text = readFileSync(file, "utf8");
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (text.includes(needle) && hasBoundaryMatch(text, needle, uuidNeedle)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const IMAGE_EXT_RE = /\.(png|jpg|jpeg|gif|webp|svg|avif|heic|heif|bmp|tiff?)(\?.*)?$/;
|
||||
const VIDEO_EXT_RE = /\.(mp4|webm|mov|avi|mkv)(\?.*)?$/;
|
||||
const DOCUMENT_EXT_RE = /\.(pdf|docx?|xlsx?|pptx?)(\?.*)?$/;
|
||||
|
||||
/**
|
||||
* classifyType classifies a URL by file extension.
|
||||
* @param {string} target
|
||||
* @returns {"image"|"video"|"document"|"article"}
|
||||
*/
|
||||
export function classifyType(target) {
|
||||
const lower = target.toLowerCase();
|
||||
if (IMAGE_EXT_RE.test(lower)) return "image";
|
||||
if (VIDEO_EXT_RE.test(lower)) return "video";
|
||||
if (DOCUMENT_EXT_RE.test(lower)) return "document";
|
||||
return "article";
|
||||
}
|
||||
Reference in new issue
Block a user