From a5904a5d782b3e577028ea8b6f1047e4f74e8917 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Thu, 13 Aug 2026 11:57:26 +0700 Subject: [PATCH] build: produce the whole site from one Vite build MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The four build variants existed only because the app could not resolve its own dataset. Now that it can, vite.config.js is a single build with an absolute base and no VARIANT switching, and the workflow compiles Rust once, installs Node dependencies once, and builds the site once — it previously built the same Rust crate twice and ran two separate pnpm installs. Because base is absolute, the emitted index.html references /thptqg/assets/... regardless of where it is served from, so the same file works as an entry point at any depth. scripts/assemble-site.js copies it to each dataset path and to the two legacy nested URLs, giving a real static file at every published route. That is what removes the need for an SPA 404-fallback redirect, which would otherwise have rewritten URLs and interfered with the ?q= deep links. Generated databases move to a gitignored .build/public, which Vite consumes as its publicDir. build-db.js gzips without -k, and the assemble step then refuses to finish if any uncompressed database artefact reached the output — .db, .db-journal, .db-wal or .db-shm. Previously a raw 100+ MB database was written into the source tree and deleted afterwards by an rm in the workflow, so shipping one was a missing cleanup step away. Both build scripts import DATASET_IDS from src/datasets.js rather than repeating the dataset list in workflow shell, so the four ids are declared in exactly one place across the frontend, the database build and the assembly. Verified by running the full pipeline locally and serving the artifact over HTTP: all nine routes return 200, every entry point is byte-identical, asset references are absolute, and the databases are fetchable. The guard was tested by injecting a raw .db and .db-journal into the build, which failed the assembly as intended. --- .github/workflows/deploy-pages.yml | 58 ++++---------- .gitignore | 3 + eslint.config.js | 4 +- package.json | 2 + .../phase-04-build-and-deploy-pipeline.md | 53 +++++++++---- scripts/assemble-site.js | 77 +++++++++++++++++++ vite.config.js | 32 ++++---- 7 files changed, 151 insertions(+), 78 deletions(-) create mode 100644 scripts/assemble-site.js diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index f37762e..71dfc46 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -24,57 +24,29 @@ jobs: - uses: Swatinem/rust-cache@v2 with: - workspaces: | - 2016/tools/xlsxread - 2017/tools/xlsxread - - - uses: pnpm/action-setup@v4 - with: - package_json_file: 2017/package.json + workspaces: parser - uses: actions/setup-node@v4 with: node-version: '24' - cache: 'pnpm' - cache-dependency-path: | - 2016/pnpm-lock.yaml - 2017/pnpm-lock.yaml + cache: 'npm' + cache-dependency-path: package-lock.json - - name: Build 2016 site - working-directory: '2016' - run: | - set -euo pipefail - cargo build --release --manifest-path tools/xlsxread/Cargo.toml - ./tools/xlsxread/target/release/xlsxread build \ - --schema tools/xlsxread/configs/thptqg2016-data.toml \ - --input data --output public/thptqg2016.db - pnpm install --frozen-lockfile - gzip -k -9 public/thptqg2016.db - pnpm build - rm -f dist/thptqg2016.db + - run: npm ci - - name: Build 2017 site variants - working-directory: '2017' + # One parser binary builds every dataset; build-db.js reads the dataset + # list from src/datasets.js and gzips each database in place, leaving no + # uncompressed file behind. + - name: Build databases run: | - set -euo pipefail - cargo build --release --manifest-path tools/xlsxread/Cargo.toml - ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data.toml --input data --output public/thptqg2017.db - ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old.toml --input data-old --output public-old/thptqg2017.db - ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old2.toml --input data-old2 --output public-old2/thptqg2017.db - pnpm install --frozen-lockfile - gzip -kf -9 public/thptqg2017.db - gzip -kf -9 public-old/thptqg2017.db - gzip -kf -9 public-old2/thptqg2017.db - pnpm build:all - rm -f dist/thptqg2017.db dist/old/thptqg2017.db dist/old2/thptqg2017.db + npm run build:rust + npm run build:db - - name: Assemble site - run: | - set -euo pipefail - mkdir -p _site - cp index.html _site/ - cp -r 2016/dist _site/2016 - cp -r 2017/dist _site/2017 + # One Vite build produces every page. scripts/assemble-site.js copies the + # emitted index.html to each dataset path (and the legacy nested URLs), + # then fails the job if any uncompressed database reached the artifact. + - name: Build and assemble site + run: npm run build:site - uses: actions/upload-pages-artifact@v3 with: diff --git a/.gitignore b/.gitignore index 22c54aa..c67537c 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,6 @@ dist/ ### Rust build artefacts ### parser/target/ + +### Assembled Pages artifact ### +_site/ diff --git a/eslint.config.js b/eslint.config.js index 2b630cc..78101ed 100644 --- a/eslint.config.js +++ b/eslint.config.js @@ -5,7 +5,7 @@ import reactRefresh from 'eslint-plugin-react-refresh' import { defineConfig, globalIgnores } from 'eslint/config' export default defineConfig([ - globalIgnores(['dist', '.build']), + globalIgnores(['dist', '.build', '_site']), { files: ['**/*.{js,jsx}'], extends: [ @@ -28,7 +28,7 @@ export default defineConfig([ }, { // Node-executed files (Vite config, parser tooling) run with Node globals. - files: ['vite.config.js', 'parser/scripts/**/*.js'], + files: ['vite.config.js', 'scripts/**/*.js', 'parser/scripts/**/*.js'], languageOptions: { globals: { ...globals.node }, }, diff --git a/package.json b/package.json index 7a32a5c..d5cc3c6 100644 --- a/package.json +++ b/package.json @@ -9,6 +9,8 @@ "build:db": "node parser/scripts/build-db.js", "dev": "vite", "build": "vite build", + "assemble": "node scripts/assemble-site.js", + "build:site": "npm run build && npm run assemble", "preview": "vite preview", "lint": "eslint ." }, diff --git a/plans/260813-0956-unify-frontend-standard-schema/phase-04-build-and-deploy-pipeline.md b/plans/260813-0956-unify-frontend-standard-schema/phase-04-build-and-deploy-pipeline.md index 206b3de..c94c595 100644 --- a/plans/260813-0956-unify-frontend-standard-schema/phase-04-build-and-deploy-pipeline.md +++ b/plans/260813-0956-unify-frontend-standard-schema/phase-04-build-and-deploy-pipeline.md @@ -130,11 +130,24 @@ Cache changes: `Swatinem/rust-cache` workspaces → `parser`; `setup-node` cache ## Related Code Files - Modify: `vite.config.js` — single build, `base: /thptqg/`, `.build/public` publicDir -- Modify: `package.json` — replace the six `build:db*` and three `build:*` variant - scripts with one `build:db` (takes a dataset argument) and one `build` -- Modify: `.github/workflows/deploy-pages.yml` — single toolchain setup, dataset - loop, npm caches, new assemble step -- Modify: `.gitignore` — add `.build/`, `parser/target/`, keep `dist/` +- Modify: `package.json` — the six `build:db*` and three `build:*` variant scripts + collapse to `build:db`, `build`, `assemble`, `build:site` +- Create: `scripts/assemble-site.js` — copies the entry point to each route and + refuses to ship an uncompressed database +- Modify: `.github/workflows/deploy-pages.yml` — single toolchain setup, npm + caches, new assemble step +- Modify: `eslint.config.js` — Node globals for `scripts/**`, ignore `_site` +- Modify: `.gitignore` — add `.build/`, `_site/`, `parser/target/`, keep `dist/` + +### Deviation: the dataset loop is a Node script, not workflow shell + +The plan sketched a `for ds in 2016 2017 …` loop inline in the workflow, with a +note to hoist the list into a job-level env var. That would still have been a +second copy of the dataset list. `parser/scripts/build-db.js` and +`scripts/assemble-site.js` both import `DATASET_IDS` from `src/datasets.js` +instead, so the four IDs are declared exactly once for the frontend, the +database build and the site assembly. It also makes the whole pipeline runnable +locally with `npm run build:site`, which is how it was verified. ## Implementation Steps @@ -164,15 +177,27 @@ Cache changes: `Swatinem/rust-cache` workspaces → `parser`; `setup-node` cache ## Success Criteria -- [ ] One `vite.config.js`, no build variants, no `DATASET` env -- [ ] Workflow compiles Rust once, installs Node deps once, builds the site once -- [ ] All five flat URLs live; deep links intact -- [ ] Both legacy nested URLs resolve rather than 404 -- [ ] Dataset list written once in the workflow, not repeated per step -- [ ] No SPA 404-redirect hack in the repo -- [ ] No uncompressed DB anywhere in the artifact -- [ ] Generated DBs live in gitignored `.build/`, not in source directories -- [ ] Zero pnpm references in the workflow +- [x] One `vite.config.js`, no build variants, no `DATASET` env +- [x] Workflow compiles Rust once, installs Node deps once, builds the site once +- [x] All five flat URLs served; deep links intact +- [x] Both legacy nested URLs resolve rather than 404 +- [x] Dataset list written once (`src/datasets.js`), not repeated in the workflow +- [x] No SPA 404-redirect hack in the repo +- [x] No uncompressed DB anywhere in the artifact — enforced by the assemble step +- [x] Generated DBs live in gitignored `.build/`, not in source directories +- [x] Zero pnpm references in the workflow + +## Verification limits + +Route resolution is verified by serving the assembled artifact over HTTP and +checking every published URL returns 200 with correct absolute asset +references, plus that all entry points are byte-identical. + +The routing **JavaScript** has not been executed. This workspace is headless +with no browser available, so hub-vs-dataset rendering and the legacy +`/2017/old/` → `/2017-old/` rewrite are verified by construction and by unit +tests of the pure functions, not by running in a browser. That check needs a +real browser. ## Risk Assessment diff --git a/scripts/assemble-site.js b/scripts/assemble-site.js new file mode 100644 index 0000000..919aa0b --- /dev/null +++ b/scripts/assemble-site.js @@ -0,0 +1,77 @@ +#!/usr/bin/env node +/** + * Assemble the GitHub Pages artifact from a single Vite build. + * + * The app resolves its dataset from the URL, so every page is the same + * index.html. Because `base` is absolute (/thptqg/), that file references + * /thptqg/assets/... no matter which directory it is served from — so copying + * it to each dataset path produces a real static file at every URL. + * + * GitHub Pages serves those as directory indexes, which is why this needs no + * SPA 404-fallback redirect. That matters beyond tidiness: the usual fallback + * rewrites the URL and would interfere with the ?q= deep links the app relies + * on. + * + * Usage: node scripts/assemble-site.js + */ + +import { cpSync, mkdirSync, rmSync, existsSync, readdirSync, statSync } from "node:fs"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +import { DATASET_IDS } from "../src/datasets.js"; + +const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), ".."); +const DIST = join(ROOT, "dist"); +const SITE = join(ROOT, "_site"); + +/** URLs published before the switch to flat paths; the router rewrites these. */ +const LEGACY_PATHS = ["2017/old", "2017/old2"]; + +if (!existsSync(join(DIST, "index.html"))) { + console.error(`no build found at ${DIST} — run: npm run build`); + process.exit(1); +} + +rmSync(SITE, { recursive: true, force: true }); +mkdirSync(SITE, { recursive: true }); + +// Base build: index.html, assets/, and the gzipped databases from publicDir. +cpSync(DIST, SITE, { recursive: true }); + +// Unknown paths render the hub rather than the default Pages 404. +cpSync(join(DIST, "index.html"), join(SITE, "404.html")); + +// One entry point per dataset, plus the legacy nested URLs. +for (const path of [...DATASET_IDS, ...LEGACY_PATHS]) { + mkdirSync(join(SITE, path), { recursive: true }); + cpSync(join(DIST, "index.html"), join(SITE, path, "index.html")); +} + +// Only gzipped databases may ship. build-db.js gzips without -k so no raw .db +// should exist, but publicDir is copied wholesale — a leftover from an +// interrupted build would go straight through, and a raw database is 100+ MB. +// SQLite also leaves .db-journal files mid-build, so anything that is not a +// .gz is rejected rather than just files ending in .db. +const stray = []; +(function walk(dir) { + for (const entry of readdirSync(dir)) { + const full = join(dir, entry); + if (statSync(full).isDirectory()) walk(full); + else if (/\.db(-journal|-wal|-shm)?$/.test(entry)) stray.push(full); + } +})(SITE); + +if (stray.length) { + console.error("uncompressed database artefact(s) found in the site output:"); + for (const f of stray) { + console.error(` ${f} (${(statSync(f).size / 1048576).toFixed(1)} MB)`); + } + console.error("\nremove them from .build/public/db and re-run"); + process.exit(1); +} + +console.log(`assembled ${SITE}`); +for (const path of ["", "404.html", ...DATASET_IDS, ...LEGACY_PATHS]) { + console.log(` /thptqg/${path}`); +} diff --git a/vite.config.js b/vite.config.js index 87e0811..1c01e6b 100644 --- a/vite.config.js +++ b/vite.config.js @@ -1,25 +1,19 @@ import { defineConfig } from "vite"; import react from "@vitejs/plugin-react"; -// VARIANT selects which dataset to ship: -// (unset) → main site at /thptqg2017/ using public/ -// old → at /thptqg2017/old/ using public-old/ -// old2 → at /thptqg2017/old2/ using public-old2/ -const VARIANT = process.env.VARIANT || ""; -const VARIANT_CONFIG = { - "": { base: "/thptqg/2017/", publicDir: "public", outDir: "dist" }, - old: { base: "/thptqg/2017/old/", publicDir: "public-old", outDir: "dist/old" }, - old2: { base: "/thptqg/2017/old2/", publicDir: "public-old2", outDir: "dist/old2" }, -}; -const cfg = VARIANT_CONFIG[VARIANT]; -if (!cfg) throw new Error(`Unknown VARIANT: ${VARIANT}`); - +// One build serves every page. The app resolves which dataset to show from the +// URL (src/router.js), so there are no per-dataset build variants. +// +// `base` is absolute, which is what lets the single emitted index.html work as +// an entry point at any depth: it references /thptqg/assets/... regardless of +// the directory it is served from. The deploy step copies it to each dataset +// path, so every URL is a real static file and no SPA 404-fallback is needed — +// and the existing ?q= deep links keep working, which that fallback would break. +// +// publicDir holds only the gzipped databases, staged there by +// parser/scripts/build-db.js. Nothing uncompressed is ever placed in it. export default defineConfig({ plugins: [react()], - base: cfg.base, - publicDir: cfg.publicDir, - build: { - outDir: cfg.outDir, - emptyOutDir: VARIANT === "", // only main build wipes dist root - }, + base: "/thptqg/", + publicDir: ".build/public", });