build: produce the whole site from one Vite build

The four build variants existed only because the app could not resolve its own
dataset. Now that it can, vite.config.js is a single build with an absolute
base and no VARIANT switching, and the workflow compiles Rust once, installs
Node dependencies once, and builds the site once — it previously built the same
Rust crate twice and ran two separate pnpm installs.

Because base is absolute, the emitted index.html references /thptqg/assets/...
regardless of where it is served from, so the same file works as an entry point
at any depth. scripts/assemble-site.js copies it to each dataset path and to
the two legacy nested URLs, giving a real static file at every published route.
That is what removes the need for an SPA 404-fallback redirect, which would
otherwise have rewritten URLs and interfered with the ?q= deep links.

Generated databases move to a gitignored .build/public, which Vite consumes as
its publicDir. build-db.js gzips without -k, and the assemble step then refuses
to finish if any uncompressed database artefact reached the output — .db,
.db-journal, .db-wal or .db-shm. Previously a raw 100+ MB database was written
into the source tree and deleted afterwards by an rm in the workflow, so
shipping one was a missing cleanup step away.

Both build scripts import DATASET_IDS from src/datasets.js rather than
repeating the dataset list in workflow shell, so the four ids are declared in
exactly one place across the frontend, the database build and the assembly.

Verified by running the full pipeline locally and serving the artifact over
HTTP: all nine routes return 200, every entry point is byte-identical, asset
references are absolute, and the databases are fetchable. The guard was tested
by injecting a raw .db and .db-journal into the build, which failed the
assembly as intended.
This commit is contained in:
tiennm99 committed 2026-08-13 11:57:26 +07:00
1 parent 003e7c8afd
commit a5904a5d78
7 files changed
+151 -78

No files matched your search

+15 -43
View File
@@ -24,57 +24,29 @@ jobs:
- uses: Swatinem/rust-cache@v2
with:
workspaces: |
2016/tools/xlsxread
2017/tools/xlsxread
- uses: pnpm/action-setup@v4
with:
package_json_file: 2017/package.json
workspaces: parser
- uses: actions/setup-node@v4
with:
node-version: '24'
cache: 'pnpm'
cache-dependency-path: |
2016/pnpm-lock.yaml
2017/pnpm-lock.yaml
cache: 'npm'
cache-dependency-path: package-lock.json
- name: Build 2016 site
working-directory: '2016'
run: |
set -euo pipefail
cargo build --release --manifest-path tools/xlsxread/Cargo.toml
./tools/xlsxread/target/release/xlsxread build \
--schema tools/xlsxread/configs/thptqg2016-data.toml \
--input data --output public/thptqg2016.db
pnpm install --frozen-lockfile
gzip -k -9 public/thptqg2016.db
pnpm build
rm -f dist/thptqg2016.db
- run: npm ci
- name: Build 2017 site variants
working-directory: '2017'
# One parser binary builds every dataset; build-db.js reads the dataset
# list from src/datasets.js and gzips each database in place, leaving no
# uncompressed file behind.
- name: Build databases
run: |
set -euo pipefail
cargo build --release --manifest-path tools/xlsxread/Cargo.toml
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data.toml --input data --output public/thptqg2017.db
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old.toml --input data-old --output public-old/thptqg2017.db
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old2.toml --input data-old2 --output public-old2/thptqg2017.db
pnpm install --frozen-lockfile
gzip -kf -9 public/thptqg2017.db
gzip -kf -9 public-old/thptqg2017.db
gzip -kf -9 public-old2/thptqg2017.db
pnpm build:all
rm -f dist/thptqg2017.db dist/old/thptqg2017.db dist/old2/thptqg2017.db
npm run build:rust
npm run build:db
- name: Assemble site
run: |
set -euo pipefail
mkdir -p _site
cp index.html _site/
cp -r 2016/dist _site/2016
cp -r 2017/dist _site/2017
# One Vite build produces every page. scripts/assemble-site.js copies the
# emitted index.html to each dataset path (and the legacy nested URLs),
# then fails the job if any uncompressed database reached the artifact.
- name: Build and assemble site
run: npm run build:site
- uses: actions/upload-pages-artifact@v3
with:
+3
View File
@@ -19,3 +19,6 @@ dist/
### Rust build artefacts ###
parser/target/
### Assembled Pages artifact ###
_site/
+2 -2
View File
@@ -5,7 +5,7 @@ import reactRefresh from 'eslint-plugin-react-refresh'
import { defineConfig, globalIgnores } from 'eslint/config'
export default defineConfig([
globalIgnores(['dist', '.build']),
globalIgnores(['dist', '.build', '_site']),
{
files: ['**/*.{js,jsx}'],
extends: [
@@ -28,7 +28,7 @@ export default defineConfig([
},
{
// Node-executed files (Vite config, parser tooling) run with Node globals.
files: ['vite.config.js', 'parser/scripts/**/*.js'],
files: ['vite.config.js', 'scripts/**/*.js', 'parser/scripts/**/*.js'],
languageOptions: {
globals: { ...globals.node },
},
+2
View File
@@ -9,6 +9,8 @@
"build:db": "node parser/scripts/build-db.js",
"dev": "vite",
"build": "vite build",
"assemble": "node scripts/assemble-site.js",
"build:site": "npm run build && npm run assemble",
"preview": "vite preview",
"lint": "eslint ."
},
@@ -130,11 +130,24 @@ Cache changes: `Swatinem/rust-cache` workspaces → `parser`; `setup-node` cache
## Related Code Files
- Modify: `vite.config.js` — single build, `base: /thptqg/`, `.build/public` publicDir
- Modify: `package.json` — replace the six `build:db*` and three `build:*` variant
scripts with one `build:db` (takes a dataset argument) and one `build`
- Modify: `.github/workflows/deploy-pages.yml` — single toolchain setup, dataset
loop, npm caches, new assemble step
- Modify: `.gitignore` — add `.build/`, `parser/target/`, keep `dist/`
- Modify: `package.json` — the six `build:db*` and three `build:*` variant scripts
collapse to `build:db`, `build`, `assemble`, `build:site`
- Create: `scripts/assemble-site.js` — copies the entry point to each route and
refuses to ship an uncompressed database
- Modify: `.github/workflows/deploy-pages.yml` — single toolchain setup, npm
caches, new assemble step
- Modify: `eslint.config.js` — Node globals for `scripts/**`, ignore `_site`
- Modify: `.gitignore` — add `.build/`, `_site/`, `parser/target/`, keep `dist/`
### Deviation: the dataset loop is a Node script, not workflow shell
The plan sketched a `for ds in 2016 2017 …` loop inline in the workflow, with a
note to hoist the list into a job-level env var. That would still have been a
second copy of the dataset list. `parser/scripts/build-db.js` and
`scripts/assemble-site.js` both import `DATASET_IDS` from `src/datasets.js`
instead, so the four IDs are declared exactly once for the frontend, the
database build and the site assembly. It also makes the whole pipeline runnable
locally with `npm run build:site`, which is how it was verified.
## Implementation Steps
@@ -164,15 +177,27 @@ Cache changes: `Swatinem/rust-cache` workspaces → `parser`; `setup-node` cache
## Success Criteria
- [ ] One `vite.config.js`, no build variants, no `DATASET` env
- [ ] Workflow compiles Rust once, installs Node deps once, builds the site once
- [ ] All five flat URLs live; deep links intact
- [ ] Both legacy nested URLs resolve rather than 404
- [ ] Dataset list written once in the workflow, not repeated per step
- [ ] No SPA 404-redirect hack in the repo
- [ ] No uncompressed DB anywhere in the artifact
- [ ] Generated DBs live in gitignored `.build/`, not in source directories
- [ ] Zero pnpm references in the workflow
- [x] One `vite.config.js`, no build variants, no `DATASET` env
- [x] Workflow compiles Rust once, installs Node deps once, builds the site once
- [x] All five flat URLs served; deep links intact
- [x] Both legacy nested URLs resolve rather than 404
- [x] Dataset list written once (`src/datasets.js`), not repeated in the workflow
- [x] No SPA 404-redirect hack in the repo
- [x] No uncompressed DB anywhere in the artifact — enforced by the assemble step
- [x] Generated DBs live in gitignored `.build/`, not in source directories
- [x] Zero pnpm references in the workflow
## Verification limits
Route resolution is verified by serving the assembled artifact over HTTP and
checking every published URL returns 200 with correct absolute asset
references, plus that all entry points are byte-identical.
The routing **JavaScript** has not been executed. This workspace is headless
with no browser available, so hub-vs-dataset rendering and the legacy
`/2017/old/` → `/2017-old/` rewrite are verified by construction and by unit
tests of the pure functions, not by running in a browser. That check needs a
real browser.
## Risk Assessment
+77
View File
@@ -0,0 +1,77 @@
#!/usr/bin/env node
/**
* Assemble the GitHub Pages artifact from a single Vite build.
*
* The app resolves its dataset from the URL, so every page is the same
* index.html. Because `base` is absolute (/thptqg/), that file references
* /thptqg/assets/... no matter which directory it is served from — so copying
* it to each dataset path produces a real static file at every URL.
*
* GitHub Pages serves those as directory indexes, which is why this needs no
* SPA 404-fallback redirect. That matters beyond tidiness: the usual fallback
* rewrites the URL and would interfere with the ?q= deep links the app relies
* on.
*
* Usage: node scripts/assemble-site.js
*/
import { cpSync, mkdirSync, rmSync, existsSync, readdirSync, statSync } from "node:fs";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { DATASET_IDS } from "../src/datasets.js";
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), "..");
const DIST = join(ROOT, "dist");
const SITE = join(ROOT, "_site");
/** URLs published before the switch to flat paths; the router rewrites these. */
const LEGACY_PATHS = ["2017/old", "2017/old2"];
if (!existsSync(join(DIST, "index.html"))) {
console.error(`no build found at ${DIST} — run: npm run build`);
process.exit(1);
}
rmSync(SITE, { recursive: true, force: true });
mkdirSync(SITE, { recursive: true });
// Base build: index.html, assets/, and the gzipped databases from publicDir.
cpSync(DIST, SITE, { recursive: true });
// Unknown paths render the hub rather than the default Pages 404.
cpSync(join(DIST, "index.html"), join(SITE, "404.html"));
// One entry point per dataset, plus the legacy nested URLs.
for (const path of [...DATASET_IDS, ...LEGACY_PATHS]) {
mkdirSync(join(SITE, path), { recursive: true });
cpSync(join(DIST, "index.html"), join(SITE, path, "index.html"));
}
// Only gzipped databases may ship. build-db.js gzips without -k so no raw .db
// should exist, but publicDir is copied wholesale — a leftover from an
// interrupted build would go straight through, and a raw database is 100+ MB.
// SQLite also leaves .db-journal files mid-build, so anything that is not a
// .gz is rejected rather than just files ending in .db.
const stray = [];
(function walk(dir) {
for (const entry of readdirSync(dir)) {
const full = join(dir, entry);
if (statSync(full).isDirectory()) walk(full);
else if (/\.db(-journal|-wal|-shm)?$/.test(entry)) stray.push(full);
}
})(SITE);
if (stray.length) {
console.error("uncompressed database artefact(s) found in the site output:");
for (const f of stray) {
console.error(` ${f} (${(statSync(f).size / 1048576).toFixed(1)} MB)`);
}
console.error("\nremove them from .build/public/db and re-run");
process.exit(1);
}
console.log(`assembled ${SITE}`);
for (const path of ["", "404.html", ...DATASET_IDS, ...LEGACY_PATHS]) {
console.log(` /thptqg/${path}`);
}
+13 -19
View File
@@ -1,25 +1,19 @@
import { defineConfig } from "vite";
import react from "@vitejs/plugin-react";
// VARIANT selects which dataset to ship:
// (unset) → main site at /thptqg2017/ using public/
// old → at /thptqg2017/old/ using public-old/
// old2 → at /thptqg2017/old2/ using public-old2/
const VARIANT = process.env.VARIANT || "";
const VARIANT_CONFIG = {
"": { base: "/thptqg/2017/", publicDir: "public", outDir: "dist" },
old: { base: "/thptqg/2017/old/", publicDir: "public-old", outDir: "dist/old" },
old2: { base: "/thptqg/2017/old2/", publicDir: "public-old2", outDir: "dist/old2" },
};
const cfg = VARIANT_CONFIG[VARIANT];
if (!cfg) throw new Error(`Unknown VARIANT: ${VARIANT}`);
// One build serves every page. The app resolves which dataset to show from the
// URL (src/router.js), so there are no per-dataset build variants.
//
// `base` is absolute, which is what lets the single emitted index.html work as
// an entry point at any depth: it references /thptqg/assets/... regardless of
// the directory it is served from. The deploy step copies it to each dataset
// path, so every URL is a real static file and no SPA 404-fallback is needed —
// and the existing ?q= deep links keep working, which that fallback would break.
//
// publicDir holds only the gzipped databases, staged there by
// parser/scripts/build-db.js. Nothing uncompressed is ever placed in it.
export default defineConfig({
plugins: [react()],
base: cfg.base,
publicDir: cfg.publicDir,
build: {
outDir: cfg.outDir,
emptyOutDir: VARIANT === "", // only main build wipes dist root
},
base: "/thptqg/",
publicDir: ".build/public",
});