From 560877e87b369322e94de45f0976e4f0350300d9 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 14 Aug 2026 14:10:58 +0700 Subject: [PATCH 1/3] chore(ci): move the actions to their node24 releases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The runner warns that the node20 action runtime is deprecated. Every action in the workflow was on it: checkout v4 -> v7 setup-go v5 -> v7 setup-node v4 -> v7 upload-pages-artifact v3 -> v5 deploy-pages v4 -> v5 The two Pages actions move together because the artifact format is shared between them. None of the inputs this workflow passes changed across those majors — the releases are ESM migrations and the runtime bump. `node-version: 24` was already the Node the build runs on; this is about the runtime the actions themselves execute in. --- .github/workflows/deploy-pages.yml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index 502a273..4a9b153 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -26,9 +26,9 @@ jobs: # explicitly rather than relying on the default. CGO_ENABLED: '0' steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - - uses: actions/setup-go@v5 + - uses: actions/setup-go@v7 with: go-version: '1.26' cache-dependency-path: | @@ -37,7 +37,7 @@ jobs: assembler/go.sum # web/ is the only npm project in the repository; the other stages are Go. - - uses: actions/setup-node@v4 + - uses: actions/setup-node@v7 with: node-version: '24' cache: 'npm' @@ -91,7 +91,7 @@ jobs: - name: Build site run: go -C assembler run ./cmd/assemble - - uses: actions/upload-pages-artifact@v3 + - uses: actions/upload-pages-artifact@v5 with: path: _site @@ -108,4 +108,4 @@ jobs: url: ${{ steps.deployment.outputs.page_url }} steps: - id: deployment - uses: actions/deploy-pages@v4 + uses: actions/deploy-pages@v5 From d69965a3fe868ec99218adce1b448b1331da0928 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 14 Aug 2026 14:13:24 +0700 Subject: [PATCH 2/3] fix(ci): take the Go version from go.mod MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit setup-go v7 exports GOTOOLCHAIN=local, so Go no longer downloads the toolchain a module asks for — whatever setup-go installed has to satisfy it. `go-version: '1.26'` resolves to the runner manifest's patch, which is 1.26.5, against modules requiring 1.26.6 since this morning's CVE bump. The parser tests stopped on that. go-version-file installs exactly what the modules declare, and removes the fourth place a Go version was written down. --- .github/workflows/deploy-pages.yml | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index 4a9b153..56ed29b 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -28,9 +28,15 @@ jobs: steps: - uses: actions/checkout@v7 + # The version comes from go.mod rather than a literal here. setup-go sets + # GOTOOLCHAIN=local, so Go will not fetch the toolchain a module asks for + # — the one installed has to satisfy it. A literal '1.26' resolves to + # whatever patch the runner manifest has, which was 1.26.5 against a + # go.mod requiring 1.26.6, and the build stopped there. All three modules + # are bumped together, so parser/go.mod speaks for them. - uses: actions/setup-go@v7 with: - go-version: '1.26' + go-version-file: parser/go.mod cache-dependency-path: | parser/go.sum crawler/go.sum From 4ba38c3d0c4f2a89d87563f4572e81599fc33897 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 14 Aug 2026 14:17:51 +0700 Subject: [PATCH 3/3] docs: clear the plans folder, keeping the reasoning that outlives it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The range-request work is merged, so its plan and the two reports behind it describe a decision already taken. What they held that the code does not is why the alternatives were turned down, so that moves into system-architecture under "Considered and not taken": chunked serverMode (GitHub Pages' ten-minute cache TTL cancels the caching it buys), sqlite-wasm-http (its shared cache needs headers Pages cannot send, and swapping now would leave two unverified variables), and substring search (no index can serve it). Everything else those documents recorded — the measurements, the query plans, the sizes — is already in docs/ and in the commits that made the changes. Git keeps the originals either way. --- docs/system-architecture.md | 18 ++ .../260814-1200-httpvfs-range-queries/plan.md | 75 -------- ...-1317-httpvfs-page-size-adoption-report.md | 78 --------- ...0814-1317-httpvfs-best-practices-report.md | 165 ------------------ 4 files changed, 18 insertions(+), 318 deletions(-) delete mode 100644 plans/260814-1200-httpvfs-range-queries/plan.md delete mode 100644 plans/reports/from-research-to-implementation-brainstorm-260814-1317-httpvfs-page-size-adoption-report.md delete mode 100644 plans/reports/web-sqlite-range-hosting-research-260814-1317-httpvfs-best-practices-report.md diff --git a/docs/system-architecture.md b/docs/system-architecture.md index fd6d5e3..d0a027c 100644 --- a/docs/system-architecture.md +++ b/docs/system-architecture.md @@ -181,6 +181,24 @@ total descending. | Routing | SvelteKit file routes, prerendered | Each dataset gets a real HTML file with its own title | | Styling | Tailwind, with tier colours as CSS variables | Tier classes are chosen at runtime, which no utility generator can see | +### Considered and not taken + +- **Chunked `serverMode`.** `sql.js-httpvfs` can split a database into parts so + a CDN caches each one whole. GitHub Pages serves everything with + `Cache-Control: max-age=600`, and every rebuild relays SQLite's pages so the + file changes even when the data does not — the caching that mode buys is + cancelled by the host. Worth revisiting behind a CDN with long TTLs, and it is + also the fallback if a single 300 MB file ever becomes a problem. +- **`sqlite-wasm-http`.** Maintained, and built on the official SQLite WASM + rather than a 2022 fork, which is the better long-term footing. Its + differentiator — a cache shared between workers — needs COOP/COEP headers that + GitHub Pages cannot send, so here it would buy maintenance alone. Deferred + until the current path has been verified in a browser, so that a swap changes + one variable rather than two. +- **Substring name search.** `LIKE '%x%'` cannot use an index, so it read the + whole 127 MB table. `name_word` keeps search by any word of a name without + it. + ## Risks and limitations - **Unindexed queries are expensive.** The SQL tab can express a query that diff --git a/plans/260814-1200-httpvfs-range-queries/plan.md b/plans/260814-1200-httpvfs-range-queries/plan.md deleted file mode 100644 index 2ba55e2..0000000 --- a/plans/260814-1200-httpvfs-range-queries/plan.md +++ /dev/null @@ -1,75 +0,0 @@ -# Serve the databases over HTTP range requests - -Status: implemented, unverified in a browser - -The whole-database download is gone. `sql.js-httpvfs` reads the pages a query -touches, so the databases ship raw as `.sqlite3` — a byte range of a gzip -stream is not a byte range of a database. - -## Why the schema had to change first - -Measured on the real 2016 database (223.5 MB before, 4 KB pages, 27 rows/page): - -| Query | Plan before | Would have fetched | -| --- | --- | --- | -| `so_bao_danh = ?` | SEARCH via PK | ~20 KB | -| `ho_ten_ascii LIKE '%x%'` | SCAN | 127 MB | -| `ho_ten_ascii LIKE 'x%'` | SCAN — the LIKE optimisation needs a NOCASE index | 127 MB | -| `COUNT(*)` | covering scan of idx_ho_ten_ascii | 20 MB | -| preset `ORDER BY toan DESC LIMIT 10` | SCAN + temp b-tree | 127 MB | - -So substring search was impossible, prefix search was no better, and the footer -count alone cost 20 MB per page load. - -## What shipped - -**Parser.** `name_word(word, so_bao_danh, ho_ten_ascii)` WITHOUT ROWID — the -table is the index — plus `name_word_freq(word, n)` and partial indexes on -`toan`, `khtn`, `khxh`. Dropped `idx_ho_ten` and `idx_ho_ten_ascii`: no query -plan could use either. - -877,460 names hold 2.87M word entries over a vocabulary of 4,397. A search asks -the frequency table which word is rarest, seeks on that one, and filters the -rest against the `ho_ten_ascii` copy inside the same b-tree — so "buu loc" still -finds "Nguyễn Bửu Lộc", in a few hundred KB. - -| Segment | 2016 | -| --- | --- | -| `student` | 137.5 MB | -| `name_word` | 98.7 MB | -| `idx_ten_cum_thi` | 38.1 MB | -| PK autoindex | 15.3 MB | -| `idx_toan` | 12.6 MB | -| **total** | **302.4 MB** (2017: 247.3 MB) | - -Written with 1 KiB pages, so a row reached by an index seek costs one 1 KB -request instead of 4 KB: 6.3 rows share a page rather than 27, which is what -turns a 100-row search from ~400 KB of row fetches into ~100 KB. - -**Assembler.** Publishes uncompressed; the size guard reads the raw size; the -stray-artifact check now rejects journals, `.db` and `.gz`. - -**Web.** `RemoteDatabase` wraps `createDbWorker`. The search tab runs with a -25 MB byte budget, the SQL tab asks for consent and then gets 250 MB, and the -bytes fetched are shown next to the query time. The footer count comes from -`datasets.json`. - -## Verified - -- Row counts unchanged: 877,460 and 861,068, both through the assembler guards. -- Every query the app issues is index-driven, checked with `EXPLAIN QUERY PLAN`: - `SEARCH w USING PRIMARY KEY (word>? AND word.sqlite3` — would eat every chunk**, `site.go:135-138,153`, `datasets.ts:91-97`, `datasets.json`, ~15 tests | -| Library swap | `sqlite.svelte.ts:1-3,50-58,76,82`, `package.json:15`; all consumers go through `RemoteDatabase`, so the blast radius is one file | - -## Options evaluated - -### A. page_size 1024 + requestChunkSize 1024 — ADOPTED - -- Upstream consensus: phiresky and mmomtchev both recommend 1024. -- Honest sizing for *our* pattern: row fetches 400 KB → 100 KB per search; - the index walk is sequential so bytes are unchanged and only the request - count rises, which prefetch read-heads collapse. Net ≈ 300 KB saved per - search — bandwidth, not latency. -- Cost: ~10 lines; file size +5% (528 → ~555 MB total, still under the 1 GB - Pages limit); both databases rebuilt. - -### B. serverMode chunked — REJECTED - -- Only benefit is CDN cache efficiency. GitHub Pages serves everything with - `Cache-Control: max-age=600`, and each deploy relays out SQLite pages anyway, - so cross-deploy caching is zero either way. -- Cost: split step, config JSON, `Clean()`/guards/`dbOf()`/`datasets.json` - rework, ~15 tests. -- Complexity buying a benefit the host cancels. Revisit only behind a CDN with - long TTLs. - -### C. swap to sqlite-wasm-http — DEFERRED - -- For: maintained (Dec 2025), official SQLite WASM instead of a 2022 fork; - matches this repo's posture on stale dependencies. Swap is one file. -- Against: its differentiator (shared cache) needs COOP/COEP headers GitHub - Pages cannot send, so we would get the synchronous fallback and the - maintenance benefit only. And the current integration has never run in a - browser — swapping now means two unverified variables and no way to tell - which broke. -- Revisit after the current build is verified live. - -## Decision - -Adopt **A only**. - -## Implementation notes - -1. `PRAGMA page_size = 1024` in `writer.OpenDB`, between `sql.Open` and - `db.Exec(schema.DDL)`. The existing VACUUM in `Finish` applies it. -2. `CHUNK_BYTES = 1024` in `web/src/lib/sqlite.svelte.ts`; it feeds - `requestChunkSize` and must equal the page size. -3. Rebuild both databases, update `dbSizeMb` in `datasets.json` to the new - sizes, re-run the assembler guards. - -## Risks - -- The benefit is arithmetic plus upstream authority, not measurement. The byte - counter in the SQL tab is the check, once deployed. -- Prefetch deliberately overfetches ahead of the cursor, so the 25 MB search - budget may trip earlier than a strict page count suggests. Tune after a real - measurement, not before. - -## Unresolved questions - -1. Does 1024 actually beat 4096 for our queries in a browser? -2. Is Fastly's caching of ranges over a 300 MB object good enough that chunked - mode stays unnecessary? diff --git a/plans/reports/web-sqlite-range-hosting-research-260814-1317-httpvfs-best-practices-report.md b/plans/reports/web-sqlite-range-hosting-research-260814-1317-httpvfs-best-practices-report.md deleted file mode 100644 index ef11e7c..0000000 --- a/plans/reports/web-sqlite-range-hosting-research-260814-1317-httpvfs-best-practices-report.md +++ /dev/null @@ -1,165 +0,0 @@ -# Research Report: sql.js-httpvfs best practices - -Conducted 2026-08-14 13:17 (Asia/Saigon). Context: two static SQLite files -(2016 = 288.6 MB, 2017 = 237.7 MB) on GitHub Pages, branch -`feat/httpvfs-range-queries`. - -## Executive summary - -Our implementation matches upstream guidance on the thing that matters most — -index design — and diverges on one measurable parameter: **page size**. Both -phiresky (sql.js-httpvfs) and mmomtchev (sqlite-wasm-http) recommend -`page_size = 1024`; we shipped 4096. For our access pattern (~100 scattered -single-row reads per search) that is a real 4× overfetch on the row-fetch half -of a query. - -Two findings reduce risk rather than add work. Prefetching with three virtual -read heads makes a sequential scan cost a *logarithmic* number of requests, so -our byte estimates hold but latency is better than assumed. And the canonical -demo hosts a **670 MiB** database on GitHub Pages, so 289 MB is precedented. - -One finding is new and worth a decision: **chunked mode** (split file + JSON -config) exists specifically to make CDN caching effective for large databases, -which matters because GitHub Pages serves everything with `Cache-Control: -max-age=600`. - -## Methodology - -- Sources: 5 (1 primary blog, 2 project READMEs, 1 recent practitioner - writeup, 1 search on Pages/Fastly caching), plus direct reading of the - installed `sql.js-httpvfs@0.8.12` bundle in a previous session. -- Date range: 2021 (canonical post) → March 2026 (practitioner writeup). -- Gemini CLI absent → WebSearch/WebFetch. - -## Key findings - -### 1. Page size: recommended 1024, we use 4096 - -Both projects say the same thing. phiresky set 1 KiB pages "to balance request -overhead against bandwidth efficiency"; sqlite-wasm-http says "it is highly -recommended to decrease your SQLite page size to 1024 bytes for maximum -performance" (`PRAGMA page_size=1024; VACUUM`). - -`requestChunkSize` must match the page size. - -Measured on our 2016 file *before* the schema change: - -| page_size | file size | -| --- | --- | -| 1024 | 235.8 MB | -| 4096 | 223.5 MB | -| 8192 | 221.8 MB | - -So 1024 costs ~5.5% file size. What it buys: a scattered row read fetches 1 KB -instead of 4 KB. Our search does ~100 of those, so the row-fetch half of a -search drops from ~400 KB to ~100 KB. The index-walk half is sequential and -benefits from prefetch either way. - -**Verdict: switch to 1024 + `requestChunkSize: 1024`.** Our workload is -dominated by scattered single-row reads, which is exactly the case small pages -serve. - -### 2. Prefetch changes request count, not bytes - -"Three separate virtual read heads" detect sequential access and grow request -sizes exponentially, so "index scans or table scans reading more than a few KiB -of data will only cause a number of requests that is logarithmic in the total -byte length." - -Consequence for our analysis: a full table scan still transfers ~127 MB (bytes -are bytes), but in tens of requests rather than tens of thousands. Our byte -budget is the right guardrail; a request-count budget would not be. - -### 3. Index design — we already comply - -- Covering indexes: put every column the query needs *in* the index, else - SQLite does "another random access (unpredictable) read and thus HTTP request - to retrieve the actual value for every data point". This is exactly why - `name_word` carries `ho_ten_ascii`. -- Column order decides which lookups are cheap. -- Verify with `EXPLAIN QUERY PLAN`; a `SCAN` means the whole table crosses the - network. We did this for every query the app issues. - -### 4. Chunked mode exists for CDN caching - -`serverMode: "chunked"` splits the database into parts (10 MB is the commonly -cited size) with a JSON config. Stated benefit: "CDN caching much more -effective" for large databases. - -Relevant because **GitHub Pages sets `Cache-Control: max-age=600`** — ten -minutes — on everything. Range responses are cached by Fastly per object; with -one 289 MB object the practical caching story is weaker than with 29 chunks -that a CDN edge can hold whole. Chunked mode also sidesteps any future per-file -concern. - -Cost: a build step to split files + emit config, and every deploy invalidates -all chunks anyway (SQLite page layout is not deterministic across rebuilds). - -### 5. Hosting facts confirmed - -- Range requests work on Pages "out of the box" — matches our own probe (206 + - correct `Content-Range`). -- CORS headers (`Access-Control-Allow-Origin`, `Access-Control-Allow-Headers: - Range`) only matter cross-origin. Ours is same-origin — non-issue. -- 670 MiB database on Pages is the canonical demo. 289 MB is not exotic. -- `Content-Encoding` remains the one fatal case: the library discards - `Content-Length` and throws when a HEAD carries a non-identity encoding. - Unknown extensions like `.sqlite3` are served `application/octet-stream` and - left alone. - -### 6. Alternatives - -| | sql.js-httpvfs | sqlite-wasm-http | -| --- | --- | --- | -| WASM base | own sql.js fork (~3.36 era) | official `@sqlite.org/sqlite-wasm` | -| Last release | 0.8.12, Sept 2022 | 1.2.0, Dec 2023; activity into Dec 2025 | -| Self-description | "demo-level code… not for high stability" | "experimental" | -| Concurrency | one worker | multiple connections, shared cache | -| Shared cache needs | — | `SharedArrayBuffer` → COOP/COEP headers | -| On GitHub Pages | works | works, but **Pages cannot set COOP/COEP**, so it falls back to the synchronous backend without shared cache | -| Module format | CJS+ESM | **ES6 only** | - -Both are self-declared experimental. sqlite-wasm-http's advantage (maintained, -official WASM) is real; its headline feature (shared cache) is unavailable on -Pages precisely because Pages cannot send cross-origin isolation headers. - -## Implementation recommendations - -1. **Change page size to 1024 and `requestChunkSize` to 1024.** Parser sets - `PRAGMA page_size=1024` before DDL; VACUUM already runs. Cost ~+5% file - size; benefit ~4× less overfetch per row read. -2. **Keep `serverMode: "full"` for now.** Chunked mode's benefit is CDN cache - efficiency, which Pages' 10-minute TTL blunts. Revisit if measured repeat- - visit cost is bad. -3. **Keep sql.js-httpvfs.** Switching to sqlite-wasm-http buys a maintained - dependency but loses nothing we use, and its differentiator does not work on - Pages. Note it as the escape hatch. -4. **Verify `Content-Encoding` after deploy** — the single fatal hosting case. -5. Keep the byte budget; drop any idea of a request-count budget. - -## Common pitfalls - -- Unindexed query → whole table over the network. `EXPLAIN QUERY PLAN` is the - check. -- Page size mismatched with `requestChunkSize` → every logical page read spans - two requests. -- Serving the database compressed → library refuses to open it. -- Assuming CDN caching helps: on Pages, `max-age=600`. - -## References - -- [Hosting SQLite databases on GitHub Pages — phiresky](https://phiresky.github.io/blog/2021/hosting-sqlite-databases-on-github-pages/) -- [phiresky/sql.js-httpvfs](https://github.com/phiresky/sql.js-httpvfs) -- [mmomtchev/sqlite-wasm-http](https://github.com/mmomtchev/sqlite-wasm-http) -- [Query SQLite on GitHub Pages with sql.js-httpvfs (Mar 2026)](https://recca0120.github.io/en/2026/03/07/sql-js-httpvfs-static-hosting/) -- [sqlite3 WebAssembly documentation](https://sqlite.org/wasm) -- [GitHub Pages asset caching discussion](https://github.com/orgs/community/discussions/11884) - -## Unresolved questions - -1. Does 1024 measurably beat 4096 *for our queries*? Only a browser with the - byte counter can answer; the estimate says yes for row fetches. -2. Does Fastly cache 206 responses for a 289 MB object well enough that chunked - mode is unnecessary? Needs a deployed measurement. -3. Does the read-head prefetch overfetch on our index-range walks (fetching - ahead beyond `LIMIT 100`)? Unknown without instrumentation.