diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index b121217..ed7aaf1 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -137,7 +137,7 @@ jobs: run: | set -euo pipefail for id in $(jq -r '.datasets[].id' datasets.json); do - url="${PAGE_URL%/}/db/${id}.sqlite3" + url="${PAGE_URL%/}/db/${id}.sqlite30" if ! magic=$(curl -sf -r 0-14 -H 'Accept-Encoding: identity;q=1, *;q=0' "$url"); then echo "::error::$url is not fetchable" exit 1 diff --git a/CLAUDE.md b/CLAUDE.md index fd837ef..b5c157c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -59,17 +59,24 @@ hashes every real input file. That is the point of it; do not skip it. used — a fallback would break the `?q=` deep links. - **`dbSizeMb` in `datasets.json` is a build guard, not just a label.** The assembler refuses to publish an artifact that falls below a ratio of it. -- **The databases ship uncompressed, as `.sqlite3`.** The browser reads - byte ranges of them, and a range of a gzip stream is not a range of the - database. -- **The file length comes from a range request, not from the host's HEAD.** - GitHub Pages gzips `application/octet-stream`, so a HEAD reports the - compressed size and `sql.js-httpvfs` refuses to open the file. Ranged reads - are unaffected — browsers must send `Accept-Encoding: identity` whenever a - request carries a `Range` header — so `web/src/lib/db-probe.js` reads the - header over a range and passes `fileLength`. Verify the way a browser asks: - `curl -sI -r 0-99 -H 'Accept-Encoding: identity' …`, never a bare `curl -sI`, - which advertises no encoding and hides the problem. +- **The databases ship uncompressed, as `.sqlite30`.** The trailing 0 is a + chunk index, not a typo — see the next point. The browser reads byte ranges + of the file, and a range of a gzip stream is not a range of the database. +- **The site uses chunked mode over a single chunk, and that is deliberate.** + GitHub Pages gzips `application/octet-stream`, so the HEAD request + `sql.js-httpvfs` sizes a file with reports the compressed length, and the + library refuses to open the file. Chunked mode is the only mode whose config + takes a length (`databaseLengthBytes`); in full mode the worker hardcodes it + to `undefined`, so a length passed there is silently dropped. One chunk holds + the database, so the index is always 0 and every request goes to + `.sqlite3` + `0`. `web/src/lib/db-probe.js` supplies the length by + reading the file header over a range request. +- **Ranged reads were never affected by the compression**, because browsers + must send `Accept-Encoding: identity` whenever a request carries a `Range` + header. Verify the way a browser asks — + `curl -s -r 0-14 -H 'Accept-Encoding: identity' …` must print + `SQLite format 3` — never a bare `curl -sI`, which advertises no encoding + and so passes whatever the host does. - **Every query the site runs must be index-driven.** Over range requests an unindexed query fetches the whole table. Hence no index on `ho_ten` (nothing can use one), `name_word` for name search, partial indexes for the score diff --git a/README.md b/README.md index 1fcbda7..06de0e5 100644 --- a/README.md +++ b/README.md @@ -42,7 +42,7 @@ app both read it and neither needs a dependency to do so; presentation stays in The dataset id is one identifier end to end: ``` -data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite30 → /thptqg/2017/ ``` ## Build diff --git a/assembler/internal/databases/databases.go b/assembler/internal/databases/databases.go index c294c46..fc22429 100644 --- a/assembler/internal/databases/databases.go +++ b/assembler/internal/databases/databases.go @@ -35,10 +35,18 @@ const driverName = "sqlite" // even if the row count somehow passed. const minSizeRatio = 0.9 -// Extension is the published suffix. Not ".db": the sql.js-httpvfs ecosystem -// uses ".sqlite3", and keeping ".db" free lets the site assembly treat any +// Extension is the published suffix. The trailing 0 is a chunk index, not a +// typo: the browser reads the file through sql.js-httpvfs in chunked mode, +// which is the only mode whose config accepts the file's length. That mode +// builds each request's URL as urlPrefix + chunkIndex, and one chunk holds the +// whole database, so the index is always 0 and the prefix is ".sqlite3". +// +// The length has to come from the config because the library otherwise takes +// it from a HEAD request, which GitHub Pages answers with the gzipped size. +// +// Not ".db" either: keeping that name free lets the site assembly treat any // stray .db or SQLite journal in the output as the leftover it is. -const Extension = ".sqlite3" +const Extension = ".sqlite30" // Paths locates the pieces this package needs. type Paths struct { diff --git a/assembler/internal/databases/databases_test.go b/assembler/internal/databases/databases_test.go index b14aada..7086165 100644 --- a/assembler/internal/databases/databases_test.go +++ b/assembler/internal/databases/databases_test.go @@ -12,9 +12,9 @@ import ( func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { dir := t.TempDir() for _, name := range []string{ - "2016.sqlite3", "2017.sqlite3", - "2017-old.sqlite3", // dropped from the registry - "2017-old2.sqlite3", // dropped from the registry + "2016.sqlite30", "2017.sqlite30", + "2017-old.sqlite30", // dropped from the registry + "2017-old2.sqlite30", // dropped from the registry "2016.db-journal", // interrupted run } { if err := os.WriteFile(filepath.Join(dir, name), []byte("x"), 0o644); err != nil { @@ -38,7 +38,7 @@ func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { } slices.Sort(left) - want := []string{"2016.sqlite3", "2017.sqlite3"} + want := []string{"2016.sqlite30", "2017.sqlite30"} if !slices.Equal(left, want) { t.Errorf("left %v, want %v", left, want) } diff --git a/assembler/internal/site/site.go b/assembler/internal/site/site.go index 5018b2c..35020dc 100644 --- a/assembler/internal/site/site.go +++ b/assembler/internal/site/site.go @@ -148,9 +148,10 @@ func checkDatabasesPresent(siteDir string, datasets []registry.Dataset) error { } // strayArtifact matches what must never reach the output: a SQLite journal from -// an interrupted run, a database under the old .db name, or a gzipped database -// from before the switch to range requests. -var strayArtifact = regexp.MustCompile(`(\.db|\.sqlite3)(-journal|-wal|-shm)$|\.db$|\.gz$`) +// an interrupted run, a database under either older name — .db, or .sqlite3 +// without the chunk index the client asks for — or a gzipped database from +// before the switch to range requests. +var strayArtifact = regexp.MustCompile(`(\.db|\.sqlite30?)(-journal|-wal|-shm)$|\.db$|\.sqlite3$|\.gz$`) // checkNoStrayArtifacts rejects leftovers that would be published. // diff --git a/assembler/internal/site/site_test.go b/assembler/internal/site/site_test.go index dae2b8e..56c2902 100644 --- a/assembler/internal/site/site_test.go +++ b/assembler/internal/site/site_test.go @@ -39,7 +39,7 @@ func write(t *testing.T, path, body string) { } func TestAssembleProducesAPageForEveryDataset(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } @@ -49,7 +49,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { filepath.Join("2016", "index.html"), filepath.Join("2017", "index.html"), filepath.Join("_app", "immutable", "entry.js"), - filepath.Join("db", "2016.sqlite3"), + filepath.Join("db", "2016.sqlite30"), } { if _, err := os.Stat(filepath.Join(p.Site, want)); err != nil { t.Errorf("missing from the artifact: %s", want) @@ -61,7 +61,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { // entry generator, which reads the same registry this does. If the two fall out // of step, that dataset's URL 404s — so the build stops instead. func TestMissingDatasetPageFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") if err := os.RemoveAll(filepath.Join(p.Dist, "2017")); err != nil { t.Fatal(err) } @@ -80,12 +80,12 @@ func TestMissingDatasetPageFailsTheBuild(t *testing.T) { // renders, every query 404s, and CI stays green. The row-count and size guards // cannot catch this — they only run when a database was built at all. func TestMissingDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3") // 2017 never built + p := fakeBuild(t, "2016.sqlite30") // 2017 never built err := Assemble(p, datasets) if err == nil { t.Fatal("expected an error when a database is missing") } - if !strings.Contains(err.Error(), "2017.sqlite3") { + if !strings.Contains(err.Error(), "2017.sqlite30") { t.Errorf("the error should name the missing database, got: %v", err) } } @@ -93,22 +93,25 @@ func TestMissingDatabaseFailsTheBuild(t *testing.T) { // TestEmptyDatabaseFailsTheBuild: a zero-byte file satisfies "exists" but is // not a database. func TestEmptyDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") - write(t, filepath.Join(p.Dist, "db", "2017.sqlite3"), "") + p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") + write(t, filepath.Join(p.Dist, "db", "2017.sqlite30"), "") if err := Assemble(p, datasets); err == nil { t.Fatal("expected an error for a zero-byte database") } } -// TestStrayArtifactFailsTheBuild: a journal means an interrupted run, a .db -// means the old naming, a .gz means a database the site could not read a range -// of — and each is 100+ MB. +// TestStrayArtifactFailsTheBuild: a journal means an interrupted run, a .db or +// a chunk-index-less .sqlite3 means an older naming the client no longer asks +// for, a .gz means a database the site could not read a range of — and each is +// 100+ MB. func TestStrayArtifactFailsTheBuild(t *testing.T) { for _, name := range []string{ - "2016.db", "2016.sqlite3-journal", "2016.sqlite3-wal", "2016.sqlite3-shm", "2016.sqlite3.gz", + "2016.db", "2016.sqlite3", + "2016.sqlite3-journal", "2016.sqlite3-wal", "2016.sqlite3-shm", "2016.sqlite3.gz", + "2016.sqlite30-journal", "2016.sqlite30-wal", "2016.sqlite30-shm", } { t.Run(name, func(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") write(t, filepath.Join(p.Dist, "db", name), "raw sqlite") err := Assemble(p, datasets) if err == nil { @@ -124,11 +127,12 @@ func TestStrayArtifactFailsTheBuild(t *testing.T) { // TestPublishedDatabaseIsNotMistakenForStray: the pattern must pass the one // file the site is built to serve. Getting this wrong would fail every build. func TestPublishedDatabaseIsNotMistakenForStray(t *testing.T) { - if strayArtifact.MatchString("2016.sqlite3") { + if strayArtifact.MatchString("2016.sqlite30") { t.Error("the published database must not be treated as a stray artifact") } for _, name := range []string{ - "2016.db", "x.sqlite3-journal", "x.sqlite3-wal", "x.sqlite3-shm", "x.sqlite3.gz", "x.db.gz", + "2016.db", "x.sqlite3", "x.sqlite3-journal", "x.sqlite3-wal", "x.sqlite3-shm", + "x.sqlite30-journal", "x.sqlite30-wal", "x.sqlite30-shm", "x.sqlite3.gz", "x.db.gz", } { if !strayArtifact.MatchString(name) { t.Errorf("%s should be rejected", name) @@ -147,7 +151,7 @@ func TestAssembleRejectsAMissingBuild(t *testing.T) { // TestAssembleIsIdempotent: the site directory is rebuilt from scratch, so a // previous run's leftovers cannot survive into the artifact. func TestAssembleIsIdempotent(t *testing.T) { - p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } diff --git a/assembler/internal/verify/verify.go b/assembler/internal/verify/verify.go index ff826ec..2964a63 100644 --- a/assembler/internal/verify/verify.go +++ b/assembler/internal/verify/verify.go @@ -135,15 +135,15 @@ func compareOne(id, dirA, dirB string) (Result, error) { return res, nil } -// open finds .sqlite3 in dir and returns a read-only handle. +// open finds .sqlite30 in dir and returns a read-only handle. // -// A gzipped database is still expanded to a temporary file rather than -// rejected: the two sides of a comparison are often a build from before the -// switch to range requests and one from after. +// The older names are still accepted, and a gzipped database is expanded to a +// temporary file rather than rejected: the two sides of a comparison are often +// a build from before a naming or format change and one from after. func open(dir, id string) (*sql.DB, func(), error) { noop := func() {} - for _, name := range []string{id + ".sqlite3", id + ".db"} { + for _, name := range []string{id + ".sqlite30", id + ".sqlite3", id + ".db"} { plain := filepath.Join(dir, name) if _, err := os.Stat(plain); err == nil { db, err := sql.Open(driverName, "file:"+plain+"?mode=ro") @@ -154,7 +154,7 @@ func open(dir, id string) (*sql.DB, func(), error) { gzPath := filepath.Join(dir, id+".db.gz") f, err := os.Open(gzPath) if err != nil { - return nil, noop, fmt.Errorf("no %s.sqlite3, %s.db or %s.db.gz in %s", id, id, id, dir) + return nil, noop, fmt.Errorf("no %s.sqlite30, %s.sqlite3, %s.db or %s.db.gz in %s", id, id, id, id, dir) } defer f.Close() diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index 518df28..06c2681 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -193,7 +193,7 @@ HTTP request. `CHUNK_BYTES` in `web/src/lib/sqlite.svelte.js` must match. ## Verifying a rebuild The assembler verifies itself: each database's row count must match the -figure in the table above, and each `.sqlite3` must be at least 90% of its usual +figure in the table above, and each `.sqlite30` must be at least 90% of its usual size, or the build fails rather than publishing. That guard is the reason a truncated dataset cannot reach the site with a green pipeline. @@ -207,7 +207,7 @@ go -C assembler run ./cmd/assemble db # rebuild go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db ``` -Each side is a directory of `.sqlite3`; a gzipped database from before the +Each side is a directory of `.sqlite30`; a gzipped database from before the switch to range requests is still expanded to a temporary file automatically. It exits non-zero on any mismatch, names the first differing rows and columns, and fails rather than skipping when a dataset is absent from either side — silently comparing one of two datasets is diff --git a/docs/deployment-guide.md b/docs/deployment-guide.md index 8d01024..a880c87 100644 --- a/docs/deployment-guide.md +++ b/docs/deployment-guide.md @@ -87,21 +87,22 @@ artifact — one missing line away from publishing it. - **Total artifact is about 552 MB**, inside the 1 GB site limit but with less headroom than before: a third dataset of this size would not fit. The fallback is `sql.js-httpvfs`'s chunked mode, which splits a database into parts. -- **Pages does compress the databases, and that is survivable.** `.sqlite3` is - unknown to Pages, so it is served as `application/octet-stream`, which is - marked compressible in `mime-db` and gzipped: a plain request returns - `Content-Encoding: gzip` and the compressed length. Ranged reads are not - affected, because the Fetch standard makes browsers send +- **Pages does compress the databases, and that is survivable.** The extension + is unknown to Pages, so the file is served as `application/octet-stream`, + which is marked compressible in `mime-db` and gzipped: a plain request + returns `Content-Encoding: gzip` and the compressed length. Ranged reads are + not affected, because the Fetch standard makes browsers send `Accept-Encoding: identity` on any request carrying a `Range` header. Only - the length probe breaks, and the site supplies `fileLength` itself instead of - trusting HEAD — see `web/src/lib/db-probe.js`. + the length probe breaks, so the site supplies the length itself instead of + trusting HEAD — see `web/src/lib/db-probe.js` and the chunked-mode note in + [system-architecture](./system-architecture.md). - **Verify the way a browser asks.** A bare `curl -sI` advertises no encoding and so reports success whatever the host does; it is what let this reach production. Check ranged reads instead, and check the bytes, not the headers: ```bash curl -s -r 0-15 -H 'Accept-Encoding: identity;q=1, *;q=0' \ - https://.github.io/thptqg/db/2016.sqlite3 | head -c 16 + https://.github.io/thptqg/db/2016.sqlite30 | head -c 16 # must print: SQLite format 3 ``` @@ -117,7 +118,7 @@ run rebuilds the older state. There is no data to migrate. | Blank page, 404 on assets | `paths.base` in `svelte.config.js` does not match the repo name | | `Failed to fetch database: 404` | Dataset id in `datasets.json` does not match the file in `db/` | | A route 404s | The site step did not run, or the id is missing from `datasets.json` | -| `Length of the file not known` | The host gzipped the un-ranged response, so HEAD reports the compressed size. The site supplies `fileLength` from a range probe; if this returns, that probe failed | +| `Length of the file not known` | The host gzipped the un-ranged response, so HEAD reports the compressed size. The site supplies `databaseLengthBytes` from a range probe; if this returns, the config is no longer reaching the worker in chunked mode | | Database fails to open | A ranged read did not return raw database bytes. The range check above must print `SQLite format 3` | | Every query is slow or huge | It is not using an index. `EXPLAIN QUERY PLAN` it: a `SCAN` means the browser is fetching the whole table | | Deploy fails on assembly | An uncompressed database artefact reached the output; the error names the files | diff --git a/docs/system-architecture.md b/docs/system-architecture.md index 5eab684..96cd913 100644 --- a/docs/system-architecture.md +++ b/docs/system-architecture.md @@ -17,10 +17,10 @@ through. `assembler/` sequences everything from the parser onwards. data//*.xls(x) │ ▼ parser/ (Go, one binary, one config per dataset) - .build/public/db/.sqlite3 + .build/public/db/.sqlite30 │ ▼ assembler/ — row count and size must match datasets.json - .build/public/db/.sqlite3 (uncompressed: ranges of a gzip stream + .build/public/db/.sqlite30 (uncompressed: ranges of a gzip stream │ are not ranges of the database) ▼ assembler/ → npm run build (SvelteKit static, assets = .build/public) web/dist/ @@ -37,7 +37,7 @@ data//*.xls(x) One identifier ties the whole pipeline together: ``` -data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite30 → /thptqg/2017/ ``` `datasets.json` at the repository root declares the ids once, with the row count @@ -171,6 +171,7 @@ total descending. | Concern | Choice | Rationale | | --- | --- | --- | | Storage | Static SQLite file, read by range request | No backend; the datasets are frozen, and a lookup needs a few pages of them | +| Reading mode | `serverMode: "chunked"` over a single chunk | The only mode whose config accepts the file length. In full mode the worker hardcodes it to `undefined` and falls back to a HEAD request, which Pages answers with the gzipped size. One chunk means the index is always 0, hence the published name `.sqlite30` | | Compression | None | A byte range of a gzip stream is not a byte range of the database | | WASM hosting | Bundled with the app | `sql.js-httpvfs` ships its own build; one less third-party runtime dependency | | Diacritics search | Pre-computed `ho_ten_ascii`, indexed word by word | `LOWER(REPLACE(...))` at query time defeats the index, and `LIKE '%x%'` reads the whole table | @@ -183,18 +184,21 @@ total descending. ### Considered and not taken -- **Chunked `serverMode`.** `sql.js-httpvfs` can split a database into parts so - a CDN caches each one whole. GitHub Pages serves everything with - `Cache-Control: max-age=600`, and every rebuild relays SQLite's pages so the - file changes even when the data does not — the caching that mode buys is - cancelled by the host. Worth revisiting behind a CDN with long TTLs, and it is - also the fallback if a single 300 MB file ever becomes a problem. +- **Splitting the database into several chunks.** The site uses chunked mode, + but over one chunk (see above). Real splitting would let a CDN cache each part + whole; GitHub Pages serves everything with `Cache-Control: max-age=600`, and + every rebuild relays SQLite's pages so the file changes even when the data + does not, so that caching is cancelled by the host. Worth revisiting behind a + CDN with long TTLs, and it is the fallback if a single 300 MB file ever + becomes a problem. - **`sqlite-wasm-http`.** Maintained, and built on the official SQLite WASM - rather than a 2022 fork, which is the better long-term footing. Its - differentiator — a cache shared between workers — needs COOP/COEP headers that - GitHub Pages cannot send, so here it would buy maintenance alone. Deferred - until the current path has been verified in a browser, so that a swap changes - one variable rather than two. + rather than a 2022 fork, which is the better long-term footing. It does not + help here: its worker sizes the file from a HEAD request's `Content-Length` + exactly as `sql.js-httpvfs` does, and its `Options` has no field for the + length, so on Pages it would silently take the gzipped size instead of + failing. Its shared-cache backend needs COOP/COEP, which Pages cannot send, + but it ships a fallback backend that does not — so isolation is not the + blocker, the missing length option is. - **Substring name search.** `LIKE '%x%'` cannot use an index, so it read the whole 127 MB table. `name_word` keeps search by any word of a name without it. @@ -209,8 +213,9 @@ total descending. it: the Fetch standard requires `Accept-Encoding: identity` on any request carrying a `Range` header. GitHub Pages *does* gzip the un-ranged response — `application/octet-stream` is compressible in `mime-db` — which is why the - file length is probed with a range request and passed as `fileLength` rather - than left to the library's HEAD. `db-probe.js` checks the returned bytes + file length is probed with a range request and passed as + `databaseLengthBytes` rather than left to the library's HEAD. + `db-probe.js` checks the returned bytes start with the SQLite magic, so a host that ever compresses a ranged response fails loudly instead of returning nonsense. - **`sql.js-httpvfs` is unmaintained** (0.8.12, September 2022) and ships its diff --git a/web/src/lib/datasets.js b/web/src/lib/datasets.js index 89d4440..8d74ee6 100644 --- a/web/src/lib/datasets.js +++ b/web/src/lib/datasets.js @@ -99,11 +99,38 @@ export function pathOf(dataset, base) { } /** - * Database URL, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.sqlite3". + * The chunk the whole database lives in. One chunk covers the file, so this is + * the only index the library ever asks for, and the assembler publishes the + * database under a name ending in it. + */ +const CHUNK_INDEX = "0"; + +/** + * Everything a database URL has except the chunk index, e.g. + * dbPrefixOf(d, "/thptqg") → "/thptqg/db/2017.sqlite3". + * + * sql.js-httpvfs reads the file in chunked mode and builds each URL as + * prefix + chunk index. One chunk holds the whole database, so the index is + * always 0 and the published file is ".sqlite30". + */ +export function dbPrefixOf(dataset, base) { + return `${base}/db/${dataset.id}.sqlite3`; +} + +/** + * The published database file, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.sqlite30". * * Uncompressed on purpose: the browser reads byte ranges of it, and a range of * a gzip stream is not a range of the database. */ export function dbOf(dataset, base) { - return `${base}/db/${dataset.id}.sqlite3`; + return `${dbPrefixOf(dataset, base)}${CHUNK_INDEX}`; +} + +/** + * Both forms of the database's location, for RemoteDatabase: the file to read, + * and the prefix the library appends the chunk index to. + */ +export function dbSourceOf(dataset, base) { + return { url: dbOf(dataset, base), urlPrefix: dbPrefixOf(dataset, base) }; } diff --git a/web/src/lib/datasets.test.js b/web/src/lib/datasets.test.js new file mode 100644 index 0000000..35e690e --- /dev/null +++ b/web/src/lib/datasets.test.js @@ -0,0 +1,28 @@ +import { describe, expect, it } from "vitest"; +import { DATASETS, dbOf, dbPrefixOf, dbSourceOf } from "./datasets.js"; + +const [dataset] = DATASETS; + +/** + * sql.js-httpvfs builds every request URL as urlPrefix + chunk index, and one + * chunk holds the whole database, so the index is always 0. If these two ever + * stop agreeing, the library asks for a file the assembler never published and + * every query 404s — which no other test would catch. + */ +describe("database location", () => { + it("puts the chunk index where the published file name ends", () => { + expect(dbOf(dataset, "/thptqg")).toBe(`${dbPrefixOf(dataset, "/thptqg")}0`); + }); + + it("names the file the assembler publishes", () => { + expect(dbOf(dataset, "/thptqg")).toBe(`/thptqg/db/${dataset.id}.sqlite30`); + }); + + it("hands RemoteDatabase both forms of the same location", () => { + const source = dbSourceOf(dataset, "/thptqg"); + expect(source).toEqual({ + url: dbOf(dataset, "/thptqg"), + urlPrefix: dbPrefixOf(dataset, "/thptqg"), + }); + }); +}); diff --git a/web/src/lib/db-probe.test.js b/web/src/lib/db-probe.test.js index 698425a..7a25348 100644 --- a/web/src/lib/db-probe.test.js +++ b/web/src/lib/db-probe.test.js @@ -58,23 +58,23 @@ describe("looksLikeSqlite", () => { describe("probeDatabase", () => { it("asks for the header by range and returns the total length", async () => { const fetchImpl = vi.fn(async () => respond(header())); - const total = await probeDatabase("/db/2016.sqlite3", 1024, fetchImpl); + const total = await probeDatabase("/db/2016.sqlite30", 1024, fetchImpl); expect(total).toBe(317096960); - expect(fetchImpl).toHaveBeenCalledWith("/db/2016.sqlite3", { + expect(fetchImpl).toHaveBeenCalledWith("/db/2016.sqlite30", { headers: { Range: "bytes=0-99" }, }); }); it("fails when the host ignores the range", async () => { const fetchImpl = async () => respond(header(), { status: 200 }); - await expect(probeDatabase("/db/2016.sqlite3", 1024, fetchImpl)).rejects.toThrow(/expected 206/); + await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).rejects.toThrow(/expected 206/); }); it("fails when the bytes are not a database", async () => { const gzip = new Uint8Array([0x1f, 0x8b, 0x08]); const fetchImpl = async () => respond(gzip); - await expect(probeDatabase("/db/2016.sqlite3", 1024, fetchImpl)).rejects.toThrow( + await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).rejects.toThrow( /not a SQLite header/, ); }); @@ -83,7 +83,7 @@ describe("probeDatabase", () => { const warn = vi.spyOn(console, "warn").mockImplementation(() => {}); const fetchImpl = async () => respond(header(4096)); - await expect(probeDatabase("/db/2016.sqlite3", 1024, fetchImpl)).resolves.toBe(317096960); + await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).resolves.toBe(317096960); expect(warn).toHaveBeenCalledWith(expect.stringMatching(/page size 4096/)); warn.mockRestore(); }); diff --git a/web/src/lib/sqlite.svelte.js b/web/src/lib/sqlite.svelte.js index 65868f7..c127153 100644 --- a/web/src/lib/sqlite.svelte.js +++ b/web/src/lib/sqlite.svelte.js @@ -53,8 +53,14 @@ export class RemoteDatabase { // Previous cumulative reading, so a query's own cost is a subtraction. #seen = { requests: 0, bytes: 0 }; - constructor(url, budgetBytes = SEARCH_BUDGET_BYTES) { - this.url = url; + /** + * `source` carries both forms of the same location: `url` is the file, and + * `urlPrefix` is what the library appends the chunk index to. They must + * agree — see dbOf/dbPrefixOf, which derive one from the other. + */ + constructor(source, budgetBytes = SEARCH_BUDGET_BYTES) { + this.url = source.url; + this.urlPrefix = source.urlPrefix; this.budgetBytes = budgetBytes; this.#opening = this.#open(); } @@ -62,15 +68,29 @@ export class RemoteDatabase { async #open() { const opened = performance.now(); try { - // Supplied rather than left to the library, which would size the file - // with a HEAD request. GitHub Pages compresses that response and reports - // the compressed length, which the library refuses to use. See db-probe. - const fileLength = await probeDatabase(this.url, CHUNK_BYTES); + // Chunked mode over a single chunk, which looks odd but is the only way + // to tell this library how long the file is: the worker reads + // databaseLengthBytes in chunked mode and hardcodes the length to + // undefined in full mode. Left to itself it sizes the file with a HEAD + // request, and GitHub Pages answers that with the gzipped length, which + // it then refuses to use. + // + // One chunk covers the whole database, so the chunk index is always 0 + // and every request goes to urlPrefix + "0" — the file the assembler + // publishes as .sqlite30. + const databaseLengthBytes = await probeDatabase(this.url, CHUNK_BYTES); const worker = await createDbWorker( [ { from: "inline", - config: { serverMode: "full", url: this.url, requestChunkSize: CHUNK_BYTES, fileLength }, + config: { + serverMode: "chunked", + urlPrefix: this.urlPrefix, + serverChunkSize: databaseLengthBytes, + databaseLengthBytes, + suffixLength: 1, + requestChunkSize: CHUNK_BYTES, + }, }, ], workerUrl, diff --git a/web/src/routes/[dataset]/+page.svelte b/web/src/routes/[dataset]/+page.svelte index 64d5b71..f613093 100644 --- a/web/src/routes/[dataset]/+page.svelte +++ b/web/src/routes/[dataset]/+page.svelte @@ -7,7 +7,7 @@ import ScoreTable from "$lib/components/score-table.svelte"; import SearchForm from "$lib/components/search-form.svelte"; import StudentDetail from "$lib/components/student-detail.svelte"; - import { dbOf } from "$lib/datasets"; + import { dbSourceOf } from "$lib/datasets"; import { isExamId } from "$lib/query-mode"; import { MAX_RESULTS, lookupExamId, searchByName } from "$lib/search"; import { PLAYGROUND_BUDGET_BYTES, RemoteDatabase, SEARCH_BUDGET_BYTES } from "$lib/sqlite.svelte"; @@ -34,7 +34,7 @@ // Opened in the browser only: $effect does not run while prerendering. A new // budget means a new worker, which costs only the header pages. $effect(() => { - const opened = new RemoteDatabase(dbOf(dataset, base), budget); + const opened = new RemoteDatabase(dbSourceOf(dataset, base), budget); db = opened; return () => { opened.close();