From dbc23c25c52e2230b0595c23c0bf19038fd7120a Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 14 Aug 2026 12:42:48 +0700 Subject: [PATCH] feat: read the databases over HTTP range requests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The browser downloaded 45 MB of gzipped SQLite before it could answer anything. Now sql.js-httpvfs asks for the pages a query touches and the databases ship uncompressed as .sqlite3 — a byte range of a gzip stream is not a byte range of a database. That only works if every query the site issues is index-driven, and measured against the real 2016 file, most were not: so_bao_danh = ? SEARCH via PK ~20 KB ho_ten_ascii LIKE '%x%' SCAN 127 MB ho_ten_ascii LIKE 'x%' SCAN 127 MB COUNT(*) covering index scan 20 MB ORDER BY toan DESC LIMIT 10 SCAN + temp b-tree 127 MB Prefix LIKE scans because SQLite's LIKE optimisation needs a NOCASE index; a range comparison does use the index. So the schema changed to suit the access pattern rather than the search changing to suit the schema. name_word holds one row per word of each name, WITHOUT ROWID so the table is the index, carrying ho_ten_ascii so a multi-word query is resolved inside a single b-tree. name_word_freq says which word of a query is rarest — the vocabulary is 4,397 words across 2.87M entries, so "buu loc" seeks on 287 entries rather than walking the 300,000 that "thi" would. Searching by any word of a name survives, at a few hundred KB a query. idx_ho_ten and idx_ho_ten_ascii are gone: no plan could use either. Partial indexes on toan, khtn and khxh cost 12 MB and keep the SQL presets off a full scan. The footer's candidate count now comes from datasets.json instead of COUNT(*). 2016 grows 223.5 MB to 288.6 MB, 2017 162.7 MB to 237.7 MB, and the site is 528 MB against the 1 GB GitHub Pages limit. Row counts are unchanged. The SQL tab is the one place a user can still write a query that reads the whole table, so it asks before it opens, runs under a byte budget that stops a runaway query, and shows what each query actually fetched. Verified: row counts through the assembler guards, every app query index-driven under EXPLAIN QUERY PLAN, and GitHub Pages returning 206 with a correct Content-Range. Not verified in a browser — this machine has none — and the library refuses to open a file the host compresses, so the deployed response headers need a look. --- CLAUDE.md | 9 + README.md | 5 +- assembler/internal/databases/databases.go | 87 ++------ .../internal/databases/databases_test.go | 50 +---- assembler/internal/site/site.go | 30 +-- assembler/internal/site/site_test.go | 51 ++--- assembler/internal/verify/verify.go | 23 ++- datasets.json | 11 +- docs/data-pipeline.md | 6 +- docs/deployment-guide.md | 30 +-- docs/project-overview.md | 6 +- docs/system-architecture.md | 47 +++-- parser/internal/schema/schema.go | 57 ++++- parser/internal/schema/schema_test.go | 42 +++- parser/internal/writer/writer.go | 72 ++++++- .../260814-1200-httpvfs-range-queries/plan.md | 71 +++++++ web/package-lock.json | 21 +- web/package.json | 4 +- web/src/lib/components/custom-query.svelte | 72 ++++--- web/src/lib/datasets.ts | 12 +- web/src/lib/search.ts | 93 +++++++++ web/src/lib/sqlite.svelte.ts | 152 ++++++++------ web/src/lib/types.ts | 5 + web/src/routes/[dataset]/+page.svelte | 195 ++++++++++-------- 24 files changed, 743 insertions(+), 408 deletions(-) create mode 100644 plans/260814-1200-httpvfs-range-queries/plan.md create mode 100644 web/src/lib/search.ts diff --git a/CLAUDE.md b/CLAUDE.md index 01a8380..a6b5962 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -58,6 +58,15 @@ hashes every real input file. That is the point of it; do not skip it. used — a fallback would break the `?q=` deep links. - **`dbSizeMb` in `datasets.json` is a build guard, not just a label.** The assembler refuses to publish an artifact that falls below a ratio of it. +- **The databases ship uncompressed, as `.sqlite3`.** The browser reads + byte ranges of them, and a range of a gzip stream is not a range of the + database. The host must not apply `Content-Encoding` either — check with + `curl -sI` after a deploy. +- **Every query the site runs must be index-driven.** Over range requests an + unindexed query fetches the whole table. Hence no index on `ho_ten` (nothing + can use one), `name_word` for name search, partial indexes for the score + presets, and the footer count read from `datasets.json` instead of + `COUNT(*)`. ## Conventions diff --git a/README.md b/README.md index afcfd9f..f578951 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,8 @@ # thptqg Tra cứu điểm thi THPT Quốc gia — exam-score lookup for Vietnam's national high -school graduation exam. Client-side SQL (sql.js) over a SQLite database built +school graduation exam. Client-side SQL over a SQLite database read in place by +HTTP range request, built from the published `.xls`/`.xlsx` score files by the Go `parser` module. Where those files come from: [data pipeline](./docs/data-pipeline.md#sources). @@ -41,7 +42,7 @@ app both read it and neither needs a dependency to do so; presentation stays in The dataset id is one identifier end to end: ``` -data/2017/ → parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ ``` ## Build diff --git a/assembler/internal/databases/databases.go b/assembler/internal/databases/databases.go index 0bf077f..c294c46 100644 --- a/assembler/internal/databases/databases.go +++ b/assembler/internal/databases/databases.go @@ -1,5 +1,4 @@ -// Package databases builds, verifies and compresses one SQLite file per -// dataset. +// Package databases builds and verifies one SQLite file per dataset. // // VERIFICATION IS THE POINT OF THIS PACKAGE, not an extra. // @@ -11,13 +10,15 @@ // // The guards below close that: a build whose row count does not match the // registry, or whose artifact is implausibly small, fails the pipeline. +// +// The databases ship uncompressed. The browser reads them a page at a time over +// HTTP range requests, and a range of a gzip stream is not a range of the +// database. package databases import ( - "compress/gzip" "database/sql" "fmt" - "io" "os" "os/exec" "path/filepath" @@ -30,10 +31,15 @@ import ( // driverName is modernc.org/sqlite's registered name. const driverName = "sqlite" -// minSizeRatio: a gzipped database far below its usual size means a truncated -// build, even if the row count somehow passed. +// minSizeRatio: a database far below its usual size means a truncated build, +// even if the row count somehow passed. const minSizeRatio = 0.9 +// Extension is the published suffix. Not ".db": the sql.js-httpvfs ecosystem +// uses ".sqlite3", and keeping ".db" free lets the site assembly treat any +// stray .db or SQLite journal in the output as the leftover it is. +const Extension = ".sqlite3" + // Paths locates the pieces this package needs. type Paths struct { // Root is the repository root. @@ -75,7 +81,7 @@ func Build(p Paths, bin string, d registry.Dataset) error { if err := os.MkdirAll(p.OutDir, 0o755); err != nil { return err } - db := filepath.Join(p.OutDir, d.ID+".db") + db := filepath.Join(p.OutDir, d.ID+Extension) cmd := exec.Command(bin, "build", @@ -99,12 +105,11 @@ func Build(p Paths, bin string, d registry.Dataset) error { } fmt.Printf(" ✓ %s: %d rows (matches expected)\n", d.ID, rows) - gz, size, err := compress(db) + st, err := os.Stat(db) if err != nil { return fmt.Errorf("%s: %w", d.ID, err) } - - sizeMb := float64(size) / 1024 / 1024 + sizeMb := float64(st.Size()) / 1024 / 1024 if min := d.DbSizeMb * minSizeRatio; sizeMb < min { return fmt.Errorf( "%s: %.1f MB is below %.1f MB (%.0f%% of the expected %.0f MB)\n"+ @@ -112,7 +117,7 @@ func Build(p Paths, bin string, d registry.Dataset) error { d.ID, sizeMb, min, minSizeRatio*100, d.DbSizeMb) } - fmt.Printf(" → %s (%.1f MB)\n\n", filepath.Base(gz), sizeMb) + fmt.Printf(" → %s (%.1f MB)\n\n", filepath.Base(db), sizeMb) return nil } @@ -131,61 +136,8 @@ func countRows(path string) (int64, error) { return n, nil } -// compress gzips path to path+".gz" and removes the original, returning the -// compressed path and its size. -// -// The source is deleted only after the compressed file is closed successfully, -// so a failure part-way through leaves the database rather than losing it. -func compress(path string) (string, int64, error) { - in, err := os.Open(path) - if err != nil { - return "", 0, err - } - defer in.Close() - - gzPath := path + ".gz" - out, err := os.Create(gzPath) - if err != nil { - return "", 0, err - } - - zw, err := gzip.NewWriterLevel(out, gzip.BestCompression) - if err != nil { - out.Close() - return "", 0, err - } - if _, err := io.Copy(zw, in); err != nil { - zw.Close() - out.Close() - os.Remove(gzPath) - return "", 0, err - } - if err := zw.Close(); err != nil { - out.Close() - os.Remove(gzPath) - return "", 0, err - } - if err := out.Close(); err != nil { - os.Remove(gzPath) - return "", 0, err - } - - if err := in.Close(); err != nil { - return "", 0, err - } - if err := os.Remove(path); err != nil { - return "", 0, fmt.Errorf("removing the uncompressed database: %w", err) - } - - st, err := os.Stat(gzPath) - if err != nil { - return "", 0, err - } - return gzPath, st.Size(), nil -} - // Clean removes staged artifacts for datasets that are no longer in the -// registry. Without this a removed dataset's .db.gz lingers in the staging +// registry. Without this a removed dataset's file lingers in the staging // directory, and the site assembly copies that directory wholesale — so the // dead database would be published again. func Clean(p Paths, keep []registry.Dataset) error { @@ -197,10 +149,9 @@ func Clean(p Paths, keep []registry.Dataset) error { return err } - wanted := make(map[string]bool, len(keep)*2) + wanted := make(map[string]bool, len(keep)) for _, d := range keep { - wanted[d.ID+".db"] = true - wanted[d.ID+".db.gz"] = true + wanted[d.ID+Extension] = true } for _, e := range entries { diff --git a/assembler/internal/databases/databases_test.go b/assembler/internal/databases/databases_test.go index 05a9c27..b14aada 100644 --- a/assembler/internal/databases/databases_test.go +++ b/assembler/internal/databases/databases_test.go @@ -1,8 +1,6 @@ package databases import ( - "compress/gzip" - "io" "os" "path/filepath" "slices" @@ -11,52 +9,12 @@ import ( "github.com/tiennm99/thptqg/assembler/internal/registry" ) -// TestCompressRoundTripsAndRemovesTheSource: only the .gz may survive, so that -// shipping a 100+ MB uncompressed database is structurally impossible rather -// than left to a cleanup step. -func TestCompressRoundTripsAndRemovesTheSource(t *testing.T) { - dir := t.TempDir() - src := filepath.Join(dir, "2016.db") - body := []byte("pretend this is a SQLite file") - if err := os.WriteFile(src, body, 0o644); err != nil { - t.Fatal(err) - } - - gzPath, size, err := compress(src) - if err != nil { - t.Fatal(err) - } - if gzPath != src+".gz" || size <= 0 { - t.Fatalf("gzPath=%q size=%d", gzPath, size) - } - if _, err := os.Stat(src); !os.IsNotExist(err) { - t.Error("the uncompressed database must not survive") - } - - f, err := os.Open(gzPath) - if err != nil { - t.Fatal(err) - } - defer f.Close() - zr, err := gzip.NewReader(f) - if err != nil { - t.Fatal(err) - } - got, err := io.ReadAll(zr) - if err != nil { - t.Fatal(err) - } - if string(got) != string(body) { - t.Errorf("round-trip gave %q", got) - } -} - func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { dir := t.TempDir() for _, name := range []string{ - "2016.db.gz", "2017.db.gz", - "2017-old.db.gz", // dropped from the registry - "2017-old2.db.gz", // dropped from the registry + "2016.sqlite3", "2017.sqlite3", + "2017-old.sqlite3", // dropped from the registry + "2017-old2.sqlite3", // dropped from the registry "2016.db-journal", // interrupted run } { if err := os.WriteFile(filepath.Join(dir, name), []byte("x"), 0o644); err != nil { @@ -80,7 +38,7 @@ func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { } slices.Sort(left) - want := []string{"2016.db.gz", "2017.db.gz"} + want := []string{"2016.sqlite3", "2017.sqlite3"} if !slices.Equal(left, want) { t.Errorf("left %v, want %v", left, want) } diff --git a/assembler/internal/site/site.go b/assembler/internal/site/site.go index 4428df2..5018b2c 100644 --- a/assembler/internal/site/site.go +++ b/assembler/internal/site/site.go @@ -20,6 +20,7 @@ import ( "regexp" "strings" + "github.com/tiennm99/thptqg/assembler/internal/databases" "github.com/tiennm99/thptqg/assembler/internal/registry" ) @@ -89,7 +90,7 @@ func Assemble(p Paths, datasets []registry.Dataset) error { if err := checkDatabasesPresent(p.Site, datasets); err != nil { return err } - if err := checkNoRawDatabases(p.Site); err != nil { + if err := checkNoStrayArtifacts(p.Site); err != nil { return err } @@ -131,10 +132,10 @@ func checkDatasetPages(siteDir string, datasets []registry.Dataset) error { func checkDatabasesPresent(siteDir string, datasets []registry.Dataset) error { var missing []string for _, d := range datasets { - gz := filepath.Join(siteDir, "db", d.ID+".db.gz") - st, err := os.Stat(gz) + file := filepath.Join(siteDir, "db", d.ID+databases.Extension) + st, err := os.Stat(file) if err != nil || st.Size() == 0 { - missing = append(missing, d.ID+".db.gz") + missing = append(missing, d.ID+databases.Extension) } } if len(missing) > 0 { @@ -146,22 +147,23 @@ func checkDatabasesPresent(siteDir string, datasets []registry.Dataset) error { return nil } -// rawDatabase matches an uncompressed SQLite artifact, including the temporary -// files SQLite leaves mid-build. -var rawDatabase = regexp.MustCompile(`\.db(-journal|-wal|-shm)?$`) +// strayArtifact matches what must never reach the output: a SQLite journal from +// an interrupted run, a database under the old .db name, or a gzipped database +// from before the switch to range requests. +var strayArtifact = regexp.MustCompile(`(\.db|\.sqlite3)(-journal|-wal|-shm)$|\.db$|\.gz$`) -// checkNoRawDatabases rejects an uncompressed database that reached the output. +// checkNoStrayArtifacts rejects leftovers that would be published. // -// The build gzips without keeping the source, so none should exist — but the -// staging directory is copied wholesale, and a leftover from an interrupted run -// would go straight through. A raw database is 100+ MB. -func checkNoRawDatabases(siteDir string) error { +// The staging directory is copied wholesale, so anything an interrupted run left +// behind goes straight through — and each of these is 100+ MB. A gzipped +// database would also be unreadable to the site, which reads byte ranges. +func checkNoStrayArtifacts(siteDir string) error { var stray []string err := filepath.WalkDir(siteDir, func(path string, d os.DirEntry, err error) error { if err != nil { return err } - if !d.IsDir() && rawDatabase.MatchString(d.Name()) { + if !d.IsDir() && strayArtifact.MatchString(d.Name()) { stray = append(stray, path) } return nil @@ -171,7 +173,7 @@ func checkNoRawDatabases(siteDir string) error { } if len(stray) > 0 { var b strings.Builder - b.WriteString("uncompressed database artefact(s) found in the site output:\n") + b.WriteString("stray database artefact(s) found in the site output:\n") for _, f := range stray { st, _ := os.Stat(f) fmt.Fprintf(&b, " %s (%.1f MB)\n", f, float64(st.Size())/1048576) diff --git a/assembler/internal/site/site_test.go b/assembler/internal/site/site_test.go index 121a49f..dae2b8e 100644 --- a/assembler/internal/site/site_test.go +++ b/assembler/internal/site/site_test.go @@ -23,7 +23,7 @@ func fakeBuild(t *testing.T, dbs ...string) Paths { } write(t, filepath.Join(dist, "_app", "immutable", "entry.js"), "console.log(1)") for _, name := range dbs { - write(t, filepath.Join(dist, "db", name), "gzipped-bytes") + write(t, filepath.Join(dist, "db", name), "sqlite-bytes") } return Paths{Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")} } @@ -39,7 +39,7 @@ func write(t *testing.T, path, body string) { } func TestAssembleProducesAPageForEveryDataset(t *testing.T) { - p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } @@ -49,7 +49,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { filepath.Join("2016", "index.html"), filepath.Join("2017", "index.html"), filepath.Join("_app", "immutable", "entry.js"), - filepath.Join("db", "2016.db.gz"), + filepath.Join("db", "2016.sqlite3"), } { if _, err := os.Stat(filepath.Join(p.Site, want)); err != nil { t.Errorf("missing from the artifact: %s", want) @@ -61,7 +61,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { // entry generator, which reads the same registry this does. If the two fall out // of step, that dataset's URL 404s — so the build stops instead. func TestMissingDatasetPageFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := os.RemoveAll(filepath.Join(p.Dist, "2017")); err != nil { t.Fatal(err) } @@ -80,12 +80,12 @@ func TestMissingDatasetPageFailsTheBuild(t *testing.T) { // renders, every query 404s, and CI stays green. The row-count and size guards // cannot catch this — they only run when a database was built at all. func TestMissingDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.db.gz") // 2017 never built + p := fakeBuild(t, "2016.sqlite3") // 2017 never built err := Assemble(p, datasets) if err == nil { t.Fatal("expected an error when a database is missing") } - if !strings.Contains(err.Error(), "2017.db.gz") { + if !strings.Contains(err.Error(), "2017.sqlite3") { t.Errorf("the error should name the missing database, got: %v", err) } } @@ -93,40 +93,45 @@ func TestMissingDatabaseFailsTheBuild(t *testing.T) { // TestEmptyDatabaseFailsTheBuild: a zero-byte file satisfies "exists" but is // not a database. func TestEmptyDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.db.gz", "2017.db.gz") - write(t, filepath.Join(p.Dist, "db", "2017.db.gz"), "") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + write(t, filepath.Join(p.Dist, "db", "2017.sqlite3"), "") if err := Assemble(p, datasets); err == nil { t.Fatal("expected an error for a zero-byte database") } } -// TestRawDatabaseFailsTheBuild: the compression step deletes its source, so a -// raw .db here means an interrupted run left one behind — and it is 100+ MB. -func TestRawDatabaseFailsTheBuild(t *testing.T) { - for _, name := range []string{"2016.db", "2016.db-journal", "2016.db-wal", "2016.db-shm"} { +// TestStrayArtifactFailsTheBuild: a journal means an interrupted run, a .db +// means the old naming, a .gz means a database the site could not read a range +// of — and each is 100+ MB. +func TestStrayArtifactFailsTheBuild(t *testing.T) { + for _, name := range []string{ + "2016.db", "2016.sqlite3-journal", "2016.sqlite3-wal", "2016.sqlite3-shm", "2016.sqlite3.gz", + } { t.Run(name, func(t *testing.T) { - p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") write(t, filepath.Join(p.Dist, "db", name), "raw sqlite") err := Assemble(p, datasets) if err == nil { t.Fatalf("expected an error for %s", name) } - if !strings.Contains(err.Error(), "uncompressed") { + if !strings.Contains(err.Error(), "stray") { t.Errorf("unexpected error: %v", err) } }) } } -// TestGzipIsNotMistakenForRaw: the reject pattern is anchored, so a .db.gz must -// pass. Getting this wrong would fail every build. -func TestGzipIsNotMistakenForRaw(t *testing.T) { - if rawDatabase.MatchString("2016.db.gz") { - t.Error("a .db.gz must not be treated as an uncompressed database") +// TestPublishedDatabaseIsNotMistakenForStray: the pattern must pass the one +// file the site is built to serve. Getting this wrong would fail every build. +func TestPublishedDatabaseIsNotMistakenForStray(t *testing.T) { + if strayArtifact.MatchString("2016.sqlite3") { + t.Error("the published database must not be treated as a stray artifact") } - for _, name := range []string{"2016.db", "x.db-journal", "x.db-wal", "x.db-shm"} { - if !rawDatabase.MatchString(name) { - t.Errorf("%s should be treated as an uncompressed artifact", name) + for _, name := range []string{ + "2016.db", "x.sqlite3-journal", "x.sqlite3-wal", "x.sqlite3-shm", "x.sqlite3.gz", "x.db.gz", + } { + if !strayArtifact.MatchString(name) { + t.Errorf("%s should be rejected", name) } } } @@ -142,7 +147,7 @@ func TestAssembleRejectsAMissingBuild(t *testing.T) { // TestAssembleIsIdempotent: the site directory is rebuilt from scratch, so a // previous run's leftovers cannot survive into the artifact. func TestAssembleIsIdempotent(t *testing.T) { - p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } diff --git a/assembler/internal/verify/verify.go b/assembler/internal/verify/verify.go index 3674889..ff826ec 100644 --- a/assembler/internal/verify/verify.go +++ b/assembler/internal/verify/verify.go @@ -44,8 +44,8 @@ type Result struct { // OK reports whether the two databases are logically identical. func (r Result) OK() bool { return len(r.Problems) == 0 } -// Compare checks every dataset in the registry, reading /.db.gz (or -// .db) from each side. +// Compare checks every dataset in the registry, reading /.sqlite3 +// from each side. func Compare(datasets []registry.Dataset, dirA, dirB string) ([]Result, error) { out := make([]Result, 0, len(datasets)) for _, d := range datasets { @@ -135,21 +135,26 @@ func compareOne(id, dirA, dirB string) (Result, error) { return res, nil } -// open finds .db.gz or .db in dir and returns a handle. A compressed -// database is expanded to a temporary file, since SQLite needs to seek. +// open finds .sqlite3 in dir and returns a read-only handle. +// +// A gzipped database is still expanded to a temporary file rather than +// rejected: the two sides of a comparison are often a build from before the +// switch to range requests and one from after. func open(dir, id string) (*sql.DB, func(), error) { noop := func() {} - plain := filepath.Join(dir, id+".db") - if _, err := os.Stat(plain); err == nil { - db, err := sql.Open(driverName, "file:"+plain+"?mode=ro") - return db, noop, err + for _, name := range []string{id + ".sqlite3", id + ".db"} { + plain := filepath.Join(dir, name) + if _, err := os.Stat(plain); err == nil { + db, err := sql.Open(driverName, "file:"+plain+"?mode=ro") + return db, noop, err + } } gzPath := filepath.Join(dir, id+".db.gz") f, err := os.Open(gzPath) if err != nil { - return nil, noop, fmt.Errorf("no %s.db or %s.db.gz in %s", id, id, dir) + return nil, noop, fmt.Errorf("no %s.sqlite3, %s.db or %s.db.gz in %s", id, id, id, dir) } defer f.Close() diff --git a/datasets.json b/datasets.json index ecd83e2..fef7161 100644 --- a/datasets.json +++ b/datasets.json @@ -8,9 +8,10 @@ " inputs are frozen exam results, so a deviation of even one row", " means something changed unintentionally and the assembler", " refuses to publish.", - " dbSizeMb usual size of the gzipped database. Required and non-zero:", - " the assembler rejects a build that comes out far smaller, and", - " the web app shows it while the download runs.", + " dbSizeMb usual size of the published database. Required and non-zero:", + " the assembler rejects a build that comes out far smaller. The", + " file is served uncompressed and read a page at a time over", + " HTTP range requests, so nothing downloads it whole.", "", "Presentation (titles, labels, SQL presets) lives in web/src/datasets.js keyed", "by id; that file throws at load if the two lists disagree." @@ -19,12 +20,12 @@ { "id": "2016", "expectedRows": 877460, - "dbSizeMb": 45 + "dbSizeMb": 289 }, { "id": "2017", "expectedRows": 861068, - "dbSizeMb": 48 + "dbSizeMb": 238 } ] } diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index 8546af6..d199eda 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -189,7 +189,7 @@ silently drops 13,720 students** (Hanoi +7,275, HCM +6,445). That is what ## Verifying a rebuild The assembler verifies itself: each database's row count must match the -figure in the table above, and each `.db.gz` must be at least 90% of its usual +figure in the table above, and each `.sqlite3` must be at least 90% of its usual size, or the build fails rather than publishing. That guard is the reason a truncated dataset cannot reach the site with a green pipeline. @@ -203,8 +203,8 @@ go -C assembler run ./cmd/assemble db # rebuild go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db ``` -Each side is a directory of `.db.gz` (or `.db`); compressed databases are -expanded to a temporary file automatically. It exits non-zero on any mismatch, +Each side is a directory of `.sqlite3`; a gzipped database from before the +switch to range requests is still expanded to a temporary file automatically. It exits non-zero on any mismatch, names the first differing rows and columns, and fails rather than skipping when a dataset is absent from either side — silently comparing one of two datasets is how a gate passes without proving anything. diff --git a/docs/deployment-guide.md b/docs/deployment-guide.md index 171baa4..0cf3049 100644 --- a/docs/deployment-guide.md +++ b/docs/deployment-guide.md @@ -75,17 +75,22 @@ artifact — one missing line away from publishing it. ## Notes -- **The gzipped database is not cacheable across deploys.** Every rebuild - produces a different `.db.gz`, because SQLite does not lay pages out - deterministically. First-visit users pay the full download; later visits hit - browser cache until the next deploy. -- **GitHub Pages caps individual files at 100 MB.** The largest gzipped database - is about 48 MB. Uncompressed they run 135–234 MB and would not fit — which is - why the browser decompresses via `DecompressionStream`. -- **No server-side compression is assumed.** The app fetches the `.gz` bytes - directly rather than relying on `Content-Encoding: gzip`; Pages does not - reliably compress arbitrary paths on the fly. -- **Total artifact is about 93 MB**, well inside the 1 GB site limit. +- **The database is not cacheable across deploys.** Every rebuild lays SQLite + pages out differently, so the file changes even when the data does not. Only + the pages a query touches are fetched, so this costs far less than it used + to, but a deploy does invalidate what a returning visitor had cached. +- **The 100 MB file limit is a Git limit, not a Pages one.** It applies to + files committed to a repository; the databases are built in CI and uploaded + as a Pages artifact, and the documented Pages limits are a 1 GB published + site and 100 GB/month of bandwidth, with no per-file figure. The two + databases are 289 MB and 238 MB. +- **Total artifact is about 528 MB**, inside the 1 GB site limit but with less + headroom than before: a third dataset of this size would not fit. The fallback + is `sql.js-httpvfs`'s chunked mode, which splits a database into parts. +- **The server must not compress the databases.** Ranges of a compressed body + address the wrong bytes, and the library refuses to open a file whose HEAD + carries a `Content-Encoding`. `.sqlite3` is an unknown type to Pages, so it is + served as `application/octet-stream` and left alone — verify after a deploy. ## Rollback @@ -99,6 +104,7 @@ run rebuilds the older state. There is no data to migrate. | Blank page, 404 on assets | `paths.base` in `svelte.config.js` does not match the repo name | | `Failed to fetch database: 404` | Dataset id in `datasets.json` does not match the file in `db/` | | A route 404s | The site step did not run, or the id is missing from `datasets.json` | -| WASM fails to load | `sql.js.org` unreachable — self-host `sql-wasm.wasm` and update `SQL_WASM_URL` in `lib/sqlite.svelte.ts` | +| Database fails to open | The host compressed it. `curl -sI …/db/.sqlite3` must show no `content-encoding`; ranges of a compressed body are unusable | +| Every query is slow or huge | It is not using an index. `EXPLAIN QUERY PLAN` it: a `SCAN` means the browser is fetching the whole table | | Deploy fails on assembly | An uncompressed database artefact reached the output; the error names the files | | Missing rows after a data update | Unknown Excel header — check the per-file row counts the parser prints | diff --git a/docs/project-overview.md b/docs/project-overview.md index d2f2951..90bc5a1 100644 --- a/docs/project-overview.md +++ b/docs/project-overview.md @@ -21,10 +21,10 @@ running entirely in the browser and hosted for free on GitHub Pages. Covers the ## Constraints -- **Zero backend.** The full database (44–48 MB gzipped per dataset) is - downloaded to the browser and queried in-process. +- **Zero backend.** The database (238–289 MB per dataset) stays on the server + and the browser reads the pages a query touches over HTTP range requests. - **Read-only.** `INSERT`/`UPDATE`/`DELETE` are rejected, so nobody is misled - into thinking edits persist. `sql.js` is in-memory anyway. + into thinking edits persist. The file is fetched, never written. - **Row caps.** 100 rows for lookups, 1000 for custom SQL, to prevent browser hangs. - **Vietnamese-first UI.** App labels and data are Vietnamese; documentation is diff --git a/docs/system-architecture.md b/docs/system-architecture.md index 544c3d6..d57b38d 100644 --- a/docs/system-architecture.md +++ b/docs/system-architecture.md @@ -1,7 +1,9 @@ # System Architecture -Static site, no backend. The browser downloads a compressed SQLite file at boot -and every query runs locally via `sql.js` (SQLite compiled to WebAssembly). +Static site, no backend. The SQLite file stays on the server and the browser +reads the pages a query touches over HTTP range requests, via `sql.js-httpvfs` +(SQLite compiled to WebAssembly behind a virtual file system). A lookup costs a +few hundred KB; nothing downloads the database. One frontend, one parser, one schema, two datasets. @@ -15,11 +17,11 @@ through. `assembler/` sequences everything from the parser onwards. data//*.xls(x) │ ▼ parser/ (Go, one binary, one config per dataset) - .build/public/db/.db - │ - ▼ assembler/ — row count must match datasets.json, then gzip - .build/public/db/.db.gz (the raw .db does not survive) + .build/public/db/.sqlite3 │ + ▼ assembler/ — row count and size must match datasets.json + .build/public/db/.sqlite3 (uncompressed: ranges of a gzip stream + │ are not ranges of the database) ▼ assembler/ → npm run build (SvelteKit static, assets = .build/public) web/dist/ │ @@ -27,7 +29,7 @@ data//*.xls(x) _site/ → GitHub Pages │ ▼ browser - sql.js (WASM) opens the .db → queries run client-side + sql.js-httpvfs asks for pages → HTTP range requests → results client-side ``` ## The dataset id @@ -35,7 +37,7 @@ data//*.xls(x) One identifier ties the whole pipeline together: ``` -data/2017/ → parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ ``` `datasets.json` at the repository root declares the ids once, with the row count @@ -44,7 +46,7 @@ because the assembler is a Go program and the web app is not, and JSON is the only format both parse without a dependency. Presentation — titles, labels, search examples, SQL presets — stays in -`web/src/datasets.js`, keyed by id. That file cross-checks the two: a registry +`web/src/lib/datasets.ts`, keyed by id. That file cross-checks the two: a registry entry with no content, or content for a dataset that was never built, throws at module load rather than rendering a page with no title or a link to a database that does not exist. @@ -168,10 +170,11 @@ total descending. | Concern | Choice | Rationale | | --- | --- | --- | -| Storage | Static SQLite file | No backend; the datasets are frozen | -| Compression | gzip in CI, `DecompressionStream` in the browser | Native API, no extra library | -| WASM hosting | `sql.js.org` CDN | Smaller self-hosted artifact | -| Diacritics search | Pre-computed `ho_ten_ascii` | `LOWER(REPLACE(...))` at query time defeats the index | +| Storage | Static SQLite file, read by range request | No backend; the datasets are frozen, and a lookup needs a few pages of them | +| Compression | None | A byte range of a gzip stream is not a byte range of the database | +| WASM hosting | Bundled with the app | `sql.js-httpvfs` ships its own build; one less third-party runtime dependency | +| Diacritics search | Pre-computed `ho_ten_ascii`, indexed word by word | `LOWER(REPLACE(...))` at query time defeats the index, and `LIKE '%x%'` reads the whole table | +| Row count in the footer | Read from `datasets.json` | `COUNT(*)` scans an index — 20 MB over range requests | | SQL safety | Leading-keyword allowlist | `sql.js` is in-memory so writes cannot persist; the allowlist prevents confusion | | Row caps | 100 (lookup), 1000 (SQL) | Keeps DOM render sizes reasonable | | Routing | SvelteKit file routes, prerendered | Each dataset gets a real HTML file with its own title | @@ -179,12 +182,16 @@ total descending. ## Risks and limitations -- **Database size.** 45–48 MB gzipped per dataset; slow links wait, mitigated by - a progress bar. -- **Browser memory.** The full database lives in RAM; older mobile devices may - run out. -- **`sql.js.org` dependency.** If that CDN is unreachable, the WASM fails to - load. Self-hosting `sql-wasm.wasm` and updating `SQL_WASM_URL` in - `web/src/lib/sqlite.svelte.ts` is the fix. +- **Unindexed queries are expensive.** The SQL tab can express a query that + walks the table, which over range requests means fetching 100+ MB. A byte + budget stops one before it gets that far, and the tab warns before it opens. +- **`Content-Encoding` breaks everything.** If the host ever compresses + `.sqlite3` on the wire, ranges address compressed bytes and + `sql.js-httpvfs` refuses to open the file. Verify after a deploy: + `curl -sI …/db/2016.sqlite3` must show no `content-encoding`. +- **`sql.js-httpvfs` is unmaintained** (0.8.12, September 2022) and ships its + own SQLite WASM. `sqlite-wasm-http`, on the official build, is the fallback. +- **Hosted size.** 528 MB for both datasets against the 1 GB GitHub Pages + limit; a third dataset of this size would not fit. - **Excel format drift.** A new source file with an unseen header layout needs a new branch in `parser/internal/ingest/detect2016.go` or a new config. diff --git a/parser/internal/schema/schema.go b/parser/internal/schema/schema.go index 95ed9bc..5ad429d 100644 --- a/parser/internal/schema/schema.go +++ b/parser/internal/schema/schema.go @@ -18,9 +18,17 @@ import "regexp" // DDL is executed verbatim after the output database is (re)created. // -// idx_ten_cum_thi is partial, so it holds zero entries on the 2017 dataset — -// where the column is always NULL — while staying useful for the 2016 -// cluster-grouping queries. Partial indexes are SQLite-specific. +// Every index here is chosen for a database read over HTTP range requests, +// where an unindexed query downloads the table. The rules that follow from +// that: +// +// - No index on ho_ten or ho_ten_ascii. Neither substring nor prefix LIKE can +// use one (SQLite's LIKE optimisation needs a NOCASE index or +// case_sensitive_like), so both scanned the whole table. name_word replaces +// them. +// - idx_ten_cum_thi is partial, so it holds zero entries on the 2017 dataset +// — where the column is always NULL — while serving the 2016 cluster +// grouping. Partial indexes are SQLite-specific. // // This text is frozen: it decides the shape of every database the parser // produces. TestDDLIsFrozen holds an independent copy so any edit has to be @@ -50,9 +58,48 @@ CREATE TABLE student ( tieng_nhat REAL, tieng_trung REAL ); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL; + +CREATE TABLE name_word ( + word TEXT NOT NULL, + so_bao_danh TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + PRIMARY KEY (word, so_bao_danh) +) WITHOUT ROWID; + +CREATE TABLE name_word_freq ( + word TEXT PRIMARY KEY, + n INTEGER NOT NULL +) WITHOUT ROWID; +` + +// PostLoadSQL runs once the student rows are in, before VACUUM. +// +// The frequency table is what lets the site pick which word of a query to seek +// on: the vocabulary is about 4,400 words and the rarest word of a real query +// matches a few hundred rows, so seeking on it and filtering the rest inside +// name_word keeps a search to a few hundred kilobytes. +// +// The three score indexes are partial for the same reason idx_ten_cum_thi is: +// each covers only the exam year that has the column, and each costs about +// 4 MB. They exist so the SQL presets that rank by these columns seek instead +// of scanning 127 MB. +const PostLoadSQL = ` +INSERT INTO name_word_freq (word, n) + SELECT word, COUNT(*) FROM name_word GROUP BY word; +CREATE INDEX idx_toan ON student(toan) WHERE toan IS NOT NULL; +CREATE INDEX idx_khtn ON student(khtn) WHERE khtn IS NOT NULL; +CREATE INDEX idx_khxh ON student(khxh) WHERE khxh IS NOT NULL; +` + +// NameWordInsertSQL adds one row per distinct word of a candidate's ASCII name. +// +// ho_ten_ascii is carried along deliberately: a query with several words seeks +// on the rarest one and filters the others against this copy, so the whole +// match happens inside one b-tree and only the surviving rows are read from +// student. +const NameWordInsertSQL = ` +INSERT OR IGNORE INTO name_word (word, so_bao_danh, ho_ten_ascii) VALUES (?, ?, ?) ` // IdentityFields are the identity columns, in INSERT parameter order. diff --git a/parser/internal/schema/schema_test.go b/parser/internal/schema/schema_test.go index d54f72b..0cbce5c 100644 --- a/parser/internal/schema/schema_test.go +++ b/parser/internal/schema/schema_test.go @@ -109,15 +109,53 @@ CREATE TABLE student ( tieng_nhat REAL, tieng_trung REAL ); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL; + +CREATE TABLE name_word ( + word TEXT NOT NULL, + so_bao_danh TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + PRIMARY KEY (word, so_bao_danh) +) WITHOUT ROWID; + +CREATE TABLE name_word_freq ( + word TEXT PRIMARY KEY, + n INTEGER NOT NULL +) WITHOUT ROWID; ` if DDL != want { t.Errorf("DDL changed\n--- got ---\n%s\n--- want ---\n%s", DDL, want) } } +// TestNoIndexOnNameColumns: the databases are read over HTTP range requests, so +// an index that no query can use is dead weight in a file the browser pages +// through. Neither substring nor prefix LIKE can use one on these columns — +// name_word is what serves name search. +func TestNoIndexOnNameColumns(t *testing.T) { + for _, dead := range []string{"idx_ho_ten ", "idx_ho_ten_ascii"} { + if strings.Contains(DDL, dead) { + t.Errorf("DDL creates %q, which no query plan can use", dead) + } + } +} + +// TestPostLoadBuildsTheSearchTables: the frequency table is what lets a search +// pick which word to seek on, and the score indexes are what keep the SQL +// presets off a full scan. +func TestPostLoadBuildsTheSearchTables(t *testing.T) { + for _, want := range []string{ + "INSERT INTO name_word_freq", + "CREATE INDEX idx_toan", + "CREATE INDEX idx_khtn", + "CREATE INDEX idx_khxh", + } { + if !strings.Contains(PostLoadSQL, want) { + t.Errorf("PostLoadSQL is missing %q", want) + } + } +} + // TestScorePatternsMatchScores exercises each pattern against the shape the // DIEM_THI cell actually carries, including the wide runs of spaces seen in the // real corpus. diff --git a/parser/internal/writer/writer.go b/parser/internal/writer/writer.go index 339918e..9f7bc0a 100644 --- a/parser/internal/writer/writer.go +++ b/parser/internal/writer/writer.go @@ -14,6 +14,7 @@ import ( "fmt" "os" "path/filepath" + "strings" "github.com/tiennm99/thptqg/parser/internal/schema" "github.com/tiennm99/thptqg/parser/internal/sqlitedb" @@ -110,10 +111,77 @@ type Stats struct { Errors uint64 } -// Finish runs VACUUM and prints the stats block. +// BuildNameIndex fills name_word from the student rows, one entry per distinct +// word of each ASCII name. // -// VACUUM must run AFTER the transaction commits — SQLite refuses it inside one. +// A second pass rather than a write alongside each insert: a repeated exam +// number replaces its earlier row, and the words of the row it replaced would +// otherwise stay behind pointing at a name that is no longer there. +func BuildNameIndex(db *sql.DB) error { + rows, err := db.Query("SELECT so_bao_danh, ho_ten_ascii FROM student") + if err != nil { + return fmt.Errorf("read names: %w", err) + } + defer rows.Close() + + tx, err := db.Begin() + if err != nil { + return fmt.Errorf("begin name index: %w", err) + } + stmt, err := tx.Prepare(schema.NameWordInsertSQL) + if err != nil { + tx.Rollback() + return fmt.Errorf("prepare name index: %w", err) + } + + var words uint64 + seen := make(map[string]struct{}, 8) + for rows.Next() { + var sbd, ascii string + if err := rows.Scan(&sbd, &ascii); err != nil { + tx.Rollback() + return fmt.Errorf("scan name: %w", err) + } + clear(seen) + for _, w := range strings.Fields(ascii) { + if _, dup := seen[w]; dup { + continue + } + seen[w] = struct{}{} + if _, err := stmt.Exec(w, sbd, ascii); err != nil { + tx.Rollback() + return fmt.Errorf("insert name word: %w", err) + } + words++ + } + } + if err := rows.Err(); err != nil { + tx.Rollback() + return fmt.Errorf("read names: %w", err) + } + if err := stmt.Close(); err != nil { + tx.Rollback() + return err + } + if err := tx.Commit(); err != nil { + return fmt.Errorf("commit name index: %w", err) + } + + if _, err := db.Exec(schema.PostLoadSQL); err != nil { + return fmt.Errorf("post-load statements: %w", err) + } + fmt.Printf("Name index: %d words\n", words) + return nil +} + +// Finish builds the derived tables, runs VACUUM and prints the stats block. +// +// VACUUM must run AFTER the transaction commits — SQLite refuses it inside one +// — and after the name index, so the file is laid out in one pass. func Finish(db *sql.DB, dbPath string, st Stats) error { + if err := BuildNameIndex(db); err != nil { + return err + } if _, err := db.Exec("VACUUM"); err != nil { return fmt.Errorf("vacuum: %w", err) } diff --git a/plans/260814-1200-httpvfs-range-queries/plan.md b/plans/260814-1200-httpvfs-range-queries/plan.md new file mode 100644 index 0000000..427f24b --- /dev/null +++ b/plans/260814-1200-httpvfs-range-queries/plan.md @@ -0,0 +1,71 @@ +# Serve the databases over HTTP range requests + +Status: implemented, unverified in a browser + +The whole-database download is gone. `sql.js-httpvfs` reads the pages a query +touches, so the databases ship raw as `.sqlite3` — a byte range of a gzip +stream is not a byte range of a database. + +## Why the schema had to change first + +Measured on the real 2016 database (223.5 MB before, 4 KB pages, 27 rows/page): + +| Query | Plan before | Would have fetched | +| --- | --- | --- | +| `so_bao_danh = ?` | SEARCH via PK | ~20 KB | +| `ho_ten_ascii LIKE '%x%'` | SCAN | 127 MB | +| `ho_ten_ascii LIKE 'x%'` | SCAN — the LIKE optimisation needs a NOCASE index | 127 MB | +| `COUNT(*)` | covering scan of idx_ho_ten_ascii | 20 MB | +| preset `ORDER BY toan DESC LIMIT 10` | SCAN + temp b-tree | 127 MB | + +So substring search was impossible, prefix search was no better, and the footer +count alone cost 20 MB per page load. + +## What shipped + +**Parser.** `name_word(word, so_bao_danh, ho_ten_ascii)` WITHOUT ROWID — the +table is the index — plus `name_word_freq(word, n)` and partial indexes on +`toan`, `khtn`, `khxh`. Dropped `idx_ho_ten` and `idx_ho_ten_ascii`: no query +plan could use either. + +877,460 names hold 2.87M word entries over a vocabulary of 4,397. A search asks +the frequency table which word is rarest, seeks on that one, and filters the +rest against the `ho_ten_ascii` copy inside the same b-tree — so "buu loc" still +finds "Nguyễn Bửu Lộc", in a few hundred KB. + +| Segment | 2016 | +| --- | --- | +| `student` | 127 MB | +| `name_word` | 97 MB | +| `idx_ten_cum_thi` | 37 MB | +| PK autoindex | 15 MB | +| score indexes | 12 MB | +| **total** | **288.6 MB** (2017: 237.7 MB) | + +**Assembler.** Publishes uncompressed; the size guard reads the raw size; the +stray-artifact check now rejects journals, `.db` and `.gz`. + +**Web.** `RemoteDatabase` wraps `createDbWorker`. The search tab runs with a +25 MB byte budget, the SQL tab asks for consent and then gets 250 MB, and the +bytes fetched are shown next to the query time. The footer count comes from +`datasets.json`. + +## Verified + +- Row counts unchanged: 877,460 and 861,068, both through the assembler guards. +- Every query the app issues is index-driven, checked with `EXPLAIN QUERY PLAN`: + `SEARCH w USING PRIMARY KEY (word>? AND word=0.10.0" } }, - "node_modules/sql.js": { - "version": "1.14.1", - "license": "MIT" + "node_modules/sql.js-httpvfs": { + "version": "0.8.12", + "resolved": "https://registry.npmjs.org/sql.js-httpvfs/-/sql.js-httpvfs-0.8.12.tgz", + "integrity": "sha512-lcEBc2q0psFRfdCx8Di22oUIkkv5MUIaVO/fGCj/Jjx6YQDKVylQEcjd7NSSbmINHTRwVkm/vWP8uuevT7Rkkw==", + "license": "Apache-2.0", + "dependencies": { + "comlink": "^4.3.0" + } }, "node_modules/stackback": { "version": "0.0.2", diff --git a/web/package.json b/web/package.json index a3f8851..fbf0a85 100644 --- a/web/package.json +++ b/web/package.json @@ -12,7 +12,7 @@ "lint": "eslint . && svelte-check --tsconfig ./tsconfig.json" }, "dependencies": { - "sql.js": "^1.14.1" + "sql.js-httpvfs": "^0.8.12" }, "devDependencies": { "@eslint/js": "^9.39.4", @@ -20,7 +20,7 @@ "@sveltejs/kit": "^2.70.2", "@sveltejs/vite-plugin-svelte": "^6.2.1", "@tailwindcss/vite": "^4.3.3", - "@types/sql.js": "^1.4.9", + "@types/sql.js": "^1.4.11", "eslint": "^9.39.4", "eslint-plugin-svelte": "^3.14.0", "globals": "^17.4.0", diff --git a/web/src/lib/components/custom-query.svelte b/web/src/lib/components/custom-query.svelte index be44eee..22d0732 100644 --- a/web/src/lib/components/custom-query.svelte +++ b/web/src/lib/components/custom-query.svelte @@ -1,5 +1,5 @@
@@ -97,7 +101,7 @@ type="button" class="btn-chip rounded-md" onclick={() => runPreset(preset.sql)} - {disabled} + disabled={disabled || running} > {preset.label} @@ -111,7 +115,7 @@ class="query-form mb-4" onsubmit={(e) => { e.preventDefault(); - execute(sql); + void execute(sql); }} > -
- {#if execTime !== null} {rows.length} kết quả · {execTime}ms {/if} + {#if db} + + Đã tải: {formatBytes(db.bytesRead)} + {/if}
@@ -145,7 +153,7 @@ [&>th]:bg-surface-alt [&>th]:px-2 [&>th]:py-2.5 [&>th]:text-left [&>th]:whitespace-nowrap" > - {#each columns as col, i (i)} + {#each columns as col (col)} {col} {/each} @@ -155,9 +163,9 @@ - {#each row as cell, ci (ci)} + {#each columns as col (col)} - {cell === null ? "NULL" : String(cell)} + {row[col] === null ? "NULL" : String(row[col])} {/each} diff --git a/web/src/lib/datasets.ts b/web/src/lib/datasets.ts index 4249d19..713c669 100644 --- a/web/src/lib/datasets.ts +++ b/web/src/lib/datasets.ts @@ -25,7 +25,7 @@ const SUBTITLE = "Dữ liệu thí sinh toàn quốc · Hỗ trợ truy vấn SQ * (crawler/internal/sources/source_.go); it is duplicated here because a Go * module and a Vite app cannot share a constant. Keep the two in step. */ -type Content = Omit & { presets: PresetGroup[] }; +type Content = Omit & { presets: PresetGroup[] }; const CONTENT: Record = { 2016: { @@ -67,6 +67,7 @@ for (const id of Object.keys(CONTENT)) { export const DATASETS: Dataset[] = registry.datasets.map((d) => ({ id: d.id, dbSizeMb: d.dbSizeMb, + rows: d.expectedRows, // Derived from expectedRows so the count the hub shows is the same number the // assembler enforces. blurb: `${d.expectedRows.toLocaleString("vi-VN")} thí sinh`, @@ -86,7 +87,12 @@ export function pathOf(dataset: Dataset, base: string): string { return `${base}/${dataset.id}/`; } -/** Gzipped database URL, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.db.gz". */ +/** + * Database URL, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.sqlite3". + * + * Uncompressed on purpose: the browser reads byte ranges of it, and a range of + * a gzip stream is not a range of the database. + */ export function dbOf(dataset: Dataset, base: string): string { - return `${base}/db/${dataset.id}.db.gz`; + return `${base}/db/${dataset.id}.sqlite3`; } diff --git a/web/src/lib/search.ts b/web/src/lib/search.ts new file mode 100644 index 0000000..76d780b --- /dev/null +++ b/web/src/lib/search.ts @@ -0,0 +1,93 @@ +import { normaliseExamId } from "./query-mode"; +import type { RemoteDatabase } from "./sqlite.svelte"; +import { toAscii } from "./to-ascii"; +import type { Student } from "./types"; + +export const MAX_RESULTS = 100; + +/** + * Name search over the name_word table. + * + * A query is matched word by word, each word as a prefix, in any order — so + * "buu loc" finds "Nguyễn Bửu Lộc". The work is arranged so that only one word + * is ever seeked on and the rest are filtered inside the same b-tree: + * + * 1. ask name_word_freq how many entries each word prefix covers (the whole + * vocabulary is ~4,400 rows, so this is a couple of pages); + * 2. seek on the rarest one — for a real name that is a few hundred to a few + * thousand entries rather than the 300,000 a word like "thi" would walk; + * 3. filter the other words against the ho_ten_ascii copy carried in + * name_word, so nothing is read from student until a row has matched; + * 4. join to student for the rows that survive, at most MAX_RESULTS of them. + * + * Every step is an index seek. A search costs a few hundred KB. + */ +export async function searchByName(db: RemoteDatabase, query: string): Promise { + const words = tokenise(query); + if (words.length === 0) return []; + + const seek = await rarest(db, words); + const others = words.filter((w) => w !== seek); + + // A word matches at a word boundary: the leading space makes the first word + // reachable by the same pattern as the rest. + const filters = others.map(() => `(' ' || w.ho_ten_ascii) LIKE ? ESCAPE '\\'`); + const sql = ` + SELECT s.* FROM name_word w + JOIN student s ON s.so_bao_danh = w.so_bao_danh + WHERE w.word >= ? AND w.word < ?${filters.length ? " AND " + filters.join(" AND ") : ""} + LIMIT ${MAX_RESULTS}`; + + return db.query(sql, [seek, upperBound(seek), ...others.map((w) => `% ${escapeLike(w)}%`)]); +} + +/** Exact lookup by exam number: a primary-key seek, a few pages. */ +export async function lookupExamId(db: RemoteDatabase, id: string): Promise { + return db.query("SELECT * FROM student WHERE so_bao_danh = ? LIMIT ?", [ + normaliseExamId(id), + MAX_RESULTS, + ]); +} + +/** Fold to ASCII and split into words, the same shape name_word was built in. */ +export function tokenise(query: string): string[] { + return toAscii(query).split(/\s+/).filter(Boolean); +} + +/** + * The word whose prefix covers the fewest entries, which is the one worth + * seeking on. One round trip for all of them. + */ +async function rarest(db: RemoteDatabase, words: string[]): Promise { + if (words.length === 1) return words[0]; + + const sql = words + .map(() => "SELECT ? AS word, COALESCE(SUM(n), 0) AS n FROM name_word_freq WHERE word >= ? AND word < ?") + .join(" UNION ALL "); + const params = words.flatMap((w) => [w, w, upperBound(w)]); + + const counts = await db.query<{ word: string; n: number }>(sql, params); + let best = words[0]; + let bestN = Infinity; + for (const { word, n } of counts) { + if (n < bestN) { + best = word; + bestN = n; + } + } + return best; +} + +/** + * The exclusive upper bound of a prefix range. U+FFFF sorts above any character + * that can follow the prefix, which is what turns "starts with" into a range + * the index can seek. + */ +function upperBound(prefix: string): string { + return prefix + "￿"; +} + +/** Escape the LIKE wildcards so a user typing % or _ searches for them. */ +function escapeLike(s: string): string { + return s.replace(/[\\%_]/g, (c) => `\\${c}`); +} diff --git a/web/src/lib/sqlite.svelte.ts b/web/src/lib/sqlite.svelte.ts index 1e70ef8..e00d10f 100644 --- a/web/src/lib/sqlite.svelte.ts +++ b/web/src/lib/sqlite.svelte.ts @@ -1,84 +1,106 @@ -import initSqlJs, { type Database } from "sql.js"; - -// The sql.js engine itself, fetched from the upstream CDN rather than bundled. -// It is a hard runtime dependency: if this URL is unreachable, initSqlJs() -// rejects and no dataset can be opened at all, whatever the database fetch does. -const SQL_WASM_URL = "https://sql.js.org/dist/sql-wasm.wasm"; +import { createDbWorker, type WorkerHttpvfs } from "sql.js-httpvfs"; +import workerUrl from "sql.js-httpvfs/dist/sqlite.worker.js?url"; +import wasmUrl from "sql.js-httpvfs/dist/sql-wasm.wasm?url"; /** - * A SQLite database loaded from a URL into sql.js, with reactive load state. + * The database is read where it lies. SQLite asks for pages, the virtual file + * system turns each into an HTTP range request, and only the pages a query + * touches ever cross the network — a few hundred KB for a lookup, against the + * 45 MB the whole file used to cost before the first query. * - * The whole file is downloaded and decompressed before the first query: a `.gz` - * URL is inflated in the browser. Nothing streams — sql.js needs the complete - * image in memory. + * That only holds while every query is index-driven. The schema exists for it: + * name_word serves name search, and the score indexes serve the SQL presets. + * An unindexed query walks the table and pulls all 100+ MB of it, which is what + * the byte budget below is for. */ -export class SqliteSource { - db = $state(null); - loading = $state(true); + +// Matches the page size the parser writes, so one request is one page. +const CHUNK_BYTES = 4096; + +/** Generous for indexed work: a name search costs well under 1 MB. */ +export const SEARCH_BUDGET_BYTES = 25 * 1024 * 1024; + +/** What the SQL tab gets once the user has accepted the cost of a scan. */ +export const PLAYGROUND_BUDGET_BYTES = 250 * 1024 * 1024; + +/** + * One remotely-paged database, with the load state the UI needs. + * + * `budgetBytes` is a hard ceiling for the worker's lifetime: past it a query + * fails instead of quietly downloading the file. Raising it means a new worker, + * which costs only the header pages. + */ +export class RemoteDatabase { + ready = $state(false); error = $state(null); - progress = $state(0); + /** Bytes fetched so far, refreshed after every query. */ + bytesRead = $state(0); - #cancelled = false; + #worker: WorkerHttpvfs | null = null; + #opening: Promise; + #closed = false; - constructor(url: string) { - void this.#load(url); + constructor( + readonly url: string, + readonly budgetBytes: number = SEARCH_BUDGET_BYTES, + ) { + this.#opening = this.#open(); } - async #load(url: string) { + async #open(): Promise { try { - const SQL = await initSqlJs({ locateFile: () => SQL_WASM_URL }); - - const response = await fetch(url); - if (!response.ok) throw new Error(`Failed to fetch database: ${response.status}`); - - const contentLength = Number(response.headers.get("Content-Length")) || 0; - const reader = response.body!.getReader(); - const chunks: Uint8Array[] = []; - let received = 0; - - for (;;) { - const { done, value } = await reader.read(); - if (done) break; - chunks.push(value); - received += value.length; - if (contentLength > 0) { - this.progress = Math.round((received / contentLength) * 100); - } - } - if (this.#cancelled) return; - - const blob = new Blob(chunks as BlobPart[]); - const bytes = url.endsWith(".gz") - ? await new Response(blob.stream().pipeThrough(new DecompressionStream("gzip"))).arrayBuffer() - : await blob.arrayBuffer(); - if (this.#cancelled) return; - - this.db = new SQL.Database(new Uint8Array(bytes)); - this.loading = false; + const worker = await createDbWorker( + [{ from: "inline", config: { serverMode: "full", url: this.url, requestChunkSize: CHUNK_BYTES } }], + workerUrl, + wasmUrl, + this.budgetBytes, + ); + if (this.#closed) throw new Error("closed"); + this.#worker = worker; + this.ready = true; + return worker; } catch (err) { - if (this.#cancelled) return; - this.error = err instanceof Error ? err.message : String(err); - this.loading = false; + if (!this.#closed) { + this.error = message(err); + this.ready = false; + } + throw err; } } - /** Release the database. Call from the owning component's teardown. */ + /** Run a query and return its rows as objects. */ + async query(sql: string, params: unknown[] = []): Promise { + const worker = this.#worker ?? (await this.#opening); + // Comlink erases the generic when it proxies the method across the worker + // boundary, so the row type is asserted here rather than inferred. + const run = worker.db.query as unknown as (sql: string, ...params: unknown[]) => Promise; + try { + return await run(sql, ...params); + } finally { + // Comlink proxies the property, so this is a round trip; worth it because + // the number is the only honest feedback about what a query cost. + this.bytesRead = await worker.worker.bytesRead; + } + } + + /** + * Drop this database. createDbWorker owns the Worker and exposes no handle to + * it, so the thread outlives this call; a page creates at most one per + * dataset and one more if the SQL budget is raised, which is why that is + * tolerable rather than a leak worth working around. + */ close() { - this.#cancelled = true; - this.db?.close(); - this.db = null; + this.#closed = true; + this.#worker = null; + this.ready = false; } } -/** Run a statement and return its rows as objects. */ -export function queryRows(db: Database, sql: string, params?: Record): T[] { - const stmt = db.prepare(sql); - try { - if (params) stmt.bind(params as never); - const rows: T[] = []; - while (stmt.step()) rows.push(stmt.getAsObject() as T); - return rows; - } finally { - stmt.free(); - } +/** True when a query failed because it would have exceeded the byte budget. */ +export function isBudgetError(err: unknown): boolean { + return /maxBytesToRead|too much data|exceeded/i.test(message(err)); +} + +function message(err: unknown): string { + return err instanceof Error ? err.message : String(err); } diff --git a/web/src/lib/types.ts b/web/src/lib/types.ts index ecf82f3..91eb320 100644 --- a/web/src/lib/types.ts +++ b/web/src/lib/types.ts @@ -35,7 +35,12 @@ export type Student = { /** A dataset as the interface needs it: registry facts plus presentation. */ export type Dataset = { id: string; + /** Size of the hosted database. Nothing downloads it whole; it is shown so a + * user knows what they are querying into. */ dbSizeMb: number; + /** Row count from the registry, so the footer never runs COUNT(*) — that + * scans an index and would cost 20 MB over range requests. */ + rows: number; blurb: string; label: string; title: string; diff --git a/web/src/routes/[dataset]/+page.svelte b/web/src/routes/[dataset]/+page.svelte index ce0720b..ac18283 100644 --- a/web/src/routes/[dataset]/+page.svelte +++ b/web/src/routes/[dataset]/+page.svelte @@ -8,59 +8,47 @@ import SearchForm from "$lib/components/search-form.svelte"; import StudentDetail from "$lib/components/student-detail.svelte"; import { dbOf } from "$lib/datasets"; - import { isExamId, normaliseExamId } from "$lib/query-mode"; - import { SqliteSource, queryRows } from "$lib/sqlite.svelte"; - import { isAsciiOnly, toAscii } from "$lib/to-ascii"; + import { isExamId } from "$lib/query-mode"; + import { MAX_RESULTS, lookupExamId, searchByName } from "$lib/search"; + import { PLAYGROUND_BUDGET_BYTES, RemoteDatabase, SEARCH_BUDGET_BYTES } from "$lib/sqlite.svelte"; import type { Student } from "$lib/types"; - const MAX_RESULTS = 100; - let { data } = $props(); const dataset = $derived(data.dataset); - let source = $state(null); + let db = $state(null); let results = $state(null); let searchError = $state(null); let activeTab = $state<"search" | "sql">("search"); - let totalCount = $state(null); + let sqlWarningOpen = $state(false); + // Raised once the user has accepted that a hand-written query may fetch a lot. + let budget = $state(SEARCH_BUDGET_BYTES); // Owned here, not in SearchForm, so it can be bound to the URL both ways. // The query string is unreadable while prerendering — there is no request — // so a deep link is picked up on the client only. let query = $state(browser ? (page.url.searchParams.get("q") ?? "") : ""); - const db = $derived(source?.db ?? null); - const loading = $derived(source?.loading ?? true); - const loadError = $derived(source?.error ?? null); - const progress = $derived(source?.progress ?? 0); - const busy = $derived(loading || !!loadError); + const opening = $derived(db !== null && !db.ready && db.error === null); + const loadError = $derived(db?.error ?? null); + const busy = $derived(!db?.ready); - // The database is per dataset, and only ever fetched in the browser: $effect - // does not run while prerendering. + // Opened in the browser only: $effect does not run while prerendering. A new + // budget means a new worker, which costs only the header pages. $effect(() => { - const opened = new SqliteSource(dbOf(dataset, base)); - source = opened; + const opened = new RemoteDatabase(dbOf(dataset, base), budget); + db = opened; return () => { opened.close(); - source = null; - results = null; - totalCount = null; + db = null; }; }); - // Candidate count for the footer. One-shot per database: it cannot change - // without a new one. - $effect(() => { - if (!db) return; - const [row] = queryRows<{ c: number }>(db, "SELECT COUNT(*) AS c FROM student"); - totalCount = row?.c ?? null; - }); - - // Hydrate a ?q= deep link as soon as the database is ready. + // Hydrate a ?q= deep link as soon as the database is open. let hydrated = false; $effect(() => { - if (!db || hydrated) return; + if (!db?.ready || hydrated) return; hydrated = true; - if (query) search(query); + if (query) void search(query); }); // Sync the query to ?q= without adding a history entry, so back still leaves @@ -73,35 +61,15 @@ replaceState(q ? `${route}?q=${encodeURIComponent(q)}` : route, page.state); } - function search(q: string) { - if (!db) return; + async function search(q: string) { + const source = db; + if (!source?.ready) return; searchError = null; query = q; writeUrlQuery(q); try { - if (isExamId(q)) { - // Letter-prefixed 2016 IDs are stored upper-case; digits are unaffected. - results = queryRows( - db, - "SELECT * FROM student WHERE so_bao_danh = $q LIMIT $limit", - { $q: normaliseExamId(q), $limit: MAX_RESULTS }, - ); - } else if (isAsciiOnly(q)) { - results = queryRows( - db, - "SELECT * FROM student WHERE ho_ten_ascii LIKE $q LIMIT $limit", - { $q: `%${toAscii(q)}%`, $limit: MAX_RESULTS }, - ); - } else { - results = queryRows( - db, - `SELECT * FROM student - WHERE ho_ten LIKE $q OR ho_ten_ascii LIKE $qn - LIMIT $limit`, - { $q: `%${q}%`, $qn: `%${toAscii(q)}%`, $limit: MAX_RESULTS }, - ); - } + results = isExamId(q) ? await lookupExamId(source, q) : await searchByName(source, q); } catch (err) { searchError = err instanceof Error ? err.message : String(err); } @@ -113,9 +81,34 @@ writeUrlQuery(""); } + function openSqlTab() { + if (budget >= PLAYGROUND_BUDGET_BYTES) { + activeTab = "sql"; + return; + } + sqlWarningOpen = true; + } + + function acceptSqlWarning() { + sqlWarningOpen = false; + // Reopening with the larger budget throws away the current worker, and with + // it the pages it had cached — a few hundred KB, refetched on demand. + budget = PLAYGROUND_BUDGET_BYTES; + activeTab = "sql"; + } + + function declineSqlWarning() { + sqlWarningOpen = false; + activeTab = "search"; + } + // Global shortcuts: Ctrl+Enter submits the SQL query, "/" focuses the search // box unless the user is already typing somewhere. function onKeydown(event: KeyboardEvent) { + if (event.key === "Escape" && sqlWarningOpen) { + declineSqlWarning(); + return; + } if (event.ctrlKey && event.key === "Enter" && activeTab === "sql") { document.querySelector(".query-form")?.requestSubmit(); return; @@ -151,18 +144,8 @@
- {#if loading} -
-

- Đang tải cơ sở dữ liệu ~{dataset.dbSizeMb} MB{progress > 0 ? ` · ${progress}%` : ""} -

-
-
-
-

- Lần đầu có thể mất 10-30 giây. Sau đó trình duyệt sẽ lưu cache và mở nhanh hơn. -

-
+ {#if opening} +

Đang mở cơ sở dữ liệu…

{/if} {#if loadError} @@ -170,20 +153,30 @@ {/if}
- {#each [{ id: "search", label: "Tra cứu" }, { id: "sql", label: "Truy vấn SQL" }] as const as tab (tab.id)} - - {/each} + +
{#if activeTab === "search"} @@ -215,16 +208,44 @@ {/if}
-
+

Nguồn: {dataset.source} - {#if totalCount !== null} - · {totalCount.toLocaleString("vi-VN")} thí sinh - {/if} - · Dữ liệu chỉ mang tính tham khảo + · {dataset.rows.toLocaleString("vi-VN")} thí sinh · Dữ liệu chỉ mang tính tham khảo

+ +{#if sqlWarningOpen} + + +{/if}