mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
The previous attempt passed fileLength in the inline config, and the
worker discarded it. sqlite.worker.ts builds the lazy file's config
itself and hardcodes:
fileLength: config.serverMode === "chunked"
? config.databaseLengthBytes
: undefined
So in full mode there is no way to supply a length, and the library
falls back to sizing the file with a HEAD request — which GitHub Pages
answers with the gzipped length, and then refuses to use.
Chunked mode is the only mode that takes a length. One chunk holds the
whole database, so the chunk index is always 0 and every request goes to
urlPrefix + "0"; the assembler therefore publishes <id>.sqlite30. The
length still comes from the range probe, which also checks the bytes are
a SQLite header.
dbPrefixOf and dbOf derive one form from the other and are handed to
RemoteDatabase together, so the prefix the library appends an index to
and the file the assembler writes cannot drift apart. A test pins that;
nothing else would catch it, because the symptom is a 404 per query.
The stray-artifact guard now also rejects a leftover <id>.sqlite3, which
after this change is a stale artifact rather than the published one.
sqlite-wasm-http was checked as an alternative and does not help: its
worker takes the size from a HEAD Content-Length too, and its options
expose no way to override it, so on Pages it would silently use the
compressed size. Its shared-cache backend wants COOP/COEP, but it ships
a fallback that does not, so isolation was never the blocker — the
architecture note claiming otherwise is corrected.
177 lines
5.7 KiB
Go
177 lines
5.7 KiB
Go
// Package databases builds and verifies one SQLite file per dataset.
|
|
//
|
|
// VERIFICATION IS THE POINT OF THIS PACKAGE, not an extra.
|
|
//
|
|
// Nothing between the parser and the published site otherwise asserts that a
|
|
// database has data in it. The parser logs a file-level failure and continues,
|
|
// returns success regardless, and finishes cleanly even at zero rows; the site
|
|
// assembly only inspects filenames. So a reader that silently under-produced
|
|
// would publish a truncated dataset with green CI and no red signal anywhere.
|
|
//
|
|
// The guards below close that: a build whose row count does not match the
|
|
// registry, or whose artifact is implausibly small, fails the pipeline.
|
|
//
|
|
// The databases ship uncompressed. The browser reads them a page at a time over
|
|
// HTTP range requests, and a range of a gzip stream is not a range of the
|
|
// database.
|
|
package databases
|
|
|
|
import (
|
|
"database/sql"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"path/filepath"
|
|
|
|
_ "modernc.org/sqlite" // pure-Go driver: the pipeline stays cgo-free
|
|
|
|
"github.com/tiennm99/thptqg/assembler/internal/registry"
|
|
)
|
|
|
|
// driverName is modernc.org/sqlite's registered name.
|
|
const driverName = "sqlite"
|
|
|
|
// minSizeRatio: a database far below its usual size means a truncated build,
|
|
// even if the row count somehow passed.
|
|
const minSizeRatio = 0.9
|
|
|
|
// Extension is the published suffix. The trailing 0 is a chunk index, not a
|
|
// typo: the browser reads the file through sql.js-httpvfs in chunked mode,
|
|
// which is the only mode whose config accepts the file's length. That mode
|
|
// builds each request's URL as urlPrefix + chunkIndex, and one chunk holds the
|
|
// whole database, so the index is always 0 and the prefix is "<id>.sqlite3".
|
|
//
|
|
// The length has to come from the config because the library otherwise takes
|
|
// it from a HEAD request, which GitHub Pages answers with the gzipped size.
|
|
//
|
|
// Not ".db" either: keeping that name free lets the site assembly treat any
|
|
// stray .db or SQLite journal in the output as the leftover it is.
|
|
const Extension = ".sqlite30"
|
|
|
|
// Paths locates the pieces this package needs.
|
|
type Paths struct {
|
|
// Root is the repository root.
|
|
Root string
|
|
// Parser is the parser module directory.
|
|
Parser string
|
|
// OutDir is where the databases are staged — the directory Vite publishes.
|
|
OutDir string
|
|
}
|
|
|
|
// DefaultPaths derives the standard layout from the repository root.
|
|
func DefaultPaths(root string) Paths {
|
|
return Paths{
|
|
Root: root,
|
|
Parser: filepath.Join(root, "parser"),
|
|
OutDir: filepath.Join(root, ".build", "public", "db"),
|
|
}
|
|
}
|
|
|
|
// BuildParser compiles the parser binary and returns its path.
|
|
//
|
|
// Compiling here rather than expecting a prebuilt binary keeps the pipeline one
|
|
// command. Go caches the work, so repeat runs cost almost nothing.
|
|
func BuildParser(p Paths) (string, error) {
|
|
bin := filepath.Join(p.Parser, "bin", "xlsxread")
|
|
cmd := exec.Command("go", "-C", p.Parser, "build", "-o", "bin/xlsxread", "./cmd/xlsxread")
|
|
cmd.Stdout, cmd.Stderr = os.Stdout, os.Stderr
|
|
if err := cmd.Run(); err != nil {
|
|
return "", fmt.Errorf("compiling the parser: %w", err)
|
|
}
|
|
return bin, nil
|
|
}
|
|
|
|
// Build runs the parser for one dataset, verifies the result and compresses it.
|
|
//
|
|
// Only the .gz survives: shipping a 100+ MB uncompressed database is made
|
|
// structurally impossible rather than left to a cleanup step.
|
|
func Build(p Paths, bin string, d registry.Dataset) error {
|
|
if err := os.MkdirAll(p.OutDir, 0o755); err != nil {
|
|
return err
|
|
}
|
|
db := filepath.Join(p.OutDir, d.ID+Extension)
|
|
|
|
cmd := exec.Command(bin,
|
|
"build",
|
|
"--schema", filepath.Join(p.Parser, "configs", d.ID+".yml"),
|
|
"--input", filepath.Join(p.Root, "data", d.ID),
|
|
"--output", db,
|
|
)
|
|
cmd.Stdout, cmd.Stderr = os.Stdout, os.Stderr
|
|
if err := cmd.Run(); err != nil {
|
|
return fmt.Errorf("%s: parser failed: %w", d.ID, err)
|
|
}
|
|
|
|
rows, err := countRows(db)
|
|
if err != nil {
|
|
return fmt.Errorf("%s: %w", d.ID, err)
|
|
}
|
|
if rows != d.ExpectedRows {
|
|
return fmt.Errorf(
|
|
"%s: row count %d, expected %d\nRefusing to publish — the build did not reproduce the known dataset",
|
|
d.ID, rows, d.ExpectedRows)
|
|
}
|
|
fmt.Printf(" ✓ %s: %d rows (matches expected)\n", d.ID, rows)
|
|
|
|
st, err := os.Stat(db)
|
|
if err != nil {
|
|
return fmt.Errorf("%s: %w", d.ID, err)
|
|
}
|
|
sizeMb := float64(st.Size()) / 1024 / 1024
|
|
if min := d.DbSizeMb * minSizeRatio; sizeMb < min {
|
|
return fmt.Errorf(
|
|
"%s: %.1f MB is below %.1f MB (%.0f%% of the expected %.0f MB)\n"+
|
|
"Refusing to publish — the artifact looks truncated",
|
|
d.ID, sizeMb, min, minSizeRatio*100, d.DbSizeMb)
|
|
}
|
|
|
|
fmt.Printf(" → %s (%.1f MB)\n\n", filepath.Base(db), sizeMb)
|
|
return nil
|
|
}
|
|
|
|
// countRows opens the database read-only and counts what was written.
|
|
func countRows(path string) (int64, error) {
|
|
conn, err := sql.Open(driverName, "file:"+path+"?mode=ro")
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
defer conn.Close()
|
|
|
|
var n int64
|
|
if err := conn.QueryRow("SELECT COUNT(*) FROM student").Scan(&n); err != nil {
|
|
return 0, fmt.Errorf("counting rows: %w", err)
|
|
}
|
|
return n, nil
|
|
}
|
|
|
|
// Clean removes staged artifacts for datasets that are no longer in the
|
|
// registry. Without this a removed dataset's file lingers in the staging
|
|
// directory, and the site assembly copies that directory wholesale — so the
|
|
// dead database would be published again.
|
|
func Clean(p Paths, keep []registry.Dataset) error {
|
|
entries, err := os.ReadDir(p.OutDir)
|
|
if os.IsNotExist(err) {
|
|
return nil
|
|
}
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
wanted := make(map[string]bool, len(keep))
|
|
for _, d := range keep {
|
|
wanted[d.ID+Extension] = true
|
|
}
|
|
|
|
for _, e := range entries {
|
|
if e.IsDir() || wanted[e.Name()] {
|
|
continue
|
|
}
|
|
full := filepath.Join(p.OutDir, e.Name())
|
|
if err := os.Remove(full); err != nil {
|
|
return err
|
|
}
|
|
fmt.Printf(" removed stale artifact %s\n", e.Name())
|
|
}
|
|
return nil
|
|
}
|