mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
GitHub Pages serves .sqlite3 as application/octet-stream, which mime-db marks compressible, so an un-ranged response comes back gzipped with the compressed length. sql.js-httpvfs sizes a file with a HEAD request, sees that length is unusable and refuses to open the database: Length of the file not known. It must either be supplied in the config or given by the HTTP server. Page reads were never affected. The Fetch standard requires browsers to send Accept-Encoding: identity on any request carrying a Range header, and the live site returns 206 with raw bytes to one. So the length is probed the same way and passed as fileLength, which is the escape hatch the library's own message points at. The probe reads the first 100 bytes, so it also checks the file starts with the SQLite magic and that its page size matches the request size — a host that ever compresses a ranged response now fails with a clear message rather than feeding the library the wrong bytes. The post-deploy check the docs prescribed could not have caught this: a bare `curl -sI` advertises no encoding, so it reports success whatever the host does. It is replaced, in the docs and in CI, by a ranged read that verifies the bytes.
151 lines
5.8 KiB
YAML
151 lines
5.8 KiB
YAML
name: Deploy to GitHub Pages
|
|
|
|
on:
|
|
push:
|
|
branches: [main]
|
|
# Pull requests run the build job only: the deploy job is guarded to main, so
|
|
# a branch can be verified end to end without touching the live site.
|
|
pull_request:
|
|
workflow_dispatch:
|
|
|
|
permissions:
|
|
contents: read
|
|
pages: write
|
|
id-token: write
|
|
|
|
# Keyed by ref, not just "pages". With one shared group a pull-request run and
|
|
# a main deploy compete for the same lane, and cancel-in-progress means the
|
|
# newer one wins: the deploy of #10 was killed 3m22s in by a PR run that
|
|
# started after it, and the site silently stayed on the previous build while
|
|
# every check stayed green. Per ref, a push still cancels its own superseded
|
|
# run, which is the case where cancelling is worth having.
|
|
concurrency:
|
|
group: pages-${{ github.ref }}
|
|
cancel-in-progress: true
|
|
|
|
jobs:
|
|
build:
|
|
runs-on: ubuntu-latest
|
|
env:
|
|
# Every Go module here is cgo-free — grate, excelize, yaml.v3, x/net,
|
|
# x/text and modernc.org/sqlite — so no C toolchain is needed. Set
|
|
# explicitly rather than relying on the default.
|
|
CGO_ENABLED: '0'
|
|
steps:
|
|
- uses: actions/checkout@v7
|
|
|
|
# The version comes from go.mod rather than a literal here. setup-go sets
|
|
# GOTOOLCHAIN=local, so Go will not fetch the toolchain a module asks for
|
|
# — the one installed has to satisfy it. A literal '1.26' resolves to
|
|
# whatever patch the runner manifest has, which was 1.26.5 against a
|
|
# go.mod requiring 1.26.6, and the build stopped there. All three modules
|
|
# are bumped together, so parser/go.mod speaks for them.
|
|
- uses: actions/setup-go@v7
|
|
with:
|
|
go-version-file: parser/go.mod
|
|
cache-dependency-path: |
|
|
parser/go.sum
|
|
crawler/go.sum
|
|
assembler/go.sum
|
|
|
|
# web/ is the only npm project in the repository; the other stages are Go.
|
|
- uses: actions/setup-node@v7
|
|
with:
|
|
node-version: '24'
|
|
cache: 'npm'
|
|
cache-dependency-path: web/package-lock.json
|
|
|
|
- name: Install web dependencies
|
|
working-directory: web
|
|
run: npm ci
|
|
|
|
# The reader-fidelity suite compares every real input file against a
|
|
# committed hash oracle, so it is the regression guard for the whole
|
|
# reader. Runs before anything is built.
|
|
- name: Test parser
|
|
run: go -C parser test ./...
|
|
|
|
# The crawler is not part of the build — it only refreshes data/ by hand.
|
|
# It is still tested here so it cannot rot unnoticed, and because its
|
|
# fixture test guards the parser: input filenames decide which row
|
|
# survives a duplicate exam number.
|
|
- name: Test crawler
|
|
run: go -C crawler test ./...
|
|
|
|
# The assembler owns every guard between a built database and the
|
|
# published site, so its tests are the ones that prove a short or missing
|
|
# database cannot ship.
|
|
- name: Test assembler
|
|
run: go -C assembler test ./...
|
|
|
|
# The tests cover the framework-free modules, including the ASCII fold
|
|
# that has to match the Go parser.
|
|
- name: Test web
|
|
working-directory: web
|
|
run: npm test
|
|
|
|
- name: Lint web
|
|
working-directory: web
|
|
run: npm run lint
|
|
|
|
# excelize carries an open advisory, and the 2017 refresh runbook feeds
|
|
# network-downloaded spreadsheets straight into the parser.
|
|
- name: Vulnerability scan
|
|
run: |
|
|
go install golang.org/x/vuln/cmd/govulncheck@latest
|
|
GOVULNCHECK="$(go env GOPATH)/bin/govulncheck"
|
|
for m in parser crawler assembler; do (cd "$m" && "$GOVULNCHECK" ./...); done
|
|
|
|
# One command runs the whole pipeline: compile the parser, build and
|
|
# verify each database against its registry row count and size, build the
|
|
# web app, and assemble _site — refusing to continue if a database is
|
|
# short, an artifact looks truncated, or one is missing entirely.
|
|
- name: Build site
|
|
run: go -C assembler run ./cmd/assemble
|
|
|
|
- uses: actions/upload-pages-artifact@v5
|
|
with:
|
|
path: _site
|
|
|
|
deploy:
|
|
# Deploy only from main. pull_request and workflow_dispatch both run on
|
|
# other branches, and publishing one would put that branch's output on the
|
|
# live site while concurrency cancel-in-progress killed an in-flight good
|
|
# deploy on the way.
|
|
if: github.ref == 'refs/heads/main'
|
|
needs: build
|
|
runs-on: ubuntu-latest
|
|
environment:
|
|
name: github-pages
|
|
url: ${{ steps.deployment.outputs.page_url }}
|
|
steps:
|
|
- id: deployment
|
|
uses: actions/deploy-pages@v5
|
|
|
|
- uses: actions/checkout@v7
|
|
|
|
# The site reads the databases a page at a time over range requests, so
|
|
# what matters is not that the host leaves the file alone in general — it
|
|
# gzips the un-ranged response, and browsers work around that by sending
|
|
# Accept-Encoding: identity whenever a request carries a Range header —
|
|
# but that a ranged read returns raw database bytes. Checking headers is
|
|
# what missed this before: a bare `curl -sI` advertises no encoding and
|
|
# so passes whatever the host does. Check the bytes instead.
|
|
- name: Verify ranged reads return database bytes
|
|
env:
|
|
PAGE_URL: ${{ steps.deployment.outputs.page_url }}
|
|
run: |
|
|
set -euo pipefail
|
|
for id in $(jq -r '.datasets[].id' datasets.json); do
|
|
url="${PAGE_URL%/}/db/${id}.sqlite3"
|
|
if ! magic=$(curl -sf -r 0-14 -H 'Accept-Encoding: identity;q=1, *;q=0' "$url"); then
|
|
echo "::error::$url is not fetchable"
|
|
exit 1
|
|
fi
|
|
if [ "$magic" != "SQLite format 3" ]; then
|
|
echo "::error::$url did not return database bytes over a range request"
|
|
exit 1
|
|
fi
|
|
echo "$url: SQLite format 3"
|
|
done
|