name: Deploy to GitHub Pages on: push: branches: [main] # Pull requests run the build job only: the deploy job is guarded to main, so # a branch can be verified end to end without touching the live site. pull_request: workflow_dispatch: permissions: contents: read pages: write id-token: write # Keyed by ref, not just "pages". With one shared group a pull-request run and # a main deploy compete for the same lane, and cancel-in-progress means the # newer one wins: the deploy of #10 was killed 3m22s in by a PR run that # started after it, and the site silently stayed on the previous build while # every check stayed green. Per ref, a push still cancels its own superseded # run, which is the case where cancelling is worth having. concurrency: group: pages-${{ github.ref }} cancel-in-progress: true jobs: build: runs-on: ubuntu-latest env: # Every Go module here is cgo-free — grate, excelize, yaml.v3, x/net, # x/text and modernc.org/sqlite — so no C toolchain is needed. Set # explicitly rather than relying on the default. CGO_ENABLED: '0' steps: - uses: actions/checkout@v7 # The version comes from go.mod rather than a literal here. setup-go sets # GOTOOLCHAIN=local, so Go will not fetch the toolchain a module asks for # — the one installed has to satisfy it. A literal '1.26' resolves to # whatever patch the runner manifest has, which was 1.26.5 against a # go.mod requiring 1.26.6, and the build stopped there. All three modules # are bumped together, so parser/go.mod speaks for them. - uses: actions/setup-go@v7 with: go-version-file: parser/go.mod cache-dependency-path: | parser/go.sum crawler/go.sum assembler/go.sum # web/ is the only npm project in the repository; the other stages are Go. - uses: actions/setup-node@v7 with: node-version: '24' cache: 'npm' cache-dependency-path: web/package-lock.json - name: Install web dependencies working-directory: web run: npm ci # The reader-fidelity suite compares every real input file against a # committed hash oracle, so it is the regression guard for the whole # reader. Runs before anything is built. - name: Test parser run: go -C parser test ./... # The crawler is not part of the build — it only refreshes data/ by hand. # It is still tested here so it cannot rot unnoticed, and because its # fixture test guards the parser: input filenames decide which row # survives a duplicate exam number. - name: Test crawler run: go -C crawler test ./... # The assembler owns every guard between a built database and the # published site, so its tests are the ones that prove a short or missing # database cannot ship. - name: Test assembler run: go -C assembler test ./... # The tests cover the framework-free modules, including the ASCII fold # that has to match the Go parser. - name: Test web working-directory: web run: npm test - name: Lint web working-directory: web run: npm run lint # excelize carries an open advisory, and the 2017 refresh runbook feeds # network-downloaded spreadsheets straight into the parser. - name: Vulnerability scan run: | go install golang.org/x/vuln/cmd/govulncheck@latest GOVULNCHECK="$(go env GOPATH)/bin/govulncheck" for m in parser crawler assembler; do (cd "$m" && "$GOVULNCHECK" ./...); done # One command runs the whole pipeline: compile the parser, build and # verify each database against its registry row count and size, build the # web app, and assemble _site — refusing to continue if a database is # short, an artifact looks truncated, or one is missing entirely. - name: Build site run: go -C assembler run ./cmd/assemble - uses: actions/upload-pages-artifact@v5 with: path: _site deploy: # Deploy only from main. pull_request and workflow_dispatch both run on # other branches, and publishing one would put that branch's output on the # live site while concurrency cancel-in-progress killed an in-flight good # deploy on the way. if: github.ref == 'refs/heads/main' needs: build runs-on: ubuntu-latest environment: name: github-pages url: ${{ steps.deployment.outputs.page_url }} steps: - id: deployment uses: actions/deploy-pages@v5 - uses: actions/checkout@v7 # The site reads the databases a page at a time over range requests, so # what matters is not that the host leaves the file alone in general — it # gzips the un-ranged response, and browsers work around that by sending # Accept-Encoding: identity whenever a request carries a Range header — # but that a ranged read returns raw database bytes. Checking headers is # what missed this before: a bare `curl -sI` advertises no encoding and # so passes whatever the host does. Check the bytes instead. - name: Verify ranged reads return database bytes env: PAGE_URL: ${{ steps.deployment.outputs.page_url }} run: | set -euo pipefail for id in $(jq -r '.datasets[].id' datasets.json); do url="${PAGE_URL%/}/db/${id}.sqlite3" if ! magic=$(curl -sf -r 0-14 -H 'Accept-Encoding: identity;q=1, *;q=0' "$url"); then echo "::error::$url is not fetchable" exit 1 fi if [ "$magic" != "SQLite format 3" ]; then echo "::error::$url did not return database bytes over a range request" exit 1 fi echo "$url: SQLite format 3" done