build: ship the dictionary from a committed corpus, refreshed monthly by pull request

This commit is contained in:
tiennm99 committed 2026-10-02 13:46:25 +07:00
1 parent a947f5ce2e
commit f37e3e69f4
13 files changed
+36670 -95

No files matched your search

+1
View File
@@ -15,6 +15,7 @@ data/*.db
data/*.db-journal
data/*.db-wal
data/*.db-shm
data/*.tmp
noitu-server
noitu-server.exe
build-dictionary
+8 -14
View File
@@ -1,9 +1,9 @@
# Tests and packaging.
#
# Nothing here downloads the upstream wordlist except the release image
# build. Everything else plays against the small database derived from
# the checked-in word sample, which goes through the same builder the real one
# does.
# Nothing here downloads the upstream wordlist. The image is built from the
# committed corpus, data/dictionary.txt, exactly as it ships; the test suites
# play against the small database derived from the checked-in word sample,
# which goes through the same builder the real one does.
#
# The wire contract has its own workflow: see proto.yml.
name: ci
@@ -159,17 +159,11 @@ jobs:
with:
persist-credentials: false
# On a release the image is built from the real upstream release, which
# is the only job in this file that downloads it. Every other run builds
# the same Dockerfile against the fixture word list, so a broken image is
# caught on the pull request rather than at release time.
# The real dictionary, from the committed corpus: the image every run
# builds is the one that ships, so a corpus the builder rejects fails
# here rather than at deploy time.
- name: Build
run: |
if [ "${{ github.event_name }}" = "release" ]; then
docker build -t noitu:ci .
else
docker build --build-arg FIXTURE_DICT=1 -t noitu:ci .
fi
run: docker build -t noitu:ci .
# CC BY-SA 4.0 applies to the derived wordlist wherever it is
# distributed, and an image is distribution. This is the assertion that
+82
View File
@@ -0,0 +1,82 @@
# Monthly dictionary refresh.
#
# Wikimedia regenerates the Wiktionary tiếng Việt dump every month. This
# downloads the current one, regenerates data/dictionary.txt from it — the
# committed corpus the image is built from — and opens a pull request against
# dev, so a new month's words arrive as a diff someone reads before it ships.
#
# The dump run that starts on the 1st takes a few days to publish, hence the
# 10th. The builder fails below its word, page and meaning-coverage floors, so
# a broken month opens no pull request rather than a corpus with holes in it.
#
# A pull request opened with GITHUB_TOKEN does not trigger ci.yml. The build
# here already verifies the database it derives; the full suite runs when dev
# is merged onward.
name: refresh-dictionary
on:
schedule:
- cron: "0 3 10 * *"
workflow_dispatch:
permissions:
contents: write
pull-requests: write
concurrency:
group: refresh-dictionary
cancel-in-progress: false
jobs:
refresh:
name: Regenerate the corpus from the current dump
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
ref: dev
- uses: actions/setup-go@v5
with:
go-version: stable
cache-dependency-path: server/go.sum
- name: Download the dump
run: make fetch-dict
- name: Regenerate the corpus
run: make refresh-dict
# What the pull request body reports: the dump's date and hash from the
# corpus header, and how many words came and went.
- name: Summarize the change
id: summary
run: |
set -eu
header() { sed -n "s/^#@$1 //p" data/dictionary.txt; }
fetched="$(header source_fetched_at)"
added="$(git diff -U0 -- data/dictionary.txt | grep -c '^+[^+#]' || true)"
removed="$(git diff -U0 -- data/dictionary.txt | grep -c '^-[^-#]' || true)"
words="$(grep -vc '^#\|^$' data/dictionary.txt)"
{
echo "date=${fetched%%T*}"
echo "body<<EOF"
echo "Regenerated \`data/dictionary.txt\` from the Wiktionary tiếng Việt dump of ${fetched}."
echo
echo "- Dump SHA-256: \`$(header source_sha256)\`"
echo "- Words in the corpus: ${words}"
echo "- Lines added: ${added}, lines removed: ${removed} (a word whose meanings changed counts as both)"
echo
echo "Opened by the monthly refresh-dictionary workflow. CI does not run on this pull request; the builder verified the database before exporting the corpus."
echo "EOF"
} >> "$GITHUB_OUTPUT"
- name: Open the pull request
uses: peter-evans/create-pull-request@v7
with:
base: dev
branch: dictionary-refresh
delete-branch: true
add-paths: data/dictionary.txt
commit-message: "chore(data): refresh the dictionary from the ${{ steps.summary.outputs.date }} dump"
title: "chore(data): refresh the dictionary from the ${{ steps.summary.outputs.date }} dump"
body: ${{ steps.summary.outputs.body }}
+6 -5
View File
@@ -1,8 +1,8 @@
# Dictionary data — never committed.
# data/viwiktionary-latest-pages-articles.xml.bz2 is the upstream dump;
# data/noitu.db is derived from it. Both are build artifacts produced by
# `make fetch-dict` and `make dict`. A plain .xml would be the dump
# decompressed by hand.
# Dictionary build artifacts. data/dictionary.txt, the corpus derived from the
# upstream dump, is committed; the dump itself
# (data/viwiktionary-latest-pages-articles.xml.bz2, from `make fetch-dict`)
# and data/noitu.db (built from the corpus by `make dict`) are not. A plain
# .xml would be the dump decompressed by hand.
data/*.bz2
data/*.bz2.part
data/*.xml
@@ -10,6 +10,7 @@ data/*.db
data/*.db-journal
data/*.db-wal
data/*.db-shm
data/*.tmp
# Go build output
/noitu-server
+10 -18
View File
@@ -1,10 +1,10 @@
# One image: the binary, the built frontend, and the derived dictionary.
#
# The upstream dump is downloaded in a builder stage and never reaches the
# final image — only the few-MB database derived from it does. That
# derived database is CC BY-SA 4.0 while the code is Apache-2.0, so it is
# copied in as its own layer alongside its licence and attribution rather than
# being embedded in the binary.
# The dictionary is built from data/dictionary.txt, the corpus committed to
# the repository, so a build downloads nothing from Wikimedia. That derived
# data is CC BY-SA 4.0 while the code is Apache-2.0, so the database is copied
# in as its own layer alongside its licence and attribution rather than being
# embedded in the binary.
# --- the frontend -----------------------------------------------------------
FROM node:24-alpine AS web
@@ -37,30 +37,22 @@ RUN CGO_ENABLED=0 go build -trimpath -o /out/build-dictionary ./cmd/build-dictio
# --- the dictionary ---------------------------------------------------------
FROM alpine:3 AS dict
# Fetched fresh, not pinned: Wikimedia regenerates the dump monthly and
# repoints `latest/`. The derived dictionary is the one thing in this image
# that cannot be rebuilt from the repository alone, so the builder records the
# SHA-256 of the file it read in the database's meta table. The Makefile uses
# the same URL for local builds, and a test asserts the two agree.
ARG DICT_URL=https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2
# Set to 1 to build from the checked-in word sample instead of downloading the
# upstream dump. That produces a playable but tiny dictionary, and exists so
# the image itself can be smoke-tested without network access.
# Set to 1 to build from the checked-in word sample instead of the corpus.
# That produces a playable but tiny dictionary, and exists so the image can be
# smoke-tested against the same words the browser suite plays.
ARG FIXTURE_DICT=0
RUN apk add --no-cache curl
WORKDIR /work
COPY --from=build /out/build-dictionary /usr/local/bin/build-dictionary
COPY testdata/fixture-words.txt ./fixture-words.txt
COPY data/dictionary.txt ./dictionary.txt
RUN set -eu; \
mkdir -p /out; \
if [ "$FIXTURE_DICT" = "1" ]; then \
build-dictionary --words ./fixture-words.txt --out /out/noitu.db --min-words 150; \
else \
curl -fsSLR -o viwiktionary-latest-pages-articles.xml.bz2 "$DICT_URL"; \
build-dictionary --dump ./viwiktionary-latest-pages-articles.xml.bz2 --out /out/noitu.db; \
build-dictionary --corpus ./dictionary.txt --out /out/noitu.db; \
fi
# --- the image --------------------------------------------------------------
+13 -7
View File
@@ -4,11 +4,13 @@
# without `make` (notably on Windows) are never blocked.
# Wikimedia regenerates the Wiktionary tiếng Việt dump monthly and repoints
# `latest/` at it; this tracks `latest/`, fetched fresh and unpinned by design,
# and the builder records the SHA-256 of what it read in the database's meta
# table. Dated directories exist should a build ever need reproducing.
# `latest/` at it; this tracks `latest/`, fetched fresh and unpinned by design.
# `refresh-dict` turns it into the committed corpus, which records the dump's
# SHA-256 in its header, and `dict` builds the database from that corpus, so
# a build never needs the download.
DICT_URL := https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2
DICT_SRC := data/viwiktionary-latest-pages-articles.xml.bz2
DICT_CORPUS := data/dictionary.txt
DICT_OUT := data/noitu.db
FIXTURE_WORDS := testdata/fixture-words.txt
FIXTURE_DB := data/fixture.db
@@ -19,11 +21,12 @@ SERVER_BIN := noitu-server
# tarball) — the same fallback -ldflags leaves unstamped Go code with anyway.
VERSION := $(shell git describe --tags --always --dirty 2>/dev/null || echo dev)
.PHONY: help fetch-dict dict fixture-dict proto proto-check server image web web-dev web-lint test test-go test-web test-e2e run clean
.PHONY: help fetch-dict refresh-dict dict fixture-dict proto proto-check server image web web-dev web-lint test test-go test-web test-e2e run clean
help:
@echo "fetch-dict download the current Wiktionary tiếng Việt dump (~61 MB) into data/"
@echo "dict derive $(DICT_OUT) from $(DICT_SRC)"
@echo "refresh-dict regenerate $(DICT_CORPUS) from $(DICT_SRC) (and build $(DICT_OUT))"
@echo "dict build $(DICT_OUT) from the committed $(DICT_CORPUS) — no download needed"
@echo "fixture-dict build the small test dictionary — no download needed"
@echo "proto regenerate the Go and JS wire types from proto/"
@echo "proto-check lint the schema and verify the committed output is in sync"
@@ -53,8 +56,11 @@ $(DICT_SRC):
@echo "$(DICT_SRC) not found — run 'make fetch-dict' first" >&2
@exit 1
dict: $(DICT_SRC)
cd server && go run ./cmd/build-dictionary --dump ../$(DICT_SRC) --out ../$(DICT_OUT)
refresh-dict: $(DICT_SRC)
cd server && go run ./cmd/build-dictionary --dump ../$(DICT_SRC) --export ../$(DICT_CORPUS) --out ../$(DICT_OUT)
dict:
cd server && go run ./cmd/build-dictionary --corpus ../$(DICT_CORPUS) --out ../$(DICT_OUT)
# The dictionary tests, end-to-end runs and CI all play against. Built from a
# checked-in word list through the same pipeline as the real one, so nothing
+18 -11
View File
@@ -176,21 +176,24 @@ is needed only to change the WebSocket schema — the generated code is committe
building and running the project does not require it.
```sh
make fetch-dict # downloads the current ~61 MB Wiktionary tiếng Việt dump into data/
make dict # derives data/noitu.db (the game's words and their meanings) from it
make dict # builds data/noitu.db (the game's words and their meanings) from the committed corpus
make test # run all tests
make run # build and start the server
```
The dump is fetched fresh, not pinned: Wikimedia regenerates it monthly and repoints
`latest/`, so two builds a month apart can differ. The database records the SHA-256 of the
file it was built from in its `meta` table. Neither the dump nor the derived database is
committed; both are build artifacts. See [`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
The dictionary is committed as text: [`data/dictionary.txt`](./data/dictionary.txt) holds
every word the builder accepted from the Wiktionary tiếng Việt dump, one per line with its
meanings, and a header naming the dump and its SHA-256. Wikimedia regenerates the dump
monthly, and the [`refresh-dictionary`](./.github/workflows/refresh-dictionary.yml) workflow
opens a pull request against `dev` with the new month's corpus on the 10th, so every change
to the word list is a reviewable diff. To refresh by hand, `make fetch-dict` (downloads the
~61 MB dump) then `make refresh-dict`. Neither the dump nor `data/noitu.db` is committed. See
[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
## Running the server
```sh
make dict # once, after fetch-dict
make dict # once
make run # builds and starts on :8080
```
@@ -247,7 +250,8 @@ dev-only URL to get wrong.
| Target | Does |
|---|---|
| `fetch-dict` | Download the current upstream Wiktionary export (~62 MB) into `data/` |
| `dict` | Derive `data/noitu.db` from the upstream export |
| `refresh-dict` | Regenerate the committed `data/dictionary.txt` from the downloaded export |
| `dict` | Build `data/noitu.db` from `data/dictionary.txt`, no download needed |
| `server` | Build the Go server binary, stamped with the version `git describe` reports |
| `image` | Build the container image, stamped the same way |
| `web` | Build the SvelteKit frontend to static assets |
@@ -270,8 +274,11 @@ dev-only URL to get wrong.
# fetch-dict (the URL is in the Makefile; -R keeps the dump's own modification time)
curl -fLR -o data/viwiktionary-latest-pages-articles.xml.bz2 "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2"
# refresh-dict (after fetch-dict; regenerates the committed corpus)
cd server && go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --export ../data/dictionary.txt --out ../data/noitu.db
# dict
cd server && go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
cd server && go run ./cmd/build-dictionary --corpus ../data/dictionary.txt --out ../data/noitu.db
# test
cd server && go vet ./... && go test ./... -race
@@ -333,13 +340,13 @@ See [`NOTICE`](./NOTICE) for the full statement.
| Artifact | License |
|---|---|
| All source code (`server/`, `web/`, `proto/`) | [Apache-2.0](./LICENSE) |
| Dictionary data (`data/noitu.db`) | [CC BY-SA 4.0](./data/LICENSE) |
| Dictionary data (`data/dictionary.txt` and the `data/noitu.db` built from it) | [CC BY-SA 4.0](./data/LICENSE) |
The dictionary is derived from the [Wiktionary tiếng Việt](https://vi.wiktionary.org/)
entries (CC BY-SA 4.0, by their contributors), read from the Wikimedia Foundation's
[monthly dump](https://dumps.wikimedia.org/viwiktionary/) of the wiki. It carries the word
forms and edited excerpts of their definitions. CC BY-SA is a **share-alike** license: any
redistribution of the derived database — including inside a container image — must carry the
redistribution of the derived corpus or database — including this repository and any container image — must carry the
same license, the attribution, and the record of modifications recorded in
[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
+20 -13
View File
@@ -13,12 +13,15 @@ the attribution and the modifications required by that license.
| Asset | [`viwiktionary-latest-pages-articles.xml.bz2`](https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2) — the Wikimedia Foundation's dump of every page of the Vietnamese Wiktionary edition with its current wikitext, ~61 MB compressed, ~43,000 pages with a Vietnamese section |
| Refresh | regenerated monthly by Wikimedia; `latest/` is repointed at each new run |
**The asset is not pinned.** Each build fetches whatever `latest/` currently points at. The
exact bytes a given `data/noitu.db` was built from are recorded in its `meta` table:
**The asset is not pinned.** Each refresh fetches whatever `latest/` currently points at,
roughly monthly. The exact bytes the committed `data/dictionary.txt` was derived from are
recorded in its `#@` header lines and carried into every `data/noitu.db` built from it, in its
`meta` table:
`source_sha256` (SHA-256 of the file as read), `source_pages` (pages with a Vietnamese
section, redirects excluded) and `source_fetched_at` (the dump's own modification time).
Two builds a month apart may differ by a few hundred words; the hash says which words and
definitions a given image shipped. Dated dumps under `dumps.wikimedia.org/viwiktionary/`
Two refreshes a month apart may differ by a few hundred words; the corpus's git history
shows exactly which words and definitions changed, and the hash says which dump a given
image shipped. Dated dumps under `dumps.wikimedia.org/viwiktionary/`
exist should a build ever need reproducing.
The attribution chain has one link before this project: Wiktionary tiếng Việt's
@@ -27,8 +30,9 @@ project's builder reads the wikitext itself.
## Modifications made by this project
`server/cmd/build-dictionary` transforms the dump into `data/noitu.db`. The derived database
is a **modified version** of the source data. Changes:
`server/cmd/build-dictionary` transforms the dump into `data/dictionary.txt`, a sorted text
corpus of the accepted words and their definition excerpts, and builds `data/noitu.db` from
that corpus. Both are a **modified version** of the source data. Changes:
1. **Section selection** — read only the Vietnamese section of each page, in either of the
two markup dialects the wiki currently uses (`{{-vie-}}` or `== {{langname|vi}} ==`).
@@ -71,9 +75,10 @@ is a **modified version** of the source data. Changes:
## Share-alike obligation
CC BY-SA 4.0 is a **share-alike** license. The derived database `data/noitu.db`, and any
distribution of it, remains licensed under **CC BY-SA 4.0** — including when it is shipped
inside a container image or any other packaged build of this project. Because the database
CC BY-SA 4.0 is a **share-alike** license. The derived corpus `data/dictionary.txt`, the
database `data/noitu.db` built from it, and any distribution of either, remain licensed under
**CC BY-SA 4.0** — including this repository, which distributes the corpus, and a container
image or any other packaged build of this project. Because the database
now redistributes edited excerpts of the entries' text and not only their headwords, the
attribution and this record of modifications travel with it wherever it goes.
@@ -85,9 +90,11 @@ keeping the two licensing regimes on separate artifacts.
## How to reproduce the derived data
```sh
make fetch-dict # downloads the current Wiktionary tiếng Việt dump (~61 MB) into data/
make dict # derives data/noitu.db from it and records the file's SHA-256 in meta
make fetch-dict # downloads the current Wiktionary tiếng Việt dump (~61 MB) into data/
make refresh-dict # derives data/dictionary.txt from it, recording the file's SHA-256 in its header
make dict # builds data/noitu.db from data/dictionary.txt
```
Neither file is committed to version control; both are build artifacts. Because `latest/`
is repointed monthly, a rebuild in a later month may not be byte-identical to an earlier one.
The corpus is committed and the dump and database are not. Because `latest/` is repointed
monthly, a refresh in a later month may not be byte-identical to an earlier one; the corpus
as committed is what every build of a given revision uses.
+36464
View File
File diff suppressed because it is too large. Load diff
+26 -12
View File
@@ -60,14 +60,14 @@ docker run -p 8080:8080 noitu:latest
`make image` runs the same build with `VERSION` filled in for you; see
"Version" below.
The build downloads the ~62 MB upstream Wiktionary export in a builder stage and
derives the ~2 MB database the game uses. Only the derived file is copied into
the final image, so the upstream export never ships. The result is a
The build turns the committed corpus, `data/dictionary.txt`, into the ~7 MB
database the game uses in a builder stage, so it downloads nothing from
Wikimedia. Only the database is copied into the final image. The result is a
distroless image of about 25 MB running as a non-root user.
Passing `--build-arg FIXTURE_DICT=1` builds the same image against the
checked-in word sample instead. It produces a playable but tiny dictionary and
exists so the image can be tested without the download; do not ship it.
checked-in word sample instead. It produces a playable but tiny dictionary;
do not ship it.
### Version
@@ -386,13 +386,27 @@ simply lost.
## Updating the dictionary
The dictionary is a build artifact, not runtime state, and the upstream dump
is fetched fresh rather than pinned: rebuilding the image picks up whatever
`dumps.wikimedia.org` currently serves under `viwiktionary/latest/`, which is
regenerated monthly, and the database's `meta` table records the SHA-256 of
the file it was built from. To update the dictionary, rebuild and redeploy.
To change the source itself, update `DICT_URL` in the `Dockerfile`, the
`Makefile` and the builder's constant (a test asserts the three agree), then
The dictionary is committed as text, `data/dictionary.txt`, and the image is
built from it, so the words a deploy ships are the words in the revision it
deploys. The upstream dump is not pinned: `dumps.wikimedia.org` regenerates
`viwiktionary/latest/` monthly, and the corpus header records the SHA-256 of
the dump it came from.
[`refresh-dictionary.yml`](../.github/workflows/refresh-dictionary.yml)
downloads the current dump on the 10th of every month, or on demand from the
Actions tab, regenerates the corpus and opens a pull request against `dev`.
Merging it, and `dev` onward to `main`, deploys the new words like any other
change. The pull request does not trigger CI, because GitHub does not run
workflows on a pull request opened with `GITHUB_TOKEN`; the builder verifies
the database before exporting the corpus, and CI runs again on the merge.
GitHub disables scheduled workflows after 60 days without a commit to the
repository, so re-enable it from the Actions tab if the pull requests stop.
To refresh by hand, run `make fetch-dict` and `make refresh-dict`, then commit
`data/dictionary.txt`.
To change the source itself, update `DICT_URL` in the `Makefile` and the
builder's constant (a test asserts the two agree), refresh the corpus, then
record what changed in `data/ATTRIBUTION.md`.
Nothing migrates, because nothing persists.
+1 -1
View File
@@ -19,7 +19,7 @@ func realDict(tb testing.TB) *dictionary.Store {
tb.Helper()
if _, err := os.Stat(realDictPath); err != nil {
tb.Skipf("real dictionary not built (run 'make fetch-dict && make dict'): %v", err)
tb.Skipf("real dictionary not built (run 'make dict'): %v", err)
}
store, err := dictionary.Open(realDictPath)
if err != nil {
+4 -4
View File
@@ -95,7 +95,7 @@ type opener struct {
// process never writes to it and never reads it again.
func Open(path string) (*Store, error) {
if _, err := os.Stat(path); err != nil {
return nil, fmt.Errorf("dictionary not found at %s — run 'make fetch-dict && make dict' first: %w", path, err)
return nil, fmt.Errorf("dictionary not found at %s — run 'make dict' first: %w", path, err)
}
db, err := sql.Open("sqlite", DSN(path, true))
@@ -170,13 +170,13 @@ func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err e
var builder string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'builder_version'`).Scan(&builder); err != nil {
if errors.Is(err, sql.ErrNoRows) {
return 0, 0, fmt.Errorf("dictionary has no builder_version: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it",
return 0, 0, fmt.Errorf("dictionary has no builder_version: it predates builder_version %s — run 'make dict' to rebuild it",
RequiredBuilderVersion)
}
return 0, 0, fmt.Errorf("read dictionary builder_version: %w", err)
}
if builder != RequiredBuilderVersion {
return 0, 0, fmt.Errorf("dictionary was built by builder_version %q, this server reads %s — run 'make fetch-dict && make dict' to rebuild it",
return 0, 0, fmt.Errorf("dictionary was built by builder_version %q, this server reads %s — run 'make dict' to rebuild it",
builder, RequiredBuilderVersion)
}
@@ -186,7 +186,7 @@ func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err e
if errors.Is(err, sql.ErrNoRows) {
// A database from before the key existed: the fix is a rebuild,
// so say so rather than naming a missing row.
return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it",
return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make dict' to rebuild it",
key, RequiredBuilderVersion)
}
return 0, fmt.Errorf("read dictionary %s: %w", key, err)
+17 -10
View File
@@ -1,8 +1,9 @@
// The upstream dump URL lives in three places: the Makefile, which
// builds it for a developer, the Dockerfile, which builds it for the image, and
// the builder, which stamps it into the database. They have to agree, or the
// container ships a dictionary nobody tested against. The docs that quote the
// URL are held to the same copy.
// The upstream dump URL lives in two places: the Makefile, which downloads it
// to refresh the committed corpus, and the builder, which stamps it into the
// corpus header and from there into the database. They have to agree, or the
// attribution record names a file nobody downloaded. The docs that quote the
// URL are held to the same copy. The image builds from the committed corpus,
// so the Dockerfile must not download the dump at all.
//
// This lives in the JavaScript suite for no better reason than that it is the
// suite that already reads other files in the repository. It is checking build
@@ -28,10 +29,16 @@ function pin(source, pattern, what) {
describe('the upstream Wiktionary dump', () => {
const makeUrl = pin(makefile, /DICT_URL\s*:?=\s*(\S+)/, 'DICT_URL in the Makefile');
const dockerUrl = pin(dockerfile, /ARG DICT_URL=(\S+)/, 'DICT_URL in the Dockerfile');
it('is the same file in both build files', () => {
expect(dockerUrl).toBe(makeUrl);
it('is never downloaded by the image build', () => {
// Only comments may mention Wikimedia; an instruction that does would
// bring back the network dependency the committed corpus removed.
const instructions = dockerfile
.split('\n')
.filter((line) => !line.trimStart().startsWith('#'))
.join('\n');
expect(instructions).not.toContain('dumps.wikimedia.org');
expect(instructions).toContain('--corpus ./dictionary.txt');
});
it('is the rolling pages-articles dump of the Vietnamese Wiktionary edition', () => {
@@ -44,8 +51,8 @@ describe('the upstream Wiktionary dump', () => {
});
it('is the URL the builder stamps into the database', () => {
// The builder records the source URL in the meta table from its own
// constant. The three copies must agree or the attribution record names
// The builder records the source URL in the corpus header from its own
// constant. The two copies must agree or the attribution record names
// a file nobody downloaded.
const builder = readFileSync(
fileURLToPath(new URL('../../server/cmd/build-dictionary/dump.go', import.meta.url)),