diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index dbffc52..a1c7e86 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -137,9 +137,10 @@ jobs: grep -qx "$required" files.txt || { echo "missing from the image: $required"; exit 1; } done - # The upstream export must never reach the final image. - if grep -q '\.jsonl$' files.txt; then - echo "the upstream wordlist leaked into the image" + # The upstream dump, compressed or not, must never reach the final + # image. + if grep -Eq '\.(bz2|xml)$' files.txt; then + echo "the upstream dump leaked into the image" exit 1 fi diff --git a/.gitignore b/.gitignore index d8ac856..8ed122e 100644 --- a/.gitignore +++ b/.gitignore @@ -1,9 +1,11 @@ # Dictionary data — never committed. -# data/kaikki-viwiktionary-vi.jsonl is the upstream export; data/noitu.db is -# derived from it. Both are build artifacts produced by `make fetch-dict` and -# `make dict`. -data/*.jsonl -data/*.jsonl.part +# data/viwiktionary-latest-pages-articles.xml.bz2 is the upstream dump; +# data/noitu.db is derived from it. Both are build artifacts produced by +# `make fetch-dict` and `make dict`. A plain .xml would be the dump +# decompressed by hand. +data/*.bz2 +data/*.bz2.part +data/*.xml data/*.db data/*.db-journal data/*.db-wal diff --git a/Dockerfile b/Dockerfile index d7d6576..d7372d4 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,7 +1,7 @@ # One image: the binary, the built frontend, and the derived dictionary. # -# The upstream wordlist is downloaded in a builder stage and never reaches the -# final image — only the ~2 MB database derived from it does. That +# The upstream dump is downloaded in a builder stage and never reaches the +# final image — only the few-MB database derived from it does. That # derived database is CC BY-SA 4.0 while the code is Apache-2.0, so it is # copied in as its own layer alongside its licence and attribution rather than # being embedded in the binary. @@ -30,15 +30,15 @@ RUN CGO_ENABLED=0 go build -trimpath -o /out/build-dictionary ./cmd/build-dictio # --- the dictionary --------------------------------------------------------- FROM alpine:3.22 AS dict -# Fetched fresh, not pinned: kaikki.org re-exports Wiktionary about weekly and -# keeps no dated snapshots. The derived wordlist is the one thing in this image +# Fetched fresh, not pinned: Wikimedia regenerates the dump monthly and +# repoints `latest/`. The derived dictionary is the one thing in this image # that cannot be rebuilt from the repository alone, so the builder records the # SHA-256 of the file it read in the database's meta table. The Makefile uses # the same URL for local builds, and a test asserts the two agree. -ARG DICT_URL=https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl +ARG DICT_URL=https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2 # Set to 1 to build from the checked-in word sample instead of downloading the -# upstream wordlist. That produces a playable but tiny dictionary, and exists so +# upstream dump. That produces a playable but tiny dictionary, and exists so # the image itself can be smoke-tested without network access. ARG FIXTURE_DICT=0 @@ -52,8 +52,8 @@ RUN set -eu; \ if [ "$FIXTURE_DICT" = "1" ]; then \ build-dictionary --words ./fixture-words.txt --out /out/noitu.db --min-words 150; \ else \ - curl -fsSL -o kaikki-viwiktionary-vi.jsonl "$DICT_URL"; \ - build-dictionary --kaikki ./kaikki-viwiktionary-vi.jsonl --out /out/noitu.db; \ + curl -fsSLR -o viwiktionary-latest-pages-articles.xml.bz2 "$DICT_URL"; \ + build-dictionary --dump ./viwiktionary-latest-pages-articles.xml.bz2 --out /out/noitu.db; \ fi # --- the image -------------------------------------------------------------- diff --git a/Makefile b/Makefile index 0f7806c..43de5fc 100644 --- a/Makefile +++ b/Makefile @@ -3,12 +3,12 @@ # Every target has a raw equivalent documented in README.md, so contributors # without `make` (notably on Windows) are never blocked. -# kaikki.org re-exports Wiktionary tiếng Việt about weekly and keeps no dated -# snapshots, so this is fetched fresh and unpinned by design: the builder -# records the SHA-256 of what it read in the database's meta table. The URL -# stays percent-encoded — the path has a space in it. -DICT_URL := https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl -DICT_SRC := data/kaikki-viwiktionary-vi.jsonl +# Wikimedia regenerates the Wiktionary tiếng Việt dump monthly and repoints +# `latest/` at it; this tracks `latest/`, fetched fresh and unpinned by design, +# and the builder records the SHA-256 of what it read in the database's meta +# table. Dated directories exist should a build ever need reproducing. +DICT_URL := https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2 +DICT_SRC := data/viwiktionary-latest-pages-articles.xml.bz2 DICT_OUT := data/noitu.db FIXTURE_WORDS := testdata/fixture-words.txt FIXTURE_DB := data/fixture.db @@ -17,7 +17,7 @@ SERVER_BIN := noitu-server .PHONY: help fetch-dict dict fixture-dict proto proto-check server web web-dev test test-go test-web test-e2e run clean help: - @echo "fetch-dict download the current upstream wordlist (~62 MB) into data/" + @echo "fetch-dict download the current Wiktionary tiếng Việt dump (~61 MB) into data/" @echo "dict derive $(DICT_OUT) from $(DICT_SRC)" @echo "fixture-dict build the small test dictionary — no download needed" @echo "proto regenerate the Go and JS wire types from proto/" @@ -30,14 +30,16 @@ help: @echo "run build and run the server locally" @echo "clean remove build artifacts (keeps downloaded dictionary)" -# Fetches whatever kaikki currently serves. -f so an HTTP error fails here -# rather than as a JSON parse error later; no resume flag, because resuming a -# file that may have changed underneath would splice two exports together; -# downloaded to a .part name and renamed only on success, so an interrupted -# fetch never leaves a truncated file for the next `make dict` to consume. +# Fetches whatever `latest/` currently points at. -f so an HTTP error fails +# here rather than as a bzip2 error later; -R keeps the server's modification +# time, which is when the dump was generated and becomes source_fetched_at; no +# resume flag, because `latest` can be repointed between two attempts and a +# resumed file would splice two months together; downloaded to a .part name +# and renamed only on success, so an interrupted fetch never leaves a truncated +# file for the next `make dict` to consume. fetch-dict: @mkdir -p data - curl -fL -o $(DICT_SRC).part $(DICT_URL) && mv $(DICT_SRC).part $(DICT_SRC) + curl -fLR -o $(DICT_SRC).part $(DICT_URL) && mv $(DICT_SRC).part $(DICT_SRC) @echo "downloaded $(DICT_SRC)" $(DICT_SRC): @@ -45,7 +47,7 @@ $(DICT_SRC): @exit 1 dict: $(DICT_SRC) - cd server && go run ./cmd/build-dictionary --kaikki ../$(DICT_SRC) --out ../$(DICT_OUT) + cd server && go run ./cmd/build-dictionary --dump ../$(DICT_SRC) --out ../$(DICT_OUT) # The dictionary tests, end-to-end runs and CI all play against. Built from a # checked-in word list through the same pipeline as the real one, so nothing diff --git a/NOTICE b/NOTICE index 24a05a3..36c7faa 100644 --- a/NOTICE +++ b/NOTICE @@ -17,16 +17,17 @@ This covers everything under server/, web/, and proto/. ------------------------------------------------------------------------------ The Vietnamese dictionary database is NOT covered by the Apache License. It is -derived from Wiktionary text and remains under the licence that text carries, -CC BY-SA 4.0: +derived from Wiktionary text — word forms and edited excerpts of their +definitions — and remains under the licence that text carries, CC BY-SA 4.0: - Affected artifacts: data/noitu.db (and data/kaikki-viwiktionary-vi.jsonl, its source) + Affected artifacts: data/noitu.db (and data/viwiktionary-latest-pages-articles.xml.bz2, + the Wikimedia dump it is built from) License text: data/LICENSE Attribution and list of changes: data/ATTRIBUTION.md Original work: Wiktionary tiếng Việt (https://vi.wiktionary.org/), by its contributors - Extraction: wiktextract, published at https://kaikki.org/viwiktionary/ - (Tatu Ylonen); fetched fresh for each build, identified + Source: https://dumps.wikimedia.org/viwiktionary/, the monthly + pages-articles dump; fetched fresh for each build, identified by the SHA-256 recorded in the database's meta table CC BY-SA 4.0 is a share-alike license. Any distribution of the derived database @@ -37,5 +38,5 @@ The two regimes are kept on separate artifacts deliberately: the database is loaded at runtime from a file and is never embedded, compiled, or linked into the Go binary. -Neither the source wordlist nor the derived database is committed to version +Neither the source dump nor the derived database is committed to version control. Both are build artifacts produced by `make fetch-dict` and `make dict`. diff --git a/README.md b/README.md index ed5af0d..e487d10 100644 --- a/README.md +++ b/README.md @@ -10,6 +10,9 @@ of the previous word**. No word may be reused. Fail to answer in time and you lo ngôn ngữ → ngữ pháp → pháp luật → luật lệ → ... ``` +Each word in the chain shows its Wiktionary meaning: the newest word's is open, and a click +on any word opens or closes its own. + Playing a word that leaves the next player nothing to answer is not itself a win. They keep the turn and lose it to the clock like any other, and are then shown a few words the position still had — or told it had none. @@ -158,17 +161,16 @@ is needed only to change the WebSocket schema — the generated code is committe building and running the project does not require it. ```sh -make fetch-dict # downloads the current ~62 MB Wiktionary export into data/ -make dict # derives data/noitu.db (the game's wordlist) from it +make fetch-dict # downloads the current ~61 MB Wiktionary tiếng Việt dump into data/ +make dict # derives data/noitu.db (the game's words and their meanings) from it make test # run all tests make run # build and start the server ``` -The export is fetched fresh, not pinned: kaikki.org re-exports Wiktionary about weekly and -keeps no dated snapshots, so two builds a week apart can differ slightly. The database -records the SHA-256 of the file it was built from in its `meta` table. Neither the export -nor the derived database is committed; both are build artifacts. See -[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md). +The dump is fetched fresh, not pinned: Wikimedia regenerates it monthly and repoints +`latest/`, so two builds a month apart can differ. The database records the SHA-256 of the +file it was built from in its `meta` table. Neither the dump nor the derived database is +committed; both are build artifacts. See [`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md). ## Running the server @@ -239,11 +241,11 @@ dev-only URL to get wrong. `make` is not installed everywhere (notably Windows). Every target is a thin wrapper: ```sh -# fetch-dict (the URL is in the Makefile; keep it percent-encoded, the path has a space) -curl -fL -o data/kaikki-viwiktionary-vi.jsonl "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl" +# fetch-dict (the URL is in the Makefile; -R keeps the dump's own modification time) +curl -fLR -o data/viwiktionary-latest-pages-articles.xml.bz2 "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2" # dict -cd server && go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db +cd server && go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db # test cd server && go vet ./... && go test ./... -race @@ -278,7 +280,7 @@ buf generate && buf lint The end-to-end suite plays against a small dictionary derived from [`testdata/fixture-words.txt`](./testdata/fixture-words.txt) through the same -builder the real one uses, so CI never downloads the upstream wordlist. +builder the real one uses, so CI never downloads the upstream dump. ## Deployment @@ -302,11 +304,12 @@ See [`NOTICE`](./NOTICE) for the full statement. | Dictionary data (`data/noitu.db`) | [CC BY-SA 4.0](./data/LICENSE) | The dictionary is derived from the [Wiktionary tiếng Việt](https://vi.wiktionary.org/) -entries (CC BY-SA 4.0, by their contributors) as extracted by -[wiktextract](https://github.com/tatuylonen/wiktextract) and published on -[kaikki.org](https://kaikki.org/viwiktionary/). CC BY-SA is a **share-alike** license: any redistribution of the derived -database — including inside a container image — must carry the same license, the attribution, -and the record of modifications recorded in [`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md). +entries (CC BY-SA 4.0, by their contributors), read from the Wikimedia Foundation's +[monthly dump](https://dumps.wikimedia.org/viwiktionary/) of the wiki. It carries the word +forms and edited excerpts of their definitions. CC BY-SA is a **share-alike** license: any +redistribution of the derived database — including inside a container image — must carry the +same license, the attribution, and the record of modifications recorded in +[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md). The database is loaded at runtime from a file and is never embedded or linked into the Go binary, keeping the two licensing regimes on separate artifacts. diff --git a/data/ATTRIBUTION.md b/data/ATTRIBUTION.md index 2ec592d..b2beb0f 100644 --- a/data/ATTRIBUTION.md +++ b/data/ATTRIBUTION.md @@ -10,31 +10,30 @@ the attribution and the modifications required by that license. |---|---| | Original work | Entries of [Wiktionary tiếng Việt](https://vi.wiktionary.org/), written by its contributors | | Original license | [CC BY-SA 4.0](https://creativecommons.org/licenses/by-sa/4.0/) (Wiktionary text is dual-licensed CC BY-SA / GFDL) — full text in [`LICENSE`](./LICENSE) | -| Extracted by | [wiktextract](https://github.com/tatuylonen/wiktextract), published on [kaikki.org](https://kaikki.org/viwiktionary/) by Tatu Ylonen; kaikki.org distributes the extracted data under the same CC BY-SA / GFDL terms as the underlying Wiktionary text | -| Asset | [`Tiếng Việt/kaikki.org-dictionary-TiếngViệt.jsonl`](https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl) — the Vietnamese-language entries of the Vietnamese Wiktionary edition, ~62 MB, ~44,000 entries | -| Refresh | kaikki re-extracts from the monthly Wikimedia dump about once a week | +| Asset | [`viwiktionary-latest-pages-articles.xml.bz2`](https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2) — the Wikimedia Foundation's dump of every page of the Vietnamese Wiktionary edition with its current wikitext, ~61 MB compressed, ~43,000 pages with a Vietnamese section | +| Refresh | regenerated monthly by Wikimedia; `latest/` is repointed at each new run | -**The asset is not pinned.** kaikki.org keeps no dated snapshots, so each build fetches the -current export. The exact bytes a given `data/noitu.db` was built from are recorded in its -`meta` table: `source_sha256` (SHA-256 of the file as read), `source_rows` (entries read) -and `source_fetched_at` (the file's modification time). Two builds a week apart may differ -by a few hundred words; the hash says which words a given image shipped. +**The asset is not pinned.** Each build fetches whatever `latest/` currently points at. The +exact bytes a given `data/noitu.db` was built from are recorded in its `meta` table: +`source_sha256` (SHA-256 of the file as read), `source_pages` (pages with a Vietnamese +section, redirects excluded) and `source_fetched_at` (the dump's own modification time). +Two builds a month apart may differ by a few hundred words; the hash says which words and +definitions a given image shipped. Dated dumps under `dumps.wikimedia.org/viwiktionary/` +exist should a build ever need reproducing. -The attribution chain has two links before this project — Wiktionary's contributors, who -wrote the entries, and wiktextract/kaikki.org, which turned the wiki markup into structured -data — and both are named here because CC BY-SA attribution belongs to the authors, not only -to the last host. kaikki.org asks users of its data to cite: -*Tatu Ylonen: Wiktextract: Wiktionary as Machine-Readable Structured Data, Proceedings of -the 13th Conference on Language Resources and Evaluation (LREC), pp. 1317–1325, Marseille, -20–25 June 2022.* +The attribution chain has one link before this project: Wiktionary tiếng Việt's +contributors, who wrote the entries. The dump is their text as the wiki stores it; this +project's builder reads the wikitext itself. ## Modifications made by this project -`server/cmd/build-dictionary` transforms the upstream export into `data/noitu.db`. The -derived database is a **modified version** of the source data. Changes: +`server/cmd/build-dictionary` transforms the dump into `data/noitu.db`. The derived database +is a **modified version** of the source data. Changes: -1. **Language selection** — kept only entries with `lang_code = "vi"`. The file is - Vietnamese-only today; any other language would be rejected and counted. +1. **Section selection** — read only the Vietnamese section of each page, in either of the + two markup dialects the wiki currently uses (`{{-vie-}}` or `== {{langname|vi}} ==`). + Pages outside the main namespace, redirects, and pages with no Vietnamese section were + skipped. Other languages' sections on the same page were not read. 2. **Length filter** — kept only words of **2 or more space-separated syllables**, as required by the nối từ game rules. Single-syllable entries were dropped. 3. **Content filter** — dropped entries containing digits or punctuation, entries using @@ -43,29 +42,40 @@ derived database is a **modified version** of the source data. Changes: Vietnamese words ("con cua") are kept. 4. **Normalization** — all words Unicode NFC-normalized, lowercased, and whitespace-collapsed. Capitalized headwords (`Hà Nội`) become lowercase entries; nothing - is removed on the basis of capitalization or part of speech. + is removed on the basis of capitalization or part of speech. Two pages whose titles + normalize to one word are merged into one entry. 5. **Spelling aliases** — added an `aliases` table mapping alternative Vietnamese spellings to canonical entries. Two kinds: competing tone placement in open oa/oe/uy syllables (`hoà` → `hòa`, `thuý` → `thúy`), and i/y alternation in Sino-Vietnamese syllables (`quí` → `quý`, `lí` → `lý`). The majority are the i/y kind. These aliases are generated by this project and are not present upstream. -6. **Added columns and tables** — `first` and `last` syllable columns, a `syllables` count, - an index on `first`, a `syllables` out-degree table, and a `meta` table recording - provenance (source URL, SHA-256, row count, fetch time, licence). All added for game - lookups. -7. **Deduplication** — an entry appears once per part of speech upstream; entries were - deduplicated by normalized word form. A generated spelling variant that is itself a real - word, or that more than one word would claim, is discarded rather than recorded as an - alias. -8. **Dropped fields** — all senses, glosses, examples, translations, pronunciations, - etymologies, categories and part-of-speech tags were discarded. The derived database - contains **only word forms**, not meanings. +6. **Definition text** — each entry carries an **excerpt and modification** of its + definitions, in a `meanings` table: the text of each `#` definition line of the + Vietnamese section, with wiki markup removed (links reduced to their display text, + formatting and references dropped, a few context and link templates unwrapped, every + other template removed whole), cut to at most **five senses of 200 characters** each, + and labelled with the Vietnamese name of the part-of-speech heading it sat under + (`danh từ`, `động từ`, …; empty when the heading was not one the builder knows). This + is not the entry as written: senses past the fifth, text past 200 characters, and + template-only definitions the builder does not understand are gone. +7. **Added columns and tables** — `first` and `last` syllable columns, a `syllables` + count, an index on `first`, a `syllables` out-degree table, the `meanings` table above, + and a `meta` table recording provenance (source URL, SHA-256, page count, fetch time, + licence). All added for game lookups. +8. **Deduplication** — a generated spelling variant that is itself a real word, or that + more than one word would claim, is discarded rather than recorded as an alias. +9. **Dropped fields** — everything else in an entry was discarded: example sentences, + quotations, translations, pronunciations, etymologies, synonyms, derived terms, + categories, images and references. The derived database carries word forms and the + definition excerpts described in item 6, and nothing else of the entry. ## Share-alike obligation CC BY-SA 4.0 is a **share-alike** license. The derived database `data/noitu.db`, and any distribution of it, remains licensed under **CC BY-SA 4.0** — including when it is shipped -inside a container image or any other packaged build of this project. +inside a container image or any other packaged build of this project. Because the database +now redistributes edited excerpts of the entries' text and not only their headwords, the +attribution and this record of modifications travel with it wherever it goes. This obligation applies to the **data only**. The source code of this project is licensed separately under Apache-2.0 (see the repository root `LICENSE` and `NOTICE`). The derived @@ -75,9 +85,9 @@ keeping the two licensing regimes on separate artifacts. ## How to reproduce the derived data ```sh -make fetch-dict # downloads kaikki's current export (~62 MB) into data/ +make fetch-dict # downloads the current Wiktionary tiếng Việt dump (~61 MB) into data/ make dict # derives data/noitu.db from it and records the file's SHA-256 in meta ``` -Neither file is committed to version control; both are build artifacts. Because the export -is refreshed upstream, a rebuild on a later day may not be byte-identical to an earlier one. +Neither file is committed to version control; both are build artifacts. Because `latest/` +is repointed monthly, a rebuild in a later month may not be byte-identical to an earlier one. diff --git a/docs/deployment.md b/docs/deployment.md index 9ad3e4c..1edac91 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -140,12 +140,13 @@ persistence, by design, in this version. ## Updating the dictionary -The wordlist is a build artifact, not runtime state, and the upstream export is -fetched fresh rather than pinned: rebuilding the image picks up whatever -kaikki.org currently serves, and the database's `meta` table records the -SHA-256 of the file it was built from. To update the dictionary, rebuild and -redeploy. To change the source itself, update `DICT_URL` in the `Dockerfile`, -the `Makefile` and the builder's constant (a test asserts the three agree), -then record what changed in `data/ATTRIBUTION.md`. +The dictionary is a build artifact, not runtime state, and the upstream dump +is fetched fresh rather than pinned: rebuilding the image picks up whatever +`dumps.wikimedia.org` currently serves under `viwiktionary/latest/`, which is +regenerated monthly, and the database's `meta` table records the SHA-256 of +the file it was built from. To update the dictionary, rebuild and redeploy. +To change the source itself, update `DICT_URL` in the `Dockerfile`, the +`Makefile` and the builder's constant (a test asserts the three agree), then +record what changed in `data/ATTRIBUTION.md`. Nothing migrates, because nothing persists. diff --git a/server/cmd/build-dictionary/dump.go b/server/cmd/build-dictionary/dump.go new file mode 100644 index 0000000..12bc6f3 --- /dev/null +++ b/server/cmd/build-dictionary/dump.go @@ -0,0 +1,270 @@ +package main + +import ( + "bufio" + "compress/bzip2" + "crypto/sha256" + "encoding/hex" + "encoding/xml" + "errors" + "fmt" + "io" + "log" + "os" + "path/filepath" + "sort" + "strings" + "time" +) + +// The upstream is the Wikimedia dump of Wiktionary tiếng Việt: every page's +// current wikitext, as one bzip2-compressed XML file regenerated monthly. +// `latest/` is a rolling pointer, fetched fresh for every build and not +// pinned, so the builder records the SHA-256 of the bytes it actually read and +// that hash is what identifies a build. Dated directories exist should +// reproducibility ever be wanted. +const dumpSourceURL = "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2" + +// dumpPage is the part of a element the builder reads. Everything +// else — contributor, timestamp, sha1 — is skipped by the decoder. +type dumpPage struct { + Title string `xml:"title"` + Ns int `xml:"ns"` + Redirect *struct { + Title string `xml:"title,attr"` + } `xml:"redirect"` + Revisions []struct { + Text string `xml:"text"` + } `xml:"revision"` +} + +// dumpProvenance identifies the bytes a build was made from. +type dumpProvenance struct { + sha256 string + // pages is the number of pages with a Vietnamese section, redirects + // excluded: the count of entries the corpus was derived from. + pages int + fetchedAt time.Time +} + +// dumpStats is what the build log reports about the dump beyond the reject +// tally: enough to see a month where a dialect vanished or a stripper rule +// started dropping everything. +type dumpStats struct { + pages int + ns0 int + redirects int + noVietnamese int + legacy int + newDialect int + bothDialects int + merged int // pages whose title normalized to a word already seen + // enders counts the {{-code-}} that closed each legacy section. Language + // codes are expected here; a heading code is one the maps are missing. + enders map[string]int + section *sectionStats +} + +// readDump streams the dump once: hashes the compressed bytes, decodes one +// page at a time, hands every Vietnamese-section title to accept() and every +// definition line to the stripper. +func readDump(path string) (map[string]entry, map[string][]sense, map[rejectReason]int, *dumpStats, dumpProvenance, error) { + var prov dumpProvenance + stats := &dumpStats{section: newSectionStats(), enders: make(map[string]int)} + + f, err := os.Open(path) + if err != nil { + return nil, nil, nil, stats, prov, fmt.Errorf("read dump: %w", err) + } + defer f.Close() + info, err := f.Stat() + if err != nil { + // A provenance row must be right or absent, never a plausible zero. + return nil, nil, nil, stats, prov, fmt.Errorf("stat dump: %w", err) + } + prov.fetchedAt = info.ModTime().UTC() + + hash := sha256.New() + compressed := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20) + if magic, err := compressed.Peek(3); err != nil || string(magic) != "BZh" { + return nil, nil, nil, stats, prov, fmt.Errorf("%s is not a bzip2 file (expected a BZh header)", path) + } + dec := xml.NewDecoder(bzip2.NewReader(compressed)) + + words := make(map[string]entry) + meanings := make(map[string][]sense) + rejects := make(map[rejectReason]int) + lastTitle := "" + + for { + tok, err := dec.Token() + if err != nil { + if errors.Is(err, io.EOF) { + break + } + return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err) + } + start, ok := tok.(xml.StartElement) + if !ok || start.Name.Local != "page" { + continue + } + var page dumpPage + if err := dec.DecodeElement(&page, &start); err != nil { + return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err) + } + lastTitle = page.Title + stats.pages++ + if page.Ns != 0 { + continue + } + stats.ns0++ + if page.Redirect != nil { + // The target page is read on its own and lowercased by accept(), + // so a case-only redirect adds nothing and any other redirect is + // an alternative title the wiki itself does not define. + stats.redirects++ + continue + } + if len(page.Revisions) == 0 { + return nil, nil, nil, stats, prov, fmt.Errorf("%s: page %q has no revision text", path, page.Title) + } + text := page.Revisions[len(page.Revisions)-1].Text + + section, dialect, both, ender := vietnameseSection(text) + if dialect == "" { + stats.noVietnamese++ + rejects[rejectNotVietnamese]++ + continue + } + if both { + stats.bothDialects++ + } + if dialect == "legacy" { + stats.legacy++ + } else { + stats.newDialect++ + } + prov.pages++ + if ender != "" { + stats.enders[ender]++ + } + + word, syllables, reason, ok := accept(page.Title) + if !ok { + rejects[reason]++ + continue + } + // After accept, so the definition counters describe words that land. + senses := definitions(section, stats.section) + if _, seen := words[word]; seen { + // Two pages whose titles normalize to one word (Việt Nam and + // việt nam): one entry, senses in page order, one cap. + stats.merged++ + } + words[word] = entry{ + word: word, + first: syllables[0], + last: syllables[len(syllables)-1], + syllables: len(syllables), + } + if len(senses) > 0 { + merged := append(meanings[word], senses...) + if len(merged) > maxSenses { + merged = merged[:maxSenses] + } + meanings[word] = merged + } + } + + // The XML decoder stops at the root's close tag; the hash must cover the + // whole file, trailing bytes included. + if _, err := io.Copy(io.Discard, compressed); err != nil { + return nil, nil, nil, stats, prov, fmt.Errorf("%s: %w", path, err) + } + prov.sha256 = hex.EncodeToString(hash.Sum(nil)) + + return words, meanings, rejects, stats, prov, nil +} + +// dumpError names where a stream failed: the last page fully read and the +// decompressed offset, so a truncated download and a malformed page are told +// apart by the message alone. +func dumpError(path, lastTitle string, offset int64, err error) error { + where := "before the first page" + if lastTitle != "" { + where = fmt.Sprintf("after page %q", lastTitle) + } + if errors.Is(err, io.ErrUnexpectedEOF) || strings.Contains(err.Error(), "unexpected EOF") { + return fmt.Errorf("%s: stream ends %s (decompressed offset %d): truncated download? %w", path, where, offset, err) + } + return fmt.Errorf("%s: %s (decompressed offset %d): %w", path, where, offset, err) +} + +// logDumpStats writes the build log lines that describe what the dump held. +func logDumpStats(stats *dumpStats) { + s := stats.section + logf := log.Printf + logf("pages %d, in the main namespace %d, redirects skipped %d, without a Vietnamese section %d", + stats.pages, stats.ns0, stats.redirects, stats.noVietnamese) + logf("Vietnamese sections: legacy {{-vie-}} %d, new == {{langname|vi}} == %d, pages with both %d, titles merged %d", + stats.legacy, stats.newDialect, stats.bothDialects, stats.merged) + logf("parts of speech: %s", formatTally(s.pos, 0)) + if len(s.unmappedPos) > 0 { + logf("headings without a label: %s", formatTally(s.unmappedPos, 20)) + } + // Language codes belong here. A heading code in this list is one the maps + // do not know, and it has been cutting sections short. + logf("codes that ended a legacy section, commonest: %s", formatTally(stats.enders, 15)) + logf("definitions kept %d (cut at %d characters: %d), dropped as empty after stripping %d", + s.defsKept, maxGlossRunes, s.defsCut, s.defsEmpty) + if len(s.dropped) > 0 { + logf("templates dropped whole, commonest: %s", formatTally(s.dropped, 10)) + } +} + +// formatTally renders counts on one log line, largest first, cut to the top +// n entries when n is positive. +func formatTally(counts map[string]int, n int) string { + type kv struct { + name string + count int + } + tally := make([]kv, 0, len(counts)) + for name, count := range counts { + if name == "" { + name = "(none)" + } + tally = append(tally, kv{name, count}) + } + sort.Slice(tally, func(i, j int) bool { + if tally[i].count != tally[j].count { + return tally[i].count > tally[j].count + } + return tally[i].name < tally[j].name + }) + if n > 0 && len(tally) > n { + tally = tally[:n] + } + parts := make([]string, len(tally)) + for i, t := range tally { + parts[i] = fmt.Sprintf("%s %d", t.name, t.count) + } + return strings.Join(parts, ", ") +} + +// dumpSourceSpec describes a dump build for the meta table. With nothing +// pinned upstream, the hash and page count of the bytes read are the +// provenance. +func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec { + return sourceSpec{ + table: "dump:" + filepath.Base(path), + url: dumpSourceURL, + license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)", + attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.", + extra: [][2]string{ + {"source_sha256", prov.sha256}, + {"source_pages", fmt.Sprint(prov.pages)}, + {"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)}, + }, + } +} diff --git a/server/cmd/build-dictionary/dump_test.go b/server/cmd/build-dictionary/dump_test.go new file mode 100644 index 0000000..cfd3778 --- /dev/null +++ b/server/cmd/build-dictionary/dump_test.go @@ -0,0 +1,149 @@ +package main + +import ( + "crypto/sha256" + "encoding/hex" + "os" + "path/filepath" + "strings" + "testing" +) + +// miniDump is a twelve-page stand-in for the Wikimedia dump, committed +// compressed beside its readable source. Go has no bzip2 writer, so the .bz2 +// is regenerated by hand: bzip2 -k9 testdata/mini-dump.xml. +const miniDump = "testdata/mini-dump.xml.bz2" + +// miniDumpCut is the same XML cut mid-page and then compressed: a valid bzip2 +// stream whose XML ends early, as distinct from a truncated download. +const miniDumpCut = "testdata/mini-dump-cut.xml.bz2" + +func TestReadDumpKeepsVietnameseSections(t *testing.T) { + words, meanings, rejects, stats, prov, err := readDump(miniDump) + if err != nil { + t.Fatal(err) + } + + var got []string + for w := range words { + got = append(got, w) + } + assertSameStrings(t, got, []string{"pháp luật", "hòa bình", "luật lệ", "ngôn ngữ", "ngữ pháp", "vô tuyến điện"}) + + if stats.pages != 12 || stats.ns0 != 11 || stats.redirects != 1 || stats.noVietnamese != 1 { + t.Errorf("pages %d ns0 %d redirects %d noVietnamese %d, want 12 11 1 1", + stats.pages, stats.ns0, stats.redirects, stats.noVietnamese) + } + if stats.legacy != 7 || stats.newDialect != 2 || stats.bothDialects != 0 { + t.Errorf("legacy %d new %d both %d, want 7 2 0", stats.legacy, stats.newDialect, stats.bothDialects) + } + if prov.pages != 9 { + t.Errorf("source pages = %d, want 9 (Vietnamese sections, redirect excluded)", prov.pages) + } + if stats.merged != 1 { + t.Errorf("merged = %d, want 1 (Hòa Bình and hòa bình)", stats.merged) + } + if rejects[rejectNotVietnamese] != 1 || rejects[rejectTooShort] != 1 || rejects[rejectDigit] != 1 { + t.Errorf("rejects = %v, want one each of not-Vietnamese, too-short, digit", rejects) + } + + assertSenses(t, meanings["pháp luật"], []sense{ + {"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."}, + {"danh từ", "(nghĩa rộng) Kỷ cương nói chung."}, + {"động từ", "(hiếm) Xử theo luật."}, + }) + // Capitalized page first, lowercase page second: senses in page order, + // the new-dialect place first. + assertSenses(t, meanings["hòa bình"], []sense{ + {"danh từ riêng", "tỉnh, Việt Nam."}, + {"danh từ", "Tình trạng không có chiến tranh."}, + {"tính từ", "Yên ổn."}, + }) + assertSenses(t, meanings["ngôn ngữ"], []sense{{"danh từ", "Hệ thống những âm, từ và quy tắc kết hợp chúng."}}) + if _, has := meanings["luật lệ"]; has { + t.Error("a definition that is only an unknown template produced a sense") + } + if stats.section.dropped["rfdef"] != 1 || stats.section.defsEmpty != 1 { + t.Errorf("dropped = %v empty = %d, want rfdef 1 and 1", stats.section.dropped, stats.section.defsEmpty) + } + if stats.section.pos["noun"] == 0 || stats.section.pos["pr-noun"] != 2 || stats.section.pos["n"] != 1 { + t.Errorf("pos tally = %v", stats.section.pos) + } +} + +func TestReadDumpHashesTheBytesItRead(t *testing.T) { + _, _, _, _, prov, err := readDump(miniDump) + if err != nil { + t.Fatal(err) + } + raw, err := os.ReadFile(miniDump) + if err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(raw) + if prov.sha256 != hex.EncodeToString(sum[:]) { + t.Errorf("sha256 = %s, want %s (the whole file)", prov.sha256, hex.EncodeToString(sum[:])) + } + if prov.fetchedAt.IsZero() { + t.Error("fetchedAt is zero") + } +} + +func TestReadDumpRejectsNonBzip2(t *testing.T) { + path := filepath.Join(t.TempDir(), "dump.xml.bz2") + if err := os.WriteFile(path, []byte(""), 0o644); err != nil { + t.Fatal(err) + } + _, _, _, _, _, err := readDump(path) + if err == nil || !strings.Contains(err.Error(), "not a bzip2 file") { + t.Fatalf("err = %v, want a message naming the missing bzip2 header", err) + } +} + +func TestReadDumpRejectsTruncatedDownload(t *testing.T) { + raw, err := os.ReadFile(miniDump) + if err != nil { + t.Fatal(err) + } + path := filepath.Join(t.TempDir(), "dump.xml.bz2") + if err := os.WriteFile(path, raw[:len(raw)/2], 0o644); err != nil { + t.Fatal(err) + } + _, _, _, _, _, err = readDump(path) + if err == nil { + t.Fatal("a half-downloaded dump was read without error") + } + if !strings.Contains(err.Error(), "truncated") { + t.Errorf("err = %v, want it to suggest a truncated download", err) + } +} + +func TestReadDumpRejectsStreamEndingMidPage(t *testing.T) { + _, _, _, _, _, err := readDump(miniDumpCut) + if err == nil { + t.Fatal("an XML stream ending mid-page was read without error") + } + if !strings.Contains(err.Error(), `after page "hello world"`) { + t.Errorf("err = %v, want it to name the last page fully read", err) + } +} + +func TestRunFailsBelowMinPages(t *testing.T) { + cfg := config{ + dump: miniDump, + out: filepath.Join(t.TempDir(), "noitu.db"), + minWords: 1, + minPages: 20000, + } + err := run(cfg) + if err == nil || !strings.Contains(err.Error(), "pages have a Vietnamese section") { + t.Fatalf("err = %v, want the page floor named", err) + } +} + +func TestFormatTally(t *testing.T) { + got := formatTally(map[string]int{"b": 2, "a": 2, "": 5, "c": 1}, 3) + if want := "(none) 5, a 2, b 2"; got != want { + t.Errorf("formatTally = %q, want %q", got, want) + } +} diff --git a/server/cmd/build-dictionary/kaikki_list.go b/server/cmd/build-dictionary/kaikki_list.go deleted file mode 100644 index 1f52b14..0000000 --- a/server/cmd/build-dictionary/kaikki_list.go +++ /dev/null @@ -1,168 +0,0 @@ -package main - -import ( - "bufio" - "bytes" - "crypto/sha256" - "encoding/hex" - "encoding/json" - "errors" - "fmt" - "io" - "os" - "path/filepath" - "sort" - "strings" - "time" -) - -// The upstream is kaikki.org's wiktextract export of Wiktionary tiếng Việt: -// one JSON object per entry, refreshed from the monthly Wikimedia dump about -// once a week. The file is fetched fresh for every build and is not pinned — -// there is no archived snapshot to pin to — so the builder records the SHA-256 -// of the bytes it actually read, and that hash is what identifies a build. -// -// The URL stays percent-encoded: the path has a space in it, and both make -// and sh would otherwise split it. -const kaikkiSourceURL = "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl" - -// kaikkiRow is the part of a wiktextract entry the game cares about. Every -// other field — senses, translations, categories — is skipped by the decoder. -type kaikkiRow struct { - Word string `json:"word"` - Pos string `json:"pos"` - LangCode string `json:"lang_code"` -} - -// kaikkiProvenance identifies the bytes a build was made from. -type kaikkiProvenance struct { - sha256 string - rows int - fetchedAt time.Time -} - -// readKaikkiList streams the export, keeps Vietnamese-language entries and -// hands their word forms to accept(). Part of speech is tallied for the build -// log but never filters: the owner's decision that capitalization removes no -// word applies equally to the "name" tag. -// -// Lines are read with bufio.Reader rather than bufio.Scanner because a row -// carries every sense and translation of its entry and can run to hundreds of -// kilobytes; a scanner's fixed cap would be a guess that eventually fails. -func readKaikkiList(path string) (map[string]entry, map[rejectReason]int, map[string]int, kaikkiProvenance, error) { - var prov kaikkiProvenance - - f, err := os.Open(path) - if err != nil { - return nil, nil, nil, prov, fmt.Errorf("read kaikki export: %w", err) - } - defer f.Close() - info, err := f.Stat() - if err != nil { - // A provenance row must be right or absent, never a plausible zero. - return nil, nil, nil, prov, fmt.Errorf("stat kaikki export: %w", err) - } - prov.fetchedAt = info.ModTime().UTC() - - hash := sha256.New() - reader := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20) - - words := make(map[string]entry) - rejects := make(map[rejectReason]int) - pos := make(map[string]int) - - lineNo := 0 - for { - line, err := reader.ReadBytes('\n') - if len(line) > 0 { - lineNo++ - if trimmed := bytes.TrimSpace(line); len(trimmed) > 0 { - // A bare literal such as null would decode into an empty row - // and be miscounted as a foreign-language entry; only objects - // are entries. - if trimmed[0] != '{' { - return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: not a JSON object", path, lineNo) - } - var row kaikkiRow - if err := json.Unmarshal(trimmed, &row); err != nil { - return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: %w", path, lineNo, err) - } - prov.rows++ - if row.LangCode != "vi" { - rejects[rejectNotVietnamese]++ - } else { - pos[row.Pos]++ - if word, syllables, reason, ok := accept(row.Word); !ok { - rejects[reason]++ - } else { - words[word] = entry{ - word: word, - first: syllables[0], - last: syllables[len(syllables)-1], - syllables: len(syllables), - } - } - } - } - } - if err != nil { - if errors.Is(err, io.EOF) { - break - } - // The failure is on the line being read: the one just counted if - // a partial line came back with the error, otherwise the next. - failed := lineNo + 1 - if len(line) > 0 { - failed = lineNo - } - return nil, nil, nil, prov, fmt.Errorf("%s:%d: %w", path, failed, err) - } - } - prov.sha256 = hex.EncodeToString(hash.Sum(nil)) - - return words, rejects, pos, prov, nil -} - -// formatPosTally renders the part-of-speech counts on one log line, largest -// first, so the build log says what kind of entries the export held. -func formatPosTally(pos map[string]int) string { - type kv struct { - name string - count int - } - tally := make([]kv, 0, len(pos)) - for name, count := range pos { - if name == "" { - name = "(none)" - } - tally = append(tally, kv{name, count}) - } - sort.Slice(tally, func(i, j int) bool { - if tally[i].count != tally[j].count { - return tally[i].count > tally[j].count - } - return tally[i].name < tally[j].name - }) - parts := make([]string, len(tally)) - for i, t := range tally { - parts[i] = fmt.Sprintf("%s %d", t.name, t.count) - } - return strings.Join(parts, ", ") -} - -// kaikkiSourceSpec describes a kaikki build for the meta table. With no commit -// or checksum pinned upstream, the hash and row count of the bytes read are the -// provenance. -func kaikkiSourceSpec(path string, prov kaikkiProvenance) sourceSpec { - return sourceSpec{ - table: "kaikki:" + filepath.Base(path), - url: kaikkiSourceURL, - license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)", - attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.", - extra: [][2]string{ - {"source_sha256", prov.sha256}, - {"source_rows", fmt.Sprint(prov.rows)}, - {"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)}, - }, - } -} diff --git a/server/cmd/build-dictionary/kaikki_list_test.go b/server/cmd/build-dictionary/kaikki_list_test.go deleted file mode 100644 index 54e3bd4..0000000 --- a/server/cmd/build-dictionary/kaikki_list_test.go +++ /dev/null @@ -1,243 +0,0 @@ -package main - -import ( - "crypto/sha256" - "encoding/hex" - "os" - "path/filepath" - "strings" - "testing" -) - -// fixtureKaikki writes a miniature stand-in for the kaikki export: the same -// JSONL shape, a handful of rows. -func fixtureKaikki(t *testing.T, lines ...string) string { - t.Helper() - path := filepath.Join(t.TempDir(), "kaikki.jsonl") - if err := os.WriteFile(path, []byte(strings.Join(lines, "\n")+"\n"), 0o644); err != nil { - t.Fatal(err) - } - return path -} - -func defaultKaikkiLines() []string { - return []string{ - `{"word": "Hà Nội", "pos": "name", "lang_code": "vi", "senses": [{"glosses": ["thủ đô"]}]}`, // capitalized, name POS — kept, lowercased - `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`, - `{"word": "học sinh", "pos": "verb", "lang_code": "vi"}`, // same word, second POS — kept once - `{"word": "student", "pos": "noun", "lang_code": "en"}`, // not Vietnamese-language — rejected and counted - `{"word": "pháp", "pos": "noun", "lang_code": "vi"}`, // single syllable — rejected downstream - ``, - } -} - -func TestKaikkiListKeepsVietnameseEntries(t *testing.T) { - words, rejects, pos, prov, err := readKaikkiList(fixtureKaikki(t, defaultKaikkiLines()...)) - if err != nil { - t.Fatal(err) - } - - var got []string - for w := range words { - got = append(got, w) - } - assertSameStrings(t, got, []string{"hà nội", "học sinh"}) - - if n := rejects[rejectNotVietnamese]; n != 1 { - t.Errorf("non-Vietnamese rejects = %d, want 1", n) - } - if n := rejects[rejectTooShort]; n != 1 { - t.Errorf("too-short rejects = %d, want 1 (pháp)", n) - } - if pos["noun"] != 2 || pos["verb"] != 1 || pos["name"] != 1 { - t.Errorf("pos tally = %v, want noun 2, verb 1, name 1 (en row excluded)", pos) - } - if prov.rows != 5 { - t.Errorf("rows = %d, want 5", prov.rows) - } -} - -func TestKaikkiListHashesTheBytesItRead(t *testing.T) { - path := fixtureKaikki(t, defaultKaikkiLines()...) - _, _, _, prov, err := readKaikkiList(path) - if err != nil { - t.Fatal(err) - } - raw, err := os.ReadFile(path) - if err != nil { - t.Fatal(err) - } - sum := sha256.Sum256(raw) - if want := hex.EncodeToString(sum[:]); prov.sha256 != want { - t.Errorf("sha256 = %s, want %s", prov.sha256, want) - } - if prov.fetchedAt.IsZero() { - t.Error("fetchedAt is zero, want the file's modification time") - } -} - -// fixtureKaikkiRaw writes exact bytes, for the shapes fixtureKaikki's trailing -// newline would hide. -func fixtureKaikkiRaw(t *testing.T, raw string) string { - t.Helper() - path := filepath.Join(t.TempDir(), "kaikki.jsonl") - if err := os.WriteFile(path, []byte(raw), 0o644); err != nil { - t.Fatal(err) - } - return path -} - -func TestKaikkiListHandlesDownloadShapes(t *testing.T) { - cases := []struct { - name string - raw string - wantWords int - wantRows int - wantErr string - }{ - {"final line without newline", - `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh mì", "pos": "noun", "lang_code": "vi"}`, - 2, 2, ""}, - {"HTTP error page instead of JSONL", "503\n", 0, 0, ":1: malformed"}, - {"cut mid-line", `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh`, 0, 0, ":2: malformed"}, - {"bare JSON literal", "null\n", 0, 0, ":1: malformed"}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - words, _, _, prov, err := readKaikkiList(fixtureKaikkiRaw(t, tc.raw)) - if tc.wantErr != "" { - if err == nil || !strings.Contains(err.Error(), tc.wantErr) { - t.Fatalf("err = %v, want one containing %q", err, tc.wantErr) - } - return - } - if err != nil { - t.Fatal(err) - } - if len(words) != tc.wantWords || prov.rows != tc.wantRows { - t.Errorf("words=%d rows=%d, want %d/%d", len(words), prov.rows, tc.wantWords, tc.wantRows) - } - }) - } -} - -func TestKaikkiListNamesMalformedLine(t *testing.T) { - path := fixtureKaikki(t, - `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`, - `{"word": "broken"`, - ) - _, _, _, _, err := readKaikkiList(path) - if err == nil { - t.Fatal("malformed line was skipped, want error") - } - if !strings.Contains(err.Error(), ":2:") { - t.Errorf("error does not name line 2: %v", err) - } -} - -func TestKaikkiListReadsLongLines(t *testing.T) { - // A real row carries every sense and translation and can exceed any - // scanner buffer; the reader must not have a line cap. - padding := strings.Repeat("x", 2<<20) - path := fixtureKaikki(t, `{"word": "học sinh", "pos": "noun", "lang_code": "vi", "note": "`+padding+`"}`) - words, _, _, _, err := readKaikkiList(path) - if err != nil { - t.Fatal(err) - } - if _, ok := words["học sinh"]; !ok { - t.Error("word on a 2 MB line was lost") - } -} - -func TestFormatPosTally(t *testing.T) { - got := formatPosTally(map[string]int{"verb": 2, "noun": 5, "": 1}) - if want := "noun 5, verb 2, (none) 1"; got != want { - t.Errorf("tally = %q, want %q", got, want) - } -} - -func TestKaikkiBuildRecordsProvenance(t *testing.T) { - out := filepath.Join(t.TempDir(), "noitu.db") - path := fixtureKaikki(t, defaultKaikkiLines()...) - if err := run(config{kaikki: path, out: out, minWords: 1}); err != nil { - t.Fatalf("run: %v", err) - } - db := openOut(t, out) - - raw, _ := os.ReadFile(path) - sum := sha256.Sum256(raw) - want := map[string]string{ - "source_url": kaikkiSourceURL, - "source_sha256": hex.EncodeToString(sum[:]), - "source_rows": "5", - "source_license": "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)", - "word_count": "2", - } - for key, value := range want { - var got string - if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&got); err != nil { - t.Errorf("meta[%q] missing: %v", key, err) - continue - } - if got != value { - t.Errorf("meta[%q] = %q, want %q", key, got, value) - } - } - for _, gone := range []string{"source_commit", "sources_kept", "sources_excluded"} { - var got string - if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, gone).Scan(&got); err == nil { - t.Errorf("meta[%q] = %q, want absent", gone, got) - } - } - var fetched string - if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_fetched_at'`).Scan(&fetched); err != nil || fetched == "" { - t.Errorf("meta[source_fetched_at] missing or empty: %v", err) - } -} - -func TestRunInputSelection(t *testing.T) { - kaikki := fixtureKaikki(t, defaultKaikkiLines()...) - words := fixtureKaikki(t, "học sinh") - out := filepath.Join(t.TempDir(), "noitu.db") - - cases := []struct { - name string - cfg config - wantErr string - }{ - {"both inputs", config{kaikki: kaikki, words: words, out: out, minWords: 1}, "mutually exclusive"}, - {"neither input", config{out: out, minWords: 1}, "no input given"}, - {"missing kaikki file", config{kaikki: filepath.Join(t.TempDir(), "absent.jsonl"), out: out, minWords: 1}, "make fetch-dict"}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - err := run(tc.cfg) - if err == nil { - t.Fatal("run succeeded, want error") - } - if !strings.Contains(err.Error(), tc.wantErr) { - t.Errorf("error %q does not mention %q", err, tc.wantErr) - } - }) - } -} - -// A fixture is hand-written data; its database must not claim the upstream's -// licence, because the server logs whatever the meta table says. -func TestWordListBuildRecordsNoUpstreamLicense(t *testing.T) { - out := filepath.Join(t.TempDir(), "noitu.db") - if err := run(config{words: fixtureKaikki(t, "học sinh", "bánh mì"), out: out, minWords: 1}); err != nil { - t.Fatalf("run: %v", err) - } - var license, url string - db := openOut(t, out) - if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil { - t.Fatal(err) - } - if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_url'`).Scan(&url); err != nil { - t.Fatal(err) - } - if strings.Contains(license, "CC BY-SA") || url != "" { - t.Errorf("fixture build claims upstream provenance: license=%q url=%q", license, url) - } -} diff --git a/server/cmd/build-dictionary/main.go b/server/cmd/build-dictionary/main.go index 65818bb..4049f5c 100644 --- a/server/cmd/build-dictionary/main.go +++ b/server/cmd/build-dictionary/main.go @@ -1,19 +1,20 @@ -// Command build-dictionary derives the game's wordlist from kaikki.org's -// wiktextract export of Wiktionary tiếng Việt. +// Command build-dictionary derives the game's wordlist and word meanings from +// the Wikimedia dump of Wiktionary tiếng Việt. // -// The upstream is a ~62 MB JSONL file: one entry per line with its senses, -// translations and part of speech. The game needs only Vietnamese word forms -// of at least two syllables, indexed by first and last syllable. This tool -// performs that reduction and records provenance in a meta table — including -// the SHA-256 of the file it read, since the upstream is fetched fresh for -// every build rather than pinned. +// The upstream is a ~61 MB bzip2-compressed XML file: every page of the wiki +// with its current wikitext, regenerated monthly. The game needs the +// Vietnamese word forms of at least two syllables, indexed by first and last +// syllable, and the plain text of each word's definitions. This tool performs +// that reduction and records provenance in a meta table — including the +// SHA-256 of the file it read, since the upstream is fetched fresh for every +// build rather than pinned. // // The derived database is a modified version of CC BY-SA 4.0 licensed data. // See data/ATTRIBUTION.md. // // Usage: // -// go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db +// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db package main import ( @@ -27,32 +28,45 @@ import ( "sort" "strings" "time" + "unicode/utf8" _ "modernc.org/sqlite" ) // builderVer changes whenever the meta table's contract does, so two databases // with different provenance rows never claim the same builder. -const builderVer = "4" +const builderVer = "5" + +// minMeaningCoverage is the share of words a dump build must carry a meaning +// for. The 2026-09-01 dump measured well above it; the floor exists to catch a +// stripper or section scanner that suddenly returns nothing, not to demand +// quality. Fixture builds are exempt: their meanings are hand-written. +const minMeaningCoverage = 0.6 type config struct { - // kaikki is the corpus: the upstream wiktextract JSONL export. - kaikki string - // words is an alternative source: a plain list, one word per line, used to - // build a small fixture database without the upstream download. + // dump is the corpus: the Wikimedia pages-articles export. + dump string + // words is an alternative source: a plain list, one word per line with an + // optional tab-separated meaning column, used to build a small fixture + // database without the upstream download. words string out string minWords int + // minPages is the floor on pages with a Vietnamese section. Distinct from + // minWords so a scanner that silently misses a dialect is caught before + // the word floor is. + minPages int } func main() { log.SetFlags(0) var cfg config - flag.StringVar(&cfg.kaikki, "kaikki", "", "upstream kaikki.org wiktextract JSONL export to read") - flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the upstream export (one word per line, # comments)") + flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read") + flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)") flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write") flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering") + flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section") flag.Parse() if err := run(cfg); err != nil { @@ -64,40 +78,47 @@ func run(cfg config) error { // Exactly one input. Picking silently between two would let a stray flag // ship a corpus nobody meant to build. switch { - case cfg.kaikki == "" && cfg.words == "": - return errors.New("no input given: pass --kaikki (the corpus) or --words (a plain list)") - case cfg.kaikki != "" && cfg.words != "": - return errors.New("--kaikki and --words are mutually exclusive") - case cfg.kaikki != "": - return runFromKaikkiList(cfg) + case cfg.dump == "" && cfg.words == "": + return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)") + case cfg.dump != "" && cfg.words != "": + return errors.New("--dump and --words are mutually exclusive") + case cfg.dump != "": + return runFromDump(cfg) default: return runFromWordList(cfg) } } -// runFromKaikkiList derives the database from the kaikki.org export, keeping -// Vietnamese-language entries and recording the hash of the bytes it read. -func runFromKaikkiList(cfg config) error { - if _, err := os.Stat(cfg.kaikki); err != nil { - return fmt.Errorf("kaikki export not found at %s — run 'make fetch-dict' first: %w", cfg.kaikki, err) +// runFromDump derives the database from the Wikimedia dump, keeping every +// page with a Vietnamese section and recording the hash of the bytes it read. +func runFromDump(cfg config) error { + if _, err := os.Stat(cfg.dump); err != nil { + return fmt.Errorf("dump not found at %s — run 'make fetch-dict' first: %w", cfg.dump, err) } - words, rejects, pos, prov, err := readKaikkiList(cfg.kaikki) + started := time.Now() + words, meanings, rejects, stats, prov, err := readDump(cfg.dump) if err != nil { return err } + logDumpStats(stats) logRejects(rejects) - log.Printf("parts of speech: %s", formatPosTally(pos)) - log.Printf("accepted %d distinct words from %s (%d rows, sha256 %s)", len(words), cfg.kaikki, prov.rows, prov.sha256) + if prov.pages < cfg.minPages { + return fmt.Errorf("only %d pages have a Vietnamese section, expected at least %d — "+ + "the dump's markup may have changed", prov.pages, cfg.minPages) + } + log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s", + len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second)) - return finish(cfg, words, kaikkiSourceSpec(cfg.kaikki, prov)) + return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true) } // finish is the tail every input mode shares: the size floor, alias // generation, the atomic write and the re-read verification. Keeping it in one // place is what stops a fixture from drifting into a different shape from the -// database production loads. -func finish(cfg config, words map[string]entry, src sourceSpec) error { +// database production loads. requireCoverage applies the meaning-coverage +// floor, which only a corpus build can be held to. +func finish(cfg config, words map[string]entry, meanings map[string][]sense, src sourceSpec, requireCoverage bool) error { if len(words) < cfg.minWords { return fmt.Errorf("only %d words survived filtering, expected at least %d — "+ "the source content may have changed", len(words), cfg.minWords) @@ -106,10 +127,10 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error { aliases, collisions := buildAliases(words) log.Printf("generated %d spelling aliases (%d skipped as ambiguous or already real words)", len(aliases), collisions) - if err := write(cfg.out, words, aliases, src); err != nil { + if err := write(cfg.out, words, meanings, aliases, src); err != nil { return err } - if err := verify(cfg.out, cfg.minWords); err != nil { + if err := verify(cfg.out, cfg.minWords, requireCoverage); err != nil { return fmt.Errorf("output failed verification: %w", err) } @@ -118,13 +139,17 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error { } // runFromWordList derives a database from a plain list of words instead of the -// upstream export. +// dump. // // It exists so tests and CI have a real dictionary to play against without the // upstream download. The filtering, alias generation, writing and verification // below are the same functions the real build uses — only the source of the // raw strings differs — so a fixture cannot drift into being shaped // differently from what production loads. +// +// A line is `word`, or `wordsensesense…` where a sense is +// `pos|gloss` or just `gloss`. The pipe never survives the stripper, so it is +// a safe separator for hand-written meanings. func runFromWordList(cfg config) error { raw, err := os.ReadFile(cfg.words) if err != nil { @@ -132,6 +157,7 @@ func runFromWordList(cfg config) error { } words := make(map[string]entry) + meanings := make(map[string][]sense) rejects := make(map[rejectReason]int) for line := range strings.Lines(string(raw)) { @@ -139,7 +165,8 @@ func runFromWordList(cfg config) error { if line == "" || strings.HasPrefix(line, "#") { continue } - word, syllables, reason, ok := accept(line) + cells := strings.Split(line, "\t") + word, syllables, reason, ok := accept(cells[0]) if !ok { rejects[reason]++ continue @@ -150,26 +177,54 @@ func runFromWordList(cfg config) error { last: syllables[len(syllables)-1], syllables: len(syllables), } + if senses := parseSenses(cells[1:]); len(senses) > 0 { + meanings[word] = senses + } } logRejects(rejects) - log.Printf("accepted %d distinct words from %s", len(words), cfg.words) + log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words) // The source spec is what lands in the meta table. Naming the list rather // than a table makes it obvious in the output which build produced a given // database — and a hand-written list carries no upstream licence, so the // fixture must not claim one. - return finish(cfg, words, sourceSpec{ + return finish(cfg, words, meanings, sourceSpec{ table: "wordlist:" + filepath.Base(cfg.words), license: "none: hand-written fixture wordlist, no upstream data", attribution: "Fixture written by this project; no third-party attribution applies.", - }) + }, false) +} + +// parseSenses reads the tab-separated meaning cells of a fixture line. A cell +// is `pos|gloss` or a bare gloss; empty cells are skipped and the cap applies +// as it does to the dump. +func parseSenses(cells []string) []sense { + var senses []sense + for _, cell := range cells { + cell = strings.TrimSpace(cell) + if cell == "" { + continue + } + s := sense{gloss: cell} + if pos, gloss, ok := strings.Cut(cell, "|"); ok { + s = sense{pos: strings.TrimSpace(pos), gloss: strings.TrimSpace(gloss)} + } + if s.gloss == "" { + continue + } + s.gloss, _ = capGloss(s.gloss) + if len(senses) < maxSenses { + senses = append(senses, s) + } + } + return senses } // verify re-opens the finished database and re-checks the invariants the game // depends on. The in-memory checks above can only prove what the builder // intended; this proves what actually landed on disk. -func verify(path string, minWords int) error { +func verify(path string, minWords int, requireCoverage bool) error { db, err := sql.Open("sqlite", "file:"+path+"?mode=ro") if err != nil { return err @@ -193,6 +248,15 @@ func verify(path string, minWords int) error { {"words whose first syllable is missing from the syllables table", `SELECT COUNT(*) FROM words w LEFT JOIN syllables s ON s.syllable = w.first WHERE s.syllable IS NULL`, func(n int) bool { return n == 0 }}, + {"meanings whose word is missing from the words table", + `SELECT COUNT(*) FROM meanings m LEFT JOIN words w ON w.word = m.word WHERE w.word IS NULL`, + func(n int) bool { return n == 0 }}, + {"meanings with an empty gloss", `SELECT COUNT(*) FROM meanings WHERE gloss = ''`, func(n int) bool { return n == 0 }}, + {"meanings over the length cap", fmt.Sprintf(`SELECT COUNT(*) FROM meanings WHERE LENGTH(gloss) > %d`, maxGlossRunes), + func(n int) bool { return n == 0 }}, + {"words with more meanings than the cap", + fmt.Sprintf(`SELECT COUNT(*) FROM (SELECT word FROM meanings GROUP BY word HAVING COUNT(*) > %d)`, maxSenses), + func(n int) bool { return n == 0 }}, } for _, c := range checks { @@ -205,6 +269,20 @@ func verify(path string, minWords int) error { } } + if requireCoverage { + var wordCount, withMeaning int + if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&wordCount); err != nil { + return err + } + if err := db.QueryRow(`SELECT COUNT(DISTINCT word) FROM meanings`).Scan(&withMeaning); err != nil { + return err + } + if float64(withMeaning) < minMeaningCoverage*float64(wordCount) { + return fmt.Errorf("only %d of %d words have a meaning, expected at least %.0f%% — "+ + "the dump's definition markup may have changed", withMeaning, wordCount, minMeaningCoverage*100) + } + } + return nil } @@ -218,7 +296,7 @@ type entry struct { // sourceSpec is what the meta table records about where the words came from. type sourceSpec struct { - // table names the input: "kaikki:" for the corpus, "wordlist:" + // table names the input: "dump:" for the corpus, "wordlist:" // for a fixture, so the output says which build produced it. table string // url is the upstream artifact; empty for fixture builds. @@ -281,7 +359,7 @@ func buildAliases(words map[string]entry) (map[string]string, int) { // partway through -- a full disk, an interrupt -- leaves an empty but // syntactically valid database where a good one used to be, which the server // would happily open and find no words in. -func write(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error { +func write(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error { tmp := path + ".tmp" if err := os.Remove(tmp); err != nil && !errors.Is(err, os.ErrNotExist) { return fmt.Errorf("remove stale temp file: %w", err) @@ -294,7 +372,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s } }() - if err := writeTo(tmp, words, aliases, src); err != nil { + if err := writeTo(tmp, words, meanings, aliases, src); err != nil { return err } @@ -311,7 +389,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s return nil } -func writeTo(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error { +func writeTo(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error { db, err := sql.Open("sqlite", "file:"+path) if err != nil { return fmt.Errorf("create output: %w", err) @@ -337,6 +415,17 @@ CREATE TABLE aliases ( canonical TEXT NOT NULL ) WITHOUT ROWID; +-- One row per sense, in page order. pos is the Vietnamese part-of-speech +-- label of the heading the definition sat under, '' when the heading was one +-- the builder does not know. No foreign key pragma: verify() checks the join. +CREATE TABLE meanings ( + word TEXT NOT NULL, + ord INTEGER NOT NULL, + pos TEXT NOT NULL, + gloss TEXT NOT NULL, + PRIMARY KEY (word, ord) +) WITHOUT ROWID; + CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL); ` if _, err := db.Exec(schema); err != nil { @@ -390,6 +479,27 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL); } } + insertMeaning, err := tx.Prepare(`INSERT INTO meanings (word, ord, pos, gloss) VALUES (?, ?, ?, ?)`) + if err != nil { + return err + } + defer insertMeaning.Close() + meaningCount := 0 + for word, senses := range meanings { + if _, isWord := words[word]; !isWord { + return fmt.Errorf("meaning for %q, which is not a word", word) + } + for ord, s := range senses { + if s.gloss == "" || utf8.RuneCountInString(s.gloss) > maxGlossRunes { + return fmt.Errorf("meaning %d of %q is empty or over the cap", ord, word) + } + if _, err := insertMeaning.Exec(word, ord, s.pos, s.gloss); err != nil { + return fmt.Errorf("insert meaning %d of %q: %w", ord, word, err) + } + meaningCount++ + } + } + insertMeta, err := tx.Prepare(`INSERT INTO meta (key, value) VALUES (?, ?)`) if err != nil { return err @@ -403,6 +513,8 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL); {"builder_version", builderVer}, {"word_count", fmt.Sprint(len(words))}, {"alias_count", fmt.Sprint(len(aliases))}, + {"meaning_count", fmt.Sprint(meaningCount)}, + {"words_with_meaning", fmt.Sprint(len(meanings))}, {"source_table", src.table}, } meta = append(meta, src.extra...) diff --git a/server/cmd/build-dictionary/main_test.go b/server/cmd/build-dictionary/main_test.go index 79ac971..918ecfc 100644 --- a/server/cmd/build-dictionary/main_test.go +++ b/server/cmd/build-dictionary/main_test.go @@ -6,47 +6,25 @@ import ( "errors" "os" "path/filepath" + "strconv" + "strings" "testing" _ "modernc.org/sqlite" ) -// fixtureSource writes a miniature stand-in for the kaikki export: the same -// JSONL shape, a handful of rows instead of 44k. Each row is a word and the -// language its Wiktionary entry is for. Tests never touch the real download. -func fixtureSource(t *testing.T, rows [][2]string) string { - t.Helper() - - lines := make([]string, 0, len(rows)) - for _, r := range rows { - lines = append(lines, `{"word": "`+r[0]+`", "pos": "noun", "lang_code": "`+r[1]+`"}`) - } - return fixtureKaikki(t, lines...) -} - -func defaultRows() [][2]string { - return [][2]string{ - {"pháp luật", "vi"}, - {"pháp luật", "vi"}, // listed twice — must dedupe to one word - {"luật lệ", "vi"}, - {"ngôn ngữ", "vi"}, - {"ngữ pháp", "vi"}, - {"hòa bình", "vi"}, - {"vô tuyến điện", "vi"}, // three syllables - {"pháp", "vi"}, // single syllable — rejected - {"covid 19", "vi"}, // digit — rejected - {"hello world", "en"}, // another language's entry — never selected - } -} - -func buildFixture(t *testing.T, rows [][2]string) string { +// buildFixture runs the whole pipeline on the committed mini dump: twelve +// pages, both dialects, a redirect, an English-only page and two pages that +// accept() rejects. Tests never touch the real download. +func buildFixture(t *testing.T) string { t.Helper() out := filepath.Join(t.TempDir(), "noitu.db") cfg := config{ - kaikki: fixtureSource(t, rows), + dump: miniDump, out: out, minWords: 1, + minPages: 1, } if err := run(cfg); err != nil { t.Fatalf("run: %v", err) @@ -64,23 +42,23 @@ func openOut(t *testing.T, path string) *sql.DB { return db } +func count(t *testing.T, db *sql.DB, query string, args ...any) int { + t.Helper() + var n int + if err := db.QueryRow(query, args...).Scan(&n); err != nil { + t.Fatalf("%s: %v", query, err) + } + return n +} + func TestBuildProducesExpectedWords(t *testing.T) { - db := openOut(t, buildFixture(t, defaultRows())) + db := openOut(t, buildFixture(t)) - var count int - if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&count); err != nil { - t.Fatal(err) + if got := count(t, db, `SELECT COUNT(*) FROM words`); got != 6 { + t.Errorf("word count = %d, want 6", got) } - if want := 6; count != want { - t.Errorf("word count = %d, want %d", count, want) - } - // Every stored word must have at least two syllables. - var short int - if err := db.QueryRow(`SELECT COUNT(*) FROM words WHERE syllables < 2`).Scan(&short); err != nil { - t.Fatal(err) - } - if short != 0 { + if short := count(t, db, `SELECT COUNT(*) FROM words WHERE syllables < 2`); short != 0 { t.Errorf("%d words have fewer than 2 syllables, want 0", short) } @@ -98,7 +76,7 @@ func TestBuildProducesExpectedWords(t *testing.T) { } func TestBuildComputesOutDegree(t *testing.T) { - db := openOut(t, buildFixture(t, defaultRows())) + db := openOut(t, buildFixture(t)) // "pháp luật" and "pháp" (rejected) mean exactly one word starts with "pháp". assertOutDegree(t, db, "pháp", 1) @@ -120,7 +98,7 @@ func assertOutDegree(t *testing.T, db *sql.DB, syllable string, want int) { } func TestBuildWritesAliases(t *testing.T) { - db := openOut(t, buildFixture(t, defaultRows())) + db := openOut(t, buildFixture(t)) var canonical string err := db.QueryRow(`SELECT canonical FROM aliases WHERE variant = ?`, "hoà bình").Scan(&canonical) @@ -132,22 +110,65 @@ func TestBuildWritesAliases(t *testing.T) { } // Every alias must point at a word that actually exists. - var orphans int - err = db.QueryRow(`SELECT COUNT(*) FROM aliases a - LEFT JOIN words w ON w.word = a.canonical - WHERE w.word IS NULL`).Scan(&orphans) - if err != nil { - t.Fatal(err) - } + orphans := count(t, db, `SELECT COUNT(*) FROM aliases a LEFT JOIN words w ON w.word = a.canonical WHERE w.word IS NULL`) if orphans != 0 { t.Errorf("%d aliases point at missing words, want 0", orphans) } } -func TestBuildRecordsProvenance(t *testing.T) { - db := openOut(t, buildFixture(t, defaultRows())) +func TestBuildWritesMeanings(t *testing.T) { + db := openOut(t, buildFixture(t)) - for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count"} { + rows, err := db.Query(`SELECT ord, pos, gloss FROM meanings WHERE word = ? ORDER BY ord`, "pháp luật") + if err != nil { + t.Fatal(err) + } + defer rows.Close() + var got []sense + for rows.Next() { + var ord int + var s sense + if err := rows.Scan(&ord, &s.pos, &s.gloss); err != nil { + t.Fatal(err) + } + if ord != len(got) { + t.Errorf("ord = %d, want %d (0-based, dense)", ord, len(got)) + } + got = append(got, s) + } + assertSenses(t, got, []sense{ + {"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."}, + {"danh từ", "(nghĩa rộng) Kỷ cương nói chung."}, + {"động từ", "(hiếm) Xử theo luật."}, + }) + + // A word whose only definition stripped to nothing has no rows, and the + // meta counts describe the table. + if n := count(t, db, `SELECT COUNT(*) FROM meanings WHERE word = ?`, "luật lệ"); n != 0 { + t.Errorf("luật lệ has %d meanings, want 0", n) + } + total := count(t, db, `SELECT COUNT(*) FROM meanings`) + withMeaning := count(t, db, `SELECT COUNT(DISTINCT word) FROM meanings`) + for key, want := range map[string]int{"meaning_count": total, "words_with_meaning": withMeaning} { + var raw string + if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil { + t.Fatalf("meta[%q]: %v", key, err) + } + if raw != strconv.Itoa(want) { + t.Errorf("meta[%q] = %s, want %d", key, raw, want) + } + } + if withMeaning != 5 { + t.Errorf("words with a meaning = %d, want 5 of 6", withMeaning) + } +} + +func TestBuildRecordsProvenance(t *testing.T) { + db := openOut(t, buildFixture(t)) + + want := map[string]string{"builder_version": builderVer, "source_url": dumpSourceURL, "source_pages": "9"} + for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count", + "builder_version", "source_sha256", "source_pages", "source_fetched_at", "meaning_count", "words_with_meaning"} { var value string if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&value); err != nil { t.Errorf("meta[%q] missing: %v", key, err) @@ -156,26 +177,42 @@ func TestBuildRecordsProvenance(t *testing.T) { if value == "" { t.Errorf("meta[%q] is empty", key) } + if w, ok := want[key]; ok && value != w { + t.Errorf("meta[%q] = %q, want %q", key, value, w) + } + } + if n := count(t, db, `SELECT COUNT(*) FROM meta WHERE key = 'source_rows'`); n != 0 { + t.Error("source_rows belonged to the previous source format and must be gone") } } -// The floor exists so a schema change upstream fails the build loudly instead +// The floor exists so a markup change upstream fails the build loudly instead // of silently shipping a near-empty dictionary. func TestBuildFailsBelowMinWords(t *testing.T) { cfg := config{ - kaikki: fixtureSource(t, defaultRows()), + dump: miniDump, out: filepath.Join(t.TempDir(), "noitu.db"), minWords: 1000, + minPages: 1, } if err := run(cfg); err == nil { t.Fatal("run succeeded with an unreachable min-words floor, want error") } } +func TestRunRequiresExactlyOneInput(t *testing.T) { + if err := run(config{}); err == nil { + t.Error("run with no input succeeded") + } + if err := run(config{dump: miniDump, words: "x.txt"}); err == nil || !strings.Contains(err.Error(), "mutually exclusive") { + t.Errorf("run with both inputs: %v", err) + } +} + // A failed build must leave the previous good database untouched. Building in // place would delete it and leave an empty file the server would happily open. func TestFailedBuildPreservesPreviousOutput(t *testing.T) { - out := buildFixture(t, defaultRows()) + out := buildFixture(t) before, err := os.ReadFile(out) if err != nil { @@ -184,9 +221,10 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) { // Same output path, but a floor no fixture can clear. cfg := config{ - kaikki: fixtureSource(t, defaultRows()), + dump: miniDump, out: out, minWords: 1000, + minPages: 1, } if err := run(cfg); err == nil { t.Fatal("run succeeded with an unreachable floor, want error") @@ -203,3 +241,127 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) { t.Error("temp database left behind after a failed build") } } + +// --- the fixture word list -------------------------------------------------- + +func writeWordList(t *testing.T, content string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "words.txt") + if err := os.WriteFile(path, []byte(content), 0o644); err != nil { + t.Fatal(err) + } + return path +} + +func TestWordListCarriesTabSeparatedMeanings(t *testing.T) { + list := writeWordList(t, "# comment\n"+ + "học sinh\tdanh từ|Người học ở trường.\tđộng từ|Đi học.\n"+ + "sinh viên\tNgười học ở trường đại học.\n"+ + "sinh hoạt\n"+ + "sinh sản\t\t\n") + out := filepath.Join(t.TempDir(), "fixture.db") + if err := run(config{words: list, out: out, minWords: 1}); err != nil { + t.Fatal(err) + } + db := openOut(t, out) + + rows, err := db.Query(`SELECT word, ord, pos, gloss FROM meanings ORDER BY word, ord`) + if err != nil { + t.Fatal(err) + } + defer rows.Close() + type row struct { + word string + ord int + s sense + } + var got []row + for rows.Next() { + var r row + if err := rows.Scan(&r.word, &r.ord, &r.s.pos, &r.s.gloss); err != nil { + t.Fatal(err) + } + got = append(got, r) + } + want := []row{ + {"học sinh", 0, sense{"danh từ", "Người học ở trường."}}, + {"học sinh", 1, sense{"động từ", "Đi học."}}, + {"sinh viên", 0, sense{"", "Người học ở trường đại học."}}, + } + if len(got) != len(want) { + t.Fatalf("meanings = %+v, want %+v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("row %d = %+v, want %+v", i, got[i], want[i]) + } + } + if n := count(t, db, `SELECT COUNT(*) FROM words`); n != 4 { + t.Errorf("words = %d, want 4 (a line without a tab is still a word)", n) + } + // Fixture builds carry no upstream licence and are exempt from the + // coverage floor: two of four words have a meaning here. + var license string + if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil { + t.Fatal(err) + } + if !strings.HasPrefix(license, "none") { + t.Errorf("fixture licence = %q, want a statement that no upstream data applies", license) + } +} + +// --- verify ----------------------------------------------------------------- + +// brokenDB writes a database that passes every schema check and then breaks +// one invariant, to prove verify() reads what is on disk. +func brokenDB(t *testing.T, extraSQL string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "broken.db") + words := map[string]entry{"pháp luật": {"pháp luật", "pháp", "luật", 2}} + meanings := map[string][]sense{"pháp luật": {{"danh từ", "Luật."}}} + if err := writeTo(path, words, meanings, nil, sourceSpec{table: "test"}); err != nil { + t.Fatal(err) + } + db, err := sql.Open("sqlite", "file:"+path) + if err != nil { + t.Fatal(err) + } + defer db.Close() + if _, err := db.Exec(extraSQL); err != nil { + t.Fatal(err) + } + return path +} + +func TestVerifyRejectsBrokenMeanings(t *testing.T) { + cases := map[string]string{ + "orphan meaning row": `INSERT INTO meanings VALUES ('không có', 0, '', 'Một nghĩa.')`, + "empty gloss": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '')`, + "over the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '` + strings.Repeat("a", maxGlossRunes+1) + `')`, + "more senses than the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', 'b'), ('pháp luật', 2, '', 'c'), + ('pháp luật', 3, '', 'd'), ('pháp luật', 4, '', 'e'), ('pháp luật', 5, '', 'f')`, + } + for name, sqlText := range cases { + t.Run(name, func(t *testing.T) { + path := brokenDB(t, sqlText) + if err := verify(path, 1, false); err == nil { + t.Error("verify passed a database that breaks a meanings invariant") + } + }) + } + if err := verify(brokenDB(t, `SELECT 1`), 1, false); err != nil { + t.Errorf("verify rejected a sound database: %v", err) + } +} + +func TestVerifyCoverageFloorAppliesToCorpusBuildsOnly(t *testing.T) { + // One word with a meaning, one without: 50%, under the floor. + path := brokenDB(t, `INSERT INTO words VALUES ('luật lệ', 'luật', 'lệ', 2); + INSERT INTO syllables VALUES ('lệ', 0); UPDATE syllables SET out_degree = 1 WHERE syllable = 'luật'`) + if err := verify(path, 1, true); err == nil || !strings.Contains(err.Error(), "have a meaning") { + t.Errorf("corpus verify with 50%% coverage: %v, want the coverage floor named", err) + } + if err := verify(path, 1, false); err != nil { + t.Errorf("fixture verify applied the coverage floor: %v", err) + } +} diff --git a/server/cmd/build-dictionary/testdata/mini-dump-cut.xml.bz2 b/server/cmd/build-dictionary/testdata/mini-dump-cut.xml.bz2 new file mode 100644 index 0000000..91f46fc Binary files /dev/null and b/server/cmd/build-dictionary/testdata/mini-dump-cut.xml.bz2 differ diff --git a/server/cmd/build-dictionary/testdata/mini-dump.xml b/server/cmd/build-dictionary/testdata/mini-dump.xml new file mode 100644 index 0000000..6ed4175 --- /dev/null +++ b/server/cmd/build-dictionary/testdata/mini-dump.xml @@ -0,0 +1,182 @@ + + + Wiktionary + viwiktionary + + + + pháp luật + 0 + 1 + + 101 + {{-vie-}} +{{-pron-}} +{{vie-pron|pháp luật}} + +{{-noun-}} +{{-dfn-}} +# [[hệ thống|Hệ thống]] các [[quy tắc]] xử sự do [[nhà nước]] đặt ra.<ref>Từ điển tiếng Việt</ref> +#: ''Tuân theo pháp luật.'' +#{{label|vi|nghĩa rộng}} [[kỷ cương|Kỷ cương]] nói chung. + +{{-verb-}} +# ''(hiếm)'' [[xử|Xử]] theo luật. + +{{-trans-}} +* {{eng}}: {{t|en|law}} + +{{-eng-}} +{{-noun-}} +# Law, in English. + + + + + Hòa Bình + 0 + 2 + + 102 + == {{langname|vi}} == +=== {{ĐM|etym}} === +Từ Hán-Việt. + +=== {{ĐM|pr-noun}} === +{{vi-pr-noun}} + +# {{place|vi|tỉnh|c/Việt Nam}}. + +== {{langname|en}} == +=== {{ĐM|pr-noun}} === +# A province of Vietnam. + + + + + hòa bình + 0 + 3 + + 103 + {{-vie-}} +{{-noun-}} +# [[tình trạng|Tình trạng]] không có [[chiến tranh]]. +{{-adj-}} +# [[yên ổn|Yên ổn]]. + + + + + mặt trời + 0 + 4 + + + 104 + #đổi [[Mặt Trời]] + + + + + hello world + 0 + 5 + + 105 + {{-eng-}} +{{-phrase-}} +# Xin chào thế giới. + + + + + Thể loại:Danh từ tiếng Việt + 14 + 6 + + 106 + {{-vie-}} +{{-noun-}} +# Not a word. + + + + + pháp + 0 + 7 + + 107 + {{-vie-}} +{{-noun-}} +# [[phép|Phép]], [[luật]]. + + + + + luật lệ + 0 + 8 + + 108 + {{-vie-}} +{{-noun-}} +# {{rfdef|vi}} + + + + + ngôn ngữ + 0 + 9 + + 109 + == {{langname|vi}} == +=== {{section|n}} === +{{vi-noun}} + +# [[hệ thống|Hệ thống]] những [[âm]], [[từ]] và [[quy tắc]] kết hợp chúng. + + + + ngữ pháp + 0 + 10 + + 110 + {{-vie-}} +{{-noun-}} +# [[toàn bộ|Toàn bộ]] những [[quy tắc]] hoạt động của các yếu tố ngôn ngữ. + + + + vô tuyến điện + 0 + 11 + + 111 + {{-vie-}} +{{-noun-}} +# [[kỹ thuật|Kỹ thuật]] truyền tin bằng [[sóng điện từ]]. + + + + covid 19 + 0 + 12 + + 112 + {{-vie-}} +{{-noun-}} +# Một [[bệnh]]. + + + diff --git a/server/cmd/build-dictionary/testdata/mini-dump.xml.bz2 b/server/cmd/build-dictionary/testdata/mini-dump.xml.bz2 new file mode 100644 index 0000000..3fd73eb Binary files /dev/null and b/server/cmd/build-dictionary/testdata/mini-dump.xml.bz2 differ diff --git a/server/cmd/build-dictionary/wikitext.go b/server/cmd/build-dictionary/wikitext.go new file mode 100644 index 0000000..7a4fc83 --- /dev/null +++ b/server/cmd/build-dictionary/wikitext.go @@ -0,0 +1,629 @@ +package main + +import ( + "html" + "regexp" + "strings" + "unicode" + "unicode/utf8" +) + +// This file reads the wikitext of one Wiktionary tiếng Việt page: it finds the +// Vietnamese section, walks its part-of-speech headings and turns each +// definition line into plain text. +// +// The wiki is mid-migration between two markup dialects and both are live +// (2026-09-01 dump: 35,885 legacy pages, 7,129 new): +// +// legacy {{-vie-}} opens the section, {{-noun-}} and kin are the headings, +// and the section ends at the next {{-xxx-}} whose code is a +// language rather than a heading. +// new == {{langname|vi}} == opens the section, === {{ĐM|noun}} === or +// === {{section|noun}} === are the headings, and the next level-2 +// heading ends it. +// +// Nothing here is a general wikitext parser. It knows exactly the shapes a +// definition line takes on this wiki and drops the rest on purpose; what +// survives is plain text, capped, safe to render as text and never as markup. + +// sense is one definition with the Vietnamese part-of-speech label of the +// heading it sat under. pos is empty when the heading was one the label map +// does not know, never a reason to drop the definition. +type sense struct { + pos string + gloss string +} + +const ( + // maxSenses and maxGlossRunes bound what one word carries to the client. + maxSenses = 5 + maxGlossRunes = 200 + // ellipsis marks a definition cut at maxGlossRunes. + ellipsis = "…" +) + +// posLabelMap maps a part-of-speech heading code, the same in both dialects +// ({{-noun-}}, {{ĐM|noun}}, {{section|noun}}, {{vi-noun}}), to the Vietnamese +// label the client shows in front of a sense. +// +// Codes and their frequencies in Vietnamese sections of the 2026-09-01 dump: +// noun 13,674 + n 647 · verb 7,595 + v 353 · adj 5,318 + adjc 526 · place 3,352 +// · pr-noun 1,397 + 329 · adv 982 · phrase 412 · proverb 295 · idiom 241 · +// interj 169 · pronoun 158 · num 123 · conj 88 · prep 88 · part 29. Note that +// "pron" on this wiki is pronunciation, not pronoun. +var posLabelMap = map[string]string{ + "noun": "danh từ", "n": "danh từ", + "verb": "động từ", "v": "động từ", "tr-verb": "động từ", "intr-verb": "động từ", "aux-verb": "động từ", + "adj": "tính từ", "adjc": "tính từ", "adjective": "tính từ", + "adv": "phó từ", "adverb": "phó từ", "advb": "phó từ", + "pr-noun": "danh từ riêng", "proper": "danh từ riêng", "propn": "danh từ riêng", "proper noun": "danh từ riêng", "name": "danh từ riêng", + "pr-adj": "tính từ riêng", + "place": "địa danh", + "pronoun": "đại từ", "per-pronoun": "đại từ", + "num": "số từ", "numeral": "số từ", + "conj": "liên từ", "conjunction": "liên từ", + "prep": "giới từ", + "interj": "thán từ", "intj": "thán từ", "interjection": "thán từ", + "part": "trợ từ", "particle": "trợ từ", + "phrase": "cụm từ", + "idiom": "thành ngữ", + "proverb": "tục ngữ", "prov": "tục ngữ", + "abbr": "viết tắt", "abr": "viết tắt", + "prefix": "tiền tố", + "suffix": "hậu tố", + "letter": "chữ cái", + "symbol": "ký hiệu", +} + +// otherSectionCodes are heading codes that are not parts of speech: they sit +// inside a language section and reset the current label without being +// counted as unmapped. Frequencies in Vietnamese sections, 2026-09-01 dump: +// pron 36,533 · ref 27,438 · trans 16,073 · paro 8,058 · etym 4,267 · syn +// 3,165 · info 1,846 · hanviet 1,601 · hanviet-t 1,441 · etymology 1,429 · +// reference 1,426 · see 959 · related 461 · drv 277 · synonym 262 · ant 236 · +// desction 128 · usage 110 · expr 92 · homo 57 · desc 54 · further 48 · forms +// 41 · note 33 · derived 28 · anagram 24 · compound 21 · redup 20 · translit 17 +// · cat 16 · antonym 15. "dfn" (4,615) is a "definitions" heading placed under a +// part-of-speech heading, so it is a heading for the section boundary but +// transparent to the label: see classifyHeading. +var otherSectionCodes = map[string]bool{ + "pron": true, "pronunciation": true, "ref": true, "reference": true, "references": true, + "trans": true, "translations": true, "paro": true, "paronym": true, "etym": true, "etymology": true, + "syn": true, "synonym": true, "ant": true, "antonym": true, "info": true, + "hanviet": true, "hanviet-t": true, "see": true, "see also": true, "related": true, "rel": true, + "related terms": true, "drv": true, "der": true, "derived": true, "derived terms": true, + "desction": true, "desc": true, "usage": true, "usage notes": true, "expr": true, "homo": true, + "further": true, "further reading": true, "forms": true, "note": true, "anagram": true, + "anagrams": true, "ana": true, "compound": true, "redup": true, "translit": true, "cat": true, + "coord": true, "coordinate": true, "alt": true, "alter": true, "alter form": true, + "alternative form": true, "alternative forms": true, "alternative script": true, + "glyph origin": true, "han": true, "nôm": true, "han character": true, "kanji": true, + "rom": true, "romanization": true, "mut": true, "participle": true, "ptcp": true, + "syllable": true, "article": true, "contr": true, "cmavo": true, "dfn": true, "com": true, + // The same sections written out in Vietnamese, as a few new-dialect pages do. + "phát âm": true, "từ nguyên": true, "từ nguyên 1": true, "từ nguyên 2": true, "tham khảo": true, + "xem thêm": true, "cách viết khác": true, "phồn thể": true, "hán-nôm": true, "hán nôm": true, + "chữ hán": true, "chữ nôm": true, "chú ý": true, "đồng nghĩa": true, "từ đồng nghĩa": true, + "bản dịch": true, "dịch": true, "dấu phụ": true, "liên kết ngoài": true, "thuật ngữ liên quan": true, + "từ tương tự": true, "meronym": true, "meronyms": true, "nguồn gốc ký tự chữ nôm": true, +} + +var ( + // legacyTemplate matches one {{-code-}} template, optionally with + // parameters: {{-noun-}}, {{-pr-noun-}}, {{-vie-|...}}. Not anchored: a few + // pages run {{-vie-}}{{-pron-}}{{vie-pron|…}}{{-place-}} together on one + // line, so a line is read as headings when it starts with one and may + // carry several. + legacyTemplate = regexp.MustCompile(`\{\{-([A-Za-z0-9-]+?)-(?:\|[^}]*)?\}\}`) + // headingLine matches == text == at any level and captures the level. + headingLine = regexp.MustCompile(`^(={2,6})\s*(.*?)\s*=+\s*$`) + // sectionTemplate captures the code of {{ĐM|code}} and {{section|code}}. + sectionTemplate = regexp.MustCompile(`\{\{(?:ĐM|đm|DM|dm|section)\|([^}|]+)`) + // headwordTemplate captures the code of {{vi-code}} / {{vie-code}} at the + // start of a line: the new dialect's headword line, which names the POS. + headwordTemplate = regexp.MustCompile(`^\{\{vie?-([a-z -]+)`) + // langnameVi is the new dialect's Vietnamese section heading text. + langnameVi = regexp.MustCompile(`^\{\{langname\|vi\}\}$`) + + htmlComment = regexp.MustCompile(`(?s)`) + refElement = regexp.MustCompile(`(?s)/]*/>|]*>.*?`) + anyTag = regexp.MustCompile(`]*>`) + spaces = regexp.MustCompile(`\s+`) +) + +// isLegacyHeading reports whether a {{-code-}} is a heading inside a language +// section. Every other code — language and script codes such as eng, tyz, +// aav-qal, Latn — ends the Vietnamese section. +func isLegacyHeading(code string) bool { + _, pos := posLabelMap[code] + return pos || otherSectionCodes[code] +} + +// vietnameseSection returns the wikitext of the page's Vietnamese section and +// which dialect opened it: "legacy", "new", or "" when the page has none. +// When both dialects open a section on one page the first one in the text +// wins and both is reported so the build log can count it. ender is the +// {{-code-}} that closed a legacy section, empty when a heading or the end of +// the page did: a heading code missing from the maps shows up there as a +// section-ending code, which is the signal that definitions are being lost. +func vietnameseSection(text string) (section, dialect string, both bool, ender string) { + lines := strings.Split(text, "\n") + legacyAt, newAt := -1, -1 + for i, line := range lines { + line = strings.TrimRight(line, "\r ") + if legacyAt < 0 && strings.HasPrefix(line, "{{-") { + for _, m := range legacyTemplate.FindAllStringSubmatchIndex(line, -1) { + if line[m[2]:m[3]] == "vie" { + legacyAt = i + // Whatever follows the marker on its own line belongs to + // the section. + lines[i] = line[m[1]:] + break + } + } + } + if newAt < 0 { + if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 && langnameVi.MatchString(m[2]) { + newAt = i + } + } + } + both = legacyAt >= 0 && newAt >= 0 + switch { + case legacyAt < 0 && newAt < 0: + return "", "", false, "" + case newAt < 0 || (legacyAt >= 0 && legacyAt < newAt): + section, ender = legacySection(lines[legacyAt:]) + return section, "legacy", both, ender + default: + return newSection(lines[newAt+1:]), "new", both, "" + } +} + +// legacySection runs from after {{-vie-}} to the next {{-xxx-}} whose code is +// not a heading, or the next level-2 heading, which is what a new-dialect +// language section on a mixed page opens with. A language code sharing a line +// with Vietnamese headings ends the section at that line; the line is lost, +// which is the conservative side of a rare shape. +func legacySection(lines []string) (section, ender string) { + for i, line := range lines { + line = strings.TrimRight(line, "\r ") + if strings.HasPrefix(line, "{{-") { + for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) { + if !isLegacyHeading(m[1]) { + return strings.Join(lines[:i], "\n"), m[1] + } + } + } + if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 { + return strings.Join(lines[:i], "\n"), "" + } + } + return strings.Join(lines, "\n"), "" +} + +// newSection runs from after == {{langname|vi}} == to the next level-2 +// heading. +func newSection(lines []string) string { + for i, line := range lines { + line = strings.TrimRight(line, "\r ") + if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 { + return strings.Join(lines[:i], "\n") + } + } + return strings.Join(lines, "\n") +} + +// headingKind says what a heading line means for the label of the +// definitions under it. +type headingKind int + +const ( + notHeading headingKind = iota + // posHeading names a part of speech: the code decides the label. + posHeading + // otherHeading is a section such as pronunciation or etymology: the label + // resets to empty and nothing is counted as unmapped. + otherHeading +) + +// classifyHeading reads one line as a heading in either dialect. +// +// {{-noun-}} legacy heading; the last of several on +// one line decides +// === {{ĐM|noun}} === new heading +// === {{section|n}} === new heading, shorthand code +// === Danh từ === new heading written out +// {{vi-noun}} / {{vie-noun}} new headword line; refines the POS only +func classifyHeading(line string) (code string, kind headingKind) { + line = strings.TrimRight(line, "\r ") + if strings.HasPrefix(line, "{{-") { + kind = notHeading + for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) { + if c, k := classifyCode(m[1]); k != notHeading { + code, kind = c, k + } + } + return code, kind + } + if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) >= 3 { + text := m[2] + if sm := sectionTemplate.FindStringSubmatch(text); sm != nil { + code = strings.TrimSpace(sm[1]) + } else { + code = strings.ToLower(text) + } + return classifyCode(code) + } + if m := headwordTemplate.FindStringSubmatch(line); m != nil { + // Only a headword template whose code is a part of speech counts; + // {{vi-pron}}, {{vi-etym-sino}} and kin are not headings. + code = strings.TrimSpace(m[1]) + if _, ok := posLabelMap[code]; ok { + return code, posHeading + } + } + return "", notHeading +} + +// classifyCode sorts a heading code seen in either dialect. "dfn" is the one +// heading that changes nothing: the wiki places {{-dfn-}} under {{-noun-}} to +// introduce the definitions, so the label above it must carry through. +func classifyCode(code string) (string, headingKind) { + switch { + case code == "dfn": + return code, notHeading + case otherSectionCodes[code]: + return code, otherHeading + } + return code, posHeading +} + +// isLabelValue reports whether a heading was written out as one of the +// Vietnamese labels already ("Danh từ"). +func isLabelValue(text string) bool { + for _, l := range posLabelMap { + if l == text { + return true + } + } + return false +} + +// posLabel maps a heading code to its Vietnamese label: through the map, or +// as itself when the heading was already written out in Vietnamese. +func posLabel(code string) (label string, mapped bool) { + if label, ok := posLabelMap[code]; ok { + return label, true + } + if isLabelValue(code) { + return code, true + } + return "", false +} + +// sectionStats counts what the scanner saw across sections, for the build log. +type sectionStats struct { + pos map[string]int // part-of-speech heading codes seen + unmappedPos map[string]int // part-of-speech heading codes with no label + defsKept int + defsEmpty int // definitions that stripped to nothing + defsCut int // definitions cut at maxGlossRunes + dropped map[string]int // template names dropped whole +} + +func newSectionStats() *sectionStats { + return §ionStats{ + pos: make(map[string]int), + unmappedPos: make(map[string]int), + dropped: make(map[string]int), + } +} + +// definitions walks the Vietnamese section and returns its senses in page +// order, at most maxSenses of them, each labelled with the part of speech of +// the heading above it. Every heading and definition is counted in stats +// whether or not it made the cut. +func definitions(section string, stats *sectionStats) []sense { + var senses []sense + pos := "" + for _, line := range strings.Split(section, "\n") { + line = strings.TrimRight(line, "\r ") + switch code, kind := classifyHeading(line); kind { + case posHeading: + stats.pos[code]++ + label, mapped := posLabel(code) + if !mapped { + stats.unmappedPos[code]++ + } + pos = label + continue + case otherHeading: + pos = "" + continue + } + if !isDefinitionLine(line) { + continue + } + gloss := stripWikitext(line[1:], stats.dropped) + if gloss == "" { + stats.defsEmpty++ + continue + } + if cut, wasCut := capGloss(gloss); wasCut { + stats.defsCut++ + gloss = cut + } + stats.defsKept++ + if len(senses) < maxSenses { + senses = append(senses, sense{pos: pos, gloss: gloss}) + } + } + return senses +} + +// isDefinitionLine accepts a top-level numbered item and nothing under it: +// "# text" and "#text" are definitions; "#: example", "#* quotation", "## sub- +// sense" and "#; term" are not. +func isDefinitionLine(line string) bool { + if len(line) < 2 || line[0] != '#' { + return false + } + switch line[1] { + case '#', ':', '*', ';': + return false + } + return true +} + +// stripWikitext turns one definition line into plain text. Lossy on purpose: +// links keep their display text, formatting goes, the handful of templates +// that carry definition text are unwrapped and every other template is +// dropped whole (its name counted in dropped when non-nil). The result is +// trimmed and whitespace-collapsed; a result with no letter or digit is empty. +func stripWikitext(s string, dropped map[string]int) string { + s = htmlComment.ReplaceAllString(s, "") + s = refElement.ReplaceAllString(s, "") + s = anyTag.ReplaceAllString(s, "") + s = stripTemplates(s, dropped) + s = stripLinks(s) + s = strings.ReplaceAll(s, "'''", "") + s = strings.ReplaceAll(s, "''", "") + s = html.UnescapeString(s) + s = strings.Map(func(r rune) rune { + switch { + case unicode.IsControl(r), unicode.Is(unicode.Cf, r): + // Cc and Cf: control characters, and format characters such as a + // bidi override or a zero-width space, which could reshape the + // rest of a rendered line. + return -1 + case unicode.IsSpace(r): + // Non-breaking and other Unicode spaces become plain ones so the + // ASCII-only collapse below catches them. + return ' ' + } + return r + }, s) + s = strings.TrimSpace(spaces.ReplaceAllString(s, " ")) + // A stray space before sentence punctuation is what unwrapping a template + // at the end of a clause leaves behind. + for _, p := range []string{" .", " ,", " ;", " :", " )"} { + s = strings.ReplaceAll(s, p, p[1:]) + } + if !strings.ContainsFunc(s, func(r rune) bool { return unicode.IsLetter(r) || unicode.IsDigit(r) }) { + return "" + } + return s +} + +// stripTemplates replaces every outermost {{...}} with its plain-text +// rendering. Nesting is tracked by depth, so a template inside a kept +// template's parameter is rendered recursively and one inside a dropped +// template goes with it. +func stripTemplates(s string, dropped map[string]int) string { + var out strings.Builder + depth := 0 + start := 0 + for i := 0; i < len(s); i++ { + switch { + case strings.HasPrefix(s[i:], "{{"): + if depth == 0 { + start = i + 2 + } + depth++ + i++ + case strings.HasPrefix(s[i:], "}}"): + // A closer with nothing open is stray markup, not text. + if depth > 0 { + depth-- + if depth == 0 { + out.WriteString(renderTemplate(s[start:i], dropped)) + } + } + i++ + case depth == 0: + out.WriteByte(s[i]) + } + } + if depth > 0 && dropped != nil { + // Unbalanced braces: whatever opened and never closed is dropped, as a + // template would be, rather than leaking half a template into a gloss. + dropped["(unclosed)"]++ + } + return out.String() +} + +// renderTemplate maps one template body (the text between {{ and }}) to plain +// text. body still contains any nested templates verbatim. +// +// The kept templates are the ones that carry definition text on this wiki: +// context labels in four spellings, links in five, place descriptions, +// non-gloss definitions and the two cross-reference templates a "dfn" section +// is usually made of. Everything else is presentation or classification. +func renderTemplate(body string, dropped map[string]int) string { + parts := splitTemplate(body) + name := strings.ToLower(strings.TrimSpace(parts[0])) + // Positional parameters only; key=value ones are presentation hints. + var params []string + for _, p := range parts[1:] { + if strings.Contains(p, "=") && !strings.Contains(p, "[[") && !strings.Contains(p, "{{") { + continue + } + params = append(params, strings.TrimSpace(stripTemplates(strings.TrimSpace(p), dropped))) + } + // A leading language code is markup, whether it is ours or a neighbour's + // pasted in: the label templates take it first, the link ones too. Only + // when something follows it, though: {{q|con}} is a one-word qualifier, + // not a language. + dropLang := func(ps []string) []string { + if len(ps) > 1 && isLangCode(ps[0]) { + return ps[1:] + } + return ps + } + switch name { + case "label", "lb", "nhãn", "context", "term", "gloss", "qualifier", "q": + params = dropLang(params) + if len(params) == 0 { + return "" + } + return "(" + strings.Join(params, ", ") + ")" + case "l", "vi-l", "w", "m", "link": + params = dropLang(params) + if len(params) == 0 { + return "" + } + return params[len(params)-1] + case "n-g", "non-gloss", "non-gloss definition": + return strings.Join(params, " ") + case "see-entry", "like-entry": + if len(params) == 0 { + return "" + } + return "Xem " + params[0] + case "place": + params = dropLang(params) + for i, p := range params { + // "c/Việt Nam" is a typed place: the type prefix is markup. + if len(p) > 2 && p[1] == '/' && p[0] >= 'a' && p[0] <= 'z' { + params[i] = p[2:] + } + } + return strings.Join(params, ", ") + } + if dropped != nil { + dropped[name]++ + } + return "" +} + +// isLangCode reports whether a template parameter is a language code rather +// than text: two or three lowercase ASCII letters, optionally with a +// hyphenated variant such as "nan-hbl". +func isLangCode(p string) bool { + if len(p) < 2 || len(p) > 11 { + return false + } + letters := 0 + for _, r := range p { + switch { + case r >= 'a' && r <= 'z': + letters++ + case r == '-': + if letters < 2 { + return false + } + letters = 0 + default: + return false + } + } + return letters >= 2 && letters <= 3 +} + +// splitTemplate splits a template body on | outside nested braces and +// brackets, so a link or template inside a parameter is not cut in two. +func splitTemplate(body string) []string { + var parts []string + depth := 0 + start := 0 + for i := 0; i < len(body); i++ { + switch { + case strings.HasPrefix(body[i:], "{{") || strings.HasPrefix(body[i:], "[["): + depth++ + i++ + case strings.HasPrefix(body[i:], "}}") || strings.HasPrefix(body[i:], "]]"): + if depth > 0 { + depth-- + } + i++ + case body[i] == '|' && depth == 0: + parts = append(parts, body[start:i]) + start = i + 1 + } + } + return append(parts, body[start:]) +} + +// stripLinks renders wiki links as their display text and drops category +// links, which are classification rather than definition. +func stripLinks(s string) string { + var out strings.Builder + for { + open := strings.Index(s, "[[") + if open < 0 { + break + } + close := strings.Index(s[open:], "]]") + if close < 0 { + break + } + out.WriteString(s[:open]) + inner := s[open+2 : open+close] + lower := strings.ToLower(inner) + if !strings.HasPrefix(lower, "thể loại:") && !strings.HasPrefix(lower, "category:") { + if bar := strings.LastIndex(inner, "|"); bar >= 0 { + inner = inner[bar+1:] + } + out.WriteString(inner) + } + s = s[open+close+2:] + } + out.WriteString(s) + s = out.String() + + // External links: [http://… label] → label; a bare URL in brackets goes. + out.Reset() + for { + open := strings.Index(s, "[http") + if open < 0 { + break + } + close := strings.Index(s[open:], "]") + if close < 0 { + break + } + out.WriteString(s[:open]) + inner := s[open+1 : open+close] + if sp := strings.IndexByte(inner, ' '); sp >= 0 { + out.WriteString(inner[sp+1:]) + } + s = s[open+close+1:] + } + out.WriteString(s) + return out.String() +} + +// capGloss cuts a definition longer than maxGlossRunes at the last space +// before the limit and marks the cut with an ellipsis. +func capGloss(s string) (string, bool) { + if utf8.RuneCountInString(s) <= maxGlossRunes { + return s, false + } + runes := []rune(s) + head := string(runes[:maxGlossRunes-utf8.RuneCountInString(ellipsis)]) + if sp := strings.LastIndexByte(head, ' '); sp > 0 { + head = head[:sp] + } + return strings.TrimRight(head, " ,;:") + ellipsis, true +} diff --git a/server/cmd/build-dictionary/wikitext_test.go b/server/cmd/build-dictionary/wikitext_test.go new file mode 100644 index 0000000..a597842 --- /dev/null +++ b/server/cmd/build-dictionary/wikitext_test.go @@ -0,0 +1,274 @@ +package main + +import ( + "strings" + "testing" + "unicode/utf8" +) + +func TestStripWikitext(t *testing.T) { + cases := []struct { + name string + in string + want string + }{ + {"links keep display text", + "[[chỗ|Chỗ]] [[râm]] [[mát]], do [[trời]] có [[mây]] hoặc do không bị [[nắng]] [[chiếu]].", + "Chỗ râm mát, do trời có mây hoặc do không bị nắng chiếu."}, + {"place template keeps its parameters and drops the type prefix", + "{{place|vi|thủ đô|c/Việt Nam}}.", + "thủ đô, Việt Nam."}, + {"label becomes a parenthesis", + "{{label|vi|thuộc lịch sử}} Một [[tỉnh]] cũ của [[Việt Nam]] vào nửa cuối thế kỷ XIX.", + "(thuộc lịch sử) Một tỉnh cũ của Việt Nam vào nửa cuối thế kỷ XIX."}, + {"Vietnamese label spellings", + "{{nhãn|vi|tin học}} {{context|cũ}} {{term|Hóa học}} Cấu trúc.", + "(tin học) (cũ) (Hóa học) Cấu trúc."}, + {"nested template inside a kept one", + "{{label|vi|{{w|Hà Nội}}}} Thủ đô.", + "(Hà Nội) Thủ đô."}, + {"unknown template dropped whole, nesting included", + "{{rfdef|vi|{{w|x}}}}", + ""}, + {"definition that is only a cross-reference", + "{{see-entry|bà la sát}}.", + "Xem bà la sát."}, + {"non-gloss definition", + "{{n-g|Trợ từ nhấn mạnh.}}", + "Trợ từ nhấn mạnh."}, + {"link templates keep the last parameter", + "{{l|vi|nói}}, {{l|vi|nói năng|nói năng (hiếm)}} và {{w|Việt Nam}}.", + "nói, nói năng (hiếm) và Việt Nam."}, + {"ref mid-sentence and a lone ref", + "Một loài [[cá]]Từ điển nước ngọt.", + "Một loài cá nước ngọt."}, + {"comment, bold, italic, entities", + "'''Rất''' ''nhanh'' và&mạnh.", + "Rất nhanh và&mạnh."}, + {"category link dropped, external link keeps label", + "Một [[thành phố]] [[Thể loại:Địa danh]] ([http://example.org trang web]).", + "Một thành phố (trang web)."}, + {"named parameters are not text", + "{{lb|vi|thơ ca|sort=x}} Câu.", + "(thơ ca) Câu."}, + {"a lone short parameter is text, not a language code", + "{{q|con}} Một loài vật, {{l|con}} là con.", + "(con) Một loài vật, con là con."}, + {"format and bidi characters are dropped", + "M\u200bột \u202enghĩa\u202c.", + "Một nghĩa."}, + {"a stray closer is not text", "Một }} nghĩa.", "Một nghĩa."}, + {"only punctuation is empty", "(...).", ""}, + {"control characters and whitespace collapse", " Một từ \t hai ", "Một từ hai"}, + {"unclosed template does not leak", "Một {{label|vi|x từ.", "Một"}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + if got := stripWikitext(c.in, nil); got != c.want { + t.Errorf("stripWikitext(%q)\n got %q\nwant %q", c.in, got, c.want) + } + }) + } +} + +func TestStripWikitextCountsDroppedTemplates(t *testing.T) { + dropped := make(map[string]int) + stripWikitext("{{rfdef|vi}} {{RfDef|vi}} {{senseid|vi|x}}", dropped) + if dropped["rfdef"] != 2 || dropped["senseid"] != 1 { + t.Errorf("dropped = %v, want rfdef 2 (case-folded), senseid 1", dropped) + } +} + +func TestCapGlossCutsAtAWordBoundary(t *testing.T) { + word := "từ " + long := strings.Repeat(word, 120) // 360 runes + got, cut := capGloss(long) + if !cut { + t.Fatal("a 360-rune gloss was not cut") + } + if n := utf8.RuneCountInString(got); n > maxGlossRunes { + t.Errorf("cut gloss is %d runes, want at most %d", n, maxGlossRunes) + } + if !strings.HasSuffix(got, "từ"+ellipsis) { + t.Errorf("cut gloss %q does not end on a whole word plus the ellipsis", got) + } + if short, cut := capGloss("ngắn"); cut || short != "ngắn" { + t.Errorf("a short gloss was changed: %q %v", short, cut) + } +} + +const legacyPage = `{{-vie-}} +{{-pron-}} +{{vie-pron|học sinh}} + +{{-noun-}} +# [[người|Người]] [[học]] ở [[nhà trường|trường]]. +#: ''Học sinh giỏi.'' +#{{label|vi|cũ}} [[môn đệ|Môn đệ]]. + +{{-verb-}} +# [[đi học|Đi học]]. + +{{-trans-}} +* {{eng}}: {{t|en|student}} + +{{-eng-}} +{{-noun-}} +# Student, in English. +` + +const newPage = `== {{langname|vi}} == +=== {{ĐM|etym}} === +Hán-Việt. + +=== {{ĐM|pr-noun}} === +{{vi-pr-noun}} + +# {{place|vi|thủ đô|c/Việt Nam}}. + +=== {{section|v}} === +# [[đi|Đi]] về thủ đô. + +=== {{ĐM|xyz}} === +# Một nghĩa dưới đề mục lạ. + +== {{langname|en}} == +=== {{ĐM|pr-noun}} === +# The capital of Vietnam. +` + +func TestVietnameseSectionLegacy(t *testing.T) { + section, dialect, both, _ := vietnameseSection(legacyPage) + if dialect != "legacy" || both { + t.Fatalf("dialect = %q both = %v, want legacy false", dialect, both) + } + if strings.Contains(section, "Student") { + t.Error("the English section leaked into the Vietnamese one") + } + if !strings.Contains(section, "{{-trans-}}") { + t.Error("the translations heading, a section heading rather than a language, cut the section short") + } +} + +func TestVietnameseSectionNew(t *testing.T) { + section, dialect, both, _ := vietnameseSection(newPage) + if dialect != "new" || both { + t.Fatalf("dialect = %q both = %v, want new false", dialect, both) + } + if strings.Contains(section, "capital of Vietnam") { + t.Error("the English section leaked into the Vietnamese one") + } + if !strings.Contains(section, "đề mục lạ") { + t.Error("a level-3 heading ended the section; only a level-2 heading may") + } +} + +func TestVietnameseSectionAbsentAndBoth(t *testing.T) { + if _, dialect, _, _ := vietnameseSection("{{-eng-}}\n{{-noun-}}\n# Word."); dialect != "" { + t.Errorf("an English-only page reported dialect %q", dialect) + } + mixed := "== {{langname|vi}} ==\n# Mới.\n" + legacyPage + section, dialect, both, _ := vietnameseSection(mixed) + if dialect != "new" || !both { + t.Errorf("dialect = %q both = %v, want the first dialect in the text and both=true", dialect, both) + } + if strings.Contains(section, "Người học") { + t.Error("the first section should end where the legacy page starts a language section") + } +} + +func TestDefinitionsLegacy(t *testing.T) { + stats := newSectionStats() + section, _, _, _ := vietnameseSection(legacyPage) + got := definitions(section, stats) + want := []sense{ + {"danh từ", "Người học ở trường."}, + {"danh từ", "(cũ) Môn đệ."}, + {"động từ", "Đi học."}, + } + assertSenses(t, got, want) + if stats.pos["noun"] != 1 || stats.pos["verb"] != 1 { + t.Errorf("pos tally = %v, want noun 1 verb 1", stats.pos) + } + if stats.pos["pron"] != 0 || stats.pos["trans"] != 0 { + t.Errorf("pronunciation and translations were tallied as parts of speech: %v", stats.pos) + } + if stats.defsKept != 3 { + t.Errorf("defsKept = %d, want 3", stats.defsKept) + } +} + +func TestDefinitionsNewDialect(t *testing.T) { + stats := newSectionStats() + section, _, _, _ := vietnameseSection(newPage) + got := definitions(section, stats) + want := []sense{ + {"danh từ riêng", "thủ đô, Việt Nam."}, + {"động từ", "Đi về thủ đô."}, + {"", "Một nghĩa dưới đề mục lạ."}, + } + assertSenses(t, got, want) + if stats.unmappedPos["xyz"] != 1 { + t.Errorf("unmapped headings = %v, want xyz 1", stats.unmappedPos) + } + if stats.unmappedPos["etym"] != 0 { + t.Errorf("etymology counted as an unmapped part of speech: %v", stats.unmappedPos) + } +} + +func TestDefinitionsHeadwordLineAndWrittenOutHeading(t *testing.T) { + section := "=== Danh từ ===\n# Một.\n{{vi-verb}}\n# Hai.\n{{vi-pron}}\n# Ba.\n=== Phát âm ===\n# Bốn." + got := definitions(section, newSectionStats()) + want := []sense{{"danh từ", "Một."}, {"động từ", "Hai."}, {"động từ", "Ba."}, {"", "Bốn."}} + assertSenses(t, got, want) +} + +func TestDefinitionsSkipsEmptyAndCaps(t *testing.T) { + stats := newSectionStats() + lines := []string{"{{-noun-}}", "# {{rfdef|vi}}", "#: not a definition", "#* nor this", "## nor this"} + for i := 0; i < 7; i++ { + lines = append(lines, "# Nghĩa số "+string(rune('a'+i))+".") + } + lines = append(lines, "# "+strings.Repeat("dài ", 80)) + got := definitions(strings.Join(lines, "\n"), stats) + if len(got) != maxSenses { + t.Fatalf("got %d senses, want the cap of %d", len(got), maxSenses) + } + if got[0].gloss != "Nghĩa số a." { + t.Errorf("first sense = %q, want the first real definition after the empty one", got[0].gloss) + } + if stats.defsEmpty != 1 || stats.dropped["rfdef"] != 1 { + t.Errorf("empty = %d dropped = %v, want 1 and rfdef 1", stats.defsEmpty, stats.dropped) + } + if stats.defsKept != 8 || stats.defsCut != 1 { + t.Errorf("kept = %d cut = %d, want 8 and 1 (counted past the cap)", stats.defsKept, stats.defsCut) + } +} + +func assertSenses(t *testing.T, got, want []sense) { + t.Helper() + if len(got) != len(want) { + t.Fatalf("got %d senses %v, want %d %v", len(got), got, len(want), want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("sense %d = %+v, want %+v", i, got[i], want[i]) + } + } +} + +// A few pages run the section marker and the headings together on one line: +// {{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}. The marker must still +// open the section and the last heading on the line must still label it. +func TestVietnameseSectionInlineHeadings(t *testing.T) { + page := "{{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}\n\n'''Thượng Hải'''\n# Thành phố lớn nhất [[Trung Quốc]].\n{{-eng-}}{{-noun-}}\n# Shanghai." + section, dialect, _, ender := vietnameseSection(page) + if dialect != "legacy" || ender != "eng" { + t.Fatalf("dialect = %q ender = %q, want legacy ended by eng", dialect, ender) + } + if strings.Contains(section, "Shanghai") { + t.Error("the English section, opened on a shared line, leaked in") + } + got := definitions(section, newSectionStats()) + assertSenses(t, got, []sense{{"địa danh", "Thành phố lớn nhất Trung Quốc."}}) +} diff --git a/server/cmd/noitu-server/main.go b/server/cmd/noitu-server/main.go index a25b7ff..5171d91 100644 --- a/server/cmd/noitu-server/main.go +++ b/server/cmd/noitu-server/main.go @@ -62,6 +62,7 @@ func run() error { "path", cfg.dbPath, "words", store.WordCount(), "aliases", store.AliasCount(), + "meanings", store.MeaningCount(), "license", store.License(), ) diff --git a/server/internal/dictionary/store.go b/server/internal/dictionary/store.go index f2729a4..91c23b0 100644 --- a/server/internal/dictionary/store.go +++ b/server/internal/dictionary/store.go @@ -21,6 +21,7 @@ import ( "iter" "net/url" "os" + "slices" "sort" "strconv" @@ -32,6 +33,10 @@ import ( // ErrNotFound is returned when a syllable has no entry in the dictionary. var ErrNotFound = errors.New("dictionary: syllable not found") +// requiredBuilderVersion is the builder whose meta contract this store reads; +// it is named in the refusal of an older database. +const requiredBuilderVersion = "5" + // wordInfo holds the two syllables the chain rule needs. Both ends are kept: // canonicalization can move either one, so the engine must never re-derive // them from what the player typed. @@ -40,6 +45,15 @@ type wordInfo struct { last string } +// Sense is one definition of a word as Wiktionary gives it: the Vietnamese +// part-of-speech label of the heading it sat under ("danh từ"), empty when the +// builder did not know the heading, and the definition as plain text. Neither +// is markup; the client renders both as text. +type Sense struct { + Pos string + Gloss string +} + // Store answers word and syllable queries against the derived dictionary. // // Every field is written once during Open and only read afterwards, and no @@ -54,6 +68,10 @@ type Store struct { // openers holds words whose last syllable has at least one continuation, // sorted by that count descending so an eligible set is always a prefix. openers []opener + // meanings holds each word's senses in page order. A few megabytes of text + // for the corpus; a per-move query would be a second code path for nothing. + meanings map[string][]Sense + meaningCount int license string } @@ -87,11 +105,12 @@ func Open(path string) (*Store, error) { aliases: make(map[string]string), byFirst: make(map[string][]string), outDegree: make(map[string]int), + meanings: make(map[string][]Sense), } // Reading meta first also rejects an unrelated database before any bulk // loading happens. - declaredWords, err := s.loadMeta(db) + declaredWords, declaredMeanings, err := s.loadMeta(db) if err != nil { return nil, err } @@ -106,7 +125,10 @@ func Open(path string) (*Store, error) { if err := s.loadAliases(db); err != nil { return nil, err } - if err := s.validate(declaredWords); err != nil { + if err := s.loadMeanings(db); err != nil { + return nil, err + } + if err := s.validate(declaredWords, declaredMeanings); err != nil { return nil, err } @@ -125,22 +147,37 @@ func dsn(path string) string { return u.String() } -func (s *Store) loadMeta(db *sql.DB) (declaredWords int, err error) { +func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err error) { // The data is CC BY-SA 4.0 and its provenance travels with it. if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&s.license); err != nil { - return 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err) + return 0, 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err) } - var raw string - if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'word_count'`).Scan(&raw); err != nil { - return 0, fmt.Errorf("read dictionary word_count: %w", err) + count := func(key string) (int, error) { + var raw string + if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil { + if errors.Is(err, sql.ErrNoRows) { + // A database from before the key existed: the fix is a rebuild, + // so say so rather than naming a missing row. + return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it", + key, requiredBuilderVersion) + } + return 0, fmt.Errorf("read dictionary %s: %w", key, err) + } + n, err := strconv.Atoi(raw) + if err != nil { + return 0, fmt.Errorf("dictionary %s %q is not a number: %w", key, raw, err) + } + return n, nil } - declaredWords, err = strconv.Atoi(raw) - if err != nil { - return 0, fmt.Errorf("dictionary word_count %q is not a number: %w", raw, err) + if declaredWords, err = count("word_count"); err != nil { + return 0, 0, err + } + if declaredMeanings, err = count("meaning_count"); err != nil { + return 0, 0, err } - return declaredWords, nil + return declaredWords, declaredMeanings, nil } func (s *Store) loadSyllables(db *sql.DB) error { @@ -204,13 +241,35 @@ func (s *Store) loadAliases(db *sql.DB) error { return rows.Err() } +func (s *Store) loadMeanings(db *sql.DB) error { + // Ordered by (word, ord), the primary key, so each word's senses arrive in + // page order and append in it. + rows, err := db.Query(`SELECT word, pos, gloss FROM meanings ORDER BY word, ord`) + if err != nil { + return fmt.Errorf("load meanings: %w", err) + } + defer rows.Close() + + for rows.Next() { + var word string + var sense Sense + if err := rows.Scan(&word, &sense.Pos, &sense.Gloss); err != nil { + return fmt.Errorf("scan meaning: %w", err) + } + s.meanings[word] = append(s.meanings[word], sense) + s.meaningCount++ + } + + return rows.Err() +} + // validate rejects a structurally valid but wrong dictionary. // // A truncated or empty database has the right schema and opens cleanly, and // the server would then start, reject every word a player types, and fail // every room creation. Checking the loaded rows against what the builder // recorded turns that into a startup failure. -func (s *Store) validate(declaredWords int) error { +func (s *Store) validate(declaredWords, declaredMeanings int) error { if len(s.words) != declaredWords { return fmt.Errorf("dictionary is incomplete: metadata declares %d words, loaded %d", declaredWords, len(s.words)) @@ -218,6 +277,17 @@ func (s *Store) validate(declaredWords int) error { if len(s.words) == 0 { return errors.New("dictionary contains no words") } + // A meanings table truncated on disk would otherwise be served silently + // as a dictionary without meanings. + if s.meaningCount != declaredMeanings { + return fmt.Errorf("dictionary is incomplete: metadata declares %d meanings, loaded %d", + declaredMeanings, s.meaningCount) + } + for word := range s.meanings { + if _, ok := s.words[word]; !ok { + return fmt.Errorf("dictionary is inconsistent: meaning for %q, which is not a word", word) + } + } // A stale syllables table would tell the bot a syllable has continuations // that WordsStartingWith cannot supply. @@ -243,6 +313,20 @@ func (s *Store) WordCount() int { return len(s.words) } // AliasCount reports how many alternative spellings are accepted. func (s *Store) AliasCount() int { return len(s.aliases) } +// MeaningCount reports how many senses the dictionary holds across all words. +func (s *Store) MeaningCount() int { return s.meaningCount } + +// Meanings returns a canonical word's senses in page order, at most five, or +// nil for a word with none. Resolve first: an alias has no senses of its own. +// The slice is a copy, so a caller cannot reach dictionary state through it. +func (s *Store) Meanings(word string) []Sense { + senses := s.meanings[word] + if len(senses) == 0 { + return nil + } + return slices.Clone(senses) +} + // License reports the licence the dictionary data is distributed under. // Callers are expected to state it at startup. func (s *Store) License() string { return s.license } diff --git a/server/internal/dictionary/store_test.go b/server/internal/dictionary/store_test.go index 51fb6e8..b570147 100644 --- a/server/internal/dictionary/store_test.go +++ b/server/internal/dictionary/store_test.go @@ -20,6 +20,7 @@ CREATE TABLE words (word TEXT PRIMARY KEY, first TEXT NOT NULL, last TEXT NOT NU CREATE INDEX idx_words_first ON words(first); CREATE TABLE syllables (syllable TEXT PRIMARY KEY, out_degree INTEGER NOT NULL) WITHOUT ROWID; CREATE TABLE aliases (variant TEXT PRIMARY KEY, canonical TEXT NOT NULL) WITHOUT ROWID; +CREATE TABLE meanings (word TEXT NOT NULL, ord INTEGER NOT NULL, pos TEXT NOT NULL, gloss TEXT NOT NULL, PRIMARY KEY (word, ord)) WITHOUT ROWID; CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL); ` @@ -40,7 +41,7 @@ func fixtureAt(tb testing.TB, dir string) string { defer db.Close() data := fixtureSchema + ` -INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7'); +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7'),('meaning_count','3'); INSERT INTO words VALUES ('pháp luật','pháp','luật',2), ('pháp lý','pháp','lý',2), @@ -55,6 +56,11 @@ INSERT INTO syllables VALUES ('pháp',2),('luật',1),('lý',1),('lệ',0),('do',0),('vô',1),('điện',0),('công',1),('dầu',0),('ngữ',1); -- "pháp lí" drifts in the LAST syllable, "luâto lệ" in the FIRST. INSERT INTO aliases VALUES ('pháp lí','pháp lý'),('luâto lệ','luật lệ'); +-- Inserted out of order to prove the store sorts by ord, not by insertion. +INSERT INTO meanings VALUES + ('pháp luật',1,'','Kỷ cương nói chung.'), + ('pháp luật',0,'danh từ','Hệ thống các quy tắc xử sự do nhà nước đặt ra.'), + ('ngữ pháp',0,'danh từ','Toàn bộ những quy tắc hoạt động của ngôn ngữ.'); ` if _, err := db.Exec(data); err != nil { tb.Fatal(err) @@ -106,7 +112,7 @@ func TestOpenWrongSchema(t *testing.T) { // every room creation. func TestOpenEmptyDictionary(t *testing.T) { path := writeDB(t, fixtureSchema+` -INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999');`) +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999'),('meaning_count','0');`) _, err := Open(path) if err == nil { @@ -121,7 +127,7 @@ INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999') // syllable has continuations that cannot be supplied. func TestOpenInconsistentOutDegree(t *testing.T) { path := writeDB(t, fixtureSchema+` -INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'); +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0'); INSERT INTO words VALUES ('pháp luật','pháp','luật',2); INSERT INTO syllables VALUES ('pháp',7),('luật',0);`) @@ -132,7 +138,7 @@ INSERT INTO syllables VALUES ('pháp',7),('luật',0);`) func TestOpenOrphanAlias(t *testing.T) { path := writeDB(t, fixtureSchema+` -INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'); +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0'); INSERT INTO words VALUES ('pháp luật','pháp','luật',2); INSERT INTO syllables VALUES ('pháp',1),('luật',0); INSERT INTO aliases VALUES ('phap luat','không tồn tại');`) @@ -541,3 +547,73 @@ func BenchmarkRandomOpeningWord(b *testing.B) { } } } + +func TestMeaningsAreOrderedAndCopied(t *testing.T) { + s := fixture(t) + + got := s.Meanings("pháp luật") + want := []Sense{ + {Pos: "danh từ", Gloss: "Hệ thống các quy tắc xử sự do nhà nước đặt ra."}, + {Pos: "", Gloss: "Kỷ cương nói chung."}, + } + if !slices.Equal(got, want) { + t.Errorf("Meanings(pháp luật) = %v, want %v (ordered by ord, not insertion)", got, want) + } + // A caller writing into the slice must not reach the store. + got[0].Gloss = "changed" + if s.Meanings("pháp luật")[0].Gloss != want[0].Gloss { + t.Error("Meanings handed out the store's own slice") + } + + if s.Meanings("pháp lý") != nil { + t.Error("a word with no senses returned a non-nil slice") + } + // An alias is not a word: callers Resolve first. + if s.Meanings("pháp lí") != nil { + t.Error("an alias returned senses of its own") + } + if s.MeaningCount() != 3 { + t.Errorf("MeaningCount = %d, want 3", s.MeaningCount()) + } +} + +// A meanings table truncated on disk must be refused at startup rather than +// served silently as a dictionary without meanings. +func TestOpenRefusesMismatchedMeaningCount(t *testing.T) { + path := writeDB(t, fixtureSchema+` +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','2'); +INSERT INTO words VALUES ('pháp luật','pháp','luật',2); +INSERT INTO syllables VALUES ('pháp',1),('luật',0); +INSERT INTO meanings VALUES ('pháp luật',0,'danh từ','Luật.');`) + + _, err := Open(path) + if err == nil || !strings.Contains(err.Error(), "meanings") { + t.Fatalf("Open = %v, want a refusal naming the meanings count", err) + } +} + +// A database built before meanings existed opens cleanly and has every table +// but one row. The refusal must say what to do, not which row is missing. +func TestOpenRefusesOlderBuilderVersion(t *testing.T) { + path := writeDB(t, fixtureSchema+` +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'); +INSERT INTO words VALUES ('pháp luật','pháp','luật',2); +INSERT INTO syllables VALUES ('pháp',1),('luật',0);`) + + _, err := Open(path) + if err == nil || !strings.Contains(err.Error(), "make dict") { + t.Fatalf("Open = %v, want a refusal that says to rebuild", err) + } +} + +func TestOpenRefusesOrphanMeaning(t *testing.T) { + path := writeDB(t, fixtureSchema+` +INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','1'); +INSERT INTO words VALUES ('pháp luật','pháp','luật',2); +INSERT INTO syllables VALUES ('pháp',1),('luật',0); +INSERT INTO meanings VALUES ('không tồn tại',0,'','Một nghĩa.');`) + + if _, err := Open(path); err == nil { + t.Fatal("Open succeeded with a meaning for a word that does not exist") + } +} diff --git a/testdata/fixture-words.txt b/testdata/fixture-words.txt index 1a849a7..48ba672 100644 --- a/testdata/fixture-words.txt +++ b/testdata/fixture-words.txt @@ -11,232 +11,238 @@ # The graph is built around one hub syllable, "sinh". It is the only syllable # with enough continuations to be chosen as an opening, so every game starts on # a word ending in "sinh" and a scripted test always knows the first answer. +# +# A line is the word, then optional tab-separated meanings; a meaning is +# `pos|gloss` or just `gloss`. Hand-written, kept short, and present on more +# than half the list so the fixture database exercises the meanings path the +# real one does. Every opening word and every "sinh" word has one, because +# those are the words the browser suite reads a meaning from. # --- openings: words ending in the hub syllable --- -học sinh -thí sinh -vệ sinh -phát sinh -khai sinh -tái sinh -dân sinh -nhân sinh -ký sinh -chúng sinh +học sinh danh từ|Người học ở trường phổ thông. +thí sinh danh từ|Người dự thi. +vệ sinh danh từ|Sự giữ gìn sạch sẽ để phòng bệnh. +phát sinh động từ|Nảy sinh, xuất hiện. +khai sinh động từ|Đăng ký việc sinh ra của một người. +tái sinh động từ|Sinh ra lần nữa; làm sống lại. +dân sinh danh từ|Đời sống của nhân dân. +nhân sinh danh từ|Cuộc sống con người. +ký sinh động từ|Sống nhờ vào cơ thể sinh vật khác. +chúng sinh danh từ|Mọi loài có sự sống, theo đạo Phật. # --- the hub: words starting with "sinh" --- -sinh viên -sinh sản -sinh hoạt -sinh học -sinh nhật -sinh vật -sinh tồn -sinh thái -sinh lý -sinh kế -sinh khí -sinh mệnh -sinh trưởng -sinh động -sinh sống -sinh thành -sinh lực -sinh quán -sinh sôi -sinh nở -sinh trắc -sinh hóa +sinh viên danh từ|Người học ở trường đại học, cao đẳng. +sinh sản động từ|Tạo ra thế hệ sau để duy trì loài. +sinh hoạt động từ|Sống và hoạt động hằng ngày. danh từ|Hoạt động tập thể có tổ chức. +sinh học danh từ|Khoa học về sự sống. +sinh nhật danh từ|Ngày kỷ niệm ngày sinh. +sinh vật danh từ|Vật có sự sống. +sinh tồn động từ|Sống và giữ được sự sống. +sinh thái danh từ|Quan hệ giữa sinh vật và môi trường. +sinh lý danh từ|Hoạt động bình thường của cơ thể sống. +sinh kế danh từ|Cách kiếm sống. +sinh khí danh từ|Sức sống, vẻ hoạt bát. +sinh mệnh danh từ|Mạng sống. +sinh trưởng động từ|Lớn lên về kích thước và khối lượng. +sinh động tính từ|Có sức sống, gợi được hình ảnh rõ. +sinh sống động từ|Sống ở một nơi. +sinh thành động từ|Sinh ra và nuôi dạy. +sinh lực danh từ|Sức sống của cơ thể. +sinh quán danh từ|Nơi sinh. +sinh sôi động từ|Sinh ra ngày càng nhiều. +sinh nở động từ|Đẻ con. +sinh trắc danh từ|Phép đo các đặc điểm sinh học của con người. +sinh hóa danh từ|Hóa học của sự sống. # --- continuations --- -viên chức +viên chức danh từ|Người làm việc trong cơ quan nhà nước. viên mãn -chức năng +chức năng danh từ|Tác dụng, vai trò của một bộ phận. chức vụ -năng lực +năng lực danh từ|Khả năng làm được việc. năng suất mãn nguyện -nguyện vọng +nguyện vọng danh từ|Điều mong muốn. vọng tưởng -sản xuất -sản phẩm +sản xuất động từ|Tạo ra của cải vật chất. +sản phẩm danh từ|Vật do lao động tạo ra. sản lượng -xuất bản +xuất bản động từ|In và phát hành sách báo. xuất phát phẩm chất lượng giác -bản đồ +bản đồ danh từ|Hình vẽ thu nhỏ một vùng đất. bản sắc đồ thị sắc thái -hoạt động +hoạt động động từ|Làm những việc có mục đích. hoạt bát -động vật -động cơ +động vật danh từ|Sinh vật có cảm giác và tự vận động được. +động cơ danh từ|Máy biến năng lượng thành chuyển động. động lực động đất -cơ bản -cơ hội -hội nghị +cơ bản tính từ|Là gốc, là nền tảng. +cơ hội danh từ|Dịp thuận lợi. +hội nghị danh từ|Cuộc họp bàn công việc. nghị luận -đất nước -nước ngoài +đất nước danh từ|Lãnh thổ của một dân tộc; quốc gia. +nước ngoài danh từ|Nước khác, ngoài nước mình. ngoài trời -trời đất +trời đất danh từ|Trời và đất; thế gian. -học tập -học phí +học tập động từ|Học và luyện tập để có hiểu biết. +học phí danh từ|Tiền trả cho việc học. học hỏi học bổng -tập trung +tập trung động từ|Dồn vào một chỗ, một việc. tập thể -trung tâm -tâm hồn +trung tâm danh từ|Điểm ở giữa; nơi tập trung hoạt động. +tâm hồn danh từ|Ý nghĩ và tình cảm của con người. hồn nhiên -nhiên liệu +nhiên liệu danh từ|Chất đốt sinh năng lượng. liệu pháp -pháp luật -pháp lý -luật sư +pháp luật danh từ|Quy tắc xử sự do nhà nước đặt ra. +pháp lý danh từ|Lý luận, căn cứ về pháp luật. +luật sư danh từ|Người bảo vệ quyền lợi trước tòa theo luật. sư phạm -phạm vi -vi phạm +phạm vi danh từ|Giới hạn của một hoạt động. +vi phạm động từ|Làm trái quy định. -nhật ký +nhật ký danh từ|Sổ ghi việc hằng ngày. nhật báo -báo cáo +báo cáo động từ|Trình bày kết quả công việc. cáo trạng -trạng thái -thái độ -thái bình +trạng thái danh từ|Tình trạng tồn tại của sự vật. +thái độ danh từ|Cách nghĩ, cách nhìn thể hiện ra ngoài. +thái bình tính từ|Yên ổn, không loạn lạc. độ cao cao nguyên -bình minh +bình minh danh từ|Lúc mặt trời mọc. minh bạch bạch tuộc -vật chất +vật chất danh từ|Cái tồn tại khách quan ngoài ý thức. vật liệu -chất lượng +chất lượng danh từ|Cái tạo nên phẩm chất của sự vật. -tồn tại +tồn tại động từ|Có thật, đang có. tồn kho tại chỗ kho tàng tàng hình -hình ảnh -ảnh hưởng +hình ảnh danh từ|Hình người, vật hiện ra hoặc được ghi lại. +ảnh hưởng động từ|Tác động đến. hưởng thụ -lý do -lý thuyết -lý luận -lý tưởng +lý do danh từ|Điều làm căn cứ để giải thích. +lý thuyết danh từ|Hệ thống tư tưởng khái quát về một lĩnh vực. +lý luận danh từ|Hệ thống luận điểm về một lĩnh vực. +lý tưởng danh từ|Mục đích cao đẹp hướng tới. thuyết minh luận điểm -tưởng tượng +tưởng tượng động từ|Tạo ra trong trí hình ảnh chưa thấy. điểm danh -danh sách +danh sách danh từ|Bảng ghi tên theo thứ tự. sách vở tượng trưng -kế hoạch -kế toán +kế hoạch danh từ|Toàn bộ dự định làm việc theo trình tự. +kế toán danh từ|Việc ghi chép, tính toán thu chi. kế tiếp hoạch định -toán học -tiếp tục +toán học danh từ|Khoa học về số, hình và cấu trúc. +tiếp tục động từ|Làm tiếp việc đang làm. định hướng -tục ngữ -hướng dẫn -ngữ pháp +tục ngữ danh từ|Câu ngắn gọn đúc kết kinh nghiệm dân gian. +hướng dẫn động từ|Chỉ bảo cách làm. +ngữ pháp danh từ|Quy tắc kết hợp từ thành câu. dẫn chứng -chứng minh +chứng minh động từ|Làm rõ là đúng bằng lý lẽ, bằng chứng. -khí hậu +khí hậu danh từ|Thời tiết trung bình nhiều năm của một vùng. khí quyển khí thế -hậu quả +hậu quả danh từ|Kết quả không hay về sau. quyển sách -thế giới +thế giới danh từ|Trái đất và mọi thứ trên đó. quả cảm -giới hạn +giới hạn danh từ|Mức không thể vượt qua. hạn chế -chế độ +chế độ danh từ|Hệ thống tổ chức chính trị, xã hội. -mệnh lệnh +mệnh lệnh danh từ|Lời sai bảo phải làm theo. mệnh đề lệnh cấm -đề tài +đề tài danh từ|Vấn đề được chọn để nghiên cứu. cấm vận -tài liệu -vận động +tài liệu danh từ|Văn bản dùng để tra cứu. +vận động động từ|Di chuyển, thay đổi vị trí. trưởng thành trưởng phòng -thành phố -thành công -thành lập -thành viên +thành phố danh từ|Đô thị lớn, đông dân. +thành công động từ|Đạt được kết quả mong muốn. +thành lập động từ|Lập nên một tổ chức. +thành viên danh từ|Người thuộc một tổ chức. phòng ban phố cổ -công việc -công nghiệp -lập trình +công việc danh từ|Việc phải làm. +công nghiệp danh từ|Ngành kinh tế sản xuất bằng máy móc. +lập trình động từ|Viết chương trình cho máy tính. việc làm nghiệp vụ -trình độ +trình độ danh từ|Mức đạt được về kiến thức, kỹ năng. làm việc -cổ điển +cổ điển tính từ|Thuộc thời xưa và có giá trị lâu dài. điển hình sống động -lực lượng +lực lượng danh từ|Sức mạnh của người hay vật. lực sĩ lượng tử sĩ quan tử tế -quan hệ -hệ thống -thống nhất +quan hệ danh từ|Sự gắn liền giữa hai hay nhiều sự vật. +hệ thống danh từ|Tập hợp các yếu tố có quan hệ với nhau. +thống nhất động từ|Hợp lại thành một khối. nhất trí -trí tuệ +trí tuệ danh từ|Khả năng nhận thức và suy nghĩ. tuệ giác quán triệt -triệt để +triệt để tính từ|Đến tận cùng, không nửa vời. để dành dành dụm sôi nổi -nổi tiếng -tiếng nói -nói chuyện +nổi tiếng tính từ|Được nhiều người biết đến. +tiếng nói danh từ|Lời nói; ngôn ngữ. +nói chuyện động từ|Trò chuyện với nhau. chuyện trò -trò chơi +trò chơi danh từ|Hoạt động để vui chơi, giải trí. chơi đùa nở hoa -hoa quả -hóa học +hoa quả danh từ|Các loại quả ăn được. +hóa học danh từ|Khoa học về chất và biến đổi của chất. hóa đơn -đơn giản +đơn giản tính từ|Không phức tạp. giản dị dị thường -thường xuyên +thường xuyên tính từ|Đều đặn, liên tục. xuyên tạc trắc nghiệm nghiệm thu -thu nhập +thu nhập danh từ|Tiền kiếm được trong một thời gian. nhập khẩu khẩu hiệu -hiệu quả +hiệu quả danh từ|Kết quả đích thực. # --- a few longer words, so syllable bonuses and the 3+ badge have cases --- -vô tuyến điện -công nghiệp hóa +vô tuyến điện danh từ|Kỹ thuật truyền tin bằng sóng điện từ. +công nghiệp hóa Quá trình phát triển công nghiệp trong nền kinh tế. tổng hợp chất -điện thoại di động +điện thoại di động danh từ|Điện thoại cầm tay dùng sóng vô tuyến. diff --git a/web/e2e/fixture-dictionary.js b/web/e2e/fixture-dictionary.js index 06c279d..4cf29a8 100644 --- a/web/e2e/fixture-dictionary.js +++ b/web/e2e/fixture-dictionary.js @@ -12,11 +12,35 @@ import { fileURLToPath } from 'node:url'; */ const listPath = fileURLToPath(new URL('../../testdata/fixture-words.txt', import.meta.url)); -const words = readFileSync(listPath, 'utf8') +// A line is the word, then optional tab-separated meanings; only the word is +// part of the graph, and the meanings are kept so a test can say what the +// chain should show for whichever word was drawn. +const lines = readFileSync(listPath, 'utf8') .split('\n') .map((line) => line.trim()) .filter((line) => line && !line.startsWith('#')); +const words = lines.map((line) => line.split('\t')[0].trim()); + +/** @type {Map} word → its first sense as the chain renders it */ +const firstSense = new Map(); +for (const line of lines) { + const [word, sense] = line.split('\t').map((cell) => cell.trim()); + if (!sense) continue; + const [pos, gloss] = sense.includes('|') ? sense.split('|') : ['', sense]; + firstSense.set(word, pos ? `(${pos}) ${gloss}` : gloss); +} + +/** + * The first sense of a fixture word, rendered as the chain renders it: + * `(pos) gloss`, or the gloss alone. Undefined for a word without one. + * + * @param {string} word + */ +export function renderedSense(word) { + return firstSense.get(word); +} + /** @type {Map} */ const byFirstSyllable = new Map(); for (const word of words) { diff --git a/web/tests/dictionary-source.test.js b/web/tests/dictionary-source.test.js index a1eb9be..6b679f2 100644 --- a/web/tests/dictionary-source.test.js +++ b/web/tests/dictionary-source.test.js @@ -1,7 +1,7 @@ -// The upstream dictionary URL lives in three places: the Makefile, which +// The upstream dump URL lives in three places: the Makefile, which // builds it for a developer, the Dockerfile, which builds it for the image, and // the builder, which stamps it into the database. They have to agree, or the -// container ships a wordlist nobody tested against. The docs that quote the +// container ships a dictionary nobody tested against. The docs that quote the // URL are held to the same copy. // // This lives in the JavaScript suite for no better reason than that it is the @@ -26,7 +26,7 @@ function pin(source, pattern, what) { return match?.[1].trim(); } -describe('the upstream dictionary export', () => { +describe('the upstream Wiktionary dump', () => { const makeUrl = pin(makefile, /DICT_URL\s*:?=\s*(\S+)/, 'DICT_URL in the Makefile'); const dockerUrl = pin(dockerfile, /ARG DICT_URL=(\S+)/, 'DICT_URL in the Dockerfile'); @@ -34,10 +34,13 @@ describe('the upstream dictionary export', () => { expect(dockerUrl).toBe(makeUrl); }); - it('is the Vietnamese-language file of the Vietnamese Wiktionary edition', () => { - // The path carries a space, so it must stay percent-encoded or make and - // sh will split it; and it must be the vi edition, not the English one. - expect(makeUrl).toMatch(/^https:\/\/kaikki\.org\/viwiktionary\/Ti%E1%BA%BFng%20Vi%E1%BB%87t\/[^\s/]+\.jsonl$/); + it('is the rolling pages-articles dump of the Vietnamese Wiktionary edition', () => { + // The vi edition, not the English one; the current-revisions file, not + // the full history; and `latest/`, which the owner chose over a dated + // pin. Any of the three changing is a decision, not a typo. + expect(makeUrl).toMatch( + /^https:\/\/dumps\.wikimedia\.org\/viwiktionary\/latest\/viwiktionary-latest-pages-articles\.xml\.bz2$/ + ); }); it('is the URL the builder stamps into the database', () => { @@ -45,10 +48,10 @@ describe('the upstream dictionary export', () => { // constant. The three copies must agree or the attribution record names // a file nobody downloaded. const builder = readFileSync( - fileURLToPath(new URL('../../server/cmd/build-dictionary/kaikki_list.go', import.meta.url)), + fileURLToPath(new URL('../../server/cmd/build-dictionary/dump.go', import.meta.url)), 'utf8' ); - const builderUrl = pin(builder, /kaikkiSourceURL\s*=\s*"([^"]+)"/, 'kaikkiSourceURL in the builder'); + const builderUrl = pin(builder, /dumpSourceURL\s*=\s*"([^"]+)"/, 'dumpSourceURL in the builder'); expect(builderUrl).toBe(makeUrl); });