mirror of
https://github.com/tiennm99/noitu.git
synced 2026-10-11 03:13:45 +00:00
feat(dict): build the corpus and word meanings from the Wikimedia viwiktionary dump
reader for both wikitext dialects; `meanings(word, ord, pos, gloss)` table; `meaning_count`/`words_with_meaning`/`source_pages` in meta, `source_rows` gone, builder_version 5; `--dump`/`--min-pages` replace `--kaikki`; attribution names the dump and the definition excerpts; 36,200 words, 96.9% with a meaning, every kaikki word kept.
This commit is contained in:
1 parent
557de1af94
commit
80f216d56b
25 files changed
+2336
-755
No files matched your search
@@ -137,9 +137,10 @@ jobs:
|
||||
grep -qx "$required" files.txt || { echo "missing from the image: $required"; exit 1; }
|
||||
done
|
||||
|
||||
# The upstream export must never reach the final image.
|
||||
if grep -q '\.jsonl$' files.txt; then
|
||||
echo "the upstream wordlist leaked into the image"
|
||||
# The upstream dump, compressed or not, must never reach the final
|
||||
# image.
|
||||
if grep -Eq '\.(bz2|xml)$' files.txt; then
|
||||
echo "the upstream dump leaked into the image"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
|
||||
+7
-5
@@ -1,9 +1,11 @@
|
||||
# Dictionary data — never committed.
|
||||
# data/kaikki-viwiktionary-vi.jsonl is the upstream export; data/noitu.db is
|
||||
# derived from it. Both are build artifacts produced by `make fetch-dict` and
|
||||
# `make dict`.
|
||||
data/*.jsonl
|
||||
data/*.jsonl.part
|
||||
# data/viwiktionary-latest-pages-articles.xml.bz2 is the upstream dump;
|
||||
# data/noitu.db is derived from it. Both are build artifacts produced by
|
||||
# `make fetch-dict` and `make dict`. A plain .xml would be the dump
|
||||
# decompressed by hand.
|
||||
data/*.bz2
|
||||
data/*.bz2.part
|
||||
data/*.xml
|
||||
data/*.db
|
||||
data/*.db-journal
|
||||
data/*.db-wal
|
||||
|
||||
+8
-8
@@ -1,7 +1,7 @@
|
||||
# One image: the binary, the built frontend, and the derived dictionary.
|
||||
#
|
||||
# The upstream wordlist is downloaded in a builder stage and never reaches the
|
||||
# final image — only the ~2 MB database derived from it does. That
|
||||
# The upstream dump is downloaded in a builder stage and never reaches the
|
||||
# final image — only the few-MB database derived from it does. That
|
||||
# derived database is CC BY-SA 4.0 while the code is Apache-2.0, so it is
|
||||
# copied in as its own layer alongside its licence and attribution rather than
|
||||
# being embedded in the binary.
|
||||
@@ -30,15 +30,15 @@ RUN CGO_ENABLED=0 go build -trimpath -o /out/build-dictionary ./cmd/build-dictio
|
||||
# --- the dictionary ---------------------------------------------------------
|
||||
FROM alpine:3.22 AS dict
|
||||
|
||||
# Fetched fresh, not pinned: kaikki.org re-exports Wiktionary about weekly and
|
||||
# keeps no dated snapshots. The derived wordlist is the one thing in this image
|
||||
# Fetched fresh, not pinned: Wikimedia regenerates the dump monthly and
|
||||
# repoints `latest/`. The derived dictionary is the one thing in this image
|
||||
# that cannot be rebuilt from the repository alone, so the builder records the
|
||||
# SHA-256 of the file it read in the database's meta table. The Makefile uses
|
||||
# the same URL for local builds, and a test asserts the two agree.
|
||||
ARG DICT_URL=https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl
|
||||
ARG DICT_URL=https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2
|
||||
|
||||
# Set to 1 to build from the checked-in word sample instead of downloading the
|
||||
# upstream wordlist. That produces a playable but tiny dictionary, and exists so
|
||||
# upstream dump. That produces a playable but tiny dictionary, and exists so
|
||||
# the image itself can be smoke-tested without network access.
|
||||
ARG FIXTURE_DICT=0
|
||||
|
||||
@@ -52,8 +52,8 @@ RUN set -eu; \
|
||||
if [ "$FIXTURE_DICT" = "1" ]; then \
|
||||
build-dictionary --words ./fixture-words.txt --out /out/noitu.db --min-words 150; \
|
||||
else \
|
||||
curl -fsSL -o kaikki-viwiktionary-vi.jsonl "$DICT_URL"; \
|
||||
build-dictionary --kaikki ./kaikki-viwiktionary-vi.jsonl --out /out/noitu.db; \
|
||||
curl -fsSLR -o viwiktionary-latest-pages-articles.xml.bz2 "$DICT_URL"; \
|
||||
build-dictionary --dump ./viwiktionary-latest-pages-articles.xml.bz2 --out /out/noitu.db; \
|
||||
fi
|
||||
|
||||
# --- the image --------------------------------------------------------------
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
# Every target has a raw equivalent documented in README.md, so contributors
|
||||
# without `make` (notably on Windows) are never blocked.
|
||||
|
||||
# kaikki.org re-exports Wiktionary tiếng Việt about weekly and keeps no dated
|
||||
# snapshots, so this is fetched fresh and unpinned by design: the builder
|
||||
# records the SHA-256 of what it read in the database's meta table. The URL
|
||||
# stays percent-encoded — the path has a space in it.
|
||||
DICT_URL := https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl
|
||||
DICT_SRC := data/kaikki-viwiktionary-vi.jsonl
|
||||
# Wikimedia regenerates the Wiktionary tiếng Việt dump monthly and repoints
|
||||
# `latest/` at it; this tracks `latest/`, fetched fresh and unpinned by design,
|
||||
# and the builder records the SHA-256 of what it read in the database's meta
|
||||
# table. Dated directories exist should a build ever need reproducing.
|
||||
DICT_URL := https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2
|
||||
DICT_SRC := data/viwiktionary-latest-pages-articles.xml.bz2
|
||||
DICT_OUT := data/noitu.db
|
||||
FIXTURE_WORDS := testdata/fixture-words.txt
|
||||
FIXTURE_DB := data/fixture.db
|
||||
@@ -17,7 +17,7 @@ SERVER_BIN := noitu-server
|
||||
.PHONY: help fetch-dict dict fixture-dict proto proto-check server web web-dev test test-go test-web test-e2e run clean
|
||||
|
||||
help:
|
||||
@echo "fetch-dict download the current upstream wordlist (~62 MB) into data/"
|
||||
@echo "fetch-dict download the current Wiktionary tiếng Việt dump (~61 MB) into data/"
|
||||
@echo "dict derive $(DICT_OUT) from $(DICT_SRC)"
|
||||
@echo "fixture-dict build the small test dictionary — no download needed"
|
||||
@echo "proto regenerate the Go and JS wire types from proto/"
|
||||
@@ -30,14 +30,16 @@ help:
|
||||
@echo "run build and run the server locally"
|
||||
@echo "clean remove build artifacts (keeps downloaded dictionary)"
|
||||
|
||||
# Fetches whatever kaikki currently serves. -f so an HTTP error fails here
|
||||
# rather than as a JSON parse error later; no resume flag, because resuming a
|
||||
# file that may have changed underneath would splice two exports together;
|
||||
# downloaded to a .part name and renamed only on success, so an interrupted
|
||||
# fetch never leaves a truncated file for the next `make dict` to consume.
|
||||
# Fetches whatever `latest/` currently points at. -f so an HTTP error fails
|
||||
# here rather than as a bzip2 error later; -R keeps the server's modification
|
||||
# time, which is when the dump was generated and becomes source_fetched_at; no
|
||||
# resume flag, because `latest` can be repointed between two attempts and a
|
||||
# resumed file would splice two months together; downloaded to a .part name
|
||||
# and renamed only on success, so an interrupted fetch never leaves a truncated
|
||||
# file for the next `make dict` to consume.
|
||||
fetch-dict:
|
||||
@mkdir -p data
|
||||
curl -fL -o $(DICT_SRC).part $(DICT_URL) && mv $(DICT_SRC).part $(DICT_SRC)
|
||||
curl -fLR -o $(DICT_SRC).part $(DICT_URL) && mv $(DICT_SRC).part $(DICT_SRC)
|
||||
@echo "downloaded $(DICT_SRC)"
|
||||
|
||||
$(DICT_SRC):
|
||||
@@ -45,7 +47,7 @@ $(DICT_SRC):
|
||||
@exit 1
|
||||
|
||||
dict: $(DICT_SRC)
|
||||
cd server && go run ./cmd/build-dictionary --kaikki ../$(DICT_SRC) --out ../$(DICT_OUT)
|
||||
cd server && go run ./cmd/build-dictionary --dump ../$(DICT_SRC) --out ../$(DICT_OUT)
|
||||
|
||||
# The dictionary tests, end-to-end runs and CI all play against. Built from a
|
||||
# checked-in word list through the same pipeline as the real one, so nothing
|
||||
|
||||
@@ -17,16 +17,17 @@ This covers everything under server/, web/, and proto/.
|
||||
------------------------------------------------------------------------------
|
||||
|
||||
The Vietnamese dictionary database is NOT covered by the Apache License. It is
|
||||
derived from Wiktionary text and remains under the licence that text carries,
|
||||
CC BY-SA 4.0:
|
||||
derived from Wiktionary text — word forms and edited excerpts of their
|
||||
definitions — and remains under the licence that text carries, CC BY-SA 4.0:
|
||||
|
||||
Affected artifacts: data/noitu.db (and data/kaikki-viwiktionary-vi.jsonl, its source)
|
||||
Affected artifacts: data/noitu.db (and data/viwiktionary-latest-pages-articles.xml.bz2,
|
||||
the Wikimedia dump it is built from)
|
||||
License text: data/LICENSE
|
||||
Attribution and
|
||||
list of changes: data/ATTRIBUTION.md
|
||||
Original work: Wiktionary tiếng Việt (https://vi.wiktionary.org/), by its contributors
|
||||
Extraction: wiktextract, published at https://kaikki.org/viwiktionary/
|
||||
(Tatu Ylonen); fetched fresh for each build, identified
|
||||
Source: https://dumps.wikimedia.org/viwiktionary/, the monthly
|
||||
pages-articles dump; fetched fresh for each build, identified
|
||||
by the SHA-256 recorded in the database's meta table
|
||||
|
||||
CC BY-SA 4.0 is a share-alike license. Any distribution of the derived database
|
||||
@@ -37,5 +38,5 @@ The two regimes are kept on separate artifacts deliberately: the database is
|
||||
loaded at runtime from a file and is never embedded, compiled, or linked into
|
||||
the Go binary.
|
||||
|
||||
Neither the source wordlist nor the derived database is committed to version
|
||||
Neither the source dump nor the derived database is committed to version
|
||||
control. Both are build artifacts produced by `make fetch-dict` and `make dict`.
|
||||
@@ -10,6 +10,9 @@ of the previous word**. No word may be reused. Fail to answer in time and you lo
|
||||
ngôn ngữ → ngữ pháp → pháp luật → luật lệ → ...
|
||||
```
|
||||
|
||||
Each word in the chain shows its Wiktionary meaning: the newest word's is open, and a click
|
||||
on any word opens or closes its own.
|
||||
|
||||
Playing a word that leaves the next player nothing to answer is not itself a win. They keep
|
||||
the turn and lose it to the clock like any other, and are then shown a few words the
|
||||
position still had — or told it had none.
|
||||
@@ -158,17 +161,16 @@ is needed only to change the WebSocket schema — the generated code is committe
|
||||
building and running the project does not require it.
|
||||
|
||||
```sh
|
||||
make fetch-dict # downloads the current ~62 MB Wiktionary export into data/
|
||||
make dict # derives data/noitu.db (the game's wordlist) from it
|
||||
make fetch-dict # downloads the current ~61 MB Wiktionary tiếng Việt dump into data/
|
||||
make dict # derives data/noitu.db (the game's words and their meanings) from it
|
||||
make test # run all tests
|
||||
make run # build and start the server
|
||||
```
|
||||
|
||||
The export is fetched fresh, not pinned: kaikki.org re-exports Wiktionary about weekly and
|
||||
keeps no dated snapshots, so two builds a week apart can differ slightly. The database
|
||||
records the SHA-256 of the file it was built from in its `meta` table. Neither the export
|
||||
nor the derived database is committed; both are build artifacts. See
|
||||
[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
|
||||
The dump is fetched fresh, not pinned: Wikimedia regenerates it monthly and repoints
|
||||
`latest/`, so two builds a month apart can differ. The database records the SHA-256 of the
|
||||
file it was built from in its `meta` table. Neither the dump nor the derived database is
|
||||
committed; both are build artifacts. See [`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
|
||||
|
||||
## Running the server
|
||||
|
||||
@@ -239,11 +241,11 @@ dev-only URL to get wrong.
|
||||
`make` is not installed everywhere (notably Windows). Every target is a thin wrapper:
|
||||
|
||||
```sh
|
||||
# fetch-dict (the URL is in the Makefile; keep it percent-encoded, the path has a space)
|
||||
curl -fL -o data/kaikki-viwiktionary-vi.jsonl "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl"
|
||||
# fetch-dict (the URL is in the Makefile; -R keeps the dump's own modification time)
|
||||
curl -fLR -o data/viwiktionary-latest-pages-articles.xml.bz2 "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2"
|
||||
|
||||
# dict
|
||||
cd server && go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db
|
||||
cd server && go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
|
||||
|
||||
# test
|
||||
cd server && go vet ./... && go test ./... -race
|
||||
@@ -278,7 +280,7 @@ buf generate && buf lint
|
||||
|
||||
The end-to-end suite plays against a small dictionary derived from
|
||||
[`testdata/fixture-words.txt`](./testdata/fixture-words.txt) through the same
|
||||
builder the real one uses, so CI never downloads the upstream wordlist.
|
||||
builder the real one uses, so CI never downloads the upstream dump.
|
||||
|
||||
## Deployment
|
||||
|
||||
@@ -302,11 +304,12 @@ See [`NOTICE`](./NOTICE) for the full statement.
|
||||
| Dictionary data (`data/noitu.db`) | [CC BY-SA 4.0](./data/LICENSE) |
|
||||
|
||||
The dictionary is derived from the [Wiktionary tiếng Việt](https://vi.wiktionary.org/)
|
||||
entries (CC BY-SA 4.0, by their contributors) as extracted by
|
||||
[wiktextract](https://github.com/tatuylonen/wiktextract) and published on
|
||||
[kaikki.org](https://kaikki.org/viwiktionary/). CC BY-SA is a **share-alike** license: any redistribution of the derived
|
||||
database — including inside a container image — must carry the same license, the attribution,
|
||||
and the record of modifications recorded in [`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
|
||||
entries (CC BY-SA 4.0, by their contributors), read from the Wikimedia Foundation's
|
||||
[monthly dump](https://dumps.wikimedia.org/viwiktionary/) of the wiki. It carries the word
|
||||
forms and edited excerpts of their definitions. CC BY-SA is a **share-alike** license: any
|
||||
redistribution of the derived database — including inside a container image — must carry the
|
||||
same license, the attribution, and the record of modifications recorded in
|
||||
[`data/ATTRIBUTION.md`](./data/ATTRIBUTION.md).
|
||||
|
||||
The database is loaded at runtime from a file and is never embedded or linked into the Go
|
||||
binary, keeping the two licensing regimes on separate artifacts.
|
||||
+45
-35
@@ -10,31 +10,30 @@ the attribution and the modifications required by that license.
|
||||
|---|---|
|
||||
| Original work | Entries of [Wiktionary tiếng Việt](https://vi.wiktionary.org/), written by its contributors |
|
||||
| Original license | [CC BY-SA 4.0](https://creativecommons.org/licenses/by-sa/4.0/) (Wiktionary text is dual-licensed CC BY-SA / GFDL) — full text in [`LICENSE`](./LICENSE) |
|
||||
| Extracted by | [wiktextract](https://github.com/tatuylonen/wiktextract), published on [kaikki.org](https://kaikki.org/viwiktionary/) by Tatu Ylonen; kaikki.org distributes the extracted data under the same CC BY-SA / GFDL terms as the underlying Wiktionary text |
|
||||
| Asset | [`Tiếng Việt/kaikki.org-dictionary-TiếngViệt.jsonl`](https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl) — the Vietnamese-language entries of the Vietnamese Wiktionary edition, ~62 MB, ~44,000 entries |
|
||||
| Refresh | kaikki re-extracts from the monthly Wikimedia dump about once a week |
|
||||
| Asset | [`viwiktionary-latest-pages-articles.xml.bz2`](https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2) — the Wikimedia Foundation's dump of every page of the Vietnamese Wiktionary edition with its current wikitext, ~61 MB compressed, ~43,000 pages with a Vietnamese section |
|
||||
| Refresh | regenerated monthly by Wikimedia; `latest/` is repointed at each new run |
|
||||
|
||||
**The asset is not pinned.** kaikki.org keeps no dated snapshots, so each build fetches the
|
||||
current export. The exact bytes a given `data/noitu.db` was built from are recorded in its
|
||||
`meta` table: `source_sha256` (SHA-256 of the file as read), `source_rows` (entries read)
|
||||
and `source_fetched_at` (the file's modification time). Two builds a week apart may differ
|
||||
by a few hundred words; the hash says which words a given image shipped.
|
||||
**The asset is not pinned.** Each build fetches whatever `latest/` currently points at. The
|
||||
exact bytes a given `data/noitu.db` was built from are recorded in its `meta` table:
|
||||
`source_sha256` (SHA-256 of the file as read), `source_pages` (pages with a Vietnamese
|
||||
section, redirects excluded) and `source_fetched_at` (the dump's own modification time).
|
||||
Two builds a month apart may differ by a few hundred words; the hash says which words and
|
||||
definitions a given image shipped. Dated dumps under `dumps.wikimedia.org/viwiktionary/`
|
||||
exist should a build ever need reproducing.
|
||||
|
||||
The attribution chain has two links before this project — Wiktionary's contributors, who
|
||||
wrote the entries, and wiktextract/kaikki.org, which turned the wiki markup into structured
|
||||
data — and both are named here because CC BY-SA attribution belongs to the authors, not only
|
||||
to the last host. kaikki.org asks users of its data to cite:
|
||||
*Tatu Ylonen: Wiktextract: Wiktionary as Machine-Readable Structured Data, Proceedings of
|
||||
the 13th Conference on Language Resources and Evaluation (LREC), pp. 1317–1325, Marseille,
|
||||
20–25 June 2022.*
|
||||
The attribution chain has one link before this project: Wiktionary tiếng Việt's
|
||||
contributors, who wrote the entries. The dump is their text as the wiki stores it; this
|
||||
project's builder reads the wikitext itself.
|
||||
|
||||
## Modifications made by this project
|
||||
|
||||
`server/cmd/build-dictionary` transforms the upstream export into `data/noitu.db`. The
|
||||
derived database is a **modified version** of the source data. Changes:
|
||||
`server/cmd/build-dictionary` transforms the dump into `data/noitu.db`. The derived database
|
||||
is a **modified version** of the source data. Changes:
|
||||
|
||||
1. **Language selection** — kept only entries with `lang_code = "vi"`. The file is
|
||||
Vietnamese-only today; any other language would be rejected and counted.
|
||||
1. **Section selection** — read only the Vietnamese section of each page, in either of the
|
||||
two markup dialects the wiki currently uses (`{{-vie-}}` or `== {{langname|vi}} ==`).
|
||||
Pages outside the main namespace, redirects, and pages with no Vietnamese section were
|
||||
skipped. Other languages' sections on the same page were not read.
|
||||
2. **Length filter** — kept only words of **2 or more space-separated syllables**, as
|
||||
required by the nối từ game rules. Single-syllable entries were dropped.
|
||||
3. **Content filter** — dropped entries containing digits or punctuation, entries using
|
||||
@@ -43,29 +42,40 @@ derived database is a **modified version** of the source data. Changes:
|
||||
Vietnamese words ("con cua") are kept.
|
||||
4. **Normalization** — all words Unicode NFC-normalized, lowercased, and
|
||||
whitespace-collapsed. Capitalized headwords (`Hà Nội`) become lowercase entries; nothing
|
||||
is removed on the basis of capitalization or part of speech.
|
||||
is removed on the basis of capitalization or part of speech. Two pages whose titles
|
||||
normalize to one word are merged into one entry.
|
||||
5. **Spelling aliases** — added an `aliases` table mapping alternative Vietnamese spellings
|
||||
to canonical entries. Two kinds: competing tone placement in open oa/oe/uy syllables
|
||||
(`hoà` → `hòa`, `thuý` → `thúy`), and i/y alternation in Sino-Vietnamese syllables
|
||||
(`quí` → `quý`, `lí` → `lý`). The majority are the i/y kind. These aliases are generated
|
||||
by this project and are not present upstream.
|
||||
6. **Added columns and tables** — `first` and `last` syllable columns, a `syllables` count,
|
||||
an index on `first`, a `syllables` out-degree table, and a `meta` table recording
|
||||
provenance (source URL, SHA-256, row count, fetch time, licence). All added for game
|
||||
lookups.
|
||||
7. **Deduplication** — an entry appears once per part of speech upstream; entries were
|
||||
deduplicated by normalized word form. A generated spelling variant that is itself a real
|
||||
word, or that more than one word would claim, is discarded rather than recorded as an
|
||||
alias.
|
||||
8. **Dropped fields** — all senses, glosses, examples, translations, pronunciations,
|
||||
etymologies, categories and part-of-speech tags were discarded. The derived database
|
||||
contains **only word forms**, not meanings.
|
||||
6. **Definition text** — each entry carries an **excerpt and modification** of its
|
||||
definitions, in a `meanings` table: the text of each `#` definition line of the
|
||||
Vietnamese section, with wiki markup removed (links reduced to their display text,
|
||||
formatting and references dropped, a few context and link templates unwrapped, every
|
||||
other template removed whole), cut to at most **five senses of 200 characters** each,
|
||||
and labelled with the Vietnamese name of the part-of-speech heading it sat under
|
||||
(`danh từ`, `động từ`, …; empty when the heading was not one the builder knows). This
|
||||
is not the entry as written: senses past the fifth, text past 200 characters, and
|
||||
template-only definitions the builder does not understand are gone.
|
||||
7. **Added columns and tables** — `first` and `last` syllable columns, a `syllables`
|
||||
count, an index on `first`, a `syllables` out-degree table, the `meanings` table above,
|
||||
and a `meta` table recording provenance (source URL, SHA-256, page count, fetch time,
|
||||
licence). All added for game lookups.
|
||||
8. **Deduplication** — a generated spelling variant that is itself a real word, or that
|
||||
more than one word would claim, is discarded rather than recorded as an alias.
|
||||
9. **Dropped fields** — everything else in an entry was discarded: example sentences,
|
||||
quotations, translations, pronunciations, etymologies, synonyms, derived terms,
|
||||
categories, images and references. The derived database carries word forms and the
|
||||
definition excerpts described in item 6, and nothing else of the entry.
|
||||
|
||||
## Share-alike obligation
|
||||
|
||||
CC BY-SA 4.0 is a **share-alike** license. The derived database `data/noitu.db`, and any
|
||||
distribution of it, remains licensed under **CC BY-SA 4.0** — including when it is shipped
|
||||
inside a container image or any other packaged build of this project.
|
||||
inside a container image or any other packaged build of this project. Because the database
|
||||
now redistributes edited excerpts of the entries' text and not only their headwords, the
|
||||
attribution and this record of modifications travel with it wherever it goes.
|
||||
|
||||
This obligation applies to the **data only**. The source code of this project is licensed
|
||||
separately under Apache-2.0 (see the repository root `LICENSE` and `NOTICE`). The derived
|
||||
@@ -75,9 +85,9 @@ keeping the two licensing regimes on separate artifacts.
|
||||
## How to reproduce the derived data
|
||||
|
||||
```sh
|
||||
make fetch-dict # downloads kaikki's current export (~62 MB) into data/
|
||||
make fetch-dict # downloads the current Wiktionary tiếng Việt dump (~61 MB) into data/
|
||||
make dict # derives data/noitu.db from it and records the file's SHA-256 in meta
|
||||
```
|
||||
|
||||
Neither file is committed to version control; both are build artifacts. Because the export
|
||||
is refreshed upstream, a rebuild on a later day may not be byte-identical to an earlier one.
|
||||
Neither file is committed to version control; both are build artifacts. Because `latest/`
|
||||
is repointed monthly, a rebuild in a later month may not be byte-identical to an earlier one.
|
||||
+8
-7
@@ -140,12 +140,13 @@ persistence, by design, in this version.
|
||||
|
||||
## Updating the dictionary
|
||||
|
||||
The wordlist is a build artifact, not runtime state, and the upstream export is
|
||||
fetched fresh rather than pinned: rebuilding the image picks up whatever
|
||||
kaikki.org currently serves, and the database's `meta` table records the
|
||||
SHA-256 of the file it was built from. To update the dictionary, rebuild and
|
||||
redeploy. To change the source itself, update `DICT_URL` in the `Dockerfile`,
|
||||
the `Makefile` and the builder's constant (a test asserts the three agree),
|
||||
then record what changed in `data/ATTRIBUTION.md`.
|
||||
The dictionary is a build artifact, not runtime state, and the upstream dump
|
||||
is fetched fresh rather than pinned: rebuilding the image picks up whatever
|
||||
`dumps.wikimedia.org` currently serves under `viwiktionary/latest/`, which is
|
||||
regenerated monthly, and the database's `meta` table records the SHA-256 of
|
||||
the file it was built from. To update the dictionary, rebuild and redeploy.
|
||||
To change the source itself, update `DICT_URL` in the `Dockerfile`, the
|
||||
`Makefile` and the builder's constant (a test asserts the three agree), then
|
||||
record what changed in `data/ATTRIBUTION.md`.
|
||||
|
||||
Nothing migrates, because nothing persists.
|
||||
@@ -0,0 +1,270 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"compress/bzip2"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/xml"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The upstream is the Wikimedia dump of Wiktionary tiếng Việt: every page's
|
||||
// current wikitext, as one bzip2-compressed XML file regenerated monthly.
|
||||
// `latest/` is a rolling pointer, fetched fresh for every build and not
|
||||
// pinned, so the builder records the SHA-256 of the bytes it actually read and
|
||||
// that hash is what identifies a build. Dated directories exist should
|
||||
// reproducibility ever be wanted.
|
||||
const dumpSourceURL = "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2"
|
||||
|
||||
// dumpPage is the part of a <page> element the builder reads. Everything
|
||||
// else — contributor, timestamp, sha1 — is skipped by the decoder.
|
||||
type dumpPage struct {
|
||||
Title string `xml:"title"`
|
||||
Ns int `xml:"ns"`
|
||||
Redirect *struct {
|
||||
Title string `xml:"title,attr"`
|
||||
} `xml:"redirect"`
|
||||
Revisions []struct {
|
||||
Text string `xml:"text"`
|
||||
} `xml:"revision"`
|
||||
}
|
||||
|
||||
// dumpProvenance identifies the bytes a build was made from.
|
||||
type dumpProvenance struct {
|
||||
sha256 string
|
||||
// pages is the number of pages with a Vietnamese section, redirects
|
||||
// excluded: the count of entries the corpus was derived from.
|
||||
pages int
|
||||
fetchedAt time.Time
|
||||
}
|
||||
|
||||
// dumpStats is what the build log reports about the dump beyond the reject
|
||||
// tally: enough to see a month where a dialect vanished or a stripper rule
|
||||
// started dropping everything.
|
||||
type dumpStats struct {
|
||||
pages int
|
||||
ns0 int
|
||||
redirects int
|
||||
noVietnamese int
|
||||
legacy int
|
||||
newDialect int
|
||||
bothDialects int
|
||||
merged int // pages whose title normalized to a word already seen
|
||||
// enders counts the {{-code-}} that closed each legacy section. Language
|
||||
// codes are expected here; a heading code is one the maps are missing.
|
||||
enders map[string]int
|
||||
section *sectionStats
|
||||
}
|
||||
|
||||
// readDump streams the dump once: hashes the compressed bytes, decodes one
|
||||
// page at a time, hands every Vietnamese-section title to accept() and every
|
||||
// definition line to the stripper.
|
||||
func readDump(path string) (map[string]entry, map[string][]sense, map[rejectReason]int, *dumpStats, dumpProvenance, error) {
|
||||
var prov dumpProvenance
|
||||
stats := &dumpStats{section: newSectionStats(), enders: make(map[string]int)}
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("read dump: %w", err)
|
||||
}
|
||||
defer f.Close()
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
// A provenance row must be right or absent, never a plausible zero.
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("stat dump: %w", err)
|
||||
}
|
||||
prov.fetchedAt = info.ModTime().UTC()
|
||||
|
||||
hash := sha256.New()
|
||||
compressed := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
|
||||
if magic, err := compressed.Peek(3); err != nil || string(magic) != "BZh" {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s is not a bzip2 file (expected a BZh header)", path)
|
||||
}
|
||||
dec := xml.NewDecoder(bzip2.NewReader(compressed))
|
||||
|
||||
words := make(map[string]entry)
|
||||
meanings := make(map[string][]sense)
|
||||
rejects := make(map[rejectReason]int)
|
||||
lastTitle := ""
|
||||
|
||||
for {
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) {
|
||||
break
|
||||
}
|
||||
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
|
||||
}
|
||||
start, ok := tok.(xml.StartElement)
|
||||
if !ok || start.Name.Local != "page" {
|
||||
continue
|
||||
}
|
||||
var page dumpPage
|
||||
if err := dec.DecodeElement(&page, &start); err != nil {
|
||||
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
|
||||
}
|
||||
lastTitle = page.Title
|
||||
stats.pages++
|
||||
if page.Ns != 0 {
|
||||
continue
|
||||
}
|
||||
stats.ns0++
|
||||
if page.Redirect != nil {
|
||||
// The target page is read on its own and lowercased by accept(),
|
||||
// so a case-only redirect adds nothing and any other redirect is
|
||||
// an alternative title the wiki itself does not define.
|
||||
stats.redirects++
|
||||
continue
|
||||
}
|
||||
if len(page.Revisions) == 0 {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s: page %q has no revision text", path, page.Title)
|
||||
}
|
||||
text := page.Revisions[len(page.Revisions)-1].Text
|
||||
|
||||
section, dialect, both, ender := vietnameseSection(text)
|
||||
if dialect == "" {
|
||||
stats.noVietnamese++
|
||||
rejects[rejectNotVietnamese]++
|
||||
continue
|
||||
}
|
||||
if both {
|
||||
stats.bothDialects++
|
||||
}
|
||||
if dialect == "legacy" {
|
||||
stats.legacy++
|
||||
} else {
|
||||
stats.newDialect++
|
||||
}
|
||||
prov.pages++
|
||||
if ender != "" {
|
||||
stats.enders[ender]++
|
||||
}
|
||||
|
||||
word, syllables, reason, ok := accept(page.Title)
|
||||
if !ok {
|
||||
rejects[reason]++
|
||||
continue
|
||||
}
|
||||
// After accept, so the definition counters describe words that land.
|
||||
senses := definitions(section, stats.section)
|
||||
if _, seen := words[word]; seen {
|
||||
// Two pages whose titles normalize to one word (Việt Nam and
|
||||
// việt nam): one entry, senses in page order, one cap.
|
||||
stats.merged++
|
||||
}
|
||||
words[word] = entry{
|
||||
word: word,
|
||||
first: syllables[0],
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
if len(senses) > 0 {
|
||||
merged := append(meanings[word], senses...)
|
||||
if len(merged) > maxSenses {
|
||||
merged = merged[:maxSenses]
|
||||
}
|
||||
meanings[word] = merged
|
||||
}
|
||||
}
|
||||
|
||||
// The XML decoder stops at the root's close tag; the hash must cover the
|
||||
// whole file, trailing bytes included.
|
||||
if _, err := io.Copy(io.Discard, compressed); err != nil {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
|
||||
|
||||
return words, meanings, rejects, stats, prov, nil
|
||||
}
|
||||
|
||||
// dumpError names where a stream failed: the last page fully read and the
|
||||
// decompressed offset, so a truncated download and a malformed page are told
|
||||
// apart by the message alone.
|
||||
func dumpError(path, lastTitle string, offset int64, err error) error {
|
||||
where := "before the first page"
|
||||
if lastTitle != "" {
|
||||
where = fmt.Sprintf("after page %q", lastTitle)
|
||||
}
|
||||
if errors.Is(err, io.ErrUnexpectedEOF) || strings.Contains(err.Error(), "unexpected EOF") {
|
||||
return fmt.Errorf("%s: stream ends %s (decompressed offset %d): truncated download? %w", path, where, offset, err)
|
||||
}
|
||||
return fmt.Errorf("%s: %s (decompressed offset %d): %w", path, where, offset, err)
|
||||
}
|
||||
|
||||
// logDumpStats writes the build log lines that describe what the dump held.
|
||||
func logDumpStats(stats *dumpStats) {
|
||||
s := stats.section
|
||||
logf := log.Printf
|
||||
logf("pages %d, in the main namespace %d, redirects skipped %d, without a Vietnamese section %d",
|
||||
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
|
||||
logf("Vietnamese sections: legacy {{-vie-}} %d, new == {{langname|vi}} == %d, pages with both %d, titles merged %d",
|
||||
stats.legacy, stats.newDialect, stats.bothDialects, stats.merged)
|
||||
logf("parts of speech: %s", formatTally(s.pos, 0))
|
||||
if len(s.unmappedPos) > 0 {
|
||||
logf("headings without a label: %s", formatTally(s.unmappedPos, 20))
|
||||
}
|
||||
// Language codes belong here. A heading code in this list is one the maps
|
||||
// do not know, and it has been cutting sections short.
|
||||
logf("codes that ended a legacy section, commonest: %s", formatTally(stats.enders, 15))
|
||||
logf("definitions kept %d (cut at %d characters: %d), dropped as empty after stripping %d",
|
||||
s.defsKept, maxGlossRunes, s.defsCut, s.defsEmpty)
|
||||
if len(s.dropped) > 0 {
|
||||
logf("templates dropped whole, commonest: %s", formatTally(s.dropped, 10))
|
||||
}
|
||||
}
|
||||
|
||||
// formatTally renders counts on one log line, largest first, cut to the top
|
||||
// n entries when n is positive.
|
||||
func formatTally(counts map[string]int, n int) string {
|
||||
type kv struct {
|
||||
name string
|
||||
count int
|
||||
}
|
||||
tally := make([]kv, 0, len(counts))
|
||||
for name, count := range counts {
|
||||
if name == "" {
|
||||
name = "(none)"
|
||||
}
|
||||
tally = append(tally, kv{name, count})
|
||||
}
|
||||
sort.Slice(tally, func(i, j int) bool {
|
||||
if tally[i].count != tally[j].count {
|
||||
return tally[i].count > tally[j].count
|
||||
}
|
||||
return tally[i].name < tally[j].name
|
||||
})
|
||||
if n > 0 && len(tally) > n {
|
||||
tally = tally[:n]
|
||||
}
|
||||
parts := make([]string, len(tally))
|
||||
for i, t := range tally {
|
||||
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// dumpSourceSpec describes a dump build for the meta table. With nothing
|
||||
// pinned upstream, the hash and page count of the bytes read are the
|
||||
// provenance.
|
||||
func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec {
|
||||
return sourceSpec{
|
||||
table: "dump:" + filepath.Base(path),
|
||||
url: dumpSourceURL,
|
||||
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
|
||||
extra: [][2]string{
|
||||
{"source_sha256", prov.sha256},
|
||||
{"source_pages", fmt.Sprint(prov.pages)},
|
||||
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// miniDump is a twelve-page stand-in for the Wikimedia dump, committed
|
||||
// compressed beside its readable source. Go has no bzip2 writer, so the .bz2
|
||||
// is regenerated by hand: bzip2 -k9 testdata/mini-dump.xml.
|
||||
const miniDump = "testdata/mini-dump.xml.bz2"
|
||||
|
||||
// miniDumpCut is the same XML cut mid-page and then compressed: a valid bzip2
|
||||
// stream whose XML ends early, as distinct from a truncated download.
|
||||
const miniDumpCut = "testdata/mini-dump-cut.xml.bz2"
|
||||
|
||||
func TestReadDumpKeepsVietnameseSections(t *testing.T) {
|
||||
words, meanings, rejects, stats, prov, err := readDump(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var got []string
|
||||
for w := range words {
|
||||
got = append(got, w)
|
||||
}
|
||||
assertSameStrings(t, got, []string{"pháp luật", "hòa bình", "luật lệ", "ngôn ngữ", "ngữ pháp", "vô tuyến điện"})
|
||||
|
||||
if stats.pages != 12 || stats.ns0 != 11 || stats.redirects != 1 || stats.noVietnamese != 1 {
|
||||
t.Errorf("pages %d ns0 %d redirects %d noVietnamese %d, want 12 11 1 1",
|
||||
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
|
||||
}
|
||||
if stats.legacy != 7 || stats.newDialect != 2 || stats.bothDialects != 0 {
|
||||
t.Errorf("legacy %d new %d both %d, want 7 2 0", stats.legacy, stats.newDialect, stats.bothDialects)
|
||||
}
|
||||
if prov.pages != 9 {
|
||||
t.Errorf("source pages = %d, want 9 (Vietnamese sections, redirect excluded)", prov.pages)
|
||||
}
|
||||
if stats.merged != 1 {
|
||||
t.Errorf("merged = %d, want 1 (Hòa Bình and hòa bình)", stats.merged)
|
||||
}
|
||||
if rejects[rejectNotVietnamese] != 1 || rejects[rejectTooShort] != 1 || rejects[rejectDigit] != 1 {
|
||||
t.Errorf("rejects = %v, want one each of not-Vietnamese, too-short, digit", rejects)
|
||||
}
|
||||
|
||||
assertSenses(t, meanings["pháp luật"], []sense{
|
||||
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
|
||||
{"động từ", "(hiếm) Xử theo luật."},
|
||||
})
|
||||
// Capitalized page first, lowercase page second: senses in page order,
|
||||
// the new-dialect place first.
|
||||
assertSenses(t, meanings["hòa bình"], []sense{
|
||||
{"danh từ riêng", "tỉnh, Việt Nam."},
|
||||
{"danh từ", "Tình trạng không có chiến tranh."},
|
||||
{"tính từ", "Yên ổn."},
|
||||
})
|
||||
assertSenses(t, meanings["ngôn ngữ"], []sense{{"danh từ", "Hệ thống những âm, từ và quy tắc kết hợp chúng."}})
|
||||
if _, has := meanings["luật lệ"]; has {
|
||||
t.Error("a definition that is only an unknown template produced a sense")
|
||||
}
|
||||
if stats.section.dropped["rfdef"] != 1 || stats.section.defsEmpty != 1 {
|
||||
t.Errorf("dropped = %v empty = %d, want rfdef 1 and 1", stats.section.dropped, stats.section.defsEmpty)
|
||||
}
|
||||
if stats.section.pos["noun"] == 0 || stats.section.pos["pr-noun"] != 2 || stats.section.pos["n"] != 1 {
|
||||
t.Errorf("pos tally = %v", stats.section.pos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpHashesTheBytesItRead(t *testing.T) {
|
||||
_, _, _, _, prov, err := readDump(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sum := sha256.Sum256(raw)
|
||||
if prov.sha256 != hex.EncodeToString(sum[:]) {
|
||||
t.Errorf("sha256 = %s, want %s (the whole file)", prov.sha256, hex.EncodeToString(sum[:]))
|
||||
}
|
||||
if prov.fetchedAt.IsZero() {
|
||||
t.Error("fetchedAt is zero")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsNonBzip2(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
|
||||
if err := os.WriteFile(path, []byte("<mediawiki></mediawiki>"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, _, _, _, _, err := readDump(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "not a bzip2 file") {
|
||||
t.Fatalf("err = %v, want a message naming the missing bzip2 header", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsTruncatedDownload(t *testing.T) {
|
||||
raw, err := os.ReadFile(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
|
||||
if err := os.WriteFile(path, raw[:len(raw)/2], 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, _, _, _, _, err = readDump(path)
|
||||
if err == nil {
|
||||
t.Fatal("a half-downloaded dump was read without error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "truncated") {
|
||||
t.Errorf("err = %v, want it to suggest a truncated download", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsStreamEndingMidPage(t *testing.T) {
|
||||
_, _, _, _, _, err := readDump(miniDumpCut)
|
||||
if err == nil {
|
||||
t.Fatal("an XML stream ending mid-page was read without error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), `after page "hello world"`) {
|
||||
t.Errorf("err = %v, want it to name the last page fully read", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailsBelowMinPages(t *testing.T) {
|
||||
cfg := config{
|
||||
dump: miniDump,
|
||||
out: filepath.Join(t.TempDir(), "noitu.db"),
|
||||
minWords: 1,
|
||||
minPages: 20000,
|
||||
}
|
||||
err := run(cfg)
|
||||
if err == nil || !strings.Contains(err.Error(), "pages have a Vietnamese section") {
|
||||
t.Fatalf("err = %v, want the page floor named", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFormatTally(t *testing.T) {
|
||||
got := formatTally(map[string]int{"b": 2, "a": 2, "": 5, "c": 1}, 3)
|
||||
if want := "(none) 5, a 2, b 2"; got != want {
|
||||
t.Errorf("formatTally = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -1,168 +0,0 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The upstream is kaikki.org's wiktextract export of Wiktionary tiếng Việt:
|
||||
// one JSON object per entry, refreshed from the monthly Wikimedia dump about
|
||||
// once a week. The file is fetched fresh for every build and is not pinned —
|
||||
// there is no archived snapshot to pin to — so the builder records the SHA-256
|
||||
// of the bytes it actually read, and that hash is what identifies a build.
|
||||
//
|
||||
// The URL stays percent-encoded: the path has a space in it, and both make
|
||||
// and sh would otherwise split it.
|
||||
const kaikkiSourceURL = "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl"
|
||||
|
||||
// kaikkiRow is the part of a wiktextract entry the game cares about. Every
|
||||
// other field — senses, translations, categories — is skipped by the decoder.
|
||||
type kaikkiRow struct {
|
||||
Word string `json:"word"`
|
||||
Pos string `json:"pos"`
|
||||
LangCode string `json:"lang_code"`
|
||||
}
|
||||
|
||||
// kaikkiProvenance identifies the bytes a build was made from.
|
||||
type kaikkiProvenance struct {
|
||||
sha256 string
|
||||
rows int
|
||||
fetchedAt time.Time
|
||||
}
|
||||
|
||||
// readKaikkiList streams the export, keeps Vietnamese-language entries and
|
||||
// hands their word forms to accept(). Part of speech is tallied for the build
|
||||
// log but never filters: the owner's decision that capitalization removes no
|
||||
// word applies equally to the "name" tag.
|
||||
//
|
||||
// Lines are read with bufio.Reader rather than bufio.Scanner because a row
|
||||
// carries every sense and translation of its entry and can run to hundreds of
|
||||
// kilobytes; a scanner's fixed cap would be a guess that eventually fails.
|
||||
func readKaikkiList(path string) (map[string]entry, map[rejectReason]int, map[string]int, kaikkiProvenance, error) {
|
||||
var prov kaikkiProvenance
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, nil, nil, prov, fmt.Errorf("read kaikki export: %w", err)
|
||||
}
|
||||
defer f.Close()
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
// A provenance row must be right or absent, never a plausible zero.
|
||||
return nil, nil, nil, prov, fmt.Errorf("stat kaikki export: %w", err)
|
||||
}
|
||||
prov.fetchedAt = info.ModTime().UTC()
|
||||
|
||||
hash := sha256.New()
|
||||
reader := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
|
||||
|
||||
words := make(map[string]entry)
|
||||
rejects := make(map[rejectReason]int)
|
||||
pos := make(map[string]int)
|
||||
|
||||
lineNo := 0
|
||||
for {
|
||||
line, err := reader.ReadBytes('\n')
|
||||
if len(line) > 0 {
|
||||
lineNo++
|
||||
if trimmed := bytes.TrimSpace(line); len(trimmed) > 0 {
|
||||
// A bare literal such as null would decode into an empty row
|
||||
// and be miscounted as a foreign-language entry; only objects
|
||||
// are entries.
|
||||
if trimmed[0] != '{' {
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: not a JSON object", path, lineNo)
|
||||
}
|
||||
var row kaikkiRow
|
||||
if err := json.Unmarshal(trimmed, &row); err != nil {
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: %w", path, lineNo, err)
|
||||
}
|
||||
prov.rows++
|
||||
if row.LangCode != "vi" {
|
||||
rejects[rejectNotVietnamese]++
|
||||
} else {
|
||||
pos[row.Pos]++
|
||||
if word, syllables, reason, ok := accept(row.Word); !ok {
|
||||
rejects[reason]++
|
||||
} else {
|
||||
words[word] = entry{
|
||||
word: word,
|
||||
first: syllables[0],
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) {
|
||||
break
|
||||
}
|
||||
// The failure is on the line being read: the one just counted if
|
||||
// a partial line came back with the error, otherwise the next.
|
||||
failed := lineNo + 1
|
||||
if len(line) > 0 {
|
||||
failed = lineNo
|
||||
}
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: %w", path, failed, err)
|
||||
}
|
||||
}
|
||||
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
|
||||
|
||||
return words, rejects, pos, prov, nil
|
||||
}
|
||||
|
||||
// formatPosTally renders the part-of-speech counts on one log line, largest
|
||||
// first, so the build log says what kind of entries the export held.
|
||||
func formatPosTally(pos map[string]int) string {
|
||||
type kv struct {
|
||||
name string
|
||||
count int
|
||||
}
|
||||
tally := make([]kv, 0, len(pos))
|
||||
for name, count := range pos {
|
||||
if name == "" {
|
||||
name = "(none)"
|
||||
}
|
||||
tally = append(tally, kv{name, count})
|
||||
}
|
||||
sort.Slice(tally, func(i, j int) bool {
|
||||
if tally[i].count != tally[j].count {
|
||||
return tally[i].count > tally[j].count
|
||||
}
|
||||
return tally[i].name < tally[j].name
|
||||
})
|
||||
parts := make([]string, len(tally))
|
||||
for i, t := range tally {
|
||||
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// kaikkiSourceSpec describes a kaikki build for the meta table. With no commit
|
||||
// or checksum pinned upstream, the hash and row count of the bytes read are the
|
||||
// provenance.
|
||||
func kaikkiSourceSpec(path string, prov kaikkiProvenance) sourceSpec {
|
||||
return sourceSpec{
|
||||
table: "kaikki:" + filepath.Base(path),
|
||||
url: kaikkiSourceURL,
|
||||
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
|
||||
extra: [][2]string{
|
||||
{"source_sha256", prov.sha256},
|
||||
{"source_rows", fmt.Sprint(prov.rows)},
|
||||
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -1,243 +0,0 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// fixtureKaikki writes a miniature stand-in for the kaikki export: the same
|
||||
// JSONL shape, a handful of rows.
|
||||
func fixtureKaikki(t *testing.T, lines ...string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
|
||||
if err := os.WriteFile(path, []byte(strings.Join(lines, "\n")+"\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func defaultKaikkiLines() []string {
|
||||
return []string{
|
||||
`{"word": "Hà Nội", "pos": "name", "lang_code": "vi", "senses": [{"glosses": ["thủ đô"]}]}`, // capitalized, name POS — kept, lowercased
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
|
||||
`{"word": "học sinh", "pos": "verb", "lang_code": "vi"}`, // same word, second POS — kept once
|
||||
`{"word": "student", "pos": "noun", "lang_code": "en"}`, // not Vietnamese-language — rejected and counted
|
||||
`{"word": "pháp", "pos": "noun", "lang_code": "vi"}`, // single syllable — rejected downstream
|
||||
``,
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListKeepsVietnameseEntries(t *testing.T) {
|
||||
words, rejects, pos, prov, err := readKaikkiList(fixtureKaikki(t, defaultKaikkiLines()...))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var got []string
|
||||
for w := range words {
|
||||
got = append(got, w)
|
||||
}
|
||||
assertSameStrings(t, got, []string{"hà nội", "học sinh"})
|
||||
|
||||
if n := rejects[rejectNotVietnamese]; n != 1 {
|
||||
t.Errorf("non-Vietnamese rejects = %d, want 1", n)
|
||||
}
|
||||
if n := rejects[rejectTooShort]; n != 1 {
|
||||
t.Errorf("too-short rejects = %d, want 1 (pháp)", n)
|
||||
}
|
||||
if pos["noun"] != 2 || pos["verb"] != 1 || pos["name"] != 1 {
|
||||
t.Errorf("pos tally = %v, want noun 2, verb 1, name 1 (en row excluded)", pos)
|
||||
}
|
||||
if prov.rows != 5 {
|
||||
t.Errorf("rows = %d, want 5", prov.rows)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListHashesTheBytesItRead(t *testing.T) {
|
||||
path := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
_, _, _, prov, err := readKaikkiList(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sum := sha256.Sum256(raw)
|
||||
if want := hex.EncodeToString(sum[:]); prov.sha256 != want {
|
||||
t.Errorf("sha256 = %s, want %s", prov.sha256, want)
|
||||
}
|
||||
if prov.fetchedAt.IsZero() {
|
||||
t.Error("fetchedAt is zero, want the file's modification time")
|
||||
}
|
||||
}
|
||||
|
||||
// fixtureKaikkiRaw writes exact bytes, for the shapes fixtureKaikki's trailing
|
||||
// newline would hide.
|
||||
func fixtureKaikkiRaw(t *testing.T, raw string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
|
||||
if err := os.WriteFile(path, []byte(raw), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestKaikkiListHandlesDownloadShapes(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
raw string
|
||||
wantWords int
|
||||
wantRows int
|
||||
wantErr string
|
||||
}{
|
||||
{"final line without newline",
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh mì", "pos": "noun", "lang_code": "vi"}`,
|
||||
2, 2, ""},
|
||||
{"HTTP error page instead of JSONL", "<html><body>503</body></html>\n", 0, 0, ":1: malformed"},
|
||||
{"cut mid-line", `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh`, 0, 0, ":2: malformed"},
|
||||
{"bare JSON literal", "null\n", 0, 0, ":1: malformed"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
words, _, _, prov, err := readKaikkiList(fixtureKaikkiRaw(t, tc.raw))
|
||||
if tc.wantErr != "" {
|
||||
if err == nil || !strings.Contains(err.Error(), tc.wantErr) {
|
||||
t.Fatalf("err = %v, want one containing %q", err, tc.wantErr)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(words) != tc.wantWords || prov.rows != tc.wantRows {
|
||||
t.Errorf("words=%d rows=%d, want %d/%d", len(words), prov.rows, tc.wantWords, tc.wantRows)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListNamesMalformedLine(t *testing.T) {
|
||||
path := fixtureKaikki(t,
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
|
||||
`{"word": "broken"`,
|
||||
)
|
||||
_, _, _, _, err := readKaikkiList(path)
|
||||
if err == nil {
|
||||
t.Fatal("malformed line was skipped, want error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), ":2:") {
|
||||
t.Errorf("error does not name line 2: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListReadsLongLines(t *testing.T) {
|
||||
// A real row carries every sense and translation and can exceed any
|
||||
// scanner buffer; the reader must not have a line cap.
|
||||
padding := strings.Repeat("x", 2<<20)
|
||||
path := fixtureKaikki(t, `{"word": "học sinh", "pos": "noun", "lang_code": "vi", "note": "`+padding+`"}`)
|
||||
words, _, _, _, err := readKaikkiList(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok := words["học sinh"]; !ok {
|
||||
t.Error("word on a 2 MB line was lost")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFormatPosTally(t *testing.T) {
|
||||
got := formatPosTally(map[string]int{"verb": 2, "noun": 5, "": 1})
|
||||
if want := "noun 5, verb 2, (none) 1"; got != want {
|
||||
t.Errorf("tally = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiBuildRecordsProvenance(t *testing.T) {
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
path := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
if err := run(config{kaikki: path, out: out, minWords: 1}); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
db := openOut(t, out)
|
||||
|
||||
raw, _ := os.ReadFile(path)
|
||||
sum := sha256.Sum256(raw)
|
||||
want := map[string]string{
|
||||
"source_url": kaikkiSourceURL,
|
||||
"source_sha256": hex.EncodeToString(sum[:]),
|
||||
"source_rows": "5",
|
||||
"source_license": "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
"word_count": "2",
|
||||
}
|
||||
for key, value := range want {
|
||||
var got string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&got); err != nil {
|
||||
t.Errorf("meta[%q] missing: %v", key, err)
|
||||
continue
|
||||
}
|
||||
if got != value {
|
||||
t.Errorf("meta[%q] = %q, want %q", key, got, value)
|
||||
}
|
||||
}
|
||||
for _, gone := range []string{"source_commit", "sources_kept", "sources_excluded"} {
|
||||
var got string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, gone).Scan(&got); err == nil {
|
||||
t.Errorf("meta[%q] = %q, want absent", gone, got)
|
||||
}
|
||||
}
|
||||
var fetched string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_fetched_at'`).Scan(&fetched); err != nil || fetched == "" {
|
||||
t.Errorf("meta[source_fetched_at] missing or empty: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunInputSelection(t *testing.T) {
|
||||
kaikki := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
words := fixtureKaikki(t, "học sinh")
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
cfg config
|
||||
wantErr string
|
||||
}{
|
||||
{"both inputs", config{kaikki: kaikki, words: words, out: out, minWords: 1}, "mutually exclusive"},
|
||||
{"neither input", config{out: out, minWords: 1}, "no input given"},
|
||||
{"missing kaikki file", config{kaikki: filepath.Join(t.TempDir(), "absent.jsonl"), out: out, minWords: 1}, "make fetch-dict"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
err := run(tc.cfg)
|
||||
if err == nil {
|
||||
t.Fatal("run succeeded, want error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), tc.wantErr) {
|
||||
t.Errorf("error %q does not mention %q", err, tc.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A fixture is hand-written data; its database must not claim the upstream's
|
||||
// licence, because the server logs whatever the meta table says.
|
||||
func TestWordListBuildRecordsNoUpstreamLicense(t *testing.T) {
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
if err := run(config{words: fixtureKaikki(t, "học sinh", "bánh mì"), out: out, minWords: 1}); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
var license, url string
|
||||
db := openOut(t, out)
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_url'`).Scan(&url); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Contains(license, "CC BY-SA") || url != "" {
|
||||
t.Errorf("fixture build claims upstream provenance: license=%q url=%q", license, url)
|
||||
}
|
||||
}
|
||||
@@ -1,19 +1,20 @@
|
||||
// Command build-dictionary derives the game's wordlist from kaikki.org's
|
||||
// wiktextract export of Wiktionary tiếng Việt.
|
||||
// Command build-dictionary derives the game's wordlist and word meanings from
|
||||
// the Wikimedia dump of Wiktionary tiếng Việt.
|
||||
//
|
||||
// The upstream is a ~62 MB JSONL file: one entry per line with its senses,
|
||||
// translations and part of speech. The game needs only Vietnamese word forms
|
||||
// of at least two syllables, indexed by first and last syllable. This tool
|
||||
// performs that reduction and records provenance in a meta table — including
|
||||
// the SHA-256 of the file it read, since the upstream is fetched fresh for
|
||||
// every build rather than pinned.
|
||||
// The upstream is a ~61 MB bzip2-compressed XML file: every page of the wiki
|
||||
// with its current wikitext, regenerated monthly. The game needs the
|
||||
// Vietnamese word forms of at least two syllables, indexed by first and last
|
||||
// syllable, and the plain text of each word's definitions. This tool performs
|
||||
// that reduction and records provenance in a meta table — including the
|
||||
// SHA-256 of the file it read, since the upstream is fetched fresh for every
|
||||
// build rather than pinned.
|
||||
//
|
||||
// The derived database is a modified version of CC BY-SA 4.0 licensed data.
|
||||
// See data/ATTRIBUTION.md.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
// go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db
|
||||
// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
|
||||
package main
|
||||
|
||||
import (
|
||||
@@ -27,32 +28,45 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// builderVer changes whenever the meta table's contract does, so two databases
|
||||
// with different provenance rows never claim the same builder.
|
||||
const builderVer = "4"
|
||||
const builderVer = "5"
|
||||
|
||||
// minMeaningCoverage is the share of words a dump build must carry a meaning
|
||||
// for. The 2026-09-01 dump measured well above it; the floor exists to catch a
|
||||
// stripper or section scanner that suddenly returns nothing, not to demand
|
||||
// quality. Fixture builds are exempt: their meanings are hand-written.
|
||||
const minMeaningCoverage = 0.6
|
||||
|
||||
type config struct {
|
||||
// kaikki is the corpus: the upstream wiktextract JSONL export.
|
||||
kaikki string
|
||||
// words is an alternative source: a plain list, one word per line, used to
|
||||
// build a small fixture database without the upstream download.
|
||||
// dump is the corpus: the Wikimedia pages-articles export.
|
||||
dump string
|
||||
// words is an alternative source: a plain list, one word per line with an
|
||||
// optional tab-separated meaning column, used to build a small fixture
|
||||
// database without the upstream download.
|
||||
words string
|
||||
out string
|
||||
minWords int
|
||||
// minPages is the floor on pages with a Vietnamese section. Distinct from
|
||||
// minWords so a scanner that silently misses a dialect is caught before
|
||||
// the word floor is.
|
||||
minPages int
|
||||
}
|
||||
|
||||
func main() {
|
||||
log.SetFlags(0)
|
||||
|
||||
var cfg config
|
||||
flag.StringVar(&cfg.kaikki, "kaikki", "", "upstream kaikki.org wiktextract JSONL export to read")
|
||||
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the upstream export (one word per line, # comments)")
|
||||
flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read")
|
||||
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)")
|
||||
flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write")
|
||||
flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering")
|
||||
flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section")
|
||||
flag.Parse()
|
||||
|
||||
if err := run(cfg); err != nil {
|
||||
@@ -64,40 +78,47 @@ func run(cfg config) error {
|
||||
// Exactly one input. Picking silently between two would let a stray flag
|
||||
// ship a corpus nobody meant to build.
|
||||
switch {
|
||||
case cfg.kaikki == "" && cfg.words == "":
|
||||
return errors.New("no input given: pass --kaikki (the corpus) or --words (a plain list)")
|
||||
case cfg.kaikki != "" && cfg.words != "":
|
||||
return errors.New("--kaikki and --words are mutually exclusive")
|
||||
case cfg.kaikki != "":
|
||||
return runFromKaikkiList(cfg)
|
||||
case cfg.dump == "" && cfg.words == "":
|
||||
return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)")
|
||||
case cfg.dump != "" && cfg.words != "":
|
||||
return errors.New("--dump and --words are mutually exclusive")
|
||||
case cfg.dump != "":
|
||||
return runFromDump(cfg)
|
||||
default:
|
||||
return runFromWordList(cfg)
|
||||
}
|
||||
}
|
||||
|
||||
// runFromKaikkiList derives the database from the kaikki.org export, keeping
|
||||
// Vietnamese-language entries and recording the hash of the bytes it read.
|
||||
func runFromKaikkiList(cfg config) error {
|
||||
if _, err := os.Stat(cfg.kaikki); err != nil {
|
||||
return fmt.Errorf("kaikki export not found at %s — run 'make fetch-dict' first: %w", cfg.kaikki, err)
|
||||
// runFromDump derives the database from the Wikimedia dump, keeping every
|
||||
// page with a Vietnamese section and recording the hash of the bytes it read.
|
||||
func runFromDump(cfg config) error {
|
||||
if _, err := os.Stat(cfg.dump); err != nil {
|
||||
return fmt.Errorf("dump not found at %s — run 'make fetch-dict' first: %w", cfg.dump, err)
|
||||
}
|
||||
|
||||
words, rejects, pos, prov, err := readKaikkiList(cfg.kaikki)
|
||||
started := time.Now()
|
||||
words, meanings, rejects, stats, prov, err := readDump(cfg.dump)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
logDumpStats(stats)
|
||||
logRejects(rejects)
|
||||
log.Printf("parts of speech: %s", formatPosTally(pos))
|
||||
log.Printf("accepted %d distinct words from %s (%d rows, sha256 %s)", len(words), cfg.kaikki, prov.rows, prov.sha256)
|
||||
if prov.pages < cfg.minPages {
|
||||
return fmt.Errorf("only %d pages have a Vietnamese section, expected at least %d — "+
|
||||
"the dump's markup may have changed", prov.pages, cfg.minPages)
|
||||
}
|
||||
log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s",
|
||||
len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second))
|
||||
|
||||
return finish(cfg, words, kaikkiSourceSpec(cfg.kaikki, prov))
|
||||
return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true)
|
||||
}
|
||||
|
||||
// finish is the tail every input mode shares: the size floor, alias
|
||||
// generation, the atomic write and the re-read verification. Keeping it in one
|
||||
// place is what stops a fixture from drifting into a different shape from the
|
||||
// database production loads.
|
||||
func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
// database production loads. requireCoverage applies the meaning-coverage
|
||||
// floor, which only a corpus build can be held to.
|
||||
func finish(cfg config, words map[string]entry, meanings map[string][]sense, src sourceSpec, requireCoverage bool) error {
|
||||
if len(words) < cfg.minWords {
|
||||
return fmt.Errorf("only %d words survived filtering, expected at least %d — "+
|
||||
"the source content may have changed", len(words), cfg.minWords)
|
||||
@@ -106,10 +127,10 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
aliases, collisions := buildAliases(words)
|
||||
log.Printf("generated %d spelling aliases (%d skipped as ambiguous or already real words)", len(aliases), collisions)
|
||||
|
||||
if err := write(cfg.out, words, aliases, src); err != nil {
|
||||
if err := write(cfg.out, words, meanings, aliases, src); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := verify(cfg.out, cfg.minWords); err != nil {
|
||||
if err := verify(cfg.out, cfg.minWords, requireCoverage); err != nil {
|
||||
return fmt.Errorf("output failed verification: %w", err)
|
||||
}
|
||||
|
||||
@@ -118,13 +139,17 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
}
|
||||
|
||||
// runFromWordList derives a database from a plain list of words instead of the
|
||||
// upstream export.
|
||||
// dump.
|
||||
//
|
||||
// It exists so tests and CI have a real dictionary to play against without the
|
||||
// upstream download. The filtering, alias generation, writing and verification
|
||||
// below are the same functions the real build uses — only the source of the
|
||||
// raw strings differs — so a fixture cannot drift into being shaped
|
||||
// differently from what production loads.
|
||||
//
|
||||
// A line is `word`, or `word<TAB>sense<TAB>sense…` where a sense is
|
||||
// `pos|gloss` or just `gloss`. The pipe never survives the stripper, so it is
|
||||
// a safe separator for hand-written meanings.
|
||||
func runFromWordList(cfg config) error {
|
||||
raw, err := os.ReadFile(cfg.words)
|
||||
if err != nil {
|
||||
@@ -132,6 +157,7 @@ func runFromWordList(cfg config) error {
|
||||
}
|
||||
|
||||
words := make(map[string]entry)
|
||||
meanings := make(map[string][]sense)
|
||||
rejects := make(map[rejectReason]int)
|
||||
|
||||
for line := range strings.Lines(string(raw)) {
|
||||
@@ -139,7 +165,8 @@ func runFromWordList(cfg config) error {
|
||||
if line == "" || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
word, syllables, reason, ok := accept(line)
|
||||
cells := strings.Split(line, "\t")
|
||||
word, syllables, reason, ok := accept(cells[0])
|
||||
if !ok {
|
||||
rejects[reason]++
|
||||
continue
|
||||
@@ -150,26 +177,54 @@ func runFromWordList(cfg config) error {
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
if senses := parseSenses(cells[1:]); len(senses) > 0 {
|
||||
meanings[word] = senses
|
||||
}
|
||||
}
|
||||
|
||||
logRejects(rejects)
|
||||
log.Printf("accepted %d distinct words from %s", len(words), cfg.words)
|
||||
log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words)
|
||||
|
||||
// The source spec is what lands in the meta table. Naming the list rather
|
||||
// than a table makes it obvious in the output which build produced a given
|
||||
// database — and a hand-written list carries no upstream licence, so the
|
||||
// fixture must not claim one.
|
||||
return finish(cfg, words, sourceSpec{
|
||||
return finish(cfg, words, meanings, sourceSpec{
|
||||
table: "wordlist:" + filepath.Base(cfg.words),
|
||||
license: "none: hand-written fixture wordlist, no upstream data",
|
||||
attribution: "Fixture written by this project; no third-party attribution applies.",
|
||||
})
|
||||
}, false)
|
||||
}
|
||||
|
||||
// parseSenses reads the tab-separated meaning cells of a fixture line. A cell
|
||||
// is `pos|gloss` or a bare gloss; empty cells are skipped and the cap applies
|
||||
// as it does to the dump.
|
||||
func parseSenses(cells []string) []sense {
|
||||
var senses []sense
|
||||
for _, cell := range cells {
|
||||
cell = strings.TrimSpace(cell)
|
||||
if cell == "" {
|
||||
continue
|
||||
}
|
||||
s := sense{gloss: cell}
|
||||
if pos, gloss, ok := strings.Cut(cell, "|"); ok {
|
||||
s = sense{pos: strings.TrimSpace(pos), gloss: strings.TrimSpace(gloss)}
|
||||
}
|
||||
if s.gloss == "" {
|
||||
continue
|
||||
}
|
||||
s.gloss, _ = capGloss(s.gloss)
|
||||
if len(senses) < maxSenses {
|
||||
senses = append(senses, s)
|
||||
}
|
||||
}
|
||||
return senses
|
||||
}
|
||||
|
||||
// verify re-opens the finished database and re-checks the invariants the game
|
||||
// depends on. The in-memory checks above can only prove what the builder
|
||||
// intended; this proves what actually landed on disk.
|
||||
func verify(path string, minWords int) error {
|
||||
func verify(path string, minWords int, requireCoverage bool) error {
|
||||
db, err := sql.Open("sqlite", "file:"+path+"?mode=ro")
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -193,6 +248,15 @@ func verify(path string, minWords int) error {
|
||||
{"words whose first syllable is missing from the syllables table",
|
||||
`SELECT COUNT(*) FROM words w LEFT JOIN syllables s ON s.syllable = w.first WHERE s.syllable IS NULL`,
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"meanings whose word is missing from the words table",
|
||||
`SELECT COUNT(*) FROM meanings m LEFT JOIN words w ON w.word = m.word WHERE w.word IS NULL`,
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"meanings with an empty gloss", `SELECT COUNT(*) FROM meanings WHERE gloss = ''`, func(n int) bool { return n == 0 }},
|
||||
{"meanings over the length cap", fmt.Sprintf(`SELECT COUNT(*) FROM meanings WHERE LENGTH(gloss) > %d`, maxGlossRunes),
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"words with more meanings than the cap",
|
||||
fmt.Sprintf(`SELECT COUNT(*) FROM (SELECT word FROM meanings GROUP BY word HAVING COUNT(*) > %d)`, maxSenses),
|
||||
func(n int) bool { return n == 0 }},
|
||||
}
|
||||
|
||||
for _, c := range checks {
|
||||
@@ -205,6 +269,20 @@ func verify(path string, minWords int) error {
|
||||
}
|
||||
}
|
||||
|
||||
if requireCoverage {
|
||||
var wordCount, withMeaning int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&wordCount); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := db.QueryRow(`SELECT COUNT(DISTINCT word) FROM meanings`).Scan(&withMeaning); err != nil {
|
||||
return err
|
||||
}
|
||||
if float64(withMeaning) < minMeaningCoverage*float64(wordCount) {
|
||||
return fmt.Errorf("only %d of %d words have a meaning, expected at least %.0f%% — "+
|
||||
"the dump's definition markup may have changed", withMeaning, wordCount, minMeaningCoverage*100)
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -218,7 +296,7 @@ type entry struct {
|
||||
|
||||
// sourceSpec is what the meta table records about where the words came from.
|
||||
type sourceSpec struct {
|
||||
// table names the input: "kaikki:<file>" for the corpus, "wordlist:<file>"
|
||||
// table names the input: "dump:<file>" for the corpus, "wordlist:<file>"
|
||||
// for a fixture, so the output says which build produced it.
|
||||
table string
|
||||
// url is the upstream artifact; empty for fixture builds.
|
||||
@@ -281,7 +359,7 @@ func buildAliases(words map[string]entry) (map[string]string, int) {
|
||||
// partway through -- a full disk, an interrupt -- leaves an empty but
|
||||
// syntactically valid database where a good one used to be, which the server
|
||||
// would happily open and find no words in.
|
||||
func write(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
|
||||
func write(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
|
||||
tmp := path + ".tmp"
|
||||
if err := os.Remove(tmp); err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fmt.Errorf("remove stale temp file: %w", err)
|
||||
@@ -294,7 +372,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
|
||||
}
|
||||
}()
|
||||
|
||||
if err := writeTo(tmp, words, aliases, src); err != nil {
|
||||
if err := writeTo(tmp, words, meanings, aliases, src); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -311,7 +389,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
|
||||
return nil
|
||||
}
|
||||
|
||||
func writeTo(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
|
||||
func writeTo(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
|
||||
db, err := sql.Open("sqlite", "file:"+path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create output: %w", err)
|
||||
@@ -337,6 +415,17 @@ CREATE TABLE aliases (
|
||||
canonical TEXT NOT NULL
|
||||
) WITHOUT ROWID;
|
||||
|
||||
-- One row per sense, in page order. pos is the Vietnamese part-of-speech
|
||||
-- label of the heading the definition sat under, '' when the heading was one
|
||||
-- the builder does not know. No foreign key pragma: verify() checks the join.
|
||||
CREATE TABLE meanings (
|
||||
word TEXT NOT NULL,
|
||||
ord INTEGER NOT NULL,
|
||||
pos TEXT NOT NULL,
|
||||
gloss TEXT NOT NULL,
|
||||
PRIMARY KEY (word, ord)
|
||||
) WITHOUT ROWID;
|
||||
|
||||
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
`
|
||||
if _, err := db.Exec(schema); err != nil {
|
||||
@@ -390,6 +479,27 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
}
|
||||
}
|
||||
|
||||
insertMeaning, err := tx.Prepare(`INSERT INTO meanings (word, ord, pos, gloss) VALUES (?, ?, ?, ?)`)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer insertMeaning.Close()
|
||||
meaningCount := 0
|
||||
for word, senses := range meanings {
|
||||
if _, isWord := words[word]; !isWord {
|
||||
return fmt.Errorf("meaning for %q, which is not a word", word)
|
||||
}
|
||||
for ord, s := range senses {
|
||||
if s.gloss == "" || utf8.RuneCountInString(s.gloss) > maxGlossRunes {
|
||||
return fmt.Errorf("meaning %d of %q is empty or over the cap", ord, word)
|
||||
}
|
||||
if _, err := insertMeaning.Exec(word, ord, s.pos, s.gloss); err != nil {
|
||||
return fmt.Errorf("insert meaning %d of %q: %w", ord, word, err)
|
||||
}
|
||||
meaningCount++
|
||||
}
|
||||
}
|
||||
|
||||
insertMeta, err := tx.Prepare(`INSERT INTO meta (key, value) VALUES (?, ?)`)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -403,6 +513,8 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
{"builder_version", builderVer},
|
||||
{"word_count", fmt.Sprint(len(words))},
|
||||
{"alias_count", fmt.Sprint(len(aliases))},
|
||||
{"meaning_count", fmt.Sprint(meaningCount)},
|
||||
{"words_with_meaning", fmt.Sprint(len(meanings))},
|
||||
{"source_table", src.table},
|
||||
}
|
||||
meta = append(meta, src.extra...)
|
||||
|
||||
@@ -6,47 +6,25 @@ import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// fixtureSource writes a miniature stand-in for the kaikki export: the same
|
||||
// JSONL shape, a handful of rows instead of 44k. Each row is a word and the
|
||||
// language its Wiktionary entry is for. Tests never touch the real download.
|
||||
func fixtureSource(t *testing.T, rows [][2]string) string {
|
||||
t.Helper()
|
||||
|
||||
lines := make([]string, 0, len(rows))
|
||||
for _, r := range rows {
|
||||
lines = append(lines, `{"word": "`+r[0]+`", "pos": "noun", "lang_code": "`+r[1]+`"}`)
|
||||
}
|
||||
return fixtureKaikki(t, lines...)
|
||||
}
|
||||
|
||||
func defaultRows() [][2]string {
|
||||
return [][2]string{
|
||||
{"pháp luật", "vi"},
|
||||
{"pháp luật", "vi"}, // listed twice — must dedupe to one word
|
||||
{"luật lệ", "vi"},
|
||||
{"ngôn ngữ", "vi"},
|
||||
{"ngữ pháp", "vi"},
|
||||
{"hòa bình", "vi"},
|
||||
{"vô tuyến điện", "vi"}, // three syllables
|
||||
{"pháp", "vi"}, // single syllable — rejected
|
||||
{"covid 19", "vi"}, // digit — rejected
|
||||
{"hello world", "en"}, // another language's entry — never selected
|
||||
}
|
||||
}
|
||||
|
||||
func buildFixture(t *testing.T, rows [][2]string) string {
|
||||
// buildFixture runs the whole pipeline on the committed mini dump: twelve
|
||||
// pages, both dialects, a redirect, an English-only page and two pages that
|
||||
// accept() rejects. Tests never touch the real download.
|
||||
func buildFixture(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, rows),
|
||||
dump: miniDump,
|
||||
out: out,
|
||||
minWords: 1,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
@@ -64,23 +42,23 @@ func openOut(t *testing.T, path string) *sql.DB {
|
||||
return db
|
||||
}
|
||||
|
||||
func count(t *testing.T, db *sql.DB, query string, args ...any) int {
|
||||
t.Helper()
|
||||
var n int
|
||||
if err := db.QueryRow(query, args...).Scan(&n); err != nil {
|
||||
t.Fatalf("%s: %v", query, err)
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func TestBuildProducesExpectedWords(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
var count int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&count); err != nil {
|
||||
t.Fatal(err)
|
||||
if got := count(t, db, `SELECT COUNT(*) FROM words`); got != 6 {
|
||||
t.Errorf("word count = %d, want 6", got)
|
||||
}
|
||||
if want := 6; count != want {
|
||||
t.Errorf("word count = %d, want %d", count, want)
|
||||
}
|
||||
|
||||
// Every stored word must have at least two syllables.
|
||||
var short int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words WHERE syllables < 2`).Scan(&short); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if short != 0 {
|
||||
if short := count(t, db, `SELECT COUNT(*) FROM words WHERE syllables < 2`); short != 0 {
|
||||
t.Errorf("%d words have fewer than 2 syllables, want 0", short)
|
||||
}
|
||||
|
||||
@@ -98,7 +76,7 @@ func TestBuildProducesExpectedWords(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestBuildComputesOutDegree(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
// "pháp luật" and "pháp" (rejected) mean exactly one word starts with "pháp".
|
||||
assertOutDegree(t, db, "pháp", 1)
|
||||
@@ -120,7 +98,7 @@ func assertOutDegree(t *testing.T, db *sql.DB, syllable string, want int) {
|
||||
}
|
||||
|
||||
func TestBuildWritesAliases(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
var canonical string
|
||||
err := db.QueryRow(`SELECT canonical FROM aliases WHERE variant = ?`, "hoà bình").Scan(&canonical)
|
||||
@@ -132,22 +110,65 @@ func TestBuildWritesAliases(t *testing.T) {
|
||||
}
|
||||
|
||||
// Every alias must point at a word that actually exists.
|
||||
var orphans int
|
||||
err = db.QueryRow(`SELECT COUNT(*) FROM aliases a
|
||||
LEFT JOIN words w ON w.word = a.canonical
|
||||
WHERE w.word IS NULL`).Scan(&orphans)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
orphans := count(t, db, `SELECT COUNT(*) FROM aliases a LEFT JOIN words w ON w.word = a.canonical WHERE w.word IS NULL`)
|
||||
if orphans != 0 {
|
||||
t.Errorf("%d aliases point at missing words, want 0", orphans)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildRecordsProvenance(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
func TestBuildWritesMeanings(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count"} {
|
||||
rows, err := db.Query(`SELECT ord, pos, gloss FROM meanings WHERE word = ? ORDER BY ord`, "pháp luật")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var got []sense
|
||||
for rows.Next() {
|
||||
var ord int
|
||||
var s sense
|
||||
if err := rows.Scan(&ord, &s.pos, &s.gloss); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if ord != len(got) {
|
||||
t.Errorf("ord = %d, want %d (0-based, dense)", ord, len(got))
|
||||
}
|
||||
got = append(got, s)
|
||||
}
|
||||
assertSenses(t, got, []sense{
|
||||
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
|
||||
{"động từ", "(hiếm) Xử theo luật."},
|
||||
})
|
||||
|
||||
// A word whose only definition stripped to nothing has no rows, and the
|
||||
// meta counts describe the table.
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM meanings WHERE word = ?`, "luật lệ"); n != 0 {
|
||||
t.Errorf("luật lệ has %d meanings, want 0", n)
|
||||
}
|
||||
total := count(t, db, `SELECT COUNT(*) FROM meanings`)
|
||||
withMeaning := count(t, db, `SELECT COUNT(DISTINCT word) FROM meanings`)
|
||||
for key, want := range map[string]int{"meaning_count": total, "words_with_meaning": withMeaning} {
|
||||
var raw string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
|
||||
t.Fatalf("meta[%q]: %v", key, err)
|
||||
}
|
||||
if raw != strconv.Itoa(want) {
|
||||
t.Errorf("meta[%q] = %s, want %d", key, raw, want)
|
||||
}
|
||||
}
|
||||
if withMeaning != 5 {
|
||||
t.Errorf("words with a meaning = %d, want 5 of 6", withMeaning)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildRecordsProvenance(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
want := map[string]string{"builder_version": builderVer, "source_url": dumpSourceURL, "source_pages": "9"}
|
||||
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count",
|
||||
"builder_version", "source_sha256", "source_pages", "source_fetched_at", "meaning_count", "words_with_meaning"} {
|
||||
var value string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&value); err != nil {
|
||||
t.Errorf("meta[%q] missing: %v", key, err)
|
||||
@@ -156,26 +177,42 @@ func TestBuildRecordsProvenance(t *testing.T) {
|
||||
if value == "" {
|
||||
t.Errorf("meta[%q] is empty", key)
|
||||
}
|
||||
if w, ok := want[key]; ok && value != w {
|
||||
t.Errorf("meta[%q] = %q, want %q", key, value, w)
|
||||
}
|
||||
}
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM meta WHERE key = 'source_rows'`); n != 0 {
|
||||
t.Error("source_rows belonged to the previous source format and must be gone")
|
||||
}
|
||||
}
|
||||
|
||||
// The floor exists so a schema change upstream fails the build loudly instead
|
||||
// The floor exists so a markup change upstream fails the build loudly instead
|
||||
// of silently shipping a near-empty dictionary.
|
||||
func TestBuildFailsBelowMinWords(t *testing.T) {
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, defaultRows()),
|
||||
dump: miniDump,
|
||||
out: filepath.Join(t.TempDir(), "noitu.db"),
|
||||
minWords: 1000,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err == nil {
|
||||
t.Fatal("run succeeded with an unreachable min-words floor, want error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRequiresExactlyOneInput(t *testing.T) {
|
||||
if err := run(config{}); err == nil {
|
||||
t.Error("run with no input succeeded")
|
||||
}
|
||||
if err := run(config{dump: miniDump, words: "x.txt"}); err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
||||
t.Errorf("run with both inputs: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A failed build must leave the previous good database untouched. Building in
|
||||
// place would delete it and leave an empty file the server would happily open.
|
||||
func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
out := buildFixture(t, defaultRows())
|
||||
out := buildFixture(t)
|
||||
|
||||
before, err := os.ReadFile(out)
|
||||
if err != nil {
|
||||
@@ -184,9 +221,10 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
|
||||
// Same output path, but a floor no fixture can clear.
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, defaultRows()),
|
||||
dump: miniDump,
|
||||
out: out,
|
||||
minWords: 1000,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err == nil {
|
||||
t.Fatal("run succeeded with an unreachable floor, want error")
|
||||
@@ -203,3 +241,127 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
t.Error("temp database left behind after a failed build")
|
||||
}
|
||||
}
|
||||
|
||||
// --- the fixture word list --------------------------------------------------
|
||||
|
||||
func writeWordList(t *testing.T, content string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "words.txt")
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestWordListCarriesTabSeparatedMeanings(t *testing.T) {
|
||||
list := writeWordList(t, "# comment\n"+
|
||||
"học sinh\tdanh từ|Người học ở trường.\tđộng từ|Đi học.\n"+
|
||||
"sinh viên\tNgười học ở trường đại học.\n"+
|
||||
"sinh hoạt\n"+
|
||||
"sinh sản\t\t\n")
|
||||
out := filepath.Join(t.TempDir(), "fixture.db")
|
||||
if err := run(config{words: list, out: out, minWords: 1}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
db := openOut(t, out)
|
||||
|
||||
rows, err := db.Query(`SELECT word, ord, pos, gloss FROM meanings ORDER BY word, ord`)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
type row struct {
|
||||
word string
|
||||
ord int
|
||||
s sense
|
||||
}
|
||||
var got []row
|
||||
for rows.Next() {
|
||||
var r row
|
||||
if err := rows.Scan(&r.word, &r.ord, &r.s.pos, &r.s.gloss); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got = append(got, r)
|
||||
}
|
||||
want := []row{
|
||||
{"học sinh", 0, sense{"danh từ", "Người học ở trường."}},
|
||||
{"học sinh", 1, sense{"động từ", "Đi học."}},
|
||||
{"sinh viên", 0, sense{"", "Người học ở trường đại học."}},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("meanings = %+v, want %+v", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("row %d = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM words`); n != 4 {
|
||||
t.Errorf("words = %d, want 4 (a line without a tab is still a word)", n)
|
||||
}
|
||||
// Fixture builds carry no upstream licence and are exempt from the
|
||||
// coverage floor: two of four words have a meaning here.
|
||||
var license string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.HasPrefix(license, "none") {
|
||||
t.Errorf("fixture licence = %q, want a statement that no upstream data applies", license)
|
||||
}
|
||||
}
|
||||
|
||||
// --- verify -----------------------------------------------------------------
|
||||
|
||||
// brokenDB writes a database that passes every schema check and then breaks
|
||||
// one invariant, to prove verify() reads what is on disk.
|
||||
func brokenDB(t *testing.T, extraSQL string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "broken.db")
|
||||
words := map[string]entry{"pháp luật": {"pháp luật", "pháp", "luật", 2}}
|
||||
meanings := map[string][]sense{"pháp luật": {{"danh từ", "Luật."}}}
|
||||
if err := writeTo(path, words, meanings, nil, sourceSpec{table: "test"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
db, err := sql.Open("sqlite", "file:"+path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer db.Close()
|
||||
if _, err := db.Exec(extraSQL); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestVerifyRejectsBrokenMeanings(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"orphan meaning row": `INSERT INTO meanings VALUES ('không có', 0, '', 'Một nghĩa.')`,
|
||||
"empty gloss": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '')`,
|
||||
"over the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '` + strings.Repeat("a", maxGlossRunes+1) + `')`,
|
||||
"more senses than the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', 'b'), ('pháp luật', 2, '', 'c'),
|
||||
('pháp luật', 3, '', 'd'), ('pháp luật', 4, '', 'e'), ('pháp luật', 5, '', 'f')`,
|
||||
}
|
||||
for name, sqlText := range cases {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
path := brokenDB(t, sqlText)
|
||||
if err := verify(path, 1, false); err == nil {
|
||||
t.Error("verify passed a database that breaks a meanings invariant")
|
||||
}
|
||||
})
|
||||
}
|
||||
if err := verify(brokenDB(t, `SELECT 1`), 1, false); err != nil {
|
||||
t.Errorf("verify rejected a sound database: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestVerifyCoverageFloorAppliesToCorpusBuildsOnly(t *testing.T) {
|
||||
// One word with a meaning, one without: 50%, under the floor.
|
||||
path := brokenDB(t, `INSERT INTO words VALUES ('luật lệ', 'luật', 'lệ', 2);
|
||||
INSERT INTO syllables VALUES ('lệ', 0); UPDATE syllables SET out_degree = 1 WHERE syllable = 'luật'`)
|
||||
if err := verify(path, 1, true); err == nil || !strings.Contains(err.Error(), "have a meaning") {
|
||||
t.Errorf("corpus verify with 50%% coverage: %v, want the coverage floor named", err)
|
||||
}
|
||||
if err := verify(path, 1, false); err != nil {
|
||||
t.Errorf("fixture verify applied the coverage floor: %v", err)
|
||||
}
|
||||
}
|
||||
Binary file not shown.
+182
@@ -0,0 +1,182 @@
|
||||
<mediawiki xmlns="http://www.mediawiki.org/xml/export-0.11/" xml:lang="vi">
|
||||
<siteinfo>
|
||||
<sitename>Wiktionary</sitename>
|
||||
<dbname>viwiktionary</dbname>
|
||||
</siteinfo>
|
||||
<!-- A legacy-dialect page: two parts of speech, an English section after
|
||||
the Vietnamese one that must not be read, and a line with no space
|
||||
after the # which is still a definition. -->
|
||||
<page>
|
||||
<title>pháp luật</title>
|
||||
<ns>0</ns>
|
||||
<id>1</id>
|
||||
<revision>
|
||||
<id>101</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-pron-}}
|
||||
{{vie-pron|pháp luật}}
|
||||
|
||||
{{-noun-}}
|
||||
{{-dfn-}}
|
||||
# [[hệ thống|Hệ thống]] các [[quy tắc]] xử sự do [[nhà nước]] đặt ra.<ref>Từ điển tiếng Việt</ref>
|
||||
#: ''Tuân theo pháp luật.''
|
||||
#{{label|vi|nghĩa rộng}} [[kỷ cương|Kỷ cương]] nói chung.
|
||||
|
||||
{{-verb-}}
|
||||
# ''(hiếm)'' [[xử|Xử]] theo luật.
|
||||
|
||||
{{-trans-}}
|
||||
* {{eng}}: {{t|en|law}}
|
||||
|
||||
{{-eng-}}
|
||||
{{-noun-}}
|
||||
# Law, in English.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A new-dialect page with a proper-noun heading, a place template and an
|
||||
English section that must not be read. -->
|
||||
<page>
|
||||
<title>Hòa Bình</title>
|
||||
<ns>0</ns>
|
||||
<id>2</id>
|
||||
<revision>
|
||||
<id>102</id>
|
||||
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
|
||||
=== {{ĐM|etym}} ===
|
||||
Từ Hán-Việt.
|
||||
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
{{vi-pr-noun}}
|
||||
|
||||
# {{place|vi|tỉnh|c/Việt Nam}}.
|
||||
|
||||
== {{langname|en}} ==
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
# A province of Vietnam.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- The lowercase page of the same word: the two merge into one entry with
|
||||
the senses in page order. -->
|
||||
<page>
|
||||
<title>hòa bình</title>
|
||||
<ns>0</ns>
|
||||
<id>3</id>
|
||||
<revision>
|
||||
<id>103</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[tình trạng|Tình trạng]] không có [[chiến tranh]].
|
||||
{{-adj-}}
|
||||
# [[yên ổn|Yên ổn]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A case-only redirect: skipped and counted. -->
|
||||
<page>
|
||||
<title>mặt trời</title>
|
||||
<ns>0</ns>
|
||||
<id>4</id>
|
||||
<redirect title="Mặt Trời" />
|
||||
<revision>
|
||||
<id>104</id>
|
||||
<text bytes="1" xml:space="preserve">#đổi [[Mặt Trời]]</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- No Vietnamese section at all. -->
|
||||
<page>
|
||||
<title>hello world</title>
|
||||
<ns>0</ns>
|
||||
<id>5</id>
|
||||
<revision>
|
||||
<id>105</id>
|
||||
<text bytes="1" xml:space="preserve">{{-eng-}}
|
||||
{{-phrase-}}
|
||||
# Xin chào thế giới.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- Not in the main namespace. -->
|
||||
<page>
|
||||
<title>Thể loại:Danh từ tiếng Việt</title>
|
||||
<ns>14</ns>
|
||||
<id>6</id>
|
||||
<revision>
|
||||
<id>106</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# Not a word.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- One syllable: rejected downstream by accept(), but its section is
|
||||
still a Vietnamese section for the page count. -->
|
||||
<page>
|
||||
<title>pháp</title>
|
||||
<ns>0</ns>
|
||||
<id>7</id>
|
||||
<revision>
|
||||
<id>107</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[phép|Phép]], [[luật]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A definition that is only a template the stripper does not know: no
|
||||
meaning, but still a word. -->
|
||||
<page>
|
||||
<title>luật lệ</title>
|
||||
<ns>0</ns>
|
||||
<id>8</id>
|
||||
<revision>
|
||||
<id>108</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# {{rfdef|vi}}</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- More graph: a shorthand heading code in the new dialect, and words that
|
||||
give the out-degree and alias tests something to check. -->
|
||||
<page>
|
||||
<title>ngôn ngữ</title>
|
||||
<ns>0</ns>
|
||||
<id>9</id>
|
||||
<revision>
|
||||
<id>109</id>
|
||||
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
|
||||
=== {{section|n}} ===
|
||||
{{vi-noun}}
|
||||
|
||||
# [[hệ thống|Hệ thống]] những [[âm]], [[từ]] và [[quy tắc]] kết hợp chúng.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>ngữ pháp</title>
|
||||
<ns>0</ns>
|
||||
<id>10</id>
|
||||
<revision>
|
||||
<id>110</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[toàn bộ|Toàn bộ]] những [[quy tắc]] hoạt động của các yếu tố ngôn ngữ.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>vô tuyến điện</title>
|
||||
<ns>0</ns>
|
||||
<id>11</id>
|
||||
<revision>
|
||||
<id>111</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[kỹ thuật|Kỹ thuật]] truyền tin bằng [[sóng điện từ]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>covid 19</title>
|
||||
<ns>0</ns>
|
||||
<id>12</id>
|
||||
<revision>
|
||||
<id>112</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# Một [[bệnh]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
</mediawiki>
|
||||
Binary file not shown.
@@ -0,0 +1,629 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"html"
|
||||
"regexp"
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// This file reads the wikitext of one Wiktionary tiếng Việt page: it finds the
|
||||
// Vietnamese section, walks its part-of-speech headings and turns each
|
||||
// definition line into plain text.
|
||||
//
|
||||
// The wiki is mid-migration between two markup dialects and both are live
|
||||
// (2026-09-01 dump: 35,885 legacy pages, 7,129 new):
|
||||
//
|
||||
// legacy {{-vie-}} opens the section, {{-noun-}} and kin are the headings,
|
||||
// and the section ends at the next {{-xxx-}} whose code is a
|
||||
// language rather than a heading.
|
||||
// new == {{langname|vi}} == opens the section, === {{ĐM|noun}} === or
|
||||
// === {{section|noun}} === are the headings, and the next level-2
|
||||
// heading ends it.
|
||||
//
|
||||
// Nothing here is a general wikitext parser. It knows exactly the shapes a
|
||||
// definition line takes on this wiki and drops the rest on purpose; what
|
||||
// survives is plain text, capped, safe to render as text and never as markup.
|
||||
|
||||
// sense is one definition with the Vietnamese part-of-speech label of the
|
||||
// heading it sat under. pos is empty when the heading was one the label map
|
||||
// does not know, never a reason to drop the definition.
|
||||
type sense struct {
|
||||
pos string
|
||||
gloss string
|
||||
}
|
||||
|
||||
const (
|
||||
// maxSenses and maxGlossRunes bound what one word carries to the client.
|
||||
maxSenses = 5
|
||||
maxGlossRunes = 200
|
||||
// ellipsis marks a definition cut at maxGlossRunes.
|
||||
ellipsis = "…"
|
||||
)
|
||||
|
||||
// posLabelMap maps a part-of-speech heading code, the same in both dialects
|
||||
// ({{-noun-}}, {{ĐM|noun}}, {{section|noun}}, {{vi-noun}}), to the Vietnamese
|
||||
// label the client shows in front of a sense.
|
||||
//
|
||||
// Codes and their frequencies in Vietnamese sections of the 2026-09-01 dump:
|
||||
// noun 13,674 + n 647 · verb 7,595 + v 353 · adj 5,318 + adjc 526 · place 3,352
|
||||
// · pr-noun 1,397 + 329 · adv 982 · phrase 412 · proverb 295 · idiom 241 ·
|
||||
// interj 169 · pronoun 158 · num 123 · conj 88 · prep 88 · part 29. Note that
|
||||
// "pron" on this wiki is pronunciation, not pronoun.
|
||||
var posLabelMap = map[string]string{
|
||||
"noun": "danh từ", "n": "danh từ",
|
||||
"verb": "động từ", "v": "động từ", "tr-verb": "động từ", "intr-verb": "động từ", "aux-verb": "động từ",
|
||||
"adj": "tính từ", "adjc": "tính từ", "adjective": "tính từ",
|
||||
"adv": "phó từ", "adverb": "phó từ", "advb": "phó từ",
|
||||
"pr-noun": "danh từ riêng", "proper": "danh từ riêng", "propn": "danh từ riêng", "proper noun": "danh từ riêng", "name": "danh từ riêng",
|
||||
"pr-adj": "tính từ riêng",
|
||||
"place": "địa danh",
|
||||
"pronoun": "đại từ", "per-pronoun": "đại từ",
|
||||
"num": "số từ", "numeral": "số từ",
|
||||
"conj": "liên từ", "conjunction": "liên từ",
|
||||
"prep": "giới từ",
|
||||
"interj": "thán từ", "intj": "thán từ", "interjection": "thán từ",
|
||||
"part": "trợ từ", "particle": "trợ từ",
|
||||
"phrase": "cụm từ",
|
||||
"idiom": "thành ngữ",
|
||||
"proverb": "tục ngữ", "prov": "tục ngữ",
|
||||
"abbr": "viết tắt", "abr": "viết tắt",
|
||||
"prefix": "tiền tố",
|
||||
"suffix": "hậu tố",
|
||||
"letter": "chữ cái",
|
||||
"symbol": "ký hiệu",
|
||||
}
|
||||
|
||||
// otherSectionCodes are heading codes that are not parts of speech: they sit
|
||||
// inside a language section and reset the current label without being
|
||||
// counted as unmapped. Frequencies in Vietnamese sections, 2026-09-01 dump:
|
||||
// pron 36,533 · ref 27,438 · trans 16,073 · paro 8,058 · etym 4,267 · syn
|
||||
// 3,165 · info 1,846 · hanviet 1,601 · hanviet-t 1,441 · etymology 1,429 ·
|
||||
// reference 1,426 · see 959 · related 461 · drv 277 · synonym 262 · ant 236 ·
|
||||
// desction 128 · usage 110 · expr 92 · homo 57 · desc 54 · further 48 · forms
|
||||
// 41 · note 33 · derived 28 · anagram 24 · compound 21 · redup 20 · translit 17
|
||||
// · cat 16 · antonym 15. "dfn" (4,615) is a "definitions" heading placed under a
|
||||
// part-of-speech heading, so it is a heading for the section boundary but
|
||||
// transparent to the label: see classifyHeading.
|
||||
var otherSectionCodes = map[string]bool{
|
||||
"pron": true, "pronunciation": true, "ref": true, "reference": true, "references": true,
|
||||
"trans": true, "translations": true, "paro": true, "paronym": true, "etym": true, "etymology": true,
|
||||
"syn": true, "synonym": true, "ant": true, "antonym": true, "info": true,
|
||||
"hanviet": true, "hanviet-t": true, "see": true, "see also": true, "related": true, "rel": true,
|
||||
"related terms": true, "drv": true, "der": true, "derived": true, "derived terms": true,
|
||||
"desction": true, "desc": true, "usage": true, "usage notes": true, "expr": true, "homo": true,
|
||||
"further": true, "further reading": true, "forms": true, "note": true, "anagram": true,
|
||||
"anagrams": true, "ana": true, "compound": true, "redup": true, "translit": true, "cat": true,
|
||||
"coord": true, "coordinate": true, "alt": true, "alter": true, "alter form": true,
|
||||
"alternative form": true, "alternative forms": true, "alternative script": true,
|
||||
"glyph origin": true, "han": true, "nôm": true, "han character": true, "kanji": true,
|
||||
"rom": true, "romanization": true, "mut": true, "participle": true, "ptcp": true,
|
||||
"syllable": true, "article": true, "contr": true, "cmavo": true, "dfn": true, "com": true,
|
||||
// The same sections written out in Vietnamese, as a few new-dialect pages do.
|
||||
"phát âm": true, "từ nguyên": true, "từ nguyên 1": true, "từ nguyên 2": true, "tham khảo": true,
|
||||
"xem thêm": true, "cách viết khác": true, "phồn thể": true, "hán-nôm": true, "hán nôm": true,
|
||||
"chữ hán": true, "chữ nôm": true, "chú ý": true, "đồng nghĩa": true, "từ đồng nghĩa": true,
|
||||
"bản dịch": true, "dịch": true, "dấu phụ": true, "liên kết ngoài": true, "thuật ngữ liên quan": true,
|
||||
"từ tương tự": true, "meronym": true, "meronyms": true, "nguồn gốc ký tự chữ nôm": true,
|
||||
}
|
||||
|
||||
var (
|
||||
// legacyTemplate matches one {{-code-}} template, optionally with
|
||||
// parameters: {{-noun-}}, {{-pr-noun-}}, {{-vie-|...}}. Not anchored: a few
|
||||
// pages run {{-vie-}}{{-pron-}}{{vie-pron|…}}{{-place-}} together on one
|
||||
// line, so a line is read as headings when it starts with one and may
|
||||
// carry several.
|
||||
legacyTemplate = regexp.MustCompile(`\{\{-([A-Za-z0-9-]+?)-(?:\|[^}]*)?\}\}`)
|
||||
// headingLine matches == text == at any level and captures the level.
|
||||
headingLine = regexp.MustCompile(`^(={2,6})\s*(.*?)\s*=+\s*$`)
|
||||
// sectionTemplate captures the code of {{ĐM|code}} and {{section|code}}.
|
||||
sectionTemplate = regexp.MustCompile(`\{\{(?:ĐM|đm|DM|dm|section)\|([^}|]+)`)
|
||||
// headwordTemplate captures the code of {{vi-code}} / {{vie-code}} at the
|
||||
// start of a line: the new dialect's headword line, which names the POS.
|
||||
headwordTemplate = regexp.MustCompile(`^\{\{vie?-([a-z -]+)`)
|
||||
// langnameVi is the new dialect's Vietnamese section heading text.
|
||||
langnameVi = regexp.MustCompile(`^\{\{langname\|vi\}\}$`)
|
||||
|
||||
htmlComment = regexp.MustCompile(`(?s)<!--.*?-->`)
|
||||
refElement = regexp.MustCompile(`(?s)<ref\b[^>/]*/>|<ref\b[^>]*>.*?</ref>`)
|
||||
anyTag = regexp.MustCompile(`</?[A-Za-z][^>]*>`)
|
||||
spaces = regexp.MustCompile(`\s+`)
|
||||
)
|
||||
|
||||
// isLegacyHeading reports whether a {{-code-}} is a heading inside a language
|
||||
// section. Every other code — language and script codes such as eng, tyz,
|
||||
// aav-qal, Latn — ends the Vietnamese section.
|
||||
func isLegacyHeading(code string) bool {
|
||||
_, pos := posLabelMap[code]
|
||||
return pos || otherSectionCodes[code]
|
||||
}
|
||||
|
||||
// vietnameseSection returns the wikitext of the page's Vietnamese section and
|
||||
// which dialect opened it: "legacy", "new", or "" when the page has none.
|
||||
// When both dialects open a section on one page the first one in the text
|
||||
// wins and both is reported so the build log can count it. ender is the
|
||||
// {{-code-}} that closed a legacy section, empty when a heading or the end of
|
||||
// the page did: a heading code missing from the maps shows up there as a
|
||||
// section-ending code, which is the signal that definitions are being lost.
|
||||
func vietnameseSection(text string) (section, dialect string, both bool, ender string) {
|
||||
lines := strings.Split(text, "\n")
|
||||
legacyAt, newAt := -1, -1
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if legacyAt < 0 && strings.HasPrefix(line, "{{-") {
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatchIndex(line, -1) {
|
||||
if line[m[2]:m[3]] == "vie" {
|
||||
legacyAt = i
|
||||
// Whatever follows the marker on its own line belongs to
|
||||
// the section.
|
||||
lines[i] = line[m[1]:]
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if newAt < 0 {
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 && langnameVi.MatchString(m[2]) {
|
||||
newAt = i
|
||||
}
|
||||
}
|
||||
}
|
||||
both = legacyAt >= 0 && newAt >= 0
|
||||
switch {
|
||||
case legacyAt < 0 && newAt < 0:
|
||||
return "", "", false, ""
|
||||
case newAt < 0 || (legacyAt >= 0 && legacyAt < newAt):
|
||||
section, ender = legacySection(lines[legacyAt:])
|
||||
return section, "legacy", both, ender
|
||||
default:
|
||||
return newSection(lines[newAt+1:]), "new", both, ""
|
||||
}
|
||||
}
|
||||
|
||||
// legacySection runs from after {{-vie-}} to the next {{-xxx-}} whose code is
|
||||
// not a heading, or the next level-2 heading, which is what a new-dialect
|
||||
// language section on a mixed page opens with. A language code sharing a line
|
||||
// with Vietnamese headings ends the section at that line; the line is lost,
|
||||
// which is the conservative side of a rare shape.
|
||||
func legacySection(lines []string) (section, ender string) {
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if strings.HasPrefix(line, "{{-") {
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
|
||||
if !isLegacyHeading(m[1]) {
|
||||
return strings.Join(lines[:i], "\n"), m[1]
|
||||
}
|
||||
}
|
||||
}
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
|
||||
return strings.Join(lines[:i], "\n"), ""
|
||||
}
|
||||
}
|
||||
return strings.Join(lines, "\n"), ""
|
||||
}
|
||||
|
||||
// newSection runs from after == {{langname|vi}} == to the next level-2
|
||||
// heading.
|
||||
func newSection(lines []string) string {
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
|
||||
return strings.Join(lines[:i], "\n")
|
||||
}
|
||||
}
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// headingKind says what a heading line means for the label of the
|
||||
// definitions under it.
|
||||
type headingKind int
|
||||
|
||||
const (
|
||||
notHeading headingKind = iota
|
||||
// posHeading names a part of speech: the code decides the label.
|
||||
posHeading
|
||||
// otherHeading is a section such as pronunciation or etymology: the label
|
||||
// resets to empty and nothing is counted as unmapped.
|
||||
otherHeading
|
||||
)
|
||||
|
||||
// classifyHeading reads one line as a heading in either dialect.
|
||||
//
|
||||
// {{-noun-}} legacy heading; the last of several on
|
||||
// one line decides
|
||||
// === {{ĐM|noun}} === new heading
|
||||
// === {{section|n}} === new heading, shorthand code
|
||||
// === Danh từ === new heading written out
|
||||
// {{vi-noun}} / {{vie-noun}} new headword line; refines the POS only
|
||||
func classifyHeading(line string) (code string, kind headingKind) {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if strings.HasPrefix(line, "{{-") {
|
||||
kind = notHeading
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
|
||||
if c, k := classifyCode(m[1]); k != notHeading {
|
||||
code, kind = c, k
|
||||
}
|
||||
}
|
||||
return code, kind
|
||||
}
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) >= 3 {
|
||||
text := m[2]
|
||||
if sm := sectionTemplate.FindStringSubmatch(text); sm != nil {
|
||||
code = strings.TrimSpace(sm[1])
|
||||
} else {
|
||||
code = strings.ToLower(text)
|
||||
}
|
||||
return classifyCode(code)
|
||||
}
|
||||
if m := headwordTemplate.FindStringSubmatch(line); m != nil {
|
||||
// Only a headword template whose code is a part of speech counts;
|
||||
// {{vi-pron}}, {{vi-etym-sino}} and kin are not headings.
|
||||
code = strings.TrimSpace(m[1])
|
||||
if _, ok := posLabelMap[code]; ok {
|
||||
return code, posHeading
|
||||
}
|
||||
}
|
||||
return "", notHeading
|
||||
}
|
||||
|
||||
// classifyCode sorts a heading code seen in either dialect. "dfn" is the one
|
||||
// heading that changes nothing: the wiki places {{-dfn-}} under {{-noun-}} to
|
||||
// introduce the definitions, so the label above it must carry through.
|
||||
func classifyCode(code string) (string, headingKind) {
|
||||
switch {
|
||||
case code == "dfn":
|
||||
return code, notHeading
|
||||
case otherSectionCodes[code]:
|
||||
return code, otherHeading
|
||||
}
|
||||
return code, posHeading
|
||||
}
|
||||
|
||||
// isLabelValue reports whether a heading was written out as one of the
|
||||
// Vietnamese labels already ("Danh từ").
|
||||
func isLabelValue(text string) bool {
|
||||
for _, l := range posLabelMap {
|
||||
if l == text {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// posLabel maps a heading code to its Vietnamese label: through the map, or
|
||||
// as itself when the heading was already written out in Vietnamese.
|
||||
func posLabel(code string) (label string, mapped bool) {
|
||||
if label, ok := posLabelMap[code]; ok {
|
||||
return label, true
|
||||
}
|
||||
if isLabelValue(code) {
|
||||
return code, true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// sectionStats counts what the scanner saw across sections, for the build log.
|
||||
type sectionStats struct {
|
||||
pos map[string]int // part-of-speech heading codes seen
|
||||
unmappedPos map[string]int // part-of-speech heading codes with no label
|
||||
defsKept int
|
||||
defsEmpty int // definitions that stripped to nothing
|
||||
defsCut int // definitions cut at maxGlossRunes
|
||||
dropped map[string]int // template names dropped whole
|
||||
}
|
||||
|
||||
func newSectionStats() *sectionStats {
|
||||
return §ionStats{
|
||||
pos: make(map[string]int),
|
||||
unmappedPos: make(map[string]int),
|
||||
dropped: make(map[string]int),
|
||||
}
|
||||
}
|
||||
|
||||
// definitions walks the Vietnamese section and returns its senses in page
|
||||
// order, at most maxSenses of them, each labelled with the part of speech of
|
||||
// the heading above it. Every heading and definition is counted in stats
|
||||
// whether or not it made the cut.
|
||||
func definitions(section string, stats *sectionStats) []sense {
|
||||
var senses []sense
|
||||
pos := ""
|
||||
for _, line := range strings.Split(section, "\n") {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
switch code, kind := classifyHeading(line); kind {
|
||||
case posHeading:
|
||||
stats.pos[code]++
|
||||
label, mapped := posLabel(code)
|
||||
if !mapped {
|
||||
stats.unmappedPos[code]++
|
||||
}
|
||||
pos = label
|
||||
continue
|
||||
case otherHeading:
|
||||
pos = ""
|
||||
continue
|
||||
}
|
||||
if !isDefinitionLine(line) {
|
||||
continue
|
||||
}
|
||||
gloss := stripWikitext(line[1:], stats.dropped)
|
||||
if gloss == "" {
|
||||
stats.defsEmpty++
|
||||
continue
|
||||
}
|
||||
if cut, wasCut := capGloss(gloss); wasCut {
|
||||
stats.defsCut++
|
||||
gloss = cut
|
||||
}
|
||||
stats.defsKept++
|
||||
if len(senses) < maxSenses {
|
||||
senses = append(senses, sense{pos: pos, gloss: gloss})
|
||||
}
|
||||
}
|
||||
return senses
|
||||
}
|
||||
|
||||
// isDefinitionLine accepts a top-level numbered item and nothing under it:
|
||||
// "# text" and "#text" are definitions; "#: example", "#* quotation", "## sub-
|
||||
// sense" and "#; term" are not.
|
||||
func isDefinitionLine(line string) bool {
|
||||
if len(line) < 2 || line[0] != '#' {
|
||||
return false
|
||||
}
|
||||
switch line[1] {
|
||||
case '#', ':', '*', ';':
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// stripWikitext turns one definition line into plain text. Lossy on purpose:
|
||||
// links keep their display text, formatting goes, the handful of templates
|
||||
// that carry definition text are unwrapped and every other template is
|
||||
// dropped whole (its name counted in dropped when non-nil). The result is
|
||||
// trimmed and whitespace-collapsed; a result with no letter or digit is empty.
|
||||
func stripWikitext(s string, dropped map[string]int) string {
|
||||
s = htmlComment.ReplaceAllString(s, "")
|
||||
s = refElement.ReplaceAllString(s, "")
|
||||
s = anyTag.ReplaceAllString(s, "")
|
||||
s = stripTemplates(s, dropped)
|
||||
s = stripLinks(s)
|
||||
s = strings.ReplaceAll(s, "'''", "")
|
||||
s = strings.ReplaceAll(s, "''", "")
|
||||
s = html.UnescapeString(s)
|
||||
s = strings.Map(func(r rune) rune {
|
||||
switch {
|
||||
case unicode.IsControl(r), unicode.Is(unicode.Cf, r):
|
||||
// Cc and Cf: control characters, and format characters such as a
|
||||
// bidi override or a zero-width space, which could reshape the
|
||||
// rest of a rendered line.
|
||||
return -1
|
||||
case unicode.IsSpace(r):
|
||||
// Non-breaking and other Unicode spaces become plain ones so the
|
||||
// ASCII-only collapse below catches them.
|
||||
return ' '
|
||||
}
|
||||
return r
|
||||
}, s)
|
||||
s = strings.TrimSpace(spaces.ReplaceAllString(s, " "))
|
||||
// A stray space before sentence punctuation is what unwrapping a template
|
||||
// at the end of a clause leaves behind.
|
||||
for _, p := range []string{" .", " ,", " ;", " :", " )"} {
|
||||
s = strings.ReplaceAll(s, p, p[1:])
|
||||
}
|
||||
if !strings.ContainsFunc(s, func(r rune) bool { return unicode.IsLetter(r) || unicode.IsDigit(r) }) {
|
||||
return ""
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// stripTemplates replaces every outermost {{...}} with its plain-text
|
||||
// rendering. Nesting is tracked by depth, so a template inside a kept
|
||||
// template's parameter is rendered recursively and one inside a dropped
|
||||
// template goes with it.
|
||||
func stripTemplates(s string, dropped map[string]int) string {
|
||||
var out strings.Builder
|
||||
depth := 0
|
||||
start := 0
|
||||
for i := 0; i < len(s); i++ {
|
||||
switch {
|
||||
case strings.HasPrefix(s[i:], "{{"):
|
||||
if depth == 0 {
|
||||
start = i + 2
|
||||
}
|
||||
depth++
|
||||
i++
|
||||
case strings.HasPrefix(s[i:], "}}"):
|
||||
// A closer with nothing open is stray markup, not text.
|
||||
if depth > 0 {
|
||||
depth--
|
||||
if depth == 0 {
|
||||
out.WriteString(renderTemplate(s[start:i], dropped))
|
||||
}
|
||||
}
|
||||
i++
|
||||
case depth == 0:
|
||||
out.WriteByte(s[i])
|
||||
}
|
||||
}
|
||||
if depth > 0 && dropped != nil {
|
||||
// Unbalanced braces: whatever opened and never closed is dropped, as a
|
||||
// template would be, rather than leaking half a template into a gloss.
|
||||
dropped["(unclosed)"]++
|
||||
}
|
||||
return out.String()
|
||||
}
|
||||
|
||||
// renderTemplate maps one template body (the text between {{ and }}) to plain
|
||||
// text. body still contains any nested templates verbatim.
|
||||
//
|
||||
// The kept templates are the ones that carry definition text on this wiki:
|
||||
// context labels in four spellings, links in five, place descriptions,
|
||||
// non-gloss definitions and the two cross-reference templates a "dfn" section
|
||||
// is usually made of. Everything else is presentation or classification.
|
||||
func renderTemplate(body string, dropped map[string]int) string {
|
||||
parts := splitTemplate(body)
|
||||
name := strings.ToLower(strings.TrimSpace(parts[0]))
|
||||
// Positional parameters only; key=value ones are presentation hints.
|
||||
var params []string
|
||||
for _, p := range parts[1:] {
|
||||
if strings.Contains(p, "=") && !strings.Contains(p, "[[") && !strings.Contains(p, "{{") {
|
||||
continue
|
||||
}
|
||||
params = append(params, strings.TrimSpace(stripTemplates(strings.TrimSpace(p), dropped)))
|
||||
}
|
||||
// A leading language code is markup, whether it is ours or a neighbour's
|
||||
// pasted in: the label templates take it first, the link ones too. Only
|
||||
// when something follows it, though: {{q|con}} is a one-word qualifier,
|
||||
// not a language.
|
||||
dropLang := func(ps []string) []string {
|
||||
if len(ps) > 1 && isLangCode(ps[0]) {
|
||||
return ps[1:]
|
||||
}
|
||||
return ps
|
||||
}
|
||||
switch name {
|
||||
case "label", "lb", "nhãn", "context", "term", "gloss", "qualifier", "q":
|
||||
params = dropLang(params)
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return "(" + strings.Join(params, ", ") + ")"
|
||||
case "l", "vi-l", "w", "m", "link":
|
||||
params = dropLang(params)
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return params[len(params)-1]
|
||||
case "n-g", "non-gloss", "non-gloss definition":
|
||||
return strings.Join(params, " ")
|
||||
case "see-entry", "like-entry":
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return "Xem " + params[0]
|
||||
case "place":
|
||||
params = dropLang(params)
|
||||
for i, p := range params {
|
||||
// "c/Việt Nam" is a typed place: the type prefix is markup.
|
||||
if len(p) > 2 && p[1] == '/' && p[0] >= 'a' && p[0] <= 'z' {
|
||||
params[i] = p[2:]
|
||||
}
|
||||
}
|
||||
return strings.Join(params, ", ")
|
||||
}
|
||||
if dropped != nil {
|
||||
dropped[name]++
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// isLangCode reports whether a template parameter is a language code rather
|
||||
// than text: two or three lowercase ASCII letters, optionally with a
|
||||
// hyphenated variant such as "nan-hbl".
|
||||
func isLangCode(p string) bool {
|
||||
if len(p) < 2 || len(p) > 11 {
|
||||
return false
|
||||
}
|
||||
letters := 0
|
||||
for _, r := range p {
|
||||
switch {
|
||||
case r >= 'a' && r <= 'z':
|
||||
letters++
|
||||
case r == '-':
|
||||
if letters < 2 {
|
||||
return false
|
||||
}
|
||||
letters = 0
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return letters >= 2 && letters <= 3
|
||||
}
|
||||
|
||||
// splitTemplate splits a template body on | outside nested braces and
|
||||
// brackets, so a link or template inside a parameter is not cut in two.
|
||||
func splitTemplate(body string) []string {
|
||||
var parts []string
|
||||
depth := 0
|
||||
start := 0
|
||||
for i := 0; i < len(body); i++ {
|
||||
switch {
|
||||
case strings.HasPrefix(body[i:], "{{") || strings.HasPrefix(body[i:], "[["):
|
||||
depth++
|
||||
i++
|
||||
case strings.HasPrefix(body[i:], "}}") || strings.HasPrefix(body[i:], "]]"):
|
||||
if depth > 0 {
|
||||
depth--
|
||||
}
|
||||
i++
|
||||
case body[i] == '|' && depth == 0:
|
||||
parts = append(parts, body[start:i])
|
||||
start = i + 1
|
||||
}
|
||||
}
|
||||
return append(parts, body[start:])
|
||||
}
|
||||
|
||||
// stripLinks renders wiki links as their display text and drops category
|
||||
// links, which are classification rather than definition.
|
||||
func stripLinks(s string) string {
|
||||
var out strings.Builder
|
||||
for {
|
||||
open := strings.Index(s, "[[")
|
||||
if open < 0 {
|
||||
break
|
||||
}
|
||||
close := strings.Index(s[open:], "]]")
|
||||
if close < 0 {
|
||||
break
|
||||
}
|
||||
out.WriteString(s[:open])
|
||||
inner := s[open+2 : open+close]
|
||||
lower := strings.ToLower(inner)
|
||||
if !strings.HasPrefix(lower, "thể loại:") && !strings.HasPrefix(lower, "category:") {
|
||||
if bar := strings.LastIndex(inner, "|"); bar >= 0 {
|
||||
inner = inner[bar+1:]
|
||||
}
|
||||
out.WriteString(inner)
|
||||
}
|
||||
s = s[open+close+2:]
|
||||
}
|
||||
out.WriteString(s)
|
||||
s = out.String()
|
||||
|
||||
// External links: [http://… label] → label; a bare URL in brackets goes.
|
||||
out.Reset()
|
||||
for {
|
||||
open := strings.Index(s, "[http")
|
||||
if open < 0 {
|
||||
break
|
||||
}
|
||||
close := strings.Index(s[open:], "]")
|
||||
if close < 0 {
|
||||
break
|
||||
}
|
||||
out.WriteString(s[:open])
|
||||
inner := s[open+1 : open+close]
|
||||
if sp := strings.IndexByte(inner, ' '); sp >= 0 {
|
||||
out.WriteString(inner[sp+1:])
|
||||
}
|
||||
s = s[open+close+1:]
|
||||
}
|
||||
out.WriteString(s)
|
||||
return out.String()
|
||||
}
|
||||
|
||||
// capGloss cuts a definition longer than maxGlossRunes at the last space
|
||||
// before the limit and marks the cut with an ellipsis.
|
||||
func capGloss(s string) (string, bool) {
|
||||
if utf8.RuneCountInString(s) <= maxGlossRunes {
|
||||
return s, false
|
||||
}
|
||||
runes := []rune(s)
|
||||
head := string(runes[:maxGlossRunes-utf8.RuneCountInString(ellipsis)])
|
||||
if sp := strings.LastIndexByte(head, ' '); sp > 0 {
|
||||
head = head[:sp]
|
||||
}
|
||||
return strings.TrimRight(head, " ,;:") + ellipsis, true
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
func TestStripWikitext(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
want string
|
||||
}{
|
||||
{"links keep display text",
|
||||
"[[chỗ|Chỗ]] [[râm]] [[mát]], do [[trời]] có [[mây]] hoặc do không bị [[nắng]] [[chiếu]].",
|
||||
"Chỗ râm mát, do trời có mây hoặc do không bị nắng chiếu."},
|
||||
{"place template keeps its parameters and drops the type prefix",
|
||||
"{{place|vi|thủ đô|c/Việt Nam}}.",
|
||||
"thủ đô, Việt Nam."},
|
||||
{"label becomes a parenthesis",
|
||||
"{{label|vi|thuộc lịch sử}} Một [[tỉnh]] cũ của [[Việt Nam]] vào nửa cuối thế kỷ XIX.",
|
||||
"(thuộc lịch sử) Một tỉnh cũ của Việt Nam vào nửa cuối thế kỷ XIX."},
|
||||
{"Vietnamese label spellings",
|
||||
"{{nhãn|vi|tin học}} {{context|cũ}} {{term|Hóa học}} Cấu trúc.",
|
||||
"(tin học) (cũ) (Hóa học) Cấu trúc."},
|
||||
{"nested template inside a kept one",
|
||||
"{{label|vi|{{w|Hà Nội}}}} Thủ đô.",
|
||||
"(Hà Nội) Thủ đô."},
|
||||
{"unknown template dropped whole, nesting included",
|
||||
"{{rfdef|vi|{{w|x}}}}",
|
||||
""},
|
||||
{"definition that is only a cross-reference",
|
||||
"{{see-entry|bà la sát}}.",
|
||||
"Xem bà la sát."},
|
||||
{"non-gloss definition",
|
||||
"{{n-g|Trợ từ nhấn mạnh.}}",
|
||||
"Trợ từ nhấn mạnh."},
|
||||
{"link templates keep the last parameter",
|
||||
"{{l|vi|nói}}, {{l|vi|nói năng|nói năng (hiếm)}} và {{w|Việt Nam}}.",
|
||||
"nói, nói năng (hiếm) và Việt Nam."},
|
||||
{"ref mid-sentence and a lone ref",
|
||||
"Một loài [[cá]]<ref>Từ điển</ref> nước ngọt<ref name=\"a\" />.",
|
||||
"Một loài cá nước ngọt."},
|
||||
{"comment, bold, italic, entities",
|
||||
"'''Rất''' ''nhanh''<!-- todo --> và&mạnh.",
|
||||
"Rất nhanh và&mạnh."},
|
||||
{"category link dropped, external link keeps label",
|
||||
"Một [[thành phố]] [[Thể loại:Địa danh]] ([http://example.org trang web]).",
|
||||
"Một thành phố (trang web)."},
|
||||
{"named parameters are not text",
|
||||
"{{lb|vi|thơ ca|sort=x}} Câu.",
|
||||
"(thơ ca) Câu."},
|
||||
{"a lone short parameter is text, not a language code",
|
||||
"{{q|con}} Một loài vật, {{l|con}} là con.",
|
||||
"(con) Một loài vật, con là con."},
|
||||
{"format and bidi characters are dropped",
|
||||
"M\u200bột \u202enghĩa\u202c.",
|
||||
"Một nghĩa."},
|
||||
{"a stray closer is not text", "Một }} nghĩa.", "Một nghĩa."},
|
||||
{"only punctuation is empty", "(...).", ""},
|
||||
{"control characters and whitespace collapse", " Một từ \t hai ", "Một từ hai"},
|
||||
{"unclosed template does not leak", "Một {{label|vi|x từ.", "Một"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := stripWikitext(c.in, nil); got != c.want {
|
||||
t.Errorf("stripWikitext(%q)\n got %q\nwant %q", c.in, got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestStripWikitextCountsDroppedTemplates(t *testing.T) {
|
||||
dropped := make(map[string]int)
|
||||
stripWikitext("{{rfdef|vi}} {{RfDef|vi}} {{senseid|vi|x}}", dropped)
|
||||
if dropped["rfdef"] != 2 || dropped["senseid"] != 1 {
|
||||
t.Errorf("dropped = %v, want rfdef 2 (case-folded), senseid 1", dropped)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCapGlossCutsAtAWordBoundary(t *testing.T) {
|
||||
word := "từ "
|
||||
long := strings.Repeat(word, 120) // 360 runes
|
||||
got, cut := capGloss(long)
|
||||
if !cut {
|
||||
t.Fatal("a 360-rune gloss was not cut")
|
||||
}
|
||||
if n := utf8.RuneCountInString(got); n > maxGlossRunes {
|
||||
t.Errorf("cut gloss is %d runes, want at most %d", n, maxGlossRunes)
|
||||
}
|
||||
if !strings.HasSuffix(got, "từ"+ellipsis) {
|
||||
t.Errorf("cut gloss %q does not end on a whole word plus the ellipsis", got)
|
||||
}
|
||||
if short, cut := capGloss("ngắn"); cut || short != "ngắn" {
|
||||
t.Errorf("a short gloss was changed: %q %v", short, cut)
|
||||
}
|
||||
}
|
||||
|
||||
const legacyPage = `{{-vie-}}
|
||||
{{-pron-}}
|
||||
{{vie-pron|học sinh}}
|
||||
|
||||
{{-noun-}}
|
||||
# [[người|Người]] [[học]] ở [[nhà trường|trường]].
|
||||
#: ''Học sinh giỏi.''
|
||||
#{{label|vi|cũ}} [[môn đệ|Môn đệ]].
|
||||
|
||||
{{-verb-}}
|
||||
# [[đi học|Đi học]].
|
||||
|
||||
{{-trans-}}
|
||||
* {{eng}}: {{t|en|student}}
|
||||
|
||||
{{-eng-}}
|
||||
{{-noun-}}
|
||||
# Student, in English.
|
||||
`
|
||||
|
||||
const newPage = `== {{langname|vi}} ==
|
||||
=== {{ĐM|etym}} ===
|
||||
Hán-Việt.
|
||||
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
{{vi-pr-noun}}
|
||||
|
||||
# {{place|vi|thủ đô|c/Việt Nam}}.
|
||||
|
||||
=== {{section|v}} ===
|
||||
# [[đi|Đi]] về thủ đô.
|
||||
|
||||
=== {{ĐM|xyz}} ===
|
||||
# Một nghĩa dưới đề mục lạ.
|
||||
|
||||
== {{langname|en}} ==
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
# The capital of Vietnam.
|
||||
`
|
||||
|
||||
func TestVietnameseSectionLegacy(t *testing.T) {
|
||||
section, dialect, both, _ := vietnameseSection(legacyPage)
|
||||
if dialect != "legacy" || both {
|
||||
t.Fatalf("dialect = %q both = %v, want legacy false", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "Student") {
|
||||
t.Error("the English section leaked into the Vietnamese one")
|
||||
}
|
||||
if !strings.Contains(section, "{{-trans-}}") {
|
||||
t.Error("the translations heading, a section heading rather than a language, cut the section short")
|
||||
}
|
||||
}
|
||||
|
||||
func TestVietnameseSectionNew(t *testing.T) {
|
||||
section, dialect, both, _ := vietnameseSection(newPage)
|
||||
if dialect != "new" || both {
|
||||
t.Fatalf("dialect = %q both = %v, want new false", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "capital of Vietnam") {
|
||||
t.Error("the English section leaked into the Vietnamese one")
|
||||
}
|
||||
if !strings.Contains(section, "đề mục lạ") {
|
||||
t.Error("a level-3 heading ended the section; only a level-2 heading may")
|
||||
}
|
||||
}
|
||||
|
||||
func TestVietnameseSectionAbsentAndBoth(t *testing.T) {
|
||||
if _, dialect, _, _ := vietnameseSection("{{-eng-}}\n{{-noun-}}\n# Word."); dialect != "" {
|
||||
t.Errorf("an English-only page reported dialect %q", dialect)
|
||||
}
|
||||
mixed := "== {{langname|vi}} ==\n# Mới.\n" + legacyPage
|
||||
section, dialect, both, _ := vietnameseSection(mixed)
|
||||
if dialect != "new" || !both {
|
||||
t.Errorf("dialect = %q both = %v, want the first dialect in the text and both=true", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "Người học") {
|
||||
t.Error("the first section should end where the legacy page starts a language section")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsLegacy(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
section, _, _, _ := vietnameseSection(legacyPage)
|
||||
got := definitions(section, stats)
|
||||
want := []sense{
|
||||
{"danh từ", "Người học ở trường."},
|
||||
{"danh từ", "(cũ) Môn đệ."},
|
||||
{"động từ", "Đi học."},
|
||||
}
|
||||
assertSenses(t, got, want)
|
||||
if stats.pos["noun"] != 1 || stats.pos["verb"] != 1 {
|
||||
t.Errorf("pos tally = %v, want noun 1 verb 1", stats.pos)
|
||||
}
|
||||
if stats.pos["pron"] != 0 || stats.pos["trans"] != 0 {
|
||||
t.Errorf("pronunciation and translations were tallied as parts of speech: %v", stats.pos)
|
||||
}
|
||||
if stats.defsKept != 3 {
|
||||
t.Errorf("defsKept = %d, want 3", stats.defsKept)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsNewDialect(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
section, _, _, _ := vietnameseSection(newPage)
|
||||
got := definitions(section, stats)
|
||||
want := []sense{
|
||||
{"danh từ riêng", "thủ đô, Việt Nam."},
|
||||
{"động từ", "Đi về thủ đô."},
|
||||
{"", "Một nghĩa dưới đề mục lạ."},
|
||||
}
|
||||
assertSenses(t, got, want)
|
||||
if stats.unmappedPos["xyz"] != 1 {
|
||||
t.Errorf("unmapped headings = %v, want xyz 1", stats.unmappedPos)
|
||||
}
|
||||
if stats.unmappedPos["etym"] != 0 {
|
||||
t.Errorf("etymology counted as an unmapped part of speech: %v", stats.unmappedPos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsHeadwordLineAndWrittenOutHeading(t *testing.T) {
|
||||
section := "=== Danh từ ===\n# Một.\n{{vi-verb}}\n# Hai.\n{{vi-pron}}\n# Ba.\n=== Phát âm ===\n# Bốn."
|
||||
got := definitions(section, newSectionStats())
|
||||
want := []sense{{"danh từ", "Một."}, {"động từ", "Hai."}, {"động từ", "Ba."}, {"", "Bốn."}}
|
||||
assertSenses(t, got, want)
|
||||
}
|
||||
|
||||
func TestDefinitionsSkipsEmptyAndCaps(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
lines := []string{"{{-noun-}}", "# {{rfdef|vi}}", "#: not a definition", "#* nor this", "## nor this"}
|
||||
for i := 0; i < 7; i++ {
|
||||
lines = append(lines, "# Nghĩa số "+string(rune('a'+i))+".")
|
||||
}
|
||||
lines = append(lines, "# "+strings.Repeat("dài ", 80))
|
||||
got := definitions(strings.Join(lines, "\n"), stats)
|
||||
if len(got) != maxSenses {
|
||||
t.Fatalf("got %d senses, want the cap of %d", len(got), maxSenses)
|
||||
}
|
||||
if got[0].gloss != "Nghĩa số a." {
|
||||
t.Errorf("first sense = %q, want the first real definition after the empty one", got[0].gloss)
|
||||
}
|
||||
if stats.defsEmpty != 1 || stats.dropped["rfdef"] != 1 {
|
||||
t.Errorf("empty = %d dropped = %v, want 1 and rfdef 1", stats.defsEmpty, stats.dropped)
|
||||
}
|
||||
if stats.defsKept != 8 || stats.defsCut != 1 {
|
||||
t.Errorf("kept = %d cut = %d, want 8 and 1 (counted past the cap)", stats.defsKept, stats.defsCut)
|
||||
}
|
||||
}
|
||||
|
||||
func assertSenses(t *testing.T, got, want []sense) {
|
||||
t.Helper()
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %d senses %v, want %d %v", len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("sense %d = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A few pages run the section marker and the headings together on one line:
|
||||
// {{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}. The marker must still
|
||||
// open the section and the last heading on the line must still label it.
|
||||
func TestVietnameseSectionInlineHeadings(t *testing.T) {
|
||||
page := "{{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}\n\n'''Thượng Hải'''\n# Thành phố lớn nhất [[Trung Quốc]].\n{{-eng-}}{{-noun-}}\n# Shanghai."
|
||||
section, dialect, _, ender := vietnameseSection(page)
|
||||
if dialect != "legacy" || ender != "eng" {
|
||||
t.Fatalf("dialect = %q ender = %q, want legacy ended by eng", dialect, ender)
|
||||
}
|
||||
if strings.Contains(section, "Shanghai") {
|
||||
t.Error("the English section, opened on a shared line, leaked in")
|
||||
}
|
||||
got := definitions(section, newSectionStats())
|
||||
assertSenses(t, got, []sense{{"địa danh", "Thành phố lớn nhất Trung Quốc."}})
|
||||
}
|
||||
@@ -62,6 +62,7 @@ func run() error {
|
||||
"path", cfg.dbPath,
|
||||
"words", store.WordCount(),
|
||||
"aliases", store.AliasCount(),
|
||||
"meanings", store.MeaningCount(),
|
||||
"license", store.License(),
|
||||
)
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ import (
|
||||
"iter"
|
||||
"net/url"
|
||||
"os"
|
||||
"slices"
|
||||
"sort"
|
||||
"strconv"
|
||||
|
||||
@@ -32,6 +33,10 @@ import (
|
||||
// ErrNotFound is returned when a syllable has no entry in the dictionary.
|
||||
var ErrNotFound = errors.New("dictionary: syllable not found")
|
||||
|
||||
// requiredBuilderVersion is the builder whose meta contract this store reads;
|
||||
// it is named in the refusal of an older database.
|
||||
const requiredBuilderVersion = "5"
|
||||
|
||||
// wordInfo holds the two syllables the chain rule needs. Both ends are kept:
|
||||
// canonicalization can move either one, so the engine must never re-derive
|
||||
// them from what the player typed.
|
||||
@@ -40,6 +45,15 @@ type wordInfo struct {
|
||||
last string
|
||||
}
|
||||
|
||||
// Sense is one definition of a word as Wiktionary gives it: the Vietnamese
|
||||
// part-of-speech label of the heading it sat under ("danh từ"), empty when the
|
||||
// builder did not know the heading, and the definition as plain text. Neither
|
||||
// is markup; the client renders both as text.
|
||||
type Sense struct {
|
||||
Pos string
|
||||
Gloss string
|
||||
}
|
||||
|
||||
// Store answers word and syllable queries against the derived dictionary.
|
||||
//
|
||||
// Every field is written once during Open and only read afterwards, and no
|
||||
@@ -54,6 +68,10 @@ type Store struct {
|
||||
// openers holds words whose last syllable has at least one continuation,
|
||||
// sorted by that count descending so an eligible set is always a prefix.
|
||||
openers []opener
|
||||
// meanings holds each word's senses in page order. A few megabytes of text
|
||||
// for the corpus; a per-move query would be a second code path for nothing.
|
||||
meanings map[string][]Sense
|
||||
meaningCount int
|
||||
|
||||
license string
|
||||
}
|
||||
@@ -87,11 +105,12 @@ func Open(path string) (*Store, error) {
|
||||
aliases: make(map[string]string),
|
||||
byFirst: make(map[string][]string),
|
||||
outDegree: make(map[string]int),
|
||||
meanings: make(map[string][]Sense),
|
||||
}
|
||||
|
||||
// Reading meta first also rejects an unrelated database before any bulk
|
||||
// loading happens.
|
||||
declaredWords, err := s.loadMeta(db)
|
||||
declaredWords, declaredMeanings, err := s.loadMeta(db)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -106,7 +125,10 @@ func Open(path string) (*Store, error) {
|
||||
if err := s.loadAliases(db); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := s.validate(declaredWords); err != nil {
|
||||
if err := s.loadMeanings(db); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := s.validate(declaredWords, declaredMeanings); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -125,22 +147,37 @@ func dsn(path string) string {
|
||||
return u.String()
|
||||
}
|
||||
|
||||
func (s *Store) loadMeta(db *sql.DB) (declaredWords int, err error) {
|
||||
func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err error) {
|
||||
// The data is CC BY-SA 4.0 and its provenance travels with it.
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&s.license); err != nil {
|
||||
return 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
|
||||
return 0, 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
|
||||
}
|
||||
|
||||
count := func(key string) (int, error) {
|
||||
var raw string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'word_count'`).Scan(&raw); err != nil {
|
||||
return 0, fmt.Errorf("read dictionary word_count: %w", err)
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
// A database from before the key existed: the fix is a rebuild,
|
||||
// so say so rather than naming a missing row.
|
||||
return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it",
|
||||
key, requiredBuilderVersion)
|
||||
}
|
||||
declaredWords, err = strconv.Atoi(raw)
|
||||
return 0, fmt.Errorf("read dictionary %s: %w", key, err)
|
||||
}
|
||||
n, err := strconv.Atoi(raw)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("dictionary word_count %q is not a number: %w", raw, err)
|
||||
return 0, fmt.Errorf("dictionary %s %q is not a number: %w", key, raw, err)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
if declaredWords, err = count("word_count"); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
if declaredMeanings, err = count("meaning_count"); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
|
||||
return declaredWords, nil
|
||||
return declaredWords, declaredMeanings, nil
|
||||
}
|
||||
|
||||
func (s *Store) loadSyllables(db *sql.DB) error {
|
||||
@@ -204,13 +241,35 @@ func (s *Store) loadAliases(db *sql.DB) error {
|
||||
return rows.Err()
|
||||
}
|
||||
|
||||
func (s *Store) loadMeanings(db *sql.DB) error {
|
||||
// Ordered by (word, ord), the primary key, so each word's senses arrive in
|
||||
// page order and append in it.
|
||||
rows, err := db.Query(`SELECT word, pos, gloss FROM meanings ORDER BY word, ord`)
|
||||
if err != nil {
|
||||
return fmt.Errorf("load meanings: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
for rows.Next() {
|
||||
var word string
|
||||
var sense Sense
|
||||
if err := rows.Scan(&word, &sense.Pos, &sense.Gloss); err != nil {
|
||||
return fmt.Errorf("scan meaning: %w", err)
|
||||
}
|
||||
s.meanings[word] = append(s.meanings[word], sense)
|
||||
s.meaningCount++
|
||||
}
|
||||
|
||||
return rows.Err()
|
||||
}
|
||||
|
||||
// validate rejects a structurally valid but wrong dictionary.
|
||||
//
|
||||
// A truncated or empty database has the right schema and opens cleanly, and
|
||||
// the server would then start, reject every word a player types, and fail
|
||||
// every room creation. Checking the loaded rows against what the builder
|
||||
// recorded turns that into a startup failure.
|
||||
func (s *Store) validate(declaredWords int) error {
|
||||
func (s *Store) validate(declaredWords, declaredMeanings int) error {
|
||||
if len(s.words) != declaredWords {
|
||||
return fmt.Errorf("dictionary is incomplete: metadata declares %d words, loaded %d",
|
||||
declaredWords, len(s.words))
|
||||
@@ -218,6 +277,17 @@ func (s *Store) validate(declaredWords int) error {
|
||||
if len(s.words) == 0 {
|
||||
return errors.New("dictionary contains no words")
|
||||
}
|
||||
// A meanings table truncated on disk would otherwise be served silently
|
||||
// as a dictionary without meanings.
|
||||
if s.meaningCount != declaredMeanings {
|
||||
return fmt.Errorf("dictionary is incomplete: metadata declares %d meanings, loaded %d",
|
||||
declaredMeanings, s.meaningCount)
|
||||
}
|
||||
for word := range s.meanings {
|
||||
if _, ok := s.words[word]; !ok {
|
||||
return fmt.Errorf("dictionary is inconsistent: meaning for %q, which is not a word", word)
|
||||
}
|
||||
}
|
||||
|
||||
// A stale syllables table would tell the bot a syllable has continuations
|
||||
// that WordsStartingWith cannot supply.
|
||||
@@ -243,6 +313,20 @@ func (s *Store) WordCount() int { return len(s.words) }
|
||||
// AliasCount reports how many alternative spellings are accepted.
|
||||
func (s *Store) AliasCount() int { return len(s.aliases) }
|
||||
|
||||
// MeaningCount reports how many senses the dictionary holds across all words.
|
||||
func (s *Store) MeaningCount() int { return s.meaningCount }
|
||||
|
||||
// Meanings returns a canonical word's senses in page order, at most five, or
|
||||
// nil for a word with none. Resolve first: an alias has no senses of its own.
|
||||
// The slice is a copy, so a caller cannot reach dictionary state through it.
|
||||
func (s *Store) Meanings(word string) []Sense {
|
||||
senses := s.meanings[word]
|
||||
if len(senses) == 0 {
|
||||
return nil
|
||||
}
|
||||
return slices.Clone(senses)
|
||||
}
|
||||
|
||||
// License reports the licence the dictionary data is distributed under.
|
||||
// Callers are expected to state it at startup.
|
||||
func (s *Store) License() string { return s.license }
|
||||
|
||||
@@ -20,6 +20,7 @@ CREATE TABLE words (word TEXT PRIMARY KEY, first TEXT NOT NULL, last TEXT NOT NU
|
||||
CREATE INDEX idx_words_first ON words(first);
|
||||
CREATE TABLE syllables (syllable TEXT PRIMARY KEY, out_degree INTEGER NOT NULL) WITHOUT ROWID;
|
||||
CREATE TABLE aliases (variant TEXT PRIMARY KEY, canonical TEXT NOT NULL) WITHOUT ROWID;
|
||||
CREATE TABLE meanings (word TEXT NOT NULL, ord INTEGER NOT NULL, pos TEXT NOT NULL, gloss TEXT NOT NULL, PRIMARY KEY (word, ord)) WITHOUT ROWID;
|
||||
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
`
|
||||
|
||||
@@ -40,7 +41,7 @@ func fixtureAt(tb testing.TB, dir string) string {
|
||||
defer db.Close()
|
||||
|
||||
data := fixtureSchema + `
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7'),('meaning_count','3');
|
||||
INSERT INTO words VALUES
|
||||
('pháp luật','pháp','luật',2),
|
||||
('pháp lý','pháp','lý',2),
|
||||
@@ -55,6 +56,11 @@ INSERT INTO syllables VALUES
|
||||
('pháp',2),('luật',1),('lý',1),('lệ',0),('do',0),('vô',1),('điện',0),('công',1),('dầu',0),('ngữ',1);
|
||||
-- "pháp lí" drifts in the LAST syllable, "luâto lệ" in the FIRST.
|
||||
INSERT INTO aliases VALUES ('pháp lí','pháp lý'),('luâto lệ','luật lệ');
|
||||
-- Inserted out of order to prove the store sorts by ord, not by insertion.
|
||||
INSERT INTO meanings VALUES
|
||||
('pháp luật',1,'','Kỷ cương nói chung.'),
|
||||
('pháp luật',0,'danh từ','Hệ thống các quy tắc xử sự do nhà nước đặt ra.'),
|
||||
('ngữ pháp',0,'danh từ','Toàn bộ những quy tắc hoạt động của ngôn ngữ.');
|
||||
`
|
||||
if _, err := db.Exec(data); err != nil {
|
||||
tb.Fatal(err)
|
||||
@@ -106,7 +112,7 @@ func TestOpenWrongSchema(t *testing.T) {
|
||||
// every room creation.
|
||||
func TestOpenEmptyDictionary(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999');`)
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999'),('meaning_count','0');`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil {
|
||||
@@ -121,7 +127,7 @@ INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999')
|
||||
// syllable has continuations that cannot be supplied.
|
||||
func TestOpenInconsistentOutDegree(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
|
||||
|
||||
@@ -132,7 +138,7 @@ INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
|
||||
|
||||
func TestOpenOrphanAlias(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO aliases VALUES ('phap luat','không tồn tại');`)
|
||||
@@ -541,3 +547,73 @@ func BenchmarkRandomOpeningWord(b *testing.B) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMeaningsAreOrderedAndCopied(t *testing.T) {
|
||||
s := fixture(t)
|
||||
|
||||
got := s.Meanings("pháp luật")
|
||||
want := []Sense{
|
||||
{Pos: "danh từ", Gloss: "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{Pos: "", Gloss: "Kỷ cương nói chung."},
|
||||
}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Errorf("Meanings(pháp luật) = %v, want %v (ordered by ord, not insertion)", got, want)
|
||||
}
|
||||
// A caller writing into the slice must not reach the store.
|
||||
got[0].Gloss = "changed"
|
||||
if s.Meanings("pháp luật")[0].Gloss != want[0].Gloss {
|
||||
t.Error("Meanings handed out the store's own slice")
|
||||
}
|
||||
|
||||
if s.Meanings("pháp lý") != nil {
|
||||
t.Error("a word with no senses returned a non-nil slice")
|
||||
}
|
||||
// An alias is not a word: callers Resolve first.
|
||||
if s.Meanings("pháp lí") != nil {
|
||||
t.Error("an alias returned senses of its own")
|
||||
}
|
||||
if s.MeaningCount() != 3 {
|
||||
t.Errorf("MeaningCount = %d, want 3", s.MeaningCount())
|
||||
}
|
||||
}
|
||||
|
||||
// A meanings table truncated on disk must be refused at startup rather than
|
||||
// served silently as a dictionary without meanings.
|
||||
func TestOpenRefusesMismatchedMeaningCount(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','2');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO meanings VALUES ('pháp luật',0,'danh từ','Luật.');`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "meanings") {
|
||||
t.Fatalf("Open = %v, want a refusal naming the meanings count", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A database built before meanings existed opens cleanly and has every table
|
||||
// but one row. The refusal must say what to do, not which row is missing.
|
||||
func TestOpenRefusesOlderBuilderVersion(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "make dict") {
|
||||
t.Fatalf("Open = %v, want a refusal that says to rebuild", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenRefusesOrphanMeaning(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','1');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO meanings VALUES ('không tồn tại',0,'','Một nghĩa.');`)
|
||||
|
||||
if _, err := Open(path); err == nil {
|
||||
t.Fatal("Open succeeded with a meaning for a word that does not exist")
|
||||
}
|
||||
}
|
||||
Vendored
+128
-122
@@ -11,232 +11,238 @@
|
||||
# The graph is built around one hub syllable, "sinh". It is the only syllable
|
||||
# with enough continuations to be chosen as an opening, so every game starts on
|
||||
# a word ending in "sinh" and a scripted test always knows the first answer.
|
||||
#
|
||||
# A line is the word, then optional tab-separated meanings; a meaning is
|
||||
# `pos|gloss` or just `gloss`. Hand-written, kept short, and present on more
|
||||
# than half the list so the fixture database exercises the meanings path the
|
||||
# real one does. Every opening word and every "sinh" word has one, because
|
||||
# those are the words the browser suite reads a meaning from.
|
||||
|
||||
# --- openings: words ending in the hub syllable ---
|
||||
học sinh
|
||||
thí sinh
|
||||
vệ sinh
|
||||
phát sinh
|
||||
khai sinh
|
||||
tái sinh
|
||||
dân sinh
|
||||
nhân sinh
|
||||
ký sinh
|
||||
chúng sinh
|
||||
học sinh danh từ|Người học ở trường phổ thông.
|
||||
thí sinh danh từ|Người dự thi.
|
||||
vệ sinh danh từ|Sự giữ gìn sạch sẽ để phòng bệnh.
|
||||
phát sinh động từ|Nảy sinh, xuất hiện.
|
||||
khai sinh động từ|Đăng ký việc sinh ra của một người.
|
||||
tái sinh động từ|Sinh ra lần nữa; làm sống lại.
|
||||
dân sinh danh từ|Đời sống của nhân dân.
|
||||
nhân sinh danh từ|Cuộc sống con người.
|
||||
ký sinh động từ|Sống nhờ vào cơ thể sinh vật khác.
|
||||
chúng sinh danh từ|Mọi loài có sự sống, theo đạo Phật.
|
||||
|
||||
# --- the hub: words starting with "sinh" ---
|
||||
sinh viên
|
||||
sinh sản
|
||||
sinh hoạt
|
||||
sinh học
|
||||
sinh nhật
|
||||
sinh vật
|
||||
sinh tồn
|
||||
sinh thái
|
||||
sinh lý
|
||||
sinh kế
|
||||
sinh khí
|
||||
sinh mệnh
|
||||
sinh trưởng
|
||||
sinh động
|
||||
sinh sống
|
||||
sinh thành
|
||||
sinh lực
|
||||
sinh quán
|
||||
sinh sôi
|
||||
sinh nở
|
||||
sinh trắc
|
||||
sinh hóa
|
||||
sinh viên danh từ|Người học ở trường đại học, cao đẳng.
|
||||
sinh sản động từ|Tạo ra thế hệ sau để duy trì loài.
|
||||
sinh hoạt động từ|Sống và hoạt động hằng ngày. danh từ|Hoạt động tập thể có tổ chức.
|
||||
sinh học danh từ|Khoa học về sự sống.
|
||||
sinh nhật danh từ|Ngày kỷ niệm ngày sinh.
|
||||
sinh vật danh từ|Vật có sự sống.
|
||||
sinh tồn động từ|Sống và giữ được sự sống.
|
||||
sinh thái danh từ|Quan hệ giữa sinh vật và môi trường.
|
||||
sinh lý danh từ|Hoạt động bình thường của cơ thể sống.
|
||||
sinh kế danh từ|Cách kiếm sống.
|
||||
sinh khí danh từ|Sức sống, vẻ hoạt bát.
|
||||
sinh mệnh danh từ|Mạng sống.
|
||||
sinh trưởng động từ|Lớn lên về kích thước và khối lượng.
|
||||
sinh động tính từ|Có sức sống, gợi được hình ảnh rõ.
|
||||
sinh sống động từ|Sống ở một nơi.
|
||||
sinh thành động từ|Sinh ra và nuôi dạy.
|
||||
sinh lực danh từ|Sức sống của cơ thể.
|
||||
sinh quán danh từ|Nơi sinh.
|
||||
sinh sôi động từ|Sinh ra ngày càng nhiều.
|
||||
sinh nở động từ|Đẻ con.
|
||||
sinh trắc danh từ|Phép đo các đặc điểm sinh học của con người.
|
||||
sinh hóa danh từ|Hóa học của sự sống.
|
||||
|
||||
# --- continuations ---
|
||||
viên chức
|
||||
viên chức danh từ|Người làm việc trong cơ quan nhà nước.
|
||||
viên mãn
|
||||
chức năng
|
||||
chức năng danh từ|Tác dụng, vai trò của một bộ phận.
|
||||
chức vụ
|
||||
năng lực
|
||||
năng lực danh từ|Khả năng làm được việc.
|
||||
năng suất
|
||||
mãn nguyện
|
||||
nguyện vọng
|
||||
nguyện vọng danh từ|Điều mong muốn.
|
||||
vọng tưởng
|
||||
|
||||
sản xuất
|
||||
sản phẩm
|
||||
sản xuất động từ|Tạo ra của cải vật chất.
|
||||
sản phẩm danh từ|Vật do lao động tạo ra.
|
||||
sản lượng
|
||||
xuất bản
|
||||
xuất bản động từ|In và phát hành sách báo.
|
||||
xuất phát
|
||||
phẩm chất
|
||||
lượng giác
|
||||
bản đồ
|
||||
bản đồ danh từ|Hình vẽ thu nhỏ một vùng đất.
|
||||
bản sắc
|
||||
đồ thị
|
||||
sắc thái
|
||||
|
||||
hoạt động
|
||||
hoạt động động từ|Làm những việc có mục đích.
|
||||
hoạt bát
|
||||
động vật
|
||||
động cơ
|
||||
động vật danh từ|Sinh vật có cảm giác và tự vận động được.
|
||||
động cơ danh từ|Máy biến năng lượng thành chuyển động.
|
||||
động lực
|
||||
động đất
|
||||
cơ bản
|
||||
cơ hội
|
||||
hội nghị
|
||||
cơ bản tính từ|Là gốc, là nền tảng.
|
||||
cơ hội danh từ|Dịp thuận lợi.
|
||||
hội nghị danh từ|Cuộc họp bàn công việc.
|
||||
nghị luận
|
||||
đất nước
|
||||
nước ngoài
|
||||
đất nước danh từ|Lãnh thổ của một dân tộc; quốc gia.
|
||||
nước ngoài danh từ|Nước khác, ngoài nước mình.
|
||||
ngoài trời
|
||||
trời đất
|
||||
trời đất danh từ|Trời và đất; thế gian.
|
||||
|
||||
học tập
|
||||
học phí
|
||||
học tập động từ|Học và luyện tập để có hiểu biết.
|
||||
học phí danh từ|Tiền trả cho việc học.
|
||||
học hỏi
|
||||
học bổng
|
||||
tập trung
|
||||
tập trung động từ|Dồn vào một chỗ, một việc.
|
||||
tập thể
|
||||
trung tâm
|
||||
tâm hồn
|
||||
trung tâm danh từ|Điểm ở giữa; nơi tập trung hoạt động.
|
||||
tâm hồn danh từ|Ý nghĩ và tình cảm của con người.
|
||||
hồn nhiên
|
||||
nhiên liệu
|
||||
nhiên liệu danh từ|Chất đốt sinh năng lượng.
|
||||
liệu pháp
|
||||
pháp luật
|
||||
pháp lý
|
||||
luật sư
|
||||
pháp luật danh từ|Quy tắc xử sự do nhà nước đặt ra.
|
||||
pháp lý danh từ|Lý luận, căn cứ về pháp luật.
|
||||
luật sư danh từ|Người bảo vệ quyền lợi trước tòa theo luật.
|
||||
sư phạm
|
||||
phạm vi
|
||||
vi phạm
|
||||
phạm vi danh từ|Giới hạn của một hoạt động.
|
||||
vi phạm động từ|Làm trái quy định.
|
||||
|
||||
nhật ký
|
||||
nhật ký danh từ|Sổ ghi việc hằng ngày.
|
||||
nhật báo
|
||||
báo cáo
|
||||
báo cáo động từ|Trình bày kết quả công việc.
|
||||
cáo trạng
|
||||
trạng thái
|
||||
thái độ
|
||||
thái bình
|
||||
trạng thái danh từ|Tình trạng tồn tại của sự vật.
|
||||
thái độ danh từ|Cách nghĩ, cách nhìn thể hiện ra ngoài.
|
||||
thái bình tính từ|Yên ổn, không loạn lạc.
|
||||
độ cao
|
||||
cao nguyên
|
||||
bình minh
|
||||
bình minh danh từ|Lúc mặt trời mọc.
|
||||
minh bạch
|
||||
bạch tuộc
|
||||
|
||||
vật chất
|
||||
vật chất danh từ|Cái tồn tại khách quan ngoài ý thức.
|
||||
vật liệu
|
||||
chất lượng
|
||||
chất lượng danh từ|Cái tạo nên phẩm chất của sự vật.
|
||||
|
||||
tồn tại
|
||||
tồn tại động từ|Có thật, đang có.
|
||||
tồn kho
|
||||
tại chỗ
|
||||
kho tàng
|
||||
tàng hình
|
||||
hình ảnh
|
||||
ảnh hưởng
|
||||
hình ảnh danh từ|Hình người, vật hiện ra hoặc được ghi lại.
|
||||
ảnh hưởng động từ|Tác động đến.
|
||||
hưởng thụ
|
||||
|
||||
lý do
|
||||
lý thuyết
|
||||
lý luận
|
||||
lý tưởng
|
||||
lý do danh từ|Điều làm căn cứ để giải thích.
|
||||
lý thuyết danh từ|Hệ thống tư tưởng khái quát về một lĩnh vực.
|
||||
lý luận danh từ|Hệ thống luận điểm về một lĩnh vực.
|
||||
lý tưởng danh từ|Mục đích cao đẹp hướng tới.
|
||||
thuyết minh
|
||||
luận điểm
|
||||
tưởng tượng
|
||||
tưởng tượng động từ|Tạo ra trong trí hình ảnh chưa thấy.
|
||||
điểm danh
|
||||
danh sách
|
||||
danh sách danh từ|Bảng ghi tên theo thứ tự.
|
||||
sách vở
|
||||
tượng trưng
|
||||
|
||||
kế hoạch
|
||||
kế toán
|
||||
kế hoạch danh từ|Toàn bộ dự định làm việc theo trình tự.
|
||||
kế toán danh từ|Việc ghi chép, tính toán thu chi.
|
||||
kế tiếp
|
||||
hoạch định
|
||||
toán học
|
||||
tiếp tục
|
||||
toán học danh từ|Khoa học về số, hình và cấu trúc.
|
||||
tiếp tục động từ|Làm tiếp việc đang làm.
|
||||
định hướng
|
||||
tục ngữ
|
||||
hướng dẫn
|
||||
ngữ pháp
|
||||
tục ngữ danh từ|Câu ngắn gọn đúc kết kinh nghiệm dân gian.
|
||||
hướng dẫn động từ|Chỉ bảo cách làm.
|
||||
ngữ pháp danh từ|Quy tắc kết hợp từ thành câu.
|
||||
dẫn chứng
|
||||
chứng minh
|
||||
chứng minh động từ|Làm rõ là đúng bằng lý lẽ, bằng chứng.
|
||||
|
||||
khí hậu
|
||||
khí hậu danh từ|Thời tiết trung bình nhiều năm của một vùng.
|
||||
khí quyển
|
||||
khí thế
|
||||
hậu quả
|
||||
hậu quả danh từ|Kết quả không hay về sau.
|
||||
quyển sách
|
||||
thế giới
|
||||
thế giới danh từ|Trái đất và mọi thứ trên đó.
|
||||
quả cảm
|
||||
giới hạn
|
||||
giới hạn danh từ|Mức không thể vượt qua.
|
||||
hạn chế
|
||||
chế độ
|
||||
chế độ danh từ|Hệ thống tổ chức chính trị, xã hội.
|
||||
|
||||
mệnh lệnh
|
||||
mệnh lệnh danh từ|Lời sai bảo phải làm theo.
|
||||
mệnh đề
|
||||
lệnh cấm
|
||||
đề tài
|
||||
đề tài danh từ|Vấn đề được chọn để nghiên cứu.
|
||||
cấm vận
|
||||
tài liệu
|
||||
vận động
|
||||
tài liệu danh từ|Văn bản dùng để tra cứu.
|
||||
vận động động từ|Di chuyển, thay đổi vị trí.
|
||||
|
||||
trưởng thành
|
||||
trưởng phòng
|
||||
thành phố
|
||||
thành công
|
||||
thành lập
|
||||
thành viên
|
||||
thành phố danh từ|Đô thị lớn, đông dân.
|
||||
thành công động từ|Đạt được kết quả mong muốn.
|
||||
thành lập động từ|Lập nên một tổ chức.
|
||||
thành viên danh từ|Người thuộc một tổ chức.
|
||||
phòng ban
|
||||
phố cổ
|
||||
công việc
|
||||
công nghiệp
|
||||
lập trình
|
||||
công việc danh từ|Việc phải làm.
|
||||
công nghiệp danh từ|Ngành kinh tế sản xuất bằng máy móc.
|
||||
lập trình động từ|Viết chương trình cho máy tính.
|
||||
việc làm
|
||||
nghiệp vụ
|
||||
trình độ
|
||||
trình độ danh từ|Mức đạt được về kiến thức, kỹ năng.
|
||||
làm việc
|
||||
cổ điển
|
||||
cổ điển tính từ|Thuộc thời xưa và có giá trị lâu dài.
|
||||
điển hình
|
||||
|
||||
sống động
|
||||
lực lượng
|
||||
lực lượng danh từ|Sức mạnh của người hay vật.
|
||||
lực sĩ
|
||||
lượng tử
|
||||
sĩ quan
|
||||
tử tế
|
||||
quan hệ
|
||||
hệ thống
|
||||
thống nhất
|
||||
quan hệ danh từ|Sự gắn liền giữa hai hay nhiều sự vật.
|
||||
hệ thống danh từ|Tập hợp các yếu tố có quan hệ với nhau.
|
||||
thống nhất động từ|Hợp lại thành một khối.
|
||||
nhất trí
|
||||
trí tuệ
|
||||
trí tuệ danh từ|Khả năng nhận thức và suy nghĩ.
|
||||
tuệ giác
|
||||
|
||||
quán triệt
|
||||
triệt để
|
||||
triệt để tính từ|Đến tận cùng, không nửa vời.
|
||||
để dành
|
||||
dành dụm
|
||||
|
||||
sôi nổi
|
||||
nổi tiếng
|
||||
tiếng nói
|
||||
nói chuyện
|
||||
nổi tiếng tính từ|Được nhiều người biết đến.
|
||||
tiếng nói danh từ|Lời nói; ngôn ngữ.
|
||||
nói chuyện động từ|Trò chuyện với nhau.
|
||||
chuyện trò
|
||||
trò chơi
|
||||
trò chơi danh từ|Hoạt động để vui chơi, giải trí.
|
||||
chơi đùa
|
||||
|
||||
nở hoa
|
||||
hoa quả
|
||||
hóa học
|
||||
hoa quả danh từ|Các loại quả ăn được.
|
||||
hóa học danh từ|Khoa học về chất và biến đổi của chất.
|
||||
hóa đơn
|
||||
đơn giản
|
||||
đơn giản tính từ|Không phức tạp.
|
||||
giản dị
|
||||
dị thường
|
||||
thường xuyên
|
||||
thường xuyên tính từ|Đều đặn, liên tục.
|
||||
xuyên tạc
|
||||
|
||||
trắc nghiệm
|
||||
nghiệm thu
|
||||
thu nhập
|
||||
thu nhập danh từ|Tiền kiếm được trong một thời gian.
|
||||
nhập khẩu
|
||||
khẩu hiệu
|
||||
hiệu quả
|
||||
hiệu quả danh từ|Kết quả đích thực.
|
||||
|
||||
# --- a few longer words, so syllable bonuses and the 3+ badge have cases ---
|
||||
vô tuyến điện
|
||||
công nghiệp hóa
|
||||
vô tuyến điện danh từ|Kỹ thuật truyền tin bằng sóng điện từ.
|
||||
công nghiệp hóa Quá trình phát triển công nghiệp trong nền kinh tế.
|
||||
tổng hợp chất
|
||||
điện thoại di động
|
||||
điện thoại di động danh từ|Điện thoại cầm tay dùng sóng vô tuyến.
|
||||
@@ -12,11 +12,35 @@ import { fileURLToPath } from 'node:url';
|
||||
*/
|
||||
const listPath = fileURLToPath(new URL('../../testdata/fixture-words.txt', import.meta.url));
|
||||
|
||||
const words = readFileSync(listPath, 'utf8')
|
||||
// A line is the word, then optional tab-separated meanings; only the word is
|
||||
// part of the graph, and the meanings are kept so a test can say what the
|
||||
// chain should show for whichever word was drawn.
|
||||
const lines = readFileSync(listPath, 'utf8')
|
||||
.split('\n')
|
||||
.map((line) => line.trim())
|
||||
.filter((line) => line && !line.startsWith('#'));
|
||||
|
||||
const words = lines.map((line) => line.split('\t')[0].trim());
|
||||
|
||||
/** @type {Map<string, string>} word → its first sense as the chain renders it */
|
||||
const firstSense = new Map();
|
||||
for (const line of lines) {
|
||||
const [word, sense] = line.split('\t').map((cell) => cell.trim());
|
||||
if (!sense) continue;
|
||||
const [pos, gloss] = sense.includes('|') ? sense.split('|') : ['', sense];
|
||||
firstSense.set(word, pos ? `(${pos}) ${gloss}` : gloss);
|
||||
}
|
||||
|
||||
/**
|
||||
* The first sense of a fixture word, rendered as the chain renders it:
|
||||
* `(pos) gloss`, or the gloss alone. Undefined for a word without one.
|
||||
*
|
||||
* @param {string} word
|
||||
*/
|
||||
export function renderedSense(word) {
|
||||
return firstSense.get(word);
|
||||
}
|
||||
|
||||
/** @type {Map<string, string[]>} */
|
||||
const byFirstSyllable = new Map();
|
||||
for (const word of words) {
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// The upstream dictionary URL lives in three places: the Makefile, which
|
||||
// The upstream dump URL lives in three places: the Makefile, which
|
||||
// builds it for a developer, the Dockerfile, which builds it for the image, and
|
||||
// the builder, which stamps it into the database. They have to agree, or the
|
||||
// container ships a wordlist nobody tested against. The docs that quote the
|
||||
// container ships a dictionary nobody tested against. The docs that quote the
|
||||
// URL are held to the same copy.
|
||||
//
|
||||
// This lives in the JavaScript suite for no better reason than that it is the
|
||||
@@ -26,7 +26,7 @@ function pin(source, pattern, what) {
|
||||
return match?.[1].trim();
|
||||
}
|
||||
|
||||
describe('the upstream dictionary export', () => {
|
||||
describe('the upstream Wiktionary dump', () => {
|
||||
const makeUrl = pin(makefile, /DICT_URL\s*:?=\s*(\S+)/, 'DICT_URL in the Makefile');
|
||||
const dockerUrl = pin(dockerfile, /ARG DICT_URL=(\S+)/, 'DICT_URL in the Dockerfile');
|
||||
|
||||
@@ -34,10 +34,13 @@ describe('the upstream dictionary export', () => {
|
||||
expect(dockerUrl).toBe(makeUrl);
|
||||
});
|
||||
|
||||
it('is the Vietnamese-language file of the Vietnamese Wiktionary edition', () => {
|
||||
// The path carries a space, so it must stay percent-encoded or make and
|
||||
// sh will split it; and it must be the vi edition, not the English one.
|
||||
expect(makeUrl).toMatch(/^https:\/\/kaikki\.org\/viwiktionary\/Ti%E1%BA%BFng%20Vi%E1%BB%87t\/[^\s/]+\.jsonl$/);
|
||||
it('is the rolling pages-articles dump of the Vietnamese Wiktionary edition', () => {
|
||||
// The vi edition, not the English one; the current-revisions file, not
|
||||
// the full history; and `latest/`, which the owner chose over a dated
|
||||
// pin. Any of the three changing is a decision, not a typo.
|
||||
expect(makeUrl).toMatch(
|
||||
/^https:\/\/dumps\.wikimedia\.org\/viwiktionary\/latest\/viwiktionary-latest-pages-articles\.xml\.bz2$/
|
||||
);
|
||||
});
|
||||
|
||||
it('is the URL the builder stamps into the database', () => {
|
||||
@@ -45,10 +48,10 @@ describe('the upstream dictionary export', () => {
|
||||
// constant. The three copies must agree or the attribution record names
|
||||
// a file nobody downloaded.
|
||||
const builder = readFileSync(
|
||||
fileURLToPath(new URL('../../server/cmd/build-dictionary/kaikki_list.go', import.meta.url)),
|
||||
fileURLToPath(new URL('../../server/cmd/build-dictionary/dump.go', import.meta.url)),
|
||||
'utf8'
|
||||
);
|
||||
const builderUrl = pin(builder, /kaikkiSourceURL\s*=\s*"([^"]+)"/, 'kaikkiSourceURL in the builder');
|
||||
const builderUrl = pin(builder, /dumpSourceURL\s*=\s*"([^"]+)"/, 'dumpSourceURL in the builder');
|
||||
expect(builderUrl).toBe(makeUrl);
|
||||
});
|
||||
|
||||
|
||||
Reference in new issue
Block a user