feat(dict): build the corpus and word meanings from the Wikimedia viwiktionary dump

reader for both wikitext dialects; `meanings(word, ord, pos, gloss)` table; `meaning_count`/`words_with_meaning`/`source_pages` in meta, `source_rows` gone, builder_version 5; `--dump`/`--min-pages` replace `--kaikki`; attribution names the dump and the definition excerpts; 36,200 words, 96.9% with a meaning, every kaikki word kept.
This commit is contained in:
tiennm99 committed 2026-09-08 22:48:26 +07:00
1 parent 557de1af94
commit 80f216d56b
25 files changed
+2338 -757

No files matched your search

+270
View File
@@ -0,0 +1,270 @@
package main
import (
"bufio"
"compress/bzip2"
"crypto/sha256"
"encoding/hex"
"encoding/xml"
"errors"
"fmt"
"io"
"log"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// The upstream is the Wikimedia dump of Wiktionary tiếng Việt: every page's
// current wikitext, as one bzip2-compressed XML file regenerated monthly.
// `latest/` is a rolling pointer, fetched fresh for every build and not
// pinned, so the builder records the SHA-256 of the bytes it actually read and
// that hash is what identifies a build. Dated directories exist should
// reproducibility ever be wanted.
const dumpSourceURL = "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2"
// dumpPage is the part of a <page> element the builder reads. Everything
// else — contributor, timestamp, sha1 — is skipped by the decoder.
type dumpPage struct {
Title string `xml:"title"`
Ns int `xml:"ns"`
Redirect *struct {
Title string `xml:"title,attr"`
} `xml:"redirect"`
Revisions []struct {
Text string `xml:"text"`
} `xml:"revision"`
}
// dumpProvenance identifies the bytes a build was made from.
type dumpProvenance struct {
sha256 string
// pages is the number of pages with a Vietnamese section, redirects
// excluded: the count of entries the corpus was derived from.
pages int
fetchedAt time.Time
}
// dumpStats is what the build log reports about the dump beyond the reject
// tally: enough to see a month where a dialect vanished or a stripper rule
// started dropping everything.
type dumpStats struct {
pages int
ns0 int
redirects int
noVietnamese int
legacy int
newDialect int
bothDialects int
merged int // pages whose title normalized to a word already seen
// enders counts the {{-code-}} that closed each legacy section. Language
// codes are expected here; a heading code is one the maps are missing.
enders map[string]int
section *sectionStats
}
// readDump streams the dump once: hashes the compressed bytes, decodes one
// page at a time, hands every Vietnamese-section title to accept() and every
// definition line to the stripper.
func readDump(path string) (map[string]entry, map[string][]sense, map[rejectReason]int, *dumpStats, dumpProvenance, error) {
var prov dumpProvenance
stats := &dumpStats{section: newSectionStats(), enders: make(map[string]int)}
f, err := os.Open(path)
if err != nil {
return nil, nil, nil, stats, prov, fmt.Errorf("read dump: %w", err)
}
defer f.Close()
info, err := f.Stat()
if err != nil {
// A provenance row must be right or absent, never a plausible zero.
return nil, nil, nil, stats, prov, fmt.Errorf("stat dump: %w", err)
}
prov.fetchedAt = info.ModTime().UTC()
hash := sha256.New()
compressed := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
if magic, err := compressed.Peek(3); err != nil || string(magic) != "BZh" {
return nil, nil, nil, stats, prov, fmt.Errorf("%s is not a bzip2 file (expected a BZh header)", path)
}
dec := xml.NewDecoder(bzip2.NewReader(compressed))
words := make(map[string]entry)
meanings := make(map[string][]sense)
rejects := make(map[rejectReason]int)
lastTitle := ""
for {
tok, err := dec.Token()
if err != nil {
if errors.Is(err, io.EOF) {
break
}
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
}
start, ok := tok.(xml.StartElement)
if !ok || start.Name.Local != "page" {
continue
}
var page dumpPage
if err := dec.DecodeElement(&page, &start); err != nil {
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
}
lastTitle = page.Title
stats.pages++
if page.Ns != 0 {
continue
}
stats.ns0++
if page.Redirect != nil {
// The target page is read on its own and lowercased by accept(),
// so a case-only redirect adds nothing and any other redirect is
// an alternative title the wiki itself does not define.
stats.redirects++
continue
}
if len(page.Revisions) == 0 {
return nil, nil, nil, stats, prov, fmt.Errorf("%s: page %q has no revision text", path, page.Title)
}
text := page.Revisions[len(page.Revisions)-1].Text
section, dialect, both, ender := vietnameseSection(text)
if dialect == "" {
stats.noVietnamese++
rejects[rejectNotVietnamese]++
continue
}
if both {
stats.bothDialects++
}
if dialect == "legacy" {
stats.legacy++
} else {
stats.newDialect++
}
prov.pages++
if ender != "" {
stats.enders[ender]++
}
word, syllables, reason, ok := accept(page.Title)
if !ok {
rejects[reason]++
continue
}
// After accept, so the definition counters describe words that land.
senses := definitions(section, stats.section)
if _, seen := words[word]; seen {
// Two pages whose titles normalize to one word (Việt Nam and
// việt nam): one entry, senses in page order, one cap.
stats.merged++
}
words[word] = entry{
word: word,
first: syllables[0],
last: syllables[len(syllables)-1],
syllables: len(syllables),
}
if len(senses) > 0 {
merged := append(meanings[word], senses...)
if len(merged) > maxSenses {
merged = merged[:maxSenses]
}
meanings[word] = merged
}
}
// The XML decoder stops at the root's close tag; the hash must cover the
// whole file, trailing bytes included.
if _, err := io.Copy(io.Discard, compressed); err != nil {
return nil, nil, nil, stats, prov, fmt.Errorf("%s: %w", path, err)
}
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
return words, meanings, rejects, stats, prov, nil
}
// dumpError names where a stream failed: the last page fully read and the
// decompressed offset, so a truncated download and a malformed page are told
// apart by the message alone.
func dumpError(path, lastTitle string, offset int64, err error) error {
where := "before the first page"
if lastTitle != "" {
where = fmt.Sprintf("after page %q", lastTitle)
}
if errors.Is(err, io.ErrUnexpectedEOF) || strings.Contains(err.Error(), "unexpected EOF") {
return fmt.Errorf("%s: stream ends %s (decompressed offset %d): truncated download? %w", path, where, offset, err)
}
return fmt.Errorf("%s: %s (decompressed offset %d): %w", path, where, offset, err)
}
// logDumpStats writes the build log lines that describe what the dump held.
func logDumpStats(stats *dumpStats) {
s := stats.section
logf := log.Printf
logf("pages %d, in the main namespace %d, redirects skipped %d, without a Vietnamese section %d",
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
logf("Vietnamese sections: legacy {{-vie-}} %d, new == {{langname|vi}} == %d, pages with both %d, titles merged %d",
stats.legacy, stats.newDialect, stats.bothDialects, stats.merged)
logf("parts of speech: %s", formatTally(s.pos, 0))
if len(s.unmappedPos) > 0 {
logf("headings without a label: %s", formatTally(s.unmappedPos, 20))
}
// Language codes belong here. A heading code in this list is one the maps
// do not know, and it has been cutting sections short.
logf("codes that ended a legacy section, commonest: %s", formatTally(stats.enders, 15))
logf("definitions kept %d (cut at %d characters: %d), dropped as empty after stripping %d",
s.defsKept, maxGlossRunes, s.defsCut, s.defsEmpty)
if len(s.dropped) > 0 {
logf("templates dropped whole, commonest: %s", formatTally(s.dropped, 10))
}
}
// formatTally renders counts on one log line, largest first, cut to the top
// n entries when n is positive.
func formatTally(counts map[string]int, n int) string {
type kv struct {
name string
count int
}
tally := make([]kv, 0, len(counts))
for name, count := range counts {
if name == "" {
name = "(none)"
}
tally = append(tally, kv{name, count})
}
sort.Slice(tally, func(i, j int) bool {
if tally[i].count != tally[j].count {
return tally[i].count > tally[j].count
}
return tally[i].name < tally[j].name
})
if n > 0 && len(tally) > n {
tally = tally[:n]
}
parts := make([]string, len(tally))
for i, t := range tally {
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
}
return strings.Join(parts, ", ")
}
// dumpSourceSpec describes a dump build for the meta table. With nothing
// pinned upstream, the hash and page count of the bytes read are the
// provenance.
func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec {
return sourceSpec{
table: "dump:" + filepath.Base(path),
url: dumpSourceURL,
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
extra: [][2]string{
{"source_sha256", prov.sha256},
{"source_pages", fmt.Sprint(prov.pages)},
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
},
}
}
+149
View File
@@ -0,0 +1,149 @@
package main
import (
"crypto/sha256"
"encoding/hex"
"os"
"path/filepath"
"strings"
"testing"
)
// miniDump is a twelve-page stand-in for the Wikimedia dump, committed
// compressed beside its readable source. Go has no bzip2 writer, so the .bz2
// is regenerated by hand: bzip2 -k9 testdata/mini-dump.xml.
const miniDump = "testdata/mini-dump.xml.bz2"
// miniDumpCut is the same XML cut mid-page and then compressed: a valid bzip2
// stream whose XML ends early, as distinct from a truncated download.
const miniDumpCut = "testdata/mini-dump-cut.xml.bz2"
func TestReadDumpKeepsVietnameseSections(t *testing.T) {
words, meanings, rejects, stats, prov, err := readDump(miniDump)
if err != nil {
t.Fatal(err)
}
var got []string
for w := range words {
got = append(got, w)
}
assertSameStrings(t, got, []string{"pháp luật", "hòa bình", "luật lệ", "ngôn ngữ", "ngữ pháp", "vô tuyến điện"})
if stats.pages != 12 || stats.ns0 != 11 || stats.redirects != 1 || stats.noVietnamese != 1 {
t.Errorf("pages %d ns0 %d redirects %d noVietnamese %d, want 12 11 1 1",
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
}
if stats.legacy != 7 || stats.newDialect != 2 || stats.bothDialects != 0 {
t.Errorf("legacy %d new %d both %d, want 7 2 0", stats.legacy, stats.newDialect, stats.bothDialects)
}
if prov.pages != 9 {
t.Errorf("source pages = %d, want 9 (Vietnamese sections, redirect excluded)", prov.pages)
}
if stats.merged != 1 {
t.Errorf("merged = %d, want 1 (Hòa Bình and hòa bình)", stats.merged)
}
if rejects[rejectNotVietnamese] != 1 || rejects[rejectTooShort] != 1 || rejects[rejectDigit] != 1 {
t.Errorf("rejects = %v, want one each of not-Vietnamese, too-short, digit", rejects)
}
assertSenses(t, meanings["pháp luật"], []sense{
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
{"động từ", "(hiếm) Xử theo luật."},
})
// Capitalized page first, lowercase page second: senses in page order,
// the new-dialect place first.
assertSenses(t, meanings["hòa bình"], []sense{
{"danh từ riêng", "tỉnh, Việt Nam."},
{"danh từ", "Tình trạng không có chiến tranh."},
{"tính từ", "Yên ổn."},
})
assertSenses(t, meanings["ngôn ngữ"], []sense{{"danh từ", "Hệ thống những âm, từ và quy tắc kết hợp chúng."}})
if _, has := meanings["luật lệ"]; has {
t.Error("a definition that is only an unknown template produced a sense")
}
if stats.section.dropped["rfdef"] != 1 || stats.section.defsEmpty != 1 {
t.Errorf("dropped = %v empty = %d, want rfdef 1 and 1", stats.section.dropped, stats.section.defsEmpty)
}
if stats.section.pos["noun"] == 0 || stats.section.pos["pr-noun"] != 2 || stats.section.pos["n"] != 1 {
t.Errorf("pos tally = %v", stats.section.pos)
}
}
func TestReadDumpHashesTheBytesItRead(t *testing.T) {
_, _, _, _, prov, err := readDump(miniDump)
if err != nil {
t.Fatal(err)
}
raw, err := os.ReadFile(miniDump)
if err != nil {
t.Fatal(err)
}
sum := sha256.Sum256(raw)
if prov.sha256 != hex.EncodeToString(sum[:]) {
t.Errorf("sha256 = %s, want %s (the whole file)", prov.sha256, hex.EncodeToString(sum[:]))
}
if prov.fetchedAt.IsZero() {
t.Error("fetchedAt is zero")
}
}
func TestReadDumpRejectsNonBzip2(t *testing.T) {
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
if err := os.WriteFile(path, []byte("<mediawiki></mediawiki>"), 0o644); err != nil {
t.Fatal(err)
}
_, _, _, _, _, err := readDump(path)
if err == nil || !strings.Contains(err.Error(), "not a bzip2 file") {
t.Fatalf("err = %v, want a message naming the missing bzip2 header", err)
}
}
func TestReadDumpRejectsTruncatedDownload(t *testing.T) {
raw, err := os.ReadFile(miniDump)
if err != nil {
t.Fatal(err)
}
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
if err := os.WriteFile(path, raw[:len(raw)/2], 0o644); err != nil {
t.Fatal(err)
}
_, _, _, _, _, err = readDump(path)
if err == nil {
t.Fatal("a half-downloaded dump was read without error")
}
if !strings.Contains(err.Error(), "truncated") {
t.Errorf("err = %v, want it to suggest a truncated download", err)
}
}
func TestReadDumpRejectsStreamEndingMidPage(t *testing.T) {
_, _, _, _, _, err := readDump(miniDumpCut)
if err == nil {
t.Fatal("an XML stream ending mid-page was read without error")
}
if !strings.Contains(err.Error(), `after page "hello world"`) {
t.Errorf("err = %v, want it to name the last page fully read", err)
}
}
func TestRunFailsBelowMinPages(t *testing.T) {
cfg := config{
dump: miniDump,
out: filepath.Join(t.TempDir(), "noitu.db"),
minWords: 1,
minPages: 20000,
}
err := run(cfg)
if err == nil || !strings.Contains(err.Error(), "pages have a Vietnamese section") {
t.Fatalf("err = %v, want the page floor named", err)
}
}
func TestFormatTally(t *testing.T) {
got := formatTally(map[string]int{"b": 2, "a": 2, "": 5, "c": 1}, 3)
if want := "(none) 5, a 2, b 2"; got != want {
t.Errorf("formatTally = %q, want %q", got, want)
}
}
-168
View File
@@ -1,168 +0,0 @@
package main
import (
"bufio"
"bytes"
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// The upstream is kaikki.org's wiktextract export of Wiktionary tiếng Việt:
// one JSON object per entry, refreshed from the monthly Wikimedia dump about
// once a week. The file is fetched fresh for every build and is not pinned —
// there is no archived snapshot to pin to — so the builder records the SHA-256
// of the bytes it actually read, and that hash is what identifies a build.
//
// The URL stays percent-encoded: the path has a space in it, and both make
// and sh would otherwise split it.
const kaikkiSourceURL = "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl"
// kaikkiRow is the part of a wiktextract entry the game cares about. Every
// other field — senses, translations, categories — is skipped by the decoder.
type kaikkiRow struct {
Word string `json:"word"`
Pos string `json:"pos"`
LangCode string `json:"lang_code"`
}
// kaikkiProvenance identifies the bytes a build was made from.
type kaikkiProvenance struct {
sha256 string
rows int
fetchedAt time.Time
}
// readKaikkiList streams the export, keeps Vietnamese-language entries and
// hands their word forms to accept(). Part of speech is tallied for the build
// log but never filters: the owner's decision that capitalization removes no
// word applies equally to the "name" tag.
//
// Lines are read with bufio.Reader rather than bufio.Scanner because a row
// carries every sense and translation of its entry and can run to hundreds of
// kilobytes; a scanner's fixed cap would be a guess that eventually fails.
func readKaikkiList(path string) (map[string]entry, map[rejectReason]int, map[string]int, kaikkiProvenance, error) {
var prov kaikkiProvenance
f, err := os.Open(path)
if err != nil {
return nil, nil, nil, prov, fmt.Errorf("read kaikki export: %w", err)
}
defer f.Close()
info, err := f.Stat()
if err != nil {
// A provenance row must be right or absent, never a plausible zero.
return nil, nil, nil, prov, fmt.Errorf("stat kaikki export: %w", err)
}
prov.fetchedAt = info.ModTime().UTC()
hash := sha256.New()
reader := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
words := make(map[string]entry)
rejects := make(map[rejectReason]int)
pos := make(map[string]int)
lineNo := 0
for {
line, err := reader.ReadBytes('\n')
if len(line) > 0 {
lineNo++
if trimmed := bytes.TrimSpace(line); len(trimmed) > 0 {
// A bare literal such as null would decode into an empty row
// and be miscounted as a foreign-language entry; only objects
// are entries.
if trimmed[0] != '{' {
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: not a JSON object", path, lineNo)
}
var row kaikkiRow
if err := json.Unmarshal(trimmed, &row); err != nil {
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: %w", path, lineNo, err)
}
prov.rows++
if row.LangCode != "vi" {
rejects[rejectNotVietnamese]++
} else {
pos[row.Pos]++
if word, syllables, reason, ok := accept(row.Word); !ok {
rejects[reason]++
} else {
words[word] = entry{
word: word,
first: syllables[0],
last: syllables[len(syllables)-1],
syllables: len(syllables),
}
}
}
}
}
if err != nil {
if errors.Is(err, io.EOF) {
break
}
// The failure is on the line being read: the one just counted if
// a partial line came back with the error, otherwise the next.
failed := lineNo + 1
if len(line) > 0 {
failed = lineNo
}
return nil, nil, nil, prov, fmt.Errorf("%s:%d: %w", path, failed, err)
}
}
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
return words, rejects, pos, prov, nil
}
// formatPosTally renders the part-of-speech counts on one log line, largest
// first, so the build log says what kind of entries the export held.
func formatPosTally(pos map[string]int) string {
type kv struct {
name string
count int
}
tally := make([]kv, 0, len(pos))
for name, count := range pos {
if name == "" {
name = "(none)"
}
tally = append(tally, kv{name, count})
}
sort.Slice(tally, func(i, j int) bool {
if tally[i].count != tally[j].count {
return tally[i].count > tally[j].count
}
return tally[i].name < tally[j].name
})
parts := make([]string, len(tally))
for i, t := range tally {
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
}
return strings.Join(parts, ", ")
}
// kaikkiSourceSpec describes a kaikki build for the meta table. With no commit
// or checksum pinned upstream, the hash and row count of the bytes read are the
// provenance.
func kaikkiSourceSpec(path string, prov kaikkiProvenance) sourceSpec {
return sourceSpec{
table: "kaikki:" + filepath.Base(path),
url: kaikkiSourceURL,
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
extra: [][2]string{
{"source_sha256", prov.sha256},
{"source_rows", fmt.Sprint(prov.rows)},
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
},
}
}
@@ -1,243 +0,0 @@
package main
import (
"crypto/sha256"
"encoding/hex"
"os"
"path/filepath"
"strings"
"testing"
)
// fixtureKaikki writes a miniature stand-in for the kaikki export: the same
// JSONL shape, a handful of rows.
func fixtureKaikki(t *testing.T, lines ...string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
if err := os.WriteFile(path, []byte(strings.Join(lines, "\n")+"\n"), 0o644); err != nil {
t.Fatal(err)
}
return path
}
func defaultKaikkiLines() []string {
return []string{
`{"word": "Hà Nội", "pos": "name", "lang_code": "vi", "senses": [{"glosses": ["thủ đô"]}]}`, // capitalized, name POS — kept, lowercased
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
`{"word": "học sinh", "pos": "verb", "lang_code": "vi"}`, // same word, second POS — kept once
`{"word": "student", "pos": "noun", "lang_code": "en"}`, // not Vietnamese-language — rejected and counted
`{"word": "pháp", "pos": "noun", "lang_code": "vi"}`, // single syllable — rejected downstream
``,
}
}
func TestKaikkiListKeepsVietnameseEntries(t *testing.T) {
words, rejects, pos, prov, err := readKaikkiList(fixtureKaikki(t, defaultKaikkiLines()...))
if err != nil {
t.Fatal(err)
}
var got []string
for w := range words {
got = append(got, w)
}
assertSameStrings(t, got, []string{"hà nội", "học sinh"})
if n := rejects[rejectNotVietnamese]; n != 1 {
t.Errorf("non-Vietnamese rejects = %d, want 1", n)
}
if n := rejects[rejectTooShort]; n != 1 {
t.Errorf("too-short rejects = %d, want 1 (pháp)", n)
}
if pos["noun"] != 2 || pos["verb"] != 1 || pos["name"] != 1 {
t.Errorf("pos tally = %v, want noun 2, verb 1, name 1 (en row excluded)", pos)
}
if prov.rows != 5 {
t.Errorf("rows = %d, want 5", prov.rows)
}
}
func TestKaikkiListHashesTheBytesItRead(t *testing.T) {
path := fixtureKaikki(t, defaultKaikkiLines()...)
_, _, _, prov, err := readKaikkiList(path)
if err != nil {
t.Fatal(err)
}
raw, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
sum := sha256.Sum256(raw)
if want := hex.EncodeToString(sum[:]); prov.sha256 != want {
t.Errorf("sha256 = %s, want %s", prov.sha256, want)
}
if prov.fetchedAt.IsZero() {
t.Error("fetchedAt is zero, want the file's modification time")
}
}
// fixtureKaikkiRaw writes exact bytes, for the shapes fixtureKaikki's trailing
// newline would hide.
func fixtureKaikkiRaw(t *testing.T, raw string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
if err := os.WriteFile(path, []byte(raw), 0o644); err != nil {
t.Fatal(err)
}
return path
}
func TestKaikkiListHandlesDownloadShapes(t *testing.T) {
cases := []struct {
name string
raw string
wantWords int
wantRows int
wantErr string
}{
{"final line without newline",
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh mì", "pos": "noun", "lang_code": "vi"}`,
2, 2, ""},
{"HTTP error page instead of JSONL", "<html><body>503</body></html>\n", 0, 0, ":1: malformed"},
{"cut mid-line", `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh`, 0, 0, ":2: malformed"},
{"bare JSON literal", "null\n", 0, 0, ":1: malformed"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
words, _, _, prov, err := readKaikkiList(fixtureKaikkiRaw(t, tc.raw))
if tc.wantErr != "" {
if err == nil || !strings.Contains(err.Error(), tc.wantErr) {
t.Fatalf("err = %v, want one containing %q", err, tc.wantErr)
}
return
}
if err != nil {
t.Fatal(err)
}
if len(words) != tc.wantWords || prov.rows != tc.wantRows {
t.Errorf("words=%d rows=%d, want %d/%d", len(words), prov.rows, tc.wantWords, tc.wantRows)
}
})
}
}
func TestKaikkiListNamesMalformedLine(t *testing.T) {
path := fixtureKaikki(t,
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
`{"word": "broken"`,
)
_, _, _, _, err := readKaikkiList(path)
if err == nil {
t.Fatal("malformed line was skipped, want error")
}
if !strings.Contains(err.Error(), ":2:") {
t.Errorf("error does not name line 2: %v", err)
}
}
func TestKaikkiListReadsLongLines(t *testing.T) {
// A real row carries every sense and translation and can exceed any
// scanner buffer; the reader must not have a line cap.
padding := strings.Repeat("x", 2<<20)
path := fixtureKaikki(t, `{"word": "học sinh", "pos": "noun", "lang_code": "vi", "note": "`+padding+`"}`)
words, _, _, _, err := readKaikkiList(path)
if err != nil {
t.Fatal(err)
}
if _, ok := words["học sinh"]; !ok {
t.Error("word on a 2 MB line was lost")
}
}
func TestFormatPosTally(t *testing.T) {
got := formatPosTally(map[string]int{"verb": 2, "noun": 5, "": 1})
if want := "noun 5, verb 2, (none) 1"; got != want {
t.Errorf("tally = %q, want %q", got, want)
}
}
func TestKaikkiBuildRecordsProvenance(t *testing.T) {
out := filepath.Join(t.TempDir(), "noitu.db")
path := fixtureKaikki(t, defaultKaikkiLines()...)
if err := run(config{kaikki: path, out: out, minWords: 1}); err != nil {
t.Fatalf("run: %v", err)
}
db := openOut(t, out)
raw, _ := os.ReadFile(path)
sum := sha256.Sum256(raw)
want := map[string]string{
"source_url": kaikkiSourceURL,
"source_sha256": hex.EncodeToString(sum[:]),
"source_rows": "5",
"source_license": "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
"word_count": "2",
}
for key, value := range want {
var got string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&got); err != nil {
t.Errorf("meta[%q] missing: %v", key, err)
continue
}
if got != value {
t.Errorf("meta[%q] = %q, want %q", key, got, value)
}
}
for _, gone := range []string{"source_commit", "sources_kept", "sources_excluded"} {
var got string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, gone).Scan(&got); err == nil {
t.Errorf("meta[%q] = %q, want absent", gone, got)
}
}
var fetched string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_fetched_at'`).Scan(&fetched); err != nil || fetched == "" {
t.Errorf("meta[source_fetched_at] missing or empty: %v", err)
}
}
func TestRunInputSelection(t *testing.T) {
kaikki := fixtureKaikki(t, defaultKaikkiLines()...)
words := fixtureKaikki(t, "học sinh")
out := filepath.Join(t.TempDir(), "noitu.db")
cases := []struct {
name string
cfg config
wantErr string
}{
{"both inputs", config{kaikki: kaikki, words: words, out: out, minWords: 1}, "mutually exclusive"},
{"neither input", config{out: out, minWords: 1}, "no input given"},
{"missing kaikki file", config{kaikki: filepath.Join(t.TempDir(), "absent.jsonl"), out: out, minWords: 1}, "make fetch-dict"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
err := run(tc.cfg)
if err == nil {
t.Fatal("run succeeded, want error")
}
if !strings.Contains(err.Error(), tc.wantErr) {
t.Errorf("error %q does not mention %q", err, tc.wantErr)
}
})
}
}
// A fixture is hand-written data; its database must not claim the upstream's
// licence, because the server logs whatever the meta table says.
func TestWordListBuildRecordsNoUpstreamLicense(t *testing.T) {
out := filepath.Join(t.TempDir(), "noitu.db")
if err := run(config{words: fixtureKaikki(t, "học sinh", "bánh mì"), out: out, minWords: 1}); err != nil {
t.Fatalf("run: %v", err)
}
var license, url string
db := openOut(t, out)
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
t.Fatal(err)
}
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_url'`).Scan(&url); err != nil {
t.Fatal(err)
}
if strings.Contains(license, "CC BY-SA") || url != "" {
t.Errorf("fixture build claims upstream provenance: license=%q url=%q", license, url)
}
}
+157 -45
View File
@@ -1,19 +1,20 @@
// Command build-dictionary derives the game's wordlist from kaikki.org's
// wiktextract export of Wiktionary tiếng Việt.
// Command build-dictionary derives the game's wordlist and word meanings from
// the Wikimedia dump of Wiktionary tiếng Việt.
//
// The upstream is a ~62 MB JSONL file: one entry per line with its senses,
// translations and part of speech. The game needs only Vietnamese word forms
// of at least two syllables, indexed by first and last syllable. This tool
// performs that reduction and records provenance in a meta table — including
// the SHA-256 of the file it read, since the upstream is fetched fresh for
// every build rather than pinned.
// The upstream is a ~61 MB bzip2-compressed XML file: every page of the wiki
// with its current wikitext, regenerated monthly. The game needs the
// Vietnamese word forms of at least two syllables, indexed by first and last
// syllable, and the plain text of each word's definitions. This tool performs
// that reduction and records provenance in a meta table — including the
// SHA-256 of the file it read, since the upstream is fetched fresh for every
// build rather than pinned.
//
// The derived database is a modified version of CC BY-SA 4.0 licensed data.
// See data/ATTRIBUTION.md.
//
// Usage:
//
// go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db
// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
package main
import (
@@ -27,32 +28,45 @@ import (
"sort"
"strings"
"time"
"unicode/utf8"
_ "modernc.org/sqlite"
)
// builderVer changes whenever the meta table's contract does, so two databases
// with different provenance rows never claim the same builder.
const builderVer = "4"
const builderVer = "5"
// minMeaningCoverage is the share of words a dump build must carry a meaning
// for. The 2026-09-01 dump measured well above it; the floor exists to catch a
// stripper or section scanner that suddenly returns nothing, not to demand
// quality. Fixture builds are exempt: their meanings are hand-written.
const minMeaningCoverage = 0.6
type config struct {
// kaikki is the corpus: the upstream wiktextract JSONL export.
kaikki string
// words is an alternative source: a plain list, one word per line, used to
// build a small fixture database without the upstream download.
// dump is the corpus: the Wikimedia pages-articles export.
dump string
// words is an alternative source: a plain list, one word per line with an
// optional tab-separated meaning column, used to build a small fixture
// database without the upstream download.
words string
out string
minWords int
// minPages is the floor on pages with a Vietnamese section. Distinct from
// minWords so a scanner that silently misses a dialect is caught before
// the word floor is.
minPages int
}
func main() {
log.SetFlags(0)
var cfg config
flag.StringVar(&cfg.kaikki, "kaikki", "", "upstream kaikki.org wiktextract JSONL export to read")
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the upstream export (one word per line, # comments)")
flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read")
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)")
flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write")
flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering")
flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section")
flag.Parse()
if err := run(cfg); err != nil {
@@ -64,40 +78,47 @@ func run(cfg config) error {
// Exactly one input. Picking silently between two would let a stray flag
// ship a corpus nobody meant to build.
switch {
case cfg.kaikki == "" && cfg.words == "":
return errors.New("no input given: pass --kaikki (the corpus) or --words (a plain list)")
case cfg.kaikki != "" && cfg.words != "":
return errors.New("--kaikki and --words are mutually exclusive")
case cfg.kaikki != "":
return runFromKaikkiList(cfg)
case cfg.dump == "" && cfg.words == "":
return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)")
case cfg.dump != "" && cfg.words != "":
return errors.New("--dump and --words are mutually exclusive")
case cfg.dump != "":
return runFromDump(cfg)
default:
return runFromWordList(cfg)
}
}
// runFromKaikkiList derives the database from the kaikki.org export, keeping
// Vietnamese-language entries and recording the hash of the bytes it read.
func runFromKaikkiList(cfg config) error {
if _, err := os.Stat(cfg.kaikki); err != nil {
return fmt.Errorf("kaikki export not found at %s — run 'make fetch-dict' first: %w", cfg.kaikki, err)
// runFromDump derives the database from the Wikimedia dump, keeping every
// page with a Vietnamese section and recording the hash of the bytes it read.
func runFromDump(cfg config) error {
if _, err := os.Stat(cfg.dump); err != nil {
return fmt.Errorf("dump not found at %s — run 'make fetch-dict' first: %w", cfg.dump, err)
}
words, rejects, pos, prov, err := readKaikkiList(cfg.kaikki)
started := time.Now()
words, meanings, rejects, stats, prov, err := readDump(cfg.dump)
if err != nil {
return err
}
logDumpStats(stats)
logRejects(rejects)
log.Printf("parts of speech: %s", formatPosTally(pos))
log.Printf("accepted %d distinct words from %s (%d rows, sha256 %s)", len(words), cfg.kaikki, prov.rows, prov.sha256)
if prov.pages < cfg.minPages {
return fmt.Errorf("only %d pages have a Vietnamese section, expected at least %d — "+
"the dump's markup may have changed", prov.pages, cfg.minPages)
}
log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s",
len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second))
return finish(cfg, words, kaikkiSourceSpec(cfg.kaikki, prov))
return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true)
}
// finish is the tail every input mode shares: the size floor, alias
// generation, the atomic write and the re-read verification. Keeping it in one
// place is what stops a fixture from drifting into a different shape from the
// database production loads.
func finish(cfg config, words map[string]entry, src sourceSpec) error {
// database production loads. requireCoverage applies the meaning-coverage
// floor, which only a corpus build can be held to.
func finish(cfg config, words map[string]entry, meanings map[string][]sense, src sourceSpec, requireCoverage bool) error {
if len(words) < cfg.minWords {
return fmt.Errorf("only %d words survived filtering, expected at least %d — "+
"the source content may have changed", len(words), cfg.minWords)
@@ -106,10 +127,10 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
aliases, collisions := buildAliases(words)
log.Printf("generated %d spelling aliases (%d skipped as ambiguous or already real words)", len(aliases), collisions)
if err := write(cfg.out, words, aliases, src); err != nil {
if err := write(cfg.out, words, meanings, aliases, src); err != nil {
return err
}
if err := verify(cfg.out, cfg.minWords); err != nil {
if err := verify(cfg.out, cfg.minWords, requireCoverage); err != nil {
return fmt.Errorf("output failed verification: %w", err)
}
@@ -118,13 +139,17 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
}
// runFromWordList derives a database from a plain list of words instead of the
// upstream export.
// dump.
//
// It exists so tests and CI have a real dictionary to play against without the
// upstream download. The filtering, alias generation, writing and verification
// below are the same functions the real build uses — only the source of the
// raw strings differs — so a fixture cannot drift into being shaped
// differently from what production loads.
//
// A line is `word`, or `word<TAB>sense<TAB>sense…` where a sense is
// `pos|gloss` or just `gloss`. The pipe never survives the stripper, so it is
// a safe separator for hand-written meanings.
func runFromWordList(cfg config) error {
raw, err := os.ReadFile(cfg.words)
if err != nil {
@@ -132,6 +157,7 @@ func runFromWordList(cfg config) error {
}
words := make(map[string]entry)
meanings := make(map[string][]sense)
rejects := make(map[rejectReason]int)
for line := range strings.Lines(string(raw)) {
@@ -139,7 +165,8 @@ func runFromWordList(cfg config) error {
if line == "" || strings.HasPrefix(line, "#") {
continue
}
word, syllables, reason, ok := accept(line)
cells := strings.Split(line, "\t")
word, syllables, reason, ok := accept(cells[0])
if !ok {
rejects[reason]++
continue
@@ -150,26 +177,54 @@ func runFromWordList(cfg config) error {
last: syllables[len(syllables)-1],
syllables: len(syllables),
}
if senses := parseSenses(cells[1:]); len(senses) > 0 {
meanings[word] = senses
}
}
logRejects(rejects)
log.Printf("accepted %d distinct words from %s", len(words), cfg.words)
log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words)
// The source spec is what lands in the meta table. Naming the list rather
// than a table makes it obvious in the output which build produced a given
// database — and a hand-written list carries no upstream licence, so the
// fixture must not claim one.
return finish(cfg, words, sourceSpec{
return finish(cfg, words, meanings, sourceSpec{
table: "wordlist:" + filepath.Base(cfg.words),
license: "none: hand-written fixture wordlist, no upstream data",
attribution: "Fixture written by this project; no third-party attribution applies.",
})
}, false)
}
// parseSenses reads the tab-separated meaning cells of a fixture line. A cell
// is `pos|gloss` or a bare gloss; empty cells are skipped and the cap applies
// as it does to the dump.
func parseSenses(cells []string) []sense {
var senses []sense
for _, cell := range cells {
cell = strings.TrimSpace(cell)
if cell == "" {
continue
}
s := sense{gloss: cell}
if pos, gloss, ok := strings.Cut(cell, "|"); ok {
s = sense{pos: strings.TrimSpace(pos), gloss: strings.TrimSpace(gloss)}
}
if s.gloss == "" {
continue
}
s.gloss, _ = capGloss(s.gloss)
if len(senses) < maxSenses {
senses = append(senses, s)
}
}
return senses
}
// verify re-opens the finished database and re-checks the invariants the game
// depends on. The in-memory checks above can only prove what the builder
// intended; this proves what actually landed on disk.
func verify(path string, minWords int) error {
func verify(path string, minWords int, requireCoverage bool) error {
db, err := sql.Open("sqlite", "file:"+path+"?mode=ro")
if err != nil {
return err
@@ -193,6 +248,15 @@ func verify(path string, minWords int) error {
{"words whose first syllable is missing from the syllables table",
`SELECT COUNT(*) FROM words w LEFT JOIN syllables s ON s.syllable = w.first WHERE s.syllable IS NULL`,
func(n int) bool { return n == 0 }},
{"meanings whose word is missing from the words table",
`SELECT COUNT(*) FROM meanings m LEFT JOIN words w ON w.word = m.word WHERE w.word IS NULL`,
func(n int) bool { return n == 0 }},
{"meanings with an empty gloss", `SELECT COUNT(*) FROM meanings WHERE gloss = ''`, func(n int) bool { return n == 0 }},
{"meanings over the length cap", fmt.Sprintf(`SELECT COUNT(*) FROM meanings WHERE LENGTH(gloss) > %d`, maxGlossRunes),
func(n int) bool { return n == 0 }},
{"words with more meanings than the cap",
fmt.Sprintf(`SELECT COUNT(*) FROM (SELECT word FROM meanings GROUP BY word HAVING COUNT(*) > %d)`, maxSenses),
func(n int) bool { return n == 0 }},
}
for _, c := range checks {
@@ -205,6 +269,20 @@ func verify(path string, minWords int) error {
}
}
if requireCoverage {
var wordCount, withMeaning int
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&wordCount); err != nil {
return err
}
if err := db.QueryRow(`SELECT COUNT(DISTINCT word) FROM meanings`).Scan(&withMeaning); err != nil {
return err
}
if float64(withMeaning) < minMeaningCoverage*float64(wordCount) {
return fmt.Errorf("only %d of %d words have a meaning, expected at least %.0f%% — "+
"the dump's definition markup may have changed", withMeaning, wordCount, minMeaningCoverage*100)
}
}
return nil
}
@@ -218,7 +296,7 @@ type entry struct {
// sourceSpec is what the meta table records about where the words came from.
type sourceSpec struct {
// table names the input: "kaikki:<file>" for the corpus, "wordlist:<file>"
// table names the input: "dump:<file>" for the corpus, "wordlist:<file>"
// for a fixture, so the output says which build produced it.
table string
// url is the upstream artifact; empty for fixture builds.
@@ -281,7 +359,7 @@ func buildAliases(words map[string]entry) (map[string]string, int) {
// partway through -- a full disk, an interrupt -- leaves an empty but
// syntactically valid database where a good one used to be, which the server
// would happily open and find no words in.
func write(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
func write(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
tmp := path + ".tmp"
if err := os.Remove(tmp); err != nil && !errors.Is(err, os.ErrNotExist) {
return fmt.Errorf("remove stale temp file: %w", err)
@@ -294,7 +372,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
}
}()
if err := writeTo(tmp, words, aliases, src); err != nil {
if err := writeTo(tmp, words, meanings, aliases, src); err != nil {
return err
}
@@ -311,7 +389,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
return nil
}
func writeTo(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
func writeTo(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
db, err := sql.Open("sqlite", "file:"+path)
if err != nil {
return fmt.Errorf("create output: %w", err)
@@ -337,6 +415,17 @@ CREATE TABLE aliases (
canonical TEXT NOT NULL
) WITHOUT ROWID;
-- One row per sense, in page order. pos is the Vietnamese part-of-speech
-- label of the heading the definition sat under, '' when the heading was one
-- the builder does not know. No foreign key pragma: verify() checks the join.
CREATE TABLE meanings (
word TEXT NOT NULL,
ord INTEGER NOT NULL,
pos TEXT NOT NULL,
gloss TEXT NOT NULL,
PRIMARY KEY (word, ord)
) WITHOUT ROWID;
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
`
if _, err := db.Exec(schema); err != nil {
@@ -390,6 +479,27 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
}
}
insertMeaning, err := tx.Prepare(`INSERT INTO meanings (word, ord, pos, gloss) VALUES (?, ?, ?, ?)`)
if err != nil {
return err
}
defer insertMeaning.Close()
meaningCount := 0
for word, senses := range meanings {
if _, isWord := words[word]; !isWord {
return fmt.Errorf("meaning for %q, which is not a word", word)
}
for ord, s := range senses {
if s.gloss == "" || utf8.RuneCountInString(s.gloss) > maxGlossRunes {
return fmt.Errorf("meaning %d of %q is empty or over the cap", ord, word)
}
if _, err := insertMeaning.Exec(word, ord, s.pos, s.gloss); err != nil {
return fmt.Errorf("insert meaning %d of %q: %w", ord, word, err)
}
meaningCount++
}
}
insertMeta, err := tx.Prepare(`INSERT INTO meta (key, value) VALUES (?, ?)`)
if err != nil {
return err
@@ -403,6 +513,8 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
{"builder_version", builderVer},
{"word_count", fmt.Sprint(len(words))},
{"alias_count", fmt.Sprint(len(aliases))},
{"meaning_count", fmt.Sprint(meaningCount)},
{"words_with_meaning", fmt.Sprint(len(meanings))},
{"source_table", src.table},
}
meta = append(meta, src.extra...)
+221 -59
View File
@@ -6,47 +6,25 @@ import (
"errors"
"os"
"path/filepath"
"strconv"
"strings"
"testing"
_ "modernc.org/sqlite"
)
// fixtureSource writes a miniature stand-in for the kaikki export: the same
// JSONL shape, a handful of rows instead of 44k. Each row is a word and the
// language its Wiktionary entry is for. Tests never touch the real download.
func fixtureSource(t *testing.T, rows [][2]string) string {
t.Helper()
lines := make([]string, 0, len(rows))
for _, r := range rows {
lines = append(lines, `{"word": "`+r[0]+`", "pos": "noun", "lang_code": "`+r[1]+`"}`)
}
return fixtureKaikki(t, lines...)
}
func defaultRows() [][2]string {
return [][2]string{
{"pháp luật", "vi"},
{"pháp luật", "vi"}, // listed twice — must dedupe to one word
{"luật lệ", "vi"},
{"ngôn ngữ", "vi"},
{"ngữ pháp", "vi"},
{"hòa bình", "vi"},
{"vô tuyến điện", "vi"}, // three syllables
{"pháp", "vi"}, // single syllable — rejected
{"covid 19", "vi"}, // digit — rejected
{"hello world", "en"}, // another language's entry — never selected
}
}
func buildFixture(t *testing.T, rows [][2]string) string {
// buildFixture runs the whole pipeline on the committed mini dump: twelve
// pages, both dialects, a redirect, an English-only page and two pages that
// accept() rejects. Tests never touch the real download.
func buildFixture(t *testing.T) string {
t.Helper()
out := filepath.Join(t.TempDir(), "noitu.db")
cfg := config{
kaikki: fixtureSource(t, rows),
dump: miniDump,
out: out,
minWords: 1,
minPages: 1,
}
if err := run(cfg); err != nil {
t.Fatalf("run: %v", err)
@@ -64,23 +42,23 @@ func openOut(t *testing.T, path string) *sql.DB {
return db
}
func count(t *testing.T, db *sql.DB, query string, args ...any) int {
t.Helper()
var n int
if err := db.QueryRow(query, args...).Scan(&n); err != nil {
t.Fatalf("%s: %v", query, err)
}
return n
}
func TestBuildProducesExpectedWords(t *testing.T) {
db := openOut(t, buildFixture(t, defaultRows()))
db := openOut(t, buildFixture(t))
var count int
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&count); err != nil {
t.Fatal(err)
if got := count(t, db, `SELECT COUNT(*) FROM words`); got != 6 {
t.Errorf("word count = %d, want 6", got)
}
if want := 6; count != want {
t.Errorf("word count = %d, want %d", count, want)
}
// Every stored word must have at least two syllables.
var short int
if err := db.QueryRow(`SELECT COUNT(*) FROM words WHERE syllables < 2`).Scan(&short); err != nil {
t.Fatal(err)
}
if short != 0 {
if short := count(t, db, `SELECT COUNT(*) FROM words WHERE syllables < 2`); short != 0 {
t.Errorf("%d words have fewer than 2 syllables, want 0", short)
}
@@ -98,7 +76,7 @@ func TestBuildProducesExpectedWords(t *testing.T) {
}
func TestBuildComputesOutDegree(t *testing.T) {
db := openOut(t, buildFixture(t, defaultRows()))
db := openOut(t, buildFixture(t))
// "pháp luật" and "pháp" (rejected) mean exactly one word starts with "pháp".
assertOutDegree(t, db, "pháp", 1)
@@ -120,7 +98,7 @@ func assertOutDegree(t *testing.T, db *sql.DB, syllable string, want int) {
}
func TestBuildWritesAliases(t *testing.T) {
db := openOut(t, buildFixture(t, defaultRows()))
db := openOut(t, buildFixture(t))
var canonical string
err := db.QueryRow(`SELECT canonical FROM aliases WHERE variant = ?`, "hoà bình").Scan(&canonical)
@@ -132,22 +110,65 @@ func TestBuildWritesAliases(t *testing.T) {
}
// Every alias must point at a word that actually exists.
var orphans int
err = db.QueryRow(`SELECT COUNT(*) FROM aliases a
LEFT JOIN words w ON w.word = a.canonical
WHERE w.word IS NULL`).Scan(&orphans)
if err != nil {
t.Fatal(err)
}
orphans := count(t, db, `SELECT COUNT(*) FROM aliases a LEFT JOIN words w ON w.word = a.canonical WHERE w.word IS NULL`)
if orphans != 0 {
t.Errorf("%d aliases point at missing words, want 0", orphans)
}
}
func TestBuildRecordsProvenance(t *testing.T) {
db := openOut(t, buildFixture(t, defaultRows()))
func TestBuildWritesMeanings(t *testing.T) {
db := openOut(t, buildFixture(t))
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count"} {
rows, err := db.Query(`SELECT ord, pos, gloss FROM meanings WHERE word = ? ORDER BY ord`, "pháp luật")
if err != nil {
t.Fatal(err)
}
defer rows.Close()
var got []sense
for rows.Next() {
var ord int
var s sense
if err := rows.Scan(&ord, &s.pos, &s.gloss); err != nil {
t.Fatal(err)
}
if ord != len(got) {
t.Errorf("ord = %d, want %d (0-based, dense)", ord, len(got))
}
got = append(got, s)
}
assertSenses(t, got, []sense{
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
{"động từ", "(hiếm) Xử theo luật."},
})
// A word whose only definition stripped to nothing has no rows, and the
// meta counts describe the table.
if n := count(t, db, `SELECT COUNT(*) FROM meanings WHERE word = ?`, "luật lệ"); n != 0 {
t.Errorf("luật lệ has %d meanings, want 0", n)
}
total := count(t, db, `SELECT COUNT(*) FROM meanings`)
withMeaning := count(t, db, `SELECT COUNT(DISTINCT word) FROM meanings`)
for key, want := range map[string]int{"meaning_count": total, "words_with_meaning": withMeaning} {
var raw string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
t.Fatalf("meta[%q]: %v", key, err)
}
if raw != strconv.Itoa(want) {
t.Errorf("meta[%q] = %s, want %d", key, raw, want)
}
}
if withMeaning != 5 {
t.Errorf("words with a meaning = %d, want 5 of 6", withMeaning)
}
}
func TestBuildRecordsProvenance(t *testing.T) {
db := openOut(t, buildFixture(t))
want := map[string]string{"builder_version": builderVer, "source_url": dumpSourceURL, "source_pages": "9"}
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count",
"builder_version", "source_sha256", "source_pages", "source_fetched_at", "meaning_count", "words_with_meaning"} {
var value string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&value); err != nil {
t.Errorf("meta[%q] missing: %v", key, err)
@@ -156,26 +177,42 @@ func TestBuildRecordsProvenance(t *testing.T) {
if value == "" {
t.Errorf("meta[%q] is empty", key)
}
if w, ok := want[key]; ok && value != w {
t.Errorf("meta[%q] = %q, want %q", key, value, w)
}
}
if n := count(t, db, `SELECT COUNT(*) FROM meta WHERE key = 'source_rows'`); n != 0 {
t.Error("source_rows belonged to the previous source format and must be gone")
}
}
// The floor exists so a schema change upstream fails the build loudly instead
// The floor exists so a markup change upstream fails the build loudly instead
// of silently shipping a near-empty dictionary.
func TestBuildFailsBelowMinWords(t *testing.T) {
cfg := config{
kaikki: fixtureSource(t, defaultRows()),
dump: miniDump,
out: filepath.Join(t.TempDir(), "noitu.db"),
minWords: 1000,
minPages: 1,
}
if err := run(cfg); err == nil {
t.Fatal("run succeeded with an unreachable min-words floor, want error")
}
}
func TestRunRequiresExactlyOneInput(t *testing.T) {
if err := run(config{}); err == nil {
t.Error("run with no input succeeded")
}
if err := run(config{dump: miniDump, words: "x.txt"}); err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
t.Errorf("run with both inputs: %v", err)
}
}
// A failed build must leave the previous good database untouched. Building in
// place would delete it and leave an empty file the server would happily open.
func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
out := buildFixture(t, defaultRows())
out := buildFixture(t)
before, err := os.ReadFile(out)
if err != nil {
@@ -184,9 +221,10 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
// Same output path, but a floor no fixture can clear.
cfg := config{
kaikki: fixtureSource(t, defaultRows()),
dump: miniDump,
out: out,
minWords: 1000,
minPages: 1,
}
if err := run(cfg); err == nil {
t.Fatal("run succeeded with an unreachable floor, want error")
@@ -203,3 +241,127 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
t.Error("temp database left behind after a failed build")
}
}
// --- the fixture word list --------------------------------------------------
func writeWordList(t *testing.T, content string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "words.txt")
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
t.Fatal(err)
}
return path
}
func TestWordListCarriesTabSeparatedMeanings(t *testing.T) {
list := writeWordList(t, "# comment\n"+
"học sinh\tdanh từ|Người học ở trường.\tđộng từ|Đi học.\n"+
"sinh viên\tNgười học ở trường đại học.\n"+
"sinh hoạt\n"+
"sinh sản\t\t\n")
out := filepath.Join(t.TempDir(), "fixture.db")
if err := run(config{words: list, out: out, minWords: 1}); err != nil {
t.Fatal(err)
}
db := openOut(t, out)
rows, err := db.Query(`SELECT word, ord, pos, gloss FROM meanings ORDER BY word, ord`)
if err != nil {
t.Fatal(err)
}
defer rows.Close()
type row struct {
word string
ord int
s sense
}
var got []row
for rows.Next() {
var r row
if err := rows.Scan(&r.word, &r.ord, &r.s.pos, &r.s.gloss); err != nil {
t.Fatal(err)
}
got = append(got, r)
}
want := []row{
{"học sinh", 0, sense{"danh từ", "Người học ở trường."}},
{"học sinh", 1, sense{"động từ", "Đi học."}},
{"sinh viên", 0, sense{"", "Người học ở trường đại học."}},
}
if len(got) != len(want) {
t.Fatalf("meanings = %+v, want %+v", got, want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("row %d = %+v, want %+v", i, got[i], want[i])
}
}
if n := count(t, db, `SELECT COUNT(*) FROM words`); n != 4 {
t.Errorf("words = %d, want 4 (a line without a tab is still a word)", n)
}
// Fixture builds carry no upstream licence and are exempt from the
// coverage floor: two of four words have a meaning here.
var license string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
t.Fatal(err)
}
if !strings.HasPrefix(license, "none") {
t.Errorf("fixture licence = %q, want a statement that no upstream data applies", license)
}
}
// --- verify -----------------------------------------------------------------
// brokenDB writes a database that passes every schema check and then breaks
// one invariant, to prove verify() reads what is on disk.
func brokenDB(t *testing.T, extraSQL string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "broken.db")
words := map[string]entry{"pháp luật": {"pháp luật", "pháp", "luật", 2}}
meanings := map[string][]sense{"pháp luật": {{"danh từ", "Luật."}}}
if err := writeTo(path, words, meanings, nil, sourceSpec{table: "test"}); err != nil {
t.Fatal(err)
}
db, err := sql.Open("sqlite", "file:"+path)
if err != nil {
t.Fatal(err)
}
defer db.Close()
if _, err := db.Exec(extraSQL); err != nil {
t.Fatal(err)
}
return path
}
func TestVerifyRejectsBrokenMeanings(t *testing.T) {
cases := map[string]string{
"orphan meaning row": `INSERT INTO meanings VALUES ('không có', 0, '', 'Một nghĩa.')`,
"empty gloss": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '')`,
"over the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '` + strings.Repeat("a", maxGlossRunes+1) + `')`,
"more senses than the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', 'b'), ('pháp luật', 2, '', 'c'),
('pháp luật', 3, '', 'd'), ('pháp luật', 4, '', 'e'), ('pháp luật', 5, '', 'f')`,
}
for name, sqlText := range cases {
t.Run(name, func(t *testing.T) {
path := brokenDB(t, sqlText)
if err := verify(path, 1, false); err == nil {
t.Error("verify passed a database that breaks a meanings invariant")
}
})
}
if err := verify(brokenDB(t, `SELECT 1`), 1, false); err != nil {
t.Errorf("verify rejected a sound database: %v", err)
}
}
func TestVerifyCoverageFloorAppliesToCorpusBuildsOnly(t *testing.T) {
// One word with a meaning, one without: 50%, under the floor.
path := brokenDB(t, `INSERT INTO words VALUES ('luật lệ', 'luật', 'lệ', 2);
INSERT INTO syllables VALUES ('lệ', 0); UPDATE syllables SET out_degree = 1 WHERE syllable = 'luật'`)
if err := verify(path, 1, true); err == nil || !strings.Contains(err.Error(), "have a meaning") {
t.Errorf("corpus verify with 50%% coverage: %v, want the coverage floor named", err)
}
if err := verify(path, 1, false); err != nil {
t.Errorf("fixture verify applied the coverage floor: %v", err)
}
}
Binary file not shown.
+182
View File
@@ -0,0 +1,182 @@
<mediawiki xmlns="http://www.mediawiki.org/xml/export-0.11/" xml:lang="vi">
<siteinfo>
<sitename>Wiktionary</sitename>
<dbname>viwiktionary</dbname>
</siteinfo>
<!-- A legacy-dialect page: two parts of speech, an English section after
the Vietnamese one that must not be read, and a line with no space
after the # which is still a definition. -->
<page>
<title>pháp luật</title>
<ns>0</ns>
<id>1</id>
<revision>
<id>101</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-pron-}}
{{vie-pron|pháp luật}}
{{-noun-}}
{{-dfn-}}
# [[hệ thống|Hệ thống]] các [[quy tắc]] xử sự do [[nhà nước]] đặt ra.&lt;ref&gt;Từ điển tiếng Việt&lt;/ref&gt;
#: ''Tuân theo pháp luật.''
#{{label|vi|nghĩa rộng}} [[kỷ cương|Kỷ cương]] nói chung.
{{-verb-}}
# ''(hiếm)'' [[xử|Xử]] theo luật.
{{-trans-}}
* {{eng}}: {{t|en|law}}
{{-eng-}}
{{-noun-}}
# Law, in English.</text>
</revision>
</page>
<!-- A new-dialect page with a proper-noun heading, a place template and an
English section that must not be read. -->
<page>
<title>Hòa Bình</title>
<ns>0</ns>
<id>2</id>
<revision>
<id>102</id>
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
=== {{ĐM|etym}} ===
Từ Hán-Việt.
=== {{ĐM|pr-noun}} ===
{{vi-pr-noun}}
# {{place|vi|tỉnh|c/Việt Nam}}.
== {{langname|en}} ==
=== {{ĐM|pr-noun}} ===
# A province of Vietnam.</text>
</revision>
</page>
<!-- The lowercase page of the same word: the two merge into one entry with
the senses in page order. -->
<page>
<title>hòa bình</title>
<ns>0</ns>
<id>3</id>
<revision>
<id>103</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# [[tình trạng|Tình trạng]] không có [[chiến tranh]].
{{-adj-}}
# [[yên ổn|Yên ổn]].</text>
</revision>
</page>
<!-- A case-only redirect: skipped and counted. -->
<page>
<title>mặt trời</title>
<ns>0</ns>
<id>4</id>
<redirect title="Mặt Trời" />
<revision>
<id>104</id>
<text bytes="1" xml:space="preserve">#đổi [[Mặt Trời]]</text>
</revision>
</page>
<!-- No Vietnamese section at all. -->
<page>
<title>hello world</title>
<ns>0</ns>
<id>5</id>
<revision>
<id>105</id>
<text bytes="1" xml:space="preserve">{{-eng-}}
{{-phrase-}}
# Xin chào thế giới.</text>
</revision>
</page>
<!-- Not in the main namespace. -->
<page>
<title>Thể loại:Danh từ tiếng Việt</title>
<ns>14</ns>
<id>6</id>
<revision>
<id>106</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# Not a word.</text>
</revision>
</page>
<!-- One syllable: rejected downstream by accept(), but its section is
still a Vietnamese section for the page count. -->
<page>
<title>pháp</title>
<ns>0</ns>
<id>7</id>
<revision>
<id>107</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# [[phép|Phép]], [[luật]].</text>
</revision>
</page>
<!-- A definition that is only a template the stripper does not know: no
meaning, but still a word. -->
<page>
<title>luật lệ</title>
<ns>0</ns>
<id>8</id>
<revision>
<id>108</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# {{rfdef|vi}}</text>
</revision>
</page>
<!-- More graph: a shorthand heading code in the new dialect, and words that
give the out-degree and alias tests something to check. -->
<page>
<title>ngôn ngữ</title>
<ns>0</ns>
<id>9</id>
<revision>
<id>109</id>
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
=== {{section|n}} ===
{{vi-noun}}
# [[hệ thống|Hệ thống]] những [[âm]], [[từ]] và [[quy tắc]] kết hợp chúng.</text>
</revision>
</page>
<page>
<title>ngữ pháp</title>
<ns>0</ns>
<id>10</id>
<revision>
<id>110</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# [[toàn bộ|Toàn bộ]] những [[quy tắc]] hoạt động của các yếu tố ngôn ngữ.</text>
</revision>
</page>
<page>
<title>vô tuyến điện</title>
<ns>0</ns>
<id>11</id>
<revision>
<id>111</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# [[kỹ thuật|Kỹ thuật]] truyền tin bằng [[sóng điện từ]].</text>
</revision>
</page>
<page>
<title>covid 19</title>
<ns>0</ns>
<id>12</id>
<revision>
<id>112</id>
<text bytes="1" xml:space="preserve">{{-vie-}}
{{-noun-}}
# Một [[bệnh]].</text>
</revision>
</page>
</mediawiki>
Binary file not shown.
+629
View File
@@ -0,0 +1,629 @@
package main
import (
"html"
"regexp"
"strings"
"unicode"
"unicode/utf8"
)
// This file reads the wikitext of one Wiktionary tiếng Việt page: it finds the
// Vietnamese section, walks its part-of-speech headings and turns each
// definition line into plain text.
//
// The wiki is mid-migration between two markup dialects and both are live
// (2026-09-01 dump: 35,885 legacy pages, 7,129 new):
//
// legacy {{-vie-}} opens the section, {{-noun-}} and kin are the headings,
// and the section ends at the next {{-xxx-}} whose code is a
// language rather than a heading.
// new == {{langname|vi}} == opens the section, === {{ĐM|noun}} === or
// === {{section|noun}} === are the headings, and the next level-2
// heading ends it.
//
// Nothing here is a general wikitext parser. It knows exactly the shapes a
// definition line takes on this wiki and drops the rest on purpose; what
// survives is plain text, capped, safe to render as text and never as markup.
// sense is one definition with the Vietnamese part-of-speech label of the
// heading it sat under. pos is empty when the heading was one the label map
// does not know, never a reason to drop the definition.
type sense struct {
pos string
gloss string
}
const (
// maxSenses and maxGlossRunes bound what one word carries to the client.
maxSenses = 5
maxGlossRunes = 200
// ellipsis marks a definition cut at maxGlossRunes.
ellipsis = "…"
)
// posLabelMap maps a part-of-speech heading code, the same in both dialects
// ({{-noun-}}, {{ĐM|noun}}, {{section|noun}}, {{vi-noun}}), to the Vietnamese
// label the client shows in front of a sense.
//
// Codes and their frequencies in Vietnamese sections of the 2026-09-01 dump:
// noun 13,674 + n 647 · verb 7,595 + v 353 · adj 5,318 + adjc 526 · place 3,352
// · pr-noun 1,397 + 329 · adv 982 · phrase 412 · proverb 295 · idiom 241 ·
// interj 169 · pronoun 158 · num 123 · conj 88 · prep 88 · part 29. Note that
// "pron" on this wiki is pronunciation, not pronoun.
var posLabelMap = map[string]string{
"noun": "danh từ", "n": "danh từ",
"verb": "động từ", "v": "động từ", "tr-verb": "động từ", "intr-verb": "động từ", "aux-verb": "động từ",
"adj": "tính từ", "adjc": "tính từ", "adjective": "tính từ",
"adv": "phó từ", "adverb": "phó từ", "advb": "phó từ",
"pr-noun": "danh từ riêng", "proper": "danh từ riêng", "propn": "danh từ riêng", "proper noun": "danh từ riêng", "name": "danh từ riêng",
"pr-adj": "tính từ riêng",
"place": "địa danh",
"pronoun": "đại từ", "per-pronoun": "đại từ",
"num": "số từ", "numeral": "số từ",
"conj": "liên từ", "conjunction": "liên từ",
"prep": "giới từ",
"interj": "thán từ", "intj": "thán từ", "interjection": "thán từ",
"part": "trợ từ", "particle": "trợ từ",
"phrase": "cụm từ",
"idiom": "thành ngữ",
"proverb": "tục ngữ", "prov": "tục ngữ",
"abbr": "viết tắt", "abr": "viết tắt",
"prefix": "tiền tố",
"suffix": "hậu tố",
"letter": "chữ cái",
"symbol": "ký hiệu",
}
// otherSectionCodes are heading codes that are not parts of speech: they sit
// inside a language section and reset the current label without being
// counted as unmapped. Frequencies in Vietnamese sections, 2026-09-01 dump:
// pron 36,533 · ref 27,438 · trans 16,073 · paro 8,058 · etym 4,267 · syn
// 3,165 · info 1,846 · hanviet 1,601 · hanviet-t 1,441 · etymology 1,429 ·
// reference 1,426 · see 959 · related 461 · drv 277 · synonym 262 · ant 236 ·
// desction 128 · usage 110 · expr 92 · homo 57 · desc 54 · further 48 · forms
// 41 · note 33 · derived 28 · anagram 24 · compound 21 · redup 20 · translit 17
// · cat 16 · antonym 15. "dfn" (4,615) is a "definitions" heading placed under a
// part-of-speech heading, so it is a heading for the section boundary but
// transparent to the label: see classifyHeading.
var otherSectionCodes = map[string]bool{
"pron": true, "pronunciation": true, "ref": true, "reference": true, "references": true,
"trans": true, "translations": true, "paro": true, "paronym": true, "etym": true, "etymology": true,
"syn": true, "synonym": true, "ant": true, "antonym": true, "info": true,
"hanviet": true, "hanviet-t": true, "see": true, "see also": true, "related": true, "rel": true,
"related terms": true, "drv": true, "der": true, "derived": true, "derived terms": true,
"desction": true, "desc": true, "usage": true, "usage notes": true, "expr": true, "homo": true,
"further": true, "further reading": true, "forms": true, "note": true, "anagram": true,
"anagrams": true, "ana": true, "compound": true, "redup": true, "translit": true, "cat": true,
"coord": true, "coordinate": true, "alt": true, "alter": true, "alter form": true,
"alternative form": true, "alternative forms": true, "alternative script": true,
"glyph origin": true, "han": true, "nôm": true, "han character": true, "kanji": true,
"rom": true, "romanization": true, "mut": true, "participle": true, "ptcp": true,
"syllable": true, "article": true, "contr": true, "cmavo": true, "dfn": true, "com": true,
// The same sections written out in Vietnamese, as a few new-dialect pages do.
"phát âm": true, "từ nguyên": true, "từ nguyên 1": true, "từ nguyên 2": true, "tham khảo": true,
"xem thêm": true, "cách viết khác": true, "phồn thể": true, "hán-nôm": true, "hán nôm": true,
"chữ hán": true, "chữ nôm": true, "chú ý": true, "đồng nghĩa": true, "từ đồng nghĩa": true,
"bản dịch": true, "dịch": true, "dấu phụ": true, "liên kết ngoài": true, "thuật ngữ liên quan": true,
"từ tương tự": true, "meronym": true, "meronyms": true, "nguồn gốc ký tự chữ nôm": true,
}
var (
// legacyTemplate matches one {{-code-}} template, optionally with
// parameters: {{-noun-}}, {{-pr-noun-}}, {{-vie-|...}}. Not anchored: a few
// pages run {{-vie-}}{{-pron-}}{{vie-pron|…}}{{-place-}} together on one
// line, so a line is read as headings when it starts with one and may
// carry several.
legacyTemplate = regexp.MustCompile(`\{\{-([A-Za-z0-9-]+?)-(?:\|[^}]*)?\}\}`)
// headingLine matches == text == at any level and captures the level.
headingLine = regexp.MustCompile(`^(={2,6})\s*(.*?)\s*=+\s*$`)
// sectionTemplate captures the code of {{ĐM|code}} and {{section|code}}.
sectionTemplate = regexp.MustCompile(`\{\{(?:ĐM|đm|DM|dm|section)\|([^}|]+)`)
// headwordTemplate captures the code of {{vi-code}} / {{vie-code}} at the
// start of a line: the new dialect's headword line, which names the POS.
headwordTemplate = regexp.MustCompile(`^\{\{vie?-([a-z -]+)`)
// langnameVi is the new dialect's Vietnamese section heading text.
langnameVi = regexp.MustCompile(`^\{\{langname\|vi\}\}$`)
htmlComment = regexp.MustCompile(`(?s)<!--.*?-->`)
refElement = regexp.MustCompile(`(?s)<ref\b[^>/]*/>|<ref\b[^>]*>.*?</ref>`)
anyTag = regexp.MustCompile(`</?[A-Za-z][^>]*>`)
spaces = regexp.MustCompile(`\s+`)
)
// isLegacyHeading reports whether a {{-code-}} is a heading inside a language
// section. Every other code — language and script codes such as eng, tyz,
// aav-qal, Latn — ends the Vietnamese section.
func isLegacyHeading(code string) bool {
_, pos := posLabelMap[code]
return pos || otherSectionCodes[code]
}
// vietnameseSection returns the wikitext of the page's Vietnamese section and
// which dialect opened it: "legacy", "new", or "" when the page has none.
// When both dialects open a section on one page the first one in the text
// wins and both is reported so the build log can count it. ender is the
// {{-code-}} that closed a legacy section, empty when a heading or the end of
// the page did: a heading code missing from the maps shows up there as a
// section-ending code, which is the signal that definitions are being lost.
func vietnameseSection(text string) (section, dialect string, both bool, ender string) {
lines := strings.Split(text, "\n")
legacyAt, newAt := -1, -1
for i, line := range lines {
line = strings.TrimRight(line, "\r ")
if legacyAt < 0 && strings.HasPrefix(line, "{{-") {
for _, m := range legacyTemplate.FindAllStringSubmatchIndex(line, -1) {
if line[m[2]:m[3]] == "vie" {
legacyAt = i
// Whatever follows the marker on its own line belongs to
// the section.
lines[i] = line[m[1]:]
break
}
}
}
if newAt < 0 {
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 && langnameVi.MatchString(m[2]) {
newAt = i
}
}
}
both = legacyAt >= 0 && newAt >= 0
switch {
case legacyAt < 0 && newAt < 0:
return "", "", false, ""
case newAt < 0 || (legacyAt >= 0 && legacyAt < newAt):
section, ender = legacySection(lines[legacyAt:])
return section, "legacy", both, ender
default:
return newSection(lines[newAt+1:]), "new", both, ""
}
}
// legacySection runs from after {{-vie-}} to the next {{-xxx-}} whose code is
// not a heading, or the next level-2 heading, which is what a new-dialect
// language section on a mixed page opens with. A language code sharing a line
// with Vietnamese headings ends the section at that line; the line is lost,
// which is the conservative side of a rare shape.
func legacySection(lines []string) (section, ender string) {
for i, line := range lines {
line = strings.TrimRight(line, "\r ")
if strings.HasPrefix(line, "{{-") {
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
if !isLegacyHeading(m[1]) {
return strings.Join(lines[:i], "\n"), m[1]
}
}
}
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
return strings.Join(lines[:i], "\n"), ""
}
}
return strings.Join(lines, "\n"), ""
}
// newSection runs from after == {{langname|vi}} == to the next level-2
// heading.
func newSection(lines []string) string {
for i, line := range lines {
line = strings.TrimRight(line, "\r ")
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
return strings.Join(lines[:i], "\n")
}
}
return strings.Join(lines, "\n")
}
// headingKind says what a heading line means for the label of the
// definitions under it.
type headingKind int
const (
notHeading headingKind = iota
// posHeading names a part of speech: the code decides the label.
posHeading
// otherHeading is a section such as pronunciation or etymology: the label
// resets to empty and nothing is counted as unmapped.
otherHeading
)
// classifyHeading reads one line as a heading in either dialect.
//
// {{-noun-}} legacy heading; the last of several on
// one line decides
// === {{ĐM|noun}} === new heading
// === {{section|n}} === new heading, shorthand code
// === Danh từ === new heading written out
// {{vi-noun}} / {{vie-noun}} new headword line; refines the POS only
func classifyHeading(line string) (code string, kind headingKind) {
line = strings.TrimRight(line, "\r ")
if strings.HasPrefix(line, "{{-") {
kind = notHeading
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
if c, k := classifyCode(m[1]); k != notHeading {
code, kind = c, k
}
}
return code, kind
}
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) >= 3 {
text := m[2]
if sm := sectionTemplate.FindStringSubmatch(text); sm != nil {
code = strings.TrimSpace(sm[1])
} else {
code = strings.ToLower(text)
}
return classifyCode(code)
}
if m := headwordTemplate.FindStringSubmatch(line); m != nil {
// Only a headword template whose code is a part of speech counts;
// {{vi-pron}}, {{vi-etym-sino}} and kin are not headings.
code = strings.TrimSpace(m[1])
if _, ok := posLabelMap[code]; ok {
return code, posHeading
}
}
return "", notHeading
}
// classifyCode sorts a heading code seen in either dialect. "dfn" is the one
// heading that changes nothing: the wiki places {{-dfn-}} under {{-noun-}} to
// introduce the definitions, so the label above it must carry through.
func classifyCode(code string) (string, headingKind) {
switch {
case code == "dfn":
return code, notHeading
case otherSectionCodes[code]:
return code, otherHeading
}
return code, posHeading
}
// isLabelValue reports whether a heading was written out as one of the
// Vietnamese labels already ("Danh từ").
func isLabelValue(text string) bool {
for _, l := range posLabelMap {
if l == text {
return true
}
}
return false
}
// posLabel maps a heading code to its Vietnamese label: through the map, or
// as itself when the heading was already written out in Vietnamese.
func posLabel(code string) (label string, mapped bool) {
if label, ok := posLabelMap[code]; ok {
return label, true
}
if isLabelValue(code) {
return code, true
}
return "", false
}
// sectionStats counts what the scanner saw across sections, for the build log.
type sectionStats struct {
pos map[string]int // part-of-speech heading codes seen
unmappedPos map[string]int // part-of-speech heading codes with no label
defsKept int
defsEmpty int // definitions that stripped to nothing
defsCut int // definitions cut at maxGlossRunes
dropped map[string]int // template names dropped whole
}
func newSectionStats() *sectionStats {
return &sectionStats{
pos: make(map[string]int),
unmappedPos: make(map[string]int),
dropped: make(map[string]int),
}
}
// definitions walks the Vietnamese section and returns its senses in page
// order, at most maxSenses of them, each labelled with the part of speech of
// the heading above it. Every heading and definition is counted in stats
// whether or not it made the cut.
func definitions(section string, stats *sectionStats) []sense {
var senses []sense
pos := ""
for _, line := range strings.Split(section, "\n") {
line = strings.TrimRight(line, "\r ")
switch code, kind := classifyHeading(line); kind {
case posHeading:
stats.pos[code]++
label, mapped := posLabel(code)
if !mapped {
stats.unmappedPos[code]++
}
pos = label
continue
case otherHeading:
pos = ""
continue
}
if !isDefinitionLine(line) {
continue
}
gloss := stripWikitext(line[1:], stats.dropped)
if gloss == "" {
stats.defsEmpty++
continue
}
if cut, wasCut := capGloss(gloss); wasCut {
stats.defsCut++
gloss = cut
}
stats.defsKept++
if len(senses) < maxSenses {
senses = append(senses, sense{pos: pos, gloss: gloss})
}
}
return senses
}
// isDefinitionLine accepts a top-level numbered item and nothing under it:
// "# text" and "#text" are definitions; "#: example", "#* quotation", "## sub-
// sense" and "#; term" are not.
func isDefinitionLine(line string) bool {
if len(line) < 2 || line[0] != '#' {
return false
}
switch line[1] {
case '#', ':', '*', ';':
return false
}
return true
}
// stripWikitext turns one definition line into plain text. Lossy on purpose:
// links keep their display text, formatting goes, the handful of templates
// that carry definition text are unwrapped and every other template is
// dropped whole (its name counted in dropped when non-nil). The result is
// trimmed and whitespace-collapsed; a result with no letter or digit is empty.
func stripWikitext(s string, dropped map[string]int) string {
s = htmlComment.ReplaceAllString(s, "")
s = refElement.ReplaceAllString(s, "")
s = anyTag.ReplaceAllString(s, "")
s = stripTemplates(s, dropped)
s = stripLinks(s)
s = strings.ReplaceAll(s, "'''", "")
s = strings.ReplaceAll(s, "''", "")
s = html.UnescapeString(s)
s = strings.Map(func(r rune) rune {
switch {
case unicode.IsControl(r), unicode.Is(unicode.Cf, r):
// Cc and Cf: control characters, and format characters such as a
// bidi override or a zero-width space, which could reshape the
// rest of a rendered line.
return -1
case unicode.IsSpace(r):
// Non-breaking and other Unicode spaces become plain ones so the
// ASCII-only collapse below catches them.
return ' '
}
return r
}, s)
s = strings.TrimSpace(spaces.ReplaceAllString(s, " "))
// A stray space before sentence punctuation is what unwrapping a template
// at the end of a clause leaves behind.
for _, p := range []string{" .", " ,", " ;", " :", " )"} {
s = strings.ReplaceAll(s, p, p[1:])
}
if !strings.ContainsFunc(s, func(r rune) bool { return unicode.IsLetter(r) || unicode.IsDigit(r) }) {
return ""
}
return s
}
// stripTemplates replaces every outermost {{...}} with its plain-text
// rendering. Nesting is tracked by depth, so a template inside a kept
// template's parameter is rendered recursively and one inside a dropped
// template goes with it.
func stripTemplates(s string, dropped map[string]int) string {
var out strings.Builder
depth := 0
start := 0
for i := 0; i < len(s); i++ {
switch {
case strings.HasPrefix(s[i:], "{{"):
if depth == 0 {
start = i + 2
}
depth++
i++
case strings.HasPrefix(s[i:], "}}"):
// A closer with nothing open is stray markup, not text.
if depth > 0 {
depth--
if depth == 0 {
out.WriteString(renderTemplate(s[start:i], dropped))
}
}
i++
case depth == 0:
out.WriteByte(s[i])
}
}
if depth > 0 && dropped != nil {
// Unbalanced braces: whatever opened and never closed is dropped, as a
// template would be, rather than leaking half a template into a gloss.
dropped["(unclosed)"]++
}
return out.String()
}
// renderTemplate maps one template body (the text between {{ and }}) to plain
// text. body still contains any nested templates verbatim.
//
// The kept templates are the ones that carry definition text on this wiki:
// context labels in four spellings, links in five, place descriptions,
// non-gloss definitions and the two cross-reference templates a "dfn" section
// is usually made of. Everything else is presentation or classification.
func renderTemplate(body string, dropped map[string]int) string {
parts := splitTemplate(body)
name := strings.ToLower(strings.TrimSpace(parts[0]))
// Positional parameters only; key=value ones are presentation hints.
var params []string
for _, p := range parts[1:] {
if strings.Contains(p, "=") && !strings.Contains(p, "[[") && !strings.Contains(p, "{{") {
continue
}
params = append(params, strings.TrimSpace(stripTemplates(strings.TrimSpace(p), dropped)))
}
// A leading language code is markup, whether it is ours or a neighbour's
// pasted in: the label templates take it first, the link ones too. Only
// when something follows it, though: {{q|con}} is a one-word qualifier,
// not a language.
dropLang := func(ps []string) []string {
if len(ps) > 1 && isLangCode(ps[0]) {
return ps[1:]
}
return ps
}
switch name {
case "label", "lb", "nhãn", "context", "term", "gloss", "qualifier", "q":
params = dropLang(params)
if len(params) == 0 {
return ""
}
return "(" + strings.Join(params, ", ") + ")"
case "l", "vi-l", "w", "m", "link":
params = dropLang(params)
if len(params) == 0 {
return ""
}
return params[len(params)-1]
case "n-g", "non-gloss", "non-gloss definition":
return strings.Join(params, " ")
case "see-entry", "like-entry":
if len(params) == 0 {
return ""
}
return "Xem " + params[0]
case "place":
params = dropLang(params)
for i, p := range params {
// "c/Việt Nam" is a typed place: the type prefix is markup.
if len(p) > 2 && p[1] == '/' && p[0] >= 'a' && p[0] <= 'z' {
params[i] = p[2:]
}
}
return strings.Join(params, ", ")
}
if dropped != nil {
dropped[name]++
}
return ""
}
// isLangCode reports whether a template parameter is a language code rather
// than text: two or three lowercase ASCII letters, optionally with a
// hyphenated variant such as "nan-hbl".
func isLangCode(p string) bool {
if len(p) < 2 || len(p) > 11 {
return false
}
letters := 0
for _, r := range p {
switch {
case r >= 'a' && r <= 'z':
letters++
case r == '-':
if letters < 2 {
return false
}
letters = 0
default:
return false
}
}
return letters >= 2 && letters <= 3
}
// splitTemplate splits a template body on | outside nested braces and
// brackets, so a link or template inside a parameter is not cut in two.
func splitTemplate(body string) []string {
var parts []string
depth := 0
start := 0
for i := 0; i < len(body); i++ {
switch {
case strings.HasPrefix(body[i:], "{{") || strings.HasPrefix(body[i:], "[["):
depth++
i++
case strings.HasPrefix(body[i:], "}}") || strings.HasPrefix(body[i:], "]]"):
if depth > 0 {
depth--
}
i++
case body[i] == '|' && depth == 0:
parts = append(parts, body[start:i])
start = i + 1
}
}
return append(parts, body[start:])
}
// stripLinks renders wiki links as their display text and drops category
// links, which are classification rather than definition.
func stripLinks(s string) string {
var out strings.Builder
for {
open := strings.Index(s, "[[")
if open < 0 {
break
}
close := strings.Index(s[open:], "]]")
if close < 0 {
break
}
out.WriteString(s[:open])
inner := s[open+2 : open+close]
lower := strings.ToLower(inner)
if !strings.HasPrefix(lower, "thể loại:") && !strings.HasPrefix(lower, "category:") {
if bar := strings.LastIndex(inner, "|"); bar >= 0 {
inner = inner[bar+1:]
}
out.WriteString(inner)
}
s = s[open+close+2:]
}
out.WriteString(s)
s = out.String()
// External links: [http://… label] → label; a bare URL in brackets goes.
out.Reset()
for {
open := strings.Index(s, "[http")
if open < 0 {
break
}
close := strings.Index(s[open:], "]")
if close < 0 {
break
}
out.WriteString(s[:open])
inner := s[open+1 : open+close]
if sp := strings.IndexByte(inner, ' '); sp >= 0 {
out.WriteString(inner[sp+1:])
}
s = s[open+close+1:]
}
out.WriteString(s)
return out.String()
}
// capGloss cuts a definition longer than maxGlossRunes at the last space
// before the limit and marks the cut with an ellipsis.
func capGloss(s string) (string, bool) {
if utf8.RuneCountInString(s) <= maxGlossRunes {
return s, false
}
runes := []rune(s)
head := string(runes[:maxGlossRunes-utf8.RuneCountInString(ellipsis)])
if sp := strings.LastIndexByte(head, ' '); sp > 0 {
head = head[:sp]
}
return strings.TrimRight(head, " ,;:") + ellipsis, true
}
@@ -0,0 +1,274 @@
package main
import (
"strings"
"testing"
"unicode/utf8"
)
func TestStripWikitext(t *testing.T) {
cases := []struct {
name string
in string
want string
}{
{"links keep display text",
"[[chỗ|Chỗ]] [[râm]] [[mát]], do [[trời]] có [[mây]] hoặc do không bị [[nắng]] [[chiếu]].",
"Chỗ râm mát, do trời có mây hoặc do không bị nắng chiếu."},
{"place template keeps its parameters and drops the type prefix",
"{{place|vi|thủ đô|c/Việt Nam}}.",
"thủ đô, Việt Nam."},
{"label becomes a parenthesis",
"{{label|vi|thuộc lịch sử}} Một [[tỉnh]] cũ của [[Việt Nam]] vào nửa cuối thế kỷ XIX.",
"(thuộc lịch sử) Một tỉnh cũ của Việt Nam vào nửa cuối thế kỷ XIX."},
{"Vietnamese label spellings",
"{{nhãn|vi|tin học}} {{context|cũ}} {{term|Hóa học}} Cấu trúc.",
"(tin học) (cũ) (Hóa học) Cấu trúc."},
{"nested template inside a kept one",
"{{label|vi|{{w|Hà Nội}}}} Thủ đô.",
"(Hà Nội) Thủ đô."},
{"unknown template dropped whole, nesting included",
"{{rfdef|vi|{{w|x}}}}",
""},
{"definition that is only a cross-reference",
"{{see-entry|bà la sát}}.",
"Xem bà la sát."},
{"non-gloss definition",
"{{n-g|Trợ từ nhấn mạnh.}}",
"Trợ từ nhấn mạnh."},
{"link templates keep the last parameter",
"{{l|vi|nói}}, {{l|vi|nói năng|nói năng (hiếm)}} và {{w|Việt Nam}}.",
"nói, nói năng (hiếm) và Việt Nam."},
{"ref mid-sentence and a lone ref",
"Một loài [[cá]]<ref>Từ điển</ref> nước ngọt<ref name=\"a\" />.",
"Một loài cá nước ngọt."},
{"comment, bold, italic, entities",
"'''Rất''' ''nhanh''<!-- todo -->&nbsp;và&amp;mạnh.",
"Rất nhanh và&mạnh."},
{"category link dropped, external link keeps label",
"Một [[thành phố]] [[Thể loại:Địa danh]] ([http://example.org trang web]).",
"Một thành phố (trang web)."},
{"named parameters are not text",
"{{lb|vi|thơ ca|sort=x}} Câu.",
"(thơ ca) Câu."},
{"a lone short parameter is text, not a language code",
"{{q|con}} Một loài vật, {{l|con}} là con.",
"(con) Một loài vật, con là con."},
{"format and bidi characters are dropped",
"M\u200bột \u202enghĩa\u202c.",
"Một nghĩa."},
{"a stray closer is not text", "Một }} nghĩa.", "Một nghĩa."},
{"only punctuation is empty", "(...).", ""},
{"control characters and whitespace collapse", " Một từ \t hai ", "Một từ hai"},
{"unclosed template does not leak", "Một {{label|vi|x từ.", "Một"},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := stripWikitext(c.in, nil); got != c.want {
t.Errorf("stripWikitext(%q)\n got %q\nwant %q", c.in, got, c.want)
}
})
}
}
func TestStripWikitextCountsDroppedTemplates(t *testing.T) {
dropped := make(map[string]int)
stripWikitext("{{rfdef|vi}} {{RfDef|vi}} {{senseid|vi|x}}", dropped)
if dropped["rfdef"] != 2 || dropped["senseid"] != 1 {
t.Errorf("dropped = %v, want rfdef 2 (case-folded), senseid 1", dropped)
}
}
func TestCapGlossCutsAtAWordBoundary(t *testing.T) {
word := "từ "
long := strings.Repeat(word, 120) // 360 runes
got, cut := capGloss(long)
if !cut {
t.Fatal("a 360-rune gloss was not cut")
}
if n := utf8.RuneCountInString(got); n > maxGlossRunes {
t.Errorf("cut gloss is %d runes, want at most %d", n, maxGlossRunes)
}
if !strings.HasSuffix(got, "từ"+ellipsis) {
t.Errorf("cut gloss %q does not end on a whole word plus the ellipsis", got)
}
if short, cut := capGloss("ngắn"); cut || short != "ngắn" {
t.Errorf("a short gloss was changed: %q %v", short, cut)
}
}
const legacyPage = `{{-vie-}}
{{-pron-}}
{{vie-pron|học sinh}}
{{-noun-}}
# [[người|Người]] [[học]] ở [[nhà trường|trường]].
#: ''Học sinh giỏi.''
#{{label|vi|cũ}} [[môn đệ|Môn đệ]].
{{-verb-}}
# [[đi học|Đi học]].
{{-trans-}}
* {{eng}}: {{t|en|student}}
{{-eng-}}
{{-noun-}}
# Student, in English.
`
const newPage = `== {{langname|vi}} ==
=== {{ĐM|etym}} ===
Hán-Việt.
=== {{ĐM|pr-noun}} ===
{{vi-pr-noun}}
# {{place|vi|thủ đô|c/Việt Nam}}.
=== {{section|v}} ===
# [[đi|Đi]] về thủ đô.
=== {{ĐM|xyz}} ===
# Một nghĩa dưới đề mục lạ.
== {{langname|en}} ==
=== {{ĐM|pr-noun}} ===
# The capital of Vietnam.
`
func TestVietnameseSectionLegacy(t *testing.T) {
section, dialect, both, _ := vietnameseSection(legacyPage)
if dialect != "legacy" || both {
t.Fatalf("dialect = %q both = %v, want legacy false", dialect, both)
}
if strings.Contains(section, "Student") {
t.Error("the English section leaked into the Vietnamese one")
}
if !strings.Contains(section, "{{-trans-}}") {
t.Error("the translations heading, a section heading rather than a language, cut the section short")
}
}
func TestVietnameseSectionNew(t *testing.T) {
section, dialect, both, _ := vietnameseSection(newPage)
if dialect != "new" || both {
t.Fatalf("dialect = %q both = %v, want new false", dialect, both)
}
if strings.Contains(section, "capital of Vietnam") {
t.Error("the English section leaked into the Vietnamese one")
}
if !strings.Contains(section, "đề mục lạ") {
t.Error("a level-3 heading ended the section; only a level-2 heading may")
}
}
func TestVietnameseSectionAbsentAndBoth(t *testing.T) {
if _, dialect, _, _ := vietnameseSection("{{-eng-}}\n{{-noun-}}\n# Word."); dialect != "" {
t.Errorf("an English-only page reported dialect %q", dialect)
}
mixed := "== {{langname|vi}} ==\n# Mới.\n" + legacyPage
section, dialect, both, _ := vietnameseSection(mixed)
if dialect != "new" || !both {
t.Errorf("dialect = %q both = %v, want the first dialect in the text and both=true", dialect, both)
}
if strings.Contains(section, "Người học") {
t.Error("the first section should end where the legacy page starts a language section")
}
}
func TestDefinitionsLegacy(t *testing.T) {
stats := newSectionStats()
section, _, _, _ := vietnameseSection(legacyPage)
got := definitions(section, stats)
want := []sense{
{"danh từ", "Người học ở trường."},
{"danh từ", "(cũ) Môn đệ."},
{"động từ", "Đi học."},
}
assertSenses(t, got, want)
if stats.pos["noun"] != 1 || stats.pos["verb"] != 1 {
t.Errorf("pos tally = %v, want noun 1 verb 1", stats.pos)
}
if stats.pos["pron"] != 0 || stats.pos["trans"] != 0 {
t.Errorf("pronunciation and translations were tallied as parts of speech: %v", stats.pos)
}
if stats.defsKept != 3 {
t.Errorf("defsKept = %d, want 3", stats.defsKept)
}
}
func TestDefinitionsNewDialect(t *testing.T) {
stats := newSectionStats()
section, _, _, _ := vietnameseSection(newPage)
got := definitions(section, stats)
want := []sense{
{"danh từ riêng", "thủ đô, Việt Nam."},
{"động từ", "Đi về thủ đô."},
{"", "Một nghĩa dưới đề mục lạ."},
}
assertSenses(t, got, want)
if stats.unmappedPos["xyz"] != 1 {
t.Errorf("unmapped headings = %v, want xyz 1", stats.unmappedPos)
}
if stats.unmappedPos["etym"] != 0 {
t.Errorf("etymology counted as an unmapped part of speech: %v", stats.unmappedPos)
}
}
func TestDefinitionsHeadwordLineAndWrittenOutHeading(t *testing.T) {
section := "=== Danh từ ===\n# Một.\n{{vi-verb}}\n# Hai.\n{{vi-pron}}\n# Ba.\n=== Phát âm ===\n# Bốn."
got := definitions(section, newSectionStats())
want := []sense{{"danh từ", "Một."}, {"động từ", "Hai."}, {"động từ", "Ba."}, {"", "Bốn."}}
assertSenses(t, got, want)
}
func TestDefinitionsSkipsEmptyAndCaps(t *testing.T) {
stats := newSectionStats()
lines := []string{"{{-noun-}}", "# {{rfdef|vi}}", "#: not a definition", "#* nor this", "## nor this"}
for i := 0; i < 7; i++ {
lines = append(lines, "# Nghĩa số "+string(rune('a'+i))+".")
}
lines = append(lines, "# "+strings.Repeat("dài ", 80))
got := definitions(strings.Join(lines, "\n"), stats)
if len(got) != maxSenses {
t.Fatalf("got %d senses, want the cap of %d", len(got), maxSenses)
}
if got[0].gloss != "Nghĩa số a." {
t.Errorf("first sense = %q, want the first real definition after the empty one", got[0].gloss)
}
if stats.defsEmpty != 1 || stats.dropped["rfdef"] != 1 {
t.Errorf("empty = %d dropped = %v, want 1 and rfdef 1", stats.defsEmpty, stats.dropped)
}
if stats.defsKept != 8 || stats.defsCut != 1 {
t.Errorf("kept = %d cut = %d, want 8 and 1 (counted past the cap)", stats.defsKept, stats.defsCut)
}
}
func assertSenses(t *testing.T, got, want []sense) {
t.Helper()
if len(got) != len(want) {
t.Fatalf("got %d senses %v, want %d %v", len(got), got, len(want), want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("sense %d = %+v, want %+v", i, got[i], want[i])
}
}
}
// A few pages run the section marker and the headings together on one line:
// {{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}. The marker must still
// open the section and the last heading on the line must still label it.
func TestVietnameseSectionInlineHeadings(t *testing.T) {
page := "{{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}\n\n'''Thượng Hải'''\n# Thành phố lớn nhất [[Trung Quốc]].\n{{-eng-}}{{-noun-}}\n# Shanghai."
section, dialect, _, ender := vietnameseSection(page)
if dialect != "legacy" || ender != "eng" {
t.Fatalf("dialect = %q ender = %q, want legacy ended by eng", dialect, ender)
}
if strings.Contains(section, "Shanghai") {
t.Error("the English section, opened on a shared line, leaked in")
}
got := definitions(section, newSectionStats())
assertSenses(t, got, []sense{{"địa danh", "Thành phố lớn nhất Trung Quốc."}})
}
+1
View File
@@ -62,6 +62,7 @@ func run() error {
"path", cfg.dbPath,
"words", store.WordCount(),
"aliases", store.AliasCount(),
"meanings", store.MeaningCount(),
"license", store.License(),
)
+96 -12
View File
@@ -21,6 +21,7 @@ import (
"iter"
"net/url"
"os"
"slices"
"sort"
"strconv"
@@ -32,6 +33,10 @@ import (
// ErrNotFound is returned when a syllable has no entry in the dictionary.
var ErrNotFound = errors.New("dictionary: syllable not found")
// requiredBuilderVersion is the builder whose meta contract this store reads;
// it is named in the refusal of an older database.
const requiredBuilderVersion = "5"
// wordInfo holds the two syllables the chain rule needs. Both ends are kept:
// canonicalization can move either one, so the engine must never re-derive
// them from what the player typed.
@@ -40,6 +45,15 @@ type wordInfo struct {
last string
}
// Sense is one definition of a word as Wiktionary gives it: the Vietnamese
// part-of-speech label of the heading it sat under ("danh từ"), empty when the
// builder did not know the heading, and the definition as plain text. Neither
// is markup; the client renders both as text.
type Sense struct {
Pos string
Gloss string
}
// Store answers word and syllable queries against the derived dictionary.
//
// Every field is written once during Open and only read afterwards, and no
@@ -54,6 +68,10 @@ type Store struct {
// openers holds words whose last syllable has at least one continuation,
// sorted by that count descending so an eligible set is always a prefix.
openers []opener
// meanings holds each word's senses in page order. A few megabytes of text
// for the corpus; a per-move query would be a second code path for nothing.
meanings map[string][]Sense
meaningCount int
license string
}
@@ -87,11 +105,12 @@ func Open(path string) (*Store, error) {
aliases: make(map[string]string),
byFirst: make(map[string][]string),
outDegree: make(map[string]int),
meanings: make(map[string][]Sense),
}
// Reading meta first also rejects an unrelated database before any bulk
// loading happens.
declaredWords, err := s.loadMeta(db)
declaredWords, declaredMeanings, err := s.loadMeta(db)
if err != nil {
return nil, err
}
@@ -106,7 +125,10 @@ func Open(path string) (*Store, error) {
if err := s.loadAliases(db); err != nil {
return nil, err
}
if err := s.validate(declaredWords); err != nil {
if err := s.loadMeanings(db); err != nil {
return nil, err
}
if err := s.validate(declaredWords, declaredMeanings); err != nil {
return nil, err
}
@@ -125,22 +147,37 @@ func dsn(path string) string {
return u.String()
}
func (s *Store) loadMeta(db *sql.DB) (declaredWords int, err error) {
func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err error) {
// The data is CC BY-SA 4.0 and its provenance travels with it.
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&s.license); err != nil {
return 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
return 0, 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
}
var raw string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'word_count'`).Scan(&raw); err != nil {
return 0, fmt.Errorf("read dictionary word_count: %w", err)
count := func(key string) (int, error) {
var raw string
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
if errors.Is(err, sql.ErrNoRows) {
// A database from before the key existed: the fix is a rebuild,
// so say so rather than naming a missing row.
return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it",
key, requiredBuilderVersion)
}
return 0, fmt.Errorf("read dictionary %s: %w", key, err)
}
n, err := strconv.Atoi(raw)
if err != nil {
return 0, fmt.Errorf("dictionary %s %q is not a number: %w", key, raw, err)
}
return n, nil
}
declaredWords, err = strconv.Atoi(raw)
if err != nil {
return 0, fmt.Errorf("dictionary word_count %q is not a number: %w", raw, err)
if declaredWords, err = count("word_count"); err != nil {
return 0, 0, err
}
if declaredMeanings, err = count("meaning_count"); err != nil {
return 0, 0, err
}
return declaredWords, nil
return declaredWords, declaredMeanings, nil
}
func (s *Store) loadSyllables(db *sql.DB) error {
@@ -204,13 +241,35 @@ func (s *Store) loadAliases(db *sql.DB) error {
return rows.Err()
}
func (s *Store) loadMeanings(db *sql.DB) error {
// Ordered by (word, ord), the primary key, so each word's senses arrive in
// page order and append in it.
rows, err := db.Query(`SELECT word, pos, gloss FROM meanings ORDER BY word, ord`)
if err != nil {
return fmt.Errorf("load meanings: %w", err)
}
defer rows.Close()
for rows.Next() {
var word string
var sense Sense
if err := rows.Scan(&word, &sense.Pos, &sense.Gloss); err != nil {
return fmt.Errorf("scan meaning: %w", err)
}
s.meanings[word] = append(s.meanings[word], sense)
s.meaningCount++
}
return rows.Err()
}
// validate rejects a structurally valid but wrong dictionary.
//
// A truncated or empty database has the right schema and opens cleanly, and
// the server would then start, reject every word a player types, and fail
// every room creation. Checking the loaded rows against what the builder
// recorded turns that into a startup failure.
func (s *Store) validate(declaredWords int) error {
func (s *Store) validate(declaredWords, declaredMeanings int) error {
if len(s.words) != declaredWords {
return fmt.Errorf("dictionary is incomplete: metadata declares %d words, loaded %d",
declaredWords, len(s.words))
@@ -218,6 +277,17 @@ func (s *Store) validate(declaredWords int) error {
if len(s.words) == 0 {
return errors.New("dictionary contains no words")
}
// A meanings table truncated on disk would otherwise be served silently
// as a dictionary without meanings.
if s.meaningCount != declaredMeanings {
return fmt.Errorf("dictionary is incomplete: metadata declares %d meanings, loaded %d",
declaredMeanings, s.meaningCount)
}
for word := range s.meanings {
if _, ok := s.words[word]; !ok {
return fmt.Errorf("dictionary is inconsistent: meaning for %q, which is not a word", word)
}
}
// A stale syllables table would tell the bot a syllable has continuations
// that WordsStartingWith cannot supply.
@@ -243,6 +313,20 @@ func (s *Store) WordCount() int { return len(s.words) }
// AliasCount reports how many alternative spellings are accepted.
func (s *Store) AliasCount() int { return len(s.aliases) }
// MeaningCount reports how many senses the dictionary holds across all words.
func (s *Store) MeaningCount() int { return s.meaningCount }
// Meanings returns a canonical word's senses in page order, at most five, or
// nil for a word with none. Resolve first: an alias has no senses of its own.
// The slice is a copy, so a caller cannot reach dictionary state through it.
func (s *Store) Meanings(word string) []Sense {
senses := s.meanings[word]
if len(senses) == 0 {
return nil
}
return slices.Clone(senses)
}
// License reports the licence the dictionary data is distributed under.
// Callers are expected to state it at startup.
func (s *Store) License() string { return s.license }
+80 -4
View File
@@ -20,6 +20,7 @@ CREATE TABLE words (word TEXT PRIMARY KEY, first TEXT NOT NULL, last TEXT NOT NU
CREATE INDEX idx_words_first ON words(first);
CREATE TABLE syllables (syllable TEXT PRIMARY KEY, out_degree INTEGER NOT NULL) WITHOUT ROWID;
CREATE TABLE aliases (variant TEXT PRIMARY KEY, canonical TEXT NOT NULL) WITHOUT ROWID;
CREATE TABLE meanings (word TEXT NOT NULL, ord INTEGER NOT NULL, pos TEXT NOT NULL, gloss TEXT NOT NULL, PRIMARY KEY (word, ord)) WITHOUT ROWID;
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
`
@@ -40,7 +41,7 @@ func fixtureAt(tb testing.TB, dir string) string {
defer db.Close()
data := fixtureSchema + `
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7');
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7'),('meaning_count','3');
INSERT INTO words VALUES
('pháp luật','pháp','luật',2),
('pháp lý','pháp','lý',2),
@@ -55,6 +56,11 @@ INSERT INTO syllables VALUES
('pháp',2),('luật',1),('lý',1),('lệ',0),('do',0),('vô',1),('điện',0),('công',1),('dầu',0),('ngữ',1);
-- "pháp lí" drifts in the LAST syllable, "luâto lệ" in the FIRST.
INSERT INTO aliases VALUES ('pháp lí','pháp lý'),('luâto lệ','luật lệ');
-- Inserted out of order to prove the store sorts by ord, not by insertion.
INSERT INTO meanings VALUES
('pháp luật',1,'','Kỷ cương nói chung.'),
('pháp luật',0,'danh từ','Hệ thống các quy tắc xử sự do nhà nước đặt ra.'),
('ngữ pháp',0,'danh từ','Toàn bộ những quy tắc hoạt động của ngôn ngữ.');
`
if _, err := db.Exec(data); err != nil {
tb.Fatal(err)
@@ -106,7 +112,7 @@ func TestOpenWrongSchema(t *testing.T) {
// every room creation.
func TestOpenEmptyDictionary(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999');`)
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999'),('meaning_count','0');`)
_, err := Open(path)
if err == nil {
@@ -121,7 +127,7 @@ INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999')
// syllable has continuations that cannot be supplied.
func TestOpenInconsistentOutDegree(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
@@ -132,7 +138,7 @@ INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
func TestOpenOrphanAlias(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
INSERT INTO aliases VALUES ('phap luat','không tồn tại');`)
@@ -541,3 +547,73 @@ func BenchmarkRandomOpeningWord(b *testing.B) {
}
}
}
func TestMeaningsAreOrderedAndCopied(t *testing.T) {
s := fixture(t)
got := s.Meanings("pháp luật")
want := []Sense{
{Pos: "danh từ", Gloss: "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
{Pos: "", Gloss: "Kỷ cương nói chung."},
}
if !slices.Equal(got, want) {
t.Errorf("Meanings(pháp luật) = %v, want %v (ordered by ord, not insertion)", got, want)
}
// A caller writing into the slice must not reach the store.
got[0].Gloss = "changed"
if s.Meanings("pháp luật")[0].Gloss != want[0].Gloss {
t.Error("Meanings handed out the store's own slice")
}
if s.Meanings("pháp lý") != nil {
t.Error("a word with no senses returned a non-nil slice")
}
// An alias is not a word: callers Resolve first.
if s.Meanings("pháp lí") != nil {
t.Error("an alias returned senses of its own")
}
if s.MeaningCount() != 3 {
t.Errorf("MeaningCount = %d, want 3", s.MeaningCount())
}
}
// A meanings table truncated on disk must be refused at startup rather than
// served silently as a dictionary without meanings.
func TestOpenRefusesMismatchedMeaningCount(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','2');
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
INSERT INTO meanings VALUES ('pháp luật',0,'danh từ','Luật.');`)
_, err := Open(path)
if err == nil || !strings.Contains(err.Error(), "meanings") {
t.Fatalf("Open = %v, want a refusal naming the meanings count", err)
}
}
// A database built before meanings existed opens cleanly and has every table
// but one row. The refusal must say what to do, not which row is missing.
func TestOpenRefusesOlderBuilderVersion(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
INSERT INTO syllables VALUES ('pháp',1),('luật',0);`)
_, err := Open(path)
if err == nil || !strings.Contains(err.Error(), "make dict") {
t.Fatalf("Open = %v, want a refusal that says to rebuild", err)
}
}
func TestOpenRefusesOrphanMeaning(t *testing.T) {
path := writeDB(t, fixtureSchema+`
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','1');
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
INSERT INTO meanings VALUES ('không tồn tại',0,'','Một nghĩa.');`)
if _, err := Open(path); err == nil {
t.Fatal("Open succeeded with a meaning for a word that does not exist")
}
}