Files
noitu/server/cmd/build-dictionary/dump_test.go
T
tiennm99 80f216d56b feat(dict): build the corpus and word meanings from the Wikimedia viwiktionary dump
reader for both wikitext dialects; `meanings(word, ord, pos, gloss)` table; `meaning_count`/`words_with_meaning`/`source_pages` in meta, `source_rows` gone, builder_version 5; `--dump`/`--min-pages` replace `--kaikki`; attribution names the dump and the definition excerpts; 36,200 words, 96.9% with a meaning, every kaikki word kept.
2026-09-08 22:48:26 +07:00

150 lines
5.0 KiB
Go

package main
import (
"crypto/sha256"
"encoding/hex"
"os"
"path/filepath"
"strings"
"testing"
)
// miniDump is a twelve-page stand-in for the Wikimedia dump, committed
// compressed beside its readable source. Go has no bzip2 writer, so the .bz2
// is regenerated by hand: bzip2 -k9 testdata/mini-dump.xml.
const miniDump = "testdata/mini-dump.xml.bz2"
// miniDumpCut is the same XML cut mid-page and then compressed: a valid bzip2
// stream whose XML ends early, as distinct from a truncated download.
const miniDumpCut = "testdata/mini-dump-cut.xml.bz2"
func TestReadDumpKeepsVietnameseSections(t *testing.T) {
words, meanings, rejects, stats, prov, err := readDump(miniDump)
if err != nil {
t.Fatal(err)
}
var got []string
for w := range words {
got = append(got, w)
}
assertSameStrings(t, got, []string{"pháp luật", "hòa bình", "luật lệ", "ngôn ngữ", "ngữ pháp", "vô tuyến điện"})
if stats.pages != 12 || stats.ns0 != 11 || stats.redirects != 1 || stats.noVietnamese != 1 {
t.Errorf("pages %d ns0 %d redirects %d noVietnamese %d, want 12 11 1 1",
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
}
if stats.legacy != 7 || stats.newDialect != 2 || stats.bothDialects != 0 {
t.Errorf("legacy %d new %d both %d, want 7 2 0", stats.legacy, stats.newDialect, stats.bothDialects)
}
if prov.pages != 9 {
t.Errorf("source pages = %d, want 9 (Vietnamese sections, redirect excluded)", prov.pages)
}
if stats.merged != 1 {
t.Errorf("merged = %d, want 1 (Hòa Bình and hòa bình)", stats.merged)
}
if rejects[rejectNotVietnamese] != 1 || rejects[rejectTooShort] != 1 || rejects[rejectDigit] != 1 {
t.Errorf("rejects = %v, want one each of not-Vietnamese, too-short, digit", rejects)
}
assertSenses(t, meanings["pháp luật"], []sense{
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
{"động từ", "(hiếm) Xử theo luật."},
})
// Capitalized page first, lowercase page second: senses in page order,
// the new-dialect place first.
assertSenses(t, meanings["hòa bình"], []sense{
{"danh từ riêng", "tỉnh, Việt Nam."},
{"danh từ", "Tình trạng không có chiến tranh."},
{"tính từ", "Yên ổn."},
})
assertSenses(t, meanings["ngôn ngữ"], []sense{{"danh từ", "Hệ thống những âm, từ và quy tắc kết hợp chúng."}})
if _, has := meanings["luật lệ"]; has {
t.Error("a definition that is only an unknown template produced a sense")
}
if stats.section.dropped["rfdef"] != 1 || stats.section.defsEmpty != 1 {
t.Errorf("dropped = %v empty = %d, want rfdef 1 and 1", stats.section.dropped, stats.section.defsEmpty)
}
if stats.section.pos["noun"] == 0 || stats.section.pos["pr-noun"] != 2 || stats.section.pos["n"] != 1 {
t.Errorf("pos tally = %v", stats.section.pos)
}
}
func TestReadDumpHashesTheBytesItRead(t *testing.T) {
_, _, _, _, prov, err := readDump(miniDump)
if err != nil {
t.Fatal(err)
}
raw, err := os.ReadFile(miniDump)
if err != nil {
t.Fatal(err)
}
sum := sha256.Sum256(raw)
if prov.sha256 != hex.EncodeToString(sum[:]) {
t.Errorf("sha256 = %s, want %s (the whole file)", prov.sha256, hex.EncodeToString(sum[:]))
}
if prov.fetchedAt.IsZero() {
t.Error("fetchedAt is zero")
}
}
func TestReadDumpRejectsNonBzip2(t *testing.T) {
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
if err := os.WriteFile(path, []byte("<mediawiki></mediawiki>"), 0o644); err != nil {
t.Fatal(err)
}
_, _, _, _, _, err := readDump(path)
if err == nil || !strings.Contains(err.Error(), "not a bzip2 file") {
t.Fatalf("err = %v, want a message naming the missing bzip2 header", err)
}
}
func TestReadDumpRejectsTruncatedDownload(t *testing.T) {
raw, err := os.ReadFile(miniDump)
if err != nil {
t.Fatal(err)
}
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
if err := os.WriteFile(path, raw[:len(raw)/2], 0o644); err != nil {
t.Fatal(err)
}
_, _, _, _, _, err = readDump(path)
if err == nil {
t.Fatal("a half-downloaded dump was read without error")
}
if !strings.Contains(err.Error(), "truncated") {
t.Errorf("err = %v, want it to suggest a truncated download", err)
}
}
func TestReadDumpRejectsStreamEndingMidPage(t *testing.T) {
_, _, _, _, _, err := readDump(miniDumpCut)
if err == nil {
t.Fatal("an XML stream ending mid-page was read without error")
}
if !strings.Contains(err.Error(), `after page "hello world"`) {
t.Errorf("err = %v, want it to name the last page fully read", err)
}
}
func TestRunFailsBelowMinPages(t *testing.T) {
cfg := config{
dump: miniDump,
out: filepath.Join(t.TempDir(), "noitu.db"),
minWords: 1,
minPages: 20000,
}
err := run(cfg)
if err == nil || !strings.Contains(err.Error(), "pages have a Vietnamese section") {
t.Fatalf("err = %v, want the page floor named", err)
}
}
func TestFormatTally(t *testing.T) {
got := formatTally(map[string]int{"b": 2, "a": 2, "": 5, "c": 1}, 3)
if want := "(none) 5, a 2, b 2"; got != want {
t.Errorf("formatTally = %q, want %q", got, want)
}
}