mirror of
https://github.com/tiennm99/noitu.git
synced 2026-10-11 03:13:45 +00:00
feat(dict): build the corpus and word meanings from the Wikimedia viwiktionary dump
reader for both wikitext dialects; `meanings(word, ord, pos, gloss)` table; `meaning_count`/`words_with_meaning`/`source_pages` in meta, `source_rows` gone, builder_version 5; `--dump`/`--min-pages` replace `--kaikki`; attribution names the dump and the definition excerpts; 36,200 words, 96.9% with a meaning, every kaikki word kept.
This commit is contained in:
1 parent
557de1af94
commit
80f216d56b
25 files changed
+2338
-757
No files matched your search
@@ -0,0 +1,270 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"compress/bzip2"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/xml"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The upstream is the Wikimedia dump of Wiktionary tiếng Việt: every page's
|
||||
// current wikitext, as one bzip2-compressed XML file regenerated monthly.
|
||||
// `latest/` is a rolling pointer, fetched fresh for every build and not
|
||||
// pinned, so the builder records the SHA-256 of the bytes it actually read and
|
||||
// that hash is what identifies a build. Dated directories exist should
|
||||
// reproducibility ever be wanted.
|
||||
const dumpSourceURL = "https://dumps.wikimedia.org/viwiktionary/latest/viwiktionary-latest-pages-articles.xml.bz2"
|
||||
|
||||
// dumpPage is the part of a <page> element the builder reads. Everything
|
||||
// else — contributor, timestamp, sha1 — is skipped by the decoder.
|
||||
type dumpPage struct {
|
||||
Title string `xml:"title"`
|
||||
Ns int `xml:"ns"`
|
||||
Redirect *struct {
|
||||
Title string `xml:"title,attr"`
|
||||
} `xml:"redirect"`
|
||||
Revisions []struct {
|
||||
Text string `xml:"text"`
|
||||
} `xml:"revision"`
|
||||
}
|
||||
|
||||
// dumpProvenance identifies the bytes a build was made from.
|
||||
type dumpProvenance struct {
|
||||
sha256 string
|
||||
// pages is the number of pages with a Vietnamese section, redirects
|
||||
// excluded: the count of entries the corpus was derived from.
|
||||
pages int
|
||||
fetchedAt time.Time
|
||||
}
|
||||
|
||||
// dumpStats is what the build log reports about the dump beyond the reject
|
||||
// tally: enough to see a month where a dialect vanished or a stripper rule
|
||||
// started dropping everything.
|
||||
type dumpStats struct {
|
||||
pages int
|
||||
ns0 int
|
||||
redirects int
|
||||
noVietnamese int
|
||||
legacy int
|
||||
newDialect int
|
||||
bothDialects int
|
||||
merged int // pages whose title normalized to a word already seen
|
||||
// enders counts the {{-code-}} that closed each legacy section. Language
|
||||
// codes are expected here; a heading code is one the maps are missing.
|
||||
enders map[string]int
|
||||
section *sectionStats
|
||||
}
|
||||
|
||||
// readDump streams the dump once: hashes the compressed bytes, decodes one
|
||||
// page at a time, hands every Vietnamese-section title to accept() and every
|
||||
// definition line to the stripper.
|
||||
func readDump(path string) (map[string]entry, map[string][]sense, map[rejectReason]int, *dumpStats, dumpProvenance, error) {
|
||||
var prov dumpProvenance
|
||||
stats := &dumpStats{section: newSectionStats(), enders: make(map[string]int)}
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("read dump: %w", err)
|
||||
}
|
||||
defer f.Close()
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
// A provenance row must be right or absent, never a plausible zero.
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("stat dump: %w", err)
|
||||
}
|
||||
prov.fetchedAt = info.ModTime().UTC()
|
||||
|
||||
hash := sha256.New()
|
||||
compressed := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
|
||||
if magic, err := compressed.Peek(3); err != nil || string(magic) != "BZh" {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s is not a bzip2 file (expected a BZh header)", path)
|
||||
}
|
||||
dec := xml.NewDecoder(bzip2.NewReader(compressed))
|
||||
|
||||
words := make(map[string]entry)
|
||||
meanings := make(map[string][]sense)
|
||||
rejects := make(map[rejectReason]int)
|
||||
lastTitle := ""
|
||||
|
||||
for {
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) {
|
||||
break
|
||||
}
|
||||
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
|
||||
}
|
||||
start, ok := tok.(xml.StartElement)
|
||||
if !ok || start.Name.Local != "page" {
|
||||
continue
|
||||
}
|
||||
var page dumpPage
|
||||
if err := dec.DecodeElement(&page, &start); err != nil {
|
||||
return nil, nil, nil, stats, prov, dumpError(path, lastTitle, dec.InputOffset(), err)
|
||||
}
|
||||
lastTitle = page.Title
|
||||
stats.pages++
|
||||
if page.Ns != 0 {
|
||||
continue
|
||||
}
|
||||
stats.ns0++
|
||||
if page.Redirect != nil {
|
||||
// The target page is read on its own and lowercased by accept(),
|
||||
// so a case-only redirect adds nothing and any other redirect is
|
||||
// an alternative title the wiki itself does not define.
|
||||
stats.redirects++
|
||||
continue
|
||||
}
|
||||
if len(page.Revisions) == 0 {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s: page %q has no revision text", path, page.Title)
|
||||
}
|
||||
text := page.Revisions[len(page.Revisions)-1].Text
|
||||
|
||||
section, dialect, both, ender := vietnameseSection(text)
|
||||
if dialect == "" {
|
||||
stats.noVietnamese++
|
||||
rejects[rejectNotVietnamese]++
|
||||
continue
|
||||
}
|
||||
if both {
|
||||
stats.bothDialects++
|
||||
}
|
||||
if dialect == "legacy" {
|
||||
stats.legacy++
|
||||
} else {
|
||||
stats.newDialect++
|
||||
}
|
||||
prov.pages++
|
||||
if ender != "" {
|
||||
stats.enders[ender]++
|
||||
}
|
||||
|
||||
word, syllables, reason, ok := accept(page.Title)
|
||||
if !ok {
|
||||
rejects[reason]++
|
||||
continue
|
||||
}
|
||||
// After accept, so the definition counters describe words that land.
|
||||
senses := definitions(section, stats.section)
|
||||
if _, seen := words[word]; seen {
|
||||
// Two pages whose titles normalize to one word (Việt Nam and
|
||||
// việt nam): one entry, senses in page order, one cap.
|
||||
stats.merged++
|
||||
}
|
||||
words[word] = entry{
|
||||
word: word,
|
||||
first: syllables[0],
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
if len(senses) > 0 {
|
||||
merged := append(meanings[word], senses...)
|
||||
if len(merged) > maxSenses {
|
||||
merged = merged[:maxSenses]
|
||||
}
|
||||
meanings[word] = merged
|
||||
}
|
||||
}
|
||||
|
||||
// The XML decoder stops at the root's close tag; the hash must cover the
|
||||
// whole file, trailing bytes included.
|
||||
if _, err := io.Copy(io.Discard, compressed); err != nil {
|
||||
return nil, nil, nil, stats, prov, fmt.Errorf("%s: %w", path, err)
|
||||
}
|
||||
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
|
||||
|
||||
return words, meanings, rejects, stats, prov, nil
|
||||
}
|
||||
|
||||
// dumpError names where a stream failed: the last page fully read and the
|
||||
// decompressed offset, so a truncated download and a malformed page are told
|
||||
// apart by the message alone.
|
||||
func dumpError(path, lastTitle string, offset int64, err error) error {
|
||||
where := "before the first page"
|
||||
if lastTitle != "" {
|
||||
where = fmt.Sprintf("after page %q", lastTitle)
|
||||
}
|
||||
if errors.Is(err, io.ErrUnexpectedEOF) || strings.Contains(err.Error(), "unexpected EOF") {
|
||||
return fmt.Errorf("%s: stream ends %s (decompressed offset %d): truncated download? %w", path, where, offset, err)
|
||||
}
|
||||
return fmt.Errorf("%s: %s (decompressed offset %d): %w", path, where, offset, err)
|
||||
}
|
||||
|
||||
// logDumpStats writes the build log lines that describe what the dump held.
|
||||
func logDumpStats(stats *dumpStats) {
|
||||
s := stats.section
|
||||
logf := log.Printf
|
||||
logf("pages %d, in the main namespace %d, redirects skipped %d, without a Vietnamese section %d",
|
||||
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
|
||||
logf("Vietnamese sections: legacy {{-vie-}} %d, new == {{langname|vi}} == %d, pages with both %d, titles merged %d",
|
||||
stats.legacy, stats.newDialect, stats.bothDialects, stats.merged)
|
||||
logf("parts of speech: %s", formatTally(s.pos, 0))
|
||||
if len(s.unmappedPos) > 0 {
|
||||
logf("headings without a label: %s", formatTally(s.unmappedPos, 20))
|
||||
}
|
||||
// Language codes belong here. A heading code in this list is one the maps
|
||||
// do not know, and it has been cutting sections short.
|
||||
logf("codes that ended a legacy section, commonest: %s", formatTally(stats.enders, 15))
|
||||
logf("definitions kept %d (cut at %d characters: %d), dropped as empty after stripping %d",
|
||||
s.defsKept, maxGlossRunes, s.defsCut, s.defsEmpty)
|
||||
if len(s.dropped) > 0 {
|
||||
logf("templates dropped whole, commonest: %s", formatTally(s.dropped, 10))
|
||||
}
|
||||
}
|
||||
|
||||
// formatTally renders counts on one log line, largest first, cut to the top
|
||||
// n entries when n is positive.
|
||||
func formatTally(counts map[string]int, n int) string {
|
||||
type kv struct {
|
||||
name string
|
||||
count int
|
||||
}
|
||||
tally := make([]kv, 0, len(counts))
|
||||
for name, count := range counts {
|
||||
if name == "" {
|
||||
name = "(none)"
|
||||
}
|
||||
tally = append(tally, kv{name, count})
|
||||
}
|
||||
sort.Slice(tally, func(i, j int) bool {
|
||||
if tally[i].count != tally[j].count {
|
||||
return tally[i].count > tally[j].count
|
||||
}
|
||||
return tally[i].name < tally[j].name
|
||||
})
|
||||
if n > 0 && len(tally) > n {
|
||||
tally = tally[:n]
|
||||
}
|
||||
parts := make([]string, len(tally))
|
||||
for i, t := range tally {
|
||||
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// dumpSourceSpec describes a dump build for the meta table. With nothing
|
||||
// pinned upstream, the hash and page count of the bytes read are the
|
||||
// provenance.
|
||||
func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec {
|
||||
return sourceSpec{
|
||||
table: "dump:" + filepath.Base(path),
|
||||
url: dumpSourceURL,
|
||||
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
|
||||
extra: [][2]string{
|
||||
{"source_sha256", prov.sha256},
|
||||
{"source_pages", fmt.Sprint(prov.pages)},
|
||||
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,149 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// miniDump is a twelve-page stand-in for the Wikimedia dump, committed
|
||||
// compressed beside its readable source. Go has no bzip2 writer, so the .bz2
|
||||
// is regenerated by hand: bzip2 -k9 testdata/mini-dump.xml.
|
||||
const miniDump = "testdata/mini-dump.xml.bz2"
|
||||
|
||||
// miniDumpCut is the same XML cut mid-page and then compressed: a valid bzip2
|
||||
// stream whose XML ends early, as distinct from a truncated download.
|
||||
const miniDumpCut = "testdata/mini-dump-cut.xml.bz2"
|
||||
|
||||
func TestReadDumpKeepsVietnameseSections(t *testing.T) {
|
||||
words, meanings, rejects, stats, prov, err := readDump(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var got []string
|
||||
for w := range words {
|
||||
got = append(got, w)
|
||||
}
|
||||
assertSameStrings(t, got, []string{"pháp luật", "hòa bình", "luật lệ", "ngôn ngữ", "ngữ pháp", "vô tuyến điện"})
|
||||
|
||||
if stats.pages != 12 || stats.ns0 != 11 || stats.redirects != 1 || stats.noVietnamese != 1 {
|
||||
t.Errorf("pages %d ns0 %d redirects %d noVietnamese %d, want 12 11 1 1",
|
||||
stats.pages, stats.ns0, stats.redirects, stats.noVietnamese)
|
||||
}
|
||||
if stats.legacy != 7 || stats.newDialect != 2 || stats.bothDialects != 0 {
|
||||
t.Errorf("legacy %d new %d both %d, want 7 2 0", stats.legacy, stats.newDialect, stats.bothDialects)
|
||||
}
|
||||
if prov.pages != 9 {
|
||||
t.Errorf("source pages = %d, want 9 (Vietnamese sections, redirect excluded)", prov.pages)
|
||||
}
|
||||
if stats.merged != 1 {
|
||||
t.Errorf("merged = %d, want 1 (Hòa Bình and hòa bình)", stats.merged)
|
||||
}
|
||||
if rejects[rejectNotVietnamese] != 1 || rejects[rejectTooShort] != 1 || rejects[rejectDigit] != 1 {
|
||||
t.Errorf("rejects = %v, want one each of not-Vietnamese, too-short, digit", rejects)
|
||||
}
|
||||
|
||||
assertSenses(t, meanings["pháp luật"], []sense{
|
||||
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
|
||||
{"động từ", "(hiếm) Xử theo luật."},
|
||||
})
|
||||
// Capitalized page first, lowercase page second: senses in page order,
|
||||
// the new-dialect place first.
|
||||
assertSenses(t, meanings["hòa bình"], []sense{
|
||||
{"danh từ riêng", "tỉnh, Việt Nam."},
|
||||
{"danh từ", "Tình trạng không có chiến tranh."},
|
||||
{"tính từ", "Yên ổn."},
|
||||
})
|
||||
assertSenses(t, meanings["ngôn ngữ"], []sense{{"danh từ", "Hệ thống những âm, từ và quy tắc kết hợp chúng."}})
|
||||
if _, has := meanings["luật lệ"]; has {
|
||||
t.Error("a definition that is only an unknown template produced a sense")
|
||||
}
|
||||
if stats.section.dropped["rfdef"] != 1 || stats.section.defsEmpty != 1 {
|
||||
t.Errorf("dropped = %v empty = %d, want rfdef 1 and 1", stats.section.dropped, stats.section.defsEmpty)
|
||||
}
|
||||
if stats.section.pos["noun"] == 0 || stats.section.pos["pr-noun"] != 2 || stats.section.pos["n"] != 1 {
|
||||
t.Errorf("pos tally = %v", stats.section.pos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpHashesTheBytesItRead(t *testing.T) {
|
||||
_, _, _, _, prov, err := readDump(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sum := sha256.Sum256(raw)
|
||||
if prov.sha256 != hex.EncodeToString(sum[:]) {
|
||||
t.Errorf("sha256 = %s, want %s (the whole file)", prov.sha256, hex.EncodeToString(sum[:]))
|
||||
}
|
||||
if prov.fetchedAt.IsZero() {
|
||||
t.Error("fetchedAt is zero")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsNonBzip2(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
|
||||
if err := os.WriteFile(path, []byte("<mediawiki></mediawiki>"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, _, _, _, _, err := readDump(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "not a bzip2 file") {
|
||||
t.Fatalf("err = %v, want a message naming the missing bzip2 header", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsTruncatedDownload(t *testing.T) {
|
||||
raw, err := os.ReadFile(miniDump)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
path := filepath.Join(t.TempDir(), "dump.xml.bz2")
|
||||
if err := os.WriteFile(path, raw[:len(raw)/2], 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
_, _, _, _, _, err = readDump(path)
|
||||
if err == nil {
|
||||
t.Fatal("a half-downloaded dump was read without error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "truncated") {
|
||||
t.Errorf("err = %v, want it to suggest a truncated download", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadDumpRejectsStreamEndingMidPage(t *testing.T) {
|
||||
_, _, _, _, _, err := readDump(miniDumpCut)
|
||||
if err == nil {
|
||||
t.Fatal("an XML stream ending mid-page was read without error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), `after page "hello world"`) {
|
||||
t.Errorf("err = %v, want it to name the last page fully read", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailsBelowMinPages(t *testing.T) {
|
||||
cfg := config{
|
||||
dump: miniDump,
|
||||
out: filepath.Join(t.TempDir(), "noitu.db"),
|
||||
minWords: 1,
|
||||
minPages: 20000,
|
||||
}
|
||||
err := run(cfg)
|
||||
if err == nil || !strings.Contains(err.Error(), "pages have a Vietnamese section") {
|
||||
t.Fatalf("err = %v, want the page floor named", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFormatTally(t *testing.T) {
|
||||
got := formatTally(map[string]int{"b": 2, "a": 2, "": 5, "c": 1}, 3)
|
||||
if want := "(none) 5, a 2, b 2"; got != want {
|
||||
t.Errorf("formatTally = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -1,168 +0,0 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The upstream is kaikki.org's wiktextract export of Wiktionary tiếng Việt:
|
||||
// one JSON object per entry, refreshed from the monthly Wikimedia dump about
|
||||
// once a week. The file is fetched fresh for every build and is not pinned —
|
||||
// there is no archived snapshot to pin to — so the builder records the SHA-256
|
||||
// of the bytes it actually read, and that hash is what identifies a build.
|
||||
//
|
||||
// The URL stays percent-encoded: the path has a space in it, and both make
|
||||
// and sh would otherwise split it.
|
||||
const kaikkiSourceURL = "https://kaikki.org/viwiktionary/Ti%E1%BA%BFng%20Vi%E1%BB%87t/kaikki.org-dictionary-Ti%E1%BA%BFngVi%E1%BB%87t.jsonl"
|
||||
|
||||
// kaikkiRow is the part of a wiktextract entry the game cares about. Every
|
||||
// other field — senses, translations, categories — is skipped by the decoder.
|
||||
type kaikkiRow struct {
|
||||
Word string `json:"word"`
|
||||
Pos string `json:"pos"`
|
||||
LangCode string `json:"lang_code"`
|
||||
}
|
||||
|
||||
// kaikkiProvenance identifies the bytes a build was made from.
|
||||
type kaikkiProvenance struct {
|
||||
sha256 string
|
||||
rows int
|
||||
fetchedAt time.Time
|
||||
}
|
||||
|
||||
// readKaikkiList streams the export, keeps Vietnamese-language entries and
|
||||
// hands their word forms to accept(). Part of speech is tallied for the build
|
||||
// log but never filters: the owner's decision that capitalization removes no
|
||||
// word applies equally to the "name" tag.
|
||||
//
|
||||
// Lines are read with bufio.Reader rather than bufio.Scanner because a row
|
||||
// carries every sense and translation of its entry and can run to hundreds of
|
||||
// kilobytes; a scanner's fixed cap would be a guess that eventually fails.
|
||||
func readKaikkiList(path string) (map[string]entry, map[rejectReason]int, map[string]int, kaikkiProvenance, error) {
|
||||
var prov kaikkiProvenance
|
||||
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return nil, nil, nil, prov, fmt.Errorf("read kaikki export: %w", err)
|
||||
}
|
||||
defer f.Close()
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
// A provenance row must be right or absent, never a plausible zero.
|
||||
return nil, nil, nil, prov, fmt.Errorf("stat kaikki export: %w", err)
|
||||
}
|
||||
prov.fetchedAt = info.ModTime().UTC()
|
||||
|
||||
hash := sha256.New()
|
||||
reader := bufio.NewReaderSize(io.TeeReader(f, hash), 1<<20)
|
||||
|
||||
words := make(map[string]entry)
|
||||
rejects := make(map[rejectReason]int)
|
||||
pos := make(map[string]int)
|
||||
|
||||
lineNo := 0
|
||||
for {
|
||||
line, err := reader.ReadBytes('\n')
|
||||
if len(line) > 0 {
|
||||
lineNo++
|
||||
if trimmed := bytes.TrimSpace(line); len(trimmed) > 0 {
|
||||
// A bare literal such as null would decode into an empty row
|
||||
// and be miscounted as a foreign-language entry; only objects
|
||||
// are entries.
|
||||
if trimmed[0] != '{' {
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: not a JSON object", path, lineNo)
|
||||
}
|
||||
var row kaikkiRow
|
||||
if err := json.Unmarshal(trimmed, &row); err != nil {
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: malformed line: %w", path, lineNo, err)
|
||||
}
|
||||
prov.rows++
|
||||
if row.LangCode != "vi" {
|
||||
rejects[rejectNotVietnamese]++
|
||||
} else {
|
||||
pos[row.Pos]++
|
||||
if word, syllables, reason, ok := accept(row.Word); !ok {
|
||||
rejects[reason]++
|
||||
} else {
|
||||
words[word] = entry{
|
||||
word: word,
|
||||
first: syllables[0],
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
if errors.Is(err, io.EOF) {
|
||||
break
|
||||
}
|
||||
// The failure is on the line being read: the one just counted if
|
||||
// a partial line came back with the error, otherwise the next.
|
||||
failed := lineNo + 1
|
||||
if len(line) > 0 {
|
||||
failed = lineNo
|
||||
}
|
||||
return nil, nil, nil, prov, fmt.Errorf("%s:%d: %w", path, failed, err)
|
||||
}
|
||||
}
|
||||
prov.sha256 = hex.EncodeToString(hash.Sum(nil))
|
||||
|
||||
return words, rejects, pos, prov, nil
|
||||
}
|
||||
|
||||
// formatPosTally renders the part-of-speech counts on one log line, largest
|
||||
// first, so the build log says what kind of entries the export held.
|
||||
func formatPosTally(pos map[string]int) string {
|
||||
type kv struct {
|
||||
name string
|
||||
count int
|
||||
}
|
||||
tally := make([]kv, 0, len(pos))
|
||||
for name, count := range pos {
|
||||
if name == "" {
|
||||
name = "(none)"
|
||||
}
|
||||
tally = append(tally, kv{name, count})
|
||||
}
|
||||
sort.Slice(tally, func(i, j int) bool {
|
||||
if tally[i].count != tally[j].count {
|
||||
return tally[i].count > tally[j].count
|
||||
}
|
||||
return tally[i].name < tally[j].name
|
||||
})
|
||||
parts := make([]string, len(tally))
|
||||
for i, t := range tally {
|
||||
parts[i] = fmt.Sprintf("%s %d", t.name, t.count)
|
||||
}
|
||||
return strings.Join(parts, ", ")
|
||||
}
|
||||
|
||||
// kaikkiSourceSpec describes a kaikki build for the meta table. With no commit
|
||||
// or checksum pinned upstream, the hash and row count of the bytes read are the
|
||||
// provenance.
|
||||
func kaikkiSourceSpec(path string, prov kaikkiProvenance) sourceSpec {
|
||||
return sourceSpec{
|
||||
table: "kaikki:" + filepath.Base(path),
|
||||
url: kaikkiSourceURL,
|
||||
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
|
||||
extra: [][2]string{
|
||||
{"source_sha256", prov.sha256},
|
||||
{"source_rows", fmt.Sprint(prov.rows)},
|
||||
{"source_fetched_at", prov.fetchedAt.Format(time.RFC3339)},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -1,243 +0,0 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// fixtureKaikki writes a miniature stand-in for the kaikki export: the same
|
||||
// JSONL shape, a handful of rows.
|
||||
func fixtureKaikki(t *testing.T, lines ...string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
|
||||
if err := os.WriteFile(path, []byte(strings.Join(lines, "\n")+"\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func defaultKaikkiLines() []string {
|
||||
return []string{
|
||||
`{"word": "Hà Nội", "pos": "name", "lang_code": "vi", "senses": [{"glosses": ["thủ đô"]}]}`, // capitalized, name POS — kept, lowercased
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
|
||||
`{"word": "học sinh", "pos": "verb", "lang_code": "vi"}`, // same word, second POS — kept once
|
||||
`{"word": "student", "pos": "noun", "lang_code": "en"}`, // not Vietnamese-language — rejected and counted
|
||||
`{"word": "pháp", "pos": "noun", "lang_code": "vi"}`, // single syllable — rejected downstream
|
||||
``,
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListKeepsVietnameseEntries(t *testing.T) {
|
||||
words, rejects, pos, prov, err := readKaikkiList(fixtureKaikki(t, defaultKaikkiLines()...))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var got []string
|
||||
for w := range words {
|
||||
got = append(got, w)
|
||||
}
|
||||
assertSameStrings(t, got, []string{"hà nội", "học sinh"})
|
||||
|
||||
if n := rejects[rejectNotVietnamese]; n != 1 {
|
||||
t.Errorf("non-Vietnamese rejects = %d, want 1", n)
|
||||
}
|
||||
if n := rejects[rejectTooShort]; n != 1 {
|
||||
t.Errorf("too-short rejects = %d, want 1 (pháp)", n)
|
||||
}
|
||||
if pos["noun"] != 2 || pos["verb"] != 1 || pos["name"] != 1 {
|
||||
t.Errorf("pos tally = %v, want noun 2, verb 1, name 1 (en row excluded)", pos)
|
||||
}
|
||||
if prov.rows != 5 {
|
||||
t.Errorf("rows = %d, want 5", prov.rows)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListHashesTheBytesItRead(t *testing.T) {
|
||||
path := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
_, _, _, prov, err := readKaikkiList(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
sum := sha256.Sum256(raw)
|
||||
if want := hex.EncodeToString(sum[:]); prov.sha256 != want {
|
||||
t.Errorf("sha256 = %s, want %s", prov.sha256, want)
|
||||
}
|
||||
if prov.fetchedAt.IsZero() {
|
||||
t.Error("fetchedAt is zero, want the file's modification time")
|
||||
}
|
||||
}
|
||||
|
||||
// fixtureKaikkiRaw writes exact bytes, for the shapes fixtureKaikki's trailing
|
||||
// newline would hide.
|
||||
func fixtureKaikkiRaw(t *testing.T, raw string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "kaikki.jsonl")
|
||||
if err := os.WriteFile(path, []byte(raw), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestKaikkiListHandlesDownloadShapes(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
raw string
|
||||
wantWords int
|
||||
wantRows int
|
||||
wantErr string
|
||||
}{
|
||||
{"final line without newline",
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh mì", "pos": "noun", "lang_code": "vi"}`,
|
||||
2, 2, ""},
|
||||
{"HTTP error page instead of JSONL", "<html><body>503</body></html>\n", 0, 0, ":1: malformed"},
|
||||
{"cut mid-line", `{"word": "học sinh", "pos": "noun", "lang_code": "vi"}` + "\n" + `{"word": "bánh`, 0, 0, ":2: malformed"},
|
||||
{"bare JSON literal", "null\n", 0, 0, ":1: malformed"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
words, _, _, prov, err := readKaikkiList(fixtureKaikkiRaw(t, tc.raw))
|
||||
if tc.wantErr != "" {
|
||||
if err == nil || !strings.Contains(err.Error(), tc.wantErr) {
|
||||
t.Fatalf("err = %v, want one containing %q", err, tc.wantErr)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(words) != tc.wantWords || prov.rows != tc.wantRows {
|
||||
t.Errorf("words=%d rows=%d, want %d/%d", len(words), prov.rows, tc.wantWords, tc.wantRows)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListNamesMalformedLine(t *testing.T) {
|
||||
path := fixtureKaikki(t,
|
||||
`{"word": "học sinh", "pos": "noun", "lang_code": "vi"}`,
|
||||
`{"word": "broken"`,
|
||||
)
|
||||
_, _, _, _, err := readKaikkiList(path)
|
||||
if err == nil {
|
||||
t.Fatal("malformed line was skipped, want error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), ":2:") {
|
||||
t.Errorf("error does not name line 2: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiListReadsLongLines(t *testing.T) {
|
||||
// A real row carries every sense and translation and can exceed any
|
||||
// scanner buffer; the reader must not have a line cap.
|
||||
padding := strings.Repeat("x", 2<<20)
|
||||
path := fixtureKaikki(t, `{"word": "học sinh", "pos": "noun", "lang_code": "vi", "note": "`+padding+`"}`)
|
||||
words, _, _, _, err := readKaikkiList(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok := words["học sinh"]; !ok {
|
||||
t.Error("word on a 2 MB line was lost")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFormatPosTally(t *testing.T) {
|
||||
got := formatPosTally(map[string]int{"verb": 2, "noun": 5, "": 1})
|
||||
if want := "noun 5, verb 2, (none) 1"; got != want {
|
||||
t.Errorf("tally = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestKaikkiBuildRecordsProvenance(t *testing.T) {
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
path := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
if err := run(config{kaikki: path, out: out, minWords: 1}); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
db := openOut(t, out)
|
||||
|
||||
raw, _ := os.ReadFile(path)
|
||||
sum := sha256.Sum256(raw)
|
||||
want := map[string]string{
|
||||
"source_url": kaikkiSourceURL,
|
||||
"source_sha256": hex.EncodeToString(sum[:]),
|
||||
"source_rows": "5",
|
||||
"source_license": "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
|
||||
"word_count": "2",
|
||||
}
|
||||
for key, value := range want {
|
||||
var got string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&got); err != nil {
|
||||
t.Errorf("meta[%q] missing: %v", key, err)
|
||||
continue
|
||||
}
|
||||
if got != value {
|
||||
t.Errorf("meta[%q] = %q, want %q", key, got, value)
|
||||
}
|
||||
}
|
||||
for _, gone := range []string{"source_commit", "sources_kept", "sources_excluded"} {
|
||||
var got string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, gone).Scan(&got); err == nil {
|
||||
t.Errorf("meta[%q] = %q, want absent", gone, got)
|
||||
}
|
||||
}
|
||||
var fetched string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_fetched_at'`).Scan(&fetched); err != nil || fetched == "" {
|
||||
t.Errorf("meta[source_fetched_at] missing or empty: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunInputSelection(t *testing.T) {
|
||||
kaikki := fixtureKaikki(t, defaultKaikkiLines()...)
|
||||
words := fixtureKaikki(t, "học sinh")
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
cfg config
|
||||
wantErr string
|
||||
}{
|
||||
{"both inputs", config{kaikki: kaikki, words: words, out: out, minWords: 1}, "mutually exclusive"},
|
||||
{"neither input", config{out: out, minWords: 1}, "no input given"},
|
||||
{"missing kaikki file", config{kaikki: filepath.Join(t.TempDir(), "absent.jsonl"), out: out, minWords: 1}, "make fetch-dict"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
err := run(tc.cfg)
|
||||
if err == nil {
|
||||
t.Fatal("run succeeded, want error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), tc.wantErr) {
|
||||
t.Errorf("error %q does not mention %q", err, tc.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A fixture is hand-written data; its database must not claim the upstream's
|
||||
// licence, because the server logs whatever the meta table says.
|
||||
func TestWordListBuildRecordsNoUpstreamLicense(t *testing.T) {
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
if err := run(config{words: fixtureKaikki(t, "học sinh", "bánh mì"), out: out, minWords: 1}); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
}
|
||||
var license, url string
|
||||
db := openOut(t, out)
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_url'`).Scan(&url); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if strings.Contains(license, "CC BY-SA") || url != "" {
|
||||
t.Errorf("fixture build claims upstream provenance: license=%q url=%q", license, url)
|
||||
}
|
||||
}
|
||||
@@ -1,19 +1,20 @@
|
||||
// Command build-dictionary derives the game's wordlist from kaikki.org's
|
||||
// wiktextract export of Wiktionary tiếng Việt.
|
||||
// Command build-dictionary derives the game's wordlist and word meanings from
|
||||
// the Wikimedia dump of Wiktionary tiếng Việt.
|
||||
//
|
||||
// The upstream is a ~62 MB JSONL file: one entry per line with its senses,
|
||||
// translations and part of speech. The game needs only Vietnamese word forms
|
||||
// of at least two syllables, indexed by first and last syllable. This tool
|
||||
// performs that reduction and records provenance in a meta table — including
|
||||
// the SHA-256 of the file it read, since the upstream is fetched fresh for
|
||||
// every build rather than pinned.
|
||||
// The upstream is a ~61 MB bzip2-compressed XML file: every page of the wiki
|
||||
// with its current wikitext, regenerated monthly. The game needs the
|
||||
// Vietnamese word forms of at least two syllables, indexed by first and last
|
||||
// syllable, and the plain text of each word's definitions. This tool performs
|
||||
// that reduction and records provenance in a meta table — including the
|
||||
// SHA-256 of the file it read, since the upstream is fetched fresh for every
|
||||
// build rather than pinned.
|
||||
//
|
||||
// The derived database is a modified version of CC BY-SA 4.0 licensed data.
|
||||
// See data/ATTRIBUTION.md.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
// go run ./cmd/build-dictionary --kaikki ../data/kaikki-viwiktionary-vi.jsonl --out ../data/noitu.db
|
||||
// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
|
||||
package main
|
||||
|
||||
import (
|
||||
@@ -27,32 +28,45 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// builderVer changes whenever the meta table's contract does, so two databases
|
||||
// with different provenance rows never claim the same builder.
|
||||
const builderVer = "4"
|
||||
const builderVer = "5"
|
||||
|
||||
// minMeaningCoverage is the share of words a dump build must carry a meaning
|
||||
// for. The 2026-09-01 dump measured well above it; the floor exists to catch a
|
||||
// stripper or section scanner that suddenly returns nothing, not to demand
|
||||
// quality. Fixture builds are exempt: their meanings are hand-written.
|
||||
const minMeaningCoverage = 0.6
|
||||
|
||||
type config struct {
|
||||
// kaikki is the corpus: the upstream wiktextract JSONL export.
|
||||
kaikki string
|
||||
// words is an alternative source: a plain list, one word per line, used to
|
||||
// build a small fixture database without the upstream download.
|
||||
// dump is the corpus: the Wikimedia pages-articles export.
|
||||
dump string
|
||||
// words is an alternative source: a plain list, one word per line with an
|
||||
// optional tab-separated meaning column, used to build a small fixture
|
||||
// database without the upstream download.
|
||||
words string
|
||||
out string
|
||||
minWords int
|
||||
// minPages is the floor on pages with a Vietnamese section. Distinct from
|
||||
// minWords so a scanner that silently misses a dialect is caught before
|
||||
// the word floor is.
|
||||
minPages int
|
||||
}
|
||||
|
||||
func main() {
|
||||
log.SetFlags(0)
|
||||
|
||||
var cfg config
|
||||
flag.StringVar(&cfg.kaikki, "kaikki", "", "upstream kaikki.org wiktextract JSONL export to read")
|
||||
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the upstream export (one word per line, # comments)")
|
||||
flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read")
|
||||
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)")
|
||||
flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write")
|
||||
flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering")
|
||||
flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section")
|
||||
flag.Parse()
|
||||
|
||||
if err := run(cfg); err != nil {
|
||||
@@ -64,40 +78,47 @@ func run(cfg config) error {
|
||||
// Exactly one input. Picking silently between two would let a stray flag
|
||||
// ship a corpus nobody meant to build.
|
||||
switch {
|
||||
case cfg.kaikki == "" && cfg.words == "":
|
||||
return errors.New("no input given: pass --kaikki (the corpus) or --words (a plain list)")
|
||||
case cfg.kaikki != "" && cfg.words != "":
|
||||
return errors.New("--kaikki and --words are mutually exclusive")
|
||||
case cfg.kaikki != "":
|
||||
return runFromKaikkiList(cfg)
|
||||
case cfg.dump == "" && cfg.words == "":
|
||||
return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)")
|
||||
case cfg.dump != "" && cfg.words != "":
|
||||
return errors.New("--dump and --words are mutually exclusive")
|
||||
case cfg.dump != "":
|
||||
return runFromDump(cfg)
|
||||
default:
|
||||
return runFromWordList(cfg)
|
||||
}
|
||||
}
|
||||
|
||||
// runFromKaikkiList derives the database from the kaikki.org export, keeping
|
||||
// Vietnamese-language entries and recording the hash of the bytes it read.
|
||||
func runFromKaikkiList(cfg config) error {
|
||||
if _, err := os.Stat(cfg.kaikki); err != nil {
|
||||
return fmt.Errorf("kaikki export not found at %s — run 'make fetch-dict' first: %w", cfg.kaikki, err)
|
||||
// runFromDump derives the database from the Wikimedia dump, keeping every
|
||||
// page with a Vietnamese section and recording the hash of the bytes it read.
|
||||
func runFromDump(cfg config) error {
|
||||
if _, err := os.Stat(cfg.dump); err != nil {
|
||||
return fmt.Errorf("dump not found at %s — run 'make fetch-dict' first: %w", cfg.dump, err)
|
||||
}
|
||||
|
||||
words, rejects, pos, prov, err := readKaikkiList(cfg.kaikki)
|
||||
started := time.Now()
|
||||
words, meanings, rejects, stats, prov, err := readDump(cfg.dump)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
logDumpStats(stats)
|
||||
logRejects(rejects)
|
||||
log.Printf("parts of speech: %s", formatPosTally(pos))
|
||||
log.Printf("accepted %d distinct words from %s (%d rows, sha256 %s)", len(words), cfg.kaikki, prov.rows, prov.sha256)
|
||||
if prov.pages < cfg.minPages {
|
||||
return fmt.Errorf("only %d pages have a Vietnamese section, expected at least %d — "+
|
||||
"the dump's markup may have changed", prov.pages, cfg.minPages)
|
||||
}
|
||||
log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s",
|
||||
len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second))
|
||||
|
||||
return finish(cfg, words, kaikkiSourceSpec(cfg.kaikki, prov))
|
||||
return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true)
|
||||
}
|
||||
|
||||
// finish is the tail every input mode shares: the size floor, alias
|
||||
// generation, the atomic write and the re-read verification. Keeping it in one
|
||||
// place is what stops a fixture from drifting into a different shape from the
|
||||
// database production loads.
|
||||
func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
// database production loads. requireCoverage applies the meaning-coverage
|
||||
// floor, which only a corpus build can be held to.
|
||||
func finish(cfg config, words map[string]entry, meanings map[string][]sense, src sourceSpec, requireCoverage bool) error {
|
||||
if len(words) < cfg.minWords {
|
||||
return fmt.Errorf("only %d words survived filtering, expected at least %d — "+
|
||||
"the source content may have changed", len(words), cfg.minWords)
|
||||
@@ -106,10 +127,10 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
aliases, collisions := buildAliases(words)
|
||||
log.Printf("generated %d spelling aliases (%d skipped as ambiguous or already real words)", len(aliases), collisions)
|
||||
|
||||
if err := write(cfg.out, words, aliases, src); err != nil {
|
||||
if err := write(cfg.out, words, meanings, aliases, src); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := verify(cfg.out, cfg.minWords); err != nil {
|
||||
if err := verify(cfg.out, cfg.minWords, requireCoverage); err != nil {
|
||||
return fmt.Errorf("output failed verification: %w", err)
|
||||
}
|
||||
|
||||
@@ -118,13 +139,17 @@ func finish(cfg config, words map[string]entry, src sourceSpec) error {
|
||||
}
|
||||
|
||||
// runFromWordList derives a database from a plain list of words instead of the
|
||||
// upstream export.
|
||||
// dump.
|
||||
//
|
||||
// It exists so tests and CI have a real dictionary to play against without the
|
||||
// upstream download. The filtering, alias generation, writing and verification
|
||||
// below are the same functions the real build uses — only the source of the
|
||||
// raw strings differs — so a fixture cannot drift into being shaped
|
||||
// differently from what production loads.
|
||||
//
|
||||
// A line is `word`, or `word<TAB>sense<TAB>sense…` where a sense is
|
||||
// `pos|gloss` or just `gloss`. The pipe never survives the stripper, so it is
|
||||
// a safe separator for hand-written meanings.
|
||||
func runFromWordList(cfg config) error {
|
||||
raw, err := os.ReadFile(cfg.words)
|
||||
if err != nil {
|
||||
@@ -132,6 +157,7 @@ func runFromWordList(cfg config) error {
|
||||
}
|
||||
|
||||
words := make(map[string]entry)
|
||||
meanings := make(map[string][]sense)
|
||||
rejects := make(map[rejectReason]int)
|
||||
|
||||
for line := range strings.Lines(string(raw)) {
|
||||
@@ -139,7 +165,8 @@ func runFromWordList(cfg config) error {
|
||||
if line == "" || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
word, syllables, reason, ok := accept(line)
|
||||
cells := strings.Split(line, "\t")
|
||||
word, syllables, reason, ok := accept(cells[0])
|
||||
if !ok {
|
||||
rejects[reason]++
|
||||
continue
|
||||
@@ -150,26 +177,54 @@ func runFromWordList(cfg config) error {
|
||||
last: syllables[len(syllables)-1],
|
||||
syllables: len(syllables),
|
||||
}
|
||||
if senses := parseSenses(cells[1:]); len(senses) > 0 {
|
||||
meanings[word] = senses
|
||||
}
|
||||
}
|
||||
|
||||
logRejects(rejects)
|
||||
log.Printf("accepted %d distinct words from %s", len(words), cfg.words)
|
||||
log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words)
|
||||
|
||||
// The source spec is what lands in the meta table. Naming the list rather
|
||||
// than a table makes it obvious in the output which build produced a given
|
||||
// database — and a hand-written list carries no upstream licence, so the
|
||||
// fixture must not claim one.
|
||||
return finish(cfg, words, sourceSpec{
|
||||
return finish(cfg, words, meanings, sourceSpec{
|
||||
table: "wordlist:" + filepath.Base(cfg.words),
|
||||
license: "none: hand-written fixture wordlist, no upstream data",
|
||||
attribution: "Fixture written by this project; no third-party attribution applies.",
|
||||
})
|
||||
}, false)
|
||||
}
|
||||
|
||||
// parseSenses reads the tab-separated meaning cells of a fixture line. A cell
|
||||
// is `pos|gloss` or a bare gloss; empty cells are skipped and the cap applies
|
||||
// as it does to the dump.
|
||||
func parseSenses(cells []string) []sense {
|
||||
var senses []sense
|
||||
for _, cell := range cells {
|
||||
cell = strings.TrimSpace(cell)
|
||||
if cell == "" {
|
||||
continue
|
||||
}
|
||||
s := sense{gloss: cell}
|
||||
if pos, gloss, ok := strings.Cut(cell, "|"); ok {
|
||||
s = sense{pos: strings.TrimSpace(pos), gloss: strings.TrimSpace(gloss)}
|
||||
}
|
||||
if s.gloss == "" {
|
||||
continue
|
||||
}
|
||||
s.gloss, _ = capGloss(s.gloss)
|
||||
if len(senses) < maxSenses {
|
||||
senses = append(senses, s)
|
||||
}
|
||||
}
|
||||
return senses
|
||||
}
|
||||
|
||||
// verify re-opens the finished database and re-checks the invariants the game
|
||||
// depends on. The in-memory checks above can only prove what the builder
|
||||
// intended; this proves what actually landed on disk.
|
||||
func verify(path string, minWords int) error {
|
||||
func verify(path string, minWords int, requireCoverage bool) error {
|
||||
db, err := sql.Open("sqlite", "file:"+path+"?mode=ro")
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -193,6 +248,15 @@ func verify(path string, minWords int) error {
|
||||
{"words whose first syllable is missing from the syllables table",
|
||||
`SELECT COUNT(*) FROM words w LEFT JOIN syllables s ON s.syllable = w.first WHERE s.syllable IS NULL`,
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"meanings whose word is missing from the words table",
|
||||
`SELECT COUNT(*) FROM meanings m LEFT JOIN words w ON w.word = m.word WHERE w.word IS NULL`,
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"meanings with an empty gloss", `SELECT COUNT(*) FROM meanings WHERE gloss = ''`, func(n int) bool { return n == 0 }},
|
||||
{"meanings over the length cap", fmt.Sprintf(`SELECT COUNT(*) FROM meanings WHERE LENGTH(gloss) > %d`, maxGlossRunes),
|
||||
func(n int) bool { return n == 0 }},
|
||||
{"words with more meanings than the cap",
|
||||
fmt.Sprintf(`SELECT COUNT(*) FROM (SELECT word FROM meanings GROUP BY word HAVING COUNT(*) > %d)`, maxSenses),
|
||||
func(n int) bool { return n == 0 }},
|
||||
}
|
||||
|
||||
for _, c := range checks {
|
||||
@@ -205,6 +269,20 @@ func verify(path string, minWords int) error {
|
||||
}
|
||||
}
|
||||
|
||||
if requireCoverage {
|
||||
var wordCount, withMeaning int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&wordCount); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := db.QueryRow(`SELECT COUNT(DISTINCT word) FROM meanings`).Scan(&withMeaning); err != nil {
|
||||
return err
|
||||
}
|
||||
if float64(withMeaning) < minMeaningCoverage*float64(wordCount) {
|
||||
return fmt.Errorf("only %d of %d words have a meaning, expected at least %.0f%% — "+
|
||||
"the dump's definition markup may have changed", withMeaning, wordCount, minMeaningCoverage*100)
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -218,7 +296,7 @@ type entry struct {
|
||||
|
||||
// sourceSpec is what the meta table records about where the words came from.
|
||||
type sourceSpec struct {
|
||||
// table names the input: "kaikki:<file>" for the corpus, "wordlist:<file>"
|
||||
// table names the input: "dump:<file>" for the corpus, "wordlist:<file>"
|
||||
// for a fixture, so the output says which build produced it.
|
||||
table string
|
||||
// url is the upstream artifact; empty for fixture builds.
|
||||
@@ -281,7 +359,7 @@ func buildAliases(words map[string]entry) (map[string]string, int) {
|
||||
// partway through -- a full disk, an interrupt -- leaves an empty but
|
||||
// syntactically valid database where a good one used to be, which the server
|
||||
// would happily open and find no words in.
|
||||
func write(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
|
||||
func write(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
|
||||
tmp := path + ".tmp"
|
||||
if err := os.Remove(tmp); err != nil && !errors.Is(err, os.ErrNotExist) {
|
||||
return fmt.Errorf("remove stale temp file: %w", err)
|
||||
@@ -294,7 +372,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
|
||||
}
|
||||
}()
|
||||
|
||||
if err := writeTo(tmp, words, aliases, src); err != nil {
|
||||
if err := writeTo(tmp, words, meanings, aliases, src); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -311,7 +389,7 @@ func write(path string, words map[string]entry, aliases map[string]string, src s
|
||||
return nil
|
||||
}
|
||||
|
||||
func writeTo(path string, words map[string]entry, aliases map[string]string, src sourceSpec) error {
|
||||
func writeTo(path string, words map[string]entry, meanings map[string][]sense, aliases map[string]string, src sourceSpec) error {
|
||||
db, err := sql.Open("sqlite", "file:"+path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create output: %w", err)
|
||||
@@ -337,6 +415,17 @@ CREATE TABLE aliases (
|
||||
canonical TEXT NOT NULL
|
||||
) WITHOUT ROWID;
|
||||
|
||||
-- One row per sense, in page order. pos is the Vietnamese part-of-speech
|
||||
-- label of the heading the definition sat under, '' when the heading was one
|
||||
-- the builder does not know. No foreign key pragma: verify() checks the join.
|
||||
CREATE TABLE meanings (
|
||||
word TEXT NOT NULL,
|
||||
ord INTEGER NOT NULL,
|
||||
pos TEXT NOT NULL,
|
||||
gloss TEXT NOT NULL,
|
||||
PRIMARY KEY (word, ord)
|
||||
) WITHOUT ROWID;
|
||||
|
||||
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
`
|
||||
if _, err := db.Exec(schema); err != nil {
|
||||
@@ -390,6 +479,27 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
}
|
||||
}
|
||||
|
||||
insertMeaning, err := tx.Prepare(`INSERT INTO meanings (word, ord, pos, gloss) VALUES (?, ?, ?, ?)`)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer insertMeaning.Close()
|
||||
meaningCount := 0
|
||||
for word, senses := range meanings {
|
||||
if _, isWord := words[word]; !isWord {
|
||||
return fmt.Errorf("meaning for %q, which is not a word", word)
|
||||
}
|
||||
for ord, s := range senses {
|
||||
if s.gloss == "" || utf8.RuneCountInString(s.gloss) > maxGlossRunes {
|
||||
return fmt.Errorf("meaning %d of %q is empty or over the cap", ord, word)
|
||||
}
|
||||
if _, err := insertMeaning.Exec(word, ord, s.pos, s.gloss); err != nil {
|
||||
return fmt.Errorf("insert meaning %d of %q: %w", ord, word, err)
|
||||
}
|
||||
meaningCount++
|
||||
}
|
||||
}
|
||||
|
||||
insertMeta, err := tx.Prepare(`INSERT INTO meta (key, value) VALUES (?, ?)`)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -403,6 +513,8 @@ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
{"builder_version", builderVer},
|
||||
{"word_count", fmt.Sprint(len(words))},
|
||||
{"alias_count", fmt.Sprint(len(aliases))},
|
||||
{"meaning_count", fmt.Sprint(meaningCount)},
|
||||
{"words_with_meaning", fmt.Sprint(len(meanings))},
|
||||
{"source_table", src.table},
|
||||
}
|
||||
meta = append(meta, src.extra...)
|
||||
|
||||
@@ -6,47 +6,25 @@ import (
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// fixtureSource writes a miniature stand-in for the kaikki export: the same
|
||||
// JSONL shape, a handful of rows instead of 44k. Each row is a word and the
|
||||
// language its Wiktionary entry is for. Tests never touch the real download.
|
||||
func fixtureSource(t *testing.T, rows [][2]string) string {
|
||||
t.Helper()
|
||||
|
||||
lines := make([]string, 0, len(rows))
|
||||
for _, r := range rows {
|
||||
lines = append(lines, `{"word": "`+r[0]+`", "pos": "noun", "lang_code": "`+r[1]+`"}`)
|
||||
}
|
||||
return fixtureKaikki(t, lines...)
|
||||
}
|
||||
|
||||
func defaultRows() [][2]string {
|
||||
return [][2]string{
|
||||
{"pháp luật", "vi"},
|
||||
{"pháp luật", "vi"}, // listed twice — must dedupe to one word
|
||||
{"luật lệ", "vi"},
|
||||
{"ngôn ngữ", "vi"},
|
||||
{"ngữ pháp", "vi"},
|
||||
{"hòa bình", "vi"},
|
||||
{"vô tuyến điện", "vi"}, // three syllables
|
||||
{"pháp", "vi"}, // single syllable — rejected
|
||||
{"covid 19", "vi"}, // digit — rejected
|
||||
{"hello world", "en"}, // another language's entry — never selected
|
||||
}
|
||||
}
|
||||
|
||||
func buildFixture(t *testing.T, rows [][2]string) string {
|
||||
// buildFixture runs the whole pipeline on the committed mini dump: twelve
|
||||
// pages, both dialects, a redirect, an English-only page and two pages that
|
||||
// accept() rejects. Tests never touch the real download.
|
||||
func buildFixture(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
out := filepath.Join(t.TempDir(), "noitu.db")
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, rows),
|
||||
dump: miniDump,
|
||||
out: out,
|
||||
minWords: 1,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err != nil {
|
||||
t.Fatalf("run: %v", err)
|
||||
@@ -64,23 +42,23 @@ func openOut(t *testing.T, path string) *sql.DB {
|
||||
return db
|
||||
}
|
||||
|
||||
func count(t *testing.T, db *sql.DB, query string, args ...any) int {
|
||||
t.Helper()
|
||||
var n int
|
||||
if err := db.QueryRow(query, args...).Scan(&n); err != nil {
|
||||
t.Fatalf("%s: %v", query, err)
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func TestBuildProducesExpectedWords(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
var count int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words`).Scan(&count); err != nil {
|
||||
t.Fatal(err)
|
||||
if got := count(t, db, `SELECT COUNT(*) FROM words`); got != 6 {
|
||||
t.Errorf("word count = %d, want 6", got)
|
||||
}
|
||||
if want := 6; count != want {
|
||||
t.Errorf("word count = %d, want %d", count, want)
|
||||
}
|
||||
|
||||
// Every stored word must have at least two syllables.
|
||||
var short int
|
||||
if err := db.QueryRow(`SELECT COUNT(*) FROM words WHERE syllables < 2`).Scan(&short); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if short != 0 {
|
||||
if short := count(t, db, `SELECT COUNT(*) FROM words WHERE syllables < 2`); short != 0 {
|
||||
t.Errorf("%d words have fewer than 2 syllables, want 0", short)
|
||||
}
|
||||
|
||||
@@ -98,7 +76,7 @@ func TestBuildProducesExpectedWords(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestBuildComputesOutDegree(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
// "pháp luật" and "pháp" (rejected) mean exactly one word starts with "pháp".
|
||||
assertOutDegree(t, db, "pháp", 1)
|
||||
@@ -120,7 +98,7 @@ func assertOutDegree(t *testing.T, db *sql.DB, syllable string, want int) {
|
||||
}
|
||||
|
||||
func TestBuildWritesAliases(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
var canonical string
|
||||
err := db.QueryRow(`SELECT canonical FROM aliases WHERE variant = ?`, "hoà bình").Scan(&canonical)
|
||||
@@ -132,22 +110,65 @@ func TestBuildWritesAliases(t *testing.T) {
|
||||
}
|
||||
|
||||
// Every alias must point at a word that actually exists.
|
||||
var orphans int
|
||||
err = db.QueryRow(`SELECT COUNT(*) FROM aliases a
|
||||
LEFT JOIN words w ON w.word = a.canonical
|
||||
WHERE w.word IS NULL`).Scan(&orphans)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
orphans := count(t, db, `SELECT COUNT(*) FROM aliases a LEFT JOIN words w ON w.word = a.canonical WHERE w.word IS NULL`)
|
||||
if orphans != 0 {
|
||||
t.Errorf("%d aliases point at missing words, want 0", orphans)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildRecordsProvenance(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t, defaultRows()))
|
||||
func TestBuildWritesMeanings(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count"} {
|
||||
rows, err := db.Query(`SELECT ord, pos, gloss FROM meanings WHERE word = ? ORDER BY ord`, "pháp luật")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var got []sense
|
||||
for rows.Next() {
|
||||
var ord int
|
||||
var s sense
|
||||
if err := rows.Scan(&ord, &s.pos, &s.gloss); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if ord != len(got) {
|
||||
t.Errorf("ord = %d, want %d (0-based, dense)", ord, len(got))
|
||||
}
|
||||
got = append(got, s)
|
||||
}
|
||||
assertSenses(t, got, []sense{
|
||||
{"danh từ", "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{"danh từ", "(nghĩa rộng) Kỷ cương nói chung."},
|
||||
{"động từ", "(hiếm) Xử theo luật."},
|
||||
})
|
||||
|
||||
// A word whose only definition stripped to nothing has no rows, and the
|
||||
// meta counts describe the table.
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM meanings WHERE word = ?`, "luật lệ"); n != 0 {
|
||||
t.Errorf("luật lệ has %d meanings, want 0", n)
|
||||
}
|
||||
total := count(t, db, `SELECT COUNT(*) FROM meanings`)
|
||||
withMeaning := count(t, db, `SELECT COUNT(DISTINCT word) FROM meanings`)
|
||||
for key, want := range map[string]int{"meaning_count": total, "words_with_meaning": withMeaning} {
|
||||
var raw string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
|
||||
t.Fatalf("meta[%q]: %v", key, err)
|
||||
}
|
||||
if raw != strconv.Itoa(want) {
|
||||
t.Errorf("meta[%q] = %s, want %d", key, raw, want)
|
||||
}
|
||||
}
|
||||
if withMeaning != 5 {
|
||||
t.Errorf("words with a meaning = %d, want 5 of 6", withMeaning)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildRecordsProvenance(t *testing.T) {
|
||||
db := openOut(t, buildFixture(t))
|
||||
|
||||
want := map[string]string{"builder_version": builderVer, "source_url": dumpSourceURL, "source_pages": "9"}
|
||||
for _, key := range []string{"source_url", "source_license", "attribution", "built_at", "word_count",
|
||||
"builder_version", "source_sha256", "source_pages", "source_fetched_at", "meaning_count", "words_with_meaning"} {
|
||||
var value string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&value); err != nil {
|
||||
t.Errorf("meta[%q] missing: %v", key, err)
|
||||
@@ -156,26 +177,42 @@ func TestBuildRecordsProvenance(t *testing.T) {
|
||||
if value == "" {
|
||||
t.Errorf("meta[%q] is empty", key)
|
||||
}
|
||||
if w, ok := want[key]; ok && value != w {
|
||||
t.Errorf("meta[%q] = %q, want %q", key, value, w)
|
||||
}
|
||||
}
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM meta WHERE key = 'source_rows'`); n != 0 {
|
||||
t.Error("source_rows belonged to the previous source format and must be gone")
|
||||
}
|
||||
}
|
||||
|
||||
// The floor exists so a schema change upstream fails the build loudly instead
|
||||
// The floor exists so a markup change upstream fails the build loudly instead
|
||||
// of silently shipping a near-empty dictionary.
|
||||
func TestBuildFailsBelowMinWords(t *testing.T) {
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, defaultRows()),
|
||||
dump: miniDump,
|
||||
out: filepath.Join(t.TempDir(), "noitu.db"),
|
||||
minWords: 1000,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err == nil {
|
||||
t.Fatal("run succeeded with an unreachable min-words floor, want error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRequiresExactlyOneInput(t *testing.T) {
|
||||
if err := run(config{}); err == nil {
|
||||
t.Error("run with no input succeeded")
|
||||
}
|
||||
if err := run(config{dump: miniDump, words: "x.txt"}); err == nil || !strings.Contains(err.Error(), "mutually exclusive") {
|
||||
t.Errorf("run with both inputs: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A failed build must leave the previous good database untouched. Building in
|
||||
// place would delete it and leave an empty file the server would happily open.
|
||||
func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
out := buildFixture(t, defaultRows())
|
||||
out := buildFixture(t)
|
||||
|
||||
before, err := os.ReadFile(out)
|
||||
if err != nil {
|
||||
@@ -184,9 +221,10 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
|
||||
// Same output path, but a floor no fixture can clear.
|
||||
cfg := config{
|
||||
kaikki: fixtureSource(t, defaultRows()),
|
||||
dump: miniDump,
|
||||
out: out,
|
||||
minWords: 1000,
|
||||
minPages: 1,
|
||||
}
|
||||
if err := run(cfg); err == nil {
|
||||
t.Fatal("run succeeded with an unreachable floor, want error")
|
||||
@@ -203,3 +241,127 @@ func TestFailedBuildPreservesPreviousOutput(t *testing.T) {
|
||||
t.Error("temp database left behind after a failed build")
|
||||
}
|
||||
}
|
||||
|
||||
// --- the fixture word list --------------------------------------------------
|
||||
|
||||
func writeWordList(t *testing.T, content string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "words.txt")
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestWordListCarriesTabSeparatedMeanings(t *testing.T) {
|
||||
list := writeWordList(t, "# comment\n"+
|
||||
"học sinh\tdanh từ|Người học ở trường.\tđộng từ|Đi học.\n"+
|
||||
"sinh viên\tNgười học ở trường đại học.\n"+
|
||||
"sinh hoạt\n"+
|
||||
"sinh sản\t\t\n")
|
||||
out := filepath.Join(t.TempDir(), "fixture.db")
|
||||
if err := run(config{words: list, out: out, minWords: 1}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
db := openOut(t, out)
|
||||
|
||||
rows, err := db.Query(`SELECT word, ord, pos, gloss FROM meanings ORDER BY word, ord`)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
type row struct {
|
||||
word string
|
||||
ord int
|
||||
s sense
|
||||
}
|
||||
var got []row
|
||||
for rows.Next() {
|
||||
var r row
|
||||
if err := rows.Scan(&r.word, &r.ord, &r.s.pos, &r.s.gloss); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got = append(got, r)
|
||||
}
|
||||
want := []row{
|
||||
{"học sinh", 0, sense{"danh từ", "Người học ở trường."}},
|
||||
{"học sinh", 1, sense{"động từ", "Đi học."}},
|
||||
{"sinh viên", 0, sense{"", "Người học ở trường đại học."}},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("meanings = %+v, want %+v", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("row %d = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
if n := count(t, db, `SELECT COUNT(*) FROM words`); n != 4 {
|
||||
t.Errorf("words = %d, want 4 (a line without a tab is still a word)", n)
|
||||
}
|
||||
// Fixture builds carry no upstream licence and are exempt from the
|
||||
// coverage floor: two of four words have a meaning here.
|
||||
var license string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&license); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.HasPrefix(license, "none") {
|
||||
t.Errorf("fixture licence = %q, want a statement that no upstream data applies", license)
|
||||
}
|
||||
}
|
||||
|
||||
// --- verify -----------------------------------------------------------------
|
||||
|
||||
// brokenDB writes a database that passes every schema check and then breaks
|
||||
// one invariant, to prove verify() reads what is on disk.
|
||||
func brokenDB(t *testing.T, extraSQL string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "broken.db")
|
||||
words := map[string]entry{"pháp luật": {"pháp luật", "pháp", "luật", 2}}
|
||||
meanings := map[string][]sense{"pháp luật": {{"danh từ", "Luật."}}}
|
||||
if err := writeTo(path, words, meanings, nil, sourceSpec{table: "test"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
db, err := sql.Open("sqlite", "file:"+path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer db.Close()
|
||||
if _, err := db.Exec(extraSQL); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
func TestVerifyRejectsBrokenMeanings(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"orphan meaning row": `INSERT INTO meanings VALUES ('không có', 0, '', 'Một nghĩa.')`,
|
||||
"empty gloss": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '')`,
|
||||
"over the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', '` + strings.Repeat("a", maxGlossRunes+1) + `')`,
|
||||
"more senses than the cap": `INSERT INTO meanings VALUES ('pháp luật', 1, '', 'b'), ('pháp luật', 2, '', 'c'),
|
||||
('pháp luật', 3, '', 'd'), ('pháp luật', 4, '', 'e'), ('pháp luật', 5, '', 'f')`,
|
||||
}
|
||||
for name, sqlText := range cases {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
path := brokenDB(t, sqlText)
|
||||
if err := verify(path, 1, false); err == nil {
|
||||
t.Error("verify passed a database that breaks a meanings invariant")
|
||||
}
|
||||
})
|
||||
}
|
||||
if err := verify(brokenDB(t, `SELECT 1`), 1, false); err != nil {
|
||||
t.Errorf("verify rejected a sound database: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestVerifyCoverageFloorAppliesToCorpusBuildsOnly(t *testing.T) {
|
||||
// One word with a meaning, one without: 50%, under the floor.
|
||||
path := brokenDB(t, `INSERT INTO words VALUES ('luật lệ', 'luật', 'lệ', 2);
|
||||
INSERT INTO syllables VALUES ('lệ', 0); UPDATE syllables SET out_degree = 1 WHERE syllable = 'luật'`)
|
||||
if err := verify(path, 1, true); err == nil || !strings.Contains(err.Error(), "have a meaning") {
|
||||
t.Errorf("corpus verify with 50%% coverage: %v, want the coverage floor named", err)
|
||||
}
|
||||
if err := verify(path, 1, false); err != nil {
|
||||
t.Errorf("fixture verify applied the coverage floor: %v", err)
|
||||
}
|
||||
}
|
||||
Binary file not shown.
+182
@@ -0,0 +1,182 @@
|
||||
<mediawiki xmlns="http://www.mediawiki.org/xml/export-0.11/" xml:lang="vi">
|
||||
<siteinfo>
|
||||
<sitename>Wiktionary</sitename>
|
||||
<dbname>viwiktionary</dbname>
|
||||
</siteinfo>
|
||||
<!-- A legacy-dialect page: two parts of speech, an English section after
|
||||
the Vietnamese one that must not be read, and a line with no space
|
||||
after the # which is still a definition. -->
|
||||
<page>
|
||||
<title>pháp luật</title>
|
||||
<ns>0</ns>
|
||||
<id>1</id>
|
||||
<revision>
|
||||
<id>101</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-pron-}}
|
||||
{{vie-pron|pháp luật}}
|
||||
|
||||
{{-noun-}}
|
||||
{{-dfn-}}
|
||||
# [[hệ thống|Hệ thống]] các [[quy tắc]] xử sự do [[nhà nước]] đặt ra.<ref>Từ điển tiếng Việt</ref>
|
||||
#: ''Tuân theo pháp luật.''
|
||||
#{{label|vi|nghĩa rộng}} [[kỷ cương|Kỷ cương]] nói chung.
|
||||
|
||||
{{-verb-}}
|
||||
# ''(hiếm)'' [[xử|Xử]] theo luật.
|
||||
|
||||
{{-trans-}}
|
||||
* {{eng}}: {{t|en|law}}
|
||||
|
||||
{{-eng-}}
|
||||
{{-noun-}}
|
||||
# Law, in English.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A new-dialect page with a proper-noun heading, a place template and an
|
||||
English section that must not be read. -->
|
||||
<page>
|
||||
<title>Hòa Bình</title>
|
||||
<ns>0</ns>
|
||||
<id>2</id>
|
||||
<revision>
|
||||
<id>102</id>
|
||||
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
|
||||
=== {{ĐM|etym}} ===
|
||||
Từ Hán-Việt.
|
||||
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
{{vi-pr-noun}}
|
||||
|
||||
# {{place|vi|tỉnh|c/Việt Nam}}.
|
||||
|
||||
== {{langname|en}} ==
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
# A province of Vietnam.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- The lowercase page of the same word: the two merge into one entry with
|
||||
the senses in page order. -->
|
||||
<page>
|
||||
<title>hòa bình</title>
|
||||
<ns>0</ns>
|
||||
<id>3</id>
|
||||
<revision>
|
||||
<id>103</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[tình trạng|Tình trạng]] không có [[chiến tranh]].
|
||||
{{-adj-}}
|
||||
# [[yên ổn|Yên ổn]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A case-only redirect: skipped and counted. -->
|
||||
<page>
|
||||
<title>mặt trời</title>
|
||||
<ns>0</ns>
|
||||
<id>4</id>
|
||||
<redirect title="Mặt Trời" />
|
||||
<revision>
|
||||
<id>104</id>
|
||||
<text bytes="1" xml:space="preserve">#đổi [[Mặt Trời]]</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- No Vietnamese section at all. -->
|
||||
<page>
|
||||
<title>hello world</title>
|
||||
<ns>0</ns>
|
||||
<id>5</id>
|
||||
<revision>
|
||||
<id>105</id>
|
||||
<text bytes="1" xml:space="preserve">{{-eng-}}
|
||||
{{-phrase-}}
|
||||
# Xin chào thế giới.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- Not in the main namespace. -->
|
||||
<page>
|
||||
<title>Thể loại:Danh từ tiếng Việt</title>
|
||||
<ns>14</ns>
|
||||
<id>6</id>
|
||||
<revision>
|
||||
<id>106</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# Not a word.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- One syllable: rejected downstream by accept(), but its section is
|
||||
still a Vietnamese section for the page count. -->
|
||||
<page>
|
||||
<title>pháp</title>
|
||||
<ns>0</ns>
|
||||
<id>7</id>
|
||||
<revision>
|
||||
<id>107</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[phép|Phép]], [[luật]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- A definition that is only a template the stripper does not know: no
|
||||
meaning, but still a word. -->
|
||||
<page>
|
||||
<title>luật lệ</title>
|
||||
<ns>0</ns>
|
||||
<id>8</id>
|
||||
<revision>
|
||||
<id>108</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# {{rfdef|vi}}</text>
|
||||
</revision>
|
||||
</page>
|
||||
<!-- More graph: a shorthand heading code in the new dialect, and words that
|
||||
give the out-degree and alias tests something to check. -->
|
||||
<page>
|
||||
<title>ngôn ngữ</title>
|
||||
<ns>0</ns>
|
||||
<id>9</id>
|
||||
<revision>
|
||||
<id>109</id>
|
||||
<text bytes="1" xml:space="preserve">== {{langname|vi}} ==
|
||||
=== {{section|n}} ===
|
||||
{{vi-noun}}
|
||||
|
||||
# [[hệ thống|Hệ thống]] những [[âm]], [[từ]] và [[quy tắc]] kết hợp chúng.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>ngữ pháp</title>
|
||||
<ns>0</ns>
|
||||
<id>10</id>
|
||||
<revision>
|
||||
<id>110</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[toàn bộ|Toàn bộ]] những [[quy tắc]] hoạt động của các yếu tố ngôn ngữ.</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>vô tuyến điện</title>
|
||||
<ns>0</ns>
|
||||
<id>11</id>
|
||||
<revision>
|
||||
<id>111</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# [[kỹ thuật|Kỹ thuật]] truyền tin bằng [[sóng điện từ]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
<page>
|
||||
<title>covid 19</title>
|
||||
<ns>0</ns>
|
||||
<id>12</id>
|
||||
<revision>
|
||||
<id>112</id>
|
||||
<text bytes="1" xml:space="preserve">{{-vie-}}
|
||||
{{-noun-}}
|
||||
# Một [[bệnh]].</text>
|
||||
</revision>
|
||||
</page>
|
||||
</mediawiki>
|
||||
Binary file not shown.
@@ -0,0 +1,629 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"html"
|
||||
"regexp"
|
||||
"strings"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// This file reads the wikitext of one Wiktionary tiếng Việt page: it finds the
|
||||
// Vietnamese section, walks its part-of-speech headings and turns each
|
||||
// definition line into plain text.
|
||||
//
|
||||
// The wiki is mid-migration between two markup dialects and both are live
|
||||
// (2026-09-01 dump: 35,885 legacy pages, 7,129 new):
|
||||
//
|
||||
// legacy {{-vie-}} opens the section, {{-noun-}} and kin are the headings,
|
||||
// and the section ends at the next {{-xxx-}} whose code is a
|
||||
// language rather than a heading.
|
||||
// new == {{langname|vi}} == opens the section, === {{ĐM|noun}} === or
|
||||
// === {{section|noun}} === are the headings, and the next level-2
|
||||
// heading ends it.
|
||||
//
|
||||
// Nothing here is a general wikitext parser. It knows exactly the shapes a
|
||||
// definition line takes on this wiki and drops the rest on purpose; what
|
||||
// survives is plain text, capped, safe to render as text and never as markup.
|
||||
|
||||
// sense is one definition with the Vietnamese part-of-speech label of the
|
||||
// heading it sat under. pos is empty when the heading was one the label map
|
||||
// does not know, never a reason to drop the definition.
|
||||
type sense struct {
|
||||
pos string
|
||||
gloss string
|
||||
}
|
||||
|
||||
const (
|
||||
// maxSenses and maxGlossRunes bound what one word carries to the client.
|
||||
maxSenses = 5
|
||||
maxGlossRunes = 200
|
||||
// ellipsis marks a definition cut at maxGlossRunes.
|
||||
ellipsis = "…"
|
||||
)
|
||||
|
||||
// posLabelMap maps a part-of-speech heading code, the same in both dialects
|
||||
// ({{-noun-}}, {{ĐM|noun}}, {{section|noun}}, {{vi-noun}}), to the Vietnamese
|
||||
// label the client shows in front of a sense.
|
||||
//
|
||||
// Codes and their frequencies in Vietnamese sections of the 2026-09-01 dump:
|
||||
// noun 13,674 + n 647 · verb 7,595 + v 353 · adj 5,318 + adjc 526 · place 3,352
|
||||
// · pr-noun 1,397 + 329 · adv 982 · phrase 412 · proverb 295 · idiom 241 ·
|
||||
// interj 169 · pronoun 158 · num 123 · conj 88 · prep 88 · part 29. Note that
|
||||
// "pron" on this wiki is pronunciation, not pronoun.
|
||||
var posLabelMap = map[string]string{
|
||||
"noun": "danh từ", "n": "danh từ",
|
||||
"verb": "động từ", "v": "động từ", "tr-verb": "động từ", "intr-verb": "động từ", "aux-verb": "động từ",
|
||||
"adj": "tính từ", "adjc": "tính từ", "adjective": "tính từ",
|
||||
"adv": "phó từ", "adverb": "phó từ", "advb": "phó từ",
|
||||
"pr-noun": "danh từ riêng", "proper": "danh từ riêng", "propn": "danh từ riêng", "proper noun": "danh từ riêng", "name": "danh từ riêng",
|
||||
"pr-adj": "tính từ riêng",
|
||||
"place": "địa danh",
|
||||
"pronoun": "đại từ", "per-pronoun": "đại từ",
|
||||
"num": "số từ", "numeral": "số từ",
|
||||
"conj": "liên từ", "conjunction": "liên từ",
|
||||
"prep": "giới từ",
|
||||
"interj": "thán từ", "intj": "thán từ", "interjection": "thán từ",
|
||||
"part": "trợ từ", "particle": "trợ từ",
|
||||
"phrase": "cụm từ",
|
||||
"idiom": "thành ngữ",
|
||||
"proverb": "tục ngữ", "prov": "tục ngữ",
|
||||
"abbr": "viết tắt", "abr": "viết tắt",
|
||||
"prefix": "tiền tố",
|
||||
"suffix": "hậu tố",
|
||||
"letter": "chữ cái",
|
||||
"symbol": "ký hiệu",
|
||||
}
|
||||
|
||||
// otherSectionCodes are heading codes that are not parts of speech: they sit
|
||||
// inside a language section and reset the current label without being
|
||||
// counted as unmapped. Frequencies in Vietnamese sections, 2026-09-01 dump:
|
||||
// pron 36,533 · ref 27,438 · trans 16,073 · paro 8,058 · etym 4,267 · syn
|
||||
// 3,165 · info 1,846 · hanviet 1,601 · hanviet-t 1,441 · etymology 1,429 ·
|
||||
// reference 1,426 · see 959 · related 461 · drv 277 · synonym 262 · ant 236 ·
|
||||
// desction 128 · usage 110 · expr 92 · homo 57 · desc 54 · further 48 · forms
|
||||
// 41 · note 33 · derived 28 · anagram 24 · compound 21 · redup 20 · translit 17
|
||||
// · cat 16 · antonym 15. "dfn" (4,615) is a "definitions" heading placed under a
|
||||
// part-of-speech heading, so it is a heading for the section boundary but
|
||||
// transparent to the label: see classifyHeading.
|
||||
var otherSectionCodes = map[string]bool{
|
||||
"pron": true, "pronunciation": true, "ref": true, "reference": true, "references": true,
|
||||
"trans": true, "translations": true, "paro": true, "paronym": true, "etym": true, "etymology": true,
|
||||
"syn": true, "synonym": true, "ant": true, "antonym": true, "info": true,
|
||||
"hanviet": true, "hanviet-t": true, "see": true, "see also": true, "related": true, "rel": true,
|
||||
"related terms": true, "drv": true, "der": true, "derived": true, "derived terms": true,
|
||||
"desction": true, "desc": true, "usage": true, "usage notes": true, "expr": true, "homo": true,
|
||||
"further": true, "further reading": true, "forms": true, "note": true, "anagram": true,
|
||||
"anagrams": true, "ana": true, "compound": true, "redup": true, "translit": true, "cat": true,
|
||||
"coord": true, "coordinate": true, "alt": true, "alter": true, "alter form": true,
|
||||
"alternative form": true, "alternative forms": true, "alternative script": true,
|
||||
"glyph origin": true, "han": true, "nôm": true, "han character": true, "kanji": true,
|
||||
"rom": true, "romanization": true, "mut": true, "participle": true, "ptcp": true,
|
||||
"syllable": true, "article": true, "contr": true, "cmavo": true, "dfn": true, "com": true,
|
||||
// The same sections written out in Vietnamese, as a few new-dialect pages do.
|
||||
"phát âm": true, "từ nguyên": true, "từ nguyên 1": true, "từ nguyên 2": true, "tham khảo": true,
|
||||
"xem thêm": true, "cách viết khác": true, "phồn thể": true, "hán-nôm": true, "hán nôm": true,
|
||||
"chữ hán": true, "chữ nôm": true, "chú ý": true, "đồng nghĩa": true, "từ đồng nghĩa": true,
|
||||
"bản dịch": true, "dịch": true, "dấu phụ": true, "liên kết ngoài": true, "thuật ngữ liên quan": true,
|
||||
"từ tương tự": true, "meronym": true, "meronyms": true, "nguồn gốc ký tự chữ nôm": true,
|
||||
}
|
||||
|
||||
var (
|
||||
// legacyTemplate matches one {{-code-}} template, optionally with
|
||||
// parameters: {{-noun-}}, {{-pr-noun-}}, {{-vie-|...}}. Not anchored: a few
|
||||
// pages run {{-vie-}}{{-pron-}}{{vie-pron|…}}{{-place-}} together on one
|
||||
// line, so a line is read as headings when it starts with one and may
|
||||
// carry several.
|
||||
legacyTemplate = regexp.MustCompile(`\{\{-([A-Za-z0-9-]+?)-(?:\|[^}]*)?\}\}`)
|
||||
// headingLine matches == text == at any level and captures the level.
|
||||
headingLine = regexp.MustCompile(`^(={2,6})\s*(.*?)\s*=+\s*$`)
|
||||
// sectionTemplate captures the code of {{ĐM|code}} and {{section|code}}.
|
||||
sectionTemplate = regexp.MustCompile(`\{\{(?:ĐM|đm|DM|dm|section)\|([^}|]+)`)
|
||||
// headwordTemplate captures the code of {{vi-code}} / {{vie-code}} at the
|
||||
// start of a line: the new dialect's headword line, which names the POS.
|
||||
headwordTemplate = regexp.MustCompile(`^\{\{vie?-([a-z -]+)`)
|
||||
// langnameVi is the new dialect's Vietnamese section heading text.
|
||||
langnameVi = regexp.MustCompile(`^\{\{langname\|vi\}\}$`)
|
||||
|
||||
htmlComment = regexp.MustCompile(`(?s)<!--.*?-->`)
|
||||
refElement = regexp.MustCompile(`(?s)<ref\b[^>/]*/>|<ref\b[^>]*>.*?</ref>`)
|
||||
anyTag = regexp.MustCompile(`</?[A-Za-z][^>]*>`)
|
||||
spaces = regexp.MustCompile(`\s+`)
|
||||
)
|
||||
|
||||
// isLegacyHeading reports whether a {{-code-}} is a heading inside a language
|
||||
// section. Every other code — language and script codes such as eng, tyz,
|
||||
// aav-qal, Latn — ends the Vietnamese section.
|
||||
func isLegacyHeading(code string) bool {
|
||||
_, pos := posLabelMap[code]
|
||||
return pos || otherSectionCodes[code]
|
||||
}
|
||||
|
||||
// vietnameseSection returns the wikitext of the page's Vietnamese section and
|
||||
// which dialect opened it: "legacy", "new", or "" when the page has none.
|
||||
// When both dialects open a section on one page the first one in the text
|
||||
// wins and both is reported so the build log can count it. ender is the
|
||||
// {{-code-}} that closed a legacy section, empty when a heading or the end of
|
||||
// the page did: a heading code missing from the maps shows up there as a
|
||||
// section-ending code, which is the signal that definitions are being lost.
|
||||
func vietnameseSection(text string) (section, dialect string, both bool, ender string) {
|
||||
lines := strings.Split(text, "\n")
|
||||
legacyAt, newAt := -1, -1
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if legacyAt < 0 && strings.HasPrefix(line, "{{-") {
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatchIndex(line, -1) {
|
||||
if line[m[2]:m[3]] == "vie" {
|
||||
legacyAt = i
|
||||
// Whatever follows the marker on its own line belongs to
|
||||
// the section.
|
||||
lines[i] = line[m[1]:]
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
if newAt < 0 {
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 && langnameVi.MatchString(m[2]) {
|
||||
newAt = i
|
||||
}
|
||||
}
|
||||
}
|
||||
both = legacyAt >= 0 && newAt >= 0
|
||||
switch {
|
||||
case legacyAt < 0 && newAt < 0:
|
||||
return "", "", false, ""
|
||||
case newAt < 0 || (legacyAt >= 0 && legacyAt < newAt):
|
||||
section, ender = legacySection(lines[legacyAt:])
|
||||
return section, "legacy", both, ender
|
||||
default:
|
||||
return newSection(lines[newAt+1:]), "new", both, ""
|
||||
}
|
||||
}
|
||||
|
||||
// legacySection runs from after {{-vie-}} to the next {{-xxx-}} whose code is
|
||||
// not a heading, or the next level-2 heading, which is what a new-dialect
|
||||
// language section on a mixed page opens with. A language code sharing a line
|
||||
// with Vietnamese headings ends the section at that line; the line is lost,
|
||||
// which is the conservative side of a rare shape.
|
||||
func legacySection(lines []string) (section, ender string) {
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if strings.HasPrefix(line, "{{-") {
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
|
||||
if !isLegacyHeading(m[1]) {
|
||||
return strings.Join(lines[:i], "\n"), m[1]
|
||||
}
|
||||
}
|
||||
}
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
|
||||
return strings.Join(lines[:i], "\n"), ""
|
||||
}
|
||||
}
|
||||
return strings.Join(lines, "\n"), ""
|
||||
}
|
||||
|
||||
// newSection runs from after == {{langname|vi}} == to the next level-2
|
||||
// heading.
|
||||
func newSection(lines []string) string {
|
||||
for i, line := range lines {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) == 2 {
|
||||
return strings.Join(lines[:i], "\n")
|
||||
}
|
||||
}
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
|
||||
// headingKind says what a heading line means for the label of the
|
||||
// definitions under it.
|
||||
type headingKind int
|
||||
|
||||
const (
|
||||
notHeading headingKind = iota
|
||||
// posHeading names a part of speech: the code decides the label.
|
||||
posHeading
|
||||
// otherHeading is a section such as pronunciation or etymology: the label
|
||||
// resets to empty and nothing is counted as unmapped.
|
||||
otherHeading
|
||||
)
|
||||
|
||||
// classifyHeading reads one line as a heading in either dialect.
|
||||
//
|
||||
// {{-noun-}} legacy heading; the last of several on
|
||||
// one line decides
|
||||
// === {{ĐM|noun}} === new heading
|
||||
// === {{section|n}} === new heading, shorthand code
|
||||
// === Danh từ === new heading written out
|
||||
// {{vi-noun}} / {{vie-noun}} new headword line; refines the POS only
|
||||
func classifyHeading(line string) (code string, kind headingKind) {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
if strings.HasPrefix(line, "{{-") {
|
||||
kind = notHeading
|
||||
for _, m := range legacyTemplate.FindAllStringSubmatch(line, -1) {
|
||||
if c, k := classifyCode(m[1]); k != notHeading {
|
||||
code, kind = c, k
|
||||
}
|
||||
}
|
||||
return code, kind
|
||||
}
|
||||
if m := headingLine.FindStringSubmatch(line); m != nil && len(m[1]) >= 3 {
|
||||
text := m[2]
|
||||
if sm := sectionTemplate.FindStringSubmatch(text); sm != nil {
|
||||
code = strings.TrimSpace(sm[1])
|
||||
} else {
|
||||
code = strings.ToLower(text)
|
||||
}
|
||||
return classifyCode(code)
|
||||
}
|
||||
if m := headwordTemplate.FindStringSubmatch(line); m != nil {
|
||||
// Only a headword template whose code is a part of speech counts;
|
||||
// {{vi-pron}}, {{vi-etym-sino}} and kin are not headings.
|
||||
code = strings.TrimSpace(m[1])
|
||||
if _, ok := posLabelMap[code]; ok {
|
||||
return code, posHeading
|
||||
}
|
||||
}
|
||||
return "", notHeading
|
||||
}
|
||||
|
||||
// classifyCode sorts a heading code seen in either dialect. "dfn" is the one
|
||||
// heading that changes nothing: the wiki places {{-dfn-}} under {{-noun-}} to
|
||||
// introduce the definitions, so the label above it must carry through.
|
||||
func classifyCode(code string) (string, headingKind) {
|
||||
switch {
|
||||
case code == "dfn":
|
||||
return code, notHeading
|
||||
case otherSectionCodes[code]:
|
||||
return code, otherHeading
|
||||
}
|
||||
return code, posHeading
|
||||
}
|
||||
|
||||
// isLabelValue reports whether a heading was written out as one of the
|
||||
// Vietnamese labels already ("Danh từ").
|
||||
func isLabelValue(text string) bool {
|
||||
for _, l := range posLabelMap {
|
||||
if l == text {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// posLabel maps a heading code to its Vietnamese label: through the map, or
|
||||
// as itself when the heading was already written out in Vietnamese.
|
||||
func posLabel(code string) (label string, mapped bool) {
|
||||
if label, ok := posLabelMap[code]; ok {
|
||||
return label, true
|
||||
}
|
||||
if isLabelValue(code) {
|
||||
return code, true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// sectionStats counts what the scanner saw across sections, for the build log.
|
||||
type sectionStats struct {
|
||||
pos map[string]int // part-of-speech heading codes seen
|
||||
unmappedPos map[string]int // part-of-speech heading codes with no label
|
||||
defsKept int
|
||||
defsEmpty int // definitions that stripped to nothing
|
||||
defsCut int // definitions cut at maxGlossRunes
|
||||
dropped map[string]int // template names dropped whole
|
||||
}
|
||||
|
||||
func newSectionStats() *sectionStats {
|
||||
return §ionStats{
|
||||
pos: make(map[string]int),
|
||||
unmappedPos: make(map[string]int),
|
||||
dropped: make(map[string]int),
|
||||
}
|
||||
}
|
||||
|
||||
// definitions walks the Vietnamese section and returns its senses in page
|
||||
// order, at most maxSenses of them, each labelled with the part of speech of
|
||||
// the heading above it. Every heading and definition is counted in stats
|
||||
// whether or not it made the cut.
|
||||
func definitions(section string, stats *sectionStats) []sense {
|
||||
var senses []sense
|
||||
pos := ""
|
||||
for _, line := range strings.Split(section, "\n") {
|
||||
line = strings.TrimRight(line, "\r ")
|
||||
switch code, kind := classifyHeading(line); kind {
|
||||
case posHeading:
|
||||
stats.pos[code]++
|
||||
label, mapped := posLabel(code)
|
||||
if !mapped {
|
||||
stats.unmappedPos[code]++
|
||||
}
|
||||
pos = label
|
||||
continue
|
||||
case otherHeading:
|
||||
pos = ""
|
||||
continue
|
||||
}
|
||||
if !isDefinitionLine(line) {
|
||||
continue
|
||||
}
|
||||
gloss := stripWikitext(line[1:], stats.dropped)
|
||||
if gloss == "" {
|
||||
stats.defsEmpty++
|
||||
continue
|
||||
}
|
||||
if cut, wasCut := capGloss(gloss); wasCut {
|
||||
stats.defsCut++
|
||||
gloss = cut
|
||||
}
|
||||
stats.defsKept++
|
||||
if len(senses) < maxSenses {
|
||||
senses = append(senses, sense{pos: pos, gloss: gloss})
|
||||
}
|
||||
}
|
||||
return senses
|
||||
}
|
||||
|
||||
// isDefinitionLine accepts a top-level numbered item and nothing under it:
|
||||
// "# text" and "#text" are definitions; "#: example", "#* quotation", "## sub-
|
||||
// sense" and "#; term" are not.
|
||||
func isDefinitionLine(line string) bool {
|
||||
if len(line) < 2 || line[0] != '#' {
|
||||
return false
|
||||
}
|
||||
switch line[1] {
|
||||
case '#', ':', '*', ';':
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// stripWikitext turns one definition line into plain text. Lossy on purpose:
|
||||
// links keep their display text, formatting goes, the handful of templates
|
||||
// that carry definition text are unwrapped and every other template is
|
||||
// dropped whole (its name counted in dropped when non-nil). The result is
|
||||
// trimmed and whitespace-collapsed; a result with no letter or digit is empty.
|
||||
func stripWikitext(s string, dropped map[string]int) string {
|
||||
s = htmlComment.ReplaceAllString(s, "")
|
||||
s = refElement.ReplaceAllString(s, "")
|
||||
s = anyTag.ReplaceAllString(s, "")
|
||||
s = stripTemplates(s, dropped)
|
||||
s = stripLinks(s)
|
||||
s = strings.ReplaceAll(s, "'''", "")
|
||||
s = strings.ReplaceAll(s, "''", "")
|
||||
s = html.UnescapeString(s)
|
||||
s = strings.Map(func(r rune) rune {
|
||||
switch {
|
||||
case unicode.IsControl(r), unicode.Is(unicode.Cf, r):
|
||||
// Cc and Cf: control characters, and format characters such as a
|
||||
// bidi override or a zero-width space, which could reshape the
|
||||
// rest of a rendered line.
|
||||
return -1
|
||||
case unicode.IsSpace(r):
|
||||
// Non-breaking and other Unicode spaces become plain ones so the
|
||||
// ASCII-only collapse below catches them.
|
||||
return ' '
|
||||
}
|
||||
return r
|
||||
}, s)
|
||||
s = strings.TrimSpace(spaces.ReplaceAllString(s, " "))
|
||||
// A stray space before sentence punctuation is what unwrapping a template
|
||||
// at the end of a clause leaves behind.
|
||||
for _, p := range []string{" .", " ,", " ;", " :", " )"} {
|
||||
s = strings.ReplaceAll(s, p, p[1:])
|
||||
}
|
||||
if !strings.ContainsFunc(s, func(r rune) bool { return unicode.IsLetter(r) || unicode.IsDigit(r) }) {
|
||||
return ""
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// stripTemplates replaces every outermost {{...}} with its plain-text
|
||||
// rendering. Nesting is tracked by depth, so a template inside a kept
|
||||
// template's parameter is rendered recursively and one inside a dropped
|
||||
// template goes with it.
|
||||
func stripTemplates(s string, dropped map[string]int) string {
|
||||
var out strings.Builder
|
||||
depth := 0
|
||||
start := 0
|
||||
for i := 0; i < len(s); i++ {
|
||||
switch {
|
||||
case strings.HasPrefix(s[i:], "{{"):
|
||||
if depth == 0 {
|
||||
start = i + 2
|
||||
}
|
||||
depth++
|
||||
i++
|
||||
case strings.HasPrefix(s[i:], "}}"):
|
||||
// A closer with nothing open is stray markup, not text.
|
||||
if depth > 0 {
|
||||
depth--
|
||||
if depth == 0 {
|
||||
out.WriteString(renderTemplate(s[start:i], dropped))
|
||||
}
|
||||
}
|
||||
i++
|
||||
case depth == 0:
|
||||
out.WriteByte(s[i])
|
||||
}
|
||||
}
|
||||
if depth > 0 && dropped != nil {
|
||||
// Unbalanced braces: whatever opened and never closed is dropped, as a
|
||||
// template would be, rather than leaking half a template into a gloss.
|
||||
dropped["(unclosed)"]++
|
||||
}
|
||||
return out.String()
|
||||
}
|
||||
|
||||
// renderTemplate maps one template body (the text between {{ and }}) to plain
|
||||
// text. body still contains any nested templates verbatim.
|
||||
//
|
||||
// The kept templates are the ones that carry definition text on this wiki:
|
||||
// context labels in four spellings, links in five, place descriptions,
|
||||
// non-gloss definitions and the two cross-reference templates a "dfn" section
|
||||
// is usually made of. Everything else is presentation or classification.
|
||||
func renderTemplate(body string, dropped map[string]int) string {
|
||||
parts := splitTemplate(body)
|
||||
name := strings.ToLower(strings.TrimSpace(parts[0]))
|
||||
// Positional parameters only; key=value ones are presentation hints.
|
||||
var params []string
|
||||
for _, p := range parts[1:] {
|
||||
if strings.Contains(p, "=") && !strings.Contains(p, "[[") && !strings.Contains(p, "{{") {
|
||||
continue
|
||||
}
|
||||
params = append(params, strings.TrimSpace(stripTemplates(strings.TrimSpace(p), dropped)))
|
||||
}
|
||||
// A leading language code is markup, whether it is ours or a neighbour's
|
||||
// pasted in: the label templates take it first, the link ones too. Only
|
||||
// when something follows it, though: {{q|con}} is a one-word qualifier,
|
||||
// not a language.
|
||||
dropLang := func(ps []string) []string {
|
||||
if len(ps) > 1 && isLangCode(ps[0]) {
|
||||
return ps[1:]
|
||||
}
|
||||
return ps
|
||||
}
|
||||
switch name {
|
||||
case "label", "lb", "nhãn", "context", "term", "gloss", "qualifier", "q":
|
||||
params = dropLang(params)
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return "(" + strings.Join(params, ", ") + ")"
|
||||
case "l", "vi-l", "w", "m", "link":
|
||||
params = dropLang(params)
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return params[len(params)-1]
|
||||
case "n-g", "non-gloss", "non-gloss definition":
|
||||
return strings.Join(params, " ")
|
||||
case "see-entry", "like-entry":
|
||||
if len(params) == 0 {
|
||||
return ""
|
||||
}
|
||||
return "Xem " + params[0]
|
||||
case "place":
|
||||
params = dropLang(params)
|
||||
for i, p := range params {
|
||||
// "c/Việt Nam" is a typed place: the type prefix is markup.
|
||||
if len(p) > 2 && p[1] == '/' && p[0] >= 'a' && p[0] <= 'z' {
|
||||
params[i] = p[2:]
|
||||
}
|
||||
}
|
||||
return strings.Join(params, ", ")
|
||||
}
|
||||
if dropped != nil {
|
||||
dropped[name]++
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// isLangCode reports whether a template parameter is a language code rather
|
||||
// than text: two or three lowercase ASCII letters, optionally with a
|
||||
// hyphenated variant such as "nan-hbl".
|
||||
func isLangCode(p string) bool {
|
||||
if len(p) < 2 || len(p) > 11 {
|
||||
return false
|
||||
}
|
||||
letters := 0
|
||||
for _, r := range p {
|
||||
switch {
|
||||
case r >= 'a' && r <= 'z':
|
||||
letters++
|
||||
case r == '-':
|
||||
if letters < 2 {
|
||||
return false
|
||||
}
|
||||
letters = 0
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return letters >= 2 && letters <= 3
|
||||
}
|
||||
|
||||
// splitTemplate splits a template body on | outside nested braces and
|
||||
// brackets, so a link or template inside a parameter is not cut in two.
|
||||
func splitTemplate(body string) []string {
|
||||
var parts []string
|
||||
depth := 0
|
||||
start := 0
|
||||
for i := 0; i < len(body); i++ {
|
||||
switch {
|
||||
case strings.HasPrefix(body[i:], "{{") || strings.HasPrefix(body[i:], "[["):
|
||||
depth++
|
||||
i++
|
||||
case strings.HasPrefix(body[i:], "}}") || strings.HasPrefix(body[i:], "]]"):
|
||||
if depth > 0 {
|
||||
depth--
|
||||
}
|
||||
i++
|
||||
case body[i] == '|' && depth == 0:
|
||||
parts = append(parts, body[start:i])
|
||||
start = i + 1
|
||||
}
|
||||
}
|
||||
return append(parts, body[start:])
|
||||
}
|
||||
|
||||
// stripLinks renders wiki links as their display text and drops category
|
||||
// links, which are classification rather than definition.
|
||||
func stripLinks(s string) string {
|
||||
var out strings.Builder
|
||||
for {
|
||||
open := strings.Index(s, "[[")
|
||||
if open < 0 {
|
||||
break
|
||||
}
|
||||
close := strings.Index(s[open:], "]]")
|
||||
if close < 0 {
|
||||
break
|
||||
}
|
||||
out.WriteString(s[:open])
|
||||
inner := s[open+2 : open+close]
|
||||
lower := strings.ToLower(inner)
|
||||
if !strings.HasPrefix(lower, "thể loại:") && !strings.HasPrefix(lower, "category:") {
|
||||
if bar := strings.LastIndex(inner, "|"); bar >= 0 {
|
||||
inner = inner[bar+1:]
|
||||
}
|
||||
out.WriteString(inner)
|
||||
}
|
||||
s = s[open+close+2:]
|
||||
}
|
||||
out.WriteString(s)
|
||||
s = out.String()
|
||||
|
||||
// External links: [http://… label] → label; a bare URL in brackets goes.
|
||||
out.Reset()
|
||||
for {
|
||||
open := strings.Index(s, "[http")
|
||||
if open < 0 {
|
||||
break
|
||||
}
|
||||
close := strings.Index(s[open:], "]")
|
||||
if close < 0 {
|
||||
break
|
||||
}
|
||||
out.WriteString(s[:open])
|
||||
inner := s[open+1 : open+close]
|
||||
if sp := strings.IndexByte(inner, ' '); sp >= 0 {
|
||||
out.WriteString(inner[sp+1:])
|
||||
}
|
||||
s = s[open+close+1:]
|
||||
}
|
||||
out.WriteString(s)
|
||||
return out.String()
|
||||
}
|
||||
|
||||
// capGloss cuts a definition longer than maxGlossRunes at the last space
|
||||
// before the limit and marks the cut with an ellipsis.
|
||||
func capGloss(s string) (string, bool) {
|
||||
if utf8.RuneCountInString(s) <= maxGlossRunes {
|
||||
return s, false
|
||||
}
|
||||
runes := []rune(s)
|
||||
head := string(runes[:maxGlossRunes-utf8.RuneCountInString(ellipsis)])
|
||||
if sp := strings.LastIndexByte(head, ' '); sp > 0 {
|
||||
head = head[:sp]
|
||||
}
|
||||
return strings.TrimRight(head, " ,;:") + ellipsis, true
|
||||
}
|
||||
@@ -0,0 +1,274 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
func TestStripWikitext(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
want string
|
||||
}{
|
||||
{"links keep display text",
|
||||
"[[chỗ|Chỗ]] [[râm]] [[mát]], do [[trời]] có [[mây]] hoặc do không bị [[nắng]] [[chiếu]].",
|
||||
"Chỗ râm mát, do trời có mây hoặc do không bị nắng chiếu."},
|
||||
{"place template keeps its parameters and drops the type prefix",
|
||||
"{{place|vi|thủ đô|c/Việt Nam}}.",
|
||||
"thủ đô, Việt Nam."},
|
||||
{"label becomes a parenthesis",
|
||||
"{{label|vi|thuộc lịch sử}} Một [[tỉnh]] cũ của [[Việt Nam]] vào nửa cuối thế kỷ XIX.",
|
||||
"(thuộc lịch sử) Một tỉnh cũ của Việt Nam vào nửa cuối thế kỷ XIX."},
|
||||
{"Vietnamese label spellings",
|
||||
"{{nhãn|vi|tin học}} {{context|cũ}} {{term|Hóa học}} Cấu trúc.",
|
||||
"(tin học) (cũ) (Hóa học) Cấu trúc."},
|
||||
{"nested template inside a kept one",
|
||||
"{{label|vi|{{w|Hà Nội}}}} Thủ đô.",
|
||||
"(Hà Nội) Thủ đô."},
|
||||
{"unknown template dropped whole, nesting included",
|
||||
"{{rfdef|vi|{{w|x}}}}",
|
||||
""},
|
||||
{"definition that is only a cross-reference",
|
||||
"{{see-entry|bà la sát}}.",
|
||||
"Xem bà la sát."},
|
||||
{"non-gloss definition",
|
||||
"{{n-g|Trợ từ nhấn mạnh.}}",
|
||||
"Trợ từ nhấn mạnh."},
|
||||
{"link templates keep the last parameter",
|
||||
"{{l|vi|nói}}, {{l|vi|nói năng|nói năng (hiếm)}} và {{w|Việt Nam}}.",
|
||||
"nói, nói năng (hiếm) và Việt Nam."},
|
||||
{"ref mid-sentence and a lone ref",
|
||||
"Một loài [[cá]]<ref>Từ điển</ref> nước ngọt<ref name=\"a\" />.",
|
||||
"Một loài cá nước ngọt."},
|
||||
{"comment, bold, italic, entities",
|
||||
"'''Rất''' ''nhanh''<!-- todo --> và&mạnh.",
|
||||
"Rất nhanh và&mạnh."},
|
||||
{"category link dropped, external link keeps label",
|
||||
"Một [[thành phố]] [[Thể loại:Địa danh]] ([http://example.org trang web]).",
|
||||
"Một thành phố (trang web)."},
|
||||
{"named parameters are not text",
|
||||
"{{lb|vi|thơ ca|sort=x}} Câu.",
|
||||
"(thơ ca) Câu."},
|
||||
{"a lone short parameter is text, not a language code",
|
||||
"{{q|con}} Một loài vật, {{l|con}} là con.",
|
||||
"(con) Một loài vật, con là con."},
|
||||
{"format and bidi characters are dropped",
|
||||
"M\u200bột \u202enghĩa\u202c.",
|
||||
"Một nghĩa."},
|
||||
{"a stray closer is not text", "Một }} nghĩa.", "Một nghĩa."},
|
||||
{"only punctuation is empty", "(...).", ""},
|
||||
{"control characters and whitespace collapse", " Một từ \t hai ", "Một từ hai"},
|
||||
{"unclosed template does not leak", "Một {{label|vi|x từ.", "Một"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := stripWikitext(c.in, nil); got != c.want {
|
||||
t.Errorf("stripWikitext(%q)\n got %q\nwant %q", c.in, got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestStripWikitextCountsDroppedTemplates(t *testing.T) {
|
||||
dropped := make(map[string]int)
|
||||
stripWikitext("{{rfdef|vi}} {{RfDef|vi}} {{senseid|vi|x}}", dropped)
|
||||
if dropped["rfdef"] != 2 || dropped["senseid"] != 1 {
|
||||
t.Errorf("dropped = %v, want rfdef 2 (case-folded), senseid 1", dropped)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCapGlossCutsAtAWordBoundary(t *testing.T) {
|
||||
word := "từ "
|
||||
long := strings.Repeat(word, 120) // 360 runes
|
||||
got, cut := capGloss(long)
|
||||
if !cut {
|
||||
t.Fatal("a 360-rune gloss was not cut")
|
||||
}
|
||||
if n := utf8.RuneCountInString(got); n > maxGlossRunes {
|
||||
t.Errorf("cut gloss is %d runes, want at most %d", n, maxGlossRunes)
|
||||
}
|
||||
if !strings.HasSuffix(got, "từ"+ellipsis) {
|
||||
t.Errorf("cut gloss %q does not end on a whole word plus the ellipsis", got)
|
||||
}
|
||||
if short, cut := capGloss("ngắn"); cut || short != "ngắn" {
|
||||
t.Errorf("a short gloss was changed: %q %v", short, cut)
|
||||
}
|
||||
}
|
||||
|
||||
const legacyPage = `{{-vie-}}
|
||||
{{-pron-}}
|
||||
{{vie-pron|học sinh}}
|
||||
|
||||
{{-noun-}}
|
||||
# [[người|Người]] [[học]] ở [[nhà trường|trường]].
|
||||
#: ''Học sinh giỏi.''
|
||||
#{{label|vi|cũ}} [[môn đệ|Môn đệ]].
|
||||
|
||||
{{-verb-}}
|
||||
# [[đi học|Đi học]].
|
||||
|
||||
{{-trans-}}
|
||||
* {{eng}}: {{t|en|student}}
|
||||
|
||||
{{-eng-}}
|
||||
{{-noun-}}
|
||||
# Student, in English.
|
||||
`
|
||||
|
||||
const newPage = `== {{langname|vi}} ==
|
||||
=== {{ĐM|etym}} ===
|
||||
Hán-Việt.
|
||||
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
{{vi-pr-noun}}
|
||||
|
||||
# {{place|vi|thủ đô|c/Việt Nam}}.
|
||||
|
||||
=== {{section|v}} ===
|
||||
# [[đi|Đi]] về thủ đô.
|
||||
|
||||
=== {{ĐM|xyz}} ===
|
||||
# Một nghĩa dưới đề mục lạ.
|
||||
|
||||
== {{langname|en}} ==
|
||||
=== {{ĐM|pr-noun}} ===
|
||||
# The capital of Vietnam.
|
||||
`
|
||||
|
||||
func TestVietnameseSectionLegacy(t *testing.T) {
|
||||
section, dialect, both, _ := vietnameseSection(legacyPage)
|
||||
if dialect != "legacy" || both {
|
||||
t.Fatalf("dialect = %q both = %v, want legacy false", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "Student") {
|
||||
t.Error("the English section leaked into the Vietnamese one")
|
||||
}
|
||||
if !strings.Contains(section, "{{-trans-}}") {
|
||||
t.Error("the translations heading, a section heading rather than a language, cut the section short")
|
||||
}
|
||||
}
|
||||
|
||||
func TestVietnameseSectionNew(t *testing.T) {
|
||||
section, dialect, both, _ := vietnameseSection(newPage)
|
||||
if dialect != "new" || both {
|
||||
t.Fatalf("dialect = %q both = %v, want new false", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "capital of Vietnam") {
|
||||
t.Error("the English section leaked into the Vietnamese one")
|
||||
}
|
||||
if !strings.Contains(section, "đề mục lạ") {
|
||||
t.Error("a level-3 heading ended the section; only a level-2 heading may")
|
||||
}
|
||||
}
|
||||
|
||||
func TestVietnameseSectionAbsentAndBoth(t *testing.T) {
|
||||
if _, dialect, _, _ := vietnameseSection("{{-eng-}}\n{{-noun-}}\n# Word."); dialect != "" {
|
||||
t.Errorf("an English-only page reported dialect %q", dialect)
|
||||
}
|
||||
mixed := "== {{langname|vi}} ==\n# Mới.\n" + legacyPage
|
||||
section, dialect, both, _ := vietnameseSection(mixed)
|
||||
if dialect != "new" || !both {
|
||||
t.Errorf("dialect = %q both = %v, want the first dialect in the text and both=true", dialect, both)
|
||||
}
|
||||
if strings.Contains(section, "Người học") {
|
||||
t.Error("the first section should end where the legacy page starts a language section")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsLegacy(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
section, _, _, _ := vietnameseSection(legacyPage)
|
||||
got := definitions(section, stats)
|
||||
want := []sense{
|
||||
{"danh từ", "Người học ở trường."},
|
||||
{"danh từ", "(cũ) Môn đệ."},
|
||||
{"động từ", "Đi học."},
|
||||
}
|
||||
assertSenses(t, got, want)
|
||||
if stats.pos["noun"] != 1 || stats.pos["verb"] != 1 {
|
||||
t.Errorf("pos tally = %v, want noun 1 verb 1", stats.pos)
|
||||
}
|
||||
if stats.pos["pron"] != 0 || stats.pos["trans"] != 0 {
|
||||
t.Errorf("pronunciation and translations were tallied as parts of speech: %v", stats.pos)
|
||||
}
|
||||
if stats.defsKept != 3 {
|
||||
t.Errorf("defsKept = %d, want 3", stats.defsKept)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsNewDialect(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
section, _, _, _ := vietnameseSection(newPage)
|
||||
got := definitions(section, stats)
|
||||
want := []sense{
|
||||
{"danh từ riêng", "thủ đô, Việt Nam."},
|
||||
{"động từ", "Đi về thủ đô."},
|
||||
{"", "Một nghĩa dưới đề mục lạ."},
|
||||
}
|
||||
assertSenses(t, got, want)
|
||||
if stats.unmappedPos["xyz"] != 1 {
|
||||
t.Errorf("unmapped headings = %v, want xyz 1", stats.unmappedPos)
|
||||
}
|
||||
if stats.unmappedPos["etym"] != 0 {
|
||||
t.Errorf("etymology counted as an unmapped part of speech: %v", stats.unmappedPos)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefinitionsHeadwordLineAndWrittenOutHeading(t *testing.T) {
|
||||
section := "=== Danh từ ===\n# Một.\n{{vi-verb}}\n# Hai.\n{{vi-pron}}\n# Ba.\n=== Phát âm ===\n# Bốn."
|
||||
got := definitions(section, newSectionStats())
|
||||
want := []sense{{"danh từ", "Một."}, {"động từ", "Hai."}, {"động từ", "Ba."}, {"", "Bốn."}}
|
||||
assertSenses(t, got, want)
|
||||
}
|
||||
|
||||
func TestDefinitionsSkipsEmptyAndCaps(t *testing.T) {
|
||||
stats := newSectionStats()
|
||||
lines := []string{"{{-noun-}}", "# {{rfdef|vi}}", "#: not a definition", "#* nor this", "## nor this"}
|
||||
for i := 0; i < 7; i++ {
|
||||
lines = append(lines, "# Nghĩa số "+string(rune('a'+i))+".")
|
||||
}
|
||||
lines = append(lines, "# "+strings.Repeat("dài ", 80))
|
||||
got := definitions(strings.Join(lines, "\n"), stats)
|
||||
if len(got) != maxSenses {
|
||||
t.Fatalf("got %d senses, want the cap of %d", len(got), maxSenses)
|
||||
}
|
||||
if got[0].gloss != "Nghĩa số a." {
|
||||
t.Errorf("first sense = %q, want the first real definition after the empty one", got[0].gloss)
|
||||
}
|
||||
if stats.defsEmpty != 1 || stats.dropped["rfdef"] != 1 {
|
||||
t.Errorf("empty = %d dropped = %v, want 1 and rfdef 1", stats.defsEmpty, stats.dropped)
|
||||
}
|
||||
if stats.defsKept != 8 || stats.defsCut != 1 {
|
||||
t.Errorf("kept = %d cut = %d, want 8 and 1 (counted past the cap)", stats.defsKept, stats.defsCut)
|
||||
}
|
||||
}
|
||||
|
||||
func assertSenses(t *testing.T, got, want []sense) {
|
||||
t.Helper()
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %d senses %v, want %d %v", len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("sense %d = %+v, want %+v", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A few pages run the section marker and the headings together on one line:
|
||||
// {{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}. The marker must still
|
||||
// open the section and the last heading on the line must still label it.
|
||||
func TestVietnameseSectionInlineHeadings(t *testing.T) {
|
||||
page := "{{-vie-}}{{-pron-}}{{vie-pron|Thượng|Hải}}{{-place-}}\n\n'''Thượng Hải'''\n# Thành phố lớn nhất [[Trung Quốc]].\n{{-eng-}}{{-noun-}}\n# Shanghai."
|
||||
section, dialect, _, ender := vietnameseSection(page)
|
||||
if dialect != "legacy" || ender != "eng" {
|
||||
t.Fatalf("dialect = %q ender = %q, want legacy ended by eng", dialect, ender)
|
||||
}
|
||||
if strings.Contains(section, "Shanghai") {
|
||||
t.Error("the English section, opened on a shared line, leaked in")
|
||||
}
|
||||
got := definitions(section, newSectionStats())
|
||||
assertSenses(t, got, []sense{{"địa danh", "Thành phố lớn nhất Trung Quốc."}})
|
||||
}
|
||||
@@ -62,6 +62,7 @@ func run() error {
|
||||
"path", cfg.dbPath,
|
||||
"words", store.WordCount(),
|
||||
"aliases", store.AliasCount(),
|
||||
"meanings", store.MeaningCount(),
|
||||
"license", store.License(),
|
||||
)
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ import (
|
||||
"iter"
|
||||
"net/url"
|
||||
"os"
|
||||
"slices"
|
||||
"sort"
|
||||
"strconv"
|
||||
|
||||
@@ -32,6 +33,10 @@ import (
|
||||
// ErrNotFound is returned when a syllable has no entry in the dictionary.
|
||||
var ErrNotFound = errors.New("dictionary: syllable not found")
|
||||
|
||||
// requiredBuilderVersion is the builder whose meta contract this store reads;
|
||||
// it is named in the refusal of an older database.
|
||||
const requiredBuilderVersion = "5"
|
||||
|
||||
// wordInfo holds the two syllables the chain rule needs. Both ends are kept:
|
||||
// canonicalization can move either one, so the engine must never re-derive
|
||||
// them from what the player typed.
|
||||
@@ -40,6 +45,15 @@ type wordInfo struct {
|
||||
last string
|
||||
}
|
||||
|
||||
// Sense is one definition of a word as Wiktionary gives it: the Vietnamese
|
||||
// part-of-speech label of the heading it sat under ("danh từ"), empty when the
|
||||
// builder did not know the heading, and the definition as plain text. Neither
|
||||
// is markup; the client renders both as text.
|
||||
type Sense struct {
|
||||
Pos string
|
||||
Gloss string
|
||||
}
|
||||
|
||||
// Store answers word and syllable queries against the derived dictionary.
|
||||
//
|
||||
// Every field is written once during Open and only read afterwards, and no
|
||||
@@ -54,6 +68,10 @@ type Store struct {
|
||||
// openers holds words whose last syllable has at least one continuation,
|
||||
// sorted by that count descending so an eligible set is always a prefix.
|
||||
openers []opener
|
||||
// meanings holds each word's senses in page order. A few megabytes of text
|
||||
// for the corpus; a per-move query would be a second code path for nothing.
|
||||
meanings map[string][]Sense
|
||||
meaningCount int
|
||||
|
||||
license string
|
||||
}
|
||||
@@ -87,11 +105,12 @@ func Open(path string) (*Store, error) {
|
||||
aliases: make(map[string]string),
|
||||
byFirst: make(map[string][]string),
|
||||
outDegree: make(map[string]int),
|
||||
meanings: make(map[string][]Sense),
|
||||
}
|
||||
|
||||
// Reading meta first also rejects an unrelated database before any bulk
|
||||
// loading happens.
|
||||
declaredWords, err := s.loadMeta(db)
|
||||
declaredWords, declaredMeanings, err := s.loadMeta(db)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -106,7 +125,10 @@ func Open(path string) (*Store, error) {
|
||||
if err := s.loadAliases(db); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := s.validate(declaredWords); err != nil {
|
||||
if err := s.loadMeanings(db); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := s.validate(declaredWords, declaredMeanings); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -125,22 +147,37 @@ func dsn(path string) string {
|
||||
return u.String()
|
||||
}
|
||||
|
||||
func (s *Store) loadMeta(db *sql.DB) (declaredWords int, err error) {
|
||||
func (s *Store) loadMeta(db *sql.DB) (declaredWords, declaredMeanings int, err error) {
|
||||
// The data is CC BY-SA 4.0 and its provenance travels with it.
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'source_license'`).Scan(&s.license); err != nil {
|
||||
return 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
|
||||
return 0, 0, fmt.Errorf("read dictionary metadata (is this a noitu.db?): %w", err)
|
||||
}
|
||||
|
||||
var raw string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = 'word_count'`).Scan(&raw); err != nil {
|
||||
return 0, fmt.Errorf("read dictionary word_count: %w", err)
|
||||
count := func(key string) (int, error) {
|
||||
var raw string
|
||||
if err := db.QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&raw); err != nil {
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
// A database from before the key existed: the fix is a rebuild,
|
||||
// so say so rather than naming a missing row.
|
||||
return 0, fmt.Errorf("dictionary has no %s: it predates builder_version %s — run 'make fetch-dict && make dict' to rebuild it",
|
||||
key, requiredBuilderVersion)
|
||||
}
|
||||
return 0, fmt.Errorf("read dictionary %s: %w", key, err)
|
||||
}
|
||||
n, err := strconv.Atoi(raw)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("dictionary %s %q is not a number: %w", key, raw, err)
|
||||
}
|
||||
return n, nil
|
||||
}
|
||||
declaredWords, err = strconv.Atoi(raw)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("dictionary word_count %q is not a number: %w", raw, err)
|
||||
if declaredWords, err = count("word_count"); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
if declaredMeanings, err = count("meaning_count"); err != nil {
|
||||
return 0, 0, err
|
||||
}
|
||||
|
||||
return declaredWords, nil
|
||||
return declaredWords, declaredMeanings, nil
|
||||
}
|
||||
|
||||
func (s *Store) loadSyllables(db *sql.DB) error {
|
||||
@@ -204,13 +241,35 @@ func (s *Store) loadAliases(db *sql.DB) error {
|
||||
return rows.Err()
|
||||
}
|
||||
|
||||
func (s *Store) loadMeanings(db *sql.DB) error {
|
||||
// Ordered by (word, ord), the primary key, so each word's senses arrive in
|
||||
// page order and append in it.
|
||||
rows, err := db.Query(`SELECT word, pos, gloss FROM meanings ORDER BY word, ord`)
|
||||
if err != nil {
|
||||
return fmt.Errorf("load meanings: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
for rows.Next() {
|
||||
var word string
|
||||
var sense Sense
|
||||
if err := rows.Scan(&word, &sense.Pos, &sense.Gloss); err != nil {
|
||||
return fmt.Errorf("scan meaning: %w", err)
|
||||
}
|
||||
s.meanings[word] = append(s.meanings[word], sense)
|
||||
s.meaningCount++
|
||||
}
|
||||
|
||||
return rows.Err()
|
||||
}
|
||||
|
||||
// validate rejects a structurally valid but wrong dictionary.
|
||||
//
|
||||
// A truncated or empty database has the right schema and opens cleanly, and
|
||||
// the server would then start, reject every word a player types, and fail
|
||||
// every room creation. Checking the loaded rows against what the builder
|
||||
// recorded turns that into a startup failure.
|
||||
func (s *Store) validate(declaredWords int) error {
|
||||
func (s *Store) validate(declaredWords, declaredMeanings int) error {
|
||||
if len(s.words) != declaredWords {
|
||||
return fmt.Errorf("dictionary is incomplete: metadata declares %d words, loaded %d",
|
||||
declaredWords, len(s.words))
|
||||
@@ -218,6 +277,17 @@ func (s *Store) validate(declaredWords int) error {
|
||||
if len(s.words) == 0 {
|
||||
return errors.New("dictionary contains no words")
|
||||
}
|
||||
// A meanings table truncated on disk would otherwise be served silently
|
||||
// as a dictionary without meanings.
|
||||
if s.meaningCount != declaredMeanings {
|
||||
return fmt.Errorf("dictionary is incomplete: metadata declares %d meanings, loaded %d",
|
||||
declaredMeanings, s.meaningCount)
|
||||
}
|
||||
for word := range s.meanings {
|
||||
if _, ok := s.words[word]; !ok {
|
||||
return fmt.Errorf("dictionary is inconsistent: meaning for %q, which is not a word", word)
|
||||
}
|
||||
}
|
||||
|
||||
// A stale syllables table would tell the bot a syllable has continuations
|
||||
// that WordsStartingWith cannot supply.
|
||||
@@ -243,6 +313,20 @@ func (s *Store) WordCount() int { return len(s.words) }
|
||||
// AliasCount reports how many alternative spellings are accepted.
|
||||
func (s *Store) AliasCount() int { return len(s.aliases) }
|
||||
|
||||
// MeaningCount reports how many senses the dictionary holds across all words.
|
||||
func (s *Store) MeaningCount() int { return s.meaningCount }
|
||||
|
||||
// Meanings returns a canonical word's senses in page order, at most five, or
|
||||
// nil for a word with none. Resolve first: an alias has no senses of its own.
|
||||
// The slice is a copy, so a caller cannot reach dictionary state through it.
|
||||
func (s *Store) Meanings(word string) []Sense {
|
||||
senses := s.meanings[word]
|
||||
if len(senses) == 0 {
|
||||
return nil
|
||||
}
|
||||
return slices.Clone(senses)
|
||||
}
|
||||
|
||||
// License reports the licence the dictionary data is distributed under.
|
||||
// Callers are expected to state it at startup.
|
||||
func (s *Store) License() string { return s.license }
|
||||
|
||||
@@ -20,6 +20,7 @@ CREATE TABLE words (word TEXT PRIMARY KEY, first TEXT NOT NULL, last TEXT NOT NU
|
||||
CREATE INDEX idx_words_first ON words(first);
|
||||
CREATE TABLE syllables (syllable TEXT PRIMARY KEY, out_degree INTEGER NOT NULL) WITHOUT ROWID;
|
||||
CREATE TABLE aliases (variant TEXT PRIMARY KEY, canonical TEXT NOT NULL) WITHOUT ROWID;
|
||||
CREATE TABLE meanings (word TEXT NOT NULL, ord INTEGER NOT NULL, pos TEXT NOT NULL, gloss TEXT NOT NULL, PRIMARY KEY (word, ord)) WITHOUT ROWID;
|
||||
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT NOT NULL);
|
||||
`
|
||||
|
||||
@@ -40,7 +41,7 @@ func fixtureAt(tb testing.TB, dir string) string {
|
||||
defer db.Close()
|
||||
|
||||
data := fixtureSchema + `
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','7'),('meaning_count','3');
|
||||
INSERT INTO words VALUES
|
||||
('pháp luật','pháp','luật',2),
|
||||
('pháp lý','pháp','lý',2),
|
||||
@@ -55,6 +56,11 @@ INSERT INTO syllables VALUES
|
||||
('pháp',2),('luật',1),('lý',1),('lệ',0),('do',0),('vô',1),('điện',0),('công',1),('dầu',0),('ngữ',1);
|
||||
-- "pháp lí" drifts in the LAST syllable, "luâto lệ" in the FIRST.
|
||||
INSERT INTO aliases VALUES ('pháp lí','pháp lý'),('luâto lệ','luật lệ');
|
||||
-- Inserted out of order to prove the store sorts by ord, not by insertion.
|
||||
INSERT INTO meanings VALUES
|
||||
('pháp luật',1,'','Kỷ cương nói chung.'),
|
||||
('pháp luật',0,'danh từ','Hệ thống các quy tắc xử sự do nhà nước đặt ra.'),
|
||||
('ngữ pháp',0,'danh từ','Toàn bộ những quy tắc hoạt động của ngôn ngữ.');
|
||||
`
|
||||
if _, err := db.Exec(data); err != nil {
|
||||
tb.Fatal(err)
|
||||
@@ -106,7 +112,7 @@ func TestOpenWrongSchema(t *testing.T) {
|
||||
// every room creation.
|
||||
func TestOpenEmptyDictionary(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999');`)
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999'),('meaning_count','0');`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil {
|
||||
@@ -121,7 +127,7 @@ INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','99999')
|
||||
// syllable has continuations that cannot be supplied.
|
||||
func TestOpenInconsistentOutDegree(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
|
||||
|
||||
@@ -132,7 +138,7 @@ INSERT INTO syllables VALUES ('pháp',7),('luật',0);`)
|
||||
|
||||
func TestOpenOrphanAlias(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','0');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO aliases VALUES ('phap luat','không tồn tại');`)
|
||||
@@ -541,3 +547,73 @@ func BenchmarkRandomOpeningWord(b *testing.B) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMeaningsAreOrderedAndCopied(t *testing.T) {
|
||||
s := fixture(t)
|
||||
|
||||
got := s.Meanings("pháp luật")
|
||||
want := []Sense{
|
||||
{Pos: "danh từ", Gloss: "Hệ thống các quy tắc xử sự do nhà nước đặt ra."},
|
||||
{Pos: "", Gloss: "Kỷ cương nói chung."},
|
||||
}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Errorf("Meanings(pháp luật) = %v, want %v (ordered by ord, not insertion)", got, want)
|
||||
}
|
||||
// A caller writing into the slice must not reach the store.
|
||||
got[0].Gloss = "changed"
|
||||
if s.Meanings("pháp luật")[0].Gloss != want[0].Gloss {
|
||||
t.Error("Meanings handed out the store's own slice")
|
||||
}
|
||||
|
||||
if s.Meanings("pháp lý") != nil {
|
||||
t.Error("a word with no senses returned a non-nil slice")
|
||||
}
|
||||
// An alias is not a word: callers Resolve first.
|
||||
if s.Meanings("pháp lí") != nil {
|
||||
t.Error("an alias returned senses of its own")
|
||||
}
|
||||
if s.MeaningCount() != 3 {
|
||||
t.Errorf("MeaningCount = %d, want 3", s.MeaningCount())
|
||||
}
|
||||
}
|
||||
|
||||
// A meanings table truncated on disk must be refused at startup rather than
|
||||
// served silently as a dictionary without meanings.
|
||||
func TestOpenRefusesMismatchedMeaningCount(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','2');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO meanings VALUES ('pháp luật',0,'danh từ','Luật.');`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "meanings") {
|
||||
t.Fatalf("Open = %v, want a refusal naming the meanings count", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A database built before meanings existed opens cleanly and has every table
|
||||
// but one row. The refusal must say what to do, not which row is missing.
|
||||
func TestOpenRefusesOlderBuilderVersion(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);`)
|
||||
|
||||
_, err := Open(path)
|
||||
if err == nil || !strings.Contains(err.Error(), "make dict") {
|
||||
t.Fatalf("Open = %v, want a refusal that says to rebuild", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenRefusesOrphanMeaning(t *testing.T) {
|
||||
path := writeDB(t, fixtureSchema+`
|
||||
INSERT INTO meta VALUES ('source_license','CC BY-SA 4.0'),('word_count','1'),('meaning_count','1');
|
||||
INSERT INTO words VALUES ('pháp luật','pháp','luật',2);
|
||||
INSERT INTO syllables VALUES ('pháp',1),('luật',0);
|
||||
INSERT INTO meanings VALUES ('không tồn tại',0,'','Một nghĩa.');`)
|
||||
|
||||
if _, err := Open(path); err == nil {
|
||||
t.Fatal("Open succeeded with a meaning for a word that does not exist")
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user