From a947f5ce2ee5d4c7712e4d9edb6643e6faba5ecf Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Fri, 2 Oct 2026 13:46:25 +0700 Subject: [PATCH] feat(build-dictionary): export accepted words as a text corpus and build from it --- server/cmd/build-dictionary/corpus.go | 160 +++++++++++++++++++++ server/cmd/build-dictionary/corpus_test.go | 133 +++++++++++++++++ server/cmd/build-dictionary/dump.go | 11 +- server/cmd/build-dictionary/main.go | 75 +++++++--- 4 files changed, 358 insertions(+), 21 deletions(-) create mode 100644 server/cmd/build-dictionary/corpus.go create mode 100644 server/cmd/build-dictionary/corpus_test.go diff --git a/server/cmd/build-dictionary/corpus.go b/server/cmd/build-dictionary/corpus.go new file mode 100644 index 0000000..ebe6c1a --- /dev/null +++ b/server/cmd/build-dictionary/corpus.go @@ -0,0 +1,160 @@ +package main + +import ( + "bufio" + "errors" + "fmt" + "log" + "os" + "path/filepath" + "sort" + "strings" + "time" +) + +// The corpus is the dictionary as committed text: what a dump build accepted, +// one word per line, so a monthly refresh is a diff a reviewer can read and +// building the image needs no download. It uses the word-list line format — +// `wordpos|gloss…` — so the same reader serves both, and every sense +// is written as `pos|gloss`, even with an empty pos, so a gloss is never +// mistaken for a label. +// +// Provenance travels in `#@key value` header lines, which the word-list +// reader skips as comments. A corpus build requires them: they are what lets +// the database still name the dump, its hash and its licence. + +const corpusHeaderPrefix = "#@" + +// The provenance keys a corpus must carry, in the order they are written. +var corpusProvenanceKeys = []string{"source_url", "source_sha256", "source_pages", "source_fetched_at"} + +const corpusPreamble = `# The noitu dictionary: every word the builder accepted from the Wiktionary +# tiếng Việt dump named below, with the meanings it kept. Derived from +# Wiktionary content under CC BY-SA 4.0; see data/ATTRIBUTION.md. +# +# Generated by ` + "`make refresh-dict`" + `; do not edit by hand. A line is the word, then +# its meanings as tab-separated pos|gloss cells, in page order. +# +` + +// writeCorpus writes the accepted words and meanings as sorted text beside +// the target and renames it into place, so an interrupted export never +// leaves half a dictionary where the committed one was. +func writeCorpus(path string, words map[string]entry, meanings map[string][]sense, prov dumpProvenance) error { + keys := make([]string, 0, len(words)) + for word := range words { + keys = append(keys, word) + } + sort.Strings(keys) + + tmp := path + ".tmp" + f, err := os.Create(tmp) + if err != nil { + return fmt.Errorf("create corpus: %w", err) + } + committed := false + defer func() { + if !committed { + _ = f.Close() + _ = os.Remove(tmp) + } + }() + + w := bufio.NewWriter(f) + _, _ = w.WriteString(corpusPreamble) + values := map[string]string{ + "source_url": dumpSourceURL, + "source_sha256": prov.sha256, + "source_pages": fmt.Sprint(prov.pages), + "source_fetched_at": prov.fetchedAt.Format(time.RFC3339), + } + for _, key := range corpusProvenanceKeys { + _, _ = fmt.Fprintf(w, "%s%s %s\n", corpusHeaderPrefix, key, values[key]) + } + _, _ = w.WriteString("\n") + + for _, word := range keys { + if err := corpusField(word); err != nil { + return fmt.Errorf("word %q: %w", word, err) + } + _, _ = w.WriteString(word) + for i, s := range meanings[word] { + if err := corpusField(s.gloss); err != nil { + return fmt.Errorf("meaning %d of %q: %w", i, word, err) + } + // A pipe in the label would move the split point, so the line + // would read back as a different sense. + if err := corpusField(s.pos); err != nil || strings.Contains(s.pos, "|") { + return fmt.Errorf("label of meaning %d of %q cannot be written as text", i, word) + } + _, _ = fmt.Fprintf(w, "\t%s|%s", s.pos, s.gloss) + } + _, _ = w.WriteString("\n") + } + + if err := w.Flush(); err != nil { + return fmt.Errorf("write corpus: %w", err) + } + if err := f.Close(); err != nil { + return fmt.Errorf("write corpus: %w", err) + } + if err := os.Rename(tmp, path); err != nil { + return fmt.Errorf("move corpus into place: %w", err) + } + committed = true + log.Printf("exported %d words to %s", len(keys), path) + return nil +} + +// corpusField rejects a value the line format cannot carry. The stripper +// should never produce one; this is where a change to it would surface, +// rather than as a corpus that reads back differently from what was written. +func corpusField(s string) error { + if strings.ContainsAny(s, "\t\r\n") { + return errors.New("contains a tab or line break") + } + if s != strings.TrimSpace(s) { + return errors.New("has leading or trailing space, which the reader would trim") + } + return nil +} + +// runFromCorpus builds the database from a committed corpus. It is held to +// what a dump build is held to — the word floor, the meaning coverage — and +// stamps the dump's provenance and licence, because the data is the dump's. +func runFromCorpus(cfg config) error { + raw, err := os.ReadFile(cfg.corpus) + if err != nil { + return fmt.Errorf("read corpus: %w", err) + } + + header := make(map[string]string) + for line := range strings.Lines(string(raw)) { + rest, ok := strings.CutPrefix(strings.TrimSpace(line), corpusHeaderPrefix) + if !ok { + continue + } + key, value, _ := strings.Cut(rest, " ") + header[key] = strings.TrimSpace(value) + } + src := sourceSpec{ + table: "corpus:" + filepath.Base(cfg.corpus), + url: header["source_url"], + license: dumpLicense, + attribution: dumpAttribution, + } + for _, key := range corpusProvenanceKeys { + if header[key] == "" { + return fmt.Errorf("%s has no %s%s line — it is not an exported corpus", cfg.corpus, corpusHeaderPrefix, key) + } + if key != "source_url" { + src.extra = append(src.extra, [2]string{key, header[key]}) + } + } + + words, meanings := readWordList(string(raw)) + log.Printf("accepted %d distinct words, %d with a meaning, from %s (dump sha256 %s)", + len(words), len(meanings), cfg.corpus, header["source_sha256"]) + + return finish(cfg, words, meanings, src, true) +} diff --git a/server/cmd/build-dictionary/corpus_test.go b/server/cmd/build-dictionary/corpus_test.go new file mode 100644 index 0000000..0e8a27c --- /dev/null +++ b/server/cmd/build-dictionary/corpus_test.go @@ -0,0 +1,133 @@ +package main + +import ( + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// exportFixture builds the mini dump with --export and returns both the +// database and the corpus it wrote. +func exportFixture(t *testing.T) (db, corpus string) { + t.Helper() + dir := t.TempDir() + db, corpus = filepath.Join(dir, "from-dump.db"), filepath.Join(dir, "dictionary.txt") + if err := run(config{dump: miniDump, export: corpus, out: db, minWords: 1, minPages: 1}); err != nil { + t.Fatalf("run: %v", err) + } + return db, corpus +} + +// The corpus is only worth committing if building from it gives back the +// database the dump gave: same words, syllables, aliases and meanings, row +// for row. +func TestCorpusRoundTripsTheDumpBuild(t *testing.T) { + fromDump, corpus := exportFixture(t) + fromCorpus := filepath.Join(t.TempDir(), "from-corpus.db") + if err := run(config{corpus: corpus, out: fromCorpus, minWords: 1}); err != nil { + t.Fatalf("run from corpus: %v", err) + } + + db := openOut(t, fromDump) + if _, err := db.Exec(`ATTACH DATABASE ? AS c`, "file:"+fromCorpus+"?mode=ro"); err != nil { + t.Fatal(err) + } + for _, table := range []string{"words", "syllables", "aliases", "meanings"} { + if n := count(t, db, `SELECT COUNT(*) FROM main.`+table); n == 0 && table != "aliases" { + t.Errorf("%s is empty, so the comparison proves nothing", table) + } + for _, q := range []string{ + `SELECT COUNT(*) FROM (SELECT * FROM main.` + table + ` EXCEPT SELECT * FROM c.` + table + `)`, + `SELECT COUNT(*) FROM (SELECT * FROM c.` + table + ` EXCEPT SELECT * FROM main.` + table + `)`, + } { + if n := count(t, db, q); n != 0 { + t.Errorf("%s differs between the dump and corpus builds: %d rows", table, n) + } + } + } +} + +// A corpus build names the dump it came from and carries its licence, not +// the fixture's "no upstream data". +func TestCorpusBuildKeepsTheDumpProvenance(t *testing.T) { + fromDump, corpus := exportFixture(t) + fromCorpus := filepath.Join(t.TempDir(), "from-corpus.db") + if err := run(config{corpus: corpus, out: fromCorpus, minWords: 1}); err != nil { + t.Fatalf("run from corpus: %v", err) + } + + meta := func(path, key string) string { + var v string + if err := openOut(t, path).QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&v); err != nil { + t.Fatalf("meta %s in %s: %v", key, path, err) + } + return v + } + for _, key := range []string{"source_url", "source_license", "attribution", "source_sha256", "source_pages", "source_fetched_at"} { + if got, want := meta(fromCorpus, key), meta(fromDump, key); got != want { + t.Errorf("meta %s = %q, want the dump build's %q", key, got, want) + } + } + if got := meta(fromCorpus, "source_table"); got != "corpus:dictionary.txt" { + t.Errorf("source_table = %q, want the corpus named", got) + } +} + +// Sorted, so a monthly refresh diffs as the words that changed rather than a +// reshuffle. +func TestCorpusIsSorted(t *testing.T) { + _, corpus := exportFixture(t) + raw, err := os.ReadFile(corpus) + if err != nil { + t.Fatal(err) + } + var prev string + for line := range strings.Lines(string(raw)) { + if strings.HasPrefix(line, "#") || strings.TrimSpace(line) == "" { + continue + } + word, _, _ := strings.Cut(strings.TrimRight(line, "\n"), "\t") + if word <= prev { + t.Fatalf("%q follows %q", word, prev) + } + prev = word + } +} + +// A plain word list is not a corpus: without the provenance header the build +// would stamp a licence and a hash it cannot vouch for. +func TestCorpusBuildRequiresTheProvenanceHeader(t *testing.T) { + list := filepath.Join(t.TempDir(), "words.txt") + if err := os.WriteFile(list, []byte("học sinh\tdanh từ|người đi học\n"), 0o644); err != nil { + t.Fatal(err) + } + err := run(config{corpus: list, out: filepath.Join(t.TempDir(), "noitu.db"), minWords: 1}) + if err == nil || !strings.Contains(err.Error(), "not an exported corpus") { + t.Fatalf("err = %v, want the missing header named", err) + } +} + +func TestExportNeedsADump(t *testing.T) { + err := run(config{words: "x.txt", export: "out.txt"}) + if err == nil || !strings.Contains(err.Error(), "--export needs --dump") { + t.Fatalf("err = %v, want --export refused without --dump", err) + } +} + +// A gloss the line format cannot carry fails the export instead of reading +// back as a different meaning. +func TestCorpusRefusesAnUnwritableGloss(t *testing.T) { + words := map[string]entry{"học sinh": {word: "học sinh", first: "học", last: "sinh", syllables: 2}} + for _, gloss := range []string{"one\ttwo", "line\nbreak", " padded"} { + meanings := map[string][]sense{"học sinh": {{pos: "danh từ", gloss: gloss}}} + path := filepath.Join(t.TempDir(), "dictionary.txt") + if err := writeCorpus(path, words, meanings, dumpProvenance{sha256: "x", pages: 1, fetchedAt: time.Now()}); err == nil { + t.Errorf("gloss %q was exported", gloss) + } + if _, err := os.Stat(path); err == nil { + t.Errorf("a failed export of %q left a corpus behind", gloss) + } + } +} diff --git a/server/cmd/build-dictionary/dump.go b/server/cmd/build-dictionary/dump.go index d60150a..f50262d 100644 --- a/server/cmd/build-dictionary/dump.go +++ b/server/cmd/build-dictionary/dump.go @@ -255,12 +255,19 @@ func formatTally(counts map[string]int, n int) string { // dumpSourceSpec describes a dump build for the meta table. With nothing // pinned upstream, the hash and page count of the bytes read are the // provenance. +// The licence and attribution of anything derived from the dump, whether +// built from it directly or from its committed corpus. +const ( + dumpLicense = "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)" + dumpAttribution = "See data/ATTRIBUTION.md for required attribution and the list of modifications." +) + func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec { return sourceSpec{ table: "dump:" + filepath.Base(path), url: dumpSourceURL, - license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)", - attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.", + license: dumpLicense, + attribution: dumpAttribution, extra: [][2]string{ {"source_sha256", prov.sha256}, {"source_pages", fmt.Sprint(prov.pages)}, diff --git a/server/cmd/build-dictionary/main.go b/server/cmd/build-dictionary/main.go index 9049cce..35ec106 100644 --- a/server/cmd/build-dictionary/main.go +++ b/server/cmd/build-dictionary/main.go @@ -12,9 +12,14 @@ // The derived database is a modified version of CC BY-SA 4.0 licensed data. // See data/ATTRIBUTION.md. // +// A dump build can also export what it accepted as data/dictionary.txt, the +// committed corpus, and a corpus build turns that text back into the same +// database without the download. That is how the image is built. +// // Usage: // -// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db +// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --export ../data/dictionary.txt --out ../data/noitu.db +// go run ./cmd/build-dictionary --corpus ../data/dictionary.txt --out ../data/noitu.db package main import ( @@ -53,7 +58,12 @@ type config struct { // words is an alternative source: a plain list, one word per line with an // optional tab-separated meaning column, used to build a small fixture // database without the upstream download. - words string + words string + // corpus is the committed text export of a dump build, read back with + // the dump's provenance and held to the dump's floors. + corpus string + // export, with dump, also writes the accepted words as a corpus. + export string out string minWords int // minPages is the floor on pages with a Vietnamese section. Distinct from @@ -68,6 +78,8 @@ func main() { var cfg config flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read") flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)") + flag.StringVar(&cfg.corpus, "corpus", "", "read a corpus exported by --export, carrying the dump's provenance") + flag.StringVar(&cfg.export, "export", "", "with --dump, also write the accepted words and meanings as a corpus text file") flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write") flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering") flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section") @@ -81,13 +93,23 @@ func main() { func run(cfg config) error { // Exactly one input. Picking silently between two would let a stray flag // ship a corpus nobody meant to build. + inputs := 0 + for _, in := range []string{cfg.dump, cfg.words, cfg.corpus} { + if in != "" { + inputs++ + } + } switch { - case cfg.dump == "" && cfg.words == "": - return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)") - case cfg.dump != "" && cfg.words != "": - return errors.New("--dump and --words are mutually exclusive") + case inputs == 0: + return errors.New("no input given: pass --dump (the upstream dump), --corpus (its committed export) or --words (a plain list)") + case inputs > 1: + return errors.New("--dump, --corpus and --words are mutually exclusive") + case cfg.export != "" && cfg.dump == "": + return errors.New("--export needs --dump: only a dump build has a corpus to export") case cfg.dump != "": return runFromDump(cfg) + case cfg.corpus != "": + return runFromCorpus(cfg) default: return runFromWordList(cfg) } @@ -114,7 +136,15 @@ func runFromDump(cfg config) error { log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s", len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second)) - return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true) + // The database is built and verified first, so a corpus is only ever + // exported from a dump that passed every floor. + if err := finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true); err != nil { + return err + } + if cfg.export == "" { + return nil + } + return writeCorpus(cfg.export, words, meanings, prov) } // finish is the tail every input mode shares: the size floor, alias @@ -160,11 +190,28 @@ func runFromWordList(cfg config) error { return fmt.Errorf("read word list: %w", err) } + words, meanings := readWordList(string(raw)) + log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words) + + // The source spec is what lands in the meta table. Naming the list rather + // than a table makes it obvious in the output which build produced a given + // database — and a hand-written list carries no upstream licence, so the + // fixture must not claim one. + return finish(cfg, words, meanings, sourceSpec{ + table: "wordlist:" + filepath.Base(cfg.words), + license: "none: hand-written fixture wordlist, no upstream data", + attribution: "Fixture written by this project; no third-party attribution applies.", + }, false) +} + +// readWordList parses the word-list line format the fixture and the corpus +// share, passing every word through accept() as a dump title would be. +func readWordList(raw string) (map[string]entry, map[string][]sense) { words := make(map[string]entry) meanings := make(map[string][]sense) rejects := make(map[rejectReason]int) - for line := range strings.Lines(string(raw)) { + for line := range strings.Lines(raw) { line = strings.TrimSpace(line) if line == "" || strings.HasPrefix(line, "#") { continue @@ -187,17 +234,7 @@ func runFromWordList(cfg config) error { } logRejects(rejects) - log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words) - - // The source spec is what lands in the meta table. Naming the list rather - // than a table makes it obvious in the output which build produced a given - // database — and a hand-written list carries no upstream licence, so the - // fixture must not claim one. - return finish(cfg, words, meanings, sourceSpec{ - table: "wordlist:" + filepath.Base(cfg.words), - license: "none: hand-written fixture wordlist, no upstream data", - attribution: "Fixture written by this project; no third-party attribution applies.", - }, false) + return words, meanings } // parseSenses reads the tab-separated meaning cells of a fixture line. A cell