feat(build-dictionary): export accepted words as a text corpus and build from it

This commit is contained in:
tiennm99 committed 2026-10-02 13:46:25 +07:00
1 parent 0cb1b74952
commit a947f5ce2e
4 files changed
+358 -21

No files matched your search

+160
View File
@@ -0,0 +1,160 @@
package main
import (
"bufio"
"errors"
"fmt"
"log"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// The corpus is the dictionary as committed text: what a dump build accepted,
// one word per line, so a monthly refresh is a diff a reviewer can read and
// building the image needs no download. It uses the word-list line format —
// `word<TAB>pos|gloss<TAB>…` — so the same reader serves both, and every sense
// is written as `pos|gloss`, even with an empty pos, so a gloss is never
// mistaken for a label.
//
// Provenance travels in `#@key value` header lines, which the word-list
// reader skips as comments. A corpus build requires them: they are what lets
// the database still name the dump, its hash and its licence.
const corpusHeaderPrefix = "#@"
// The provenance keys a corpus must carry, in the order they are written.
var corpusProvenanceKeys = []string{"source_url", "source_sha256", "source_pages", "source_fetched_at"}
const corpusPreamble = `# The noitu dictionary: every word the builder accepted from the Wiktionary
# tiếng Việt dump named below, with the meanings it kept. Derived from
# Wiktionary content under CC BY-SA 4.0; see data/ATTRIBUTION.md.
#
# Generated by ` + "`make refresh-dict`" + `; do not edit by hand. A line is the word, then
# its meanings as tab-separated pos|gloss cells, in page order.
#
`
// writeCorpus writes the accepted words and meanings as sorted text beside
// the target and renames it into place, so an interrupted export never
// leaves half a dictionary where the committed one was.
func writeCorpus(path string, words map[string]entry, meanings map[string][]sense, prov dumpProvenance) error {
keys := make([]string, 0, len(words))
for word := range words {
keys = append(keys, word)
}
sort.Strings(keys)
tmp := path + ".tmp"
f, err := os.Create(tmp)
if err != nil {
return fmt.Errorf("create corpus: %w", err)
}
committed := false
defer func() {
if !committed {
_ = f.Close()
_ = os.Remove(tmp)
}
}()
w := bufio.NewWriter(f)
_, _ = w.WriteString(corpusPreamble)
values := map[string]string{
"source_url": dumpSourceURL,
"source_sha256": prov.sha256,
"source_pages": fmt.Sprint(prov.pages),
"source_fetched_at": prov.fetchedAt.Format(time.RFC3339),
}
for _, key := range corpusProvenanceKeys {
_, _ = fmt.Fprintf(w, "%s%s %s\n", corpusHeaderPrefix, key, values[key])
}
_, _ = w.WriteString("\n")
for _, word := range keys {
if err := corpusField(word); err != nil {
return fmt.Errorf("word %q: %w", word, err)
}
_, _ = w.WriteString(word)
for i, s := range meanings[word] {
if err := corpusField(s.gloss); err != nil {
return fmt.Errorf("meaning %d of %q: %w", i, word, err)
}
// A pipe in the label would move the split point, so the line
// would read back as a different sense.
if err := corpusField(s.pos); err != nil || strings.Contains(s.pos, "|") {
return fmt.Errorf("label of meaning %d of %q cannot be written as text", i, word)
}
_, _ = fmt.Fprintf(w, "\t%s|%s", s.pos, s.gloss)
}
_, _ = w.WriteString("\n")
}
if err := w.Flush(); err != nil {
return fmt.Errorf("write corpus: %w", err)
}
if err := f.Close(); err != nil {
return fmt.Errorf("write corpus: %w", err)
}
if err := os.Rename(tmp, path); err != nil {
return fmt.Errorf("move corpus into place: %w", err)
}
committed = true
log.Printf("exported %d words to %s", len(keys), path)
return nil
}
// corpusField rejects a value the line format cannot carry. The stripper
// should never produce one; this is where a change to it would surface,
// rather than as a corpus that reads back differently from what was written.
func corpusField(s string) error {
if strings.ContainsAny(s, "\t\r\n") {
return errors.New("contains a tab or line break")
}
if s != strings.TrimSpace(s) {
return errors.New("has leading or trailing space, which the reader would trim")
}
return nil
}
// runFromCorpus builds the database from a committed corpus. It is held to
// what a dump build is held to — the word floor, the meaning coverage — and
// stamps the dump's provenance and licence, because the data is the dump's.
func runFromCorpus(cfg config) error {
raw, err := os.ReadFile(cfg.corpus)
if err != nil {
return fmt.Errorf("read corpus: %w", err)
}
header := make(map[string]string)
for line := range strings.Lines(string(raw)) {
rest, ok := strings.CutPrefix(strings.TrimSpace(line), corpusHeaderPrefix)
if !ok {
continue
}
key, value, _ := strings.Cut(rest, " ")
header[key] = strings.TrimSpace(value)
}
src := sourceSpec{
table: "corpus:" + filepath.Base(cfg.corpus),
url: header["source_url"],
license: dumpLicense,
attribution: dumpAttribution,
}
for _, key := range corpusProvenanceKeys {
if header[key] == "" {
return fmt.Errorf("%s has no %s%s line — it is not an exported corpus", cfg.corpus, corpusHeaderPrefix, key)
}
if key != "source_url" {
src.extra = append(src.extra, [2]string{key, header[key]})
}
}
words, meanings := readWordList(string(raw))
log.Printf("accepted %d distinct words, %d with a meaning, from %s (dump sha256 %s)",
len(words), len(meanings), cfg.corpus, header["source_sha256"])
return finish(cfg, words, meanings, src, true)
}
+133
View File
@@ -0,0 +1,133 @@
package main
import (
"os"
"path/filepath"
"strings"
"testing"
"time"
)
// exportFixture builds the mini dump with --export and returns both the
// database and the corpus it wrote.
func exportFixture(t *testing.T) (db, corpus string) {
t.Helper()
dir := t.TempDir()
db, corpus = filepath.Join(dir, "from-dump.db"), filepath.Join(dir, "dictionary.txt")
if err := run(config{dump: miniDump, export: corpus, out: db, minWords: 1, minPages: 1}); err != nil {
t.Fatalf("run: %v", err)
}
return db, corpus
}
// The corpus is only worth committing if building from it gives back the
// database the dump gave: same words, syllables, aliases and meanings, row
// for row.
func TestCorpusRoundTripsTheDumpBuild(t *testing.T) {
fromDump, corpus := exportFixture(t)
fromCorpus := filepath.Join(t.TempDir(), "from-corpus.db")
if err := run(config{corpus: corpus, out: fromCorpus, minWords: 1}); err != nil {
t.Fatalf("run from corpus: %v", err)
}
db := openOut(t, fromDump)
if _, err := db.Exec(`ATTACH DATABASE ? AS c`, "file:"+fromCorpus+"?mode=ro"); err != nil {
t.Fatal(err)
}
for _, table := range []string{"words", "syllables", "aliases", "meanings"} {
if n := count(t, db, `SELECT COUNT(*) FROM main.`+table); n == 0 && table != "aliases" {
t.Errorf("%s is empty, so the comparison proves nothing", table)
}
for _, q := range []string{
`SELECT COUNT(*) FROM (SELECT * FROM main.` + table + ` EXCEPT SELECT * FROM c.` + table + `)`,
`SELECT COUNT(*) FROM (SELECT * FROM c.` + table + ` EXCEPT SELECT * FROM main.` + table + `)`,
} {
if n := count(t, db, q); n != 0 {
t.Errorf("%s differs between the dump and corpus builds: %d rows", table, n)
}
}
}
}
// A corpus build names the dump it came from and carries its licence, not
// the fixture's "no upstream data".
func TestCorpusBuildKeepsTheDumpProvenance(t *testing.T) {
fromDump, corpus := exportFixture(t)
fromCorpus := filepath.Join(t.TempDir(), "from-corpus.db")
if err := run(config{corpus: corpus, out: fromCorpus, minWords: 1}); err != nil {
t.Fatalf("run from corpus: %v", err)
}
meta := func(path, key string) string {
var v string
if err := openOut(t, path).QueryRow(`SELECT value FROM meta WHERE key = ?`, key).Scan(&v); err != nil {
t.Fatalf("meta %s in %s: %v", key, path, err)
}
return v
}
for _, key := range []string{"source_url", "source_license", "attribution", "source_sha256", "source_pages", "source_fetched_at"} {
if got, want := meta(fromCorpus, key), meta(fromDump, key); got != want {
t.Errorf("meta %s = %q, want the dump build's %q", key, got, want)
}
}
if got := meta(fromCorpus, "source_table"); got != "corpus:dictionary.txt" {
t.Errorf("source_table = %q, want the corpus named", got)
}
}
// Sorted, so a monthly refresh diffs as the words that changed rather than a
// reshuffle.
func TestCorpusIsSorted(t *testing.T) {
_, corpus := exportFixture(t)
raw, err := os.ReadFile(corpus)
if err != nil {
t.Fatal(err)
}
var prev string
for line := range strings.Lines(string(raw)) {
if strings.HasPrefix(line, "#") || strings.TrimSpace(line) == "" {
continue
}
word, _, _ := strings.Cut(strings.TrimRight(line, "\n"), "\t")
if word <= prev {
t.Fatalf("%q follows %q", word, prev)
}
prev = word
}
}
// A plain word list is not a corpus: without the provenance header the build
// would stamp a licence and a hash it cannot vouch for.
func TestCorpusBuildRequiresTheProvenanceHeader(t *testing.T) {
list := filepath.Join(t.TempDir(), "words.txt")
if err := os.WriteFile(list, []byte("học sinh\tdanh từ|người đi học\n"), 0o644); err != nil {
t.Fatal(err)
}
err := run(config{corpus: list, out: filepath.Join(t.TempDir(), "noitu.db"), minWords: 1})
if err == nil || !strings.Contains(err.Error(), "not an exported corpus") {
t.Fatalf("err = %v, want the missing header named", err)
}
}
func TestExportNeedsADump(t *testing.T) {
err := run(config{words: "x.txt", export: "out.txt"})
if err == nil || !strings.Contains(err.Error(), "--export needs --dump") {
t.Fatalf("err = %v, want --export refused without --dump", err)
}
}
// A gloss the line format cannot carry fails the export instead of reading
// back as a different meaning.
func TestCorpusRefusesAnUnwritableGloss(t *testing.T) {
words := map[string]entry{"học sinh": {word: "học sinh", first: "học", last: "sinh", syllables: 2}}
for _, gloss := range []string{"one\ttwo", "line\nbreak", " padded"} {
meanings := map[string][]sense{"học sinh": {{pos: "danh từ", gloss: gloss}}}
path := filepath.Join(t.TempDir(), "dictionary.txt")
if err := writeCorpus(path, words, meanings, dumpProvenance{sha256: "x", pages: 1, fetchedAt: time.Now()}); err == nil {
t.Errorf("gloss %q was exported", gloss)
}
if _, err := os.Stat(path); err == nil {
t.Errorf("a failed export of %q left a corpus behind", gloss)
}
}
}
+9 -2
View File
@@ -255,12 +255,19 @@ func formatTally(counts map[string]int, n int) string {
// dumpSourceSpec describes a dump build for the meta table. With nothing
// pinned upstream, the hash and page count of the bytes read are the
// provenance.
// The licence and attribution of anything derived from the dump, whether
// built from it directly or from its committed corpus.
const (
dumpLicense = "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)"
dumpAttribution = "See data/ATTRIBUTION.md for required attribution and the list of modifications."
)
func dumpSourceSpec(path string, prov dumpProvenance) sourceSpec {
return sourceSpec{
table: "dump:" + filepath.Base(path),
url: dumpSourceURL,
license: "CC BY-SA 4.0 (https://creativecommons.org/licenses/by-sa/4.0/)",
attribution: "See data/ATTRIBUTION.md for required attribution and the list of modifications.",
license: dumpLicense,
attribution: dumpAttribution,
extra: [][2]string{
{"source_sha256", prov.sha256},
{"source_pages", fmt.Sprint(prov.pages)},
+56 -19
View File
@@ -12,9 +12,14 @@
// The derived database is a modified version of CC BY-SA 4.0 licensed data.
// See data/ATTRIBUTION.md.
//
// A dump build can also export what it accepted as data/dictionary.txt, the
// committed corpus, and a corpus build turns that text back into the same
// database without the download. That is how the image is built.
//
// Usage:
//
// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --out ../data/noitu.db
// go run ./cmd/build-dictionary --dump ../data/viwiktionary-latest-pages-articles.xml.bz2 --export ../data/dictionary.txt --out ../data/noitu.db
// go run ./cmd/build-dictionary --corpus ../data/dictionary.txt --out ../data/noitu.db
package main
import (
@@ -53,7 +58,12 @@ type config struct {
// words is an alternative source: a plain list, one word per line with an
// optional tab-separated meaning column, used to build a small fixture
// database without the upstream download.
words string
words string
// corpus is the committed text export of a dump build, read back with
// the dump's provenance and held to the dump's floors.
corpus string
// export, with dump, also writes the accepted words as a corpus.
export string
out string
minWords int
// minPages is the floor on pages with a Vietnamese section. Distinct from
@@ -68,6 +78,8 @@ func main() {
var cfg config
flag.StringVar(&cfg.dump, "dump", "", "upstream Wikimedia pages-articles.xml.bz2 dump to read")
flag.StringVar(&cfg.words, "words", "", "read a plain word list instead of the dump (one word per line, optional tab-separated meanings, # comments)")
flag.StringVar(&cfg.corpus, "corpus", "", "read a corpus exported by --export, carrying the dump's provenance")
flag.StringVar(&cfg.export, "export", "", "with --dump, also write the accepted words and meanings as a corpus text file")
flag.StringVar(&cfg.out, "out", "../data/noitu.db", "derived database to write")
flag.IntVar(&cfg.minWords, "min-words", 30000, "fail if fewer words survive filtering")
flag.IntVar(&cfg.minPages, "min-pages", 20000, "fail if the dump has fewer pages with a Vietnamese section")
@@ -81,13 +93,23 @@ func main() {
func run(cfg config) error {
// Exactly one input. Picking silently between two would let a stray flag
// ship a corpus nobody meant to build.
inputs := 0
for _, in := range []string{cfg.dump, cfg.words, cfg.corpus} {
if in != "" {
inputs++
}
}
switch {
case cfg.dump == "" && cfg.words == "":
return errors.New("no input given: pass --dump (the corpus) or --words (a plain list)")
case cfg.dump != "" && cfg.words != "":
return errors.New("--dump and --words are mutually exclusive")
case inputs == 0:
return errors.New("no input given: pass --dump (the upstream dump), --corpus (its committed export) or --words (a plain list)")
case inputs > 1:
return errors.New("--dump, --corpus and --words are mutually exclusive")
case cfg.export != "" && cfg.dump == "":
return errors.New("--export needs --dump: only a dump build has a corpus to export")
case cfg.dump != "":
return runFromDump(cfg)
case cfg.corpus != "":
return runFromCorpus(cfg)
default:
return runFromWordList(cfg)
}
@@ -114,7 +136,15 @@ func runFromDump(cfg config) error {
log.Printf("accepted %d distinct words, %d with a meaning, from %s (%d pages, sha256 %s) in %s",
len(words), len(meanings), cfg.dump, prov.pages, prov.sha256, time.Since(started).Round(time.Second))
return finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true)
// The database is built and verified first, so a corpus is only ever
// exported from a dump that passed every floor.
if err := finish(cfg, words, meanings, dumpSourceSpec(cfg.dump, prov), true); err != nil {
return err
}
if cfg.export == "" {
return nil
}
return writeCorpus(cfg.export, words, meanings, prov)
}
// finish is the tail every input mode shares: the size floor, alias
@@ -160,11 +190,28 @@ func runFromWordList(cfg config) error {
return fmt.Errorf("read word list: %w", err)
}
words, meanings := readWordList(string(raw))
log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words)
// The source spec is what lands in the meta table. Naming the list rather
// than a table makes it obvious in the output which build produced a given
// database — and a hand-written list carries no upstream licence, so the
// fixture must not claim one.
return finish(cfg, words, meanings, sourceSpec{
table: "wordlist:" + filepath.Base(cfg.words),
license: "none: hand-written fixture wordlist, no upstream data",
attribution: "Fixture written by this project; no third-party attribution applies.",
}, false)
}
// readWordList parses the word-list line format the fixture and the corpus
// share, passing every word through accept() as a dump title would be.
func readWordList(raw string) (map[string]entry, map[string][]sense) {
words := make(map[string]entry)
meanings := make(map[string][]sense)
rejects := make(map[rejectReason]int)
for line := range strings.Lines(string(raw)) {
for line := range strings.Lines(raw) {
line = strings.TrimSpace(line)
if line == "" || strings.HasPrefix(line, "#") {
continue
@@ -187,17 +234,7 @@ func runFromWordList(cfg config) error {
}
logRejects(rejects)
log.Printf("accepted %d distinct words, %d with a meaning, from %s", len(words), len(meanings), cfg.words)
// The source spec is what lands in the meta table. Naming the list rather
// than a table makes it obvious in the output which build produced a given
// database — and a hand-written list carries no upstream licence, so the
// fixture must not claim one.
return finish(cfg, words, meanings, sourceSpec{
table: "wordlist:" + filepath.Base(cfg.words),
license: "none: hand-written fixture wordlist, no upstream data",
attribution: "Fixture written by this project; no third-party attribution applies.",
}, false)
return words, meanings
}
// parseSenses reads the tab-separated meaning cells of a fixture line. A cell