Files
noitu/server/cmd/build-dictionary/corpus.go
T

161 lines
5.2 KiB
Go

package main
import (
"bufio"
"errors"
"fmt"
"log"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// The corpus is the dictionary as committed text: what a dump build accepted,
// one word per line, so a monthly refresh is a diff a reviewer can read and
// building the image needs no download. It uses the word-list line format —
// `word<TAB>pos|gloss<TAB>…` — so the same reader serves both, and every sense
// is written as `pos|gloss`, even with an empty pos, so a gloss is never
// mistaken for a label.
//
// Provenance travels in `#@key value` header lines, which the word-list
// reader skips as comments. A corpus build requires them: they are what lets
// the database still name the dump, its hash and its licence.
const corpusHeaderPrefix = "#@"
// The provenance keys a corpus must carry, in the order they are written.
var corpusProvenanceKeys = []string{"source_url", "source_sha256", "source_pages", "source_fetched_at"}
const corpusPreamble = `# The noitu dictionary: every word the builder accepted from the Wiktionary
# tiếng Việt dump named below, with the meanings it kept. Derived from
# Wiktionary content under CC BY-SA 4.0; see data/ATTRIBUTION.md.
#
# Generated by ` + "`make refresh-dict`" + `; do not edit by hand. A line is the word, then
# its meanings as tab-separated pos|gloss cells, in page order.
#
`
// writeCorpus writes the accepted words and meanings as sorted text beside
// the target and renames it into place, so an interrupted export never
// leaves half a dictionary where the committed one was.
func writeCorpus(path string, words map[string]entry, meanings map[string][]sense, prov dumpProvenance) error {
keys := make([]string, 0, len(words))
for word := range words {
keys = append(keys, word)
}
sort.Strings(keys)
tmp := path + ".tmp"
f, err := os.Create(tmp)
if err != nil {
return fmt.Errorf("create corpus: %w", err)
}
committed := false
defer func() {
if !committed {
_ = f.Close()
_ = os.Remove(tmp)
}
}()
w := bufio.NewWriter(f)
_, _ = w.WriteString(corpusPreamble)
values := map[string]string{
"source_url": dumpSourceURL,
"source_sha256": prov.sha256,
"source_pages": fmt.Sprint(prov.pages),
"source_fetched_at": prov.fetchedAt.Format(time.RFC3339),
}
for _, key := range corpusProvenanceKeys {
_, _ = fmt.Fprintf(w, "%s%s %s\n", corpusHeaderPrefix, key, values[key])
}
_, _ = w.WriteString("\n")
for _, word := range keys {
if err := corpusField(word); err != nil {
return fmt.Errorf("word %q: %w", word, err)
}
_, _ = w.WriteString(word)
for i, s := range meanings[word] {
if err := corpusField(s.gloss); err != nil {
return fmt.Errorf("meaning %d of %q: %w", i, word, err)
}
// A pipe in the label would move the split point, so the line
// would read back as a different sense.
if err := corpusField(s.pos); err != nil || strings.Contains(s.pos, "|") {
return fmt.Errorf("label of meaning %d of %q cannot be written as text", i, word)
}
_, _ = fmt.Fprintf(w, "\t%s|%s", s.pos, s.gloss)
}
_, _ = w.WriteString("\n")
}
if err := w.Flush(); err != nil {
return fmt.Errorf("write corpus: %w", err)
}
if err := f.Close(); err != nil {
return fmt.Errorf("write corpus: %w", err)
}
if err := os.Rename(tmp, path); err != nil {
return fmt.Errorf("move corpus into place: %w", err)
}
committed = true
log.Printf("exported %d words to %s", len(keys), path)
return nil
}
// corpusField rejects a value the line format cannot carry. The stripper
// should never produce one; this is where a change to it would surface,
// rather than as a corpus that reads back differently from what was written.
func corpusField(s string) error {
if strings.ContainsAny(s, "\t\r\n") {
return errors.New("contains a tab or line break")
}
if s != strings.TrimSpace(s) {
return errors.New("has leading or trailing space, which the reader would trim")
}
return nil
}
// runFromCorpus builds the database from a committed corpus. It is held to
// what a dump build is held to — the word floor, the meaning coverage — and
// stamps the dump's provenance and licence, because the data is the dump's.
func runFromCorpus(cfg config) error {
raw, err := os.ReadFile(cfg.corpus)
if err != nil {
return fmt.Errorf("read corpus: %w", err)
}
header := make(map[string]string)
for line := range strings.Lines(string(raw)) {
rest, ok := strings.CutPrefix(strings.TrimSpace(line), corpusHeaderPrefix)
if !ok {
continue
}
key, value, _ := strings.Cut(rest, " ")
header[key] = strings.TrimSpace(value)
}
src := sourceSpec{
table: "corpus:" + filepath.Base(cfg.corpus),
url: header["source_url"],
license: dumpLicense,
attribution: dumpAttribution,
}
for _, key := range corpusProvenanceKeys {
if header[key] == "" {
return fmt.Errorf("%s has no %s%s line — it is not an exported corpus", cfg.corpus, corpusHeaderPrefix, key)
}
if key != "source_url" {
src.extra = append(src.extra, [2]string{key, header[key]})
}
}
words, meanings := readWordList(string(raw))
log.Printf("accepted %d distinct words, %d with a meaning, from %s (dump sha256 %s)",
len(words), len(meanings), cfg.corpus, header["source_sha256"])
return finish(cfg, words, meanings, src, true)
}