mirror of
https://github.com/tiennm99/noitu.git
synced 2026-10-11 03:13:45 +00:00
161 lines
5.2 KiB
Go
161 lines
5.2 KiB
Go
package main
|
|
|
|
import (
|
|
"bufio"
|
|
"errors"
|
|
"fmt"
|
|
"log"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// The corpus is the dictionary as committed text: what a dump build accepted,
|
|
// one word per line, so a monthly refresh is a diff a reviewer can read and
|
|
// building the image needs no download. It uses the word-list line format —
|
|
// `word<TAB>pos|gloss<TAB>…` — so the same reader serves both, and every sense
|
|
// is written as `pos|gloss`, even with an empty pos, so a gloss is never
|
|
// mistaken for a label.
|
|
//
|
|
// Provenance travels in `#@key value` header lines, which the word-list
|
|
// reader skips as comments. A corpus build requires them: they are what lets
|
|
// the database still name the dump, its hash and its licence.
|
|
|
|
const corpusHeaderPrefix = "#@"
|
|
|
|
// The provenance keys a corpus must carry, in the order they are written.
|
|
var corpusProvenanceKeys = []string{"source_url", "source_sha256", "source_pages", "source_fetched_at"}
|
|
|
|
const corpusPreamble = `# The noitu dictionary: every word the builder accepted from the Wiktionary
|
|
# tiếng Việt dump named below, with the meanings it kept. Derived from
|
|
# Wiktionary content under CC BY-SA 4.0; see data/ATTRIBUTION.md.
|
|
#
|
|
# Generated by ` + "`make refresh-dict`" + `; do not edit by hand. A line is the word, then
|
|
# its meanings as tab-separated pos|gloss cells, in page order.
|
|
#
|
|
`
|
|
|
|
// writeCorpus writes the accepted words and meanings as sorted text beside
|
|
// the target and renames it into place, so an interrupted export never
|
|
// leaves half a dictionary where the committed one was.
|
|
func writeCorpus(path string, words map[string]entry, meanings map[string][]sense, prov dumpProvenance) error {
|
|
keys := make([]string, 0, len(words))
|
|
for word := range words {
|
|
keys = append(keys, word)
|
|
}
|
|
sort.Strings(keys)
|
|
|
|
tmp := path + ".tmp"
|
|
f, err := os.Create(tmp)
|
|
if err != nil {
|
|
return fmt.Errorf("create corpus: %w", err)
|
|
}
|
|
committed := false
|
|
defer func() {
|
|
if !committed {
|
|
_ = f.Close()
|
|
_ = os.Remove(tmp)
|
|
}
|
|
}()
|
|
|
|
w := bufio.NewWriter(f)
|
|
_, _ = w.WriteString(corpusPreamble)
|
|
values := map[string]string{
|
|
"source_url": dumpSourceURL,
|
|
"source_sha256": prov.sha256,
|
|
"source_pages": fmt.Sprint(prov.pages),
|
|
"source_fetched_at": prov.fetchedAt.Format(time.RFC3339),
|
|
}
|
|
for _, key := range corpusProvenanceKeys {
|
|
_, _ = fmt.Fprintf(w, "%s%s %s\n", corpusHeaderPrefix, key, values[key])
|
|
}
|
|
_, _ = w.WriteString("\n")
|
|
|
|
for _, word := range keys {
|
|
if err := corpusField(word); err != nil {
|
|
return fmt.Errorf("word %q: %w", word, err)
|
|
}
|
|
_, _ = w.WriteString(word)
|
|
for i, s := range meanings[word] {
|
|
if err := corpusField(s.gloss); err != nil {
|
|
return fmt.Errorf("meaning %d of %q: %w", i, word, err)
|
|
}
|
|
// A pipe in the label would move the split point, so the line
|
|
// would read back as a different sense.
|
|
if err := corpusField(s.pos); err != nil || strings.Contains(s.pos, "|") {
|
|
return fmt.Errorf("label of meaning %d of %q cannot be written as text", i, word)
|
|
}
|
|
_, _ = fmt.Fprintf(w, "\t%s|%s", s.pos, s.gloss)
|
|
}
|
|
_, _ = w.WriteString("\n")
|
|
}
|
|
|
|
if err := w.Flush(); err != nil {
|
|
return fmt.Errorf("write corpus: %w", err)
|
|
}
|
|
if err := f.Close(); err != nil {
|
|
return fmt.Errorf("write corpus: %w", err)
|
|
}
|
|
if err := os.Rename(tmp, path); err != nil {
|
|
return fmt.Errorf("move corpus into place: %w", err)
|
|
}
|
|
committed = true
|
|
log.Printf("exported %d words to %s", len(keys), path)
|
|
return nil
|
|
}
|
|
|
|
// corpusField rejects a value the line format cannot carry. The stripper
|
|
// should never produce one; this is where a change to it would surface,
|
|
// rather than as a corpus that reads back differently from what was written.
|
|
func corpusField(s string) error {
|
|
if strings.ContainsAny(s, "\t\r\n") {
|
|
return errors.New("contains a tab or line break")
|
|
}
|
|
if s != strings.TrimSpace(s) {
|
|
return errors.New("has leading or trailing space, which the reader would trim")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// runFromCorpus builds the database from a committed corpus. It is held to
|
|
// what a dump build is held to — the word floor, the meaning coverage — and
|
|
// stamps the dump's provenance and licence, because the data is the dump's.
|
|
func runFromCorpus(cfg config) error {
|
|
raw, err := os.ReadFile(cfg.corpus)
|
|
if err != nil {
|
|
return fmt.Errorf("read corpus: %w", err)
|
|
}
|
|
|
|
header := make(map[string]string)
|
|
for line := range strings.Lines(string(raw)) {
|
|
rest, ok := strings.CutPrefix(strings.TrimSpace(line), corpusHeaderPrefix)
|
|
if !ok {
|
|
continue
|
|
}
|
|
key, value, _ := strings.Cut(rest, " ")
|
|
header[key] = strings.TrimSpace(value)
|
|
}
|
|
src := sourceSpec{
|
|
table: "corpus:" + filepath.Base(cfg.corpus),
|
|
url: header["source_url"],
|
|
license: dumpLicense,
|
|
attribution: dumpAttribution,
|
|
}
|
|
for _, key := range corpusProvenanceKeys {
|
|
if header[key] == "" {
|
|
return fmt.Errorf("%s has no %s%s line — it is not an exported corpus", cfg.corpus, corpusHeaderPrefix, key)
|
|
}
|
|
if key != "source_url" {
|
|
src.extra = append(src.extra, [2]string{key, header[key]})
|
|
}
|
|
}
|
|
|
|
words, meanings := readWordList(string(raw))
|
|
log.Printf("accepted %d distinct words, %d with a meaning, from %s (dump sha256 %s)",
|
|
len(words), len(meanings), cfg.corpus, header["source_sha256"])
|
|
|
|
return finish(cfg, words, meanings, src, true)
|
|
}
|