Files
thptqg/parser/cmd/dumpcells/main.go
T
tiennm99 c359a0b444 refactor: one directory per pipeline stage, and an assembler to drive them
The repository now reads as the pipeline it is: crawler fetches, parser
converts, assembler verifies and publishes, with data/ and web/ as the stores
they hand work through. go-parser is renamed parser now that there is no other.

The assembler replaces build-db.js and assemble-site.js. It compiles the
parser, builds and verifies each database, compresses it, runs the Vite build
and assembles _site — one command, and the only place that knows the order.

It also closes a real hole: nothing previously asserted that a database reached
the site. An empty staging directory assembled happily, so every page rendered,
every query 404d and CI stayed green. The row-count and size guards could not
catch that, since they only run when a database was built at all.

Removing Node from the root forced the dataset list out of web/src/datasets.js,
which the assembler cannot import. datasets.json is now the registry both sides
read — JSON because Go and the browser both parse it without a dependency —
while presentation stays in the web app, keyed by id and cross-checked against
the registry so a half-added dataset fails instead of half-working.

Guards verified by making each one fail: a missing database, and an expected
row count one higher than the truth.
2026-08-13 22:50:05 +07:00

89 lines
2.1 KiB
Go

// Command dumpcells emits the canonical cell rendering of a spreadsheet, for
// comparison against the Rust/calamine ground truth produced by
// parser/examples/dump_cells.rs.
//
// The canonical stream carries geometry and rendered cell values only. The
// calamine Data variant is deliberately excluded: Data::Empty and
// Data::String("") both render "" and both count as blank everywhere
// downstream, so the distinction cannot affect the database.
//
// Usage: dumpcells <spreadsheet> [out-file]
package main
import (
"bufio"
"fmt"
"os"
"strings"
"github.com/tiennm99/thptqg/parser/internal/reader"
)
// escape mirrors the Rust dumper so field separators can never break the format.
func escape(s string) string {
var b strings.Builder
b.Grow(len(s))
for _, ch := range s {
switch ch {
case '\\':
b.WriteString(`\\`)
case '\t':
b.WriteString(`\t`)
case '\n':
b.WriteString(`\n`)
case '\r':
b.WriteString(`\r`)
default:
b.WriteRune(ch)
}
}
return b.String()
}
func main() {
if len(os.Args) < 2 {
fmt.Fprintln(os.Stderr, "usage: dumpcells <spreadsheet> [out-file]")
os.Exit(2)
}
path := os.Args[1]
out := os.Stdout
if len(os.Args) > 2 {
f, err := os.Create(os.Args[2])
if err != nil {
fmt.Fprintf(os.Stderr, "create: %v\n", err)
os.Exit(1)
}
defer f.Close()
out = f
}
w := bufio.NewWriterSize(out, 1<<20)
defer w.Flush()
wb, err := reader.Open(path)
if err != nil {
fmt.Fprintf(os.Stderr, "open: %v\n", err)
os.Exit(1)
}
defer wb.Close()
sheets := wb.Sheets()
fmt.Fprintf(w, "FILE\t%s\n", escape(path))
fmt.Fprintf(w, "SHEETCOUNT\t%d\n", len(sheets))
for _, sh := range sheets {
fmt.Fprintf(w, "SHEET\t%d\t%s\t%d\t%d\n", sh.Index, escape(sh.Name), sh.Height, sh.Width)
err := wb.EachRow(sh.Index, func(s reader.Sheet, rowIdx int, row []reader.Cell) error {
fmt.Fprintf(w, "ROW\t%d\t%d\t%d\n", s.Index, rowIdx, len(row))
for c, cell := range row {
fmt.Fprintf(w, "CELL\t%d\t%d\t%d\t%s\n", s.Index, rowIdx, c, escape(cell.Str))
}
return nil
})
if err != nil {
fmt.Fprintf(os.Stderr, "rows: %v\n", err)
os.Exit(1)
}
}
}