From cd4b07f9cb9401b7a6782953d36e6e59f3473f19 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Thu, 13 Aug 2026 11:16:24 +0700 Subject: [PATCH 1/7] refactor(parser): define one canonical student schema for all datasets MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 2016 and 2017 parsers were the same crate with divergent SQL. DDL, INSERT and the subject regex table lived in each dataset's TOML config, so four copies had to be kept in step by hand — which is how the two schemas drifted apart. Move all of it into src/schema.rs as a single 22-column definition: 6 identity columns (adding ten_cum_thi and gioi_tinh, previously 2016-only) and 16 subject columns (the union of both exam years). Columns a dataset carries no data for bind NULL, costing ~1 byte per row. This collapses writer.rs to one insert path: insert_row_2016, SCORE_FIELDS_2016, SCORE_FIELDS_2017 and the SCORE_FIELDS alias all go away. Configs shrink to the parse rules that genuinely vary per dataset — sheet mode, column indices, SBD validation, header tokens, blank-row stripping — and carry no SQL at all. Config parsing now denies unknown fields, so a leftover [schema] block fails loudly instead of looking effective while schema.rs drives the build. idx_ten_cum_thi is partial, so it holds zero entries on the three 2017 datasets where the column is always NULL. Adds db-stats.js and verify-parity.js to prove no data moved: row counts, per-column non-NULL counts and a deterministic student sample are compared against databases built from the previous code. Verified across all four datasets — row counts, every pre-existing column count, and all sampled students are identical. Database size grows 1.4-2.2%. --- .../xlsxread/configs/thptqg2016-data.toml | 64 +- .../xlsxread/configs/thptqg2017-data-old.toml | 65 +- .../configs/thptqg2017-data-old2.toml | 26 + .../xlsxread/configs/thptqg2017-data.toml | 65 +- 2016/tools/xlsxread/src/config.rs | 56 +- 2016/tools/xlsxread/src/format_detect_2016.rs | 9 +- 2016/tools/xlsxread/src/lib.rs | 1 + 2016/tools/xlsxread/src/main.rs | 16 +- 2016/tools/xlsxread/src/schema.rs | 213 + 2016/tools/xlsxread/src/transform.rs | 25 +- 2016/tools/xlsxread/src/writer.rs | 110 +- 2016/tools/xlsxread/tests/golden.rs | 12 +- 2017/scripts/db-stats.js | 79 + 2017/scripts/verify-parity.js | 157 + ...e-01-standard-schema-and-unified-parser.md | 170 + .../phase-02-repo-restructure.md | 149 + ...3-unified-frontend-and-dataset-registry.md | 249 + .../phase-04-build-and-deploy-pipeline.md | 185 + .../phase-05-parity-verification-and-docs.md | 128 + .../plan.md | 184 + plans/reports/parser-parity-baseline.json | 3910 ++++++++++++++ plans/reports/parser-parity-current.json | 4686 +++++++++++++++++ 22 files changed, 10218 insertions(+), 341 deletions(-) create mode 100644 2016/tools/xlsxread/configs/thptqg2017-data-old2.toml create mode 100644 2016/tools/xlsxread/src/schema.rs create mode 100644 2017/scripts/db-stats.js create mode 100644 2017/scripts/verify-parity.js create mode 100644 plans/260813-0956-unify-frontend-standard-schema/phase-01-standard-schema-and-unified-parser.md create mode 100644 plans/260813-0956-unify-frontend-standard-schema/phase-02-repo-restructure.md create mode 100644 plans/260813-0956-unify-frontend-standard-schema/phase-03-unified-frontend-and-dataset-registry.md create mode 100644 plans/260813-0956-unify-frontend-standard-schema/phase-04-build-and-deploy-pipeline.md create mode 100644 plans/260813-0956-unify-frontend-standard-schema/phase-05-parity-verification-and-docs.md create mode 100644 plans/260813-0956-unify-frontend-standard-schema/plan.md create mode 100644 plans/reports/parser-parity-baseline.json create mode 100644 plans/reports/parser-parity-current.json diff --git a/2016/tools/xlsxread/configs/thptqg2016-data.toml b/2016/tools/xlsxread/configs/thptqg2016-data.toml index 646ad68..6bde781 100644 --- a/2016/tools/xlsxread/configs/thptqg2016-data.toml +++ b/2016/tools/xlsxread/configs/thptqg2016-data.toml @@ -1,4 +1,4 @@ -# Config for thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files. +# thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files. # # Three column layouts exist across the 119 files; the binary selects the # right one per-file at runtime via format_detection = "thptqg2016": @@ -7,15 +7,13 @@ # mapped header SOBAODANH|SBD + DIEM_THI — most provinces # default no header; positional 6-col layout — remaining files # -# Schema differences from thptqg2017: -# + ten_cum_thi TEXT (exam-cluster name, TEN_CUMTHI column) -# + gioi_tinh TEXT (gender: "Nam"/"Nữ", GIOI_TINH column) -# + tieng_duc REAL (German) -# + tieng_nhat REAL (Japanese) -# - khtn, khxh, tieng_nga (not in 2016 dataset) +# This is the only dataset that populates ten_cum_thi and gioi_tinh, and the +# only one whose files carry Tiếng Đức / Tiếng Nhật scores. # # sheet_mode = "all": several provinces overflow into Sheet2 (65k Excel row cap). # strip_blank_rows = false: no blank-row anomaly observed in this dataset. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. format_detection = "thptqg2016" @@ -32,55 +30,3 @@ require_nonempty_sbd = true # Tokens that identify a header row by first-cell content (uppercased). # Covers both SOBAODANH-style and SBD-style headers. tokens = ["SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - ten_cum_thi TEXT, - gioi_tinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - lich_su REAL, - dia_ly REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_duc REAL, - tieng_nhat REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_duc = 'Tiếng Đức:\s*(\d+(?:\.\d+)?)' -tieng_nhat = 'Tiếng Nhật:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, - lich_su, dia_ly, - tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/configs/thptqg2017-data-old.toml b/2016/tools/xlsxread/configs/thptqg2017-data-old.toml index 34d59c0..f21f8aa 100644 --- a/2016/tools/xlsxread/configs/thptqg2017-data-old.toml +++ b/2016/tools/xlsxread/configs/thptqg2017-data-old.toml @@ -1,8 +1,10 @@ -# Test-only config used by golden integration tests (data-old variant). -# Same column layout as thptqg2017-data.toml but with: -# - sheet_mode = "first" (reads only first sheet) -# - require_numeric_sbd = true (rejects non-digit SBDs) -# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols). +# thptqg2017 data-old/ — 63 .xlsx files (pre-baotintuc refresh). +# +# sheet_mode = "first": single-sheet workbooks, never hit the 65k row cap. +# SBD validation: require ^\d+$ (build-database-old.js:55 guard). +# strip_blank_rows = false: no explicit blank-skip in build-database-old.js. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. [reader] sheet_mode = "first" @@ -21,56 +23,3 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - khtn REAL, - lich_su REAL, - dia_ly REAL, - gdcd REAL, - khxh REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_nga REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' -khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, - lich_su, dia_ly, gdcd, khxh, - tieng_anh, tieng_phap, tieng_nga, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml b/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml new file mode 100644 index 0000000..c4fd6b7 --- /dev/null +++ b/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml @@ -0,0 +1,26 @@ +# thptqg2017 data-old2/ — 54 .xlsx files (corrected-export set). +# +# sheet_mode = "all": 24.HCM_UTLQ.xlsx overflows into Sheet2 (+6,446 rows). +# SBD validation: require ^\d+$ (build-database-old2.js:57 guard). +# strip_blank_rows = true: skip fully blank rows BEFORE counting sourceRows +# (build-database-old2.js:50-51 checks blank before sourceRows++). +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. + +[reader] +sheet_mode = "all" +strip_blank_rows = true + +[columns] +ho_ten = 0 +ngay_sinh = 1 +so_bao_danh = 2 +diem_thi = 3 + +[validation] +require_numeric_sbd = true +require_nonempty_name = true +require_nonempty_sbd = true + +[header] +tokens = ["HO_TEN", "HỌ TÊN", "STT"] diff --git a/2016/tools/xlsxread/configs/thptqg2017-data.toml b/2016/tools/xlsxread/configs/thptqg2017-data.toml index 892fc6f..99dccbb 100644 --- a/2016/tools/xlsxread/configs/thptqg2017-data.toml +++ b/2016/tools/xlsxread/configs/thptqg2017-data.toml @@ -1,8 +1,10 @@ -# Test-only config used by golden integration tests. -# Matches the fixture row layout produced by tests/golden.rs: write_xlsx() -# header: HO_TEN(0) NGAY_SINH(1) SO_BAO_DANH(2) DIEM_THI(3) -# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols) so run_build_cmd -# in golden.rs can call insert_row(..., SCORE_FIELDS) without modification. +# thptqg2017 data/ — 63 .xls files from baotintuc.vn (current generation). +# +# sheet_mode = "all": Hà Nội and HCM overflow into Sheet2 (65k Excel row cap). +# SBD validation: no numeric guard — build-database.js did not apply ^\d+$. +# strip_blank_rows = false: no blank-row anomaly in this dataset. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. [reader] sheet_mode = "all" @@ -21,56 +23,3 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - khtn REAL, - lich_su REAL, - dia_ly REAL, - gdcd REAL, - khxh REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_nga REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' -khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, - lich_su, dia_ly, gdcd, khxh, - tieng_anh, tieng_phap, tieng_nga, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/src/config.rs b/2016/tools/xlsxread/src/config.rs index e2aaa8e..c23e090 100644 --- a/2016/tools/xlsxread/src/config.rs +++ b/2016/tools/xlsxread/src/config.rs @@ -1,4 +1,3 @@ -use std::collections::HashMap; use std::fs; use std::path::Path; @@ -10,7 +9,14 @@ use crate::error::BuildError; // Top-level dataset configuration loaded from a .toml file // --------------------------------------------------------------------------- +/// Per-dataset parse rules. +/// +/// Deliberately carries no SQL. The table shape, the INSERT and the subject +/// regexes are identical for every dataset and live in `crate::schema` — keeping +/// them here meant four copies of the same DDL, which is how the 2016 and 2017 +/// schemas drifted apart. #[derive(Debug, Deserialize, Clone)] +#[serde(deny_unknown_fields)] pub struct DatasetConfig { pub reader: ReaderCfg, /// Fixed column indices. Optional when format_detection handles per-file mapping. @@ -18,10 +24,6 @@ pub struct DatasetConfig { pub columns: Option, pub validation: ValidationCfg, pub header: HeaderCfg, - pub schema: SchemaCfg, - /// field name → regex source string (one entry per scoreable subject) - pub scores: HashMap, - pub insert: InsertCfg, /// When set to "thptqg2016", enables per-file format auto-detection. /// Each file's header row is inspected at runtime to choose the right /// column layout (separate-scores / mapped / default-positional). @@ -68,18 +70,6 @@ pub struct HeaderCfg { pub tokens: Vec, } -#[derive(Debug, Deserialize, Clone)] -pub struct SchemaCfg { - /// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX) - pub ddl: String, -} - -#[derive(Debug, Deserialize, Clone)] -pub struct InsertCfg { - /// Parameterised INSERT OR REPLACE SQL using :named_param style - pub sql: String, -} - // --------------------------------------------------------------------------- // Loader // --------------------------------------------------------------------------- @@ -119,16 +109,6 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)" "#; #[test] @@ -142,11 +122,20 @@ sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)" assert!(!cfg.validation.require_numeric_sbd); assert!(cfg.validation.require_nonempty_name); assert_eq!(cfg.header.tokens.len(), 3); - assert!(cfg.scores.contains_key("toan")); - assert!(cfg.scores.contains_key("ngu_van")); assert!(cfg.format_detection.is_none()); } + /// A config carrying leftover SQL sections must be rejected rather than + /// silently ignored — otherwise a stale [schema] block would look effective + /// while `crate::schema` was actually driving the build. + #[test] + fn config_rejects_leftover_sql_sections() { + let with_ddl = format!( + "{SAMPLE_TOML}\n[schema]\nddl = \"CREATE TABLE student (so_bao_danh TEXT);\"\n" + ); + assert!(toml::from_str::(&with_ddl).is_err()); + } + #[test] fn config_first_sheet_mode() { let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#); @@ -171,15 +160,6 @@ require_nonempty_sbd = true [header] tokens = ["SBD", "SOBAODANH", "STT"] - -[schema] -ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (?)" "#; let cfg: DatasetConfig = toml::from_str(toml_str).expect("parse failed"); assert_eq!(cfg.format_detection.as_deref(), Some("thptqg2016")); diff --git a/2016/tools/xlsxread/src/format_detect_2016.rs b/2016/tools/xlsxread/src/format_detect_2016.rs index 36c2bfd..768a88c 100644 --- a/2016/tools/xlsxread/src/format_detect_2016.rs +++ b/2016/tools/xlsxread/src/format_detect_2016.rs @@ -337,20 +337,13 @@ pub fn process_row_2016( #[cfg(test)] mod tests { use super::*; - use std::collections::HashMap; fn s(v: &str) -> Data { Data::String(v.to_string()) } fn make_patterns() -> CompiledPatterns { - let mut map = HashMap::new(); - map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into()); - map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_anh".into(), r"Tiếng Anh:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_duc".into(), r"Tiếng Đức:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_nhat".into(), r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)".into()); - CompiledPatterns::new(&map).unwrap() + CompiledPatterns::new().unwrap() } // --- detect_format: branch 1 — separate-scores --- diff --git a/2016/tools/xlsxread/src/lib.rs b/2016/tools/xlsxread/src/lib.rs index b61560c..ce7f5e4 100644 --- a/2016/tools/xlsxread/src/lib.rs +++ b/2016/tools/xlsxread/src/lib.rs @@ -6,5 +6,6 @@ pub mod config; pub mod error; pub mod format_detect_2016; pub mod reader; +pub mod schema; pub mod transform; pub mod writer; diff --git a/2016/tools/xlsxread/src/main.rs b/2016/tools/xlsxread/src/main.rs index 0736ed7..fd8b342 100644 --- a/2016/tools/xlsxread/src/main.rs +++ b/2016/tools/xlsxread/src/main.rs @@ -25,9 +25,7 @@ use xlsxread::format_detect_2016::{ }; use xlsxread::reader::{is_all_blank, process_file}; use xlsxread::transform::{validate_row, CompiledPatterns, SkipReason}; -use xlsxread::writer::{ - finish_db, insert_row, insert_row_2016, open_db, SCORE_FIELDS, SCORE_FIELDS_2016, -}; +use xlsxread::writer::{finish_db, insert_row, open_db}; fn main() -> Result<()> { let cli = Cli::parse(); @@ -79,7 +77,7 @@ fn run_build_standard( output_path: &Path, ) -> Result<()> { let patterns = - CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?; + CompiledPatterns::new().with_context(|| "Failed to compile score regexes")?; let mut files: Vec = std::fs::read_dir(input_dir) .with_context(|| format!("Cannot read input dir: {}", input_dir.display()))? @@ -109,7 +107,7 @@ fn run_build_standard( files.len() ); - let conn = open_db(output_path, cfg) + let conn = open_db(output_path) .with_context(|| format!("Failed to open DB: {}", output_path.display()))?; let mut total_source_rows: u64 = 0; @@ -159,7 +157,7 @@ fn run_build_standard( let parsed = xlsxread::transform::transform_row(raw, cfg, &patterns); - match insert_row(&conn, &cfg.insert.sql, &parsed, SCORE_FIELDS) { + match insert_row(&conn, &parsed) { Ok(()) => file_rows += 1, Err(e) => { file_errors += 1; @@ -216,7 +214,7 @@ fn run_build_2016( output_path: &Path, ) -> Result<()> { let patterns = - CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?; + CompiledPatterns::new().with_context(|| "Failed to compile score regexes")?; let mut files: Vec = std::fs::read_dir(input_dir) .with_context(|| format!("Cannot read input dir: {}", input_dir.display()))? @@ -246,7 +244,7 @@ fn run_build_2016( files.len() ); - let conn = open_db(output_path, cfg) + let conn = open_db(output_path) .with_context(|| format!("Failed to open DB: {}", output_path.display()))?; let mut total_source_rows: u64 = 0; @@ -361,7 +359,7 @@ fn process_file_2016( // Row was empty/invalid — skipped (mirrors JS `if (!record) continue`) } Some(parsed) => { - match insert_row_2016(conn, &cfg.insert.sql, &parsed, SCORE_FIELDS_2016) { + match insert_row(conn, &parsed) { Ok(()) => file_rows += 1, Err(e) => { *total_errors += 1; diff --git a/2016/tools/xlsxread/src/schema.rs b/2016/tools/xlsxread/src/schema.rs new file mode 100644 index 0000000..6eaf739 --- /dev/null +++ b/2016/tools/xlsxread/src/schema.rs @@ -0,0 +1,213 @@ +//! Canonical `student` table definition — the single source of truth for the +//! SQL shape of every dataset. +//! +//! Every dataset (2016, 2017, 2017-old, 2017-old2) is written into this same +//! 22-column table. Columns a dataset has no data for bind NULL, which costs +//! ~1 byte per row in SQLite's record header. +//! +//! Column provenance: +//! ten_cum_thi, gioi_tinh, tieng_duc, tieng_nhat → 2016 only +//! khtn, khxh, gdcd, tieng_nga → 2017 datasets only +//! everything else → both +//! +//! Before this module existed the DDL, the INSERT statement and the subject +//! regex table were duplicated across four TOML configs, which is how the 2016 +//! and 2017 schemas drifted apart in the first place. The configs now carry only +//! per-dataset parse rules. +// --------------------------------------------------------------------------- +// DDL +// --------------------------------------------------------------------------- + +/// Executed verbatim after the output database is (re)created. +/// +/// `idx_ten_cum_thi` is partial so it holds zero entries on the three 2017 +/// datasets — where the column is always NULL — while staying fully useful for +/// the 2016 cluster-grouping queries. +pub const DDL: &str = " +CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + ngay_sinh TEXT, + ten_cum_thi TEXT, + gioi_tinh TEXT, + toan REAL, + ngu_van REAL, + vat_ly REAL, + hoa_hoc REAL, + sinh_hoc REAL, + khtn REAL, + lich_su REAL, + dia_ly REAL, + gdcd REAL, + khxh REAL, + tieng_anh REAL, + tieng_phap REAL, + tieng_nga REAL, + tieng_duc REAL, + tieng_nhat REAL, + tieng_trung REAL +); +CREATE INDEX idx_ho_ten ON student(ho_ten); +CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); +CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL; +"; + +// --------------------------------------------------------------------------- +// Column order +// --------------------------------------------------------------------------- + +/// Identity columns, in INSERT parameter order. +pub const IDENTITY_FIELDS: &[&str] = &[ + "so_bao_danh", + "ho_ten", + "ho_ten_ascii", + "ngay_sinh", + "ten_cum_thi", + "gioi_tinh", +]; + +/// Subject columns, in INSERT parameter order. Bound as NULL when a row has no +/// score for that subject. +pub const SCORE_FIELDS: &[&str] = &[ + "toan", + "ngu_van", + "vat_ly", + "hoa_hoc", + "sinh_hoc", + "khtn", + "lich_su", + "dia_ly", + "gdcd", + "khxh", + "tieng_anh", + "tieng_phap", + "tieng_nga", + "tieng_duc", + "tieng_nhat", + "tieng_trung", +]; + +/// Total bound parameters per row. +pub const PARAM_COUNT: usize = IDENTITY_FIELDS.len() + SCORE_FIELDS.len(); + +// --------------------------------------------------------------------------- +// INSERT +// --------------------------------------------------------------------------- + +/// Positional INSERT matching IDENTITY_FIELDS followed by SCORE_FIELDS. +/// +/// `OR REPLACE` preserves the pre-existing behaviour where a repeated SBD +/// overwrites the earlier row rather than aborting the transaction. +pub const INSERT_SQL: &str = " +INSERT OR REPLACE INTO student + (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh, + toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, + lich_su, dia_ly, gdcd, khxh, + tieng_anh, tieng_phap, tieng_nga, tieng_duc, tieng_nhat, tieng_trung) +VALUES + (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) +"; + +// --------------------------------------------------------------------------- +// Subject score patterns +// --------------------------------------------------------------------------- + +/// Regex per subject, applied to the DIEM_THI cell text. +/// +/// Every pattern runs against every dataset. A subject absent from a given exam +/// year simply never matches and stays NULL — 2016 source files contain no +/// "KHTN:" or "Tiếng Nga:" tokens, and 2017 files contain no "Tiếng Đức:" or +/// "Tiếng Nhật:". The parity check asserts those counts are exactly zero rather +/// than assuming it. +/// +/// Order here is irrelevant (matching is by name); SCORE_FIELDS fixes the +/// INSERT order. +pub const SCORE_PATTERNS: &[(&str, &str)] = &[ + ("toan", r"Toán:\s*(\d+(?:\.\d+)?)"), + ("ngu_van", r"Ngữ văn:\s*(\d+(?:\.\d+)?)"), + ("vat_ly", r"Vật lí:\s*(\d+(?:\.\d+)?)"), + ("hoa_hoc", r"Hóa học:\s*(\d+(?:\.\d+)?)"), + ("sinh_hoc", r"Sinh học:\s*(\d+(?:\.\d+)?)"), + ("khtn", r"KHTN:\s*(\d+(?:\.\d+)?)"), + ("lich_su", r"Lịch sử:\s*(\d+(?:\.\d+)?)"), + ("dia_ly", r"Địa lí:\s*(\d+(?:\.\d+)?)"), + ("gdcd", r"GDCD:\s*(\d+(?:\.\d+)?)"), + ("khxh", r"KHXH:\s*(\d+(?:\.\d+)?)"), + ("tieng_anh", r"Tiếng Anh:\s*(\d+(?:\.\d+)?)"), + ("tieng_phap", r"Tiếng Pháp:\s*(\d+(?:\.\d+)?)"), + ("tieng_nga", r"Tiếng Nga:\s*(\d+(?:\.\d+)?)"), + ("tieng_duc", r"Tiếng Đức:\s*(\d+(?:\.\d+)?)"), + ("tieng_nhat", r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)"), + ("tieng_trung", r"Tiếng Trung:\s*(\d+(?:\.\d+)?)"), +]; + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +#[cfg(test)] +mod tests { + use super::*; + + /// The INSERT placeholder count, the named column list and the field + /// constants must agree, or rows silently land in the wrong columns. + #[test] + fn insert_matches_field_order() { + assert_eq!(PARAM_COUNT, 22); + assert_eq!(INSERT_SQL.matches('?').count(), PARAM_COUNT); + + let named = INSERT_SQL + .split_once('(') + .and_then(|(_, rest)| rest.split_once(')')) + .map(|(cols, _)| cols) + .expect("INSERT must contain a column list"); + + let listed: Vec<&str> = named + .split(',') + .map(str::trim) + .filter(|s| !s.is_empty()) + .collect(); + + let expected: Vec<&str> = IDENTITY_FIELDS + .iter() + .chain(SCORE_FIELDS.iter()) + .copied() + .collect(); + + assert_eq!(listed, expected); + } + + /// Every subject column must have a pattern and vice versa. + #[test] + fn score_patterns_cover_score_fields() { + assert_eq!(SCORE_PATTERNS.len(), SCORE_FIELDS.len()); + for (field, _) in SCORE_PATTERNS { + assert!( + SCORE_FIELDS.contains(field), + "pattern {field} has no column" + ); + } + for field in SCORE_FIELDS { + assert!( + SCORE_PATTERNS.iter().any(|(f, _)| f == field), + "column {field} has no pattern" + ); + } + } + + /// Every column named in the DDL must be bound by the INSERT. + #[test] + fn ddl_columns_match_insert() { + for field in IDENTITY_FIELDS.iter().chain(SCORE_FIELDS.iter()) { + assert!(DDL.contains(field), "DDL missing column {field}"); + } + } + + #[test] + fn score_patterns_compile() { + for (field, src) in SCORE_PATTERNS { + regex::Regex::new(src).unwrap_or_else(|e| panic!("{field}: {e}")); + } + } +} diff --git a/2016/tools/xlsxread/src/transform.rs b/2016/tools/xlsxread/src/transform.rs index c688be4..1ea2d68 100644 --- a/2016/tools/xlsxread/src/transform.rs +++ b/2016/tools/xlsxread/src/transform.rs @@ -15,22 +15,25 @@ use crate::error::BuildError; // --------------------------------------------------------------------------- pub struct CompiledPatterns { - /// Ordered list so INSERT column order is deterministic + /// One compiled regex per subject column in `schema::SCORE_PATTERNS`. pub patterns: Vec<(String, Regex)>, } impl CompiledPatterns { - pub fn new(scores: &HashMap) -> Result { - let mut patterns = Vec::with_capacity(scores.len()); - for (field, src) in scores { + /// Compile the canonical subject patterns once at startup. + /// + /// All 16 patterns run against every dataset. A subject that did not exist + /// in a given exam year simply never matches and stays NULL — the parity + /// check asserts those counts are exactly zero rather than assuming it. + pub fn new() -> Result { + let mut patterns = Vec::with_capacity(crate::schema::SCORE_PATTERNS.len()); + for (field, src) in crate::schema::SCORE_PATTERNS { let re = Regex::new(src).map_err(|e| BuildError::Regex { - pattern: src.clone(), + pattern: (*src).to_string(), source: e, })?; - patterns.push((field.clone(), re)); + patterns.push(((*field).to_string(), re)); } - // Sort for deterministic order across HashMap iteration - patterns.sort_by(|a, b| a.0.cmp(&b.0)); Ok(Self { patterns }) } } @@ -314,11 +317,7 @@ mod tests { // --- Score parsing tests --- fn make_patterns() -> CompiledPatterns { - let mut map = HashMap::new(); - map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into()); - map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into()); - map.insert("vat_ly".into(), r"Vật lí:\s*(\d+(?:\.\d+)?)".into()); - CompiledPatterns::new(&map).unwrap() + CompiledPatterns::new().unwrap() } #[test] diff --git a/2016/tools/xlsxread/src/writer.rs b/2016/tools/xlsxread/src/writer.rs index 07a83f0..e8fc17a 100644 --- a/2016/tools/xlsxread/src/writer.rs +++ b/2016/tools/xlsxread/src/writer.rs @@ -3,27 +3,24 @@ /// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js. /// Stats output lines match the JS stdout exactly so existing CI log-greps still work. /// -/// thptqg2016 differences vs thptqg2017: -/// - INSERT includes ten_cum_thi and gioi_tinh after ngay_sinh -/// - Score columns are toan/ngu_van/vat_ly/hoa_hoc/sinh_hoc/lich_su/dia_ly/ -/// tieng_anh/tieng_phap/tieng_duc/tieng_nhat/tieng_trung -/// (no khtn, khxh, tieng_nga) +/// Every dataset writes the same canonical table (see `crate::schema`), so there +/// is exactly one insert path. Columns a dataset carries no data for bind NULL. use std::fs; use std::path::Path; use rusqlite::{params_from_iter, Connection, ToSql}; -use crate::config::DatasetConfig; use crate::error::BuildError; +use crate::schema; use crate::transform::ParsedRow; // --------------------------------------------------------------------------- // DB initialisation — mirrors build-lib.js createDb (delete + recreate) // --------------------------------------------------------------------------- -/// Open (or recreate) the output SQLite database, execute the DDL from config, +/// Open (or recreate) the output SQLite database, execute the canonical DDL, /// and return the open connection ready for inserts. -pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result { +pub fn open_db(db_path: &Path) -> Result { // Mirror Node behaviour: delete existing file before creating (build-lib.js:54) if db_path.exists() { fs::remove_file(db_path).map_err(|e| BuildError::Io { @@ -43,92 +40,22 @@ pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result Result<(), BuildError> { + let mut params: Vec> = Vec::with_capacity(schema::PARAM_COUNT); -/// thptqg2016 score columns (12 fields: tieng_duc/tieng_nhat present; no khtn/khxh/tieng_nga). -pub const SCORE_FIELDS_2016: &[&str] = &[ - "toan", - "ngu_van", - "vat_ly", - "hoa_hoc", - "sinh_hoc", - "lich_su", - "dia_ly", - "tieng_anh", - "tieng_phap", - "tieng_duc", - "tieng_nhat", - "tieng_trung", -]; - -/// Alias kept so thptqg2017 callers that import SCORE_FIELDS continue to compile. -pub const SCORE_FIELDS: &[&str] = SCORE_FIELDS_2017; - -// --------------------------------------------------------------------------- -// Insert a single parsed row (thptqg2017 schema — no ten_cum_thi / gioi_tinh) -// --------------------------------------------------------------------------- - -/// Bind all fields from `row` into the prepared statement and execute it. -/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, -pub fn insert_row( - conn: &Connection, - sql: &str, - row: &ParsedRow, - score_fields: &[&str], -) -> Result<(), BuildError> { - let mut params: Vec> = Vec::with_capacity(4 + score_fields.len()); - params.push(Box::new(row.so_bao_danh.clone())); - params.push(Box::new(row.ho_ten.clone())); - params.push(Box::new(row.ho_ten_ascii.clone())); - params.push(Box::new(row.ngay_sinh.clone())); - - for field in score_fields { - let val: Option = row.scores.get(*field).copied(); - params.push(Box::new(val)); - } - - conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?; - Ok(()) -} - -// --------------------------------------------------------------------------- -// Insert a single parsed row (thptqg2016 schema — includes ten_cum_thi + gioi_tinh) -// --------------------------------------------------------------------------- - -/// Bind all fields from `row` into the prepared statement for the thptqg2016 schema. -/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, -/// gioi_tinh, -pub fn insert_row_2016( - conn: &Connection, - sql: &str, - row: &ParsedRow, - score_fields: &[&str], -) -> Result<(), BuildError> { - let mut params: Vec> = Vec::with_capacity(6 + score_fields.len()); params.push(Box::new(row.so_bao_danh.clone())); params.push(Box::new(row.ho_ten.clone())); params.push(Box::new(row.ho_ten_ascii.clone())); @@ -136,12 +63,17 @@ pub fn insert_row_2016( params.push(Box::new(row.ten_cum_thi.clone())); params.push(Box::new(row.gioi_tinh.clone())); - for field in score_fields { + for field in schema::SCORE_FIELDS { let val: Option = row.scores.get(*field).copied(); params.push(Box::new(val)); } - conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?; + debug_assert_eq!(params.len(), schema::PARAM_COUNT); + + conn.execute( + schema::INSERT_SQL, + params_from_iter(params.iter().map(|p| p.as_ref())), + )?; Ok(()) } diff --git a/2016/tools/xlsxread/tests/golden.rs b/2016/tools/xlsxread/tests/golden.rs index 0038945..47d1309 100644 --- a/2016/tools/xlsxread/tests/golden.rs +++ b/2016/tools/xlsxread/tests/golden.rs @@ -581,8 +581,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { let cfg = xlsxread::config::load_config(&cfg_path) .unwrap_or_else(|e| panic!("load config {config_name}: {e}")); - let patterns = - xlsxread::transform::CompiledPatterns::new(&cfg.scores).expect("compile patterns"); + let patterns = xlsxread::transform::CompiledPatterns::new().expect("compile patterns"); // Collect files let mut files: Vec = std::fs::read_dir(input_dir) @@ -602,7 +601,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { .collect(); files.sort(); - let conn = xlsxread::writer::open_db(db_path, &cfg).expect("open db"); + let conn = xlsxread::writer::open_db(db_path).expect("open db"); conn.execute_batch("BEGIN").unwrap(); for file in &files { @@ -632,12 +631,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { return; } let row = xlsxread::transform::transform_row(raw, &cfg, &patterns); - let _ = xlsxread::writer::insert_row( - &conn, - &cfg.insert.sql, - &row, - xlsxread::writer::SCORE_FIELDS, - ); + let _ = xlsxread::writer::insert_row(&conn, &row); }) .expect("process file"); } diff --git a/2017/scripts/db-stats.js b/2017/scripts/db-stats.js new file mode 100644 index 0000000..8582821 --- /dev/null +++ b/2017/scripts/db-stats.js @@ -0,0 +1,79 @@ +#!/usr/bin/env node +/** + * Dump per-dataset statistics from built SQLite databases as JSON. + * + * Used twice: once against the pre-refactor databases to capture a baseline, + * and again after the schema unification. Comparing the two outputs is what + * proves no score data was lost or invented. + * + * Schema-agnostic on purpose — columns come from PRAGMA table_info, so the same + * script runs against the old 18/20-column tables and the new 22-column one. + * + * Usage: + * node db-stats.js