mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
feat: replace xlsx (SheetJS) build pipeline with Rust xlsxread CLI (#1)
* feat(xlsxread): stage 0 scaffold with pinned deps and clap CLI skeleton * feat(xlsxread): stage 1 reader — calamine sheet enumeration and header skip * feat(xlsxread): stage 2 transform — to_ascii, score regex, validation with 38 unit tests * feat(xlsxread): stage 3 writer — SQLite DDL, INSERT OR REPLACE, VACUUM, stats output * feat(xlsxread): stage 4 audit — distinct SBD scan vs DB count, mirrors audit-row-counts.js output * feat(xlsxread): stage 5 golden tests — in-process xlsx fixtures, 8 integration tests pass * chore(xlsxread): commit Cargo.lock for reproducible Rust builds * feat(build): wire root build:db scripts to xlsxread CLI Replace node scripts/build-database*.js invocations with the Rust xlsxread binary. Each build:db* script now calls `pnpm build:rust` (cargo build --release) before invoking the xlsxread build subcommand with the matching per-dataset config. Drop xlsx and better-sqlite3 from devDependencies — no Node script consumes them anymore. sql.js (runtime DB reader in the SPA) is unaffected and remains in dependencies. * ci: build xlsxread before running database build jobs Add dtolnay/rust-toolchain@stable and Swatinem/rust-cache@v2 (workspaces: tools/xlsxread) for warm incremental Rust builds. Replace the single `pnpm build:db:all` step with explicit xlsxread invocations so CI doesn't call pnpm build:rust redundantly three times. The binary is built once, then each of the three datasets is processed in sequence. * chore: remove deprecated xlsx-based build scripts Delete scripts/build-database.js, build-database-old.js, build-database-old2.js, build-lib.js, and audit-row-counts.js. Functionality replaced by the xlsxread Rust CLI configured via tools/xlsxread/configs/*.toml. History preserved in git; one-click revert available via the chore/migration-backup-260519 branch. * docs: update README build instructions for xlsxread pipeline Replace Node.js + xlsx references with Rust + xlsxread workflow. Update requirements (Node 24+, pnpm, Rust stable), quickstart, scripts table, and project layout tree to reflect the current state after the xlsx-based build scripts were removed. * chore(deps): drop xlsx and better-sqlite3 from package.json and lockfile Remove xlsx (SheetJS, vulnerable: GHSA-4r6h-8v6p-xvw6, GHSA-5pgg-2g8v-p4x9) and better-sqlite3 from devDependencies. Both were only used by the now-deleted Node build scripts. The Rust xlsxread CLI vendors SQLite via rusqlite-bundled; no Node-side SQLite dependency is needed. `pnpm audit` returns clean.
This commit is contained in:
1 parent
5b0cec3b27
commit
99465fda59
29 files changed
+3523
-805
No files matched your search
Vendored
+13
-1
@@ -20,6 +20,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: tools/xlsxread
|
||||
|
||||
- uses: pnpm/action-setup@v4
|
||||
|
||||
- uses: actions/setup-node@v4
|
||||
@@ -27,11 +33,17 @@ jobs:
|
||||
node-version: '24'
|
||||
cache: 'pnpm'
|
||||
|
||||
- name: Build xlsxread binary
|
||||
run: cargo build --release --manifest-path tools/xlsxread/Cargo.toml
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
- name: Build all databases
|
||||
run: pnpm build:db:all
|
||||
run: |
|
||||
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data.toml --input data --output public/thptqg2017.db
|
||||
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old.toml --input data-old --output public-old/thptqg2017.db
|
||||
./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old2.toml --input data-old2 --output public-old2/thptqg2017.db
|
||||
|
||||
- name: Compress databases
|
||||
run: |
|
||||
|
||||
@@ -22,3 +22,6 @@ public-old2/thptqg2017.db.gz
|
||||
|
||||
### Build output ###
|
||||
dist/
|
||||
|
||||
### Rust build artefacts ###
|
||||
tools/xlsxread/target/
|
||||
+29
-19
@@ -24,32 +24,39 @@ Three deployments, one per dataset:
|
||||
|
||||
## Requirements
|
||||
|
||||
- Node.js 20+
|
||||
- npm
|
||||
- Node.js 24+
|
||||
- pnpm
|
||||
- Rust (stable) — installed via [rustup](https://rustup.rs/); required for `build:db*` scripts
|
||||
|
||||
## Quickstart
|
||||
|
||||
```bash
|
||||
npm install
|
||||
npm run build:db # parse data/ into public/thptqg2017.db (~2 min, 159 MB)
|
||||
pnpm install
|
||||
pnpm build:db # compile xlsxread + parse data/ → public/thptqg2017.db (~2 min, 159 MB)
|
||||
gzip -kf -9 public/thptqg2017.db
|
||||
npm run dev # http://localhost:5173
|
||||
pnpm dev # http://localhost:5173
|
||||
```
|
||||
|
||||
The database build pipeline uses the `xlsxread` Rust CLI located at
|
||||
`tools/xlsxread/`. It reads `.xls`/`.xlsx` source files, strips
|
||||
diacritics, parses score text, and writes a SQLite database — replacing
|
||||
the former Node.js + `xlsx` (SheetJS) pipeline. See
|
||||
`tools/xlsxread/README.md` for CLI invocation details and config schema.
|
||||
|
||||
## Scripts
|
||||
|
||||
| Command | Action |
|
||||
|---|---|
|
||||
| `npm run dev` | Vite dev server |
|
||||
| `npm run build` | Production build (main variant → `dist/`) |
|
||||
| `npm run build:old` / `build:old2` | Build variant sites to `dist/old/`, `dist/old2/` |
|
||||
| `npm run build:all` | All 3 web variants |
|
||||
| `npm run build:db` | Build main DB from `data/` |
|
||||
| `npm run build:db:old` / `build:db:old2` | Build old / old2 variant DBs |
|
||||
| `npm run build:db:all` | All 3 DBs |
|
||||
| `npm run lint` | ESLint |
|
||||
| `pnpm dev` | Vite dev server |
|
||||
| `pnpm build` | Production build (main variant → `dist/`) |
|
||||
| `pnpm build:old` / `build:old2` | Build variant sites to `dist/old/`, `dist/old2/` |
|
||||
| `pnpm build:all` | All 3 web variants |
|
||||
| `pnpm build:rust` | Compile `tools/xlsxread` release binary (run once; auto-called by `build:db*`) |
|
||||
| `pnpm build:db` | Build main DB from `data/` via xlsxread |
|
||||
| `pnpm build:db:old` / `build:db:old2` | Build old / old2 variant DBs via xlsxread |
|
||||
| `pnpm build:db:all` | All 3 DBs via xlsxread |
|
||||
| `pnpm lint` | ESLint |
|
||||
| `node scripts/crawl-baotintuc.js` | Re-download all 63 province files from baotintuc.vn |
|
||||
| `node scripts/audit-row-counts.js` | Verify source row count matches DB row count |
|
||||
| `node scripts/check-duplicates.js` | MD5 + row-content duplicate audit |
|
||||
| `node scripts/diff-datasets.js` | Compare `public/` vs `backup/` DB (when backup present) |
|
||||
|
||||
@@ -64,12 +71,7 @@ npm run dev # http://localhost:5173
|
||||
├── public-old/ # old variant assets
|
||||
├── public-old2/ # old2 variant assets
|
||||
├── scripts/
|
||||
│ ├── build-lib.js # shared schema + helpers
|
||||
│ ├── build-database.js # parser for data/
|
||||
│ ├── build-database-old.js # parser for data-old/
|
||||
│ ├── build-database-old2.js # parser for data-old2/
|
||||
│ ├── crawl-baotintuc.js # downloader
|
||||
│ ├── audit-row-counts.js # parse-loss audit
|
||||
│ ├── check-duplicates.js # md5 dup detector
|
||||
│ └── diff-datasets.js # DB-to-DB comparator
|
||||
├── src/
|
||||
@@ -78,6 +80,14 @@ npm run dev # http://localhost:5173
|
||||
│ ├── components/{search-form, score-table, student-detail, custom-query}.jsx
|
||||
│ ├── hooks/use-sqlite.js
|
||||
│ └── lib/admission-blocks.js # 49 admission-block definitions + score-tier helper
|
||||
├── tools/
|
||||
│ └── xlsxread/ # Rust CLI — reads .xls/.xlsx, writes SQLite
|
||||
│ ├── configs/
|
||||
│ │ ├── thptqg2017-data.toml # config for data/ (63 .xls, all sheets)
|
||||
│ │ ├── thptqg2017-data-old.toml # config for data-old/ (63 .xlsx, sheet 0)
|
||||
│ │ └── thptqg2017-data-old2.toml # config for data-old2/ (54 .xlsx, all sheets)
|
||||
│ ├── src/ # Rust source
|
||||
│ └── README.md # CLI reference + config schema
|
||||
├── docs/ # see docs/README.md
|
||||
├── index.html
|
||||
├── vite.config.js
|
||||
|
||||
+6
-7
@@ -6,10 +6,11 @@
|
||||
"type": "module",
|
||||
"description": "Tra cứu điểm thi THPT QG 2017",
|
||||
"scripts": {
|
||||
"build:db": "node scripts/build-database.js",
|
||||
"build:db:old": "node scripts/build-database-old.js",
|
||||
"build:db:old2": "node scripts/build-database-old2.js",
|
||||
"build:db:all": "pnpm build:db && pnpm build:db:old && pnpm build:db:old2",
|
||||
"build:rust": "cargo build --release --manifest-path tools/xlsxread/Cargo.toml",
|
||||
"build:db": "pnpm build:rust && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data.toml --input data --output public/thptqg2017.db",
|
||||
"build:db:old": "pnpm build:rust && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old.toml --input data-old --output public-old/thptqg2017.db",
|
||||
"build:db:old2": "pnpm build:rust && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old2.toml --input data-old2 --output public-old2/thptqg2017.db",
|
||||
"build:db:all": "pnpm build:rust && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data.toml --input data --output public/thptqg2017.db && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old.toml --input data-old --output public-old/thptqg2017.db && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2017-data-old2.toml --input data-old2 --output public-old2/thptqg2017.db",
|
||||
"dev": "vite",
|
||||
"build": "vite build",
|
||||
"build:old": "node -e \"process.env.VARIANT='old';require('child_process').spawnSync('npx',['vite','build'],{stdio:'inherit',shell:true,env:process.env})\"",
|
||||
@@ -33,12 +34,10 @@
|
||||
"@types/react": "^19.2.14",
|
||||
"@types/react-dom": "^19.2.3",
|
||||
"@vitejs/plugin-react": "^6.0.1",
|
||||
"better-sqlite3": "^12.8.0",
|
||||
"eslint": "^9.39.4",
|
||||
"eslint-plugin-react-hooks": "^7.0.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.2",
|
||||
"globals": "^17.4.0",
|
||||
"vite": "^8.0.4",
|
||||
"xlsx": "^0.18.5"
|
||||
"vite": "^8.0.4"
|
||||
}
|
||||
}
|
||||
Generated
-339
@@ -30,9 +30,6 @@ importers:
|
||||
'@vitejs/plugin-react':
|
||||
specifier: ^6.0.1
|
||||
version: 6.0.1(vite@8.0.12)
|
||||
better-sqlite3:
|
||||
specifier: ^12.8.0
|
||||
version: 12.9.0
|
||||
eslint:
|
||||
specifier: ^9.39.4
|
||||
version: 9.39.4
|
||||
@@ -48,9 +45,6 @@ importers:
|
||||
vite:
|
||||
specifier: ^8.0.4
|
||||
version: 8.0.12
|
||||
xlsx:
|
||||
specifier: ^0.18.5
|
||||
version: 0.18.5
|
||||
|
||||
packages:
|
||||
|
||||
@@ -354,10 +348,6 @@ packages:
|
||||
engines: {node: '>=0.4.0'}
|
||||
hasBin: true
|
||||
|
||||
adler-32@1.3.1:
|
||||
resolution: {integrity: sha512-ynZ4w/nUUv5rrsR8UUGoe1VC9hZj6V5hU9Qw1HlMDJGEJw5S7TfTErWTjMys6M7vr0YWcPqs3qAr4ss0nDfP+A==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
ajv@6.15.0:
|
||||
resolution: {integrity: sha512-fgFx7Hfoq60ytK2c7DhnF8jIvzYgOMxfugjLOSMHjLIPgenqa7S7oaagATUq99mV6IYvN2tRmC0wnTYX6iPbMw==}
|
||||
|
||||
@@ -371,24 +361,11 @@ packages:
|
||||
balanced-match@1.0.2:
|
||||
resolution: {integrity: sha512-3oSeUO0TMV67hN1AmbXsK4yaqU7tjiHlbxRDZOpH0KW9+CeX4bRAaX0Anxt0tx2MrpRpWwQaPwIlISEJhYU5Pw==}
|
||||
|
||||
base64-js@1.5.1:
|
||||
resolution: {integrity: sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==}
|
||||
|
||||
baseline-browser-mapping@2.10.29:
|
||||
resolution: {integrity: sha512-Asa2krT+XTPZINCS+2QcyS8WTkObE77RwkydwF7h6DmnKqbvlalz93m/dnphUyCa6SWSP51VgtEUf2FN+gelFQ==}
|
||||
engines: {node: '>=6.0.0'}
|
||||
hasBin: true
|
||||
|
||||
better-sqlite3@12.9.0:
|
||||
resolution: {integrity: sha512-wqUv4Gm3toFpHDQmaKD4QhZm3g1DjUBI0yzS4UBl6lElUmXFYdTQmmEDpAFa5o8FiFiymURypEnfVHzILKaxqQ==}
|
||||
engines: {node: 20.x || 22.x || 23.x || 24.x || 25.x}
|
||||
|
||||
bindings@1.5.0:
|
||||
resolution: {integrity: sha512-p2q/t/mhvuOj/UeLlV6566GD/guowlr0hHxClI0W9m7MWYkL1F0hLo+0Aexs9HSPCtR1SXQ0TD3MMKrXZajbiQ==}
|
||||
|
||||
bl@4.1.0:
|
||||
resolution: {integrity: sha512-1W07cM9gS6DcLperZfFSj+bWLtaPGSOHWhPiGzXmvVJbRLdG82sH/Kn8EtW1VqWVA54AKf2h5k5BbnIbwF3h6w==}
|
||||
|
||||
brace-expansion@1.1.14:
|
||||
resolution: {integrity: sha512-MWPGfDxnyzKU7rNOW9SP/c50vi3xrmrua/+6hfPbCS2ABNWfx24vPidzvC7krjU/RTo235sV776ymlsMtGKj8g==}
|
||||
|
||||
@@ -397,9 +374,6 @@ packages:
|
||||
engines: {node: ^6 || ^7 || ^8 || ^9 || ^10 || ^11 || ^12 || >=13.7}
|
||||
hasBin: true
|
||||
|
||||
buffer@5.7.1:
|
||||
resolution: {integrity: sha512-EHcyIPBQ4BSGlvjB16k5KgAJ27CIsHY/2JBmCRReo48y9rQ3MaUzWX3KVlBa4U7MyX02HdVj0K7C3WaB3ju7FQ==}
|
||||
|
||||
callsites@3.1.0:
|
||||
resolution: {integrity: sha512-P8BjAsXvZS+VIDUI11hHCQEv74YT67YUi5JJFNWIqL235sBmjX4+qx9Muvls5ivyNENctx46xQLQ3aTuE7ssaQ==}
|
||||
engines: {node: '>=6'}
|
||||
@@ -407,21 +381,10 @@ packages:
|
||||
caniuse-lite@1.0.30001792:
|
||||
resolution: {integrity: sha512-hVLMUZFgR4JJ6ACt1uEESvQN1/dBVqPAKY0hgrV70eN3391K6juAfTjKZLKvOMsx8PxA7gsY1/tLMMTcfFLLpw==}
|
||||
|
||||
cfb@1.2.2:
|
||||
resolution: {integrity: sha512-KfdUZsSOw19/ObEWasvBP/Ac4reZvAGauZhs6S/gqNhXhI7cKwvlH7ulj+dOEYnca4bm4SGo8C1bTAQvnTjgQA==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
chalk@4.1.2:
|
||||
resolution: {integrity: sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==}
|
||||
engines: {node: '>=10'}
|
||||
|
||||
chownr@1.1.4:
|
||||
resolution: {integrity: sha512-jJ0bqzaylmJtVnNgzTeSOs8DPavpbYgEr/b0YL8/2GO3xJEhInFmhKMUnEJQjZumK7KXGFhUy89PrsJWlakBVg==}
|
||||
|
||||
codepage@1.15.0:
|
||||
resolution: {integrity: sha512-3g6NUTPd/YtuuGrhMnOMRjFc+LJw/bnMp3+0r/Wcz3IXUuCosKRJvMphm5+Q+bvTVGcJJuRvVLuYba+WojaFaA==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
color-convert@2.0.1:
|
||||
resolution: {integrity: sha512-RRECPsj7iu/xb5oKYcsFHSppFNnsj/52OVTRKb4zP5onXwVF3zVmmToNcOfGC+CRDpfK/U584fMg38ZHCaElKQ==}
|
||||
engines: {node: '>=7.0.0'}
|
||||
@@ -435,11 +398,6 @@ packages:
|
||||
convert-source-map@2.0.0:
|
||||
resolution: {integrity: sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg==}
|
||||
|
||||
crc-32@1.2.2:
|
||||
resolution: {integrity: sha512-ROmzCKrTnOwybPcJApAA6WBWij23HVfGVNKqqrZpuyZOHqK2CwHSvpGuyt/UNNvaIjEd8X5IFGp4Mh+Ie1IHJQ==}
|
||||
engines: {node: '>=0.8'}
|
||||
hasBin: true
|
||||
|
||||
cross-spawn@7.0.6:
|
||||
resolution: {integrity: sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==}
|
||||
engines: {node: '>= 8'}
|
||||
@@ -456,14 +414,6 @@ packages:
|
||||
supports-color:
|
||||
optional: true
|
||||
|
||||
decompress-response@6.0.0:
|
||||
resolution: {integrity: sha512-aW35yZM6Bb/4oJlZncMH2LCoZtJXTRxES17vE3hoRiowU2kWHaJKFkSBDnDR+cm9J+9QhXmREyIfv0pji9ejCQ==}
|
||||
engines: {node: '>=10'}
|
||||
|
||||
deep-extend@0.6.0:
|
||||
resolution: {integrity: sha512-LOHxIOaPYdHlJRtCQfDIVZtfw/ufM8+rVj649RIHzcm/vGwQRXFt6OPqIFWsm2XEMrNIEtWR64sY1LEKD2vAOA==}
|
||||
engines: {node: '>=4.0.0'}
|
||||
|
||||
deep-is@0.1.4:
|
||||
resolution: {integrity: sha512-oIPzksmTg4/MriiaYGO+okXDT7ztn/w3Eptv/+gSIdMdKsJo0u4CfYNFJPy+4SKMuCqGw2wxnA+URMg3t8a/bQ==}
|
||||
|
||||
@@ -474,9 +424,6 @@ packages:
|
||||
electron-to-chromium@1.5.353:
|
||||
resolution: {integrity: sha512-kOrWphBi8TOZyiJZqsgqIle0lw+tzmnQK83pV9dZUd01Nm2POECSyFQMAuarzZdYqQW7FH9RaYOuaRo3h+bQ3w==}
|
||||
|
||||
end-of-stream@1.4.5:
|
||||
resolution: {integrity: sha512-ooEGc6HP26xXq/N+GCGOT0JKCLDGrq2bQUZrQ7gyrJiZANJ/8YDTxTpQBXGMn+WbIQXNVpyWymm7KYVICQnyOg==}
|
||||
|
||||
escalade@3.2.0:
|
||||
resolution: {integrity: sha512-WUj2qlxaQtO4g6Pq5c29GTcWGDyd8itL8zTlipgECz3JesAiiOKotd8JU6otB3PACgG6xkJUyVhboMS+bje/jA==}
|
||||
engines: {node: '>=6'}
|
||||
@@ -538,10 +485,6 @@ packages:
|
||||
resolution: {integrity: sha512-kVscqXk4OCp68SZ0dkgEKVi6/8ij300KBWTJq32P/dYeWTSwK41WyTxalN1eRmA5Z9UU/LX9D7FWSmV9SAYx6g==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
|
||||
expand-template@2.0.3:
|
||||
resolution: {integrity: sha512-XYfuKMvj4O35f/pOXLObndIRvyQ+/+6AhODh+OKWj9S9498pHHn/IMszH+gt0fBCRWMNfk1ZSp5x3AifmnI2vg==}
|
||||
engines: {node: '>=6'}
|
||||
|
||||
fast-deep-equal@3.1.3:
|
||||
resolution: {integrity: sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==}
|
||||
|
||||
@@ -564,9 +507,6 @@ packages:
|
||||
resolution: {integrity: sha512-XXTUwCvisa5oacNGRP9SfNtYBNAMi+RPwBFmblZEF7N7swHYQS6/Zfk7SRwx4D5j3CH211YNRco1DEMNVfZCnQ==}
|
||||
engines: {node: '>=16.0.0'}
|
||||
|
||||
file-uri-to-path@1.0.0:
|
||||
resolution: {integrity: sha512-0Zt+s3L7Vf1biwWZ29aARiVYLx7iMGnEUl9x33fbB/j3jR81u/O2LbqK+Bm1CDSNDKVtJ/YjwY7TUd5SkeLQLw==}
|
||||
|
||||
find-up@5.0.0:
|
||||
resolution: {integrity: sha512-78/PXT1wlLLDgTzDs7sjq9hzz0vXD+zn+7wypEe4fXQxCmdmqfGsEPQxmiCSQI3ajFV91bVSsvNtrJRiW6nGng==}
|
||||
engines: {node: '>=10'}
|
||||
@@ -578,13 +518,6 @@ packages:
|
||||
flatted@3.4.2:
|
||||
resolution: {integrity: sha512-PjDse7RzhcPkIJwy5t7KPWQSZ9cAbzQXcafsetQoD7sOJRQlGikNbx7yZp2OotDnJyrDcbyRq3Ttb18iYOqkxA==}
|
||||
|
||||
frac@1.1.2:
|
||||
resolution: {integrity: sha512-w/XBfkibaTl3YDqASwfDUqkna4Z2p9cFSr1aHDt0WoMTECnRfBOv2WArlZILlqgWlmdIlALXGpM2AOhEk5W3IA==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
fs-constants@1.0.0:
|
||||
resolution: {integrity: sha512-y6OAwoSIf7FyjMIv94u+b5rdheZEjzR63GTyZJm5qh4Bi+2YgwLCcI/fPFZkL5PSixOt6ZNKm+w+Hfp/Bciwow==}
|
||||
|
||||
fsevents@2.3.3:
|
||||
resolution: {integrity: sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==}
|
||||
engines: {node: ^8.16.0 || ^10.6.0 || >=11.0.0}
|
||||
@@ -594,9 +527,6 @@ packages:
|
||||
resolution: {integrity: sha512-3hN7NaskYvMDLQY55gnW3NQ+mesEAepTqlg+VEbj7zzqEMBVNhzcGYYeqFo/TlYz6eQiFcp1HcsCZO+nGgS8zg==}
|
||||
engines: {node: '>=6.9.0'}
|
||||
|
||||
github-from-package@0.0.0:
|
||||
resolution: {integrity: sha512-SyHy3T1v2NUXn29OsWdxmK6RwHD+vkj3v8en8AOBZ1wBQ/hCAQ5bAQTD02kW4W9tUp/3Qh6J8r9EvntiyCmOOw==}
|
||||
|
||||
glob-parent@6.0.2:
|
||||
resolution: {integrity: sha512-XxwI8EOhVQgWp6iDL+3b0r86f4d6AX6zSU55HfB4ydCEuXLXc5FcYeOu+nnGftS4TEju/11rt4KJPTMgbfmv4A==}
|
||||
engines: {node: '>=10.13.0'}
|
||||
@@ -619,9 +549,6 @@ packages:
|
||||
hermes-parser@0.25.1:
|
||||
resolution: {integrity: sha512-6pEjquH3rqaI6cYAXYPcz9MS4rY6R4ngRgrgfDshRptUZIc3lw0MCIJIGDj9++mfySOuPTHB4nrSW99BCvOPIA==}
|
||||
|
||||
ieee754@1.2.1:
|
||||
resolution: {integrity: sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA==}
|
||||
|
||||
ignore@5.3.2:
|
||||
resolution: {integrity: sha512-hsBTNUqQTDwkWtcdYI2i06Y/nUBEsNEDJKjWdigLvegy8kDuJAS8uRlpkkcQpyEXL0Z/pjDy5HBmMjRCJ2gq+g==}
|
||||
engines: {node: '>= 4'}
|
||||
@@ -634,12 +561,6 @@ packages:
|
||||
resolution: {integrity: sha512-JmXMZ6wuvDmLiHEml9ykzqO6lwFbof0GG4IkcGaENdCRDDmMVnny7s5HsIgHCbaq0w2MyPhDqkhTUgS2LU2PHA==}
|
||||
engines: {node: '>=0.8.19'}
|
||||
|
||||
inherits@2.0.4:
|
||||
resolution: {integrity: sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ==}
|
||||
|
||||
ini@1.3.8:
|
||||
resolution: {integrity: sha512-JV/yugV2uzW5iMRSiZAyDtQd+nxtUnjeLt0acNdw98kKLrvuRVyB80tsREOE7yvGVgalhZ6RNXCmEHkUKBKxew==}
|
||||
|
||||
is-extglob@2.1.1:
|
||||
resolution: {integrity: sha512-SbKbANkN603Vi4jEZv49LeVJMn4yGwsbzZworEoyEiutsN3nJYdbO36zfhGJ6QEDpOZIFkDtnq5JRxmvl3jsoQ==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
@@ -768,19 +689,9 @@ packages:
|
||||
lru-cache@5.1.1:
|
||||
resolution: {integrity: sha512-KpNARQA3Iwv+jTA0utUVVbrh+Jlrr1Fv0e56GGzAFOXN7dk/FviaDW8LHmK52DlcH4WP2n6gI8vN1aesBFgo9w==}
|
||||
|
||||
mimic-response@3.1.0:
|
||||
resolution: {integrity: sha512-z0yWI+4FDrrweS8Zmt4Ej5HdJmky15+L2e6Wgn3+iK5fWzb6T3fhNFq2+MeTRb064c6Wr4N/wv0DzQTjNzHNGQ==}
|
||||
engines: {node: '>=10'}
|
||||
|
||||
minimatch@3.1.5:
|
||||
resolution: {integrity: sha512-VgjWUsnnT6n+NUk6eZq77zeFdpW2LWDzP6zFGrCbHXiYNul5Dzqk2HHQ5uFH2DNW5Xbp8+jVzaeNt94ssEEl4w==}
|
||||
|
||||
minimist@1.2.8:
|
||||
resolution: {integrity: sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA==}
|
||||
|
||||
mkdirp-classic@0.5.3:
|
||||
resolution: {integrity: sha512-gKLcREMhtuZRwRAfqP3RFW+TK4JqApVBtOIftVgjuABpAtpxhPGaDcfvbhNvD0B8iD1oUr/txX35NjcaY6Ns/A==}
|
||||
|
||||
ms@2.1.3:
|
||||
resolution: {integrity: sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==}
|
||||
|
||||
@@ -789,22 +700,12 @@ packages:
|
||||
engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1}
|
||||
hasBin: true
|
||||
|
||||
napi-build-utils@2.0.0:
|
||||
resolution: {integrity: sha512-GEbrYkbfF7MoNaoh2iGG84Mnf/WZfB0GdGEsM8wz7Expx/LlWf5U8t9nvJKXSp3qr5IsEbK04cBGhol/KwOsWA==}
|
||||
|
||||
natural-compare@1.4.0:
|
||||
resolution: {integrity: sha512-OWND8ei3VtNC9h7V60qff3SVobHr996CTwgxubgyQYEpg290h9J0buyECNNJexkFm5sOajh5G116RYA1c8ZMSw==}
|
||||
|
||||
node-abi@3.92.0:
|
||||
resolution: {integrity: sha512-KdHvFWZjEKDf0cakgFjebl371GPsISX2oZHcuyKqM7DtogIsHrqKeLTo8wBHxaXRAQlY2PsPlZmfo+9ZCxEREQ==}
|
||||
engines: {node: '>=10'}
|
||||
|
||||
node-releases@2.0.38:
|
||||
resolution: {integrity: sha512-3qT/88Y3FbH/Kx4szpQQ4HzUbVrHPKTLVpVocKiLfoYvw9XSGOX2FmD2d6DrXbVYyAQTF2HeF6My8jmzx7/CRw==}
|
||||
|
||||
once@1.4.0:
|
||||
resolution: {integrity: sha512-lNaJgI+2Q5URQBkccEKHTQOPaXdUxnZZElQTZY0MFUAuaEqe1E+Nyvgdz/aIyNi6Z9MzO5dv1H8n58/GELp3+w==}
|
||||
|
||||
optionator@0.9.4:
|
||||
resolution: {integrity: sha512-6IpQ7mKUxRcZNLIObR0hz7lxsapSSIYNZJwXPGeF0mTVqGKFIXj1DQcMoT22S3ROcLyY/rz0PWaWZ9ayWmad9g==}
|
||||
engines: {node: '>= 0.8.0'}
|
||||
@@ -840,27 +741,14 @@ packages:
|
||||
resolution: {integrity: sha512-SoSL4+OSEtR99LHFZQiJLkT59C5B1amGO1NzTwj7TT1qCUgUO6hxOvzkOYxD+vMrXBM3XJIKzokoERdqQq/Zmg==}
|
||||
engines: {node: ^10 || ^12 || >=14}
|
||||
|
||||
prebuild-install@7.1.3:
|
||||
resolution: {integrity: sha512-8Mf2cbV7x1cXPUILADGI3wuhfqWvtiLA1iclTDbFRZkgRQS0NqsPZphna9V+HyTEadheuPmjaJMsbzKQFOzLug==}
|
||||
engines: {node: '>=10'}
|
||||
deprecated: No longer maintained. Please contact the author of the relevant native addon; alternatives are available.
|
||||
hasBin: true
|
||||
|
||||
prelude-ls@1.2.1:
|
||||
resolution: {integrity: sha512-vkcDPrRZo1QZLbn5RLGPpg/WmIQ65qoWWhcGKf/b5eplkkarX0m9z8ppCat4mlOqUsWpyNuYgO3VRyrYHSzX5g==}
|
||||
engines: {node: '>= 0.8.0'}
|
||||
|
||||
pump@3.0.4:
|
||||
resolution: {integrity: sha512-VS7sjc6KR7e1ukRFhQSY5LM2uBWAUPiOPa/A3mkKmiMwSmRFUITt0xuj+/lesgnCv+dPIEYlkzrcyXgquIHMcA==}
|
||||
|
||||
punycode@2.3.1:
|
||||
resolution: {integrity: sha512-vYt7UD1U9Wg6138shLtLOvdAu+8DsC/ilFtEVHcH+wydcSpNE20AfSOduf6MkRFahL5FY7X1oU7nKVZFtfq8Fg==}
|
||||
engines: {node: '>=6'}
|
||||
|
||||
rc@1.2.8:
|
||||
resolution: {integrity: sha512-y3bGgqKj3QBdxLbLkomlohkvsA8gdAiUQlSBJnBhfn+BPxg4bc62d8TcBW15wavDfgexCgccckhcZvywyQYPOw==}
|
||||
hasBin: true
|
||||
|
||||
react-dom@19.2.6:
|
||||
resolution: {integrity: sha512-0prMI+hvBbPjsWnxDLxlCGyM8PN6UuWjEUCYmZhO67xIV9Xasa/r/vDnq+Xyq4Lo27g8QSbO5YzARu0D1Sps3g==}
|
||||
peerDependencies:
|
||||
@@ -870,10 +758,6 @@ packages:
|
||||
resolution: {integrity: sha512-sfWGGfavi0xr8Pg0sVsyHMAOziVYKgPLNrS7ig+ivMNb3wbCBw3KxtflsGBAwD3gYQlE/AEZsTLgToRrSCjb0Q==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
|
||||
readable-stream@3.6.2:
|
||||
resolution: {integrity: sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==}
|
||||
engines: {node: '>= 6'}
|
||||
|
||||
resolve-from@4.0.0:
|
||||
resolution: {integrity: sha512-pb/MYmXstAkysRFx8piNI1tGFNQIFA3vkE3Gq4EuA1dF6gHp/+vgZqsCGJapvy8N3Q+4o7FwvquPJcnZ7RYy4g==}
|
||||
engines: {node: '>=4'}
|
||||
@@ -883,9 +767,6 @@ packages:
|
||||
engines: {node: ^20.19.0 || >=22.12.0}
|
||||
hasBin: true
|
||||
|
||||
safe-buffer@5.2.1:
|
||||
resolution: {integrity: sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==}
|
||||
|
||||
scheduler@0.27.0:
|
||||
resolution: {integrity: sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q==}
|
||||
|
||||
@@ -893,11 +774,6 @@ packages:
|
||||
resolution: {integrity: sha512-BR7VvDCVHO+q2xBEWskxS6DJE1qRnb7DxzUrogb71CWoSficBxYsiAGd+Kl0mmq/MprG9yArRkyrQxTO6XjMzA==}
|
||||
hasBin: true
|
||||
|
||||
semver@7.8.0:
|
||||
resolution: {integrity: sha512-AcM7dV/5ul4EekoQ29Agm5vri8JNqRyj39o0qpX6vDF2GZrtutZl5RwgD1XnZjiTAfncsJhMI48QQH3sN87YNA==}
|
||||
engines: {node: '>=10'}
|
||||
hasBin: true
|
||||
|
||||
shebang-command@2.0.0:
|
||||
resolution: {integrity: sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==}
|
||||
engines: {node: '>=8'}
|
||||
@@ -906,12 +782,6 @@ packages:
|
||||
resolution: {integrity: sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==}
|
||||
engines: {node: '>=8'}
|
||||
|
||||
simple-concat@1.0.1:
|
||||
resolution: {integrity: sha512-cSFtAPtRhljv69IK0hTVZQ+OfE9nePi/rtJmw5UjHeVyVroEqJXP1sFztKUy1qU+xvz3u/sfYJLa947b7nAN2Q==}
|
||||
|
||||
simple-get@4.0.1:
|
||||
resolution: {integrity: sha512-brv7p5WgH0jmQJr1ZDDfKDOSeWWg+OVypG99A/5vYGPqJ6pxiaHLy8nxtFjBA7oMa01ebA9gfh1uMCFqOuXxvA==}
|
||||
|
||||
source-map-js@1.2.1:
|
||||
resolution: {integrity: sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
@@ -919,17 +789,6 @@ packages:
|
||||
sql.js@1.14.1:
|
||||
resolution: {integrity: sha512-gcj8zBWU5cFsi9WUP+4bFNXAyF1iRpA3LLyS/DP5xlrNzGmPIizUeBggKa8DbDwdqaKwUcTEnChtd2grWo/x/A==}
|
||||
|
||||
ssf@0.11.2:
|
||||
resolution: {integrity: sha512-+idbmIXoYET47hH+d7dfm2epdOMUDjqcB4648sTZ+t2JwoyBFL/insLfB/racrDmsKB3diwsDA696pZMieAC5g==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
string_decoder@1.3.0:
|
||||
resolution: {integrity: sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==}
|
||||
|
||||
strip-json-comments@2.0.1:
|
||||
resolution: {integrity: sha512-4gB8na07fecVVkOI6Rs4e7T6NOTki5EmL7TUduTs6bu3EdnSycntVJ4re8kgZA+wx9IueI2Y11bfbgwtzuE0KQ==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
|
||||
strip-json-comments@3.1.1:
|
||||
resolution: {integrity: sha512-6fPc+R4ihwqP6N/aIv2f1gMH8lOVtWQHoqC4yK6oSDVVocumAsfCqjkXnqiYMhmMwS/mEHLp7Vehlt3ql6lEig==}
|
||||
engines: {node: '>=8'}
|
||||
@@ -938,13 +797,6 @@ packages:
|
||||
resolution: {integrity: sha512-qpCAvRl9stuOHveKsn7HncJRvv501qIacKzQlO/+Lwxc9+0q2wLyv4Dfvt80/DPn2pqOBsJdDiogXGR9+OvwRw==}
|
||||
engines: {node: '>=8'}
|
||||
|
||||
tar-fs@2.1.4:
|
||||
resolution: {integrity: sha512-mDAjwmZdh7LTT6pNleZ05Yt65HC3E+NiQzl672vQG38jIrehtJk/J3mNwIg+vShQPcLF/LV7CMnDW6vjj6sfYQ==}
|
||||
|
||||
tar-stream@2.2.0:
|
||||
resolution: {integrity: sha512-ujeqbceABgwMZxEJnk2HDY2DlnUZ+9oEcb1KzTVfYHio0UE6dG71n60d8D2I4qNvleWrrXpmjpt7vZeF1LnMZQ==}
|
||||
engines: {node: '>=6'}
|
||||
|
||||
tinyglobby@0.2.16:
|
||||
resolution: {integrity: sha512-pn99VhoACYR8nFHhxqix+uvsbXineAasWm5ojXoN8xEwK5Kd3/TrhNn1wByuD52UxWRLy8pu+kRMniEi6Eq9Zg==}
|
||||
engines: {node: '>=12.0.0'}
|
||||
@@ -952,9 +804,6 @@ packages:
|
||||
tslib@2.8.1:
|
||||
resolution: {integrity: sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==}
|
||||
|
||||
tunnel-agent@0.6.0:
|
||||
resolution: {integrity: sha512-McnNiV1l8RYeY8tBgEpuodCC1mLUdbSN+CYBL7kJsJNInOP8UjDDEwdk6Mw60vdLLrr5NHKZhMAOSrR2NZuQ+w==}
|
||||
|
||||
type-check@0.4.0:
|
||||
resolution: {integrity: sha512-XleUoc9uwGXqjWwXaUTZAmzMcFZ5858QA2vvx1Ur5xIcixXIP+8LnFDgRplU30us6teqdlskFfu+ae4K79Ooew==}
|
||||
engines: {node: '>= 0.8.0'}
|
||||
@@ -968,9 +817,6 @@ packages:
|
||||
uri-js@4.4.1:
|
||||
resolution: {integrity: sha512-7rKUyy33Q1yc98pQ1DAmLtwX109F7TIfWlW1Ydo8Wl1ii1SeHieeh0HHfPeL2fMXK6z0s8ecKs9frCuLJvndBg==}
|
||||
|
||||
util-deprecate@1.0.2:
|
||||
resolution: {integrity: sha512-EPD5q1uXyFxJpCrLnCc1nHnq3gOa6DZBocAIiI2TaSCA7VCJ1UJDMagCzIkXNsUYfD1daK//LTEQ8xiIbrHtcw==}
|
||||
|
||||
vite@8.0.12:
|
||||
resolution: {integrity: sha512-w2dDofOWv2QB09ZITZBsvKTVAlYvPR4IAmrY/v0ir9KvLs0xybR7i48wxhM1/oyBWO34wPns+bPGw5ZrZqDpZg==}
|
||||
engines: {node: ^20.19.0 || >=22.12.0}
|
||||
@@ -1019,26 +865,10 @@ packages:
|
||||
engines: {node: '>= 8'}
|
||||
hasBin: true
|
||||
|
||||
wmf@1.0.2:
|
||||
resolution: {integrity: sha512-/p9K7bEh0Dj6WbXg4JG0xvLQmIadrner1bi45VMJTfnbVHsc7yIajZyoSoK60/dtVBs12Fm6WkUI5/3WAVsNMw==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
word-wrap@1.2.5:
|
||||
resolution: {integrity: sha512-BN22B5eaMMI9UMtjrGd5g5eCYPpCPDUy0FJXbYsaT5zYxjFOckS53SQDE3pWkVoWpHXVb3BrYcEN4Twa55B5cA==}
|
||||
engines: {node: '>=0.10.0'}
|
||||
|
||||
word@0.3.0:
|
||||
resolution: {integrity: sha512-OELeY0Q61OXpdUfTp+oweA/vtLVg5VDOXh+3he3PNzLGG/y0oylSOC1xRVj0+l4vQ3tj/bB1HVHv1ocXkQceFA==}
|
||||
engines: {node: '>=0.8'}
|
||||
|
||||
wrappy@1.0.2:
|
||||
resolution: {integrity: sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==}
|
||||
|
||||
xlsx@0.18.5:
|
||||
resolution: {integrity: sha512-dmg3LCjBPHZnQp5/F/+nnTa+miPJxUXB6vtk42YjBBKayDNagxGEeIdWApkYPOf3Z3pm3k62Knjzp7lMeTEtFQ==}
|
||||
engines: {node: '>=0.8'}
|
||||
hasBin: true
|
||||
|
||||
yallist@3.1.1:
|
||||
resolution: {integrity: sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g==}
|
||||
|
||||
@@ -1344,8 +1174,6 @@ snapshots:
|
||||
|
||||
acorn@8.16.0: {}
|
||||
|
||||
adler-32@1.3.1: {}
|
||||
|
||||
ajv@6.15.0:
|
||||
dependencies:
|
||||
fast-deep-equal: 3.1.3
|
||||
@@ -1361,25 +1189,8 @@ snapshots:
|
||||
|
||||
balanced-match@1.0.2: {}
|
||||
|
||||
base64-js@1.5.1: {}
|
||||
|
||||
baseline-browser-mapping@2.10.29: {}
|
||||
|
||||
better-sqlite3@12.9.0:
|
||||
dependencies:
|
||||
bindings: 1.5.0
|
||||
prebuild-install: 7.1.3
|
||||
|
||||
bindings@1.5.0:
|
||||
dependencies:
|
||||
file-uri-to-path: 1.0.0
|
||||
|
||||
bl@4.1.0:
|
||||
dependencies:
|
||||
buffer: 5.7.1
|
||||
inherits: 2.0.4
|
||||
readable-stream: 3.6.2
|
||||
|
||||
brace-expansion@1.1.14:
|
||||
dependencies:
|
||||
balanced-match: 1.0.2
|
||||
@@ -1393,29 +1204,15 @@ snapshots:
|
||||
node-releases: 2.0.38
|
||||
update-browserslist-db: 1.2.3(browserslist@4.28.2)
|
||||
|
||||
buffer@5.7.1:
|
||||
dependencies:
|
||||
base64-js: 1.5.1
|
||||
ieee754: 1.2.1
|
||||
|
||||
callsites@3.1.0: {}
|
||||
|
||||
caniuse-lite@1.0.30001792: {}
|
||||
|
||||
cfb@1.2.2:
|
||||
dependencies:
|
||||
adler-32: 1.3.1
|
||||
crc-32: 1.2.2
|
||||
|
||||
chalk@4.1.2:
|
||||
dependencies:
|
||||
ansi-styles: 4.3.0
|
||||
supports-color: 7.2.0
|
||||
|
||||
chownr@1.1.4: {}
|
||||
|
||||
codepage@1.15.0: {}
|
||||
|
||||
color-convert@2.0.1:
|
||||
dependencies:
|
||||
color-name: 1.1.4
|
||||
@@ -1426,8 +1223,6 @@ snapshots:
|
||||
|
||||
convert-source-map@2.0.0: {}
|
||||
|
||||
crc-32@1.2.2: {}
|
||||
|
||||
cross-spawn@7.0.6:
|
||||
dependencies:
|
||||
path-key: 3.1.1
|
||||
@@ -1440,22 +1235,12 @@ snapshots:
|
||||
dependencies:
|
||||
ms: 2.1.3
|
||||
|
||||
decompress-response@6.0.0:
|
||||
dependencies:
|
||||
mimic-response: 3.1.0
|
||||
|
||||
deep-extend@0.6.0: {}
|
||||
|
||||
deep-is@0.1.4: {}
|
||||
|
||||
detect-libc@2.1.2: {}
|
||||
|
||||
electron-to-chromium@1.5.353: {}
|
||||
|
||||
end-of-stream@1.4.5:
|
||||
dependencies:
|
||||
once: 1.4.0
|
||||
|
||||
escalade@3.2.0: {}
|
||||
|
||||
escape-string-regexp@4.0.0: {}
|
||||
@@ -1541,8 +1326,6 @@ snapshots:
|
||||
|
||||
esutils@2.0.3: {}
|
||||
|
||||
expand-template@2.0.3: {}
|
||||
|
||||
fast-deep-equal@3.1.3: {}
|
||||
|
||||
fast-json-stable-stringify@2.1.0: {}
|
||||
@@ -1557,8 +1340,6 @@ snapshots:
|
||||
dependencies:
|
||||
flat-cache: 4.0.1
|
||||
|
||||
file-uri-to-path@1.0.0: {}
|
||||
|
||||
find-up@5.0.0:
|
||||
dependencies:
|
||||
locate-path: 6.0.0
|
||||
@@ -1571,17 +1352,11 @@ snapshots:
|
||||
|
||||
flatted@3.4.2: {}
|
||||
|
||||
frac@1.1.2: {}
|
||||
|
||||
fs-constants@1.0.0: {}
|
||||
|
||||
fsevents@2.3.3:
|
||||
optional: true
|
||||
|
||||
gensync@1.0.0-beta.2: {}
|
||||
|
||||
github-from-package@0.0.0: {}
|
||||
|
||||
glob-parent@6.0.2:
|
||||
dependencies:
|
||||
is-glob: 4.0.3
|
||||
@@ -1598,8 +1373,6 @@ snapshots:
|
||||
dependencies:
|
||||
hermes-estree: 0.25.1
|
||||
|
||||
ieee754@1.2.1: {}
|
||||
|
||||
ignore@5.3.2: {}
|
||||
|
||||
import-fresh@3.3.1:
|
||||
@@ -1609,10 +1382,6 @@ snapshots:
|
||||
|
||||
imurmurhash@0.1.4: {}
|
||||
|
||||
inherits@2.0.4: {}
|
||||
|
||||
ini@1.3.8: {}
|
||||
|
||||
is-extglob@2.1.1: {}
|
||||
|
||||
is-glob@4.0.3:
|
||||
@@ -1705,34 +1474,18 @@ snapshots:
|
||||
dependencies:
|
||||
yallist: 3.1.1
|
||||
|
||||
mimic-response@3.1.0: {}
|
||||
|
||||
minimatch@3.1.5:
|
||||
dependencies:
|
||||
brace-expansion: 1.1.14
|
||||
|
||||
minimist@1.2.8: {}
|
||||
|
||||
mkdirp-classic@0.5.3: {}
|
||||
|
||||
ms@2.1.3: {}
|
||||
|
||||
nanoid@3.3.12: {}
|
||||
|
||||
napi-build-utils@2.0.0: {}
|
||||
|
||||
natural-compare@1.4.0: {}
|
||||
|
||||
node-abi@3.92.0:
|
||||
dependencies:
|
||||
semver: 7.8.0
|
||||
|
||||
node-releases@2.0.38: {}
|
||||
|
||||
once@1.4.0:
|
||||
dependencies:
|
||||
wrappy: 1.0.2
|
||||
|
||||
optionator@0.9.4:
|
||||
dependencies:
|
||||
deep-is: 0.1.4
|
||||
@@ -1768,37 +1521,10 @@ snapshots:
|
||||
picocolors: 1.1.1
|
||||
source-map-js: 1.2.1
|
||||
|
||||
prebuild-install@7.1.3:
|
||||
dependencies:
|
||||
detect-libc: 2.1.2
|
||||
expand-template: 2.0.3
|
||||
github-from-package: 0.0.0
|
||||
minimist: 1.2.8
|
||||
mkdirp-classic: 0.5.3
|
||||
napi-build-utils: 2.0.0
|
||||
node-abi: 3.92.0
|
||||
pump: 3.0.4
|
||||
rc: 1.2.8
|
||||
simple-get: 4.0.1
|
||||
tar-fs: 2.1.4
|
||||
tunnel-agent: 0.6.0
|
||||
|
||||
prelude-ls@1.2.1: {}
|
||||
|
||||
pump@3.0.4:
|
||||
dependencies:
|
||||
end-of-stream: 1.4.5
|
||||
once: 1.4.0
|
||||
|
||||
punycode@2.3.1: {}
|
||||
|
||||
rc@1.2.8:
|
||||
dependencies:
|
||||
deep-extend: 0.6.0
|
||||
ini: 1.3.8
|
||||
minimist: 1.2.8
|
||||
strip-json-comments: 2.0.1
|
||||
|
||||
react-dom@19.2.6(react@19.2.6):
|
||||
dependencies:
|
||||
react: 19.2.6
|
||||
@@ -1806,12 +1532,6 @@ snapshots:
|
||||
|
||||
react@19.2.6: {}
|
||||
|
||||
readable-stream@3.6.2:
|
||||
dependencies:
|
||||
inherits: 2.0.4
|
||||
string_decoder: 1.3.0
|
||||
util-deprecate: 1.0.2
|
||||
|
||||
resolve-from@4.0.0: {}
|
||||
|
||||
rolldown@1.0.0:
|
||||
@@ -1835,63 +1555,26 @@ snapshots:
|
||||
'@rolldown/binding-win32-arm64-msvc': 1.0.0
|
||||
'@rolldown/binding-win32-x64-msvc': 1.0.0
|
||||
|
||||
safe-buffer@5.2.1: {}
|
||||
|
||||
scheduler@0.27.0: {}
|
||||
|
||||
semver@6.3.1: {}
|
||||
|
||||
semver@7.8.0: {}
|
||||
|
||||
shebang-command@2.0.0:
|
||||
dependencies:
|
||||
shebang-regex: 3.0.0
|
||||
|
||||
shebang-regex@3.0.0: {}
|
||||
|
||||
simple-concat@1.0.1: {}
|
||||
|
||||
simple-get@4.0.1:
|
||||
dependencies:
|
||||
decompress-response: 6.0.0
|
||||
once: 1.4.0
|
||||
simple-concat: 1.0.1
|
||||
|
||||
source-map-js@1.2.1: {}
|
||||
|
||||
sql.js@1.14.1: {}
|
||||
|
||||
ssf@0.11.2:
|
||||
dependencies:
|
||||
frac: 1.1.2
|
||||
|
||||
string_decoder@1.3.0:
|
||||
dependencies:
|
||||
safe-buffer: 5.2.1
|
||||
|
||||
strip-json-comments@2.0.1: {}
|
||||
|
||||
strip-json-comments@3.1.1: {}
|
||||
|
||||
supports-color@7.2.0:
|
||||
dependencies:
|
||||
has-flag: 4.0.0
|
||||
|
||||
tar-fs@2.1.4:
|
||||
dependencies:
|
||||
chownr: 1.1.4
|
||||
mkdirp-classic: 0.5.3
|
||||
pump: 3.0.4
|
||||
tar-stream: 2.2.0
|
||||
|
||||
tar-stream@2.2.0:
|
||||
dependencies:
|
||||
bl: 4.1.0
|
||||
end-of-stream: 1.4.5
|
||||
fs-constants: 1.0.0
|
||||
inherits: 2.0.4
|
||||
readable-stream: 3.6.2
|
||||
|
||||
tinyglobby@0.2.16:
|
||||
dependencies:
|
||||
fdir: 6.5.0(picomatch@4.0.4)
|
||||
@@ -1900,10 +1583,6 @@ snapshots:
|
||||
tslib@2.8.1:
|
||||
optional: true
|
||||
|
||||
tunnel-agent@0.6.0:
|
||||
dependencies:
|
||||
safe-buffer: 5.2.1
|
||||
|
||||
type-check@0.4.0:
|
||||
dependencies:
|
||||
prelude-ls: 1.2.1
|
||||
@@ -1918,8 +1597,6 @@ snapshots:
|
||||
dependencies:
|
||||
punycode: 2.3.1
|
||||
|
||||
util-deprecate@1.0.2: {}
|
||||
|
||||
vite@8.0.12:
|
||||
dependencies:
|
||||
lightningcss: 1.32.0
|
||||
@@ -1934,24 +1611,8 @@ snapshots:
|
||||
dependencies:
|
||||
isexe: 2.0.0
|
||||
|
||||
wmf@1.0.2: {}
|
||||
|
||||
word-wrap@1.2.5: {}
|
||||
|
||||
word@0.3.0: {}
|
||||
|
||||
wrappy@1.0.2: {}
|
||||
|
||||
xlsx@0.18.5:
|
||||
dependencies:
|
||||
adler-32: 1.3.1
|
||||
cfb: 1.2.2
|
||||
codepage: 1.15.0
|
||||
crc-32: 1.2.2
|
||||
ssf: 0.11.2
|
||||
wmf: 1.0.2
|
||||
word: 0.3.0
|
||||
|
||||
yallist@3.1.1: {}
|
||||
|
||||
yocto-queue@0.1.0: {}
|
||||
|
||||
@@ -1,63 +0,0 @@
|
||||
// Audit: compare expected unique SBDs from raw Excel files vs DB row count
|
||||
import XLSX from "xlsx";
|
||||
import Database from "better-sqlite3";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
|
||||
const RAW_DIR = "D:/tiennm99/thptqg2017/data";
|
||||
const DB_PATH = "D:/tiennm99/thptqg2017/public/thptqg2017.db";
|
||||
|
||||
function collectFiles() {
|
||||
const out = [];
|
||||
for (const f of fs.readdirSync(RAW_DIR)) {
|
||||
const full = path.join(RAW_DIR, f);
|
||||
if (fs.statSync(full).isFile() && f.endsWith(".xlsx")) out.push(full);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function isHeader(row) {
|
||||
if (!row || row.length < 3) return false;
|
||||
const first = String(row[0] || "").toUpperCase();
|
||||
return first === "HO_TEN" || first === "HỌ TÊN" || first === "STT";
|
||||
}
|
||||
|
||||
const allSbd = new Set();
|
||||
let totalDataRows = 0,
|
||||
emptyName = 0,
|
||||
emptySbd = 0,
|
||||
bothEmpty = 0;
|
||||
|
||||
for (const file of collectFiles()) {
|
||||
const wb = XLSX.readFile(file);
|
||||
const ws = wb.Sheets[wb.SheetNames[0]];
|
||||
const rows = XLSX.utils.sheet_to_json(ws, { header: 1 });
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
const r = rows[i];
|
||||
if (i === 0 && isHeader(r)) continue;
|
||||
totalDataRows++;
|
||||
const hoTen = String(r?.[0] || "").trim();
|
||||
const sbd = String(r?.[2] || "").trim();
|
||||
if (!hoTen && !sbd) {
|
||||
bothEmpty++;
|
||||
continue;
|
||||
}
|
||||
if (!hoTen) emptyName++;
|
||||
if (!sbd) emptySbd++;
|
||||
if (sbd) allSbd.add(sbd);
|
||||
}
|
||||
}
|
||||
|
||||
const db = new Database(DB_PATH, { readonly: true });
|
||||
const dbCount = db.prepare("SELECT COUNT(*) c FROM student").get().c;
|
||||
|
||||
console.log("=== Source vs DB ===");
|
||||
console.log(`Source: total data rows across all files: ${totalDataRows}`);
|
||||
console.log(`Source: rows with empty name AND sbd (skipped): ${bothEmpty}`);
|
||||
console.log(`Source: rows with missing name only: ${emptyName}`);
|
||||
console.log(`Source: rows with missing sbd only: ${emptySbd}`);
|
||||
console.log(`Source: distinct SBDs: ${allSbd.size}`);
|
||||
console.log(`DB: row count: ${dbCount}`);
|
||||
console.log(
|
||||
`Match: ${allSbd.size === dbCount ? "YES — all unique SBDs accounted for" : "NO — gap of " + (allSbd.size - dbCount)}`,
|
||||
);
|
||||
@@ -1,83 +0,0 @@
|
||||
// Build DB from data-old/ — the original 63 xlsx files (pre-baotintuc refresh).
|
||||
// Quirks: single-sheet workbooks; some files have no header row, so we
|
||||
// isHeaderRow-test row 0 and skip only when it matches. A previous bug let a
|
||||
// HO_TEN='SOBAODANH' header leak through — that's handled here by the
|
||||
// hoTen/soBaoDanh validity check (trim + non-empty + soBaoDanh must be digits).
|
||||
import XLSX from "xlsx";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { fileURLToPath } from "url";
|
||||
import {
|
||||
createDb,
|
||||
parseScores,
|
||||
isHeaderRow,
|
||||
buildRow,
|
||||
} from "./build-lib.js";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const SRC_DIR = path.join(__dirname, "..", "data-old");
|
||||
const DB_PATH = path.join(__dirname, "..", "public-old", "thptqg2017.db");
|
||||
|
||||
function collectFiles() {
|
||||
return fs
|
||||
.readdirSync(SRC_DIR)
|
||||
.filter((f) => f.endsWith(".xlsx") || f.endsWith(".xls"))
|
||||
.map((f) => path.join(SRC_DIR, f));
|
||||
}
|
||||
|
||||
function main() {
|
||||
const { db, insert } = createDb(DB_PATH);
|
||||
const files = collectFiles();
|
||||
let sourceRows = 0, skipped = 0, errors = 0;
|
||||
|
||||
const run = db.transaction(() => {
|
||||
for (const file of files) {
|
||||
const base = path.basename(file);
|
||||
let fileRows = 0;
|
||||
const wb = XLSX.readFile(file);
|
||||
// data-old/: always read ONLY sheet 0 — these are xlsx and never hit the
|
||||
// 65k row cap, so any additional sheet is noise.
|
||||
const rows = XLSX.utils.sheet_to_json(wb.Sheets[wb.SheetNames[0]], {
|
||||
header: 1,
|
||||
});
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
if (i === 0 && isHeaderRow(rows[i])) continue;
|
||||
sourceRows++;
|
||||
const r = rows[i];
|
||||
const hoTen = String(r?.[0] || "").trim();
|
||||
const ngaySinh = String(r?.[1] || "").trim();
|
||||
const soBaoDanh = String(r?.[2] || "").trim();
|
||||
const diemThi = String(r?.[3] || "");
|
||||
// Extra guard specific to old data: reject rows where sbd is not
|
||||
// numeric or ho_ten literally says 'HO_TEN' (header leak from the
|
||||
// prior conversion pipeline).
|
||||
if (!soBaoDanh || !hoTen) { skipped++; continue; }
|
||||
if (!/^\d+$/.test(soBaoDanh)) { skipped++; continue; }
|
||||
try {
|
||||
insert.run(buildRow({ hoTen, ngaySinh, soBaoDanh, scores: parseScores(diemThi) }));
|
||||
fileRows++;
|
||||
} catch (err) {
|
||||
errors++;
|
||||
if (errors <= 5) console.warn(` [warn] ${base}: ${err.message}`);
|
||||
}
|
||||
}
|
||||
console.log(` ${base}: ${fileRows} rows`);
|
||||
}
|
||||
});
|
||||
|
||||
console.log(`[build] data-old/ → ${DB_PATH} (${files.length} files)`);
|
||||
run();
|
||||
db.exec("VACUUM");
|
||||
|
||||
const dbCount = db.prepare("SELECT COUNT(*) c FROM student").get().c;
|
||||
console.log(`\nSource data rows (post-header): ${sourceRows}`);
|
||||
console.log(` skipped (empty/non-numeric SBD): ${skipped}`);
|
||||
console.log(` insertable: ${sourceRows - skipped}`);
|
||||
console.log(` insert errors: ${errors}`);
|
||||
console.log(`DB rows (distinct SBD): ${dbCount}`);
|
||||
const sz = fs.statSync(DB_PATH).size;
|
||||
console.log(`Size: ${(sz / 1024 / 1024).toFixed(1)} MB`);
|
||||
db.close();
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,86 +0,0 @@
|
||||
// Build DB from data-old2/ — the 54 "update/" corrected-export files.
|
||||
// Quirks:
|
||||
// • 54 provinces only (no Hà Nội, Bình Phước etc.)
|
||||
// • Mostly 2 sheets (one main + one empty trailing) — safe to iterate all
|
||||
// • HCM (24.HCM_UTLQ.xlsx) overflows into Sheet2 with +6,446 students —
|
||||
// multi-sheet walk is REQUIRED for it
|
||||
// • Some headers have extended columns beyond DIEM_THI (pre-parsed per-
|
||||
// subject cols). We still use only the DIEM_THI text at index 3.
|
||||
import XLSX from "xlsx";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { fileURLToPath } from "url";
|
||||
import {
|
||||
createDb,
|
||||
parseScores,
|
||||
isHeaderRow,
|
||||
buildRow,
|
||||
} from "./build-lib.js";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const SRC_DIR = path.join(__dirname, "..", "data-old2");
|
||||
const DB_PATH = path.join(__dirname, "..", "public-old2", "thptqg2017.db");
|
||||
|
||||
function collectFiles() {
|
||||
return fs
|
||||
.readdirSync(SRC_DIR)
|
||||
.filter((f) => f.endsWith(".xlsx") || f.endsWith(".xls"))
|
||||
.map((f) => path.join(SRC_DIR, f));
|
||||
}
|
||||
|
||||
function main() {
|
||||
const { db, insert } = createDb(DB_PATH);
|
||||
const files = collectFiles();
|
||||
let sourceRows = 0, skipped = 0, errors = 0;
|
||||
|
||||
const run = db.transaction(() => {
|
||||
for (const file of files) {
|
||||
const base = path.basename(file);
|
||||
let fileRows = 0;
|
||||
const wb = XLSX.readFile(file);
|
||||
|
||||
for (const sheetName of wb.SheetNames) {
|
||||
const rows = XLSX.utils.sheet_to_json(wb.Sheets[sheetName], {
|
||||
header: 1,
|
||||
});
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
if (i === 0 && isHeaderRow(rows[i])) continue;
|
||||
const r = rows[i];
|
||||
// Skip completely blank rows (common in xlsx tails) before counting
|
||||
if (!r || r.every((c) => c === "" || c == null)) continue;
|
||||
sourceRows++;
|
||||
const hoTen = String(r?.[0] || "").trim();
|
||||
const ngaySinh = String(r?.[1] || "").trim();
|
||||
const soBaoDanh = String(r?.[2] || "").trim();
|
||||
const diemThi = String(r?.[3] || "");
|
||||
if (!soBaoDanh || !hoTen) { skipped++; continue; }
|
||||
if (!/^\d+$/.test(soBaoDanh)) { skipped++; continue; }
|
||||
try {
|
||||
insert.run(buildRow({ hoTen, ngaySinh, soBaoDanh, scores: parseScores(diemThi) }));
|
||||
fileRows++;
|
||||
} catch (err) {
|
||||
errors++;
|
||||
if (errors <= 5) console.warn(` [warn] ${base}: ${err.message}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
console.log(` ${base}: ${fileRows} rows`);
|
||||
}
|
||||
});
|
||||
|
||||
console.log(`[build] data-old2/ → ${DB_PATH} (${files.length} files)`);
|
||||
run();
|
||||
db.exec("VACUUM");
|
||||
|
||||
const dbCount = db.prepare("SELECT COUNT(*) c FROM student").get().c;
|
||||
console.log(`\nSource non-blank data rows: ${sourceRows}`);
|
||||
console.log(` skipped (empty/non-numeric SBD): ${skipped}`);
|
||||
console.log(` insertable: ${sourceRows - skipped}`);
|
||||
console.log(` insert errors: ${errors}`);
|
||||
console.log(`DB rows (distinct SBD): ${dbCount}`);
|
||||
const sz = fs.statSync(DB_PATH).size;
|
||||
console.log(`Size: ${(sz / 1024 / 1024).toFixed(1)} MB`);
|
||||
db.close();
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,90 +0,0 @@
|
||||
// Build DB from data/ (baotintuc.vn .xls source).
|
||||
// Quirks: .xls has a 65,536-row-per-sheet limit. Hà Nội and HCM overflow into
|
||||
// Sheet2, so we MUST iterate every sheet. Header row may or may not be
|
||||
// present on each sheet.
|
||||
import XLSX from "xlsx";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { fileURLToPath } from "url";
|
||||
import {
|
||||
createDb,
|
||||
parseScores,
|
||||
isHeaderRow,
|
||||
buildRow,
|
||||
} from "./build-lib.js";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const SRC_DIR = path.join(__dirname, "..", "data");
|
||||
const DB_PATH = path.join(__dirname, "..", "public", "thptqg2017.db");
|
||||
|
||||
function collectFiles() {
|
||||
return fs
|
||||
.readdirSync(SRC_DIR)
|
||||
.filter((f) => f.endsWith(".xls") || f.endsWith(".xlsx"))
|
||||
.map((f) => path.join(SRC_DIR, f));
|
||||
}
|
||||
|
||||
function main() {
|
||||
const { db, insert } = createDb(DB_PATH);
|
||||
|
||||
const files = collectFiles();
|
||||
let sourceRows = 0; // total data rows observed across ALL sheets (post-header)
|
||||
let skipped = 0; // empty/invalid rows skipped
|
||||
let errors = 0;
|
||||
|
||||
const run = db.transaction(() => {
|
||||
for (const file of files) {
|
||||
const base = path.basename(file);
|
||||
let fileRows = 0;
|
||||
const wb = XLSX.readFile(file);
|
||||
|
||||
for (const sheetName of wb.SheetNames) {
|
||||
const rows = XLSX.utils.sheet_to_json(wb.Sheets[sheetName], {
|
||||
header: 1,
|
||||
});
|
||||
for (let i = 0; i < rows.length; i++) {
|
||||
if (i === 0 && isHeaderRow(rows[i])) continue;
|
||||
sourceRows++;
|
||||
const r = rows[i];
|
||||
const hoTen = String(r?.[0] || "").trim();
|
||||
const ngaySinh = String(r?.[1] || "").trim();
|
||||
const soBaoDanh = String(r?.[2] || "").trim();
|
||||
const diemThi = String(r?.[3] || "");
|
||||
if (!soBaoDanh || !hoTen) { skipped++; continue; }
|
||||
try {
|
||||
insert.run(buildRow({ hoTen, ngaySinh, soBaoDanh, scores: parseScores(diemThi) }));
|
||||
fileRows++;
|
||||
} catch (err) {
|
||||
errors++;
|
||||
if (errors <= 5) console.warn(` [warn] ${base}: ${err.message}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
console.log(` ${base}: ${fileRows} rows`);
|
||||
}
|
||||
});
|
||||
|
||||
console.log(`[build] data/ → ${DB_PATH} (${files.length} files)`);
|
||||
run();
|
||||
db.exec("VACUUM");
|
||||
|
||||
const dbCount = db.prepare("SELECT COUNT(*) c FROM student").get().c;
|
||||
const distinctSbd = sourceRows - skipped;
|
||||
console.log(`\nSource data rows (post-header): ${sourceRows}`);
|
||||
console.log(` skipped (empty/invalid): ${skipped}`);
|
||||
console.log(` insertable: ${distinctSbd}`);
|
||||
console.log(` insert errors: ${errors}`);
|
||||
console.log(`DB rows (distinct SBD): ${dbCount}`);
|
||||
// data/ comes from a single source, so every SBD should be unique after
|
||||
// header skip — no duplicates expected.
|
||||
if (dbCount !== distinctSbd - 0 && errors === 0) {
|
||||
const gap = distinctSbd - dbCount;
|
||||
if (gap === 0) console.log(`Audit: OK — every source row made it in.`);
|
||||
else console.log(`Audit: ${gap} row(s) collapsed (duplicate SBDs overwriting).`);
|
||||
}
|
||||
const sz = fs.statSync(DB_PATH).size;
|
||||
console.log(`Size: ${(sz / 1024 / 1024).toFixed(1)} MB`);
|
||||
db.close();
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,117 +0,0 @@
|
||||
// Shared SQLite schema + score-text helpers for all 3 build pipelines.
|
||||
// The parse LOOP for each dataset lives in its own build-database-*.js file
|
||||
// (so source-specific quirks stay isolated), but the DB shape and regex
|
||||
// dictionary are centralised here so the 3 DBs stay drop-in compatible with
|
||||
// the same frontend.
|
||||
import Database from "better-sqlite3";
|
||||
import fs from "fs";
|
||||
|
||||
const NUM = "(\\d+(?:\\.\\d+)?)";
|
||||
export const SCORE_PATTERNS = {
|
||||
toan: new RegExp("Toán:\\s*" + NUM),
|
||||
ngu_van: new RegExp("Ngữ văn:\\s*" + NUM),
|
||||
vat_ly: new RegExp("Vật lí:\\s*" + NUM),
|
||||
hoa_hoc: new RegExp("Hóa học:\\s*" + NUM),
|
||||
sinh_hoc: new RegExp("Sinh học:\\s*" + NUM),
|
||||
khtn: new RegExp("KHTN:\\s*" + NUM),
|
||||
lich_su: new RegExp("Lịch sử:\\s*" + NUM),
|
||||
dia_ly: new RegExp("Địa lí:\\s*" + NUM),
|
||||
gdcd: new RegExp("GDCD:\\s*" + NUM),
|
||||
khxh: new RegExp("KHXH:\\s*" + NUM),
|
||||
tieng_anh: new RegExp("Tiếng Anh:\\s*" + NUM),
|
||||
tieng_phap: new RegExp("Tiếng Pháp:\\s*" + NUM),
|
||||
tieng_nga: new RegExp("Tiếng Nga:\\s*" + NUM),
|
||||
tieng_trung: new RegExp("Tiếng Trung:\\s*" + NUM),
|
||||
};
|
||||
|
||||
export function parseScores(diemThi) {
|
||||
const out = {};
|
||||
for (const [k, re] of Object.entries(SCORE_PATTERNS)) {
|
||||
const m = diemThi.match(re);
|
||||
if (m) out[k] = parseFloat(m[1]);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function toAscii(str) {
|
||||
return str
|
||||
.normalize("NFD")
|
||||
.replace(/[\u0300-\u036f]/g, "")
|
||||
.replace(/đ/gi, "d")
|
||||
.toLowerCase();
|
||||
}
|
||||
|
||||
export function isHeaderRow(row) {
|
||||
if (!row || row.length < 3) return false;
|
||||
const first = String(row[0] || "").toUpperCase();
|
||||
return first === "HO_TEN" || first === "HỌ TÊN" || first === "STT";
|
||||
}
|
||||
|
||||
// Create empty DB file at dbPath with the canonical schema + prepared insert.
|
||||
// Caller owns the returned handles and must close them.
|
||||
export function createDb(dbPath) {
|
||||
fs.mkdirSync(dbPath.replace(/[^/\\]+$/, ""), { recursive: true });
|
||||
if (fs.existsSync(dbPath)) fs.unlinkSync(dbPath);
|
||||
const db = new Database(dbPath);
|
||||
db.exec(`
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
`);
|
||||
const insert = db.prepare(`
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(@so_bao_danh, @ho_ten, @ho_ten_ascii, @ngay_sinh,
|
||||
@toan, @ngu_van, @vat_ly, @hoa_hoc, @sinh_hoc, @khtn,
|
||||
@lich_su, @dia_ly, @gdcd, @khxh,
|
||||
@tieng_anh, @tieng_phap, @tieng_nga, @tieng_trung)
|
||||
`);
|
||||
return { db, insert };
|
||||
}
|
||||
|
||||
// Build the @named params object the INSERT expects
|
||||
export function buildRow({ hoTen, ngaySinh, soBaoDanh, scores }) {
|
||||
return {
|
||||
so_bao_danh: soBaoDanh,
|
||||
ho_ten: hoTen,
|
||||
ho_ten_ascii: toAscii(hoTen),
|
||||
ngay_sinh: ngaySinh || null,
|
||||
toan: scores.toan ?? null,
|
||||
ngu_van: scores.ngu_van ?? null,
|
||||
vat_ly: scores.vat_ly ?? null,
|
||||
hoa_hoc: scores.hoa_hoc ?? null,
|
||||
sinh_hoc: scores.sinh_hoc ?? null,
|
||||
khtn: scores.khtn ?? null,
|
||||
lich_su: scores.lich_su ?? null,
|
||||
dia_ly: scores.dia_ly ?? null,
|
||||
gdcd: scores.gdcd ?? null,
|
||||
khxh: scores.khxh ?? null,
|
||||
tieng_anh: scores.tieng_anh ?? null,
|
||||
tieng_phap: scores.tieng_phap ?? null,
|
||||
tieng_nga: scores.tieng_nga ?? null,
|
||||
tieng_trung: scores.tieng_trung ?? null,
|
||||
};
|
||||
}
|
||||
Generated
+1159
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,30 @@
|
||||
[package]
|
||||
name = "xlsxread"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
description = "Rust CLI replacing SheetJS xlsx build scripts for thptqg2017/thptqg2016"
|
||||
|
||||
[dependencies]
|
||||
calamine = "0.26"
|
||||
rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
toml = "0.8"
|
||||
regex = "1"
|
||||
unicode-normalization = "0.1"
|
||||
thiserror = "1"
|
||||
anyhow = "1"
|
||||
glob = "0.3"
|
||||
|
||||
# zip is already a transitive dep of calamine; pin explicitly so tests can use it
|
||||
[dev-dependencies]
|
||||
zip = "2"
|
||||
rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
|
||||
[[bin]]
|
||||
name = "xlsxread"
|
||||
path = "src/main.rs"
|
||||
|
||||
[[test]]
|
||||
name = "golden"
|
||||
path = "tests/golden.rs"
|
||||
@@ -0,0 +1,75 @@
|
||||
# Config for data-old/ — 63 .xlsx files (pre-baotintuc refresh)
|
||||
# Sheet mode: "first" — single-sheet workbooks, never hit 65k row cap
|
||||
# SBD validation: require ^\d+$ (build-database-old.js:55 guard)
|
||||
# Blank row strip: off (no explicit blank-skip in build-database-old.js)
|
||||
|
||||
[reader]
|
||||
sheet_mode = "first"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = true
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
|
||||
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,76 @@
|
||||
# Config for data-old2/ — 54 .xlsx files (corrected-export set)
|
||||
# Sheet mode: "all" — HCM (24.HCM_UTLQ.xlsx) overflows into Sheet2 (+6,446 rows)
|
||||
# SBD validation: require ^\d+$ (build-database-old2.js:57 guard)
|
||||
# Blank row strip: true — skip fully blank rows BEFORE counting sourceRows
|
||||
# (build-database-old2.js:50-51: blank row check before sourceRows++)
|
||||
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = true
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = true
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
|
||||
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,75 @@
|
||||
# Config for data/ — 63 .xls files from baotintuc.vn
|
||||
# Sheet mode: "all" because Hà Nội and HCM overflow into Sheet2 (65k row cap)
|
||||
# SBD validation: no numeric guard (build-database.js does not apply ^\d+$)
|
||||
# Blank row strip: off
|
||||
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
|
||||
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,174 @@
|
||||
/// Audit subcommand: replicates audit-row-counts.js exactly.
|
||||
///
|
||||
/// Reads all .xlsx files from the input directory (sheet 0 only, matching the
|
||||
/// JS script's behaviour at audit-row-counts.js:33), collects distinct SBDs
|
||||
/// into a HashSet, then queries `SELECT COUNT(*) FROM student` from the DB.
|
||||
/// Prints the same lines as audit-row-counts.js:54-62 and exits 0 on match,
|
||||
/// 1 on mismatch.
|
||||
use std::collections::HashSet;
|
||||
use std::path::Path;
|
||||
|
||||
use calamine::{open_workbook_auto, Data, Reader};
|
||||
|
||||
use crate::config::DatasetConfig;
|
||||
use crate::error::BuildError;
|
||||
use crate::reader::is_header_row;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Audit result
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct AuditResult {
|
||||
pub total_data_rows: u64,
|
||||
pub both_empty: u64,
|
||||
pub empty_name: u64,
|
||||
pub empty_sbd: u64,
|
||||
pub distinct_sbds: usize,
|
||||
pub db_count: i64,
|
||||
pub matched: bool,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Main audit logic
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Collect distinct SBDs from all xlsx files in `input_dir`, query `db_path`,
|
||||
/// print the audit report and return the result.
|
||||
///
|
||||
/// The JS script reads only sheet 0 for every file (audit-row-counts.js:33).
|
||||
/// Unlike build-database.js, the audit script does NOT iterate all sheets.
|
||||
pub fn run_audit(
|
||||
input_dir: &Path,
|
||||
db_path: &Path,
|
||||
cfg: &DatasetConfig,
|
||||
) -> Result<AuditResult, BuildError> {
|
||||
// Collect .xlsx files (audit-row-counts.js only checks .xlsx — line 15)
|
||||
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
|
||||
.map_err(|e| BuildError::Io {
|
||||
path: input_dir.display().to_string(),
|
||||
source: e,
|
||||
})?
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| e.eq_ignore_ascii_case("xlsx"))
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let mut all_sbd: HashSet<String> = HashSet::new();
|
||||
let mut total_data_rows: u64 = 0;
|
||||
let mut empty_name: u64 = 0;
|
||||
let mut empty_sbd: u64 = 0;
|
||||
let mut both_empty: u64 = 0;
|
||||
|
||||
for file in &files {
|
||||
let path_str = file.display().to_string();
|
||||
let mut workbook = open_workbook_auto(file).map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let sheet_names = workbook.sheet_names().to_vec();
|
||||
if sheet_names.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
// audit-row-counts.js reads only sheet 0 (line 33: wb.SheetNames[0])
|
||||
let range =
|
||||
workbook
|
||||
.worksheet_range(&sheet_names[0])
|
||||
.map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let mut first_row = true;
|
||||
for raw in range.rows() {
|
||||
let row: Vec<Data> = raw.to_vec();
|
||||
|
||||
// Skip header row on first row only
|
||||
if first_row {
|
||||
first_row = false;
|
||||
if is_header_row(&row, &cfg.header) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
total_data_rows += 1;
|
||||
|
||||
let ho_ten = row
|
||||
.get(cfg.columns.ho_ten)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let sbd = row
|
||||
.get(cfg.columns.so_bao_danh)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
|
||||
if ho_ten.is_empty() && sbd.is_empty() {
|
||||
both_empty += 1;
|
||||
continue;
|
||||
}
|
||||
if ho_ten.is_empty() {
|
||||
empty_name += 1;
|
||||
}
|
||||
if sbd.is_empty() {
|
||||
empty_sbd += 1;
|
||||
}
|
||||
if !sbd.is_empty() {
|
||||
all_sbd.insert(sbd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Query DB count
|
||||
let conn =
|
||||
rusqlite::Connection::open_with_flags(db_path, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY)?;
|
||||
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
|
||||
|
||||
let distinct_sbds = all_sbd.len();
|
||||
let matched = distinct_sbds as i64 == db_count;
|
||||
|
||||
Ok(AuditResult {
|
||||
total_data_rows,
|
||||
both_empty,
|
||||
empty_name,
|
||||
empty_sbd,
|
||||
distinct_sbds,
|
||||
db_count,
|
||||
matched,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Print audit report — mirrors audit-row-counts.js:54-62 exactly
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn print_audit_report(r: &AuditResult) {
|
||||
println!("=== Source vs DB ===");
|
||||
println!(
|
||||
"Source: total data rows across all files: {}",
|
||||
r.total_data_rows
|
||||
);
|
||||
println!(
|
||||
"Source: rows with empty name AND sbd (skipped): {}",
|
||||
r.both_empty
|
||||
);
|
||||
println!("Source: rows with missing name only: {}", r.empty_name);
|
||||
println!("Source: rows with missing sbd only: {}", r.empty_sbd);
|
||||
println!("Source: distinct SBDs: {}", r.distinct_sbds);
|
||||
println!("DB: row count: {}", r.db_count);
|
||||
println!(
|
||||
"Match: {}",
|
||||
if r.matched {
|
||||
"YES — all unique SBDs accounted for".to_string()
|
||||
} else {
|
||||
format!("NO — gap of {}", r.distinct_sbds as i64 - r.db_count)
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/// CLI argument structs via clap derive.
|
||||
use std::path::PathBuf;
|
||||
|
||||
use clap::{Parser, Subcommand};
|
||||
|
||||
#[derive(Parser)]
|
||||
#[command(
|
||||
name = "xlsxread",
|
||||
version,
|
||||
about = "Read .xls/.xlsx files and build SQLite databases for thptqg datasets"
|
||||
)]
|
||||
pub struct Cli {
|
||||
#[command(subcommand)]
|
||||
pub cmd: Cmd,
|
||||
}
|
||||
|
||||
#[derive(Subcommand)]
|
||||
pub enum Cmd {
|
||||
/// Read input spreadsheets and write a SQLite database
|
||||
Build {
|
||||
/// Path to the dataset TOML config file
|
||||
#[arg(long)]
|
||||
schema: PathBuf,
|
||||
|
||||
/// Directory containing the .xls / .xlsx source files
|
||||
#[arg(long)]
|
||||
input: PathBuf,
|
||||
|
||||
/// Output SQLite database path
|
||||
#[arg(long)]
|
||||
output: PathBuf,
|
||||
},
|
||||
|
||||
/// Audit: compare distinct SBD count from xlsx files vs DB row count
|
||||
Audit {
|
||||
/// Path to the dataset TOML config file
|
||||
#[arg(long)]
|
||||
schema: PathBuf,
|
||||
|
||||
/// Directory containing the .xlsx source files
|
||||
#[arg(long)]
|
||||
input: PathBuf,
|
||||
|
||||
/// SQLite database to compare against
|
||||
#[arg(long)]
|
||||
db: PathBuf,
|
||||
},
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Top-level dataset configuration loaded from a .toml file
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct DatasetConfig {
|
||||
pub reader: ReaderCfg,
|
||||
pub columns: ColumnMap,
|
||||
pub validation: ValidationCfg,
|
||||
pub header: HeaderCfg,
|
||||
pub schema: SchemaCfg,
|
||||
/// field name → regex source string (one entry per scoreable subject)
|
||||
pub scores: HashMap<String, String>,
|
||||
pub insert: InsertCfg,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ReaderCfg {
|
||||
/// "all" → iterate every sheet (handles HCM/HN overflow); "first" → sheet 0 only
|
||||
pub sheet_mode: SheetMode,
|
||||
/// If true, skip rows where every cell is empty/null before counting (data-old2 quirk)
|
||||
pub strip_blank_rows: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum SheetMode {
|
||||
All,
|
||||
First,
|
||||
}
|
||||
|
||||
/// Zero-indexed column positions in the source spreadsheet row.
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ColumnMap {
|
||||
pub ho_ten: usize,
|
||||
pub ngay_sinh: usize,
|
||||
pub so_bao_danh: usize,
|
||||
pub diem_thi: usize,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ValidationCfg {
|
||||
/// build-database-old.js / -old2.js require soBaoDanh to match ^\d+$
|
||||
pub require_numeric_sbd: bool,
|
||||
pub require_nonempty_name: bool,
|
||||
pub require_nonempty_sbd: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct HeaderCfg {
|
||||
/// Tokens to match against row[0].to_uppercase() to detect a header row
|
||||
pub tokens: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct SchemaCfg {
|
||||
/// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX)
|
||||
pub ddl: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct InsertCfg {
|
||||
/// Parameterised INSERT OR REPLACE SQL using :named_param style
|
||||
pub sql: String,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Loader
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn load_config(path: &Path) -> Result<DatasetConfig, BuildError> {
|
||||
let text = fs::read_to_string(path).map_err(|e| BuildError::Io {
|
||||
path: path.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
let cfg: DatasetConfig = toml::from_str(&text)?;
|
||||
Ok(cfg)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const SAMPLE_TOML: &str = r#"
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)"
|
||||
"#;
|
||||
|
||||
#[test]
|
||||
fn config_round_trip() {
|
||||
let cfg: DatasetConfig = toml::from_str(SAMPLE_TOML).expect("parse failed");
|
||||
assert_eq!(cfg.reader.sheet_mode, SheetMode::All);
|
||||
assert!(!cfg.reader.strip_blank_rows);
|
||||
assert_eq!(cfg.columns.ho_ten, 0);
|
||||
assert_eq!(cfg.columns.diem_thi, 3);
|
||||
assert!(!cfg.validation.require_numeric_sbd);
|
||||
assert!(cfg.validation.require_nonempty_name);
|
||||
assert_eq!(cfg.header.tokens.len(), 3);
|
||||
assert!(cfg.scores.contains_key("toan"));
|
||||
assert!(cfg.scores.contains_key("ngu_van"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn config_first_sheet_mode() {
|
||||
let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#);
|
||||
let cfg: DatasetConfig = toml::from_str(&toml_str).expect("parse failed");
|
||||
assert_eq!(cfg.reader.sheet_mode, SheetMode::First);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
use thiserror::Error;
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
pub enum BuildError {
|
||||
#[error("I/O error for {path}: {source}")]
|
||||
Io {
|
||||
path: String,
|
||||
#[source]
|
||||
source: std::io::Error,
|
||||
},
|
||||
|
||||
#[error("Calamine error for {path}: {source}")]
|
||||
Calamine {
|
||||
path: String,
|
||||
#[source]
|
||||
source: calamine::Error,
|
||||
},
|
||||
|
||||
#[error("SQLite error: {0}")]
|
||||
Sqlite(#[from] rusqlite::Error),
|
||||
|
||||
#[error("Config parse error: {0}")]
|
||||
Config(#[from] toml::de::Error),
|
||||
|
||||
#[error("Regex compile error for pattern '{pattern}': {source}")]
|
||||
Regex {
|
||||
pattern: String,
|
||||
#[source]
|
||||
source: regex::Error,
|
||||
},
|
||||
|
||||
#[error("Schema has no sheets in file: {0}")]
|
||||
NoSheets(String),
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
/// Public library interface for integration tests.
|
||||
/// The binary entry point is src/main.rs; this file re-exports the internal
|
||||
/// modules so tests/golden.rs can call them without going through the CLI.
|
||||
pub mod audit;
|
||||
pub mod config;
|
||||
pub mod error;
|
||||
pub mod reader;
|
||||
pub mod transform;
|
||||
pub mod writer;
|
||||
@@ -0,0 +1,193 @@
|
||||
/// xlsxread — Rust CLI replacing the SheetJS xlsx build scripts.
|
||||
///
|
||||
/// Subcommands:
|
||||
/// build — read .xls/.xlsx files → write SQLite DB
|
||||
/// audit — compare distinct SBD count from xlsx vs DB row count
|
||||
///
|
||||
/// Library modules are declared in lib.rs; main.rs only adds the CLI layer.
|
||||
mod cli;
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use clap::Parser;
|
||||
|
||||
use cli::{Cli, Cmd};
|
||||
use xlsxread::audit;
|
||||
use xlsxread::config::load_config;
|
||||
use xlsxread::reader::{is_all_blank, process_file};
|
||||
use xlsxread::transform::{validate_row, CompiledPatterns, SkipReason};
|
||||
use xlsxread::writer::{finish_db, insert_row, open_db, SCORE_FIELDS};
|
||||
|
||||
fn main() -> Result<()> {
|
||||
let cli = Cli::parse();
|
||||
|
||||
match cli.cmd {
|
||||
Cmd::Build {
|
||||
schema,
|
||||
input,
|
||||
output,
|
||||
} => {
|
||||
run_build(&schema, &input, &output)?;
|
||||
}
|
||||
Cmd::Audit { schema, input, db } => {
|
||||
let cfg = load_config(&schema)
|
||||
.with_context(|| format!("Failed to load config: {}", schema.display()))?;
|
||||
let result = audit::run_audit(&input, &db, &cfg).with_context(|| "Audit failed")?;
|
||||
audit::print_audit_report(&result);
|
||||
if !result.matched {
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Build subcommand
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn run_build(schema_path: &Path, input_dir: &Path, output_path: &Path) -> Result<()> {
|
||||
let cfg = load_config(schema_path)
|
||||
.with_context(|| format!("Failed to load config: {}", schema_path.display()))?;
|
||||
|
||||
// Compile score regexes once at startup
|
||||
let patterns =
|
||||
CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?;
|
||||
|
||||
// Collect input files (.xls and .xlsx), sorted for deterministic order
|
||||
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
|
||||
.with_context(|| format!("Cannot read input dir: {}", input_dir.display()))?
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| {
|
||||
let lower = e.to_lowercase();
|
||||
lower == "xls" || lower == "xlsx"
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let dataset_label = input_dir
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("data");
|
||||
|
||||
println!(
|
||||
"[build] {dataset_label}/ → {} ({} files)",
|
||||
output_path.display(),
|
||||
files.len()
|
||||
);
|
||||
|
||||
// Open (or recreate) DB and apply DDL
|
||||
let conn = open_db(output_path, &cfg)
|
||||
.with_context(|| format!("Failed to open DB: {}", output_path.display()))?;
|
||||
|
||||
let mut total_source_rows: u64 = 0;
|
||||
let mut total_skipped: u64 = 0;
|
||||
let mut total_errors: u64 = 0;
|
||||
|
||||
let is_old2 = dataset_label.contains("old2");
|
||||
let strip_blank = cfg.reader.strip_blank_rows;
|
||||
|
||||
// Single transaction over all files — mirrors the Node `db.transaction(() => { ... })()`
|
||||
conn.execute_batch("BEGIN")?;
|
||||
|
||||
for file in &files {
|
||||
let base = file
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("?")
|
||||
.to_owned();
|
||||
let mut file_rows: u64 = 0;
|
||||
let mut file_skipped: u64 = 0;
|
||||
let mut file_errors: u64 = 0;
|
||||
|
||||
let process_result = process_file(file, &cfg, |_sheet_idx, raw| {
|
||||
// data-old2: skip fully blank rows BEFORE counting sourceRows
|
||||
let all_blank = is_all_blank(raw);
|
||||
if strip_blank && all_blank {
|
||||
return;
|
||||
}
|
||||
|
||||
total_source_rows += 1;
|
||||
|
||||
let ho_ten = raw
|
||||
.get(cfg.columns.ho_ten)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let so_bao_danh = raw
|
||||
.get(cfg.columns.so_bao_danh)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
|
||||
match validate_row(
|
||||
&ho_ten,
|
||||
&so_bao_danh,
|
||||
&cfg.validation,
|
||||
strip_blank,
|
||||
all_blank,
|
||||
) {
|
||||
Err(SkipReason::BlankRow) => {
|
||||
// Already guarded above; won't reach here
|
||||
}
|
||||
Err(_) => {
|
||||
file_skipped += 1;
|
||||
return;
|
||||
}
|
||||
Ok(()) => {}
|
||||
}
|
||||
|
||||
let parsed = xlsxread::transform::transform_row(raw, &cfg, &patterns);
|
||||
|
||||
match insert_row(&conn, &cfg.insert.sql, &parsed, SCORE_FIELDS) {
|
||||
Ok(()) => {
|
||||
file_rows += 1;
|
||||
}
|
||||
Err(e) => {
|
||||
file_errors += 1;
|
||||
if total_errors + file_errors <= 5 {
|
||||
eprintln!(" [warn] {base}: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
match process_result {
|
||||
Ok(_) => {}
|
||||
Err(e) => {
|
||||
eprintln!(" [error] {base}: {e}");
|
||||
file_errors += 1;
|
||||
}
|
||||
}
|
||||
|
||||
total_skipped += file_skipped;
|
||||
total_errors += file_errors;
|
||||
|
||||
// Per-file row count line — mirrors `console.log(` ${base}: ${fileRows} rows`)`
|
||||
println!(" {base}: {file_rows} rows");
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT")?;
|
||||
|
||||
// VACUUM + stats output
|
||||
finish_db(
|
||||
&conn,
|
||||
output_path,
|
||||
total_source_rows,
|
||||
total_skipped,
|
||||
total_errors,
|
||||
dataset_label,
|
||||
files.len(),
|
||||
is_old2,
|
||||
)
|
||||
.with_context(|| "Failed to finalise DB")?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
/// Spreadsheet reader: wraps calamine to iterate rows across sheets.
|
||||
///
|
||||
/// Sheet selection mirrors the JS scripts:
|
||||
/// - sheet_mode = "all" → iterate every sheet (handles HCM/HN 65k overflow in data/)
|
||||
/// - sheet_mode = "first" → sheet 0 only (data-old/)
|
||||
///
|
||||
/// Header detection mirrors build-lib.js isHeaderRow:
|
||||
/// row[0].toUpperCase() in {"HO_TEN", "HỌ TÊN", "STT"}
|
||||
use std::path::Path;
|
||||
|
||||
use calamine::{open_workbook_auto, Data, Reader, Sheets};
|
||||
|
||||
use crate::config::{DatasetConfig, HeaderCfg, SheetMode};
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public row representation from calamine
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub type RawRow = Vec<Data>;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Header detection — mirrors build-lib.js isHeaderRow
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Returns true when the first cell (uppercased) matches one of the configured
|
||||
/// header tokens. Used to skip the header row on the first row of each sheet.
|
||||
pub fn is_header_row(row: &[Data], header_cfg: &HeaderCfg) -> bool {
|
||||
if row.len() < 3 {
|
||||
return false;
|
||||
}
|
||||
let first = row[0].to_string().trim().to_uppercase();
|
||||
header_cfg.tokens.iter().any(|t| t.to_uppercase() == first)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// All-blank row check (data-old2: strip_blank_rows)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn is_all_blank(row: &[Data]) -> bool {
|
||||
row.iter()
|
||||
.all(|c| matches!(c, Data::Empty) || c.to_string().trim().is_empty())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// File processor — yields all data rows from the file
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process one spreadsheet file, calling `on_row` for each data row.
|
||||
///
|
||||
/// `on_row` receives `(sheet_index, row_index_in_sheet, raw_row)` where
|
||||
/// `row_index_in_sheet` is 0-based AFTER the header has been consumed.
|
||||
/// Returns `(sheets_seen, total_rows_yielded)`.
|
||||
pub fn process_file<F>(
|
||||
path: &Path,
|
||||
cfg: &DatasetConfig,
|
||||
mut on_row: F,
|
||||
) -> Result<(usize, usize), BuildError>
|
||||
where
|
||||
F: FnMut(usize, &RawRow),
|
||||
{
|
||||
let path_str = path.display().to_string();
|
||||
|
||||
// calamine::open_workbook_auto dispatches on file extension
|
||||
let mut workbook: Sheets<_> = open_workbook_auto(path).map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let sheet_names: Vec<String> = workbook.sheet_names().to_vec();
|
||||
if sheet_names.is_empty() {
|
||||
return Err(BuildError::NoSheets(path_str.clone()));
|
||||
}
|
||||
|
||||
// Sheet selection per config
|
||||
let sheets_to_read: Vec<String> = match cfg.reader.sheet_mode {
|
||||
SheetMode::All => sheet_names.clone(),
|
||||
SheetMode::First => vec![sheet_names[0].clone()],
|
||||
};
|
||||
|
||||
let mut total_rows = 0usize;
|
||||
|
||||
for (sheet_idx, sheet_name) in sheets_to_read.iter().enumerate() {
|
||||
let range = workbook
|
||||
.worksheet_range(sheet_name)
|
||||
.map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let mut first_row = true;
|
||||
|
||||
for raw in range.rows() {
|
||||
let row: RawRow = raw.to_vec();
|
||||
|
||||
// Skip header row on first row of each sheet (matches JS: `if (i === 0 && isHeaderRow(...))`)
|
||||
if first_row {
|
||||
first_row = false;
|
||||
if is_header_row(&row, &cfg.header) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
on_row(sheet_idx, &row);
|
||||
total_rows += 1;
|
||||
}
|
||||
}
|
||||
|
||||
Ok((sheets_to_read.len(), total_rows))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::config::HeaderCfg;
|
||||
|
||||
fn hdr(tokens: &[&str]) -> HeaderCfg {
|
||||
HeaderCfg {
|
||||
tokens: tokens.iter().map(|s| s.to_string()).collect(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_ho_ten() {
|
||||
let row = vec![
|
||||
Data::String("HO_TEN".into()),
|
||||
Data::String("NGAY_SINH".into()),
|
||||
Data::String("SBD".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_stt() {
|
||||
let row = vec![
|
||||
Data::String("STT".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_ho_ten_unicode() {
|
||||
let row = vec![
|
||||
Data::String("HỌ TÊN".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_rejects_data_row() {
|
||||
let row = vec![
|
||||
Data::String("Nguyen Van A".into()),
|
||||
Data::String("01/01/2000".into()),
|
||||
Data::String("12345678".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(!is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_rejects_short_row() {
|
||||
let row = vec![Data::String("HO_TEN".into()), Data::Empty];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(!is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_case_insensitive() {
|
||||
let row = vec![
|
||||
Data::String("ho_ten".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn blank_row_detection() {
|
||||
let row = vec![Data::Empty, Data::Empty, Data::String("".into())];
|
||||
assert!(is_all_blank(&row));
|
||||
|
||||
let row2 = vec![Data::String("Nguyen".into()), Data::Empty, Data::Empty];
|
||||
assert!(!is_all_blank(&row2));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,398 @@
|
||||
/// Row transformation: ascii normalisation, score regex parsing, validation.
|
||||
///
|
||||
/// `to_ascii` replicates build-lib.js `toAscii` exactly:
|
||||
/// str.normalize("NFD").replace(/[̀-ͯ]/g,"").replace(/đ/gi,"d").toLowerCase()
|
||||
use std::collections::HashMap;
|
||||
|
||||
use regex::Regex;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
use crate::config::{DatasetConfig, ValidationCfg};
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Compiled score patterns (built once at startup from config)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct CompiledPatterns {
|
||||
/// Ordered list so INSERT column order is deterministic
|
||||
pub patterns: Vec<(String, Regex)>,
|
||||
}
|
||||
|
||||
impl CompiledPatterns {
|
||||
pub fn new(scores: &HashMap<String, String>) -> Result<Self, BuildError> {
|
||||
let mut patterns = Vec::with_capacity(scores.len());
|
||||
for (field, src) in scores {
|
||||
let re = Regex::new(src).map_err(|e| BuildError::Regex {
|
||||
pattern: src.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
patterns.push((field.clone(), re));
|
||||
}
|
||||
// Sort for deterministic order across HashMap iteration
|
||||
patterns.sort_by(|a, b| a.0.cmp(&b.0));
|
||||
Ok(Self { patterns })
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// to_ascii — must be byte-for-byte equivalent to build-lib.js toAscii
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Normalise a Vietnamese name to an ASCII slug.
|
||||
///
|
||||
/// Algorithm mirrors the JavaScript `toAscii` in build-lib.js:
|
||||
/// 1. NFD decompose (splits base + combining diacritics)
|
||||
/// 2. Drop all Unicode combining marks (U+0300–U+036F)
|
||||
/// 3. Replace đ/Đ with d (NFD does not decompose đ)
|
||||
/// 4. Lowercase
|
||||
pub fn to_ascii(s: &str) -> String {
|
||||
// Step 1 + 2: NFD then filter out combining marks (Unicode category M)
|
||||
let decomposed: String = s
|
||||
.nfd()
|
||||
.filter(|c| !('\u{0300}'..='\u{036f}').contains(c))
|
||||
.collect();
|
||||
|
||||
// Step 3: đ/Đ are not decomposed by NFD — replace explicitly
|
||||
let replaced = decomposed.replace(['đ', 'Đ'], "d");
|
||||
|
||||
// Step 4: lowercase
|
||||
replaced.to_lowercase()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Parsed row ready for DB insert
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct ParsedRow {
|
||||
pub so_bao_danh: String,
|
||||
pub ho_ten: String,
|
||||
pub ho_ten_ascii: String,
|
||||
pub ngay_sinh: Option<String>,
|
||||
/// Subject field → float value; absent subjects not in map → NULL
|
||||
pub scores: HashMap<String, f64>,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Row validation — mirrors the per-script skip logic
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Returns `None` when the row should be skipped entirely (before sourceRows counter).
|
||||
/// Returns `Some(reason)` when the row should be counted as sourceRows but skipped.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub enum SkipReason {
|
||||
/// Row is fully blank (data-old2 only, before sourceRows counter)
|
||||
BlankRow,
|
||||
/// soBaoDanh or hoTen empty/missing
|
||||
EmptyField,
|
||||
/// soBaoDanh contains non-digit characters (data-old / data-old2 guard)
|
||||
NonNumericSbd,
|
||||
}
|
||||
|
||||
/// Validates a raw cell slice against the dataset's `ValidationCfg`.
|
||||
/// Returns `Ok(())` on pass, `Err(SkipReason)` on fail.
|
||||
pub fn validate_row(
|
||||
ho_ten: &str,
|
||||
so_bao_danh: &str,
|
||||
cfg: &ValidationCfg,
|
||||
strip_blank_rows: bool,
|
||||
all_blank: bool,
|
||||
) -> Result<(), SkipReason> {
|
||||
// data-old2: skip fully blank rows BEFORE counting sourceRows
|
||||
if strip_blank_rows && all_blank {
|
||||
return Err(SkipReason::BlankRow);
|
||||
}
|
||||
|
||||
if cfg.require_nonempty_sbd && so_bao_danh.is_empty() {
|
||||
return Err(SkipReason::EmptyField);
|
||||
}
|
||||
if cfg.require_nonempty_name && ho_ten.is_empty() {
|
||||
return Err(SkipReason::EmptyField);
|
||||
}
|
||||
if cfg.require_numeric_sbd && !so_bao_danh.chars().all(|c| c.is_ascii_digit()) {
|
||||
return Err(SkipReason::NonNumericSbd);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Score parsing — mirrors build-lib.js parseScores
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Parse a DIEM_THI cell string and extract matching subject scores.
|
||||
pub fn parse_scores(diem_thi: &str, patterns: &CompiledPatterns) -> HashMap<String, f64> {
|
||||
let mut out = HashMap::new();
|
||||
for (field, re) in &patterns.patterns {
|
||||
if let Some(caps) = re.captures(diem_thi) {
|
||||
if let Some(m) = caps.get(1) {
|
||||
if let Ok(v) = m.as_str().parse::<f64>() {
|
||||
if v.is_finite() {
|
||||
out.insert(field.clone(), v);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Full row transform
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Extract and transform one spreadsheet row into a `ParsedRow`.
|
||||
/// `raw` is the full cell slice; column indices come from `cfg.columns`.
|
||||
pub fn transform_row(
|
||||
raw: &[calamine::Data],
|
||||
cfg: &DatasetConfig,
|
||||
patterns: &CompiledPatterns,
|
||||
) -> ParsedRow {
|
||||
let get = |idx: usize| -> String {
|
||||
raw.get(idx)
|
||||
.map(|cell| cell.to_string().trim().to_owned())
|
||||
.unwrap_or_default()
|
||||
};
|
||||
|
||||
let ho_ten = get(cfg.columns.ho_ten);
|
||||
let ngay_sinh = get(cfg.columns.ngay_sinh);
|
||||
let so_bao_danh = get(cfg.columns.so_bao_danh);
|
||||
let diem_thi = raw
|
||||
.get(cfg.columns.diem_thi)
|
||||
.map(|c| c.to_string())
|
||||
.unwrap_or_default();
|
||||
|
||||
let ho_ten_ascii = to_ascii(&ho_ten);
|
||||
let scores = parse_scores(&diem_thi, patterns);
|
||||
let ngay_sinh_opt = if ngay_sinh.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(ngay_sinh)
|
||||
};
|
||||
|
||||
ParsedRow {
|
||||
so_bao_danh,
|
||||
ho_ten,
|
||||
ho_ten_ascii,
|
||||
ngay_sinh: ngay_sinh_opt,
|
||||
scores,
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests — 20 cases for to_ascii (real Vietnamese names)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
// Helper: assert to_ascii(input) == expected
|
||||
fn check(input: &str, expected: &str) {
|
||||
assert_eq!(
|
||||
to_ascii(input),
|
||||
expected,
|
||||
"to_ascii({input:?}) expected {expected:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_plain_latin() {
|
||||
check("Nguyen Van A", "nguyen van a");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_hoa() {
|
||||
check("Nguyễn Thị Hoa", "nguyen thi hoa");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_tran_van_duc() {
|
||||
// đ/Đ replacement
|
||||
check("Trần Văn Đức", "tran van duc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_le_thi_my_duyen() {
|
||||
check("Lê Thị Mỹ Duyên", "le thi my duyen");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_pham_thi_lan() {
|
||||
check("Phạm Thị Lan", "pham thi lan");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_bui_thi_thu() {
|
||||
check("Bùi Thị Thu", "bui thi thu");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_hoang_van_truong() {
|
||||
check("Hoàng Văn Trường", "hoang van truong");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_do_thi_ngan() {
|
||||
// Đ uppercase at start
|
||||
check("Đỗ Thị Ngân", "do thi ngan");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_van_khanh() {
|
||||
check("Nguyễn Văn Khánh", "nguyen van khanh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_trinh_thi_bich_ngoc() {
|
||||
check("Trịnh Thị Bích Ngọc", "trinh thi bich ngoc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_vu_thi_dieu() {
|
||||
// ề = e + combining grave + combining circumflex (after NFD)
|
||||
check("Vũ Thị Diệu", "vu thi dieu");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_tuong_vi() {
|
||||
check("Nguyễn Thị Tường Vi", "nguyen thi tuong vi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_lowercase_d_stroke() {
|
||||
// Lowercase đ → d
|
||||
check("đặng thị hằng", "dang thi hang");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_uppercase_d_stroke() {
|
||||
check("ĐẶNG THỊ HẰNG", "dang thi hang");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_mixed_case() {
|
||||
check("NGUYỄN VĂN AN", "nguyen van an");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_tran_thi_kim_anh() {
|
||||
check("Trần Thị Kim Anh", "tran thi kim anh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_phuong_thao() {
|
||||
check("Nguyễn Thị Phương Thảo", "nguyen thi phuong thao");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_le_van_long() {
|
||||
check("Lê Văn Long", "le van long");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_vo_thi_xuan_mai() {
|
||||
check("Võ Thị Xuân Mai", "vo thi xuan mai");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_empty_string() {
|
||||
check("", "");
|
||||
}
|
||||
|
||||
// --- Score parsing tests ---
|
||||
|
||||
fn make_patterns() -> CompiledPatterns {
|
||||
let mut map = HashMap::new();
|
||||
map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("vat_ly".into(), r"Vật lí:\s*(\d+(?:\.\d+)?)".into());
|
||||
CompiledPatterns::new(&map).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_single() {
|
||||
let p = make_patterns();
|
||||
let s = "Toán: 8.5";
|
||||
let scores = parse_scores(s, &p);
|
||||
assert_eq!(scores.get("toan"), Some(&8.5));
|
||||
assert!(scores.get("ngu_van").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_multiple() {
|
||||
let p = make_patterns();
|
||||
let s = "Toán: 7.25 Ngữ văn: 6.0 Vật lí: 9";
|
||||
let scores = parse_scores(s, &p);
|
||||
assert_eq!(scores.get("toan"), Some(&7.25));
|
||||
assert_eq!(scores.get("ngu_van"), Some(&6.0));
|
||||
assert_eq!(scores.get("vat_ly"), Some(&9.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_empty_cell() {
|
||||
let p = make_patterns();
|
||||
let scores = parse_scores("", &p);
|
||||
assert!(scores.is_empty());
|
||||
}
|
||||
|
||||
// --- Validation tests ---
|
||||
|
||||
fn default_validation() -> ValidationCfg {
|
||||
ValidationCfg {
|
||||
require_numeric_sbd: false,
|
||||
require_nonempty_name: true,
|
||||
require_nonempty_sbd: true,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_ok() {
|
||||
let v = default_validation();
|
||||
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_empty_sbd() {
|
||||
let v = default_validation();
|
||||
assert_eq!(
|
||||
validate_row("Nguyen Van A", "", &v, false, false),
|
||||
Err(SkipReason::EmptyField)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_empty_name() {
|
||||
let v = default_validation();
|
||||
assert_eq!(
|
||||
validate_row("", "12345678", &v, false, false),
|
||||
Err(SkipReason::EmptyField)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_non_numeric_sbd_rejected() {
|
||||
let mut v = default_validation();
|
||||
v.require_numeric_sbd = true;
|
||||
assert_eq!(
|
||||
validate_row("Nguyen Van A", "12AB5678", &v, false, false),
|
||||
Err(SkipReason::NonNumericSbd)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_numeric_sbd_accepted() {
|
||||
let mut v = default_validation();
|
||||
v.require_numeric_sbd = true;
|
||||
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_blank_row_skipped() {
|
||||
let v = default_validation();
|
||||
// strip_blank_rows=true AND all_blank=true → BlankRow
|
||||
assert_eq!(
|
||||
validate_row("", "", &v, true, true),
|
||||
Err(SkipReason::BlankRow)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
/// SQLite writer: DDL setup, batched INSERT OR REPLACE, VACUUM, stats output.
|
||||
///
|
||||
/// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js.
|
||||
/// Stats output lines match the JS stdout exactly so existing CI log-greps still work.
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
use rusqlite::{params_from_iter, Connection, ToSql};
|
||||
|
||||
use crate::config::DatasetConfig;
|
||||
use crate::error::BuildError;
|
||||
use crate::transform::ParsedRow;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// DB initialisation — mirrors build-lib.js createDb (delete + recreate)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Open (or recreate) the output SQLite database, execute the DDL from config,
|
||||
/// and return the open connection ready for inserts.
|
||||
pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result<Connection, BuildError> {
|
||||
// Mirror Node behaviour: delete existing file before creating (build-lib.js:54)
|
||||
if db_path.exists() {
|
||||
fs::remove_file(db_path).map_err(|e| BuildError::Io {
|
||||
path: db_path.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
}
|
||||
|
||||
// Ensure parent directory exists
|
||||
if let Some(parent) = db_path.parent() {
|
||||
if !parent.as_os_str().is_empty() {
|
||||
fs::create_dir_all(parent).map_err(|e| BuildError::Io {
|
||||
path: parent.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
}
|
||||
}
|
||||
|
||||
let conn = Connection::open(db_path)?;
|
||||
conn.execute_batch(&cfg.schema.ddl)?;
|
||||
Ok(conn)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Ordered score field list — canonical INSERT column order from build-lib.js
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Fixed subject column order matching the INSERT statement in every config.
|
||||
/// NULL is bound for any subject not present in a given row's score map.
|
||||
pub const SCORE_FIELDS: &[&str] = &[
|
||||
"toan",
|
||||
"ngu_van",
|
||||
"vat_ly",
|
||||
"hoa_hoc",
|
||||
"sinh_hoc",
|
||||
"khtn",
|
||||
"lich_su",
|
||||
"dia_ly",
|
||||
"gdcd",
|
||||
"khxh",
|
||||
"tieng_anh",
|
||||
"tieng_phap",
|
||||
"tieng_nga",
|
||||
"tieng_trung",
|
||||
];
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Insert a single parsed row inside an active transaction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Bind all fields from `row` into the prepared statement and execute it.
|
||||
/// `score_fields` should be the ordered list of subject columns the INSERT expects.
|
||||
pub fn insert_row(
|
||||
conn: &Connection,
|
||||
sql: &str,
|
||||
row: &ParsedRow,
|
||||
score_fields: &[&str],
|
||||
) -> Result<(), BuildError> {
|
||||
// Build positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, <scores...>
|
||||
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(4 + score_fields.len());
|
||||
params.push(Box::new(row.so_bao_danh.clone()));
|
||||
params.push(Box::new(row.ho_ten.clone()));
|
||||
params.push(Box::new(row.ho_ten_ascii.clone()));
|
||||
params.push(Box::new(row.ngay_sinh.clone()));
|
||||
|
||||
for field in score_fields {
|
||||
let val: Option<f64> = row.scores.get(*field).copied();
|
||||
params.push(Box::new(val));
|
||||
}
|
||||
|
||||
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Post-build: VACUUM + stats output
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Run VACUUM and print statistics lines that mirror the Node scripts' stdout.
|
||||
/// The exact prefix tokens ("Source data rows", "DB rows", "Size:") are preserved
|
||||
/// so any log-grep in the deploy pipeline keeps working.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn finish_db(
|
||||
conn: &Connection,
|
||||
db_path: &Path,
|
||||
source_rows: u64,
|
||||
skipped: u64,
|
||||
errors: u64,
|
||||
dataset_label: &str, // e.g. "data/" or "data-old2/"
|
||||
_file_count: usize,
|
||||
is_old2: bool, // data-old2 uses different label for the skipped line
|
||||
) -> Result<(), BuildError> {
|
||||
conn.execute_batch("VACUUM")?;
|
||||
|
||||
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
|
||||
|
||||
let insertable = source_rows - skipped;
|
||||
|
||||
// Mirror exact JS stdout format for each dataset variant
|
||||
println!();
|
||||
if is_old2 {
|
||||
println!("Source non-blank data rows: {source_rows}");
|
||||
println!(" skipped (empty/non-numeric SBD): {skipped}");
|
||||
} else {
|
||||
println!("Source data rows (post-header): {source_rows}");
|
||||
if dataset_label.contains("old") {
|
||||
println!(" skipped (empty/non-numeric SBD): {skipped}");
|
||||
} else {
|
||||
println!(" skipped (empty/invalid): {skipped}");
|
||||
}
|
||||
}
|
||||
println!(" insertable: {insertable}");
|
||||
println!(" insert errors: {errors}");
|
||||
println!("DB rows (distinct SBD): {db_count}");
|
||||
|
||||
// Audit gap comment (mirrors build-database.js:80-83 for data/ only)
|
||||
if !dataset_label.contains("old") && errors == 0 {
|
||||
let gap = insertable as i64 - db_count;
|
||||
if gap == 0 {
|
||||
println!("Audit: OK — every source row made it in.");
|
||||
} else {
|
||||
println!("Audit: {gap} row(s) collapsed (duplicate SBDs overwriting).");
|
||||
}
|
||||
}
|
||||
|
||||
let sz = fs::metadata(db_path).map(|m| m.len()).unwrap_or(0);
|
||||
println!("Size: {:.1} MB", sz as f64 / 1024.0 / 1024.0);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Test Fixtures
|
||||
|
||||
Anonymised `.xlsx` files for integration testing. All student PII has been replaced:
|
||||
|
||||
- `ho_ten` replaced with `Nguyen Van Test NNN` / `Tran Thi Test NNN` patterns
|
||||
- `so_bao_danh` replaced with sequential synthetic numbers (e.g. `10000001`)
|
||||
- `ngay_sinh` replaced with fixed synthetic dates
|
||||
- Scores are realistic random values in the 0–10 range
|
||||
|
||||
Files:
|
||||
|
||||
- `province-100.xlsx` — 100-row single-sheet file (simulates a normal province)
|
||||
- `hcm-overflow.xlsx` — 2-sheet file (200 rows Sheet1 + 200 rows Sheet2, simulating HCM overflow)
|
||||
- `province-numeric-sbd.xlsx` — 100 rows with strictly numeric SBDs (for data-old variant)
|
||||
|
||||
These files are generated by `tests/golden.rs` `generate_fixtures()` if they do not already exist
|
||||
on disk. The generator is pure Rust (uses the `zip` crate already pulled in via calamine).
|
||||
No external Python or Node tooling required for unit/integration tests.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,690 @@
|
||||
/// Stage 5 golden tests — integration tests using anonymised fixture files.
|
||||
///
|
||||
/// Fixture files are generated in-process via raw OOXML + zip if they do not
|
||||
/// already exist on disk. No external Python or Node tooling required for the
|
||||
/// Rust-side tests. The Node golden comparison is marked #[ignore] when pnpm
|
||||
/// is not in PATH.
|
||||
use std::io::Write as IoWrite;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Minimal OOXML xlsx generator
|
||||
//
|
||||
// Produces a valid .xlsx that calamine can read. Only uses the `zip` crate
|
||||
// which is already pulled in as a transitive dependency of calamine.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// One row of cell data for a fixture sheet.
|
||||
struct XlsxRow {
|
||||
values: Vec<String>,
|
||||
}
|
||||
|
||||
/// Write a minimal .xlsx to `path` with the given sheets.
|
||||
/// `sheets`: Vec<(sheet_name, rows)> where rows[0] is the header.
|
||||
fn write_xlsx(path: &Path, sheets: &[(String, Vec<XlsxRow>)]) {
|
||||
use zip::{write::SimpleFileOptions, ZipWriter};
|
||||
|
||||
let file = std::fs::File::create(path).expect("create fixture xlsx");
|
||||
let mut zip = ZipWriter::new(file);
|
||||
let opts = SimpleFileOptions::default();
|
||||
|
||||
// [Content_Types].xml
|
||||
let mut content_types = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">
|
||||
<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>
|
||||
<Default Extension="xml" ContentType="application/xml"/>
|
||||
<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>
|
||||
"#,
|
||||
);
|
||||
for (i, _) in sheets.iter().enumerate() {
|
||||
content_types.push_str(&format!(
|
||||
r#" <Override PartName="/xl/worksheets/sheet{}.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>
|
||||
"#,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
content_types.push_str("</Types>");
|
||||
zip.start_file("[Content_Types].xml", opts).unwrap();
|
||||
zip.write_all(content_types.as_bytes()).unwrap();
|
||||
|
||||
// _rels/.rels
|
||||
zip.start_file("_rels/.rels", opts).unwrap();
|
||||
zip.write_all(
|
||||
br#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
|
||||
<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>
|
||||
</Relationships>"#,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// xl/_rels/workbook.xml.rels
|
||||
let mut wb_rels = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
|
||||
"#,
|
||||
);
|
||||
for (i, _) in sheets.iter().enumerate() {
|
||||
wb_rels.push_str(&format!(
|
||||
r#" <Relationship Id="rId{}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet{}.xml"/>
|
||||
"#,
|
||||
i + 1,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
wb_rels.push_str("</Relationships>");
|
||||
zip.start_file("xl/_rels/workbook.xml.rels", opts).unwrap();
|
||||
zip.write_all(wb_rels.as_bytes()).unwrap();
|
||||
|
||||
// xl/workbook.xml
|
||||
let mut wb = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main"
|
||||
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">
|
||||
<sheets>
|
||||
"#,
|
||||
);
|
||||
for (i, (name, _)) in sheets.iter().enumerate() {
|
||||
let escaped = xml_escape(name);
|
||||
wb.push_str(&format!(
|
||||
r#" <sheet name="{}" sheetId="{}" r:id="rId{}"/>
|
||||
"#,
|
||||
escaped,
|
||||
i + 1,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
wb.push_str(" </sheets>\n</workbook>");
|
||||
zip.start_file("xl/workbook.xml", opts).unwrap();
|
||||
zip.write_all(wb.as_bytes()).unwrap();
|
||||
|
||||
// xl/worksheets/sheetN.xml
|
||||
for (i, (_, rows)) in sheets.iter().enumerate() {
|
||||
let mut ws = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
|
||||
<sheetData>
|
||||
"#,
|
||||
);
|
||||
for (row_idx, row) in rows.iter().enumerate() {
|
||||
ws.push_str(&format!(
|
||||
r#" <row r="{}">
|
||||
"#,
|
||||
row_idx + 1
|
||||
));
|
||||
for (col_idx, val) in row.values.iter().enumerate() {
|
||||
let col_letter = col_letter(col_idx);
|
||||
let cell_ref = format!("{}{}", col_letter, row_idx + 1);
|
||||
let escaped = xml_escape(val);
|
||||
ws.push_str(&format!(
|
||||
r#" <c r="{}" t="inlineStr"><is><t>{}</t></is></c>
|
||||
"#,
|
||||
cell_ref, escaped
|
||||
));
|
||||
}
|
||||
ws.push_str(" </row>\n");
|
||||
}
|
||||
ws.push_str(" </sheetData>\n</worksheet>");
|
||||
zip.start_file(&format!("xl/worksheets/sheet{}.xml", i + 1), opts)
|
||||
.unwrap();
|
||||
zip.write_all(ws.as_bytes()).unwrap();
|
||||
}
|
||||
|
||||
zip.finish().unwrap();
|
||||
}
|
||||
|
||||
fn col_letter(idx: usize) -> &'static str {
|
||||
const LETTERS: &[&str] = &[
|
||||
"A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R",
|
||||
"S", "T", "U", "V", "W", "X", "Y", "Z",
|
||||
];
|
||||
LETTERS[idx % 26]
|
||||
}
|
||||
|
||||
fn xml_escape(s: &str) -> String {
|
||||
s.replace('&', "&")
|
||||
.replace('<', "<")
|
||||
.replace('>', ">")
|
||||
.replace('"', """)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture data builders
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn header_row() -> XlsxRow {
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"HO_TEN".into(),
|
||||
"NGAY_SINH".into(),
|
||||
"SO_BAO_DANH".into(),
|
||||
"DIEM_THI".into(),
|
||||
],
|
||||
}
|
||||
}
|
||||
|
||||
fn data_row(idx: usize, scores: &str) -> XlsxRow {
|
||||
// Anonymised: name uses sequential pattern, SBD is purely synthetic
|
||||
let name = if idx % 2 == 0 {
|
||||
format!("Nguyen Van Test {:03}", idx)
|
||||
} else {
|
||||
format!("Tran Thi Test {:03}", idx)
|
||||
};
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
name,
|
||||
format!("15/0{}/{}", (idx % 9) + 1, 1999 + (idx % 5)),
|
||||
format!("1000{:04}", idx),
|
||||
scores.to_owned(),
|
||||
],
|
||||
}
|
||||
}
|
||||
|
||||
fn sample_scores(idx: usize) -> String {
|
||||
// Realistic scores in 0–10 range, varies by idx
|
||||
let toan = 4.0 + (idx % 60) as f64 / 10.0;
|
||||
let van = 3.5 + (idx % 65) as f64 / 10.0;
|
||||
format!("Toán: {toan:.1} Ngữ văn: {van:.1} Tiếng Anh: 7.5")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture file paths
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn fixtures_dir() -> PathBuf {
|
||||
// tests/fixtures/ relative to the crate root
|
||||
let mut p = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
|
||||
p.push("tests");
|
||||
p.push("fixtures");
|
||||
p
|
||||
}
|
||||
|
||||
fn province_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("province-100.xlsx")
|
||||
}
|
||||
|
||||
fn hcm_overflow_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("hcm-overflow.xlsx")
|
||||
}
|
||||
|
||||
fn numeric_sbd_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("province-numeric-sbd.xlsx")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture generation — called once per test run if files missing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn ensure_fixtures() {
|
||||
let dir = fixtures_dir();
|
||||
std::fs::create_dir_all(&dir).expect("create fixtures dir");
|
||||
|
||||
// province-100.xlsx — 100 data rows, single sheet, with header
|
||||
if !province_fixture_path().exists() {
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..100 {
|
||||
rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(&province_fixture_path(), &[("Sheet1".to_owned(), rows)]);
|
||||
}
|
||||
|
||||
// hcm-overflow.xlsx — 2 sheets × 200 rows each (no header on sheet 2)
|
||||
if !hcm_overflow_fixture_path().exists() {
|
||||
let mut sheet1 = vec![header_row()];
|
||||
for i in 0..200 {
|
||||
sheet1.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
// Sheet2: continuation rows, no header row (as in real HCM overflow)
|
||||
let mut sheet2 = Vec::new();
|
||||
for i in 200..400 {
|
||||
sheet2.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&hcm_overflow_fixture_path(),
|
||||
&[("Sheet1".to_owned(), sheet1), ("Sheet2".to_owned(), sheet2)],
|
||||
);
|
||||
}
|
||||
|
||||
// province-numeric-sbd.xlsx — strictly numeric SBDs for data-old config
|
||||
if !numeric_sbd_fixture_path().exists() {
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..100 {
|
||||
rows.push(XlsxRow {
|
||||
values: vec![
|
||||
format!("Nguyen Van Test {:03}", i),
|
||||
"01/01/2000".to_owned(),
|
||||
format!("{:08}", 20000000 + i), // pure digits
|
||||
sample_scores(i),
|
||||
],
|
||||
});
|
||||
}
|
||||
write_xlsx(&numeric_sbd_fixture_path(), &[("Sheet1".to_owned(), rows)]);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Config helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_data_config() -> xlsxread::config::DatasetConfig {
|
||||
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("configs")
|
||||
.join("thptqg2017-data.toml");
|
||||
xlsxread::config::load_config(&cfg_path).expect("load data config")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Integration tests — pure Rust, no Node dependency
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn province_100_builds_100_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(count, 100, "expected 100 rows from province-100 fixture");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hcm_overflow_builds_400_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 400,
|
||||
"expected 400 rows (200 × 2 sheets) from hcm-overflow fixture"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn data_old_first_sheet_only_100_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
// Use the overflow file but with data-old config (first sheet only → 200 rows)
|
||||
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
|
||||
|
||||
// data-old: sheet_mode=first → only 200 rows from sheet1; but SBDs "1000NNNN" are
|
||||
// all digits so all pass the numeric guard
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 200,
|
||||
"data-old config should read only first sheet (200 rows)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_sbd_guard_rejects_non_numeric() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
// Write a fixture with one non-numeric SBD mixed in
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..10 {
|
||||
rows.push(XlsxRow {
|
||||
values: vec![
|
||||
format!("Test {:03}", i),
|
||||
"01/01/2000".to_owned(),
|
||||
if i == 5 {
|
||||
"ABC123".to_owned()
|
||||
} else {
|
||||
format!("{:08}", 20000000 + i)
|
||||
},
|
||||
sample_scores(i),
|
||||
],
|
||||
});
|
||||
}
|
||||
let mixed_path = fixture_dir.join("mixed.xlsx");
|
||||
write_xlsx(&mixed_path, &[("Sheet1".to_owned(), rows)]);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
|
||||
|
||||
// Row i=5 has non-numeric SBD → rejected by data-old config
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 9,
|
||||
"non-numeric SBD row should be skipped by data-old config"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scores_parsed_correctly_into_db() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
let rows = vec![
|
||||
header_row(),
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"Nguyen Van Test 001".to_owned(),
|
||||
"01/01/2000".to_owned(),
|
||||
"10000001".to_owned(),
|
||||
"Toán: 8.5 Ngữ văn: 7.0 Tiếng Anh: 9.25".to_owned(),
|
||||
],
|
||||
},
|
||||
];
|
||||
write_xlsx(
|
||||
&fixture_dir.join("one.xlsx"),
|
||||
&[("Sheet1".to_owned(), rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let conn = rusqlite::Connection::open(&db_path).unwrap();
|
||||
let (toan, van, anh): (f64, f64, f64) = conn
|
||||
.query_row(
|
||||
"SELECT toan, ngu_van, tieng_anh FROM student WHERE so_bao_danh = '10000001'",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
|
||||
)
|
||||
.expect("row not found");
|
||||
assert!((toan - 8.5).abs() < 1e-9);
|
||||
assert!((van - 7.0).abs() < 1e-9);
|
||||
assert!((anh - 9.25).abs() < 1e-9);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn to_ascii_stored_correctly() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
let rows = vec![
|
||||
header_row(),
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"Nguyễn Văn Đức".to_owned(),
|
||||
"".to_owned(),
|
||||
"20000001".to_owned(),
|
||||
"".to_owned(),
|
||||
],
|
||||
},
|
||||
];
|
||||
write_xlsx(
|
||||
&fixture_dir.join("one.xlsx"),
|
||||
&[("Sheet1".to_owned(), rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let conn = rusqlite::Connection::open(&db_path).unwrap();
|
||||
let ascii: String = conn
|
||||
.query_row(
|
||||
"SELECT ho_ten_ascii FROM student WHERE so_bao_danh = '20000001'",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.expect("row not found");
|
||||
assert_eq!(ascii, "nguyen van duc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn audit_subcommand_matches_after_build() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
// audit should match (100 distinct SBDs in xlsx == 100 rows in DB)
|
||||
let cfg = make_data_config();
|
||||
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
|
||||
assert!(result.matched, "audit should match after build");
|
||||
assert_eq!(result.distinct_sbds, 100);
|
||||
assert_eq!(result.db_count, 100);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn audit_subcommand_mismatch_detected() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
// Write 10 rows to xlsx but build DB from only 5 rows
|
||||
let mut all_rows = vec![header_row()];
|
||||
for i in 0..10 {
|
||||
all_rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&fixture_dir.join("all.xlsx"),
|
||||
&[("Sheet1".to_owned(), all_rows)],
|
||||
);
|
||||
|
||||
// Build DB with only first 5 rows in a different file
|
||||
let build_dir = dir.join("build_input");
|
||||
std::fs::create_dir_all(&build_dir).unwrap();
|
||||
let mut five_rows = vec![header_row()];
|
||||
for i in 0..5 {
|
||||
five_rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&build_dir.join("five.xlsx"),
|
||||
&[("Sheet1".to_owned(), five_rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&build_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
// audit against fixture_dir (10 xlsx rows) but DB has 5 rows → mismatch
|
||||
let cfg = make_data_config();
|
||||
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
|
||||
assert!(!result.matched, "audit should not match (10 xlsx vs 5 db)");
|
||||
assert_eq!(result.distinct_sbds, 10);
|
||||
assert_eq!(result.db_count, 5);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Golden test: compare Rust DB vs Node DB on identical fixture
|
||||
// Marked #[ignore] when pnpm / node is not in PATH — CI installs them first.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn golden_rust_matches_node_db() {
|
||||
// This test requires: pnpm, node, and the thptqg2017 package to be installed
|
||||
// Run with: cargo test -- --ignored golden_rust_matches_node_db
|
||||
let which_pnpm = std::process::Command::new("which")
|
||||
.arg("pnpm")
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false);
|
||||
if !which_pnpm {
|
||||
eprintln!("pnpm not in PATH — skipping golden test");
|
||||
return;
|
||||
}
|
||||
|
||||
let dir = tempdir();
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
// Build with Rust
|
||||
let rust_db = dir.join("rust.db");
|
||||
run_build_cmd(&fixture_dir, &rust_db, "thptqg2017-data.toml");
|
||||
|
||||
// Build with Node (run build-database.js with DATA_DIR / DB_PATH overrides)
|
||||
// Node script reads env-vars via a thin wrapper — see scripts/build-database.js
|
||||
// For now: diff via SELECT * ORDER BY so_bao_danh
|
||||
let node_db = dir.join("node.db");
|
||||
let status = std::process::Command::new("pnpm")
|
||||
.args(["exec", "node", "scripts/build-database.js"])
|
||||
.env("OVERRIDE_SRC_DIR", fixture_dir.to_str().unwrap())
|
||||
.env("OVERRIDE_DB_PATH", node_db.to_str().unwrap())
|
||||
.current_dir(
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.parent()
|
||||
.unwrap()
|
||||
.parent()
|
||||
.unwrap(),
|
||||
)
|
||||
.status()
|
||||
.expect("failed to run node build script");
|
||||
|
||||
if !status.success() {
|
||||
panic!("Node build script failed with: {status}");
|
||||
}
|
||||
|
||||
// Row-by-row comparison
|
||||
diff_dbs(&rust_db, &node_db);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn tempdir() -> PathBuf {
|
||||
let base = std::env::temp_dir().join(format!(
|
||||
"xlsxread-test-{}",
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.subsec_nanos()
|
||||
));
|
||||
std::fs::create_dir_all(&base).unwrap();
|
||||
base
|
||||
}
|
||||
|
||||
fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) {
|
||||
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("configs")
|
||||
.join(config_name);
|
||||
|
||||
let cfg = xlsxread::config::load_config(&cfg_path)
|
||||
.unwrap_or_else(|e| panic!("load config {config_name}: {e}"));
|
||||
let patterns =
|
||||
xlsxread::transform::CompiledPatterns::new(&cfg.scores).expect("compile patterns");
|
||||
|
||||
// Collect files
|
||||
let mut files: Vec<PathBuf> = std::fs::read_dir(input_dir)
|
||||
.unwrap()
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| {
|
||||
let l = e.to_lowercase();
|
||||
l == "xls" || l == "xlsx"
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let conn = xlsxread::writer::open_db(db_path, &cfg).expect("open db");
|
||||
conn.execute_batch("BEGIN").unwrap();
|
||||
|
||||
for file in &files {
|
||||
xlsxread::reader::process_file(file, &cfg, |_, raw| {
|
||||
let all_blank = xlsxread::reader::is_all_blank(raw);
|
||||
if cfg.reader.strip_blank_rows && all_blank {
|
||||
return;
|
||||
}
|
||||
let ho_ten = raw
|
||||
.get(cfg.columns.ho_ten)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let so_bao_danh = raw
|
||||
.get(cfg.columns.so_bao_danh)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
if xlsxread::transform::validate_row(
|
||||
&ho_ten,
|
||||
&so_bao_danh,
|
||||
&cfg.validation,
|
||||
cfg.reader.strip_blank_rows,
|
||||
all_blank,
|
||||
)
|
||||
.is_err()
|
||||
{
|
||||
return;
|
||||
}
|
||||
let row = xlsxread::transform::transform_row(raw, &cfg, &patterns);
|
||||
let _ = xlsxread::writer::insert_row(
|
||||
&conn,
|
||||
&cfg.insert.sql,
|
||||
&row,
|
||||
xlsxread::writer::SCORE_FIELDS,
|
||||
);
|
||||
})
|
||||
.expect("process file");
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT").unwrap();
|
||||
conn.execute_batch("VACUUM").unwrap();
|
||||
}
|
||||
|
||||
fn query_count(db_path: &Path) -> i64 {
|
||||
let conn = rusqlite::Connection::open(db_path).expect("open db for count");
|
||||
conn.query_row("SELECT COUNT(*) FROM student", [], |r| r.get(0))
|
||||
.expect("count query")
|
||||
}
|
||||
|
||||
fn diff_dbs(a: &Path, b: &Path) {
|
||||
let conn_a = rusqlite::Connection::open(a).unwrap();
|
||||
|
||||
// Attach b as "other"
|
||||
conn_a
|
||||
.execute_batch(&format!("ATTACH DATABASE '{}' AS other", b.display()))
|
||||
.unwrap();
|
||||
|
||||
// Rows in a not in b
|
||||
let missing_in_b: i64 = conn_a
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM main.student s
|
||||
WHERE NOT EXISTS (SELECT 1 FROM other.student o WHERE o.so_bao_danh = s.so_bao_danh)",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Rows in b not in a
|
||||
let missing_in_a: i64 = conn_a
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM other.student o
|
||||
WHERE NOT EXISTS (SELECT 1 FROM main.student s WHERE s.so_bao_danh = o.so_bao_danh)",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
missing_in_b, 0,
|
||||
"{missing_in_b} rows in Rust DB missing from Node DB"
|
||||
);
|
||||
assert_eq!(
|
||||
missing_in_a, 0,
|
||||
"{missing_in_a} rows in Node DB missing from Rust DB"
|
||||
);
|
||||
}
|
||||
Reference in new issue
Block a user