diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index 052f987..a8ae901 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -23,9 +23,9 @@ jobs: build: runs-on: ubuntu-latest env: - # The parser is pure Go — grate, excelize, yaml.v3 and modernc.org/sqlite - # are all cgo-free — so no C toolchain is needed. Set explicitly rather - # than relying on the default. + # Every Go module here is cgo-free — grate, excelize, yaml.v3, x/net, + # x/text and modernc.org/sqlite — so no C toolchain is needed. Set + # explicitly rather than relying on the default. CGO_ENABLED: '0' steps: - uses: actions/checkout@v4 @@ -34,29 +34,37 @@ jobs: with: go-version: '1.26' cache-dependency-path: | - go-parser/go.sum + parser/go.sum crawler/go.sum + assembler/go.sum + # web/ is the only npm project in the repository; the other stages are Go. - uses: actions/setup-node@v4 with: node-version: '24' cache: 'npm' - cache-dependency-path: package-lock.json + cache-dependency-path: web/package-lock.json - - run: npm ci + - name: Install web dependencies + working-directory: web + run: npm ci - # The reader-fidelity suite compares all 299 real input files against a + # The reader-fidelity suite compares every real input file against a # committed hash oracle, so it is the regression guard for the whole # reader. Runs before anything is built. - name: Test parser - run: npm run test:go + run: go -C parser test ./... # The crawler is not part of the build — it only refreshes data/ by hand. - # It is still compiled and tested here so it cannot rot unnoticed, and - # because its filename test guards the parser: input filenames decide - # which row survives a duplicate exam number. + # It is still tested here so it cannot rot unnoticed, and because its + # fixture test guards the parser: input filenames decide which row + # survives a duplicate exam number. - name: Test crawler - run: npm run test:crawler + run: go -C crawler test ./... + + - name: Lint web + working-directory: web + run: npm run lint # excelize carries an open advisory, and the 2017 refresh runbook feeds # network-downloaded spreadsheets straight into the parser. @@ -64,22 +72,14 @@ jobs: run: | go install golang.org/x/vuln/cmd/govulncheck@latest GOVULNCHECK="$(go env GOPATH)/bin/govulncheck" - (cd go-parser && "$GOVULNCHECK" ./...) - (cd crawler && "$GOVULNCHECK" ./...) + for m in parser crawler assembler; do (cd "$m" && "$GOVULNCHECK" ./...); done - # One parser binary builds every dataset; build-db.js reads the dataset - # list from web/src/datasets.js, verifies each database against its row - # count, then gzips it in place leaving no uncompressed file behind. - - name: Build databases - run: | - npm run build:go - npm run build:db - - # One Vite build produces every page. web/scripts/assemble-site.js copies - # the emitted index.html to each dataset path (and the legacy nested - # URLs), then fails the job if an uncompressed database reached _site. - - name: Build and assemble site - run: npm run build:site + # One command runs the whole pipeline: compile the parser, build and + # verify each database against its registry row count, compress it, build + # the web app, and assemble _site — refusing to continue if a database is + # short, an artifact looks truncated, or one is missing entirely. + - name: Build site + run: go -C assembler run ./cmd/assemble - uses: actions/upload-pages-artifact@v3 with: diff --git a/.gitignore b/.gitignore index 1ea055c..35f1649 100644 --- a/.gitignore +++ b/.gitignore @@ -17,11 +17,9 @@ dist/ ### Generated databases and Vite publicDir staging ### .build/ -### Rust build artefacts ### - ### Assembled Pages artifact ### _site/ -# Go parser build output and regenerable ground-truth dumps -go-parser/bin/ -go-parser/testdata/dumps/ +# Parser build output and regenerable ground-truth dumps +parser/bin/ +parser/testdata/dumps/ diff --git a/README.md b/README.md index 7b82751..627f52a 100644 --- a/README.md +++ b/README.md @@ -17,49 +17,57 @@ in git history. ## Layout +The repository is one directory per pipeline stage, plus the two stores they +pass between them. + ``` -web/ the frontend — one Vite app serving both datasets and the hub - src/datasets.js the dataset ids and their per-dataset content - src/router.js pathname → dataset - scripts/ site assembly -crawler/ Go — re-fetches the source spreadsheets - internal/sources/ per dataset: which article to read, how to name its files - internal/article/ pulls the download links out of that article - internal/fetch/ concurrent, resumable downloading -go-parser/ Go — Excel to SQLite - internal/schema/ canonical 22-column table: DDL, INSERT, subject regexes - configs/.yml per-dataset parse rules only, no SQL - scripts/ database build, parity verification -data// raw Excel files, one directory per dataset -docs/ architecture, data pipeline, deployment +crawler/ Go — re-fetches the source spreadsheets → data/ +parser/ Go — Excel to SQLite data/ → .db +assembler/ Go — verifies, compresses, builds, assembles .db + web/ → _site/ +web/ npm — the frontend, one Vite app for every dataset +data// raw Excel files, one directory per dataset +datasets.json the registry: which datasets exist, and their expected size +docs/ architecture, data pipeline, deployment ``` -`web/` is the only npm workspace; `crawler/` and `go-parser/` are independent Go -modules. The one cross-boundary import is `web/src/datasets.js`, which -`go-parser/scripts/build-db.js` reads for the dataset list and expected sizes. +Each stage runs on its own and hands its output to the next through the stores. +`web/` is the only npm project; the three stages are independent Go modules. + +`datasets.json` is the contract between them. It is JSON because Go and the Vite +app both read it and neither needs a dependency to do so; presentation stays in +`web/src/datasets.js`, keyed by id, which fails loudly if the two disagree. The dataset id is one identifier end to end: ``` -data/2017/ → go-parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ ``` ## Build ```bash -npm ci -npm run build:go # compile the parser -npm run build:db # build + gzip both databases (add an id for just one) -npm run build:site # one Vite build, then assemble into _site/ +(cd web && npm ci) +go -C assembler run ./cmd/assemble # databases, then the site, into _site/ npx serve _site ``` +That one command compiles the parser, builds and verifies each database against +its registry row count, compresses it, builds the web app and assembles `_site` — +refusing to continue if a database is short, an artifact looks truncated, or one +is missing altogether. Sub-steps when iterating: + +```bash +go -C assembler run ./cmd/assemble db 2017 # one database +go -C assembler run ./cmd/assemble site # web build and _site only +(cd web && npm run dev) # the app against staged databases +``` + The source spreadsheets are committed, so a crawl is only needed to refresh them: ```bash -npm run crawl:2016 # re-fetch data/2016/ -npm run crawl:2017 # re-fetch data/2017/ +go -C crawler run ./cmd/crawl 2016 +go -C crawler run ./cmd/crawl 2017 ``` Each reads the download links out of the article that published the dataset, so @@ -72,12 +80,14 @@ Pushing to `main` runs the same steps in ## Adding a dataset 1. Put the Excel files in `data//` -2. Add `go-parser/configs/.yml` — sheet mode, column indices, validation +2. Add `parser/configs/.yml` — sheet mode, column indices, validation guards. No SQL; the schema is canonical. -3. Add an entry to `DATASETS` in `web/src/datasets.js` +3. Add an entry to `datasets.json` with its expected row count and size +4. Add the matching presentation to `CONTENT` in `web/src/datasets.js` -Everything else follows: the build script, the site assembly and the router all -read that one list, and the UI adapts to whichever columns the dataset fills. +Everything else follows: the assembler, the router and the hub all read the +registry, and the UI adapts to whichever columns the dataset fills. Steps 3 and 4 +check each other, so forgetting either one fails rather than half-working. ## Docs diff --git a/assembler/cmd/assemble/main.go b/assembler/cmd/assemble/main.go new file mode 100644 index 0000000..adaa95b --- /dev/null +++ b/assembler/cmd/assemble/main.go @@ -0,0 +1,129 @@ +// Command assemble turns source data and the web app into the directory +// GitHub Pages publishes. +// +// assemble # databases, then the site +// assemble db # databases only (add ids to limit: assemble db 2017) +// assemble site # web build and _site only, reusing staged databases +// +// It sequences the other stages rather than doing their work: the parser reads +// spreadsheets, Vite bundles the app, and this decides what runs, checks what +// came out, and refuses to publish anything that looks wrong. +package main + +import ( + "fmt" + "os" + "path/filepath" + + "github.com/tiennm99/thptqg/assembler/internal/databases" + "github.com/tiennm99/thptqg/assembler/internal/registry" + "github.com/tiennm99/thptqg/assembler/internal/site" +) + +func main() { + if err := run(os.Args[1:]); err != nil { + fmt.Fprintf(os.Stderr, "assemble: %v\n", err) + os.Exit(1) + } +} + +func usage() { + fmt.Fprint(os.Stderr, `usage: assemble [step] [dataset...] + +Steps: + (none) databases, then the site + db build, verify and compress the databases + site build the web app and assemble _site + +Naming datasets limits the db step to those; the site step always covers all of +them, since a partial site would publish links to databases it did not build. +`) +} + +func run(args []string) error { + step := "" + if len(args) > 0 { + switch args[0] { + case "db", "site": + step, args = args[0], args[1:] + case "-h", "--help": + usage() + return nil + default: + // Bare dataset ids are a natural thing to type; treat them as the + // db step rather than rejecting them. + step = "db" + } + } + + root, err := repoRoot() + if err != nil { + return err + } + + all, err := registry.Load(root) + if err != nil { + return err + } + + if step == "" || step == "db" { + selected, err := registry.Select(all, args) + if err != nil { + return err + } + if err := buildDatabases(root, all, selected); err != nil { + return err + } + } + + if step == "" || step == "site" { + sp := site.DefaultPaths(root) + if err := site.BuildWeb(sp); err != nil { + return err + } + if err := site.Assemble(sp, all); err != nil { + return err + } + } + return nil +} + +func buildDatabases(root string, all, selected []registry.Dataset) error { + p := databases.DefaultPaths(root) + + // Sweep first: a dataset dropped from the registry leaves its .db.gz behind, + // and the site assembly copies the staging directory wholesale. + if err := databases.Clean(p, all); err != nil { + return err + } + + bin, err := databases.BuildParser(p) + if err != nil { + return err + } + for _, d := range selected { + if err := databases.Build(p, bin, d); err != nil { + return err + } + } + return nil +} + +// repoRoot walks up from the working directory to the directory holding +// datasets.json, so the command works from anywhere in the tree. +func repoRoot() (string, error) { + dir, err := os.Getwd() + if err != nil { + return "", err + } + for { + if _, err := os.Stat(filepath.Join(dir, "datasets.json")); err == nil { + return dir, nil + } + parent := filepath.Dir(dir) + if parent == dir { + return "", fmt.Errorf("no datasets.json found in any parent of the working directory") + } + dir = parent + } +} diff --git a/assembler/go.mod b/assembler/go.mod new file mode 100644 index 0000000..481fa27 --- /dev/null +++ b/assembler/go.mod @@ -0,0 +1,17 @@ +module github.com/tiennm99/thptqg/assembler + +go 1.26.5 + +require modernc.org/sqlite v1.56.0 + +require ( + github.com/dustin/go-humanize v1.0.1 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/mattn/go-isatty v0.0.24 // indirect + github.com/ncruces/go-strftime v1.0.0 // indirect + github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect + golang.org/x/sys v0.47.0 // indirect + modernc.org/libc v1.74.4 // indirect + modernc.org/mathutil v1.7.1 // indirect + modernc.org/memory v1.11.0 // indirect +) diff --git a/assembler/go.sum b/assembler/go.sum new file mode 100644 index 0000000..1932692 --- /dev/null +++ b/assembler/go.sum @@ -0,0 +1,50 @@ +github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY= +github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3 h1:LMLX+LgTNWpfvCBdFebv6EsYotImrt/Ppc5cXIriCSo= +github.com/google/pprof v0.0.0-20260802141513-ef3492d7dac3/go.mod h1:jl5iWTm0/hd5PjEYEOuwAJ57L/CibdZfrqZ5XA5GrCk= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k= +github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM= +github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= +github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w= +github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE= +github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo= +golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ= +golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0= +golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM= +golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q= +golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA= +modernc.org/cc/v4 v4.29.1 h1:MKgdCV3WykTSPqpVrnxdEDS0HEd2FHpKZDzxzU5LyeI= +modernc.org/cc/v4 v4.29.1/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI= +modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU= +modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk= +modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM= +modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU= +modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI= +modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito= +modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI= +modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY= +modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks= +modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI= +modernc.org/libc v1.74.4 h1:fX1Omw4o2/1C2iRkkIsrQTasJQldLhRmuPreXLoWs9k= +modernc.org/libc v1.74.4/go.mod h1:eeQAS9W3sZeKYMFubydxJpII9ybHWshk+7or7bLG9co= +modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU= +modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg= +modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI= +modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw= +modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg= +modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns= +modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w= +modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE= +modernc.org/sqlite v1.56.0 h1:/D8e2RfFqoy/Zc6PuC76U28zFwmI/sYx1Kjm4yEn9e0= +modernc.org/sqlite v1.56.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ= +modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0= +modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A= +modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y= +modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM= diff --git a/assembler/internal/databases/databases.go b/assembler/internal/databases/databases.go new file mode 100644 index 0000000..0bf077f --- /dev/null +++ b/assembler/internal/databases/databases.go @@ -0,0 +1,217 @@ +// Package databases builds, verifies and compresses one SQLite file per +// dataset. +// +// VERIFICATION IS THE POINT OF THIS PACKAGE, not an extra. +// +// Nothing between the parser and the published site otherwise asserts that a +// database has data in it. The parser logs a file-level failure and continues, +// returns success regardless, and finishes cleanly even at zero rows; the site +// assembly only inspects filenames. So a reader that silently under-produced +// would publish a truncated dataset with green CI and no red signal anywhere. +// +// The guards below close that: a build whose row count does not match the +// registry, or whose artifact is implausibly small, fails the pipeline. +package databases + +import ( + "compress/gzip" + "database/sql" + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + + _ "modernc.org/sqlite" // pure-Go driver: the pipeline stays cgo-free + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +// driverName is modernc.org/sqlite's registered name. +const driverName = "sqlite" + +// minSizeRatio: a gzipped database far below its usual size means a truncated +// build, even if the row count somehow passed. +const minSizeRatio = 0.9 + +// Paths locates the pieces this package needs. +type Paths struct { + // Root is the repository root. + Root string + // Parser is the parser module directory. + Parser string + // OutDir is where the databases are staged — the directory Vite publishes. + OutDir string +} + +// DefaultPaths derives the standard layout from the repository root. +func DefaultPaths(root string) Paths { + return Paths{ + Root: root, + Parser: filepath.Join(root, "parser"), + OutDir: filepath.Join(root, ".build", "public", "db"), + } +} + +// BuildParser compiles the parser binary and returns its path. +// +// Compiling here rather than expecting a prebuilt binary keeps the pipeline one +// command. Go caches the work, so repeat runs cost almost nothing. +func BuildParser(p Paths) (string, error) { + bin := filepath.Join(p.Parser, "bin", "xlsxread") + cmd := exec.Command("go", "-C", p.Parser, "build", "-o", "bin/xlsxread", "./cmd/xlsxread") + cmd.Stdout, cmd.Stderr = os.Stdout, os.Stderr + if err := cmd.Run(); err != nil { + return "", fmt.Errorf("compiling the parser: %w", err) + } + return bin, nil +} + +// Build runs the parser for one dataset, verifies the result and compresses it. +// +// Only the .gz survives: shipping a 100+ MB uncompressed database is made +// structurally impossible rather than left to a cleanup step. +func Build(p Paths, bin string, d registry.Dataset) error { + if err := os.MkdirAll(p.OutDir, 0o755); err != nil { + return err + } + db := filepath.Join(p.OutDir, d.ID+".db") + + cmd := exec.Command(bin, + "build", + "--schema", filepath.Join(p.Parser, "configs", d.ID+".yml"), + "--input", filepath.Join(p.Root, "data", d.ID), + "--output", db, + ) + cmd.Stdout, cmd.Stderr = os.Stdout, os.Stderr + if err := cmd.Run(); err != nil { + return fmt.Errorf("%s: parser failed: %w", d.ID, err) + } + + rows, err := countRows(db) + if err != nil { + return fmt.Errorf("%s: %w", d.ID, err) + } + if rows != d.ExpectedRows { + return fmt.Errorf( + "%s: row count %d, expected %d\nRefusing to publish — the build did not reproduce the known dataset", + d.ID, rows, d.ExpectedRows) + } + fmt.Printf(" ✓ %s: %d rows (matches expected)\n", d.ID, rows) + + gz, size, err := compress(db) + if err != nil { + return fmt.Errorf("%s: %w", d.ID, err) + } + + sizeMb := float64(size) / 1024 / 1024 + if min := d.DbSizeMb * minSizeRatio; sizeMb < min { + return fmt.Errorf( + "%s: %.1f MB is below %.1f MB (%.0f%% of the expected %.0f MB)\n"+ + "Refusing to publish — the artifact looks truncated", + d.ID, sizeMb, min, minSizeRatio*100, d.DbSizeMb) + } + + fmt.Printf(" → %s (%.1f MB)\n\n", filepath.Base(gz), sizeMb) + return nil +} + +// countRows opens the database read-only and counts what was written. +func countRows(path string) (int64, error) { + conn, err := sql.Open(driverName, "file:"+path+"?mode=ro") + if err != nil { + return 0, err + } + defer conn.Close() + + var n int64 + if err := conn.QueryRow("SELECT COUNT(*) FROM student").Scan(&n); err != nil { + return 0, fmt.Errorf("counting rows: %w", err) + } + return n, nil +} + +// compress gzips path to path+".gz" and removes the original, returning the +// compressed path and its size. +// +// The source is deleted only after the compressed file is closed successfully, +// so a failure part-way through leaves the database rather than losing it. +func compress(path string) (string, int64, error) { + in, err := os.Open(path) + if err != nil { + return "", 0, err + } + defer in.Close() + + gzPath := path + ".gz" + out, err := os.Create(gzPath) + if err != nil { + return "", 0, err + } + + zw, err := gzip.NewWriterLevel(out, gzip.BestCompression) + if err != nil { + out.Close() + return "", 0, err + } + if _, err := io.Copy(zw, in); err != nil { + zw.Close() + out.Close() + os.Remove(gzPath) + return "", 0, err + } + if err := zw.Close(); err != nil { + out.Close() + os.Remove(gzPath) + return "", 0, err + } + if err := out.Close(); err != nil { + os.Remove(gzPath) + return "", 0, err + } + + if err := in.Close(); err != nil { + return "", 0, err + } + if err := os.Remove(path); err != nil { + return "", 0, fmt.Errorf("removing the uncompressed database: %w", err) + } + + st, err := os.Stat(gzPath) + if err != nil { + return "", 0, err + } + return gzPath, st.Size(), nil +} + +// Clean removes staged artifacts for datasets that are no longer in the +// registry. Without this a removed dataset's .db.gz lingers in the staging +// directory, and the site assembly copies that directory wholesale — so the +// dead database would be published again. +func Clean(p Paths, keep []registry.Dataset) error { + entries, err := os.ReadDir(p.OutDir) + if os.IsNotExist(err) { + return nil + } + if err != nil { + return err + } + + wanted := make(map[string]bool, len(keep)*2) + for _, d := range keep { + wanted[d.ID+".db"] = true + wanted[d.ID+".db.gz"] = true + } + + for _, e := range entries { + if e.IsDir() || wanted[e.Name()] { + continue + } + full := filepath.Join(p.OutDir, e.Name()) + if err := os.Remove(full); err != nil { + return err + } + fmt.Printf(" removed stale artifact %s\n", e.Name()) + } + return nil +} diff --git a/assembler/internal/databases/databases_test.go b/assembler/internal/databases/databases_test.go new file mode 100644 index 0000000..05a9c27 --- /dev/null +++ b/assembler/internal/databases/databases_test.go @@ -0,0 +1,108 @@ +package databases + +import ( + "compress/gzip" + "io" + "os" + "path/filepath" + "slices" + "testing" + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +// TestCompressRoundTripsAndRemovesTheSource: only the .gz may survive, so that +// shipping a 100+ MB uncompressed database is structurally impossible rather +// than left to a cleanup step. +func TestCompressRoundTripsAndRemovesTheSource(t *testing.T) { + dir := t.TempDir() + src := filepath.Join(dir, "2016.db") + body := []byte("pretend this is a SQLite file") + if err := os.WriteFile(src, body, 0o644); err != nil { + t.Fatal(err) + } + + gzPath, size, err := compress(src) + if err != nil { + t.Fatal(err) + } + if gzPath != src+".gz" || size <= 0 { + t.Fatalf("gzPath=%q size=%d", gzPath, size) + } + if _, err := os.Stat(src); !os.IsNotExist(err) { + t.Error("the uncompressed database must not survive") + } + + f, err := os.Open(gzPath) + if err != nil { + t.Fatal(err) + } + defer f.Close() + zr, err := gzip.NewReader(f) + if err != nil { + t.Fatal(err) + } + got, err := io.ReadAll(zr) + if err != nil { + t.Fatal(err) + } + if string(got) != string(body) { + t.Errorf("round-trip gave %q", got) + } +} + +func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { + dir := t.TempDir() + for _, name := range []string{ + "2016.db.gz", "2017.db.gz", + "2017-old.db.gz", // dropped from the registry + "2017-old2.db.gz", // dropped from the registry + "2016.db-journal", // interrupted run + } { + if err := os.WriteFile(filepath.Join(dir, name), []byte("x"), 0o644); err != nil { + t.Fatal(err) + } + } + + p := Paths{OutDir: dir} + keep := []registry.Dataset{{ID: "2016"}, {ID: "2017"}} + if err := Clean(p, keep); err != nil { + t.Fatal(err) + } + + entries, err := os.ReadDir(dir) + if err != nil { + t.Fatal(err) + } + var left []string + for _, e := range entries { + left = append(left, e.Name()) + } + slices.Sort(left) + + want := []string{"2016.db.gz", "2017.db.gz"} + if !slices.Equal(left, want) { + t.Errorf("left %v, want %v", left, want) + } +} + +// TestCleanToleratesAnAbsentStagingDirectory: a fresh checkout has never built +// anything, and that is not an error. +func TestCleanToleratesAnAbsentStagingDirectory(t *testing.T) { + p := Paths{OutDir: filepath.Join(t.TempDir(), "never-created")} + if err := Clean(p, nil); err != nil { + t.Errorf("Clean on a missing directory should succeed, got %v", err) + } +} + +func TestDefaultPaths(t *testing.T) { + p := DefaultPaths("/repo") + if p.Parser != filepath.Join("/repo", "parser") { + t.Errorf("Parser = %q", p.Parser) + } + // The staging directory must be the one Vite publishes, or the databases + // never reach the site. + if p.OutDir != filepath.Join("/repo", ".build", "public", "db") { + t.Errorf("OutDir = %q", p.OutDir) + } +} diff --git a/assembler/internal/registry/registry.go b/assembler/internal/registry/registry.go new file mode 100644 index 0000000..7622185 --- /dev/null +++ b/assembler/internal/registry/registry.go @@ -0,0 +1,80 @@ +// Package registry reads the repository-root datasets.json. +// +// That file is the one place every stage agrees on what exists. The Vite app +// reads it too, which is why it is JSON: Go and the browser both parse it +// without a dependency. +package registry + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" +) + +// Dataset is one entry in the registry. +type Dataset struct { + ID string `json:"id"` + // ExpectedRows is exact. The inputs are frozen historical exam results, so + // a deviation of even one row means something changed that nobody intended. + ExpectedRows int64 `json:"expectedRows"` + // DbSizeMb is the usual size of the gzipped database, used to catch a + // build that produced a plausible row count but a truncated artifact. + DbSizeMb float64 `json:"dbSizeMb"` +} + +type file struct { + Datasets []Dataset `json:"datasets"` +} + +// Load reads datasets.json from the repository root. +func Load(root string) ([]Dataset, error) { + path := filepath.Join(root, "datasets.json") + b, err := os.ReadFile(path) + if err != nil { + return nil, fmt.Errorf("cannot read the dataset registry: %w", err) + } + + var f file + if err := json.Unmarshal(b, &f); err != nil { + return nil, fmt.Errorf("%s: %w", path, err) + } + if len(f.Datasets) == 0 { + return nil, fmt.Errorf("%s declares no datasets", path) + } + + for _, d := range f.Datasets { + switch { + case d.ID == "": + return nil, fmt.Errorf("%s: a dataset has no id", path) + case d.ExpectedRows <= 0: + return nil, fmt.Errorf("%s: %s has no expectedRows; the build guard needs it", path, d.ID) + case d.DbSizeMb <= 0: + return nil, fmt.Errorf("%s: %s has no dbSizeMb; the size guard needs it", path, d.ID) + } + } + return f.Datasets, nil +} + +// Select returns the named datasets, or all of them when none are named. +func Select(all []Dataset, ids []string) ([]Dataset, error) { + if len(ids) == 0 { + return all, nil + } + byID := make(map[string]Dataset, len(all)) + known := make([]string, 0, len(all)) + for _, d := range all { + byID[d.ID] = d + known = append(known, d.ID) + } + + out := make([]Dataset, 0, len(ids)) + for _, id := range ids { + d, ok := byID[id] + if !ok { + return nil, fmt.Errorf("unknown dataset %q (known: %v)", id, known) + } + out = append(out, d) + } + return out, nil +} diff --git a/assembler/internal/registry/registry_test.go b/assembler/internal/registry/registry_test.go new file mode 100644 index 0000000..e1c8d06 --- /dev/null +++ b/assembler/internal/registry/registry_test.go @@ -0,0 +1,93 @@ +package registry + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func writeRegistry(t *testing.T, body string) string { + t.Helper() + dir := t.TempDir() + if err := os.WriteFile(filepath.Join(dir, "datasets.json"), []byte(body), 0o644); err != nil { + t.Fatal(err) + } + return dir +} + +func TestLoadsTheRealRegistry(t *testing.T) { + root := filepath.Join("..", "..", "..") + got, err := Load(root) + if err != nil { + t.Fatal(err) + } + if len(got) == 0 { + t.Fatal("no datasets") + } + for _, d := range got { + if d.ID == "" || d.ExpectedRows <= 0 || d.DbSizeMb <= 0 { + t.Errorf("incomplete entry: %+v", d) + } + // Every declared dataset must have somewhere to read from and a parse + // config, or the build fails much later with a worse message. + for _, p := range []string{ + filepath.Join(root, "data", d.ID), + filepath.Join(root, "parser", "configs", d.ID+".yml"), + } { + if _, err := os.Stat(p); err != nil { + t.Errorf("%s: missing %s", d.ID, p) + } + } + } +} + +// TestIncompleteEntriesAreRejected: a dataset missing either guard figure would +// otherwise publish unverified. Both are load-bearing, so neither may default. +func TestIncompleteEntriesAreRejected(t *testing.T) { + for name, body := range map[string]string{ + "no expectedRows": `{"datasets":[{"id":"x","dbSizeMb":1}]}`, + "no dbSizeMb": `{"datasets":[{"id":"x","expectedRows":1}]}`, + "no id": `{"datasets":[{"expectedRows":1,"dbSizeMb":1}]}`, + "zero rows": `{"datasets":[{"id":"x","expectedRows":0,"dbSizeMb":1}]}`, + "empty": `{"datasets":[]}`, + } { + t.Run(name, func(t *testing.T) { + if _, err := Load(writeRegistry(t, body)); err == nil { + t.Error("expected an error") + } + }) + } +} + +func TestMissingAndMalformedRegistry(t *testing.T) { + if _, err := Load(t.TempDir()); err == nil { + t.Error("a missing registry must fail") + } + if _, err := Load(writeRegistry(t, "{not json")); err == nil { + t.Error("a malformed registry must fail") + } +} + +func TestSelect(t *testing.T) { + all := []Dataset{{ID: "2016"}, {ID: "2017"}} + + got, err := Select(all, nil) + if err != nil || len(got) != 2 { + t.Errorf("no ids should select everything: %v %v", got, err) + } + + got, err = Select(all, []string{"2017"}) + if err != nil || len(got) != 1 || got[0].ID != "2017" { + t.Errorf("Select(2017) = %v, %v", got, err) + } + + // A typo must not silently build nothing. + _, err = Select(all, []string{"2018"}) + if err == nil { + t.Fatal("an unknown id must fail") + } + if !strings.Contains(err.Error(), "2018") || !strings.Contains(err.Error(), "2016") { + t.Errorf("the error should name the bad id and the known ones, got: %v", err) + } +} diff --git a/assembler/internal/site/site.go b/assembler/internal/site/site.go new file mode 100644 index 0000000..f799bb4 --- /dev/null +++ b/assembler/internal/site/site.go @@ -0,0 +1,210 @@ +// Package site turns the web app and the staged databases into the directory +// GitHub Pages publishes. +// +// The app resolves its dataset from the URL, so every page is the same +// index.html. Because Vite's `base` is absolute (/thptqg/), that file references +// /thptqg/assets/... no matter which directory it is served from — so copying it +// to each dataset path produces a real static file at every URL. +// +// GitHub Pages serves those as directory indexes, which is why this needs no +// SPA 404-fallback redirect. That matters beyond tidiness: the usual fallback +// rewrites the URL and would interfere with the ?q= deep links the app relies +// on. +package site + +import ( + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + "regexp" + "strings" + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +// Paths locates the pieces this package needs. +type Paths struct { + Root string + // Web is the Vite project directory. + Web string + // Dist is where Vite emits, inside the web workspace. + Dist string + // Site is the artifact the deploy action uploads, at the repository root. + Site string +} + +// DefaultPaths derives the standard layout from the repository root. +func DefaultPaths(root string) Paths { + web := filepath.Join(root, "web") + return Paths{ + Root: root, + Web: web, + Dist: filepath.Join(web, "dist"), + Site: filepath.Join(root, "_site"), + } +} + +// BuildWeb runs the Vite build. +// +// Shelling out to npm is not a wart: Vite is a Node tool, and web/ is the only +// npm project left in the repository. This stage owns the sequencing, not the +// bundling. +func BuildWeb(p Paths) error { + cmd := exec.Command("npm", "run", "build") + cmd.Dir = p.Web + cmd.Stdout, cmd.Stderr = os.Stdout, os.Stderr + if err := cmd.Run(); err != nil { + return fmt.Errorf("vite build: %w", err) + } + return nil +} + +// Assemble copies the build to one directory per dataset and checks the result. +func Assemble(p Paths, datasets []registry.Dataset) error { + index := filepath.Join(p.Dist, "index.html") + if _, err := os.Stat(index); err != nil { + return fmt.Errorf("no build found at %s — run the web build first", p.Dist) + } + + if err := os.RemoveAll(p.Site); err != nil { + return err + } + if err := os.MkdirAll(p.Site, 0o755); err != nil { + return err + } + + // Base build: index.html, assets/, and the gzipped databases from publicDir. + if err := copyTree(p.Dist, p.Site); err != nil { + return err + } + + // Unknown paths render the hub rather than the default Pages 404. + if err := copyFile(index, filepath.Join(p.Site, "404.html")); err != nil { + return err + } + + // One entry point per dataset. + for _, d := range datasets { + dir := filepath.Join(p.Site, d.ID) + if err := os.MkdirAll(dir, 0o755); err != nil { + return err + } + if err := copyFile(index, filepath.Join(dir, "index.html")); err != nil { + return err + } + } + + if err := checkDatabasesPresent(p.Site, datasets); err != nil { + return err + } + if err := checkNoRawDatabases(p.Site); err != nil { + return err + } + + fmt.Printf("assembled %s\n", p.Site) + fmt.Printf(" /thptqg/\n /thptqg/404.html\n") + for _, d := range datasets { + fmt.Printf(" /thptqg/%s\n", d.ID) + } + return nil +} + +// checkDatabasesPresent: every dataset must have shipped its database. +// +// Without this the site assembles happily with an empty db/ directory — every +// page renders, every query 404s, and CI stays green. That is the failure this +// catches; the size and row-count guards only run when a database was built at +// all. +func checkDatabasesPresent(siteDir string, datasets []registry.Dataset) error { + var missing []string + for _, d := range datasets { + gz := filepath.Join(siteDir, "db", d.ID+".db.gz") + st, err := os.Stat(gz) + if err != nil || st.Size() == 0 { + missing = append(missing, d.ID+".db.gz") + } + } + if len(missing) > 0 { + return fmt.Errorf( + "no database in the site output for: %s\n"+ + "Every page would render and every query would 404. Build the databases first", + strings.Join(missing, ", ")) + } + return nil +} + +// rawDatabase matches an uncompressed SQLite artifact, including the temporary +// files SQLite leaves mid-build. +var rawDatabase = regexp.MustCompile(`\.db(-journal|-wal|-shm)?$`) + +// checkNoRawDatabases rejects an uncompressed database that reached the output. +// +// The build gzips without keeping the source, so none should exist — but the +// staging directory is copied wholesale, and a leftover from an interrupted run +// would go straight through. A raw database is 100+ MB. +func checkNoRawDatabases(siteDir string) error { + var stray []string + err := filepath.WalkDir(siteDir, func(path string, d os.DirEntry, err error) error { + if err != nil { + return err + } + if !d.IsDir() && rawDatabase.MatchString(d.Name()) { + stray = append(stray, path) + } + return nil + }) + if err != nil { + return err + } + if len(stray) > 0 { + var b strings.Builder + b.WriteString("uncompressed database artefact(s) found in the site output:\n") + for _, f := range stray { + st, _ := os.Stat(f) + fmt.Fprintf(&b, " %s (%.1f MB)\n", f, float64(st.Size())/1048576) + } + b.WriteString("remove them from the staging directory and re-run") + return fmt.Errorf("%s", b.String()) + } + return nil +} + +func copyTree(src, dst string) error { + return filepath.WalkDir(src, func(path string, d os.DirEntry, err error) error { + if err != nil { + return err + } + rel, err := filepath.Rel(src, path) + if err != nil { + return err + } + target := filepath.Join(dst, rel) + if d.IsDir() { + return os.MkdirAll(target, 0o755) + } + return copyFile(path, target) + }) +} + +func copyFile(src, dst string) error { + in, err := os.Open(src) + if err != nil { + return err + } + defer in.Close() + + if err := os.MkdirAll(filepath.Dir(dst), 0o755); err != nil { + return err + } + out, err := os.Create(dst) + if err != nil { + return err + } + if _, err := io.Copy(out, in); err != nil { + out.Close() + return err + } + return out.Close() +} diff --git a/assembler/internal/site/site_test.go b/assembler/internal/site/site_test.go new file mode 100644 index 0000000..33b0d67 --- /dev/null +++ b/assembler/internal/site/site_test.go @@ -0,0 +1,137 @@ +package site + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +var datasets = []registry.Dataset{{ID: "2016"}, {ID: "2017"}} + +// fakeBuild stands in for a Vite build: an index.html, an asset, and whatever +// databases the caller wants staged. +func fakeBuild(t *testing.T, dbs ...string) Paths { + t.Helper() + root := t.TempDir() + dist := filepath.Join(root, "web", "dist") + write(t, filepath.Join(dist, "index.html"), "app") + write(t, filepath.Join(dist, "assets", "index.js"), "console.log(1)") + for _, name := range dbs { + write(t, filepath.Join(dist, "db", name), "gzipped-bytes") + } + return Paths{Root: root, Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")} +} + +func write(t *testing.T, path, body string) { + t.Helper() + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte(body), 0o644); err != nil { + t.Fatal(err) + } +} + +func TestAssembleProducesAPageForEveryDataset(t *testing.T) { + p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + if err := Assemble(p, datasets); err != nil { + t.Fatal(err) + } + for _, want := range []string{ + "index.html", + "404.html", + filepath.Join("2016", "index.html"), + filepath.Join("2017", "index.html"), + filepath.Join("assets", "index.js"), + filepath.Join("db", "2016.db.gz"), + } { + if _, err := os.Stat(filepath.Join(p.Site, want)); err != nil { + t.Errorf("missing from the artifact: %s", want) + } + } +} + +// TestMissingDatabaseFailsTheBuild is the guard that closes the widest hole. +// +// Without it the site assembles happily with an empty db/ directory: every page +// renders, every query 404s, and CI stays green. The row-count and size guards +// cannot catch this — they only run when a database was built at all. +func TestMissingDatabaseFailsTheBuild(t *testing.T) { + p := fakeBuild(t, "2016.db.gz") // 2017 never built + err := Assemble(p, datasets) + if err == nil { + t.Fatal("expected an error when a database is missing") + } + if !strings.Contains(err.Error(), "2017.db.gz") { + t.Errorf("the error should name the missing database, got: %v", err) + } +} + +// TestEmptyDatabaseFailsTheBuild: a zero-byte file satisfies "exists" but is +// not a database. +func TestEmptyDatabaseFailsTheBuild(t *testing.T) { + p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + write(t, filepath.Join(p.Dist, "db", "2017.db.gz"), "") + if err := Assemble(p, datasets); err == nil { + t.Fatal("expected an error for a zero-byte database") + } +} + +// TestRawDatabaseFailsTheBuild: the compression step deletes its source, so a +// raw .db here means an interrupted run left one behind — and it is 100+ MB. +func TestRawDatabaseFailsTheBuild(t *testing.T) { + for _, name := range []string{"2016.db", "2016.db-journal", "2016.db-wal", "2016.db-shm"} { + t.Run(name, func(t *testing.T) { + p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + write(t, filepath.Join(p.Dist, "db", name), "raw sqlite") + err := Assemble(p, datasets) + if err == nil { + t.Fatalf("expected an error for %s", name) + } + if !strings.Contains(err.Error(), "uncompressed") { + t.Errorf("unexpected error: %v", err) + } + }) + } +} + +// TestGzipIsNotMistakenForRaw: the reject pattern is anchored, so a .db.gz must +// pass. Getting this wrong would fail every build. +func TestGzipIsNotMistakenForRaw(t *testing.T) { + if rawDatabase.MatchString("2016.db.gz") { + t.Error("a .db.gz must not be treated as an uncompressed database") + } + for _, name := range []string{"2016.db", "x.db-journal", "x.db-wal", "x.db-shm"} { + if !rawDatabase.MatchString(name) { + t.Errorf("%s should be treated as an uncompressed artifact", name) + } + } +} + +func TestAssembleRejectsAMissingBuild(t *testing.T) { + root := t.TempDir() + p := Paths{Root: root, Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")} + if err := Assemble(p, datasets); err == nil { + t.Fatal("expected an error when there is no Vite build") + } +} + +// TestAssembleIsIdempotent: the site directory is rebuilt from scratch, so a +// previous run's leftovers cannot survive into the artifact. +func TestAssembleIsIdempotent(t *testing.T) { + p := fakeBuild(t, "2016.db.gz", "2017.db.gz") + if err := Assemble(p, datasets); err != nil { + t.Fatal(err) + } + stale := filepath.Join(p.Site, "2015", "index.html") + write(t, stale, "old dataset") + if err := Assemble(p, datasets); err != nil { + t.Fatal(err) + } + if _, err := os.Stat(stale); !os.IsNotExist(err) { + t.Error("a directory from a previous run survived into the new artifact") + } +} diff --git a/crawler/cmd/crawl/main.go b/crawler/cmd/crawl/main.go index f05dc9b..c6c394a 100644 --- a/crawler/cmd/crawl/main.go +++ b/crawler/cmd/crawl/main.go @@ -1,6 +1,6 @@ // Command crawl downloads a dataset's source spreadsheets into data//. // -// The argument is the dataset id, the same one used by go-parser's configs and +// The argument is the dataset id, the same one used by the parser's configs and // the published site paths. // // crawl 2016 # 119 exam-cluster files diff --git a/crawler/internal/article/article_test.go b/crawler/internal/article/article_test.go index 254bae3..6bcb0f5 100644 --- a/crawler/internal/article/article_test.go +++ b/crawler/internal/article/article_test.go @@ -66,7 +66,7 @@ func TestFiltersByExtension(t *testing.T) { } // TestFileIgnoresQueryString: a query string is not part of the filename, and -// writing one to disk would produce a name go-parser never sees. +// writing one to disk would produce a name the parser never sees. func TestFileIgnoresQueryString(t *testing.T) { got := extract(t, `x`, ".xlsx") if len(got) != 1 { diff --git a/crawler/internal/fetch/fetch.go b/crawler/internal/fetch/fetch.go index 5e34d7e..292f72e 100644 --- a/crawler/internal/fetch/fetch.go +++ b/crawler/internal/fetch/fetch.go @@ -141,7 +141,7 @@ func Run(ctx context.Context, items []Item, opts Options) ([]Result, error) { // // A .part left behind by an abrupt kill needs no cleanup: it does not satisfy // the skip check, os.Create truncates it, and the next run re-fetches the file. -// go-parser ignores it in the meantime, since it reads only .xls and .xlsx. +// The parser ignores it in the meantime, since it reads only .xls and .xlsx. func download(ctx context.Context, client *http.Client, it Item, headers map[string]string) Result { if st, err := os.Stat(it.Path); err == nil && st.Size() > 0 { return Result{Item: it, Status: StatusSkip, Size: st.Size()} diff --git a/crawler/internal/sources/source_2016.go b/crawler/internal/sources/source_2016.go index 680b31e..3e2f1a8 100644 --- a/crawler/internal/sources/source_2016.go +++ b/crawler/internal/sources/source_2016.go @@ -16,11 +16,11 @@ import ( // Dest keeps the server's own filename verbatim — a 32-hex content hash, the // cluster slug, then a millisecond timestamp. Two reasons not to prettify it: // -// - go-parser sorts inputs bytewise and inserts last-wins, so filenames decide +// - parser sorts inputs bytewise and inserts last-wins, so filenames decide // which row survives a duplicate exam number. That is live here, not // hypothetical: 877,464 source rows collapse to 877,461, so three rows' // contents depend on this ordering. -// - go-parser/testdata/reader-fidelity-hashes.tsv is keyed by full path, and +// - parser/testdata/reader-fidelity-hashes.tsv is keyed by full path, and // it is frozen — it was produced by the Rust reader, which no longer exists. // // Unlike 2017 this needs no transliteration: the name comes from the URL, so it diff --git a/crawler/internal/sources/source_2017.go b/crawler/internal/sources/source_2017.go index 4f9d088..d6abb9e 100644 --- a/crawler/internal/sources/source_2017.go +++ b/crawler/internal/sources/source_2017.go @@ -53,7 +53,7 @@ var nonAlphanumeric = regexp.MustCompile(`[^a-z0-9]+`) // // The article writes the names with diacritics, so this has to strip them. // Combining marks are removed by the literal range U+0300–U+036F rather than by -// the unicode.Mn category, matching what go-parser does to build ho_ten_ascii. +// the unicode.Mn category, matching what parser does to build ho_ten_ascii. func slug(name string) string { var b strings.Builder for _, r := range norm.NFD.String(name) { diff --git a/crawler/internal/sources/sources.go b/crawler/internal/sources/sources.go index c18da33..5ddaf5d 100644 --- a/crawler/internal/sources/sources.go +++ b/crawler/internal/sources/sources.go @@ -15,7 +15,7 @@ import ( // Source is one crawlable dataset. type Source struct { // ID is the dataset id: the subcommand, the directory under data/, and the - // go-parser config name, all at once. Keeping it single means a source + // parser config name, all at once. Keeping it single means a source // cannot be pointed at the wrong dataset's directory. ID string Summary string @@ -32,14 +32,14 @@ type Source struct { // WantFiles is how many links the article is expected to yield. A page that // suddenly yields fewer has changed shape, and silently crawling a partial - // dataset is the failure this exists to prevent — go-parser would happily + // dataset is the failure this exists to prevent — parser would happily // build a short database and only the row-count guard would catch it, after // the fact. WantFiles int // Dest names the local file for one discovered link. // - // This is the load-bearing part. go-parser sorts input files bytewise and + // This is the load-bearing part. parser sorts input files bytewise and // inserts last-wins, so the names chosen here decide which row survives a // duplicate exam number. Two sources answer it differently and both have a // reason: see source_2016.go and source_2017.go. diff --git a/crawler/internal/sources/sources_test.go b/crawler/internal/sources/sources_test.go index b4586f8..593ea18 100644 --- a/crawler/internal/sources/sources_test.go +++ b/crawler/internal/sources/sources_test.go @@ -49,7 +49,7 @@ func resolveFixture(t *testing.T, src Source) []File { // TestReproducesFilesOnDisk is the guard that matters. // -// go-parser sorts its inputs and inserts last-wins, so filenames decide which +// parser sorts its inputs and inserts last-wins, so filenames decide which // row survives a duplicate exam number. If extraction or naming drifted, a // re-crawl could rebuild a database with the same row count and different // content, which the row-count guard in build-db.js would not catch. diff --git a/datasets.json b/datasets.json new file mode 100644 index 0000000..aab7660 --- /dev/null +++ b/datasets.json @@ -0,0 +1,31 @@ +{ + "_comment": [ + "The dataset registry: the one place every stage agrees on what exists.", + "", + " crawler/ fills data//", + " parser/ reads data// with configs/.yml, writes .db", + " assembler/ verifies, compresses and publishes it as db/.db.gz", + " web/ serves it at /thptqg//", + "", + "It is JSON rather than a module because Go and the Vite app both read it,", + "and JSON is the only format both parse without a dependency. Presentation", + "(titles, labels, SQL presets) stays in web/src/datasets.js keyed by id;", + "that file fails loudly if the two lists disagree.", + "", + "expectedRows is exact, not approximate. The inputs are frozen historical", + "exam results, so a deviation of even one row means something changed that", + "nobody intended, and the assembler refuses to publish." + ], + "datasets": [ + { + "id": "2016", + "expectedRows": 877461, + "dbSizeMb": 44 + }, + { + "id": "2017", + "expectedRows": 861068, + "dbSizeMb": 48 + } + ] +} diff --git a/docs/README.md b/docs/README.md index b2ad89a..0c9ef0a 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,6 +1,6 @@ # Docs -- [`project-overview.md`](./project-overview.md) — goal, scope, constraints, the four datasets, history +- [`project-overview.md`](./project-overview.md) — goal, scope, constraints, the datasets, history - [`system-architecture.md`](./system-architecture.md) — data flow, canonical schema, routing, how one frontend serves both exam years - [`data-pipeline.md`](./data-pipeline.md) — Excel parse quirks, per-dataset formats, overflow-sheet gotcha, expected row counts, verifying a rebuild - [`deployment-guide.md`](./deployment-guide.md) — GitHub Pages workflow, adding a dataset, rollback, troubleshooting diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index c842ed8..4ee55db 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -2,10 +2,10 @@ From raw Excel files to a compressed SQLite file the browser can load. -One Go binary (`go-parser/`) builds every dataset. What differs per dataset is +One Go binary (`parser/`) builds every dataset. What differs per dataset is parse rules only — sheet strategy, column layout, validation guards — declared -in `go-parser/configs/.yml`. The table shape, the INSERT and the subject -regexes are canonical and live in `go-parser/internal/schema/schema.go`. +in `parser/configs/.yml`. The table shape, the INSERT and the subject +regexes are canonical and live in `parser/internal/schema/schema.go`. ## Sources @@ -19,9 +19,9 @@ build — the source files are committed, so a crawl only refreshes them. Both runs are idempotent: files already present are skipped. ```bash -npm run crawl:2016 # crawler/internal/sources/source_2016.go -npm run crawl:2017 # crawler/internal/sources/source_2017.go -npm run crawl:2017 -- --list # read the article, list the files, download none +go -C crawler run ./cmd/crawl 2016 # sources/source_2016.go +go -C crawler run ./cmd/crawl 2017 # sources/source_2017.go +go -C crawler run ./cmd/crawl 2017 --list # list only, download nothing ``` **2017** comes from the article @@ -55,17 +55,17 @@ says how to name what it finds there: downloading. Because `Article` is read at run time, `--list` needs network access too. -`WantFiles` exists because a partial crawl is otherwise silent: go-parser will +`WantFiles` exists because a partial crawl is otherwise silent: parser will build a short database from whatever files are present, and only the row-count guard would notice, after the fact. A page that changes shape stops the crawl instead. ### Local filenames are load-bearing -`go-parser` sorts its input files and inserts with `INSERT OR REPLACE`, which is +`parser` sorts its input files and inserts with `INSERT OR REPLACE`, which is last-wins, so **filenames decide which row survives a duplicate exam number**. A re-crawl that names files differently can produce a database with the same row -count and different content, which the row-count guard in `build-db.js` would +count and different content, which the assembler's row-count guard would not catch. Each source therefore pins its local names, and @@ -103,7 +103,7 @@ which is why only 2016 populates those columns. ## Score text parsing -`SCORE_PATTERNS` in `go-parser/internal/schema/schema.go` defines one regex per subject, and +`SCORE_PATTERNS` in `parser/internal/schema/schema.go` defines one regex per subject, and **all 16 run against every dataset**. A subject a given exam year did not offer simply never matches and stays NULL. @@ -161,18 +161,18 @@ silently drops 13,720 students** (Hanoi +7,275, HCM +6,445). That is what ## Verifying a rebuild -`npm run build:db` verifies itself: each database's row count must match the +The assembler verifies itself: each database's row count must match the figure in the table above, and each `.db.gz` must be at least 90% of its usual size, or the build fails rather than publishing. That guard is the reason a truncated dataset cannot reach the site with a green pipeline. -For a deeper check, `go-parser/scripts/differential-parity.mjs` compares two sets +For a deeper check, `parser/scripts/differential-parity.mjs` compares two sets of databases field-by-field — row counts, per-column non-NULL counts, a full-table SHA-256 over every row ordered by `so_bao_danh`, schema metadata, and build stdout: ```bash -node go-parser/scripts/differential-parity.mjs \ +node parser/scripts/differential-parity.mjs \ --rust /path/to/a-{id}.db --go /path/to/b-{id}.db ``` @@ -180,16 +180,16 @@ It exits non-zero on any mismatch and fails loudly if a dataset is missing rathe than skipping it. Written for the Rust-to-Go migration, it works for any two builds. Uses the built-in `node:sqlite`, so it needs no dependencies. -`go-parser/internal/reader` additionally carries a frozen oracle of per-file -cell-dump hashes covering all 299 inputs; `npm run test:go` fails if any single +`parser/internal/reader` additionally carries a frozen oracle of per-file +cell-dump hashes covering all 182 inputs; `go -C parser test ./...` fails if any single cell of any input file reads differently. ## Refreshing the 2017 data ```bash rm data/2017/*.xls -npm run crawl:2017 -npm run build:db 2017 +go -C crawler run ./cmd/crawl 2017 +go -C assembler run ./cmd/assemble db 2017 ``` The row-count guard in `build:db` confirms the rebuild matches the expected diff --git a/docs/deployment-guide.md b/docs/deployment-guide.md index 88b8342..3f4cf4d 100644 --- a/docs/deployment-guide.md +++ b/docs/deployment-guide.md @@ -7,14 +7,14 @@ One-time setup: **Settings → Pages → Source: GitHub Actions**. ## What the workflow does -1. Checkout, Go toolchain, Node 24, `npm ci` -2. `npm run build:go` — one parser binary -3. `npm run build:db` — builds and gzips both databases into - `.build/public/db/` -4. `npm run build:site` — one Vite build, then `web/scripts/assemble-site.js` -5. `actions/upload-pages-artifact` + `actions/deploy-pages` +1. Checkout, Go toolchain, Node 24, `npm ci` in `web/` +2. Parser and crawler test suites, web lint, `govulncheck` over all three modules +3. `go -C assembler run ./cmd/assemble` — the whole pipeline: compile the + parser, build and verify each database, compress it into `.build/public/db/`, + run the Vite build, assemble `_site/` +4. `actions/upload-pages-artifact` + `actions/deploy-pages` -The database build dominates the runtime: roughly 419 MB of Excel is parsed on +The database build dominates the runtime: roughly 348 MB of Excel is parsed on every deploy. ## Resulting URLs @@ -32,17 +32,16 @@ they now render the hub via `404.html`. ## Local reproduction ```bash -npm ci -npm run build:go -npm run build:db # both; pass an id to build just one -npm run build:site # vite build + assemble into _site/ +(cd web && npm ci) +go -C assembler run ./cmd/assemble npx serve _site ``` -To rebuild a single dataset: +To rebuild a single dataset, or only the site: ```bash -npm run build:db 2017 +go -C assembler run ./cmd/assemble db 2017 +go -C assembler run ./cmd/assemble site ``` ## Base path @@ -54,19 +53,21 @@ up as a blank page with 404s on `/assets/...`. ## Adding a dataset 1. Put the Excel files in `data//` -2. Add `go-parser/configs/.yml` with the parse rules — sheet mode, column +2. Add `parser/configs/.yml` with the parse rules — sheet mode, column indices, SBD validation, header tokens, blank-row stripping. No SQL: the - schema is canonical and lives in `go-parser/internal/schema/schema.go` -3. Add an entry to `DATASETS` in `web/src/datasets.js` + schema is canonical and lives in `parser/internal/schema/schema.go` +3. Add an entry to `datasets.json` — id, `expectedRows`, `dbSizeMb` +4. Add its presentation to `CONTENT` in `web/src/datasets.js` -Nothing else. The build script, the site assembly and the router all read that -one list, and the frontend adapts to whichever columns the dataset populates. +Nothing else. The assembler and the router both read the registry, and the +frontend adapts to whichever columns the dataset populates. The last two steps +check each other, so forgetting either fails rather than half-working. ## Why no uncompressed database can ship -`build-db.js` runs `gzip -9` **without** `-k`, so the raw file does not survive -the build. `assemble-site.js` then fails the job if any `.db`, `.db-journal`, -`.db-wal` or `.db-shm` reached the output. +The assembler deletes the source once compression succeeds, so the raw file +does not survive the build, and it then fails the job if any `.db`, +`.db-journal`, `.db-wal` or `.db-shm` reached the output. Both guards exist because the previous pipeline wrote a 100+ MB uncompressed database into the source tree and relied on an `rm` step to keep it out of the @@ -96,8 +97,8 @@ run rebuilds the older state. There is no data to migrate. | Symptom | Typical cause | | --- | --- | | Blank page, 404 on assets | `base` in `vite.config.js` does not match the repo name | -| `Failed to fetch database: 404` | Dataset id in `web/src/datasets.js` does not match the file in `db/` | -| A route 404s | `assemble-site.js` did not run, or the id is missing from `DATASETS` | +| `Failed to fetch database: 404` | Dataset id in `datasets.json` does not match the file in `db/` | +| A route 404s | The site step did not run, or the id is missing from `datasets.json` | | WASM fails to load | `sql.js.org` unreachable — self-host `sql-wasm.wasm` and update `SQL_WASM_URL` in `use-sqlite.js` | | Deploy fails on assembly | An uncompressed database artefact reached the output; the error names the files | | Missing rows after a data update | Unknown Excel header — check the per-file row counts the parser prints | diff --git a/docs/system-architecture.md b/docs/system-architecture.md index 803b756..5de66fb 100644 --- a/docs/system-architecture.md +++ b/docs/system-architecture.md @@ -7,20 +7,23 @@ One frontend, one parser, one schema, two datasets. ## Data flow +Each stage is a directory; `data/` and `_site/` are the stores they hand work +through. `assembler/` sequences everything from the parser onwards. + ``` - ▲ crawler/ (Go — manual refresh only, never part of the build) + ▲ crawler/ (Go — manual refresh only, never part of the build) data//*.xls(x) │ - ▼ go-parser/ (Go, one binary, one config per dataset) + ▼ parser/ (Go, one binary, one config per dataset) .build/public/db/.db │ - ▼ gzip -9 (no -k: the raw file does not survive) - .build/public/db/.db.gz + ▼ assembler/ — row count must match datasets.json, then gzip + .build/public/db/.db.gz (the raw .db does not survive) │ - ▼ vite build (root = web/, publicDir = .build/public) + ▼ assembler/ → vite build (root = web/, publicDir = .build/public) web/dist/ │ - ▼ web/scripts/assemble-site.js + ▼ assembler/ — one index.html per dataset; every database must be present _site/ → GitHub Pages │ ▼ browser @@ -32,16 +35,19 @@ data//*.xls(x) One identifier ties the whole pipeline together: ``` -data/2017/ → go-parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.db.gz → /thptqg/2017/ ``` -`web/src/datasets.js` declares the two ids once. The frontend, the database -build (`go-parser/scripts/build-db.js`) and the site assembly all import that -list, so adding a dataset means adding one entry and one config file. +`datasets.json` at the repository root declares the ids once, with the row count +and artifact size the assembler enforces. It is JSON rather than a module +because the assembler is a Go program and the Vite app is not, and JSON is the +only format both parse without a dependency. -That import crosses a package boundary — go-parser reaches into the web -workspace for it. It stays there because the same entries also carry the UI's -labels and SQL presets, and splitting them would mean two lists to keep in step. +Presentation — titles, labels, search examples, SQL presets — stays in +`web/src/datasets.js`, keyed by id. That file cross-checks the two: a registry +entry with no content, or content for a dataset that was never built, throws at +module load rather than rendering a page with no title or a link to a database +that does not exist. | id | Exam | Rows | Source | | --- | --- | --- | --- | @@ -50,7 +56,7 @@ labels and SQL presets, and splitting them would mean two lists to keep in step. ## Canonical schema -Defined once in `go-parser/internal/schema/schema.go` — DDL, INSERT, column order and the 16 +Defined once in `parser/internal/schema/schema.go` — DDL, INSERT, column order and the 16 subject regexes. The two YAML configs carry no SQL at all, only per-dataset parse rules. Config parsing sets `KnownFields(true)`, so a leftover `schema:` block fails loudly instead of looking effective while `schema.go` drives the @@ -178,4 +184,4 @@ total descending. load. Self-hosting `sql-wasm.wasm` and updating `SQL_WASM_URL` in `use-sqlite.js` is the fix. - **Excel format drift.** A new source file with an unseen header layout needs a - new branch in `go-parser/internal/ingest/detect2016.go` or a new config. + new branch in `parser/internal/ingest/detect2016.go` or a new config. diff --git a/go-parser/scripts/build-db.js b/go-parser/scripts/build-db.js deleted file mode 100644 index 51bca9d..0000000 --- a/go-parser/scripts/build-db.js +++ /dev/null @@ -1,124 +0,0 @@ -#!/usr/bin/env node -/** - * Build the SQLite database for one or all datasets, verify it, then gzip it. - * - * The dataset list comes from web/src/datasets.js so it is written in exactly - * one place. - * - * Output goes to .build/public/db/ — the directory Vite copies as its publicDir. - * Only the .gz survives: shipping a 100+ MB uncompressed database is made - * structurally impossible rather than left to a cleanup step. - * - * VERIFICATION IS THE POINT OF THIS SCRIPT, not an extra. - * - * Until now nothing between the parser and the public site asserted that a - * database actually had data in it. The parser logs a file-level failure and - * continues, returns success regardless, and finishes cleanly even at zero rows; - * this script gzipped whatever it got; and web/scripts/assemble-site.js greps - * *filenames* for stray .db files. So a reader that silently under-produced - * would publish a truncated dataset with green CI and no red signal anywhere. - * - * The guard below closes that: a build whose row count does not match the known - * figure, or whose artifact is implausibly small, fails the pipeline. - * - * Usage: - * node go-parser/scripts/build-db.js # all four datasets - * node go-parser/scripts/build-db.js 2017-old # just one - */ - -import { execFileSync } from "node:child_process"; -import { mkdirSync, rmSync, existsSync, statSync } from "node:fs"; -import { dirname, resolve } from "node:path"; -import { fileURLToPath } from "node:url"; -import { DatabaseSync } from "node:sqlite"; - -import { DATASET_IDS, DATASETS } from "../../web/src/datasets.js"; - -const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), "../.."); -const BIN = resolve(ROOT, "go-parser/bin/xlsxread"); -const OUT_DIR = resolve(ROOT, ".build/public/db"); - -/** - * Known-good row counts, from docs/data-pipeline.md. - * - * The inputs are frozen historical exam results, so these are exact, not - * approximate. A deviation of even one row means something changed that nobody - * intended — treat it as a build failure, not a warning. - */ -const EXPECTED_ROWS = { - "2016": 877461, - "2017": 861068, -}; - -/** A gzipped database far below its usual size means a truncated build. */ -const MIN_SIZE_RATIO = 0.9; - -const requested = process.argv.slice(2); -const unknown = requested.filter((id) => !DATASET_IDS.includes(id)); -if (unknown.length) { - console.error(`unknown dataset(s): ${unknown.join(", ")}`); - console.error(`known: ${DATASET_IDS.join(", ")}`); - process.exit(2); -} -const targets = requested.length ? requested : DATASET_IDS; - -if (!existsSync(BIN)) { - console.error(`parser binary not found at ${BIN}`); - console.error("run: npm run build:go"); - process.exit(1); -} - -mkdirSync(OUT_DIR, { recursive: true }); - -for (const id of targets) { - const db = resolve(OUT_DIR, `${id}.db`); - - // execFileSync throws on a non-zero exit, so a parser failure aborts the run. - execFileSync( - BIN, - [ - "build", - "--schema", - resolve(ROOT, `go-parser/configs/${id}.yml`), - "--input", - resolve(ROOT, `data/${id}`), - "--output", - db, - ], - { stdio: "inherit" }, - ); - - // --- guard: the database must contain what it is supposed to contain --- - const expected = EXPECTED_ROWS[id]; - if (expected === undefined) { - console.error(`no expected row count recorded for ${id}; add one to EXPECTED_ROWS`); - process.exit(1); - } - const conn = new DatabaseSync(db, { readOnly: true }); - const actual = conn.prepare("SELECT COUNT(*) c FROM student").get().c; - conn.close(); - if (actual !== expected) { - console.error(`\n${id}: row count ${actual}, expected ${expected}`); - console.error("Refusing to publish — the build did not reproduce the known dataset."); - process.exit(1); - } - console.log(` ✓ ${id}: ${actual} rows (matches expected)`); - - // -9 without -k: the raw .db must not reach the published artifact. - rmSync(`${db}.gz`, { force: true }); - execFileSync("gzip", ["-9", db], { stdio: "inherit" }); - - const gz = `${db}.gz`; - const sizeMb = statSync(gz).size / 1024 / 1024; - const nominal = DATASETS.find((d) => d.id === id)?.dbSizeMb; - if (nominal && sizeMb < nominal * MIN_SIZE_RATIO) { - console.error( - `\n${id}: ${sizeMb.toFixed(1)} MB is below ${(nominal * MIN_SIZE_RATIO).toFixed(1)} MB ` + - `(${MIN_SIZE_RATIO * 100}% of the expected ${nominal} MB)`, - ); - console.error("Refusing to publish — the artifact looks truncated."); - process.exit(1); - } - - console.log(` → db/${id}.db.gz (${sizeMb.toFixed(1)} MB)\n`); -} diff --git a/package.json b/package.json deleted file mode 100644 index 55325af..0000000 --- a/package.json +++ /dev/null @@ -1,36 +0,0 @@ -{ - "name": "thptqg", - "private": true, - "version": "1.0.0", - "type": "module", - "description": "Tra cứu điểm thi THPT Quốc gia — 2016 và 2017", - "workspaces": [ - "web" - ], - "scripts": { - "crawl:2016": "go -C crawler run ./cmd/crawl 2016", - "crawl:2017": "go -C crawler run ./cmd/crawl 2017", - "test:crawler": "go -C crawler test ./...", - "build:go": "go -C go-parser build -o bin/xlsxread ./cmd/xlsxread", - "build:db": "node go-parser/scripts/build-db.js", - "test:go": "go -C go-parser test ./...", - "dev": "npm run dev -w web", - "build": "npm run build -w web", - "preview": "npm run preview -w web", - "assemble": "node web/scripts/assemble-site.js", - "build:site": "npm run build && npm run assemble", - "lint": "eslint ." - }, - "repository": { - "type": "git", - "url": "git+https://github.com/tiennm99/thptqg.git" - }, - "license": "ISC", - "devDependencies": { - "@eslint/js": "^9.39.4", - "eslint": "^9.39.4", - "eslint-plugin-react-hooks": "^7.0.1", - "eslint-plugin-react-refresh": "^0.5.2", - "globals": "^17.4.0" - } -} diff --git a/go-parser/README.md b/parser/README.md similarity index 58% rename from go-parser/README.md rename to parser/README.md index 9d25b24..519afd2 100644 --- a/go-parser/README.md +++ b/parser/README.md @@ -1,17 +1,24 @@ -# go-parser +# parser Reads the `.xls`/`.xlsx` source spreadsheets in `data/` and writes one SQLite database per dataset. ```bash -npm run build:go # compile go-parser/bin/xlsxread -npm run build:db # build + verify + gzip all four datasets -npm run test:go # unit tests + the 299-file reader-fidelity suite +go -C parser build -o bin/xlsxread ./cmd/xlsxread # compile +go -C parser test ./... # unit tests + the reader-fidelity suite ``` ``` -xlsxread build --schema go-parser/configs/.yml --input data/ --output -xlsxread audit --schema go-parser/configs/.yml --input data/ --db +xlsxread build --schema parser/configs/.yml --input data/ --output +xlsxread audit --schema parser/configs/.yml --input data/ --db +``` + +This stage only produces a database. Verifying it against the expected row +count, compressing it and publishing it belong to `assembler/`, which compiles +this binary and drives it per dataset: + +```bash +go -C assembler run ./cmd/assemble db ``` ## Layout @@ -32,16 +39,18 @@ reader independently verifiable against a hash oracle. ## Provenance -This is a port of a Rust crate that lived at `parser/` until the Go -implementation reached full parity. Source comments cite the original by file -and line (`parser/src/transform.rs:56` and similar); those paths resolve at the -tag **`pre-go-parser-removal`**, the last commit containing the Rust code. +This is a port of a Rust crate that occupied this same path until the Go +implementation reached full parity, when it was built alongside as `go-parser/` +and moved back here once the Rust was removed. Source comments cite the original +by file and line (`parser/src/transform.rs:56` and similar) — those refer to the +Rust tree and resolve at the tag **`pre-go-parser-removal`**, the last commit +containing it. The port was gated on a field-by-field comparison of both implementations across -all four datasets — 3,265,641 rows with identical full-table SHA-256, identical -per-column non-NULL counts, identical schema metadata and identical build -stdout. `scripts/differential-parity.mjs` is that comparator and still runs -against any two sets of databases. +the four datasets that existed then — 3,265,641 rows with identical full-table +SHA-256, identical per-column non-NULL counts, identical schema metadata and +identical build stdout. `scripts/differential-parity.mjs` is that comparator and +still runs against any two sets of databases. Behaviour was matched bug-for-bug, deliberately. Several quirks look like defects and are load-bearing for the published data: @@ -61,7 +70,7 @@ Each has a test naming it, so none can be tidied away by accident. `testdata/reader-fidelity-hashes.tsv` holds a SHA-256 per input file over a canonical dump of every cell of every sheet. It is **frozen**: it was produced by the Rust reader, which no longer exists, so it cannot be regenerated. It -still fails if any single cell of any of the 299 files reads differently. +still fails if any single cell of any of the 182 files reads differently. -`npm run build:db` refuses to publish a database whose row count does not match +The assembler refuses to publish a database whose row count does not match the known figure, or whose artifact is under 90% of its usual size. diff --git a/go-parser/cmd/dumpcells/main.go b/parser/cmd/dumpcells/main.go similarity index 97% rename from go-parser/cmd/dumpcells/main.go rename to parser/cmd/dumpcells/main.go index 802e853..e209b49 100644 --- a/go-parser/cmd/dumpcells/main.go +++ b/parser/cmd/dumpcells/main.go @@ -16,7 +16,7 @@ import ( "os" "strings" - "github.com/tiennm99/thptqg/go-parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/reader" ) // escape mirrors the Rust dumper so field separators can never break the format. diff --git a/go-parser/cmd/xlsxread/main.go b/parser/cmd/xlsxread/main.go similarity index 94% rename from go-parser/cmd/xlsxread/main.go rename to parser/cmd/xlsxread/main.go index 1a07b84..cf7dc89 100644 --- a/go-parser/cmd/xlsxread/main.go +++ b/parser/cmd/xlsxread/main.go @@ -16,9 +16,9 @@ import ( "fmt" "os" - "github.com/tiennm99/thptqg/go-parser/internal/audit" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/ingest" + "github.com/tiennm99/thptqg/parser/internal/audit" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/ingest" ) func usage() { diff --git a/go-parser/configs/2016.yml b/parser/configs/2016.yml similarity index 96% rename from go-parser/configs/2016.yml rename to parser/configs/2016.yml index f823d44..b2a938d 100644 --- a/go-parser/configs/2016.yml +++ b/parser/configs/2016.yml @@ -13,7 +13,7 @@ # sheet_mode = "all": several provinces overflow into Sheet2 (65k Excel row cap). # strip_blank_rows = false: no blank-row anomaly observed in this dataset. # -# Table shape, INSERT and subject regexes are canonical — see go-parser/internal/schema/schema.go. +# Table shape, INSERT and subject regexes are canonical — see parser/internal/schema/schema.go. format_detection: thptqg2016 diff --git a/go-parser/configs/2017.yml b/parser/configs/2017.yml similarity index 94% rename from go-parser/configs/2017.yml rename to parser/configs/2017.yml index 49dcff6..ea87f94 100644 --- a/go-parser/configs/2017.yml +++ b/parser/configs/2017.yml @@ -4,7 +4,7 @@ # SBD validation: no numeric guard — build-database.js did not apply ^\d+$. # strip_blank_rows = false: no blank-row anomaly in this dataset. # -# Table shape, INSERT and subject regexes are canonical — see go-parser/internal/schema/schema.go. +# Table shape, INSERT and subject regexes are canonical — see parser/internal/schema/schema.go. reader: sheet_mode: all diff --git a/go-parser/go.mod b/parser/go.mod similarity index 95% rename from go-parser/go.mod rename to parser/go.mod index 23ffa38..7ad9076 100644 --- a/go-parser/go.mod +++ b/parser/go.mod @@ -1,4 +1,4 @@ -module github.com/tiennm99/thptqg/go-parser +module github.com/tiennm99/thptqg/parser go 1.26.5 diff --git a/go-parser/go.sum b/parser/go.sum similarity index 100% rename from go-parser/go.sum rename to parser/go.sum diff --git a/go-parser/internal/audit/audit.go b/parser/internal/audit/audit.go similarity index 94% rename from go-parser/internal/audit/audit.go rename to parser/internal/audit/audit.go index e182207..7185ef5 100644 --- a/go-parser/internal/audit/audit.go +++ b/parser/internal/audit/audit.go @@ -18,10 +18,10 @@ import ( "sort" "strings" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/ingest" - "github.com/tiennm99/thptqg/go-parser/internal/reader" - "github.com/tiennm99/thptqg/go-parser/internal/sqlitedb" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/ingest" + "github.com/tiennm99/thptqg/parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/sqlitedb" ) // Result carries the audit counters. diff --git a/go-parser/internal/config/config.go b/parser/internal/config/config.go similarity index 100% rename from go-parser/internal/config/config.go rename to parser/internal/config/config.go diff --git a/go-parser/internal/config/config_test.go b/parser/internal/config/config_test.go similarity index 97% rename from go-parser/internal/config/config_test.go rename to parser/internal/config/config_test.go index 28f91fd..9e7e195 100644 --- a/go-parser/internal/config/config_test.go +++ b/parser/internal/config/config_test.go @@ -152,7 +152,7 @@ func TestLoadRealConfigs(t *testing.T) { } for id, w := range want { t.Run(id, func(t *testing.T) { - cfg, err := Load(filepath.Join(root, "go-parser", "configs", id+".yml")) + cfg, err := Load(filepath.Join(root, "parser", "configs", id+".yml")) if err != nil { t.Fatalf("load: %v", err) } @@ -214,7 +214,7 @@ func repoRoot(t *testing.T) string { t.Fatalf("getwd: %v", err) } for i := 0; i < 6; i++ { - if _, err := os.Stat(filepath.Join(dir, "go-parser", "configs")); err == nil { + if _, err := os.Stat(filepath.Join(dir, "parser", "configs")); err == nil { return dir } dir = filepath.Dir(dir) diff --git a/go-parser/internal/ingest/detect2016.go b/parser/internal/ingest/detect2016.go similarity index 97% rename from go-parser/internal/ingest/detect2016.go rename to parser/internal/ingest/detect2016.go index 8a5c375..1cf43a2 100644 --- a/go-parser/internal/ingest/detect2016.go +++ b/parser/internal/ingest/detect2016.go @@ -7,10 +7,10 @@ import ( "strconv" "strings" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/reader" - "github.com/tiennm99/thptqg/go-parser/internal/transform" - "github.com/tiennm99/thptqg/go-parser/internal/writer" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/transform" + "github.com/tiennm99/thptqg/parser/internal/writer" ) // The 2016 dataset's 119 files were produced by inconsistent tooling and use diff --git a/go-parser/internal/ingest/detect2016_test.go b/parser/internal/ingest/detect2016_test.go similarity index 99% rename from go-parser/internal/ingest/detect2016_test.go rename to parser/internal/ingest/detect2016_test.go index a8efb73..93c3b48 100644 --- a/go-parser/internal/ingest/detect2016_test.go +++ b/parser/internal/ingest/detect2016_test.go @@ -3,7 +3,7 @@ package ingest import ( "testing" - "github.com/tiennm99/thptqg/go-parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/reader" ) // Ports the 11 tests in parser/src/format_detect_2016.rs, plus a guard per quirk. diff --git a/go-parser/internal/ingest/ingest.go b/parser/internal/ingest/ingest.go similarity index 96% rename from go-parser/internal/ingest/ingest.go rename to parser/internal/ingest/ingest.go index 4d17d04..377a1d0 100644 --- a/go-parser/internal/ingest/ingest.go +++ b/parser/internal/ingest/ingest.go @@ -14,10 +14,10 @@ import ( "sort" "strings" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/reader" - "github.com/tiennm99/thptqg/go-parser/internal/transform" - "github.com/tiennm99/thptqg/go-parser/internal/writer" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/transform" + "github.com/tiennm99/thptqg/parser/internal/writer" ) // IsHeaderRow reports whether row is a header, by matching its uppercased first diff --git a/go-parser/internal/ingest/ingest_test.go b/parser/internal/ingest/ingest_test.go similarity index 98% rename from go-parser/internal/ingest/ingest_test.go rename to parser/internal/ingest/ingest_test.go index 8d06f3d..22a4e06 100644 --- a/go-parser/internal/ingest/ingest_test.go +++ b/parser/internal/ingest/ingest_test.go @@ -5,7 +5,7 @@ import ( "path/filepath" "testing" - "github.com/tiennm99/thptqg/go-parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/reader" ) // Ports the 7 tests in parser/src/reader.rs:116-197. They were listed under diff --git a/go-parser/internal/reader/fidelity_test.go b/parser/internal/reader/fidelity_test.go similarity index 96% rename from go-parser/internal/reader/fidelity_test.go rename to parser/internal/reader/fidelity_test.go index 7389cda..133f096 100644 --- a/go-parser/internal/reader/fidelity_test.go +++ b/parser/internal/reader/fidelity_test.go @@ -10,7 +10,7 @@ import ( "strings" "testing" - "github.com/tiennm99/thptqg/go-parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/reader" ) // TestReaderFidelity asserts the Go reader reproduces calamine byte-for-byte on @@ -33,7 +33,7 @@ func TestReaderFidelity(t *testing.T) { t.Skip("-short: skipping the 299-file corpus sweep") } root := repoRoot(t) - manifest := filepath.Join(root, "go-parser", "testdata", "reader-fidelity-hashes.tsv") + manifest := filepath.Join(root, "parser", "testdata", "reader-fidelity-hashes.tsv") f, err := os.Open(manifest) if err != nil { diff --git a/go-parser/internal/reader/reader.go b/parser/internal/reader/reader.go similarity index 100% rename from go-parser/internal/reader/reader.go rename to parser/internal/reader/reader.go diff --git a/go-parser/internal/reader/xls.go b/parser/internal/reader/xls.go similarity index 100% rename from go-parser/internal/reader/xls.go rename to parser/internal/reader/xls.go diff --git a/go-parser/internal/reader/xlsx.go b/parser/internal/reader/xlsx.go similarity index 100% rename from go-parser/internal/reader/xlsx.go rename to parser/internal/reader/xlsx.go diff --git a/go-parser/internal/reader/xlsx_fixups.go b/parser/internal/reader/xlsx_fixups.go similarity index 100% rename from go-parser/internal/reader/xlsx_fixups.go rename to parser/internal/reader/xlsx_fixups.go diff --git a/go-parser/internal/schema/schema.go b/parser/internal/schema/schema.go similarity index 100% rename from go-parser/internal/schema/schema.go rename to parser/internal/schema/schema.go diff --git a/go-parser/internal/schema/schema_test.go b/parser/internal/schema/schema_test.go similarity index 100% rename from go-parser/internal/schema/schema_test.go rename to parser/internal/schema/schema_test.go diff --git a/go-parser/internal/sqlitedb/sqlitedb.go b/parser/internal/sqlitedb/sqlitedb.go similarity index 100% rename from go-parser/internal/sqlitedb/sqlitedb.go rename to parser/internal/sqlitedb/sqlitedb.go diff --git a/go-parser/internal/transform/ascii_crosscheck_test.go b/parser/internal/transform/ascii_crosscheck_test.go similarity index 93% rename from go-parser/internal/transform/ascii_crosscheck_test.go rename to parser/internal/transform/ascii_crosscheck_test.go index 9e88bca..7ab5279 100644 --- a/go-parser/internal/transform/ascii_crosscheck_test.go +++ b/parser/internal/transform/ascii_crosscheck_test.go @@ -5,8 +5,8 @@ import ( "os" "testing" - _ "github.com/tiennm99/thptqg/go-parser/internal/sqlitedb" - "github.com/tiennm99/thptqg/go-parser/internal/transform" + _ "github.com/tiennm99/thptqg/parser/internal/sqlitedb" + "github.com/tiennm99/thptqg/parser/internal/transform" ) // TestToAsciiAgainstRustOutput cross-checks ToAscii against Rust on real data. diff --git a/go-parser/internal/transform/transform.go b/parser/internal/transform/transform.go similarity index 97% rename from go-parser/internal/transform/transform.go rename to parser/internal/transform/transform.go index c205b2f..92af37b 100644 --- a/go-parser/internal/transform/transform.go +++ b/parser/internal/transform/transform.go @@ -14,9 +14,9 @@ import ( "golang.org/x/text/unicode/norm" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/reader" - "github.com/tiennm99/thptqg/go-parser/internal/schema" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/schema" ) // ToAscii normalises a Vietnamese name to an ASCII slug. diff --git a/go-parser/internal/transform/transform_test.go b/parser/internal/transform/transform_test.go similarity index 98% rename from go-parser/internal/transform/transform_test.go rename to parser/internal/transform/transform_test.go index a509285..0e2e5e2 100644 --- a/go-parser/internal/transform/transform_test.go +++ b/parser/internal/transform/transform_test.go @@ -3,8 +3,8 @@ package transform import ( "testing" - "github.com/tiennm99/thptqg/go-parser/internal/config" - "github.com/tiennm99/thptqg/go-parser/internal/reader" + "github.com/tiennm99/thptqg/parser/internal/config" + "github.com/tiennm99/thptqg/parser/internal/reader" ) // Ports every test in parser/src/transform.rs's test module (:201-409) — 29 in diff --git a/go-parser/internal/writer/writer.go b/parser/internal/writer/writer.go similarity index 96% rename from go-parser/internal/writer/writer.go rename to parser/internal/writer/writer.go index e6b943a..a820759 100644 --- a/go-parser/internal/writer/writer.go +++ b/parser/internal/writer/writer.go @@ -15,9 +15,9 @@ import ( "os" "path/filepath" - "github.com/tiennm99/thptqg/go-parser/internal/schema" - "github.com/tiennm99/thptqg/go-parser/internal/sqlitedb" - "github.com/tiennm99/thptqg/go-parser/internal/transform" + "github.com/tiennm99/thptqg/parser/internal/schema" + "github.com/tiennm99/thptqg/parser/internal/sqlitedb" + "github.com/tiennm99/thptqg/parser/internal/transform" ) // OpenDB deletes any existing database at dbPath, recreates it, and executes the diff --git a/go-parser/scripts/differential-parity.mjs b/parser/scripts/differential-parity.mjs similarity index 100% rename from go-parser/scripts/differential-parity.mjs rename to parser/scripts/differential-parity.mjs diff --git a/go-parser/testdata/reader-fidelity-hashes.tsv b/parser/testdata/reader-fidelity-hashes.tsv similarity index 100% rename from go-parser/testdata/reader-fidelity-hashes.tsv rename to parser/testdata/reader-fidelity-hashes.tsv diff --git a/eslint.config.js b/web/eslint.config.js similarity index 65% rename from eslint.config.js rename to web/eslint.config.js index bbaa88e..809a70e 100644 --- a/eslint.config.js +++ b/web/eslint.config.js @@ -4,10 +4,11 @@ import reactHooks from 'eslint-plugin-react-hooks' import reactRefresh from 'eslint-plugin-react-refresh' import { defineConfig, globalIgnores } from 'eslint/config' +// Scoped to this workspace. The other stages are Go, and the one remaining +// Node file outside it (parser/scripts/differential-parity.mjs, a diagnostic +// run by hand) is out of reach of a config that lives here. export default defineConfig([ - // '**/dist', not 'dist': the build output moved to web/dist when the app - // became a workspace, and a root-anchored pattern would stop matching it. - globalIgnores(['**/dist', '.build', '_site']), + globalIgnores(['dist']), { files: ['**/*.{js,jsx}'], extends: [ @@ -29,9 +30,8 @@ export default defineConfig([ }, }, { - // Node-executed files (Vite config, site assembly, parser tooling) run with - // Node globals. The crawler is Go, so it has nothing here. - files: ['web/vite.config.js', 'web/scripts/**/*.js', 'go-parser/scripts/**/*.js'], + // The Vite config runs under Node, not in the browser. + files: ['vite.config.js'], languageOptions: { globals: { ...globals.node }, }, diff --git a/package-lock.json b/web/package-lock.json similarity index 99% rename from package-lock.json rename to web/package-lock.json index e755a6a..f997505 100644 --- a/package-lock.json +++ b/web/package-lock.json @@ -1,22 +1,27 @@ { - "name": "thptqg", + "name": "thptqg-web", "version": "1.0.0", "lockfileVersion": 3, "requires": true, "packages": { "": { - "name": "thptqg", + "name": "thptqg-web", "version": "1.0.0", - "license": "ISC", - "workspaces": [ - "web" - ], + "dependencies": { + "react": "^19.2.4", + "react-dom": "^19.2.4", + "sql.js": "^1.14.1" + }, "devDependencies": { "@eslint/js": "^9.39.4", + "@types/react": "^19.2.14", + "@types/react-dom": "^19.2.3", + "@vitejs/plugin-react": "^6.0.1", "eslint": "^9.39.4", "eslint-plugin-react-hooks": "^7.0.1", "eslint-plugin-react-refresh": "^0.5.2", - "globals": "^17.4.0" + "globals": "^17.4.0", + "vite": "^8.0.16" } }, "node_modules/@babel/code-frame": { @@ -936,9 +941,9 @@ "license": "MIT" }, "node_modules/baseline-browser-mapping": { - "version": "2.11.13", - "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.11.13.tgz", - "integrity": "sha512-k9HNuUVMlqVjQ9UHzfPjIqiDbWw7WqT1AoT7GL8VwvF3r0ZfArtgiSPAlmupyNquNgOJHTuH4CKYf8ttMTWBTQ==", + "version": "2.11.14", + "resolved": "https://registry.npmjs.org/baseline-browser-mapping/-/baseline-browser-mapping-2.11.14.tgz", + "integrity": "sha512-JyJ954WzuIR8/FFzX0o5krdSTrBAkcCSRfWSleRsIHSWV+cZe2FI1PKggVkFke1hBldRs+LRxUczzE9iPmgZww==", "dev": true, "license": "Apache-2.0", "bin": { @@ -2341,10 +2346,6 @@ "node": ">=8" } }, - "node_modules/thptqg-web": { - "resolved": "web", - "link": true - }, "node_modules/tinyglobby": { "version": "0.2.17", "resolved": "https://registry.npmjs.org/tinyglobby/-/tinyglobby-0.2.17.tgz", @@ -2562,21 +2563,6 @@ "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" } - }, - "web": { - "name": "thptqg-web", - "version": "1.0.0", - "dependencies": { - "react": "^19.2.4", - "react-dom": "^19.2.4", - "sql.js": "^1.14.1" - }, - "devDependencies": { - "@types/react": "^19.2.14", - "@types/react-dom": "^19.2.3", - "@vitejs/plugin-react": "^6.0.1", - "vite": "^8.0.16" - } } } } diff --git a/web/package.json b/web/package.json index 3459e34..e020bfe 100644 --- a/web/package.json +++ b/web/package.json @@ -3,11 +3,12 @@ "private": true, "version": "1.0.0", "type": "module", - "description": "Frontend — one Vite app serving all four datasets and the hub", + "description": "Frontend — one Vite app serving every dataset and the hub", "scripts": { "dev": "vite", "build": "vite build", - "preview": "vite preview" + "preview": "vite preview", + "lint": "eslint ." }, "dependencies": { "react": "^19.2.4", @@ -15,9 +16,14 @@ "sql.js": "^1.14.1" }, "devDependencies": { + "@eslint/js": "^9.39.4", "@types/react": "^19.2.14", "@types/react-dom": "^19.2.3", "@vitejs/plugin-react": "^6.0.1", + "eslint": "^9.39.4", + "eslint-plugin-react-hooks": "^7.0.1", + "eslint-plugin-react-refresh": "^0.5.2", + "globals": "^17.4.0", "vite": "^8.0.16" } } diff --git a/web/scripts/assemble-site.js b/web/scripts/assemble-site.js deleted file mode 100644 index 6907d30..0000000 --- a/web/scripts/assemble-site.js +++ /dev/null @@ -1,80 +0,0 @@ -#!/usr/bin/env node -/** - * Assemble the GitHub Pages artifact from a single Vite build. - * - * The app resolves its dataset from the URL, so every page is the same - * index.html. Because `base` is absolute (/thptqg/), that file references - * /thptqg/assets/... no matter which directory it is served from — so copying - * it to each dataset path produces a real static file at every URL. - * - * GitHub Pages serves those as directory indexes, which is why this needs no - * SPA 404-fallback redirect. That matters beyond tidiness: the usual fallback - * rewrites the URL and would interfere with the ?q= deep links the app relies - * on. - * - * Usage: node web/scripts/assemble-site.js - */ - -import { cpSync, mkdirSync, rmSync, existsSync, readdirSync, statSync } from "node:fs"; -import { dirname, join, resolve } from "node:path"; -import { fileURLToPath } from "node:url"; - -import { DATASET_IDS } from "../src/datasets.js"; - -// The build output belongs to this workspace; the Pages artifact does not. The -// deploy action uploads _site from the repository root, so that one climbs out. -const WEB = resolve(dirname(fileURLToPath(import.meta.url)), ".."); -const DIST = join(WEB, "dist"); -const SITE = join(WEB, "..", "_site"); - -// The two legacy nested URLs (2017/old, 2017/old2) are gone with the datasets -// they addressed. Unknown paths render the hub via 404.html, so old bookmarks -// land somewhere useful rather than on a Pages error page. - -if (!existsSync(join(DIST, "index.html"))) { - console.error(`no build found at ${DIST} — run: npm run build`); - process.exit(1); -} - -rmSync(SITE, { recursive: true, force: true }); -mkdirSync(SITE, { recursive: true }); - -// Base build: index.html, assets/, and the gzipped databases from publicDir. -cpSync(DIST, SITE, { recursive: true }); - -// Unknown paths render the hub rather than the default Pages 404. -cpSync(join(DIST, "index.html"), join(SITE, "404.html")); - -// One entry point per dataset. -for (const path of DATASET_IDS) { - mkdirSync(join(SITE, path), { recursive: true }); - cpSync(join(DIST, "index.html"), join(SITE, path, "index.html")); -} - -// Only gzipped databases may ship. build-db.js gzips without -k so no raw .db -// should exist, but publicDir is copied wholesale — a leftover from an -// interrupted build would go straight through, and a raw database is 100+ MB. -// SQLite also leaves .db-journal files mid-build, so anything that is not a -// .gz is rejected rather than just files ending in .db. -const stray = []; -(function walk(dir) { - for (const entry of readdirSync(dir)) { - const full = join(dir, entry); - if (statSync(full).isDirectory()) walk(full); - else if (/\.db(-journal|-wal|-shm)?$/.test(entry)) stray.push(full); - } -})(SITE); - -if (stray.length) { - console.error("uncompressed database artefact(s) found in the site output:"); - for (const f of stray) { - console.error(` ${f} (${(statSync(f).size / 1048576).toFixed(1)} MB)`); - } - console.error("\nremove them from .build/public/db and re-run"); - process.exit(1); -} - -console.log(`assembled ${SITE}`); -for (const path of ["", "404.html", ...DATASET_IDS]) { - console.log(` /thptqg/${path}`); -} diff --git a/web/src/datasets.js b/web/src/datasets.js index d370b23..07fc3af 100644 --- a/web/src/datasets.js +++ b/web/src/datasets.js @@ -1,64 +1,77 @@ /** - * The four datasets this site serves. + * The datasets this site serves. * * `id` is the single identifier used end to end: * - * data// → go-parser/configs/.yml → db/.db.gz → /thptqg// + * data// → parser/configs/.yml → db/.db.gz → /thptqg// * * Site path and database URL are derived from `id` rather than stored, so a * dataset cannot be misconfigured into pointing at the wrong database. * - * Imported by the Vite app *and* by go-parser/scripts/build-db.js under plain - * Node, so this module must stay free of `import.meta.env` and any Vite-only - * syntax. Callers pass the base URL in explicitly for that reason. - * - * That cross-package import is why this file cannot move into a Vite-only - * corner of the app: go-parser reads DATASET_IDS and dbSizeMb straight from - * here, and reaches across the workspace boundary to do it. + * Which datasets exist is not decided here. That lives in the repository-root + * `datasets.json`, which the assembler reads too — it is a Go program and + * cannot import this module. This file supplies only what the interface needs: + * titles, labels, search examples and SQL presets, keyed by id. */ -// Extension is required: this module is also imported by plain Node -// (go-parser/scripts/build-db.js), which does not resolve extensionless paths. +import registry from "../../datasets.json"; + import { PRESETS_2016, PRESETS_2017 } from "./lib/sql-presets.js"; const SUBTITLE = "Dữ liệu thí sinh toàn quốc · Hỗ trợ truy vấn SQL tùy chỉnh"; -export const DATASETS = [ - { - id: "2016", - year: 2016, +/** Presentation, keyed by the ids declared in datasets.json. */ +const CONTENT = { + 2016: { label: "Kỳ thi 2016", blurb: "877.461 thí sinh", title: "Tra cứu điểm thi THPT Quốc gia 2016", subtitle: SUBTITLE, source: "Bộ GD&ĐT", - dbSizeMb: 44, examples: ["17006021", "Nguyễn Thị Hoa"], presets: PRESETS_2016, }, - { - id: "2017", - year: 2017, + 2017: { label: "Kỳ thi 2017", blurb: "861.068 thí sinh", title: "Tra cứu điểm thi THPT Quốc gia 2017", subtitle: SUBTITLE, source: "baotintuc.vn", - dbSizeMb: 48, examples: ["49008235", "Nguyễn Minh Tiến"], presets: PRESETS_2017, }, -]; +}; + +// A dataset in the registry with no content here would render a page with no +// title and no presets; content here for a dataset that no longer exists would +// offer a link to a database that was never built. Neither should reach a +// browser, so both fail at module load. +for (const { id } of registry.datasets) { + if (!CONTENT[id]) { + throw new Error(`datasets.json declares "${id}" but web/src/datasets.js has no content for it`); + } +} +for (const id of Object.keys(CONTENT)) { + if (!registry.datasets.some((d) => d.id === id)) { + throw new Error(`web/src/datasets.js has content for "${id}", which datasets.json does not declare`); + } +} + +export const DATASETS = registry.datasets.map((d) => ({ + id: d.id, + dbSizeMb: d.dbSizeMb, + ...CONTENT[d.id], +})); /** Dataset IDs in build order. */ export const DATASET_IDS = DATASETS.map((d) => d.id); -/** Site path for a dataset, e.g. pathOf(d, "/thptqg/") → "/thptqg/2017-old/". */ +/** Site path for a dataset, e.g. pathOf(d, "/thptqg/") → "/thptqg/2017/". */ export function pathOf(dataset, base) { return `${base}${dataset.id}/`; } -/** Gzipped database URL, e.g. dbOf(d, "/thptqg/") → "/thptqg/db/2017-old.db.gz". */ +/** Gzipped database URL, e.g. dbOf(d, "/thptqg/") → "/thptqg/db/2017.db.gz". */ export function dbOf(dataset, base) { return `${base}db/${dataset.id}.db.gz`; } diff --git a/web/src/lib/subjects.js b/web/src/lib/subjects.js index 86841e4..ea9efa9 100644 --- a/web/src/lib/subjects.js +++ b/web/src/lib/subjects.js @@ -1,7 +1,7 @@ /** * The 16 subject columns of the canonical schema, in display order. * - * Mirrors `SCORE_FIELDS` in go-parser/internal/schema/schema.go. Previously this list was + * Mirrors `SCORE_FIELDS` in parser/internal/schema/schema.go. Previously this list was * maintained separately in score-table.jsx and student-detail.jsx, which is how * they drifted out of sync with each other and with the database. * diff --git a/web/vite.config.js b/web/vite.config.js index 10b02f6..fa99143 100644 --- a/web/vite.config.js +++ b/web/vite.config.js @@ -11,10 +11,10 @@ import react from "@vitejs/plugin-react"; // and the existing ?q= deep links keep working, which that fallback would break. // // publicDir holds only the gzipped databases, staged there by -// go-parser/scripts/build-db.js. Nothing uncompressed is ever placed in it. +// parser/scripts/build-db.js. Nothing uncompressed is ever placed in it. // // It sits at the repository root rather than inside this workspace because -// go-parser writes it, so the path has to climb out of web/. Vite resolves +// parser writes it, so the path has to climb out of web/. Vite resolves // publicDir against the project root, which is this directory. // // outDir is left at its default, so the build lands in web/dist and this