diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index 426f1ee..3d6ced3 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -162,7 +162,7 @@ jobs: run: | set -euo pipefail for id in $(jq -r '.datasets[].id' datasets.json); do - url="${PAGE_URL%/}/db/${id}.sqlite30" + url="${PAGE_URL%/}/db/${id}.sqlite3" if ! magic=$(curl -sf -r 0-14 -H 'Accept-Encoding: identity;q=1, *;q=0' "$url"); then echo "::error::$url is not fetchable" exit 1 diff --git a/CLAUDE.md b/CLAUDE.md index 7557f03..6ad9ab4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -57,31 +57,30 @@ hashes every real input file. That is the point of it; do not skip it. prerendered to its own file and asset URLs stay absolute, so the copy of `index.html` serving as `404.html` works at any depth. No SPA 404-fallback is used — a fallback would break the `?q=` deep links. -- **`dbSizeMb` in `datasets.json` is a build guard, not just a label.** The - assembler refuses to publish an artifact that falls below a ratio of it. -- **The databases ship uncompressed, as `.sqlite30`.** The trailing 0 is a - chunk index, not a typo — see the next point. The browser reads byte ranges - of the file, and a range of a gzip stream is not a range of the database. -- **The site uses chunked mode over a single chunk, and that is deliberate.** - GitHub Pages gzips `application/octet-stream`, so the HEAD request - `sql.js-httpvfs` sizes a file with reports the compressed length, and the - library refuses to open the file. Chunked mode is the only mode whose config - takes a length (`databaseLengthBytes`); in full mode the worker hardcodes it - to `undefined`, so a length passed there is silently dropped. One chunk holds - the database, so the index is always 0 and every request goes to - `.sqlite3` + `0`. `web/src/lib/db-probe.js` supplies the length by - reading the file header over a range request. -- **Ranged reads were never affected by the compression**, because browsers - must send `Accept-Encoding: identity` whenever a request carries a `Range` - header. Verify the way a browser asks — - `curl -s -r 0-14 -H 'Accept-Encoding: identity' …` must print - `SQLite format 3` — never a bare `curl -sI`, which advertises no encoding - and so passes whatever the host does. -- **Every query the site runs must be index-driven.** Over range requests an - unindexed query fetches the whole table. Hence no index on `ho_ten` (nothing - can use one), `name_word` for name search, partial indexes for the score - presets, and the footer count read from `datasets.json` instead of - `COUNT(*)`. +- **`dbSizeMb` in `datasets.json` is a build guard and a user-facing figure.** + The assembler refuses to publish an artifact that falls below a ratio of it, + and the download gate shows it as the memory the tab will need. +- **The browser downloads the whole database and queries it in memory.** The + dataset page is gated behind that download: `download-gate.svelte` states the + transfer and memory cost and offers nothing but the download, because there + is no answer to give without it. `sql.js` holds the file in WebAssembly + memory for as long as the tab is open, so the memory figure is RAM, not disk. +- **The download is kept in Cache Storage, versioned by ETag** (`db-cache.js`). + A later visit opens the stored copy without asking, since consent was given + once and reuse costs no network; a redeploy changes the ETag, so the new + version replaces the old rather than being served stale. When the server + cannot be reached at all, any stored version is used — which is what lets the + site answer offline. +- **The schema carries no secondary indexes, deliberately.** An index saves a + scan that already takes a few hundred milliseconds in memory, and costs every + visitor megabytes of download. An earlier design read the file over HTTP + range requests and needed the opposite — a `name_word` table with one row per + word of every name, plus partial score indexes — which was more than half the + published file: 288 MB became 142 MB when they went. +- **The databases ship uncompressed, as `.sqlite3`.** GitHub Pages gzips + them on the wire anyway, which is where the transfer figure comes from + (142 MB stored, 31 MB delivered), so publishing a `.gz` would only mean + decompressing twice. ## Conventions diff --git a/README.md b/README.md index 4e0f15c..951e28f 100644 --- a/README.md +++ b/README.md @@ -42,7 +42,7 @@ app both read it and neither needs a dependency to do so; presentation stays in The dataset id is one identifier end to end: ``` -data/2017/ → parser/configs/2017.yml → db/2017.sqlite30 → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ ``` ## Build diff --git a/assembler/internal/databases/databases.go b/assembler/internal/databases/databases.go index 6cefeb4..7e30b86 100644 --- a/assembler/internal/databases/databases.go +++ b/assembler/internal/databases/databases.go @@ -35,18 +35,10 @@ const driverName = "sqlite" // even if the row count somehow passed. const minSizeRatio = 0.9 -// Extension is the published suffix. The trailing 0 is a chunk index, not a -// typo: the browser reads the file through sql.js-httpvfs in chunked mode, -// which is the only mode whose config accepts the file's length. That mode -// builds each request's URL as urlPrefix + chunkIndex, and one chunk holds the -// whole database, so the index is always 0 and the prefix is ".sqlite3". -// -// The length has to come from the config because the library otherwise takes -// it from a HEAD request, which GitHub Pages answers with the gzipped size. -// -// Not ".db" either: keeping that name free lets the site assembly treat any -// stray .db or SQLite journal in the output as the leftover it is. -const Extension = ".sqlite30" +// Extension is the published suffix. Not ".db": keeping that name free lets +// the site assembly treat any stray .db or SQLite journal in the output as the +// leftover it is. +const Extension = ".sqlite3" // Paths locates the pieces this package needs. type Paths struct { diff --git a/assembler/internal/databases/databases_test.go b/assembler/internal/databases/databases_test.go index 7086165..b14aada 100644 --- a/assembler/internal/databases/databases_test.go +++ b/assembler/internal/databases/databases_test.go @@ -12,9 +12,9 @@ import ( func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { dir := t.TempDir() for _, name := range []string{ - "2016.sqlite30", "2017.sqlite30", - "2017-old.sqlite30", // dropped from the registry - "2017-old2.sqlite30", // dropped from the registry + "2016.sqlite3", "2017.sqlite3", + "2017-old.sqlite3", // dropped from the registry + "2017-old2.sqlite3", // dropped from the registry "2016.db-journal", // interrupted run } { if err := os.WriteFile(filepath.Join(dir, name), []byte("x"), 0o644); err != nil { @@ -38,7 +38,7 @@ func TestCleanRemovesOnlyDroppedDatasets(t *testing.T) { } slices.Sort(left) - want := []string{"2016.sqlite30", "2017.sqlite30"} + want := []string{"2016.sqlite3", "2017.sqlite3"} if !slices.Equal(left, want) { t.Errorf("left %v, want %v", left, want) } diff --git a/assembler/internal/site/site.go b/assembler/internal/site/site.go index eb8a010..dbca298 100644 --- a/assembler/internal/site/site.go +++ b/assembler/internal/site/site.go @@ -147,11 +147,10 @@ func checkDatabasesPresent(siteDir string, datasets []registry.Dataset) error { return nil } -// strayArtifact matches what must never reach the output: a SQLite journal from -// an interrupted run, a database under either older name — .db, or .sqlite3 -// without the chunk index the client asks for — or a gzipped database from -// before the switch to range requests. -var strayArtifact = regexp.MustCompile(`(\.db|\.sqlite30?)(-journal|-wal|-shm)$|\.db$|\.sqlite3$|\.gz$`) +// strayArtifact matches what must never reach the output: a SQLite journal +// from an interrupted run, a database under the old .db name, or one under the +// .sqlite30 name a former range-request client asked for. Each is 100+ MB. +var strayArtifact = regexp.MustCompile(`(\.db|\.sqlite30?)(-journal|-wal|-shm)$|\.db$|\.sqlite30$|\.gz$`) // checkNoStrayArtifacts rejects leftovers that would be published. // diff --git a/assembler/internal/site/site_test.go b/assembler/internal/site/site_test.go index 56c2902..5b524be 100644 --- a/assembler/internal/site/site_test.go +++ b/assembler/internal/site/site_test.go @@ -39,7 +39,7 @@ func write(t *testing.T, path, body string) { } func TestAssembleProducesAPageForEveryDataset(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } @@ -49,7 +49,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { filepath.Join("2016", "index.html"), filepath.Join("2017", "index.html"), filepath.Join("_app", "immutable", "entry.js"), - filepath.Join("db", "2016.sqlite30"), + filepath.Join("db", "2016.sqlite3"), } { if _, err := os.Stat(filepath.Join(p.Site, want)); err != nil { t.Errorf("missing from the artifact: %s", want) @@ -61,7 +61,7 @@ func TestAssembleProducesAPageForEveryDataset(t *testing.T) { // entry generator, which reads the same registry this does. If the two fall out // of step, that dataset's URL 404s — so the build stops instead. func TestMissingDatasetPageFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := os.RemoveAll(filepath.Join(p.Dist, "2017")); err != nil { t.Fatal(err) } @@ -80,12 +80,12 @@ func TestMissingDatasetPageFailsTheBuild(t *testing.T) { // renders, every query 404s, and CI stays green. The row-count and size guards // cannot catch this — they only run when a database was built at all. func TestMissingDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30") // 2017 never built + p := fakeBuild(t, "2016.sqlite3") // 2017 never built err := Assemble(p, datasets) if err == nil { t.Fatal("expected an error when a database is missing") } - if !strings.Contains(err.Error(), "2017.sqlite30") { + if !strings.Contains(err.Error(), "2017.sqlite3") { t.Errorf("the error should name the missing database, got: %v", err) } } @@ -93,25 +93,23 @@ func TestMissingDatabaseFailsTheBuild(t *testing.T) { // TestEmptyDatabaseFailsTheBuild: a zero-byte file satisfies "exists" but is // not a database. func TestEmptyDatabaseFailsTheBuild(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") - write(t, filepath.Join(p.Dist, "db", "2017.sqlite30"), "") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") + write(t, filepath.Join(p.Dist, "db", "2017.sqlite3"), "") if err := Assemble(p, datasets); err == nil { t.Fatal("expected an error for a zero-byte database") } } // TestStrayArtifactFailsTheBuild: a journal means an interrupted run, a .db or -// a chunk-index-less .sqlite3 means an older naming the client no longer asks -// for, a .gz means a database the site could not read a range of — and each is -// 100+ MB. +// a .sqlite30 means a name from an earlier design that no client asks for now, +// a .gz means a database left compressed — and each is 100+ MB. func TestStrayArtifactFailsTheBuild(t *testing.T) { for _, name := range []string{ - "2016.db", "2016.sqlite3", + "2016.db", "2016.sqlite30", "2016.sqlite3-journal", "2016.sqlite3-wal", "2016.sqlite3-shm", "2016.sqlite3.gz", - "2016.sqlite30-journal", "2016.sqlite30-wal", "2016.sqlite30-shm", } { t.Run(name, func(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") write(t, filepath.Join(p.Dist, "db", name), "raw sqlite") err := Assemble(p, datasets) if err == nil { @@ -127,12 +125,12 @@ func TestStrayArtifactFailsTheBuild(t *testing.T) { // TestPublishedDatabaseIsNotMistakenForStray: the pattern must pass the one // file the site is built to serve. Getting this wrong would fail every build. func TestPublishedDatabaseIsNotMistakenForStray(t *testing.T) { - if strayArtifact.MatchString("2016.sqlite30") { + if strayArtifact.MatchString("2016.sqlite3") { t.Error("the published database must not be treated as a stray artifact") } for _, name := range []string{ - "2016.db", "x.sqlite3", "x.sqlite3-journal", "x.sqlite3-wal", "x.sqlite3-shm", - "x.sqlite30-journal", "x.sqlite30-wal", "x.sqlite30-shm", "x.sqlite3.gz", "x.db.gz", + "2016.db", "x.sqlite30", "x.sqlite3-journal", "x.sqlite3-wal", "x.sqlite3-shm", + "x.sqlite30-journal", "x.sqlite3.gz", "x.db.gz", } { if !strayArtifact.MatchString(name) { t.Errorf("%s should be rejected", name) @@ -151,7 +149,7 @@ func TestAssembleRejectsAMissingBuild(t *testing.T) { // TestAssembleIsIdempotent: the site directory is rebuilt from scratch, so a // previous run's leftovers cannot survive into the artifact. func TestAssembleIsIdempotent(t *testing.T) { - p := fakeBuild(t, "2016.sqlite30", "2017.sqlite30") + p := fakeBuild(t, "2016.sqlite3", "2017.sqlite3") if err := Assemble(p, datasets); err != nil { t.Fatal(err) } diff --git a/assembler/internal/verify/verify.go b/assembler/internal/verify/verify.go index 2964a63..567a333 100644 --- a/assembler/internal/verify/verify.go +++ b/assembler/internal/verify/verify.go @@ -135,7 +135,7 @@ func compareOne(id, dirA, dirB string) (Result, error) { return res, nil } -// open finds .sqlite30 in dir and returns a read-only handle. +// open finds .sqlite3 in dir and returns a read-only handle. // // The older names are still accepted, and a gzipped database is expanded to a // temporary file rather than rejected: the two sides of a comparison are often @@ -143,7 +143,7 @@ func compareOne(id, dirA, dirB string) (Result, error) { func open(dir, id string) (*sql.DB, func(), error) { noop := func() {} - for _, name := range []string{id + ".sqlite30", id + ".sqlite3", id + ".db"} { + for _, name := range []string{id + ".sqlite3", id + ".sqlite30", id + ".db"} { plain := filepath.Join(dir, name) if _, err := os.Stat(plain); err == nil { db, err := sql.Open(driverName, "file:"+plain+"?mode=ro") @@ -154,7 +154,7 @@ func open(dir, id string) (*sql.DB, func(), error) { gzPath := filepath.Join(dir, id+".db.gz") f, err := os.Open(gzPath) if err != nil { - return nil, noop, fmt.Errorf("no %s.sqlite30, %s.sqlite3, %s.db or %s.db.gz in %s", id, id, id, id, dir) + return nil, noop, fmt.Errorf("no %s.sqlite3, %s.sqlite30, %s.db or %s.db.gz in %s", id, id, id, id, dir) } defer f.Close() diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index 6a029e6..d33cba6 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -199,7 +199,7 @@ exists. ## Verifying a rebuild The assembler verifies itself: each database's row count must match the -figure in the table above, and each `.sqlite30` must be at least 90% of its usual +figure in the table above, and each `.sqlite3` must be at least 90% of its usual size, or the build fails rather than publishing. That guard is the reason a truncated dataset cannot reach the site with a green pipeline. @@ -213,8 +213,8 @@ go -C assembler run ./cmd/assemble db # rebuild go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db ``` -Each side is a directory of `.sqlite30`; a gzipped database from before the -switch to range requests is still expanded to a temporary file automatically. It exits non-zero on any mismatch, +Each side is a directory of `.sqlite3`; a database under an older name, or +a gzipped one, is still opened automatically. It exits non-zero on any mismatch, names the first differing rows and columns, and fails rather than skipping when a dataset is absent from either side — silently comparing one of two datasets is how a gate passes without proving anything. diff --git a/docs/deployment-guide.md b/docs/deployment-guide.md index 3c653da..81ff3ca 100644 --- a/docs/deployment-guide.md +++ b/docs/deployment-guide.md @@ -106,26 +106,21 @@ from publishing something it did not mean to. files committed to a repository; the databases are built in CI and uploaded as a Pages artifact, and the documented Pages limits are a 1 GB published site and 100 GB/month of bandwidth, with no per-file figure. The two - databases are 302 MB and 247 MB. -- **Total artifact is about 552 MB**, inside the 1 GB site limit but with less - headroom than before: a third dataset of this size would not fit. The fallback - is `sql.js-httpvfs`'s chunked mode, which splits a database into parts. -- **Pages does compress the databases, and that is survivable.** The extension - is unknown to Pages, so the file is served as `application/octet-stream`, - which is marked compressible in `mime-db` and gzipped: a plain request - returns `Content-Encoding: gzip` and the compressed length. Ranged reads are - not affected, because the Fetch standard makes browsers send - `Accept-Encoding: identity` on any request carrying a `Range` header. Only - the length probe breaks, so the site supplies the length itself instead of - trusting HEAD — see `web/src/lib/db-probe.js` and the chunked-mode note in - [system-architecture](./system-architecture.md). -- **Verify the way a browser asks.** A bare `curl -sI` advertises no encoding - and so reports success whatever the host does; it is what let this reach - production. Check ranged reads instead, and check the bytes, not the headers: + databases are 142 MB and 119 MB. +- **Total artifact is about 263 MB**, comfortably inside the 1 GB site limit. + Dropping the search index the range-request design needed halved both files. +- **Pages compresses the databases on the wire, which is a benefit here.** The + extension is unknown to Pages, so the file is served as + `application/octet-stream`, which `mime-db` marks compressible, and the + browser decompresses it transparently: 142 MB stored becomes about 31 MB + delivered for 2016, and 119 MB becomes about 36 MB for 2017. Note that the + smaller database is the larger download: the two compress differently, so + neither transfer figure can be derived from the stored size. +- **Verify the bytes, not the headers.** A bare `curl -sI` reports success + whatever the host does. Read the file's first bytes instead: ```bash - curl -s -r 0-15 -H 'Accept-Encoding: identity;q=1, *;q=0' \ - https://.github.io/thptqg/db/2016.sqlite30 | head -c 16 + curl -s -r 0-15 https://.github.io/thptqg/db/2016.sqlite3 | head -c 16 # must print: SQLite format 3 ``` @@ -141,9 +136,8 @@ run rebuilds the older state. There is no data to migrate. | Blank page, 404 on assets | `paths.base` in `svelte.config.js` does not match the repo name | | `Failed to fetch database: 404` | Dataset id in `datasets.json` does not match the file in `db/` | | A route 404s | The site step did not run, or the id is missing from `datasets.json` | -| `Length of the file not known` | The host gzipped the un-ranged response, so HEAD reports the compressed size. The site supplies `databaseLengthBytes` from a range probe; if this returns, the config is no longer reaching the worker in chunked mode | -| Database fails to open | A ranged read did not return raw database bytes. The range check above must print `SQLite format 3` | -| Every query is slow or huge | It is not using an index. `EXPLAIN QUERY PLAN` it: a `SCAN` means the browser is fetching the whole table | -| Deploy fails on assembly | An uncompressed database artefact reached the output; the error names the files | +| Download gate never finishes | The file is not being served, or the tab ran out of memory holding it. The check above must print `SQLite format 3` | +| Tab crashes on a phone | The database needs 142 MB of memory for 2016, 119 MB for 2017; a low-memory device may have the tab killed | +| Deploy fails on assembly | A stray database artefact reached the output — a journal, a `.gz`, or a name from an earlier design; the error names the files | | Missing rows after a data update | Unknown Excel header — check the per-file row counts the parser prints | | Visitor still sees old data after a deploy | Their stored copy is keyed by ETag, so this should not happen. Confirm the deployed file's `ETag` header actually changed | diff --git a/docs/system-architecture.md b/docs/system-architecture.md index 54a5dbd..e4b0c30 100644 --- a/docs/system-architecture.md +++ b/docs/system-architecture.md @@ -24,11 +24,11 @@ through. `assembler/` sequences everything from the parser onwards. data//*.xls(x) │ ▼ parser/ (Go, one binary, one config per dataset) - .build/public/db/.sqlite30 + .build/public/db/.sqlite3 │ ▼ assembler/ — row count and size must match datasets.json - .build/public/db/.sqlite30 (uncompressed: ranges of a gzip stream - │ are not ranges of the database) + .build/public/db/.sqlite3 (uncompressed: the host gzips it on the + │ wire, so a .gz would decompress twice) ▼ assembler/ → npm run build (SvelteKit static, assets = .build/public) web/dist/ │ @@ -71,7 +71,7 @@ figure is reported separately as the transfer size. One identifier ties the whole pipeline together: ``` -data/2017/ → parser/configs/2017.yml → db/2017.sqlite30 → /thptqg/2017/ +data/2017/ → parser/configs/2017.yml → db/2017.sqlite3 → /thptqg/2017/ ``` `datasets.json` at the repository root declares the ids once, with the row count @@ -204,57 +204,56 @@ total descending. | Concern | Choice | Rationale | | --- | --- | --- | -| Storage | Static SQLite file, read by range request | No backend; the datasets are frozen, and a lookup needs a few pages of them | -| Reading mode | `serverMode: "chunked"` over a single chunk | The only mode whose config accepts the file length. In full mode the worker hardcodes it to `undefined` and falls back to a HEAD request, which Pages answers with the gzipped size. One chunk means the index is always 0, hence the published name `.sqlite30` | -| Compression | None | A byte range of a gzip stream is not a byte range of the database | -| WASM hosting | Bundled with the app | `sql.js-httpvfs` ships its own build; one less third-party runtime dependency | -| Diacritics search | Pre-computed `ho_ten_ascii`, indexed word by word | `LOWER(REPLACE(...))` at query time defeats the index, and `LIKE '%x%'` reads the whole table | -| Row count in the footer | Read from `datasets.json` | `COUNT(*)` scans an index — 20 MB over range requests | -| Page size | 1 KiB, matched by `requestChunkSize` | One HTTP request is one page; a row fetched by seek costs 1 KB rather than 4 KB, for about 5% more file | -| SQL safety | Leading-keyword allowlist | `sql.js` is in-memory so writes cannot persist; the allowlist prevents confusion | +| Storage | Static SQLite file, downloaded whole | No backend; the datasets are frozen, and one large transfer is something browsers and CDNs are both good at | +| Download gate | Blocking, no dismiss | The page has no answers before the file arrives, and 31 MB of someone's mobile data should be asked for rather than spent silently | +| Keeping the download | Cache Storage, versioned by ETag | The transfer is paid once per device instead of once per visit; the ETag is what stops a redeploy being answered with last week's data | +| Searching | On submit, never while typing | A name search scans every row, so a keystroke-per-search would run hundreds of full scans to answer one question | +| Compression | None published | The host gzips on the wire, so a `.gz` artifact would only be decompressed twice | +| WASM hosting | Bundled with the app | `sql.js` ships its own build; one less third-party runtime dependency | +| Diacritics search | Pre-computed `ho_ten_ascii` | `LOWER(REPLACE(...))` at query time is far slower over 877,460 rows than a column computed once at build | +| Secondary indexes | None | In memory a full scan costs a few hundred milliseconds; an index costs every visitor megabytes of download. The name index alone was 146 MB | +| Row count in the footer | Read from `datasets.json` | Costs nothing and is the same number the assembler enforces | +| Page size | 4 KiB | SQLite's default, and nothing on the client depends on it any more | +| SQL safety | Leading-keyword allowlist | The copy is the visitor's own, so this guards their session against a typo rather than protecting data | | Row caps | 100 (lookup), 1000 (SQL) | Keeps DOM render sizes reasonable | | Routing | SvelteKit file routes, prerendered | Each dataset gets a real HTML file with its own title | | Styling | Tailwind, with tier colours as CSS variables | Tier classes are chosen at runtime, which no utility generator can see | ### Considered and not taken -- **Splitting the database into several chunks.** The site uses chunked mode, - but over one chunk (see above). Real splitting would let a CDN cache each part - whole; GitHub Pages serves everything with `Cache-Control: max-age=600`, and - every rebuild relays SQLite's pages so the file changes even when the data - does not, so that caching is cancelled by the host. Worth revisiting behind a - CDN with long TTLs, and it is the fallback if a single 300 MB file ever - becomes a problem. -- **`sqlite-wasm-http`.** Maintained, and built on the official SQLite WASM - rather than a 2022 fork, which is the better long-term footing. It does not - help here: its worker sizes the file from a HEAD request's `Content-Length` - exactly as `sql.js-httpvfs` does, and its `Options` has no field for the - length, so on Pages it would silently take the gzipped size instead of - failing. Its shared-cache backend needs COOP/COEP, which Pages cannot send, - but it ships a fallback backend that does not — so isolation is not the - blocker, the missing length option is. -- **Substring name search.** `LIKE '%x%'` cannot use an index, so it read the - whole 127 MB table. `name_word` keeps search by any word of a name without - it. +- **Reading the file over HTTP range requests** (`sql.js-httpvfs`), which this + site did until it was measured. Two costs killed it. Reads are serial — + the worker uses synchronous XHR — so a name search that touched 390 pages + waited 17 seconds to move 608 KB, and roughly one request per result row is a + floor no page size removes. And the first visitor after each deploy waited + ~26 s for the CDN to fill its cache with a 288 MB object. The download pays + once, up front, visibly. +- **`sqlite-wasm-http`.** Built on the official SQLite WASM and maintained, + but it sizes the file from a HEAD request's `Content-Length` and exposes no + option to override it, so on a host that gzips it would silently use the + compressed size. Same class of problem, less recourse. +- **DuckDB-WASM over Parquet.** Genuinely maintained, async, parallel range + requests. Its binaries are 32–37 MB before the Parquet extension, which is + more than the entire database download for a phone looking up one score. +- **Static pre-generated shards**, one file per exam-number bucket. The most + robust option and the fastest single lookup, but it cannot answer arbitrary + SQL, and a static file cannot stop early the way `LIMIT` does — a common + Vietnamese surname would mean fetching a very large posting list. ## Risks and limitations -- **Unindexed queries are expensive.** The SQL tab can express a query that - walks the table, which over range requests means fetching 100+ MB. A byte - budget stops one before it gets that far, and the tab warns before it opens. -- **`Content-Encoding` on a ranged response would break everything.** A range - of a compressed body addresses the wrong bytes. In practice browsers prevent - it: the Fetch standard requires `Accept-Encoding: identity` on any request - carrying a `Range` header. GitHub Pages *does* gzip the un-ranged response — - `application/octet-stream` is compressible in `mime-db` — which is why the - file length is probed with a range request and passed as - `databaseLengthBytes` rather than left to the library's HEAD. - `db-probe.js` checks the returned bytes - start with the SQLite magic, so a host that ever compresses a ranged response - fails loudly instead of returning nonsense. -- **`sql.js-httpvfs` is unmaintained** (0.8.12, September 2022) and ships its - own SQLite WASM. `sqlite-wasm-http`, on the official build, is the fallback. -- **Hosted size.** 552 MB for both datasets against the 1 GB GitHub Pages +- **Memory is the binding constraint.** The database lives in the tab's + WebAssembly memory for as long as the page is open: 142 MB for 2016, 119 MB + for 2017. A low-memory phone may have the tab killed, which is why the gate + states the figure before the download starts. +- **A deploy costs returning visitors the transfer again.** Every rebuild lays + SQLite pages out differently, so the file — and its ETag — changes even when + the data does not. The stored copy is then a stale version and is replaced. +- **The stored copy can be evicted.** Cache Storage is subject to the browser's + own storage pressure, so a device short on disk falls back to downloading. +- **A visitor who will not download cannot use the site.** That is the + deliberate shape of the gate, and it makes the first impression a 31 MB ask. +- **Hosted size.** 526 MB for both datasets against the 1 GB GitHub Pages limit; a third dataset of this size would not fit. - **Excel format drift.** A new source file with an unseen header layout needs a new branch in `parser/internal/ingest/detect2016.go` or a new config. diff --git a/web/src/lib/datasets.js b/web/src/lib/datasets.js index 16e0f94..3339d72 100644 --- a/web/src/lib/datasets.js +++ b/web/src/lib/datasets.js @@ -99,39 +99,12 @@ export function pathOf(dataset, base) { } /** - * The chunk the whole database lives in. One chunk covers the file, so this is - * the only index the library ever asks for, and the assembler publishes the - * database under a name ending in it. - */ -const CHUNK_INDEX = "0"; - -/** - * Everything a database URL has except the chunk index, e.g. - * dbPrefixOf(d, "/thptqg") → "/thptqg/db/2017.sqlite3". - * - * sql.js-httpvfs reads the file in chunked mode and builds each URL as - * prefix + chunk index. One chunk holds the whole database, so the index is - * always 0 and the published file is ".sqlite30". - */ -export function dbPrefixOf(dataset, base) { - return `${base}/db/${dataset.id}.sqlite3`; -} - -/** - * The published database file, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.sqlite30". + * The published database, e.g. dbOf(d, "/thptqg") → "/thptqg/db/2017.sqlite3". * * The browser downloads this whole file and queries it in memory. The server * compresses it on the wire, which is why the transfer is a third of the size * the registry records. */ export function dbOf(dataset, base) { - return `${dbPrefixOf(dataset, base)}${CHUNK_INDEX}`; -} - -/** - * Both forms of the database's location, for RemoteDatabase: the file to read, - * and the prefix the library appends the chunk index to. - */ -export function dbSourceOf(dataset, base) { - return { url: dbOf(dataset, base), urlPrefix: dbPrefixOf(dataset, base) }; + return `${base}/db/${dataset.id}.sqlite3`; } diff --git a/web/src/lib/datasets.test.js b/web/src/lib/datasets.test.js index 35e690e..bd7627b 100644 --- a/web/src/lib/datasets.test.js +++ b/web/src/lib/datasets.test.js @@ -1,28 +1,14 @@ import { describe, expect, it } from "vitest"; -import { DATASETS, dbOf, dbPrefixOf, dbSourceOf } from "./datasets.js"; +import { DATASETS, dbOf, pathOf } from "./datasets.js"; const [dataset] = DATASETS; -/** - * sql.js-httpvfs builds every request URL as urlPrefix + chunk index, and one - * chunk holds the whole database, so the index is always 0. If these two ever - * stop agreeing, the library asks for a file the assembler never published and - * every query 404s — which no other test would catch. - */ -describe("database location", () => { - it("puts the chunk index where the published file name ends", () => { - expect(dbOf(dataset, "/thptqg")).toBe(`${dbPrefixOf(dataset, "/thptqg")}0`); +describe("dataset locations", () => { + it("derives the site path from the id", () => { + expect(pathOf(dataset, "/thptqg")).toBe(`/thptqg/${dataset.id}/`); }); - it("names the file the assembler publishes", () => { - expect(dbOf(dataset, "/thptqg")).toBe(`/thptqg/db/${dataset.id}.sqlite30`); - }); - - it("hands RemoteDatabase both forms of the same location", () => { - const source = dbSourceOf(dataset, "/thptqg"); - expect(source).toEqual({ - url: dbOf(dataset, "/thptqg"), - urlPrefix: dbPrefixOf(dataset, "/thptqg"), - }); + it("names the database file the assembler publishes", () => { + expect(dbOf(dataset, "/thptqg")).toBe(`/thptqg/db/${dataset.id}.sqlite3`); }); }); diff --git a/web/src/lib/db-probe.test.js b/web/src/lib/db-probe.test.js deleted file mode 100644 index 7a25348..0000000 --- a/web/src/lib/db-probe.test.js +++ /dev/null @@ -1,90 +0,0 @@ -import { describe, expect, it, vi } from "vitest"; -import { looksLikeSqlite, parseTotalBytes, probeDatabase, readPageSize } from "./db-probe.js"; - -/** The first bytes of a real database: magic, NUL, then the page size at 16. */ -function header(pageSize = 1024, magic = "SQLite format 3") { - const bytes = new Uint8Array(100); - for (let i = 0; i < magic.length; i += 1) bytes[i] = magic.charCodeAt(i); - bytes[16] = pageSize >> 8; - bytes[17] = pageSize & 0xff; - return bytes; -} - -function respond(bytes, { status = 206, contentRange = `bytes 0-99/${317096960}` } = {}) { - return { - status, - headers: { get: (name) => (name.toLowerCase() === "content-range" ? contentRange : null) }, - arrayBuffer: async () => bytes.buffer, - }; -} - -describe("parseTotalBytes", () => { - it("takes the total from a Content-Range", () => { - expect(parseTotalBytes("bytes 0-99/317096960")).toBe(317096960); - }); - - it("refuses an unknown total", () => { - expect(() => parseTotalBytes("bytes 0-99/*")).toThrow(/how large/); - expect(() => parseTotalBytes(null)).toThrow(/absent/); - }); -}); - -describe("readPageSize", () => { - it("reads the two big-endian bytes at offset 16", () => { - expect(readPageSize(header(1024))).toBe(1024); - expect(readPageSize(header(4096))).toBe(4096); - }); - - it("treats 1 as 65536, as the file format does", () => { - expect(readPageSize(header(1))).toBe(65536); - }); -}); - -describe("looksLikeSqlite", () => { - it("accepts a real header", () => { - expect(looksLikeSqlite(header())).toBe(true); - }); - - it("rejects a gzip stream, which is what a compressing host returns", () => { - const gzip = new Uint8Array([0x1f, 0x8b, 0x08, 0x00, 0x00, 0x00, 0x00, 0x00]); - expect(looksLikeSqlite(gzip)).toBe(false); - }); - - it("requires the NUL that terminates the magic", () => { - expect(looksLikeSqlite(header(1024, "SQLite format 3x"))).toBe(false); - }); -}); - -describe("probeDatabase", () => { - it("asks for the header by range and returns the total length", async () => { - const fetchImpl = vi.fn(async () => respond(header())); - const total = await probeDatabase("/db/2016.sqlite30", 1024, fetchImpl); - - expect(total).toBe(317096960); - expect(fetchImpl).toHaveBeenCalledWith("/db/2016.sqlite30", { - headers: { Range: "bytes=0-99" }, - }); - }); - - it("fails when the host ignores the range", async () => { - const fetchImpl = async () => respond(header(), { status: 200 }); - await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).rejects.toThrow(/expected 206/); - }); - - it("fails when the bytes are not a database", async () => { - const gzip = new Uint8Array([0x1f, 0x8b, 0x08]); - const fetchImpl = async () => respond(gzip); - await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).rejects.toThrow( - /not a SQLite header/, - ); - }); - - it("warns, but continues, when the page size is not the request size", async () => { - const warn = vi.spyOn(console, "warn").mockImplementation(() => {}); - const fetchImpl = async () => respond(header(4096)); - - await expect(probeDatabase("/db/2016.sqlite30", 1024, fetchImpl)).resolves.toBe(317096960); - expect(warn).toHaveBeenCalledWith(expect.stringMatching(/page size 4096/)); - warn.mockRestore(); - }); -}); diff --git a/web/src/lib/sqlite.svelte.js b/web/src/lib/sqlite.svelte.js deleted file mode 100644 index c127153..0000000 --- a/web/src/lib/sqlite.svelte.js +++ /dev/null @@ -1,191 +0,0 @@ -import { createDbWorker } from "sql.js-httpvfs"; -import workerUrl from "sql.js-httpvfs/dist/sqlite.worker.js?url"; -import wasmUrl from "sql.js-httpvfs/dist/sql-wasm.wasm?url"; -import { probeDatabase } from "./db-probe.js"; - -/** - * The database is read where it lies. SQLite asks for pages, the virtual file - * system turns each into an HTTP range request, and only the pages a query - * touches ever cross the network — a few hundred KB for a lookup, against the - * 45 MB the whole file used to cost before the first query. - * - * That only holds while every query is index-driven. The schema exists for it: - * name_word serves name search, and the score indexes serve the SQL presets. - * An unindexed query walks the table and pulls all 100+ MB of it, which is what - * the byte budget below is for. - */ - -// Must equal the page size the parser writes (PRAGMA page_size in -// parser/internal/writer/writer.go), so one request is exactly one page. A -// mismatch makes every logical page read span two requests. -const CHUNK_BYTES = 1024; - -/** Generous for indexed work: a name search costs well under 1 MB. */ -export const SEARCH_BUDGET_BYTES = 25 * 1024 * 1024; - -/** What the SQL tab gets once the user has accepted the cost of a scan. */ -export const PLAYGROUND_BUDGET_BYTES = 250 * 1024 * 1024; - -/** - * What one query cost over the network: `{ requests, bytes, ms }`. - */ - -/** - * One remotely-paged database, with the load state the UI needs. - * - * `budgetBytes` is a hard ceiling for the worker's lifetime: past it a query - * fails instead of quietly downloading the file. Raising it means a new worker, - * which costs only the header pages. - */ -export class RemoteDatabase { - ready = $state(false); - error = $state(null); - /** Bytes fetched by this database so far, prefetch included. */ - bytesRead = $state(0); - /** HTTP range requests issued so far. */ - requests = $state(0); - /** What the most recent query cost on its own. */ - lastCost = $state(null); - - #worker = null; - #opening; - #closed = false; - // Previous cumulative reading, so a query's own cost is a subtraction. - #seen = { requests: 0, bytes: 0 }; - - /** - * `source` carries both forms of the same location: `url` is the file, and - * `urlPrefix` is what the library appends the chunk index to. They must - * agree — see dbOf/dbPrefixOf, which derive one from the other. - */ - constructor(source, budgetBytes = SEARCH_BUDGET_BYTES) { - this.url = source.url; - this.urlPrefix = source.urlPrefix; - this.budgetBytes = budgetBytes; - this.#opening = this.#open(); - } - - async #open() { - const opened = performance.now(); - try { - // Chunked mode over a single chunk, which looks odd but is the only way - // to tell this library how long the file is: the worker reads - // databaseLengthBytes in chunked mode and hardcodes the length to - // undefined in full mode. Left to itself it sizes the file with a HEAD - // request, and GitHub Pages answers that with the gzipped length, which - // it then refuses to use. - // - // One chunk covers the whole database, so the chunk index is always 0 - // and every request goes to urlPrefix + "0" — the file the assembler - // publishes as .sqlite30. - const databaseLengthBytes = await probeDatabase(this.url, CHUNK_BYTES); - const worker = await createDbWorker( - [ - { - from: "inline", - config: { - serverMode: "chunked", - urlPrefix: this.urlPrefix, - serverChunkSize: databaseLengthBytes, - databaseLengthBytes, - suffixLength: 1, - requestChunkSize: CHUNK_BYTES, - }, - }, - ], - workerUrl, - wasmUrl, - this.budgetBytes, - ); - if (this.#closed) throw new Error("closed"); - this.#worker = worker; - this.ready = true; - // Opening is not free either: the header and schema pages are read before - // any query runs, and that shows up in every later session total. - await this.#account(worker, `open ${this.url}`, performance.now() - opened); - return worker; - } catch (err) { - if (!this.#closed) { - this.error = message(err); - this.ready = false; - } - throw err; - } - } - - /** - * Run a query and return its rows as objects. - * - * `label` names the query in the console trace — the only way to see what a - * search actually costs, since the byte count depends on how much the read - * heads prefetched, not just on the pages the plan needed. - */ - async query(sql, params = [], label) { - const worker = this.#worker ?? (await this.#opening); - const started = performance.now(); - try { - return await worker.db.query(sql, ...params); - } finally { - await this.#account(worker, label ?? firstLine(sql), performance.now() - started); - } - } - - /** - * Read the cumulative counters and report the delta. - * - * getStats() rather than the worker's `bytesRead`: that one is the budget - * counter and resets itself to zero when a query trips the ceiling, so it - * would under-report exactly when the number matters most. - */ - async #account(worker, label, ms) { - const stats = await worker.worker.getStats().catch(() => null); - if (!stats) return; - - const cost = { - requests: stats.totalRequests - this.#seen.requests, - bytes: stats.totalFetchedBytes - this.#seen.bytes, - ms, - }; - this.#seen = { requests: stats.totalRequests, bytes: stats.totalFetchedBytes }; - this.requests = stats.totalRequests; - this.bytesRead = stats.totalFetchedBytes; - this.lastCost = cost; - - console.info( - `[httpvfs] ${label} — ${cost.requests} request(s), ${formatBytes(cost.bytes)}, ${ms.toFixed(0)} ms` + - ` · session ${stats.totalRequests} request(s), ${formatBytes(stats.totalFetchedBytes)}` + - ` of ${formatBytes(stats.totalBytes)}`, - ); - } - - /** - * Drop this database. createDbWorker owns the Worker and exposes no handle to - * it, so the thread outlives this call; a page creates at most one per - * dataset and one more if the SQL budget is raised, which is why that is - * tolerable rather than a leak worth working around. - */ - close() { - this.#closed = true; - this.#worker = null; - this.ready = false; - } -} - -/** True when a query failed because it would have exceeded the byte budget. */ -export function isBudgetError(err) { - return /maxBytesToRead|too much data|exceeded/i.test(message(err)); -} - -function message(err) { - return err instanceof Error ? err.message : String(err); -} - -/** Enough of a query to recognise it in the console. */ -function firstLine(sql) { - const line = sql.trim().split("\n")[0]; - return line.length > 70 ? `${line.slice(0, 70)}…` : line; -} - -export function formatBytes(n) { - return n < 1024 * 1024 ? `${Math.round(n / 1024)} KB` : `${(n / 1048576).toFixed(1)} MB`; -} diff --git a/web/src/routes/[dataset]/+page.svelte b/web/src/routes/[dataset]/+page.svelte index dfe17bb..1f1823b 100644 --- a/web/src/routes/[dataset]/+page.svelte +++ b/web/src/routes/[dataset]/+page.svelte @@ -7,7 +7,9 @@ import ScoreTable from "$lib/components/score-table.svelte"; import SearchForm from "$lib/components/search-form.svelte"; import StudentDetail from "$lib/components/student-detail.svelte"; - import { dbSourceOf } from "$lib/datasets"; + import { LocalDatabase, formatBytes } from "$lib/database.svelte"; + import { forgetAll } from "$lib/db-cache"; + import { dbOf } from "$lib/datasets"; import { isExamId } from "$lib/query-mode"; import { MAX_RESULTS, lookupExamId, searchByName } from "$lib/search"; @@ -31,7 +33,7 @@ // Created in the browser only: $effect does not run while prerendering. The // download itself waits for the visitor to accept it. $effect(() => { - const opened = new RemoteDatabase(dbSourceOf(dataset, base), budget); + const opened = new LocalDatabase(dbOf(dataset, base), dataset.dbSizeMb * 1024 * 1024); db = opened; // Consults the cache, and opens a stored copy without asking: consent was // given the first time, and reusing it costs nothing.