Files
thptqg/parser/internal/schema/schema_test.go
T
tiennm99 dbc23c25c5 feat: read the databases over HTTP range requests
The browser downloaded 45 MB of gzipped SQLite before it could answer
anything. Now sql.js-httpvfs asks for the pages a query touches and the
databases ship uncompressed as <id>.sqlite3 — a byte range of a gzip
stream is not a byte range of a database.

That only works if every query the site issues is index-driven, and
measured against the real 2016 file, most were not:

  so_bao_danh = ?              SEARCH via PK          ~20 KB
  ho_ten_ascii LIKE '%x%'      SCAN                   127 MB
  ho_ten_ascii LIKE 'x%'       SCAN                   127 MB
  COUNT(*)                     covering index scan     20 MB
  ORDER BY toan DESC LIMIT 10  SCAN + temp b-tree     127 MB

Prefix LIKE scans because SQLite's LIKE optimisation needs a NOCASE
index; a range comparison does use the index. So the schema changed to
suit the access pattern rather than the search changing to suit the
schema.

name_word holds one row per word of each name, WITHOUT ROWID so the
table is the index, carrying ho_ten_ascii so a multi-word query is
resolved inside a single b-tree. name_word_freq says which word of a
query is rarest — the vocabulary is 4,397 words across 2.87M entries, so
"buu loc" seeks on 287 entries rather than walking the 300,000 that
"thi" would. Searching by any word of a name survives, at a few hundred
KB a query.

idx_ho_ten and idx_ho_ten_ascii are gone: no plan could use either.
Partial indexes on toan, khtn and khxh cost 12 MB and keep the SQL
presets off a full scan. The footer's candidate count now comes from
datasets.json instead of COUNT(*).

2016 grows 223.5 MB to 288.6 MB, 2017 162.7 MB to 237.7 MB, and the site
is 528 MB against the 1 GB GitHub Pages limit. Row counts are unchanged.

The SQL tab is the one place a user can still write a query that reads
the whole table, so it asks before it opens, runs under a byte budget
that stops a runaway query, and shows what each query actually fetched.

Verified: row counts through the assembler guards, every app query
index-driven under EXPLAIN QUERY PLAN, and GitHub Pages returning 206
with a correct Content-Range. Not verified in a browser — this machine
has none — and the library refuses to open a file the host compresses,
so the deployed response headers need a look.
2026-08-14 12:42:48 +07:00

186 lines
5.3 KiB
Go

package schema
import (
"strings"
"testing"
)
// These tests stop the DDL, the INSERT column list and the field-order constants
// from drifting apart — a drift that silently lands values in the wrong columns.
func TestInsertMatchesFieldOrder(t *testing.T) {
if ParamCount != 22 {
t.Errorf("ParamCount = %d, want 22", ParamCount)
}
if got := strings.Count(InsertSQL, "?"); got != ParamCount {
t.Errorf("INSERT placeholders = %d, want %d", got, ParamCount)
}
open := strings.Index(InsertSQL, "(")
closeIdx := strings.Index(InsertSQL, ")")
if open < 0 || closeIdx < 0 {
t.Fatal("INSERT must contain a column list")
}
var listed []string
for _, c := range strings.Split(InsertSQL[open+1:closeIdx], ",") {
if c = strings.TrimSpace(c); c != "" {
listed = append(listed, c)
}
}
want := append(append([]string{}, IdentityFields...), ScoreFields...)
if len(listed) != len(want) {
t.Fatalf("INSERT lists %d columns, want %d", len(listed), len(want))
}
for i := range want {
if listed[i] != want[i] {
t.Errorf("column %d: INSERT has %q, field order has %q", i, listed[i], want[i])
}
}
}
func TestScorePatternsCoverScoreFields(t *testing.T) {
if len(ScorePatterns) != len(ScoreFields) {
t.Fatalf("%d patterns for %d score columns", len(ScorePatterns), len(ScoreFields))
}
inFields := make(map[string]bool, len(ScoreFields))
for _, f := range ScoreFields {
inFields[f] = true
}
for field := range ScorePatterns {
if !inFields[field] {
t.Errorf("pattern %q has no column", field)
}
}
for _, field := range ScoreFields {
if _, ok := ScorePatterns[field]; !ok {
t.Errorf("column %q has no pattern", field)
}
}
}
func TestDDLColumnsMatchInsert(t *testing.T) {
for _, field := range append(append([]string{}, IdentityFields...), ScoreFields...) {
if !strings.Contains(DDL, field) {
t.Errorf("DDL missing column %q", field)
}
}
}
// TestScorePatternsCompile: compilation happens in the package initialiser, so
// reaching this point already proves it; the explicit checks guard against an
// empty or partial table.
func TestScorePatternsCompile(t *testing.T) {
for field, re := range ScorePatterns {
if re == nil {
t.Errorf("pattern %q is nil", field)
}
}
}
// TestDDLIsFrozen compares DDL against an independent copy of the exact text,
// down to the byte. It catches column, type and index changes that every
// row-level check would still pass — a database can be structurally different
// and look fine one row at a time. Update the copy below only when the schema
// change is intended.
func TestDDLIsFrozen(t *testing.T) {
const want = `
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
ten_cum_thi TEXT,
gioi_tinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_duc REAL,
tieng_nhat REAL,
tieng_trung REAL
);
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL;
CREATE TABLE name_word (
word TEXT NOT NULL,
so_bao_danh TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
PRIMARY KEY (word, so_bao_danh)
) WITHOUT ROWID;
CREATE TABLE name_word_freq (
word TEXT PRIMARY KEY,
n INTEGER NOT NULL
) WITHOUT ROWID;
`
if DDL != want {
t.Errorf("DDL changed\n--- got ---\n%s\n--- want ---\n%s", DDL, want)
}
}
// TestNoIndexOnNameColumns: the databases are read over HTTP range requests, so
// an index that no query can use is dead weight in a file the browser pages
// through. Neither substring nor prefix LIKE can use one on these columns —
// name_word is what serves name search.
func TestNoIndexOnNameColumns(t *testing.T) {
for _, dead := range []string{"idx_ho_ten ", "idx_ho_ten_ascii"} {
if strings.Contains(DDL, dead) {
t.Errorf("DDL creates %q, which no query plan can use", dead)
}
}
}
// TestPostLoadBuildsTheSearchTables: the frequency table is what lets a search
// pick which word to seek on, and the score indexes are what keep the SQL
// presets off a full scan.
func TestPostLoadBuildsTheSearchTables(t *testing.T) {
for _, want := range []string{
"INSERT INTO name_word_freq",
"CREATE INDEX idx_toan",
"CREATE INDEX idx_khtn",
"CREATE INDEX idx_khxh",
} {
if !strings.Contains(PostLoadSQL, want) {
t.Errorf("PostLoadSQL is missing %q", want)
}
}
}
// TestScorePatternsMatchScores exercises each pattern against the shape the
// DIEM_THI cell actually carries, including the wide runs of spaces seen in the
// real corpus.
func TestScorePatternsMatchScores(t *testing.T) {
const cell = "Toán: 8.50 Ngữ văn: 7.00 Tiếng Đức: 9 KHXH: 5.58 "
cases := map[string]string{
"toan": "8.50",
"ngu_van": "7.00",
"tieng_duc": "9",
"khxh": "5.58",
"tieng_nhat": "", // absent from the cell -> no match
}
for field, want := range cases {
re, ok := ScorePatterns[field]
if !ok {
t.Fatalf("no pattern for %q", field)
}
m := re.FindStringSubmatch(cell)
got := ""
if m != nil {
got = m[1]
}
if got != want {
t.Errorf("%s: matched %q, want %q", field, got, want)
}
}
}