mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-04 22:13:49 +00:00
The browser downloaded 45 MB of gzipped SQLite before it could answer anything. Now sql.js-httpvfs asks for the pages a query touches and the databases ship uncompressed as <id>.sqlite3 — a byte range of a gzip stream is not a byte range of a database. That only works if every query the site issues is index-driven, and measured against the real 2016 file, most were not: so_bao_danh = ? SEARCH via PK ~20 KB ho_ten_ascii LIKE '%x%' SCAN 127 MB ho_ten_ascii LIKE 'x%' SCAN 127 MB COUNT(*) covering index scan 20 MB ORDER BY toan DESC LIMIT 10 SCAN + temp b-tree 127 MB Prefix LIKE scans because SQLite's LIKE optimisation needs a NOCASE index; a range comparison does use the index. So the schema changed to suit the access pattern rather than the search changing to suit the schema. name_word holds one row per word of each name, WITHOUT ROWID so the table is the index, carrying ho_ten_ascii so a multi-word query is resolved inside a single b-tree. name_word_freq says which word of a query is rarest — the vocabulary is 4,397 words across 2.87M entries, so "buu loc" seeks on 287 entries rather than walking the 300,000 that "thi" would. Searching by any word of a name survives, at a few hundred KB a query. idx_ho_ten and idx_ho_ten_ascii are gone: no plan could use either. Partial indexes on toan, khtn and khxh cost 12 MB and keep the SQL presets off a full scan. The footer's candidate count now comes from datasets.json instead of COUNT(*). 2016 grows 223.5 MB to 288.6 MB, 2017 162.7 MB to 237.7 MB, and the site is 528 MB against the 1 GB GitHub Pages limit. Row counts are unchanged. The SQL tab is the one place a user can still write a query that reads the whole table, so it asks before it opens, runs under a byte budget that stops a runaway query, and shows what each query actually fetched. Verified: row counts through the assembler guards, every app query index-driven under EXPLAIN QUERY PLAN, and GitHub Pages returning 206 with a correct Content-Range. Not verified in a browser — this machine has none — and the library refuses to open a file the host compresses, so the deployed response headers need a look.
186 lines
5.3 KiB
Go
186 lines
5.3 KiB
Go
package schema
|
|
|
|
import (
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// These tests stop the DDL, the INSERT column list and the field-order constants
|
|
// from drifting apart — a drift that silently lands values in the wrong columns.
|
|
|
|
func TestInsertMatchesFieldOrder(t *testing.T) {
|
|
if ParamCount != 22 {
|
|
t.Errorf("ParamCount = %d, want 22", ParamCount)
|
|
}
|
|
if got := strings.Count(InsertSQL, "?"); got != ParamCount {
|
|
t.Errorf("INSERT placeholders = %d, want %d", got, ParamCount)
|
|
}
|
|
|
|
open := strings.Index(InsertSQL, "(")
|
|
closeIdx := strings.Index(InsertSQL, ")")
|
|
if open < 0 || closeIdx < 0 {
|
|
t.Fatal("INSERT must contain a column list")
|
|
}
|
|
var listed []string
|
|
for _, c := range strings.Split(InsertSQL[open+1:closeIdx], ",") {
|
|
if c = strings.TrimSpace(c); c != "" {
|
|
listed = append(listed, c)
|
|
}
|
|
}
|
|
|
|
want := append(append([]string{}, IdentityFields...), ScoreFields...)
|
|
if len(listed) != len(want) {
|
|
t.Fatalf("INSERT lists %d columns, want %d", len(listed), len(want))
|
|
}
|
|
for i := range want {
|
|
if listed[i] != want[i] {
|
|
t.Errorf("column %d: INSERT has %q, field order has %q", i, listed[i], want[i])
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestScorePatternsCoverScoreFields(t *testing.T) {
|
|
if len(ScorePatterns) != len(ScoreFields) {
|
|
t.Fatalf("%d patterns for %d score columns", len(ScorePatterns), len(ScoreFields))
|
|
}
|
|
inFields := make(map[string]bool, len(ScoreFields))
|
|
for _, f := range ScoreFields {
|
|
inFields[f] = true
|
|
}
|
|
for field := range ScorePatterns {
|
|
if !inFields[field] {
|
|
t.Errorf("pattern %q has no column", field)
|
|
}
|
|
}
|
|
for _, field := range ScoreFields {
|
|
if _, ok := ScorePatterns[field]; !ok {
|
|
t.Errorf("column %q has no pattern", field)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestDDLColumnsMatchInsert(t *testing.T) {
|
|
for _, field := range append(append([]string{}, IdentityFields...), ScoreFields...) {
|
|
if !strings.Contains(DDL, field) {
|
|
t.Errorf("DDL missing column %q", field)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestScorePatternsCompile: compilation happens in the package initialiser, so
|
|
// reaching this point already proves it; the explicit checks guard against an
|
|
// empty or partial table.
|
|
func TestScorePatternsCompile(t *testing.T) {
|
|
for field, re := range ScorePatterns {
|
|
if re == nil {
|
|
t.Errorf("pattern %q is nil", field)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestDDLIsFrozen compares DDL against an independent copy of the exact text,
|
|
// down to the byte. It catches column, type and index changes that every
|
|
// row-level check would still pass — a database can be structurally different
|
|
// and look fine one row at a time. Update the copy below only when the schema
|
|
// change is intended.
|
|
func TestDDLIsFrozen(t *testing.T) {
|
|
const want = `
|
|
CREATE TABLE student (
|
|
so_bao_danh TEXT PRIMARY KEY,
|
|
ho_ten TEXT NOT NULL,
|
|
ho_ten_ascii TEXT NOT NULL,
|
|
ngay_sinh TEXT,
|
|
ten_cum_thi TEXT,
|
|
gioi_tinh TEXT,
|
|
toan REAL,
|
|
ngu_van REAL,
|
|
vat_ly REAL,
|
|
hoa_hoc REAL,
|
|
sinh_hoc REAL,
|
|
khtn REAL,
|
|
lich_su REAL,
|
|
dia_ly REAL,
|
|
gdcd REAL,
|
|
khxh REAL,
|
|
tieng_anh REAL,
|
|
tieng_phap REAL,
|
|
tieng_nga REAL,
|
|
tieng_duc REAL,
|
|
tieng_nhat REAL,
|
|
tieng_trung REAL
|
|
);
|
|
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL;
|
|
|
|
CREATE TABLE name_word (
|
|
word TEXT NOT NULL,
|
|
so_bao_danh TEXT NOT NULL,
|
|
ho_ten_ascii TEXT NOT NULL,
|
|
PRIMARY KEY (word, so_bao_danh)
|
|
) WITHOUT ROWID;
|
|
|
|
CREATE TABLE name_word_freq (
|
|
word TEXT PRIMARY KEY,
|
|
n INTEGER NOT NULL
|
|
) WITHOUT ROWID;
|
|
`
|
|
if DDL != want {
|
|
t.Errorf("DDL changed\n--- got ---\n%s\n--- want ---\n%s", DDL, want)
|
|
}
|
|
}
|
|
|
|
// TestNoIndexOnNameColumns: the databases are read over HTTP range requests, so
|
|
// an index that no query can use is dead weight in a file the browser pages
|
|
// through. Neither substring nor prefix LIKE can use one on these columns —
|
|
// name_word is what serves name search.
|
|
func TestNoIndexOnNameColumns(t *testing.T) {
|
|
for _, dead := range []string{"idx_ho_ten ", "idx_ho_ten_ascii"} {
|
|
if strings.Contains(DDL, dead) {
|
|
t.Errorf("DDL creates %q, which no query plan can use", dead)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestPostLoadBuildsTheSearchTables: the frequency table is what lets a search
|
|
// pick which word to seek on, and the score indexes are what keep the SQL
|
|
// presets off a full scan.
|
|
func TestPostLoadBuildsTheSearchTables(t *testing.T) {
|
|
for _, want := range []string{
|
|
"INSERT INTO name_word_freq",
|
|
"CREATE INDEX idx_toan",
|
|
"CREATE INDEX idx_khtn",
|
|
"CREATE INDEX idx_khxh",
|
|
} {
|
|
if !strings.Contains(PostLoadSQL, want) {
|
|
t.Errorf("PostLoadSQL is missing %q", want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestScorePatternsMatchScores exercises each pattern against the shape the
|
|
// DIEM_THI cell actually carries, including the wide runs of spaces seen in the
|
|
// real corpus.
|
|
func TestScorePatternsMatchScores(t *testing.T) {
|
|
const cell = "Toán: 8.50 Ngữ văn: 7.00 Tiếng Đức: 9 KHXH: 5.58 "
|
|
cases := map[string]string{
|
|
"toan": "8.50",
|
|
"ngu_van": "7.00",
|
|
"tieng_duc": "9",
|
|
"khxh": "5.58",
|
|
"tieng_nhat": "", // absent from the cell -> no match
|
|
}
|
|
for field, want := range cases {
|
|
re, ok := ScorePatterns[field]
|
|
if !ok {
|
|
t.Fatalf("no pattern for %q", field)
|
|
}
|
|
m := re.FindStringSubmatch(cell)
|
|
got := ""
|
|
if m != nil {
|
|
got = m[1]
|
|
}
|
|
if got != want {
|
|
t.Errorf("%s: matched %q, want %q", field, got, want)
|
|
}
|
|
}
|
|
}
|