mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-05 16:14:22 +00:00
Four of the 119 files in data/2016 publish one score column per subject instead of a DIEM_THI sentence, and none of them was being read correctly. The ĐH Công nghiệp Thực phẩm file puts a three-row ministry title block above its header, so no header was recognised and the positional fallback shifted every column by one: the serial number became so_bao_danh, the exam number became ho_ten, the name became ngay_sinh, and the national ID became the score cell. All 7,833 rows were unusable. The three ĐH Cần Thơ files name an SBD column but no DIEM_THI, so they fell to the same fallback: surname into ngay_sinh, given name into ten_cum_thi, birth date into the score cell, and 12,152 candidates with no scores at all. Both are now read by FormatSubjectColumns, which resolves identity and one column per subject from the header. The header is searched for in the first five rows, so a title block no longer hides it. The Cần Thơ score columns are numbered rather than named. They follow the order the exam was sat — each morning an essay paper, each afternoon a multiple-choice one — which is what identifies them: columns 1/3/5/7 quantise to 0.25 and 2/4/6/8 do not, and each column's mean lands within 0.5 of the same subject's mean across the rest of the dataset. The foreign language is filed under the subject its N1..N6 code names. Gender now accepts the 0/1 encoding those files use: of the rows marked 1, 53% carry "Thị" in the name against 1% of those marked 0. Birth dates in the compact ddmmyy form are expanded so the column holds one format. A score of 0 is stored rather than dropped, recovering 302 real scores that a JavaScript falsy check had been turning into NULL. Row count falls by one, to 877,460: the removed row is the title line "ĐƠN VỊ: / TRƯỜNG ĐẠI HỌC CÔNG NGHIỆP THỰC PHẨM TP. HỒ CHÍ MINH", which had been stored as a student. The dataset has no duplicate exam numbers; the three rows previously described as collapsing were that same file's title and header lines being counted and then rejected. Also drops behaviour that existed only to match the parser this one replaced: the inert "SINH " header token, the untrimmed diem_thi cell, an unreachable blank-row branch, and a cross-check test against a database that can no longer exist. None of them changes output. Verified by rebuilding both datasets: 877,460 and 861,068 rows, both artifacts through the assembler's row and size guards, and the reader fidelity suite unchanged across all 182 files.
228 lines
5.9 KiB
Go
228 lines
5.9 KiB
Go
// Package ingest owns all dataset policy: which sheets to read, which rows are
|
|
// headers, which are blank, and how rows are counted.
|
|
//
|
|
// The reader deliberately has none of this — it reports every sheet and every
|
|
// row verbatim, which is what keeps its fidelity independently testable.
|
|
package ingest
|
|
|
|
import (
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"sort"
|
|
"strings"
|
|
|
|
"github.com/tiennm99/thptqg/parser/internal/config"
|
|
"github.com/tiennm99/thptqg/parser/internal/reader"
|
|
"github.com/tiennm99/thptqg/parser/internal/transform"
|
|
"github.com/tiennm99/thptqg/parser/internal/writer"
|
|
)
|
|
|
|
// IsHeaderRow reports whether row is a header, by matching its uppercased first
|
|
// cell against the configured tokens.
|
|
//
|
|
// Rows shorter than 3 cells are never headers — a 1- or 2-cell row is a stray
|
|
// fragment, not a real header.
|
|
func IsHeaderRow(row []reader.Cell, tokens []string) bool {
|
|
if len(row) < 3 {
|
|
return false
|
|
}
|
|
first := strings.ToUpper(strings.TrimSpace(row[0].Str))
|
|
for _, t := range tokens {
|
|
if strings.ToUpper(t) == first {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// IsAllBlank reports whether every cell is empty or whitespace-only.
|
|
//
|
|
// Compares on Str only. Cell.IsEmpty is diagnostic: an absent cell and an empty
|
|
// string cell both render "" and both count as blank here, so branching on the
|
|
// flag would invent a distinction nothing downstream acts on.
|
|
func IsAllBlank(row []reader.Cell) bool {
|
|
for _, c := range row {
|
|
if strings.TrimSpace(c.Str) != "" {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
// InputFiles lists a dataset directory's spreadsheets, bytewise-sorted on the
|
|
// full path.
|
|
//
|
|
// The sort is load-bearing, not cosmetic: INSERT OR REPLACE is last-wins, so
|
|
// file order decides which row survives a duplicate SBD.
|
|
func InputFiles(dir string) ([]string, error) {
|
|
entries, err := os.ReadDir(dir)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("cannot read input dir %s: %w", dir, err)
|
|
}
|
|
var out []string
|
|
for _, e := range entries {
|
|
if e.IsDir() {
|
|
continue
|
|
}
|
|
switch strings.ToLower(filepath.Ext(e.Name())) {
|
|
case ".xls", ".xlsx":
|
|
out = append(out, filepath.Join(dir, e.Name()))
|
|
}
|
|
}
|
|
sort.Strings(out)
|
|
return out, nil
|
|
}
|
|
|
|
// DatasetLabel derives the build-log label from the input directory basename.
|
|
func DatasetLabel(inputDir string) string {
|
|
base := filepath.Base(strings.TrimRight(inputDir, string(filepath.Separator)))
|
|
if base == "" || base == "." || base == string(filepath.Separator) {
|
|
return "data"
|
|
}
|
|
return base
|
|
}
|
|
|
|
// RowFn consumes one data row of one sheet, after header skipping.
|
|
type RowFn func(sheetIdx int, row []reader.Cell)
|
|
|
|
// ProcessFile applies sheet selection and per-sheet header skipping, invoking fn
|
|
// for every remaining row.
|
|
//
|
|
// The header check is per SHEET, not per file: firstRow resets inside the sheet
|
|
// loop, so a workbook whose second sheet repeats the header has it skipped
|
|
// there too.
|
|
func ProcessFile(path string, cfg *config.DatasetConfig, fn RowFn) error {
|
|
wb, err := reader.Open(path)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer wb.Close()
|
|
|
|
sheets := wb.Sheets()
|
|
if len(sheets) == 0 {
|
|
return fmt.Errorf("no sheets in %s", path)
|
|
}
|
|
if cfg.Reader.SheetMode == config.SheetModeFirst {
|
|
sheets = sheets[:1]
|
|
}
|
|
|
|
for _, sh := range sheets {
|
|
firstRow := true
|
|
err := wb.EachRow(sh.Index, func(s reader.Sheet, _ int, row []reader.Cell) error {
|
|
if firstRow {
|
|
firstRow = false
|
|
if IsHeaderRow(row, cfg.Header.Tokens) {
|
|
return nil
|
|
}
|
|
}
|
|
fn(s.Index, row)
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// Standard runs the fixed-column path for the 2017-family datasets.
|
|
func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error {
|
|
files, err := InputFiles(inputDir)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
label := DatasetLabel(inputDir)
|
|
fmt.Printf("[build] %s/ → %s (%d files)\n", label, outputPath, len(files))
|
|
|
|
db, err := writer.OpenDB(outputPath)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
defer db.Close()
|
|
|
|
stripBlank := cfg.Reader.StripBlankRows
|
|
|
|
// One transaction spans the whole dataset directory.
|
|
tx, err := db.Begin()
|
|
if err != nil {
|
|
return fmt.Errorf("begin: %w", err)
|
|
}
|
|
ins, err := writer.Prepare(tx)
|
|
if err != nil {
|
|
tx.Rollback()
|
|
return err
|
|
}
|
|
|
|
var st writer.Stats
|
|
for _, file := range files {
|
|
base := filepath.Base(file)
|
|
var fileRows, fileSkipped, fileErrors uint64
|
|
|
|
procErr := ProcessFile(file, cfg, func(_ int, row []reader.Cell) {
|
|
allBlank := IsAllBlank(row)
|
|
// Blank rows drop out BEFORE the source-row counter, so they never
|
|
// reach the source total.
|
|
if stripBlank && allBlank {
|
|
return
|
|
}
|
|
st.SourceRows++
|
|
|
|
hoTen, soBaoDanh := "", ""
|
|
if cols := cfg.Columns; cols != nil {
|
|
hoTen = cellAt(row, cols.HoTen)
|
|
soBaoDanh = cellAt(row, cols.SoBaoDanh)
|
|
}
|
|
|
|
if transform.ValidateRow(hoTen, soBaoDanh, &cfg.Validation, stripBlank, allBlank) != transform.SkipNone {
|
|
fileSkipped++
|
|
return
|
|
}
|
|
|
|
parsed, err := transform.TransformRow(row, cfg)
|
|
if err != nil {
|
|
fileErrors++
|
|
return
|
|
}
|
|
if err := ins.Insert(parsed); err != nil {
|
|
fileErrors++
|
|
// Only the first five insert warnings print.
|
|
if st.Errors+fileErrors <= 5 {
|
|
fmt.Fprintf(os.Stderr, " [warn] %s: %v\n", base, err)
|
|
}
|
|
return
|
|
}
|
|
fileRows++
|
|
})
|
|
if procErr != nil {
|
|
// A file that cannot be read is logged and counted, never fatal —
|
|
// one corrupt file must not abandon the batch.
|
|
fmt.Fprintf(os.Stderr, " [error] %s: %v\n", base, procErr)
|
|
fileErrors++
|
|
}
|
|
|
|
st.Skipped += fileSkipped
|
|
st.Errors += fileErrors
|
|
fmt.Printf(" %s: %d rows\n", base, fileRows)
|
|
}
|
|
|
|
if err := ins.Close(); err != nil {
|
|
tx.Rollback()
|
|
return err
|
|
}
|
|
if err := tx.Commit(); err != nil {
|
|
return fmt.Errorf("commit: %w", err)
|
|
}
|
|
|
|
// VACUUM only after COMMIT — SQLite refuses it inside a transaction.
|
|
return writer.Finish(db, outputPath, st)
|
|
}
|
|
|
|
// cellAt returns the trimmed cell at idx, or "" when idx is out of range.
|
|
func cellAt(row []reader.Cell, idx int) string {
|
|
if idx < 0 || idx >= len(row) {
|
|
return ""
|
|
}
|
|
return strings.TrimSpace(row[idx].Str)
|
|
}
|