Files
miti99bot/scripts/build-semantle-words.js
T
tiennm99 866f6d663f feat(semantle): source target pool from google-10000-english dictionary
The ~250-word hand-curated TARGET_POOL was too small for long-term play.
Replaces it with a build-script-generated dictionary:

- scripts/build-semantle-words.js fetches first20hours/google-10000-english
  (no-swears variant), filters to 4–10 ASCII letters, drops the top-200
  most frequent function words, and writes src/modules/semantle/words-data.js
  as a static ES-module export.
- wordlist.js now just re-exports that data via TARGET_POOL + pickFromPool.
- package.json: new build:semantle-words script; chained into `npm run build`
  alongside build:wordle-data so `npm run deploy` regenerates automatically.

Pool size: ~250 → 7953 words. Same ConceptNet verify-and-fallback flow, so
low-quality picks still cost at most one extra concept lookup.
2026-04-22 23:12:07 +07:00

68 lines
2.4 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* @file build-semantle-words — fetches a common-English word list (10k
* by Google Ngram frequency, curated by first20hours) and writes the
* 410 letter alphabetic subset to src/modules/semantle/words-data.js.
*
* Why this list:
* - Sorted by frequency → easy top-N trimming to drop function words
* (`the`, `of`, `and`) that make terrible guessing targets.
* - Already de-swear-ed. No further blocklist needed.
* - Tiny (~90 KB raw, ~30 KB gzipped) — comfortably inside the Worker
* size budget.
*
* Source: https://github.com/first20hours/google-10000-english
* Credits: Josh Kaufman (first20hours) — list derived from Peter Norvig's
* Google Ngram analysis.
*
* Usage:
* node scripts/build-semantle-words.js
*/
import { writeFileSync } from "node:fs";
import { resolve } from "node:path";
const SOURCE_URL =
"https://raw.githubusercontent.com/first20hours/google-10000-english/master/google-10000-english-no-swears.txt";
// Skip the top-N most frequent words — too common, lousy puzzles.
const SKIP_TOP_N = 200;
const MIN_LEN = 4;
const MAX_LEN = 10;
const root = resolve(import.meta.dirname, "..");
const dst = resolve(root, "src/modules/semantle/words-data.js");
const res = await fetch(SOURCE_URL);
if (!res.ok) throw new Error(`fetch failed: ${res.status} ${res.statusText}`);
const text = await res.text();
const lines = text.split(/\r?\n/).map((w) => w.trim().toLowerCase());
// Preserve original frequency order while filtering — downstream consumers
// can still sample uniformly, but the index itself stays a frequency rank.
const words = Array.from(
new Set(
lines
.slice(SKIP_TOP_N)
.filter((w) => w.length >= MIN_LEN && w.length <= MAX_LEN && /^[a-z]+$/.test(w)),
),
);
if (words.length === 0) throw new Error("no words parsed from source");
const body = words.map((w) => ` "${w}",`).join("\n");
const out = [
"// Auto-generated from https://github.com/first20hours/google-10000-english",
"// Credits: Josh Kaufman (first20hours) — common English words by Google Ngram frequency.",
`// Filter: ${MIN_LEN}${MAX_LEN} ASCII letters, skip top ${SKIP_TOP_N} most common.`,
"// Regenerate with: node scripts/build-semantle-words.js",
"export default [",
body,
"];",
"",
].join("\n");
writeFileSync(dst, out);
console.log(`wrote ${dst} (${words.length} words)`);