feat(dict): every roadmap word is a card, tagged with its unit
The reworked app's learner model hangs off cards: recall evidence, the phase-review checklist, the practice set, and the gate's "words he has met". So every word a unit introduces has to be studiable — and 16 of the 371 were not. Eleven existed only as sentence chunks or dictionary rows outside the review deck, and five (봐 읽어 갔어 봤어 먹었어) nowhere at all. The build now marks exactly one reviewable lemma per roadmap word with the unit that introduces it. Where the deck has the word, its row is chosen deterministically (deck order, then source, then part of speech) — the artifact tagged whichever card came last, which put the evidence for 이, 눈 and 저 on the wrong meaning. The sixteen get a `curriculum` lemma of their own, glossed from the curated verb they conjugate (자 is "sleep", the 반말 of 자다 — not the dictionary's "ruler"), else from the sentence that uses them, else the dictionary. Lemmas also carry their topic, which the vocabulary filters need. dict:assert gains the guarantee: 371/371 roadmap words as one card each. Migration 7 adds the two columns; the rows arrive with the dictionary reload a changed build now triggers on its own. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -6,53 +6,53 @@
|
|||||||
"frequencyForms": 688129
|
"frequencyForms": 688129
|
||||||
},
|
},
|
||||||
"totals": {
|
"totals": {
|
||||||
"lemmas": 30520,
|
"lemmas": 30536,
|
||||||
"surfaces": 43879
|
"surfaces": 43895
|
||||||
},
|
},
|
||||||
"bands": [
|
"bands": [
|
||||||
{
|
{
|
||||||
"band": 0,
|
"band": 0,
|
||||||
"file": "band-0.json.gz",
|
"file": "band-0.json.gz",
|
||||||
"lemmas": 801,
|
"lemmas": 802,
|
||||||
"surfaces": 1176,
|
"surfaces": 1177,
|
||||||
"bytes": 21332,
|
"bytes": 24358,
|
||||||
"sha256": "e631e95e0263995c23ee58a5a3ed7cadf11d7b10566449192d644ccb2be7fe04",
|
"sha256": "ec95a68245700a6dd9aee519378b9d86713484efe888f8450578f5c9315a48c6",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"band": 1,
|
"band": 1,
|
||||||
"file": "band-1.json.gz",
|
"file": "band-1.json.gz",
|
||||||
"lemmas": 1240,
|
"lemmas": 1246,
|
||||||
"surfaces": 2335,
|
"surfaces": 2341,
|
||||||
"bytes": 48215,
|
"bytes": 48984,
|
||||||
"sha256": "f4ab15154ed2cefd0aa86346d874c6f3b54c8455e6c58f1b20cd83750b8402e0",
|
"sha256": "7f3aa9965d3c893efb3c6ffa96f68cce219b00c1bd146a9b847f4dbc25ec203b",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"band": 2,
|
"band": 2,
|
||||||
"file": "band-2.json.gz",
|
"file": "band-2.json.gz",
|
||||||
"lemmas": 1450,
|
"lemmas": 1456,
|
||||||
"surfaces": 2590,
|
"surfaces": 2596,
|
||||||
"bytes": 54698,
|
"bytes": 55436,
|
||||||
"sha256": "2a48e603909cd969423cc444cfac215e7a2e51c6129a80e0779cbeed0c81ebd1",
|
"sha256": "4420ee704d1e16de6d9e5e721b3e9656482f7fd9f7930b9276ca1a9299031532",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"band": 3,
|
"band": 3,
|
||||||
"file": "band-3.json.gz",
|
"file": "band-3.json.gz",
|
||||||
"lemmas": 1972,
|
"lemmas": 1974,
|
||||||
"surfaces": 3373,
|
"surfaces": 3375,
|
||||||
"bytes": 72713,
|
"bytes": 73309,
|
||||||
"sha256": "1b5bf666475b301930a99b2c23a8d7a9e585a3cbd40a4787feeebdf96e6ad2af",
|
"sha256": "64789873ede0cb9876fc01d45b89c144a69c28fa87bd6e7734d26936626a87b7",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"band": 4,
|
"band": 4,
|
||||||
"file": "band-4.json.gz",
|
"file": "band-4.json.gz",
|
||||||
"lemmas": 2987,
|
"lemmas": 2988,
|
||||||
"surfaces": 4775,
|
"surfaces": 4776,
|
||||||
"bytes": 106990,
|
"bytes": 107822,
|
||||||
"sha256": "a1ec6eb219b3b1c303c26296d47e5341329e9773b7a30432729288edbda91957",
|
"sha256": "ec8c31ddf5080f7a6f0a1c92ac4067e764815122d9bdc244bd6b717d8a228d5e",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -60,8 +60,8 @@
|
|||||||
"file": "band-5.json.gz",
|
"file": "band-5.json.gz",
|
||||||
"lemmas": 6963,
|
"lemmas": 6963,
|
||||||
"surfaces": 10347,
|
"surfaces": 10347,
|
||||||
"bytes": 245251,
|
"bytes": 247327,
|
||||||
"sha256": "1f8576b0d5686b67aece0a5b7a2bf291e5680f3fb7f43ec95a49cf7839d5730d",
|
"sha256": "de9b3c2daec3d5086a5711e6384a7239036fb3fd9e69ce4bfeea89458698731d",
|
||||||
"reference": false
|
"reference": false
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -69,8 +69,8 @@
|
|||||||
"file": "band-6.json.gz",
|
"file": "band-6.json.gz",
|
||||||
"lemmas": 15107,
|
"lemmas": 15107,
|
||||||
"surfaces": 19283,
|
"surfaces": 19283,
|
||||||
"bytes": 499732,
|
"bytes": 507133,
|
||||||
"sha256": "f9e2ed69ddb4515192d6e24a8abe863656c993414f6e28948ef59d78c6d0056c",
|
"sha256": "2a893ac64e46af015db7a3a6a29043848284e989e3b92c6f026059d6c3f11d51",
|
||||||
"reference": true
|
"reference": true
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
|
|||||||
Binary file not shown.
@@ -177,6 +177,19 @@ export const MIGRATIONS: Migration[] = [
|
|||||||
`,
|
`,
|
||||||
run: rekeyLemmas,
|
run: rekeyLemmas,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
id: 7,
|
||||||
|
name: "curriculum words are cards — a lemma knows its topic and the unit that introduces it",
|
||||||
|
sql: /* sql */ `
|
||||||
|
-- Reference data from the band files, like the rest of lemma: not
|
||||||
|
-- synced, no updated_at. The build marks exactly one reviewable lemma
|
||||||
|
-- per roadmap word with its unit. The rows arrive with the next band
|
||||||
|
-- load, which a changed dictionary triggers on its own.
|
||||||
|
ALTER TABLE lemma ADD COLUMN topic TEXT;
|
||||||
|
ALTER TABLE lemma ADD COLUMN unit_id TEXT;
|
||||||
|
CREATE INDEX IF NOT EXISTS lemma_unit ON lemma(unit_id);
|
||||||
|
`,
|
||||||
|
},
|
||||||
];
|
];
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -385,6 +385,10 @@ export interface LemmaRow {
|
|||||||
gloss_ko: string;
|
gloss_ko: string;
|
||||||
unit_band: number;
|
unit_band: number;
|
||||||
source: string;
|
source: string;
|
||||||
|
/** The deck topic, or "수업 <unit> <name>" for a curriculum word. */
|
||||||
|
topic?: string | null;
|
||||||
|
/** Set on exactly one lemma per roadmap word: the unit that introduces it. */
|
||||||
|
unit_id?: string | null;
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface SurfaceRow {
|
export interface SurfaceRow {
|
||||||
@@ -414,10 +418,10 @@ export async function insertBand(
|
|||||||
surfaces: SurfaceRow[],
|
surfaces: SurfaceRow[],
|
||||||
): Promise<void> {
|
): Promise<void> {
|
||||||
await db.tx(async (tx) => {
|
await db.tx(async (tx) => {
|
||||||
const lemmaChunk = chunkFor(9);
|
const lemmaChunk = chunkFor(11);
|
||||||
for (let i = 0; i < lemmas.length; i += lemmaChunk) {
|
for (let i = 0; i < lemmas.length; i += lemmaChunk) {
|
||||||
const slice = lemmas.slice(i, i + lemmaChunk);
|
const slice = lemmas.slice(i, i + lemmaChunk);
|
||||||
const values = slice.map(() => "(?,?,?,?,?,?,?,?,?)").join(",");
|
const values = slice.map(() => "(?,?,?,?,?,?,?,?,?,?,?)").join(",");
|
||||||
const params: Params = slice.flatMap((l) => [
|
const params: Params = slice.flatMap((l) => [
|
||||||
l.id,
|
l.id,
|
||||||
l.headword,
|
l.headword,
|
||||||
@@ -428,10 +432,12 @@ export async function insertBand(
|
|||||||
l.gloss_ko,
|
l.gloss_ko,
|
||||||
l.unit_band,
|
l.unit_band,
|
||||||
l.source,
|
l.source,
|
||||||
|
l.topic ?? null,
|
||||||
|
l.unit_id ?? null,
|
||||||
]);
|
]);
|
||||||
await tx.run(
|
await tx.run(
|
||||||
`INSERT OR REPLACE INTO lemma
|
`INSERT OR REPLACE INTO lemma
|
||||||
(id, headword, pos, freq_rank, level, gloss_en, gloss_ko, unit_band, source)
|
(id, headword, pos, freq_rank, level, gloss_en, gloss_ko, unit_band, source, topic, unit_id)
|
||||||
VALUES ${values}`,
|
VALUES ${values}`,
|
||||||
params,
|
params,
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -20,6 +20,9 @@ export interface DeckEntry {
|
|||||||
pos: string;
|
pos: string;
|
||||||
glossEn: string;
|
glossEn: string;
|
||||||
source: string;
|
source: string;
|
||||||
|
topic: string | null;
|
||||||
|
/** The unit that introduces this word, for a roadmap word's card. */
|
||||||
|
unitId: string | null;
|
||||||
card: Card | null;
|
card: Card | null;
|
||||||
status: CardStatus;
|
status: CardStatus;
|
||||||
}
|
}
|
||||||
@@ -44,7 +47,10 @@ function toCard(row: Record<string, unknown>): Card | null {
|
|||||||
*/
|
*/
|
||||||
// 'custom' is here so the learner's own words are reviewable like any
|
// 'custom' is here so the learner's own words are reviewable like any
|
||||||
// other — adding a word you cannot then study would be pointless.
|
// other — adding a word you cannot then study would be pointless.
|
||||||
const REVIEWABLE = "('curated', 'sfx', 'grammar', 'custom')";
|
// 'curriculum' is a roadmap word the curated deck does not hold (먹어, 봤어):
|
||||||
|
// every word a unit introduces is a card, because the recall evidence and
|
||||||
|
// the phase review both hang off one.
|
||||||
|
const REVIEWABLE = "('curated', 'sfx', 'grammar', 'custom', 'curriculum')";
|
||||||
const SENTENCE_SOURCE = "('sentence')";
|
const SENTENCE_SOURCE = "('sentence')";
|
||||||
|
|
||||||
export interface DeckOptions {
|
export interface DeckOptions {
|
||||||
@@ -63,7 +69,8 @@ function sourceClause(opts: DeckOptions): string {
|
|||||||
|
|
||||||
export async function deck(db: Db, opts: DeckOptions = {}): Promise<DeckEntry[]> {
|
export async function deck(db: Db, opts: DeckOptions = {}): Promise<DeckEntry[]> {
|
||||||
const rows = await db.all<Record<string, unknown>>(
|
const rows = await db.all<Record<string, unknown>>(
|
||||||
`SELECT l.id AS lemmaId, l.headword, l.pos, l.gloss_en AS glossEn, l.source, ${CARD_COLUMNS}
|
`SELECT l.id AS lemmaId, l.headword, l.pos, l.gloss_en AS glossEn, l.source,
|
||||||
|
l.topic, l.unit_id AS unitId, ${CARD_COLUMNS}
|
||||||
FROM lemma l LEFT JOIN card c ON c.lemma_id = l.id
|
FROM lemma l LEFT JOIN card c ON c.lemma_id = l.id
|
||||||
WHERE ${sourceClause(opts)}
|
WHERE ${sourceClause(opts)}
|
||||||
ORDER BY l.headword`,
|
ORDER BY l.headword`,
|
||||||
@@ -77,6 +84,8 @@ export async function deck(db: Db, opts: DeckOptions = {}): Promise<DeckEntry[]>
|
|||||||
pos: r.pos as string,
|
pos: r.pos as string,
|
||||||
glossEn: r.glossEn as string,
|
glossEn: r.glossEn as string,
|
||||||
source: r.source as string,
|
source: r.source as string,
|
||||||
|
topic: (r.topic as string | null) ?? null,
|
||||||
|
unitId: (r.unitId as string | null) ?? null,
|
||||||
card,
|
card,
|
||||||
status: statusOf(card),
|
status: statusOf(card),
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -39,7 +39,8 @@ const SELECT_VIA_SURFACE = `
|
|||||||
/** Prefer a curated gloss, then a common word, then anything. */
|
/** Prefer a curated gloss, then a common word, then anything. */
|
||||||
const RANKED = `
|
const RANKED = `
|
||||||
ORDER BY CASE l.source WHEN 'curated' THEN 0 WHEN 'grammar' THEN 1
|
ORDER BY CASE l.source WHEN 'curated' THEN 0 WHEN 'grammar' THEN 1
|
||||||
WHEN 'sentence' THEN 2 WHEN 'sfx' THEN 3 ELSE 4 END,
|
WHEN 'curriculum' THEN 2 WHEN 'sentence' THEN 3
|
||||||
|
WHEN 'sfx' THEN 4 ELSE 5 END,
|
||||||
l.freq_rank IS NULL, l.freq_rank`;
|
l.freq_rank IS NULL, l.freq_rank`;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -80,7 +80,7 @@ async function readBandWords(db: Db, band: number, ceiling: number): Promise<str
|
|||||||
`SELECT DISTINCT headword FROM lemma
|
`SELECT DISTINCT headword FROM lemma
|
||||||
WHERE unit_band <= ? AND unit_band < ?
|
WHERE unit_band <= ? AND unit_band < ?
|
||||||
AND (freq_rank IS NOT NULL AND freq_rank <= ?
|
AND (freq_rank IS NOT NULL AND freq_rank <= ?
|
||||||
OR source IN ('curated','grammar','sentence','sfx'))
|
OR source IN ('curated','grammar','sentence','sfx','curriculum'))
|
||||||
ORDER BY freq_rank IS NULL, freq_rank
|
ORDER BY freq_rank IS NULL, freq_rank
|
||||||
LIMIT ?`,
|
LIMIT ?`,
|
||||||
[band, REFERENCE_BAND, ceiling, VOCAB_CAP * 3],
|
[band, REFERENCE_BAND, ceiling, VOCAB_CAP * 3],
|
||||||
|
|||||||
@@ -65,6 +65,24 @@ describe("loading a band", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("curriculum words", () => {
|
||||||
|
it("arrive as studiable cards, each tagged with the unit that introduces it", async () => {
|
||||||
|
serve(shipped);
|
||||||
|
const { ensureBands } = await loader();
|
||||||
|
await ensureBands(db, 1);
|
||||||
|
const { deck } = await import("@app/domain/cards.js");
|
||||||
|
const entries = await deck(db);
|
||||||
|
|
||||||
|
const eat = entries.find((e) => e.headword === "먹어");
|
||||||
|
expect(eat).toMatchObject({ source: "curriculum", unitId: "2.3", glossEn: "eat" });
|
||||||
|
expect(eat!.lemmaId).toBe(lemmaId("먹어", "form"));
|
||||||
|
|
||||||
|
const tagged = entries.filter((e) => e.unitId?.startsWith("1."));
|
||||||
|
const phase1 = (await import("@app/domain/gate.js")).UNITS.filter((u) => u.phase === 1).flatMap((u) => u.words ?? []);
|
||||||
|
expect(tagged.map((e) => e.headword).sort()).toEqual([...new Set(phase1)].sort());
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
describe("a rebuilt dictionary", () => {
|
describe("a rebuilt dictionary", () => {
|
||||||
it("reloads, and every card and custom word still names its word", async () => {
|
it("reloads, and every card and custom word still names its word", async () => {
|
||||||
serve(shipped);
|
serve(shipped);
|
||||||
|
|||||||
@@ -31,6 +31,7 @@ async function loadLexicon() {
|
|||||||
|
|
||||||
const lemmas = new Map(); // headword -> [pos]
|
const lemmas = new Map(); // headword -> [pos]
|
||||||
const surfaces = new Map(); // form -> analysis
|
const surfaces = new Map(); // form -> analysis
|
||||||
|
const cards = new Map(); // roadmap word -> [{ unit, source }], from unit_id
|
||||||
|
|
||||||
for (const band of manifest.bands) {
|
for (const band of manifest.bands) {
|
||||||
const raw = gunzipSync(await readFile(DICT + band.file));
|
const raw = gunzipSync(await readFile(DICT + band.file));
|
||||||
@@ -40,6 +41,8 @@ async function loadLexicon() {
|
|||||||
const S = data.columns.surface;
|
const S = data.columns.surface;
|
||||||
const hw = L.indexOf("headword");
|
const hw = L.indexOf("headword");
|
||||||
const pos = L.indexOf("pos");
|
const pos = L.indexOf("pos");
|
||||||
|
const unit = L.indexOf("unit_id");
|
||||||
|
const source = L.indexOf("source");
|
||||||
const form = S.indexOf("form");
|
const form = S.indexOf("form");
|
||||||
const analysis = S.indexOf("analysis");
|
const analysis = S.indexOf("analysis");
|
||||||
|
|
||||||
@@ -47,13 +50,17 @@ async function loadLexicon() {
|
|||||||
const w = row[hw];
|
const w = row[hw];
|
||||||
if (!lemmas.has(w)) lemmas.set(w, []);
|
if (!lemmas.has(w)) lemmas.set(w, []);
|
||||||
lemmas.get(w).push(row[pos]);
|
lemmas.get(w).push(row[pos]);
|
||||||
|
if (unit !== -1 && row[unit]) {
|
||||||
|
if (!cards.has(w)) cards.set(w, []);
|
||||||
|
cards.get(w).push({ unit: row[unit], source: row[source] });
|
||||||
|
}
|
||||||
}
|
}
|
||||||
for (const row of data.surfaces) {
|
for (const row of data.surfaces) {
|
||||||
if (!surfaces.has(row[form])) surfaces.set(row[form], row[analysis]);
|
if (!surfaces.has(row[form])) surfaces.set(row[form], row[analysis]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return { manifest, lemmas, surfaces };
|
return { manifest, lemmas, surfaces, cards };
|
||||||
}
|
}
|
||||||
|
|
||||||
const resolves = (lex, word) => lex.lemmas.has(word) || lex.surfaces.has(word);
|
const resolves = (lex, word) => lex.lemmas.has(word) || lex.surfaces.has(word);
|
||||||
@@ -91,6 +98,21 @@ async function main() {
|
|||||||
const deckWords = Object.values(deck.topics).flat();
|
const deckWords = Object.values(deck.topics).flat();
|
||||||
for (const [w] of deckWords) check("deck", "deck.json", w);
|
for (const [w] of deckWords) check("deck", "deck.json", w);
|
||||||
|
|
||||||
|
/* 4. every roadmap word is exactly ONE reviewable card, tagged with the
|
||||||
|
unit that introduces it — the recall evidence, the phase review and
|
||||||
|
the gate's met words all hang off that card */
|
||||||
|
const REVIEWABLE = new Set(["curated", "grammar", "sfx", "curriculum"]);
|
||||||
|
let cardWords = 0;
|
||||||
|
for (const u of units) {
|
||||||
|
for (const w of u.words ?? []) {
|
||||||
|
cardWords++;
|
||||||
|
const tagged = lex.cards.get(w) ?? [];
|
||||||
|
if (tagged.length !== 1 || tagged[0].unit !== u.id || !REVIEWABLE.has(tagged[0].source)) {
|
||||||
|
failures.push({ kind: "card", where: u.id, word: `${w} (${tagged.length} tagged)` });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* ── report ── */
|
/* ── report ── */
|
||||||
const src = lex.manifest.builtWith.dictionary;
|
const src = lex.manifest.builtWith.dictionary;
|
||||||
console.log(`Dictionary: ${src} — ${lex.manifest.totals.lemmas} lemmas, ` +
|
console.log(`Dictionary: ${src} — ${lex.manifest.totals.lemmas} lemmas, ` +
|
||||||
@@ -111,6 +133,7 @@ async function main() {
|
|||||||
report("Roadmap words", roadmapWords, "roadmap");
|
report("Roadmap words", roadmapWords, "roadmap");
|
||||||
report("Spiral targets", revisits, "revisit");
|
report("Spiral targets", revisits, "revisit");
|
||||||
report("Deck words", deckWords.length, "deck");
|
report("Deck words", deckWords.length, "deck");
|
||||||
|
report("Roadmap words as one card each", cardWords, "card");
|
||||||
|
|
||||||
if (failures.length) {
|
if (failures.length) {
|
||||||
console.log(`\nFAIL — ${failures.length} words cannot be glossed.`);
|
console.log(`\nFAIL — ${failures.length} words cannot be glossed.`);
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ import { createHash } from "node:crypto";
|
|||||||
import { fileURLToPath } from "node:url";
|
import { fileURLToPath } from "node:url";
|
||||||
import { DatabaseSync } from "node:sqlite";
|
import { DatabaseSync } from "node:sqlite";
|
||||||
|
|
||||||
import { surfaceForms } from "../../lib/conjugation.js";
|
import { surfaceForms, haeche, past } from "../../lib/conjugation.js";
|
||||||
import { flatten } from "../../lib/gate.js";
|
import { flatten } from "../../lib/gate.js";
|
||||||
import { bandOf, bandForUnit, REFERENCE_BAND, BANDS } from "../../shared/bands.mjs";
|
import { bandOf, bandForUnit, REFERENCE_BAND, BANDS } from "../../shared/bands.mjs";
|
||||||
import { lemmaId } from "../../shared/lemma-id.mjs";
|
import { lemmaId } from "../../shared/lemma-id.mjs";
|
||||||
@@ -153,11 +153,13 @@ async function loadGlossExtra(lex) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/** deck.json — 386 curated words. These glosses beat the dictionary's. */
|
/** deck.json — 386 curated words. These glosses beat the dictionary's. */
|
||||||
async function loadDeck(lex) {
|
async function loadDeck(lex, deckOrder) {
|
||||||
const deck = await readJson(`${ROOT}data/deck.json`);
|
const deck = await readJson(`${ROOT}data/deck.json`);
|
||||||
let n = 0;
|
let n = 0;
|
||||||
for (const [topic, rows] of Object.entries(deck.topics)) {
|
for (const [topic, rows] of Object.entries(deck.topics)) {
|
||||||
for (const [headword, , gloss, pos] of rows) {
|
for (const [headword, , gloss, pos] of rows) {
|
||||||
|
const key = lex.key(headword, pos === "phrase" ? "phrase" : pos);
|
||||||
|
if (!deckOrder.has(key)) deckOrder.set(key, deckOrder.size);
|
||||||
lex.add({
|
lex.add({
|
||||||
headword,
|
headword,
|
||||||
pos: pos === "phrase" ? "phrase" : pos,
|
pos: pos === "phrase" ? "phrase" : pos,
|
||||||
@@ -228,6 +230,89 @@ async function loadGrammar(lex) {
|
|||||||
return g.entries.length;
|
return g.entries.length;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* ── curriculum words are cards ────────────────────────────────────────
|
||||||
|
Every word a unit introduces must be studiable: the recall evidence, the
|
||||||
|
phase-review checklist, the practice set and the gate's "met words" all
|
||||||
|
hang off a card. So each roadmap word gets exactly ONE reviewable lemma
|
||||||
|
marked with the unit that introduces it.
|
||||||
|
|
||||||
|
Sixteen roadmap words have no reviewable row — inflected forms (먹어,
|
||||||
|
봤어) and words the dictionary only knows as something else (자 as
|
||||||
|
"ruler", where unit 2.3 means the 반말 of 자다). Those get a lemma of
|
||||||
|
their own, glossed from the curated verb they conjugate, else from the
|
||||||
|
sentence that uses them, else from the dictionary. */
|
||||||
|
|
||||||
|
/** Sources the review deck draws from. */
|
||||||
|
const REVIEWABLE = new Set(["curated", "grammar", "sfx", "curriculum"]);
|
||||||
|
|
||||||
|
/** A deterministic preference among several reviewable rows for one word. */
|
||||||
|
const POS_ORDER = ["noun", "verb", "adj", "adv", "pron", "num", "det", "particle", "ending", "phrase", "word"];
|
||||||
|
const posRank = (pos) => {
|
||||||
|
const i = POS_ORDER.indexOf(pos);
|
||||||
|
return i === -1 ? POS_ORDER.length : i;
|
||||||
|
};
|
||||||
|
|
||||||
|
function markCurriculum(lex, units, deckOrder) {
|
||||||
|
/* 반말 forms of the curated verbs: 봐 is 보다, 먹었어 is 먹다 in the past. */
|
||||||
|
const formGloss = new Map();
|
||||||
|
for (const e of lex.entries()) {
|
||||||
|
if ((e.pos !== "verb" && e.pos !== "adj") || !REVIEWABLE.has(e.source) || !e.gloss_en) continue;
|
||||||
|
const g = String(e.gloss_en).replace(/^to be /, "").replace(/^to /, "");
|
||||||
|
const present = haeche(e.headword);
|
||||||
|
if (!present) continue;
|
||||||
|
if (!formGloss.has(present)) formGloss.set(present, { gloss: g, note: `반말, from ${e.headword}` });
|
||||||
|
const was = past(present);
|
||||||
|
if (was && !formGloss.has(was)) formGloss.set(was, { gloss: `${g} (past)`, note: `반말 past, from ${e.headword}` });
|
||||||
|
}
|
||||||
|
|
||||||
|
let chosen = 0;
|
||||||
|
let made = 0;
|
||||||
|
const seen = new Set();
|
||||||
|
for (const u of units) {
|
||||||
|
for (const w of u.words ?? []) {
|
||||||
|
if (seen.has(w)) continue;
|
||||||
|
seen.add(w);
|
||||||
|
const topic = `수업 ${u.id} ${u.ko}`;
|
||||||
|
const rows = lex.forHeadword(w);
|
||||||
|
const reviewable = rows
|
||||||
|
.filter((e) => REVIEWABLE.has(e.source))
|
||||||
|
.sort(
|
||||||
|
(a, b) =>
|
||||||
|
(deckOrder.get(lex.key(a.headword, a.pos)) ?? Infinity) -
|
||||||
|
(deckOrder.get(lex.key(b.headword, b.pos)) ?? Infinity) ||
|
||||||
|
["curated", "grammar", "sfx", "curriculum"].indexOf(a.source) -
|
||||||
|
["curated", "grammar", "sfx", "curriculum"].indexOf(b.source) ||
|
||||||
|
posRank(a.pos) - posRank(b.pos) ||
|
||||||
|
a.pos.localeCompare(b.pos),
|
||||||
|
);
|
||||||
|
|
||||||
|
if (reviewable.length) {
|
||||||
|
const e = reviewable[0];
|
||||||
|
lex.add({ ...e, unit_id: u.id, topic: e.topic ?? topic });
|
||||||
|
chosen++;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
const form = formGloss.get(w);
|
||||||
|
const chunk = rows.find((e) => e.source === "sentence" && e.gloss_en);
|
||||||
|
const other = rows.find((e) => e.gloss_en);
|
||||||
|
lex.add({
|
||||||
|
headword: w,
|
||||||
|
pos: form ? "form" : "word",
|
||||||
|
gloss_en: form?.gloss ?? chunk?.gloss_en ?? other?.gloss_en ?? "",
|
||||||
|
gloss_ko: "",
|
||||||
|
note: form?.note ?? "",
|
||||||
|
level: null,
|
||||||
|
source: "curriculum",
|
||||||
|
topic,
|
||||||
|
unit_id: u.id,
|
||||||
|
});
|
||||||
|
made++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
log(` curriculum: ${chosen + made} roadmap words as cards (${made} given a lemma of their own)`);
|
||||||
|
}
|
||||||
|
|
||||||
/* ── build ────────────────────────────────────────────────────────── */
|
/* ── build ────────────────────────────────────────────────────────── */
|
||||||
|
|
||||||
async function main() {
|
async function main() {
|
||||||
@@ -236,11 +321,16 @@ async function main() {
|
|||||||
await mkdir(OUT, { recursive: true });
|
await mkdir(OUT, { recursive: true });
|
||||||
|
|
||||||
const lex = new Lexicon();
|
const lex = new Lexicon();
|
||||||
|
const deckOrder = new Map();
|
||||||
const dict = await loadDictionary(lex);
|
const dict = await loadDictionary(lex);
|
||||||
await loadGlossExtra(lex);
|
await loadGlossExtra(lex);
|
||||||
await loadDeck(lex);
|
await loadDeck(lex, deckOrder);
|
||||||
await loadSentencesAndSfx(lex);
|
await loadSentencesAndSfx(lex);
|
||||||
await loadGrammar(lex);
|
await loadGrammar(lex);
|
||||||
|
|
||||||
|
const curriculum = await readJson(`${ROOT}data/curriculum.json`);
|
||||||
|
const units = flatten(curriculum);
|
||||||
|
markCurriculum(lex, units, deckOrder);
|
||||||
log(` merged: ${lex.size} distinct (headword, pos)\n`);
|
log(` merged: ${lex.size} distinct (headword, pos)\n`);
|
||||||
|
|
||||||
/* Frequency, via the inverted join — see freq-forms.mjs. */
|
/* Frequency, via the inverted join — see freq-forms.mjs. */
|
||||||
@@ -252,8 +342,6 @@ async function main() {
|
|||||||
|
|
||||||
/* Which unit introduces a word, so curriculum vocabulary is admitted by
|
/* Which unit introduces a word, so curriculum vocabulary is admitted by
|
||||||
the band of its own unit rather than by its raw frequency. */
|
the band of its own unit rather than by its raw frequency. */
|
||||||
const curriculum = await readJson(`${ROOT}data/curriculum.json`);
|
|
||||||
const units = flatten(curriculum);
|
|
||||||
const introducedIn = new Map();
|
const introducedIn = new Map();
|
||||||
for (const u of units) {
|
for (const u of units) {
|
||||||
for (const w of u.words ?? []) if (!introducedIn.has(w)) introducedIn.set(w, u.id);
|
for (const w of u.words ?? []) if (!introducedIn.has(w)) introducedIn.set(w, u.id);
|
||||||
@@ -297,6 +385,8 @@ async function main() {
|
|||||||
gloss_ko: e.gloss_ko ?? "",
|
gloss_ko: e.gloss_ko ?? "",
|
||||||
unit_band: band,
|
unit_band: band,
|
||||||
source: e.source,
|
source: e.source,
|
||||||
|
topic: e.topic ?? null,
|
||||||
|
unit_id: e.unit_id ?? null,
|
||||||
});
|
});
|
||||||
|
|
||||||
/* The surface index. Strictly surfaceForms() plus the headword itself —
|
/* The surface index. Strictly surfaceForms() plus the headword itself —
|
||||||
@@ -341,11 +431,11 @@ async function main() {
|
|||||||
band,
|
band,
|
||||||
format: 2,
|
format: 2,
|
||||||
columns: {
|
columns: {
|
||||||
lemma: ["headword", "pos", "freq_rank", "level", "gloss_en", "gloss_ko", "unit_band", "source"],
|
lemma: ["headword", "pos", "freq_rank", "level", "gloss_en", "gloss_ko", "unit_band", "source", "topic", "unit_id"],
|
||||||
surface: ["form", "lemma", "analysis"],
|
surface: ["form", "lemma", "analysis"],
|
||||||
},
|
},
|
||||||
lemmas: data.lemmas.map((l) => [
|
lemmas: data.lemmas.map((l) => [
|
||||||
l.headword, l.pos, l.freq_rank, l.level, l.gloss_en, l.gloss_ko, l.unit_band, l.source,
|
l.headword, l.pos, l.freq_rank, l.level, l.gloss_en, l.gloss_ko, l.unit_band, l.source, l.topic, l.unit_id,
|
||||||
]),
|
]),
|
||||||
surfaces: data.surfaces.map((s) => [s.form, rowOf.get(s.lemma_id), s.analysis]),
|
surfaces: data.surfaces.map((s) => [s.form, rowOf.get(s.lemma_id), s.analysis]),
|
||||||
};
|
};
|
||||||
@@ -379,17 +469,18 @@ async function main() {
|
|||||||
seed.exec(`
|
seed.exec(`
|
||||||
CREATE TABLE lemma (id INTEGER PRIMARY KEY, headword TEXT NOT NULL, pos TEXT NOT NULL,
|
CREATE TABLE lemma (id INTEGER PRIMARY KEY, headword TEXT NOT NULL, pos TEXT NOT NULL,
|
||||||
freq_rank INTEGER, level TEXT, gloss_en TEXT NOT NULL DEFAULT '',
|
freq_rank INTEGER, level TEXT, gloss_en TEXT NOT NULL DEFAULT '',
|
||||||
gloss_ko TEXT NOT NULL DEFAULT '', unit_band INTEGER NOT NULL DEFAULT 0, source TEXT NOT NULL);
|
gloss_ko TEXT NOT NULL DEFAULT '', unit_band INTEGER NOT NULL DEFAULT 0, source TEXT NOT NULL,
|
||||||
|
topic TEXT, unit_id TEXT);
|
||||||
CREATE TABLE surface (form TEXT NOT NULL, lemma_id INTEGER NOT NULL, analysis TEXT NOT NULL,
|
CREATE TABLE surface (form TEXT NOT NULL, lemma_id INTEGER NOT NULL, analysis TEXT NOT NULL,
|
||||||
PRIMARY KEY (form, lemma_id)) WITHOUT ROWID;
|
PRIMARY KEY (form, lemma_id)) WITHOUT ROWID;
|
||||||
CREATE INDEX surface_form ON surface(form);
|
CREATE INDEX surface_form ON surface(form);
|
||||||
`);
|
`);
|
||||||
const band0 = byBand.get(0) ?? { lemmas: [], surfaces: [] };
|
const band0 = byBand.get(0) ?? { lemmas: [], surfaces: [] };
|
||||||
const li = seed.prepare("INSERT INTO lemma VALUES (?,?,?,?,?,?,?,?,?)");
|
const li = seed.prepare("INSERT INTO lemma VALUES (?,?,?,?,?,?,?,?,?,?,?)");
|
||||||
const si = seed.prepare("INSERT OR IGNORE INTO surface VALUES (?,?,?)");
|
const si = seed.prepare("INSERT OR IGNORE INTO surface VALUES (?,?,?)");
|
||||||
seed.exec("BEGIN");
|
seed.exec("BEGIN");
|
||||||
for (const l of band0.lemmas)
|
for (const l of band0.lemmas)
|
||||||
li.run(l.id, l.headword, l.pos, l.freq_rank, l.level, l.gloss_en, l.gloss_ko, l.unit_band, l.source);
|
li.run(l.id, l.headword, l.pos, l.freq_rank, l.level, l.gloss_en, l.gloss_ko, l.unit_band, l.source, l.topic, l.unit_id);
|
||||||
for (const s of band0.surfaces) si.run(s.form, s.lemma_id, s.analysis);
|
for (const s of band0.surfaces) si.run(s.form, s.lemma_id, s.analysis);
|
||||||
seed.exec("COMMIT");
|
seed.exec("COMMIT");
|
||||||
seed.close();
|
seed.close();
|
||||||
|
|||||||
Reference in New Issue
Block a user