diff --git a/.gitignore b/.gitignore index cb4f8db..34175dd 100644 --- a/.gitignore +++ b/.gitignore @@ -13,4 +13,6 @@ __pycache__/ data-pipeline/kaikki-source-files/ data-pipeline/db/staging.db data-pipeline/.env +data-pipeline/responses/ +data-pipeline/rejections/ .aider* diff --git a/data-pipeline/gemini.ts b/data-pipeline/gemini.ts new file mode 100644 index 0000000..9006e73 --- /dev/null +++ b/data-pipeline/gemini.ts @@ -0,0 +1,171 @@ +import { + DIFFICULTY_LEVELS, + NOUN_GENDERS, + type SupportedLanguageCode, + type SupportedPos, +} from "@lila/shared"; + +export const DEFAULT_MODEL = "gemini-3.6-flash"; + +const API_BASE = "https://generativelanguage.googleapis.com/v1beta/models"; +const MAX_ATTEMPTS = 5; +const RETRYABLE_STATUS = new Set([429, 500, 503]); + +type ResponseSchema = Record; + +/** + * OpenAPI-subset schema for Gemini structured output: an array of word + * entries matching design-doc §6.3, with enums narrowed to this batch's + * source/POS/target languages. + */ +export const buildEntriesResponseSchema = ( + sourceLanguage: SupportedLanguageCode, + pos: SupportedPos, + targetLanguages: readonly SupportedLanguageCode[], +): ResponseSchema => ({ + type: "ARRAY", + items: { + type: "OBJECT", + properties: { + headword: { type: "STRING" }, + language: { type: "STRING", enum: [sourceLanguage] }, + pos: { type: "STRING", enum: [pos] }, + senses: { + type: "ARRAY", + items: { + type: "OBJECT", + properties: { + sense_index: { type: "INTEGER" }, + difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] }, + definitions: { type: "ARRAY", items: { type: "STRING" } }, + examples: { type: "ARRAY", items: { type: "STRING" } }, + translations: { + type: "ARRAY", + items: { + type: "OBJECT", + properties: { + target_language: { + type: "STRING", + enum: [...targetLanguages], + }, + word: { type: "STRING" }, + gender: { + type: "STRING", + enum: [...NOUN_GENDERS], + nullable: true, + }, + difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] }, + }, + required: ["target_language", "word", "gender", "difficulty"], + }, + }, + }, + required: [ + "sense_index", + "difficulty", + "definitions", + "examples", + "translations", + ], + }, + }, + }, + required: ["headword", "language", "pos", "senses"], + }, +}); + +const sleep = (ms: number): Promise => + new Promise((resolve) => setTimeout(resolve, ms)); + +const retryDelayMs = (body: string, attempt: number): number => { + const match = body.match(/"retryDelay":\s*"(\d+(?:\.\d+)?)s"/); + if (match?.[1] !== undefined) { + return Math.ceil(Number(match[1]) * 1000) + 500; + } + return 2 ** attempt * 2000; +}; + +const extractText = (body: unknown): string => { + if (isRecord(body)) { + const candidates = body["candidates"]; + if (Array.isArray(candidates) && isRecord(candidates[0])) { + const candidate = candidates[0]; + const finishReason = candidate["finishReason"]; + if (finishReason !== undefined && finishReason !== "STOP") { + throw new Error( + `Gemini stopped early: finishReason=${JSON.stringify(finishReason)}`, + ); + } + const content = candidate["content"]; + if (isRecord(content)) { + const parts = content["parts"]; + if (Array.isArray(parts) && isRecord(parts[0])) { + const text = parts[0]["text"]; + if (typeof text === "string") return text; + } + } + } + } + throw new Error("Gemini response contained no text candidate"); +}; + +const isRecord = (value: unknown): value is Record => + typeof value === "object" && value !== null && !Array.isArray(value); + +export const generateContent = async ( + apiKey: string, + model: string, + prompt: string, + responseSchema: ResponseSchema, +): Promise => { + let lastError = ""; + for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) { + let response: Response; + try { + response = await fetch(`${API_BASE}/${model}:generateContent`, { + method: "POST", + headers: { + "Content-Type": "application/json", + "x-goog-api-key": apiKey, + }, + body: JSON.stringify({ + contents: [{ role: "user", parts: [{ text: prompt }] }], + generationConfig: { + responseMimeType: "application/json", + responseSchema, + temperature: 0.2, + }, + }), + }); + } catch (error) { + lastError = `network error: ${error instanceof Error ? error.message : String(error)}`; + await sleep(2 ** attempt * 2000); + continue; + } + + if (response.ok) { + return extractText(await response.json()); + } + + const body = await response.text(); + lastError = `HTTP ${response.status}: ${body.slice(0, 500)}`; + if (!RETRYABLE_STATUS.has(response.status)) { + break; + } + await sleep(retryDelayMs(body, attempt)); + } + throw new Error(`Gemini request failed after retries — ${lastError}`); +}; + +/** Parse the model's JSON text into an array of unknown entries. */ +export const parseEntries = (rawText: string): unknown[] => { + const stripped = rawText + .trim() + .replace(/^```(?:json)?\s*/i, "") + .replace(/\s*```$/, ""); + const parsed: unknown = JSON.parse(stripped); + if (!Array.isArray(parsed)) { + throw new Error("Gemini response is not a JSON array"); + } + return parsed; +}; diff --git a/data-pipeline/pipeline.ts b/data-pipeline/pipeline.ts index 4379edc..baf40df 100644 --- a/data-pipeline/pipeline.ts +++ b/data-pipeline/pipeline.ts @@ -1,45 +1,327 @@ -// pipeline.ts pseudo code +import { appendFileSync, mkdirSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { parseArgs } from "node:util"; +import { + SUPPORTED_LANGUAGE_CODES, + SUPPORTED_POS, + type SupportedLanguageCode, + type SupportedPos, +} from "@lila/shared"; +import { + DEFAULT_MODEL, + buildEntriesResponseSchema, + generateContent, + parseEntries, +} from "./gemini.js"; +import { loadPromptTemplate, renderPrompt } from "./promptTemplate.js"; +import { discoverSourceLists, type SourceList } from "./sourceLists.js"; +import { + countStagedRows, + getStagedHeadwords, + openStaging, + stageEntry, +} from "./staging.js"; +import { validateEntry, type ValidationContext } from "./validate.js"; -/* -step 1: discover source lists +const ROOT = import.meta.dirname; +const SOURCE_DATA_DIR = path.join(ROOT, "source-data"); +const STAGING_DB_PATH = path.join(ROOT, "db", "staging.db"); +const STAGING_SCHEMA_PATH = path.join(ROOT, "db", "schema.sql"); +const RESPONSES_DIR = path.join(ROOT, "responses"); +const REJECTIONS_DIR = path.join(ROOT, "rejections"); -this will give us an array of objects with this schema -sourceLanguage: 'en' | 'de' | 'es' | 'fr' | 'it' -pos: 'noun' | 'verb' | 'adjective' | 'adverb' -words: string[]; -filePath: string; +type CliOptions = { + langs: SupportedLanguageCode[] | null; + pos: SupportedPos; + maxBatches: number | null; + batchSize: number; + delayMs: number; + dryRun: boolean; +}; -the terminal output should be something like: +type ListStats = { + list: SourceList; + pending: number; + batchesRun: number; + staged: number; + skipped: number; + rejected: number; + failedBatches: number; +}; -found 5 source lists: +const parseCli = (): CliOptions => { + // pnpm forwards the "--" separator itself (pnpm pipeline:run -- --langs …); + // drop it so the flags after it are parsed as flags, not positionals. + const args = process.argv.slice(2); + if (args[0] === "--") args.shift(); + const { values } = parseArgs({ + args, + options: { + langs: { type: "string" }, + pos: { type: "string", default: "noun" }, + "max-batches": { type: "string" }, + "batch-size": { type: "string", default: "20" }, + "delay-ms": { type: "string", default: "6000" }, + "dry-run": { type: "boolean", default: false }, + }, + }); -de: noun -en: noun + const pos = values.pos as SupportedPos; + if (!(SUPPORTED_POS as readonly string[]).includes(pos)) { + throw new Error(`--pos must be one of ${SUPPORTED_POS.join(", ")}`); + } + let langs: SupportedLanguageCode[] | null = null; + if (values.langs !== undefined) { + langs = values.langs.split(",").map((code) => { + const trimmed = code.trim() as SupportedLanguageCode; + if (!(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(trimmed)) { + throw new Error(`--langs: unknown language code "${trimmed}"`); + } + return trimmed; + }); + } + return { + langs, + pos, + maxBatches: + values["max-batches"] !== undefined + ? Number(values["max-batches"]) + : null, + batchSize: Number(values["batch-size"]), + delayMs: Number(values["delay-ms"]), + dryRun: values["dry-run"], + }; +}; -and so on +const chunk = (items: readonly T[], size: number): T[][] => { + const chunks: T[][] = []; + for (let i = 0; i < items.length; i += size) { + chunks.push(items.slice(i, i + size)); + } + return chunks; +}; -later on, it will also contain it: noun, verb, adjective etc - */ +const sleep = (ms: number): Promise => + new Promise((resolve) => setTimeout(resolve, ms)); -/* +const rejectionFile = (list: SourceList): string => + path.join(REJECTIONS_DIR, `${list.sourceLanguage}-${list.pos}.jsonl`); -step 2: validating source lists +const logRejection = ( + list: SourceList, + headword: string | null, + errors: string[], + entry: unknown, +): void => { + mkdirSync(REJECTIONS_DIR, { recursive: true }); + const line = JSON.stringify({ + at: new Date().toISOString(), + sourceLanguage: list.sourceLanguage, + pos: list.pos, + headword, + errors, + entry, + }); + appendFileSync(rejectionFile(list), `${line}\n`); +}; -a small script that trims whitespaces, removes duplicated words etc +const saveRawResponse = ( + list: SourceList, + batchIndex: number, + model: string, + words: readonly string[], + targetLanguages: readonly SupportedLanguageCode[], + rawText: string, +): void => { + mkdirSync(RESPONSES_DIR, { recursive: true }); + const stamp = new Date().toISOString().replaceAll(":", "-"); + const file = path.join( + RESPONSES_DIR, + `${list.sourceLanguage}-${list.pos}-${stamp}-batch${batchIndex}.json`, + ); + writeFileSync( + file, + JSON.stringify( + { + model, + sourceLanguage: list.sourceLanguage, + pos: list.pos, + targetLanguages, + words, + receivedAt: new Date().toISOString(), + rawText, + }, + null, + 2, + ), + ); +}; -terminal output: summary of how many words per pos per language were found +const headwordOf = (entry: unknown): string | null => { + if (typeof entry === "object" && entry !== null && !Array.isArray(entry)) { + const headword = (entry as Record)["headword"]; + if (typeof headword === "string") return headword; + } + return null; +}; -*/ +const main = async (): Promise => { + const options = parseCli(); + const model = process.env["GEMINI_MODEL"] ?? DEFAULT_MODEL; + const apiKey = process.env["GEMINI_API_KEY"]; + if (apiKey === undefined && !options.dryRun) { + throw new Error("GEMINI_API_KEY is not set (data-pipeline/.env)"); + } -/* + const allLists = discoverSourceLists(SOURCE_DATA_DIR); + console.log(`found ${allLists.length} source lists:\n`); + for (const list of allLists) { + console.log( + ` ${list.sourceLanguage}: ${list.pos} (${list.words.length} unique words)`, + ); + } -step 3: writing to database? + const lists = allLists.filter( + (list) => + list.pos === options.pos && + (options.langs === null || options.langs.includes(list.sourceLanguage)), + ); + if (lists.length === 0) { + console.log("\nnothing matches the requested --langs/--pos, exiting"); + return; + } -my thought: ill restart the pipeline several times during testing, and when adding more wordlists with other pos or extending the exisiting noun lists -eventually the lists will contain tens or hundreds of thousands of words -how do we prevent reading and processing the same words multiple times? -if we read and validate+normalize the wordlists and write them to the database, we could then read from the database fill the missing translations etc -and not read the same words from the same text files multiple times? + const template = loadPromptTemplate(); + const db = openStaging(STAGING_DB_PATH, STAGING_SCHEMA_PATH); + const allStats: ListStats[] = []; + let firstApiCall = true; -if we do this, we have to adjust the database schema because there are several notNull() rows inside -*/ + for (const list of lists) { + const staged = getStagedHeadwords(db, list.sourceLanguage, list.pos); + const pending = list.words.filter((word) => !staged.has(word)); + const targetLanguages = SUPPORTED_LANGUAGE_CODES.filter( + (code) => code !== list.sourceLanguage, + ); + const batches = chunk(pending, options.batchSize).slice( + 0, + options.maxBatches ?? Number.POSITIVE_INFINITY, + ); + + console.log( + `\n${list.sourceLanguage}/${list.pos}: ${list.words.length} unique, ${staged.size} already staged, ${pending.length} pending → running ${batches.length} batch(es)`, + ); + const stats: ListStats = { + list, + pending: pending.length, + batchesRun: 0, + staged: 0, + skipped: 0, + rejected: 0, + failedBatches: 0, + }; + allStats.push(stats); + + for (const [batchIndex, words] of batches.entries()) { + const prompt = renderPrompt(template, { + sourceLanguage: list.sourceLanguage, + pos: list.pos, + targetLanguages, + words, + }); + + if (options.dryRun) { + console.log( + ` [dry-run] batch ${batchIndex + 1}/${batches.length}: ${words.join(", ")}`, + ); + stats.batchesRun++; + continue; + } + + if (!firstApiCall) { + await sleep(options.delayMs); + } + firstApiCall = false; + + try { + const rawText = await generateContent( + apiKey as string, + model, + prompt, + buildEntriesResponseSchema( + list.sourceLanguage, + list.pos, + targetLanguages, + ), + ); + saveRawResponse( + list, + batchIndex + 1, + model, + words, + targetLanguages, + rawText, + ); + + const entries = parseEntries(rawText); + const ctx: ValidationContext = { + sourceLanguage: list.sourceLanguage, + pos: list.pos, + targetLanguages, + inputWords: new Set(words), + }; + const covered = new Set(); + for (const entry of entries) { + const result = validateEntry(entry, ctx); + if (result.status === "valid") { + stageEntry(db, result.entry); + covered.add(result.entry.headword); + stats.staged++; + } else if (result.status === "empty") { + covered.add(result.headword); + stats.skipped++; + console.log( + ` skipped "${result.headword}" (no valid ${list.pos} senses)`, + ); + } else { + const headword = headwordOf(entry); + if (headword !== null) covered.add(headword); + logRejection(list, headword, result.errors, entry); + stats.rejected++; + } + } + for (const word of words) { + if (!covered.has(word)) { + logRejection(list, word, ["missing from Gemini response"], null); + stats.rejected++; + } + } + stats.batchesRun++; + console.log( + ` batch ${batchIndex + 1}/${batches.length} done — ${stats.staged} staged, ${stats.rejected} rejected, ${stats.skipped} skipped`, + ); + } catch (error) { + stats.failedBatches++; + console.error( + ` batch ${batchIndex + 1}/${batches.length} FAILED: ${error instanceof Error ? error.message : String(error)}`, + ); + } + } + } + + console.log("\n— summary —"); + for (const stats of allStats) { + console.log( + `${stats.list.sourceLanguage}/${stats.list.pos}: ${stats.staged} staged, ${stats.skipped} skipped, ${stats.rejected} rejected, ${stats.failedBatches} failed batch(es), ${stats.pending - stats.staged - stats.skipped - stats.rejected} still pending`, + ); + } + const totals = countStagedRows(db); + console.log( + `staging.db totals: ${totals.words} words, ${totals.senses} senses, ${totals.translations} translations`, + ); + db.close(); +}; + +main().catch((error: unknown) => { + console.error(error instanceof Error ? error.message : error); + process.exitCode = 1; +}); diff --git a/data-pipeline/prompt b/data-pipeline/prompt index 09275b0..c6d7301 100644 --- a/data-pipeline/prompt +++ b/data-pipeline/prompt @@ -1,31 +1,11 @@ You are a multilingual lexicographer generating vocabulary data for a language-learning app. -Source language: spanish -Part of speech: noun -Target languages: en, it, de, fr +Source language: {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}") +Part of speech: {{POS}} +Target languages: {{TARGET_LANGUAGE_CODES}} Input words: -suelo -pared -techo -portón -valla -esquina -centro -borde -superficie -centro -suburbio -medioambiente -oportunidad -ventaja -decisión -paciencia -comportamiento -industria -conocimiento -solución - +{{INPUT_WORDS}} Return ONLY valid JSON. Do not include markdown fences. @@ -39,8 +19,8 @@ Each word object must have this shape: { "headword": string, - "language": "es", - "pos": "noun", + "language": "{{SOURCE_LANGUAGE_CODE}}", + "pos": "{{POS}}", "senses": [ { "sense_index": number, @@ -49,7 +29,7 @@ Each word object must have this shape: "examples": string[], "translations": [ { - "target_language": "de" | "it" | "en" | "fr", + "target_language": {{TARGET_LANGUAGE_UNION}}, "word": string, "gender": "masculine" | "feminine" | "neuter" | null, "difficulty": "easy" | "medium" | "hard" @@ -62,36 +42,36 @@ Each word object must have this shape: Rules: 1. The headword must be exactly one of the input words. -2. language must be "en". -3. pos must be "noun". +2. language must be "{{SOURCE_LANGUAGE_CODE}}". +3. pos must be "{{POS}}". 4. Include only common, learner-relevant senses. 5. Most words should have 1 sense. 6. Polysemous words may have 2 or 3 senses. 7. Do not include rare, archaic, highly technical, or literary senses unless they are common. 8. sense_index must start at 0 and increase by 1. -9. definitions must be written in English. -10. examples must be written in English. +9. definitions must be written in {{SOURCE_LANGUAGE_NAME}}. +10. examples must be written in {{SOURCE_LANGUAGE_NAME}}. 11. Include 1 or 2 definitions per sense. 12. Include 1 or 2 example sentences per sense. 13. Each definition must be student-friendly and at most 15 words. 14. Each example should naturally contain the headword or a clear form of it. -15. Every sense must have translations for all target languages: de, it, es, fr. -16. Do not include English as a target_language. +15. Every sense must have translations for all target languages: {{TARGET_LANGUAGE_CODES}}. +16. Do not include {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}") as a target_language. 17. You may include up to 2 translations per target language per sense if they are genuinely common synonyms or difficulty variants. 18. Do not include more than 2 translations per target language per sense. 19. Do not duplicate the same translation word for the same target language within one sense. -20. Use the base dictionary form of the translated noun. +20. Use the base dictionary form of the translated {{POS}}. 21. Do not include articles or determiners in translations. 22. German translation nouns must be capitalized. 23. Spanish, French, and Italian translation nouns should be lowercase unless they are proper nouns. 24. For target_language "de", gender must be "masculine", "feminine", or "neuter". 25. For target_language "it", "es", or "fr", gender must be "masculine" or "feminine". -26. For this prompt, gender must never be null. +26. gender must be null if and only if target_language is "en". 27. difficulty must be one of: "easy", "medium", "hard". 28. senses.difficulty describes how common or advanced the meaning is. 29. translations.difficulty describes how difficult the specific target-language word is for a learner. 30. A common translation like "Bank" may be easy, while a formal synonym like "Geldinstitut" may be medium. -31. If a word cannot be treated as a valid English noun, return it with "senses": []. +31. If a word cannot be treated as a valid {{SOURCE_LANGUAGE_NAME}} {{POS}}, return it with "senses": []. Difficulty calibration: @@ -101,7 +81,7 @@ Difficulty calibration: Do not include CEFR levels in the output. -Example output shape for the English noun "bank": +Example output shape for the English noun "bank" (illustrative of the JSON shape only — your definitions and examples must be in {{SOURCE_LANGUAGE_NAME}}): [ { diff --git a/data-pipeline/promptTemplate.ts b/data-pipeline/promptTemplate.ts new file mode 100644 index 0000000..3ef38d3 --- /dev/null +++ b/data-pipeline/promptTemplate.ts @@ -0,0 +1,56 @@ +import { readFileSync } from "node:fs"; +import path from "node:path"; +import type { SupportedLanguageCode, SupportedPos } from "@lila/shared"; + +export const LANGUAGE_NAMES: Record = { + en: "English", + it: "Italian", + de: "German", + fr: "French", + es: "Spanish", +}; + +export type PromptParams = { + sourceLanguage: SupportedLanguageCode; + pos: SupportedPos; + targetLanguages: readonly SupportedLanguageCode[]; + words: readonly string[]; +}; + +export const loadPromptTemplate = (): string => + readFileSync(path.join(import.meta.dirname, "prompt"), "utf-8"); + +export const renderPrompt = ( + template: string, + params: PromptParams, +): string => { + const { sourceLanguage, pos, targetLanguages, words } = params; + if (words.length === 0) { + throw new Error("renderPrompt: empty word batch"); + } + if ( + targetLanguages.length === 0 || + targetLanguages.includes(sourceLanguage) + ) { + throw new Error( + `renderPrompt: target languages must be non-empty and exclude the source language (got ${targetLanguages.join(", ")})`, + ); + } + + const rendered = template + .replaceAll("{{SOURCE_LANGUAGE_NAME}}", LANGUAGE_NAMES[sourceLanguage]) + .replaceAll("{{SOURCE_LANGUAGE_CODE}}", sourceLanguage) + .replaceAll("{{POS}}", pos) + .replaceAll("{{TARGET_LANGUAGE_CODES}}", targetLanguages.join(", ")) + .replaceAll( + "{{TARGET_LANGUAGE_UNION}}", + targetLanguages.map((code) => `"${code}"`).join(" | "), + ) + .replaceAll("{{INPUT_WORDS}}", words.join("\n")); + + const leftover = rendered.match(/\{\{[A-Z_]+\}\}/); + if (leftover) { + throw new Error(`renderPrompt: unreplaced placeholder ${leftover[0]}`); + } + return rendered; +}; diff --git a/data-pipeline/sourceLists.ts b/data-pipeline/sourceLists.ts new file mode 100644 index 0000000..520b2ca --- /dev/null +++ b/data-pipeline/sourceLists.ts @@ -0,0 +1,56 @@ +import { readdirSync, readFileSync } from "node:fs"; +import path from "node:path"; +import { + SUPPORTED_LANGUAGE_CODES, + SUPPORTED_POS, + type SupportedLanguageCode, + type SupportedPos, +} from "@lila/shared"; + +export type SourceList = { + sourceLanguage: SupportedLanguageCode; + pos: SupportedPos; + words: string[]; + filePath: string; +}; + +const isLanguageCode = (value: string): value is SupportedLanguageCode => + (SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(value); + +const isPos = (value: string): value is SupportedPos => + (SUPPORTED_POS as readonly string[]).includes(value); + +export const normalizeWords = (lines: readonly string[]): string[] => { + const seen = new Set(); + const words: string[] = []; + for (const line of lines) { + const word = line.trim(); + if (word === "" || seen.has(word)) continue; + seen.add(word); + words.push(word); + } + return words; +}; + +export const discoverSourceLists = (rootDir: string): SourceList[] => { + const lists: SourceList[] = []; + const languageDirs = readdirSync(rootDir, { withFileTypes: true }) + .filter((entry) => entry.isDirectory() && isLanguageCode(entry.name)) + .map((entry) => entry.name as SupportedLanguageCode) + .sort(); + + for (const language of languageDirs) { + const languageDir = path.join(rootDir, language); + const posFiles = readdirSync(languageDir, { withFileTypes: true }) + .filter((entry) => entry.isFile() && isPos(entry.name)) + .map((entry) => entry.name as SupportedPos) + .sort(); + + for (const pos of posFiles) { + const filePath = path.join(languageDir, pos); + const words = normalizeWords(readFileSync(filePath, "utf-8").split("\n")); + lists.push({ sourceLanguage: language, pos, words, filePath }); + } + } + return lists; +}; diff --git a/data-pipeline/staging.ts b/data-pipeline/staging.ts new file mode 100644 index 0000000..cfb266d --- /dev/null +++ b/data-pipeline/staging.ts @@ -0,0 +1,123 @@ +import { randomUUID } from "node:crypto"; +import { readFileSync } from "node:fs"; +import Database from "better-sqlite3"; +import type { SupportedLanguageCode, SupportedPos } from "@lila/shared"; +import type { GeminiWordEntry } from "./validate.js"; + +const STAGING_TABLES = ["words", "senses", "translations"] as const; + +export const openStaging = ( + dbPath: string, + schemaPath: string, +): Database.Database => { + const db = new Database(dbPath); + db.pragma("foreign_keys = ON"); + + const existing = db + .prepare<[], { name: string }>( + `SELECT name FROM sqlite_master WHERE type = 'table' AND name IN ('words', 'senses', 'translations')`, + ) + .all() + .map((row) => row.name); + + if (existing.length === 0) { + db.exec(readFileSync(schemaPath, "utf-8")); + } else if (existing.length !== STAGING_TABLES.length) { + db.close(); + throw new Error( + `staging database at ${dbPath} has a partial schema (found: ${existing.join(", ")}) — fix or delete it`, + ); + } + return db; +}; + +export const getStagedHeadwords = ( + db: Database.Database, + language: SupportedLanguageCode, + pos: SupportedPos, +): Set => { + const rows = db + .prepare< + [string, string], + { headword: string } + >(`SELECT headword FROM words WHERE language_code = ? AND pos = ?`) + .all(language, pos); + return new Set(rows.map((row) => row.headword)); +}; + +/** + * Writes one validated entry (word + senses + translations) atomically. + * Returns "already-staged" without writing if the word exists — a partial + * word can never be left behind, so no NOT NULL relaxation is needed. + */ +export const stageEntry = ( + db: Database.Database, + entry: GeminiWordEntry, +): "staged" | "already-staged" => { + const insert = db.transaction((): "staged" | "already-staged" => { + const existing = db + .prepare< + [string, string, string], + { id: string } + >(`SELECT id FROM words WHERE headword = ? AND language_code = ? AND pos = ?`) + .get(entry.headword, entry.language, entry.pos); + if (existing !== undefined) { + return "already-staged"; + } + + const wordId = randomUUID(); + db.prepare( + `INSERT INTO words (id, headword, language_code, pos) VALUES (?, ?, ?, ?)`, + ).run(wordId, entry.headword, entry.language, entry.pos); + + const insertSense = db.prepare( + `INSERT INTO senses (id, word_id, sense_index, difficulty, definitions, examples) + VALUES (?, ?, ?, ?, ?, ?)`, + ); + const insertTranslation = db.prepare( + `INSERT INTO translations (id, sense_id, target_language_code, translation, gender, difficulty) + VALUES (?, ?, ?, ?, ?, ?)`, + ); + + for (const sense of entry.senses) { + const senseId = randomUUID(); + insertSense.run( + senseId, + wordId, + sense.sense_index, + sense.difficulty, + JSON.stringify(sense.definitions), + JSON.stringify(sense.examples), + ); + for (const translation of sense.translations) { + insertTranslation.run( + randomUUID(), + senseId, + translation.target_language, + translation.word, + translation.gender, + translation.difficulty, + ); + } + } + return "staged"; + }); + return insert(); +}; + +export type StagingCounts = { + words: number; + senses: number; + translations: number; +}; + +export const countStagedRows = (db: Database.Database): StagingCounts => { + const count = (table: (typeof STAGING_TABLES)[number]): number => + db.prepare<[], { n: number }>(`SELECT COUNT(*) AS n FROM ${table}`).get() + ?.n ?? 0; + return { + words: count("words"), + senses: count("senses"), + translations: count("translations"), + }; +}; diff --git a/data-pipeline/tests/promptTemplate.test.ts b/data-pipeline/tests/promptTemplate.test.ts new file mode 100644 index 0000000..59cb578 --- /dev/null +++ b/data-pipeline/tests/promptTemplate.test.ts @@ -0,0 +1,57 @@ +import { describe, it, expect } from "vitest"; +import { + loadPromptTemplate, + renderPrompt, + type PromptParams, +} from "../promptTemplate.js"; + +const params: PromptParams = { + sourceLanguage: "es", + pos: "noun", + targetLanguages: ["en", "it", "de", "fr"], + words: ["suelo", "pared"], +}; + +describe("renderPrompt", () => { + it("substitutes every placeholder from the checked-in template", () => { + const rendered = renderPrompt(loadPromptTemplate(), params); + expect(rendered).toContain('Source language: Spanish ("es")'); + expect(rendered).toContain('language must be "es"'); + expect(rendered).toContain("Target languages: en, it, de, fr"); + expect(rendered).toContain('"en" | "it" | "de" | "fr"'); + expect(rendered).toContain("suelo\npared"); + expect(rendered).toContain("valid Spanish noun"); + expect(rendered).not.toMatch(/\{\{[A-Z_]+\}\}/); + }); + + it("does not tell the model to exclude a non-source language as target", () => { + const rendered = renderPrompt(loadPromptTemplate(), params); + expect(rendered).toContain( + 'Do not include Spanish ("es") as a target_language.', + ); + expect(rendered).not.toContain( + "Do not include English as a target_language", + ); + }); + + it("throws when the source language is listed as a target", () => { + expect(() => + renderPrompt(loadPromptTemplate(), { + ...params, + targetLanguages: ["en", "es"], + }), + ).toThrow(/exclude the source language/); + }); + + it("throws on an empty word batch", () => { + expect(() => + renderPrompt(loadPromptTemplate(), { ...params, words: [] }), + ).toThrow(/empty word batch/); + }); + + it("throws on unreplaced placeholders", () => { + expect(() => renderPrompt("hello {{UNKNOWN_TOKEN}}", params)).toThrow( + /UNKNOWN_TOKEN/, + ); + }); +}); diff --git a/data-pipeline/tests/sourceLists.test.ts b/data-pipeline/tests/sourceLists.test.ts new file mode 100644 index 0000000..4d1675a --- /dev/null +++ b/data-pipeline/tests/sourceLists.test.ts @@ -0,0 +1,54 @@ +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { describe, it, expect, afterAll } from "vitest"; +import { discoverSourceLists, normalizeWords } from "../sourceLists.js"; + +describe("normalizeWords", () => { + it("trims whitespace and drops empty lines", () => { + expect(normalizeWords([" Haus ", "", " ", "Tür"])).toEqual([ + "Haus", + "Tür", + ]); + }); + + it("dedups while preserving first-occurrence order", () => { + expect(normalizeWords(["centro", "borde", "centro", "suelo"])).toEqual([ + "centro", + "borde", + "suelo", + ]); + }); +}); + +describe("discoverSourceLists", () => { + const root = mkdtempSync(path.join(tmpdir(), "lila-source-data-")); + afterAll(() => rmSync(root, { recursive: true, force: true })); + + it("finds only supported language/pos paths and normalizes their words", () => { + mkdirSync(path.join(root, "de")); + mkdirSync(path.join(root, "es")); + mkdirSync(path.join(root, "german")); // unsupported dir name — ignored + writeFileSync( + path.join(root, "de", "noun"), + "Haus\nTür\nHaus\n\n Tisch \n", + ); + writeFileSync(path.join(root, "es", "noun"), "casa\npared\n"); + writeFileSync(path.join(root, "es", "nouns"), "ignored\n"); // unsupported pos name + writeFileSync(path.join(root, "german", "noun"), "ignored\n"); + writeFileSync(path.join(root, "stray.txt"), "ignored\n"); + + const lists = discoverSourceLists(root); + expect(lists).toHaveLength(2); + expect(lists[0]).toMatchObject({ + sourceLanguage: "de", + pos: "noun", + words: ["Haus", "Tür", "Tisch"], + }); + expect(lists[1]).toMatchObject({ + sourceLanguage: "es", + pos: "noun", + words: ["casa", "pared"], + }); + }); +}); diff --git a/data-pipeline/tests/staging.test.ts b/data-pipeline/tests/staging.test.ts new file mode 100644 index 0000000..4d3c234 --- /dev/null +++ b/data-pipeline/tests/staging.test.ts @@ -0,0 +1,106 @@ +import path from "node:path"; +import { describe, it, expect, beforeEach } from "vitest"; +import type Database from "better-sqlite3"; +import { + countStagedRows, + getStagedHeadwords, + openStaging, + stageEntry, +} from "../staging.js"; +import type { GeminiWordEntry } from "../validate.js"; + +const SCHEMA_PATH = path.join(import.meta.dirname, "..", "db", "schema.sql"); + +const entry: GeminiWordEntry = { + headword: "casa", + language: "es", + pos: "noun", + senses: [ + { + sense_index: 0, + difficulty: "easy", + definitions: ["Un edificio para vivir."], + examples: ["Compraron una casa en la ciudad."], + translations: [ + { + target_language: "en", + word: "house", + gender: null, + difficulty: "easy", + }, + { + target_language: "it", + word: "casa", + gender: "feminine", + difficulty: "easy", + }, + { + target_language: "de", + word: "Haus", + gender: "neuter", + difficulty: "easy", + }, + { + target_language: "fr", + word: "maison", + gender: "feminine", + difficulty: "easy", + }, + ], + }, + ], +}; + +describe("staging", () => { + let db: Database.Database; + beforeEach(() => { + db = openStaging(":memory:", SCHEMA_PATH); + }); + + it("creates the schema from db/schema.sql on an empty database", () => { + expect(countStagedRows(db)).toEqual({ + words: 0, + senses: 0, + translations: 0, + }); + }); + + it("stages a word with its senses and translations atomically", () => { + expect(stageEntry(db, entry)).toBe("staged"); + expect(countStagedRows(db)).toEqual({ + words: 1, + senses: 1, + translations: 4, + }); + + const row = db + .prepare< + [], + { definitions: string; examples: string } + >(`SELECT definitions, examples FROM senses`) + .get(); + expect(JSON.parse(row?.definitions ?? "")).toEqual([ + "Un edificio para vivir.", + ]); + expect(JSON.parse(row?.examples ?? "")).toEqual([ + "Compraron una casa en la ciudad.", + ]); + }); + + it("is idempotent: staging the same word twice writes nothing new", () => { + expect(stageEntry(db, entry)).toBe("staged"); + expect(stageEntry(db, entry)).toBe("already-staged"); + expect(countStagedRows(db)).toEqual({ + words: 1, + senses: 1, + translations: 4, + }); + }); + + it("reports staged headwords per language and pos", () => { + stageEntry(db, entry); + expect(getStagedHeadwords(db, "es", "noun")).toEqual(new Set(["casa"])); + expect(getStagedHeadwords(db, "de", "noun")).toEqual(new Set()); + expect(getStagedHeadwords(db, "es", "verb")).toEqual(new Set()); + }); +}); diff --git a/data-pipeline/tests/validate.test.ts b/data-pipeline/tests/validate.test.ts new file mode 100644 index 0000000..53512d8 --- /dev/null +++ b/data-pipeline/tests/validate.test.ts @@ -0,0 +1,285 @@ +import { describe, it, expect } from "vitest"; +import { validateEntry, type ValidationContext } from "../validate.js"; + +const ctx: ValidationContext = { + sourceLanguage: "es", + pos: "noun", + targetLanguages: ["en", "it", "de", "fr"], + inputWords: new Set(["casa", "banco"]), +}; + +type Translation = { + target_language: string; + word: string; + gender: string | null; + difficulty: string; +}; + +const translations = (): Translation[] => [ + { target_language: "en", word: "house", gender: null, difficulty: "easy" }, + { + target_language: "it", + word: "casa", + gender: "feminine", + difficulty: "easy", + }, + { target_language: "de", word: "Haus", gender: "neuter", difficulty: "easy" }, + { + target_language: "fr", + word: "maison", + gender: "feminine", + difficulty: "easy", + }, +]; + +const sense = ( + overrides: Record = {}, +): Record => ({ + sense_index: 0, + difficulty: "easy", + definitions: ["Un edificio para vivir."], + examples: ["Compraron una casa en la ciudad."], + translations: translations(), + ...overrides, +}); + +const entry = ( + overrides: Record = {}, +): Record => ({ + headword: "casa", + language: "es", + pos: "noun", + senses: [sense()], + ...overrides, +}); + +const errorsOf = (raw: unknown): string[] => { + const result = validateEntry(raw, ctx); + return result.status === "invalid" ? result.errors : []; +}; + +describe("validateEntry", () => { + it("accepts a fully valid entry", () => { + const result = validateEntry(entry(), ctx); + expect(result.status).toBe("valid"); + }); + + it("treats senses: [] as empty (word skipped, not rejected)", () => { + const result = validateEntry(entry({ senses: [] }), ctx); + expect(result).toEqual({ status: "empty", headword: "casa" }); + }); + + it("rejects senses: [] when the rest of the entry is invalid", () => { + const result = validateEntry(entry({ senses: [], language: "en" }), ctx); + expect(result.status).toBe("invalid"); + }); + + it("rejects non-object input", () => { + expect(validateEntry("casa", ctx).status).toBe("invalid"); + expect(validateEntry(null, ctx).status).toBe("invalid"); + expect(validateEntry([entry()], ctx).status).toBe("invalid"); + }); + + it("rejects a headword that was not in the input batch", () => { + expect(errorsOf(entry({ headword: "perro" }))).toContainEqual( + expect.stringContaining("not in the input batch"), + ); + }); + + it("rejects a wrong source language", () => { + expect(errorsOf(entry({ language: "en" }))).toContainEqual( + expect.stringContaining('language must be "es"'), + ); + }); + + it("rejects a wrong pos", () => { + expect(errorsOf(entry({ pos: "verb" }))).toContainEqual( + expect.stringContaining('pos must be "noun"'), + ); + }); + + it("rejects more than 3 senses", () => { + const senses = [0, 1, 2, 3].map((i) => sense({ sense_index: i })); + expect(errorsOf(entry({ senses }))).toContainEqual( + expect.stringContaining("at most 3"), + ); + }); + + it("rejects non-sequential sense_index", () => { + const senses = [sense({ sense_index: 0 }), sense({ sense_index: 2 })]; + expect(errorsOf(entry({ senses }))).toContainEqual( + expect.stringContaining("sense_index must be 1"), + ); + }); + + it("rejects an unknown difficulty", () => { + expect( + errorsOf(entry({ senses: [sense({ difficulty: "intermediate" })] })), + ).toContainEqual(expect.stringContaining("difficulty must be one of")); + }); + + it("rejects empty definitions and examples", () => { + expect( + errorsOf(entry({ senses: [sense({ definitions: [] })] })), + ).toContainEqual( + expect.stringContaining("definitions must be a non-empty array"), + ); + expect( + errorsOf(entry({ senses: [sense({ examples: [""] })] })), + ).toContainEqual( + expect.stringContaining("examples must contain only non-empty strings"), + ); + }); + + it("rejects a non-null gender for English targets", () => { + const bad = translations(); + bad[0] = { + target_language: "en", + word: "house", + gender: "feminine", + difficulty: "easy", + }; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual( + expect.stringContaining('gender must be null for target "en"'), + ); + }); + + it("rejects a null gender for German targets", () => { + const bad = translations(); + bad[2] = { + target_language: "de", + word: "Haus", + gender: null, + difficulty: "easy", + }; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual(expect.stringContaining('for target "de"')); + }); + + it("rejects neuter for Romance-language targets", () => { + const bad = translations(); + bad[3] = { + target_language: "fr", + word: "maison", + gender: "neuter", + difficulty: "easy", + }; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual(expect.stringContaining('for target "fr"')); + }); + + it("rejects an invented gender value", () => { + const bad = translations(); + bad[2] = { + target_language: "de", + word: "Haus", + gender: "common", + difficulty: "easy", + }; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })).length, + ).toBeGreaterThan(0); + }); + + it("rejects a missing target language", () => { + const partial = translations().filter((t) => t.target_language !== "fr"); + expect( + errorsOf(entry({ senses: [sense({ translations: partial })] })), + ).toContainEqual( + expect.stringContaining('missing translation for target language "fr"'), + ); + }); + + it("rejects the source language as a target", () => { + const bad = [ + ...translations(), + { + target_language: "es", + word: "hogar", + gender: "masculine", + difficulty: "easy", + }, + ]; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual(expect.stringContaining("target_language must be one of")); + }); + + it("rejects more than 2 translations for one target language", () => { + const bad = [ + ...translations(), + { + target_language: "de", + word: "Gebäude", + gender: "neuter", + difficulty: "medium", + }, + { + target_language: "de", + word: "Heim", + gender: "neuter", + difficulty: "medium", + }, + ]; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual( + expect.stringContaining( + 'more than 2 translations for target language "de"', + ), + ); + }); + + it("rejects duplicate translation words for one target language", () => { + const bad = [ + ...translations(), + { + target_language: "de", + word: "Haus", + gender: "neuter", + difficulty: "medium", + }, + ]; + expect( + errorsOf(entry({ senses: [sense({ translations: bad })] })), + ).toContainEqual( + expect.stringContaining( + 'duplicate translation word for target language "de"', + ), + ); + }); + + it("rejects a translation difficulty below the sense difficulty", () => { + const easyTranslations = translations(); + expect( + errorsOf( + entry({ + senses: [ + sense({ difficulty: "medium", translations: easyTranslations }), + ], + }), + ), + ).toContainEqual( + expect.stringContaining('is lower than sense difficulty "medium"'), + ); + }); + + it("allows a translation difficulty above the sense difficulty", () => { + const harder = translations(); + harder[2] = { + target_language: "de", + word: "Geldinstitut", + gender: "neuter", + difficulty: "medium", + }; + const result = validateEntry( + entry({ senses: [sense({ translations: harder })] }), + ctx, + ); + expect(result.status).toBe("valid"); + }); +}); diff --git a/data-pipeline/validate.ts b/data-pipeline/validate.ts new file mode 100644 index 0000000..75d5f04 --- /dev/null +++ b/data-pipeline/validate.ts @@ -0,0 +1,239 @@ +import { + DIFFICULTY_LEVELS, + NOUN_GENDERS, + type DifficultyLevel, + type NounGender, + type SupportedLanguageCode, + type SupportedPos, +} from "@lila/shared"; + +export type GeminiTranslation = { + target_language: SupportedLanguageCode; + word: string; + gender: NounGender | null; + difficulty: DifficultyLevel; +}; + +export type GeminiSense = { + sense_index: number; + difficulty: DifficultyLevel; + definitions: string[]; + examples: string[]; + translations: GeminiTranslation[]; +}; + +export type GeminiWordEntry = { + headword: string; + language: SupportedLanguageCode; + pos: SupportedPos; + senses: GeminiSense[]; +}; + +export type ValidationContext = { + sourceLanguage: SupportedLanguageCode; + pos: SupportedPos; + targetLanguages: readonly SupportedLanguageCode[]; + inputWords: ReadonlySet; +}; + +/** + * "empty" is the contract's way of saying "not a valid word of this POS" + * (senses: []) — the word is skipped, not rejected. + */ +export type ValidationResult = + | { status: "valid"; entry: GeminiWordEntry } + | { status: "empty"; headword: string } + | { status: "invalid"; errors: string[] }; + +const isRecord = (value: unknown): value is Record => + typeof value === "object" && value !== null && !Array.isArray(value); + +const isNonEmptyString = (value: unknown): value is string => + typeof value === "string" && value.trim() !== ""; + +const isDifficulty = (value: unknown): value is DifficultyLevel => + (DIFFICULTY_LEVELS as readonly unknown[]).includes(value); + +const difficultyRank = (level: DifficultyLevel): number => + DIFFICULTY_LEVELS.indexOf(level); + +const validGendersFor = ( + target: SupportedLanguageCode, +): readonly (NounGender | null)[] => { + if (target === "en") return [null]; + if (target === "de") return NOUN_GENDERS; + return ["masculine", "feminine"]; +}; + +const checkStringArray = ( + value: unknown, + label: string, + errors: string[], +): void => { + if (!Array.isArray(value) || value.length === 0) { + errors.push(`${label} must be a non-empty array`); + return; + } + if (!value.every(isNonEmptyString)) { + errors.push(`${label} must contain only non-empty strings`); + } +}; + +const checkTranslation = ( + raw: unknown, + label: string, + senseDifficulty: DifficultyLevel | null, + ctx: ValidationContext, + errors: string[], +): void => { + if (!isRecord(raw)) { + errors.push(`${label} must be an object`); + return; + } + const target = raw["target_language"]; + if (!(ctx.targetLanguages as readonly unknown[]).includes(target)) { + errors.push( + `${label}: target_language must be one of ${ctx.targetLanguages.join(", ")}`, + ); + } + if (!isNonEmptyString(raw["word"])) { + errors.push(`${label}: word must be a non-empty string`); + } + const difficulty = raw["difficulty"]; + if (!isDifficulty(difficulty)) { + errors.push( + `${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`, + ); + } else if ( + senseDifficulty !== null && + difficultyRank(difficulty) < difficultyRank(senseDifficulty) + ) { + errors.push( + `${label}: difficulty "${difficulty}" is lower than sense difficulty "${senseDifficulty}"`, + ); + } + if ( + typeof target === "string" && + (ctx.targetLanguages as readonly string[]).includes(target) + ) { + const gender = raw["gender"]; + const allowed = validGendersFor(target as SupportedLanguageCode); + if (!(allowed as readonly unknown[]).includes(gender)) { + errors.push( + `${label}: gender must be ${allowed.map((g) => g ?? "null").join(" or ")} for target "${target}"`, + ); + } + } +}; + +const checkSense = ( + raw: unknown, + index: number, + ctx: ValidationContext, + errors: string[], +): void => { + const label = `senses[${index}]`; + if (!isRecord(raw)) { + errors.push(`${label} must be an object`); + return; + } + if (raw["sense_index"] !== index) { + errors.push(`${label}: sense_index must be ${index} (sequential from 0)`); + } + const senseDifficulty = isDifficulty(raw["difficulty"]) + ? raw["difficulty"] + : null; + if (senseDifficulty === null) { + errors.push( + `${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`, + ); + } + checkStringArray(raw["definitions"], `${label}.definitions`, errors); + checkStringArray(raw["examples"], `${label}.examples`, errors); + + const translations = raw["translations"]; + if (!Array.isArray(translations) || translations.length === 0) { + errors.push(`${label}.translations must be a non-empty array`); + return; + } + translations.forEach((translation, i) => { + checkTranslation( + translation, + `${label}.translations[${i}]`, + senseDifficulty, + ctx, + errors, + ); + }); + + const wordsPerTarget = new Map(); + for (const translation of translations) { + if (!isRecord(translation)) continue; + const target = translation["target_language"]; + const word = translation["word"]; + if (typeof target !== "string" || typeof word !== "string") continue; + const words = wordsPerTarget.get(target) ?? []; + words.push(word); + wordsPerTarget.set(target, words); + } + for (const target of ctx.targetLanguages) { + const words = wordsPerTarget.get(target) ?? []; + if (words.length === 0) { + errors.push( + `${label}: missing translation for target language "${target}"`, + ); + } + if (words.length > 2) { + errors.push( + `${label}: more than 2 translations for target language "${target}"`, + ); + } + if (new Set(words).size !== words.length) { + errors.push( + `${label}: duplicate translation word for target language "${target}"`, + ); + } + } +}; + +export const validateEntry = ( + raw: unknown, + ctx: ValidationContext, +): ValidationResult => { + const errors: string[] = []; + if (!isRecord(raw)) { + return { status: "invalid", errors: ["entry must be an object"] }; + } + + const headword = raw["headword"]; + if (!isNonEmptyString(headword)) { + errors.push("headword must be a non-empty string"); + } else if (!ctx.inputWords.has(headword)) { + errors.push(`headword "${headword}" is not in the input batch`); + } + if (raw["language"] !== ctx.sourceLanguage) { + errors.push(`language must be "${ctx.sourceLanguage}"`); + } + if (raw["pos"] !== ctx.pos) { + errors.push(`pos must be "${ctx.pos}"`); + } + + const senses = raw["senses"]; + if (!Array.isArray(senses)) { + errors.push("senses must be an array"); + } else if (senses.length === 0) { + if (errors.length === 0 && isNonEmptyString(headword)) { + return { status: "empty", headword }; + } + } else { + if (senses.length > 3) { + errors.push("senses must contain at most 3 entries"); + } + senses.forEach((sense, i) => checkSense(sense, i, ctx, errors)); + } + + if (errors.length > 0) { + return { status: "invalid", errors }; + } + return { status: "valid", entry: raw as GeminiWordEntry }; +};