implementing phase 3 pipeline: gemini structured output, validation, sqlite staging
Prompt is now a template (fixes the hardcoded en/es leftovers in rules 2, 3, 15, 16, 26, 31). pipeline.ts replaces the pseudocode: wordlist normalization, skip-already-staged idempotency, batches of 20 against gemini-3.6-flash with responseSchema, raw responses persisted per batch, per-entry validation with rejection log, one transaction per word into db/staging.db. Flags: --langs --pos --max-batches --delay-ms --dry-run. Smoke run: 40/40 words staged (de+es, one batch each), 0 rejections. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
303bb9388c
commit
37c978e230
12 changed files with 1477 additions and 66 deletions
171
data-pipeline/gemini.ts
Normal file
171
data-pipeline/gemini.ts
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
import {
|
||||
DIFFICULTY_LEVELS,
|
||||
NOUN_GENDERS,
|
||||
type SupportedLanguageCode,
|
||||
type SupportedPos,
|
||||
} from "@lila/shared";
|
||||
|
||||
export const DEFAULT_MODEL = "gemini-3.6-flash";
|
||||
|
||||
const API_BASE = "https://generativelanguage.googleapis.com/v1beta/models";
|
||||
const MAX_ATTEMPTS = 5;
|
||||
const RETRYABLE_STATUS = new Set([429, 500, 503]);
|
||||
|
||||
type ResponseSchema = Record<string, unknown>;
|
||||
|
||||
/**
|
||||
* OpenAPI-subset schema for Gemini structured output: an array of word
|
||||
* entries matching design-doc §6.3, with enums narrowed to this batch's
|
||||
* source/POS/target languages.
|
||||
*/
|
||||
export const buildEntriesResponseSchema = (
|
||||
sourceLanguage: SupportedLanguageCode,
|
||||
pos: SupportedPos,
|
||||
targetLanguages: readonly SupportedLanguageCode[],
|
||||
): ResponseSchema => ({
|
||||
type: "ARRAY",
|
||||
items: {
|
||||
type: "OBJECT",
|
||||
properties: {
|
||||
headword: { type: "STRING" },
|
||||
language: { type: "STRING", enum: [sourceLanguage] },
|
||||
pos: { type: "STRING", enum: [pos] },
|
||||
senses: {
|
||||
type: "ARRAY",
|
||||
items: {
|
||||
type: "OBJECT",
|
||||
properties: {
|
||||
sense_index: { type: "INTEGER" },
|
||||
difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] },
|
||||
definitions: { type: "ARRAY", items: { type: "STRING" } },
|
||||
examples: { type: "ARRAY", items: { type: "STRING" } },
|
||||
translations: {
|
||||
type: "ARRAY",
|
||||
items: {
|
||||
type: "OBJECT",
|
||||
properties: {
|
||||
target_language: {
|
||||
type: "STRING",
|
||||
enum: [...targetLanguages],
|
||||
},
|
||||
word: { type: "STRING" },
|
||||
gender: {
|
||||
type: "STRING",
|
||||
enum: [...NOUN_GENDERS],
|
||||
nullable: true,
|
||||
},
|
||||
difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] },
|
||||
},
|
||||
required: ["target_language", "word", "gender", "difficulty"],
|
||||
},
|
||||
},
|
||||
},
|
||||
required: [
|
||||
"sense_index",
|
||||
"difficulty",
|
||||
"definitions",
|
||||
"examples",
|
||||
"translations",
|
||||
],
|
||||
},
|
||||
},
|
||||
},
|
||||
required: ["headword", "language", "pos", "senses"],
|
||||
},
|
||||
});
|
||||
|
||||
const sleep = (ms: number): Promise<void> =>
|
||||
new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
const retryDelayMs = (body: string, attempt: number): number => {
|
||||
const match = body.match(/"retryDelay":\s*"(\d+(?:\.\d+)?)s"/);
|
||||
if (match?.[1] !== undefined) {
|
||||
return Math.ceil(Number(match[1]) * 1000) + 500;
|
||||
}
|
||||
return 2 ** attempt * 2000;
|
||||
};
|
||||
|
||||
const extractText = (body: unknown): string => {
|
||||
if (isRecord(body)) {
|
||||
const candidates = body["candidates"];
|
||||
if (Array.isArray(candidates) && isRecord(candidates[0])) {
|
||||
const candidate = candidates[0];
|
||||
const finishReason = candidate["finishReason"];
|
||||
if (finishReason !== undefined && finishReason !== "STOP") {
|
||||
throw new Error(
|
||||
`Gemini stopped early: finishReason=${JSON.stringify(finishReason)}`,
|
||||
);
|
||||
}
|
||||
const content = candidate["content"];
|
||||
if (isRecord(content)) {
|
||||
const parts = content["parts"];
|
||||
if (Array.isArray(parts) && isRecord(parts[0])) {
|
||||
const text = parts[0]["text"];
|
||||
if (typeof text === "string") return text;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new Error("Gemini response contained no text candidate");
|
||||
};
|
||||
|
||||
const isRecord = (value: unknown): value is Record<string, unknown> =>
|
||||
typeof value === "object" && value !== null && !Array.isArray(value);
|
||||
|
||||
export const generateContent = async (
|
||||
apiKey: string,
|
||||
model: string,
|
||||
prompt: string,
|
||||
responseSchema: ResponseSchema,
|
||||
): Promise<string> => {
|
||||
let lastError = "";
|
||||
for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) {
|
||||
let response: Response;
|
||||
try {
|
||||
response = await fetch(`${API_BASE}/${model}:generateContent`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"x-goog-api-key": apiKey,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
contents: [{ role: "user", parts: [{ text: prompt }] }],
|
||||
generationConfig: {
|
||||
responseMimeType: "application/json",
|
||||
responseSchema,
|
||||
temperature: 0.2,
|
||||
},
|
||||
}),
|
||||
});
|
||||
} catch (error) {
|
||||
lastError = `network error: ${error instanceof Error ? error.message : String(error)}`;
|
||||
await sleep(2 ** attempt * 2000);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (response.ok) {
|
||||
return extractText(await response.json());
|
||||
}
|
||||
|
||||
const body = await response.text();
|
||||
lastError = `HTTP ${response.status}: ${body.slice(0, 500)}`;
|
||||
if (!RETRYABLE_STATUS.has(response.status)) {
|
||||
break;
|
||||
}
|
||||
await sleep(retryDelayMs(body, attempt));
|
||||
}
|
||||
throw new Error(`Gemini request failed after retries — ${lastError}`);
|
||||
};
|
||||
|
||||
/** Parse the model's JSON text into an array of unknown entries. */
|
||||
export const parseEntries = (rawText: string): unknown[] => {
|
||||
const stripped = rawText
|
||||
.trim()
|
||||
.replace(/^```(?:json)?\s*/i, "")
|
||||
.replace(/\s*```$/, "");
|
||||
const parsed: unknown = JSON.parse(stripped);
|
||||
if (!Array.isArray(parsed)) {
|
||||
throw new Error("Gemini response is not a JSON array");
|
||||
}
|
||||
return parsed;
|
||||
};
|
||||
|
|
@ -1,45 +1,327 @@
|
|||
// pipeline.ts pseudo code
|
||||
import { appendFileSync, mkdirSync, writeFileSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { parseArgs } from "node:util";
|
||||
import {
|
||||
SUPPORTED_LANGUAGE_CODES,
|
||||
SUPPORTED_POS,
|
||||
type SupportedLanguageCode,
|
||||
type SupportedPos,
|
||||
} from "@lila/shared";
|
||||
import {
|
||||
DEFAULT_MODEL,
|
||||
buildEntriesResponseSchema,
|
||||
generateContent,
|
||||
parseEntries,
|
||||
} from "./gemini.js";
|
||||
import { loadPromptTemplate, renderPrompt } from "./promptTemplate.js";
|
||||
import { discoverSourceLists, type SourceList } from "./sourceLists.js";
|
||||
import {
|
||||
countStagedRows,
|
||||
getStagedHeadwords,
|
||||
openStaging,
|
||||
stageEntry,
|
||||
} from "./staging.js";
|
||||
import { validateEntry, type ValidationContext } from "./validate.js";
|
||||
|
||||
/*
|
||||
step 1: discover source lists
|
||||
const ROOT = import.meta.dirname;
|
||||
const SOURCE_DATA_DIR = path.join(ROOT, "source-data");
|
||||
const STAGING_DB_PATH = path.join(ROOT, "db", "staging.db");
|
||||
const STAGING_SCHEMA_PATH = path.join(ROOT, "db", "schema.sql");
|
||||
const RESPONSES_DIR = path.join(ROOT, "responses");
|
||||
const REJECTIONS_DIR = path.join(ROOT, "rejections");
|
||||
|
||||
this will give us an array of objects with this schema
|
||||
sourceLanguage: 'en' | 'de' | 'es' | 'fr' | 'it'
|
||||
pos: 'noun' | 'verb' | 'adjective' | 'adverb'
|
||||
words: string[];
|
||||
filePath: string;
|
||||
type CliOptions = {
|
||||
langs: SupportedLanguageCode[] | null;
|
||||
pos: SupportedPos;
|
||||
maxBatches: number | null;
|
||||
batchSize: number;
|
||||
delayMs: number;
|
||||
dryRun: boolean;
|
||||
};
|
||||
|
||||
the terminal output should be something like:
|
||||
type ListStats = {
|
||||
list: SourceList;
|
||||
pending: number;
|
||||
batchesRun: number;
|
||||
staged: number;
|
||||
skipped: number;
|
||||
rejected: number;
|
||||
failedBatches: number;
|
||||
};
|
||||
|
||||
found 5 source lists:
|
||||
const parseCli = (): CliOptions => {
|
||||
// pnpm forwards the "--" separator itself (pnpm pipeline:run -- --langs …);
|
||||
// drop it so the flags after it are parsed as flags, not positionals.
|
||||
const args = process.argv.slice(2);
|
||||
if (args[0] === "--") args.shift();
|
||||
const { values } = parseArgs({
|
||||
args,
|
||||
options: {
|
||||
langs: { type: "string" },
|
||||
pos: { type: "string", default: "noun" },
|
||||
"max-batches": { type: "string" },
|
||||
"batch-size": { type: "string", default: "20" },
|
||||
"delay-ms": { type: "string", default: "6000" },
|
||||
"dry-run": { type: "boolean", default: false },
|
||||
},
|
||||
});
|
||||
|
||||
de: noun
|
||||
en: noun
|
||||
const pos = values.pos as SupportedPos;
|
||||
if (!(SUPPORTED_POS as readonly string[]).includes(pos)) {
|
||||
throw new Error(`--pos must be one of ${SUPPORTED_POS.join(", ")}`);
|
||||
}
|
||||
let langs: SupportedLanguageCode[] | null = null;
|
||||
if (values.langs !== undefined) {
|
||||
langs = values.langs.split(",").map((code) => {
|
||||
const trimmed = code.trim() as SupportedLanguageCode;
|
||||
if (!(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(trimmed)) {
|
||||
throw new Error(`--langs: unknown language code "${trimmed}"`);
|
||||
}
|
||||
return trimmed;
|
||||
});
|
||||
}
|
||||
return {
|
||||
langs,
|
||||
pos,
|
||||
maxBatches:
|
||||
values["max-batches"] !== undefined
|
||||
? Number(values["max-batches"])
|
||||
: null,
|
||||
batchSize: Number(values["batch-size"]),
|
||||
delayMs: Number(values["delay-ms"]),
|
||||
dryRun: values["dry-run"],
|
||||
};
|
||||
};
|
||||
|
||||
and so on
|
||||
const chunk = <T>(items: readonly T[], size: number): T[][] => {
|
||||
const chunks: T[][] = [];
|
||||
for (let i = 0; i < items.length; i += size) {
|
||||
chunks.push(items.slice(i, i + size));
|
||||
}
|
||||
return chunks;
|
||||
};
|
||||
|
||||
later on, it will also contain it: noun, verb, adjective etc
|
||||
*/
|
||||
const sleep = (ms: number): Promise<void> =>
|
||||
new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
/*
|
||||
const rejectionFile = (list: SourceList): string =>
|
||||
path.join(REJECTIONS_DIR, `${list.sourceLanguage}-${list.pos}.jsonl`);
|
||||
|
||||
step 2: validating source lists
|
||||
const logRejection = (
|
||||
list: SourceList,
|
||||
headword: string | null,
|
||||
errors: string[],
|
||||
entry: unknown,
|
||||
): void => {
|
||||
mkdirSync(REJECTIONS_DIR, { recursive: true });
|
||||
const line = JSON.stringify({
|
||||
at: new Date().toISOString(),
|
||||
sourceLanguage: list.sourceLanguage,
|
||||
pos: list.pos,
|
||||
headword,
|
||||
errors,
|
||||
entry,
|
||||
});
|
||||
appendFileSync(rejectionFile(list), `${line}\n`);
|
||||
};
|
||||
|
||||
a small script that trims whitespaces, removes duplicated words etc
|
||||
const saveRawResponse = (
|
||||
list: SourceList,
|
||||
batchIndex: number,
|
||||
model: string,
|
||||
words: readonly string[],
|
||||
targetLanguages: readonly SupportedLanguageCode[],
|
||||
rawText: string,
|
||||
): void => {
|
||||
mkdirSync(RESPONSES_DIR, { recursive: true });
|
||||
const stamp = new Date().toISOString().replaceAll(":", "-");
|
||||
const file = path.join(
|
||||
RESPONSES_DIR,
|
||||
`${list.sourceLanguage}-${list.pos}-${stamp}-batch${batchIndex}.json`,
|
||||
);
|
||||
writeFileSync(
|
||||
file,
|
||||
JSON.stringify(
|
||||
{
|
||||
model,
|
||||
sourceLanguage: list.sourceLanguage,
|
||||
pos: list.pos,
|
||||
targetLanguages,
|
||||
words,
|
||||
receivedAt: new Date().toISOString(),
|
||||
rawText,
|
||||
},
|
||||
null,
|
||||
2,
|
||||
),
|
||||
);
|
||||
};
|
||||
|
||||
terminal output: summary of how many words per pos per language were found
|
||||
const headwordOf = (entry: unknown): string | null => {
|
||||
if (typeof entry === "object" && entry !== null && !Array.isArray(entry)) {
|
||||
const headword = (entry as Record<string, unknown>)["headword"];
|
||||
if (typeof headword === "string") return headword;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
*/
|
||||
const main = async (): Promise<void> => {
|
||||
const options = parseCli();
|
||||
const model = process.env["GEMINI_MODEL"] ?? DEFAULT_MODEL;
|
||||
const apiKey = process.env["GEMINI_API_KEY"];
|
||||
if (apiKey === undefined && !options.dryRun) {
|
||||
throw new Error("GEMINI_API_KEY is not set (data-pipeline/.env)");
|
||||
}
|
||||
|
||||
/*
|
||||
const allLists = discoverSourceLists(SOURCE_DATA_DIR);
|
||||
console.log(`found ${allLists.length} source lists:\n`);
|
||||
for (const list of allLists) {
|
||||
console.log(
|
||||
` ${list.sourceLanguage}: ${list.pos} (${list.words.length} unique words)`,
|
||||
);
|
||||
}
|
||||
|
||||
step 3: writing to database?
|
||||
const lists = allLists.filter(
|
||||
(list) =>
|
||||
list.pos === options.pos &&
|
||||
(options.langs === null || options.langs.includes(list.sourceLanguage)),
|
||||
);
|
||||
if (lists.length === 0) {
|
||||
console.log("\nnothing matches the requested --langs/--pos, exiting");
|
||||
return;
|
||||
}
|
||||
|
||||
my thought: ill restart the pipeline several times during testing, and when adding more wordlists with other pos or extending the exisiting noun lists
|
||||
eventually the lists will contain tens or hundreds of thousands of words
|
||||
how do we prevent reading and processing the same words multiple times?
|
||||
if we read and validate+normalize the wordlists and write them to the database, we could then read from the database fill the missing translations etc
|
||||
and not read the same words from the same text files multiple times?
|
||||
const template = loadPromptTemplate();
|
||||
const db = openStaging(STAGING_DB_PATH, STAGING_SCHEMA_PATH);
|
||||
const allStats: ListStats[] = [];
|
||||
let firstApiCall = true;
|
||||
|
||||
if we do this, we have to adjust the database schema because there are several notNull() rows inside
|
||||
*/
|
||||
for (const list of lists) {
|
||||
const staged = getStagedHeadwords(db, list.sourceLanguage, list.pos);
|
||||
const pending = list.words.filter((word) => !staged.has(word));
|
||||
const targetLanguages = SUPPORTED_LANGUAGE_CODES.filter(
|
||||
(code) => code !== list.sourceLanguage,
|
||||
);
|
||||
const batches = chunk(pending, options.batchSize).slice(
|
||||
0,
|
||||
options.maxBatches ?? Number.POSITIVE_INFINITY,
|
||||
);
|
||||
|
||||
console.log(
|
||||
`\n${list.sourceLanguage}/${list.pos}: ${list.words.length} unique, ${staged.size} already staged, ${pending.length} pending → running ${batches.length} batch(es)`,
|
||||
);
|
||||
const stats: ListStats = {
|
||||
list,
|
||||
pending: pending.length,
|
||||
batchesRun: 0,
|
||||
staged: 0,
|
||||
skipped: 0,
|
||||
rejected: 0,
|
||||
failedBatches: 0,
|
||||
};
|
||||
allStats.push(stats);
|
||||
|
||||
for (const [batchIndex, words] of batches.entries()) {
|
||||
const prompt = renderPrompt(template, {
|
||||
sourceLanguage: list.sourceLanguage,
|
||||
pos: list.pos,
|
||||
targetLanguages,
|
||||
words,
|
||||
});
|
||||
|
||||
if (options.dryRun) {
|
||||
console.log(
|
||||
` [dry-run] batch ${batchIndex + 1}/${batches.length}: ${words.join(", ")}`,
|
||||
);
|
||||
stats.batchesRun++;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!firstApiCall) {
|
||||
await sleep(options.delayMs);
|
||||
}
|
||||
firstApiCall = false;
|
||||
|
||||
try {
|
||||
const rawText = await generateContent(
|
||||
apiKey as string,
|
||||
model,
|
||||
prompt,
|
||||
buildEntriesResponseSchema(
|
||||
list.sourceLanguage,
|
||||
list.pos,
|
||||
targetLanguages,
|
||||
),
|
||||
);
|
||||
saveRawResponse(
|
||||
list,
|
||||
batchIndex + 1,
|
||||
model,
|
||||
words,
|
||||
targetLanguages,
|
||||
rawText,
|
||||
);
|
||||
|
||||
const entries = parseEntries(rawText);
|
||||
const ctx: ValidationContext = {
|
||||
sourceLanguage: list.sourceLanguage,
|
||||
pos: list.pos,
|
||||
targetLanguages,
|
||||
inputWords: new Set(words),
|
||||
};
|
||||
const covered = new Set<string>();
|
||||
for (const entry of entries) {
|
||||
const result = validateEntry(entry, ctx);
|
||||
if (result.status === "valid") {
|
||||
stageEntry(db, result.entry);
|
||||
covered.add(result.entry.headword);
|
||||
stats.staged++;
|
||||
} else if (result.status === "empty") {
|
||||
covered.add(result.headword);
|
||||
stats.skipped++;
|
||||
console.log(
|
||||
` skipped "${result.headword}" (no valid ${list.pos} senses)`,
|
||||
);
|
||||
} else {
|
||||
const headword = headwordOf(entry);
|
||||
if (headword !== null) covered.add(headword);
|
||||
logRejection(list, headword, result.errors, entry);
|
||||
stats.rejected++;
|
||||
}
|
||||
}
|
||||
for (const word of words) {
|
||||
if (!covered.has(word)) {
|
||||
logRejection(list, word, ["missing from Gemini response"], null);
|
||||
stats.rejected++;
|
||||
}
|
||||
}
|
||||
stats.batchesRun++;
|
||||
console.log(
|
||||
` batch ${batchIndex + 1}/${batches.length} done — ${stats.staged} staged, ${stats.rejected} rejected, ${stats.skipped} skipped`,
|
||||
);
|
||||
} catch (error) {
|
||||
stats.failedBatches++;
|
||||
console.error(
|
||||
` batch ${batchIndex + 1}/${batches.length} FAILED: ${error instanceof Error ? error.message : String(error)}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log("\n— summary —");
|
||||
for (const stats of allStats) {
|
||||
console.log(
|
||||
`${stats.list.sourceLanguage}/${stats.list.pos}: ${stats.staged} staged, ${stats.skipped} skipped, ${stats.rejected} rejected, ${stats.failedBatches} failed batch(es), ${stats.pending - stats.staged - stats.skipped - stats.rejected} still pending`,
|
||||
);
|
||||
}
|
||||
const totals = countStagedRows(db);
|
||||
console.log(
|
||||
`staging.db totals: ${totals.words} words, ${totals.senses} senses, ${totals.translations} translations`,
|
||||
);
|
||||
db.close();
|
||||
};
|
||||
|
||||
main().catch((error: unknown) => {
|
||||
console.error(error instanceof Error ? error.message : error);
|
||||
process.exitCode = 1;
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,31 +1,11 @@
|
|||
You are a multilingual lexicographer generating vocabulary data for a language-learning app.
|
||||
|
||||
Source language: spanish
|
||||
Part of speech: noun
|
||||
Target languages: en, it, de, fr
|
||||
Source language: {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}")
|
||||
Part of speech: {{POS}}
|
||||
Target languages: {{TARGET_LANGUAGE_CODES}}
|
||||
|
||||
Input words:
|
||||
suelo
|
||||
pared
|
||||
techo
|
||||
portón
|
||||
valla
|
||||
esquina
|
||||
centro
|
||||
borde
|
||||
superficie
|
||||
centro
|
||||
suburbio
|
||||
medioambiente
|
||||
oportunidad
|
||||
ventaja
|
||||
decisión
|
||||
paciencia
|
||||
comportamiento
|
||||
industria
|
||||
conocimiento
|
||||
solución
|
||||
|
||||
{{INPUT_WORDS}}
|
||||
|
||||
Return ONLY valid JSON.
|
||||
Do not include markdown fences.
|
||||
|
|
@ -39,8 +19,8 @@ Each word object must have this shape:
|
|||
|
||||
{
|
||||
"headword": string,
|
||||
"language": "es",
|
||||
"pos": "noun",
|
||||
"language": "{{SOURCE_LANGUAGE_CODE}}",
|
||||
"pos": "{{POS}}",
|
||||
"senses": [
|
||||
{
|
||||
"sense_index": number,
|
||||
|
|
@ -49,7 +29,7 @@ Each word object must have this shape:
|
|||
"examples": string[],
|
||||
"translations": [
|
||||
{
|
||||
"target_language": "de" | "it" | "en" | "fr",
|
||||
"target_language": {{TARGET_LANGUAGE_UNION}},
|
||||
"word": string,
|
||||
"gender": "masculine" | "feminine" | "neuter" | null,
|
||||
"difficulty": "easy" | "medium" | "hard"
|
||||
|
|
@ -62,36 +42,36 @@ Each word object must have this shape:
|
|||
Rules:
|
||||
|
||||
1. The headword must be exactly one of the input words.
|
||||
2. language must be "en".
|
||||
3. pos must be "noun".
|
||||
2. language must be "{{SOURCE_LANGUAGE_CODE}}".
|
||||
3. pos must be "{{POS}}".
|
||||
4. Include only common, learner-relevant senses.
|
||||
5. Most words should have 1 sense.
|
||||
6. Polysemous words may have 2 or 3 senses.
|
||||
7. Do not include rare, archaic, highly technical, or literary senses unless they are common.
|
||||
8. sense_index must start at 0 and increase by 1.
|
||||
9. definitions must be written in English.
|
||||
10. examples must be written in English.
|
||||
9. definitions must be written in {{SOURCE_LANGUAGE_NAME}}.
|
||||
10. examples must be written in {{SOURCE_LANGUAGE_NAME}}.
|
||||
11. Include 1 or 2 definitions per sense.
|
||||
12. Include 1 or 2 example sentences per sense.
|
||||
13. Each definition must be student-friendly and at most 15 words.
|
||||
14. Each example should naturally contain the headword or a clear form of it.
|
||||
15. Every sense must have translations for all target languages: de, it, es, fr.
|
||||
16. Do not include English as a target_language.
|
||||
15. Every sense must have translations for all target languages: {{TARGET_LANGUAGE_CODES}}.
|
||||
16. Do not include {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}") as a target_language.
|
||||
17. You may include up to 2 translations per target language per sense if they are genuinely common synonyms or difficulty variants.
|
||||
18. Do not include more than 2 translations per target language per sense.
|
||||
19. Do not duplicate the same translation word for the same target language within one sense.
|
||||
20. Use the base dictionary form of the translated noun.
|
||||
20. Use the base dictionary form of the translated {{POS}}.
|
||||
21. Do not include articles or determiners in translations.
|
||||
22. German translation nouns must be capitalized.
|
||||
23. Spanish, French, and Italian translation nouns should be lowercase unless they are proper nouns.
|
||||
24. For target_language "de", gender must be "masculine", "feminine", or "neuter".
|
||||
25. For target_language "it", "es", or "fr", gender must be "masculine" or "feminine".
|
||||
26. For this prompt, gender must never be null.
|
||||
26. gender must be null if and only if target_language is "en".
|
||||
27. difficulty must be one of: "easy", "medium", "hard".
|
||||
28. senses.difficulty describes how common or advanced the meaning is.
|
||||
29. translations.difficulty describes how difficult the specific target-language word is for a learner.
|
||||
30. A common translation like "Bank" may be easy, while a formal synonym like "Geldinstitut" may be medium.
|
||||
31. If a word cannot be treated as a valid English noun, return it with "senses": [].
|
||||
31. If a word cannot be treated as a valid {{SOURCE_LANGUAGE_NAME}} {{POS}}, return it with "senses": [].
|
||||
|
||||
Difficulty calibration:
|
||||
|
||||
|
|
@ -101,7 +81,7 @@ Difficulty calibration:
|
|||
|
||||
Do not include CEFR levels in the output.
|
||||
|
||||
Example output shape for the English noun "bank":
|
||||
Example output shape for the English noun "bank" (illustrative of the JSON shape only — your definitions and examples must be in {{SOURCE_LANGUAGE_NAME}}):
|
||||
|
||||
[
|
||||
{
|
||||
|
|
|
|||
56
data-pipeline/promptTemplate.ts
Normal file
56
data-pipeline/promptTemplate.ts
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
import { readFileSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import type { SupportedLanguageCode, SupportedPos } from "@lila/shared";
|
||||
|
||||
export const LANGUAGE_NAMES: Record<SupportedLanguageCode, string> = {
|
||||
en: "English",
|
||||
it: "Italian",
|
||||
de: "German",
|
||||
fr: "French",
|
||||
es: "Spanish",
|
||||
};
|
||||
|
||||
export type PromptParams = {
|
||||
sourceLanguage: SupportedLanguageCode;
|
||||
pos: SupportedPos;
|
||||
targetLanguages: readonly SupportedLanguageCode[];
|
||||
words: readonly string[];
|
||||
};
|
||||
|
||||
export const loadPromptTemplate = (): string =>
|
||||
readFileSync(path.join(import.meta.dirname, "prompt"), "utf-8");
|
||||
|
||||
export const renderPrompt = (
|
||||
template: string,
|
||||
params: PromptParams,
|
||||
): string => {
|
||||
const { sourceLanguage, pos, targetLanguages, words } = params;
|
||||
if (words.length === 0) {
|
||||
throw new Error("renderPrompt: empty word batch");
|
||||
}
|
||||
if (
|
||||
targetLanguages.length === 0 ||
|
||||
targetLanguages.includes(sourceLanguage)
|
||||
) {
|
||||
throw new Error(
|
||||
`renderPrompt: target languages must be non-empty and exclude the source language (got ${targetLanguages.join(", ")})`,
|
||||
);
|
||||
}
|
||||
|
||||
const rendered = template
|
||||
.replaceAll("{{SOURCE_LANGUAGE_NAME}}", LANGUAGE_NAMES[sourceLanguage])
|
||||
.replaceAll("{{SOURCE_LANGUAGE_CODE}}", sourceLanguage)
|
||||
.replaceAll("{{POS}}", pos)
|
||||
.replaceAll("{{TARGET_LANGUAGE_CODES}}", targetLanguages.join(", "))
|
||||
.replaceAll(
|
||||
"{{TARGET_LANGUAGE_UNION}}",
|
||||
targetLanguages.map((code) => `"${code}"`).join(" | "),
|
||||
)
|
||||
.replaceAll("{{INPUT_WORDS}}", words.join("\n"));
|
||||
|
||||
const leftover = rendered.match(/\{\{[A-Z_]+\}\}/);
|
||||
if (leftover) {
|
||||
throw new Error(`renderPrompt: unreplaced placeholder ${leftover[0]}`);
|
||||
}
|
||||
return rendered;
|
||||
};
|
||||
56
data-pipeline/sourceLists.ts
Normal file
56
data-pipeline/sourceLists.ts
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
import { readdirSync, readFileSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import {
|
||||
SUPPORTED_LANGUAGE_CODES,
|
||||
SUPPORTED_POS,
|
||||
type SupportedLanguageCode,
|
||||
type SupportedPos,
|
||||
} from "@lila/shared";
|
||||
|
||||
export type SourceList = {
|
||||
sourceLanguage: SupportedLanguageCode;
|
||||
pos: SupportedPos;
|
||||
words: string[];
|
||||
filePath: string;
|
||||
};
|
||||
|
||||
const isLanguageCode = (value: string): value is SupportedLanguageCode =>
|
||||
(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(value);
|
||||
|
||||
const isPos = (value: string): value is SupportedPos =>
|
||||
(SUPPORTED_POS as readonly string[]).includes(value);
|
||||
|
||||
export const normalizeWords = (lines: readonly string[]): string[] => {
|
||||
const seen = new Set<string>();
|
||||
const words: string[] = [];
|
||||
for (const line of lines) {
|
||||
const word = line.trim();
|
||||
if (word === "" || seen.has(word)) continue;
|
||||
seen.add(word);
|
||||
words.push(word);
|
||||
}
|
||||
return words;
|
||||
};
|
||||
|
||||
export const discoverSourceLists = (rootDir: string): SourceList[] => {
|
||||
const lists: SourceList[] = [];
|
||||
const languageDirs = readdirSync(rootDir, { withFileTypes: true })
|
||||
.filter((entry) => entry.isDirectory() && isLanguageCode(entry.name))
|
||||
.map((entry) => entry.name as SupportedLanguageCode)
|
||||
.sort();
|
||||
|
||||
for (const language of languageDirs) {
|
||||
const languageDir = path.join(rootDir, language);
|
||||
const posFiles = readdirSync(languageDir, { withFileTypes: true })
|
||||
.filter((entry) => entry.isFile() && isPos(entry.name))
|
||||
.map((entry) => entry.name as SupportedPos)
|
||||
.sort();
|
||||
|
||||
for (const pos of posFiles) {
|
||||
const filePath = path.join(languageDir, pos);
|
||||
const words = normalizeWords(readFileSync(filePath, "utf-8").split("\n"));
|
||||
lists.push({ sourceLanguage: language, pos, words, filePath });
|
||||
}
|
||||
}
|
||||
return lists;
|
||||
};
|
||||
123
data-pipeline/staging.ts
Normal file
123
data-pipeline/staging.ts
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
import { randomUUID } from "node:crypto";
|
||||
import { readFileSync } from "node:fs";
|
||||
import Database from "better-sqlite3";
|
||||
import type { SupportedLanguageCode, SupportedPos } from "@lila/shared";
|
||||
import type { GeminiWordEntry } from "./validate.js";
|
||||
|
||||
const STAGING_TABLES = ["words", "senses", "translations"] as const;
|
||||
|
||||
export const openStaging = (
|
||||
dbPath: string,
|
||||
schemaPath: string,
|
||||
): Database.Database => {
|
||||
const db = new Database(dbPath);
|
||||
db.pragma("foreign_keys = ON");
|
||||
|
||||
const existing = db
|
||||
.prepare<[], { name: string }>(
|
||||
`SELECT name FROM sqlite_master WHERE type = 'table' AND name IN ('words', 'senses', 'translations')`,
|
||||
)
|
||||
.all()
|
||||
.map((row) => row.name);
|
||||
|
||||
if (existing.length === 0) {
|
||||
db.exec(readFileSync(schemaPath, "utf-8"));
|
||||
} else if (existing.length !== STAGING_TABLES.length) {
|
||||
db.close();
|
||||
throw new Error(
|
||||
`staging database at ${dbPath} has a partial schema (found: ${existing.join(", ")}) — fix or delete it`,
|
||||
);
|
||||
}
|
||||
return db;
|
||||
};
|
||||
|
||||
export const getStagedHeadwords = (
|
||||
db: Database.Database,
|
||||
language: SupportedLanguageCode,
|
||||
pos: SupportedPos,
|
||||
): Set<string> => {
|
||||
const rows = db
|
||||
.prepare<
|
||||
[string, string],
|
||||
{ headword: string }
|
||||
>(`SELECT headword FROM words WHERE language_code = ? AND pos = ?`)
|
||||
.all(language, pos);
|
||||
return new Set(rows.map((row) => row.headword));
|
||||
};
|
||||
|
||||
/**
|
||||
* Writes one validated entry (word + senses + translations) atomically.
|
||||
* Returns "already-staged" without writing if the word exists — a partial
|
||||
* word can never be left behind, so no NOT NULL relaxation is needed.
|
||||
*/
|
||||
export const stageEntry = (
|
||||
db: Database.Database,
|
||||
entry: GeminiWordEntry,
|
||||
): "staged" | "already-staged" => {
|
||||
const insert = db.transaction((): "staged" | "already-staged" => {
|
||||
const existing = db
|
||||
.prepare<
|
||||
[string, string, string],
|
||||
{ id: string }
|
||||
>(`SELECT id FROM words WHERE headword = ? AND language_code = ? AND pos = ?`)
|
||||
.get(entry.headword, entry.language, entry.pos);
|
||||
if (existing !== undefined) {
|
||||
return "already-staged";
|
||||
}
|
||||
|
||||
const wordId = randomUUID();
|
||||
db.prepare(
|
||||
`INSERT INTO words (id, headword, language_code, pos) VALUES (?, ?, ?, ?)`,
|
||||
).run(wordId, entry.headword, entry.language, entry.pos);
|
||||
|
||||
const insertSense = db.prepare(
|
||||
`INSERT INTO senses (id, word_id, sense_index, difficulty, definitions, examples)
|
||||
VALUES (?, ?, ?, ?, ?, ?)`,
|
||||
);
|
||||
const insertTranslation = db.prepare(
|
||||
`INSERT INTO translations (id, sense_id, target_language_code, translation, gender, difficulty)
|
||||
VALUES (?, ?, ?, ?, ?, ?)`,
|
||||
);
|
||||
|
||||
for (const sense of entry.senses) {
|
||||
const senseId = randomUUID();
|
||||
insertSense.run(
|
||||
senseId,
|
||||
wordId,
|
||||
sense.sense_index,
|
||||
sense.difficulty,
|
||||
JSON.stringify(sense.definitions),
|
||||
JSON.stringify(sense.examples),
|
||||
);
|
||||
for (const translation of sense.translations) {
|
||||
insertTranslation.run(
|
||||
randomUUID(),
|
||||
senseId,
|
||||
translation.target_language,
|
||||
translation.word,
|
||||
translation.gender,
|
||||
translation.difficulty,
|
||||
);
|
||||
}
|
||||
}
|
||||
return "staged";
|
||||
});
|
||||
return insert();
|
||||
};
|
||||
|
||||
export type StagingCounts = {
|
||||
words: number;
|
||||
senses: number;
|
||||
translations: number;
|
||||
};
|
||||
|
||||
export const countStagedRows = (db: Database.Database): StagingCounts => {
|
||||
const count = (table: (typeof STAGING_TABLES)[number]): number =>
|
||||
db.prepare<[], { n: number }>(`SELECT COUNT(*) AS n FROM ${table}`).get()
|
||||
?.n ?? 0;
|
||||
return {
|
||||
words: count("words"),
|
||||
senses: count("senses"),
|
||||
translations: count("translations"),
|
||||
};
|
||||
};
|
||||
57
data-pipeline/tests/promptTemplate.test.ts
Normal file
57
data-pipeline/tests/promptTemplate.test.ts
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
import { describe, it, expect } from "vitest";
|
||||
import {
|
||||
loadPromptTemplate,
|
||||
renderPrompt,
|
||||
type PromptParams,
|
||||
} from "../promptTemplate.js";
|
||||
|
||||
const params: PromptParams = {
|
||||
sourceLanguage: "es",
|
||||
pos: "noun",
|
||||
targetLanguages: ["en", "it", "de", "fr"],
|
||||
words: ["suelo", "pared"],
|
||||
};
|
||||
|
||||
describe("renderPrompt", () => {
|
||||
it("substitutes every placeholder from the checked-in template", () => {
|
||||
const rendered = renderPrompt(loadPromptTemplate(), params);
|
||||
expect(rendered).toContain('Source language: Spanish ("es")');
|
||||
expect(rendered).toContain('language must be "es"');
|
||||
expect(rendered).toContain("Target languages: en, it, de, fr");
|
||||
expect(rendered).toContain('"en" | "it" | "de" | "fr"');
|
||||
expect(rendered).toContain("suelo\npared");
|
||||
expect(rendered).toContain("valid Spanish noun");
|
||||
expect(rendered).not.toMatch(/\{\{[A-Z_]+\}\}/);
|
||||
});
|
||||
|
||||
it("does not tell the model to exclude a non-source language as target", () => {
|
||||
const rendered = renderPrompt(loadPromptTemplate(), params);
|
||||
expect(rendered).toContain(
|
||||
'Do not include Spanish ("es") as a target_language.',
|
||||
);
|
||||
expect(rendered).not.toContain(
|
||||
"Do not include English as a target_language",
|
||||
);
|
||||
});
|
||||
|
||||
it("throws when the source language is listed as a target", () => {
|
||||
expect(() =>
|
||||
renderPrompt(loadPromptTemplate(), {
|
||||
...params,
|
||||
targetLanguages: ["en", "es"],
|
||||
}),
|
||||
).toThrow(/exclude the source language/);
|
||||
});
|
||||
|
||||
it("throws on an empty word batch", () => {
|
||||
expect(() =>
|
||||
renderPrompt(loadPromptTemplate(), { ...params, words: [] }),
|
||||
).toThrow(/empty word batch/);
|
||||
});
|
||||
|
||||
it("throws on unreplaced placeholders", () => {
|
||||
expect(() => renderPrompt("hello {{UNKNOWN_TOKEN}}", params)).toThrow(
|
||||
/UNKNOWN_TOKEN/,
|
||||
);
|
||||
});
|
||||
});
|
||||
54
data-pipeline/tests/sourceLists.test.ts
Normal file
54
data-pipeline/tests/sourceLists.test.ts
Normal file
|
|
@ -0,0 +1,54 @@
|
|||
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { describe, it, expect, afterAll } from "vitest";
|
||||
import { discoverSourceLists, normalizeWords } from "../sourceLists.js";
|
||||
|
||||
describe("normalizeWords", () => {
|
||||
it("trims whitespace and drops empty lines", () => {
|
||||
expect(normalizeWords([" Haus ", "", " ", "Tür"])).toEqual([
|
||||
"Haus",
|
||||
"Tür",
|
||||
]);
|
||||
});
|
||||
|
||||
it("dedups while preserving first-occurrence order", () => {
|
||||
expect(normalizeWords(["centro", "borde", "centro", "suelo"])).toEqual([
|
||||
"centro",
|
||||
"borde",
|
||||
"suelo",
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("discoverSourceLists", () => {
|
||||
const root = mkdtempSync(path.join(tmpdir(), "lila-source-data-"));
|
||||
afterAll(() => rmSync(root, { recursive: true, force: true }));
|
||||
|
||||
it("finds only supported language/pos paths and normalizes their words", () => {
|
||||
mkdirSync(path.join(root, "de"));
|
||||
mkdirSync(path.join(root, "es"));
|
||||
mkdirSync(path.join(root, "german")); // unsupported dir name — ignored
|
||||
writeFileSync(
|
||||
path.join(root, "de", "noun"),
|
||||
"Haus\nTür\nHaus\n\n Tisch \n",
|
||||
);
|
||||
writeFileSync(path.join(root, "es", "noun"), "casa\npared\n");
|
||||
writeFileSync(path.join(root, "es", "nouns"), "ignored\n"); // unsupported pos name
|
||||
writeFileSync(path.join(root, "german", "noun"), "ignored\n");
|
||||
writeFileSync(path.join(root, "stray.txt"), "ignored\n");
|
||||
|
||||
const lists = discoverSourceLists(root);
|
||||
expect(lists).toHaveLength(2);
|
||||
expect(lists[0]).toMatchObject({
|
||||
sourceLanguage: "de",
|
||||
pos: "noun",
|
||||
words: ["Haus", "Tür", "Tisch"],
|
||||
});
|
||||
expect(lists[1]).toMatchObject({
|
||||
sourceLanguage: "es",
|
||||
pos: "noun",
|
||||
words: ["casa", "pared"],
|
||||
});
|
||||
});
|
||||
});
|
||||
106
data-pipeline/tests/staging.test.ts
Normal file
106
data-pipeline/tests/staging.test.ts
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
import path from "node:path";
|
||||
import { describe, it, expect, beforeEach } from "vitest";
|
||||
import type Database from "better-sqlite3";
|
||||
import {
|
||||
countStagedRows,
|
||||
getStagedHeadwords,
|
||||
openStaging,
|
||||
stageEntry,
|
||||
} from "../staging.js";
|
||||
import type { GeminiWordEntry } from "../validate.js";
|
||||
|
||||
const SCHEMA_PATH = path.join(import.meta.dirname, "..", "db", "schema.sql");
|
||||
|
||||
const entry: GeminiWordEntry = {
|
||||
headword: "casa",
|
||||
language: "es",
|
||||
pos: "noun",
|
||||
senses: [
|
||||
{
|
||||
sense_index: 0,
|
||||
difficulty: "easy",
|
||||
definitions: ["Un edificio para vivir."],
|
||||
examples: ["Compraron una casa en la ciudad."],
|
||||
translations: [
|
||||
{
|
||||
target_language: "en",
|
||||
word: "house",
|
||||
gender: null,
|
||||
difficulty: "easy",
|
||||
},
|
||||
{
|
||||
target_language: "it",
|
||||
word: "casa",
|
||||
gender: "feminine",
|
||||
difficulty: "easy",
|
||||
},
|
||||
{
|
||||
target_language: "de",
|
||||
word: "Haus",
|
||||
gender: "neuter",
|
||||
difficulty: "easy",
|
||||
},
|
||||
{
|
||||
target_language: "fr",
|
||||
word: "maison",
|
||||
gender: "feminine",
|
||||
difficulty: "easy",
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
describe("staging", () => {
|
||||
let db: Database.Database;
|
||||
beforeEach(() => {
|
||||
db = openStaging(":memory:", SCHEMA_PATH);
|
||||
});
|
||||
|
||||
it("creates the schema from db/schema.sql on an empty database", () => {
|
||||
expect(countStagedRows(db)).toEqual({
|
||||
words: 0,
|
||||
senses: 0,
|
||||
translations: 0,
|
||||
});
|
||||
});
|
||||
|
||||
it("stages a word with its senses and translations atomically", () => {
|
||||
expect(stageEntry(db, entry)).toBe("staged");
|
||||
expect(countStagedRows(db)).toEqual({
|
||||
words: 1,
|
||||
senses: 1,
|
||||
translations: 4,
|
||||
});
|
||||
|
||||
const row = db
|
||||
.prepare<
|
||||
[],
|
||||
{ definitions: string; examples: string }
|
||||
>(`SELECT definitions, examples FROM senses`)
|
||||
.get();
|
||||
expect(JSON.parse(row?.definitions ?? "")).toEqual([
|
||||
"Un edificio para vivir.",
|
||||
]);
|
||||
expect(JSON.parse(row?.examples ?? "")).toEqual([
|
||||
"Compraron una casa en la ciudad.",
|
||||
]);
|
||||
});
|
||||
|
||||
it("is idempotent: staging the same word twice writes nothing new", () => {
|
||||
expect(stageEntry(db, entry)).toBe("staged");
|
||||
expect(stageEntry(db, entry)).toBe("already-staged");
|
||||
expect(countStagedRows(db)).toEqual({
|
||||
words: 1,
|
||||
senses: 1,
|
||||
translations: 4,
|
||||
});
|
||||
});
|
||||
|
||||
it("reports staged headwords per language and pos", () => {
|
||||
stageEntry(db, entry);
|
||||
expect(getStagedHeadwords(db, "es", "noun")).toEqual(new Set(["casa"]));
|
||||
expect(getStagedHeadwords(db, "de", "noun")).toEqual(new Set());
|
||||
expect(getStagedHeadwords(db, "es", "verb")).toEqual(new Set());
|
||||
});
|
||||
});
|
||||
285
data-pipeline/tests/validate.test.ts
Normal file
285
data-pipeline/tests/validate.test.ts
Normal file
|
|
@ -0,0 +1,285 @@
|
|||
import { describe, it, expect } from "vitest";
|
||||
import { validateEntry, type ValidationContext } from "../validate.js";
|
||||
|
||||
const ctx: ValidationContext = {
|
||||
sourceLanguage: "es",
|
||||
pos: "noun",
|
||||
targetLanguages: ["en", "it", "de", "fr"],
|
||||
inputWords: new Set(["casa", "banco"]),
|
||||
};
|
||||
|
||||
type Translation = {
|
||||
target_language: string;
|
||||
word: string;
|
||||
gender: string | null;
|
||||
difficulty: string;
|
||||
};
|
||||
|
||||
const translations = (): Translation[] => [
|
||||
{ target_language: "en", word: "house", gender: null, difficulty: "easy" },
|
||||
{
|
||||
target_language: "it",
|
||||
word: "casa",
|
||||
gender: "feminine",
|
||||
difficulty: "easy",
|
||||
},
|
||||
{ target_language: "de", word: "Haus", gender: "neuter", difficulty: "easy" },
|
||||
{
|
||||
target_language: "fr",
|
||||
word: "maison",
|
||||
gender: "feminine",
|
||||
difficulty: "easy",
|
||||
},
|
||||
];
|
||||
|
||||
const sense = (
|
||||
overrides: Record<string, unknown> = {},
|
||||
): Record<string, unknown> => ({
|
||||
sense_index: 0,
|
||||
difficulty: "easy",
|
||||
definitions: ["Un edificio para vivir."],
|
||||
examples: ["Compraron una casa en la ciudad."],
|
||||
translations: translations(),
|
||||
...overrides,
|
||||
});
|
||||
|
||||
const entry = (
|
||||
overrides: Record<string, unknown> = {},
|
||||
): Record<string, unknown> => ({
|
||||
headword: "casa",
|
||||
language: "es",
|
||||
pos: "noun",
|
||||
senses: [sense()],
|
||||
...overrides,
|
||||
});
|
||||
|
||||
const errorsOf = (raw: unknown): string[] => {
|
||||
const result = validateEntry(raw, ctx);
|
||||
return result.status === "invalid" ? result.errors : [];
|
||||
};
|
||||
|
||||
describe("validateEntry", () => {
|
||||
it("accepts a fully valid entry", () => {
|
||||
const result = validateEntry(entry(), ctx);
|
||||
expect(result.status).toBe("valid");
|
||||
});
|
||||
|
||||
it("treats senses: [] as empty (word skipped, not rejected)", () => {
|
||||
const result = validateEntry(entry({ senses: [] }), ctx);
|
||||
expect(result).toEqual({ status: "empty", headword: "casa" });
|
||||
});
|
||||
|
||||
it("rejects senses: [] when the rest of the entry is invalid", () => {
|
||||
const result = validateEntry(entry({ senses: [], language: "en" }), ctx);
|
||||
expect(result.status).toBe("invalid");
|
||||
});
|
||||
|
||||
it("rejects non-object input", () => {
|
||||
expect(validateEntry("casa", ctx).status).toBe("invalid");
|
||||
expect(validateEntry(null, ctx).status).toBe("invalid");
|
||||
expect(validateEntry([entry()], ctx).status).toBe("invalid");
|
||||
});
|
||||
|
||||
it("rejects a headword that was not in the input batch", () => {
|
||||
expect(errorsOf(entry({ headword: "perro" }))).toContainEqual(
|
||||
expect.stringContaining("not in the input batch"),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects a wrong source language", () => {
|
||||
expect(errorsOf(entry({ language: "en" }))).toContainEqual(
|
||||
expect.stringContaining('language must be "es"'),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects a wrong pos", () => {
|
||||
expect(errorsOf(entry({ pos: "verb" }))).toContainEqual(
|
||||
expect.stringContaining('pos must be "noun"'),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects more than 3 senses", () => {
|
||||
const senses = [0, 1, 2, 3].map((i) => sense({ sense_index: i }));
|
||||
expect(errorsOf(entry({ senses }))).toContainEqual(
|
||||
expect.stringContaining("at most 3"),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects non-sequential sense_index", () => {
|
||||
const senses = [sense({ sense_index: 0 }), sense({ sense_index: 2 })];
|
||||
expect(errorsOf(entry({ senses }))).toContainEqual(
|
||||
expect.stringContaining("sense_index must be 1"),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects an unknown difficulty", () => {
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ difficulty: "intermediate" })] })),
|
||||
).toContainEqual(expect.stringContaining("difficulty must be one of"));
|
||||
});
|
||||
|
||||
it("rejects empty definitions and examples", () => {
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ definitions: [] })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining("definitions must be a non-empty array"),
|
||||
);
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ examples: [""] })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining("examples must contain only non-empty strings"),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects a non-null gender for English targets", () => {
|
||||
const bad = translations();
|
||||
bad[0] = {
|
||||
target_language: "en",
|
||||
word: "house",
|
||||
gender: "feminine",
|
||||
difficulty: "easy",
|
||||
};
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining('gender must be null for target "en"'),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects a null gender for German targets", () => {
|
||||
const bad = translations();
|
||||
bad[2] = {
|
||||
target_language: "de",
|
||||
word: "Haus",
|
||||
gender: null,
|
||||
difficulty: "easy",
|
||||
};
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(expect.stringContaining('for target "de"'));
|
||||
});
|
||||
|
||||
it("rejects neuter for Romance-language targets", () => {
|
||||
const bad = translations();
|
||||
bad[3] = {
|
||||
target_language: "fr",
|
||||
word: "maison",
|
||||
gender: "neuter",
|
||||
difficulty: "easy",
|
||||
};
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(expect.stringContaining('for target "fr"'));
|
||||
});
|
||||
|
||||
it("rejects an invented gender value", () => {
|
||||
const bad = translations();
|
||||
bad[2] = {
|
||||
target_language: "de",
|
||||
word: "Haus",
|
||||
gender: "common",
|
||||
difficulty: "easy",
|
||||
};
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })).length,
|
||||
).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("rejects a missing target language", () => {
|
||||
const partial = translations().filter((t) => t.target_language !== "fr");
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: partial })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining('missing translation for target language "fr"'),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects the source language as a target", () => {
|
||||
const bad = [
|
||||
...translations(),
|
||||
{
|
||||
target_language: "es",
|
||||
word: "hogar",
|
||||
gender: "masculine",
|
||||
difficulty: "easy",
|
||||
},
|
||||
];
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(expect.stringContaining("target_language must be one of"));
|
||||
});
|
||||
|
||||
it("rejects more than 2 translations for one target language", () => {
|
||||
const bad = [
|
||||
...translations(),
|
||||
{
|
||||
target_language: "de",
|
||||
word: "Gebäude",
|
||||
gender: "neuter",
|
||||
difficulty: "medium",
|
||||
},
|
||||
{
|
||||
target_language: "de",
|
||||
word: "Heim",
|
||||
gender: "neuter",
|
||||
difficulty: "medium",
|
||||
},
|
||||
];
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining(
|
||||
'more than 2 translations for target language "de"',
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects duplicate translation words for one target language", () => {
|
||||
const bad = [
|
||||
...translations(),
|
||||
{
|
||||
target_language: "de",
|
||||
word: "Haus",
|
||||
gender: "neuter",
|
||||
difficulty: "medium",
|
||||
},
|
||||
];
|
||||
expect(
|
||||
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||
).toContainEqual(
|
||||
expect.stringContaining(
|
||||
'duplicate translation word for target language "de"',
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects a translation difficulty below the sense difficulty", () => {
|
||||
const easyTranslations = translations();
|
||||
expect(
|
||||
errorsOf(
|
||||
entry({
|
||||
senses: [
|
||||
sense({ difficulty: "medium", translations: easyTranslations }),
|
||||
],
|
||||
}),
|
||||
),
|
||||
).toContainEqual(
|
||||
expect.stringContaining('is lower than sense difficulty "medium"'),
|
||||
);
|
||||
});
|
||||
|
||||
it("allows a translation difficulty above the sense difficulty", () => {
|
||||
const harder = translations();
|
||||
harder[2] = {
|
||||
target_language: "de",
|
||||
word: "Geldinstitut",
|
||||
gender: "neuter",
|
||||
difficulty: "medium",
|
||||
};
|
||||
const result = validateEntry(
|
||||
entry({ senses: [sense({ translations: harder })] }),
|
||||
ctx,
|
||||
);
|
||||
expect(result.status).toBe("valid");
|
||||
});
|
||||
});
|
||||
239
data-pipeline/validate.ts
Normal file
239
data-pipeline/validate.ts
Normal file
|
|
@ -0,0 +1,239 @@
|
|||
import {
|
||||
DIFFICULTY_LEVELS,
|
||||
NOUN_GENDERS,
|
||||
type DifficultyLevel,
|
||||
type NounGender,
|
||||
type SupportedLanguageCode,
|
||||
type SupportedPos,
|
||||
} from "@lila/shared";
|
||||
|
||||
export type GeminiTranslation = {
|
||||
target_language: SupportedLanguageCode;
|
||||
word: string;
|
||||
gender: NounGender | null;
|
||||
difficulty: DifficultyLevel;
|
||||
};
|
||||
|
||||
export type GeminiSense = {
|
||||
sense_index: number;
|
||||
difficulty: DifficultyLevel;
|
||||
definitions: string[];
|
||||
examples: string[];
|
||||
translations: GeminiTranslation[];
|
||||
};
|
||||
|
||||
export type GeminiWordEntry = {
|
||||
headword: string;
|
||||
language: SupportedLanguageCode;
|
||||
pos: SupportedPos;
|
||||
senses: GeminiSense[];
|
||||
};
|
||||
|
||||
export type ValidationContext = {
|
||||
sourceLanguage: SupportedLanguageCode;
|
||||
pos: SupportedPos;
|
||||
targetLanguages: readonly SupportedLanguageCode[];
|
||||
inputWords: ReadonlySet<string>;
|
||||
};
|
||||
|
||||
/**
|
||||
* "empty" is the contract's way of saying "not a valid word of this POS"
|
||||
* (senses: []) — the word is skipped, not rejected.
|
||||
*/
|
||||
export type ValidationResult =
|
||||
| { status: "valid"; entry: GeminiWordEntry }
|
||||
| { status: "empty"; headword: string }
|
||||
| { status: "invalid"; errors: string[] };
|
||||
|
||||
const isRecord = (value: unknown): value is Record<string, unknown> =>
|
||||
typeof value === "object" && value !== null && !Array.isArray(value);
|
||||
|
||||
const isNonEmptyString = (value: unknown): value is string =>
|
||||
typeof value === "string" && value.trim() !== "";
|
||||
|
||||
const isDifficulty = (value: unknown): value is DifficultyLevel =>
|
||||
(DIFFICULTY_LEVELS as readonly unknown[]).includes(value);
|
||||
|
||||
const difficultyRank = (level: DifficultyLevel): number =>
|
||||
DIFFICULTY_LEVELS.indexOf(level);
|
||||
|
||||
const validGendersFor = (
|
||||
target: SupportedLanguageCode,
|
||||
): readonly (NounGender | null)[] => {
|
||||
if (target === "en") return [null];
|
||||
if (target === "de") return NOUN_GENDERS;
|
||||
return ["masculine", "feminine"];
|
||||
};
|
||||
|
||||
const checkStringArray = (
|
||||
value: unknown,
|
||||
label: string,
|
||||
errors: string[],
|
||||
): void => {
|
||||
if (!Array.isArray(value) || value.length === 0) {
|
||||
errors.push(`${label} must be a non-empty array`);
|
||||
return;
|
||||
}
|
||||
if (!value.every(isNonEmptyString)) {
|
||||
errors.push(`${label} must contain only non-empty strings`);
|
||||
}
|
||||
};
|
||||
|
||||
const checkTranslation = (
|
||||
raw: unknown,
|
||||
label: string,
|
||||
senseDifficulty: DifficultyLevel | null,
|
||||
ctx: ValidationContext,
|
||||
errors: string[],
|
||||
): void => {
|
||||
if (!isRecord(raw)) {
|
||||
errors.push(`${label} must be an object`);
|
||||
return;
|
||||
}
|
||||
const target = raw["target_language"];
|
||||
if (!(ctx.targetLanguages as readonly unknown[]).includes(target)) {
|
||||
errors.push(
|
||||
`${label}: target_language must be one of ${ctx.targetLanguages.join(", ")}`,
|
||||
);
|
||||
}
|
||||
if (!isNonEmptyString(raw["word"])) {
|
||||
errors.push(`${label}: word must be a non-empty string`);
|
||||
}
|
||||
const difficulty = raw["difficulty"];
|
||||
if (!isDifficulty(difficulty)) {
|
||||
errors.push(
|
||||
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
|
||||
);
|
||||
} else if (
|
||||
senseDifficulty !== null &&
|
||||
difficultyRank(difficulty) < difficultyRank(senseDifficulty)
|
||||
) {
|
||||
errors.push(
|
||||
`${label}: difficulty "${difficulty}" is lower than sense difficulty "${senseDifficulty}"`,
|
||||
);
|
||||
}
|
||||
if (
|
||||
typeof target === "string" &&
|
||||
(ctx.targetLanguages as readonly string[]).includes(target)
|
||||
) {
|
||||
const gender = raw["gender"];
|
||||
const allowed = validGendersFor(target as SupportedLanguageCode);
|
||||
if (!(allowed as readonly unknown[]).includes(gender)) {
|
||||
errors.push(
|
||||
`${label}: gender must be ${allowed.map((g) => g ?? "null").join(" or ")} for target "${target}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const checkSense = (
|
||||
raw: unknown,
|
||||
index: number,
|
||||
ctx: ValidationContext,
|
||||
errors: string[],
|
||||
): void => {
|
||||
const label = `senses[${index}]`;
|
||||
if (!isRecord(raw)) {
|
||||
errors.push(`${label} must be an object`);
|
||||
return;
|
||||
}
|
||||
if (raw["sense_index"] !== index) {
|
||||
errors.push(`${label}: sense_index must be ${index} (sequential from 0)`);
|
||||
}
|
||||
const senseDifficulty = isDifficulty(raw["difficulty"])
|
||||
? raw["difficulty"]
|
||||
: null;
|
||||
if (senseDifficulty === null) {
|
||||
errors.push(
|
||||
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
|
||||
);
|
||||
}
|
||||
checkStringArray(raw["definitions"], `${label}.definitions`, errors);
|
||||
checkStringArray(raw["examples"], `${label}.examples`, errors);
|
||||
|
||||
const translations = raw["translations"];
|
||||
if (!Array.isArray(translations) || translations.length === 0) {
|
||||
errors.push(`${label}.translations must be a non-empty array`);
|
||||
return;
|
||||
}
|
||||
translations.forEach((translation, i) => {
|
||||
checkTranslation(
|
||||
translation,
|
||||
`${label}.translations[${i}]`,
|
||||
senseDifficulty,
|
||||
ctx,
|
||||
errors,
|
||||
);
|
||||
});
|
||||
|
||||
const wordsPerTarget = new Map<string, string[]>();
|
||||
for (const translation of translations) {
|
||||
if (!isRecord(translation)) continue;
|
||||
const target = translation["target_language"];
|
||||
const word = translation["word"];
|
||||
if (typeof target !== "string" || typeof word !== "string") continue;
|
||||
const words = wordsPerTarget.get(target) ?? [];
|
||||
words.push(word);
|
||||
wordsPerTarget.set(target, words);
|
||||
}
|
||||
for (const target of ctx.targetLanguages) {
|
||||
const words = wordsPerTarget.get(target) ?? [];
|
||||
if (words.length === 0) {
|
||||
errors.push(
|
||||
`${label}: missing translation for target language "${target}"`,
|
||||
);
|
||||
}
|
||||
if (words.length > 2) {
|
||||
errors.push(
|
||||
`${label}: more than 2 translations for target language "${target}"`,
|
||||
);
|
||||
}
|
||||
if (new Set(words).size !== words.length) {
|
||||
errors.push(
|
||||
`${label}: duplicate translation word for target language "${target}"`,
|
||||
);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
export const validateEntry = (
|
||||
raw: unknown,
|
||||
ctx: ValidationContext,
|
||||
): ValidationResult => {
|
||||
const errors: string[] = [];
|
||||
if (!isRecord(raw)) {
|
||||
return { status: "invalid", errors: ["entry must be an object"] };
|
||||
}
|
||||
|
||||
const headword = raw["headword"];
|
||||
if (!isNonEmptyString(headword)) {
|
||||
errors.push("headword must be a non-empty string");
|
||||
} else if (!ctx.inputWords.has(headword)) {
|
||||
errors.push(`headword "${headword}" is not in the input batch`);
|
||||
}
|
||||
if (raw["language"] !== ctx.sourceLanguage) {
|
||||
errors.push(`language must be "${ctx.sourceLanguage}"`);
|
||||
}
|
||||
if (raw["pos"] !== ctx.pos) {
|
||||
errors.push(`pos must be "${ctx.pos}"`);
|
||||
}
|
||||
|
||||
const senses = raw["senses"];
|
||||
if (!Array.isArray(senses)) {
|
||||
errors.push("senses must be an array");
|
||||
} else if (senses.length === 0) {
|
||||
if (errors.length === 0 && isNonEmptyString(headword)) {
|
||||
return { status: "empty", headword };
|
||||
}
|
||||
} else {
|
||||
if (senses.length > 3) {
|
||||
errors.push("senses must contain at most 3 entries");
|
||||
}
|
||||
senses.forEach((sense, i) => checkSense(sense, i, ctx, errors));
|
||||
}
|
||||
|
||||
if (errors.length > 0) {
|
||||
return { status: "invalid", errors };
|
||||
}
|
||||
return { status: "valid", entry: raw as GeminiWordEntry };
|
||||
};
|
||||
Loading…
Add table
Add a link
Reference in a new issue