implementing phase 3 pipeline: gemini structured output, validation, sqlite staging
Prompt is now a template (fixes the hardcoded en/es leftovers in rules 2, 3, 15, 16, 26, 31). pipeline.ts replaces the pseudocode: wordlist normalization, skip-already-staged idempotency, batches of 20 against gemini-3.6-flash with responseSchema, raw responses persisted per batch, per-entry validation with rejection log, one transaction per word into db/staging.db. Flags: --langs --pos --max-batches --delay-ms --dry-run. Smoke run: 40/40 words staged (de+es, one batch each), 0 rejections. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
303bb9388c
commit
37c978e230
12 changed files with 1477 additions and 66 deletions
2
.gitignore
vendored
2
.gitignore
vendored
|
|
@ -13,4 +13,6 @@ __pycache__/
|
||||||
data-pipeline/kaikki-source-files/
|
data-pipeline/kaikki-source-files/
|
||||||
data-pipeline/db/staging.db
|
data-pipeline/db/staging.db
|
||||||
data-pipeline/.env
|
data-pipeline/.env
|
||||||
|
data-pipeline/responses/
|
||||||
|
data-pipeline/rejections/
|
||||||
.aider*
|
.aider*
|
||||||
|
|
|
||||||
171
data-pipeline/gemini.ts
Normal file
171
data-pipeline/gemini.ts
Normal file
|
|
@ -0,0 +1,171 @@
|
||||||
|
import {
|
||||||
|
DIFFICULTY_LEVELS,
|
||||||
|
NOUN_GENDERS,
|
||||||
|
type SupportedLanguageCode,
|
||||||
|
type SupportedPos,
|
||||||
|
} from "@lila/shared";
|
||||||
|
|
||||||
|
export const DEFAULT_MODEL = "gemini-3.6-flash";
|
||||||
|
|
||||||
|
const API_BASE = "https://generativelanguage.googleapis.com/v1beta/models";
|
||||||
|
const MAX_ATTEMPTS = 5;
|
||||||
|
const RETRYABLE_STATUS = new Set([429, 500, 503]);
|
||||||
|
|
||||||
|
type ResponseSchema = Record<string, unknown>;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* OpenAPI-subset schema for Gemini structured output: an array of word
|
||||||
|
* entries matching design-doc §6.3, with enums narrowed to this batch's
|
||||||
|
* source/POS/target languages.
|
||||||
|
*/
|
||||||
|
export const buildEntriesResponseSchema = (
|
||||||
|
sourceLanguage: SupportedLanguageCode,
|
||||||
|
pos: SupportedPos,
|
||||||
|
targetLanguages: readonly SupportedLanguageCode[],
|
||||||
|
): ResponseSchema => ({
|
||||||
|
type: "ARRAY",
|
||||||
|
items: {
|
||||||
|
type: "OBJECT",
|
||||||
|
properties: {
|
||||||
|
headword: { type: "STRING" },
|
||||||
|
language: { type: "STRING", enum: [sourceLanguage] },
|
||||||
|
pos: { type: "STRING", enum: [pos] },
|
||||||
|
senses: {
|
||||||
|
type: "ARRAY",
|
||||||
|
items: {
|
||||||
|
type: "OBJECT",
|
||||||
|
properties: {
|
||||||
|
sense_index: { type: "INTEGER" },
|
||||||
|
difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] },
|
||||||
|
definitions: { type: "ARRAY", items: { type: "STRING" } },
|
||||||
|
examples: { type: "ARRAY", items: { type: "STRING" } },
|
||||||
|
translations: {
|
||||||
|
type: "ARRAY",
|
||||||
|
items: {
|
||||||
|
type: "OBJECT",
|
||||||
|
properties: {
|
||||||
|
target_language: {
|
||||||
|
type: "STRING",
|
||||||
|
enum: [...targetLanguages],
|
||||||
|
},
|
||||||
|
word: { type: "STRING" },
|
||||||
|
gender: {
|
||||||
|
type: "STRING",
|
||||||
|
enum: [...NOUN_GENDERS],
|
||||||
|
nullable: true,
|
||||||
|
},
|
||||||
|
difficulty: { type: "STRING", enum: [...DIFFICULTY_LEVELS] },
|
||||||
|
},
|
||||||
|
required: ["target_language", "word", "gender", "difficulty"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
required: [
|
||||||
|
"sense_index",
|
||||||
|
"difficulty",
|
||||||
|
"definitions",
|
||||||
|
"examples",
|
||||||
|
"translations",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
required: ["headword", "language", "pos", "senses"],
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
const sleep = (ms: number): Promise<void> =>
|
||||||
|
new Promise((resolve) => setTimeout(resolve, ms));
|
||||||
|
|
||||||
|
const retryDelayMs = (body: string, attempt: number): number => {
|
||||||
|
const match = body.match(/"retryDelay":\s*"(\d+(?:\.\d+)?)s"/);
|
||||||
|
if (match?.[1] !== undefined) {
|
||||||
|
return Math.ceil(Number(match[1]) * 1000) + 500;
|
||||||
|
}
|
||||||
|
return 2 ** attempt * 2000;
|
||||||
|
};
|
||||||
|
|
||||||
|
const extractText = (body: unknown): string => {
|
||||||
|
if (isRecord(body)) {
|
||||||
|
const candidates = body["candidates"];
|
||||||
|
if (Array.isArray(candidates) && isRecord(candidates[0])) {
|
||||||
|
const candidate = candidates[0];
|
||||||
|
const finishReason = candidate["finishReason"];
|
||||||
|
if (finishReason !== undefined && finishReason !== "STOP") {
|
||||||
|
throw new Error(
|
||||||
|
`Gemini stopped early: finishReason=${JSON.stringify(finishReason)}`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
const content = candidate["content"];
|
||||||
|
if (isRecord(content)) {
|
||||||
|
const parts = content["parts"];
|
||||||
|
if (Array.isArray(parts) && isRecord(parts[0])) {
|
||||||
|
const text = parts[0]["text"];
|
||||||
|
if (typeof text === "string") return text;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw new Error("Gemini response contained no text candidate");
|
||||||
|
};
|
||||||
|
|
||||||
|
const isRecord = (value: unknown): value is Record<string, unknown> =>
|
||||||
|
typeof value === "object" && value !== null && !Array.isArray(value);
|
||||||
|
|
||||||
|
export const generateContent = async (
|
||||||
|
apiKey: string,
|
||||||
|
model: string,
|
||||||
|
prompt: string,
|
||||||
|
responseSchema: ResponseSchema,
|
||||||
|
): Promise<string> => {
|
||||||
|
let lastError = "";
|
||||||
|
for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) {
|
||||||
|
let response: Response;
|
||||||
|
try {
|
||||||
|
response = await fetch(`${API_BASE}/${model}:generateContent`, {
|
||||||
|
method: "POST",
|
||||||
|
headers: {
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"x-goog-api-key": apiKey,
|
||||||
|
},
|
||||||
|
body: JSON.stringify({
|
||||||
|
contents: [{ role: "user", parts: [{ text: prompt }] }],
|
||||||
|
generationConfig: {
|
||||||
|
responseMimeType: "application/json",
|
||||||
|
responseSchema,
|
||||||
|
temperature: 0.2,
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
} catch (error) {
|
||||||
|
lastError = `network error: ${error instanceof Error ? error.message : String(error)}`;
|
||||||
|
await sleep(2 ** attempt * 2000);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (response.ok) {
|
||||||
|
return extractText(await response.json());
|
||||||
|
}
|
||||||
|
|
||||||
|
const body = await response.text();
|
||||||
|
lastError = `HTTP ${response.status}: ${body.slice(0, 500)}`;
|
||||||
|
if (!RETRYABLE_STATUS.has(response.status)) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
await sleep(retryDelayMs(body, attempt));
|
||||||
|
}
|
||||||
|
throw new Error(`Gemini request failed after retries — ${lastError}`);
|
||||||
|
};
|
||||||
|
|
||||||
|
/** Parse the model's JSON text into an array of unknown entries. */
|
||||||
|
export const parseEntries = (rawText: string): unknown[] => {
|
||||||
|
const stripped = rawText
|
||||||
|
.trim()
|
||||||
|
.replace(/^```(?:json)?\s*/i, "")
|
||||||
|
.replace(/\s*```$/, "");
|
||||||
|
const parsed: unknown = JSON.parse(stripped);
|
||||||
|
if (!Array.isArray(parsed)) {
|
||||||
|
throw new Error("Gemini response is not a JSON array");
|
||||||
|
}
|
||||||
|
return parsed;
|
||||||
|
};
|
||||||
|
|
@ -1,45 +1,327 @@
|
||||||
// pipeline.ts pseudo code
|
import { appendFileSync, mkdirSync, writeFileSync } from "node:fs";
|
||||||
|
import path from "node:path";
|
||||||
|
import { parseArgs } from "node:util";
|
||||||
|
import {
|
||||||
|
SUPPORTED_LANGUAGE_CODES,
|
||||||
|
SUPPORTED_POS,
|
||||||
|
type SupportedLanguageCode,
|
||||||
|
type SupportedPos,
|
||||||
|
} from "@lila/shared";
|
||||||
|
import {
|
||||||
|
DEFAULT_MODEL,
|
||||||
|
buildEntriesResponseSchema,
|
||||||
|
generateContent,
|
||||||
|
parseEntries,
|
||||||
|
} from "./gemini.js";
|
||||||
|
import { loadPromptTemplate, renderPrompt } from "./promptTemplate.js";
|
||||||
|
import { discoverSourceLists, type SourceList } from "./sourceLists.js";
|
||||||
|
import {
|
||||||
|
countStagedRows,
|
||||||
|
getStagedHeadwords,
|
||||||
|
openStaging,
|
||||||
|
stageEntry,
|
||||||
|
} from "./staging.js";
|
||||||
|
import { validateEntry, type ValidationContext } from "./validate.js";
|
||||||
|
|
||||||
/*
|
const ROOT = import.meta.dirname;
|
||||||
step 1: discover source lists
|
const SOURCE_DATA_DIR = path.join(ROOT, "source-data");
|
||||||
|
const STAGING_DB_PATH = path.join(ROOT, "db", "staging.db");
|
||||||
|
const STAGING_SCHEMA_PATH = path.join(ROOT, "db", "schema.sql");
|
||||||
|
const RESPONSES_DIR = path.join(ROOT, "responses");
|
||||||
|
const REJECTIONS_DIR = path.join(ROOT, "rejections");
|
||||||
|
|
||||||
this will give us an array of objects with this schema
|
type CliOptions = {
|
||||||
sourceLanguage: 'en' | 'de' | 'es' | 'fr' | 'it'
|
langs: SupportedLanguageCode[] | null;
|
||||||
pos: 'noun' | 'verb' | 'adjective' | 'adverb'
|
pos: SupportedPos;
|
||||||
words: string[];
|
maxBatches: number | null;
|
||||||
filePath: string;
|
batchSize: number;
|
||||||
|
delayMs: number;
|
||||||
|
dryRun: boolean;
|
||||||
|
};
|
||||||
|
|
||||||
the terminal output should be something like:
|
type ListStats = {
|
||||||
|
list: SourceList;
|
||||||
|
pending: number;
|
||||||
|
batchesRun: number;
|
||||||
|
staged: number;
|
||||||
|
skipped: number;
|
||||||
|
rejected: number;
|
||||||
|
failedBatches: number;
|
||||||
|
};
|
||||||
|
|
||||||
found 5 source lists:
|
const parseCli = (): CliOptions => {
|
||||||
|
// pnpm forwards the "--" separator itself (pnpm pipeline:run -- --langs …);
|
||||||
|
// drop it so the flags after it are parsed as flags, not positionals.
|
||||||
|
const args = process.argv.slice(2);
|
||||||
|
if (args[0] === "--") args.shift();
|
||||||
|
const { values } = parseArgs({
|
||||||
|
args,
|
||||||
|
options: {
|
||||||
|
langs: { type: "string" },
|
||||||
|
pos: { type: "string", default: "noun" },
|
||||||
|
"max-batches": { type: "string" },
|
||||||
|
"batch-size": { type: "string", default: "20" },
|
||||||
|
"delay-ms": { type: "string", default: "6000" },
|
||||||
|
"dry-run": { type: "boolean", default: false },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
de: noun
|
const pos = values.pos as SupportedPos;
|
||||||
en: noun
|
if (!(SUPPORTED_POS as readonly string[]).includes(pos)) {
|
||||||
|
throw new Error(`--pos must be one of ${SUPPORTED_POS.join(", ")}`);
|
||||||
|
}
|
||||||
|
let langs: SupportedLanguageCode[] | null = null;
|
||||||
|
if (values.langs !== undefined) {
|
||||||
|
langs = values.langs.split(",").map((code) => {
|
||||||
|
const trimmed = code.trim() as SupportedLanguageCode;
|
||||||
|
if (!(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(trimmed)) {
|
||||||
|
throw new Error(`--langs: unknown language code "${trimmed}"`);
|
||||||
|
}
|
||||||
|
return trimmed;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
langs,
|
||||||
|
pos,
|
||||||
|
maxBatches:
|
||||||
|
values["max-batches"] !== undefined
|
||||||
|
? Number(values["max-batches"])
|
||||||
|
: null,
|
||||||
|
batchSize: Number(values["batch-size"]),
|
||||||
|
delayMs: Number(values["delay-ms"]),
|
||||||
|
dryRun: values["dry-run"],
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
and so on
|
const chunk = <T>(items: readonly T[], size: number): T[][] => {
|
||||||
|
const chunks: T[][] = [];
|
||||||
|
for (let i = 0; i < items.length; i += size) {
|
||||||
|
chunks.push(items.slice(i, i + size));
|
||||||
|
}
|
||||||
|
return chunks;
|
||||||
|
};
|
||||||
|
|
||||||
later on, it will also contain it: noun, verb, adjective etc
|
const sleep = (ms: number): Promise<void> =>
|
||||||
*/
|
new Promise((resolve) => setTimeout(resolve, ms));
|
||||||
|
|
||||||
/*
|
const rejectionFile = (list: SourceList): string =>
|
||||||
|
path.join(REJECTIONS_DIR, `${list.sourceLanguage}-${list.pos}.jsonl`);
|
||||||
|
|
||||||
step 2: validating source lists
|
const logRejection = (
|
||||||
|
list: SourceList,
|
||||||
|
headword: string | null,
|
||||||
|
errors: string[],
|
||||||
|
entry: unknown,
|
||||||
|
): void => {
|
||||||
|
mkdirSync(REJECTIONS_DIR, { recursive: true });
|
||||||
|
const line = JSON.stringify({
|
||||||
|
at: new Date().toISOString(),
|
||||||
|
sourceLanguage: list.sourceLanguage,
|
||||||
|
pos: list.pos,
|
||||||
|
headword,
|
||||||
|
errors,
|
||||||
|
entry,
|
||||||
|
});
|
||||||
|
appendFileSync(rejectionFile(list), `${line}\n`);
|
||||||
|
};
|
||||||
|
|
||||||
a small script that trims whitespaces, removes duplicated words etc
|
const saveRawResponse = (
|
||||||
|
list: SourceList,
|
||||||
|
batchIndex: number,
|
||||||
|
model: string,
|
||||||
|
words: readonly string[],
|
||||||
|
targetLanguages: readonly SupportedLanguageCode[],
|
||||||
|
rawText: string,
|
||||||
|
): void => {
|
||||||
|
mkdirSync(RESPONSES_DIR, { recursive: true });
|
||||||
|
const stamp = new Date().toISOString().replaceAll(":", "-");
|
||||||
|
const file = path.join(
|
||||||
|
RESPONSES_DIR,
|
||||||
|
`${list.sourceLanguage}-${list.pos}-${stamp}-batch${batchIndex}.json`,
|
||||||
|
);
|
||||||
|
writeFileSync(
|
||||||
|
file,
|
||||||
|
JSON.stringify(
|
||||||
|
{
|
||||||
|
model,
|
||||||
|
sourceLanguage: list.sourceLanguage,
|
||||||
|
pos: list.pos,
|
||||||
|
targetLanguages,
|
||||||
|
words,
|
||||||
|
receivedAt: new Date().toISOString(),
|
||||||
|
rawText,
|
||||||
|
},
|
||||||
|
null,
|
||||||
|
2,
|
||||||
|
),
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
terminal output: summary of how many words per pos per language were found
|
const headwordOf = (entry: unknown): string | null => {
|
||||||
|
if (typeof entry === "object" && entry !== null && !Array.isArray(entry)) {
|
||||||
|
const headword = (entry as Record<string, unknown>)["headword"];
|
||||||
|
if (typeof headword === "string") return headword;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
|
||||||
*/
|
const main = async (): Promise<void> => {
|
||||||
|
const options = parseCli();
|
||||||
|
const model = process.env["GEMINI_MODEL"] ?? DEFAULT_MODEL;
|
||||||
|
const apiKey = process.env["GEMINI_API_KEY"];
|
||||||
|
if (apiKey === undefined && !options.dryRun) {
|
||||||
|
throw new Error("GEMINI_API_KEY is not set (data-pipeline/.env)");
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
const allLists = discoverSourceLists(SOURCE_DATA_DIR);
|
||||||
|
console.log(`found ${allLists.length} source lists:\n`);
|
||||||
|
for (const list of allLists) {
|
||||||
|
console.log(
|
||||||
|
` ${list.sourceLanguage}: ${list.pos} (${list.words.length} unique words)`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
step 3: writing to database?
|
const lists = allLists.filter(
|
||||||
|
(list) =>
|
||||||
|
list.pos === options.pos &&
|
||||||
|
(options.langs === null || options.langs.includes(list.sourceLanguage)),
|
||||||
|
);
|
||||||
|
if (lists.length === 0) {
|
||||||
|
console.log("\nnothing matches the requested --langs/--pos, exiting");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
my thought: ill restart the pipeline several times during testing, and when adding more wordlists with other pos or extending the exisiting noun lists
|
const template = loadPromptTemplate();
|
||||||
eventually the lists will contain tens or hundreds of thousands of words
|
const db = openStaging(STAGING_DB_PATH, STAGING_SCHEMA_PATH);
|
||||||
how do we prevent reading and processing the same words multiple times?
|
const allStats: ListStats[] = [];
|
||||||
if we read and validate+normalize the wordlists and write them to the database, we could then read from the database fill the missing translations etc
|
let firstApiCall = true;
|
||||||
and not read the same words from the same text files multiple times?
|
|
||||||
|
|
||||||
if we do this, we have to adjust the database schema because there are several notNull() rows inside
|
for (const list of lists) {
|
||||||
*/
|
const staged = getStagedHeadwords(db, list.sourceLanguage, list.pos);
|
||||||
|
const pending = list.words.filter((word) => !staged.has(word));
|
||||||
|
const targetLanguages = SUPPORTED_LANGUAGE_CODES.filter(
|
||||||
|
(code) => code !== list.sourceLanguage,
|
||||||
|
);
|
||||||
|
const batches = chunk(pending, options.batchSize).slice(
|
||||||
|
0,
|
||||||
|
options.maxBatches ?? Number.POSITIVE_INFINITY,
|
||||||
|
);
|
||||||
|
|
||||||
|
console.log(
|
||||||
|
`\n${list.sourceLanguage}/${list.pos}: ${list.words.length} unique, ${staged.size} already staged, ${pending.length} pending → running ${batches.length} batch(es)`,
|
||||||
|
);
|
||||||
|
const stats: ListStats = {
|
||||||
|
list,
|
||||||
|
pending: pending.length,
|
||||||
|
batchesRun: 0,
|
||||||
|
staged: 0,
|
||||||
|
skipped: 0,
|
||||||
|
rejected: 0,
|
||||||
|
failedBatches: 0,
|
||||||
|
};
|
||||||
|
allStats.push(stats);
|
||||||
|
|
||||||
|
for (const [batchIndex, words] of batches.entries()) {
|
||||||
|
const prompt = renderPrompt(template, {
|
||||||
|
sourceLanguage: list.sourceLanguage,
|
||||||
|
pos: list.pos,
|
||||||
|
targetLanguages,
|
||||||
|
words,
|
||||||
|
});
|
||||||
|
|
||||||
|
if (options.dryRun) {
|
||||||
|
console.log(
|
||||||
|
` [dry-run] batch ${batchIndex + 1}/${batches.length}: ${words.join(", ")}`,
|
||||||
|
);
|
||||||
|
stats.batchesRun++;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!firstApiCall) {
|
||||||
|
await sleep(options.delayMs);
|
||||||
|
}
|
||||||
|
firstApiCall = false;
|
||||||
|
|
||||||
|
try {
|
||||||
|
const rawText = await generateContent(
|
||||||
|
apiKey as string,
|
||||||
|
model,
|
||||||
|
prompt,
|
||||||
|
buildEntriesResponseSchema(
|
||||||
|
list.sourceLanguage,
|
||||||
|
list.pos,
|
||||||
|
targetLanguages,
|
||||||
|
),
|
||||||
|
);
|
||||||
|
saveRawResponse(
|
||||||
|
list,
|
||||||
|
batchIndex + 1,
|
||||||
|
model,
|
||||||
|
words,
|
||||||
|
targetLanguages,
|
||||||
|
rawText,
|
||||||
|
);
|
||||||
|
|
||||||
|
const entries = parseEntries(rawText);
|
||||||
|
const ctx: ValidationContext = {
|
||||||
|
sourceLanguage: list.sourceLanguage,
|
||||||
|
pos: list.pos,
|
||||||
|
targetLanguages,
|
||||||
|
inputWords: new Set(words),
|
||||||
|
};
|
||||||
|
const covered = new Set<string>();
|
||||||
|
for (const entry of entries) {
|
||||||
|
const result = validateEntry(entry, ctx);
|
||||||
|
if (result.status === "valid") {
|
||||||
|
stageEntry(db, result.entry);
|
||||||
|
covered.add(result.entry.headword);
|
||||||
|
stats.staged++;
|
||||||
|
} else if (result.status === "empty") {
|
||||||
|
covered.add(result.headword);
|
||||||
|
stats.skipped++;
|
||||||
|
console.log(
|
||||||
|
` skipped "${result.headword}" (no valid ${list.pos} senses)`,
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
const headword = headwordOf(entry);
|
||||||
|
if (headword !== null) covered.add(headword);
|
||||||
|
logRejection(list, headword, result.errors, entry);
|
||||||
|
stats.rejected++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (const word of words) {
|
||||||
|
if (!covered.has(word)) {
|
||||||
|
logRejection(list, word, ["missing from Gemini response"], null);
|
||||||
|
stats.rejected++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stats.batchesRun++;
|
||||||
|
console.log(
|
||||||
|
` batch ${batchIndex + 1}/${batches.length} done — ${stats.staged} staged, ${stats.rejected} rejected, ${stats.skipped} skipped`,
|
||||||
|
);
|
||||||
|
} catch (error) {
|
||||||
|
stats.failedBatches++;
|
||||||
|
console.error(
|
||||||
|
` batch ${batchIndex + 1}/${batches.length} FAILED: ${error instanceof Error ? error.message : String(error)}`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
console.log("\n— summary —");
|
||||||
|
for (const stats of allStats) {
|
||||||
|
console.log(
|
||||||
|
`${stats.list.sourceLanguage}/${stats.list.pos}: ${stats.staged} staged, ${stats.skipped} skipped, ${stats.rejected} rejected, ${stats.failedBatches} failed batch(es), ${stats.pending - stats.staged - stats.skipped - stats.rejected} still pending`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
const totals = countStagedRows(db);
|
||||||
|
console.log(
|
||||||
|
`staging.db totals: ${totals.words} words, ${totals.senses} senses, ${totals.translations} translations`,
|
||||||
|
);
|
||||||
|
db.close();
|
||||||
|
};
|
||||||
|
|
||||||
|
main().catch((error: unknown) => {
|
||||||
|
console.error(error instanceof Error ? error.message : error);
|
||||||
|
process.exitCode = 1;
|
||||||
|
});
|
||||||
|
|
|
||||||
|
|
@ -1,31 +1,11 @@
|
||||||
You are a multilingual lexicographer generating vocabulary data for a language-learning app.
|
You are a multilingual lexicographer generating vocabulary data for a language-learning app.
|
||||||
|
|
||||||
Source language: spanish
|
Source language: {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}")
|
||||||
Part of speech: noun
|
Part of speech: {{POS}}
|
||||||
Target languages: en, it, de, fr
|
Target languages: {{TARGET_LANGUAGE_CODES}}
|
||||||
|
|
||||||
Input words:
|
Input words:
|
||||||
suelo
|
{{INPUT_WORDS}}
|
||||||
pared
|
|
||||||
techo
|
|
||||||
portón
|
|
||||||
valla
|
|
||||||
esquina
|
|
||||||
centro
|
|
||||||
borde
|
|
||||||
superficie
|
|
||||||
centro
|
|
||||||
suburbio
|
|
||||||
medioambiente
|
|
||||||
oportunidad
|
|
||||||
ventaja
|
|
||||||
decisión
|
|
||||||
paciencia
|
|
||||||
comportamiento
|
|
||||||
industria
|
|
||||||
conocimiento
|
|
||||||
solución
|
|
||||||
|
|
||||||
|
|
||||||
Return ONLY valid JSON.
|
Return ONLY valid JSON.
|
||||||
Do not include markdown fences.
|
Do not include markdown fences.
|
||||||
|
|
@ -39,8 +19,8 @@ Each word object must have this shape:
|
||||||
|
|
||||||
{
|
{
|
||||||
"headword": string,
|
"headword": string,
|
||||||
"language": "es",
|
"language": "{{SOURCE_LANGUAGE_CODE}}",
|
||||||
"pos": "noun",
|
"pos": "{{POS}}",
|
||||||
"senses": [
|
"senses": [
|
||||||
{
|
{
|
||||||
"sense_index": number,
|
"sense_index": number,
|
||||||
|
|
@ -49,7 +29,7 @@ Each word object must have this shape:
|
||||||
"examples": string[],
|
"examples": string[],
|
||||||
"translations": [
|
"translations": [
|
||||||
{
|
{
|
||||||
"target_language": "de" | "it" | "en" | "fr",
|
"target_language": {{TARGET_LANGUAGE_UNION}},
|
||||||
"word": string,
|
"word": string,
|
||||||
"gender": "masculine" | "feminine" | "neuter" | null,
|
"gender": "masculine" | "feminine" | "neuter" | null,
|
||||||
"difficulty": "easy" | "medium" | "hard"
|
"difficulty": "easy" | "medium" | "hard"
|
||||||
|
|
@ -62,36 +42,36 @@ Each word object must have this shape:
|
||||||
Rules:
|
Rules:
|
||||||
|
|
||||||
1. The headword must be exactly one of the input words.
|
1. The headword must be exactly one of the input words.
|
||||||
2. language must be "en".
|
2. language must be "{{SOURCE_LANGUAGE_CODE}}".
|
||||||
3. pos must be "noun".
|
3. pos must be "{{POS}}".
|
||||||
4. Include only common, learner-relevant senses.
|
4. Include only common, learner-relevant senses.
|
||||||
5. Most words should have 1 sense.
|
5. Most words should have 1 sense.
|
||||||
6. Polysemous words may have 2 or 3 senses.
|
6. Polysemous words may have 2 or 3 senses.
|
||||||
7. Do not include rare, archaic, highly technical, or literary senses unless they are common.
|
7. Do not include rare, archaic, highly technical, or literary senses unless they are common.
|
||||||
8. sense_index must start at 0 and increase by 1.
|
8. sense_index must start at 0 and increase by 1.
|
||||||
9. definitions must be written in English.
|
9. definitions must be written in {{SOURCE_LANGUAGE_NAME}}.
|
||||||
10. examples must be written in English.
|
10. examples must be written in {{SOURCE_LANGUAGE_NAME}}.
|
||||||
11. Include 1 or 2 definitions per sense.
|
11. Include 1 or 2 definitions per sense.
|
||||||
12. Include 1 or 2 example sentences per sense.
|
12. Include 1 or 2 example sentences per sense.
|
||||||
13. Each definition must be student-friendly and at most 15 words.
|
13. Each definition must be student-friendly and at most 15 words.
|
||||||
14. Each example should naturally contain the headword or a clear form of it.
|
14. Each example should naturally contain the headword or a clear form of it.
|
||||||
15. Every sense must have translations for all target languages: de, it, es, fr.
|
15. Every sense must have translations for all target languages: {{TARGET_LANGUAGE_CODES}}.
|
||||||
16. Do not include English as a target_language.
|
16. Do not include {{SOURCE_LANGUAGE_NAME}} ("{{SOURCE_LANGUAGE_CODE}}") as a target_language.
|
||||||
17. You may include up to 2 translations per target language per sense if they are genuinely common synonyms or difficulty variants.
|
17. You may include up to 2 translations per target language per sense if they are genuinely common synonyms or difficulty variants.
|
||||||
18. Do not include more than 2 translations per target language per sense.
|
18. Do not include more than 2 translations per target language per sense.
|
||||||
19. Do not duplicate the same translation word for the same target language within one sense.
|
19. Do not duplicate the same translation word for the same target language within one sense.
|
||||||
20. Use the base dictionary form of the translated noun.
|
20. Use the base dictionary form of the translated {{POS}}.
|
||||||
21. Do not include articles or determiners in translations.
|
21. Do not include articles or determiners in translations.
|
||||||
22. German translation nouns must be capitalized.
|
22. German translation nouns must be capitalized.
|
||||||
23. Spanish, French, and Italian translation nouns should be lowercase unless they are proper nouns.
|
23. Spanish, French, and Italian translation nouns should be lowercase unless they are proper nouns.
|
||||||
24. For target_language "de", gender must be "masculine", "feminine", or "neuter".
|
24. For target_language "de", gender must be "masculine", "feminine", or "neuter".
|
||||||
25. For target_language "it", "es", or "fr", gender must be "masculine" or "feminine".
|
25. For target_language "it", "es", or "fr", gender must be "masculine" or "feminine".
|
||||||
26. For this prompt, gender must never be null.
|
26. gender must be null if and only if target_language is "en".
|
||||||
27. difficulty must be one of: "easy", "medium", "hard".
|
27. difficulty must be one of: "easy", "medium", "hard".
|
||||||
28. senses.difficulty describes how common or advanced the meaning is.
|
28. senses.difficulty describes how common or advanced the meaning is.
|
||||||
29. translations.difficulty describes how difficult the specific target-language word is for a learner.
|
29. translations.difficulty describes how difficult the specific target-language word is for a learner.
|
||||||
30. A common translation like "Bank" may be easy, while a formal synonym like "Geldinstitut" may be medium.
|
30. A common translation like "Bank" may be easy, while a formal synonym like "Geldinstitut" may be medium.
|
||||||
31. If a word cannot be treated as a valid English noun, return it with "senses": [].
|
31. If a word cannot be treated as a valid {{SOURCE_LANGUAGE_NAME}} {{POS}}, return it with "senses": [].
|
||||||
|
|
||||||
Difficulty calibration:
|
Difficulty calibration:
|
||||||
|
|
||||||
|
|
@ -101,7 +81,7 @@ Difficulty calibration:
|
||||||
|
|
||||||
Do not include CEFR levels in the output.
|
Do not include CEFR levels in the output.
|
||||||
|
|
||||||
Example output shape for the English noun "bank":
|
Example output shape for the English noun "bank" (illustrative of the JSON shape only — your definitions and examples must be in {{SOURCE_LANGUAGE_NAME}}):
|
||||||
|
|
||||||
[
|
[
|
||||||
{
|
{
|
||||||
|
|
|
||||||
56
data-pipeline/promptTemplate.ts
Normal file
56
data-pipeline/promptTemplate.ts
Normal file
|
|
@ -0,0 +1,56 @@
|
||||||
|
import { readFileSync } from "node:fs";
|
||||||
|
import path from "node:path";
|
||||||
|
import type { SupportedLanguageCode, SupportedPos } from "@lila/shared";
|
||||||
|
|
||||||
|
export const LANGUAGE_NAMES: Record<SupportedLanguageCode, string> = {
|
||||||
|
en: "English",
|
||||||
|
it: "Italian",
|
||||||
|
de: "German",
|
||||||
|
fr: "French",
|
||||||
|
es: "Spanish",
|
||||||
|
};
|
||||||
|
|
||||||
|
export type PromptParams = {
|
||||||
|
sourceLanguage: SupportedLanguageCode;
|
||||||
|
pos: SupportedPos;
|
||||||
|
targetLanguages: readonly SupportedLanguageCode[];
|
||||||
|
words: readonly string[];
|
||||||
|
};
|
||||||
|
|
||||||
|
export const loadPromptTemplate = (): string =>
|
||||||
|
readFileSync(path.join(import.meta.dirname, "prompt"), "utf-8");
|
||||||
|
|
||||||
|
export const renderPrompt = (
|
||||||
|
template: string,
|
||||||
|
params: PromptParams,
|
||||||
|
): string => {
|
||||||
|
const { sourceLanguage, pos, targetLanguages, words } = params;
|
||||||
|
if (words.length === 0) {
|
||||||
|
throw new Error("renderPrompt: empty word batch");
|
||||||
|
}
|
||||||
|
if (
|
||||||
|
targetLanguages.length === 0 ||
|
||||||
|
targetLanguages.includes(sourceLanguage)
|
||||||
|
) {
|
||||||
|
throw new Error(
|
||||||
|
`renderPrompt: target languages must be non-empty and exclude the source language (got ${targetLanguages.join(", ")})`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
const rendered = template
|
||||||
|
.replaceAll("{{SOURCE_LANGUAGE_NAME}}", LANGUAGE_NAMES[sourceLanguage])
|
||||||
|
.replaceAll("{{SOURCE_LANGUAGE_CODE}}", sourceLanguage)
|
||||||
|
.replaceAll("{{POS}}", pos)
|
||||||
|
.replaceAll("{{TARGET_LANGUAGE_CODES}}", targetLanguages.join(", "))
|
||||||
|
.replaceAll(
|
||||||
|
"{{TARGET_LANGUAGE_UNION}}",
|
||||||
|
targetLanguages.map((code) => `"${code}"`).join(" | "),
|
||||||
|
)
|
||||||
|
.replaceAll("{{INPUT_WORDS}}", words.join("\n"));
|
||||||
|
|
||||||
|
const leftover = rendered.match(/\{\{[A-Z_]+\}\}/);
|
||||||
|
if (leftover) {
|
||||||
|
throw new Error(`renderPrompt: unreplaced placeholder ${leftover[0]}`);
|
||||||
|
}
|
||||||
|
return rendered;
|
||||||
|
};
|
||||||
56
data-pipeline/sourceLists.ts
Normal file
56
data-pipeline/sourceLists.ts
Normal file
|
|
@ -0,0 +1,56 @@
|
||||||
|
import { readdirSync, readFileSync } from "node:fs";
|
||||||
|
import path from "node:path";
|
||||||
|
import {
|
||||||
|
SUPPORTED_LANGUAGE_CODES,
|
||||||
|
SUPPORTED_POS,
|
||||||
|
type SupportedLanguageCode,
|
||||||
|
type SupportedPos,
|
||||||
|
} from "@lila/shared";
|
||||||
|
|
||||||
|
export type SourceList = {
|
||||||
|
sourceLanguage: SupportedLanguageCode;
|
||||||
|
pos: SupportedPos;
|
||||||
|
words: string[];
|
||||||
|
filePath: string;
|
||||||
|
};
|
||||||
|
|
||||||
|
const isLanguageCode = (value: string): value is SupportedLanguageCode =>
|
||||||
|
(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(value);
|
||||||
|
|
||||||
|
const isPos = (value: string): value is SupportedPos =>
|
||||||
|
(SUPPORTED_POS as readonly string[]).includes(value);
|
||||||
|
|
||||||
|
export const normalizeWords = (lines: readonly string[]): string[] => {
|
||||||
|
const seen = new Set<string>();
|
||||||
|
const words: string[] = [];
|
||||||
|
for (const line of lines) {
|
||||||
|
const word = line.trim();
|
||||||
|
if (word === "" || seen.has(word)) continue;
|
||||||
|
seen.add(word);
|
||||||
|
words.push(word);
|
||||||
|
}
|
||||||
|
return words;
|
||||||
|
};
|
||||||
|
|
||||||
|
export const discoverSourceLists = (rootDir: string): SourceList[] => {
|
||||||
|
const lists: SourceList[] = [];
|
||||||
|
const languageDirs = readdirSync(rootDir, { withFileTypes: true })
|
||||||
|
.filter((entry) => entry.isDirectory() && isLanguageCode(entry.name))
|
||||||
|
.map((entry) => entry.name as SupportedLanguageCode)
|
||||||
|
.sort();
|
||||||
|
|
||||||
|
for (const language of languageDirs) {
|
||||||
|
const languageDir = path.join(rootDir, language);
|
||||||
|
const posFiles = readdirSync(languageDir, { withFileTypes: true })
|
||||||
|
.filter((entry) => entry.isFile() && isPos(entry.name))
|
||||||
|
.map((entry) => entry.name as SupportedPos)
|
||||||
|
.sort();
|
||||||
|
|
||||||
|
for (const pos of posFiles) {
|
||||||
|
const filePath = path.join(languageDir, pos);
|
||||||
|
const words = normalizeWords(readFileSync(filePath, "utf-8").split("\n"));
|
||||||
|
lists.push({ sourceLanguage: language, pos, words, filePath });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return lists;
|
||||||
|
};
|
||||||
123
data-pipeline/staging.ts
Normal file
123
data-pipeline/staging.ts
Normal file
|
|
@ -0,0 +1,123 @@
|
||||||
|
import { randomUUID } from "node:crypto";
|
||||||
|
import { readFileSync } from "node:fs";
|
||||||
|
import Database from "better-sqlite3";
|
||||||
|
import type { SupportedLanguageCode, SupportedPos } from "@lila/shared";
|
||||||
|
import type { GeminiWordEntry } from "./validate.js";
|
||||||
|
|
||||||
|
const STAGING_TABLES = ["words", "senses", "translations"] as const;
|
||||||
|
|
||||||
|
export const openStaging = (
|
||||||
|
dbPath: string,
|
||||||
|
schemaPath: string,
|
||||||
|
): Database.Database => {
|
||||||
|
const db = new Database(dbPath);
|
||||||
|
db.pragma("foreign_keys = ON");
|
||||||
|
|
||||||
|
const existing = db
|
||||||
|
.prepare<[], { name: string }>(
|
||||||
|
`SELECT name FROM sqlite_master WHERE type = 'table' AND name IN ('words', 'senses', 'translations')`,
|
||||||
|
)
|
||||||
|
.all()
|
||||||
|
.map((row) => row.name);
|
||||||
|
|
||||||
|
if (existing.length === 0) {
|
||||||
|
db.exec(readFileSync(schemaPath, "utf-8"));
|
||||||
|
} else if (existing.length !== STAGING_TABLES.length) {
|
||||||
|
db.close();
|
||||||
|
throw new Error(
|
||||||
|
`staging database at ${dbPath} has a partial schema (found: ${existing.join(", ")}) — fix or delete it`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
return db;
|
||||||
|
};
|
||||||
|
|
||||||
|
export const getStagedHeadwords = (
|
||||||
|
db: Database.Database,
|
||||||
|
language: SupportedLanguageCode,
|
||||||
|
pos: SupportedPos,
|
||||||
|
): Set<string> => {
|
||||||
|
const rows = db
|
||||||
|
.prepare<
|
||||||
|
[string, string],
|
||||||
|
{ headword: string }
|
||||||
|
>(`SELECT headword FROM words WHERE language_code = ? AND pos = ?`)
|
||||||
|
.all(language, pos);
|
||||||
|
return new Set(rows.map((row) => row.headword));
|
||||||
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Writes one validated entry (word + senses + translations) atomically.
|
||||||
|
* Returns "already-staged" without writing if the word exists — a partial
|
||||||
|
* word can never be left behind, so no NOT NULL relaxation is needed.
|
||||||
|
*/
|
||||||
|
export const stageEntry = (
|
||||||
|
db: Database.Database,
|
||||||
|
entry: GeminiWordEntry,
|
||||||
|
): "staged" | "already-staged" => {
|
||||||
|
const insert = db.transaction((): "staged" | "already-staged" => {
|
||||||
|
const existing = db
|
||||||
|
.prepare<
|
||||||
|
[string, string, string],
|
||||||
|
{ id: string }
|
||||||
|
>(`SELECT id FROM words WHERE headword = ? AND language_code = ? AND pos = ?`)
|
||||||
|
.get(entry.headword, entry.language, entry.pos);
|
||||||
|
if (existing !== undefined) {
|
||||||
|
return "already-staged";
|
||||||
|
}
|
||||||
|
|
||||||
|
const wordId = randomUUID();
|
||||||
|
db.prepare(
|
||||||
|
`INSERT INTO words (id, headword, language_code, pos) VALUES (?, ?, ?, ?)`,
|
||||||
|
).run(wordId, entry.headword, entry.language, entry.pos);
|
||||||
|
|
||||||
|
const insertSense = db.prepare(
|
||||||
|
`INSERT INTO senses (id, word_id, sense_index, difficulty, definitions, examples)
|
||||||
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
||||||
|
);
|
||||||
|
const insertTranslation = db.prepare(
|
||||||
|
`INSERT INTO translations (id, sense_id, target_language_code, translation, gender, difficulty)
|
||||||
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
||||||
|
);
|
||||||
|
|
||||||
|
for (const sense of entry.senses) {
|
||||||
|
const senseId = randomUUID();
|
||||||
|
insertSense.run(
|
||||||
|
senseId,
|
||||||
|
wordId,
|
||||||
|
sense.sense_index,
|
||||||
|
sense.difficulty,
|
||||||
|
JSON.stringify(sense.definitions),
|
||||||
|
JSON.stringify(sense.examples),
|
||||||
|
);
|
||||||
|
for (const translation of sense.translations) {
|
||||||
|
insertTranslation.run(
|
||||||
|
randomUUID(),
|
||||||
|
senseId,
|
||||||
|
translation.target_language,
|
||||||
|
translation.word,
|
||||||
|
translation.gender,
|
||||||
|
translation.difficulty,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "staged";
|
||||||
|
});
|
||||||
|
return insert();
|
||||||
|
};
|
||||||
|
|
||||||
|
export type StagingCounts = {
|
||||||
|
words: number;
|
||||||
|
senses: number;
|
||||||
|
translations: number;
|
||||||
|
};
|
||||||
|
|
||||||
|
export const countStagedRows = (db: Database.Database): StagingCounts => {
|
||||||
|
const count = (table: (typeof STAGING_TABLES)[number]): number =>
|
||||||
|
db.prepare<[], { n: number }>(`SELECT COUNT(*) AS n FROM ${table}`).get()
|
||||||
|
?.n ?? 0;
|
||||||
|
return {
|
||||||
|
words: count("words"),
|
||||||
|
senses: count("senses"),
|
||||||
|
translations: count("translations"),
|
||||||
|
};
|
||||||
|
};
|
||||||
57
data-pipeline/tests/promptTemplate.test.ts
Normal file
57
data-pipeline/tests/promptTemplate.test.ts
Normal file
|
|
@ -0,0 +1,57 @@
|
||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import {
|
||||||
|
loadPromptTemplate,
|
||||||
|
renderPrompt,
|
||||||
|
type PromptParams,
|
||||||
|
} from "../promptTemplate.js";
|
||||||
|
|
||||||
|
const params: PromptParams = {
|
||||||
|
sourceLanguage: "es",
|
||||||
|
pos: "noun",
|
||||||
|
targetLanguages: ["en", "it", "de", "fr"],
|
||||||
|
words: ["suelo", "pared"],
|
||||||
|
};
|
||||||
|
|
||||||
|
describe("renderPrompt", () => {
|
||||||
|
it("substitutes every placeholder from the checked-in template", () => {
|
||||||
|
const rendered = renderPrompt(loadPromptTemplate(), params);
|
||||||
|
expect(rendered).toContain('Source language: Spanish ("es")');
|
||||||
|
expect(rendered).toContain('language must be "es"');
|
||||||
|
expect(rendered).toContain("Target languages: en, it, de, fr");
|
||||||
|
expect(rendered).toContain('"en" | "it" | "de" | "fr"');
|
||||||
|
expect(rendered).toContain("suelo\npared");
|
||||||
|
expect(rendered).toContain("valid Spanish noun");
|
||||||
|
expect(rendered).not.toMatch(/\{\{[A-Z_]+\}\}/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("does not tell the model to exclude a non-source language as target", () => {
|
||||||
|
const rendered = renderPrompt(loadPromptTemplate(), params);
|
||||||
|
expect(rendered).toContain(
|
||||||
|
'Do not include Spanish ("es") as a target_language.',
|
||||||
|
);
|
||||||
|
expect(rendered).not.toContain(
|
||||||
|
"Do not include English as a target_language",
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("throws when the source language is listed as a target", () => {
|
||||||
|
expect(() =>
|
||||||
|
renderPrompt(loadPromptTemplate(), {
|
||||||
|
...params,
|
||||||
|
targetLanguages: ["en", "es"],
|
||||||
|
}),
|
||||||
|
).toThrow(/exclude the source language/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("throws on an empty word batch", () => {
|
||||||
|
expect(() =>
|
||||||
|
renderPrompt(loadPromptTemplate(), { ...params, words: [] }),
|
||||||
|
).toThrow(/empty word batch/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("throws on unreplaced placeholders", () => {
|
||||||
|
expect(() => renderPrompt("hello {{UNKNOWN_TOKEN}}", params)).toThrow(
|
||||||
|
/UNKNOWN_TOKEN/,
|
||||||
|
);
|
||||||
|
});
|
||||||
|
});
|
||||||
54
data-pipeline/tests/sourceLists.test.ts
Normal file
54
data-pipeline/tests/sourceLists.test.ts
Normal file
|
|
@ -0,0 +1,54 @@
|
||||||
|
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||||
|
import { tmpdir } from "node:os";
|
||||||
|
import path from "node:path";
|
||||||
|
import { describe, it, expect, afterAll } from "vitest";
|
||||||
|
import { discoverSourceLists, normalizeWords } from "../sourceLists.js";
|
||||||
|
|
||||||
|
describe("normalizeWords", () => {
|
||||||
|
it("trims whitespace and drops empty lines", () => {
|
||||||
|
expect(normalizeWords([" Haus ", "", " ", "Tür"])).toEqual([
|
||||||
|
"Haus",
|
||||||
|
"Tür",
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("dedups while preserving first-occurrence order", () => {
|
||||||
|
expect(normalizeWords(["centro", "borde", "centro", "suelo"])).toEqual([
|
||||||
|
"centro",
|
||||||
|
"borde",
|
||||||
|
"suelo",
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("discoverSourceLists", () => {
|
||||||
|
const root = mkdtempSync(path.join(tmpdir(), "lila-source-data-"));
|
||||||
|
afterAll(() => rmSync(root, { recursive: true, force: true }));
|
||||||
|
|
||||||
|
it("finds only supported language/pos paths and normalizes their words", () => {
|
||||||
|
mkdirSync(path.join(root, "de"));
|
||||||
|
mkdirSync(path.join(root, "es"));
|
||||||
|
mkdirSync(path.join(root, "german")); // unsupported dir name — ignored
|
||||||
|
writeFileSync(
|
||||||
|
path.join(root, "de", "noun"),
|
||||||
|
"Haus\nTür\nHaus\n\n Tisch \n",
|
||||||
|
);
|
||||||
|
writeFileSync(path.join(root, "es", "noun"), "casa\npared\n");
|
||||||
|
writeFileSync(path.join(root, "es", "nouns"), "ignored\n"); // unsupported pos name
|
||||||
|
writeFileSync(path.join(root, "german", "noun"), "ignored\n");
|
||||||
|
writeFileSync(path.join(root, "stray.txt"), "ignored\n");
|
||||||
|
|
||||||
|
const lists = discoverSourceLists(root);
|
||||||
|
expect(lists).toHaveLength(2);
|
||||||
|
expect(lists[0]).toMatchObject({
|
||||||
|
sourceLanguage: "de",
|
||||||
|
pos: "noun",
|
||||||
|
words: ["Haus", "Tür", "Tisch"],
|
||||||
|
});
|
||||||
|
expect(lists[1]).toMatchObject({
|
||||||
|
sourceLanguage: "es",
|
||||||
|
pos: "noun",
|
||||||
|
words: ["casa", "pared"],
|
||||||
|
});
|
||||||
|
});
|
||||||
|
});
|
||||||
106
data-pipeline/tests/staging.test.ts
Normal file
106
data-pipeline/tests/staging.test.ts
Normal file
|
|
@ -0,0 +1,106 @@
|
||||||
|
import path from "node:path";
|
||||||
|
import { describe, it, expect, beforeEach } from "vitest";
|
||||||
|
import type Database from "better-sqlite3";
|
||||||
|
import {
|
||||||
|
countStagedRows,
|
||||||
|
getStagedHeadwords,
|
||||||
|
openStaging,
|
||||||
|
stageEntry,
|
||||||
|
} from "../staging.js";
|
||||||
|
import type { GeminiWordEntry } from "../validate.js";
|
||||||
|
|
||||||
|
const SCHEMA_PATH = path.join(import.meta.dirname, "..", "db", "schema.sql");
|
||||||
|
|
||||||
|
const entry: GeminiWordEntry = {
|
||||||
|
headword: "casa",
|
||||||
|
language: "es",
|
||||||
|
pos: "noun",
|
||||||
|
senses: [
|
||||||
|
{
|
||||||
|
sense_index: 0,
|
||||||
|
difficulty: "easy",
|
||||||
|
definitions: ["Un edificio para vivir."],
|
||||||
|
examples: ["Compraron una casa en la ciudad."],
|
||||||
|
translations: [
|
||||||
|
{
|
||||||
|
target_language: "en",
|
||||||
|
word: "house",
|
||||||
|
gender: null,
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
target_language: "it",
|
||||||
|
word: "casa",
|
||||||
|
gender: "feminine",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
target_language: "de",
|
||||||
|
word: "Haus",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
target_language: "fr",
|
||||||
|
word: "maison",
|
||||||
|
gender: "feminine",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
],
|
||||||
|
};
|
||||||
|
|
||||||
|
describe("staging", () => {
|
||||||
|
let db: Database.Database;
|
||||||
|
beforeEach(() => {
|
||||||
|
db = openStaging(":memory:", SCHEMA_PATH);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("creates the schema from db/schema.sql on an empty database", () => {
|
||||||
|
expect(countStagedRows(db)).toEqual({
|
||||||
|
words: 0,
|
||||||
|
senses: 0,
|
||||||
|
translations: 0,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("stages a word with its senses and translations atomically", () => {
|
||||||
|
expect(stageEntry(db, entry)).toBe("staged");
|
||||||
|
expect(countStagedRows(db)).toEqual({
|
||||||
|
words: 1,
|
||||||
|
senses: 1,
|
||||||
|
translations: 4,
|
||||||
|
});
|
||||||
|
|
||||||
|
const row = db
|
||||||
|
.prepare<
|
||||||
|
[],
|
||||||
|
{ definitions: string; examples: string }
|
||||||
|
>(`SELECT definitions, examples FROM senses`)
|
||||||
|
.get();
|
||||||
|
expect(JSON.parse(row?.definitions ?? "")).toEqual([
|
||||||
|
"Un edificio para vivir.",
|
||||||
|
]);
|
||||||
|
expect(JSON.parse(row?.examples ?? "")).toEqual([
|
||||||
|
"Compraron una casa en la ciudad.",
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("is idempotent: staging the same word twice writes nothing new", () => {
|
||||||
|
expect(stageEntry(db, entry)).toBe("staged");
|
||||||
|
expect(stageEntry(db, entry)).toBe("already-staged");
|
||||||
|
expect(countStagedRows(db)).toEqual({
|
||||||
|
words: 1,
|
||||||
|
senses: 1,
|
||||||
|
translations: 4,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("reports staged headwords per language and pos", () => {
|
||||||
|
stageEntry(db, entry);
|
||||||
|
expect(getStagedHeadwords(db, "es", "noun")).toEqual(new Set(["casa"]));
|
||||||
|
expect(getStagedHeadwords(db, "de", "noun")).toEqual(new Set());
|
||||||
|
expect(getStagedHeadwords(db, "es", "verb")).toEqual(new Set());
|
||||||
|
});
|
||||||
|
});
|
||||||
285
data-pipeline/tests/validate.test.ts
Normal file
285
data-pipeline/tests/validate.test.ts
Normal file
|
|
@ -0,0 +1,285 @@
|
||||||
|
import { describe, it, expect } from "vitest";
|
||||||
|
import { validateEntry, type ValidationContext } from "../validate.js";
|
||||||
|
|
||||||
|
const ctx: ValidationContext = {
|
||||||
|
sourceLanguage: "es",
|
||||||
|
pos: "noun",
|
||||||
|
targetLanguages: ["en", "it", "de", "fr"],
|
||||||
|
inputWords: new Set(["casa", "banco"]),
|
||||||
|
};
|
||||||
|
|
||||||
|
type Translation = {
|
||||||
|
target_language: string;
|
||||||
|
word: string;
|
||||||
|
gender: string | null;
|
||||||
|
difficulty: string;
|
||||||
|
};
|
||||||
|
|
||||||
|
const translations = (): Translation[] => [
|
||||||
|
{ target_language: "en", word: "house", gender: null, difficulty: "easy" },
|
||||||
|
{
|
||||||
|
target_language: "it",
|
||||||
|
word: "casa",
|
||||||
|
gender: "feminine",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
{ target_language: "de", word: "Haus", gender: "neuter", difficulty: "easy" },
|
||||||
|
{
|
||||||
|
target_language: "fr",
|
||||||
|
word: "maison",
|
||||||
|
gender: "feminine",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
const sense = (
|
||||||
|
overrides: Record<string, unknown> = {},
|
||||||
|
): Record<string, unknown> => ({
|
||||||
|
sense_index: 0,
|
||||||
|
difficulty: "easy",
|
||||||
|
definitions: ["Un edificio para vivir."],
|
||||||
|
examples: ["Compraron una casa en la ciudad."],
|
||||||
|
translations: translations(),
|
||||||
|
...overrides,
|
||||||
|
});
|
||||||
|
|
||||||
|
const entry = (
|
||||||
|
overrides: Record<string, unknown> = {},
|
||||||
|
): Record<string, unknown> => ({
|
||||||
|
headword: "casa",
|
||||||
|
language: "es",
|
||||||
|
pos: "noun",
|
||||||
|
senses: [sense()],
|
||||||
|
...overrides,
|
||||||
|
});
|
||||||
|
|
||||||
|
const errorsOf = (raw: unknown): string[] => {
|
||||||
|
const result = validateEntry(raw, ctx);
|
||||||
|
return result.status === "invalid" ? result.errors : [];
|
||||||
|
};
|
||||||
|
|
||||||
|
describe("validateEntry", () => {
|
||||||
|
it("accepts a fully valid entry", () => {
|
||||||
|
const result = validateEntry(entry(), ctx);
|
||||||
|
expect(result.status).toBe("valid");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("treats senses: [] as empty (word skipped, not rejected)", () => {
|
||||||
|
const result = validateEntry(entry({ senses: [] }), ctx);
|
||||||
|
expect(result).toEqual({ status: "empty", headword: "casa" });
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects senses: [] when the rest of the entry is invalid", () => {
|
||||||
|
const result = validateEntry(entry({ senses: [], language: "en" }), ctx);
|
||||||
|
expect(result.status).toBe("invalid");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects non-object input", () => {
|
||||||
|
expect(validateEntry("casa", ctx).status).toBe("invalid");
|
||||||
|
expect(validateEntry(null, ctx).status).toBe("invalid");
|
||||||
|
expect(validateEntry([entry()], ctx).status).toBe("invalid");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a headword that was not in the input batch", () => {
|
||||||
|
expect(errorsOf(entry({ headword: "perro" }))).toContainEqual(
|
||||||
|
expect.stringContaining("not in the input batch"),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a wrong source language", () => {
|
||||||
|
expect(errorsOf(entry({ language: "en" }))).toContainEqual(
|
||||||
|
expect.stringContaining('language must be "es"'),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a wrong pos", () => {
|
||||||
|
expect(errorsOf(entry({ pos: "verb" }))).toContainEqual(
|
||||||
|
expect.stringContaining('pos must be "noun"'),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects more than 3 senses", () => {
|
||||||
|
const senses = [0, 1, 2, 3].map((i) => sense({ sense_index: i }));
|
||||||
|
expect(errorsOf(entry({ senses }))).toContainEqual(
|
||||||
|
expect.stringContaining("at most 3"),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects non-sequential sense_index", () => {
|
||||||
|
const senses = [sense({ sense_index: 0 }), sense({ sense_index: 2 })];
|
||||||
|
expect(errorsOf(entry({ senses }))).toContainEqual(
|
||||||
|
expect.stringContaining("sense_index must be 1"),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects an unknown difficulty", () => {
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ difficulty: "intermediate" })] })),
|
||||||
|
).toContainEqual(expect.stringContaining("difficulty must be one of"));
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects empty definitions and examples", () => {
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ definitions: [] })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining("definitions must be a non-empty array"),
|
||||||
|
);
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ examples: [""] })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining("examples must contain only non-empty strings"),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a non-null gender for English targets", () => {
|
||||||
|
const bad = translations();
|
||||||
|
bad[0] = {
|
||||||
|
target_language: "en",
|
||||||
|
word: "house",
|
||||||
|
gender: "feminine",
|
||||||
|
difficulty: "easy",
|
||||||
|
};
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining('gender must be null for target "en"'),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a null gender for German targets", () => {
|
||||||
|
const bad = translations();
|
||||||
|
bad[2] = {
|
||||||
|
target_language: "de",
|
||||||
|
word: "Haus",
|
||||||
|
gender: null,
|
||||||
|
difficulty: "easy",
|
||||||
|
};
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(expect.stringContaining('for target "de"'));
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects neuter for Romance-language targets", () => {
|
||||||
|
const bad = translations();
|
||||||
|
bad[3] = {
|
||||||
|
target_language: "fr",
|
||||||
|
word: "maison",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "easy",
|
||||||
|
};
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(expect.stringContaining('for target "fr"'));
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects an invented gender value", () => {
|
||||||
|
const bad = translations();
|
||||||
|
bad[2] = {
|
||||||
|
target_language: "de",
|
||||||
|
word: "Haus",
|
||||||
|
gender: "common",
|
||||||
|
difficulty: "easy",
|
||||||
|
};
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })).length,
|
||||||
|
).toBeGreaterThan(0);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a missing target language", () => {
|
||||||
|
const partial = translations().filter((t) => t.target_language !== "fr");
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: partial })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining('missing translation for target language "fr"'),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects the source language as a target", () => {
|
||||||
|
const bad = [
|
||||||
|
...translations(),
|
||||||
|
{
|
||||||
|
target_language: "es",
|
||||||
|
word: "hogar",
|
||||||
|
gender: "masculine",
|
||||||
|
difficulty: "easy",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(expect.stringContaining("target_language must be one of"));
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects more than 2 translations for one target language", () => {
|
||||||
|
const bad = [
|
||||||
|
...translations(),
|
||||||
|
{
|
||||||
|
target_language: "de",
|
||||||
|
word: "Gebäude",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "medium",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
target_language: "de",
|
||||||
|
word: "Heim",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "medium",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining(
|
||||||
|
'more than 2 translations for target language "de"',
|
||||||
|
),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects duplicate translation words for one target language", () => {
|
||||||
|
const bad = [
|
||||||
|
...translations(),
|
||||||
|
{
|
||||||
|
target_language: "de",
|
||||||
|
word: "Haus",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "medium",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
expect(
|
||||||
|
errorsOf(entry({ senses: [sense({ translations: bad })] })),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining(
|
||||||
|
'duplicate translation word for target language "de"',
|
||||||
|
),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a translation difficulty below the sense difficulty", () => {
|
||||||
|
const easyTranslations = translations();
|
||||||
|
expect(
|
||||||
|
errorsOf(
|
||||||
|
entry({
|
||||||
|
senses: [
|
||||||
|
sense({ difficulty: "medium", translations: easyTranslations }),
|
||||||
|
],
|
||||||
|
}),
|
||||||
|
),
|
||||||
|
).toContainEqual(
|
||||||
|
expect.stringContaining('is lower than sense difficulty "medium"'),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("allows a translation difficulty above the sense difficulty", () => {
|
||||||
|
const harder = translations();
|
||||||
|
harder[2] = {
|
||||||
|
target_language: "de",
|
||||||
|
word: "Geldinstitut",
|
||||||
|
gender: "neuter",
|
||||||
|
difficulty: "medium",
|
||||||
|
};
|
||||||
|
const result = validateEntry(
|
||||||
|
entry({ senses: [sense({ translations: harder })] }),
|
||||||
|
ctx,
|
||||||
|
);
|
||||||
|
expect(result.status).toBe("valid");
|
||||||
|
});
|
||||||
|
});
|
||||||
239
data-pipeline/validate.ts
Normal file
239
data-pipeline/validate.ts
Normal file
|
|
@ -0,0 +1,239 @@
|
||||||
|
import {
|
||||||
|
DIFFICULTY_LEVELS,
|
||||||
|
NOUN_GENDERS,
|
||||||
|
type DifficultyLevel,
|
||||||
|
type NounGender,
|
||||||
|
type SupportedLanguageCode,
|
||||||
|
type SupportedPos,
|
||||||
|
} from "@lila/shared";
|
||||||
|
|
||||||
|
export type GeminiTranslation = {
|
||||||
|
target_language: SupportedLanguageCode;
|
||||||
|
word: string;
|
||||||
|
gender: NounGender | null;
|
||||||
|
difficulty: DifficultyLevel;
|
||||||
|
};
|
||||||
|
|
||||||
|
export type GeminiSense = {
|
||||||
|
sense_index: number;
|
||||||
|
difficulty: DifficultyLevel;
|
||||||
|
definitions: string[];
|
||||||
|
examples: string[];
|
||||||
|
translations: GeminiTranslation[];
|
||||||
|
};
|
||||||
|
|
||||||
|
export type GeminiWordEntry = {
|
||||||
|
headword: string;
|
||||||
|
language: SupportedLanguageCode;
|
||||||
|
pos: SupportedPos;
|
||||||
|
senses: GeminiSense[];
|
||||||
|
};
|
||||||
|
|
||||||
|
export type ValidationContext = {
|
||||||
|
sourceLanguage: SupportedLanguageCode;
|
||||||
|
pos: SupportedPos;
|
||||||
|
targetLanguages: readonly SupportedLanguageCode[];
|
||||||
|
inputWords: ReadonlySet<string>;
|
||||||
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* "empty" is the contract's way of saying "not a valid word of this POS"
|
||||||
|
* (senses: []) — the word is skipped, not rejected.
|
||||||
|
*/
|
||||||
|
export type ValidationResult =
|
||||||
|
| { status: "valid"; entry: GeminiWordEntry }
|
||||||
|
| { status: "empty"; headword: string }
|
||||||
|
| { status: "invalid"; errors: string[] };
|
||||||
|
|
||||||
|
const isRecord = (value: unknown): value is Record<string, unknown> =>
|
||||||
|
typeof value === "object" && value !== null && !Array.isArray(value);
|
||||||
|
|
||||||
|
const isNonEmptyString = (value: unknown): value is string =>
|
||||||
|
typeof value === "string" && value.trim() !== "";
|
||||||
|
|
||||||
|
const isDifficulty = (value: unknown): value is DifficultyLevel =>
|
||||||
|
(DIFFICULTY_LEVELS as readonly unknown[]).includes(value);
|
||||||
|
|
||||||
|
const difficultyRank = (level: DifficultyLevel): number =>
|
||||||
|
DIFFICULTY_LEVELS.indexOf(level);
|
||||||
|
|
||||||
|
const validGendersFor = (
|
||||||
|
target: SupportedLanguageCode,
|
||||||
|
): readonly (NounGender | null)[] => {
|
||||||
|
if (target === "en") return [null];
|
||||||
|
if (target === "de") return NOUN_GENDERS;
|
||||||
|
return ["masculine", "feminine"];
|
||||||
|
};
|
||||||
|
|
||||||
|
const checkStringArray = (
|
||||||
|
value: unknown,
|
||||||
|
label: string,
|
||||||
|
errors: string[],
|
||||||
|
): void => {
|
||||||
|
if (!Array.isArray(value) || value.length === 0) {
|
||||||
|
errors.push(`${label} must be a non-empty array`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!value.every(isNonEmptyString)) {
|
||||||
|
errors.push(`${label} must contain only non-empty strings`);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const checkTranslation = (
|
||||||
|
raw: unknown,
|
||||||
|
label: string,
|
||||||
|
senseDifficulty: DifficultyLevel | null,
|
||||||
|
ctx: ValidationContext,
|
||||||
|
errors: string[],
|
||||||
|
): void => {
|
||||||
|
if (!isRecord(raw)) {
|
||||||
|
errors.push(`${label} must be an object`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const target = raw["target_language"];
|
||||||
|
if (!(ctx.targetLanguages as readonly unknown[]).includes(target)) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: target_language must be one of ${ctx.targetLanguages.join(", ")}`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (!isNonEmptyString(raw["word"])) {
|
||||||
|
errors.push(`${label}: word must be a non-empty string`);
|
||||||
|
}
|
||||||
|
const difficulty = raw["difficulty"];
|
||||||
|
if (!isDifficulty(difficulty)) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
|
||||||
|
);
|
||||||
|
} else if (
|
||||||
|
senseDifficulty !== null &&
|
||||||
|
difficultyRank(difficulty) < difficultyRank(senseDifficulty)
|
||||||
|
) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: difficulty "${difficulty}" is lower than sense difficulty "${senseDifficulty}"`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (
|
||||||
|
typeof target === "string" &&
|
||||||
|
(ctx.targetLanguages as readonly string[]).includes(target)
|
||||||
|
) {
|
||||||
|
const gender = raw["gender"];
|
||||||
|
const allowed = validGendersFor(target as SupportedLanguageCode);
|
||||||
|
if (!(allowed as readonly unknown[]).includes(gender)) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: gender must be ${allowed.map((g) => g ?? "null").join(" or ")} for target "${target}"`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const checkSense = (
|
||||||
|
raw: unknown,
|
||||||
|
index: number,
|
||||||
|
ctx: ValidationContext,
|
||||||
|
errors: string[],
|
||||||
|
): void => {
|
||||||
|
const label = `senses[${index}]`;
|
||||||
|
if (!isRecord(raw)) {
|
||||||
|
errors.push(`${label} must be an object`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (raw["sense_index"] !== index) {
|
||||||
|
errors.push(`${label}: sense_index must be ${index} (sequential from 0)`);
|
||||||
|
}
|
||||||
|
const senseDifficulty = isDifficulty(raw["difficulty"])
|
||||||
|
? raw["difficulty"]
|
||||||
|
: null;
|
||||||
|
if (senseDifficulty === null) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
checkStringArray(raw["definitions"], `${label}.definitions`, errors);
|
||||||
|
checkStringArray(raw["examples"], `${label}.examples`, errors);
|
||||||
|
|
||||||
|
const translations = raw["translations"];
|
||||||
|
if (!Array.isArray(translations) || translations.length === 0) {
|
||||||
|
errors.push(`${label}.translations must be a non-empty array`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
translations.forEach((translation, i) => {
|
||||||
|
checkTranslation(
|
||||||
|
translation,
|
||||||
|
`${label}.translations[${i}]`,
|
||||||
|
senseDifficulty,
|
||||||
|
ctx,
|
||||||
|
errors,
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
const wordsPerTarget = new Map<string, string[]>();
|
||||||
|
for (const translation of translations) {
|
||||||
|
if (!isRecord(translation)) continue;
|
||||||
|
const target = translation["target_language"];
|
||||||
|
const word = translation["word"];
|
||||||
|
if (typeof target !== "string" || typeof word !== "string") continue;
|
||||||
|
const words = wordsPerTarget.get(target) ?? [];
|
||||||
|
words.push(word);
|
||||||
|
wordsPerTarget.set(target, words);
|
||||||
|
}
|
||||||
|
for (const target of ctx.targetLanguages) {
|
||||||
|
const words = wordsPerTarget.get(target) ?? [];
|
||||||
|
if (words.length === 0) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: missing translation for target language "${target}"`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (words.length > 2) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: more than 2 translations for target language "${target}"`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if (new Set(words).size !== words.length) {
|
||||||
|
errors.push(
|
||||||
|
`${label}: duplicate translation word for target language "${target}"`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
export const validateEntry = (
|
||||||
|
raw: unknown,
|
||||||
|
ctx: ValidationContext,
|
||||||
|
): ValidationResult => {
|
||||||
|
const errors: string[] = [];
|
||||||
|
if (!isRecord(raw)) {
|
||||||
|
return { status: "invalid", errors: ["entry must be an object"] };
|
||||||
|
}
|
||||||
|
|
||||||
|
const headword = raw["headword"];
|
||||||
|
if (!isNonEmptyString(headword)) {
|
||||||
|
errors.push("headword must be a non-empty string");
|
||||||
|
} else if (!ctx.inputWords.has(headword)) {
|
||||||
|
errors.push(`headword "${headword}" is not in the input batch`);
|
||||||
|
}
|
||||||
|
if (raw["language"] !== ctx.sourceLanguage) {
|
||||||
|
errors.push(`language must be "${ctx.sourceLanguage}"`);
|
||||||
|
}
|
||||||
|
if (raw["pos"] !== ctx.pos) {
|
||||||
|
errors.push(`pos must be "${ctx.pos}"`);
|
||||||
|
}
|
||||||
|
|
||||||
|
const senses = raw["senses"];
|
||||||
|
if (!Array.isArray(senses)) {
|
||||||
|
errors.push("senses must be an array");
|
||||||
|
} else if (senses.length === 0) {
|
||||||
|
if (errors.length === 0 && isNonEmptyString(headword)) {
|
||||||
|
return { status: "empty", headword };
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
if (senses.length > 3) {
|
||||||
|
errors.push("senses must contain at most 3 entries");
|
||||||
|
}
|
||||||
|
senses.forEach((sense, i) => checkSense(sense, i, ctx, errors));
|
||||||
|
}
|
||||||
|
|
||||||
|
if (errors.length > 0) {
|
||||||
|
return { status: "invalid", errors };
|
||||||
|
}
|
||||||
|
return { status: "valid", entry: raw as GeminiWordEntry };
|
||||||
|
};
|
||||||
Loading…
Add table
Add a link
Reference in a new issue