import { DIFFICULTY_LEVELS, NOUN_GENDERS, type DifficultyLevel, type NounGender, type SupportedLanguageCode, type SupportedPos, } from "@lila/shared"; export type GeminiTranslation = { target_language: SupportedLanguageCode; word: string; gender: NounGender | null; difficulty: DifficultyLevel; }; export type GeminiSense = { sense_index: number; difficulty: DifficultyLevel; definitions: string[]; examples: string[]; translations: GeminiTranslation[]; }; export type GeminiWordEntry = { headword: string; language: SupportedLanguageCode; pos: SupportedPos; senses: GeminiSense[]; }; export type ValidationContext = { sourceLanguage: SupportedLanguageCode; pos: SupportedPos; targetLanguages: readonly SupportedLanguageCode[]; inputWords: ReadonlySet; }; /** * "empty" is the contract's way of saying "not a valid word of this POS" * (senses: []) — the word is skipped, not rejected. * * A "valid" result carries the *normalized* entry, which may differ from the * input — see `applySenseDifficultyFloor`. `normalizations` is a human-readable * record of any repair applied, so the pipeline can report how often the model * needed correcting instead of silently papering over it. */ export type ValidationResult = | { status: "valid"; entry: GeminiWordEntry; normalizations: string[] } | { status: "empty"; headword: string } | { status: "invalid"; errors: string[] }; const isRecord = (value: unknown): value is Record => typeof value === "object" && value !== null && !Array.isArray(value); const isNonEmptyString = (value: unknown): value is string => typeof value === "string" && value.trim() !== ""; const isDifficulty = (value: unknown): value is DifficultyLevel => (DIFFICULTY_LEVELS as readonly unknown[]).includes(value); const difficultyRank = (level: DifficultyLevel): number => DIFFICULTY_LEVELS.indexOf(level); const validGendersFor = ( target: SupportedLanguageCode, ): readonly (NounGender | null)[] => { if (target === "en") return [null]; if (target === "de") return NOUN_GENDERS; return ["masculine", "feminine"]; }; const checkStringArray = ( value: unknown, label: string, errors: string[], ): void => { if (!Array.isArray(value) || value.length === 0) { errors.push(`${label} must be a non-empty array`); return; } if (!value.every(isNonEmptyString)) { errors.push(`${label} must contain only non-empty strings`); } }; const checkTranslation = ( raw: unknown, label: string, ctx: ValidationContext, errors: string[], ): void => { if (!isRecord(raw)) { errors.push(`${label} must be an object`); return; } const target = raw["target_language"]; if (!(ctx.targetLanguages as readonly unknown[]).includes(target)) { errors.push( `${label}: target_language must be one of ${ctx.targetLanguages.join(", ")}`, ); } if (!isNonEmptyString(raw["word"])) { errors.push(`${label}: word must be a non-empty string`); } // A translation ranked below its sense is not an error — the sense is the // derived value and gets floored to match. See applySenseDifficultyFloor. if (!isDifficulty(raw["difficulty"])) { errors.push( `${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`, ); } if ( typeof target === "string" && (ctx.targetLanguages as readonly string[]).includes(target) ) { const gender = raw["gender"]; const allowed = validGendersFor(target as SupportedLanguageCode); if (!(allowed as readonly unknown[]).includes(gender)) { errors.push( `${label}: gender must be ${allowed.map((g) => g ?? "null").join(" or ")} for target "${target}"`, ); } } }; const checkSense = ( raw: unknown, index: number, ctx: ValidationContext, errors: string[], ): void => { const label = `senses[${index}]`; if (!isRecord(raw)) { errors.push(`${label} must be an object`); return; } if (raw["sense_index"] !== index) { errors.push(`${label}: sense_index must be ${index} (sequential from 0)`); } const senseDifficulty = isDifficulty(raw["difficulty"]) ? raw["difficulty"] : null; if (senseDifficulty === null) { errors.push( `${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`, ); } checkStringArray(raw["definitions"], `${label}.definitions`, errors); checkStringArray(raw["examples"], `${label}.examples`, errors); const translations = raw["translations"]; if (!Array.isArray(translations) || translations.length === 0) { errors.push(`${label}.translations must be a non-empty array`); return; } translations.forEach((translation, i) => { checkTranslation(translation, `${label}.translations[${i}]`, ctx, errors); }); const wordsPerTarget = new Map(); for (const translation of translations) { if (!isRecord(translation)) continue; const target = translation["target_language"]; const word = translation["word"]; if (typeof target !== "string" || typeof word !== "string") continue; const words = wordsPerTarget.get(target) ?? []; words.push(word); wordsPerTarget.set(target, words); } for (const target of ctx.targetLanguages) { const words = wordsPerTarget.get(target) ?? []; if (words.length === 0) { errors.push( `${label}: missing translation for target language "${target}"`, ); } if (words.length > 2) { errors.push( `${label}: more than 2 translations for target language "${target}"`, ); } if (new Set(words).size !== words.length) { errors.push( `${label}: duplicate translation word for target language "${target}"`, ); } } }; /** * Lowers a sense's difficulty to that of its easiest translation when the model * tagged the sense higher. * * Why this is a repair and not a rejection: design-doc §5.1 filters sense * difficulty as a *ceiling* and translation difficulty as an *exact* target, so * a translation ranked below its own sense can never be served — the sense is * gated out at exactly the level where that translation would be the answer. * The row is dead data. The prompt already defines sense difficulty as the * easiest translation difficulty in the sense, which makes it a derived value * rather than an independent judgement, so it is recomputed here instead of * discarding an otherwise-good entry. * * The floor only ever lowers. Raising a sense to match its translations would * gate a concept out of levels it belongs in, and would collapse the * concept-vs-word distinction that the two difficulty columns exist to express * (design-doc §4). */ const applySenseDifficultyFloor = ( entry: GeminiWordEntry, normalizations: string[], ): GeminiWordEntry => ({ ...entry, senses: entry.senses.map((sense) => { const easiest = sense.translations.reduce( (lowest, translation) => difficultyRank(translation.difficulty) < difficultyRank(lowest) ? translation.difficulty : lowest, sense.difficulty, ); if (easiest === sense.difficulty) return sense; normalizations.push( `senses[${sense.sense_index}]: difficulty "${sense.difficulty}" → "${easiest}" (floored to easiest translation)`, ); return { ...sense, difficulty: easiest }; }), }); export const validateEntry = ( raw: unknown, ctx: ValidationContext, ): ValidationResult => { const errors: string[] = []; if (!isRecord(raw)) { return { status: "invalid", errors: ["entry must be an object"] }; } const headword = raw["headword"]; if (!isNonEmptyString(headword)) { errors.push("headword must be a non-empty string"); } else if (!ctx.inputWords.has(headword)) { errors.push(`headword "${headword}" is not in the input batch`); } if (raw["language"] !== ctx.sourceLanguage) { errors.push(`language must be "${ctx.sourceLanguage}"`); } if (raw["pos"] !== ctx.pos) { errors.push(`pos must be "${ctx.pos}"`); } const senses = raw["senses"]; if (!Array.isArray(senses)) { errors.push("senses must be an array"); } else if (senses.length === 0) { if (errors.length === 0 && isNonEmptyString(headword)) { return { status: "empty", headword }; } } else { if (senses.length > 3) { errors.push("senses must contain at most 3 entries"); } senses.forEach((sense, i) => checkSense(sense, i, ctx, errors)); } if (errors.length > 0) { return { status: "invalid", errors }; } const normalizations: string[] = []; return { status: "valid", entry: applySenseDifficultyFloor(raw as GeminiWordEntry, normalizations), normalizations, }; };