lila/data-pipeline/validate.ts
lila 5ee594334c flooring sense difficulty instead of rejecting the entry
Every rejection in the German run was the same rule: a translation ranked
below its own sense. The model tags a sense "medium" while correctly
tagging some translations "easy" — the translations are right and the
derived sense label is wrong, but the whole entry was discarded.

The prompt defines sense difficulty as the easiest translation difficulty
in that sense, so it is a derived value rather than an independent
judgement. validate.ts now recomputes it via applySenseDifficultyFloor.

The floor only ever lowers. Raising a sense to match its translations
would gate a concept out of levels it belongs in and collapse the
concept-vs-word distinction the two difficulty columns exist to express
(design-doc section 4).

- validate.ts: drop the cross-field rejection, add the floor; the valid
  result now carries "normalizations" so repairs are reported, not silent
- pipeline.ts: count and print normalizations per batch and in the summary
- replay.ts: new, re-validates responses/ with the current rules and no
  API calls; --write stages recovered entries, --langs and --verbose
- tests: six cases covering the floor, replacing the old rejection test

Replaying all 38 saved responses took the reject rate from 20 entries to
zero. staging.db now holds 746 words / 774 senses / 3,436 translations
with no sense ranked above its easiest translation.

Docs also record the API quota ceiling found today: the free tier allows
about 20 requests/day, not the 1,000 previously assumed.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-20 12:34:37 +02:00

275 lines
8.6 KiB
TypeScript

import {
DIFFICULTY_LEVELS,
NOUN_GENDERS,
type DifficultyLevel,
type NounGender,
type SupportedLanguageCode,
type SupportedPos,
} from "@lila/shared";
export type GeminiTranslation = {
target_language: SupportedLanguageCode;
word: string;
gender: NounGender | null;
difficulty: DifficultyLevel;
};
export type GeminiSense = {
sense_index: number;
difficulty: DifficultyLevel;
definitions: string[];
examples: string[];
translations: GeminiTranslation[];
};
export type GeminiWordEntry = {
headword: string;
language: SupportedLanguageCode;
pos: SupportedPos;
senses: GeminiSense[];
};
export type ValidationContext = {
sourceLanguage: SupportedLanguageCode;
pos: SupportedPos;
targetLanguages: readonly SupportedLanguageCode[];
inputWords: ReadonlySet<string>;
};
/**
* "empty" is the contract's way of saying "not a valid word of this POS"
* (senses: []) — the word is skipped, not rejected.
*
* A "valid" result carries the *normalized* entry, which may differ from the
* input — see `applySenseDifficultyFloor`. `normalizations` is a human-readable
* record of any repair applied, so the pipeline can report how often the model
* needed correcting instead of silently papering over it.
*/
export type ValidationResult =
| { status: "valid"; entry: GeminiWordEntry; normalizations: string[] }
| { status: "empty"; headword: string }
| { status: "invalid"; errors: string[] };
const isRecord = (value: unknown): value is Record<string, unknown> =>
typeof value === "object" && value !== null && !Array.isArray(value);
const isNonEmptyString = (value: unknown): value is string =>
typeof value === "string" && value.trim() !== "";
const isDifficulty = (value: unknown): value is DifficultyLevel =>
(DIFFICULTY_LEVELS as readonly unknown[]).includes(value);
const difficultyRank = (level: DifficultyLevel): number =>
DIFFICULTY_LEVELS.indexOf(level);
const validGendersFor = (
target: SupportedLanguageCode,
): readonly (NounGender | null)[] => {
if (target === "en") return [null];
if (target === "de") return NOUN_GENDERS;
return ["masculine", "feminine"];
};
const checkStringArray = (
value: unknown,
label: string,
errors: string[],
): void => {
if (!Array.isArray(value) || value.length === 0) {
errors.push(`${label} must be a non-empty array`);
return;
}
if (!value.every(isNonEmptyString)) {
errors.push(`${label} must contain only non-empty strings`);
}
};
const checkTranslation = (
raw: unknown,
label: string,
ctx: ValidationContext,
errors: string[],
): void => {
if (!isRecord(raw)) {
errors.push(`${label} must be an object`);
return;
}
const target = raw["target_language"];
if (!(ctx.targetLanguages as readonly unknown[]).includes(target)) {
errors.push(
`${label}: target_language must be one of ${ctx.targetLanguages.join(", ")}`,
);
}
if (!isNonEmptyString(raw["word"])) {
errors.push(`${label}: word must be a non-empty string`);
}
// A translation ranked below its sense is not an error — the sense is the
// derived value and gets floored to match. See applySenseDifficultyFloor.
if (!isDifficulty(raw["difficulty"])) {
errors.push(
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
);
}
if (
typeof target === "string" &&
(ctx.targetLanguages as readonly string[]).includes(target)
) {
const gender = raw["gender"];
const allowed = validGendersFor(target as SupportedLanguageCode);
if (!(allowed as readonly unknown[]).includes(gender)) {
errors.push(
`${label}: gender must be ${allowed.map((g) => g ?? "null").join(" or ")} for target "${target}"`,
);
}
}
};
const checkSense = (
raw: unknown,
index: number,
ctx: ValidationContext,
errors: string[],
): void => {
const label = `senses[${index}]`;
if (!isRecord(raw)) {
errors.push(`${label} must be an object`);
return;
}
if (raw["sense_index"] !== index) {
errors.push(`${label}: sense_index must be ${index} (sequential from 0)`);
}
const senseDifficulty = isDifficulty(raw["difficulty"])
? raw["difficulty"]
: null;
if (senseDifficulty === null) {
errors.push(
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
);
}
checkStringArray(raw["definitions"], `${label}.definitions`, errors);
checkStringArray(raw["examples"], `${label}.examples`, errors);
const translations = raw["translations"];
if (!Array.isArray(translations) || translations.length === 0) {
errors.push(`${label}.translations must be a non-empty array`);
return;
}
translations.forEach((translation, i) => {
checkTranslation(translation, `${label}.translations[${i}]`, ctx, errors);
});
const wordsPerTarget = new Map<string, string[]>();
for (const translation of translations) {
if (!isRecord(translation)) continue;
const target = translation["target_language"];
const word = translation["word"];
if (typeof target !== "string" || typeof word !== "string") continue;
const words = wordsPerTarget.get(target) ?? [];
words.push(word);
wordsPerTarget.set(target, words);
}
for (const target of ctx.targetLanguages) {
const words = wordsPerTarget.get(target) ?? [];
if (words.length === 0) {
errors.push(
`${label}: missing translation for target language "${target}"`,
);
}
if (words.length > 2) {
errors.push(
`${label}: more than 2 translations for target language "${target}"`,
);
}
if (new Set(words).size !== words.length) {
errors.push(
`${label}: duplicate translation word for target language "${target}"`,
);
}
}
};
/**
* Lowers a sense's difficulty to that of its easiest translation when the model
* tagged the sense higher.
*
* Why this is a repair and not a rejection: design-doc §5.1 filters sense
* difficulty as a *ceiling* and translation difficulty as an *exact* target, so
* a translation ranked below its own sense can never be served — the sense is
* gated out at exactly the level where that translation would be the answer.
* The row is dead data. The prompt already defines sense difficulty as the
* easiest translation difficulty in the sense, which makes it a derived value
* rather than an independent judgement, so it is recomputed here instead of
* discarding an otherwise-good entry.
*
* The floor only ever lowers. Raising a sense to match its translations would
* gate a concept out of levels it belongs in, and would collapse the
* concept-vs-word distinction that the two difficulty columns exist to express
* (design-doc §4).
*/
const applySenseDifficultyFloor = (
entry: GeminiWordEntry,
normalizations: string[],
): GeminiWordEntry => ({
...entry,
senses: entry.senses.map((sense) => {
const easiest = sense.translations.reduce<DifficultyLevel>(
(lowest, translation) =>
difficultyRank(translation.difficulty) < difficultyRank(lowest)
? translation.difficulty
: lowest,
sense.difficulty,
);
if (easiest === sense.difficulty) return sense;
normalizations.push(
`senses[${sense.sense_index}]: difficulty "${sense.difficulty}" → "${easiest}" (floored to easiest translation)`,
);
return { ...sense, difficulty: easiest };
}),
});
export const validateEntry = (
raw: unknown,
ctx: ValidationContext,
): ValidationResult => {
const errors: string[] = [];
if (!isRecord(raw)) {
return { status: "invalid", errors: ["entry must be an object"] };
}
const headword = raw["headword"];
if (!isNonEmptyString(headword)) {
errors.push("headword must be a non-empty string");
} else if (!ctx.inputWords.has(headword)) {
errors.push(`headword "${headword}" is not in the input batch`);
}
if (raw["language"] !== ctx.sourceLanguage) {
errors.push(`language must be "${ctx.sourceLanguage}"`);
}
if (raw["pos"] !== ctx.pos) {
errors.push(`pos must be "${ctx.pos}"`);
}
const senses = raw["senses"];
if (!Array.isArray(senses)) {
errors.push("senses must be an array");
} else if (senses.length === 0) {
if (errors.length === 0 && isNonEmptyString(headword)) {
return { status: "empty", headword };
}
} else {
if (senses.length > 3) {
errors.push("senses must contain at most 3 entries");
}
senses.forEach((sense, i) => checkSense(sense, i, ctx, errors));
}
if (errors.length > 0) {
return { status: "invalid", errors };
}
const normalizations: string[] = [];
return {
status: "valid",
entry: applySenseDifficultyFloor(raw as GeminiWordEntry, normalizations),
normalizations,
};
};