flooring sense difficulty instead of rejecting the entry

Every rejection in the German run was the same rule: a translation ranked
below its own sense. The model tags a sense "medium" while correctly
tagging some translations "easy" — the translations are right and the
derived sense label is wrong, but the whole entry was discarded.

The prompt defines sense difficulty as the easiest translation difficulty
in that sense, so it is a derived value rather than an independent
judgement. validate.ts now recomputes it via applySenseDifficultyFloor.

The floor only ever lowers. Raising a sense to match its translations
would gate a concept out of levels it belongs in and collapse the
concept-vs-word distinction the two difficulty columns exist to express
(design-doc section 4).

- validate.ts: drop the cross-field rejection, add the floor; the valid
  result now carries "normalizations" so repairs are reported, not silent
- pipeline.ts: count and print normalizations per batch and in the summary
- replay.ts: new, re-validates responses/ with the current rules and no
  API calls; --write stages recovered entries, --langs and --verbose
- tests: six cases covering the floor, replacing the old rejection test

Replaying all 38 saved responses took the reject rate from 20 entries to
zero. staging.db now holds 746 words / 774 senses / 3,436 translations
with no sense ranked above its easiest translation.

Docs also record the API quota ceiling found today: the free tier allows
about 20 requests/day, not the 1,000 previously assumed.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
lila 2026-08-20 12:34:37 +02:00
parent da9cdbfa1b
commit 5ee594334c
7 changed files with 454 additions and 50 deletions

View file

@ -6,7 +6,8 @@
"scripts": {
"test": "vitest run",
"test:watch": "vitest",
"pipeline:run": "tsx --env-file .env pipeline.ts"
"pipeline:run": "tsx --env-file .env pipeline.ts",
"pipeline:replay": "tsx replay.ts"
},
"dependencies": {
"@lila/shared": "workspace:*",

View file

@ -46,6 +46,7 @@ type ListStats = {
staged: number;
skipped: number;
rejected: number;
normalized: number;
failedBatches: number;
};
@ -217,6 +218,7 @@ const main = async (): Promise<void> => {
staged: 0,
skipped: 0,
rejected: 0,
normalized: 0,
failedBatches: 0,
};
allStats.push(stats);
@ -276,6 +278,12 @@ const main = async (): Promise<void> => {
stageEntry(db, result.entry);
covered.add(result.entry.headword);
stats.staged++;
if (result.normalizations.length > 0) {
stats.normalized++;
for (const note of result.normalizations) {
console.log(` normalized "${result.entry.headword}" ${note}`);
}
}
} else if (result.status === "empty") {
covered.add(result.headword);
stats.skipped++;
@ -311,7 +319,7 @@ const main = async (): Promise<void> => {
console.log("\n— summary —");
for (const stats of allStats) {
console.log(
`${stats.list.sourceLanguage}/${stats.list.pos}: ${stats.staged} staged, ${stats.skipped} skipped, ${stats.rejected} rejected, ${stats.failedBatches} failed batch(es), ${stats.pending - stats.staged - stats.skipped - stats.rejected} still pending`,
`${stats.list.sourceLanguage}/${stats.list.pos}: ${stats.staged} staged (${stats.normalized} normalized), ${stats.skipped} skipped, ${stats.rejected} rejected, ${stats.failedBatches} failed batch(es), ${stats.pending - stats.staged - stats.skipped - stats.rejected} still pending`,
);
}
const totals = countStagedRows(db);

226
data-pipeline/replay.ts Normal file
View file

@ -0,0 +1,226 @@
/**
* Re-validates saved Gemini responses from `responses/` without calling the API.
*
* Every batch response is written to disk before it is parsed, so a change to
* the validation rules can be applied retroactively to everything already
* generated. Use this after editing `validate.ts` to see what the change would
* do, and to recover entries that the old rules rejected.
*
* Reports by default; pass --write to stage recovered entries into staging.db.
* Staging is idempotent, so re-running is safe.
*/
import { readdirSync, readFileSync } from "node:fs";
import path from "node:path";
import { parseArgs } from "node:util";
import {
SUPPORTED_LANGUAGE_CODES,
SUPPORTED_POS,
type SupportedLanguageCode,
type SupportedPos,
} from "@lila/shared";
import { parseEntries } from "./gemini.js";
import { openStaging, stageEntry, countStagedRows } from "./staging.js";
import { validateEntry, type ValidationContext } from "./validate.js";
const ROOT = import.meta.dirname;
const RESPONSES_DIR = path.join(ROOT, "responses");
const STAGING_DB_PATH = path.join(ROOT, "db", "staging.db");
const STAGING_SCHEMA_PATH = path.join(ROOT, "db", "schema.sql");
type SavedResponse = {
sourceLanguage: SupportedLanguageCode;
pos: SupportedPos;
targetLanguages: SupportedLanguageCode[];
words: string[];
rawText: string;
};
type Tally = {
files: number;
unreadable: number;
entries: number;
valid: number;
normalized: number;
empty: number;
invalid: number;
staged: number;
alreadyStaged: number;
};
const isRecord = (value: unknown): value is Record<string, unknown> =>
typeof value === "object" && value !== null && !Array.isArray(value);
const isStringArray = (value: unknown): value is string[] =>
Array.isArray(value) && value.every((item) => typeof item === "string");
/** Saved files are trusted-but-verified: they are ours, but they are on disk. */
const parseSavedResponse = (raw: unknown): SavedResponse | null => {
if (!isRecord(raw)) return null;
const { sourceLanguage, pos, targetLanguages, words, rawText } = raw;
if (
typeof sourceLanguage !== "string" ||
!(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(sourceLanguage) ||
typeof pos !== "string" ||
!(SUPPORTED_POS as readonly string[]).includes(pos) ||
!isStringArray(targetLanguages) ||
!isStringArray(words) ||
typeof rawText !== "string"
) {
return null;
}
return {
sourceLanguage: sourceLanguage as SupportedLanguageCode,
pos: pos as SupportedPos,
targetLanguages: targetLanguages as SupportedLanguageCode[],
words,
rawText,
};
};
const main = (): void => {
const { values } = parseArgs({
args:
process.argv[2] === "--" ? process.argv.slice(3) : process.argv.slice(2),
options: {
write: { type: "boolean", default: false },
langs: { type: "string" },
verbose: { type: "boolean", default: false },
},
});
const langFilter =
values.langs === undefined
? null
: new Set(values.langs.split(",").map((code) => code.trim()));
let files: string[];
try {
files = readdirSync(RESPONSES_DIR)
.filter((name) => name.endsWith(".json"))
.sort();
} catch {
console.error(`no responses directory at ${RESPONSES_DIR}`);
process.exitCode = 1;
return;
}
if (files.length === 0) {
console.log("no saved responses to replay");
return;
}
const db = values.write
? openStaging(STAGING_DB_PATH, STAGING_SCHEMA_PATH)
: null;
const tally: Tally = {
files: 0,
unreadable: 0,
entries: 0,
valid: 0,
normalized: 0,
empty: 0,
invalid: 0,
staged: 0,
alreadyStaged: 0,
};
const recovered: string[] = [];
const stillInvalid = new Map<string, string[]>();
for (const name of files) {
let saved: SavedResponse | null;
try {
saved = parseSavedResponse(
JSON.parse(readFileSync(path.join(RESPONSES_DIR, name), "utf-8")),
);
} catch {
saved = null;
}
if (saved === null) {
tally.unreadable++;
console.warn(` skipping unreadable response file: ${name}`);
continue;
}
if (langFilter !== null && !langFilter.has(saved.sourceLanguage)) continue;
tally.files++;
let entries: unknown[];
try {
entries = parseEntries(saved.rawText);
} catch (error) {
tally.unreadable++;
console.warn(
` skipping ${name}: ${error instanceof Error ? error.message : String(error)}`,
);
continue;
}
const ctx: ValidationContext = {
sourceLanguage: saved.sourceLanguage,
pos: saved.pos,
targetLanguages: saved.targetLanguages,
inputWords: new Set(saved.words),
};
for (const entry of entries) {
tally.entries++;
const result = validateEntry(entry, ctx);
if (result.status === "valid") {
tally.valid++;
if (result.normalizations.length > 0) {
tally.normalized++;
recovered.push(
`${saved.sourceLanguage}/${result.entry.headword}: ${result.normalizations.join("; ")}`,
);
}
if (db !== null) {
const outcome = stageEntry(db, result.entry);
if (outcome === "staged") tally.staged++;
else tally.alreadyStaged++;
}
} else if (result.status === "empty") {
tally.empty++;
} else {
tally.invalid++;
const key = result.errors.join(" | ");
stillInvalid.set(key, [...(stillInvalid.get(key) ?? []), name]);
}
}
}
if (values.verbose && recovered.length > 0) {
console.log("\n— normalized entries —");
for (const line of recovered) console.log(` ${line}`);
}
if (stillInvalid.size > 0) {
console.log("\n— still invalid —");
for (const [errors, sources] of stillInvalid) {
console.log(` (${sources.length}×) ${errors}`);
}
}
console.log("\n— replay summary —");
console.log(` response files replayed: ${tally.files}`);
if (tally.unreadable > 0)
console.log(` unreadable files: ${tally.unreadable}`);
console.log(` entries seen: ${tally.entries}`);
console.log(` valid: ${tally.valid}`);
console.log(` of which normalized: ${tally.normalized}`);
console.log(` empty (skipped): ${tally.empty}`);
console.log(` invalid: ${tally.invalid}`);
if (db !== null) {
console.log(` newly staged: ${tally.staged}`);
console.log(` already staged: ${tally.alreadyStaged}`);
const totals = countStagedRows(db);
console.log(
` staging.db totals: ${totals.words} words, ${totals.senses} senses, ${totals.translations} translations`,
);
db.close();
} else {
console.log("\n (report only — pass --write to stage recovered entries)");
}
};
main();

View file

@ -1,5 +1,9 @@
import { describe, it, expect } from "vitest";
import { validateEntry, type ValidationContext } from "../validate.js";
import {
validateEntry,
type ValidationContext,
type ValidationResult,
} from "../validate.js";
const ctx: ValidationContext = {
sourceLanguage: "es",
@ -253,21 +257,6 @@ describe("validateEntry", () => {
);
});
it("rejects a translation difficulty below the sense difficulty", () => {
const easyTranslations = translations();
expect(
errorsOf(
entry({
senses: [
sense({ difficulty: "medium", translations: easyTranslations }),
],
}),
),
).toContainEqual(
expect.stringContaining('is lower than sense difficulty "medium"'),
);
});
it("allows a translation difficulty above the sense difficulty", () => {
const harder = translations();
harder[2] = {
@ -282,4 +271,97 @@ describe("validateEntry", () => {
);
expect(result.status).toBe("valid");
});
describe("sense difficulty floor", () => {
const senseDifficultyOf = (result: ValidationResult): string =>
result.status === "valid"
? (result.entry.senses[0]?.difficulty ?? "")
: "";
it("floors a sense to its easiest translation instead of rejecting", () => {
// The real "Ellbogen" case: sense tagged medium, but some translations
// are correctly easy. The entry is good; only the derived label is wrong.
const mixed = translations();
mixed[1] = {
target_language: "it",
word: "gomito",
gender: "masculine",
difficulty: "medium",
};
const result = validateEntry(
entry({
senses: [sense({ difficulty: "medium", translations: mixed })],
}),
ctx,
);
expect(result.status).toBe("valid");
expect(senseDifficultyOf(result)).toBe("easy");
});
it("reports what it changed", () => {
const result = validateEntry(
entry({ senses: [sense({ difficulty: "hard" })] }),
ctx,
);
expect(result.status).toBe("valid");
if (result.status !== "valid") return;
expect(result.normalizations).toHaveLength(1);
expect(result.normalizations[0]).toContain('"hard" → "easy"');
});
it("leaves an already-consistent sense untouched", () => {
const result = validateEntry(entry(), ctx);
expect(result.status).toBe("valid");
if (result.status !== "valid") return;
expect(result.normalizations).toEqual([]);
expect(senseDifficultyOf(result)).toBe("easy");
});
it("never raises a sense above its own label", () => {
// All translations medium, sense easy: the concept stays easy, because
// sense difficulty gates the meaning, not the word (design-doc §4).
const allMedium = translations().map((t) => ({
...t,
difficulty: "medium",
}));
const result = validateEntry(
entry({ senses: [sense({ translations: allMedium })] }),
ctx,
);
expect(result.status).toBe("valid");
if (result.status !== "valid") return;
expect(result.normalizations).toEqual([]);
expect(senseDifficultyOf(result)).toBe("easy");
});
it("floors each sense independently", () => {
const easySense = sense({ sense_index: 0, difficulty: "medium" });
const hardSense = sense({
sense_index: 1,
difficulty: "hard",
translations: translations().map((t) => ({
...t,
difficulty: "medium",
})),
});
const result = validateEntry(
entry({ senses: [easySense, hardSense] }),
ctx,
);
expect(result.status).toBe("valid");
if (result.status !== "valid") return;
expect(result.entry.senses.map((s) => s.difficulty)).toEqual([
"easy",
"medium",
]);
expect(result.normalizations).toHaveLength(2);
});
it("does not mutate the input entry", () => {
const input = entry({ senses: [sense({ difficulty: "medium" })] });
validateEntry(input, ctx);
const senses = input["senses"] as Record<string, unknown>[];
expect(senses[0]?.["difficulty"]).toBe("medium");
});
});
});

View file

@ -39,9 +39,14 @@ export type ValidationContext = {
/**
* "empty" is the contract's way of saying "not a valid word of this POS"
* (senses: []) — the word is skipped, not rejected.
*
* A "valid" result carries the *normalized* entry, which may differ from the
* input — see `applySenseDifficultyFloor`. `normalizations` is a human-readable
* record of any repair applied, so the pipeline can report how often the model
* needed correcting instead of silently papering over it.
*/
export type ValidationResult =
| { status: "valid"; entry: GeminiWordEntry }
| { status: "valid"; entry: GeminiWordEntry; normalizations: string[] }
| { status: "empty"; headword: string }
| { status: "invalid"; errors: string[] };
@ -82,7 +87,6 @@ const checkStringArray = (
const checkTranslation = (
raw: unknown,
label: string,
senseDifficulty: DifficultyLevel | null,
ctx: ValidationContext,
errors: string[],
): void => {
@ -99,18 +103,12 @@ const checkTranslation = (
if (!isNonEmptyString(raw["word"])) {
errors.push(`${label}: word must be a non-empty string`);
}
const difficulty = raw["difficulty"];
if (!isDifficulty(difficulty)) {
// A translation ranked below its sense is not an error — the sense is the
// derived value and gets floored to match. See applySenseDifficultyFloor.
if (!isDifficulty(raw["difficulty"])) {
errors.push(
`${label}: difficulty must be one of ${DIFFICULTY_LEVELS.join(", ")}`,
);
} else if (
senseDifficulty !== null &&
difficultyRank(difficulty) < difficultyRank(senseDifficulty)
) {
errors.push(
`${label}: difficulty "${difficulty}" is lower than sense difficulty "${senseDifficulty}"`,
);
}
if (
typeof target === "string" &&
@ -157,13 +155,7 @@ const checkSense = (
return;
}
translations.forEach((translation, i) => {
checkTranslation(
translation,
`${label}.translations[${i}]`,
senseDifficulty,
ctx,
errors,
);
checkTranslation(translation, `${label}.translations[${i}]`, ctx, errors);
});
const wordsPerTarget = new Map<string, string[]>();
@ -196,6 +188,45 @@ const checkSense = (
}
};
/**
* Lowers a sense's difficulty to that of its easiest translation when the model
* tagged the sense higher.
*
* Why this is a repair and not a rejection: design-doc §5.1 filters sense
* difficulty as a *ceiling* and translation difficulty as an *exact* target, so
* a translation ranked below its own sense can never be served — the sense is
* gated out at exactly the level where that translation would be the answer.
* The row is dead data. The prompt already defines sense difficulty as the
* easiest translation difficulty in the sense, which makes it a derived value
* rather than an independent judgement, so it is recomputed here instead of
* discarding an otherwise-good entry.
*
* The floor only ever lowers. Raising a sense to match its translations would
* gate a concept out of levels it belongs in, and would collapse the
* concept-vs-word distinction that the two difficulty columns exist to express
* (design-doc §4).
*/
const applySenseDifficultyFloor = (
entry: GeminiWordEntry,
normalizations: string[],
): GeminiWordEntry => ({
...entry,
senses: entry.senses.map((sense) => {
const easiest = sense.translations.reduce<DifficultyLevel>(
(lowest, translation) =>
difficultyRank(translation.difficulty) < difficultyRank(lowest)
? translation.difficulty
: lowest,
sense.difficulty,
);
if (easiest === sense.difficulty) return sense;
normalizations.push(
`senses[${sense.sense_index}]: difficulty "${sense.difficulty}" → "${easiest}" (floored to easiest translation)`,
);
return { ...sense, difficulty: easiest };
}),
});
export const validateEntry = (
raw: unknown,
ctx: ValidationContext,
@ -235,5 +266,10 @@ export const validateEntry = (
if (errors.length > 0) {
return { status: "invalid", errors };
}
return { status: "valid", entry: raw as GeminiWordEntry };
const normalizations: string[] = [];
return {
status: "valid",
entry: applySenseDifficultyFloor(raw as GeminiWordEntry, normalizations),
normalizations,
};
};