implementing phase 3 pipeline: gemini structured output, validation, sqlite staging

Prompt is now a template (fixes the hardcoded en/es leftovers in rules
2, 3, 15, 16, 26, 31). pipeline.ts replaces the pseudocode: wordlist
normalization, skip-already-staged idempotency, batches of 20 against
gemini-3.6-flash with responseSchema, raw responses persisted per batch,
per-entry validation with rejection log, one transaction per word into
db/staging.db. Flags: --langs --pos --max-batches --delay-ms --dry-run.

Smoke run: 40/40 words staged (de+es, one batch each), 0 rejections.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
lila 2026-08-09 19:04:04 +02:00
parent 303bb9388c
commit 37c978e230
12 changed files with 1477 additions and 66 deletions

View file

@ -0,0 +1,56 @@
import { readdirSync, readFileSync } from "node:fs";
import path from "node:path";
import {
SUPPORTED_LANGUAGE_CODES,
SUPPORTED_POS,
type SupportedLanguageCode,
type SupportedPos,
} from "@lila/shared";
export type SourceList = {
sourceLanguage: SupportedLanguageCode;
pos: SupportedPos;
words: string[];
filePath: string;
};
const isLanguageCode = (value: string): value is SupportedLanguageCode =>
(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(value);
const isPos = (value: string): value is SupportedPos =>
(SUPPORTED_POS as readonly string[]).includes(value);
export const normalizeWords = (lines: readonly string[]): string[] => {
const seen = new Set<string>();
const words: string[] = [];
for (const line of lines) {
const word = line.trim();
if (word === "" || seen.has(word)) continue;
seen.add(word);
words.push(word);
}
return words;
};
export const discoverSourceLists = (rootDir: string): SourceList[] => {
const lists: SourceList[] = [];
const languageDirs = readdirSync(rootDir, { withFileTypes: true })
.filter((entry) => entry.isDirectory() && isLanguageCode(entry.name))
.map((entry) => entry.name as SupportedLanguageCode)
.sort();
for (const language of languageDirs) {
const languageDir = path.join(rootDir, language);
const posFiles = readdirSync(languageDir, { withFileTypes: true })
.filter((entry) => entry.isFile() && isPos(entry.name))
.map((entry) => entry.name as SupportedPos)
.sort();
for (const pos of posFiles) {
const filePath = path.join(languageDir, pos);
const words = normalizeWords(readFileSync(filePath, "utf-8").split("\n"));
lists.push({ sourceLanguage: language, pos, words, filePath });
}
}
return lists;
};