Prompt is now a template (fixes the hardcoded en/es leftovers in rules 2, 3, 15, 16, 26, 31). pipeline.ts replaces the pseudocode: wordlist normalization, skip-already-staged idempotency, batches of 20 against gemini-3.6-flash with responseSchema, raw responses persisted per batch, per-entry validation with rejection log, one transaction per word into db/staging.db. Flags: --langs --pos --max-batches --delay-ms --dry-run. Smoke run: 40/40 words staged (de+es, one batch each), 0 rejections. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
56 lines
1.7 KiB
TypeScript
56 lines
1.7 KiB
TypeScript
import { readdirSync, readFileSync } from "node:fs";
|
|
import path from "node:path";
|
|
import {
|
|
SUPPORTED_LANGUAGE_CODES,
|
|
SUPPORTED_POS,
|
|
type SupportedLanguageCode,
|
|
type SupportedPos,
|
|
} from "@lila/shared";
|
|
|
|
export type SourceList = {
|
|
sourceLanguage: SupportedLanguageCode;
|
|
pos: SupportedPos;
|
|
words: string[];
|
|
filePath: string;
|
|
};
|
|
|
|
const isLanguageCode = (value: string): value is SupportedLanguageCode =>
|
|
(SUPPORTED_LANGUAGE_CODES as readonly string[]).includes(value);
|
|
|
|
const isPos = (value: string): value is SupportedPos =>
|
|
(SUPPORTED_POS as readonly string[]).includes(value);
|
|
|
|
export const normalizeWords = (lines: readonly string[]): string[] => {
|
|
const seen = new Set<string>();
|
|
const words: string[] = [];
|
|
for (const line of lines) {
|
|
const word = line.trim();
|
|
if (word === "" || seen.has(word)) continue;
|
|
seen.add(word);
|
|
words.push(word);
|
|
}
|
|
return words;
|
|
};
|
|
|
|
export const discoverSourceLists = (rootDir: string): SourceList[] => {
|
|
const lists: SourceList[] = [];
|
|
const languageDirs = readdirSync(rootDir, { withFileTypes: true })
|
|
.filter((entry) => entry.isDirectory() && isLanguageCode(entry.name))
|
|
.map((entry) => entry.name as SupportedLanguageCode)
|
|
.sort();
|
|
|
|
for (const language of languageDirs) {
|
|
const languageDir = path.join(rootDir, language);
|
|
const posFiles = readdirSync(languageDir, { withFileTypes: true })
|
|
.filter((entry) => entry.isFile() && isPos(entry.name))
|
|
.map((entry) => entry.name as SupportedPos)
|
|
.sort();
|
|
|
|
for (const pos of posFiles) {
|
|
const filePath = path.join(languageDir, pos);
|
|
const words = normalizeWords(readFileSync(filePath, "utf-8").split("\n"));
|
|
lists.push({ sourceLanguage: language, pos, words, filePath });
|
|
}
|
|
}
|
|
return lists;
|
|
};
|