lila/repomix-output.xml
2026-07-06 17:12:25 +02:00

1987 lines
57 KiB
XML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

This file is a merged representation of the entire codebase, combined into a single document by Repomix.
<file_summary>
This section contains a summary of this file.
<purpose>
This file contains a packed representation of the entire repository's contents.
It is designed to be easily consumable by AI systems for analysis, code review,
or other automated processes.
</purpose>
<file_format>
The content is organized as follows:
1. This summary section
2. Repository information
3. Directory structure
4. Repository files (if enabled)
5. Multiple file entries, each consisting of:
- File path as an attribute
- Full contents of the file
</file_format>
<usage_guidelines>
- This file should be treated as read-only. Any changes should be made to the
original repository files, not this packed version.
- When processing this file, use the file path to distinguish
between different files in the repository.
- Be aware that this file may contain sensitive information. Handle it with
the same level of security as you would the original repository.
</usage_guidelines>
<notes>
- Some files may have been excluded based on .gitignore rules and Repomix's configuration
- Binary files are not included in this packed representation. Please refer to the Repository Structure section for a complete list of file paths, including binary files
- Files matching patterns in .gitignore are excluded
- Files matching default ignore patterns are excluded
- Files are sorted by Git change count (files with more changes are at the bottom)
</notes>
</file_summary>
<directory_structure>
config/
batch.ts
constants.ts
llm.ts
prompt.ts
providers.ts
source-data/
english/
nouns
utils/
llm-adapters/
factory.ts
gemini.ts
openai-compatible.ts
types.ts
check-if-json-exists.ts
check-llm-server.ts
cli.ts
create-base-json.ts
create-line-reader.ts
create-output-dirs.ts
delete-file.ts
enrich-word.ts
get-word-file-path.ts
merge-enriched-data.ts
pipeline-timer.ts
progress-tracker.ts
scanning-source-files.ts
verify-enriched-file.ts
write-json-file.ts
.env.example
.pipeline-config.json
package.json
pipeline.ts
tsconfig.json
vitest.config.ts
</directory_structure>
<files>
This section contains the contents of the repository's files.
<file path="config/batch.ts">
// Runtime-populated by pipeline.ts after CLI initialization
export const BATCH_CONFIG = { size: 4, maxRetries: 3 };
</file>
<file path="config/constants.ts">
export const LANG_MAP: Record<string, string> = {
english: "en",
italian: "it",
german: "de",
french: "fr",
spanish: "es",
};
export const POS_MAP: Record<string, string> = {
nouns: "noun",
verbs: "verb",
adverbs: "adverb",
adjectives: "adjective",
};
export const ALL_LANGUAGES = ["en", "de", "it", "es", "fr"];
</file>
<file path="config/llm.ts">
import type { OnlineProvider } from "./providers.js";
export type LlmProvider = "local" | OnlineProvider;
// Runtime-populated by pipeline.ts after CLI initialization
export const LLM_CONFIG = {
provider: "local" as LlmProvider,
url: "http://127.0.0.1:8080/v1/chat/completions",
model: undefined as string | undefined,
};
</file>
<file path="config/prompt.ts">
export function buildSystemPrompt(
sourceLanguage: string,
pos: string,
targetLanguages: string[],
): string {
return `You are a multilingual dictionary engine. Output ONLY a JSON object. No markdown, no explanations.
For each ${sourceLanguage} ${pos} provided, generate 1-2 distinct senses.
CEFR difficulty mapping:
- A1/A2 → easy
- B1/B2 → medium
- C1/C2 → hard
Each sense must have:
- sense: student-friendly definition, max 15 words
- example: natural sentence using the word
- difficulty_level: easy, medium, or hard
- translations: object with keys ${targetLanguages.join(", ")}; each value is an array of {word, gender} where gender MUST be masculine, feminine, or neuter. Use null ONLY if the language has no grammatical gender for that word.
Output format: JSON object where keys are the input words, values are arrays of sense objects.
Example for ["house"]:
{
"house": [
{
"sense": "A building for human habitation.",
"example": "They bought a house in the city.",
"difficulty_level": "easy",
"translations": {
"de": [{"word": "Haus", "gender": "neuter"}],
"it": [{"word": "casa", "gender": "feminine"}],
"es": [{"word": "casa", "gender": "feminine"}],
"fr": [{"word": "maison", "gender": "feminine"}]
}
}
]
}
`;
}
</file>
<file path="config/providers.ts">
export type ProviderMeta = {
name: string;
envVar: string;
url: string;
requiresKey: boolean;
models: string[];
};
export const ONLINE_PROVIDERS: Record<string, ProviderMeta> = {
gemini: {
name: "Gemini",
envVar: "GEMINI_API_KEY",
url: "https://generativelanguage.googleapis.com/v1beta",
requiresKey: true,
models: ["gemini-2.5-flash", "gemini-2.5-pro"],
},
deepseek: {
name: "DeepSeek",
envVar: "DEEPSEEK_API_KEY",
url: "https://api.deepseek.com/v1/chat/completions",
requiresKey: true,
models: ["deepseek-chat", "deepseek-reasoner"],
},
openrouter: {
name: "OpenRouter",
envVar: "OPENROUTER_API_KEY",
url: "https://openrouter.ai/api/v1/chat/completions",
requiresKey: true,
models: [
"openai/gpt-oss-120b:free",
"google/gemma-4-31b-it:free",
"qwen/qwen3-next-80b-a3b-instruct:free",
"meta-llama/llama-3.3-70b-instruct:free",
"anthropic/claude-sonnet-4",
"google/gemini-2.5-flash",
"deepseek/deepseek-chat-v3",
],
},
groq: {
name: "Groq",
envVar: "GROQ_API_KEY",
url: "https://api.groq.com/openai/v1/chat/completions",
requiresKey: true,
models: ["llama-3.3-70b-versatile", "gemma2-9b-it", "mixtral-8x7b-32768"],
},
} as const;
export const LOCAL_PROVIDER: ProviderMeta = {
name: "Local (llama.cpp / ollama / lm-studio)",
envVar: "",
url: "http://127.0.0.1:8080/v1/chat/completions",
requiresKey: false,
models: [],
};
export type OnlineProvider = keyof typeof ONLINE_PROVIDERS;
</file>
<file path="source-data/english/nouns">
house
time
water
year
people
day
way
man
woman
child
work
life
world
hand
eye
book
friend
school
city
family
</file>
<file path="utils/llm-adapters/factory.ts">
import { LLM_CONFIG } from "../../config/llm.js";
import { OpenAiCompatibleAdapter } from "./openai-compatible.js";
import { GeminiAdapter } from "./gemini.js";
import type { LlmAdapter } from "./types.js";
export function createAdapter(): LlmAdapter {
switch (LLM_CONFIG.provider) {
case "local":
return new OpenAiCompatibleAdapter(
LLM_CONFIG.url,
undefined,
LLM_CONFIG.model,
);
case "openrouter":
return new OpenAiCompatibleAdapter(
LLM_CONFIG.url,
process.env["OPENROUTER_API_KEY"],
LLM_CONFIG.model,
);
case "deepseek":
return new OpenAiCompatibleAdapter(
LLM_CONFIG.url,
process.env["DEEPSEEK_API_KEY"],
LLM_CONFIG.model,
);
case "groq":
return new OpenAiCompatibleAdapter(
LLM_CONFIG.url,
process.env["GROQ_API_KEY"],
LLM_CONFIG.model,
);
case "gemini": {
const apiKey = process.env["GEMINI_API_KEY"];
if (!apiKey) throw new Error("GEMINI_API_KEY env var not set");
if (!LLM_CONFIG.model)
throw new Error("LLM_CONFIG.model required for gemini");
return new GeminiAdapter(apiKey, LLM_CONFIG.model);
}
default:
throw new Error(`Unknown provider: ${LLM_CONFIG.provider}`);
}
}
</file>
<file path="utils/llm-adapters/gemini.ts">
import type { LlmAdapter } from "./types.js";
interface GeminiResponse {
candidates: Array<{ content: { parts: Array<{ text: string }> } }>;
usageMetadata: {
promptTokenCount: number;
candidatesTokenCount: number;
totalTokenCount: number;
};
}
export class GeminiAdapter implements LlmAdapter {
private apiKey: string;
private model: string;
constructor(apiKey: string, model: string) {
this.apiKey = apiKey;
this.model = model;
}
async call(
words: string[],
systemPrompt: string,
): Promise<{
content: string;
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
}> {
const url = `https://generativelanguage.googleapis.com/v1beta/models/${this.model}:generateContent?key=${this.apiKey}`;
const payload = {
systemInstruction: { parts: [{ text: systemPrompt }] },
contents: [
{ role: "user", parts: [{ text: "Words: " + JSON.stringify(words) }] },
],
generationConfig: {
temperature: 0.1,
topP: 0.9,
maxOutputTokens: Math.ceil(words.length * 250 * 1.2),
},
};
const startTime = Date.now();
const response = await fetch(url, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(payload),
});
const totalTimeMs = Date.now() - startTime;
if (!response.ok) {
throw new Error(`Gemini API responded with status: ${response.status}`);
}
const json = (await response.json()) as GeminiResponse;
const content = json.candidates[0]?.content?.parts[0]?.text;
if (!content) {
throw new Error("Gemini response content is empty");
}
const promptTokens = json.usageMetadata.promptTokenCount;
const completionTokens = json.usageMetadata.candidatesTokenCount;
return {
content,
promptTokens,
completionTokens,
totalTokens: json.usageMetadata.totalTokenCount,
promptTimeMs: null,
completionTimeMs: null,
totalTimeMs,
};
}
}
</file>
<file path="utils/llm-adapters/openai-compatible.ts">
import type { LlmAdapter } from "./types.js";
interface OpenAiResponse {
choices: Array<{ message: { content: string } }>;
usage: {
prompt_tokens: number;
completion_tokens: number;
total_tokens: number;
};
timings?: { prompt_ms: number; predicted_ms: number };
}
export class OpenAiCompatibleAdapter implements LlmAdapter {
private url: string;
private apiKey: string | undefined;
private model: string | undefined;
constructor(url: string, apiKey?: string, model?: string) {
this.url = url;
this.apiKey = apiKey;
this.model = model;
}
async call(
words: string[],
systemPrompt: string,
): Promise<{
content: string;
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
}> {
const payload: Record<string, unknown> = {
messages: [
{ role: "system", content: systemPrompt },
{ role: "user", content: JSON.stringify(words) },
],
temperature: 0.1,
top_p: 0.9,
max_tokens: Math.ceil(words.length * 250 * 1.2),
};
if (this.model) {
payload["model"] = this.model;
}
const headers: Record<string, string> = {
"Content-Type": "application/json",
};
if (this.apiKey) {
headers["Authorization"] = `Bearer ${this.apiKey}`;
}
const startTime = Date.now();
const response = await fetch(this.url, {
method: "POST",
headers,
body: JSON.stringify(payload),
});
const totalTimeMs = Date.now() - startTime;
if (!response.ok) {
throw new Error(`LLM server responded with status: ${response.status}`);
}
const json = (await response.json()) as OpenAiResponse;
const content = json.choices[0]?.message?.content;
if (!content) {
throw new Error("LLM response content is empty");
}
const promptTokens = json.usage.prompt_tokens;
const completionTokens = json.usage.completion_tokens;
return {
content,
promptTokens,
completionTokens,
totalTokens: json.usage.total_tokens,
promptTimeMs: json.timings?.prompt_ms ?? null,
completionTimeMs: json.timings?.predicted_ms ?? null,
totalTimeMs,
};
}
}
</file>
<file path="utils/llm-adapters/types.ts">
export interface LlmAdapter {
call(
words: string[],
systemPrompt: string,
): Promise<{
content: string;
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
}>;
}
</file>
<file path="utils/check-if-json-exists.ts">
import fs from "fs";
import path from "path";
/**
* Checks if a JSON file for the given word exists AND contains enriched data.
* Returns false for skeleton files (missing senses array).
*/
export function isWordProcessed(word: string, outputDir: string): boolean {
const targetFilePath = path.join(outputDir, `${word}.json`);
if (!fs.existsSync(targetFilePath)) {
return false;
}
try {
const content = fs.readFileSync(targetFilePath, "utf-8");
const data = JSON.parse(content) as Record<string, unknown>;
return Array.isArray(data["senses"]) && data["senses"].length > 0;
} catch (_error: unknown) {
// Corrupted file => treat as not processed
return false;
}
}
</file>
<file path="utils/check-llm-server.ts">
import { LLM_CONFIG } from "../config/llm.js";
/**
* Pings the local llama.cpp server to ensure it's up, running, and has a model loaded.
* If the server is offline or still loading, it terminates the pipeline gracefully.
* Skipped entirely when using a cloud provider.
*/
export async function checkLlmServer(
url = "http://127.0.0.1:8080/health",
): Promise<void> {
if (LLM_CONFIG.provider !== "local") {
console.log("🌐 Using cloud provider — skipping local health check.");
return;
}
try {
const response = await fetch(url);
// llama.cpp returns a 503 status if the server is up but the model weights are still loading
if (response.status === 503) {
throw new Error(
"Local AI engine is starting up, but the model is still loading into memory. " +
"Please wait a minute for the weights to load, then run the pipeline again.",
);
}
// Parse the JSON health response (expected: { status: "ok" })
const data = (await response.json()) as { status?: string };
if (response.ok && data.status === "ok") {
console.log("🟢 Local AI engine is connected and ready for inference!");
return;
}
// Catch-all for unexpected active server responses
throw new Error(
`Unknown response from local AI engine health check (Status: ${response.status}).`,
);
} catch (error: unknown) {
if (error instanceof Error && error.message.includes("Local AI engine")) {
throw error; // Re-throw our own errors
}
throw new Error(
`Could not connect to the local AI engine at ${url}. ` +
"Make sure your './llama-server' command is actively running in another terminal tab.",
{ cause: error },
);
}
}
</file>
<file path="utils/cli.ts">
import { createInterface } from "node:readline";
import { existsSync, readFileSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import {
ONLINE_PROVIDERS,
LOCAL_PROVIDER,
type OnlineProvider,
} from "../config/providers.js";
// ── Types ──────────────────────────────────────────────────────────────────
export interface PipelineConfig {
provider: "local" | OnlineProvider;
url: string;
model: string | undefined;
batchSize: number;
maxRetries: number;
}
interface SavedConfig {
provider: PipelineConfig["provider"];
model: string;
batchSize: number;
maxRetries: number;
}
// ── Helpers ────────────────────────────────────────────────────────────────
function getConfigPath(): string {
return join(import.meta.dirname, "..", ".pipeline-config.json");
}
function loadLastConfig(): SavedConfig | null {
const path = getConfigPath();
if (!existsSync(path)) return null;
try {
const raw = readFileSync(path, "utf-8");
return JSON.parse(raw) as SavedConfig;
} catch {
return null;
}
}
function saveConfig(config: SavedConfig): void {
writeFileSync(getConfigPath(), JSON.stringify(config, null, 2));
}
function ask(
rl: ReturnType<typeof createInterface>,
prompt: string,
): Promise<string> {
return new Promise((resolve) => {
rl.question(prompt, resolve);
});
}
function printLine(char = "─", length = 50): void {
console.log(char.repeat(length));
}
function formatProviderLabel(p: string): string {
const meta = p === "local" ? LOCAL_PROVIDER : ONLINE_PROVIDERS[p];
return meta ? meta.name : p;
}
// ── Validation ─────────────────────────────────────────────────────────────
function validateBatchSize(input: string): number {
const n = parseInt(input.trim(), 10);
if (Number.isNaN(n) || n < 1 || n > 20) {
throw new Error("Batch size must be an integer between 1 and 20");
}
return n;
}
function checkApiKey(provider: string): void {
const meta = ONLINE_PROVIDERS[provider];
if (!meta) return;
const key = process.env[meta.envVar];
if (!key) {
console.error(`\n ❌ Missing API key: ${meta.envVar} is not set.`);
console.error(` Export it before running the pipeline:`);
console.error(` export ${meta.envVar}=your_key_here\n`);
process.exit(1);
}
}
// ── Prompt flows ────────────────────────────────────────────────────────────
async function promptProviderType(
rl: ReturnType<typeof createInterface>,
): Promise<"local" | "online"> {
console.log("\nSelect provider type:");
console.log(" [1] Local (llama.cpp, ollama, lm-studio, etc.)");
console.log(" [2] Online API (Gemini, DeepSeek, OpenRouter, Groq)");
while (true) {
const choice = (await ask(rl, "Choice [1/2]: ")).trim();
if (choice === "1") return "local";
if (choice === "2") return "online";
console.log(" Invalid choice. Enter 1 or 2.");
}
}
async function promptOnlineProvider(
rl: ReturnType<typeof createInterface>,
): Promise<string> {
console.log("\nSelect online provider:");
const entries = Object.entries(ONLINE_PROVIDERS);
entries.forEach(([_key, meta], i) => {
const hasKey = process.env[meta.envVar] ? "✓" : "✗";
console.log(` [${i + 1}] ${meta.name} (${hasKey} ${meta.envVar})`);
});
while (true) {
const choice = (await ask(rl, `Choice [1-${entries.length}]: `)).trim();
const idx = parseInt(choice, 10) - 1;
if (idx >= 0 && idx < entries.length) {
const entry = entries[idx]!;
const provider = entry[0];
checkApiKey(provider);
return provider;
}
console.log(` Invalid choice. Enter 1-${entries.length}.`);
}
}
async function promptModel(
rl: ReturnType<typeof createInterface>,
provider: string,
): Promise<string> {
if (provider === "local") {
console.log("\nLocal provider selected.");
console.log(" Using: http://127.0.0.1:8080/v1/chat/completions");
const model = (
await ask(rl, "Model name (optional, press Enter to skip): ")
).trim();
return model || "local-model";
}
const meta = ONLINE_PROVIDERS[provider];
if (!meta) {
throw new Error(`Unknown provider: ${provider}`);
}
console.log(`\nSelect model for ${meta.name}:`);
meta.models.forEach((m, i) => console.log(` [${i + 1}] ${m}`));
console.log(` [${meta.models.length + 1}] Other (type manually)`);
while (true) {
const choice = (
await ask(rl, `Choice [1-${meta.models.length + 1}]: `)
).trim();
const idx = parseInt(choice, 10) - 1;
if (idx >= 0 && idx < meta.models.length) {
return meta.models[idx]!;
}
if (idx === meta.models.length) {
const custom = (await ask(rl, "Enter model name: ")).trim();
if (custom) return custom;
console.log(" Model name cannot be empty.");
continue;
}
console.log(` Invalid choice. Enter 1-${meta.models.length + 1}.`);
}
}
async function promptBatchSize(
rl: ReturnType<typeof createInterface>,
): Promise<number> {
console.log("\nBatch size: how many words to enrich per LLM call.");
console.log(" Recommended: 26 for complex languages, 48 for simple.");
while (true) {
const input = (await ask(rl, "Batch size [1-20, default 4]: ")).trim();
if (!input) return 4;
try {
return validateBatchSize(input);
} catch (err) {
console.log(` ${(err as Error).message}`);
}
}
}
async function promptConfirm(
rl: ReturnType<typeof createInterface>,
config: PipelineConfig,
): Promise<boolean> {
console.log("\n");
printLine();
console.log(" CONFIGURATION SUMMARY");
printLine();
console.log(` Provider: ${formatProviderLabel(config.provider)}`);
console.log(` URL: ${config.url}`);
console.log(` Model: ${config.model ?? "(none)"}`);
console.log(` Batch: ${config.batchSize} words/call`);
console.log(` Retries: ${config.maxRetries}`);
printLine();
const answer = (await ask(rl, "\nProceed with this configuration? [Y/n]: "))
.trim()
.toLowerCase();
return answer === "" || answer === "y" || answer === "yes";
}
// ── Main export ────────────────────────────────────────────────────────────
export async function runCli(): Promise<PipelineConfig> {
const rl = createInterface({ input: process.stdin, output: process.stdout });
try {
const lastConfig = loadLastConfig();
// ── Startup menu ─────────────────────────────────────────────────────────
console.log("\n");
printLine("═", 50);
console.log(" PIPELINE CONFIGURATION");
printLine("═", 50);
if (lastConfig) {
console.log("\nLast used configuration:");
console.log(` Provider: ${formatProviderLabel(lastConfig.provider)}`);
console.log(` Model: ${lastConfig.model}`);
console.log(` Batch: ${lastConfig.batchSize}`);
} else {
console.log("\nNo previous configuration found.");
}
console.log(
"\n[1] Use last config" +
(lastConfig ? "" : " (not available)") +
" [2] Configure new run",
);
let useLast = false;
if (lastConfig) {
while (true) {
const choice = (await ask(rl, "Choice [1/2]: ")).trim();
if (choice === "1") {
useLast = true;
break;
}
if (choice === "2") break;
console.log(" Invalid choice. Enter 1 or 2.");
}
} else {
// No last config, auto-select new run
console.log("Auto-selecting: Configure new run");
await ask(rl, "Press Enter to continue...");
}
// ── Build config ────────────────────────────────────────────────────────
let config: PipelineConfig;
if (useLast && lastConfig) {
// Re-validate API key before reusing
if (lastConfig.provider !== "local") {
checkApiKey(lastConfig.provider);
}
const meta =
lastConfig.provider === "local"
? LOCAL_PROVIDER
: ONLINE_PROVIDERS[lastConfig.provider];
config = {
provider: lastConfig.provider,
url: meta?.url ?? LOCAL_PROVIDER.url,
model: lastConfig.model,
batchSize: lastConfig.batchSize,
maxRetries: lastConfig.maxRetries,
};
} else {
// New run flow
const providerType = await promptProviderType(rl);
let provider: string;
let url: string;
if (providerType === "local") {
provider = "local";
url = LOCAL_PROVIDER.url;
} else {
provider = await promptOnlineProvider(rl);
url = ONLINE_PROVIDERS[provider]!.url;
}
const model = await promptModel(rl, provider);
const batchSize = await promptBatchSize(rl);
config = {
provider: provider,
url,
model: model || undefined,
batchSize,
maxRetries: 3,
};
// Confirm before saving
const confirmed = await promptConfirm(rl, config);
if (!confirmed) {
console.log("\n ❌ Configuration cancelled. Exiting.\n");
process.exit(0);
}
// Save for next time
saveConfig({
provider: config.provider,
model: config.model ?? "",
batchSize: config.batchSize,
maxRetries: config.maxRetries,
});
console.log("\n ✓ Configuration saved to .pipeline-config.json");
}
console.log("\n");
return config;
} finally {
rl.close();
}
}
</file>
<file path="utils/create-base-json.ts">
import fs from "fs";
import path from "path";
import { LANG_MAP, POS_MAP } from "../config/constants.js";
/**
* Creates the base JSON file with word, language, and pos.
* No logging — the orchestrator handles all console output.
*/
export function createBaseJson(
word: string,
outputDir: string,
rawLanguage: string,
rawPos: string,
): void {
const targetFilePath = path.join(outputDir, `${word}.json`);
const dbLanguage = LANG_MAP[rawLanguage] || rawLanguage;
const dbPos = POS_MAP[rawPos] || rawPos;
const initialData = { word, language: dbLanguage, pos: dbPos };
fs.writeFileSync(
targetFilePath,
JSON.stringify(initialData, null, 2),
"utf-8",
);
}
</file>
<file path="utils/create-line-reader.ts">
import fs from "fs";
import readline from "readline";
/**
* Creates a line-by-line reader stream for a given file path.
*/
export function createLineReader(sourcePath: string): readline.Interface {
const fileStream = fs.createReadStream(sourcePath, "utf-8");
return readline.createInterface({ input: fileStream, crlfDelay: Infinity });
}
</file>
<file path="utils/create-output-dirs.ts">
import fs from "fs";
import type { Wordlist } from "./scanning-source-files.js";
/**
* Takes a list of scanned datasets and creates their output folders if missing.
*/
export function ensureOutputFolders(wordlists: Wordlist[]): void {
for (const wordlist of wordlists) {
if (!fs.existsSync(wordlist.outputDir)) {
fs.mkdirSync(wordlist.outputDir, { recursive: true });
console.log(
`📁 Created target folder: worddata/${wordlist.language}/${wordlist.pos}`,
);
}
}
console.log(
"✅ All required output directories have been verified and created successfully.",
);
}
</file>
<file path="utils/delete-file.ts">
import fs from "fs";
/**
* Deletes a file if it exists. Silently ignores missing files.
*/
export function deleteFileIfExists(filePath: string): void {
if (fs.existsSync(filePath)) {
fs.unlinkSync(filePath);
}
}
</file>
<file path="utils/enrich-word.ts">
import { buildSystemPrompt } from "../config/prompt.js";
import { createAdapter } from "./llm-adapters/factory.js";
import { BATCH_CONFIG } from "../config/batch.js";
import type { Language, Pos, EnrichedSense } from "./merge-enriched-data.js";
import { LANG_MAP, POS_MAP, ALL_LANGUAGES } from "../config/constants.js";
interface LlmResponse {
content: string;
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
}
export interface EnrichmentResult {
results: Map<string, EnrichedSense[]>;
metrics: {
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
};
}
/**
* Calls the LLM with the enrichment prompt.
* Returns the response content and timing metrics.
*/
async function callLlm(
words: string[],
rawLanguage: string,
rawPos: string,
): Promise<LlmResponse> {
const adapter = createAdapter();
const sourceCode = LANG_MAP[rawLanguage] || rawLanguage;
const targetLanguages = ALL_LANGUAGES.filter((lang) => lang !== sourceCode);
const prompt = buildSystemPrompt(rawLanguage, rawPos, targetLanguages);
return adapter.call(words, prompt);
}
/**
* Strips markdown code blocks and extracts the JSON object from raw LLM output.
* Throws if no valid JSON object braces are found.
*/
function sanitizeLlmOutput(raw: string): string {
const cleaned = raw.replace(/```json\s*/g, "").replace(/```\s*$/g, "");
const start = cleaned.indexOf("{");
const end = cleaned.lastIndexOf("}");
if (start === -1 || end === -1) {
throw new Error("No JSON object found in LLM output");
}
return cleaned.slice(start, end + 1);
}
function validateSense(item: unknown, word: string, index: number): void {
if (typeof item !== "object" || item === null || Array.isArray(item)) {
throw new Error(`Sense ${index} for "${word}" is not an object`);
}
const sense = item as Record<string, unknown>;
if (typeof sense["sense"] !== "string" || !sense["sense"]) {
throw new Error(`Sense ${index} for "${word}": missing or invalid "sense"`);
}
if (typeof sense["example"] !== "string" || !sense["example"]) {
throw new Error(
`Sense ${index} for "${word}": missing or invalid "example"`,
);
}
if (
!["easy", "medium", "hard"].includes(sense["difficulty_level"] as string)
) {
throw new Error(`Sense ${index} for "${word}": invalid "difficulty_level"`);
}
if (
typeof sense["translations"] !== "object" ||
sense["translations"] === null
) {
throw new Error(`Sense ${index} for "${word}": missing "translations"`);
}
const trans = sense["translations"] as Record<string, unknown>;
for (const lang of ["de", "it", "es", "fr"]) {
if (!Array.isArray(trans[lang])) {
throw new Error(
`Sense ${index} for "${word}": missing or invalid "${lang}" translations`,
);
}
for (let j = 0; j < (trans[lang] as unknown[]).length; j++) {
const t = (trans[lang] as unknown[])[j] as Record<string, unknown>;
if (typeof t["word"] !== "string" || !t["word"]) {
throw new Error(
`Sense ${index} for "${word}": ${lang}[${j}] missing "word"`,
);
}
if (
!["masculine", "feminine", "neuter", null].includes(
t["gender"] as string | null,
)
) {
throw new Error(
`Sense ${index} for "${word}": ${lang}[${j}] invalid "gender"`,
);
}
}
}
}
/**
* Parses the LLM response string into a JavaScript object.
* Throws if the response is not valid JSON or not an object with expected keys.
*/
export function parseLlmResponse(
rawJson: string,
expectedWords: string[],
): Record<string, unknown> {
let parsed: unknown;
try {
const sanitized = sanitizeLlmOutput(rawJson);
parsed = JSON.parse(sanitized);
} catch (error: unknown) {
throw new Error(`Failed to parse LLM output as JSON: ${rawJson}`, {
cause: error,
});
}
if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) {
throw new Error("LLM output is not a JSON object");
}
const obj = parsed as Record<string, unknown>;
for (const word of expectedWords) {
if (!(word in obj)) {
throw new Error(`Missing key in LLM output: "${word}"`);
}
if (!Array.isArray(obj[word]) || (obj[word] as unknown[]).length === 0) {
throw new Error(`LLM output for "${word}" is not a non-empty array`);
}
// Validate each sense in the array
const senses = obj[word] as unknown[];
for (let i = 0; i < senses.length; i++) {
validateSense(senses[i], word, i);
}
}
return obj;
}
/**
* Takes parsed LLM output and builds final enriched objects with composite IDs.
*/
export function buildEnrichedData(
parsed: Record<string, unknown>,
rawLanguage: string,
rawPos: string,
): Map<string, EnrichedSense[]> {
const language = (LANG_MAP[rawLanguage] || rawLanguage) as Language;
const pos = (POS_MAP[rawPos] || rawPos) as Pos;
const results = new Map<string, EnrichedSense[]>();
for (const [word, sensesArray] of Object.entries(parsed)) {
const senses = (sensesArray as unknown[]).map((item, index) => {
const sense = item as Omit<
EnrichedSense,
"id" | "word" | "language" | "pos"
>;
return {
id: `${word}:${language}:${pos}:${index}`,
word,
language,
pos,
...sense,
} as EnrichedSense;
});
results.set(word, senses);
}
return results;
}
/**
* Enriches a batch of words by calling the LLM, parsing the response, and building final data.
*/
export async function enrichWord(
words: string[],
rawLanguage: string,
rawPos: string,
): Promise<EnrichmentResult> {
const llmResponse = await callLlm(words, rawLanguage, rawPos);
const parsed = parseLlmResponse(llmResponse.content, words);
const results = buildEnrichedData(parsed, rawLanguage, rawPos);
return {
results,
metrics: {
promptTokens: llmResponse.promptTokens,
completionTokens: llmResponse.completionTokens,
totalTokens: llmResponse.totalTokens,
promptTimeMs: llmResponse.promptTimeMs,
completionTimeMs: llmResponse.completionTimeMs,
totalTimeMs: llmResponse.totalTimeMs,
},
};
}
/**
* Enriches a batch of words with retry and split-on-failure logic.
* Retries up to BATCH_CONFIG.maxRetries times, then splits batch in half and retries each half.
* Continues splitting until batch size is 1, then throws if still failing.
*/
export async function enrichWordWithRetry(
words: string[],
rawLanguage: string,
rawPos: string,
attempt: number = 1,
): Promise<EnrichmentResult> {
try {
return await enrichWord(words, rawLanguage, rawPos);
} catch (error: unknown) {
const errorMessage = error instanceof Error ? error.message : String(error);
if (words.length === 1) {
throw new Error(
`Failed to enrich word "${words[0]}" after ${attempt} attempts: ${errorMessage}`,
{ cause: error },
);
}
if (attempt < BATCH_CONFIG.maxRetries) {
console.log(
` Retry ${attempt}/${BATCH_CONFIG.maxRetries} for batch [${words.join(", ")}]: ${errorMessage}`,
);
return enrichWordWithRetry(words, rawLanguage, rawPos, attempt + 1);
}
// Max retries reached split and retry
console.log(
` Splitting batch [${words.join(", ")}] after ${BATCH_CONFIG.maxRetries} failed attempts`,
);
const half = Math.ceil(words.length / 2);
const left = words.slice(0, half);
const right = words.slice(half);
const leftResult = await enrichWordWithRetry(left, rawLanguage, rawPos, 1);
const rightResult = await enrichWordWithRetry(
right,
rawLanguage,
rawPos,
1,
);
// Merge results
const merged = new Map([...leftResult.results, ...rightResult.results]);
const mergedMetrics = {
promptTokens:
leftResult.metrics.promptTokens + rightResult.metrics.promptTokens,
completionTokens:
leftResult.metrics.completionTokens +
rightResult.metrics.completionTokens,
totalTokens:
leftResult.metrics.totalTokens + rightResult.metrics.totalTokens,
promptTimeMs:
(leftResult.metrics.promptTimeMs ?? 0) +
(rightResult.metrics.promptTimeMs ?? 0),
completionTimeMs:
(leftResult.metrics.completionTimeMs ?? 0) +
(rightResult.metrics.completionTimeMs ?? 0),
totalTimeMs:
leftResult.metrics.totalTimeMs + rightResult.metrics.totalTimeMs,
};
return { results: merged, metrics: mergedMetrics };
}
}
</file>
<file path="utils/get-word-file-path.ts">
import path from "path";
export function getWordFilePath(word: string, outputDir: string): string {
return path.join(outputDir, `${word}.json`);
}
</file>
<file path="utils/merge-enriched-data.ts">
import { LLM_CONFIG } from "../config/llm.js";
export type Language = "en" | "de" | "it" | "es" | "fr";
export type Pos = "noun" | "verb" | "adjective" | "adverb";
export type Gender = "masculine" | "feminine" | "neuter" | null;
export type Difficulty = "easy" | "medium" | "hard";
export interface Translation {
word: string;
gender: Gender;
}
export interface EnrichedSense {
id: string;
word: string;
language: Language;
pos: Pos;
sense: string;
example: string;
difficulty_level: Difficulty;
translations: {
de: Translation[];
it: Translation[];
es: Translation[];
fr: Translation[];
};
}
/**
* Merges skeleton data with enriched LLM senses into the final pipeline output.
*/
export function mergeEnrichedData(
word: string,
senses: EnrichedSense[],
): Record<string, unknown> {
return {
word,
language: senses[0]?.language ?? "en",
pos: senses[0]?.pos ?? "noun",
senses,
enrichedAt: new Date().toISOString(),
model: LLM_CONFIG.model ?? "unknown",
};
}
</file>
<file path="utils/pipeline-timer.ts">
interface LlmMetrics {
promptTokens: number;
completionTokens: number;
totalTokens: number;
promptTimeMs: number | null;
completionTimeMs: number | null;
totalTimeMs: number;
}
interface PipelineMetrics {
startTime: Date;
endTime?: Date;
wordsProcessed: number;
wordsSkipped: number;
wordsFailed: number;
llmCalls: number;
totalPromptTokens: number;
totalCompletionTokens: number;
totalTokens: number;
totalPromptTimeMs: number;
totalCompletionTimeMs: number;
totalTimeMs: number;
currentWordStartTime?: Date;
}
/**
* Simple timer and metrics tracker for the pipeline.
* Tracks both pipeline throughput and LLM performance.
*/
export class PipelineTimer {
private metrics: PipelineMetrics;
constructor() {
this.metrics = {
startTime: new Date(),
wordsProcessed: 0,
wordsSkipped: 0,
wordsFailed: 0,
llmCalls: 0,
totalPromptTokens: 0,
totalCompletionTokens: 0,
totalTokens: 0,
totalPromptTimeMs: 0,
totalCompletionTimeMs: 0,
totalTimeMs: 0,
};
}
startWord(): void {
this.metrics.currentWordStartTime = new Date();
}
getWordDurationMs(): number {
if (!this.metrics.currentWordStartTime) return 0;
return new Date().getTime() - this.metrics.currentWordStartTime.getTime();
}
recordProcessed(llmMetrics?: LlmMetrics): void {
this.metrics.wordsProcessed++;
if (llmMetrics) {
this.metrics.llmCalls++;
this.metrics.totalPromptTokens += llmMetrics.promptTokens;
this.metrics.totalCompletionTokens += llmMetrics.completionTokens;
this.metrics.totalTokens += llmMetrics.totalTokens;
if (llmMetrics.promptTimeMs !== null) {
this.metrics.totalPromptTimeMs += llmMetrics.promptTimeMs;
}
if (llmMetrics.completionTimeMs !== null) {
this.metrics.totalCompletionTimeMs += llmMetrics.completionTimeMs;
}
this.metrics.totalTimeMs += llmMetrics.totalTimeMs;
}
}
recordSkipped(): void {
this.metrics.wordsSkipped++;
}
recordFailed(): void {
this.metrics.wordsFailed++;
}
stop(): void {
this.metrics.endTime = new Date();
}
getWordTiming(): string {
const durationMs = this.getWordDurationMs();
const durationSec = (durationMs / 1000).toFixed(1);
return `⏱️ Word took ${durationSec}s`;
}
getEta(totalWords: number): string {
const processed = this.metrics.wordsProcessed;
const remaining = totalWords - processed - this.metrics.wordsSkipped;
if (processed === 0 || remaining <= 0) return "ETA: calculating...";
const elapsedMs = new Date().getTime() - this.metrics.startTime.getTime();
const avgMsPerWord = elapsedMs / processed;
const etaMs = avgMsPerWord * remaining;
const etaMin = Math.round(etaMs / 60000);
const etaHour = (etaMs / 3600000).toFixed(1);
if (etaMin < 60) {
return `ETA: ${etaMin} min`;
}
return `ETA: ${etaHour} hours`;
}
getSummary(): string {
const end = this.metrics.endTime || new Date();
const durationMs = end.getTime() - this.metrics.startTime.getTime();
const durationSec = (durationMs / 1000).toFixed(1);
const total =
this.metrics.wordsProcessed +
this.metrics.wordsSkipped +
this.metrics.wordsFailed;
const throughput =
this.metrics.wordsProcessed > 0
? (this.metrics.wordsProcessed / (durationMs / 1000)).toFixed(2)
: "0";
const avgPromptTokens =
this.metrics.llmCalls > 0
? (this.metrics.totalPromptTokens / this.metrics.llmCalls).toFixed(0)
: "0";
const avgCompletionTokens =
this.metrics.llmCalls > 0
? (this.metrics.totalCompletionTokens / this.metrics.llmCalls).toFixed(
0,
)
: "0";
const avgTotalTimeMs =
this.metrics.llmCalls > 0
? (this.metrics.totalTimeMs / this.metrics.llmCalls).toFixed(0)
: "0";
const unifiedThroughput =
this.metrics.totalTimeMs > 0
? (
this.metrics.totalTokens /
(this.metrics.totalTimeMs / 1000)
).toFixed(1)
: "N/A";
const hasDetailedTimings =
this.metrics.totalPromptTimeMs > 0 ||
this.metrics.totalCompletionTimeMs > 0;
const avgPromptSpeed =
this.metrics.totalPromptTimeMs > 0
? (
this.metrics.totalPromptTokens /
(this.metrics.totalPromptTimeMs / 1000)
).toFixed(1)
: "N/A";
const avgCompletionSpeed =
this.metrics.totalCompletionTimeMs > 0
? (
this.metrics.totalCompletionTokens /
(this.metrics.totalCompletionTimeMs / 1000)
).toFixed(1)
: "N/A";
const lines = [
`⏱️ Pipeline Summary`,
` Duration: ${durationSec}s`,
` Processed: ${this.metrics.wordsProcessed}`,
` Skipped: ${this.metrics.wordsSkipped}`,
` Failed: ${this.metrics.wordsFailed}`,
` Total: ${total}`,
` Throughput: ${throughput} words/sec`,
``,
`🤖 LLM Metrics`,
` Calls: ${this.metrics.llmCalls}`,
` Avg prompt tokens: ${avgPromptTokens}`,
` Avg completion tokens: ${avgCompletionTokens}`,
` Avg total tokens: ${avgPromptTokens + avgCompletionTokens}`,
` Avg total request time: ${avgTotalTimeMs}ms`,
` Avg throughput: ${unifiedThroughput} tok/s`,
];
if (hasDetailedTimings) {
lines.push(
``,
` [Local breakdown]`,
` Avg prompt speed: ${avgPromptSpeed} tok/s`,
` Avg completion speed: ${avgCompletionSpeed} tok/s`,
);
}
return lines.join("\n");
}
}
</file>
<file path="utils/progress-tracker.ts">
/**
* Simple progress tracker for pipeline execution.
*/
export class ProgressTracker {
private current: number;
private failed: number;
private total: number;
constructor(total: number) {
this.current = 0;
this.failed = 0;
this.total = total;
}
next(): number {
this.current++;
return this.current;
}
recordFailed(): void {
this.failed++;
}
format(label: string): string {
return `[${this.current}/${this.total}] (${this.failed} failed) ${label}`;
}
}
</file>
<file path="utils/scanning-source-files.ts">
import fs from "fs";
import path from "path";
// Define a simple shape for what a discovered dataset looks like
export interface Wordlist {
language: string;
pos: string;
sourcePath: string;
outputDir: string;
}
/**
* Scans the source-data directory to find all available word lists.
*/
export function scanSourceData(baseDir: string): Wordlist[] {
const sourceBaseDir = path.join(baseDir, "source-data");
const discoveredWordlists: Wordlist[] = [];
// Safety check: if there's no source-data folder, return an empty array
if (!fs.existsSync(sourceBaseDir)) {
return discoveredWordlists;
}
// 1. Read the language directories (e.g., ['english'])
const languages = fs.readdirSync(sourceBaseDir);
for (const lang of languages) {
const langFolderPath = path.join(sourceBaseDir, lang);
// Make sure it's a directory, not a stray file
if (!fs.statSync(langFolderPath).isDirectory()) continue;
// 2. Read the files inside the language folder (e.g., ['nouns'])
const posFiles = fs.readdirSync(langFolderPath);
for (const pos of posFiles) {
const fullSourcePath = path.join(langFolderPath, pos);
// Make sure it's a file (like your extensionless "nouns" file)
if (!fs.statSync(fullSourcePath).isFile()) continue;
// 3. Package everything into a flat item and add it to our array
discoveredWordlists.push({
language: lang,
pos: pos,
sourcePath: fullSourcePath,
outputDir: path.join(baseDir, "worddata", lang, pos),
});
}
}
// show summary
console.log(
`✅ Scan complete! Found ${discoveredWordlists.length} wordlist(s):`,
);
for (const list of discoveredWordlists) {
console.log(` • ${list.language.toUpperCase()} (${list.pos})`);
}
return discoveredWordlists;
}
</file>
<file path="utils/verify-enriched-file.ts">
import fs from "fs";
interface VerificationResult {
valid: boolean;
errors: string[];
}
/**
* Verifies that an enriched JSON file matches the expected schema.
* Returns detailed error messages for any violations.
*/
export function verifyEnrichedFile(filePath: string): VerificationResult {
const errors: string[] = [];
if (!fs.existsSync(filePath)) {
return { valid: false, errors: ["File does not exist"] };
}
let data: unknown;
try {
data = JSON.parse(fs.readFileSync(filePath, "utf-8"));
} catch (_error: unknown) {
return { valid: false, errors: ["Invalid JSON syntax"] };
}
if (typeof data !== "object" || data === null || Array.isArray(data)) {
return { valid: false, errors: ["Root must be an object"] };
}
const obj = data as Record<string, unknown>;
// Required top-level fields
const requiredFields = ["word", "language", "pos", "senses"];
for (const field of requiredFields) {
if (!(field in obj)) {
errors.push(`Missing required field: "${field}"`);
}
}
// Validate senses array
if (!Array.isArray(obj["senses"])) {
errors.push('"senses" must be an array');
} else if (obj["senses"].length === 0) {
errors.push('"senses" array cannot be empty');
} else {
for (let i = 0; i < obj["senses"].length; i++) {
const sense = obj["senses"][i] as Record<string, unknown>;
const sensePrefix = `senses[${i}]`;
if (!sense["sense"] || typeof sense["sense"] !== "string") {
errors.push(`${sensePrefix}: missing or invalid "sense"`);
}
if (!sense["example"] || typeof sense["example"] !== "string") {
errors.push(`${sensePrefix}: missing or invalid "example"`);
}
if (
!["easy", "medium", "hard"].includes(
sense["difficulty_level"] as string,
)
) {
errors.push(`${sensePrefix}: invalid "difficulty_level"`);
}
if (!sense["translations"] || typeof sense["translations"] !== "object") {
errors.push(`${sensePrefix}: missing "translations"`);
} else {
const trans = sense["translations"] as Record<string, unknown>;
for (const lang of ["de", "it", "es", "fr"]) {
if (!Array.isArray(trans[lang])) {
errors.push(
`${sensePrefix}: missing or invalid "${lang}" translations`,
);
} else {
for (let j = 0; j < (trans[lang] as unknown[]).length; j++) {
const t = (trans[lang] as unknown[])[j] as Record<
string,
unknown
>;
if (!t["word"] || typeof t["word"] !== "string") {
errors.push(`${sensePrefix}.${lang}[${j}]: missing "word"`);
}
if (
!["masculine", "feminine", "neuter", null].includes(
t["gender"] as string | null,
)
) {
errors.push(`${sensePrefix}.${lang}[${j}]: invalid "gender"`);
}
}
}
}
}
}
}
return { valid: errors.length === 0, errors };
}
</file>
<file path="utils/write-json-file.ts">
import fs from "fs";
/**
* Writes data as formatted JSON to a file path.
* Safely catches and re-throws file system errors.
*/
export function writeJsonFile(filePath: string, data: unknown): void {
const tempPath = `${filePath}.tmp`;
fs.writeFileSync(tempPath, JSON.stringify(data, null, 2), "utf-8");
fs.renameSync(tempPath, filePath);
}
</file>
<file path=".pipeline-config.json">
{
"provider": "openrouter",
"model": "google/gemini-2.5-flash",
"batchSize": 5,
"maxRetries": 3
}
</file>
<file path="pipeline.ts">
import { isWordProcessed } from "./utils/check-if-json-exists.js";
import { createBaseJson } from "./utils/create-base-json.js";
import { ensureOutputFolders } from "./utils/create-output-dirs.js";
import { scanSourceData } from "./utils/scanning-source-files.js";
import { createLineReader } from "./utils/create-line-reader.js";
import { checkLlmServer } from "./utils/check-llm-server.js";
import { getWordFilePath } from "./utils/get-word-file-path.js";
import { mergeEnrichedData } from "./utils/merge-enriched-data.js";
import { enrichWordWithRetry } from "./utils/enrich-word.js";
import { writeJsonFile } from "./utils/write-json-file.js";
import { deleteFileIfExists } from "./utils/delete-file.js";
import { PipelineTimer } from "./utils/pipeline-timer.js";
import { ProgressTracker } from "./utils/progress-tracker.js";
import { verifyEnrichedFile } from "./utils/verify-enriched-file.js";
import { runCli } from "./utils/cli.js";
import type { PipelineConfig } from "./utils/cli.js";
import { LLM_CONFIG } from "./config/llm.js";
import { BATCH_CONFIG } from "./config/batch.js";
// Runtime config accessor for other modules
let RUNTIME_CONFIG: PipelineConfig;
export function getRuntimeConfig(): PipelineConfig {
return RUNTIME_CONFIG;
}
async function main() {
// ── Interactive CLI ──────────────────────────────────────────────────────
RUNTIME_CONFIG = await runCli();
// Populate shared config objects so existing imports keep working
LLM_CONFIG.provider = RUNTIME_CONFIG.provider;
LLM_CONFIG.url = RUNTIME_CONFIG.url;
LLM_CONFIG.model = RUNTIME_CONFIG.model;
BATCH_CONFIG.size = RUNTIME_CONFIG.batchSize;
BATCH_CONFIG.maxRetries = RUNTIME_CONFIG.maxRetries;
console.log("Starting data pipeline...\n");
console.log(`Provider: ${RUNTIME_CONFIG.provider}`);
console.log(`Model: ${RUNTIME_CONFIG.model ?? "(none)"}`);
console.log(`Batch: ${RUNTIME_CONFIG.batchSize} words/call\n`);
const timer = new PipelineTimer();
// step 1: scanning for source files
console.log("\n step 1: scanning the source files...");
const wordlists = scanSourceData(import.meta.dirname);
// step 2: ensuring output folders exist
console.log("\n step 2: creating necessary output folders...");
ensureOutputFolders(wordlists);
// step 3: check to verify the local AI engine is ready before touching anything
console.log("\n step 3: verifying local AI engine status...");
try {
await checkLlmServer();
} catch (error: unknown) {
const message = error instanceof Error ? error.message : String(error);
console.error(`\n ❌ ${message}`);
process.exit(1);
}
// Step 4: Loop through the wordlists array
console.log("\n step 4: looping through the wordlists...");
for (const wordlist of wordlists) {
console.log(
`\nReading list: [${wordlist.language.toUpperCase()}] -> [${wordlist.pos.toUpperCase()}]`,
);
const rl = createLineReader(wordlist.sourcePath);
// Collect words and count them
const words: string[] = [];
for await (const line of rl) {
const word = line.trim().toLowerCase();
if (word) words.push(word);
}
// Filter out already-processed words
const unprocessedWords = words.filter(
(word) => !isWordProcessed(word, wordlist.outputDir),
);
const skippedCount = words.length - unprocessedWords.length;
if (skippedCount > 0) {
console.log(` Skipped ${skippedCount} already-processed words`);
}
const progress = new ProgressTracker(unprocessedWords.length);
// Step 5: Process in batches
for (let i = 0; i < unprocessedWords.length; i += BATCH_CONFIG.size) {
const batch = unprocessedWords.slice(i, i + BATCH_CONFIG.size);
const batchNum = Math.floor(i / BATCH_CONFIG.size) + 1;
const totalBatches = Math.ceil(
unprocessedWords.length / BATCH_CONFIG.size,
);
const batchLabel = `Batch ${batchNum}/${totalBatches}`;
console.log(`\n ${batchLabel}: [${batch.join(", ")}]`);
// Create skeletons for all words in batch
for (const word of batch) {
createBaseJson(
word,
wordlist.outputDir,
wordlist.language,
wordlist.pos,
);
}
timer.startWord();
try {
// Step 6: enrich batch with senses (with retry/split)
const result = await enrichWordWithRetry(
batch,
wordlist.language,
wordlist.pos,
);
// Step 7: write each word's result
for (const [word, senses] of result.results) {
progress.next();
console.log(
` ${progress.format(`Enriched and saved: ${word}.json`)}`,
);
const targetFilePath = getWordFilePath(word, wordlist.outputDir);
const enrichedData = mergeEnrichedData(word, senses);
writeJsonFile(targetFilePath, enrichedData);
// Verify the generated file
const verification = verifyEnrichedFile(targetFilePath);
if (!verification.valid) {
console.error(` Warning: Schema violations in ${word}.json:`);
for (const error of verification.errors) {
console.error(` - ${error}`);
}
}
timer.recordProcessed({
promptTokens: result.metrics.promptTokens,
completionTokens: result.metrics.completionTokens,
totalTokens: result.metrics.totalTokens,
promptTimeMs: result.metrics.promptTimeMs,
completionTimeMs: result.metrics.completionTimeMs,
totalTimeMs: result.metrics.totalTimeMs,
});
}
console.log(` ${timer.getWordTiming()}`);
// Show ETA every 5 batches or on the last batch
if (batchNum % 5 === 0 || batchNum === totalBatches) {
console.log(` 📊 ${timer.getEta(unprocessedWords.length)}`);
}
} catch (error: unknown) {
const errorMessage =
error instanceof Error ? error.message : String(error);
console.error(
` Failed to enrich batch [${batch.join(", ")}]: ${errorMessage}`,
);
// Cleanup: delete any partially-written files for the failed batch
for (const word of batch) {
const targetFilePath = getWordFilePath(word, wordlist.outputDir);
deleteFileIfExists(targetFilePath);
console.log(` Removed incomplete file: ${word}.json`);
progress.recordFailed();
}
timer.recordFailed();
}
}
}
timer.stop();
console.log("\n" + timer.getSummary());
console.log("\nGlobal data pipeline run completed successfully.");
}
// Fire the orchestrator block
main().catch((err) => {
console.error("Critical unexpected pipeline failure:", err);
});
</file>
<file path="vitest.config.ts">
import { defineConfig } from "vitest/config";
export default defineConfig({
test: {
environment: "node",
globals: true,
include: ["tests/**/*.test.ts"],
exclude: ["**/dist/**", "**/node_modules/**"],
testTimeout: 60_000,
},
});
</file>
<file path="tsconfig.json">
{
"extends": "../tsconfig.base.json",
"compilerOptions": {
"module": "NodeNext",
"moduleResolution": "NodeNext",
"outDir": "dist",
"rootDir": ".",
"types": ["node"]
},
"references": [{ "path": "../packages/shared" }],
"include": ["./**/*", "vitest.config.ts"]
}
</file>
<file path=".env.example">
# OpenRouter API key — required for OpenRouter providers
# Get one at https://openrouter.ai/keys
OPENROUTER_API_KEY=
# Anthropic API key — required for Anthropic provider (reference baseline only)
# Get one at https://console.anthropic.com/
ANTHROPIC_API_KEY=
</file>
<file path="package.json">
{
"name": "@lila/pipeline",
"version": "1.0.0",
"private": true,
"type": "module",
"scripts": {
"db:reset": "tsx db/reset.ts",
"extract": "tsx stage-1-extract/scripts/extract.ts",
"reverse-link": "tsx stage-2-reverse-link/scripts/reverse-link.ts",
"db:import": "tsx db/import.ts",
"db:init": "tsx db/init.ts",
"test": "vitest run",
"test:watch": "vitest",
"pipeline:run": "tsx --env-file .env pipeline.ts"
},
"dependencies": {
"@lila/shared": "workspace:*",
"better-sqlite3": "^12.9.0"
},
"devDependencies": {
"@types/better-sqlite3": "^7.6.13",
"@types/node": "^24.12.0",
"tsx": "^4.21.0",
"typescript": "^5.9.3",
"vitest": "^4.1.0"
}
}
</file>
</files>