diff --git a/data-pipeline/pipeline.ts b/data-pipeline/pipeline.ts index 7f28677..b1daecd 100644 --- a/data-pipeline/pipeline.ts +++ b/data-pipeline/pipeline.ts @@ -29,7 +29,13 @@ async function main() { // step 3: check to verify the local AI engine is ready before touching anything console.log("\n step 3: verifying local AI engine status..."); - await checkLlmServer(); + try { + await checkLlmServer(); + } catch (error: unknown) { + const message = error instanceof Error ? error.message : String(error); + console.error(`\n ❌ ${message}`); + process.exit(1); + } // Step 4: Loop through the wordlists array console.log("\n step 4: looping through the wordlists..."); @@ -134,7 +140,7 @@ async function main() { ` Failed to enrich batch [${batch.join(", ")}]: ${errorMessage}`, ); - // Cleanup: delete skeleton files for failed batch + // Cleanup: delete any partially-written files for the failed batch for (const word of batch) { const targetFilePath = getWordFilePath(word, wordlist.outputDir); deleteFileIfExists(targetFilePath); diff --git a/data-pipeline/utils/check-llm-server.ts b/data-pipeline/utils/check-llm-server.ts index f2de396..b3f4f96 100644 --- a/data-pipeline/utils/check-llm-server.ts +++ b/data-pipeline/utils/check-llm-server.ts @@ -18,13 +18,10 @@ export async function checkLlmServer( // llama.cpp returns a 503 status if the server is up but the model weights are still loading if (response.status === 503) { - console.error( - "\n ⏳ Local AI engine is starting up, but the model is still loading into memory.", + throw new Error( + "Local AI engine is starting up, but the model is still loading into memory. " + + "Please wait a minute for the weights to load, then run the pipeline again.", ); - console.error( - "👉 Please wait a minute for the weights to load, then run the pipeline again.\n", - ); - process.exit(1); } // Parse the JSON health response (expected: { status: "ok" }) @@ -36,16 +33,17 @@ export async function checkLlmServer( } // Catch-all for unexpected active server responses - console.error( - `\n ❌ Unknown response from local AI engine health check (Status: ${response.status}).`, + throw new Error( + `Unknown response from local AI engine health check (Status: ${response.status}).`, ); - process.exit(1); - } catch (_error: unknown) { - console.error("\n ❌ Could not connect to the local AI engine."); - console.error(`🔗 Attempted endpoint: ${url}`); - console.error( - "👉 Make sure your './llama-server' command is actively running in another terminal tab!\n", + } catch (error: unknown) { + if (error instanceof Error && error.message.includes("Local AI engine")) { + throw error; // Re-throw our own errors + } + throw new Error( + `Could not connect to the local AI engine at ${url}. ` + + "Make sure your './llama-server' command is actively running in another terminal tab.", + { cause: error }, ); - process.exit(1); } } diff --git a/documentation/DATA_PIPELINE.md b/documentation/DATA_PIPELINE.md index 9271551..f4d6768 100644 --- a/documentation/DATA_PIPELINE.md +++ b/documentation/DATA_PIPELINE.md @@ -19,26 +19,28 @@ 7. [Batching Strategy](#7-batching-strategy) 8. [Hardware Constraints](#8-hardware-constraints) 9. [Testing & Quality Assurance](#9-testing--quality-assurance) -10. [Future Extensions & Roadmap](#10-future-extensions--roadmap) -11. [Decisions Log](#11-decisions-log) -12. [Known Issues & Dev Notes](#12-known-issues--dev-notes) -13. [How to Run](#13-how-to-run) -14. [Roadmap](#14-roadmap) +10. [Interactive CLI](#10-interactive-cli) +11. [Future Extensions & Roadmap](#11-future-extensions--roadmap) +12. [Decisions Log](#12-decisions-log) +13. [Known Issues & Dev Notes](#13-known-issues--dev-notes) +14. [How to Run](#14-how-to-run) +15. [Roadmap](#15-roadmap) --- ## Quick Reference -| What | Where | -| ------------- | ----------------------------------- | -| Entry point | `pipeline.ts` | -| LLM config | `config/llm.ts` | -| System prompt | `config/prompt.ts` | -| Batch config | `config/batch.ts` | -| Output schema | `utils/merge-enriched-data.ts` | -| LLM adapters | `utils/llm-adapters/` | -| Current model | `qwen2.5-1.5b-instruct-q4_k_m.gguf` | -| Target scale | 100,000+ words | +| What | Where | +| ---------------- | ------------------------------------------ | +| Entry point | `pipeline.ts` | +| LLM config | `config/llm.ts` | +| System prompt | `config/prompt.ts` — `buildSystemPrompt()` | +| Batch config | `config/batch.ts` | +| Shared constants | `config/constants.ts` | +| Output schema | `utils/merge-enriched-data.ts` | +| LLM adapters | `utils/llm-adapters/` | +| Current model | `qwen2.5-1.5b-instruct-q4_k_m.gguf` | +| Target scale | 100,000+ words | --- @@ -49,9 +51,9 @@ The Lila Data Pipeline is a TypeScript-based batch processing system that enrich - One or more **senses** (definitions) - A **natural example sentence** per sense - A **CEFR-based difficulty level** (`easy` / `medium` / `hard`) -- **Translations** into German, Italian, Spanish, and French, each with grammatical **gender** +- **Translations** into all target languages except the source, each with grammatical **gender** -The pipeline is designed to scale to **100,000+ words** across multiple languages and parts of speech (nouns, verbs, adjectives, adverbs). It is currently in active development: the core architecture is stable, the LLM integration layer supports both local and cloud providers via a pluggable adapter pattern, and a configurable batching system with retry/split logic is fully implemented. +The pipeline is designed to scale to **100,000+ words** across multiple languages and parts of speech (nouns, verbs, adjectives, adverbs). It supports both local inference (llama.cpp) and cloud providers (Gemini, DeepSeek, OpenRouter, Groq) via a pluggable adapter pattern, with an interactive CLI for provider selection. ### Key Design Principles @@ -62,6 +64,7 @@ The pipeline is designed to scale to **100,000+ words** across multiple language | **Resumable & idempotent** | Each word writes to its own JSON file. The pipeline skips already-processed words on restart. | | **Configurable batching** | Batch size is a single config value (`config/batch.ts`). The pipeline adapts without code changes. | | **Provider-agnostic** | LLM adapters abstract local, OpenRouter, DeepSeek, and Gemini behind a single interface. | +| **Honest metrics** | Local models report detailed prompt/completion timing. Cloud providers report total request time only — no fake breakdowns. | ### Open Question: Gender Accuracy @@ -81,8 +84,10 @@ No decision made. Gender handling will be determined by the 20-word quality tort - Local LLM integration via llama.cpp server (OpenAI-compatible API) - **Cloud provider adapters**: Gemini, DeepSeek, OpenRouter via `utils/llm-adapters/` - **Batching with retry/split**: configurable batch size, exponential split-on-failure (4 → 2 → 1) -- Schema validation for generated JSON -- Progress tracking and timing metrics +- **Honest timing**: unified throughput for all providers, detailed breakdown only for local +- **Auto-target languages**: prompt dynamically excludes source language from targets +- **Schema validation**: validates LLM response structure before file writes +- **Interactive CLI**: planned — provider/model/batch selection with saved config - **In progress:** Evaluating local models (Qwen2.5-1.5B tested; Qwen2.5-3B download pending) - **Pending:** 20-word quality torture suite (will decide gender approach) - **Pending:** Online API evaluation (Gemini free tier, DeepSeek, Groq) @@ -90,7 +95,7 @@ No decision made. Gender handling will be determined by the 20-word quality tort ### One-Line Architecture ``` -source wordlists -> llama.cpp server (LLM) -> merge senses -> verify schema -> write .json +source wordlists -> LLM adapter (local or cloud) -> merge senses -> verify schema -> write .json | [gender: LLM-generated, accuracy TBD] ``` @@ -101,12 +106,13 @@ source wordlists -> llama.cpp server (LLM) -> merge senses -> verify schema -> w | ----------------------------------------- | ----------------------------------------------------------------------------------- | | `pipeline.ts` | Orchestrator. Scans sources, loops words, coordinates all stages | | `config/llm.ts` | Provider selection, API URL, model name | -| `config/prompt.ts` | System prompt sent to the LLM | +| `config/prompt.ts` | `buildSystemPrompt()` — dynamic prompt with auto-target languages | | `config/batch.ts` | Batch size and max retry count | +| `config/constants.ts` | Shared `LANG_MAP`, `POS_MAP`, `ALL_LANGUAGES` | | `utils/enrich-word.ts` | Calls LLM via adapter, parses response, builds `EnrichedSense[]`, retry/split logic | -| `utils/merge-enriched-data.ts` | Merges skeleton + enriched senses into final JSON; defines TypeScript schema | +| `utils/merge-enriched-data.ts` | Merges skeleton + enriched senses into final JSON | | `utils/verify-enriched-file.ts` | Schema validation (required fields, types, gender enum) | -| `utils/check-llm-server.ts` | Health check before pipeline starts | +| `utils/check-llm-server.ts` | Health check for local server; skipped for cloud providers | | `utils/scanning-source-files.ts` | Discovers wordlists from `source-data/` directory | | `utils/create-base-json.ts` | Writes skeleton `{word, language, pos}` files | | `utils/write-json-file.ts` | Atomic `.tmp` → rename writes | @@ -116,7 +122,7 @@ source wordlists -> llama.cpp server (LLM) -> merge senses -> verify schema -> w | `utils/delete-file.ts` | Cleanup helper for failed batches | | `utils/get-word-file-path.ts` | Path construction helper | | `utils/progress-tracker.ts` | `[current/total]` formatting for console output | -| `utils/pipeline-timer.ts` | Per-word and global timing + LLM token metrics | +| `utils/pipeline-timer.ts` | Timing + token metrics; unified throughput for all providers | | `utils/llm-adapters/factory.ts` | Creates the right adapter based on `LLM_CONFIG.provider` | | `utils/llm-adapters/types.ts` | `LlmAdapter` interface | | `utils/llm-adapters/openai-compatible.ts` | Local llama.cpp, OpenRouter, DeepSeek | @@ -165,11 +171,11 @@ The LLM fills the gap: it generates **pedagogical content** (student-friendly de The pipeline is **direction-agnostic**. A wordlist is defined by: - **Source language**: the language of the input words -- **Target languages**: the languages to translate into +- **Target languages**: all other languages in the system (auto-derived from `ALL_LANGUAGES` minus source) Current focus: **English -> German/Italian/Spanish/French** -Planned directions include **German -> French**, **Italian -> Spanish**, etc. The LLM prompt and output schema support any combination — the only change is the source wordlist and the target languages specified in the prompt. +Planned directions include **German -> French**, **Italian -> Spanish**, etc. The LLM prompt and output schema support any combination — the only change is the source wordlist. Target languages are computed automatically. ### Why 100,000+ Words? @@ -201,6 +207,7 @@ Generating 100,000 entries with an LLM introduces risks: | JSON parse failures | Retry + split logic, schema validation, cleanup on failure | | Model drift (online APIs) | Version pinning, local fallback | | Provider downtime | Adapter pattern allows hot-swapping providers | +| Malformed LLM responses | `validateSense()` catches bad data before file writes | ### Why TypeScript + Node? @@ -224,15 +231,15 @@ Generating 100,000 entries with an LLM introduces risks: ``` Scan sources -> Check LLM -> Loop wordlists -> Stream words -> Skip processed - -> Create skeletons (batch) -> Call LLM -> Parse JSON -> Retry/split on failure - -> Merge -> Write atomically -> Verify schema -> Log metrics + -> Create skeletons (batch) -> Call LLM -> Parse JSON -> Validate senses + -> Retry/split on failure -> Merge -> Write atomically -> Verify schema -> Log metrics ``` ### Resumability - **Skip existing**: `check-if-json-exists.ts` checks if `{word}.json` exists with non-empty `senses` - **Atomic writes**: `.tmp` -> rename in `write-json-file.ts`, no partial files on crash -- **Cleanup on failure**: Deletes skeleton files for failed batches, continues to next batch +- **Cleanup on failure**: Deletes partially-written files for failed batches, continues to next batch ### Directory Structure @@ -241,13 +248,14 @@ data-pipeline/ |-- pipeline.ts # Entry point / orchestrator |-- config/ | |-- llm.ts # Provider, API URL, model name -| |-- prompt.ts # System prompt +| |-- prompt.ts # buildSystemPrompt() — dynamic prompt | |-- batch.ts # Batch size and retry config +| |-- constants.ts # LANG_MAP, POS_MAP, ALL_LANGUAGES |-- utils/ | |-- enrich-word.ts # LLM call, parse, retry/split | |-- merge-enriched-data.ts # Schema types + merge logic | |-- verify-enriched-file.ts # Schema validation -| |-- check-llm-server.ts # Health check +| |-- check-llm-server.ts # Health check (local only) | |-- scanning-source-files.ts # Source discovery | |-- create-base-json.ts # Skeleton writer | |-- write-json-file.ts # Atomic JSON writer @@ -281,18 +289,22 @@ Full TypeScript interfaces: `utils/merge-enriched-data.ts`. ### Error Handling -| Failure | Behavior | -| ------------------------------ | ------------------------------------------------------- | -| LLM server offline | Hard fail at startup (`check-llm-server.ts`) | -| LLM returns bad JSON | Retry up to 3 times, then split batch. Log and continue | -| Schema validation fails | Log warnings, keep file | -| Individual batch fails | Does not stop pipeline; cleans up skeletons | -| Individual word fails (size 1) | Log and continue to next word | +| Failure | Behavior | +| ------------------------------ | -------------------------------------------------------- | +| LLM server offline | Hard fail at startup (`check-llm-server.ts`, local only) | +| LLM returns bad JSON | Retry up to 3 times, then split batch. Log and continue | +| LLM returns malformed senses | `validateSense()` catches it before file write | +| Schema validation fails | Log warnings, keep file | +| Individual batch fails | Does not stop pipeline; cleans up partial files | +| Individual word fails (size 1) | Log and continue to next word | ### Metrics Per-run: words processed/skipped/failed, duration, throughput, LLM token counts and speeds. See `utils/pipeline-timer.ts`. +**Unified throughput** (all providers): total tokens / total request time +**Detailed breakdown** (local only): prompt speed vs completion speed + --- ## 4. Current Implementation @@ -308,23 +320,24 @@ Per-run: words processed/skipped/failed, duration, throughput, LLM token counts ### Configuration -| File | Purpose | -| ------------------ | ---------------------------------------------------------------- | ------------ | ---------- | ------------------------- | -| `config/llm.ts` | `provider` (`local` | `openrouter` | `deepseek` | `gemini`), `url`, `model` | -| `config/prompt.ts` | System prompt with CEFR mapping, required fields, example output | -| `config/batch.ts` | `BATCH_CONFIG.size` (words per call), `maxRetries` | +| File | Purpose | +| --------------------- | --------------------------------------------------------- | ------------ | ---------- | ------------------------- | +| `config/llm.ts` | `provider` (`local` | `openrouter` | `deepseek` | `gemini`), `url`, `model` | +| `config/prompt.ts` | `buildSystemPrompt(sourceLanguage, pos, targetLanguages)` | +| `config/batch.ts` | `BATCH_CONFIG.size` (words per call), `maxRetries` | +| `config/constants.ts` | `LANG_MAP`, `POS_MAP`, `ALL_LANGUAGES` | ### Key Modules -| File | Responsibility | -| ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | -| `utils/enrich-word.ts` | Calls LLM via adapter, strips markdown, parses JSON array, builds `EnrichedSense[]` with composite IDs, retry/split logic | -| `utils/merge-enriched-data.ts` | Merges skeleton `{word, language, pos}` with LLM senses, adds `enrichedAt` and `model` | -| `utils/verify-enriched-file.ts` | Schema validation: required fields, array lengths, gender enum, translation structure | -| `utils/pipeline-timer.ts` | Tracks per-word and global metrics (duration, tokens, throughput) | -| `utils/llm-adapters/factory.ts` | Creates adapter based on `LLM_CONFIG.provider` | -| `utils/llm-adapters/openai-compatible.ts` | OpenAI chat completions API for local llama.cpp, OpenRouter, DeepSeek | -| `utils/llm-adapters/gemini.ts` | Google Gemini `generateContent` API | +| File | Responsibility | +| ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | +| `utils/enrich-word.ts` | Calls LLM via adapter, strips markdown, parses JSON, validates senses, builds `EnrichedSense[]`, retry/split logic | +| `utils/merge-enriched-data.ts` | Merges skeleton with LLM senses, adds `enrichedAt` and `model` | +| `utils/verify-enriched-file.ts` | Schema validation: required fields, array lengths, gender enum, translation structure | +| `utils/pipeline-timer.ts` | Tracks per-word and global metrics; unified throughput for all providers | +| `utils/llm-adapters/factory.ts` | Creates adapter based on `LLM_CONFIG.provider` | +| `utils/llm-adapters/openai-compatible.ts` | OpenAI chat completions API for local llama.cpp, OpenRouter, DeepSeek | +| `utils/llm-adapters/gemini.ts` | Google Gemini `generateContent` API with `systemInstruction` | ### Current Model @@ -353,7 +366,7 @@ The server flags evolved through trial and error on the target hardware (Intel i | Flag | Value Tried | Result | Why | | ----------------- | ----------------------------------- | ------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------- | | `-m` | `qwen3.5-4b-q4_k_m.gguf` | Works, ~47s/word | Baseline. Correct genders. 2.6GB file, tight on VRAM. | -| `-m` | `Ministral-3b-instruct.Q4_K_M.gguf` | **Broken** | Tokenizer mismatch (Tekken). Outputs gibberish regardless of template. See [Known Issues](#12-known-issues--dev-notes). | +| `-m` | `Ministral-3b-instruct.Q4_K_M.gguf` | **Broken** | Tokenizer mismatch (Tekken). Outputs gibberish regardless of template. See [Known Issues](#13-known-issues--dev-notes). | | `-m` | `qwen2.5-1.5b-instruct-q4_k_m.gguf` | Works, ~8s/word | Current. Fast but gender accuracy degraded. | | `-ngl` | `999` | Keeps | Offload all layers to GPU. Required for any speed. | | `-c` | `4096` | Wasteful | 4K context for 300-token dictionary entries wastes ~400MB VRAM. | @@ -420,8 +433,7 @@ The server flags evolved through trial and error on the target hardware (Intel i - **Gender accuracy**: Qwen2.5-1.5B systematically defaults to `neuter` for Romance languages. Under evaluation whether larger models fix this. - **Single POS**: Only nouns tested. Verbs/adjectives/adverbs need prompt adjustments. -- **Hardcoded model name**: `merge-enriched-data.ts` hardcodes `"qwen3.5-4b-q4_k_m"` regardless of actual model used. -- **Duplicated mappings**: `LANG_MAP`/`POS_MAP` exist in three separate files. +- **Pre-scanning wordlists**: Entire file read into memory before processing. Inefficient for 100k words. See [Refactor Notes](#refactor-notes). --- @@ -715,20 +727,118 @@ For each word and each candidate model: - Each sense: `sense` (string), `example` (string), `difficulty_level` in {easy, medium, hard} - Each translation: `word` (string), `gender` in {masculine, feminine, neuter, null} +Additionally, `validateSense()` in `enrich-word.ts` catches malformed senses **before** file writes, triggering retry/split instead of writing bad data. + --- -## 10. Future Extensions & Roadmap +## 10. Interactive CLI + +### Overview + +The pipeline includes an interactive CLI that asks the user to select provider, model, and batch size on each run. No editing of `config/llm.ts` required. + +### Flow + +``` +$ npx tsx pipeline.ts + +🌐 Lila Data Pipeline +───────────────────── + +[1] Use last config: online → gemini → gemini-2.5-flash-lite → batch 50 +[2] Configure new run + +> 2 + +Provider type: + [1] Local (llama.cpp) + [2] Online API + +> 2 + +Online provider: + [1] Gemini + [2] DeepSeek + [3] OpenRouter + [4] Groq + +> 1 + +Model: + [1] gemini-2.5-flash-lite (recommended for cost) + [2] gemini-2.5-pro (recommended for quality) + +> 1 + +Batch size: + [1] 1 (safest, slowest) + [2] 5 (recommended for local) + [3] 10 + [4] 20 + [5] 50 (recommended for Gemini) + [6] Custom + +> 5 + +✅ Configuration: + Provider: gemini + Model: gemini-2.5-flash-lite + API key: GEMINI_API_KEY found in environment ✓ + Batch size: 50 + +Start pipeline with these settings? [Y/n] +> Y +``` + +### Saved Config + +On first run, after confirming, write to `.pipeline-config.json`: + +```json +{ + "provider": "gemini", + "model": "gemini-2.5-flash-lite", + "batchSize": 50, + "lastRun": "2026-07-06T13:54:00Z" +} +``` + +Next run shows `[1] Use last config` at the top. + +### API Key Rules + +- **Never prompt for keys.** Check `process.env` for the provider's key. +- **If missing:** Print which env var is needed, then exit. +- **Supported env vars:** `GEMINI_API_KEY`, `DEEPSEEK_API_KEY`, `OPENROUTER_API_KEY` + +### Batch Size Recommendations + +| Provider | Recommended | Rationale | +| ------------------ | ----------- | ---------------------------------- | +| `local` (GTX 950M) | 5 | VRAM-limited, KV cache pressure | +| `local` (RTX 4090) | 20 | Fast, more VRAM | +| `gemini` | 50 | Free tier: 1,500 req/day | +| `deepseek` | 20 | 5M free tokens, balance speed/cost | +| `groq` | 50 | Very fast, rate limits generous | +| `openrouter` | 10 | 200 req/day free tier | + +--- + +## 11. Future Extensions & Roadmap ### Near-Term (Next 2-4 Weeks) -| Item | Status | Notes | -| ------------------------ | ------------ | --------------------------------------------------------------- | -| Configurable batching | **Complete** | Single `BATCH_CONFIG.size` value, retry/split logic implemented | -| 20-word torture suite | Pending | Decides gender approach and model selection | -| Qwen2.5-3B evaluation | Pending | Download and test | -| Online API testing | Pending | Gemini free tier, DeepSeek, Groq | -| Fix hardcoded model name | Pending | `merge-enriched-data.ts` hardcodes `"qwen3.5-4b"` | -| Extract shared constants | Pending | `LANG_MAP`/`POS_MAP` duplicated in 3 files | +| Item | Status | Notes | +| ---------------------- | ------------ | ------------------------------------------------------------- | +| Configurable batching | **Complete** | `config/batch.ts` with `size` and `maxRetries` | +| Retry + split logic | **Complete** | `enrichWordWithRetry`: 3 retries, then halve batch | +| Honest timing metrics | **Complete** | Unified throughput for all providers, detailed only for local | +| Auto-target languages | **Complete** | `buildSystemPrompt()` excludes source from targets | +| Validate LLM responses | **Complete** | `validateSense()` catches bad data before writes | +| Interactive CLI | Planned | Provider/model/batch selection with saved config | +| 20-word torture suite | Pending | Decides gender approach and model selection | +| Qwen2.5-3B evaluation | Pending | Download and test | +| Online API testing | Pending | Gemini free tier, DeepSeek, Groq | ### Medium-Term (1-3 Months) @@ -751,27 +861,30 @@ For each word and each candidate model: --- -## 11. Decisions Log +## 12. Decisions Log -| Date | Decision | Context | Rationale | -| ---------- | --------------------------------- | ---------------------------------------------------- | --------------------------------------------------------------------------- | -| 2026-01-04 | TanStack Router for frontend | Previous project used React Router | Simpler, type-safe routing for the trainer app | -| 2026-01-04 | Vite dev server (no Nginx) | Docker setup for glossa-web | Nginx unnecessary for dev; Vite handles HMR and proxying | -| 2026-01-17 | Backend answer verification | Security vulnerability: correctAnswer exposed in API | Moved verification to server-side, shared schemas | -| 2026-03-26 | Multi-stage Docker builds | glossa-api and glossa-web containers | Smaller images, faster deploys | -| 2026-06-16 | llama.cpp for local LLM | Need local inference on old laptop | GGUF format, OpenAI-compatible API, no dependencies | -| 2026-06-16 | Q4_K_M quantization | Balance size vs quality | Q4_K_M is the community standard for 4-bit inference | -| 2026-06-16 | `-c 2048` context | Default was 4096 | Dictionary entries need ~500 tokens max; frees VRAM | -| 2026-06-16 | `-t 2` physical cores | Default was 4 (HT threads) | Hyperthreading hurts llama.cpp performance | -| 2026-06-17 | Qwen2.5-1.5B as current model | Qwen3.5-4B too slow (47s/word) | 6x speedup (8s/word), quality under evaluation | -| 2026-06-17 | Skip Gemma 4 | E2B Q4_K_M is 3.46GB | Does not fit in 4GB VRAM; lower quants sacrifice quality | -| 2026-06-17 | Skip Ministral-3B | Tokenizer mismatch (Tekken) | Outputs gibberish regardless of template; not fixable without re-conversion | -| 2026-07-06 | Adapter pattern for LLM providers | Need to evaluate local vs cloud | `utils/llm-adapters/` with factory + types + per-provider implementations | -| 2026-07-06 | Retry + split batching | LLM JSON parse failures on larger batches | `enrichWordWithRetry` retries 3 times, then halves batch until size 1 | +| Date | Decision | Context | Rationale | +| ---------- | ----------------------------------- | -------------------------------------------------------- | --------------------------------------------------------------------------- | +| 2026-01-04 | TanStack Router for frontend | Previous project used React Router | Simpler, type-safe routing for the trainer app | +| 2026-01-04 | Vite dev server (no Nginx) | Docker setup for glossa-web | Nginx unnecessary for dev; Vite handles HMR and proxying | +| 2026-01-17 | Backend answer verification | Security vulnerability: correctAnswer exposed in API | Moved verification to server-side, shared schemas | +| 2026-03-26 | Multi-stage Docker builds | glossa-api and glossa-web containers | Smaller images, faster deploys | +| 2026-06-16 | llama.cpp for local LLM | Need local inference on old laptop | GGUF format, OpenAI-compatible API, no dependencies | +| 2026-06-16 | Q4_K_M quantization | Balance size vs quality | Q4_K_M is the community standard for 4-bit inference | +| 2026-06-16 | `-c 2048` context | Default was 4096 | Dictionary entries need ~500 tokens max; frees VRAM | +| 2026-06-16 | `-t 2` physical cores | Default was 4 (HT threads) | Hyperthreading hurts llama.cpp performance | +| 2026-06-17 | Qwen2.5-1.5B as current model | Qwen3.5-4B too slow (47s/word) | 6x speedup (8s/word), quality under evaluation | +| 2026-06-17 | Skip Gemma 4 | E2B Q4_K_M is 3.46GB | Does not fit in 4GB VRAM; lower quants sacrifice quality | +| 2026-06-17 | Skip Ministral-3B | Tokenizer mismatch (Tekken) | Outputs gibberish regardless of template; not fixable without re-conversion | +| 2026-07-06 | Adapter pattern for LLM providers | Need to evaluate local vs cloud | `utils/llm-adapters/` with factory + types + per-provider implementations | +| 2026-07-06 | Retry + split batching | LLM JSON parse failures on larger batches | `enrichWordWithRetry` retries 3 times, then halves batch until size 1 | +| 2026-07-06 | Honest timing metrics | Cloud providers don't expose prompt/completion breakdown | Unified `totalTimeMs` for all; detailed breakdown only when available | +| 2026-07-06 | Auto-target languages | Prompt hardcoded English -> de/it/es/fr | `ALL_LANGUAGES` minus source = targets; works for any source language | +| 2026-07-06 | Validate LLM responses before write | Bad data was written then warned about | `validateSense()` catches malformed responses early, triggers retry | --- -## 12. Known Issues & Dev Notes +## 13. Known Issues & Dev Notes ### glossa-web (Frontend) @@ -782,15 +895,12 @@ For each word and each candidate model: ### Data Pipeline -| Issue | Details | Severity | -| --------------------------------- | ----------------------------------------------------------------------------------------------- | -------------------------------- | -| Ministral-3B tokenizer mismatch | Tekken tokenizer not properly converted to GGUF. Model outputs gibberish. | Blocker - abandoned | -| Qwen2.5-1.5B gender hallucination | Systematic `neuter` default for Romance languages. | Under evaluation | -| Hardcoded model name | `merge-enriched-data.ts` always writes `"qwen3.5-4b-q4_k_m"` regardless of actual model. | Minor - fix before production | -| Duplicated LANG_MAP/POS_MAP | Identical mapping objects in `create-base-json.ts`, `merge-enriched-data.ts`, `enrich-word.ts`. | Minor - refactor risk | -| OpenAI-compatible timings | `json.timings` is llama.cpp-specific. Will break for OpenRouter/DeepSeek. | Medium - needs graceful fallback | -| Single POS tested | Only nouns validated. Verbs/adjectives need prompt changes. | Known limitation | -| Batch metrics averaging | Split-and-merge averages tokens/sec instead of weighting by token count. | Minor - summary stats only | +| Issue | Details | Severity | +| --------------------------------- | --------------------------------------------------------------------------- | ----------------------------------- | +| Ministral-3B tokenizer mismatch | Tekken tokenizer not properly converted to GGUF. Model outputs gibberish. | Blocker - abandoned | +| Qwen2.5-1.5B gender hallucination | Systematic `neuter` default for Romance languages. | Under evaluation | +| Pre-scanning wordlists | Entire file read into memory before processing. Inefficient for 100k words. | Medium - streaming refactor planned | +| Single POS tested | Only nouns validated. Verbs/adjectives need prompt changes. | Known limitation | ### Hardware @@ -800,9 +910,15 @@ For each word and each candidate model: | Maxwell GPU aging | No Flash Attention support, bandwidth-starved. | | Laptop thermals | Cannot run 24/7 for weeks. Batch processing required. | +### Refactor Notes + +| Note | File | Context | +| --------------------- | ------------- | -------------------------------------------------------------------------- | +| Streaming vs pre-scan | `pipeline.ts` | For 100k words, stream and batch on-the-fly instead of reading entire file | + --- -## 13. How to Run +## 14. How to Run ### Prerequisites @@ -836,37 +952,35 @@ cd ~/Downloads/llama.cpp --prio 2 ``` -### Configure Provider - -Edit `config/llm.ts`: - -```typescript -// Local -export const LLM_CONFIG = { - provider: "local" as const, - url: "http://127.0.0.1:8080/v1/chat/completions", - model: undefined, -}; - -// Gemini -export const LLM_CONFIG = { - provider: "gemini" as const, - url: "", - model: "gemini-2.5-flash-lite", -}; -``` - -### Run the Pipeline +### Run the Pipeline (Interactive CLI) ```bash cd /path/to/data-pipeline npx tsx pipeline.ts ``` +Follow the prompts to select provider, model, and batch size. + +### Run with Last Config + +```bash +cd /path/to/data-pipeline +npx tsx pipeline.ts +# Select [1] Use last config +``` + ### Expected Output ``` -Starting data pipeline... +🌐 Lila Data Pipeline +───────────────────── + +[1] Use last config: local → qwen2.5-1.5b → batch 5 +[2] Configure new run + +> 1 + +🟢 Local AI engine is connected and ready for inference! step 1: scanning the source files... ✅ Scan complete! Found 1 wordlist(s): @@ -901,6 +1015,11 @@ Reading list: [ENGLISH] -> [NOUNS] Calls: 1 Avg prompt tokens: 312 Avg completion tokens: 524 + Avg total tokens: 836 + Avg total request time: 32800ms + Avg throughput: 25.5 tok/s + + [Local breakdown] Avg prompt speed: 548.2 tok/s Avg completion speed: 18.8 tok/s @@ -909,7 +1028,7 @@ Global data pipeline run completed successfully. --- -## 14. Roadmap +## 15. Roadmap ### Phase 1: Batching (Complete) @@ -917,6 +1036,8 @@ Global data pipeline run completed successfully. | ------------------------------------- | ------------ | ------------------------------------------------------------------------------------ | | Implement configurable batch size | **Complete** | `config/batch.ts` with `size` and `maxRetries` | | Implement retry + split logic | **Complete** | `enrichWordWithRetry`: 3 retries, then halve batch | +| Honest timing metrics | **Complete** | Unified throughput for all providers | +| Validate LLM responses | **Complete** | `validateSense()` catches bad data before writes | | Verify batching doesn't break quality | Pending | Run 20-word torture suite on Qwen2.5-1.5B with batch sizes 1, 5, 15. Compare output. | | Measure speedup vs batch size | Pending | Track throughput at 1, 5, 15 on local hardware. | @@ -924,7 +1045,20 @@ Global data pipeline run completed successfully. --- -### Phase 2: Model Selection +### Phase 2: Interactive CLI + +| Task | Status | Notes | +| --------------------- | ------- | ----------------------------------------- | +| Design prompt flow | Planned | Provider → model → batch size → confirm | +| Implement CLI module | Planned | Use `readline` or `inquirer` for prompts | +| Save/load config | Planned | `.pipeline-config.json` | +| Wire into pipeline.ts | Planned | Replace static config with runtime config | + +**Goal:** No editing of TypeScript files to switch providers. + +--- + +### Phase 3: Model Selection | Task | Status | Notes | | ------------------------------------------------ | ------- | ---------------------------------------------------------------------------------------- | @@ -939,11 +1073,11 @@ Global data pipeline run completed successfully. --- -### Phase 3: Scale +### Phase 4: Scale | Task | Status | Notes | | ------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------- | -| Run 100k word pipeline | Pending | Estimated time depends on Phase 2 decision: ~10 days (local 1.5B) to ~1.5 days (Gemini batched free) to ~3-4 hours (Groq). | +| Run 100k word pipeline | Pending | Estimated time depends on Phase 3 decision: ~10 days (local 1.5B) to ~1.5 days (Gemini batched free) to ~3-4 hours (Groq). | | Spot-check output quality | Pending | Random sample of 100 entries. | | Fix gender if needed | Pending | Kaikki lookup post-processing if LLM gender remains unreliable. | | Handle failures & retries | Pending | Exponential backoff, split-and-retry for batch failures. | @@ -952,7 +1086,7 @@ Global data pipeline run completed successfully. --- -### Phase 4: Extend +### Phase 5: Extend | Task | Status | Notes | | ---------------------------- | ------- | ------------------------------------------------------------------------------------------- | @@ -969,13 +1103,11 @@ Global data pipeline run completed successfully. ### Backlog (Unscheduled) -| Task | Context | -| ------------------------------------------- | ------------------------------------------------------------------------------------------- | -| Batch API discounts | Gemini, Qwen, Azure offer 50% off for 24h SLA. Relevant if running recurring large batches. | -| Model auto-switching | Fallback to online API if local server fails mid-run. | -| Community open-source | Clean up, document, publish for other language learners. | -| Prometheus metrics | `--metrics` flag on llama-server for automated performance tracking. | -| `-c 1024` / `-b 256` experiments | Further VRAM optimization on GTX 950M. Low priority if moving to cloud. | -| Extract shared LANG_MAP/POS_MAP | Single source of truth for language/pos mappings. | -| Fix hardcoded model name | Pass actual model name through enrichment chain. | -| Graceful timing fallback for cloud adapters | Handle missing `timings` field in OpenRouter/DeepSeek responses. | +| Task | Context | +| -------------------------------- | ------------------------------------------------------------------------------------------- | +| Batch API discounts | Gemini, Qwen, Azure offer 50% off for 24h SLA. Relevant if running recurring large batches. | +| Model auto-switching | Fallback to online API if local server fails mid-run. | +| Community open-source | Clean up, document, publish for other language learners. | +| Prometheus metrics | `--metrics` flag on llama-server for automated performance tracking. | +| `-c 1024` / `-b 256` experiments | Further VRAM optimization on GTX 950M. Low priority if moving to cloud. | +| Streaming wordlist processing | Read file line-by-line and batch on-the-fly. Eliminates pre-scan memory usage. |