Spaces:
Build error
Build error
refactor: remove deprecated apkg_generator module and tests
Browse files- docs/superpowers/plans/2026-06-14-csv-for-anki-export.md +1016 -0
- export/apkg_generator.py +0 -383
- tests/apkg_generator_test.py +0 -587
docs/superpowers/plans/2026-06-14-csv-for-anki-export.md
ADDED
|
@@ -0,0 +1,1016 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# CSV-for-Anki Export Implementation Plan
|
| 2 |
+
|
| 3 |
+
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
| 4 |
+
|
| 5 |
+
**Goal:** Replace the broken `.apkg` export with a reliable Anki-compatible CSV export that uses HTML-embedded media references and Anki's native text-file import mechanism.
|
| 6 |
+
|
| 7 |
+
**Architecture:** Create a new `export/csv_for_anki.py` module mirroring `csv_export.py`'s pattern (same function signature, same media naming convention). The module writes a 2-column CSV (`Front`/`Back`) with HTML `<img>` and `<audio>` tags referencing files in a `collection.media/` subfolder, then zips it. The old `apkg_generator.py` is deleted. UI wiring in `app.py` and `widgets.py` is updated to call the new handler.
|
| 8 |
+
|
| 9 |
+
**Tech Stack:** Python stdlib (`csv`, `zipfile`, `shutil`, `html`, `pathlib`) — no new dependencies.
|
| 10 |
+
|
| 11 |
+
---
|
| 12 |
+
|
| 13 |
+
## File Map
|
| 14 |
+
|
| 15 |
+
| Action | File | Responsibility |
|
| 16 |
+
|---|---|---|
|
| 17 |
+
| **Create** | `export/csv_for_anki.py` | New Anki CSV export module (~120 lines) |
|
| 18 |
+
| **Create** | `tests/csv_for_anki_test.py` | Tests for the new module (~80 lines) |
|
| 19 |
+
| **Modify** | `app.py:340-360` | Replace `_handle_export_apkg()` with `_handle_export_csv_for_anki()` |
|
| 20 |
+
| **Modify** | `frontend/ui/widgets.py:~100 lines` | Rename event handler, change `file_types` on `export_apkg_file` from `.apkg` to `.zip` |
|
| 21 |
+
| **Delete** | `export/apkg_generator.py` | Replaced entirely by new module |
|
| 22 |
+
| **Delete** | `tests/apkg_generator_test.py` | Replaced by new test file |
|
| 23 |
+
|
| 24 |
+
---
|
| 25 |
+
|
| 26 |
+
### Task 1: Create `export/csv_for_anki.py` with core helpers
|
| 27 |
+
|
| 28 |
+
**Files:**
|
| 29 |
+
- Create: `export/csv_for_anki.py`
|
| 30 |
+
|
| 31 |
+
- [ ] **Step 1: Write the module skeleton with shared helpers**
|
| 32 |
+
|
| 33 |
+
```python
|
| 34 |
+
"""EuropaLex CSV-for-Anki Export — creates an Anki-compatible zip with HTML media references.
|
| 35 |
+
|
| 36 |
+
Produces a zipped folder containing:
|
| 37 |
+
{folder_name}/
|
| 38 |
+
cards.csv (2 columns: Front, Back — HTML embedded)
|
| 39 |
+
collection.media/ (media files following Anki convention)
|
| 40 |
+
|
| 41 |
+
Uses Anki's native text-file import mechanism with embedded media via
|
| 42 |
+
relative paths in <img> and <audio> tags.
|
| 43 |
+
"""
|
| 44 |
+
|
| 45 |
+
import csv
|
| 46 |
+
import html
|
| 47 |
+
import shutil
|
| 48 |
+
from pathlib import Path
|
| 49 |
+
|
| 50 |
+
# ISO 639-1 language abbreviation mapping — mirrors csv_export.py exactly
|
| 51 |
+
_LANGUAGE_ABBREVS: dict[str, str] = {
|
| 52 |
+
"Latvian": "LV",
|
| 53 |
+
"Spanish": "ES",
|
| 54 |
+
"French": "FR",
|
| 55 |
+
"German": "DE",
|
| 56 |
+
"Polish": "PL",
|
| 57 |
+
"Italian": "IT",
|
| 58 |
+
"Portuguese": "PT",
|
| 59 |
+
"Finnish": "FI",
|
| 60 |
+
}
|
| 61 |
+
|
| 62 |
+
# Project root for resolving relative paths — mirrors csv_export.py
|
| 63 |
+
_PROJECT_ROOT = Path(__file__).resolve().parent.parent
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def _sanitize_folder_name(scenario: str) -> str:
|
| 67 |
+
"""Convert scenario text to a filesystem-safe folder name slug.
|
| 68 |
+
|
| 69 |
+
Same implementation as csv_export._sanitize_folder_name for consistency.
|
| 70 |
+
|
| 71 |
+
Args:
|
| 72 |
+
scenario: Free-form scenario/topic string from the user.
|
| 73 |
+
|
| 74 |
+
Returns:
|
| 75 |
+
Sanitized slug suitable for use as a directory name.
|
| 76 |
+
"""
|
| 77 |
+
import re
|
| 78 |
+
slug = scenario.strip().lower()
|
| 79 |
+
slug = re.sub(r'[^a-z0-9\s_]', '', slug) # remove special chars
|
| 80 |
+
slug = re.sub(r'\s+', '_', slug) # spaces → underscores
|
| 81 |
+
slug = re.sub(r'_+', '_', slug) # collapse multiple underscores
|
| 82 |
+
return slug.strip('_')
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def _get_language_abbrev(language: str) -> str:
|
| 86 |
+
"""Return the ISO 639-1 abbreviation for a language name.
|
| 87 |
+
|
| 88 |
+
Args:
|
| 89 |
+
language: Language name (e.g., 'Latvian', 'Spanish').
|
| 90 |
+
|
| 91 |
+
Returns:
|
| 92 |
+
Two-letter ISO 639-1 code.
|
| 93 |
+
|
| 94 |
+
Raises:
|
| 95 |
+
ValueError: If the language is not in the mapping.
|
| 96 |
+
"""
|
| 97 |
+
if language not in _LANGUAGE_ABBREVS:
|
| 98 |
+
raise ValueError(
|
| 99 |
+
f"Unknown language '{language}'. "
|
| 100 |
+
f"Supported: {', '.join(sorted(_LANGUAGE_ABBREVS.keys()))}"
|
| 101 |
+
)
|
| 102 |
+
return _LANGUAGE_ABBREVS[language]
|
| 103 |
+
```
|
| 104 |
+
|
| 105 |
+
- [ ] **Step 2: Run test to verify it fails**
|
| 106 |
+
|
| 107 |
+
Run: `uv run pytest tests/csv_for_anki_test.py -v --tb=short`
|
| 108 |
+
Expected: FAIL with "ModuleNotFoundError: No module named 'export.csv_for_anki'" (file doesn't exist yet) or import errors for the missing functions.
|
| 109 |
+
|
| 110 |
+
- [ ] **Step 3: Write minimal implementation — just imports and stubs**
|
| 111 |
+
|
| 112 |
+
Add to `export/csv_for_anki.py`:
|
| 113 |
+
|
| 114 |
+
```python
|
| 115 |
+
def export_csv_for_anki(
|
| 116 |
+
cards: list[dict],
|
| 117 |
+
scenario: str,
|
| 118 |
+
cefr_level: str,
|
| 119 |
+
target_language: str,
|
| 120 |
+
) -> str:
|
| 121 |
+
"""Export cards as an Anki-compatible CSV zip with HTML media references.
|
| 122 |
+
|
| 123 |
+
Args:
|
| 124 |
+
cards: List of card dicts with keys: 'text', 'translation',
|
| 125 |
+
'audio_path' (str or None), 'image_path' (str or None).
|
| 126 |
+
scenario: Free-form scenario/topic string.
|
| 127 |
+
cefr_level: CEFR level string (e.g., 'A2', 'B1').
|
| 128 |
+
target_language: Target language name (e.g., 'Latvian').
|
| 129 |
+
|
| 130 |
+
Returns:
|
| 131 |
+
Absolute path to the generated .zip file.
|
| 132 |
+
|
| 133 |
+
Raises:
|
| 134 |
+
ValueError: If no cards provided or target_language not supported.
|
| 135 |
+
"""
|
| 136 |
+
if not cards:
|
| 137 |
+
raise ValueError("No cards provided for Anki CSV export")
|
| 138 |
+
|
| 139 |
+
lang_abbrev = _get_language_abbrev(target_language)
|
| 140 |
+
scenario_slug = _sanitize_folder_name(scenario)
|
| 141 |
+
folder_name = f"{scenario_slug}_{cefr_level}_{lang_abbrev}"
|
| 142 |
+
|
| 143 |
+
# Resolve output directory (same pattern as csv_export.py)
|
| 144 |
+
export_base = _PROJECT_ROOT / ".local" / "models" / "output" / "export"
|
| 145 |
+
export_base.mkdir(parents=True, exist_ok=True)
|
| 146 |
+
|
| 147 |
+
export_dir = export_base / folder_name
|
| 148 |
+
export_dir.mkdir(parents=True, exist_ok=True)
|
| 149 |
+
|
| 150 |
+
media_dir = export_dir / "collection.media"
|
| 151 |
+
media_dir.mkdir(parents=True, exist_ok=True)
|
| 152 |
+
|
| 153 |
+
# Build CSV rows and copy media files
|
| 154 |
+
csv_path = export_dir / "cards.csv"
|
| 155 |
+
with open(csv_path, 'w', newline='', encoding='utf-8') as csvfile:
|
| 156 |
+
writer = csv.writer(csvfile)
|
| 157 |
+
writer.writerow(['Front', 'Back'])
|
| 158 |
+
|
| 159 |
+
for i, card in enumerate(cards):
|
| 160 |
+
front_html = _build_front_html(
|
| 161 |
+
translation=card.get("translation", ""),
|
| 162 |
+
audio_path=card.get("audio_path"),
|
| 163 |
+
image_path=card.get("image_path"),
|
| 164 |
+
scenario_slug=scenario_slug,
|
| 165 |
+
cefr_level=cefr_level,
|
| 166 |
+
lang_abbrev=lang_abbrev,
|
| 167 |
+
card_index=i,
|
| 168 |
+
)
|
| 169 |
+
back_text = card.get("text", "")
|
| 170 |
+
writer.writerow([front_html, back_text])
|
| 171 |
+
|
| 172 |
+
# Create zip archive
|
| 173 |
+
zip_path = shutil.make_archive(
|
| 174 |
+
str(export_base / folder_name),
|
| 175 |
+
'zip',
|
| 176 |
+
export_dir,
|
| 177 |
+
)
|
| 178 |
+
|
| 179 |
+
return str(zip_path)
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
def _build_front_html(
|
| 183 |
+
translation: str,
|
| 184 |
+
audio_path: str | None,
|
| 185 |
+
image_path: str | None,
|
| 186 |
+
scenario_slug: str,
|
| 187 |
+
cefr_level: str,
|
| 188 |
+
lang_abbrev: str,
|
| 189 |
+
card_index: int,
|
| 190 |
+
) -> str:
|
| 191 |
+
"""Build the HTML string for the card front field.
|
| 192 |
+
|
| 193 |
+
Format: <b>translation</b><br>[<img>]<br>[<audio>]
|
| 194 |
+
Tags are omitted entirely if media paths are None/missing.
|
| 195 |
+
|
| 196 |
+
Args:
|
| 197 |
+
translation: Target-language text (HTML-escaped).
|
| 198 |
+
audio_path: Path to TTS .wav file or None.
|
| 199 |
+
image_path: Path to illustration .png file or None.
|
| 200 |
+
scenario_slug: Sanitized scenario name for media filename.
|
| 201 |
+
cefr_level: CEFR level string.
|
| 202 |
+
lang_abbrev: ISO 639-1 language code.
|
| 203 |
+
card_index: Zero-based card index.
|
| 204 |
+
|
| 205 |
+
Returns:
|
| 206 |
+
HTML string for the Front field.
|
| 207 |
+
"""
|
| 208 |
+
base_name = f"{scenario_slug}_{cefr_level}_{lang_abbrev}"
|
| 209 |
+
parts = [f"<b>{html.escape(translation)}</b>"]
|
| 210 |
+
|
| 211 |
+
# Copy image file if path exists, add <img> tag
|
| 212 |
+
if image_path and Path(image_path).exists():
|
| 213 |
+
media_filename = f"{base_name}_{card_index}.png"
|
| 214 |
+
shutil.copy2(image_path, str(media_dir / media_filename))
|
| 215 |
+
parts.append(f'<img src="collection.media/{media_filename}">')
|
| 216 |
+
|
| 217 |
+
# Copy audio file if path exists, add <audio> tag
|
| 218 |
+
if audio_path and Path(audio_path).exists():
|
| 219 |
+
media_filename = f"{base_name}_{card_index}.wav"
|
| 220 |
+
shutil.copy2(audio_path, str(media_dir / media_filename))
|
| 221 |
+
parts.append(f'<audio controls src="collection.media/{media_filename}"></audio>')
|
| 222 |
+
|
| 223 |
+
return "<br>".join(parts)
|
| 224 |
+
```
|
| 225 |
+
|
| 226 |
+
Wait — that references `media_dir` which is a local variable in the outer function. Fix by making `_build_front_html` accept the export directory:
|
| 227 |
+
|
| 228 |
+
Actually, let me restructure slightly so `_build_front_html` receives the media directory path and returns only the HTML string. The media copying happens inside it.
|
| 229 |
+
|
| 230 |
+
```python
|
| 231 |
+
def _copy_media_file(src_path: str | None, dest_dir: Path, base_name: str, card_index: int) -> str | None:
|
| 232 |
+
"""Copy a media file to the export media directory and return its filename, or None.
|
| 233 |
+
|
| 234 |
+
Args:
|
| 235 |
+
src_path: Source file path or None.
|
| 236 |
+
dest_dir: Destination media directory (collection.media/).
|
| 237 |
+
base_name: Filename prefix ({scenario}_{CEFR}_{LANG}).
|
| 238 |
+
card_index: Zero-based card index for the filename suffix.
|
| 239 |
+
|
| 240 |
+
Returns:
|
| 241 |
+
Bare media filename (e.g., 'slug_A2_LV_0.wav') or None if skipped.
|
| 242 |
+
"""
|
| 243 |
+
if not src_path or not Path(src_path).exists():
|
| 244 |
+
return None
|
| 245 |
+
ext = Path(src_path).suffix.lower()
|
| 246 |
+
media_filename = f"{base_name}_{card_index}{ext}"
|
| 247 |
+
shutil.copy2(src_path, str(dest_dir / media_filename))
|
| 248 |
+
return media_filename
|
| 249 |
+
|
| 250 |
+
|
| 251 |
+
def _build_front_html(
|
| 252 |
+
translation: str,
|
| 253 |
+
audio_path: str | None,
|
| 254 |
+
image_path: str | None,
|
| 255 |
+
export_dir: Path,
|
| 256 |
+
card_index: int,
|
| 257 |
+
) -> str:
|
| 258 |
+
"""Build the HTML string for the card front field.
|
| 259 |
+
|
| 260 |
+
Args:
|
| 261 |
+
translation: Target-language text (will be HTML-escaped).
|
| 262 |
+
audio_path: Path to TTS .wav file or None.
|
| 263 |
+
image_path: Path to illustration .png file or None.
|
| 264 |
+
export_dir: Export directory containing collection.media/.
|
| 265 |
+
card_index: Zero-based card index.
|
| 266 |
+
|
| 267 |
+
Returns:
|
| 268 |
+
HTML string for the Front field.
|
| 269 |
+
"""
|
| 270 |
+
media_dir = export_dir / "collection.media"
|
| 271 |
+
parts = [f"<b>{html.escape(translation)}</b>"]
|
| 272 |
+
|
| 273 |
+
if image_path:
|
| 274 |
+
fname = _copy_media_file(image_path, media_dir, "", card_index)
|
| 275 |
+
if fname:
|
| 276 |
+
parts.append(f'<img src="collection.media/{fname}">')
|
| 277 |
+
|
| 278 |
+
if audio_path:
|
| 279 |
+
fname = _copy_media_file(audio_path, media_dir, "", card_index)
|
| 280 |
+
if fname:
|
| 281 |
+
parts.append(f'<audio controls src="collection.media/{fname}"></audio>')
|
| 282 |
+
|
| 283 |
+
return "<br>".join(parts)
|
| 284 |
+
```
|
| 285 |
+
|
| 286 |
+
Hmm, that loses the base_name. Let me fix `_copy_media_file`:
|
| 287 |
+
|
| 288 |
+
```python
|
| 289 |
+
def _copy_media_file(src_path: str | None, dest_dir: Path, filename_prefix: str, card_index: int, ext: str) -> str | None:
|
| 290 |
+
"""Copy a media file to the export media directory and return its filename, or None.
|
| 291 |
+
|
| 292 |
+
Args:
|
| 293 |
+
src_path: Source file path or None.
|
| 294 |
+
dest_dir: Destination media directory (collection.media/).
|
| 295 |
+
filename_prefix: Filename prefix ({scenario}_{CEFR}_{LANG}).
|
| 296 |
+
card_index: Zero-based card index for the filename suffix.
|
| 297 |
+
ext: File extension including dot (e.g., '.wav', '.png').
|
| 298 |
+
|
| 299 |
+
Returns:
|
| 300 |
+
Bare media filename or None if skipped.
|
| 301 |
+
"""
|
| 302 |
+
if not src_path or not Path(src_path).exists():
|
| 303 |
+
return None
|
| 304 |
+
media_filename = f"{filename_prefix}_{card_index}{ext}"
|
| 305 |
+
shutil.copy2(src_path, str(dest_dir / media_filename))
|
| 306 |
+
return media_filename
|
| 307 |
+
```
|
| 308 |
+
|
| 309 |
+
And `_build_front_html`:
|
| 310 |
+
|
| 311 |
+
```python
|
| 312 |
+
def _build_front_html(
|
| 313 |
+
translation: str,
|
| 314 |
+
audio_path: str | None,
|
| 315 |
+
image_path: str | None,
|
| 316 |
+
export_dir: Path,
|
| 317 |
+
base_name: str,
|
| 318 |
+
card_index: int,
|
| 319 |
+
) -> str:
|
| 320 |
+
"""Build the HTML string for the card front field.
|
| 321 |
+
|
| 322 |
+
Args:
|
| 323 |
+
translation: Target-language text (will be HTML-escaped).
|
| 324 |
+
audio_path: Path to TTS .wav file or None.
|
| 325 |
+
image_path: Path to illustration .png file or None.
|
| 326 |
+
export_dir: Export directory containing collection.media/.
|
| 327 |
+
base_name: Filename prefix ({scenario}_{CEFR}_{LANG}).
|
| 328 |
+
card_index: Zero-based card index.
|
| 329 |
+
|
| 330 |
+
Returns:
|
| 331 |
+
HTML string for the Front field.
|
| 332 |
+
"""
|
| 333 |
+
media_dir = export_dir / "collection.media"
|
| 334 |
+
parts = [f"<b>{html.escape(translation)}</b>"]
|
| 335 |
+
|
| 336 |
+
if image_path:
|
| 337 |
+
fname = _copy_media_file(image_path, media_dir, base_name, card_index, ".png")
|
| 338 |
+
if fname:
|
| 339 |
+
parts.append(f'<img src="collection.media/{fname}">')
|
| 340 |
+
|
| 341 |
+
if audio_path:
|
| 342 |
+
fname = _copy_media_file(audio_path, media_dir, base_name, card_index, ".wav")
|
| 343 |
+
if fname:
|
| 344 |
+
parts.append(f'<audio controls src="collection.media/{fname}"></audio>')
|
| 345 |
+
|
| 346 |
+
return "<br>".join(parts)
|
| 347 |
+
```
|
| 348 |
+
|
| 349 |
+
And the CSV writer loop becomes:
|
| 350 |
+
|
| 351 |
+
```python
|
| 352 |
+
for i, card in enumerate(cards):
|
| 353 |
+
base_name = f"{scenario_slug}_{cefr_level}_{lang_abbrev}"
|
| 354 |
+
front_html = _build_front_html(
|
| 355 |
+
translation=card.get("translation", ""),
|
| 356 |
+
audio_path=card.get("audio_path"),
|
| 357 |
+
image_path=card.get("image_path"),
|
| 358 |
+
export_dir=export_dir,
|
| 359 |
+
base_name=base_name,
|
| 360 |
+
card_index=i,
|
| 361 |
+
)
|
| 362 |
+
back_text = card.get("text", "")
|
| 363 |
+
writer.writerow([front_html, back_text])
|
| 364 |
+
```
|
| 365 |
+
|
| 366 |
+
- [ ] **Step 4: Run test to verify it passes**
|
| 367 |
+
|
| 368 |
+
Run: `uv run pytest tests/csv_for_anki_test.py -v --tb=short`
|
| 369 |
+
Expected: All 3 tests PASS.
|
| 370 |
+
|
| 371 |
+
- [ ] **Step 5: Commit**
|
| 372 |
+
|
| 373 |
+
```bash
|
| 374 |
+
git add export/csv_for_anki.py
|
| 375 |
+
git commit -m "feat: add csv_for_anki module skeleton with helpers and core export function"
|
| 376 |
+
```
|
| 377 |
+
|
| 378 |
+
---
|
| 379 |
+
|
| 380 |
+
### Task 2: Write unit tests for `csv_for_anki.py`
|
| 381 |
+
|
| 382 |
+
**Files:**
|
| 383 |
+
- Test: `tests/csv_for_anki_test.py`
|
| 384 |
+
|
| 385 |
+
- [ ] **Step 1: Write the test file**
|
| 386 |
+
|
| 387 |
+
Create `tests/csv_for_anki_test.py`:
|
| 388 |
+
|
| 389 |
+
```python
|
| 390 |
+
"""Tests for export/csv_for_anki.py — Anki-compatible CSV export with HTML media references."""
|
| 391 |
+
|
| 392 |
+
import csv
|
| 393 |
+
import zipfile
|
| 394 |
+
from pathlib import Path
|
| 395 |
+
|
| 396 |
+
import pytest
|
| 397 |
+
|
| 398 |
+
# Import functions under test
|
| 399 |
+
from export.csv_for_anki import (
|
| 400 |
+
_LANGUAGE_ABBREVS,
|
| 401 |
+
_copy_media_file,
|
| 402 |
+
_get_language_abbrev,
|
| 403 |
+
_sanitize_folder_name,
|
| 404 |
+
export_csv_for_anki,
|
| 405 |
+
)
|
| 406 |
+
|
| 407 |
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent
|
| 408 |
+
|
| 409 |
+
|
| 410 |
+
@pytest.fixture
|
| 411 |
+
def sample_cards():
|
| 412 |
+
"""Sample card data matching the structure from Phase 2."""
|
| 413 |
+
return [
|
| 414 |
+
{
|
| 415 |
+
"text": "I love eating fresh fruits.",
|
| 416 |
+
"translation": "Me encanta comer frutas frescas.",
|
| 417 |
+
"audio_path": str(PROJECT_ROOT / "tests" / "test_outputs" / "audio" / "audio_0.wav"),
|
| 418 |
+
"image_path": str(PROJECT_ROOT / "tests" / "test_outputs" / "images" / "image_0.png"),
|
| 419 |
+
},
|
| 420 |
+
{
|
| 421 |
+
"text": "She enjoys cooking pasta.",
|
| 422 |
+
"translation": "Le encanta cocinar pasta.",
|
| 423 |
+
"audio_path": None,
|
| 424 |
+
"image_path": str(PROJECT_ROOT / "tests" / "test_outputs" / "images" / "image_1.png"),
|
| 425 |
+
},
|
| 426 |
+
{
|
| 427 |
+
"text": "The chef prepared a delicious meal.",
|
| 428 |
+
"translation": "El chef preparó una comida deliciosa.",
|
| 429 |
+
"audio_path": str(PROJECT_ROOT / "tests" / "test_outputs" / "audio" / "audio_2.wav"),
|
| 430 |
+
"image_path": None,
|
| 431 |
+
},
|
| 432 |
+
]
|
| 433 |
+
|
| 434 |
+
|
| 435 |
+
@pytest.fixture
|
| 436 |
+
def tmp_export_base(monkeypatch, tmp_path):
|
| 437 |
+
"""Fixture that patches _PROJECT_ROOT to tmp_path and returns the export base dir.
|
| 438 |
+
|
| 439 |
+
The code creates .local/models/output/export/ under _PROJECT_ROOT,
|
| 440 |
+
so files end up at tmp_path/.local/models/output/export/{folder}/cards.csv.
|
| 441 |
+
This fixture returns that export base for easy test assertions.
|
| 442 |
+
"""
|
| 443 |
+
import export.csv_for_anki as mod
|
| 444 |
+
monkeypatch.setattr(mod, '_PROJECT_ROOT', tmp_path)
|
| 445 |
+
return tmp_path / ".local" / "models" / "output" / "export"
|
| 446 |
+
|
| 447 |
+
|
| 448 |
+
class TestSanitizeFolderName:
|
| 449 |
+
"""Tests for _sanitize_folder_name helper."""
|
| 450 |
+
|
| 451 |
+
def test_simple_scenario(self):
|
| 452 |
+
assert _sanitize_folder_name("ordering coffee") == "ordering_coffee"
|
| 453 |
+
|
| 454 |
+
def test_special_chars_removed(self):
|
| 455 |
+
assert _sanitize_folder_name("hello! world?") == "hello_world"
|
| 456 |
+
|
| 457 |
+
def test_multiple_spaces_collapsed(self):
|
| 458 |
+
assert _sanitize_folder_name("many spaces here") == "many_spaces_here"
|
| 459 |
+
|
| 460 |
+
|
| 461 |
+
class TestLanguageAbbrevMapping:
|
| 462 |
+
"""Tests for _get_language_abbreviation helper."""
|
| 463 |
+
|
| 464 |
+
def test_all_languages_mapped(self):
|
| 465 |
+
expected = {
|
| 466 |
+
"Latvian": "LV", "Spanish": "ES", "French": "FR",
|
| 467 |
+
"German": "DE", "Polish": "PL", "Italian": "IT",
|
| 468 |
+
"Portuguese": "PT", "Finnish": "FI",
|
| 469 |
+
}
|
| 470 |
+
assert _LANGUAGE_ABBREVS == expected
|
| 471 |
+
|
| 472 |
+
def test_invalid_language_raises(self):
|
| 473 |
+
with pytest.raises(ValueError, match="Unknown language"):
|
| 474 |
+
_get_language_abbrev("Japanese")
|
| 475 |
+
|
| 476 |
+
|
| 477 |
+
class TestCopyMediaFile:
|
| 478 |
+
"""Tests for _copy_media_file helper."""
|
| 479 |
+
|
| 480 |
+
def test_copies_file_with_correct_name(self, tmp_path):
|
| 481 |
+
src = PROJECT_ROOT / "tests" / "test_outputs" / "audio" / "audio_0.wav"
|
| 482 |
+
dest_dir = tmp_path / "media"
|
| 483 |
+
dest_dir.mkdir()
|
| 484 |
+
result = _copy_media_file(str(src), dest_dir, "slug_A2_LV", 0, ".wav")
|
| 485 |
+
assert result == "slug_A2_LV_0.wav"
|
| 486 |
+
assert (dest_dir / "slug_A2_LV_0.wav").exists()
|
| 487 |
+
|
| 488 |
+
def test_returns_none_for_missing_path(self, tmp_path):
|
| 489 |
+
dest_dir = tmp_path / "media"
|
| 490 |
+
dest_dir.mkdir()
|
| 491 |
+
result = _copy_media_file("/nonexistent/file.wav", dest_dir, "prefix", 0, ".wav")
|
| 492 |
+
assert result is None
|
| 493 |
+
|
| 494 |
+
def test_returns_none_for_none_path(self, tmp_path):
|
| 495 |
+
dest_dir = tmp_path / "media"
|
| 496 |
+
dest_dir.mkdir()
|
| 497 |
+
result = _copy_media_file(None, dest_dir, "prefix", 0, ".wav")
|
| 498 |
+
assert result is None
|
| 499 |
+
|
| 500 |
+
|
| 501 |
+
class TestExportCsvForAnki:
|
| 502 |
+
"""Tests for the main export_csv_for_anki function."""
|
| 503 |
+
|
| 504 |
+
def test_export_returns_zip_path(self, sample_cards, tmp_export_base):
|
| 505 |
+
"""Function returns a valid file path string."""
|
| 506 |
+
zip_path = export_csv_for_anki(
|
| 507 |
+
cards=sample_cards,
|
| 508 |
+
scenario="ordering coffee",
|
| 509 |
+
cefr_level="A2",
|
| 510 |
+
target_language="Latvian",
|
| 511 |
+
)
|
| 512 |
+
assert isinstance(zip_path, str)
|
| 513 |
+
assert Path(zip_path).is_absolute()
|
| 514 |
+
assert zip_path.endswith('.zip')
|
| 515 |
+
|
| 516 |
+
def test_zip_contains_cards_csv(self, sample_cards, tmp_export_base):
|
| 517 |
+
"""CSV file exists inside the zip archive."""
|
| 518 |
+
zip_path = export_csv_for_anki(
|
| 519 |
+
cards=sample_cards,
|
| 520 |
+
scenario="ordering coffee",
|
| 521 |
+
cefr_level="A2",
|
| 522 |
+
target_language="Latvian",
|
| 523 |
+
)
|
| 524 |
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
| 525 |
+
names = zf.namelist()
|
| 526 |
+
assert any('cards.csv' in n for n in names)
|
| 527 |
+
|
| 528 |
+
def test_csv_has_front_back_columns(self, sample_cards, tmp_export_base):
|
| 529 |
+
"""CSV header row is exactly ['Front', 'Back']."""
|
| 530 |
+
export_csv_for_anki(
|
| 531 |
+
cards=sample_cards,
|
| 532 |
+
scenario="test topic",
|
| 533 |
+
cefr_level="B1",
|
| 534 |
+
target_language="Spanish",
|
| 535 |
+
)
|
| 536 |
+
csv_file = tmp_export_base / "test_topic_B1_ES" / "cards.csv"
|
| 537 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 538 |
+
reader = csv.reader(f)
|
| 539 |
+
header = next(reader)
|
| 540 |
+
assert header == ['Front', 'Back']
|
| 541 |
+
|
| 542 |
+
def test_csv_row_count_matches_cards(self, sample_cards, tmp_export_base):
|
| 543 |
+
"""One data row per card (excluding header)."""
|
| 544 |
+
export_csv_for_anki(
|
| 545 |
+
cards=sample_cards,
|
| 546 |
+
scenario="test topic",
|
| 547 |
+
cefr_level="B1",
|
| 548 |
+
target_language="Spanish",
|
| 549 |
+
)
|
| 550 |
+
csv_file = tmp_export_base / "test_topic_B1_ES" / "cards.csv"
|
| 551 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 552 |
+
reader = csv.reader(f)
|
| 553 |
+
next(reader) # skip header
|
| 554 |
+
rows = list(reader)
|
| 555 |
+
assert len(rows) == 3
|
| 556 |
+
|
| 557 |
+
def test_front_field_contains_translation(self, sample_cards, tmp_export_base):
|
| 558 |
+
"""Front HTML contains the translated text."""
|
| 559 |
+
export_csv_for_anki(
|
| 560 |
+
cards=sample_cards,
|
| 561 |
+
scenario="test topic",
|
| 562 |
+
cefr_level="B1",
|
| 563 |
+
target_language="Spanish",
|
| 564 |
+
)
|
| 565 |
+
csv_file = tmp_export_base / "test_topic_B1_ES" / "cards.csv"
|
| 566 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 567 |
+
reader = csv.reader(f)
|
| 568 |
+
next(reader) # skip header
|
| 569 |
+
row = next(reader)
|
| 570 |
+
assert "Me encanta comer frutas frescas." in row[0]
|
| 571 |
+
|
| 572 |
+
def test_front_field_contains_image_tag(self, sample_cards, tmp_export_base):
|
| 573 |
+
"""Front HTML contains <img> tag when image path exists."""
|
| 574 |
+
export_csv_for_anki(
|
| 575 |
+
cards=sample_cards,
|
| 576 |
+
scenario="ordering coffee",
|
| 577 |
+
cefr_level="A2",
|
| 578 |
+
target_language="Latvian",
|
| 579 |
+
)
|
| 580 |
+
csv_file = tmp_export_base / "ordering_coffee_A2_LV" / "cards.csv"
|
| 581 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 582 |
+
reader = csv.reader(f)
|
| 583 |
+
next(reader) # skip header
|
| 584 |
+
row = next(reader) # card 0 has image
|
| 585 |
+
assert '<img src="collection.media/' in row[0]
|
| 586 |
+
|
| 587 |
+
def test_front_field_contains_audio_tag(self, sample_cards, tmp_export_base):
|
| 588 |
+
"""Front HTML contains <audio> tag when audio path exists."""
|
| 589 |
+
export_csv_for_anki(
|
| 590 |
+
cards=sample_cards,
|
| 591 |
+
scenario="ordering coffee",
|
| 592 |
+
cefr_level="A2",
|
| 593 |
+
target_language="Latvian",
|
| 594 |
+
)
|
| 595 |
+
csv_file = tmp_export_base / "ordering_coffee_A2_LV" / "cards.csv"
|
| 596 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 597 |
+
reader = csv.reader(f)
|
| 598 |
+
next(reader) # skip header
|
| 599 |
+
row = next(reader) # card 0 has audio
|
| 600 |
+
assert '<audio controls src="collection.media/' in row[0]
|
| 601 |
+
|
| 602 |
+
def test_back_field_contains_english(self, sample_cards, tmp_export_base):
|
| 603 |
+
"""Back field contains the English source text."""
|
| 604 |
+
export_csv_for_anki(
|
| 605 |
+
cards=sample_cards,
|
| 606 |
+
scenario="test topic",
|
| 607 |
+
cefr_level="B1",
|
| 608 |
+
target_language="Spanish",
|
| 609 |
+
)
|
| 610 |
+
csv_file = tmp_export_base / "test_topic_B1_ES" / "cards.csv"
|
| 611 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 612 |
+
reader = csv.reader(f)
|
| 613 |
+
next(reader) # skip header
|
| 614 |
+
row = next(reader)
|
| 615 |
+
assert "I love eating fresh fruits." in row[1]
|
| 616 |
+
|
| 617 |
+
def test_media_files_copied_to_export(self, sample_cards, tmp_export_base):
|
| 618 |
+
"""Media files are copied into the zip archive."""
|
| 619 |
+
zip_path = export_csv_for_anki(
|
| 620 |
+
cards=sample_cards,
|
| 621 |
+
scenario="ordering coffee",
|
| 622 |
+
cefr_level="A2",
|
| 623 |
+
target_language="Latvian",
|
| 624 |
+
)
|
| 625 |
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
| 626 |
+
names = zf.namelist()
|
| 627 |
+
# Card 0 has both audio and image
|
| 628 |
+
assert any('ordering_coffee_A2_LV_0.wav' in n for n in names)
|
| 629 |
+
assert any('ordering_coffee_A2_LV_0.png' in n for n in names)
|
| 630 |
+
# Card 1 has only image
|
| 631 |
+
assert any('ordering_coffee_A2_LV_1.png' in n for n in names)
|
| 632 |
+
# Card 2 has only audio
|
| 633 |
+
assert any('ordering_coffee_A2_LV_2.wav' in n for n in names)
|
| 634 |
+
|
| 635 |
+
def test_media_in_collection_media_folder(self, sample_cards, tmp_export_base):
|
| 636 |
+
"""Media files are placed under collection.media/ inside the zip."""
|
| 637 |
+
zip_path = export_csv_for_anki(
|
| 638 |
+
cards=sample_cards,
|
| 639 |
+
scenario="ordering coffee",
|
| 640 |
+
cefr_level="A2",
|
| 641 |
+
target_language="Latvian",
|
| 642 |
+
)
|
| 643 |
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
| 644 |
+
names = zf.namelist()
|
| 645 |
+
media_files = [n for n in names if n.startswith('collection.media/')]
|
| 646 |
+
assert len(media_files) > 0
|
| 647 |
+
# All media files should be under collection.media/
|
| 648 |
+
assert all(n.startswith('collection.media/') for n in media_files)
|
| 649 |
+
|
| 650 |
+
def test_missing_media_skipped_gracefully(self, tmp_export_base):
|
| 651 |
+
"""No error when audio/image path is None or file doesn't exist."""
|
| 652 |
+
cards_no_media = [
|
| 653 |
+
{
|
| 654 |
+
"text": "No media card",
|
| 655 |
+
"translation": "Sin multimedia",
|
| 656 |
+
"audio_path": None,
|
| 657 |
+
"image_path": None,
|
| 658 |
+
},
|
| 659 |
+
{
|
| 660 |
+
"text": "Missing file card",
|
| 661 |
+
"translation": "Falta archivo",
|
| 662 |
+
"audio_path": "/nonexistent/path.wav",
|
| 663 |
+
"image_path": "/nonexistent/path.png",
|
| 664 |
+
},
|
| 665 |
+
]
|
| 666 |
+
zip_path = export_csv_for_anki(
|
| 667 |
+
cards=cards_no_media,
|
| 668 |
+
scenario="no media test",
|
| 669 |
+
cefr_level="A1",
|
| 670 |
+
target_language="Spanish",
|
| 671 |
+
)
|
| 672 |
+
assert Path(zip_path).exists()
|
| 673 |
+
|
| 674 |
+
def test_html_escaping(self, tmp_export_base):
|
| 675 |
+
"""Special characters in translation text are properly HTML-escaped."""
|
| 676 |
+
cards_with_special = [
|
| 677 |
+
{
|
| 678 |
+
"text": "Hello <world>",
|
| 679 |
+
"translation": "¡Hola & mundo!",
|
| 680 |
+
"audio_path": None,
|
| 681 |
+
"image_path": None,
|
| 682 |
+
},
|
| 683 |
+
]
|
| 684 |
+
export_csv_for_anki(
|
| 685 |
+
cards=cards_with_special,
|
| 686 |
+
scenario="greetings",
|
| 687 |
+
cefr_level="A1",
|
| 688 |
+
target_language="Spanish",
|
| 689 |
+
)
|
| 690 |
+
csv_file = tmp_export_base / "greetings_A1_ES" / "cards.csv"
|
| 691 |
+
with open(csv_file, 'r', encoding='utf-8') as f:
|
| 692 |
+
reader = csv.reader(f)
|
| 693 |
+
next(reader) # skip header
|
| 694 |
+
row = next(reader)
|
| 695 |
+
# Translation should be HTML-escaped in the <b> tag
|
| 696 |
+
assert "<world>" in row[0] or "Hola" in row[0]
|
| 697 |
+
assert "&" in row[0]
|
| 698 |
+
|
| 699 |
+
def test_empty_cards_raises_valueerror(self):
|
| 700 |
+
"""Raises ValueError when no cards provided."""
|
| 701 |
+
with pytest.raises(ValueError, match="No cards"):
|
| 702 |
+
export_csv_for_anki([], "test", "A1", "Spanish")
|
| 703 |
+
|
| 704 |
+
def test_zip_structure_has_folder_inside(self, sample_cards, tmp_export_base):
|
| 705 |
+
"""Zip contains a top-level folder (not just flat files)."""
|
| 706 |
+
zip_path = export_csv_for_anki(
|
| 707 |
+
cards=sample_cards,
|
| 708 |
+
scenario="ordering coffee",
|
| 709 |
+
cefr_level="A2",
|
| 710 |
+
target_language="Latvian",
|
| 711 |
+
)
|
| 712 |
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
| 713 |
+
names = zf.namelist()
|
| 714 |
+
# Should have folder prefix like ordering_coffee_A2_LV/cards.csv
|
| 715 |
+
assert any('ordering_coffee_A2_LV' in n for n in names)
|
| 716 |
+
```
|
| 717 |
+
|
| 718 |
+
- [ ] **Step 2: Run tests to verify they all fail first**
|
| 719 |
+
|
| 720 |
+
Run: `uv run pytest tests/csv_for_anki_test.py -v --tb=short`
|
| 721 |
+
Expected: FAIL — import errors (functions not yet implemented), or assertion failures if skeleton is present.
|
| 722 |
+
|
| 723 |
+
- [ ] **Step 3: Run tests against the implementation from Task 1**
|
| 724 |
+
|
| 725 |
+
Run: `uv run pytest tests/csv_for_anki_test.py -v --tb=short`
|
| 726 |
+
Expected: All ~20 tests PASS.
|
| 727 |
+
|
| 728 |
+
- [ ] **Step 4: Commit**
|
| 729 |
+
|
| 730 |
+
```bash
|
| 731 |
+
git add tests/csv_for_anki_test.py
|
| 732 |
+
git commit -m "test: add comprehensive tests for csv_for_anki module"
|
| 733 |
+
```
|
| 734 |
+
|
| 735 |
+
---
|
| 736 |
+
|
| 737 |
+
### Task 3: Wire up `app.py` — replace `_handle_export_apkg` with `_handle_export_csv_for_anki`
|
| 738 |
+
|
| 739 |
+
**Files:**
|
| 740 |
+
- Modify: `app.py:~340-360` (the `_handle_export_apkg` function)
|
| 741 |
+
|
| 742 |
+
- [ ] **Step 1: Replace the handler function**
|
| 743 |
+
|
| 744 |
+
Replace the existing `_handle_export_apkg()` function in `app.py`:
|
| 745 |
+
|
| 746 |
+
```python
|
| 747 |
+
# OLD CODE (delete):
|
| 748 |
+
def _handle_export_apkg(
|
| 749 |
+
scenario: str,
|
| 750 |
+
cefr_level: str,
|
| 751 |
+
target_language: str,
|
| 752 |
+
) -> str | None:
|
| 753 |
+
"""Export current cards as an Anki package (.apkg).
|
| 754 |
+
|
| 755 |
+
Returns the absolute path to the generated .apkg file for Gradio DownloadButton.
|
| 756 |
+
Returns None if no cards to export or export failed.
|
| 757 |
+
"""
|
| 758 |
+
if not _current_cards:
|
| 759 |
+
logger.warning("APKG export: no cards to export")
|
| 760 |
+
return None
|
| 761 |
+
|
| 762 |
+
try:
|
| 763 |
+
from core.types import CEFRLevel
|
| 764 |
+
from export.apkg_generator import generate_apkg_package
|
| 765 |
+
|
| 766 |
+
cefr = CEFRLevel(cefr_level)
|
| 767 |
+
apkg_path = generate_apkg_package(_current_cards, scenario, cefr_level, target_language)
|
| 768 |
+
return apkg_path
|
| 769 |
+
except Exception as e:
|
| 770 |
+
logger.error("APKG export failed: %s", e, exc_info=True)
|
| 771 |
+
return None
|
| 772 |
+
```
|
| 773 |
+
|
| 774 |
+
With NEW CODE:
|
| 775 |
+
|
| 776 |
+
```python
|
| 777 |
+
def _handle_export_csv_for_anki(
|
| 778 |
+
scenario: str,
|
| 779 |
+
cefr_level: str,
|
| 780 |
+
target_language: str,
|
| 781 |
+
) -> str | None:
|
| 782 |
+
"""Export current cards as an Anki-compatible CSV zip.
|
| 783 |
+
|
| 784 |
+
Returns the absolute path to the generated .zip file for Gradio DownloadButton.
|
| 785 |
+
Returns None if no cards to export or export failed.
|
| 786 |
+
"""
|
| 787 |
+
if not _current_cards:
|
| 788 |
+
logger.warning("Anki CSV export: no cards to export")
|
| 789 |
+
return None
|
| 790 |
+
|
| 791 |
+
try:
|
| 792 |
+
from core.types import CEFRLevel
|
| 793 |
+
from export.csv_for_anki import export_csv_for_anki
|
| 794 |
+
|
| 795 |
+
cefr = CEFRLevel(cefr_level)
|
| 796 |
+
zip_path = export_csv_for_anki(_current_cards, scenario, cefr_level, target_language)
|
| 797 |
+
return zip_path
|
| 798 |
+
except Exception as e:
|
| 799 |
+
logger.error("Anki CSV export failed: %s", e, exc_info=True)
|
| 800 |
+
return None
|
| 801 |
+
```
|
| 802 |
+
|
| 803 |
+
- [ ] **Step 2: Run smoke test**
|
| 804 |
+
|
| 805 |
+
Run: `uv run pytest tests/smoke_test.py -v`
|
| 806 |
+
Expected: PASS — verifies all imports still work (the old import of `apkg_generator` is gone, new import of `csv_for_anki` works).
|
| 807 |
+
|
| 808 |
+
- [ ] **Step 3: Commit**
|
| 809 |
+
|
| 810 |
+
```bash
|
| 811 |
+
git add app.py
|
| 812 |
+
git commit -m "refactor: replace _handle_export_apkg with _handle_export_csv_for_anki"
|
| 813 |
+
```
|
| 814 |
+
|
| 815 |
+
---
|
| 816 |
+
|
| 817 |
+
### Task 4: Wire up `widgets.py` — rename handler, update file types, update state transitions
|
| 818 |
+
|
| 819 |
+
**Files:**
|
| 820 |
+
- Modify: `frontend/ui/widgets.py` (event handler rename, file type change, state transition updates)
|
| 821 |
+
|
| 822 |
+
- [ ] **Step 1: Rename the event handler function**
|
| 823 |
+
|
| 824 |
+
Find and replace in `frontend/ui/widgets.py`:
|
| 825 |
+
|
| 826 |
+
```python
|
| 827 |
+
# OLD (delete):
|
| 828 |
+
def _handle_export_apkg_event(scenario: str, cefr_level: str, target_language: str):
|
| 829 |
+
"""Export current cards as Anki package (.apkg).
|
| 830 |
+
|
| 831 |
+
Sets the generated .apkg file path as the value of export_apkg_file component,
|
| 832 |
+
which Gradio renders as a downloadable file link.
|
| 833 |
+
"""
|
| 834 |
+
from frontend.ui.cards import generate_progress_html
|
| 835 |
+
|
| 836 |
+
if not _app_module._current_cards:
|
| 837 |
+
return generate_progress_html(0, "\u26a0\ufe0f No cards to export."), None, gr.File(visible=False)
|
| 838 |
+
|
| 839 |
+
try:
|
| 840 |
+
apkg_path = _app_module._handle_export_apkg(scenario, cefr_level, target_language)
|
| 841 |
+
if apkg_path is None:
|
| 842 |
+
return generate_progress_html(0, "\u26a0\ufe0f Export failed."), None, gr.File(visible=False)
|
| 843 |
+
# Show the file for download — gr.File component renders it as a clickable link
|
| 844 |
+
return generate_progress_html(100, "Export complete! Click the file below to download."), apkg_path, gr.File(visible=True)
|
| 845 |
+
except Exception as e:
|
| 846 |
+
logger = logging.getLogger(__name__)
|
| 847 |
+
logger.error("APKG export failed: %s", e, exc_info=True)
|
| 848 |
+
return generate_progress_html(0, f"\u26a0\ufe0f Export failed: {e}"), None, gr.File(visible=False)
|
| 849 |
+
|
| 850 |
+
# APKG Export button click — generates .apkg and shows it in gr.File for download
|
| 851 |
+
export_apkg_btn.click(
|
| 852 |
+
fn=_handle_export_apkg_event,
|
| 853 |
+
inputs=[scenario_input, cefr_dropdown, language_dropdown],
|
| 854 |
+
outputs=[progress_html, export_apkg_file, export_apkg_file],
|
| 855 |
+
)
|
| 856 |
+
```
|
| 857 |
+
|
| 858 |
+
With NEW:
|
| 859 |
+
|
| 860 |
+
```python
|
| 861 |
+
def _handle_export_csv_for_anki_event(scenario: str, cefr_level: str, target_language: str):
|
| 862 |
+
"""Export current cards as Anki-compatible CSV zip.
|
| 863 |
+
|
| 864 |
+
Sets the generated .zip file path as the value of export_apkg_file component,
|
| 865 |
+
which Gradio renders as a downloadable file link.
|
| 866 |
+
"""
|
| 867 |
+
from frontend.ui.cards import generate_progress_html
|
| 868 |
+
|
| 869 |
+
if not _app_module._current_cards:
|
| 870 |
+
return generate_progress_html(0, "\u26a0\ufe0f No cards to export."), None, gr.File(visible=False)
|
| 871 |
+
|
| 872 |
+
try:
|
| 873 |
+
zip_path = _app_module._handle_export_csv_for_anki(scenario, cefr_level, target_language)
|
| 874 |
+
if zip_path is None:
|
| 875 |
+
return generate_progress_html(0, "\u26a0\ufe0f Export failed."), None, gr.File(visible=False)
|
| 876 |
+
# Show the file for download — gr.File component renders it as a clickable link
|
| 877 |
+
return generate_progress_html(100, "Export complete! Click the file below to download."), zip_path, gr.File(visible=True)
|
| 878 |
+
except Exception as e:
|
| 879 |
+
logger = logging.getLogger(__name__)
|
| 880 |
+
logger.error("Anki CSV export failed: %s", e, exc_info=True)
|
| 881 |
+
return generate_progress_html(0, f"\u26a0\ufe0f Export failed: {e}"), None, gr.File(visible=False)
|
| 882 |
+
|
| 883 |
+
# Anki CSV Export button click — generates zip and shows it in gr.File for download
|
| 884 |
+
export_apkg_btn.click(
|
| 885 |
+
fn=_handle_export_csv_for_anki_event,
|
| 886 |
+
inputs=[scenario_input, cefr_dropdown, language_dropdown],
|
| 887 |
+
outputs=[progress_html, export_apkg_file, export_apkg_file],
|
| 888 |
+
)
|
| 889 |
+
```
|
| 890 |
+
|
| 891 |
+
- [ ] **Step 2: Update `file_types` on the `export_apkg_file` Gradio component**
|
| 892 |
+
|
| 893 |
+
Find this in the widget creation section (around line ~200):
|
| 894 |
+
|
| 895 |
+
```python
|
| 896 |
+
# OLD:
|
| 897 |
+
export_apkg_file = gr.File(
|
| 898 |
+
label="Download Anki Cards", file_types=[".apkg"], visible=False
|
| 899 |
+
)
|
| 900 |
+
```
|
| 901 |
+
|
| 902 |
+
Replace with:
|
| 903 |
+
|
| 904 |
+
```python
|
| 905 |
+
# NEW:
|
| 906 |
+
export_apkg_file = gr.File(
|
| 907 |
+
label="Download Anki Cards", file_types=[".zip"], visible=False
|
| 908 |
+
)
|
| 909 |
+
```
|
| 910 |
+
|
| 911 |
+
- [ ] **Step 3: Update `_on_media_generation_complete` to reference the new handler name in log messages**
|
| 912 |
+
|
| 913 |
+
Find `_on_media_generation_complete` and verify it doesn't reference "APKG" — it currently just enables buttons, so no change needed. Confirm by reading the function body; if it only returns Gradio component updates (no string messages), skip this step.
|
| 914 |
+
|
| 915 |
+
- [ ] **Step 4: Run smoke test**
|
| 916 |
+
|
| 917 |
+
Run: `uv run pytest tests/smoke_test.py -v`
|
| 918 |
+
Expected: PASS — verifies all imports work and the Gradio app can be constructed.
|
| 919 |
+
|
| 920 |
+
- [ ] **Step 5: Commit**
|
| 921 |
+
|
| 922 |
+
```bash
|
| 923 |
+
git add frontend/ui/widgets.py
|
| 924 |
+
git commit -m "refactor: rename export handler to csv_for_anki, update file_types to .zip"
|
| 925 |
+
```
|
| 926 |
+
|
| 927 |
+
---
|
| 928 |
+
|
| 929 |
+
### Task 5: Delete old files and run full test suite
|
| 930 |
+
|
| 931 |
+
**Files:**
|
| 932 |
+
- Delete: `export/apkg_generator.py`
|
| 933 |
+
- Delete: `tests/apkg_generator_test.py`
|
| 934 |
+
|
| 935 |
+
- [ ] **Step 1: Remove old files**
|
| 936 |
+
|
| 937 |
+
```bash
|
| 938 |
+
rm export/apkg_generator.py
|
| 939 |
+
rm tests/apkg_generator_test.py
|
| 940 |
+
```
|
| 941 |
+
|
| 942 |
+
- [ ] **Step 2: Run full test suite**
|
| 943 |
+
|
| 944 |
+
Run: `uv run pytest tests/ -v`
|
| 945 |
+
Expected: All tests PASS — the old apkg tests are gone, new csv_for_anki tests pass.
|
| 946 |
+
|
| 947 |
+
- [ ] **Step 3: Verify no remaining references to deleted code**
|
| 948 |
+
|
| 949 |
+
Run: `grep -rn "apkg_generator\|generate_apkg_package\|_handle_export_apkg\b" --include="*.py" .`
|
| 950 |
+
Expected: No matches (all references have been replaced).
|
| 951 |
+
|
| 952 |
+
- [ ] **Step 4: Commit**
|
| 953 |
+
|
| 954 |
+
```bash
|
| 955 |
+
git rm export/apkg_generator.py tests/apkg_generator_test.py
|
| 956 |
+
git commit -m "refactor: remove deprecated apkg_generator module and tests"
|
| 957 |
+
```
|
| 958 |
+
|
| 959 |
+
---
|
| 960 |
+
|
| 961 |
+
### Task 6: Final verification — smoke test + full suite + app launch check
|
| 962 |
+
|
| 963 |
+
**Files:** None (verification only)
|
| 964 |
+
|
| 965 |
+
- [ ] **Step 1: Run the full test suite one final time**
|
| 966 |
+
|
| 967 |
+
Run: `uv run pytest tests/ -v`
|
| 968 |
+
Expected: All tests PASS.
|
| 969 |
+
|
| 970 |
+
- [ ] **Step 2: Verify import chain is clean**
|
| 971 |
+
|
| 972 |
+
Run: `python -c "from export.csv_for_anki import export_csv_for_anki; from app import _handle_export_csv_for_anki; print('OK')"`
|
| 973 |
+
Expected: `OK` printed with no errors.
|
| 974 |
+
|
| 975 |
+
- [ ] **Step 3: Verify the Gradio app constructs without errors**
|
| 976 |
+
|
| 977 |
+
Run: `python -c "from frontend.ui.widgets import build_ui; demo = build_ui(); print('App constructed OK')"`
|
| 978 |
+
Expected: `App constructed OK` printed with no errors.
|
| 979 |
+
|
| 980 |
+
- [ ] **Step 4: Final commit**
|
| 981 |
+
|
| 982 |
+
```bash
|
| 983 |
+
git add -A
|
| 984 |
+
git commit -m "test: verify full test suite passes and app constructs cleanly"
|
| 985 |
+
```
|
| 986 |
+
|
| 987 |
+
---
|
| 988 |
+
|
| 989 |
+
## Self-Review Checklist
|
| 990 |
+
|
| 991 |
+
**1. Spec coverage:**
|
| 992 |
+
- ✅ New `csv_for_anki.py` module with 2-column CSV (Front/Back) → Task 1
|
| 993 |
+
- ✅ HTML-embedded `<img>` and `<audio>` tags in Front field → Task 1 (`_build_front_html`)
|
| 994 |
+
- ✅ `collection.media/` subfolder for media files → Task 1 (`media_dir = export_dir / "collection.media"`)
|
| 995 |
+
- ✅ Media naming convention `{scenario_slug}_{CEFR}_{LANG}_{index}.{ext}` → Task 1 (`_copy_media_file`)
|
| 996 |
+
- ✅ Same function signature as existing exports → Task 1 (matches `export_csv_zip` signature)
|
| 997 |
+
- ✅ HTML escaping of translation text → Task 1 (`html.escape(translation)`)
|
| 998 |
+
- ✅ Omit empty tags (no `<img>` when no image) → Task 1 (conditional `if fname:`)
|
| 999 |
+
- ✅ UI button repurposed (same button, new handler) → Task 4
|
| 1000 |
+
- ✅ `file_types` changed from `.apkg` to `.zip` → Task 4
|
| 1001 |
+
- ✅ Old `apkg_generator.py` deleted → Task 5
|
| 1002 |
+
- ✅ Old tests deleted → Task 5
|
| 1003 |
+
|
| 1004 |
+
**2. Placeholder scan:** No "TBD", "TODO", "implement later", or "similar to" patterns found. All code is concrete.
|
| 1005 |
+
|
| 1006 |
+
**3. Type consistency:**
|
| 1007 |
+
- Function signature matches spec: `export_csv_for_anki(cards, scenario, cefr_level, target_language) -> str`
|
| 1008 |
+
- Card dict keys match existing code: `text`, `translation`, `audio_path`, `image_path`
|
| 1009 |
+
- `_copy_media_file` helper uses explicit `ext` parameter (not derived from path) to avoid ambiguity
|
| 1010 |
+
- Language abbrev mapping is identical between `csv_export.py` and `csv_for_anki.py`
|
| 1011 |
+
|
| 1012 |
+
**4. Edge cases covered in tests:**
|
| 1013 |
+
- Missing media files (None paths, non-existent paths) → `test_missing_media_skipped_gracefully`
|
| 1014 |
+
- HTML escaping of special characters → `test_html_escaping`
|
| 1015 |
+
- Empty cards list → `test_empty_cards_raises_valueerror`
|
| 1016 |
+
- Media deduplication: NOT explicitly tested — but the new module copies per-card (not shared across cards), so no dedup needed. Each card gets its own copy in `collection.media/`.
|
export/apkg_generator.py
DELETED
|
@@ -1,383 +0,0 @@
|
|
| 1 |
-
"""EuropaLex .apkg Generator — Creates Anki package files from card data.
|
| 2 |
-
|
| 3 |
-
Uses genanki for database structure and post-processes the generated zip
|
| 4 |
-
to inject media files (.wav, .png) with correct MD5-hashed filenames
|
| 5 |
-
and update the media JSON manifest.
|
| 6 |
-
"""
|
| 7 |
-
|
| 8 |
-
import hashlib
|
| 9 |
-
import html
|
| 10 |
-
import json
|
| 11 |
-
import logging
|
| 12 |
-
from pathlib import Path
|
| 13 |
-
|
| 14 |
-
logger = logging.getLogger(__name__)
|
| 15 |
-
|
| 16 |
-
# ─── Model Definition ──────────────────────────────────────────────
|
| 17 |
-
|
| 18 |
-
MODEL_ID = 1607392319 # Hardcoded unique ID (30-bit unsigned int)
|
| 19 |
-
MODEL_NAME = "EuropaLex Flashcard"
|
| 20 |
-
FIELDS = [
|
| 21 |
-
{"name": "Translation"}, # Front side: target language text
|
| 22 |
-
{"name": "English"}, # Back side: English source text
|
| 23 |
-
{"name": "Audio"}, # HTML audio tag for TTS
|
| 24 |
-
{"name": "Image"}, # HTML img tag for illustration
|
| 25 |
-
]
|
| 26 |
-
TEMPLATE = {
|
| 27 |
-
"name": "Card 1",
|
| 28 |
-
"qfmt": "{{Translation}}\n{{Image}}\n{{Audio}}", # front side
|
| 29 |
-
"afmt": "{{FrontSide}}<hr id=answer>{{English}}", # back side
|
| 30 |
-
}
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
def _create_model() -> "genanki.Model":
|
| 34 |
-
"""Create a genanki Model with EuropaLex field definitions.
|
| 35 |
-
|
| 36 |
-
Returns:
|
| 37 |
-
Configured genanki.Model instance.
|
| 38 |
-
"""
|
| 39 |
-
import genanki
|
| 40 |
-
return genanki.Model(
|
| 41 |
-
model_id=MODEL_ID,
|
| 42 |
-
name=MODEL_NAME,
|
| 43 |
-
fields=FIELDS,
|
| 44 |
-
templates=[TEMPLATE],
|
| 45 |
-
)
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
def _extract_filename(path: str | None) -> str:
|
| 49 |
-
"""Extract bare filename from a path string, or return empty string.
|
| 50 |
-
|
| 51 |
-
Args:
|
| 52 |
-
path: File path string or None.
|
| 53 |
-
|
| 54 |
-
Returns:
|
| 55 |
-
Bare filename (e.g., 'hello_A2_LV_0.wav') or empty string.
|
| 56 |
-
"""
|
| 57 |
-
if not path:
|
| 58 |
-
return ""
|
| 59 |
-
return Path(path).name
|
| 60 |
-
|
| 61 |
-
|
| 62 |
-
def _create_note(
|
| 63 |
-
model: "genanki.Model",
|
| 64 |
-
translation: str,
|
| 65 |
-
english: str,
|
| 66 |
-
audio_path: str | None = None,
|
| 67 |
-
image_path: str | None = None,
|
| 68 |
-
) -> "genanki.Note":
|
| 69 |
-
"""Create a genanki Note with EuropaLex field mapping.
|
| 70 |
-
|
| 71 |
-
Fields are HTML-escaped. Media references use original filenames
|
| 72 |
-
(Anki resolves them to hashed files in the package).
|
| 73 |
-
|
| 74 |
-
Args:
|
| 75 |
-
model: The genanki.Model this note belongs to.
|
| 76 |
-
translation: Target-language text (front side).
|
| 77 |
-
english: English source text (back side).
|
| 78 |
-
audio_path: Path to TTS .wav file or None.
|
| 79 |
-
image_path: Path to illustration .png file or None.
|
| 80 |
-
|
| 81 |
-
Returns:
|
| 82 |
-
Configured genanki.Note instance.
|
| 83 |
-
"""
|
| 84 |
-
import genanki
|
| 85 |
-
|
| 86 |
-
# HTML-escape text fields
|
| 87 |
-
translation_escaped = html.escape(translation) if translation else ""
|
| 88 |
-
english_escaped = html.escape(english) if english else ""
|
| 89 |
-
|
| 90 |
-
# Build audio field: <audio controls src="filename.wav"> or empty
|
| 91 |
-
audio_filename = _extract_filename(audio_path)
|
| 92 |
-
audio_field = (
|
| 93 |
-
f'<audio controls src="{audio_filename}"></audio>'
|
| 94 |
-
if audio_filename
|
| 95 |
-
else ""
|
| 96 |
-
)
|
| 97 |
-
|
| 98 |
-
# Build image field: <img src="filename.png" style="max-width:100%"> or empty
|
| 99 |
-
image_filename = _extract_filename(image_path)
|
| 100 |
-
image_field = (
|
| 101 |
-
f'<img src="{image_filename}" style="max-width:100%">'
|
| 102 |
-
if image_filename
|
| 103 |
-
else ""
|
| 104 |
-
)
|
| 105 |
-
|
| 106 |
-
return genanki.Note(
|
| 107 |
-
model=model,
|
| 108 |
-
fields=[translation_escaped, english_escaped, audio_field, image_field],
|
| 109 |
-
)
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
def _sanitize_folder_name(scenario: str) -> str:
|
| 113 |
-
"""Convert scenario text to a filesystem-safe folder name slug.
|
| 114 |
-
|
| 115 |
-
Lowercase, remove special characters (keep alphanumeric, spaces, underscores),
|
| 116 |
-
replace spaces with underscores, collapse multiple spaces, strip leading/trailing underscores.
|
| 117 |
-
|
| 118 |
-
Args:
|
| 119 |
-
scenario: Free-form scenario string.
|
| 120 |
-
|
| 121 |
-
Returns:
|
| 122 |
-
Slug suitable for use as a folder or deck name.
|
| 123 |
-
"""
|
| 124 |
-
import re
|
| 125 |
-
slug = scenario.lower()
|
| 126 |
-
slug = re.sub(r'[^a-z0-9_ ]', '', slug)
|
| 127 |
-
slug = re.sub(r'\s+', '_', slug)
|
| 128 |
-
slug = slug.strip('_')
|
| 129 |
-
return slug
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
def _get_language_abbrev(language: str) -> str:
|
| 133 |
-
"""Return the ISO 639-1 abbreviation for a language name.
|
| 134 |
-
|
| 135 |
-
Args:
|
| 136 |
-
language: Language name (e.g., 'Latvian', 'Spanish').
|
| 137 |
-
|
| 138 |
-
Returns:
|
| 139 |
-
Two-letter ISO 639-1 code.
|
| 140 |
-
|
| 141 |
-
Raises:
|
| 142 |
-
ValueError: If the language is not in the supported mapping.
|
| 143 |
-
"""
|
| 144 |
-
_LANGUAGE_ABBREVS = {
|
| 145 |
-
"Latvian": "LV",
|
| 146 |
-
"Spanish": "ES",
|
| 147 |
-
"French": "FR",
|
| 148 |
-
"German": "DE",
|
| 149 |
-
"Polish": "PL",
|
| 150 |
-
"Italian": "IT",
|
| 151 |
-
"Portuguese": "PT",
|
| 152 |
-
"Finnish": "FI",
|
| 153 |
-
}
|
| 154 |
-
if language not in _LANGUAGE_ABBREVS:
|
| 155 |
-
raise ValueError(
|
| 156 |
-
f"Unknown language '{language}'. "
|
| 157 |
-
f"Supported: {', '.join(sorted(_LANGUAGE_ABBREVS.keys()))}"
|
| 158 |
-
)
|
| 159 |
-
return _LANGUAGE_ABBREVS[language]
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
def _create_package(
|
| 163 |
-
notes: list["genanki.Note"],
|
| 164 |
-
scenario: str,
|
| 165 |
-
cefr_level: str,
|
| 166 |
-
target_language: str,
|
| 167 |
-
) -> str:
|
| 168 |
-
"""Create a genanki Package (.apkg) from notes and return its path.
|
| 169 |
-
|
| 170 |
-
Args:
|
| 171 |
-
notes: List of genanki.Note instances.
|
| 172 |
-
scenario: Free-form scenario/topic string (used in deck name).
|
| 173 |
-
cefr_level: CEFR level string (e.g., 'A2', 'B1').
|
| 174 |
-
target_language: Target language name (e.g., 'Latvian').
|
| 175 |
-
|
| 176 |
-
Returns:
|
| 177 |
-
Absolute path to the generated .apkg file.
|
| 178 |
-
"""
|
| 179 |
-
import genanki
|
| 180 |
-
import tempfile
|
| 181 |
-
import uuid
|
| 182 |
-
|
| 183 |
-
# Build deck name using same convention as CSV export
|
| 184 |
-
scenario_slug = _sanitize_folder_name(scenario)
|
| 185 |
-
lang_abbrev = _get_language_abbrev(target_language)
|
| 186 |
-
deck_name = f"{scenario_slug}_{cefr_level}_{lang_abbrev}"
|
| 187 |
-
|
| 188 |
-
deck = genanki.Deck(
|
| 189 |
-
deck_id=int(uuid.uuid4().hex[:8], 16),
|
| 190 |
-
name=deck_name,
|
| 191 |
-
)
|
| 192 |
-
|
| 193 |
-
for note in notes:
|
| 194 |
-
deck.add_note(note)
|
| 195 |
-
|
| 196 |
-
# Write to temp dir — caller decides where to save
|
| 197 |
-
with tempfile.NamedTemporaryFile(suffix='.apkg', delete=False) as f:
|
| 198 |
-
pkg = genanki.Package(deck)
|
| 199 |
-
pkg.write_to_file(f.name)
|
| 200 |
-
return f.name
|
| 201 |
-
|
| 202 |
-
|
| 203 |
-
def _fix_conf_model(existing_entries: dict[str, bytes]) -> None:
|
| 204 |
-
"""Fix conf.curModel in the Anki database to reference our MODEL_ID.
|
| 205 |
-
|
| 206 |
-
genanki sets conf.curModel to a random default model ID that doesn't
|
| 207 |
-
match our custom MODEL_ID. Without this fix, Anki can't find the model
|
| 208 |
-
during import and throws "A number was invalid or out of range".
|
| 209 |
-
|
| 210 |
-
Modifies existing_entries in-place with the fixed database bytes.
|
| 211 |
-
|
| 212 |
-
Args:
|
| 213 |
-
existing_entries: Dict of zip entry name → bytes (from _inject_media).
|
| 214 |
-
"""
|
| 215 |
-
if 'collection.anki2' not in existing_entries:
|
| 216 |
-
return
|
| 217 |
-
|
| 218 |
-
import sqlite3
|
| 219 |
-
import tempfile
|
| 220 |
-
|
| 221 |
-
tmp_db = tempfile.NamedTemporaryFile(suffix='.db', delete=False)
|
| 222 |
-
tmp_db.write(existing_entries['collection.anki2'])
|
| 223 |
-
tmp_db.close()
|
| 224 |
-
|
| 225 |
-
conn = sqlite3.connect(str(tmp_db.name))
|
| 226 |
-
cur = conn.cursor()
|
| 227 |
-
cur.execute('SELECT conf FROM col')
|
| 228 |
-
conf_raw = cur.fetchone()[0]
|
| 229 |
-
conf = json.loads(conf_raw)
|
| 230 |
-
conf['curModel'] = str(MODEL_ID)
|
| 231 |
-
cur.execute('UPDATE col SET conf = ? WHERE id = 1', (json.dumps(conf),))
|
| 232 |
-
conn.commit()
|
| 233 |
-
conn.close()
|
| 234 |
-
|
| 235 |
-
existing_entries['collection.anki2'] = Path(tmp_db.name).read_bytes()
|
| 236 |
-
Path(tmp_db.name).unlink(missing_ok=True)
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
def _inject_media(
|
| 240 |
-
apkg_path: str,
|
| 241 |
-
cards: list[dict],
|
| 242 |
-
) -> None:
|
| 243 |
-
"""Inject media files (.wav, .png) into an existing .apkg zip.
|
| 244 |
-
|
| 245 |
-
For each unique audio/image path in cards:
|
| 246 |
-
1. Compute MD5 hash of file content (Anki's media naming convention)
|
| 247 |
-
2. Write the file into the zip under the hashed name
|
| 248 |
-
3. Update the media JSON manifest: {hash.ext} → {original_filename.ext}
|
| 249 |
-
|
| 250 |
-
Also fixes conf.curModel to reference our custom MODEL_ID, which
|
| 251 |
-
genanki leaves pointing to a non-existent default model.
|
| 252 |
-
|
| 253 |
-
Deduplicates by content hash — same file injected only once.
|
| 254 |
-
Skips files that don't exist on disk (logged as warning).
|
| 255 |
-
|
| 256 |
-
Args:
|
| 257 |
-
apkg_path: Path to the .apkg zip file generated by genanki.
|
| 258 |
-
cards: List of card dicts with 'audio_path' and 'image_path' keys.
|
| 259 |
-
"""
|
| 260 |
-
import zipfile
|
| 261 |
-
|
| 262 |
-
# Collect unique media files to inject (dedup by absolute path)
|
| 263 |
-
seen_paths = set()
|
| 264 |
-
media_files = [] # list of (source_path, original_filename, ext)
|
| 265 |
-
|
| 266 |
-
for card in cards:
|
| 267 |
-
for path_key in ('audio_path', 'image_path'):
|
| 268 |
-
src = card.get(path_key)
|
| 269 |
-
if not src or not Path(src).exists():
|
| 270 |
-
continue
|
| 271 |
-
abs_src = str(Path(src).resolve())
|
| 272 |
-
if abs_src in seen_paths:
|
| 273 |
-
continue
|
| 274 |
-
seen_paths.add(abs_src)
|
| 275 |
-
|
| 276 |
-
ext = Path(src).suffix.lower()
|
| 277 |
-
if ext not in ('.wav', '.png'):
|
| 278 |
-
logger.warning("Skipping unsupported media type: %s", src)
|
| 279 |
-
continue
|
| 280 |
-
|
| 281 |
-
original_filename = Path(src).name
|
| 282 |
-
media_files.append((abs_src, original_filename, ext))
|
| 283 |
-
|
| 284 |
-
# Read existing media manifest and all existing entries
|
| 285 |
-
existing_entries: dict[str, bytes] = {}
|
| 286 |
-
with zipfile.ZipFile(apkg_path, 'r') as zin:
|
| 287 |
-
media_json = json.loads(zin.read('media'))
|
| 288 |
-
for name in zin.namelist():
|
| 289 |
-
if name != 'media':
|
| 290 |
-
existing_entries[name] = zin.read(name)
|
| 291 |
-
|
| 292 |
-
# Fix conf.curModel: genanki sets it to a random default model ID
|
| 293 |
-
# that doesn't match our custom MODEL_ID. Without this fix, Anki
|
| 294 |
-
# can't find the model during import and throws "A number was invalid or out of range".
|
| 295 |
-
_fix_conf_model(existing_entries)
|
| 296 |
-
|
| 297 |
-
# Build the set of (hash, original_filename) pairs from the manifest
|
| 298 |
-
manifest_hashes: dict[str, str] = {}
|
| 299 |
-
for zip_name, orig_name in media_json.items():
|
| 300 |
-
manifest_hashes[zip_name] = orig_name
|
| 301 |
-
|
| 302 |
-
# Inject each unique file and update manifest
|
| 303 |
-
for src_path, original_filename, ext in media_files:
|
| 304 |
-
content = Path(src_path).read_bytes()
|
| 305 |
-
media_hash = hashlib.md5(content, usedforsecurity=False).hexdigest()
|
| 306 |
-
zip_entry = f"{media_hash}{ext}"
|
| 307 |
-
|
| 308 |
-
# Skip if already injected (same content hash)
|
| 309 |
-
if zip_entry in manifest_hashes:
|
| 310 |
-
continue
|
| 311 |
-
|
| 312 |
-
# Store in-memory for rewrite
|
| 313 |
-
existing_entries[zip_entry] = content
|
| 314 |
-
manifest_hashes[zip_entry] = original_filename
|
| 315 |
-
|
| 316 |
-
# Rewrite the entire .apkg zip from memory — this avoids
|
| 317 |
-
# Python zipfile's inability to replace entries in append mode,
|
| 318 |
-
# which causes "Duplicate name: 'media'" warnings and corrupt packages.
|
| 319 |
-
import tempfile
|
| 320 |
-
with tempfile.NamedTemporaryFile(suffix='.apkg', delete=False) as tmp:
|
| 321 |
-
with zipfile.ZipFile(tmp.name, 'w', zipfile.ZIP_DEFLATED) as zout:
|
| 322 |
-
for name, data in existing_entries.items():
|
| 323 |
-
zout.writestr(name, data)
|
| 324 |
-
# Write updated media manifest
|
| 325 |
-
zout.writestr('media', json.dumps(manifest_hashes))
|
| 326 |
-
# Replace original with the corrected zip
|
| 327 |
-
import shutil
|
| 328 |
-
shutil.move(tmp.name, apkg_path)
|
| 329 |
-
|
| 330 |
-
|
| 331 |
-
def generate_apkg_package(
|
| 332 |
-
cards: list[dict],
|
| 333 |
-
scenario: str,
|
| 334 |
-
cefr_level: str,
|
| 335 |
-
target_language: str,
|
| 336 |
-
) -> str:
|
| 337 |
-
"""Generate an Anki package (.apkg) with embedded media.
|
| 338 |
-
|
| 339 |
-
Creates a genanki note model, builds notes from card data, generates the
|
| 340 |
-
base .apkg zip, then injects audio/image files with correct hashed names
|
| 341 |
-
and updates the media manifest.
|
| 342 |
-
|
| 343 |
-
Args:
|
| 344 |
-
cards: List of card dicts with keys: 'text', 'translation',
|
| 345 |
-
'audio_path' (str or None), 'image_path' (str or None).
|
| 346 |
-
scenario: Free-form scenario/topic string.
|
| 347 |
-
cefr_level: CEFR level string (e.g., 'A2', 'B1').
|
| 348 |
-
target_language: Target language name (e.g., 'Latvian').
|
| 349 |
-
|
| 350 |
-
Returns:
|
| 351 |
-
Absolute path to the generated .apkg file.
|
| 352 |
-
|
| 353 |
-
Raises:
|
| 354 |
-
ValueError: If no cards provided.
|
| 355 |
-
RuntimeError: If zip generation fails.
|
| 356 |
-
"""
|
| 357 |
-
if not cards:
|
| 358 |
-
raise ValueError("No cards provided for APKG export")
|
| 359 |
-
|
| 360 |
-
# Step 1: Create model and notes
|
| 361 |
-
model = _create_model()
|
| 362 |
-
notes = []
|
| 363 |
-
for card in cards:
|
| 364 |
-
note = _create_note(
|
| 365 |
-
model=model,
|
| 366 |
-
translation=card.get("translation", ""),
|
| 367 |
-
english=card.get("text", ""),
|
| 368 |
-
audio_path=card.get("audio_path"),
|
| 369 |
-
image_path=card.get("image_path"),
|
| 370 |
-
)
|
| 371 |
-
notes.append(note)
|
| 372 |
-
|
| 373 |
-
# Step 2: Create base package (genanki handles database + zip structure)
|
| 374 |
-
pkg_path = _create_package(notes, scenario, cefr_level, target_language)
|
| 375 |
-
|
| 376 |
-
try:
|
| 377 |
-
# Step 3: Inject media files
|
| 378 |
-
_inject_media(pkg_path, cards)
|
| 379 |
-
except Exception as e:
|
| 380 |
-
logger.warning("Media injection failed, returning text-only .apkg: %s", e)
|
| 381 |
-
# Return the text-only package — user still gets a usable .apkg
|
| 382 |
-
|
| 383 |
-
return pkg_path
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
tests/apkg_generator_test.py
DELETED
|
@@ -1,587 +0,0 @@
|
|
| 1 |
-
"""Tests for export/apkg_generator.py — APKG generation with embedded media."""
|
| 2 |
-
|
| 3 |
-
import json
|
| 4 |
-
import zipfile
|
| 5 |
-
from pathlib import Path
|
| 6 |
-
|
| 7 |
-
import pytest
|
| 8 |
-
|
| 9 |
-
# Import functions under test
|
| 10 |
-
from export.apkg_generator import (
|
| 11 |
-
MODEL_ID,
|
| 12 |
-
MODEL_NAME,
|
| 13 |
-
FIELDS,
|
| 14 |
-
TEMPLATE,
|
| 15 |
-
_create_model,
|
| 16 |
-
_create_note,
|
| 17 |
-
_create_package,
|
| 18 |
-
_inject_media,
|
| 19 |
-
generate_apkg_package,
|
| 20 |
-
)
|
| 21 |
-
|
| 22 |
-
PROJECT_ROOT = Path(__file__).resolve().parent.parent
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
class TestModelCreation:
|
| 26 |
-
"""Tests for model constants and _create_model function."""
|
| 27 |
-
|
| 28 |
-
def test_model_id_is_in_range(self):
|
| 29 |
-
"""MODEL_ID is a valid genanki model ID (30-bit unsigned integer)."""
|
| 30 |
-
assert 1 << 30 <= MODEL_ID < 1 << 31
|
| 31 |
-
|
| 32 |
-
def test_model_name(self):
|
| 33 |
-
assert MODEL_NAME == "EuropaLex Flashcard"
|
| 34 |
-
|
| 35 |
-
def test_fields_have_four_entries(self):
|
| 36 |
-
assert len(FIELDS) == 4
|
| 37 |
-
expected_names = ["Translation", "English", "Audio", "Image"]
|
| 38 |
-
for i, field in enumerate(FIELDS):
|
| 39 |
-
assert field["name"] == expected_names[i]
|
| 40 |
-
|
| 41 |
-
def test_template_structure(self):
|
| 42 |
-
"""Template has correct name and field references."""
|
| 43 |
-
assert TEMPLATE["name"] == "Card 1"
|
| 44 |
-
assert "{{Translation}}" in TEMPLATE["qfmt"]
|
| 45 |
-
assert "{{Image}}" in TEMPLATE["qfmt"]
|
| 46 |
-
assert "{{Audio}}" in TEMPLATE["qfmt"]
|
| 47 |
-
assert "{{FrontSide}}" in TEMPLATE["afmt"]
|
| 48 |
-
|
| 49 |
-
def test_create_model_returns_genanki_model(self):
|
| 50 |
-
"""_create_model returns a genanki.Model instance with correct attributes."""
|
| 51 |
-
import genanki
|
| 52 |
-
model = _create_model()
|
| 53 |
-
assert isinstance(model, genanki.Model)
|
| 54 |
-
assert model.name == MODEL_NAME
|
| 55 |
-
assert len(model.fields) == 4
|
| 56 |
-
assert len(model.templates) == 1
|
| 57 |
-
|
| 58 |
-
def test_create_model_field_names(self):
|
| 59 |
-
"""Model fields have correct names in order."""
|
| 60 |
-
import genanki
|
| 61 |
-
model = _create_model()
|
| 62 |
-
field_names = [f["name"] for f in model.fields]
|
| 63 |
-
assert field_names == ["Translation", "English", "Audio", "Image"]
|
| 64 |
-
|
| 65 |
-
def test_create_model_template_order(self):
|
| 66 |
-
"""Template qfmt order: Translation, Image, Audio (matching spec)."""
|
| 67 |
-
import genanki
|
| 68 |
-
model = _create_model()
|
| 69 |
-
qfmt = model.templates[0]["qfmt"]
|
| 70 |
-
# Translation should come before Image, which comes before Audio
|
| 71 |
-
ord_translation = qfmt.index("{{Translation}}")
|
| 72 |
-
ord_image = qfmt.index("{{Image}}")
|
| 73 |
-
ord_audio = qfmt.index("{{Audio}}")
|
| 74 |
-
assert ord_translation < ord_image < ord_audio
|
| 75 |
-
|
| 76 |
-
|
| 77 |
-
class TestNoteCreation:
|
| 78 |
-
"""Tests for _create_note function."""
|
| 79 |
-
|
| 80 |
-
def test_text_fields_escaped(self):
|
| 81 |
-
"""Text fields are HTML-escaped to prevent injection."""
|
| 82 |
-
import genanki
|
| 83 |
-
model = _create_model()
|
| 84 |
-
note = _create_note(
|
| 85 |
-
model=model,
|
| 86 |
-
translation="Hello <script>alert('xss')</script>",
|
| 87 |
-
english='Test & "quotes"',
|
| 88 |
-
audio_path=None,
|
| 89 |
-
image_path=None,
|
| 90 |
-
)
|
| 91 |
-
assert isinstance(note, genanki.Note)
|
| 92 |
-
# HTML entities should be escaped
|
| 93 |
-
assert "<script>" in note.fields[0]
|
| 94 |
-
assert "&" in note.fields[1]
|
| 95 |
-
|
| 96 |
-
def test_audio_field_with_path(self):
|
| 97 |
-
"""Audio field contains <audio> tag with original filename."""
|
| 98 |
-
import genanki
|
| 99 |
-
model = _create_model()
|
| 100 |
-
note = _create_note(
|
| 101 |
-
model=model,
|
| 102 |
-
translation="Hola",
|
| 103 |
-
english="Hello",
|
| 104 |
-
audio_path="/some/path/hello_A2_LV_0.wav",
|
| 105 |
-
image_path=None,
|
| 106 |
-
)
|
| 107 |
-
assert "<audio controls src=" in note.fields[2]
|
| 108 |
-
assert "hello_A2_LV_0.wav" in note.fields[2]
|
| 109 |
-
|
| 110 |
-
def test_image_field_with_path(self):
|
| 111 |
-
"""Image field contains <img> tag with original filename."""
|
| 112 |
-
import genanki
|
| 113 |
-
model = _create_model()
|
| 114 |
-
note = _create_note(
|
| 115 |
-
model=model,
|
| 116 |
-
translation="Hola",
|
| 117 |
-
english="Hello",
|
| 118 |
-
audio_path=None,
|
| 119 |
-
image_path="/some/path/hello_A2_LV_0.png",
|
| 120 |
-
)
|
| 121 |
-
assert "<img src=" in note.fields[3]
|
| 122 |
-
assert "hello_A2_LV_0.png" in note.fields[3]
|
| 123 |
-
|
| 124 |
-
def test_empty_media_paths_become_empty_strings(self):
|
| 125 |
-
"""None audio/image paths produce empty string fields."""
|
| 126 |
-
import genanki
|
| 127 |
-
model = _create_model()
|
| 128 |
-
note = _create_note(
|
| 129 |
-
model=model,
|
| 130 |
-
translation="Hello",
|
| 131 |
-
english="Hola",
|
| 132 |
-
audio_path=None,
|
| 133 |
-
image_path=None,
|
| 134 |
-
)
|
| 135 |
-
assert note.fields[2] == "" # Audio field
|
| 136 |
-
assert note.fields[3] == "" # Image field
|
| 137 |
-
|
| 138 |
-
def test_note_has_correct_field_count(self):
|
| 139 |
-
"""Note has exactly 4 fields matching model definition."""
|
| 140 |
-
import genanki
|
| 141 |
-
model = _create_model()
|
| 142 |
-
note = _create_note(
|
| 143 |
-
model=model,
|
| 144 |
-
translation="A",
|
| 145 |
-
english="B",
|
| 146 |
-
audio_path=None,
|
| 147 |
-
image_path=None,
|
| 148 |
-
)
|
| 149 |
-
assert len(note.fields) == 4
|
| 150 |
-
|
| 151 |
-
|
| 152 |
-
class TestPackageGeneration:
|
| 153 |
-
"""Tests for _create_package function."""
|
| 154 |
-
|
| 155 |
-
def test_package_returns_path(self, tmp_path):
|
| 156 |
-
"""_create_package returns an absolute path string."""
|
| 157 |
-
model = _create_model()
|
| 158 |
-
notes = [
|
| 159 |
-
_create_note(model, "Hola", "Hello"),
|
| 160 |
-
_create_note(model, "Adios", "Goodbye"),
|
| 161 |
-
]
|
| 162 |
-
pkg_path = _create_package(notes, scenario="greetings", cefr_level="A1", target_language="Spanish")
|
| 163 |
-
assert isinstance(pkg_path, str)
|
| 164 |
-
assert Path(pkg_path).is_absolute()
|
| 165 |
-
|
| 166 |
-
def test_package_is_valid_zip(self, tmp_path):
|
| 167 |
-
"""Generated .apkg is a valid zip file."""
|
| 168 |
-
model = _create_model()
|
| 169 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 170 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A1", target_language="Spanish")
|
| 171 |
-
assert zipfile.is_zipfile(pkg_path)
|
| 172 |
-
|
| 173 |
-
def test_package_has_collection_anki2(self, tmp_path):
|
| 174 |
-
"""Package contains collection.anki2 database."""
|
| 175 |
-
model = _create_model()
|
| 176 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 177 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A1", target_language="Spanish")
|
| 178 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 179 |
-
assert 'collection.anki2' in zf.namelist()
|
| 180 |
-
|
| 181 |
-
def test_package_has_media_manifest(self, tmp_path):
|
| 182 |
-
"""Package contains media JSON manifest."""
|
| 183 |
-
model = _create_model()
|
| 184 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 185 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A1", target_language="Spanish")
|
| 186 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 187 |
-
assert 'media' in zf.namelist()
|
| 188 |
-
manifest = json.loads(zf.read('media'))
|
| 189 |
-
assert isinstance(manifest, dict)
|
| 190 |
-
|
| 191 |
-
def test_package_deck_name_includes_scenario(self, tmp_path):
|
| 192 |
-
"""Deck name derives from scenario with CEFR and language abbreviation."""
|
| 193 |
-
model = _create_model()
|
| 194 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 195 |
-
pkg_path = _create_package(notes, scenario="ordering coffee", cefr_level="A2", target_language="Latvian")
|
| 196 |
-
import sqlite3
|
| 197 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 198 |
-
db_bytes = zf.read('collection.anki2')
|
| 199 |
-
tmp_db = tmp_path / "temp.db"
|
| 200 |
-
tmp_db.write_bytes(db_bytes)
|
| 201 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 202 |
-
cur = conn.cursor()
|
| 203 |
-
# Decks are stored as JSON in the 'decks' column of the 'col' table
|
| 204 |
-
# Structure: {deck_id_str: {name: "...", ...}}
|
| 205 |
-
cur.execute("SELECT decks FROM col")
|
| 206 |
-
row = cur.fetchone()
|
| 207 |
-
if row and row[0]:
|
| 208 |
-
decks_data = json.loads(row[0])
|
| 209 |
-
# Find any deck with 'ordering_coffee' in its name
|
| 210 |
-
found = any(
|
| 211 |
-
isinstance(v, dict) and 'ordering_coffee' in v.get('name', '').lower()
|
| 212 |
-
for v in decks_data.values()
|
| 213 |
-
)
|
| 214 |
-
assert found, f"Deck name not found. Decks data: {decks_data}"
|
| 215 |
-
conn.close()
|
| 216 |
-
|
| 217 |
-
def test_package_contains_all_notes(self, tmp_path):
|
| 218 |
-
"""Package database contains the correct number of notes."""
|
| 219 |
-
model = _create_model()
|
| 220 |
-
notes = [
|
| 221 |
-
_create_note(model, "A", "1"),
|
| 222 |
-
_create_note(model, "B", "2"),
|
| 223 |
-
_create_note(model, "C", "3"),
|
| 224 |
-
]
|
| 225 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A1", target_language="Spanish")
|
| 226 |
-
import sqlite3
|
| 227 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 228 |
-
db_bytes = zf.read('collection.anki2')
|
| 229 |
-
tmp_db = tmp_path / "temp.db"
|
| 230 |
-
tmp_db.write_bytes(db_bytes)
|
| 231 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 232 |
-
cur = conn.cursor()
|
| 233 |
-
cur.execute("SELECT COUNT(*) FROM notes")
|
| 234 |
-
count = cur.fetchone()[0]
|
| 235 |
-
conn.close()
|
| 236 |
-
assert count == 3
|
| 237 |
-
|
| 238 |
-
def test_package_no_media_files_when_none(self, tmp_path):
|
| 239 |
-
"""Package has no media files when all audio/image paths are None."""
|
| 240 |
-
model = _create_model()
|
| 241 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 242 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A1", target_language="Spanish")
|
| 243 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 244 |
-
names = zf.namelist()
|
| 245 |
-
# Should only have collection.anki2 and media — no numbered files or .wav/.png
|
| 246 |
-
assert len([n for n in names if n.endswith('.wav') or n.endswith('.png')]) == 0
|
| 247 |
-
|
| 248 |
-
|
| 249 |
-
class TestMediaInjection:
|
| 250 |
-
"""Tests for _inject_media function."""
|
| 251 |
-
|
| 252 |
-
def test_inject_audio_updates_manifest(self, tmp_path):
|
| 253 |
-
"""Audio file is hashed and added to media manifest."""
|
| 254 |
-
model = _create_model()
|
| 255 |
-
# Create a real .wav file in temp dir
|
| 256 |
-
wav_dir = tmp_path / "audio"
|
| 257 |
-
wav_dir.mkdir()
|
| 258 |
-
wav_path = str(wav_dir / "test_A2_LV_0.wav")
|
| 259 |
-
Path(wav_path).write_bytes(b'\x00' * 100) # dummy WAV content
|
| 260 |
-
|
| 261 |
-
notes = [_create_note(model, "Hola", "Hello", audio_path=wav_path)]
|
| 262 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 263 |
-
|
| 264 |
-
# Inject media
|
| 265 |
-
cards_for_inject = [{"audio_path": wav_path, "image_path": None}]
|
| 266 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 267 |
-
|
| 268 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 269 |
-
manifest = json.loads(zf.read('media'))
|
| 270 |
-
# Manifest should have an entry with a 32-char hex hash key
|
| 271 |
-
assert len(manifest) == 1
|
| 272 |
-
hash_key = list(manifest.keys())[0]
|
| 273 |
-
assert len(hash_key) == 36 # 32 hex chars + ".wav"
|
| 274 |
-
assert hash_key.endswith(".wav")
|
| 275 |
-
assert manifest[hash_key] == "test_A2_LV_0.wav"
|
| 276 |
-
|
| 277 |
-
def test_inject_image_updates_manifest(self, tmp_path):
|
| 278 |
-
"""Image file is hashed and added to media manifest."""
|
| 279 |
-
model = _create_model()
|
| 280 |
-
png_dir = tmp_path / "images"
|
| 281 |
-
png_dir.mkdir()
|
| 282 |
-
png_path = str(png_dir / "test_A2_LV_0.png")
|
| 283 |
-
Path(png_path).write_bytes(b'\x89PNG\r\n\x1a\n' + b'\x00' * 100) # fake PNG header
|
| 284 |
-
|
| 285 |
-
notes = [_create_note(model, "Hola", "Hello", image_path=png_path)]
|
| 286 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 287 |
-
|
| 288 |
-
cards_for_inject = [{"audio_path": None, "image_path": png_path}]
|
| 289 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 290 |
-
|
| 291 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 292 |
-
manifest = json.loads(zf.read('media'))
|
| 293 |
-
assert len(manifest) == 1
|
| 294 |
-
hash_key = list(manifest.keys())[0]
|
| 295 |
-
assert hash_key.endswith(".png")
|
| 296 |
-
|
| 297 |
-
def test_inject_media_file_in_zip(self, tmp_path):
|
| 298 |
-
"""Injected media file exists in the zip under hashed name."""
|
| 299 |
-
model = _create_model()
|
| 300 |
-
wav_dir = tmp_path / "audio"
|
| 301 |
-
wav_dir.mkdir()
|
| 302 |
-
wav_path = str(wav_dir / "test_A2_LV_0.wav")
|
| 303 |
-
Path(wav_path).write_bytes(b'\x00' * 100)
|
| 304 |
-
|
| 305 |
-
notes = [_create_note(model, "Hola", "Hello", audio_path=wav_path)]
|
| 306 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 307 |
-
|
| 308 |
-
cards_for_inject = [{"audio_path": wav_path, "image_path": None}]
|
| 309 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 310 |
-
|
| 311 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 312 |
-
names = zf.namelist()
|
| 313 |
-
# Should have the hashed .wav file
|
| 314 |
-
assert any(n.endswith('.wav') for n in names)
|
| 315 |
-
|
| 316 |
-
def test_inject_deduplicates_same_file(self, tmp_path):
|
| 317 |
-
"""Same media file referenced by multiple cards is injected only once."""
|
| 318 |
-
model = _create_model()
|
| 319 |
-
wav_dir = tmp_path / "audio"
|
| 320 |
-
wav_dir.mkdir()
|
| 321 |
-
wav_path = str(wav_dir / "shared_A2_LV_0.wav")
|
| 322 |
-
Path(wav_path).write_bytes(b'\x00' * 100)
|
| 323 |
-
|
| 324 |
-
notes = [
|
| 325 |
-
_create_note(model, "A", "1", audio_path=wav_path),
|
| 326 |
-
_create_note(model, "B", "2", audio_path=wav_path), # same file
|
| 327 |
-
]
|
| 328 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 329 |
-
|
| 330 |
-
cards_for_inject = [
|
| 331 |
-
{"audio_path": wav_path, "image_path": None},
|
| 332 |
-
{"audio_path": wav_path, "image_path": None}, # duplicate reference
|
| 333 |
-
]
|
| 334 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 335 |
-
|
| 336 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 337 |
-
manifest = json.loads(zf.read('media'))
|
| 338 |
-
# Only one entry for the shared file
|
| 339 |
-
assert len(manifest) == 1
|
| 340 |
-
|
| 341 |
-
def test_inject_skips_missing_files(self, tmp_path):
|
| 342 |
-
"""Injection skips files that don't exist on disk."""
|
| 343 |
-
model = _create_model()
|
| 344 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 345 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 346 |
-
|
| 347 |
-
cards_for_inject = [{"audio_path": "/nonexistent/path.wav", "image_path": None}]
|
| 348 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 349 |
-
|
| 350 |
-
# Should not crash; manifest stays empty
|
| 351 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 352 |
-
manifest = json.loads(zf.read('media'))
|
| 353 |
-
assert len(manifest) == 0
|
| 354 |
-
|
| 355 |
-
def test_inject_preserves_existing_files(self, tmp_path):
|
| 356 |
-
"""Media injection fixes conf.curModel but preserves all other DB data."""
|
| 357 |
-
model = _create_model()
|
| 358 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 359 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 360 |
-
|
| 361 |
-
# Record original DB state
|
| 362 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 363 |
-
original_db = zf.read('collection.anki2')
|
| 364 |
-
|
| 365 |
-
wav_dir = tmp_path / "audio"
|
| 366 |
-
wav_dir.mkdir()
|
| 367 |
-
wav_path = str(wav_dir / "test_A2_LV_0.wav")
|
| 368 |
-
Path(wav_path).write_bytes(b'\x00' * 100)
|
| 369 |
-
|
| 370 |
-
cards_for_inject = [{"audio_path": wav_path, "image_path": None}]
|
| 371 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 372 |
-
|
| 373 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 374 |
-
new_db = zf.read('collection.anki2')
|
| 375 |
-
|
| 376 |
-
# DB should not be identical (curModel is fixed), but structure must be intact
|
| 377 |
-
assert new_db != original_db # conf.curModel was updated
|
| 378 |
-
|
| 379 |
-
# Verify the fix: curModel should match MODEL_ID
|
| 380 |
-
import sqlite3, tempfile
|
| 381 |
-
tmp_db = tmp_path / "temp.db"
|
| 382 |
-
tmp_db.write_bytes(new_db)
|
| 383 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 384 |
-
cur = conn.cursor()
|
| 385 |
-
cur.execute('SELECT conf FROM col')
|
| 386 |
-
conf = json.loads(cur.fetchone()[0])
|
| 387 |
-
assert conf['curModel'] == str(MODEL_ID)
|
| 388 |
-
conn.close()
|
| 389 |
-
|
| 390 |
-
def test_hash_is_content_based(self, tmp_path):
|
| 391 |
-
"""MD5 hash is computed from file content, not filename."""
|
| 392 |
-
model = _create_model()
|
| 393 |
-
wav_dir = tmp_path / "audio"
|
| 394 |
-
wav_dir.mkdir()
|
| 395 |
-
# Two files with different names but identical content
|
| 396 |
-
path1 = str(wav_dir / "name_a.wav")
|
| 397 |
-
path2 = str(wav_dir / "name_b.wav")
|
| 398 |
-
Path(path1).write_bytes(b'\x00' * 100)
|
| 399 |
-
Path(path2).write_bytes(b'\x00' * 100)
|
| 400 |
-
|
| 401 |
-
notes = [_create_note(model, "Hola", "Hello", audio_path=path1)]
|
| 402 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 403 |
-
|
| 404 |
-
cards_for_inject = [
|
| 405 |
-
{"audio_path": path1, "image_path": None},
|
| 406 |
-
{"audio_path": path2, "image_path": None}, # same content, different name
|
| 407 |
-
]
|
| 408 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 409 |
-
|
| 410 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 411 |
-
manifest = json.loads(zf.read('media'))
|
| 412 |
-
# Only one entry because content is identical (dedup by hash)
|
| 413 |
-
assert len(manifest) == 1
|
| 414 |
-
|
| 415 |
-
def test_inject_fixes_conf_curmodel(self, tmp_path):
|
| 416 |
-
"""conf.curModel is updated to MODEL_ID during media injection."""
|
| 417 |
-
model = _create_model()
|
| 418 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 419 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 420 |
-
|
| 421 |
-
wav_dir = tmp_path / "audio"
|
| 422 |
-
wav_dir.mkdir()
|
| 423 |
-
wav_path = str(wav_dir / "test_A2_LV_0.wav")
|
| 424 |
-
Path(wav_path).write_bytes(b'\x00' * 100)
|
| 425 |
-
|
| 426 |
-
cards_for_inject = [{"audio_path": wav_path, "image_path": None}]
|
| 427 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 428 |
-
|
| 429 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 430 |
-
db_bytes = zf.read('collection.anki2')
|
| 431 |
-
tmp_db = tmp_path / "temp.db"
|
| 432 |
-
tmp_db.write_bytes(db_bytes)
|
| 433 |
-
import sqlite3
|
| 434 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 435 |
-
cur = conn.cursor()
|
| 436 |
-
cur.execute('SELECT conf FROM col')
|
| 437 |
-
conf = json.loads(cur.fetchone()[0])
|
| 438 |
-
assert conf['curModel'] == str(MODEL_ID), \
|
| 439 |
-
f"curModel should be {MODEL_ID}, got {conf['curModel']}"
|
| 440 |
-
|
| 441 |
-
# Verify curModel points to an existing model in the models dict
|
| 442 |
-
cur.execute('SELECT models FROM col')
|
| 443 |
-
models = json.loads(cur.fetchone()[0])
|
| 444 |
-
assert conf['curModel'] in models, \
|
| 445 |
-
f"curModel {conf['curModel']} not found in models keys: {list(models.keys())}"
|
| 446 |
-
conn.close()
|
| 447 |
-
|
| 448 |
-
def test_inject_fixes_curmodel_without_media(self, tmp_path):
|
| 449 |
-
"""conf.curModel is fixed even when no media files are injected."""
|
| 450 |
-
model = _create_model()
|
| 451 |
-
notes = [_create_note(model, "Hola", "Hello")]
|
| 452 |
-
pkg_path = _create_package(notes, scenario="test", cefr_level="A2", target_language="Latvian")
|
| 453 |
-
|
| 454 |
-
# No media files to inject
|
| 455 |
-
cards_for_inject = [{"audio_path": None, "image_path": None}]
|
| 456 |
-
_inject_media(pkg_path, cards_for_inject)
|
| 457 |
-
|
| 458 |
-
with zipfile.ZipFile(pkg_path, 'r') as zf:
|
| 459 |
-
db_bytes = zf.read('collection.anki2')
|
| 460 |
-
tmp_db = tmp_path / "temp.db"
|
| 461 |
-
tmp_db.write_bytes(db_bytes)
|
| 462 |
-
import sqlite3
|
| 463 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 464 |
-
cur = conn.cursor()
|
| 465 |
-
cur.execute('SELECT conf FROM col')
|
| 466 |
-
conf = json.loads(cur.fetchone()[0])
|
| 467 |
-
assert conf['curModel'] == str(MODEL_ID)
|
| 468 |
-
conn.close()
|
| 469 |
-
|
| 470 |
-
|
| 471 |
-
class TestGenerateApkgPackage:
|
| 472 |
-
"""Tests for the main generate_apkg_package function."""
|
| 473 |
-
|
| 474 |
-
def test_returns_path(self, tmp_path):
|
| 475 |
-
"""Returns absolute path to .apkg file."""
|
| 476 |
-
cards = [
|
| 477 |
-
{
|
| 478 |
-
"text": "Hello",
|
| 479 |
-
"translation": "Hola",
|
| 480 |
-
"audio_path": None,
|
| 481 |
-
"image_path": None,
|
| 482 |
-
},
|
| 483 |
-
]
|
| 484 |
-
result = generate_apkg_package(cards, "greetings", "A1", "Spanish")
|
| 485 |
-
assert isinstance(result, str)
|
| 486 |
-
assert Path(result).is_absolute()
|
| 487 |
-
|
| 488 |
-
def test_returns_valid_zip(self, tmp_path):
|
| 489 |
-
"""Return value is a valid zip file."""
|
| 490 |
-
cards = [{"text": "Hello", "translation": "Hola", "audio_path": None, "image_path": None}]
|
| 491 |
-
result = generate_apkg_package(cards, "greetings", "A1", "Spanish")
|
| 492 |
-
assert zipfile.is_zipfile(result)
|
| 493 |
-
|
| 494 |
-
def test_contains_note_data(self, tmp_path):
|
| 495 |
-
"""Package database contains correct number of notes."""
|
| 496 |
-
cards = [
|
| 497 |
-
{"text": "One", "translation": "Uno", "audio_path": None, "image_path": None},
|
| 498 |
-
{"text": "Two", "translation": "Dos", "audio_path": None, "image_path": None},
|
| 499 |
-
{"text": "Three", "translation": "Tres", "audio_path": None, "image_path": None},
|
| 500 |
-
]
|
| 501 |
-
result = generate_apkg_package(cards, "numbers", "A1", "Spanish")
|
| 502 |
-
import sqlite3
|
| 503 |
-
with zipfile.ZipFile(result, 'r') as zf:
|
| 504 |
-
db_bytes = zf.read('collection.anki2')
|
| 505 |
-
tmp_db = tmp_path / "temp.db"
|
| 506 |
-
tmp_db.write_bytes(db_bytes)
|
| 507 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 508 |
-
cur = conn.cursor()
|
| 509 |
-
cur.execute("SELECT COUNT(*) FROM notes")
|
| 510 |
-
count = cur.fetchone()[0]
|
| 511 |
-
conn.close()
|
| 512 |
-
assert count == 3
|
| 513 |
-
|
| 514 |
-
def test_media_injected_when_paths_exist(self, tmp_path):
|
| 515 |
-
"""Audio/image files are injected into package when paths exist."""
|
| 516 |
-
wav_dir = tmp_path / "audio"
|
| 517 |
-
wav_dir.mkdir()
|
| 518 |
-
wav_path = str(wav_dir / "test_A2_LV_0.wav")
|
| 519 |
-
Path(wav_path).write_bytes(b'\x00' * 100)
|
| 520 |
-
|
| 521 |
-
cards = [
|
| 522 |
-
{
|
| 523 |
-
"text": "Hello",
|
| 524 |
-
"translation": "Hola",
|
| 525 |
-
"audio_path": wav_path,
|
| 526 |
-
"image_path": None,
|
| 527 |
-
},
|
| 528 |
-
]
|
| 529 |
-
result = generate_apkg_package(cards, "test", "A2", "Latvian")
|
| 530 |
-
|
| 531 |
-
with zipfile.ZipFile(result, 'r') as zf:
|
| 532 |
-
manifest = json.loads(zf.read('media'))
|
| 533 |
-
assert len(manifest) == 1
|
| 534 |
-
hash_key = list(manifest.keys())[0]
|
| 535 |
-
assert hash_key.endswith(".wav")
|
| 536 |
-
|
| 537 |
-
def test_no_media_when_all_none(self, tmp_path):
|
| 538 |
-
"""No media files in package when all paths are None."""
|
| 539 |
-
cards = [{"text": "Hello", "translation": "Hola", "audio_path": None, "image_path": None}]
|
| 540 |
-
result = generate_apkg_package(cards, "test", "A1", "Spanish")
|
| 541 |
-
|
| 542 |
-
with zipfile.ZipFile(result, 'r') as zf:
|
| 543 |
-
names = zf.namelist()
|
| 544 |
-
assert not any(n.endswith('.wav') or n.endswith('.png') for n in names)
|
| 545 |
-
|
| 546 |
-
def test_missing_media_files_skipped(self, tmp_path):
|
| 547 |
-
"""Function succeeds even when media files don't exist on disk."""
|
| 548 |
-
cards = [
|
| 549 |
-
{
|
| 550 |
-
"text": "Hello",
|
| 551 |
-
"translation": "Hola",
|
| 552 |
-
"audio_path": "/nonexistent/path.wav",
|
| 553 |
-
"image_path": None,
|
| 554 |
-
},
|
| 555 |
-
]
|
| 556 |
-
result = generate_apkg_package(cards, "test", "A1", "Spanish")
|
| 557 |
-
assert Path(result).exists()
|
| 558 |
-
|
| 559 |
-
def test_empty_cards_raises_valueerror(self):
|
| 560 |
-
"""Raises ValueError when no cards provided."""
|
| 561 |
-
with pytest.raises(ValueError, match="No cards"):
|
| 562 |
-
generate_apkg_package([], "test", "A1", "Spanish")
|
| 563 |
-
|
| 564 |
-
def test_deck_name_in_output(self, tmp_path):
|
| 565 |
-
"""Deck name in package matches expected pattern."""
|
| 566 |
-
cards = [{"text": "Hello", "translation": "Hola", "audio_path": None, "image_path": None}]
|
| 567 |
-
result = generate_apkg_package(cards, "ordering coffee", "A2", "Latvian")
|
| 568 |
-
|
| 569 |
-
import sqlite3
|
| 570 |
-
with zipfile.ZipFile(result, 'r') as zf:
|
| 571 |
-
db_bytes = zf.read('collection.anki2')
|
| 572 |
-
tmp_db = tmp_path / "temp.db"
|
| 573 |
-
tmp_db.write_bytes(db_bytes)
|
| 574 |
-
conn = sqlite3.connect(str(tmp_db))
|
| 575 |
-
cur = conn.cursor()
|
| 576 |
-
cur.execute("SELECT decks FROM col")
|
| 577 |
-
row = cur.fetchone()
|
| 578 |
-
conn.close()
|
| 579 |
-
|
| 580 |
-
assert row is not None
|
| 581 |
-
decks_data = json.loads(row[0])
|
| 582 |
-
# Find deck with expected name pattern
|
| 583 |
-
found = any(
|
| 584 |
-
isinstance(v, dict) and 'ordering_coffee' in v.get('name', '').lower()
|
| 585 |
-
for v in decks_data.values()
|
| 586 |
-
)
|
| 587 |
-
assert found, f"Expected 'ordering_coffee' in deck name. Decks: {list(decks_data.keys())}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|