diff --git a/README.md b/README.md index 5a2be81..982d414 100644 --- a/README.md +++ b/README.md @@ -199,7 +199,78 @@ hablar 9 見る (見た) 6 ``` -### YouTube +### Words from TSV Export + +Extract vocabulary from an Anki TSV export file instead of using AnkiConnect: + +```shell +saiki words --lang es --input Español.txt --field 2 --output words_es_content.txt --debug words_es_debug.tsv +``` + +Additional file-based options: + +```shell +saiki words --lang es --input Español.txt --field 2 --include-proper-nouns +saiki words --lang es --input Español.txt --field 2 --function-words +saiki words --lang es --input Español.txt --field 2 --lemma-corrections my_fixes.tsv +saiki words --lang es --input Español.txt --field 2 --bad-lemma-file bad_lemmas.txt +``` + +When `--input` is provided, `--field` specifies the 1-based column index of the +Spanish text (default 2). The audio column (index 1) and tags column are +automatically skipped. File-based NLP extraction is currently Spanish-specific; +the older AnkiConnect-based `saiki words jp` flow is unchanged. + +The file-based pipeline: +- Parses Anki `#` header lines (`#separator:tab`, `#html:true`, `#tags column:N`) +- Uses Python's `csv` module for robust TSV parsing +- Removes `[sound:...mp3]` markers, HTML tags, and English glosses after + `

` from the field text +- Applies safe Spanish lemmatisation (multi-word lemmas like `ayudar yo` or + `lavar él` are rejected; bad lemmas like `comar` -> `comer` are corrected) +- Tracks POS counts, surface forms, example sentences, and source line numbers +- Tracks original spaCy lemmas and correction/fallback reasons for debugging + +Output files produced: +- `words_es_content.txt` — cleaned content vocabulary (NOUN, VERB, ADJ, ADV) +- `words_es_debug.tsv` — per-lemma debug info +- `words_es_proper_nouns.txt` — proper nouns (only with `--include-proper-nouns`) +- `words_es_function_words.txt` — function words (only with `--function-words`) +- `words_es_suspicious_tokens.txt` — lemmas that required correction + +Debug TSV example: + +```text +lemma count pos_counts top_surface_forms example_sentences source_lines original_lemmas lemma_statuses status +comer 8 VERB:8 como, come, comen Yo como manzanas.; Ustedes los comen con arroz. 70,979 comar, comer corrected:comar->comer:1, ok:7 +``` + +### Lint Spanish Cards + +Check an Anki TSV export for suspicious or awkward Spanish: + +```shell +saiki lint-anki-es --input Español.txt --field 2 --output suspicious_cards_es.tsv +``` + +Detects known errors and rule-based suspicious patterns and produces a TSV +report with: source line, original text, cleaned text, reason, and suggested +fix. + +### Compare + +Compare deck vocabulary against a target list: + +```shell +saiki compare --deck-words words_es_content.txt --target-words target_es_top_1000.txt --output missing_from_deck.txt +saiki compare --deck-words words_es_content.txt --target-words target_es_top_1000.txt --min-frequency 3 +``` + +Normalises case and optionally strips accents for matching, so `cómo` and +`como` are treated as the same word. Use `--min-frequency` to ignore accidental +low-frequency words in the deck. + + Mine vocabulary or sentence rows from YouTube subtitles. diff --git a/src/saiki/anki_tsv.py b/src/saiki/anki_tsv.py new file mode 100644 index 0000000..cddc7e3 --- /dev/null +++ b/src/saiki/anki_tsv.py @@ -0,0 +1,124 @@ +"""Parse Anki TSV export files. + +Anki exports are tab-separated files with ``#`` header lines describing +the separator, HTML mode, and tag column. This module parses those +headers and extracts structured records. +""" + +from __future__ import annotations + +import csv +from typing import Any + +SEPARATOR_ALIASES = { + "tab": "\t", + "space": " ", + "comma": ",", + "semicolon": ";", +} + + +def _resolve_separator(raw: str) -> str: + """Map Anki separator names to actual delimiter characters.""" + return SEPARATOR_ALIASES.get(raw.strip().lower(), raw.strip()) + + +def parse_anki_headers(path: str) -> dict[str, Any]: + """Read ``#`` -prefixed metadata lines from an Anki export file. + + Stops at the first non-header line. Returns a dict with optional keys + ``separator``, ``html``, and ``tags_column``. + """ + info: dict[str, Any] = {} + with open(path, "r", encoding="utf-8") as fh: + for line in fh: + line = line.rstrip("\n\r") + if not line.startswith("#"): + break + if line.startswith("#separator:"): + raw = line[len("#separator:"):].strip() + info["separator"] = _resolve_separator(raw) + elif line.startswith("#html:"): + info["html"] = line[len("#html:"):].strip().lower() == "true" + elif line.startswith("#tags column:"): + info["tags_column"] = line[len("#tags column:"):].strip() + return info + + +def parse_anki_tsv( + path: str, + field_index: int = 2, + include_tags: bool = False, +) -> list[dict[str, Any]]: + """Parse an Anki TSV export file into structured records. + + Parameters + ---------- + path: + Path to the Anki TSV export file. + field_index: + 1-based column index of the target text field (default 2). + include_tags: + Whether to include the tags column in output records. + + Returns + ------- + A list of record dicts with keys: + ``text``, ``tags`` (if *include_tags* is True), ``line_number``, + and ``raw_line``. + + Header lines (``#``-prefixed) are skipped. The actual separator + is detected from ``#separator:`` and defaults to tab. + """ + if field_index < 1: + raise ValueError(f"field_index must be 1-based and >= 1, got {field_index}") + + header = parse_anki_headers(path) + separator = header.get("separator", "\t") + + records: list[dict[str, Any]] = [] + col_idx = field_index - 1 # convert to 0-based + + tags_column: int | None = None + if include_tags and "tags_column" in header: + try: + tags_column = int(header["tags_column"]) - 1 + except ValueError as exc: + raise ValueError( + f"Invalid #tags column value: {header['tags_column']!r}" + ) from exc + if tags_column < 0: + raise ValueError( + f"#tags column must be 1-based and >= 1, got {header['tags_column']!r}" + ) + + with open(path, "r", encoding="utf-8") as fh: + reader = csv.reader(fh, delimiter=separator) + line_number = 0 + for row in reader: + line_number += 1 + # skip header rows and completely empty rows + if not row: + continue + if row and row[0].startswith("#"): + continue + + if col_idx >= len(row): + raise ValueError( + f"Line {line_number}: field index {field_index} not found " + f"in row with {len(row)} column(s)" + ) + + text = row[col_idx].strip() + record: dict[str, Any] = { + "text": text, + "line_number": line_number, + "raw_line": separator.join(row), + } + if include_tags and tags_column is not None: + record["tags"] = ( + row[tags_column].strip() if tags_column < len(row) else "" + ) + records.append(record) + + return records diff --git a/src/saiki/cli.py b/src/saiki/cli.py index e57a61a..7c9041a 100644 --- a/src/saiki/cli.py +++ b/src/saiki/cli.py @@ -15,7 +15,7 @@ from .importer import ( supported_tts_backends, synthesize_tts_sample, ) -from .words import compare_word_files, extract_words +from .words import compare_word_files, compare_word_lists, extract_words, extract_words_from_file, lint_anki_cards_main from .youtube import run_youtube @@ -85,22 +85,48 @@ def build_parser(config: Config | None = None) -> argparse.ArgumentParser: audio.add_argument("--media-dir") audio.add_argument("--copy-only-new", action="store_true") - words = sub.add_parser("words", help="Extract frequent words from Anki.") - words.add_argument("lang", choices=choices) + words = sub.add_parser("words", help="Extract frequent words from Anki or from a TSV export file.") + words.add_argument("lang", nargs="?", choices=choices) + words.add_argument("--lang", dest="lang_option", choices=choices, help="Language code.") + words.add_argument("--input", help="Anki TSV export file (instead of AnkiConnect).") group = words.add_mutually_exclusive_group() group.add_argument("--query") group.add_argument("--deck", action="append") - words.add_argument("--field") + words.add_argument("--field", help="Anki field name (AnkiConnect) or 1-based column index (TSV file).") words.add_argument("--min-freq", type=int, default=2) words.add_argument("--outdir") - words.add_argument("--out") + words.add_argument("--out", "--output", dest="out") words.add_argument("--full-field", action="store_true") words.add_argument("--spacy-model") + words.add_argument("--include-proper-nouns", action="store_true", help="Include proper nouns (PROPN).") + words.add_argument("--function-words", action="store_true", help="Extract function words separately.") + words.add_argument("--debug", help="Path for debug TSV output.") + words.add_argument("--lemma-corrections", help="Path to badgood TSV of lemma corrections.") + words.add_argument("--bad-lemma-file", help="Path to file listing lemmas to skip.") + words.add_argument("--include-tags", action="store_true", help="Include tags column in parsing.") + words.add_argument("--debug-min-freq", type=int, default=0, + help="Minimum frequency for debug output (default: all).") + words.add_argument("--no-clean", action="store_true", + help="Skip Anki field text cleaning before NLP.") compare = sub.add_parser("compare-words", help="Print words in source that are not in known.") compare.add_argument("source") compare.add_argument("known") + compare_new = sub.add_parser("compare", help="Compare deck vocabulary against a target list.") + compare_new.add_argument("--deck-words", required=True, help="Path to deck word list.") + compare_new.add_argument("--target-words", required=True, help="Path to target vocabulary list.") + compare_new.add_argument("--output", help="Write missing words to this file.") + compare_new.add_argument("--seen-output", help="Write seen words to this file.") + compare_new.add_argument("--min-frequency", type=int, default=0, + help="Minimum frequency for deck words to count.") + + lint_es = sub.add_parser("lint-anki-es", help="Check Spanish Anki cards for suspicious patterns.") + lint_es.add_argument("--input", required=True, help="Anki TSV export file.") + lint_es.add_argument("--field", type=int, default=2, help="1-based column index (default 2).") + lint_es.add_argument("--output", default="suspicious_cards_es.tsv", help="Output TSV path.") + lint_es.add_argument("--verbose", action="store_true", help="Output all cards, not just suspicious ones.") + youtube = sub.add_parser("youtube", help="Mine a YouTube transcript.") youtube.add_argument("lang", choices=choices) youtube.add_argument("video") @@ -144,6 +170,19 @@ def main(argv: list[str] | None = None) -> int: parser = build_parser(config) args = parser.parse_args(argv) + if args.command == "words": + lang_option = getattr(args, "lang_option", None) + if args.lang and lang_option and args.lang != lang_option: + print( + f"Error: conflicting language values: {args.lang!r} and {lang_option!r}", + file=sys.stderr, + ) + return 2 + args.lang = lang_option or args.lang + if not args.lang: + print("Error: words requires a language, e.g. 'saiki words es' or 'saiki words --lang es'", file=sys.stderr) + return 2 + if args.command == "audio": result = extract_audio(config, args.lang, args.outdir, args.media_dir, args.copy_only_new, args.concat) print(f"Copied {result['copied']} files") @@ -154,14 +193,62 @@ def main(argv: list[str] | None = None) -> int: return 0 if args.command == "words": - result = extract_words( - config, args.lang, args.query, args.deck, args.field, args.min_freq, - args.outdir, args.out, args.full_field, args.spacy_model, - ) - print(f"Query: {result['query']}") - print(f"Found {result['notes']} notes") - print(f"Extracted {result['unique']} unique entries") - print(f"Wrote {result['written']} entries to: {result['out']}") + if args.input: + # File-based extraction + if config.language_name(args.lang) != "spanish": + print( + "Error: TSV file-based word extraction is currently Spanish-only.", + file=sys.stderr, + ) + return 1 + if args.field and not args.field.isdigit(): + print( + f"Error: --field must be a 1-based column index (integer) " + f"when --input is used, got: {args.field!r}", + file=sys.stderr, + ) + return 1 + field_index = int(args.field) if args.field else 2 + if field_index < 1: + print( + f"Error: --field must be a 1-based column index >= 1, got: {field_index}", + file=sys.stderr, + ) + return 1 + result = extract_words_from_file( + config, args.lang, args.input, + field_index=field_index, + min_freq=args.min_freq, + outdir=args.outdir, + out=args.out, + include_proper_nouns=args.include_proper_nouns, + function_words=args.function_words, + debug=args.debug, + lemma_corrections=args.lemma_corrections, + bad_lemma_file=args.bad_lemma_file, + spacy_model=args.spacy_model, + include_tags=args.include_tags, + debug_min_freq=args.debug_min_freq, + no_clean=args.no_clean, + ) + print(f"Parsed {result['records']} records") + print(f"Wrote {result['written']} content entries to: {result['out']}") + if result.get("debug"): + print(f"Debug output: {result['debug']}") + if result.get("proper_nouns"): + print(f"Proper nouns: {result['proper_nouns']}") + if result.get("function_words"): + print(f"Function words: {result['function_words']}") + print(f"Suspicious tokens: {result.get('suspicious', '')}") + else: + result = extract_words( + config, args.lang, args.query, args.deck, args.field, args.min_freq, + args.outdir, args.out, args.full_field, args.spacy_model, + ) + print(f"Query: {result['query']}") + print(f"Found {result['notes']} notes") + print(f"Extracted {result['unique']} unique entries") + print(f"Wrote {result['written']} entries to: {result['out']}") return 0 if args.command == "compare-words": @@ -169,6 +256,47 @@ def main(argv: list[str] | None = None) -> int: print(line) return 0 + if args.command == "compare": + result = compare_word_lists( + args.deck_words, + args.target_words, + output_path=args.output, + min_frequency=args.min_frequency, + seen_output_path=args.seen_output, + ) + print(f"Total missing from deck: {result['total_missing']}") + seen_count = len(result["seen"]) + print(f"Total seen in deck: {seen_count}") + if result["missing"]: + print("Missing words:") + for w in result["missing"][:20]: + print(f" {w}") + if len(result["missing"]) > 20: + print(f" ... and {len(result['missing']) - 20} more") + if args.output: + print(f"Missing words written to: {args.output}") + if args.seen_output: + print(f"Seen words written to: {args.seen_output}") + return 0 + + if args.command == "lint-anki-es": + if args.field < 1: + print( + f"Error: --field must be a 1-based column index >= 1, got: {args.field}", + file=sys.stderr, + ) + return 1 + result = lint_anki_cards_main( + args.input, + field_index=args.field, + output_path=args.output, + verbose=args.verbose, + ) + print(f"Checked {result['records']} cards") + print(f"Found {result['issues']} suspicious card(s)") + print(f"Report written to: {result['path']}") + return 0 + if args.command == "youtube": result = run_youtube( config, args.lang, args.video, args.mode, args.top, args.no_stopwords, diff --git a/src/saiki/spanish.py b/src/saiki/spanish.py new file mode 100644 index 0000000..0733326 --- /dev/null +++ b/src/saiki/spanish.py @@ -0,0 +1,719 @@ +"""Spanish-specific NLP utilities for vocabulary extraction and card linting.""" + +from __future__ import annotations + +import html +import re +from collections import defaultdict +from dataclasses import dataclass +from typing import Any, Callable + +# ── Regex patterns ───────────────────────────────────────────────── + +SOUND_RE = re.compile(r"\[sound:[^\]]*\]") +MEDIA_RE = re.compile(r"\[\s*(?:sound|image|media):[^\]]*\]", re.IGNORECASE) +MEDIA_FILENAME_RE = re.compile( + r"[^/\s]+\.(?:mp3|ogg|wav|m4a|flac|jpg|jpeg|png|gif|webp|mp4|webm)\b", + re.IGNORECASE, +) +BR_RE = re.compile(r"<\s*br\s*/?\s*>", re.IGNORECASE) +DOUBLE_BR_RE = re.compile( + r"<\s*br\s*/?\s*>\s*<\s*br\s*/?\s*>", re.IGNORECASE +) +HTML_TAG_RE = re.compile(r"<[^>]+>") +URL_RE = re.compile(r"https?://\S+|www\.\S+", re.IGNORECASE) +EMAIL_RE = re.compile(r"\S+@\S+\.\S+") +MULTI_WS_RE = re.compile(r"[ \t]+") +SPANISH_WORD_RE = re.compile(r"^[a-záéíóúüñ]+$", re.IGNORECASE) +BAD_TOKEN_TEXT_RE = re.compile(r"[\d_/@#\\|+*=<>~`^]") +LEMMA_PUNCT_RE = re.compile(r"[.,;:!?¿¡\"'()\[\]{}<>\-_/\\|`~@#$%^&*+=\d]") + +_ENGLISH_PATTERNS: list[str] = [ + r"\bI\s+(?:am|have|had|will|would|could|should|was|were|did|do|don" + r"|can|need|like|want|think|know|love|hate|see|hear|eat|drink" + r"|go|come|take|give|make|say|tell|ask|get|put)\b", + r"\byou\s+(?:are|have|had|will|would|could|should|were|did|do|don" + r"|can|need|like|want|think|know|love|hate|see|hear)\b", + r"\b(the|to|and|in|of|a|an|is|it|for|with|on|at|by|from|or|be|this|that)\b", + r"\bsee you later\b", + r"\bI love you\b", + r"\bI want you\b", + r"\bknowledge/understanding\b", + r"\b(?:see|watch|read|write|speak|listen|learn|study|teach|explain)\s+" + r"(?:you|him|her|it|us|them|me)\b", +] +ENGLISH_RE = re.compile("|".join(_ENGLISH_PATTERNS), re.IGNORECASE) + +# ── Lemma corrections ────────────────────────────────────────────── + +BUILTIN_LEMMA_CORRECTIONS: dict[str, str] = { + "comar": "comer", + "acabir": "acabar", + "deliciós": "delicioso", + "llover/llover": "llover", +} + +PRIR_SURFACE_FORMS = {"pide", "pido", "piden", "pidiendo", "pedir"} + + +@dataclass(frozen=True) +class LemmaInfo: + """A selected lemma plus audit metadata about how it was chosen.""" + + lemma: str + original_lemma: str + surface: str + status: str = "ok" + +# ── Known card issues (exact substring → fix → explanation) ──────── + +KNOWN_CARD_ISSUES: list[tuple[str, str, str]] = [ + ( + "Tú sabes nada", + "Tú no sabes nada.", + "Missing negation 'no' before 'sabes'", + ), + ( + "Dulce sueños", + "Dulces sueños.", + "'Dulce' should agree in number with 'sueños'", + ), + ( + "Ella soña con viajar.", + "Ella sueña con viajar.", + "'Soña' → 'sueña' (present indicative of soñar)", + ), + ( + "Siento lastima", + "Siento lástima", + "'lastima' should be 'lástima' with accent", + ), + ( + "Tengo uno libro", + "Tengo un libro", + "'uno' → 'un' before singular masculine noun", + ), + ( + "Tengo veintiuno años", + "Tengo veintiún años", + "'veintiuno' → 'veintiún' before plural noun", + ), + ( + "Dile a tu madre que feliz cumpleaños", + 'Dile a tu madre: "Feliz cumpleaños."', + "Missing colon/quotes or restructured phrasing", + ), + ( + "La pareja está en un bote observando a los peces nadan.", + "La pareja está en un bote observando cómo nadan los peces.", + "Missing 'cómo' or subordinating conjunction", + ), + ( + "Los perros pequeños le vuelven loca.", + "Los perros pequeños la vuelven loca.", + "'le' → 'la' (direct object, not indirect)", + ), +] + +# ── Regex-based suspicious patterns ──────────────────────────────── +# (pattern, description, suggested_fix_or_None) + +SUSPICIOUS_PATTERNS: list[tuple[str, str, str | None]] = [ + ( + r"AI-generated text-to-speech", + "Metadata contamination: AI-generated TTS label in text field", + None, + ), + ( + r"\bsabes\s+nada\b(?!.*\bno\b)", + "Missing negation: 'sabes nada' without 'no'", + "Add 'no' before 'sabes': 'No sabes nada.'", + ), + ( + r"\buno\s+(?:libro|coche|casa|perro|gato|hombre|mujer|niño" + r"|amigo|día|año|mes|semana|minuto|segundo)\b", + "'uno' instead of 'un' before masculine singular noun", + "Replace 'uno' with 'un'", + ), + ( + r"\bveintiuno\s+(?:años|días|meses|semanas|horas|minutos|segundos)\b", + "'veintiuno' instead of 'veintiún' before plural noun", + "Replace 'veintiuno' with 'veintiún'", + ), + ( + r"observando\s+a\s+.*\bnadan\b", + "Possible missing 'cómo' in 'observando cómo ...' construction", + "Consider adding 'cómo': 'observando cómo ...'", + ), + ( + r"\ble\s+vuelve\w*\s+loc[ao]s?\b", + "'le vuelve/vuelven loco/a' may need a direct object pronoun", + "Replace 'le' with lo/la/los/las as appropriate", + ), + ( + r"\ble\s+vuelven\s+loc[ao]s?\b", + "'le vuelven loco/a' may need a direct object pronoun", + "Replace 'le' with lo/la/los/las as appropriate", + ), + ( + r"\bNecesito\s+[a-záéíóúüñ]{2,}\b(?!\s+(?:un|una|el|la|los" + r"|las|al|del|mi|tu|su|nuestro))\s*$", + "Missing article before the noun after 'Necesito'", + None, + ), + ( + r"\bQu[ée]\s+es\s+la\s+respuesta\b", + "'Qué es la respuesta' → 'Cuál es la respuesta'", + "Replace 'Qué' with 'Cuál'", + ), + ( + r"\bpeludos?\s+animales?\b", + "'peludo(s) animal(es)' → 'animal(es) peludo(s)' (adjective placement)", + "Place adjective after noun", + ), + ( + r"\bhembras\s+y\s+varones\b", + "'hembras y varones' — consider 'niños y niñas' or 'chicos y chicas'", + None, + ), + ( + r"\bt[ií]o\s+alto\s+con\s+uniforme\b", + "'tío alto con uniforme' — check intent (Spain slang vs 'hombre')", + None, + ), + ( + r"\brepresentante\s+adulto\b", + "'representante adulto' → 'adulto responsable'", + "Consider 'adulto responsable'", + ), + ( + r"\bpropiedad\s+a\s+este\s+representante\b", + "'propiedad a este representante' — unusual phrasing", + "Consider revising wording", + ), +] + +CONTENT_POS = frozenset({"NOUN", "VERB", "ADJ", "ADV"}) +FUNCTION_POS = frozenset( + {"PRON", "ADP", "CCONJ", "SCONJ", "DET", "AUX", "PART", "INTJ"} +) + +# ── Text cleaning ────────────────────────────────────────────────── + + +def _clean_single_field_part(text: str) -> str: + """Clean one segment of an Anki field (no English-gloss filtering).""" + text = BR_RE.sub(" ", text) + text = HTML_TAG_RE.sub("", text) + text = html.unescape(text) + text = URL_RE.sub("", text) + text = EMAIL_RE.sub("", text) + return text.strip() + + +def _looks_like_english_gloss(text: str) -> bool: + """Heuristic: does *text* read like an English gloss/translation?""" + stripped = text.strip().strip("()[]()【】") + if not stripped: + return True + return bool(ENGLISH_RE.search(stripped)) + + +def clean_anki_field_text(text: str) -> str: + """Clean Anki field text for Spanish NLP. + + * Removes ``[sound:...]`` and ``[media:...]`` markers. + * Splits on ``

`` to separate Spanish from English glosses. + * Strips HTML tags and unescapes entities. + * Removes URLs, emails. + * Drops segments that look like English translations. + * Normalises whitespace. + * Preserves Spanish accents and meaningful content. + """ + if not text: + return "" + + text = SOUND_RE.sub("", text) + text = MEDIA_RE.sub("", text) + + # Split on double
to isolate Spanish from glosses + parts = DOUBLE_BR_RE.split(text) + cleaned_parts: list[str] = [] + for i, part in enumerate(parts): + cleaned = _clean_single_field_part(part) + if not cleaned: + continue + if i > 0 and _looks_like_english_gloss(cleaned): + continue + cleaned_parts.append(cleaned) + + result = " ".join(cleaned_parts) + result = MULTI_WS_RE.sub(" ", result).strip() + return result + + +# ── Safe lemmatisation ───────────────────────────────────────────── + + +def load_lemma_corrections(path: str) -> dict[str, str]: + """Load a ``badgood`` TSV of manual lemma fixes.""" + corrections: dict[str, str] = {} + with open(path, "r", encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if not line or line.startswith("#"): + continue + parts = line.split("\t") + if len(parts) >= 2: + bad = parts[0].strip().lower() + good = parts[1].strip().lower() + if bad and good: + corrections[bad] = good + return corrections + + +def load_bad_lemmas(path: str) -> set[str]: + """Load a list of lemmas (one per line) to skip and report.""" + bad: set[str] = set() + with open(path, "r", encoding="utf-8") as fh: + for line in fh: + lemma = line.strip() + if lemma and not lemma.startswith("#"): + bad.add(lemma.lower()) + return bad + + +def safe_spanish_lemma( + token, + extra_corrections: dict[str, str] | None = None, +) -> str: + """Return only the selected lemma for callers that do not need metadata.""" + return safe_spanish_lemma_info(token, extra_corrections=extra_corrections).lemma + + +def safe_spanish_lemma_info( + token, + extra_corrections: dict[str, str] | None = None, +) -> LemmaInfo: + """Return a safe Spanish lemma for a spaCy token. + + Falls back to ``token.text.lower()`` when the spaCy lemma is empty, + multi-word, contains punctuation, or looks non-Spanish. Applies a + built-in and optionally an external correction map, and records why + fallback/correction happened. + """ + manual = dict(BUILTIN_LEMMA_CORRECTIONS) + if extra_corrections: + manual.update({k.lower(): v.lower() for k, v in extra_corrections.items()}) + + lemma = (token.lemma_ or "").strip().lower() + surface = (token.text or "").strip().lower() + + # Manual correction map wins + if lemma in manual: + return LemmaInfo( + manual[lemma], + original_lemma=lemma, + surface=surface, + status=f"corrected:{lemma}->{manual[lemma]}", + ) + + # Empty or one-char fallback + if not lemma or len(lemma) <= 1: + return LemmaInfo( + surface or lemma, + original_lemma=lemma, + surface=surface, + status="fallback_empty_or_short_lemma", + ) + + # Multi-word lemma → suspicious spaCy output + if " " in lemma: + return LemmaInfo( + surface, + original_lemma=lemma, + surface=surface, + status="fallback_multi_word_lemma", + ) + + # Punctuation or digits in lemma → fall back + if LEMMA_PUNCT_RE.search(lemma): + return LemmaInfo( + surface, + original_lemma=lemma, + surface=surface, + status="fallback_bad_lemma_characters", + ) + + # Special case: "prir" might be spaCy failing on pedir forms + if lemma == "prir": + if surface in PRIR_SURFACE_FORMS: + return LemmaInfo( + "pedir", + original_lemma=lemma, + surface=surface, + status="corrected:prir->pedir", + ) + # otherwise return surface so it shows up in debug as suspicious + return LemmaInfo( + surface, + original_lemma=lemma, + surface=surface, + status="suspicious_prir", + ) + + if not SPANISH_WORD_RE.match(lemma): + return LemmaInfo( + surface, + original_lemma=lemma, + surface=surface, + status="fallback_non_spanish_lemma", + ) + + return LemmaInfo(lemma, original_lemma=lemma, surface=surface) + + +# ── Token filters ────────────────────────────────────────────────── + + +def spanish_content_filter( + token, + include_proper_nouns: bool = False, +) -> bool: + """Return True for Spanish content words (NOUN, VERB, ADJ, ADV). + + Excludes punctuation, digits, symbols, URLs, emails, media + filenames, and one-character junk. + """ + text = (token.text or "").strip() + if not text: + return False + if token.is_punct or token.is_space or token.is_digit: + return False + if token.is_currency: + return False + if token.like_url or token.like_email: + return False + if MEDIA_FILENAME_RE.search(text): + return False + if BAD_TOKEN_TEXT_RE.search(text): + return False + if len(text) <= 1 and text not in ("a", "y", "e", "o", "u", "él"): + return False + pos = getattr(token, "pos_", "") or "" + if pos in CONTENT_POS: + return True + if include_proper_nouns and pos == "PROPN": + return True + return False + + +def spanish_function_word_filter(token) -> bool: + """Return True for Spanish function words (PRON, ADP, CCONJ, …).""" + text = (token.text or "").strip() + if not text: + return False + if token.is_punct or token.is_space or token.like_url or token.like_email: + return False + if MEDIA_FILENAME_RE.search(text) or BAD_TOKEN_TEXT_RE.search(text): + return False + pos = getattr(token, "pos_", "") or "" + return pos in FUNCTION_POS + + +# ── Suspicious-card detection ────────────────────────────────────── + + +def check_card_for_issues(text: str, cleaned: str) -> list[dict[str, Any]]: + """Check a single card's text for suspicious patterns. + + Returns a list of issue dicts (normally 0 or 1 per card, but one + card can match multiple patterns). + """ + issues: list[dict[str, Any]] = [] + seen_reasons: set[str] = set() + + for bad_text, fix, explanation in KNOWN_CARD_ISSUES: + if bad_text in text: + if explanation not in seen_reasons: + seen_reasons.add(explanation) + issues.append({ + "reason": explanation, + "suggested_fix": fix, + }) + + for pattern, description, fix in SUSPICIOUS_PATTERNS: + if re.search(pattern, text, re.IGNORECASE): + if description not in seen_reasons: + seen_reasons.add(description) + issues.append({ + "reason": description, + "suggested_fix": fix or "", + }) + + # Check for English after

+ parts = DOUBLE_BR_RE.split(text) + for part in parts[1:]: + cleaned_part = _clean_single_field_part(part) + if _looks_like_english_gloss(cleaned_part): + desc = "English gloss after

" + if desc not in seen_reasons: + seen_reasons.add(desc) + issues.append({"reason": desc, "suggested_fix": ""}) + + # Check for missing final punctuation in full-sentence cards + stripped_text = text.rstrip() + if ( + stripped_text + and not stripped_text.endswith((".", "!", "?", "…", '"', "'")) + and stripped_text[0].isupper() + and len(stripped_text.split()) >= 3 + ): + desc = "Missing final punctuation for full-sentence card" + if desc not in seen_reasons: + seen_reasons.add(desc) + issues.append({"reason": desc, "suggested_fix": ""}) + + return issues + + +def lint_anki_cards( + records: list[dict[str, Any]], + verbose: bool = False, +) -> list[dict[str, Any]]: + """Run all suspicious-card checks across parsed Anki records. + + When *verbose* is True every record is returned, not just those + with issues. + + Returns a list of issue dicts with keys: + ``source_line``, ``original_text``, ``cleaned_text``, + ``reason``, ``suggested_fix``. + """ + results: list[dict[str, Any]] = [] + for record in records: + text = record.get("text", "") + if not text: + if verbose: + results.append({ + "source_line": record.get("line_number", ""), + "original_text": record.get("raw_line", ""), + "cleaned_text": "", + "reason": "empty field", + "suggested_fix": "", + }) + continue + cleaned = clean_anki_field_text(text) + issues = check_card_for_issues(text, cleaned) + if verbose or issues: + if issues: + combined_reason = "; ".join( + sorted({i["reason"] for i in issues}) + ) + suggested = "" + for i in issues: + if i.get("suggested_fix"): + suggested = i["suggested_fix"] + break + else: + combined_reason = "" + suggested = "" + results.append({ + "source_line": record.get("line_number", ""), + "original_text": record.get("raw_line", text), + "cleaned_text": cleaned, + "reason": combined_reason, + "suggested_fix": suggested, + }) + + return results + + +# ── Detailed extraction helpers ───────────────────────────────────── + + +def extract_detailed_counts( + texts: list[dict[str, Any]], + nlp, + token_filter: Callable, + lemma_fn: Callable, + max_examples: int = 3, + clean: bool = True, +) -> dict[str, dict[str, Any]]: + """Build detailed per-lemma statistics from a list of ``{text, line_number}`` records. + + When *clean* is False the raw text is passed to the NLP pipeline + without going through ``clean_anki_field_text``. + + Returns :: + + {lemma: {"count": int, + "pos_counts": {POS: int, …}, + "surface_forms": [str, …], + "example_sentences": [str, …], + "source_lines": [str, …]}} + """ + stats: dict[str, dict[str, Any]] = {} + + for record in texts: + raw = record.get("text", "") + line_number = record.get("line_number", 0) + if not raw: + continue + txt = clean_anki_field_text(raw) if clean else raw.strip() + if not txt: + continue + + doc = nlp(txt) + for token in doc: + if not token_filter(token): + continue + lemma_value = lemma_fn(token) + if lemma_value is None: + continue + if isinstance(lemma_value, LemmaInfo): + lemma = lemma_value.lemma + original_lemma = lemma_value.original_lemma + lemma_status = lemma_value.status + else: + lemma = str(lemma_value) + original_lemma = (getattr(token, "lemma_", "") or "").strip().lower() + lemma_status = "ok" + if not lemma: + continue + + if lemma not in stats: + stats[lemma] = { + "count": 0, + "pos_counts": defaultdict(int), + "surface_forms": set(), + "original_lemmas": set(), + "lemma_statuses": defaultdict(int), + "example_sentences": [], + "source_lines": [], + } + + s = stats[lemma] + s["count"] += 1 + s["pos_counts"][token.pos_] += 1 + s["surface_forms"].add(token.text.lower()) + if original_lemma: + s["original_lemmas"].add(original_lemma) + s["lemma_statuses"][lemma_status] += 1 + if len(s["example_sentences"]) < max_examples: + sent_text = txt + if sent_text not in s["example_sentences"]: + s["example_sentences"].append(sent_text) + if line_number: + s["source_lines"].append(str(line_number)) + + # Convert defaultdicts/sets to plain types for serialisation + for lemma, s in stats.items(): + s["pos_counts"] = dict(s["pos_counts"]) + s["surface_forms"] = sorted(s["surface_forms"]) + s["original_lemmas"] = sorted(s.get("original_lemmas", [])) + s["lemma_statuses"] = dict(s.get("lemma_statuses", {})) + s["source_lines"] = list(dict.fromkeys(s["source_lines"])) # dedup, preserve order + + return stats + + +# ── Output helpers ───────────────────────────────────────────────── + + +def write_debug_tsv( + stats: dict[str, dict[str, Any]], + path: str, + bad_lemmas: set[str] | None = None, + min_freq: int = 0, +) -> None: + """Write a debug TSV with columns:: + + lemma count pos_counts top_surface_forms example_sentences source_lines + + Appends a special line per lemma in *bad_lemmas* with an + ``(EXCLUDED)`` marker. When *min_freq* > 0 only lemmas with + count >= min_freq are included. + """ + import os + + os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True) + + bad = bad_lemmas or set() + + with open(path, "w", encoding="utf-8") as fh: + fh.write( + "lemma\tcount\tpos_counts\ttop_surface_forms" + "\texample_sentences\tsource_lines\toriginal_lemmas" + "\tlemma_statuses\tstatus\n" + ) + sorted_lemmas = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0])) + for lemma, s in sorted_lemmas: + if min_freq > 0 and s["count"] < min_freq: + continue + pos_counts_str = ", ".join( + f"{pos}:{cnt}" + for pos, cnt in sorted( + s["pos_counts"].items(), key=lambda x: -x[1] + ) + ) + surfaces_str = ", ".join(s["surface_forms"]) + examples_str = "; ".join(s["example_sentences"]) + lines_str = ", ".join(s["source_lines"]) + original_lemmas_str = ", ".join(s.get("original_lemmas", [])) + lemma_statuses_str = ", ".join( + f"{status}:{cnt}" + for status, cnt in sorted( + s.get("lemma_statuses", {}).items(), + key=lambda x: (-x[1], x[0]), + ) + ) + + status = "EXCLUDED" if lemma in bad else "" + fh.write( + f"{lemma}\t{s['count']}\t{pos_counts_str}\t{surfaces_str}" + f"\t{examples_str}\t{lines_str}\t{original_lemmas_str}" + f"\t{lemma_statuses_str}\t{status}\n" + ) + + +def write_suspicious_tokens( + stats: dict[str, dict[str, Any]], + path: str, + bad_lemmas: set[str] | None = None, +) -> None: + """Write a file of suspicious lemmas (multi-word, punctuation, etc.) + + Any lemma that was corrected, fell back from a suspicious spaCy lemma, + or appears in the user blocklist is listed here. + """ + import os + + os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True) + bad = bad_lemmas or set() + + with open(path, "w", encoding="utf-8") as fh: + fh.write("lemma\tcount\tsurface_forms\toriginal_lemmas\tnote\n") + fh.write( + "# Lemmas that required correction or fallback are listed here\n" + ) + for lemma, s in sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0])): + statuses = { + status: count + for status, count in s.get("lemma_statuses", {}).items() + if status != "ok" + } + if lemma in bad: + statuses["blocked_bad_lemma"] = s["count"] + if not statuses: + continue + surfaces_str = ", ".join(s["surface_forms"]) + originals_str = ", ".join(s.get("original_lemmas", [])) + note = ", ".join( + f"{status}:{count}" + for status, count in sorted(statuses.items()) + ) + fh.write( + f"{lemma}\t{s['count']}\t{surfaces_str}" + f"\t{originals_str}\t{note}\n" + ) diff --git a/src/saiki/words.py b/src/saiki/words.py index 5c7ff3d..02ac565 100644 --- a/src/saiki/words.py +++ b/src/saiki/words.py @@ -5,14 +5,29 @@ from __future__ import annotations import logging import os from collections import Counter -from typing import Callable import regex as re +from typing import Any, Callable from .ankiconnect import anki_request +from .anki_tsv import parse_anki_tsv from .config import Config +from .spanish import ( + clean_anki_field_text, + extract_detailed_counts, + lint_anki_cards, + load_bad_lemmas, + load_lemma_corrections, + safe_spanish_lemma_info, + spanish_content_filter, + spanish_function_word_filter, + write_debug_tsv, + write_suspicious_tokens, +) from .text import extract_first_visible_line, extract_visible_text, normalize_word_key +ACCENT_MAP = str.maketrans("áéíóúüñÁÉÍÓÚÜÑ", "aeiouunAEIOUUN") + JAPANESE_CHAR_RE = re.compile(r"[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}ー]+") JAPANESE_PARTICLES = { "は", "が", "を", "に", "へ", "で", "と", "や", "も", "から", "まで", "より", "ば", "なら", @@ -202,3 +217,311 @@ def extract_words( counter = extract_counts(notes, field_name, nlp, profile["token_filter"], profile["output_format"], full_field) written = write_counts(counter, out_path, min_freq) return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path} + + +def _write_content_words( + stats: dict[str, dict[str, Any]], + path: str, + min_freq: int, + bad_lemmas: set[str] | None = None, +) -> int: + """Write a sorted ``lemma count`` file from detailed stats.""" + bad = bad_lemmas or set() + items = [ + (lemma, data["count"]) + for lemma, data in stats.items() + if data["count"] >= min_freq and lemma not in bad + ] + items.sort(key=lambda x: (-x[1], x[0])) + os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True) + with open(path, "w", encoding="utf-8") as f: + for word, freq in items: + f.write(f"{word} {freq}\n") + return len(items) + + +def _write_function_words( + stats: dict[str, dict[str, Any]], + path: str, +) -> int: + """Write a function-word frequency file from detailed stats.""" + os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True) + items = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0])) + with open(path, "w", encoding="utf-8") as f: + for word, data in items: + f.write(f"{word} {data['count']}\n") + return len(items) + + +def extract_words_from_file( + config: Config, + lang: str, + input_path: str, + field_index: int = 2, + min_freq: int = 2, + outdir: str | None = None, + out: str | None = None, + include_proper_nouns: bool = False, + function_words: bool = False, + debug: str | None = None, + lemma_corrections: str | None = None, + bad_lemma_file: str | None = None, + spacy_model: str | None = None, + include_tags: bool = False, + debug_min_freq: int = 0, + no_clean: bool = False, +) -> dict[str, Any]: + """Extract frequent Spanish vocabulary from an Anki TSV export file. + + Parameters + ---------- + config: + Saiki configuration. + lang: + Language code (e.g. ``"es"``). + input_path: + Path to the Anki TSV export file. + field_index: + 1-based column index of the Spanish text (default 2). + min_freq: + Minimum frequency to include in output. + outdir: + Output directory (defaults to config word output root). + out: + Output filename or path for content words. + include_proper_nouns: + Include ``PROPN`` tokens. + function_words: + Also extract a separate function-word list. + debug: + Path for the debug TSV. + lemma_corrections: + Path to a ``bad\\tgood`` TSV of lemma corrections. + bad_lemma_file: + Path to a list of lemmas to skip. + spacy_model: + SpaCy model name override. + + Returns + ------- + A dict with keys ``records``, ``stats``, ``written``, ``out``, + ``debug``, ``proper_nouns``, ``function_words``, ``suspicious``. + """ + language_bucket = config.language_name(lang) + out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket) + out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}_content.txt") + model_name = spacy_model or str(config.language(lang).get("word_model")) + nlp = load_spacy_model(model_name) + + corrections = None + if lemma_corrections: + corrections = load_lemma_corrections(lemma_corrections) + bad_lemmas = None + if bad_lemma_file: + bad_lemmas = load_bad_lemmas(bad_lemma_file) + + records = parse_anki_tsv( + input_path, + field_index=field_index, + include_tags=include_tags, + ) + if not records: + return {"records": 0, "stats": {}, "written": 0, "out": out_path} + + # Build token filter + def _content_filter(token) -> bool: + return spanish_content_filter(token, include_proper_nouns=include_proper_nouns) + + stats = extract_detailed_counts( + records, nlp, _content_filter, + lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections), + clean=not no_clean, + ) + + written = _write_content_words(stats, out_path, min_freq, bad_lemmas=bad_lemmas) + + result: dict[str, Any] = { + "records": len(records), + "stats": stats, + "written": written, + "out": out_path, + } + + # Debug output + if debug: + debug_path = os.path.expanduser(debug) + write_debug_tsv(stats, debug_path, bad_lemmas=bad_lemmas, min_freq=debug_min_freq) + result["debug"] = debug_path + + # Proper nouns + if include_proper_nouns: + proper_path = os.path.join(out_dir, f"words_{lang}_proper_nouns.txt") + proper_items = [ + (lemma, data["count"]) + for lemma, data in stats.items() + if data["pos_counts"].get("PROPN", 0) > 0 + ] + proper_items.sort(key=lambda x: (-x[1], x[0])) + os.makedirs(os.path.dirname(os.path.abspath(proper_path)), exist_ok=True) + with open(proper_path, "w", encoding="utf-8") as f: + for w, c in proper_items: + f.write(f"{w} {c}\n") + result["proper_nouns"] = proper_path + + # Function words (separate extraction pass) + if function_words: + fw_path = os.path.join(out_dir, f"words_{lang}_function_words.txt") + fw_stats = extract_detailed_counts( + records, nlp, spanish_function_word_filter, + lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections), + ) + fw_written = _write_function_words(fw_stats, fw_path) + result["function_words"] = fw_path + result["function_words_written"] = fw_written + + # Suspicious tokens + susp_path = os.path.join(out_dir, f"words_{lang}_suspicious_tokens.txt") + write_suspicious_tokens(stats, susp_path, bad_lemmas=bad_lemmas) + result["suspicious"] = susp_path + + return result + + +def lint_anki_cards_main( + input_path: str, + field_index: int = 2, + output_path: str = "suspicious_cards_es.tsv", + verbose: bool = False, +) -> dict[str, Any]: + """Read an Anki TSV export and produce a suspicious-card report. + + When *verbose* is True every record is written to the report, not + only those with issues. + + Returns ``{"records": …, "issues": …, "path": output_path}``. + """ + records = parse_anki_tsv(input_path, field_index=field_index) + issues = lint_anki_cards(records, verbose=verbose) + + os.makedirs(os.path.dirname(os.path.abspath(output_path)) or ".", exist_ok=True) + with open(output_path, "w", encoding="utf-8") as fh: + fh.write( + "source_line\toriginal_text\tcleaned_text\treason\tsuggested_fix\n" + ) + for issue in issues: + fh.write( + f"{issue['source_line']}\t" + f"{_tsv_escape(issue['original_text'])}\t" + f"{_tsv_escape(issue['cleaned_text'])}\t" + f"{_tsv_escape(issue['reason'])}\t" + f"{_tsv_escape(issue['suggested_fix'])}\n" + ) + + return {"records": len(records), "issues": len(issues), "path": output_path} + + +def _tsv_escape(value: str) -> str: + """Escape a value for TSV output (backslash-escape tabs and newlines).""" + return value.replace("\\", "\\\\").replace("\t", "\\t").replace("\n", "\\n") + + +def normalize_word_for_comparison(word: str) -> str: + """Normalise a word for file-comparison: lowercase, strip accents.""" + return word.lower().translate(ACCENT_MAP) + + +def read_word_file_normalized(path: str) -> dict[str, set[str]]: + """Read a word-frequency file returning ``{normalised: {display_forms}}``. + + Also returns a flat set of normalised forms. + """ + display_map: dict[str, set[str]] = {} + with open(os.path.expanduser(path), "r", encoding="utf-8") as fh: + for line in fh: + stripped = line.strip() + if not stripped: + continue + word = stripped.rsplit(" ", 1)[0] + key = normalize_word_for_comparison(word) + display_map.setdefault(key, set()).add(word) + return display_map + + +def compare_word_lists( + deck_words_path: str, + target_words_path: str, + output_path: str | None = None, + min_frequency: int = 0, + seen_output_path: str | None = None, +) -> dict[str, Any]: + """Compare a deck word list against a target vocabulary list. + + Parameters + ---------- + deck_words_path: + Path to the deck's extracted word list (``lemma count`` format). + target_words_path: + Path to the target vocabulary list (same format). + output_path: + If given, write missing-from-deck words here. + min_frequency: + Only consider deck words with frequency >= this value. + seen_output_path: + If given, write seen-in-deck words here. + + Returns + ------- + ``{"missing": [str], "seen": [str], "total_missing": int}`` + """ + # Read target words + target_map = read_word_file_normalized(target_words_path) + target_keys = set(target_map.keys()) + + # Read deck words, optionally filtering by frequency + deck_map: dict[str, set[str]] = {} + with open(os.path.expanduser(deck_words_path), "r", encoding="utf-8") as fh: + for line in fh: + stripped = line.strip() + if not stripped: + continue + parts = stripped.rsplit(" ", 1) + word = parts[0] + freq = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else 1 + if freq < min_frequency: + continue + key = normalize_word_for_comparison(word) + deck_map.setdefault(key, set()).add(word) + deck_keys = set(deck_map.keys()) + + missing_keys = target_keys - deck_keys + seen_keys = target_keys & deck_keys + + missing_entries: list[str] = [] + for key in sorted(missing_keys): + for display in sorted(target_map[key]): + missing_entries.append(display) + + seen_entries: list[str] = [] + for key in sorted(seen_keys): + for display in sorted(target_map[key]): + seen_entries.append(display) + + if output_path: + out = os.path.expanduser(output_path) + os.makedirs(os.path.dirname(os.path.abspath(out)) or ".", exist_ok=True) + with open(out, "w", encoding="utf-8") as fh: + for entry in missing_entries: + fh.write(f"{entry}\n") + + if seen_output_path: + seen_out = os.path.expanduser(seen_output_path) + os.makedirs(os.path.dirname(os.path.abspath(seen_out)) or ".", exist_ok=True) + with open(seen_out, "w", encoding="utf-8") as fh: + for entry in seen_entries: + fh.write(f"{entry}\n") + + return { + "missing": missing_entries, + "seen": seen_entries, + "total_missing": len(missing_entries), + } diff --git a/tests/test_core.py b/tests/test_core.py index 001c648..4ae5d83 100644 --- a/tests/test_core.py +++ b/tests/test_core.py @@ -12,7 +12,9 @@ SRC_DIR = Path(__file__).resolve().parents[1] / "src" if str(SRC_DIR) not in sys.path: sys.path.insert(0, str(SRC_DIR)) +from saiki.anki_tsv import parse_anki_headers, parse_anki_tsv from saiki.audio import build_playlist, resolve_media_paths +from saiki.cli import build_parser, main from saiki.config import Config, DEFAULT_CONFIG, deep_merge from saiki.importer import ( PreparedTtsBackend, @@ -24,8 +26,28 @@ from saiki.importer import ( synthesize_tts_sample, supported_tts_backends, ) +from saiki.spanish import ( + BUILTIN_LEMMA_CORRECTIONS, + check_card_for_issues, + clean_anki_field_text, + extract_detailed_counts, + lint_anki_cards, + load_lemma_corrections, + safe_spanish_lemma, + safe_spanish_lemma_info, + spanish_content_filter, + spanish_function_word_filter, + write_debug_tsv, + write_suspicious_tokens, +) from saiki.text import extract_first_visible_line, extract_visible_text -from saiki.words import build_query_from_decks, compare_word_files, read_word_file +from saiki.words import ( + build_query_from_decks, + compare_word_files, + compare_word_lists, + normalize_word_for_comparison, + read_word_file, +) from saiki.youtube import TranscriptLine, extract_video_id, sentence_vocab, write_sentence_export @@ -81,6 +103,630 @@ class WordsTests(unittest.TestCase): self.assertEqual(compare_word_files(source, known), ["hablar 2"]) +class AnkiTsvTests(unittest.TestCase): + def test_parse_headers_separator_tab(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#separator:tab\n#html:true\n#tags column:3\ncol1\tcol2\tcol3\n") + headers = parse_anki_headers(path) + self.assertEqual(headers["separator"], "\t") + self.assertTrue(headers["html"]) + self.assertEqual(headers["tags_column"], "3") + + def test_parse_headers_default_separator(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("# some comment\ncol1\tcol2\n") + headers = parse_anki_headers(path) + self.assertNotIn("separator", headers) + + def test_parse_tsv_skips_header_lines(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#separator:tab\n#html:true\nA\tB\tC\nD\tE\tF\n") + records = parse_anki_tsv(path, field_index=2) + self.assertEqual(len(records), 2) + self.assertEqual(records[0]["text"], "B") + self.assertEqual(records[1]["text"], "E") + + def test_parse_tsv_field_index_1_based(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("alpha\tbeta\tgamma\n") + records = parse_anki_tsv(path, field_index=1) + self.assertEqual(records[0]["text"], "alpha") + records = parse_anki_tsv(path, field_index=3) + self.assertEqual(records[0]["text"], "gamma") + + def test_parse_tsv_default_field_2_ignores_sound_field(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#separator:tab\n#tags column:3\n") + f.write("[sound:es_001.mp3]\tYo como manzanas.\tanki-tag\n") + records = parse_anki_tsv(path) + self.assertEqual(records[0]["text"], "Yo como manzanas.") + self.assertNotIn("tags", records[0]) + + def test_parse_tsv_rejects_zero_field_index(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("alpha\tbeta\tgamma\n") + with self.assertRaises(ValueError): + parse_anki_tsv(path, field_index=0) + + def test_parse_tsv_rejects_missing_field(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("alpha\tbeta\n") + with self.assertRaises(ValueError): + parse_anki_tsv(path, field_index=3) + + def test_parse_tsv_line_numbers(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#header\nuno\tdos\nthree\tfour\n") + records = parse_anki_tsv(path, field_index=2) + self.assertEqual(records[0]["line_number"], 2) + self.assertEqual(records[1]["line_number"], 3) + + def test_parse_tsv_include_tags(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#tags column:3\nx\ty\ttag_a\n") + records = parse_anki_tsv(path, field_index=2, include_tags=True) + self.assertEqual(records[0]["tags"], "tag_a") + + def test_parse_tsv_without_tags(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "test.txt") + with open(path, "w", encoding="utf-8") as f: + f.write("#tags column:3\nx\ty\ttag_a\n") + records = parse_anki_tsv(path, field_index=2, include_tags=False) + self.assertNotIn("tags", records[0]) + + +class CliValidationTests(unittest.TestCase): + """Test CLI-level validation against the real parser/main paths.""" + + def test_words_accepts_lang_option_and_output_alias(self): + with patch("saiki.cli.importlib.metadata.version", return_value="0"): + parser = build_parser() + args = parser.parse_args([ + "words", "--lang", "es", "--input", "deck.tsv", "--output", "out.txt" + ]) + self.assertEqual(args.lang_option, "es") + self.assertEqual(args.out, "out.txt") + + def test_words_input_rejects_zero_field(self): + from io import StringIO + + with patch("saiki.cli.importlib.metadata.version", return_value="0"): + with patch("sys.stderr", new_callable=StringIO) as stderr: + code = main(["words", "--lang", "es", "--input", "deck.tsv", "--field", "0"]) + self.assertEqual(code, 1) + self.assertIn("1-based", stderr.getvalue()) + + def test_words_input_is_spanish_only(self): + from io import StringIO + + with patch("saiki.cli.importlib.metadata.version", return_value="0"): + with patch("sys.stderr", new_callable=StringIO) as stderr: + code = main(["words", "jp", "--input", "deck.tsv", "--field", "2"]) + self.assertEqual(code, 1) + self.assertIn("Spanish-only", stderr.getvalue()) + + +class SpanishTextCleaningTests(unittest.TestCase): + def test_remove_sound_marker(self): + self.assertEqual( + clean_anki_field_text("[sound:es_001.mp3] Hola mundo"), + "Hola mundo", + ) + + def test_strip_html_and_unescape(self): + result = clean_anki_field_text("Para el lunes...") + self.assertEqual(result, "Para el lunes...") + + def test_english_after_brbr_removed(self): + result = clean_anki_field_text("hasta luego

(see you later)") + self.assertEqual(result, "hasta luego") + + def test_english_gloss_after_brbr_removed(self): + result = clean_anki_field_text("Te quiero.

I love you / I want you") + self.assertEqual(result, "Te quiero.") + + def test_spanish_after_brbr_preserved(self): + result = clean_anki_field_text("Hola

Otra línea") + self.assertEqual(result, "Hola Otra línea") + + def test_normalize_whitespace(self): + result = clean_anki_field_text(" Hola mundo ") + self.assertEqual(result, "Hola mundo") + + def test_br_variants(self): + result = clean_anki_field_text("Line1
Line2
Line3") + self.assertEqual(result, "Line1 Line2 Line3") + + def test_preserves_accents(self): + result = clean_anki_field_text("camión corazón") + self.assertEqual(result, "camión corazón") + + def test_cleaned_text_empty_input(self): + self.assertEqual(clean_anki_field_text(""), "") + self.assertEqual(clean_anki_field_text(None), "") + + +class MockToken: + """Minimal mock of a spaCy token for lemmatisation tests.""" + + def __init__(self, text="", lemma="", pos="", **kwargs): + self.text = text + self.lemma_ = lemma + self.pos_ = pos + self.is_stop = kwargs.get("is_stop", False) + self.is_alpha = kwargs.get("is_alpha", True) + self.is_punct = kwargs.get("is_punct", False) + self.is_space = kwargs.get("is_space", False) + self.is_digit = kwargs.get("is_digit", False) + self.is_currency = kwargs.get("is_currency", False) + self.like_url = kwargs.get("like_url", False) + self.like_email = kwargs.get("like_email", False) + self.sent = None + + +class SpanishLemmaTests(unittest.TestCase): + def test_builtin_correction_comar(self): + token = MockToken(text="como", lemma="comar") + self.assertEqual(safe_spanish_lemma(token), "comer") + + def test_builtin_correction_acabir(self): + token = MockToken(text="acabas", lemma="acabir") + self.assertEqual(safe_spanish_lemma(token), "acabar") + + def test_builtin_correction_delicius(self): + token = MockToken(text="delicioso", lemma="deliciós") + self.assertEqual(safe_spanish_lemma(token), "delicioso") + + def test_builtin_correction_llover_slash(self): + token = MockToken(text="llueve", lemma="llover/llover") + self.assertEqual(safe_spanish_lemma(token), "llover") + + def test_multi_word_lemma_falls_back(self): + token = MockToken(text="ayudar", lemma="ayudar yo") + self.assertEqual(safe_spanish_lemma(token), "ayudar") + + def test_multi_word_lemma_lavar_el(self): + token = MockToken(text="lava", lemma="lavar él") + self.assertEqual(safe_spanish_lemma(token), "lava") + + def test_prir_corrected_when_source_is_pedir(self): + for form in ["pide", "pido", "piden", "pidiendo", "pedir"]: + token = MockToken(text=form, lemma="prir") + self.assertEqual(safe_spanish_lemma(token), "pedir") + + def test_prir_not_corrected_for_unknown_form(self): + token = MockToken(text="prir", lemma="prir") + self.assertEqual(safe_spanish_lemma(token), "prir") + + def test_empty_lemma_falls_back(self): + token = MockToken(text="hola", lemma="") + self.assertEqual(safe_spanish_lemma(token), "hola") + + def test_lemma_with_punctuation_falls_back(self): + token = MockToken(text="comiendo", lemma="comiendo,") + self.assertEqual(safe_spanish_lemma(token), "comiendo") + + def test_normal_lemma_passes_through(self): + token = MockToken(text="comiendo", lemma="comer") + self.assertEqual(safe_spanish_lemma(token), "comer") + + def test_lemma_info_tracks_correction_reason(self): + token = MockToken(text="como", lemma="comar") + info = safe_spanish_lemma_info(token) + self.assertEqual(info.lemma, "comer") + self.assertEqual(info.original_lemma, "comar") + self.assertIn("corrected", info.status) + + def test_non_spanish_lemma_falls_back(self): + token = MockToken(text="hola", lemma="hello_world") + info = safe_spanish_lemma_info(token) + self.assertEqual(info.lemma, "hola") + self.assertEqual(info.status, "fallback_bad_lemma_characters") + + def test_extra_corrections_override_builtin(self): + extra = {"comar": "comprar"} + token = MockToken(text="como", lemma="comar") + self.assertEqual( + safe_spanish_lemma(token, extra_corrections=extra), + "comprar", + ) + + def test_load_lemma_corrections(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "corrections.tsv") + with open(path, "w", encoding="utf-8") as f: + f.write("malo\tbueno\nfeo\tbonito\n") + corrections = load_lemma_corrections(path) + self.assertEqual(corrections, {"malo": "bueno", "feo": "bonito"}) + + def test_load_lemma_corrections_skips_comments(self): + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "corrections.tsv") + with open(path, "w", encoding="utf-8") as f: + f.write("# comment\nmalo\tbueno\n") + corrections = load_lemma_corrections(path) + self.assertEqual(corrections, {"malo": "bueno"}) + + +class SpanishFilterTests(unittest.TestCase): + def test_content_filter_keeps_noun(self): + token = MockToken(text="casa", lemma="casa", pos="NOUN") + self.assertTrue(spanish_content_filter(token)) + + def test_content_filter_keeps_verb(self): + token = MockToken(text="come", lemma="comer", pos="VERB") + self.assertTrue(spanish_content_filter(token)) + + def test_content_filter_keeps_adj(self): + token = MockToken(text="grande", lemma="grande", pos="ADJ") + self.assertTrue(spanish_content_filter(token)) + + def test_content_filter_keeps_adv(self): + token = MockToken(text="bien", lemma="bien", pos="ADV") + self.assertTrue(spanish_content_filter(token)) + + def test_content_filter_excludes_propn_by_default(self): + token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN") + self.assertFalse(spanish_content_filter(token)) + + def test_content_filter_includes_propn_with_flag(self): + token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN") + self.assertTrue( + spanish_content_filter(token, include_proper_nouns=True) + ) + + def test_content_filter_excludes_punct(self): + token = MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True) + self.assertFalse(spanish_content_filter(token)) + + def test_content_filter_excludes_digit(self): + token = MockToken(text="123", lemma="123", pos="NUM", is_digit=True) + self.assertFalse(spanish_content_filter(token)) + + def test_content_filter_excludes_url(self): + token = MockToken( + text="http://x.com", lemma="http://x.com", pos="X", like_url=True + ) + self.assertFalse(spanish_content_filter(token)) + + def test_content_filter_excludes_media_filename(self): + token = MockToken(text="es_001.mp3", lemma="es_001.mp3", pos="NOUN") + self.assertFalse(spanish_content_filter(token)) + + def test_content_filter_excludes_one_char_non_vowel(self): + token = MockToken(text="x", lemma="x", pos="NOUN") + self.assertFalse(spanish_content_filter(token)) + + def test_function_word_filter_keeps_preposition(self): + token = MockToken(text="para", lemma="para", pos="ADP") + self.assertTrue(spanish_function_word_filter(token)) + + def test_function_word_filter_keeps_conjunction(self): + token = MockToken(text="y", lemma="y", pos="CCONJ") + self.assertTrue(spanish_function_word_filter(token)) + + def test_function_word_filter_rejects_noun(self): + token = MockToken(text="casa", lemma="casa", pos="NOUN") + self.assertFalse(spanish_function_word_filter(token)) + + +class SpanishLintTests(unittest.TestCase): + def _check(self, text: str) -> list: + return check_card_for_issues(text, clean_anki_field_text(text)) + + def test_detect_tu_sabes_nada(self): + issues = self._check("Tú sabes nada") + self.assertTrue(any("Missing negation" in i["reason"] for i in issues)) + + def test_detect_dulce_suenos(self): + issues = self._check("Dulce sueños") + self.assertTrue( + any("agree in number" in i["reason"] for i in issues) + ) + + def test_detect_ella_sona(self): + issues = self._check("Ella soña con viajar.") + self.assertTrue( + any("present indicative" in i["reason"] for i in issues) + ) + + def test_detect_tengo_uno_libro(self): + issues = self._check("Tengo uno libro en la mesa.") + self.assertTrue( + any("instead of 'un'" in i["reason"] for i in issues) + ) + + def test_detect_observando_nadan(self): + issues = self._check( + "La pareja está en un bote observando a los peces nadan." + ) + self.assertTrue( + any("'cómo'" in i["reason"] for i in issues) + ) + + def test_detect_le_vuelven_loca(self): + issues = self._check("Los perros pequeños le vuelven loca.") + self.assertTrue( + any("direct object" in i["reason"] for i in issues) + ) + + def test_detect_ai_generated_tts(self): + issues = self._check("AI-generated text-to-speech") + self.assertTrue( + any("Metadata contamination" in i["reason"] for i in issues) + ) + + def test_detect_necesito_missing_article(self): + issues = self._check("Necesito jardinero") + self.assertTrue( + any("Missing article" in i["reason"] for i in issues) + ) + + def test_detect_le_vuelven_loco(self): + issues = self._check("Estos problemas le vuelven loco.") + self.assertTrue( + any("direct object" in i["reason"] or "direct object pronoun" in i["reason"] for i in issues) + ) + + def test_veintiuno_anos(self): + issues = self._check("Tengo veintiuno años de edad.") + self.assertTrue( + any("veintiún" in i["suggested_fix"] for i in issues) + ) + + def test_lint_anki_cards_multiple_records(self): + records = [ + {"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"}, + {"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"}, + {"text": "Ella soña con viajar.", "line_number": 3, "raw_line": "Ella soña con viajar."}, + ] + results = lint_anki_cards(records) + self.assertEqual(len(results), 2) + reasons = results[0]["reason"] + results[1]["reason"] + self.assertIn("Missing negation", reasons) + self.assertIn("present indicative", reasons) + + +class SpanishDetailedCountsTests(unittest.TestCase): + def test_extract_detailed_counts_basic(self): + records = [ + {"text": "Yo como manzanas.", "line_number": 1}, + {"text": "Tú comes pan.", "line_number": 2}, + ] + + class MockDoc: + def __init__(self, tokens): + self.tokens = tokens + + def __iter__(self): + return iter(self.tokens) + + class MockNLP: + def __call__(self, text): + if "como" in text: + return MockDoc([ + MockToken(text="Yo", lemma="yo", pos="PRON"), + MockToken(text="como", lemma="comar", pos="VERB"), + MockToken(text="manzanas", lemma="manzana", pos="NOUN"), + MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True), + ]) + return MockDoc([ + MockToken(text="Tú", lemma="tú", pos="PRON"), + MockToken(text="comes", lemma="comer", pos="VERB"), + MockToken(text="pan", lemma="pan", pos="NOUN"), + MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True), + ]) + + stats = extract_detailed_counts( + records, + MockNLP(), + lambda t: t.pos_ in {"NOUN", "VERB", "ADJ", "ADV"}, + safe_spanish_lemma, + ) + + # "comar" should be corrected to "comer" + self.assertIn("comer", stats) + # "comer" should have count 2 (comes + como -> comer) + self.assertEqual(stats["comer"]["count"], 2) + self.assertIn("manzana", stats) + self.assertIn("pan", stats) + + def test_debug_tsv_output(self): + stats = { + "comer": { + "count": 3, + "pos_counts": {"VERB": 3}, + "surface_forms": ["como", "comes", "come"], + "example_sentences": ["Yo como.", "Tú comes."], + "source_lines": ["1", "2", "3"], + } + } + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "debug.tsv") + write_debug_tsv(stats, path) + with open(path, "r", encoding="utf-8") as f: + content = f.read() + self.assertIn("comer", content) + self.assertIn("VERB:3", content) + self.assertIn("como, comes, come", content) + self.assertIn("Yo como.", content) + self.assertIn("1, 2, 3", content) + + def test_suspicious_tokens_output(self): + stats = { + "comer": { + "count": 3, + "pos_counts": {"VERB": 3}, + "surface_forms": ["como", "comes"], + "original_lemmas": ["comar", "comer"], + "lemma_statuses": {"corrected:comar->comer": 1, "ok": 2}, + "example_sentences": ["Yo como."], + "source_lines": ["1"], + } + } + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "suspicious.tsv") + write_suspicious_tokens(stats, path) + with open(path, "r", encoding="utf-8") as f: + content = f.read() + self.assertIn("comer", content) + self.assertIn("comar", content) + self.assertIn("corrected:comar->comer", content) + + def test_debug_tsv_min_freq_filters(self): + stats = { + "comer": {"count": 5, "pos_counts": {"VERB": 5}, + "surface_forms": ["come"], "example_sentences": ["Yo come."], + "source_lines": ["1"]}, + "hablar": {"count": 2, "pos_counts": {"VERB": 2}, + "surface_forms": ["habla"], "example_sentences": ["Él habla."], + "source_lines": ["2"]}, + } + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "debug.tsv") + write_debug_tsv(stats, path, min_freq=3) + with open(path, "r", encoding="utf-8") as f: + content = f.read() + self.assertIn("comer", content) + self.assertNotIn("hablar", content) + + def test_extract_detailed_counts_no_clean(self): + class MockDoc: + def __init__(self, tokens): + self.tokens = tokens + def __iter__(self): + return iter(self.tokens) + + class MockNLP: + def __call__(self, text): + return MockDoc([ + MockToken(text="[sound:x.mp3]", lemma="[sound:x.mp3]", pos="X"), + ]) + + records = [{"text": "[sound:x.mp3]", "line_number": 1}] + stats = extract_detailed_counts( + records, MockNLP(), + lambda t: True, + lambda t: t.text.lower(), + clean=False, + ) + # With clean=False, [sound:x.mp3] is NOT stripped + self.assertIn("[sound:x.mp3]", stats) + + stats_clean = extract_detailed_counts( + records, MockNLP(), + lambda t: True, + lambda t: t.text.lower(), + clean=True, + ) + # With clean=True, the sound marker is stripped → empty text → no stats + self.assertEqual(len(stats_clean), 0) + + def test_lint_anki_cards_verbose_includes_all(self): + records = [ + {"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"}, + {"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"}, + ] + # Without verbose – only suspicious + results = lint_anki_cards(records, verbose=False) + self.assertEqual(len(results), 1) + + # With verbose – both records + results = lint_anki_cards(records, verbose=True) + self.assertEqual(len(results), 2) + + def test_lint_anki_cards_verbose_empty_field(self): + records = [ + {"text": "", "line_number": 1, "raw_line": ""}, + ] + results = lint_anki_cards(records, verbose=True) + self.assertEqual(len(results), 1) + self.assertEqual(results[0]["reason"], "empty field") + + +class CompareWordListsTests(unittest.TestCase): + def test_normalize_word_for_comparison(self): + self.assertEqual(normalize_word_for_comparison("camión"), "camion") + self.assertEqual(normalize_word_for_comparison("Corazón"), "corazon") + self.assertEqual(normalize_word_for_comparison("ESPAÑOL"), "espanol") + + def test_compare_word_lists_identifies_missing(self): + with tempfile.TemporaryDirectory() as tmp: + deck = os.path.join(tmp, "deck.txt") + target = os.path.join(tmp, "target.txt") + with open(deck, "w", encoding="utf-8") as f: + f.write("comer 5\nhablar 3\n") + with open(target, "w", encoding="utf-8") as f: + f.write("comer 10\nvivir 8\ncorrer 6\n") + result = compare_word_lists(deck, target) + self.assertIn("vivir", result["missing"]) + self.assertIn("correr", result["missing"]) + self.assertNotIn("comer", result["missing"]) + + def test_compare_word_lists_with_min_frequency(self): + with tempfile.TemporaryDirectory() as tmp: + deck = os.path.join(tmp, "deck.txt") + target = os.path.join(tmp, "target.txt") + with open(deck, "w", encoding="utf-8") as f: + f.write("comer 1\nhablar 5\n") + with open(target, "w", encoding="utf-8") as f: + f.write("comer 10\nhablar 10\n") + result = compare_word_lists(deck, target, min_frequency=3) + # "comer" has freq 1 < 3, so it's excluded from deck entirely + # meaning it should be "missing" from deck's perspective + self.assertIn("comer", result["missing"]) + self.assertNotIn("hablar", result["missing"]) + self.assertIn("hablar", result["seen"]) + + def test_compare_word_lists_writes_output(self): + with tempfile.TemporaryDirectory() as tmp: + deck = os.path.join(tmp, "deck.txt") + target = os.path.join(tmp, "target.txt") + out = os.path.join(tmp, "output.txt") + with open(deck, "w", encoding="utf-8") as f: + f.write("comer 5\n") + with open(target, "w", encoding="utf-8") as f: + f.write("comer 10\nvivir 8\n") + result = compare_word_lists(deck, target, output_path=out) + self.assertEqual(result["total_missing"], 1) + with open(out, "r", encoding="utf-8") as f: + self.assertIn("vivir", f.read()) + + def test_compare_word_lists_writes_seen_output(self): + with tempfile.TemporaryDirectory() as tmp: + deck = os.path.join(tmp, "deck.txt") + target = os.path.join(tmp, "target.txt") + seen_out = os.path.join(tmp, "seen.txt") + with open(deck, "w", encoding="utf-8") as f: + f.write("comer 5\n") + with open(target, "w", encoding="utf-8") as f: + f.write("comer 10\nvivir 8\n") + result = compare_word_lists(deck, target, seen_output_path=seen_out) + self.assertIn("comer", result["seen"]) + with open(seen_out, "r", encoding="utf-8") as f: + self.assertIn("comer", f.read()) + self.assertNotIn("vivir", f.read()) + + class YoutubeTests(unittest.TestCase): def test_extract_video_id(self): self.assertEqual(extract_video_id("https://youtu.be/abc123"), "abc123")