Remove Spanish-specific TSV tooling

This commit is contained in:
2026-06-15 16:45:26 -04:00
parent 67594835d2
commit 43a6b7d93d
8 changed files with 309 additions and 1246 deletions
+2
View File
@@ -5,3 +5,5 @@ venv/
*.egg-info/ *.egg-info/
dist/ dist/
build/ build/
.pytest_cache/
.mypy_cache/
+15 -38
View File
@@ -204,59 +204,36 @@ hablar 9
Extract vocabulary from an Anki TSV export file instead of using AnkiConnect: Extract vocabulary from an Anki TSV export file instead of using AnkiConnect:
```shell ```shell
saiki words --lang es --input Español.txt --field 2 --output words_es_content.txt --debug words_es_debug.tsv saiki words --lang es --input Español.txt --field 2 --field-section first --output words_es_content.txt --debug words_es_debug.tsv
``` saiki words --lang jp --input Japanese.txt --field 2 --output words_jp_content.txt
Additional file-based options:
```shell
saiki words --lang es --input Español.txt --field 2 --include-proper-nouns
saiki words --lang es --input Español.txt --field 2 --function-words
saiki words --lang es --input Español.txt --field 2 --lemma-corrections my_fixes.tsv
saiki words --lang es --input Español.txt --field 2 --bad-lemma-file bad_lemmas.txt
``` ```
When `--input` is provided, `--field` specifies the 1-based column index of the When `--input` is provided, `--field` specifies the 1-based column index of the
Spanish text (default 2). The audio column (index 1) and tags column are text field (default 2). Only that column is mined; other columns such as audio
automatically skipped. File-based NLP extraction is currently Spanish-specific; or Anki tags are ignored. By default, all blank-line-separated sections inside
the older AnkiConnect-based `saiki words jp` flow is unchanged. the selected field are kept. Use `--field-section first` when your card format
stores target-language text before a translation or note in the same field.
The file-based pipeline: The file-based pipeline:
- Parses Anki `#` header lines (`#separator:tab`, `#html:true`, `#tags column:N`) - Parses Anki `#` header lines (`#separator:tab`, `#html:true`, `#tags column:N`)
- Uses Python's `csv` module for robust TSV parsing - Uses Python's `csv` module for robust TSV parsing
- Removes `[sound:...mp3]` markers, HTML tags, and English glosses after - Removes `[sound:...mp3]` markers, HTML tags, URLs, and email addresses from
`<br><br>` from the field text the field text
- Applies safe Spanish lemmatisation (multi-word lemmas like `ayudar yo` or - Uses the configured language's spaCy model, token filter, and output format
`lavar él` are rejected; bad lemmas like `comar` -> `comer` are corrected)
- Tracks POS counts, surface forms, example sentences, and source line numbers - Tracks POS counts, surface forms, example sentences, and source line numbers
- Tracks original spaCy lemmas and correction/fallback reasons for debugging - Tracks original spaCy lemmas for debugging
Output files produced: Output files produced:
- `words_es_content.txt` — cleaned content vocabulary (NOUN, VERB, ADJ, ADV) - `words_<lang>_content.txt` — cleaned vocabulary
- `words_es_debug.tsv` — per-lemma debug info - `words_<lang>_debug.tsv` — per-entry debug info (only with `--debug`)
- `words_es_proper_nouns.txt` — proper nouns (only with `--include-proper-nouns`)
- `words_es_function_words.txt` — function words (only with `--function-words`)
- `words_es_suspicious_tokens.txt` — lemmas that required correction
Debug TSV example: Debug TSV example:
```text ```text
lemma count pos_counts top_surface_forms example_sentences source_lines original_lemmas lemma_statuses status entry count pos_counts top_surface_forms example_sentences source_lines original_lemmas
comer 8 VERB:8 como, come, comen Yo como manzanas.; Ustedes los comen con arroz. 70,979 comar, comer corrected:comar->comer:1, ok:7 comer 8 VERB:8 como, come, comen Yo como manzanas.; Ustedes los comen con arroz. 70,979 comer
``` ```
### Lint Spanish Cards
Check an Anki TSV export for suspicious or awkward Spanish:
```shell
saiki lint-anki-es --input Español.txt --field 2 --output suspicious_cards_es.tsv
```
Detects known errors and rule-based suspicious patterns and produces a TSV
report with: source line, original text, cleaned text, reason, and suggested
fix.
### Compare ### Compare
Compare deck vocabulary against a target list: Compare deck vocabulary against a target list:
@@ -270,7 +247,7 @@ Normalises case and optionally strips accents for matching, so `cómo` and
`como` are treated as the same word. Use `--min-frequency` to ignore accidental `como` are treated as the same word. Use `--min-frequency` to ignore accidental
low-frequency words in the deck. low-frequency words in the deck.
### YouTube
Mine vocabulary or sentence rows from YouTube subtitles. Mine vocabulary or sentence rows from YouTube subtitles.
+4 -2
View File
@@ -67,7 +67,7 @@ def parse_anki_tsv(
``text``, ``tags`` (if *include_tags* is True), ``line_number``, ``text``, ``tags`` (if *include_tags* is True), ``line_number``,
and ``raw_line``. and ``raw_line``.
Header lines (``#``-prefixed) are skipped. The actual separator Leading header lines (``#``-prefixed) are skipped. The actual separator
is detected from ``#separator:`` and defaults to tab. is detected from ``#separator:`` and defaults to tab.
""" """
if field_index < 1: if field_index < 1:
@@ -95,13 +95,15 @@ def parse_anki_tsv(
with open(path, "r", encoding="utf-8") as fh: with open(path, "r", encoding="utf-8") as fh:
reader = csv.reader(fh, delimiter=separator) reader = csv.reader(fh, delimiter=separator)
line_number = 0 line_number = 0
in_headers = True
for row in reader: for row in reader:
line_number += 1 line_number += 1
# skip header rows and completely empty rows # skip header rows and completely empty rows
if not row: if not row:
continue continue
if row and row[0].startswith("#"): if in_headers and row[0].startswith("#"):
continue continue
in_headers = False
if col_idx >= len(row): if col_idx >= len(row):
raise ValueError( raise ValueError(
+9 -47
View File
@@ -15,7 +15,7 @@ from .importer import (
supported_tts_backends, supported_tts_backends,
synthesize_tts_sample, synthesize_tts_sample,
) )
from .words import compare_word_files, compare_word_lists, extract_words, extract_words_from_file, lint_anki_cards_main from .words import compare_word_files, compare_word_lists, extract_words, extract_words_from_file
from .youtube import run_youtube from .youtube import run_youtube
@@ -98,12 +98,13 @@ def build_parser(config: Config | None = None) -> argparse.ArgumentParser:
words.add_argument("--out", "--output", dest="out") words.add_argument("--out", "--output", dest="out")
words.add_argument("--full-field", action="store_true") words.add_argument("--full-field", action="store_true")
words.add_argument("--spacy-model") words.add_argument("--spacy-model")
words.add_argument("--include-proper-nouns", action="store_true", help="Include proper nouns (PROPN).")
words.add_argument("--function-words", action="store_true", help="Extract function words separately.")
words.add_argument("--debug", help="Path for debug TSV output.") words.add_argument("--debug", help="Path for debug TSV output.")
words.add_argument("--lemma-corrections", help="Path to bad<TAB>good TSV of lemma corrections.") words.add_argument(
words.add_argument("--bad-lemma-file", help="Path to file listing lemmas to skip.") "--field-section",
words.add_argument("--include-tags", action="store_true", help="Include tags column in parsing.") choices=["all", "first"],
default="all",
help="Which blank-line-separated TSV field section to mine.",
)
words.add_argument("--debug-min-freq", type=int, default=0, words.add_argument("--debug-min-freq", type=int, default=0,
help="Minimum frequency for debug output (default: all).") help="Minimum frequency for debug output (default: all).")
words.add_argument("--no-clean", action="store_true", words.add_argument("--no-clean", action="store_true",
@@ -121,12 +122,6 @@ def build_parser(config: Config | None = None) -> argparse.ArgumentParser:
compare_new.add_argument("--min-frequency", type=int, default=0, compare_new.add_argument("--min-frequency", type=int, default=0,
help="Minimum frequency for deck words to count.") help="Minimum frequency for deck words to count.")
lint_es = sub.add_parser("lint-anki-es", help="Check Spanish Anki cards for suspicious patterns.")
lint_es.add_argument("--input", required=True, help="Anki TSV export file.")
lint_es.add_argument("--field", type=int, default=2, help="1-based column index (default 2).")
lint_es.add_argument("--output", default="suspicious_cards_es.tsv", help="Output TSV path.")
lint_es.add_argument("--verbose", action="store_true", help="Output all cards, not just suspicious ones.")
youtube = sub.add_parser("youtube", help="Mine a YouTube transcript.") youtube = sub.add_parser("youtube", help="Mine a YouTube transcript.")
youtube.add_argument("lang", choices=choices) youtube.add_argument("lang", choices=choices)
youtube.add_argument("video") youtube.add_argument("video")
@@ -195,12 +190,6 @@ def main(argv: list[str] | None = None) -> int:
if args.command == "words": if args.command == "words":
if args.input: if args.input:
# File-based extraction # File-based extraction
if config.language_name(args.lang) != "spanish":
print(
"Error: TSV file-based word extraction is currently Spanish-only.",
file=sys.stderr,
)
return 1
if args.field and not args.field.isdigit(): if args.field and not args.field.isdigit():
print( print(
f"Error: --field must be a 1-based column index (integer) " f"Error: --field must be a 1-based column index (integer) "
@@ -221,25 +210,16 @@ def main(argv: list[str] | None = None) -> int:
min_freq=args.min_freq, min_freq=args.min_freq,
outdir=args.outdir, outdir=args.outdir,
out=args.out, out=args.out,
include_proper_nouns=args.include_proper_nouns,
function_words=args.function_words,
debug=args.debug, debug=args.debug,
lemma_corrections=args.lemma_corrections,
bad_lemma_file=args.bad_lemma_file,
spacy_model=args.spacy_model, spacy_model=args.spacy_model,
include_tags=args.include_tags,
debug_min_freq=args.debug_min_freq, debug_min_freq=args.debug_min_freq,
no_clean=args.no_clean, no_clean=args.no_clean,
field_section=args.field_section,
) )
print(f"Parsed {result['records']} records") print(f"Parsed {result['records']} records")
print(f"Wrote {result['written']} content entries to: {result['out']}") print(f"Wrote {result['written']} entries to: {result['out']}")
if result.get("debug"): if result.get("debug"):
print(f"Debug output: {result['debug']}") print(f"Debug output: {result['debug']}")
if result.get("proper_nouns"):
print(f"Proper nouns: {result['proper_nouns']}")
if result.get("function_words"):
print(f"Function words: {result['function_words']}")
print(f"Suspicious tokens: {result.get('suspicious', '')}")
else: else:
result = extract_words( result = extract_words(
config, args.lang, args.query, args.deck, args.field, args.min_freq, config, args.lang, args.query, args.deck, args.field, args.min_freq,
@@ -279,24 +259,6 @@ def main(argv: list[str] | None = None) -> int:
print(f"Seen words written to: {args.seen_output}") print(f"Seen words written to: {args.seen_output}")
return 0 return 0
if args.command == "lint-anki-es":
if args.field < 1:
print(
f"Error: --field must be a 1-based column index >= 1, got: {args.field}",
file=sys.stderr,
)
return 1
result = lint_anki_cards_main(
args.input,
field_index=args.field,
output_path=args.output,
verbose=args.verbose,
)
print(f"Checked {result['records']} cards")
print(f"Found {result['issues']} suspicious card(s)")
print(f"Report written to: {result['path']}")
return 0
if args.command == "youtube": if args.command == "youtube":
result = run_youtube( result = run_youtube(
config, args.lang, args.video, args.mode, args.top, args.no_stopwords, config, args.lang, args.video, args.mode, args.top, args.no_stopwords,
+1 -1
View File
@@ -13,7 +13,7 @@ from typing import Any
try: try:
import yaml import yaml
except Exception: # pragma: no cover - handled when config files are loaded except ImportError: # pragma: no cover - handled when config files are loaded
yaml = None yaml = None
-719
View File
@@ -1,719 +0,0 @@
"""Spanish-specific NLP utilities for vocabulary extraction and card linting."""
from __future__ import annotations
import html
import re
from collections import defaultdict
from dataclasses import dataclass
from typing import Any, Callable
# ── Regex patterns ─────────────────────────────────────────────────
SOUND_RE = re.compile(r"\[sound:[^\]]*\]")
MEDIA_RE = re.compile(r"\[\s*(?:sound|image|media):[^\]]*\]", re.IGNORECASE)
MEDIA_FILENAME_RE = re.compile(
r"[^/\s]+\.(?:mp3|ogg|wav|m4a|flac|jpg|jpeg|png|gif|webp|mp4|webm)\b",
re.IGNORECASE,
)
BR_RE = re.compile(r"<\s*br\s*/?\s*>", re.IGNORECASE)
DOUBLE_BR_RE = re.compile(
r"<\s*br\s*/?\s*>\s*<\s*br\s*/?\s*>", re.IGNORECASE
)
HTML_TAG_RE = re.compile(r"<[^>]+>")
URL_RE = re.compile(r"https?://\S+|www\.\S+", re.IGNORECASE)
EMAIL_RE = re.compile(r"\S+@\S+\.\S+")
MULTI_WS_RE = re.compile(r"[ \t]+")
SPANISH_WORD_RE = re.compile(r"^[a-záéíóúüñ]+$", re.IGNORECASE)
BAD_TOKEN_TEXT_RE = re.compile(r"[\d_/@#\\|+*=<>~`^]")
LEMMA_PUNCT_RE = re.compile(r"[.,;:!?¿¡\"'()\[\]{}<>\-_/\\|`~@#$%^&*+=\d]")
_ENGLISH_PATTERNS: list[str] = [
r"\bI\s+(?:am|have|had|will|would|could|should|was|were|did|do|don"
r"|can|need|like|want|think|know|love|hate|see|hear|eat|drink"
r"|go|come|take|give|make|say|tell|ask|get|put)\b",
r"\byou\s+(?:are|have|had|will|would|could|should|were|did|do|don"
r"|can|need|like|want|think|know|love|hate|see|hear)\b",
r"\b(the|to|and|in|of|a|an|is|it|for|with|on|at|by|from|or|be|this|that)\b",
r"\bsee you later\b",
r"\bI love you\b",
r"\bI want you\b",
r"\bknowledge/understanding\b",
r"\b(?:see|watch|read|write|speak|listen|learn|study|teach|explain)\s+"
r"(?:you|him|her|it|us|them|me)\b",
]
ENGLISH_RE = re.compile("|".join(_ENGLISH_PATTERNS), re.IGNORECASE)
# ── Lemma corrections ──────────────────────────────────────────────
BUILTIN_LEMMA_CORRECTIONS: dict[str, str] = {
"comar": "comer",
"acabir": "acabar",
"deliciós": "delicioso",
"llover/llover": "llover",
}
PRIR_SURFACE_FORMS = {"pide", "pido", "piden", "pidiendo", "pedir"}
@dataclass(frozen=True)
class LemmaInfo:
"""A selected lemma plus audit metadata about how it was chosen."""
lemma: str
original_lemma: str
surface: str
status: str = "ok"
# ── Known card issues (exact substring → fix → explanation) ────────
KNOWN_CARD_ISSUES: list[tuple[str, str, str]] = [
(
"Tú sabes nada",
"Tú no sabes nada.",
"Missing negation 'no' before 'sabes'",
),
(
"Dulce sueños",
"Dulces sueños.",
"'Dulce' should agree in number with 'sueños'",
),
(
"Ella soña con viajar.",
"Ella sueña con viajar.",
"'Soña''sueña' (present indicative of soñar)",
),
(
"Siento lastima",
"Siento lástima",
"'lastima' should be 'lástima' with accent",
),
(
"Tengo uno libro",
"Tengo un libro",
"'uno''un' before singular masculine noun",
),
(
"Tengo veintiuno años",
"Tengo veintiún años",
"'veintiuno''veintiún' before plural noun",
),
(
"Dile a tu madre que feliz cumpleaños",
'Dile a tu madre: "Feliz cumpleaños."',
"Missing colon/quotes or restructured phrasing",
),
(
"La pareja está en un bote observando a los peces nadan.",
"La pareja está en un bote observando cómo nadan los peces.",
"Missing 'cómo' or subordinating conjunction",
),
(
"Los perros pequeños le vuelven loca.",
"Los perros pequeños la vuelven loca.",
"'le''la' (direct object, not indirect)",
),
]
# ── Regex-based suspicious patterns ────────────────────────────────
# (pattern, description, suggested_fix_or_None)
SUSPICIOUS_PATTERNS: list[tuple[str, str, str | None]] = [
(
r"AI-generated text-to-speech",
"Metadata contamination: AI-generated TTS label in text field",
None,
),
(
r"\bsabes\s+nada\b(?!.*\bno\b)",
"Missing negation: 'sabes nada' without 'no'",
"Add 'no' before 'sabes': 'No sabes nada.'",
),
(
r"\buno\s+(?:libro|coche|casa|perro|gato|hombre|mujer|niño"
r"|amigo|día|año|mes|semana|minuto|segundo)\b",
"'uno' instead of 'un' before masculine singular noun",
"Replace 'uno' with 'un'",
),
(
r"\bveintiuno\s+(?:años|días|meses|semanas|horas|minutos|segundos)\b",
"'veintiuno' instead of 'veintiún' before plural noun",
"Replace 'veintiuno' with 'veintiún'",
),
(
r"observando\s+a\s+.*\bnadan\b",
"Possible missing 'cómo' in 'observando cómo ...' construction",
"Consider adding 'cómo': 'observando cómo ...'",
),
(
r"\ble\s+vuelve\w*\s+loc[ao]s?\b",
"'le vuelve/vuelven loco/a' may need a direct object pronoun",
"Replace 'le' with lo/la/los/las as appropriate",
),
(
r"\ble\s+vuelven\s+loc[ao]s?\b",
"'le vuelven loco/a' may need a direct object pronoun",
"Replace 'le' with lo/la/los/las as appropriate",
),
(
r"\bNecesito\s+[a-záéíóúüñ]{2,}\b(?!\s+(?:un|una|el|la|los"
r"|las|al|del|mi|tu|su|nuestro))\s*$",
"Missing article before the noun after 'Necesito'",
None,
),
(
r"\bQu[ée]\s+es\s+la\s+respuesta\b",
"'Qué es la respuesta''Cuál es la respuesta'",
"Replace 'Qué' with 'Cuál'",
),
(
r"\bpeludos?\s+animales?\b",
"'peludo(s) animal(es)''animal(es) peludo(s)' (adjective placement)",
"Place adjective after noun",
),
(
r"\bhembras\s+y\s+varones\b",
"'hembras y varones' — consider 'niños y niñas' or 'chicos y chicas'",
None,
),
(
r"\bt[ií]o\s+alto\s+con\s+uniforme\b",
"'tío alto con uniforme' — check intent (Spain slang vs 'hombre')",
None,
),
(
r"\brepresentante\s+adulto\b",
"'representante adulto''adulto responsable'",
"Consider 'adulto responsable'",
),
(
r"\bpropiedad\s+a\s+este\s+representante\b",
"'propiedad a este representante' — unusual phrasing",
"Consider revising wording",
),
]
CONTENT_POS = frozenset({"NOUN", "VERB", "ADJ", "ADV"})
FUNCTION_POS = frozenset(
{"PRON", "ADP", "CCONJ", "SCONJ", "DET", "AUX", "PART", "INTJ"}
)
# ── Text cleaning ──────────────────────────────────────────────────
def _clean_single_field_part(text: str) -> str:
"""Clean one segment of an Anki field (no English-gloss filtering)."""
text = BR_RE.sub(" ", text)
text = HTML_TAG_RE.sub("", text)
text = html.unescape(text)
text = URL_RE.sub("", text)
text = EMAIL_RE.sub("", text)
return text.strip()
def _looks_like_english_gloss(text: str) -> bool:
"""Heuristic: does *text* read like an English gloss/translation?"""
stripped = text.strip().strip("()[]()【】")
if not stripped:
return True
return bool(ENGLISH_RE.search(stripped))
def clean_anki_field_text(text: str) -> str:
"""Clean Anki field text for Spanish NLP.
* Removes ``[sound:...]`` and ``[media:...]`` markers.
* Splits on ``<br><br>`` to separate Spanish from English glosses.
* Strips HTML tags and unescapes entities.
* Removes URLs, emails.
* Drops segments that look like English translations.
* Normalises whitespace.
* Preserves Spanish accents and meaningful content.
"""
if not text:
return ""
text = SOUND_RE.sub("", text)
text = MEDIA_RE.sub("", text)
# Split on double <br> to isolate Spanish from glosses
parts = DOUBLE_BR_RE.split(text)
cleaned_parts: list[str] = []
for i, part in enumerate(parts):
cleaned = _clean_single_field_part(part)
if not cleaned:
continue
if i > 0 and _looks_like_english_gloss(cleaned):
continue
cleaned_parts.append(cleaned)
result = " ".join(cleaned_parts)
result = MULTI_WS_RE.sub(" ", result).strip()
return result
# ── Safe lemmatisation ─────────────────────────────────────────────
def load_lemma_corrections(path: str) -> dict[str, str]:
"""Load a ``bad<TAB>good`` TSV of manual lemma fixes."""
corrections: dict[str, str] = {}
with open(path, "r", encoding="utf-8") as fh:
for line in fh:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) >= 2:
bad = parts[0].strip().lower()
good = parts[1].strip().lower()
if bad and good:
corrections[bad] = good
return corrections
def load_bad_lemmas(path: str) -> set[str]:
"""Load a list of lemmas (one per line) to skip and report."""
bad: set[str] = set()
with open(path, "r", encoding="utf-8") as fh:
for line in fh:
lemma = line.strip()
if lemma and not lemma.startswith("#"):
bad.add(lemma.lower())
return bad
def safe_spanish_lemma(
token,
extra_corrections: dict[str, str] | None = None,
) -> str:
"""Return only the selected lemma for callers that do not need metadata."""
return safe_spanish_lemma_info(token, extra_corrections=extra_corrections).lemma
def safe_spanish_lemma_info(
token,
extra_corrections: dict[str, str] | None = None,
) -> LemmaInfo:
"""Return a safe Spanish lemma for a spaCy token.
Falls back to ``token.text.lower()`` when the spaCy lemma is empty,
multi-word, contains punctuation, or looks non-Spanish. Applies a
built-in and optionally an external correction map, and records why
fallback/correction happened.
"""
manual = dict(BUILTIN_LEMMA_CORRECTIONS)
if extra_corrections:
manual.update({k.lower(): v.lower() for k, v in extra_corrections.items()})
lemma = (token.lemma_ or "").strip().lower()
surface = (token.text or "").strip().lower()
# Manual correction map wins
if lemma in manual:
return LemmaInfo(
manual[lemma],
original_lemma=lemma,
surface=surface,
status=f"corrected:{lemma}->{manual[lemma]}",
)
# Empty or one-char fallback
if not lemma or len(lemma) <= 1:
return LemmaInfo(
surface or lemma,
original_lemma=lemma,
surface=surface,
status="fallback_empty_or_short_lemma",
)
# Multi-word lemma → suspicious spaCy output
if " " in lemma:
return LemmaInfo(
surface,
original_lemma=lemma,
surface=surface,
status="fallback_multi_word_lemma",
)
# Punctuation or digits in lemma → fall back
if LEMMA_PUNCT_RE.search(lemma):
return LemmaInfo(
surface,
original_lemma=lemma,
surface=surface,
status="fallback_bad_lemma_characters",
)
# Special case: "prir" might be spaCy failing on pedir forms
if lemma == "prir":
if surface in PRIR_SURFACE_FORMS:
return LemmaInfo(
"pedir",
original_lemma=lemma,
surface=surface,
status="corrected:prir->pedir",
)
# otherwise return surface so it shows up in debug as suspicious
return LemmaInfo(
surface,
original_lemma=lemma,
surface=surface,
status="suspicious_prir",
)
if not SPANISH_WORD_RE.match(lemma):
return LemmaInfo(
surface,
original_lemma=lemma,
surface=surface,
status="fallback_non_spanish_lemma",
)
return LemmaInfo(lemma, original_lemma=lemma, surface=surface)
# ── Token filters ──────────────────────────────────────────────────
def spanish_content_filter(
token,
include_proper_nouns: bool = False,
) -> bool:
"""Return True for Spanish content words (NOUN, VERB, ADJ, ADV).
Excludes punctuation, digits, symbols, URLs, emails, media
filenames, and one-character junk.
"""
text = (token.text or "").strip()
if not text:
return False
if token.is_punct or token.is_space or token.is_digit:
return False
if token.is_currency:
return False
if token.like_url or token.like_email:
return False
if MEDIA_FILENAME_RE.search(text):
return False
if BAD_TOKEN_TEXT_RE.search(text):
return False
if len(text) <= 1 and text not in ("a", "y", "e", "o", "u", "él"):
return False
pos = getattr(token, "pos_", "") or ""
if pos in CONTENT_POS:
return True
if include_proper_nouns and pos == "PROPN":
return True
return False
def spanish_function_word_filter(token) -> bool:
"""Return True for Spanish function words (PRON, ADP, CCONJ, …)."""
text = (token.text or "").strip()
if not text:
return False
if token.is_punct or token.is_space or token.like_url or token.like_email:
return False
if MEDIA_FILENAME_RE.search(text) or BAD_TOKEN_TEXT_RE.search(text):
return False
pos = getattr(token, "pos_", "") or ""
return pos in FUNCTION_POS
# ── Suspicious-card detection ──────────────────────────────────────
def check_card_for_issues(text: str, cleaned: str) -> list[dict[str, Any]]:
"""Check a single card's text for suspicious patterns.
Returns a list of issue dicts (normally 0 or 1 per card, but one
card can match multiple patterns).
"""
issues: list[dict[str, Any]] = []
seen_reasons: set[str] = set()
for bad_text, fix, explanation in KNOWN_CARD_ISSUES:
if bad_text in text:
if explanation not in seen_reasons:
seen_reasons.add(explanation)
issues.append({
"reason": explanation,
"suggested_fix": fix,
})
for pattern, description, fix in SUSPICIOUS_PATTERNS:
if re.search(pattern, text, re.IGNORECASE):
if description not in seen_reasons:
seen_reasons.add(description)
issues.append({
"reason": description,
"suggested_fix": fix or "",
})
# Check for English after <br><br>
parts = DOUBLE_BR_RE.split(text)
for part in parts[1:]:
cleaned_part = _clean_single_field_part(part)
if _looks_like_english_gloss(cleaned_part):
desc = "English gloss after <br><br>"
if desc not in seen_reasons:
seen_reasons.add(desc)
issues.append({"reason": desc, "suggested_fix": ""})
# Check for missing final punctuation in full-sentence cards
stripped_text = text.rstrip()
if (
stripped_text
and not stripped_text.endswith((".", "!", "?", "", '"', "'"))
and stripped_text[0].isupper()
and len(stripped_text.split()) >= 3
):
desc = "Missing final punctuation for full-sentence card"
if desc not in seen_reasons:
seen_reasons.add(desc)
issues.append({"reason": desc, "suggested_fix": ""})
return issues
def lint_anki_cards(
records: list[dict[str, Any]],
verbose: bool = False,
) -> list[dict[str, Any]]:
"""Run all suspicious-card checks across parsed Anki records.
When *verbose* is True every record is returned, not just those
with issues.
Returns a list of issue dicts with keys:
``source_line``, ``original_text``, ``cleaned_text``,
``reason``, ``suggested_fix``.
"""
results: list[dict[str, Any]] = []
for record in records:
text = record.get("text", "")
if not text:
if verbose:
results.append({
"source_line": record.get("line_number", ""),
"original_text": record.get("raw_line", ""),
"cleaned_text": "",
"reason": "empty field",
"suggested_fix": "",
})
continue
cleaned = clean_anki_field_text(text)
issues = check_card_for_issues(text, cleaned)
if verbose or issues:
if issues:
combined_reason = "; ".join(
sorted({i["reason"] for i in issues})
)
suggested = ""
for i in issues:
if i.get("suggested_fix"):
suggested = i["suggested_fix"]
break
else:
combined_reason = ""
suggested = ""
results.append({
"source_line": record.get("line_number", ""),
"original_text": record.get("raw_line", text),
"cleaned_text": cleaned,
"reason": combined_reason,
"suggested_fix": suggested,
})
return results
# ── Detailed extraction helpers ─────────────────────────────────────
def extract_detailed_counts(
texts: list[dict[str, Any]],
nlp,
token_filter: Callable,
lemma_fn: Callable,
max_examples: int = 3,
clean: bool = True,
) -> dict[str, dict[str, Any]]:
"""Build detailed per-lemma statistics from a list of ``{text, line_number}`` records.
When *clean* is False the raw text is passed to the NLP pipeline
without going through ``clean_anki_field_text``.
Returns ::
{lemma: {"count": int,
"pos_counts": {POS: int, },
"surface_forms": [str, ],
"example_sentences": [str, ],
"source_lines": [str, ]}}
"""
stats: dict[str, dict[str, Any]] = {}
for record in texts:
raw = record.get("text", "")
line_number = record.get("line_number", 0)
if not raw:
continue
txt = clean_anki_field_text(raw) if clean else raw.strip()
if not txt:
continue
doc = nlp(txt)
for token in doc:
if not token_filter(token):
continue
lemma_value = lemma_fn(token)
if lemma_value is None:
continue
if isinstance(lemma_value, LemmaInfo):
lemma = lemma_value.lemma
original_lemma = lemma_value.original_lemma
lemma_status = lemma_value.status
else:
lemma = str(lemma_value)
original_lemma = (getattr(token, "lemma_", "") or "").strip().lower()
lemma_status = "ok"
if not lemma:
continue
if lemma not in stats:
stats[lemma] = {
"count": 0,
"pos_counts": defaultdict(int),
"surface_forms": set(),
"original_lemmas": set(),
"lemma_statuses": defaultdict(int),
"example_sentences": [],
"source_lines": [],
}
s = stats[lemma]
s["count"] += 1
s["pos_counts"][token.pos_] += 1
s["surface_forms"].add(token.text.lower())
if original_lemma:
s["original_lemmas"].add(original_lemma)
s["lemma_statuses"][lemma_status] += 1
if len(s["example_sentences"]) < max_examples:
sent_text = txt
if sent_text not in s["example_sentences"]:
s["example_sentences"].append(sent_text)
if line_number:
s["source_lines"].append(str(line_number))
# Convert defaultdicts/sets to plain types for serialisation
for lemma, s in stats.items():
s["pos_counts"] = dict(s["pos_counts"])
s["surface_forms"] = sorted(s["surface_forms"])
s["original_lemmas"] = sorted(s.get("original_lemmas", []))
s["lemma_statuses"] = dict(s.get("lemma_statuses", {}))
s["source_lines"] = list(dict.fromkeys(s["source_lines"])) # dedup, preserve order
return stats
# ── Output helpers ─────────────────────────────────────────────────
def write_debug_tsv(
stats: dict[str, dict[str, Any]],
path: str,
bad_lemmas: set[str] | None = None,
min_freq: int = 0,
) -> None:
"""Write a debug TSV with columns::
lemma count pos_counts top_surface_forms example_sentences source_lines
Appends a special line per lemma in *bad_lemmas* with an
``(EXCLUDED)`` marker. When *min_freq* > 0 only lemmas with
count >= min_freq are included.
"""
import os
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
bad = bad_lemmas or set()
with open(path, "w", encoding="utf-8") as fh:
fh.write(
"lemma\tcount\tpos_counts\ttop_surface_forms"
"\texample_sentences\tsource_lines\toriginal_lemmas"
"\tlemma_statuses\tstatus\n"
)
sorted_lemmas = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
for lemma, s in sorted_lemmas:
if min_freq > 0 and s["count"] < min_freq:
continue
pos_counts_str = ", ".join(
f"{pos}:{cnt}"
for pos, cnt in sorted(
s["pos_counts"].items(), key=lambda x: -x[1]
)
)
surfaces_str = ", ".join(s["surface_forms"])
examples_str = "; ".join(s["example_sentences"])
lines_str = ", ".join(s["source_lines"])
original_lemmas_str = ", ".join(s.get("original_lemmas", []))
lemma_statuses_str = ", ".join(
f"{status}:{cnt}"
for status, cnt in sorted(
s.get("lemma_statuses", {}).items(),
key=lambda x: (-x[1], x[0]),
)
)
status = "EXCLUDED" if lemma in bad else ""
fh.write(
f"{lemma}\t{s['count']}\t{pos_counts_str}\t{surfaces_str}"
f"\t{examples_str}\t{lines_str}\t{original_lemmas_str}"
f"\t{lemma_statuses_str}\t{status}\n"
)
def write_suspicious_tokens(
stats: dict[str, dict[str, Any]],
path: str,
bad_lemmas: set[str] | None = None,
) -> None:
"""Write a file of suspicious lemmas (multi-word, punctuation, etc.)
Any lemma that was corrected, fell back from a suspicious spaCy lemma,
or appears in the user blocklist is listed here.
"""
import os
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
bad = bad_lemmas or set()
with open(path, "w", encoding="utf-8") as fh:
fh.write("lemma\tcount\tsurface_forms\toriginal_lemmas\tnote\n")
fh.write(
"# Lemmas that required correction or fallback are listed here\n"
)
for lemma, s in sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0])):
statuses = {
status: count
for status, count in s.get("lemma_statuses", {}).items()
if status != "ok"
}
if lemma in bad:
statuses["blocked_bad_lemma"] = s["count"]
if not statuses:
continue
surfaces_str = ", ".join(s["surface_forms"])
originals_str = ", ".join(s.get("original_lemmas", []))
note = ", ".join(
f"{status}:{count}"
for status, count in sorted(statuses.items())
)
fh.write(
f"{lemma}\t{s['count']}\t{surfaces_str}"
f"\t{originals_str}\t{note}\n"
)
+174 -143
View File
@@ -2,9 +2,9 @@
from __future__ import annotations from __future__ import annotations
import logging import html
import os import os
from collections import Counter from collections import Counter, defaultdict
import regex as re import regex as re
from typing import Any, Callable from typing import Any, Callable
@@ -12,18 +12,6 @@ from typing import Any, Callable
from .ankiconnect import anki_request from .ankiconnect import anki_request
from .anki_tsv import parse_anki_tsv from .anki_tsv import parse_anki_tsv
from .config import Config from .config import Config
from .spanish import (
clean_anki_field_text,
extract_detailed_counts,
lint_anki_cards,
load_bad_lemmas,
load_lemma_corrections,
safe_spanish_lemma_info,
spanish_content_filter,
spanish_function_word_filter,
write_debug_tsv,
write_suspicious_tokens,
)
from .text import extract_first_visible_line, extract_visible_text, normalize_word_key from .text import extract_first_visible_line, extract_visible_text, normalize_word_key
ACCENT_MAP = str.maketrans("áéíóúüñÁÉÍÓÚÜÑ", "aeiouunAEIOUUN") ACCENT_MAP = str.maketrans("áéíóúüñÁÉÍÓÚÜÑ", "aeiouunAEIOUUN")
@@ -38,19 +26,67 @@ JAPANESE_GRAMMAR_EXCLUDE = {
"", "", "ます", "れる", "てる", "", "", "しまう", "いる", "ない", "なる", "ある", "", "です", "", "", "ます", "れる", "てる", "", "", "しまう", "いる", "ない", "なる", "ある", "", "です",
} }
JAPANESE_ALLOWED_POS = {"NOUN", "PROPN", "VERB", "ADJ"} JAPANESE_ALLOWED_POS = {"NOUN", "PROPN", "VERB", "ADJ"}
SPANISH_ALLOWED_POS = {"NOUN", "VERB", "ADJ", "ADV"}
FIELD_SECTION_CHOICES = {"all", "first"}
SOUND_RE = re.compile(r"\[sound:[^\]]*\]")
def setup_logging(logfile: str) -> None: MEDIA_RE = re.compile(r"\[\s*(?:sound|image|media):[^\]]*\]", re.IGNORECASE)
"""Configure file logging for word extraction scripts.""" BR_RE = re.compile(r"<\s*br\s*/?\s*>", re.IGNORECASE)
os.makedirs(os.path.dirname(os.path.abspath(logfile)), exist_ok=True) DOUBLE_BR_RE = re.compile(
logging.basicConfig(filename=logfile, level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") r"<\s*br\s*/?\s*>\s*<\s*br\s*/?\s*>", re.IGNORECASE
)
HTML_TAG_RE = re.compile(r"<[^>]+>")
URL_RE = re.compile(r"https?://\S+|www\.\S+", re.IGNORECASE)
EMAIL_RE = re.compile(r"\S+@\S+\.\S+")
MULTI_WS_RE = re.compile(r"[ \t]+")
def build_query_from_decks(decks: list[str]) -> str: def build_query_from_decks(decks: list[str]) -> str:
"""Build an Anki search query that matches any configured deck.""" """Build an Anki search query that matches any configured deck."""
return " OR ".join(f'deck:"{d}"' for d in decks) return " OR ".join(f'deck:"{d}"' for d in decks)
def _clean_single_field_part(text: str) -> str:
"""Clean one segment of an Anki field before NLP."""
text = BR_RE.sub(" ", text)
text = HTML_TAG_RE.sub("", text)
text = html.unescape(text)
text = URL_RE.sub("", text)
text = EMAIL_RE.sub("", text)
return text.strip()
def clean_anki_field_text(text: str | None, field_section: str = "all") -> str:
"""Clean Anki field text before NLP.
Removes media markers, strips HTML/URLs/emails, normalizes spaces, and can
optionally keep only the first blank-line-separated field section.
"""
if not text:
return ""
if field_section not in FIELD_SECTION_CHOICES:
raise ValueError(
f"field_section must be one of {sorted(FIELD_SECTION_CHOICES)}, "
f"got {field_section!r}"
)
text = SOUND_RE.sub("", text)
text = MEDIA_RE.sub("", text)
parts = DOUBLE_BR_RE.split(text)
if field_section == "first":
parts = parts[:1]
cleaned_parts: list[str] = []
for part in parts:
cleaned = _clean_single_field_part(part)
if not cleaned:
continue
cleaned_parts.append(cleaned)
result = " ".join(cleaned_parts)
return MULTI_WS_RE.sub(" ", result).strip()
def japanese_filter(token) -> bool: def japanese_filter(token) -> bool:
"""Return whether a spaCy token is useful Japanese vocabulary. """Return whether a spaCy token is useful Japanese vocabulary.
@@ -74,7 +110,13 @@ def japanese_filter(token) -> bool:
def spanish_filter(token) -> bool: def spanish_filter(token) -> bool:
"""Return whether a spaCy token is useful Spanish vocabulary.""" """Return whether a spaCy token is useful Spanish vocabulary."""
return bool(getattr(token, "is_alpha", False)) and not bool(getattr(token, "is_stop", False)) if not bool(getattr(token, "is_alpha", False)):
return False
if bool(getattr(token, "is_stop", False)):
return False
if bool(getattr(token, "like_url", False)) or bool(getattr(token, "like_email", False)):
return False
return getattr(token, "pos_", None) in SPANISH_ALLOWED_POS
def spanish_format(token) -> str: def spanish_format(token) -> str:
@@ -219,18 +261,109 @@ def extract_words(
return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path} return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path}
def extract_detailed_counts(
texts: list[dict[str, Any]],
nlp,
token_filter: Callable,
output_format: Callable,
max_examples: int = 3,
clean: bool = True,
field_section: str = "all",
) -> dict[str, dict[str, Any]]:
"""Build detailed per-entry statistics from parsed TSV records."""
stats: dict[str, dict[str, Any]] = {}
for record in texts:
raw = record.get("text", "")
line_number = record.get("line_number", 0)
if not raw:
continue
text = clean_anki_field_text(raw, field_section=field_section) if clean else raw.strip()
if not text:
continue
for token in nlp(text):
if not token_filter(token):
continue
key = str(output_format(token)).strip()
if not key:
continue
if key not in stats:
stats[key] = {
"count": 0,
"pos_counts": defaultdict(int),
"surface_forms": set(),
"original_lemmas": set(),
"example_sentences": [],
"source_lines": [],
}
item = stats[key]
item["count"] += 1
item["pos_counts"][getattr(token, "pos_", "") or ""] += 1
item["surface_forms"].add((getattr(token, "text", "") or "").lower())
original_lemma = (getattr(token, "lemma_", "") or "").strip().lower()
if original_lemma:
item["original_lemmas"].add(original_lemma)
if len(item["example_sentences"]) < max_examples and text not in item["example_sentences"]:
item["example_sentences"].append(text)
if line_number:
item["source_lines"].append(str(line_number))
for item in stats.values():
item["pos_counts"] = dict(item["pos_counts"])
item["surface_forms"] = sorted(item["surface_forms"])
item["original_lemmas"] = sorted(item["original_lemmas"])
item["source_lines"] = list(dict.fromkeys(item["source_lines"]))
return stats
def write_debug_tsv(
stats: dict[str, dict[str, Any]],
path: str,
min_freq: int = 0,
) -> None:
"""Write a debug TSV for detailed extraction statistics."""
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
with open(path, "w", encoding="utf-8") as fh:
fh.write(
"entry\tcount\tpos_counts\ttop_surface_forms"
"\texample_sentences\tsource_lines\toriginal_lemmas\n"
)
sorted_entries = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
for entry, item in sorted_entries:
if min_freq > 0 and item["count"] < min_freq:
continue
pos_counts = ", ".join(
f"{pos}:{count}"
for pos, count in sorted(
item["pos_counts"].items(), key=lambda x: -x[1]
)
if pos
)
surfaces = ", ".join(item["surface_forms"])
examples = "; ".join(item["example_sentences"])
lines = ", ".join(item["source_lines"])
original_lemmas = ", ".join(item.get("original_lemmas", []))
fh.write(
f"{entry}\t{item['count']}\t{pos_counts}\t{surfaces}"
f"\t{examples}\t{lines}\t{original_lemmas}\n"
)
def _write_content_words( def _write_content_words(
stats: dict[str, dict[str, Any]], stats: dict[str, dict[str, Any]],
path: str, path: str,
min_freq: int, min_freq: int,
bad_lemmas: set[str] | None = None,
) -> int: ) -> int:
"""Write a sorted ``lemma count`` file from detailed stats.""" """Write a sorted ``entry count`` file from detailed stats."""
bad = bad_lemmas or set()
items = [ items = [
(lemma, data["count"]) (entry, data["count"])
for lemma, data in stats.items() for entry, data in stats.items()
if data["count"] >= min_freq and lemma not in bad if data["count"] >= min_freq
] ]
items.sort(key=lambda x: (-x[1], x[0])) items.sort(key=lambda x: (-x[1], x[0]))
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True) os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
@@ -240,19 +373,6 @@ def _write_content_words(
return len(items) return len(items)
def _write_function_words(
stats: dict[str, dict[str, Any]],
path: str,
) -> int:
"""Write a function-word frequency file from detailed stats."""
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
items = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
with open(path, "w", encoding="utf-8") as f:
for word, data in items:
f.write(f"{word} {data['count']}\n")
return len(items)
def extract_words_from_file( def extract_words_from_file(
config: Config, config: Config,
lang: str, lang: str,
@@ -261,84 +381,64 @@ def extract_words_from_file(
min_freq: int = 2, min_freq: int = 2,
outdir: str | None = None, outdir: str | None = None,
out: str | None = None, out: str | None = None,
include_proper_nouns: bool = False,
function_words: bool = False,
debug: str | None = None, debug: str | None = None,
lemma_corrections: str | None = None,
bad_lemma_file: str | None = None,
spacy_model: str | None = None, spacy_model: str | None = None,
include_tags: bool = False,
debug_min_freq: int = 0, debug_min_freq: int = 0,
no_clean: bool = False, no_clean: bool = False,
field_section: str = "all",
) -> dict[str, Any]: ) -> dict[str, Any]:
"""Extract frequent Spanish vocabulary from an Anki TSV export file. """Extract frequent vocabulary from an Anki TSV export file.
Parameters Parameters
---------- ----------
config: config:
Saiki configuration. Saiki configuration.
lang: lang:
Language code (e.g. ``"es"``). Language code (e.g. ``"es"`` or ``"jp"``).
input_path: input_path:
Path to the Anki TSV export file. Path to the Anki TSV export file.
field_index: field_index:
1-based column index of the Spanish text (default 2). 1-based column index of the text field (default 2).
min_freq: min_freq:
Minimum frequency to include in output. Minimum frequency to include in output.
outdir: outdir:
Output directory (defaults to config word output root). Output directory (defaults to config word output root).
out: out:
Output filename or path for content words. Output filename or path for content words.
include_proper_nouns:
Include ``PROPN`` tokens.
function_words:
Also extract a separate function-word list.
debug: debug:
Path for the debug TSV. Path for the debug TSV.
lemma_corrections:
Path to a ``bad\\tgood`` TSV of lemma corrections.
bad_lemma_file:
Path to a list of lemmas to skip.
spacy_model: spacy_model:
SpaCy model name override. SpaCy model name override.
field_section:
Which blank-line-separated field section to mine when cleaning:
``"all"`` or ``"first"``.
Returns Returns
------- -------
A dict with keys ``records``, ``stats``, ``written``, ``out``, A dict with keys ``records``, ``stats``, ``written``, ``out``, and
``debug``, ``proper_nouns``, ``function_words``, ``suspicious``. optional debug/supporting output paths.
""" """
language_bucket = config.language_name(lang) language_bucket = config.language_name(lang)
profile = LANGUAGE_PROFILES[language_bucket]
out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket) out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket)
out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}_content.txt") out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}_content.txt")
model_name = spacy_model or str(config.language(lang).get("word_model")) model_name = spacy_model or str(config.language(lang).get("word_model"))
nlp = load_spacy_model(model_name) nlp = load_spacy_model(model_name)
corrections = None
if lemma_corrections:
corrections = load_lemma_corrections(lemma_corrections)
bad_lemmas = None
if bad_lemma_file:
bad_lemmas = load_bad_lemmas(bad_lemma_file)
records = parse_anki_tsv( records = parse_anki_tsv(
input_path, input_path,
field_index=field_index, field_index=field_index,
include_tags=include_tags,
) )
if not records: if not records:
return {"records": 0, "stats": {}, "written": 0, "out": out_path} return {"records": 0, "stats": {}, "written": 0, "out": out_path}
# Build token filter
def _content_filter(token) -> bool:
return spanish_content_filter(token, include_proper_nouns=include_proper_nouns)
stats = extract_detailed_counts( stats = extract_detailed_counts(
records, nlp, _content_filter, records, nlp, profile["token_filter"], profile["output_format"],
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
clean=not no_clean, clean=not no_clean,
field_section=field_section,
) )
written = _write_content_words(stats, out_path, min_freq, bad_lemmas=bad_lemmas) written = _write_content_words(stats, out_path, min_freq)
result: dict[str, Any] = { result: dict[str, Any] = {
"records": len(records), "records": len(records),
@@ -350,81 +450,12 @@ def extract_words_from_file(
# Debug output # Debug output
if debug: if debug:
debug_path = os.path.expanduser(debug) debug_path = os.path.expanduser(debug)
write_debug_tsv(stats, debug_path, bad_lemmas=bad_lemmas, min_freq=debug_min_freq) write_debug_tsv(stats, debug_path, min_freq=debug_min_freq)
result["debug"] = debug_path result["debug"] = debug_path
# Proper nouns
if include_proper_nouns:
proper_path = os.path.join(out_dir, f"words_{lang}_proper_nouns.txt")
proper_items = [
(lemma, data["count"])
for lemma, data in stats.items()
if data["pos_counts"].get("PROPN", 0) > 0
]
proper_items.sort(key=lambda x: (-x[1], x[0]))
os.makedirs(os.path.dirname(os.path.abspath(proper_path)), exist_ok=True)
with open(proper_path, "w", encoding="utf-8") as f:
for w, c in proper_items:
f.write(f"{w} {c}\n")
result["proper_nouns"] = proper_path
# Function words (separate extraction pass)
if function_words:
fw_path = os.path.join(out_dir, f"words_{lang}_function_words.txt")
fw_stats = extract_detailed_counts(
records, nlp, spanish_function_word_filter,
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
)
fw_written = _write_function_words(fw_stats, fw_path)
result["function_words"] = fw_path
result["function_words_written"] = fw_written
# Suspicious tokens
susp_path = os.path.join(out_dir, f"words_{lang}_suspicious_tokens.txt")
write_suspicious_tokens(stats, susp_path, bad_lemmas=bad_lemmas)
result["suspicious"] = susp_path
return result return result
def lint_anki_cards_main(
input_path: str,
field_index: int = 2,
output_path: str = "suspicious_cards_es.tsv",
verbose: bool = False,
) -> dict[str, Any]:
"""Read an Anki TSV export and produce a suspicious-card report.
When *verbose* is True every record is written to the report, not
only those with issues.
Returns ``{"records": , "issues": , "path": output_path}``.
"""
records = parse_anki_tsv(input_path, field_index=field_index)
issues = lint_anki_cards(records, verbose=verbose)
os.makedirs(os.path.dirname(os.path.abspath(output_path)) or ".", exist_ok=True)
with open(output_path, "w", encoding="utf-8") as fh:
fh.write(
"source_line\toriginal_text\tcleaned_text\treason\tsuggested_fix\n"
)
for issue in issues:
fh.write(
f"{issue['source_line']}\t"
f"{_tsv_escape(issue['original_text'])}\t"
f"{_tsv_escape(issue['cleaned_text'])}\t"
f"{_tsv_escape(issue['reason'])}\t"
f"{_tsv_escape(issue['suggested_fix'])}\n"
)
return {"records": len(records), "issues": len(issues), "path": output_path}
def _tsv_escape(value: str) -> str:
"""Escape a value for TSV output (backslash-escape tabs and newlines)."""
return value.replace("\\", "\\\\").replace("\t", "\\t").replace("\n", "\\n")
def normalize_word_for_comparison(word: str) -> str: def normalize_word_for_comparison(word: str) -> str:
"""Normalise a word for file-comparison: lowercase, strip accents.""" """Normalise a word for file-comparison: lowercase, strip accents."""
return word.lower().translate(ACCENT_MAP) return word.lower().translate(ACCENT_MAP)
+104 -296
View File
@@ -26,27 +26,18 @@ from saiki.importer import (
synthesize_tts_sample, synthesize_tts_sample,
supported_tts_backends, supported_tts_backends,
) )
from saiki.spanish import (
BUILTIN_LEMMA_CORRECTIONS,
check_card_for_issues,
clean_anki_field_text,
extract_detailed_counts,
lint_anki_cards,
load_lemma_corrections,
safe_spanish_lemma,
safe_spanish_lemma_info,
spanish_content_filter,
spanish_function_word_filter,
write_debug_tsv,
write_suspicious_tokens,
)
from saiki.text import extract_first_visible_line, extract_visible_text from saiki.text import extract_first_visible_line, extract_visible_text
from saiki.words import ( from saiki.words import (
build_query_from_decks, build_query_from_decks,
clean_anki_field_text,
compare_word_files, compare_word_files,
compare_word_lists, compare_word_lists,
extract_detailed_counts,
extract_words_from_file,
normalize_word_for_comparison, normalize_word_for_comparison,
read_word_file, read_word_file,
spanish_filter,
write_debug_tsv,
) )
from saiki.youtube import TranscriptLine, extract_video_id, sentence_vocab, write_sentence_export from saiki.youtube import TranscriptLine, extract_video_id, sentence_vocab, write_sentence_export
@@ -177,6 +168,15 @@ class AnkiTsvTests(unittest.TestCase):
self.assertEqual(records[0]["line_number"], 2) self.assertEqual(records[0]["line_number"], 2)
self.assertEqual(records[1]["line_number"], 3) self.assertEqual(records[1]["line_number"], 3)
def test_parse_tsv_preserves_hash_data_rows_after_headers(self):
with tempfile.TemporaryDirectory() as tmp:
path = os.path.join(tmp, "test.txt")
with open(path, "w", encoding="utf-8") as f:
f.write("#separator:tab\nalpha\tbeta\n#hashtag\tvalue\n")
records = parse_anki_tsv(path, field_index=1)
self.assertEqual([r["text"] for r in records], ["alpha", "#hashtag"])
self.assertEqual(records[1]["line_number"], 3)
def test_parse_tsv_include_tags(self): def test_parse_tsv_include_tags(self):
with tempfile.TemporaryDirectory() as tmp: with tempfile.TemporaryDirectory() as tmp:
path = os.path.join(tmp, "test.txt") path = os.path.join(tmp, "test.txt")
@@ -215,17 +215,19 @@ class CliValidationTests(unittest.TestCase):
self.assertEqual(code, 1) self.assertEqual(code, 1)
self.assertIn("1-based", stderr.getvalue()) self.assertIn("1-based", stderr.getvalue())
def test_words_input_is_spanish_only(self): def test_words_input_accepts_japanese_tsv(self):
from io import StringIO from io import StringIO
with patch("saiki.cli.importlib.metadata.version", return_value="0"): with patch("saiki.cli.importlib.metadata.version", return_value="0"):
with patch("sys.stderr", new_callable=StringIO) as stderr: with patch("saiki.cli.extract_words_from_file") as extract:
code = main(["words", "jp", "--input", "deck.tsv", "--field", "2"]) extract.return_value = {"records": 0, "written": 0, "out": "out.txt"}
self.assertEqual(code, 1) with patch("sys.stdout", new_callable=StringIO):
self.assertIn("Spanish-only", stderr.getvalue()) code = main(["words", "jp", "--input", "deck.tsv", "--field", "2"])
self.assertEqual(code, 0)
extract.assert_called_once()
class SpanishTextCleaningTests(unittest.TestCase): class AnkiFieldCleaningTests(unittest.TestCase):
def test_remove_sound_marker(self): def test_remove_sound_marker(self):
self.assertEqual( self.assertEqual(
clean_anki_field_text("[sound:es_001.mp3] Hola mundo"), clean_anki_field_text("[sound:es_001.mp3] Hola mundo"),
@@ -236,17 +238,23 @@ class SpanishTextCleaningTests(unittest.TestCase):
result = clean_anki_field_text("<span>Para el lunes...</span>") result = clean_anki_field_text("<span>Para el lunes...</span>")
self.assertEqual(result, "Para el lunes...") self.assertEqual(result, "Para el lunes...")
def test_english_after_brbr_removed(self): def test_double_br_sections_are_preserved_by_default(self):
result = clean_anki_field_text("hasta luego<br><br>(see you later)") result = clean_anki_field_text("Target sentence<br><br>(Translation sentence)")
self.assertEqual(result, "hasta luego") self.assertEqual(result, "Target sentence (Translation sentence)")
def test_english_gloss_after_brbr_removed(self): def test_first_field_section_drops_later_sections_without_phrase_rules(self):
result = clean_anki_field_text("Te quiero.<br><br>I love you / I want you") result = clean_anki_field_text("Target sentence<br><br>Translation sentence")
self.assertEqual(result, "Te quiero.") self.assertEqual(result, "Target sentence Translation sentence")
def test_spanish_after_brbr_preserved(self): result = clean_anki_field_text(
result = clean_anki_field_text("Hola<br><br>Otra línea") "Target sentence<br><br>Translation sentence",
self.assertEqual(result, "Hola Otra línea") field_section="first",
)
self.assertEqual(result, "Target sentence")
def test_later_field_sections_are_preserved_by_default(self):
result = clean_anki_field_text("First line<br><br>Second line")
self.assertEqual(result, "First line Second line")
def test_normalize_whitespace(self): def test_normalize_whitespace(self):
result = clean_anki_field_text(" Hola mundo ") result = clean_anki_field_text(" Hola mundo ")
@@ -283,230 +291,13 @@ class MockToken:
self.sent = None self.sent = None
class SpanishLemmaTests(unittest.TestCase): class DetailedCountsTests(unittest.TestCase):
def test_builtin_correction_comar(self): def test_spanish_filter_uses_content_pos(self):
token = MockToken(text="como", lemma="comar") self.assertTrue(spanish_filter(MockToken(text="casa", lemma="casa", pos="NOUN")))
self.assertEqual(safe_spanish_lemma(token), "comer") self.assertTrue(spanish_filter(MockToken(text="rápido", lemma="rápido", pos="ADV")))
self.assertFalse(spanish_filter(MockToken(text="Madrid", lemma="Madrid", pos="PROPN")))
self.assertFalse(spanish_filter(MockToken(text="para", lemma="para", pos="ADP")))
def test_builtin_correction_acabir(self):
token = MockToken(text="acabas", lemma="acabir")
self.assertEqual(safe_spanish_lemma(token), "acabar")
def test_builtin_correction_delicius(self):
token = MockToken(text="delicioso", lemma="deliciós")
self.assertEqual(safe_spanish_lemma(token), "delicioso")
def test_builtin_correction_llover_slash(self):
token = MockToken(text="llueve", lemma="llover/llover")
self.assertEqual(safe_spanish_lemma(token), "llover")
def test_multi_word_lemma_falls_back(self):
token = MockToken(text="ayudar", lemma="ayudar yo")
self.assertEqual(safe_spanish_lemma(token), "ayudar")
def test_multi_word_lemma_lavar_el(self):
token = MockToken(text="lava", lemma="lavar él")
self.assertEqual(safe_spanish_lemma(token), "lava")
def test_prir_corrected_when_source_is_pedir(self):
for form in ["pide", "pido", "piden", "pidiendo", "pedir"]:
token = MockToken(text=form, lemma="prir")
self.assertEqual(safe_spanish_lemma(token), "pedir")
def test_prir_not_corrected_for_unknown_form(self):
token = MockToken(text="prir", lemma="prir")
self.assertEqual(safe_spanish_lemma(token), "prir")
def test_empty_lemma_falls_back(self):
token = MockToken(text="hola", lemma="")
self.assertEqual(safe_spanish_lemma(token), "hola")
def test_lemma_with_punctuation_falls_back(self):
token = MockToken(text="comiendo", lemma="comiendo,")
self.assertEqual(safe_spanish_lemma(token), "comiendo")
def test_normal_lemma_passes_through(self):
token = MockToken(text="comiendo", lemma="comer")
self.assertEqual(safe_spanish_lemma(token), "comer")
def test_lemma_info_tracks_correction_reason(self):
token = MockToken(text="como", lemma="comar")
info = safe_spanish_lemma_info(token)
self.assertEqual(info.lemma, "comer")
self.assertEqual(info.original_lemma, "comar")
self.assertIn("corrected", info.status)
def test_non_spanish_lemma_falls_back(self):
token = MockToken(text="hola", lemma="hello_world")
info = safe_spanish_lemma_info(token)
self.assertEqual(info.lemma, "hola")
self.assertEqual(info.status, "fallback_bad_lemma_characters")
def test_extra_corrections_override_builtin(self):
extra = {"comar": "comprar"}
token = MockToken(text="como", lemma="comar")
self.assertEqual(
safe_spanish_lemma(token, extra_corrections=extra),
"comprar",
)
def test_load_lemma_corrections(self):
with tempfile.TemporaryDirectory() as tmp:
path = os.path.join(tmp, "corrections.tsv")
with open(path, "w", encoding="utf-8") as f:
f.write("malo\tbueno\nfeo\tbonito\n")
corrections = load_lemma_corrections(path)
self.assertEqual(corrections, {"malo": "bueno", "feo": "bonito"})
def test_load_lemma_corrections_skips_comments(self):
with tempfile.TemporaryDirectory() as tmp:
path = os.path.join(tmp, "corrections.tsv")
with open(path, "w", encoding="utf-8") as f:
f.write("# comment\nmalo\tbueno\n")
corrections = load_lemma_corrections(path)
self.assertEqual(corrections, {"malo": "bueno"})
class SpanishFilterTests(unittest.TestCase):
def test_content_filter_keeps_noun(self):
token = MockToken(text="casa", lemma="casa", pos="NOUN")
self.assertTrue(spanish_content_filter(token))
def test_content_filter_keeps_verb(self):
token = MockToken(text="come", lemma="comer", pos="VERB")
self.assertTrue(spanish_content_filter(token))
def test_content_filter_keeps_adj(self):
token = MockToken(text="grande", lemma="grande", pos="ADJ")
self.assertTrue(spanish_content_filter(token))
def test_content_filter_keeps_adv(self):
token = MockToken(text="bien", lemma="bien", pos="ADV")
self.assertTrue(spanish_content_filter(token))
def test_content_filter_excludes_propn_by_default(self):
token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN")
self.assertFalse(spanish_content_filter(token))
def test_content_filter_includes_propn_with_flag(self):
token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN")
self.assertTrue(
spanish_content_filter(token, include_proper_nouns=True)
)
def test_content_filter_excludes_punct(self):
token = MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True)
self.assertFalse(spanish_content_filter(token))
def test_content_filter_excludes_digit(self):
token = MockToken(text="123", lemma="123", pos="NUM", is_digit=True)
self.assertFalse(spanish_content_filter(token))
def test_content_filter_excludes_url(self):
token = MockToken(
text="http://x.com", lemma="http://x.com", pos="X", like_url=True
)
self.assertFalse(spanish_content_filter(token))
def test_content_filter_excludes_media_filename(self):
token = MockToken(text="es_001.mp3", lemma="es_001.mp3", pos="NOUN")
self.assertFalse(spanish_content_filter(token))
def test_content_filter_excludes_one_char_non_vowel(self):
token = MockToken(text="x", lemma="x", pos="NOUN")
self.assertFalse(spanish_content_filter(token))
def test_function_word_filter_keeps_preposition(self):
token = MockToken(text="para", lemma="para", pos="ADP")
self.assertTrue(spanish_function_word_filter(token))
def test_function_word_filter_keeps_conjunction(self):
token = MockToken(text="y", lemma="y", pos="CCONJ")
self.assertTrue(spanish_function_word_filter(token))
def test_function_word_filter_rejects_noun(self):
token = MockToken(text="casa", lemma="casa", pos="NOUN")
self.assertFalse(spanish_function_word_filter(token))
class SpanishLintTests(unittest.TestCase):
def _check(self, text: str) -> list:
return check_card_for_issues(text, clean_anki_field_text(text))
def test_detect_tu_sabes_nada(self):
issues = self._check("Tú sabes nada")
self.assertTrue(any("Missing negation" in i["reason"] for i in issues))
def test_detect_dulce_suenos(self):
issues = self._check("Dulce sueños")
self.assertTrue(
any("agree in number" in i["reason"] for i in issues)
)
def test_detect_ella_sona(self):
issues = self._check("Ella soña con viajar.")
self.assertTrue(
any("present indicative" in i["reason"] for i in issues)
)
def test_detect_tengo_uno_libro(self):
issues = self._check("Tengo uno libro en la mesa.")
self.assertTrue(
any("instead of 'un'" in i["reason"] for i in issues)
)
def test_detect_observando_nadan(self):
issues = self._check(
"La pareja está en un bote observando a los peces nadan."
)
self.assertTrue(
any("'cómo'" in i["reason"] for i in issues)
)
def test_detect_le_vuelven_loca(self):
issues = self._check("Los perros pequeños le vuelven loca.")
self.assertTrue(
any("direct object" in i["reason"] for i in issues)
)
def test_detect_ai_generated_tts(self):
issues = self._check("AI-generated text-to-speech")
self.assertTrue(
any("Metadata contamination" in i["reason"] for i in issues)
)
def test_detect_necesito_missing_article(self):
issues = self._check("Necesito jardinero")
self.assertTrue(
any("Missing article" in i["reason"] for i in issues)
)
def test_detect_le_vuelven_loco(self):
issues = self._check("Estos problemas le vuelven loco.")
self.assertTrue(
any("direct object" in i["reason"] or "direct object pronoun" in i["reason"] for i in issues)
)
def test_veintiuno_anos(self):
issues = self._check("Tengo veintiuno años de edad.")
self.assertTrue(
any("veintiún" in i["suggested_fix"] for i in issues)
)
def test_lint_anki_cards_multiple_records(self):
records = [
{"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"},
{"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"},
{"text": "Ella soña con viajar.", "line_number": 3, "raw_line": "Ella soña con viajar."},
]
results = lint_anki_cards(records)
self.assertEqual(len(results), 2)
reasons = results[0]["reason"] + results[1]["reason"]
self.assertIn("Missing negation", reasons)
self.assertIn("present indicative", reasons)
class SpanishDetailedCountsTests(unittest.TestCase):
def test_extract_detailed_counts_basic(self): def test_extract_detailed_counts_basic(self):
records = [ records = [
{"text": "Yo como manzanas.", "line_number": 1}, {"text": "Yo como manzanas.", "line_number": 1},
@@ -525,7 +316,7 @@ class SpanishDetailedCountsTests(unittest.TestCase):
if "como" in text: if "como" in text:
return MockDoc([ return MockDoc([
MockToken(text="Yo", lemma="yo", pos="PRON"), MockToken(text="Yo", lemma="yo", pos="PRON"),
MockToken(text="como", lemma="comar", pos="VERB"), MockToken(text="como", lemma="comer", pos="VERB"),
MockToken(text="manzanas", lemma="manzana", pos="NOUN"), MockToken(text="manzanas", lemma="manzana", pos="NOUN"),
MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True), MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True),
]) ])
@@ -540,12 +331,10 @@ class SpanishDetailedCountsTests(unittest.TestCase):
records, records,
MockNLP(), MockNLP(),
lambda t: t.pos_ in {"NOUN", "VERB", "ADJ", "ADV"}, lambda t: t.pos_ in {"NOUN", "VERB", "ADJ", "ADV"},
safe_spanish_lemma, lambda t: (t.lemma_ or t.text).lower(),
) )
# "comar" should be corrected to "comer"
self.assertIn("comer", stats) self.assertIn("comer", stats)
# "comer" should have count 2 (comes + como -> comer)
self.assertEqual(stats["comer"]["count"], 2) self.assertEqual(stats["comer"]["count"], 2)
self.assertIn("manzana", stats) self.assertIn("manzana", stats)
self.assertIn("pan", stats) self.assertIn("pan", stats)
@@ -571,27 +360,6 @@ class SpanishDetailedCountsTests(unittest.TestCase):
self.assertIn("Yo como.", content) self.assertIn("Yo como.", content)
self.assertIn("1, 2, 3", content) self.assertIn("1, 2, 3", content)
def test_suspicious_tokens_output(self):
stats = {
"comer": {
"count": 3,
"pos_counts": {"VERB": 3},
"surface_forms": ["como", "comes"],
"original_lemmas": ["comar", "comer"],
"lemma_statuses": {"corrected:comar->comer": 1, "ok": 2},
"example_sentences": ["Yo como."],
"source_lines": ["1"],
}
}
with tempfile.TemporaryDirectory() as tmp:
path = os.path.join(tmp, "suspicious.tsv")
write_suspicious_tokens(stats, path)
with open(path, "r", encoding="utf-8") as f:
content = f.read()
self.assertIn("comer", content)
self.assertIn("comar", content)
self.assertIn("corrected:comar->comer", content)
def test_debug_tsv_min_freq_filters(self): def test_debug_tsv_min_freq_filters(self):
stats = { stats = {
"comer": {"count": 5, "pos_counts": {"VERB": 5}, "comer": {"count": 5, "pos_counts": {"VERB": 5},
@@ -638,29 +406,69 @@ class SpanishDetailedCountsTests(unittest.TestCase):
lambda t: t.text.lower(), lambda t: t.text.lower(),
clean=True, clean=True,
) )
# With clean=True, the sound marker is stripped → empty text → no stats # With clean=True, the sound marker is stripped, so there is no text to count.
self.assertEqual(len(stats_clean), 0) self.assertEqual(len(stats_clean), 0)
def test_lint_anki_cards_verbose_includes_all(self): def test_extract_detailed_counts_can_use_first_field_section(self):
records = [ class MockDoc:
{"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"}, def __init__(self, tokens):
{"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"}, self.tokens = tokens
] def __iter__(self):
# Without verbose only suspicious return iter(self.tokens)
results = lint_anki_cards(records, verbose=False)
self.assertEqual(len(results), 1)
# With verbose both records class MockNLP:
results = lint_anki_cards(records, verbose=True) def __call__(self, text):
self.assertEqual(len(results), 2) return MockDoc([
MockToken(text=part, lemma=part.lower(), pos="NOUN")
for part in text.split()
])
def test_lint_anki_cards_verbose_empty_field(self): records = [{"text": "Target<br><br>Translation", "line_number": 1}]
records = [ stats = extract_detailed_counts(
{"text": "", "line_number": 1, "raw_line": ""}, records,
] MockNLP(),
results = lint_anki_cards(records, verbose=True) lambda t: True,
self.assertEqual(len(results), 1) lambda t: t.lemma_,
self.assertEqual(results[0]["reason"], "empty field") field_section="first",
)
self.assertIn("target", stats)
self.assertNotIn("translation", stats)
def test_extract_words_from_file_uses_language_profile(self):
class MockDoc:
def __init__(self, tokens):
self.tokens = tokens
def __iter__(self):
return iter(self.tokens)
class MockNLP:
def __call__(self, text):
return MockDoc([
MockToken(text="", lemma="", pos="NOUN"),
MockToken(text="です", lemma="です", pos="AUX"),
])
with tempfile.TemporaryDirectory() as tmp:
source = os.path.join(tmp, "deck.tsv")
out = os.path.join(tmp, "words_jp.txt")
with open(source, "w", encoding="utf-8") as f:
f.write("#separator:tab\n[sound:jp_001.mp3]\t猫です。\n")
with patch("saiki.words.load_spacy_model", return_value=MockNLP()):
result = extract_words_from_file(
Config(deepcopy(DEFAULT_CONFIG)),
"jp",
source,
out=out,
min_freq=1,
)
with open(out, "r", encoding="utf-8") as f:
content = f.read()
self.assertEqual(result["records"], 1)
self.assertIn("猫 1", content)
self.assertNotIn("です", content)
class CompareWordListsTests(unittest.TestCase): class CompareWordListsTests(unittest.TestCase):