Improve Spanish Anki vocabulary tooling
This commit is contained in:
@@ -199,7 +199,78 @@ hablar 9
|
|||||||
見る (見た) 6
|
見る (見た) 6
|
||||||
```
|
```
|
||||||
|
|
||||||
### YouTube
|
### Words from TSV Export
|
||||||
|
|
||||||
|
Extract vocabulary from an Anki TSV export file instead of using AnkiConnect:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
saiki words --lang es --input Español.txt --field 2 --output words_es_content.txt --debug words_es_debug.tsv
|
||||||
|
```
|
||||||
|
|
||||||
|
Additional file-based options:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
saiki words --lang es --input Español.txt --field 2 --include-proper-nouns
|
||||||
|
saiki words --lang es --input Español.txt --field 2 --function-words
|
||||||
|
saiki words --lang es --input Español.txt --field 2 --lemma-corrections my_fixes.tsv
|
||||||
|
saiki words --lang es --input Español.txt --field 2 --bad-lemma-file bad_lemmas.txt
|
||||||
|
```
|
||||||
|
|
||||||
|
When `--input` is provided, `--field` specifies the 1-based column index of the
|
||||||
|
Spanish text (default 2). The audio column (index 1) and tags column are
|
||||||
|
automatically skipped. File-based NLP extraction is currently Spanish-specific;
|
||||||
|
the older AnkiConnect-based `saiki words jp` flow is unchanged.
|
||||||
|
|
||||||
|
The file-based pipeline:
|
||||||
|
- Parses Anki `#` header lines (`#separator:tab`, `#html:true`, `#tags column:N`)
|
||||||
|
- Uses Python's `csv` module for robust TSV parsing
|
||||||
|
- Removes `[sound:...mp3]` markers, HTML tags, and English glosses after
|
||||||
|
`<br><br>` from the field text
|
||||||
|
- Applies safe Spanish lemmatisation (multi-word lemmas like `ayudar yo` or
|
||||||
|
`lavar él` are rejected; bad lemmas like `comar` -> `comer` are corrected)
|
||||||
|
- Tracks POS counts, surface forms, example sentences, and source line numbers
|
||||||
|
- Tracks original spaCy lemmas and correction/fallback reasons for debugging
|
||||||
|
|
||||||
|
Output files produced:
|
||||||
|
- `words_es_content.txt` — cleaned content vocabulary (NOUN, VERB, ADJ, ADV)
|
||||||
|
- `words_es_debug.tsv` — per-lemma debug info
|
||||||
|
- `words_es_proper_nouns.txt` — proper nouns (only with `--include-proper-nouns`)
|
||||||
|
- `words_es_function_words.txt` — function words (only with `--function-words`)
|
||||||
|
- `words_es_suspicious_tokens.txt` — lemmas that required correction
|
||||||
|
|
||||||
|
Debug TSV example:
|
||||||
|
|
||||||
|
```text
|
||||||
|
lemma count pos_counts top_surface_forms example_sentences source_lines original_lemmas lemma_statuses status
|
||||||
|
comer 8 VERB:8 como, come, comen Yo como manzanas.; Ustedes los comen con arroz. 70,979 comar, comer corrected:comar->comer:1, ok:7
|
||||||
|
```
|
||||||
|
|
||||||
|
### Lint Spanish Cards
|
||||||
|
|
||||||
|
Check an Anki TSV export for suspicious or awkward Spanish:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
saiki lint-anki-es --input Español.txt --field 2 --output suspicious_cards_es.tsv
|
||||||
|
```
|
||||||
|
|
||||||
|
Detects known errors and rule-based suspicious patterns and produces a TSV
|
||||||
|
report with: source line, original text, cleaned text, reason, and suggested
|
||||||
|
fix.
|
||||||
|
|
||||||
|
### Compare
|
||||||
|
|
||||||
|
Compare deck vocabulary against a target list:
|
||||||
|
|
||||||
|
```shell
|
||||||
|
saiki compare --deck-words words_es_content.txt --target-words target_es_top_1000.txt --output missing_from_deck.txt
|
||||||
|
saiki compare --deck-words words_es_content.txt --target-words target_es_top_1000.txt --min-frequency 3
|
||||||
|
```
|
||||||
|
|
||||||
|
Normalises case and optionally strips accents for matching, so `cómo` and
|
||||||
|
`como` are treated as the same word. Use `--min-frequency` to ignore accidental
|
||||||
|
low-frequency words in the deck.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Mine vocabulary or sentence rows from YouTube subtitles.
|
Mine vocabulary or sentence rows from YouTube subtitles.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,124 @@
|
|||||||
|
"""Parse Anki TSV export files.
|
||||||
|
|
||||||
|
Anki exports are tab-separated files with ``#`` header lines describing
|
||||||
|
the separator, HTML mode, and tag column. This module parses those
|
||||||
|
headers and extracts structured records.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
SEPARATOR_ALIASES = {
|
||||||
|
"tab": "\t",
|
||||||
|
"space": " ",
|
||||||
|
"comma": ",",
|
||||||
|
"semicolon": ";",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_separator(raw: str) -> str:
|
||||||
|
"""Map Anki separator names to actual delimiter characters."""
|
||||||
|
return SEPARATOR_ALIASES.get(raw.strip().lower(), raw.strip())
|
||||||
|
|
||||||
|
|
||||||
|
def parse_anki_headers(path: str) -> dict[str, Any]:
|
||||||
|
"""Read ``#`` -prefixed metadata lines from an Anki export file.
|
||||||
|
|
||||||
|
Stops at the first non-header line. Returns a dict with optional keys
|
||||||
|
``separator``, ``html``, and ``tags_column``.
|
||||||
|
"""
|
||||||
|
info: dict[str, Any] = {}
|
||||||
|
with open(path, "r", encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
line = line.rstrip("\n\r")
|
||||||
|
if not line.startswith("#"):
|
||||||
|
break
|
||||||
|
if line.startswith("#separator:"):
|
||||||
|
raw = line[len("#separator:"):].strip()
|
||||||
|
info["separator"] = _resolve_separator(raw)
|
||||||
|
elif line.startswith("#html:"):
|
||||||
|
info["html"] = line[len("#html:"):].strip().lower() == "true"
|
||||||
|
elif line.startswith("#tags column:"):
|
||||||
|
info["tags_column"] = line[len("#tags column:"):].strip()
|
||||||
|
return info
|
||||||
|
|
||||||
|
|
||||||
|
def parse_anki_tsv(
|
||||||
|
path: str,
|
||||||
|
field_index: int = 2,
|
||||||
|
include_tags: bool = False,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Parse an Anki TSV export file into structured records.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
path:
|
||||||
|
Path to the Anki TSV export file.
|
||||||
|
field_index:
|
||||||
|
1-based column index of the target text field (default 2).
|
||||||
|
include_tags:
|
||||||
|
Whether to include the tags column in output records.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
A list of record dicts with keys:
|
||||||
|
``text``, ``tags`` (if *include_tags* is True), ``line_number``,
|
||||||
|
and ``raw_line``.
|
||||||
|
|
||||||
|
Header lines (``#``-prefixed) are skipped. The actual separator
|
||||||
|
is detected from ``#separator:`` and defaults to tab.
|
||||||
|
"""
|
||||||
|
if field_index < 1:
|
||||||
|
raise ValueError(f"field_index must be 1-based and >= 1, got {field_index}")
|
||||||
|
|
||||||
|
header = parse_anki_headers(path)
|
||||||
|
separator = header.get("separator", "\t")
|
||||||
|
|
||||||
|
records: list[dict[str, Any]] = []
|
||||||
|
col_idx = field_index - 1 # convert to 0-based
|
||||||
|
|
||||||
|
tags_column: int | None = None
|
||||||
|
if include_tags and "tags_column" in header:
|
||||||
|
try:
|
||||||
|
tags_column = int(header["tags_column"]) - 1
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"Invalid #tags column value: {header['tags_column']!r}"
|
||||||
|
) from exc
|
||||||
|
if tags_column < 0:
|
||||||
|
raise ValueError(
|
||||||
|
f"#tags column must be 1-based and >= 1, got {header['tags_column']!r}"
|
||||||
|
)
|
||||||
|
|
||||||
|
with open(path, "r", encoding="utf-8") as fh:
|
||||||
|
reader = csv.reader(fh, delimiter=separator)
|
||||||
|
line_number = 0
|
||||||
|
for row in reader:
|
||||||
|
line_number += 1
|
||||||
|
# skip header rows and completely empty rows
|
||||||
|
if not row:
|
||||||
|
continue
|
||||||
|
if row and row[0].startswith("#"):
|
||||||
|
continue
|
||||||
|
|
||||||
|
if col_idx >= len(row):
|
||||||
|
raise ValueError(
|
||||||
|
f"Line {line_number}: field index {field_index} not found "
|
||||||
|
f"in row with {len(row)} column(s)"
|
||||||
|
)
|
||||||
|
|
||||||
|
text = row[col_idx].strip()
|
||||||
|
record: dict[str, Any] = {
|
||||||
|
"text": text,
|
||||||
|
"line_number": line_number,
|
||||||
|
"raw_line": separator.join(row),
|
||||||
|
}
|
||||||
|
if include_tags and tags_column is not None:
|
||||||
|
record["tags"] = (
|
||||||
|
row[tags_column].strip() if tags_column < len(row) else ""
|
||||||
|
)
|
||||||
|
records.append(record)
|
||||||
|
|
||||||
|
return records
|
||||||
+141
-13
@@ -15,7 +15,7 @@ from .importer import (
|
|||||||
supported_tts_backends,
|
supported_tts_backends,
|
||||||
synthesize_tts_sample,
|
synthesize_tts_sample,
|
||||||
)
|
)
|
||||||
from .words import compare_word_files, extract_words
|
from .words import compare_word_files, compare_word_lists, extract_words, extract_words_from_file, lint_anki_cards_main
|
||||||
from .youtube import run_youtube
|
from .youtube import run_youtube
|
||||||
|
|
||||||
|
|
||||||
@@ -85,22 +85,48 @@ def build_parser(config: Config | None = None) -> argparse.ArgumentParser:
|
|||||||
audio.add_argument("--media-dir")
|
audio.add_argument("--media-dir")
|
||||||
audio.add_argument("--copy-only-new", action="store_true")
|
audio.add_argument("--copy-only-new", action="store_true")
|
||||||
|
|
||||||
words = sub.add_parser("words", help="Extract frequent words from Anki.")
|
words = sub.add_parser("words", help="Extract frequent words from Anki or from a TSV export file.")
|
||||||
words.add_argument("lang", choices=choices)
|
words.add_argument("lang", nargs="?", choices=choices)
|
||||||
|
words.add_argument("--lang", dest="lang_option", choices=choices, help="Language code.")
|
||||||
|
words.add_argument("--input", help="Anki TSV export file (instead of AnkiConnect).")
|
||||||
group = words.add_mutually_exclusive_group()
|
group = words.add_mutually_exclusive_group()
|
||||||
group.add_argument("--query")
|
group.add_argument("--query")
|
||||||
group.add_argument("--deck", action="append")
|
group.add_argument("--deck", action="append")
|
||||||
words.add_argument("--field")
|
words.add_argument("--field", help="Anki field name (AnkiConnect) or 1-based column index (TSV file).")
|
||||||
words.add_argument("--min-freq", type=int, default=2)
|
words.add_argument("--min-freq", type=int, default=2)
|
||||||
words.add_argument("--outdir")
|
words.add_argument("--outdir")
|
||||||
words.add_argument("--out")
|
words.add_argument("--out", "--output", dest="out")
|
||||||
words.add_argument("--full-field", action="store_true")
|
words.add_argument("--full-field", action="store_true")
|
||||||
words.add_argument("--spacy-model")
|
words.add_argument("--spacy-model")
|
||||||
|
words.add_argument("--include-proper-nouns", action="store_true", help="Include proper nouns (PROPN).")
|
||||||
|
words.add_argument("--function-words", action="store_true", help="Extract function words separately.")
|
||||||
|
words.add_argument("--debug", help="Path for debug TSV output.")
|
||||||
|
words.add_argument("--lemma-corrections", help="Path to bad<TAB>good TSV of lemma corrections.")
|
||||||
|
words.add_argument("--bad-lemma-file", help="Path to file listing lemmas to skip.")
|
||||||
|
words.add_argument("--include-tags", action="store_true", help="Include tags column in parsing.")
|
||||||
|
words.add_argument("--debug-min-freq", type=int, default=0,
|
||||||
|
help="Minimum frequency for debug output (default: all).")
|
||||||
|
words.add_argument("--no-clean", action="store_true",
|
||||||
|
help="Skip Anki field text cleaning before NLP.")
|
||||||
|
|
||||||
compare = sub.add_parser("compare-words", help="Print words in source that are not in known.")
|
compare = sub.add_parser("compare-words", help="Print words in source that are not in known.")
|
||||||
compare.add_argument("source")
|
compare.add_argument("source")
|
||||||
compare.add_argument("known")
|
compare.add_argument("known")
|
||||||
|
|
||||||
|
compare_new = sub.add_parser("compare", help="Compare deck vocabulary against a target list.")
|
||||||
|
compare_new.add_argument("--deck-words", required=True, help="Path to deck word list.")
|
||||||
|
compare_new.add_argument("--target-words", required=True, help="Path to target vocabulary list.")
|
||||||
|
compare_new.add_argument("--output", help="Write missing words to this file.")
|
||||||
|
compare_new.add_argument("--seen-output", help="Write seen words to this file.")
|
||||||
|
compare_new.add_argument("--min-frequency", type=int, default=0,
|
||||||
|
help="Minimum frequency for deck words to count.")
|
||||||
|
|
||||||
|
lint_es = sub.add_parser("lint-anki-es", help="Check Spanish Anki cards for suspicious patterns.")
|
||||||
|
lint_es.add_argument("--input", required=True, help="Anki TSV export file.")
|
||||||
|
lint_es.add_argument("--field", type=int, default=2, help="1-based column index (default 2).")
|
||||||
|
lint_es.add_argument("--output", default="suspicious_cards_es.tsv", help="Output TSV path.")
|
||||||
|
lint_es.add_argument("--verbose", action="store_true", help="Output all cards, not just suspicious ones.")
|
||||||
|
|
||||||
youtube = sub.add_parser("youtube", help="Mine a YouTube transcript.")
|
youtube = sub.add_parser("youtube", help="Mine a YouTube transcript.")
|
||||||
youtube.add_argument("lang", choices=choices)
|
youtube.add_argument("lang", choices=choices)
|
||||||
youtube.add_argument("video")
|
youtube.add_argument("video")
|
||||||
@@ -144,6 +170,19 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
parser = build_parser(config)
|
parser = build_parser(config)
|
||||||
args = parser.parse_args(argv)
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
if args.command == "words":
|
||||||
|
lang_option = getattr(args, "lang_option", None)
|
||||||
|
if args.lang and lang_option and args.lang != lang_option:
|
||||||
|
print(
|
||||||
|
f"Error: conflicting language values: {args.lang!r} and {lang_option!r}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 2
|
||||||
|
args.lang = lang_option or args.lang
|
||||||
|
if not args.lang:
|
||||||
|
print("Error: words requires a language, e.g. 'saiki words es' or 'saiki words --lang es'", file=sys.stderr)
|
||||||
|
return 2
|
||||||
|
|
||||||
if args.command == "audio":
|
if args.command == "audio":
|
||||||
result = extract_audio(config, args.lang, args.outdir, args.media_dir, args.copy_only_new, args.concat)
|
result = extract_audio(config, args.lang, args.outdir, args.media_dir, args.copy_only_new, args.concat)
|
||||||
print(f"Copied {result['copied']} files")
|
print(f"Copied {result['copied']} files")
|
||||||
@@ -154,14 +193,62 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
if args.command == "words":
|
if args.command == "words":
|
||||||
result = extract_words(
|
if args.input:
|
||||||
config, args.lang, args.query, args.deck, args.field, args.min_freq,
|
# File-based extraction
|
||||||
args.outdir, args.out, args.full_field, args.spacy_model,
|
if config.language_name(args.lang) != "spanish":
|
||||||
)
|
print(
|
||||||
print(f"Query: {result['query']}")
|
"Error: TSV file-based word extraction is currently Spanish-only.",
|
||||||
print(f"Found {result['notes']} notes")
|
file=sys.stderr,
|
||||||
print(f"Extracted {result['unique']} unique entries")
|
)
|
||||||
print(f"Wrote {result['written']} entries to: {result['out']}")
|
return 1
|
||||||
|
if args.field and not args.field.isdigit():
|
||||||
|
print(
|
||||||
|
f"Error: --field must be a 1-based column index (integer) "
|
||||||
|
f"when --input is used, got: {args.field!r}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 1
|
||||||
|
field_index = int(args.field) if args.field else 2
|
||||||
|
if field_index < 1:
|
||||||
|
print(
|
||||||
|
f"Error: --field must be a 1-based column index >= 1, got: {field_index}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 1
|
||||||
|
result = extract_words_from_file(
|
||||||
|
config, args.lang, args.input,
|
||||||
|
field_index=field_index,
|
||||||
|
min_freq=args.min_freq,
|
||||||
|
outdir=args.outdir,
|
||||||
|
out=args.out,
|
||||||
|
include_proper_nouns=args.include_proper_nouns,
|
||||||
|
function_words=args.function_words,
|
||||||
|
debug=args.debug,
|
||||||
|
lemma_corrections=args.lemma_corrections,
|
||||||
|
bad_lemma_file=args.bad_lemma_file,
|
||||||
|
spacy_model=args.spacy_model,
|
||||||
|
include_tags=args.include_tags,
|
||||||
|
debug_min_freq=args.debug_min_freq,
|
||||||
|
no_clean=args.no_clean,
|
||||||
|
)
|
||||||
|
print(f"Parsed {result['records']} records")
|
||||||
|
print(f"Wrote {result['written']} content entries to: {result['out']}")
|
||||||
|
if result.get("debug"):
|
||||||
|
print(f"Debug output: {result['debug']}")
|
||||||
|
if result.get("proper_nouns"):
|
||||||
|
print(f"Proper nouns: {result['proper_nouns']}")
|
||||||
|
if result.get("function_words"):
|
||||||
|
print(f"Function words: {result['function_words']}")
|
||||||
|
print(f"Suspicious tokens: {result.get('suspicious', '')}")
|
||||||
|
else:
|
||||||
|
result = extract_words(
|
||||||
|
config, args.lang, args.query, args.deck, args.field, args.min_freq,
|
||||||
|
args.outdir, args.out, args.full_field, args.spacy_model,
|
||||||
|
)
|
||||||
|
print(f"Query: {result['query']}")
|
||||||
|
print(f"Found {result['notes']} notes")
|
||||||
|
print(f"Extracted {result['unique']} unique entries")
|
||||||
|
print(f"Wrote {result['written']} entries to: {result['out']}")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
if args.command == "compare-words":
|
if args.command == "compare-words":
|
||||||
@@ -169,6 +256,47 @@ def main(argv: list[str] | None = None) -> int:
|
|||||||
print(line)
|
print(line)
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
if args.command == "compare":
|
||||||
|
result = compare_word_lists(
|
||||||
|
args.deck_words,
|
||||||
|
args.target_words,
|
||||||
|
output_path=args.output,
|
||||||
|
min_frequency=args.min_frequency,
|
||||||
|
seen_output_path=args.seen_output,
|
||||||
|
)
|
||||||
|
print(f"Total missing from deck: {result['total_missing']}")
|
||||||
|
seen_count = len(result["seen"])
|
||||||
|
print(f"Total seen in deck: {seen_count}")
|
||||||
|
if result["missing"]:
|
||||||
|
print("Missing words:")
|
||||||
|
for w in result["missing"][:20]:
|
||||||
|
print(f" {w}")
|
||||||
|
if len(result["missing"]) > 20:
|
||||||
|
print(f" ... and {len(result['missing']) - 20} more")
|
||||||
|
if args.output:
|
||||||
|
print(f"Missing words written to: {args.output}")
|
||||||
|
if args.seen_output:
|
||||||
|
print(f"Seen words written to: {args.seen_output}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if args.command == "lint-anki-es":
|
||||||
|
if args.field < 1:
|
||||||
|
print(
|
||||||
|
f"Error: --field must be a 1-based column index >= 1, got: {args.field}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return 1
|
||||||
|
result = lint_anki_cards_main(
|
||||||
|
args.input,
|
||||||
|
field_index=args.field,
|
||||||
|
output_path=args.output,
|
||||||
|
verbose=args.verbose,
|
||||||
|
)
|
||||||
|
print(f"Checked {result['records']} cards")
|
||||||
|
print(f"Found {result['issues']} suspicious card(s)")
|
||||||
|
print(f"Report written to: {result['path']}")
|
||||||
|
return 0
|
||||||
|
|
||||||
if args.command == "youtube":
|
if args.command == "youtube":
|
||||||
result = run_youtube(
|
result = run_youtube(
|
||||||
config, args.lang, args.video, args.mode, args.top, args.no_stopwords,
|
config, args.lang, args.video, args.mode, args.top, args.no_stopwords,
|
||||||
|
|||||||
@@ -0,0 +1,719 @@
|
|||||||
|
"""Spanish-specific NLP utilities for vocabulary extraction and card linting."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
import re
|
||||||
|
from collections import defaultdict
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Any, Callable
|
||||||
|
|
||||||
|
# ── Regex patterns ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
SOUND_RE = re.compile(r"\[sound:[^\]]*\]")
|
||||||
|
MEDIA_RE = re.compile(r"\[\s*(?:sound|image|media):[^\]]*\]", re.IGNORECASE)
|
||||||
|
MEDIA_FILENAME_RE = re.compile(
|
||||||
|
r"[^/\s]+\.(?:mp3|ogg|wav|m4a|flac|jpg|jpeg|png|gif|webp|mp4|webm)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
BR_RE = re.compile(r"<\s*br\s*/?\s*>", re.IGNORECASE)
|
||||||
|
DOUBLE_BR_RE = re.compile(
|
||||||
|
r"<\s*br\s*/?\s*>\s*<\s*br\s*/?\s*>", re.IGNORECASE
|
||||||
|
)
|
||||||
|
HTML_TAG_RE = re.compile(r"<[^>]+>")
|
||||||
|
URL_RE = re.compile(r"https?://\S+|www\.\S+", re.IGNORECASE)
|
||||||
|
EMAIL_RE = re.compile(r"\S+@\S+\.\S+")
|
||||||
|
MULTI_WS_RE = re.compile(r"[ \t]+")
|
||||||
|
SPANISH_WORD_RE = re.compile(r"^[a-záéíóúüñ]+$", re.IGNORECASE)
|
||||||
|
BAD_TOKEN_TEXT_RE = re.compile(r"[\d_/@#\\|+*=<>~`^]")
|
||||||
|
LEMMA_PUNCT_RE = re.compile(r"[.,;:!?¿¡\"'()\[\]{}<>\-_/\\|`~@#$%^&*+=\d]")
|
||||||
|
|
||||||
|
_ENGLISH_PATTERNS: list[str] = [
|
||||||
|
r"\bI\s+(?:am|have|had|will|would|could|should|was|were|did|do|don"
|
||||||
|
r"|can|need|like|want|think|know|love|hate|see|hear|eat|drink"
|
||||||
|
r"|go|come|take|give|make|say|tell|ask|get|put)\b",
|
||||||
|
r"\byou\s+(?:are|have|had|will|would|could|should|were|did|do|don"
|
||||||
|
r"|can|need|like|want|think|know|love|hate|see|hear)\b",
|
||||||
|
r"\b(the|to|and|in|of|a|an|is|it|for|with|on|at|by|from|or|be|this|that)\b",
|
||||||
|
r"\bsee you later\b",
|
||||||
|
r"\bI love you\b",
|
||||||
|
r"\bI want you\b",
|
||||||
|
r"\bknowledge/understanding\b",
|
||||||
|
r"\b(?:see|watch|read|write|speak|listen|learn|study|teach|explain)\s+"
|
||||||
|
r"(?:you|him|her|it|us|them|me)\b",
|
||||||
|
]
|
||||||
|
ENGLISH_RE = re.compile("|".join(_ENGLISH_PATTERNS), re.IGNORECASE)
|
||||||
|
|
||||||
|
# ── Lemma corrections ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
BUILTIN_LEMMA_CORRECTIONS: dict[str, str] = {
|
||||||
|
"comar": "comer",
|
||||||
|
"acabir": "acabar",
|
||||||
|
"deliciós": "delicioso",
|
||||||
|
"llover/llover": "llover",
|
||||||
|
}
|
||||||
|
|
||||||
|
PRIR_SURFACE_FORMS = {"pide", "pido", "piden", "pidiendo", "pedir"}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class LemmaInfo:
|
||||||
|
"""A selected lemma plus audit metadata about how it was chosen."""
|
||||||
|
|
||||||
|
lemma: str
|
||||||
|
original_lemma: str
|
||||||
|
surface: str
|
||||||
|
status: str = "ok"
|
||||||
|
|
||||||
|
# ── Known card issues (exact substring → fix → explanation) ────────
|
||||||
|
|
||||||
|
KNOWN_CARD_ISSUES: list[tuple[str, str, str]] = [
|
||||||
|
(
|
||||||
|
"Tú sabes nada",
|
||||||
|
"Tú no sabes nada.",
|
||||||
|
"Missing negation 'no' before 'sabes'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Dulce sueños",
|
||||||
|
"Dulces sueños.",
|
||||||
|
"'Dulce' should agree in number with 'sueños'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Ella soña con viajar.",
|
||||||
|
"Ella sueña con viajar.",
|
||||||
|
"'Soña' → 'sueña' (present indicative of soñar)",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Siento lastima",
|
||||||
|
"Siento lástima",
|
||||||
|
"'lastima' should be 'lástima' with accent",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Tengo uno libro",
|
||||||
|
"Tengo un libro",
|
||||||
|
"'uno' → 'un' before singular masculine noun",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Tengo veintiuno años",
|
||||||
|
"Tengo veintiún años",
|
||||||
|
"'veintiuno' → 'veintiún' before plural noun",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Dile a tu madre que feliz cumpleaños",
|
||||||
|
'Dile a tu madre: "Feliz cumpleaños."',
|
||||||
|
"Missing colon/quotes or restructured phrasing",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"La pareja está en un bote observando a los peces nadan.",
|
||||||
|
"La pareja está en un bote observando cómo nadan los peces.",
|
||||||
|
"Missing 'cómo' or subordinating conjunction",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Los perros pequeños le vuelven loca.",
|
||||||
|
"Los perros pequeños la vuelven loca.",
|
||||||
|
"'le' → 'la' (direct object, not indirect)",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
# ── Regex-based suspicious patterns ────────────────────────────────
|
||||||
|
# (pattern, description, suggested_fix_or_None)
|
||||||
|
|
||||||
|
SUSPICIOUS_PATTERNS: list[tuple[str, str, str | None]] = [
|
||||||
|
(
|
||||||
|
r"AI-generated text-to-speech",
|
||||||
|
"Metadata contamination: AI-generated TTS label in text field",
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bsabes\s+nada\b(?!.*\bno\b)",
|
||||||
|
"Missing negation: 'sabes nada' without 'no'",
|
||||||
|
"Add 'no' before 'sabes': 'No sabes nada.'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\buno\s+(?:libro|coche|casa|perro|gato|hombre|mujer|niño"
|
||||||
|
r"|amigo|día|año|mes|semana|minuto|segundo)\b",
|
||||||
|
"'uno' instead of 'un' before masculine singular noun",
|
||||||
|
"Replace 'uno' with 'un'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bveintiuno\s+(?:años|días|meses|semanas|horas|minutos|segundos)\b",
|
||||||
|
"'veintiuno' instead of 'veintiún' before plural noun",
|
||||||
|
"Replace 'veintiuno' with 'veintiún'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"observando\s+a\s+.*\bnadan\b",
|
||||||
|
"Possible missing 'cómo' in 'observando cómo ...' construction",
|
||||||
|
"Consider adding 'cómo': 'observando cómo ...'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\ble\s+vuelve\w*\s+loc[ao]s?\b",
|
||||||
|
"'le vuelve/vuelven loco/a' may need a direct object pronoun",
|
||||||
|
"Replace 'le' with lo/la/los/las as appropriate",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\ble\s+vuelven\s+loc[ao]s?\b",
|
||||||
|
"'le vuelven loco/a' may need a direct object pronoun",
|
||||||
|
"Replace 'le' with lo/la/los/las as appropriate",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bNecesito\s+[a-záéíóúüñ]{2,}\b(?!\s+(?:un|una|el|la|los"
|
||||||
|
r"|las|al|del|mi|tu|su|nuestro))\s*$",
|
||||||
|
"Missing article before the noun after 'Necesito'",
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bQu[ée]\s+es\s+la\s+respuesta\b",
|
||||||
|
"'Qué es la respuesta' → 'Cuál es la respuesta'",
|
||||||
|
"Replace 'Qué' with 'Cuál'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bpeludos?\s+animales?\b",
|
||||||
|
"'peludo(s) animal(es)' → 'animal(es) peludo(s)' (adjective placement)",
|
||||||
|
"Place adjective after noun",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bhembras\s+y\s+varones\b",
|
||||||
|
"'hembras y varones' — consider 'niños y niñas' or 'chicos y chicas'",
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bt[ií]o\s+alto\s+con\s+uniforme\b",
|
||||||
|
"'tío alto con uniforme' — check intent (Spain slang vs 'hombre')",
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\brepresentante\s+adulto\b",
|
||||||
|
"'representante adulto' → 'adulto responsable'",
|
||||||
|
"Consider 'adulto responsable'",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
r"\bpropiedad\s+a\s+este\s+representante\b",
|
||||||
|
"'propiedad a este representante' — unusual phrasing",
|
||||||
|
"Consider revising wording",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
CONTENT_POS = frozenset({"NOUN", "VERB", "ADJ", "ADV"})
|
||||||
|
FUNCTION_POS = frozenset(
|
||||||
|
{"PRON", "ADP", "CCONJ", "SCONJ", "DET", "AUX", "PART", "INTJ"}
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── Text cleaning ──────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _clean_single_field_part(text: str) -> str:
|
||||||
|
"""Clean one segment of an Anki field (no English-gloss filtering)."""
|
||||||
|
text = BR_RE.sub(" ", text)
|
||||||
|
text = HTML_TAG_RE.sub("", text)
|
||||||
|
text = html.unescape(text)
|
||||||
|
text = URL_RE.sub("", text)
|
||||||
|
text = EMAIL_RE.sub("", text)
|
||||||
|
return text.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _looks_like_english_gloss(text: str) -> bool:
|
||||||
|
"""Heuristic: does *text* read like an English gloss/translation?"""
|
||||||
|
stripped = text.strip().strip("()[]()【】")
|
||||||
|
if not stripped:
|
||||||
|
return True
|
||||||
|
return bool(ENGLISH_RE.search(stripped))
|
||||||
|
|
||||||
|
|
||||||
|
def clean_anki_field_text(text: str) -> str:
|
||||||
|
"""Clean Anki field text for Spanish NLP.
|
||||||
|
|
||||||
|
* Removes ``[sound:...]`` and ``[media:...]`` markers.
|
||||||
|
* Splits on ``<br><br>`` to separate Spanish from English glosses.
|
||||||
|
* Strips HTML tags and unescapes entities.
|
||||||
|
* Removes URLs, emails.
|
||||||
|
* Drops segments that look like English translations.
|
||||||
|
* Normalises whitespace.
|
||||||
|
* Preserves Spanish accents and meaningful content.
|
||||||
|
"""
|
||||||
|
if not text:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
text = SOUND_RE.sub("", text)
|
||||||
|
text = MEDIA_RE.sub("", text)
|
||||||
|
|
||||||
|
# Split on double <br> to isolate Spanish from glosses
|
||||||
|
parts = DOUBLE_BR_RE.split(text)
|
||||||
|
cleaned_parts: list[str] = []
|
||||||
|
for i, part in enumerate(parts):
|
||||||
|
cleaned = _clean_single_field_part(part)
|
||||||
|
if not cleaned:
|
||||||
|
continue
|
||||||
|
if i > 0 and _looks_like_english_gloss(cleaned):
|
||||||
|
continue
|
||||||
|
cleaned_parts.append(cleaned)
|
||||||
|
|
||||||
|
result = " ".join(cleaned_parts)
|
||||||
|
result = MULTI_WS_RE.sub(" ", result).strip()
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
# ── Safe lemmatisation ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def load_lemma_corrections(path: str) -> dict[str, str]:
|
||||||
|
"""Load a ``bad<TAB>good`` TSV of manual lemma fixes."""
|
||||||
|
corrections: dict[str, str] = {}
|
||||||
|
with open(path, "r", encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
line = line.strip()
|
||||||
|
if not line or line.startswith("#"):
|
||||||
|
continue
|
||||||
|
parts = line.split("\t")
|
||||||
|
if len(parts) >= 2:
|
||||||
|
bad = parts[0].strip().lower()
|
||||||
|
good = parts[1].strip().lower()
|
||||||
|
if bad and good:
|
||||||
|
corrections[bad] = good
|
||||||
|
return corrections
|
||||||
|
|
||||||
|
|
||||||
|
def load_bad_lemmas(path: str) -> set[str]:
|
||||||
|
"""Load a list of lemmas (one per line) to skip and report."""
|
||||||
|
bad: set[str] = set()
|
||||||
|
with open(path, "r", encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
lemma = line.strip()
|
||||||
|
if lemma and not lemma.startswith("#"):
|
||||||
|
bad.add(lemma.lower())
|
||||||
|
return bad
|
||||||
|
|
||||||
|
|
||||||
|
def safe_spanish_lemma(
|
||||||
|
token,
|
||||||
|
extra_corrections: dict[str, str] | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Return only the selected lemma for callers that do not need metadata."""
|
||||||
|
return safe_spanish_lemma_info(token, extra_corrections=extra_corrections).lemma
|
||||||
|
|
||||||
|
|
||||||
|
def safe_spanish_lemma_info(
|
||||||
|
token,
|
||||||
|
extra_corrections: dict[str, str] | None = None,
|
||||||
|
) -> LemmaInfo:
|
||||||
|
"""Return a safe Spanish lemma for a spaCy token.
|
||||||
|
|
||||||
|
Falls back to ``token.text.lower()`` when the spaCy lemma is empty,
|
||||||
|
multi-word, contains punctuation, or looks non-Spanish. Applies a
|
||||||
|
built-in and optionally an external correction map, and records why
|
||||||
|
fallback/correction happened.
|
||||||
|
"""
|
||||||
|
manual = dict(BUILTIN_LEMMA_CORRECTIONS)
|
||||||
|
if extra_corrections:
|
||||||
|
manual.update({k.lower(): v.lower() for k, v in extra_corrections.items()})
|
||||||
|
|
||||||
|
lemma = (token.lemma_ or "").strip().lower()
|
||||||
|
surface = (token.text or "").strip().lower()
|
||||||
|
|
||||||
|
# Manual correction map wins
|
||||||
|
if lemma in manual:
|
||||||
|
return LemmaInfo(
|
||||||
|
manual[lemma],
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status=f"corrected:{lemma}->{manual[lemma]}",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Empty or one-char fallback
|
||||||
|
if not lemma or len(lemma) <= 1:
|
||||||
|
return LemmaInfo(
|
||||||
|
surface or lemma,
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="fallback_empty_or_short_lemma",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Multi-word lemma → suspicious spaCy output
|
||||||
|
if " " in lemma:
|
||||||
|
return LemmaInfo(
|
||||||
|
surface,
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="fallback_multi_word_lemma",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Punctuation or digits in lemma → fall back
|
||||||
|
if LEMMA_PUNCT_RE.search(lemma):
|
||||||
|
return LemmaInfo(
|
||||||
|
surface,
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="fallback_bad_lemma_characters",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Special case: "prir" might be spaCy failing on pedir forms
|
||||||
|
if lemma == "prir":
|
||||||
|
if surface in PRIR_SURFACE_FORMS:
|
||||||
|
return LemmaInfo(
|
||||||
|
"pedir",
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="corrected:prir->pedir",
|
||||||
|
)
|
||||||
|
# otherwise return surface so it shows up in debug as suspicious
|
||||||
|
return LemmaInfo(
|
||||||
|
surface,
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="suspicious_prir",
|
||||||
|
)
|
||||||
|
|
||||||
|
if not SPANISH_WORD_RE.match(lemma):
|
||||||
|
return LemmaInfo(
|
||||||
|
surface,
|
||||||
|
original_lemma=lemma,
|
||||||
|
surface=surface,
|
||||||
|
status="fallback_non_spanish_lemma",
|
||||||
|
)
|
||||||
|
|
||||||
|
return LemmaInfo(lemma, original_lemma=lemma, surface=surface)
|
||||||
|
|
||||||
|
|
||||||
|
# ── Token filters ──────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def spanish_content_filter(
|
||||||
|
token,
|
||||||
|
include_proper_nouns: bool = False,
|
||||||
|
) -> bool:
|
||||||
|
"""Return True for Spanish content words (NOUN, VERB, ADJ, ADV).
|
||||||
|
|
||||||
|
Excludes punctuation, digits, symbols, URLs, emails, media
|
||||||
|
filenames, and one-character junk.
|
||||||
|
"""
|
||||||
|
text = (token.text or "").strip()
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
if token.is_punct or token.is_space or token.is_digit:
|
||||||
|
return False
|
||||||
|
if token.is_currency:
|
||||||
|
return False
|
||||||
|
if token.like_url or token.like_email:
|
||||||
|
return False
|
||||||
|
if MEDIA_FILENAME_RE.search(text):
|
||||||
|
return False
|
||||||
|
if BAD_TOKEN_TEXT_RE.search(text):
|
||||||
|
return False
|
||||||
|
if len(text) <= 1 and text not in ("a", "y", "e", "o", "u", "él"):
|
||||||
|
return False
|
||||||
|
pos = getattr(token, "pos_", "") or ""
|
||||||
|
if pos in CONTENT_POS:
|
||||||
|
return True
|
||||||
|
if include_proper_nouns and pos == "PROPN":
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def spanish_function_word_filter(token) -> bool:
|
||||||
|
"""Return True for Spanish function words (PRON, ADP, CCONJ, …)."""
|
||||||
|
text = (token.text or "").strip()
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
if token.is_punct or token.is_space or token.like_url or token.like_email:
|
||||||
|
return False
|
||||||
|
if MEDIA_FILENAME_RE.search(text) or BAD_TOKEN_TEXT_RE.search(text):
|
||||||
|
return False
|
||||||
|
pos = getattr(token, "pos_", "") or ""
|
||||||
|
return pos in FUNCTION_POS
|
||||||
|
|
||||||
|
|
||||||
|
# ── Suspicious-card detection ──────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def check_card_for_issues(text: str, cleaned: str) -> list[dict[str, Any]]:
|
||||||
|
"""Check a single card's text for suspicious patterns.
|
||||||
|
|
||||||
|
Returns a list of issue dicts (normally 0 or 1 per card, but one
|
||||||
|
card can match multiple patterns).
|
||||||
|
"""
|
||||||
|
issues: list[dict[str, Any]] = []
|
||||||
|
seen_reasons: set[str] = set()
|
||||||
|
|
||||||
|
for bad_text, fix, explanation in KNOWN_CARD_ISSUES:
|
||||||
|
if bad_text in text:
|
||||||
|
if explanation not in seen_reasons:
|
||||||
|
seen_reasons.add(explanation)
|
||||||
|
issues.append({
|
||||||
|
"reason": explanation,
|
||||||
|
"suggested_fix": fix,
|
||||||
|
})
|
||||||
|
|
||||||
|
for pattern, description, fix in SUSPICIOUS_PATTERNS:
|
||||||
|
if re.search(pattern, text, re.IGNORECASE):
|
||||||
|
if description not in seen_reasons:
|
||||||
|
seen_reasons.add(description)
|
||||||
|
issues.append({
|
||||||
|
"reason": description,
|
||||||
|
"suggested_fix": fix or "",
|
||||||
|
})
|
||||||
|
|
||||||
|
# Check for English after <br><br>
|
||||||
|
parts = DOUBLE_BR_RE.split(text)
|
||||||
|
for part in parts[1:]:
|
||||||
|
cleaned_part = _clean_single_field_part(part)
|
||||||
|
if _looks_like_english_gloss(cleaned_part):
|
||||||
|
desc = "English gloss after <br><br>"
|
||||||
|
if desc not in seen_reasons:
|
||||||
|
seen_reasons.add(desc)
|
||||||
|
issues.append({"reason": desc, "suggested_fix": ""})
|
||||||
|
|
||||||
|
# Check for missing final punctuation in full-sentence cards
|
||||||
|
stripped_text = text.rstrip()
|
||||||
|
if (
|
||||||
|
stripped_text
|
||||||
|
and not stripped_text.endswith((".", "!", "?", "…", '"', "'"))
|
||||||
|
and stripped_text[0].isupper()
|
||||||
|
and len(stripped_text.split()) >= 3
|
||||||
|
):
|
||||||
|
desc = "Missing final punctuation for full-sentence card"
|
||||||
|
if desc not in seen_reasons:
|
||||||
|
seen_reasons.add(desc)
|
||||||
|
issues.append({"reason": desc, "suggested_fix": ""})
|
||||||
|
|
||||||
|
return issues
|
||||||
|
|
||||||
|
|
||||||
|
def lint_anki_cards(
|
||||||
|
records: list[dict[str, Any]],
|
||||||
|
verbose: bool = False,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Run all suspicious-card checks across parsed Anki records.
|
||||||
|
|
||||||
|
When *verbose* is True every record is returned, not just those
|
||||||
|
with issues.
|
||||||
|
|
||||||
|
Returns a list of issue dicts with keys:
|
||||||
|
``source_line``, ``original_text``, ``cleaned_text``,
|
||||||
|
``reason``, ``suggested_fix``.
|
||||||
|
"""
|
||||||
|
results: list[dict[str, Any]] = []
|
||||||
|
for record in records:
|
||||||
|
text = record.get("text", "")
|
||||||
|
if not text:
|
||||||
|
if verbose:
|
||||||
|
results.append({
|
||||||
|
"source_line": record.get("line_number", ""),
|
||||||
|
"original_text": record.get("raw_line", ""),
|
||||||
|
"cleaned_text": "",
|
||||||
|
"reason": "empty field",
|
||||||
|
"suggested_fix": "",
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
cleaned = clean_anki_field_text(text)
|
||||||
|
issues = check_card_for_issues(text, cleaned)
|
||||||
|
if verbose or issues:
|
||||||
|
if issues:
|
||||||
|
combined_reason = "; ".join(
|
||||||
|
sorted({i["reason"] for i in issues})
|
||||||
|
)
|
||||||
|
suggested = ""
|
||||||
|
for i in issues:
|
||||||
|
if i.get("suggested_fix"):
|
||||||
|
suggested = i["suggested_fix"]
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
combined_reason = ""
|
||||||
|
suggested = ""
|
||||||
|
results.append({
|
||||||
|
"source_line": record.get("line_number", ""),
|
||||||
|
"original_text": record.get("raw_line", text),
|
||||||
|
"cleaned_text": cleaned,
|
||||||
|
"reason": combined_reason,
|
||||||
|
"suggested_fix": suggested,
|
||||||
|
})
|
||||||
|
|
||||||
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
# ── Detailed extraction helpers ─────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def extract_detailed_counts(
|
||||||
|
texts: list[dict[str, Any]],
|
||||||
|
nlp,
|
||||||
|
token_filter: Callable,
|
||||||
|
lemma_fn: Callable,
|
||||||
|
max_examples: int = 3,
|
||||||
|
clean: bool = True,
|
||||||
|
) -> dict[str, dict[str, Any]]:
|
||||||
|
"""Build detailed per-lemma statistics from a list of ``{text, line_number}`` records.
|
||||||
|
|
||||||
|
When *clean* is False the raw text is passed to the NLP pipeline
|
||||||
|
without going through ``clean_anki_field_text``.
|
||||||
|
|
||||||
|
Returns ::
|
||||||
|
|
||||||
|
{lemma: {"count": int,
|
||||||
|
"pos_counts": {POS: int, …},
|
||||||
|
"surface_forms": [str, …],
|
||||||
|
"example_sentences": [str, …],
|
||||||
|
"source_lines": [str, …]}}
|
||||||
|
"""
|
||||||
|
stats: dict[str, dict[str, Any]] = {}
|
||||||
|
|
||||||
|
for record in texts:
|
||||||
|
raw = record.get("text", "")
|
||||||
|
line_number = record.get("line_number", 0)
|
||||||
|
if not raw:
|
||||||
|
continue
|
||||||
|
txt = clean_anki_field_text(raw) if clean else raw.strip()
|
||||||
|
if not txt:
|
||||||
|
continue
|
||||||
|
|
||||||
|
doc = nlp(txt)
|
||||||
|
for token in doc:
|
||||||
|
if not token_filter(token):
|
||||||
|
continue
|
||||||
|
lemma_value = lemma_fn(token)
|
||||||
|
if lemma_value is None:
|
||||||
|
continue
|
||||||
|
if isinstance(lemma_value, LemmaInfo):
|
||||||
|
lemma = lemma_value.lemma
|
||||||
|
original_lemma = lemma_value.original_lemma
|
||||||
|
lemma_status = lemma_value.status
|
||||||
|
else:
|
||||||
|
lemma = str(lemma_value)
|
||||||
|
original_lemma = (getattr(token, "lemma_", "") or "").strip().lower()
|
||||||
|
lemma_status = "ok"
|
||||||
|
if not lemma:
|
||||||
|
continue
|
||||||
|
|
||||||
|
if lemma not in stats:
|
||||||
|
stats[lemma] = {
|
||||||
|
"count": 0,
|
||||||
|
"pos_counts": defaultdict(int),
|
||||||
|
"surface_forms": set(),
|
||||||
|
"original_lemmas": set(),
|
||||||
|
"lemma_statuses": defaultdict(int),
|
||||||
|
"example_sentences": [],
|
||||||
|
"source_lines": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
s = stats[lemma]
|
||||||
|
s["count"] += 1
|
||||||
|
s["pos_counts"][token.pos_] += 1
|
||||||
|
s["surface_forms"].add(token.text.lower())
|
||||||
|
if original_lemma:
|
||||||
|
s["original_lemmas"].add(original_lemma)
|
||||||
|
s["lemma_statuses"][lemma_status] += 1
|
||||||
|
if len(s["example_sentences"]) < max_examples:
|
||||||
|
sent_text = txt
|
||||||
|
if sent_text not in s["example_sentences"]:
|
||||||
|
s["example_sentences"].append(sent_text)
|
||||||
|
if line_number:
|
||||||
|
s["source_lines"].append(str(line_number))
|
||||||
|
|
||||||
|
# Convert defaultdicts/sets to plain types for serialisation
|
||||||
|
for lemma, s in stats.items():
|
||||||
|
s["pos_counts"] = dict(s["pos_counts"])
|
||||||
|
s["surface_forms"] = sorted(s["surface_forms"])
|
||||||
|
s["original_lemmas"] = sorted(s.get("original_lemmas", []))
|
||||||
|
s["lemma_statuses"] = dict(s.get("lemma_statuses", {}))
|
||||||
|
s["source_lines"] = list(dict.fromkeys(s["source_lines"])) # dedup, preserve order
|
||||||
|
|
||||||
|
return stats
|
||||||
|
|
||||||
|
|
||||||
|
# ── Output helpers ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def write_debug_tsv(
|
||||||
|
stats: dict[str, dict[str, Any]],
|
||||||
|
path: str,
|
||||||
|
bad_lemmas: set[str] | None = None,
|
||||||
|
min_freq: int = 0,
|
||||||
|
) -> None:
|
||||||
|
"""Write a debug TSV with columns::
|
||||||
|
|
||||||
|
lemma count pos_counts top_surface_forms example_sentences source_lines
|
||||||
|
|
||||||
|
Appends a special line per lemma in *bad_lemmas* with an
|
||||||
|
``(EXCLUDED)`` marker. When *min_freq* > 0 only lemmas with
|
||||||
|
count >= min_freq are included.
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
|
||||||
|
|
||||||
|
bad = bad_lemmas or set()
|
||||||
|
|
||||||
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
|
fh.write(
|
||||||
|
"lemma\tcount\tpos_counts\ttop_surface_forms"
|
||||||
|
"\texample_sentences\tsource_lines\toriginal_lemmas"
|
||||||
|
"\tlemma_statuses\tstatus\n"
|
||||||
|
)
|
||||||
|
sorted_lemmas = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
|
||||||
|
for lemma, s in sorted_lemmas:
|
||||||
|
if min_freq > 0 and s["count"] < min_freq:
|
||||||
|
continue
|
||||||
|
pos_counts_str = ", ".join(
|
||||||
|
f"{pos}:{cnt}"
|
||||||
|
for pos, cnt in sorted(
|
||||||
|
s["pos_counts"].items(), key=lambda x: -x[1]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
surfaces_str = ", ".join(s["surface_forms"])
|
||||||
|
examples_str = "; ".join(s["example_sentences"])
|
||||||
|
lines_str = ", ".join(s["source_lines"])
|
||||||
|
original_lemmas_str = ", ".join(s.get("original_lemmas", []))
|
||||||
|
lemma_statuses_str = ", ".join(
|
||||||
|
f"{status}:{cnt}"
|
||||||
|
for status, cnt in sorted(
|
||||||
|
s.get("lemma_statuses", {}).items(),
|
||||||
|
key=lambda x: (-x[1], x[0]),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
status = "EXCLUDED" if lemma in bad else ""
|
||||||
|
fh.write(
|
||||||
|
f"{lemma}\t{s['count']}\t{pos_counts_str}\t{surfaces_str}"
|
||||||
|
f"\t{examples_str}\t{lines_str}\t{original_lemmas_str}"
|
||||||
|
f"\t{lemma_statuses_str}\t{status}\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_suspicious_tokens(
|
||||||
|
stats: dict[str, dict[str, Any]],
|
||||||
|
path: str,
|
||||||
|
bad_lemmas: set[str] | None = None,
|
||||||
|
) -> None:
|
||||||
|
"""Write a file of suspicious lemmas (multi-word, punctuation, etc.)
|
||||||
|
|
||||||
|
Any lemma that was corrected, fell back from a suspicious spaCy lemma,
|
||||||
|
or appears in the user blocklist is listed here.
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
|
||||||
|
bad = bad_lemmas or set()
|
||||||
|
|
||||||
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
|
fh.write("lemma\tcount\tsurface_forms\toriginal_lemmas\tnote\n")
|
||||||
|
fh.write(
|
||||||
|
"# Lemmas that required correction or fallback are listed here\n"
|
||||||
|
)
|
||||||
|
for lemma, s in sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0])):
|
||||||
|
statuses = {
|
||||||
|
status: count
|
||||||
|
for status, count in s.get("lemma_statuses", {}).items()
|
||||||
|
if status != "ok"
|
||||||
|
}
|
||||||
|
if lemma in bad:
|
||||||
|
statuses["blocked_bad_lemma"] = s["count"]
|
||||||
|
if not statuses:
|
||||||
|
continue
|
||||||
|
surfaces_str = ", ".join(s["surface_forms"])
|
||||||
|
originals_str = ", ".join(s.get("original_lemmas", []))
|
||||||
|
note = ", ".join(
|
||||||
|
f"{status}:{count}"
|
||||||
|
for status, count in sorted(statuses.items())
|
||||||
|
)
|
||||||
|
fh.write(
|
||||||
|
f"{lemma}\t{s['count']}\t{surfaces_str}"
|
||||||
|
f"\t{originals_str}\t{note}\n"
|
||||||
|
)
|
||||||
+324
-1
@@ -5,14 +5,29 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
from collections import Counter
|
from collections import Counter
|
||||||
from typing import Callable
|
|
||||||
|
|
||||||
import regex as re
|
import regex as re
|
||||||
|
from typing import Any, Callable
|
||||||
|
|
||||||
from .ankiconnect import anki_request
|
from .ankiconnect import anki_request
|
||||||
|
from .anki_tsv import parse_anki_tsv
|
||||||
from .config import Config
|
from .config import Config
|
||||||
|
from .spanish import (
|
||||||
|
clean_anki_field_text,
|
||||||
|
extract_detailed_counts,
|
||||||
|
lint_anki_cards,
|
||||||
|
load_bad_lemmas,
|
||||||
|
load_lemma_corrections,
|
||||||
|
safe_spanish_lemma_info,
|
||||||
|
spanish_content_filter,
|
||||||
|
spanish_function_word_filter,
|
||||||
|
write_debug_tsv,
|
||||||
|
write_suspicious_tokens,
|
||||||
|
)
|
||||||
from .text import extract_first_visible_line, extract_visible_text, normalize_word_key
|
from .text import extract_first_visible_line, extract_visible_text, normalize_word_key
|
||||||
|
|
||||||
|
ACCENT_MAP = str.maketrans("áéíóúüñÁÉÍÓÚÜÑ", "aeiouunAEIOUUN")
|
||||||
|
|
||||||
JAPANESE_CHAR_RE = re.compile(r"[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}ー]+")
|
JAPANESE_CHAR_RE = re.compile(r"[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}ー]+")
|
||||||
JAPANESE_PARTICLES = {
|
JAPANESE_PARTICLES = {
|
||||||
"は", "が", "を", "に", "へ", "で", "と", "や", "も", "から", "まで", "より", "ば", "なら",
|
"は", "が", "を", "に", "へ", "で", "と", "や", "も", "から", "まで", "より", "ば", "なら",
|
||||||
@@ -202,3 +217,311 @@ def extract_words(
|
|||||||
counter = extract_counts(notes, field_name, nlp, profile["token_filter"], profile["output_format"], full_field)
|
counter = extract_counts(notes, field_name, nlp, profile["token_filter"], profile["output_format"], full_field)
|
||||||
written = write_counts(counter, out_path, min_freq)
|
written = write_counts(counter, out_path, min_freq)
|
||||||
return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path}
|
return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path}
|
||||||
|
|
||||||
|
|
||||||
|
def _write_content_words(
|
||||||
|
stats: dict[str, dict[str, Any]],
|
||||||
|
path: str,
|
||||||
|
min_freq: int,
|
||||||
|
bad_lemmas: set[str] | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Write a sorted ``lemma count`` file from detailed stats."""
|
||||||
|
bad = bad_lemmas or set()
|
||||||
|
items = [
|
||||||
|
(lemma, data["count"])
|
||||||
|
for lemma, data in stats.items()
|
||||||
|
if data["count"] >= min_freq and lemma not in bad
|
||||||
|
]
|
||||||
|
items.sort(key=lambda x: (-x[1], x[0]))
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
for word, freq in items:
|
||||||
|
f.write(f"{word} {freq}\n")
|
||||||
|
return len(items)
|
||||||
|
|
||||||
|
|
||||||
|
def _write_function_words(
|
||||||
|
stats: dict[str, dict[str, Any]],
|
||||||
|
path: str,
|
||||||
|
) -> int:
|
||||||
|
"""Write a function-word frequency file from detailed stats."""
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
|
||||||
|
items = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
for word, data in items:
|
||||||
|
f.write(f"{word} {data['count']}\n")
|
||||||
|
return len(items)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_words_from_file(
|
||||||
|
config: Config,
|
||||||
|
lang: str,
|
||||||
|
input_path: str,
|
||||||
|
field_index: int = 2,
|
||||||
|
min_freq: int = 2,
|
||||||
|
outdir: str | None = None,
|
||||||
|
out: str | None = None,
|
||||||
|
include_proper_nouns: bool = False,
|
||||||
|
function_words: bool = False,
|
||||||
|
debug: str | None = None,
|
||||||
|
lemma_corrections: str | None = None,
|
||||||
|
bad_lemma_file: str | None = None,
|
||||||
|
spacy_model: str | None = None,
|
||||||
|
include_tags: bool = False,
|
||||||
|
debug_min_freq: int = 0,
|
||||||
|
no_clean: bool = False,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Extract frequent Spanish vocabulary from an Anki TSV export file.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
config:
|
||||||
|
Saiki configuration.
|
||||||
|
lang:
|
||||||
|
Language code (e.g. ``"es"``).
|
||||||
|
input_path:
|
||||||
|
Path to the Anki TSV export file.
|
||||||
|
field_index:
|
||||||
|
1-based column index of the Spanish text (default 2).
|
||||||
|
min_freq:
|
||||||
|
Minimum frequency to include in output.
|
||||||
|
outdir:
|
||||||
|
Output directory (defaults to config word output root).
|
||||||
|
out:
|
||||||
|
Output filename or path for content words.
|
||||||
|
include_proper_nouns:
|
||||||
|
Include ``PROPN`` tokens.
|
||||||
|
function_words:
|
||||||
|
Also extract a separate function-word list.
|
||||||
|
debug:
|
||||||
|
Path for the debug TSV.
|
||||||
|
lemma_corrections:
|
||||||
|
Path to a ``bad\\tgood`` TSV of lemma corrections.
|
||||||
|
bad_lemma_file:
|
||||||
|
Path to a list of lemmas to skip.
|
||||||
|
spacy_model:
|
||||||
|
SpaCy model name override.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
A dict with keys ``records``, ``stats``, ``written``, ``out``,
|
||||||
|
``debug``, ``proper_nouns``, ``function_words``, ``suspicious``.
|
||||||
|
"""
|
||||||
|
language_bucket = config.language_name(lang)
|
||||||
|
out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket)
|
||||||
|
out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}_content.txt")
|
||||||
|
model_name = spacy_model or str(config.language(lang).get("word_model"))
|
||||||
|
nlp = load_spacy_model(model_name)
|
||||||
|
|
||||||
|
corrections = None
|
||||||
|
if lemma_corrections:
|
||||||
|
corrections = load_lemma_corrections(lemma_corrections)
|
||||||
|
bad_lemmas = None
|
||||||
|
if bad_lemma_file:
|
||||||
|
bad_lemmas = load_bad_lemmas(bad_lemma_file)
|
||||||
|
|
||||||
|
records = parse_anki_tsv(
|
||||||
|
input_path,
|
||||||
|
field_index=field_index,
|
||||||
|
include_tags=include_tags,
|
||||||
|
)
|
||||||
|
if not records:
|
||||||
|
return {"records": 0, "stats": {}, "written": 0, "out": out_path}
|
||||||
|
|
||||||
|
# Build token filter
|
||||||
|
def _content_filter(token) -> bool:
|
||||||
|
return spanish_content_filter(token, include_proper_nouns=include_proper_nouns)
|
||||||
|
|
||||||
|
stats = extract_detailed_counts(
|
||||||
|
records, nlp, _content_filter,
|
||||||
|
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
|
||||||
|
clean=not no_clean,
|
||||||
|
)
|
||||||
|
|
||||||
|
written = _write_content_words(stats, out_path, min_freq, bad_lemmas=bad_lemmas)
|
||||||
|
|
||||||
|
result: dict[str, Any] = {
|
||||||
|
"records": len(records),
|
||||||
|
"stats": stats,
|
||||||
|
"written": written,
|
||||||
|
"out": out_path,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Debug output
|
||||||
|
if debug:
|
||||||
|
debug_path = os.path.expanduser(debug)
|
||||||
|
write_debug_tsv(stats, debug_path, bad_lemmas=bad_lemmas, min_freq=debug_min_freq)
|
||||||
|
result["debug"] = debug_path
|
||||||
|
|
||||||
|
# Proper nouns
|
||||||
|
if include_proper_nouns:
|
||||||
|
proper_path = os.path.join(out_dir, f"words_{lang}_proper_nouns.txt")
|
||||||
|
proper_items = [
|
||||||
|
(lemma, data["count"])
|
||||||
|
for lemma, data in stats.items()
|
||||||
|
if data["pos_counts"].get("PROPN", 0) > 0
|
||||||
|
]
|
||||||
|
proper_items.sort(key=lambda x: (-x[1], x[0]))
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(proper_path)), exist_ok=True)
|
||||||
|
with open(proper_path, "w", encoding="utf-8") as f:
|
||||||
|
for w, c in proper_items:
|
||||||
|
f.write(f"{w} {c}\n")
|
||||||
|
result["proper_nouns"] = proper_path
|
||||||
|
|
||||||
|
# Function words (separate extraction pass)
|
||||||
|
if function_words:
|
||||||
|
fw_path = os.path.join(out_dir, f"words_{lang}_function_words.txt")
|
||||||
|
fw_stats = extract_detailed_counts(
|
||||||
|
records, nlp, spanish_function_word_filter,
|
||||||
|
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
|
||||||
|
)
|
||||||
|
fw_written = _write_function_words(fw_stats, fw_path)
|
||||||
|
result["function_words"] = fw_path
|
||||||
|
result["function_words_written"] = fw_written
|
||||||
|
|
||||||
|
# Suspicious tokens
|
||||||
|
susp_path = os.path.join(out_dir, f"words_{lang}_suspicious_tokens.txt")
|
||||||
|
write_suspicious_tokens(stats, susp_path, bad_lemmas=bad_lemmas)
|
||||||
|
result["suspicious"] = susp_path
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def lint_anki_cards_main(
|
||||||
|
input_path: str,
|
||||||
|
field_index: int = 2,
|
||||||
|
output_path: str = "suspicious_cards_es.tsv",
|
||||||
|
verbose: bool = False,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Read an Anki TSV export and produce a suspicious-card report.
|
||||||
|
|
||||||
|
When *verbose* is True every record is written to the report, not
|
||||||
|
only those with issues.
|
||||||
|
|
||||||
|
Returns ``{"records": …, "issues": …, "path": output_path}``.
|
||||||
|
"""
|
||||||
|
records = parse_anki_tsv(input_path, field_index=field_index)
|
||||||
|
issues = lint_anki_cards(records, verbose=verbose)
|
||||||
|
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(output_path)) or ".", exist_ok=True)
|
||||||
|
with open(output_path, "w", encoding="utf-8") as fh:
|
||||||
|
fh.write(
|
||||||
|
"source_line\toriginal_text\tcleaned_text\treason\tsuggested_fix\n"
|
||||||
|
)
|
||||||
|
for issue in issues:
|
||||||
|
fh.write(
|
||||||
|
f"{issue['source_line']}\t"
|
||||||
|
f"{_tsv_escape(issue['original_text'])}\t"
|
||||||
|
f"{_tsv_escape(issue['cleaned_text'])}\t"
|
||||||
|
f"{_tsv_escape(issue['reason'])}\t"
|
||||||
|
f"{_tsv_escape(issue['suggested_fix'])}\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
return {"records": len(records), "issues": len(issues), "path": output_path}
|
||||||
|
|
||||||
|
|
||||||
|
def _tsv_escape(value: str) -> str:
|
||||||
|
"""Escape a value for TSV output (backslash-escape tabs and newlines)."""
|
||||||
|
return value.replace("\\", "\\\\").replace("\t", "\\t").replace("\n", "\\n")
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_word_for_comparison(word: str) -> str:
|
||||||
|
"""Normalise a word for file-comparison: lowercase, strip accents."""
|
||||||
|
return word.lower().translate(ACCENT_MAP)
|
||||||
|
|
||||||
|
|
||||||
|
def read_word_file_normalized(path: str) -> dict[str, set[str]]:
|
||||||
|
"""Read a word-frequency file returning ``{normalised: {display_forms}}``.
|
||||||
|
|
||||||
|
Also returns a flat set of normalised forms.
|
||||||
|
"""
|
||||||
|
display_map: dict[str, set[str]] = {}
|
||||||
|
with open(os.path.expanduser(path), "r", encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
stripped = line.strip()
|
||||||
|
if not stripped:
|
||||||
|
continue
|
||||||
|
word = stripped.rsplit(" ", 1)[0]
|
||||||
|
key = normalize_word_for_comparison(word)
|
||||||
|
display_map.setdefault(key, set()).add(word)
|
||||||
|
return display_map
|
||||||
|
|
||||||
|
|
||||||
|
def compare_word_lists(
|
||||||
|
deck_words_path: str,
|
||||||
|
target_words_path: str,
|
||||||
|
output_path: str | None = None,
|
||||||
|
min_frequency: int = 0,
|
||||||
|
seen_output_path: str | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Compare a deck word list against a target vocabulary list.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
deck_words_path:
|
||||||
|
Path to the deck's extracted word list (``lemma count`` format).
|
||||||
|
target_words_path:
|
||||||
|
Path to the target vocabulary list (same format).
|
||||||
|
output_path:
|
||||||
|
If given, write missing-from-deck words here.
|
||||||
|
min_frequency:
|
||||||
|
Only consider deck words with frequency >= this value.
|
||||||
|
seen_output_path:
|
||||||
|
If given, write seen-in-deck words here.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
``{"missing": [str], "seen": [str], "total_missing": int}``
|
||||||
|
"""
|
||||||
|
# Read target words
|
||||||
|
target_map = read_word_file_normalized(target_words_path)
|
||||||
|
target_keys = set(target_map.keys())
|
||||||
|
|
||||||
|
# Read deck words, optionally filtering by frequency
|
||||||
|
deck_map: dict[str, set[str]] = {}
|
||||||
|
with open(os.path.expanduser(deck_words_path), "r", encoding="utf-8") as fh:
|
||||||
|
for line in fh:
|
||||||
|
stripped = line.strip()
|
||||||
|
if not stripped:
|
||||||
|
continue
|
||||||
|
parts = stripped.rsplit(" ", 1)
|
||||||
|
word = parts[0]
|
||||||
|
freq = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else 1
|
||||||
|
if freq < min_frequency:
|
||||||
|
continue
|
||||||
|
key = normalize_word_for_comparison(word)
|
||||||
|
deck_map.setdefault(key, set()).add(word)
|
||||||
|
deck_keys = set(deck_map.keys())
|
||||||
|
|
||||||
|
missing_keys = target_keys - deck_keys
|
||||||
|
seen_keys = target_keys & deck_keys
|
||||||
|
|
||||||
|
missing_entries: list[str] = []
|
||||||
|
for key in sorted(missing_keys):
|
||||||
|
for display in sorted(target_map[key]):
|
||||||
|
missing_entries.append(display)
|
||||||
|
|
||||||
|
seen_entries: list[str] = []
|
||||||
|
for key in sorted(seen_keys):
|
||||||
|
for display in sorted(target_map[key]):
|
||||||
|
seen_entries.append(display)
|
||||||
|
|
||||||
|
if output_path:
|
||||||
|
out = os.path.expanduser(output_path)
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(out)) or ".", exist_ok=True)
|
||||||
|
with open(out, "w", encoding="utf-8") as fh:
|
||||||
|
for entry in missing_entries:
|
||||||
|
fh.write(f"{entry}\n")
|
||||||
|
|
||||||
|
if seen_output_path:
|
||||||
|
seen_out = os.path.expanduser(seen_output_path)
|
||||||
|
os.makedirs(os.path.dirname(os.path.abspath(seen_out)) or ".", exist_ok=True)
|
||||||
|
with open(seen_out, "w", encoding="utf-8") as fh:
|
||||||
|
for entry in seen_entries:
|
||||||
|
fh.write(f"{entry}\n")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"missing": missing_entries,
|
||||||
|
"seen": seen_entries,
|
||||||
|
"total_missing": len(missing_entries),
|
||||||
|
}
|
||||||
|
|||||||
+647
-1
@@ -12,7 +12,9 @@ SRC_DIR = Path(__file__).resolve().parents[1] / "src"
|
|||||||
if str(SRC_DIR) not in sys.path:
|
if str(SRC_DIR) not in sys.path:
|
||||||
sys.path.insert(0, str(SRC_DIR))
|
sys.path.insert(0, str(SRC_DIR))
|
||||||
|
|
||||||
|
from saiki.anki_tsv import parse_anki_headers, parse_anki_tsv
|
||||||
from saiki.audio import build_playlist, resolve_media_paths
|
from saiki.audio import build_playlist, resolve_media_paths
|
||||||
|
from saiki.cli import build_parser, main
|
||||||
from saiki.config import Config, DEFAULT_CONFIG, deep_merge
|
from saiki.config import Config, DEFAULT_CONFIG, deep_merge
|
||||||
from saiki.importer import (
|
from saiki.importer import (
|
||||||
PreparedTtsBackend,
|
PreparedTtsBackend,
|
||||||
@@ -24,8 +26,28 @@ from saiki.importer import (
|
|||||||
synthesize_tts_sample,
|
synthesize_tts_sample,
|
||||||
supported_tts_backends,
|
supported_tts_backends,
|
||||||
)
|
)
|
||||||
|
from saiki.spanish import (
|
||||||
|
BUILTIN_LEMMA_CORRECTIONS,
|
||||||
|
check_card_for_issues,
|
||||||
|
clean_anki_field_text,
|
||||||
|
extract_detailed_counts,
|
||||||
|
lint_anki_cards,
|
||||||
|
load_lemma_corrections,
|
||||||
|
safe_spanish_lemma,
|
||||||
|
safe_spanish_lemma_info,
|
||||||
|
spanish_content_filter,
|
||||||
|
spanish_function_word_filter,
|
||||||
|
write_debug_tsv,
|
||||||
|
write_suspicious_tokens,
|
||||||
|
)
|
||||||
from saiki.text import extract_first_visible_line, extract_visible_text
|
from saiki.text import extract_first_visible_line, extract_visible_text
|
||||||
from saiki.words import build_query_from_decks, compare_word_files, read_word_file
|
from saiki.words import (
|
||||||
|
build_query_from_decks,
|
||||||
|
compare_word_files,
|
||||||
|
compare_word_lists,
|
||||||
|
normalize_word_for_comparison,
|
||||||
|
read_word_file,
|
||||||
|
)
|
||||||
from saiki.youtube import TranscriptLine, extract_video_id, sentence_vocab, write_sentence_export
|
from saiki.youtube import TranscriptLine, extract_video_id, sentence_vocab, write_sentence_export
|
||||||
|
|
||||||
|
|
||||||
@@ -81,6 +103,630 @@ class WordsTests(unittest.TestCase):
|
|||||||
self.assertEqual(compare_word_files(source, known), ["hablar 2"])
|
self.assertEqual(compare_word_files(source, known), ["hablar 2"])
|
||||||
|
|
||||||
|
|
||||||
|
class AnkiTsvTests(unittest.TestCase):
|
||||||
|
def test_parse_headers_separator_tab(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#separator:tab\n#html:true\n#tags column:3\ncol1\tcol2\tcol3\n")
|
||||||
|
headers = parse_anki_headers(path)
|
||||||
|
self.assertEqual(headers["separator"], "\t")
|
||||||
|
self.assertTrue(headers["html"])
|
||||||
|
self.assertEqual(headers["tags_column"], "3")
|
||||||
|
|
||||||
|
def test_parse_headers_default_separator(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("# some comment\ncol1\tcol2\n")
|
||||||
|
headers = parse_anki_headers(path)
|
||||||
|
self.assertNotIn("separator", headers)
|
||||||
|
|
||||||
|
def test_parse_tsv_skips_header_lines(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#separator:tab\n#html:true\nA\tB\tC\nD\tE\tF\n")
|
||||||
|
records = parse_anki_tsv(path, field_index=2)
|
||||||
|
self.assertEqual(len(records), 2)
|
||||||
|
self.assertEqual(records[0]["text"], "B")
|
||||||
|
self.assertEqual(records[1]["text"], "E")
|
||||||
|
|
||||||
|
def test_parse_tsv_field_index_1_based(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("alpha\tbeta\tgamma\n")
|
||||||
|
records = parse_anki_tsv(path, field_index=1)
|
||||||
|
self.assertEqual(records[0]["text"], "alpha")
|
||||||
|
records = parse_anki_tsv(path, field_index=3)
|
||||||
|
self.assertEqual(records[0]["text"], "gamma")
|
||||||
|
|
||||||
|
def test_parse_tsv_default_field_2_ignores_sound_field(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#separator:tab\n#tags column:3\n")
|
||||||
|
f.write("[sound:es_001.mp3]\tYo como manzanas.\tanki-tag\n")
|
||||||
|
records = parse_anki_tsv(path)
|
||||||
|
self.assertEqual(records[0]["text"], "Yo como manzanas.")
|
||||||
|
self.assertNotIn("tags", records[0])
|
||||||
|
|
||||||
|
def test_parse_tsv_rejects_zero_field_index(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("alpha\tbeta\tgamma\n")
|
||||||
|
with self.assertRaises(ValueError):
|
||||||
|
parse_anki_tsv(path, field_index=0)
|
||||||
|
|
||||||
|
def test_parse_tsv_rejects_missing_field(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("alpha\tbeta\n")
|
||||||
|
with self.assertRaises(ValueError):
|
||||||
|
parse_anki_tsv(path, field_index=3)
|
||||||
|
|
||||||
|
def test_parse_tsv_line_numbers(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#header\nuno\tdos\nthree\tfour\n")
|
||||||
|
records = parse_anki_tsv(path, field_index=2)
|
||||||
|
self.assertEqual(records[0]["line_number"], 2)
|
||||||
|
self.assertEqual(records[1]["line_number"], 3)
|
||||||
|
|
||||||
|
def test_parse_tsv_include_tags(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#tags column:3\nx\ty\ttag_a\n")
|
||||||
|
records = parse_anki_tsv(path, field_index=2, include_tags=True)
|
||||||
|
self.assertEqual(records[0]["tags"], "tag_a")
|
||||||
|
|
||||||
|
def test_parse_tsv_without_tags(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "test.txt")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("#tags column:3\nx\ty\ttag_a\n")
|
||||||
|
records = parse_anki_tsv(path, field_index=2, include_tags=False)
|
||||||
|
self.assertNotIn("tags", records[0])
|
||||||
|
|
||||||
|
|
||||||
|
class CliValidationTests(unittest.TestCase):
|
||||||
|
"""Test CLI-level validation against the real parser/main paths."""
|
||||||
|
|
||||||
|
def test_words_accepts_lang_option_and_output_alias(self):
|
||||||
|
with patch("saiki.cli.importlib.metadata.version", return_value="0"):
|
||||||
|
parser = build_parser()
|
||||||
|
args = parser.parse_args([
|
||||||
|
"words", "--lang", "es", "--input", "deck.tsv", "--output", "out.txt"
|
||||||
|
])
|
||||||
|
self.assertEqual(args.lang_option, "es")
|
||||||
|
self.assertEqual(args.out, "out.txt")
|
||||||
|
|
||||||
|
def test_words_input_rejects_zero_field(self):
|
||||||
|
from io import StringIO
|
||||||
|
|
||||||
|
with patch("saiki.cli.importlib.metadata.version", return_value="0"):
|
||||||
|
with patch("sys.stderr", new_callable=StringIO) as stderr:
|
||||||
|
code = main(["words", "--lang", "es", "--input", "deck.tsv", "--field", "0"])
|
||||||
|
self.assertEqual(code, 1)
|
||||||
|
self.assertIn("1-based", stderr.getvalue())
|
||||||
|
|
||||||
|
def test_words_input_is_spanish_only(self):
|
||||||
|
from io import StringIO
|
||||||
|
|
||||||
|
with patch("saiki.cli.importlib.metadata.version", return_value="0"):
|
||||||
|
with patch("sys.stderr", new_callable=StringIO) as stderr:
|
||||||
|
code = main(["words", "jp", "--input", "deck.tsv", "--field", "2"])
|
||||||
|
self.assertEqual(code, 1)
|
||||||
|
self.assertIn("Spanish-only", stderr.getvalue())
|
||||||
|
|
||||||
|
|
||||||
|
class SpanishTextCleaningTests(unittest.TestCase):
|
||||||
|
def test_remove_sound_marker(self):
|
||||||
|
self.assertEqual(
|
||||||
|
clean_anki_field_text("[sound:es_001.mp3] Hola mundo"),
|
||||||
|
"Hola mundo",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_strip_html_and_unescape(self):
|
||||||
|
result = clean_anki_field_text("<span>Para el lunes...</span>")
|
||||||
|
self.assertEqual(result, "Para el lunes...")
|
||||||
|
|
||||||
|
def test_english_after_brbr_removed(self):
|
||||||
|
result = clean_anki_field_text("hasta luego<br><br>(see you later)")
|
||||||
|
self.assertEqual(result, "hasta luego")
|
||||||
|
|
||||||
|
def test_english_gloss_after_brbr_removed(self):
|
||||||
|
result = clean_anki_field_text("Te quiero.<br><br>I love you / I want you")
|
||||||
|
self.assertEqual(result, "Te quiero.")
|
||||||
|
|
||||||
|
def test_spanish_after_brbr_preserved(self):
|
||||||
|
result = clean_anki_field_text("Hola<br><br>Otra línea")
|
||||||
|
self.assertEqual(result, "Hola Otra línea")
|
||||||
|
|
||||||
|
def test_normalize_whitespace(self):
|
||||||
|
result = clean_anki_field_text(" Hola mundo ")
|
||||||
|
self.assertEqual(result, "Hola mundo")
|
||||||
|
|
||||||
|
def test_br_variants(self):
|
||||||
|
result = clean_anki_field_text("Line1<br/>Line2<br />Line3")
|
||||||
|
self.assertEqual(result, "Line1 Line2 Line3")
|
||||||
|
|
||||||
|
def test_preserves_accents(self):
|
||||||
|
result = clean_anki_field_text("camión corazón")
|
||||||
|
self.assertEqual(result, "camión corazón")
|
||||||
|
|
||||||
|
def test_cleaned_text_empty_input(self):
|
||||||
|
self.assertEqual(clean_anki_field_text(""), "")
|
||||||
|
self.assertEqual(clean_anki_field_text(None), "")
|
||||||
|
|
||||||
|
|
||||||
|
class MockToken:
|
||||||
|
"""Minimal mock of a spaCy token for lemmatisation tests."""
|
||||||
|
|
||||||
|
def __init__(self, text="", lemma="", pos="", **kwargs):
|
||||||
|
self.text = text
|
||||||
|
self.lemma_ = lemma
|
||||||
|
self.pos_ = pos
|
||||||
|
self.is_stop = kwargs.get("is_stop", False)
|
||||||
|
self.is_alpha = kwargs.get("is_alpha", True)
|
||||||
|
self.is_punct = kwargs.get("is_punct", False)
|
||||||
|
self.is_space = kwargs.get("is_space", False)
|
||||||
|
self.is_digit = kwargs.get("is_digit", False)
|
||||||
|
self.is_currency = kwargs.get("is_currency", False)
|
||||||
|
self.like_url = kwargs.get("like_url", False)
|
||||||
|
self.like_email = kwargs.get("like_email", False)
|
||||||
|
self.sent = None
|
||||||
|
|
||||||
|
|
||||||
|
class SpanishLemmaTests(unittest.TestCase):
|
||||||
|
def test_builtin_correction_comar(self):
|
||||||
|
token = MockToken(text="como", lemma="comar")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "comer")
|
||||||
|
|
||||||
|
def test_builtin_correction_acabir(self):
|
||||||
|
token = MockToken(text="acabas", lemma="acabir")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "acabar")
|
||||||
|
|
||||||
|
def test_builtin_correction_delicius(self):
|
||||||
|
token = MockToken(text="delicioso", lemma="deliciós")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "delicioso")
|
||||||
|
|
||||||
|
def test_builtin_correction_llover_slash(self):
|
||||||
|
token = MockToken(text="llueve", lemma="llover/llover")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "llover")
|
||||||
|
|
||||||
|
def test_multi_word_lemma_falls_back(self):
|
||||||
|
token = MockToken(text="ayudar", lemma="ayudar yo")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "ayudar")
|
||||||
|
|
||||||
|
def test_multi_word_lemma_lavar_el(self):
|
||||||
|
token = MockToken(text="lava", lemma="lavar él")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "lava")
|
||||||
|
|
||||||
|
def test_prir_corrected_when_source_is_pedir(self):
|
||||||
|
for form in ["pide", "pido", "piden", "pidiendo", "pedir"]:
|
||||||
|
token = MockToken(text=form, lemma="prir")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "pedir")
|
||||||
|
|
||||||
|
def test_prir_not_corrected_for_unknown_form(self):
|
||||||
|
token = MockToken(text="prir", lemma="prir")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "prir")
|
||||||
|
|
||||||
|
def test_empty_lemma_falls_back(self):
|
||||||
|
token = MockToken(text="hola", lemma="")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "hola")
|
||||||
|
|
||||||
|
def test_lemma_with_punctuation_falls_back(self):
|
||||||
|
token = MockToken(text="comiendo", lemma="comiendo,")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "comiendo")
|
||||||
|
|
||||||
|
def test_normal_lemma_passes_through(self):
|
||||||
|
token = MockToken(text="comiendo", lemma="comer")
|
||||||
|
self.assertEqual(safe_spanish_lemma(token), "comer")
|
||||||
|
|
||||||
|
def test_lemma_info_tracks_correction_reason(self):
|
||||||
|
token = MockToken(text="como", lemma="comar")
|
||||||
|
info = safe_spanish_lemma_info(token)
|
||||||
|
self.assertEqual(info.lemma, "comer")
|
||||||
|
self.assertEqual(info.original_lemma, "comar")
|
||||||
|
self.assertIn("corrected", info.status)
|
||||||
|
|
||||||
|
def test_non_spanish_lemma_falls_back(self):
|
||||||
|
token = MockToken(text="hola", lemma="hello_world")
|
||||||
|
info = safe_spanish_lemma_info(token)
|
||||||
|
self.assertEqual(info.lemma, "hola")
|
||||||
|
self.assertEqual(info.status, "fallback_bad_lemma_characters")
|
||||||
|
|
||||||
|
def test_extra_corrections_override_builtin(self):
|
||||||
|
extra = {"comar": "comprar"}
|
||||||
|
token = MockToken(text="como", lemma="comar")
|
||||||
|
self.assertEqual(
|
||||||
|
safe_spanish_lemma(token, extra_corrections=extra),
|
||||||
|
"comprar",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_load_lemma_corrections(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "corrections.tsv")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("malo\tbueno\nfeo\tbonito\n")
|
||||||
|
corrections = load_lemma_corrections(path)
|
||||||
|
self.assertEqual(corrections, {"malo": "bueno", "feo": "bonito"})
|
||||||
|
|
||||||
|
def test_load_lemma_corrections_skips_comments(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "corrections.tsv")
|
||||||
|
with open(path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("# comment\nmalo\tbueno\n")
|
||||||
|
corrections = load_lemma_corrections(path)
|
||||||
|
self.assertEqual(corrections, {"malo": "bueno"})
|
||||||
|
|
||||||
|
|
||||||
|
class SpanishFilterTests(unittest.TestCase):
|
||||||
|
def test_content_filter_keeps_noun(self):
|
||||||
|
token = MockToken(text="casa", lemma="casa", pos="NOUN")
|
||||||
|
self.assertTrue(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_keeps_verb(self):
|
||||||
|
token = MockToken(text="come", lemma="comer", pos="VERB")
|
||||||
|
self.assertTrue(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_keeps_adj(self):
|
||||||
|
token = MockToken(text="grande", lemma="grande", pos="ADJ")
|
||||||
|
self.assertTrue(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_keeps_adv(self):
|
||||||
|
token = MockToken(text="bien", lemma="bien", pos="ADV")
|
||||||
|
self.assertTrue(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_excludes_propn_by_default(self):
|
||||||
|
token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN")
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_includes_propn_with_flag(self):
|
||||||
|
token = MockToken(text="Madrid", lemma="Madrid", pos="PROPN")
|
||||||
|
self.assertTrue(
|
||||||
|
spanish_content_filter(token, include_proper_nouns=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_content_filter_excludes_punct(self):
|
||||||
|
token = MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True)
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_excludes_digit(self):
|
||||||
|
token = MockToken(text="123", lemma="123", pos="NUM", is_digit=True)
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_excludes_url(self):
|
||||||
|
token = MockToken(
|
||||||
|
text="http://x.com", lemma="http://x.com", pos="X", like_url=True
|
||||||
|
)
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_excludes_media_filename(self):
|
||||||
|
token = MockToken(text="es_001.mp3", lemma="es_001.mp3", pos="NOUN")
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_content_filter_excludes_one_char_non_vowel(self):
|
||||||
|
token = MockToken(text="x", lemma="x", pos="NOUN")
|
||||||
|
self.assertFalse(spanish_content_filter(token))
|
||||||
|
|
||||||
|
def test_function_word_filter_keeps_preposition(self):
|
||||||
|
token = MockToken(text="para", lemma="para", pos="ADP")
|
||||||
|
self.assertTrue(spanish_function_word_filter(token))
|
||||||
|
|
||||||
|
def test_function_word_filter_keeps_conjunction(self):
|
||||||
|
token = MockToken(text="y", lemma="y", pos="CCONJ")
|
||||||
|
self.assertTrue(spanish_function_word_filter(token))
|
||||||
|
|
||||||
|
def test_function_word_filter_rejects_noun(self):
|
||||||
|
token = MockToken(text="casa", lemma="casa", pos="NOUN")
|
||||||
|
self.assertFalse(spanish_function_word_filter(token))
|
||||||
|
|
||||||
|
|
||||||
|
class SpanishLintTests(unittest.TestCase):
|
||||||
|
def _check(self, text: str) -> list:
|
||||||
|
return check_card_for_issues(text, clean_anki_field_text(text))
|
||||||
|
|
||||||
|
def test_detect_tu_sabes_nada(self):
|
||||||
|
issues = self._check("Tú sabes nada")
|
||||||
|
self.assertTrue(any("Missing negation" in i["reason"] for i in issues))
|
||||||
|
|
||||||
|
def test_detect_dulce_suenos(self):
|
||||||
|
issues = self._check("Dulce sueños")
|
||||||
|
self.assertTrue(
|
||||||
|
any("agree in number" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_ella_sona(self):
|
||||||
|
issues = self._check("Ella soña con viajar.")
|
||||||
|
self.assertTrue(
|
||||||
|
any("present indicative" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_tengo_uno_libro(self):
|
||||||
|
issues = self._check("Tengo uno libro en la mesa.")
|
||||||
|
self.assertTrue(
|
||||||
|
any("instead of 'un'" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_observando_nadan(self):
|
||||||
|
issues = self._check(
|
||||||
|
"La pareja está en un bote observando a los peces nadan."
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
any("'cómo'" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_le_vuelven_loca(self):
|
||||||
|
issues = self._check("Los perros pequeños le vuelven loca.")
|
||||||
|
self.assertTrue(
|
||||||
|
any("direct object" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_ai_generated_tts(self):
|
||||||
|
issues = self._check("AI-generated text-to-speech")
|
||||||
|
self.assertTrue(
|
||||||
|
any("Metadata contamination" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_necesito_missing_article(self):
|
||||||
|
issues = self._check("Necesito jardinero")
|
||||||
|
self.assertTrue(
|
||||||
|
any("Missing article" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_detect_le_vuelven_loco(self):
|
||||||
|
issues = self._check("Estos problemas le vuelven loco.")
|
||||||
|
self.assertTrue(
|
||||||
|
any("direct object" in i["reason"] or "direct object pronoun" in i["reason"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_veintiuno_anos(self):
|
||||||
|
issues = self._check("Tengo veintiuno años de edad.")
|
||||||
|
self.assertTrue(
|
||||||
|
any("veintiún" in i["suggested_fix"] for i in issues)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_lint_anki_cards_multiple_records(self):
|
||||||
|
records = [
|
||||||
|
{"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"},
|
||||||
|
{"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"},
|
||||||
|
{"text": "Ella soña con viajar.", "line_number": 3, "raw_line": "Ella soña con viajar."},
|
||||||
|
]
|
||||||
|
results = lint_anki_cards(records)
|
||||||
|
self.assertEqual(len(results), 2)
|
||||||
|
reasons = results[0]["reason"] + results[1]["reason"]
|
||||||
|
self.assertIn("Missing negation", reasons)
|
||||||
|
self.assertIn("present indicative", reasons)
|
||||||
|
|
||||||
|
|
||||||
|
class SpanishDetailedCountsTests(unittest.TestCase):
|
||||||
|
def test_extract_detailed_counts_basic(self):
|
||||||
|
records = [
|
||||||
|
{"text": "Yo como manzanas.", "line_number": 1},
|
||||||
|
{"text": "Tú comes pan.", "line_number": 2},
|
||||||
|
]
|
||||||
|
|
||||||
|
class MockDoc:
|
||||||
|
def __init__(self, tokens):
|
||||||
|
self.tokens = tokens
|
||||||
|
|
||||||
|
def __iter__(self):
|
||||||
|
return iter(self.tokens)
|
||||||
|
|
||||||
|
class MockNLP:
|
||||||
|
def __call__(self, text):
|
||||||
|
if "como" in text:
|
||||||
|
return MockDoc([
|
||||||
|
MockToken(text="Yo", lemma="yo", pos="PRON"),
|
||||||
|
MockToken(text="como", lemma="comar", pos="VERB"),
|
||||||
|
MockToken(text="manzanas", lemma="manzana", pos="NOUN"),
|
||||||
|
MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True),
|
||||||
|
])
|
||||||
|
return MockDoc([
|
||||||
|
MockToken(text="Tú", lemma="tú", pos="PRON"),
|
||||||
|
MockToken(text="comes", lemma="comer", pos="VERB"),
|
||||||
|
MockToken(text="pan", lemma="pan", pos="NOUN"),
|
||||||
|
MockToken(text=".", lemma=".", pos="PUNCT", is_punct=True),
|
||||||
|
])
|
||||||
|
|
||||||
|
stats = extract_detailed_counts(
|
||||||
|
records,
|
||||||
|
MockNLP(),
|
||||||
|
lambda t: t.pos_ in {"NOUN", "VERB", "ADJ", "ADV"},
|
||||||
|
safe_spanish_lemma,
|
||||||
|
)
|
||||||
|
|
||||||
|
# "comar" should be corrected to "comer"
|
||||||
|
self.assertIn("comer", stats)
|
||||||
|
# "comer" should have count 2 (comes + como -> comer)
|
||||||
|
self.assertEqual(stats["comer"]["count"], 2)
|
||||||
|
self.assertIn("manzana", stats)
|
||||||
|
self.assertIn("pan", stats)
|
||||||
|
|
||||||
|
def test_debug_tsv_output(self):
|
||||||
|
stats = {
|
||||||
|
"comer": {
|
||||||
|
"count": 3,
|
||||||
|
"pos_counts": {"VERB": 3},
|
||||||
|
"surface_forms": ["como", "comes", "come"],
|
||||||
|
"example_sentences": ["Yo como.", "Tú comes."],
|
||||||
|
"source_lines": ["1", "2", "3"],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "debug.tsv")
|
||||||
|
write_debug_tsv(stats, path)
|
||||||
|
with open(path, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
self.assertIn("comer", content)
|
||||||
|
self.assertIn("VERB:3", content)
|
||||||
|
self.assertIn("como, comes, come", content)
|
||||||
|
self.assertIn("Yo como.", content)
|
||||||
|
self.assertIn("1, 2, 3", content)
|
||||||
|
|
||||||
|
def test_suspicious_tokens_output(self):
|
||||||
|
stats = {
|
||||||
|
"comer": {
|
||||||
|
"count": 3,
|
||||||
|
"pos_counts": {"VERB": 3},
|
||||||
|
"surface_forms": ["como", "comes"],
|
||||||
|
"original_lemmas": ["comar", "comer"],
|
||||||
|
"lemma_statuses": {"corrected:comar->comer": 1, "ok": 2},
|
||||||
|
"example_sentences": ["Yo como."],
|
||||||
|
"source_lines": ["1"],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "suspicious.tsv")
|
||||||
|
write_suspicious_tokens(stats, path)
|
||||||
|
with open(path, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
self.assertIn("comer", content)
|
||||||
|
self.assertIn("comar", content)
|
||||||
|
self.assertIn("corrected:comar->comer", content)
|
||||||
|
|
||||||
|
def test_debug_tsv_min_freq_filters(self):
|
||||||
|
stats = {
|
||||||
|
"comer": {"count": 5, "pos_counts": {"VERB": 5},
|
||||||
|
"surface_forms": ["come"], "example_sentences": ["Yo come."],
|
||||||
|
"source_lines": ["1"]},
|
||||||
|
"hablar": {"count": 2, "pos_counts": {"VERB": 2},
|
||||||
|
"surface_forms": ["habla"], "example_sentences": ["Él habla."],
|
||||||
|
"source_lines": ["2"]},
|
||||||
|
}
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "debug.tsv")
|
||||||
|
write_debug_tsv(stats, path, min_freq=3)
|
||||||
|
with open(path, "r", encoding="utf-8") as f:
|
||||||
|
content = f.read()
|
||||||
|
self.assertIn("comer", content)
|
||||||
|
self.assertNotIn("hablar", content)
|
||||||
|
|
||||||
|
def test_extract_detailed_counts_no_clean(self):
|
||||||
|
class MockDoc:
|
||||||
|
def __init__(self, tokens):
|
||||||
|
self.tokens = tokens
|
||||||
|
def __iter__(self):
|
||||||
|
return iter(self.tokens)
|
||||||
|
|
||||||
|
class MockNLP:
|
||||||
|
def __call__(self, text):
|
||||||
|
return MockDoc([
|
||||||
|
MockToken(text="[sound:x.mp3]", lemma="[sound:x.mp3]", pos="X"),
|
||||||
|
])
|
||||||
|
|
||||||
|
records = [{"text": "[sound:x.mp3]", "line_number": 1}]
|
||||||
|
stats = extract_detailed_counts(
|
||||||
|
records, MockNLP(),
|
||||||
|
lambda t: True,
|
||||||
|
lambda t: t.text.lower(),
|
||||||
|
clean=False,
|
||||||
|
)
|
||||||
|
# With clean=False, [sound:x.mp3] is NOT stripped
|
||||||
|
self.assertIn("[sound:x.mp3]", stats)
|
||||||
|
|
||||||
|
stats_clean = extract_detailed_counts(
|
||||||
|
records, MockNLP(),
|
||||||
|
lambda t: True,
|
||||||
|
lambda t: t.text.lower(),
|
||||||
|
clean=True,
|
||||||
|
)
|
||||||
|
# With clean=True, the sound marker is stripped → empty text → no stats
|
||||||
|
self.assertEqual(len(stats_clean), 0)
|
||||||
|
|
||||||
|
def test_lint_anki_cards_verbose_includes_all(self):
|
||||||
|
records = [
|
||||||
|
{"text": "Hola mundo", "line_number": 1, "raw_line": "Hola mundo"},
|
||||||
|
{"text": "Tú sabes nada", "line_number": 2, "raw_line": "Tú sabes nada"},
|
||||||
|
]
|
||||||
|
# Without verbose – only suspicious
|
||||||
|
results = lint_anki_cards(records, verbose=False)
|
||||||
|
self.assertEqual(len(results), 1)
|
||||||
|
|
||||||
|
# With verbose – both records
|
||||||
|
results = lint_anki_cards(records, verbose=True)
|
||||||
|
self.assertEqual(len(results), 2)
|
||||||
|
|
||||||
|
def test_lint_anki_cards_verbose_empty_field(self):
|
||||||
|
records = [
|
||||||
|
{"text": "", "line_number": 1, "raw_line": ""},
|
||||||
|
]
|
||||||
|
results = lint_anki_cards(records, verbose=True)
|
||||||
|
self.assertEqual(len(results), 1)
|
||||||
|
self.assertEqual(results[0]["reason"], "empty field")
|
||||||
|
|
||||||
|
|
||||||
|
class CompareWordListsTests(unittest.TestCase):
|
||||||
|
def test_normalize_word_for_comparison(self):
|
||||||
|
self.assertEqual(normalize_word_for_comparison("camión"), "camion")
|
||||||
|
self.assertEqual(normalize_word_for_comparison("Corazón"), "corazon")
|
||||||
|
self.assertEqual(normalize_word_for_comparison("ESPAÑOL"), "espanol")
|
||||||
|
|
||||||
|
def test_compare_word_lists_identifies_missing(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
deck = os.path.join(tmp, "deck.txt")
|
||||||
|
target = os.path.join(tmp, "target.txt")
|
||||||
|
with open(deck, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 5\nhablar 3\n")
|
||||||
|
with open(target, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 10\nvivir 8\ncorrer 6\n")
|
||||||
|
result = compare_word_lists(deck, target)
|
||||||
|
self.assertIn("vivir", result["missing"])
|
||||||
|
self.assertIn("correr", result["missing"])
|
||||||
|
self.assertNotIn("comer", result["missing"])
|
||||||
|
|
||||||
|
def test_compare_word_lists_with_min_frequency(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
deck = os.path.join(tmp, "deck.txt")
|
||||||
|
target = os.path.join(tmp, "target.txt")
|
||||||
|
with open(deck, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 1\nhablar 5\n")
|
||||||
|
with open(target, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 10\nhablar 10\n")
|
||||||
|
result = compare_word_lists(deck, target, min_frequency=3)
|
||||||
|
# "comer" has freq 1 < 3, so it's excluded from deck entirely
|
||||||
|
# meaning it should be "missing" from deck's perspective
|
||||||
|
self.assertIn("comer", result["missing"])
|
||||||
|
self.assertNotIn("hablar", result["missing"])
|
||||||
|
self.assertIn("hablar", result["seen"])
|
||||||
|
|
||||||
|
def test_compare_word_lists_writes_output(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
deck = os.path.join(tmp, "deck.txt")
|
||||||
|
target = os.path.join(tmp, "target.txt")
|
||||||
|
out = os.path.join(tmp, "output.txt")
|
||||||
|
with open(deck, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 5\n")
|
||||||
|
with open(target, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 10\nvivir 8\n")
|
||||||
|
result = compare_word_lists(deck, target, output_path=out)
|
||||||
|
self.assertEqual(result["total_missing"], 1)
|
||||||
|
with open(out, "r", encoding="utf-8") as f:
|
||||||
|
self.assertIn("vivir", f.read())
|
||||||
|
|
||||||
|
def test_compare_word_lists_writes_seen_output(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
deck = os.path.join(tmp, "deck.txt")
|
||||||
|
target = os.path.join(tmp, "target.txt")
|
||||||
|
seen_out = os.path.join(tmp, "seen.txt")
|
||||||
|
with open(deck, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 5\n")
|
||||||
|
with open(target, "w", encoding="utf-8") as f:
|
||||||
|
f.write("comer 10\nvivir 8\n")
|
||||||
|
result = compare_word_lists(deck, target, seen_output_path=seen_out)
|
||||||
|
self.assertIn("comer", result["seen"])
|
||||||
|
with open(seen_out, "r", encoding="utf-8") as f:
|
||||||
|
self.assertIn("comer", f.read())
|
||||||
|
self.assertNotIn("vivir", f.read())
|
||||||
|
|
||||||
|
|
||||||
class YoutubeTests(unittest.TestCase):
|
class YoutubeTests(unittest.TestCase):
|
||||||
def test_extract_video_id(self):
|
def test_extract_video_id(self):
|
||||||
self.assertEqual(extract_video_id("https://youtu.be/abc123"), "abc123")
|
self.assertEqual(extract_video_id("https://youtu.be/abc123"), "abc123")
|
||||||
|
|||||||
Reference in New Issue
Block a user