528 lines
19 KiB
Python
528 lines
19 KiB
Python
"""Extract and compare language-learning vocabulary."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import os
|
|
from collections import Counter
|
|
|
|
import regex as re
|
|
from typing import Any, Callable
|
|
|
|
from .ankiconnect import anki_request
|
|
from .anki_tsv import parse_anki_tsv
|
|
from .config import Config
|
|
from .spanish import (
|
|
clean_anki_field_text,
|
|
extract_detailed_counts,
|
|
lint_anki_cards,
|
|
load_bad_lemmas,
|
|
load_lemma_corrections,
|
|
safe_spanish_lemma_info,
|
|
spanish_content_filter,
|
|
spanish_function_word_filter,
|
|
write_debug_tsv,
|
|
write_suspicious_tokens,
|
|
)
|
|
from .text import extract_first_visible_line, extract_visible_text, normalize_word_key
|
|
|
|
ACCENT_MAP = str.maketrans("áéíóúüñÁÉÍÓÚÜÑ", "aeiouunAEIOUUN")
|
|
|
|
JAPANESE_CHAR_RE = re.compile(r"[\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Han}ー]+")
|
|
JAPANESE_PARTICLES = {
|
|
"は", "が", "を", "に", "へ", "で", "と", "や", "も", "から", "まで", "より", "ば", "なら",
|
|
"の", "ね", "よ", "ぞ", "ぜ", "さ", "わ", "か", "な", "って", "とき", "ってば", "けど", "けれど",
|
|
"しかし", "でも", "ながら", "ほど", "し", "もの", "こと", "ところ", "よう", "らしい", "られる",
|
|
}
|
|
JAPANESE_GRAMMAR_EXCLUDE = {
|
|
"て", "た", "ます", "れる", "てる", "ぬ", "ん", "しまう", "いる", "ない", "なる", "ある", "だ", "です",
|
|
}
|
|
JAPANESE_ALLOWED_POS = {"NOUN", "PROPN", "VERB", "ADJ"}
|
|
|
|
|
|
def setup_logging(logfile: str) -> None:
|
|
"""Configure file logging for word extraction scripts."""
|
|
os.makedirs(os.path.dirname(os.path.abspath(logfile)), exist_ok=True)
|
|
logging.basicConfig(filename=logfile, level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
|
|
|
|
|
def build_query_from_decks(decks: list[str]) -> str:
|
|
"""Build an Anki search query that matches any configured deck."""
|
|
return " OR ".join(f'deck:"{d}"' for d in decks)
|
|
|
|
|
|
def japanese_filter(token) -> bool:
|
|
"""Return whether a spaCy token is useful Japanese vocabulary.
|
|
|
|
The filter is intentionally conservative: it keeps content words and drops
|
|
common particles, helper grammar, stop words, URLs, and obvious HTML debris.
|
|
"""
|
|
text = (token.text or "").strip()
|
|
lemma = (token.lemma_ or "").strip()
|
|
if not text or not JAPANESE_CHAR_RE.fullmatch(text):
|
|
return False
|
|
if lemma in JAPANESE_GRAMMAR_EXCLUDE or text in JAPANESE_PARTICLES:
|
|
return False
|
|
if getattr(token, "pos_", None) not in JAPANESE_ALLOWED_POS:
|
|
return False
|
|
if getattr(token, "is_stop", False) or getattr(token, "like_url", False) or getattr(token, "like_email", False):
|
|
return False
|
|
if any(c in text for c in "<>=/\\:&%"):
|
|
return False
|
|
return text not in {"ruby", "rt", "div", "br", "nbsp", "href", "strong", "a"}
|
|
|
|
|
|
def spanish_filter(token) -> bool:
|
|
"""Return whether a spaCy token is useful Spanish vocabulary."""
|
|
return bool(getattr(token, "is_alpha", False)) and not bool(getattr(token, "is_stop", False))
|
|
|
|
|
|
def spanish_format(token) -> str:
|
|
"""Normalize a Spanish token to its lowercase lemma."""
|
|
return (token.lemma_ or token.text or "").lower().strip()
|
|
|
|
|
|
def japanese_format(token) -> str:
|
|
"""Format a Japanese token as lemma plus surface form when they differ."""
|
|
lemma = (token.lemma_ or "").strip()
|
|
surface = (token.text or "").strip()
|
|
if lemma and surface and lemma != surface:
|
|
return f"{lemma} ({surface})"
|
|
return lemma or surface
|
|
|
|
|
|
LANGUAGE_PROFILES = {
|
|
"spanish": {"token_filter": spanish_filter, "output_format": spanish_format},
|
|
"japanese": {"token_filter": japanese_filter, "output_format": japanese_format},
|
|
}
|
|
|
|
|
|
def load_spacy_model(model_name: str):
|
|
"""Load a spaCy model with installation-oriented error messages."""
|
|
try:
|
|
import spacy # type: ignore
|
|
except Exception as e:
|
|
raise RuntimeError("Failed to import spaCy. Use a Python version supported by spaCy.") from e
|
|
try:
|
|
return spacy.load(model_name)
|
|
except Exception as e:
|
|
raise RuntimeError(f"Failed to load spaCy model '{model_name}'. Try: python -m spacy download {model_name}") from e
|
|
|
|
|
|
def get_notes(query: str, config: Config, request: Callable = anki_request) -> list[dict]:
|
|
"""Fetch Anki note details matching a search query."""
|
|
note_ids = request("findNotes", url=config.anki_connect_url, query=query) or []
|
|
if not note_ids:
|
|
return []
|
|
return request("notesInfo", url=config.anki_connect_url, notes=note_ids) or []
|
|
|
|
|
|
def extract_counts(
|
|
notes: list[dict],
|
|
field_name: str,
|
|
nlp,
|
|
token_filter: Callable,
|
|
output_format: Callable,
|
|
use_full_field: bool,
|
|
) -> Counter:
|
|
"""Count formatted vocabulary items across Anki notes."""
|
|
counter: Counter = Counter()
|
|
for note in notes:
|
|
fields = note.get("fields", {}) or {}
|
|
raw_val = (fields.get(field_name, {}) or {}).get("value", "") or ""
|
|
text = extract_visible_text(raw_val) if use_full_field else extract_first_visible_line(raw_val)
|
|
if not text:
|
|
continue
|
|
for token in nlp(text):
|
|
if token_filter(token):
|
|
key = output_format(token)
|
|
if key:
|
|
counter[key] += 1
|
|
return counter
|
|
|
|
|
|
def write_counts(counter: Counter, out_path: str, min_freq: int) -> int:
|
|
"""Write a sorted ``word frequency`` list and return the number of rows."""
|
|
items = [(w, c) for (w, c) in counter.items() if c >= min_freq]
|
|
items.sort(key=lambda x: (-x[1], x[0]))
|
|
os.makedirs(os.path.dirname(os.path.abspath(out_path)), exist_ok=True)
|
|
with open(out_path, "w", encoding="utf-8") as f:
|
|
for word, freq in items:
|
|
f.write(f"{word} {freq}\n")
|
|
return len(items)
|
|
|
|
|
|
def read_word_file(path: str) -> set[str]:
|
|
"""Read a ``word frequency`` file into normalized word keys."""
|
|
words: set[str] = set()
|
|
with open(os.path.expanduser(path), "r", encoding="utf-8") as fh:
|
|
for line in fh:
|
|
stripped = line.strip()
|
|
if not stripped:
|
|
continue
|
|
word = stripped.rsplit(" ", 1)[0]
|
|
words.add(normalize_word_key(word))
|
|
return words
|
|
|
|
|
|
def compare_word_files(source_path: str, known_path: str) -> list[str]:
|
|
"""Return source rows whose normalized word is not in the known list."""
|
|
known = read_word_file(known_path)
|
|
new_words: list[str] = []
|
|
with open(os.path.expanduser(source_path), "r", encoding="utf-8") as fh:
|
|
for line in fh:
|
|
stripped = line.strip()
|
|
if not stripped:
|
|
continue
|
|
word = stripped.rsplit(" ", 1)[0]
|
|
if normalize_word_key(word) not in known:
|
|
new_words.append(stripped)
|
|
return new_words
|
|
|
|
|
|
def extract_words(
|
|
config: Config,
|
|
lang: str,
|
|
query: str | None = None,
|
|
decks: list[str] | None = None,
|
|
field: str | None = None,
|
|
min_freq: int = 2,
|
|
outdir: str | None = None,
|
|
out: str | None = None,
|
|
full_field: bool = False,
|
|
spacy_model: str | None = None,
|
|
request: Callable = anki_request,
|
|
) -> dict[str, object]:
|
|
"""Extract frequent vocabulary from configured Anki cards.
|
|
|
|
The function accepts explicit query/deck/field overrides for CLI use, but
|
|
defaults to the selected language config. Its dictionary return value keeps
|
|
the CLI output simple and gives tests stable fields to assert against.
|
|
"""
|
|
language_bucket = config.language_name(lang)
|
|
profile = LANGUAGE_PROFILES[language_bucket]
|
|
search_query = query or build_query_from_decks(decks or config.decks_for(lang))
|
|
out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket)
|
|
out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}.txt")
|
|
model_name = spacy_model or str(config.language(lang).get("word_model"))
|
|
nlp = load_spacy_model(model_name)
|
|
notes = get_notes(search_query, config, request=request)
|
|
if notes:
|
|
fields0 = (notes[0].get("fields", {}) or {})
|
|
field_name = field or config.field_for(lang)
|
|
if field_name not in fields0:
|
|
raise RuntimeError(f"Field '{field_name}' not found. Available fields: {list(fields0.keys())}")
|
|
else:
|
|
field_name = field or config.field_for(lang)
|
|
counter = extract_counts(notes, field_name, nlp, profile["token_filter"], profile["output_format"], full_field)
|
|
written = write_counts(counter, out_path, min_freq)
|
|
return {"query": search_query, "notes": len(notes), "unique": len(counter), "written": written, "out": out_path}
|
|
|
|
|
|
def _write_content_words(
|
|
stats: dict[str, dict[str, Any]],
|
|
path: str,
|
|
min_freq: int,
|
|
bad_lemmas: set[str] | None = None,
|
|
) -> int:
|
|
"""Write a sorted ``lemma count`` file from detailed stats."""
|
|
bad = bad_lemmas or set()
|
|
items = [
|
|
(lemma, data["count"])
|
|
for lemma, data in stats.items()
|
|
if data["count"] >= min_freq and lemma not in bad
|
|
]
|
|
items.sort(key=lambda x: (-x[1], x[0]))
|
|
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
for word, freq in items:
|
|
f.write(f"{word} {freq}\n")
|
|
return len(items)
|
|
|
|
|
|
def _write_function_words(
|
|
stats: dict[str, dict[str, Any]],
|
|
path: str,
|
|
) -> int:
|
|
"""Write a function-word frequency file from detailed stats."""
|
|
os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
|
|
items = sorted(stats.items(), key=lambda x: (-x[1]["count"], x[0]))
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
for word, data in items:
|
|
f.write(f"{word} {data['count']}\n")
|
|
return len(items)
|
|
|
|
|
|
def extract_words_from_file(
|
|
config: Config,
|
|
lang: str,
|
|
input_path: str,
|
|
field_index: int = 2,
|
|
min_freq: int = 2,
|
|
outdir: str | None = None,
|
|
out: str | None = None,
|
|
include_proper_nouns: bool = False,
|
|
function_words: bool = False,
|
|
debug: str | None = None,
|
|
lemma_corrections: str | None = None,
|
|
bad_lemma_file: str | None = None,
|
|
spacy_model: str | None = None,
|
|
include_tags: bool = False,
|
|
debug_min_freq: int = 0,
|
|
no_clean: bool = False,
|
|
) -> dict[str, Any]:
|
|
"""Extract frequent Spanish vocabulary from an Anki TSV export file.
|
|
|
|
Parameters
|
|
----------
|
|
config:
|
|
Saiki configuration.
|
|
lang:
|
|
Language code (e.g. ``"es"``).
|
|
input_path:
|
|
Path to the Anki TSV export file.
|
|
field_index:
|
|
1-based column index of the Spanish text (default 2).
|
|
min_freq:
|
|
Minimum frequency to include in output.
|
|
outdir:
|
|
Output directory (defaults to config word output root).
|
|
out:
|
|
Output filename or path for content words.
|
|
include_proper_nouns:
|
|
Include ``PROPN`` tokens.
|
|
function_words:
|
|
Also extract a separate function-word list.
|
|
debug:
|
|
Path for the debug TSV.
|
|
lemma_corrections:
|
|
Path to a ``bad\\tgood`` TSV of lemma corrections.
|
|
bad_lemma_file:
|
|
Path to a list of lemmas to skip.
|
|
spacy_model:
|
|
SpaCy model name override.
|
|
|
|
Returns
|
|
-------
|
|
A dict with keys ``records``, ``stats``, ``written``, ``out``,
|
|
``debug``, ``proper_nouns``, ``function_words``, ``suspicious``.
|
|
"""
|
|
language_bucket = config.language_name(lang)
|
|
out_dir = os.path.expanduser(outdir) if outdir else os.path.join(config.word_output_root, language_bucket)
|
|
out_path = os.path.expanduser(out) if out else os.path.join(out_dir, f"words_{lang}_content.txt")
|
|
model_name = spacy_model or str(config.language(lang).get("word_model"))
|
|
nlp = load_spacy_model(model_name)
|
|
|
|
corrections = None
|
|
if lemma_corrections:
|
|
corrections = load_lemma_corrections(lemma_corrections)
|
|
bad_lemmas = None
|
|
if bad_lemma_file:
|
|
bad_lemmas = load_bad_lemmas(bad_lemma_file)
|
|
|
|
records = parse_anki_tsv(
|
|
input_path,
|
|
field_index=field_index,
|
|
include_tags=include_tags,
|
|
)
|
|
if not records:
|
|
return {"records": 0, "stats": {}, "written": 0, "out": out_path}
|
|
|
|
# Build token filter
|
|
def _content_filter(token) -> bool:
|
|
return spanish_content_filter(token, include_proper_nouns=include_proper_nouns)
|
|
|
|
stats = extract_detailed_counts(
|
|
records, nlp, _content_filter,
|
|
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
|
|
clean=not no_clean,
|
|
)
|
|
|
|
written = _write_content_words(stats, out_path, min_freq, bad_lemmas=bad_lemmas)
|
|
|
|
result: dict[str, Any] = {
|
|
"records": len(records),
|
|
"stats": stats,
|
|
"written": written,
|
|
"out": out_path,
|
|
}
|
|
|
|
# Debug output
|
|
if debug:
|
|
debug_path = os.path.expanduser(debug)
|
|
write_debug_tsv(stats, debug_path, bad_lemmas=bad_lemmas, min_freq=debug_min_freq)
|
|
result["debug"] = debug_path
|
|
|
|
# Proper nouns
|
|
if include_proper_nouns:
|
|
proper_path = os.path.join(out_dir, f"words_{lang}_proper_nouns.txt")
|
|
proper_items = [
|
|
(lemma, data["count"])
|
|
for lemma, data in stats.items()
|
|
if data["pos_counts"].get("PROPN", 0) > 0
|
|
]
|
|
proper_items.sort(key=lambda x: (-x[1], x[0]))
|
|
os.makedirs(os.path.dirname(os.path.abspath(proper_path)), exist_ok=True)
|
|
with open(proper_path, "w", encoding="utf-8") as f:
|
|
for w, c in proper_items:
|
|
f.write(f"{w} {c}\n")
|
|
result["proper_nouns"] = proper_path
|
|
|
|
# Function words (separate extraction pass)
|
|
if function_words:
|
|
fw_path = os.path.join(out_dir, f"words_{lang}_function_words.txt")
|
|
fw_stats = extract_detailed_counts(
|
|
records, nlp, spanish_function_word_filter,
|
|
lambda t: safe_spanish_lemma_info(t, extra_corrections=corrections),
|
|
)
|
|
fw_written = _write_function_words(fw_stats, fw_path)
|
|
result["function_words"] = fw_path
|
|
result["function_words_written"] = fw_written
|
|
|
|
# Suspicious tokens
|
|
susp_path = os.path.join(out_dir, f"words_{lang}_suspicious_tokens.txt")
|
|
write_suspicious_tokens(stats, susp_path, bad_lemmas=bad_lemmas)
|
|
result["suspicious"] = susp_path
|
|
|
|
return result
|
|
|
|
|
|
def lint_anki_cards_main(
|
|
input_path: str,
|
|
field_index: int = 2,
|
|
output_path: str = "suspicious_cards_es.tsv",
|
|
verbose: bool = False,
|
|
) -> dict[str, Any]:
|
|
"""Read an Anki TSV export and produce a suspicious-card report.
|
|
|
|
When *verbose* is True every record is written to the report, not
|
|
only those with issues.
|
|
|
|
Returns ``{"records": …, "issues": …, "path": output_path}``.
|
|
"""
|
|
records = parse_anki_tsv(input_path, field_index=field_index)
|
|
issues = lint_anki_cards(records, verbose=verbose)
|
|
|
|
os.makedirs(os.path.dirname(os.path.abspath(output_path)) or ".", exist_ok=True)
|
|
with open(output_path, "w", encoding="utf-8") as fh:
|
|
fh.write(
|
|
"source_line\toriginal_text\tcleaned_text\treason\tsuggested_fix\n"
|
|
)
|
|
for issue in issues:
|
|
fh.write(
|
|
f"{issue['source_line']}\t"
|
|
f"{_tsv_escape(issue['original_text'])}\t"
|
|
f"{_tsv_escape(issue['cleaned_text'])}\t"
|
|
f"{_tsv_escape(issue['reason'])}\t"
|
|
f"{_tsv_escape(issue['suggested_fix'])}\n"
|
|
)
|
|
|
|
return {"records": len(records), "issues": len(issues), "path": output_path}
|
|
|
|
|
|
def _tsv_escape(value: str) -> str:
|
|
"""Escape a value for TSV output (backslash-escape tabs and newlines)."""
|
|
return value.replace("\\", "\\\\").replace("\t", "\\t").replace("\n", "\\n")
|
|
|
|
|
|
def normalize_word_for_comparison(word: str) -> str:
|
|
"""Normalise a word for file-comparison: lowercase, strip accents."""
|
|
return word.lower().translate(ACCENT_MAP)
|
|
|
|
|
|
def read_word_file_normalized(path: str) -> dict[str, set[str]]:
|
|
"""Read a word-frequency file returning ``{normalised: {display_forms}}``.
|
|
|
|
Also returns a flat set of normalised forms.
|
|
"""
|
|
display_map: dict[str, set[str]] = {}
|
|
with open(os.path.expanduser(path), "r", encoding="utf-8") as fh:
|
|
for line in fh:
|
|
stripped = line.strip()
|
|
if not stripped:
|
|
continue
|
|
word = stripped.rsplit(" ", 1)[0]
|
|
key = normalize_word_for_comparison(word)
|
|
display_map.setdefault(key, set()).add(word)
|
|
return display_map
|
|
|
|
|
|
def compare_word_lists(
|
|
deck_words_path: str,
|
|
target_words_path: str,
|
|
output_path: str | None = None,
|
|
min_frequency: int = 0,
|
|
seen_output_path: str | None = None,
|
|
) -> dict[str, Any]:
|
|
"""Compare a deck word list against a target vocabulary list.
|
|
|
|
Parameters
|
|
----------
|
|
deck_words_path:
|
|
Path to the deck's extracted word list (``lemma count`` format).
|
|
target_words_path:
|
|
Path to the target vocabulary list (same format).
|
|
output_path:
|
|
If given, write missing-from-deck words here.
|
|
min_frequency:
|
|
Only consider deck words with frequency >= this value.
|
|
seen_output_path:
|
|
If given, write seen-in-deck words here.
|
|
|
|
Returns
|
|
-------
|
|
``{"missing": [str], "seen": [str], "total_missing": int}``
|
|
"""
|
|
# Read target words
|
|
target_map = read_word_file_normalized(target_words_path)
|
|
target_keys = set(target_map.keys())
|
|
|
|
# Read deck words, optionally filtering by frequency
|
|
deck_map: dict[str, set[str]] = {}
|
|
with open(os.path.expanduser(deck_words_path), "r", encoding="utf-8") as fh:
|
|
for line in fh:
|
|
stripped = line.strip()
|
|
if not stripped:
|
|
continue
|
|
parts = stripped.rsplit(" ", 1)
|
|
word = parts[0]
|
|
freq = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else 1
|
|
if freq < min_frequency:
|
|
continue
|
|
key = normalize_word_for_comparison(word)
|
|
deck_map.setdefault(key, set()).add(word)
|
|
deck_keys = set(deck_map.keys())
|
|
|
|
missing_keys = target_keys - deck_keys
|
|
seen_keys = target_keys & deck_keys
|
|
|
|
missing_entries: list[str] = []
|
|
for key in sorted(missing_keys):
|
|
for display in sorted(target_map[key]):
|
|
missing_entries.append(display)
|
|
|
|
seen_entries: list[str] = []
|
|
for key in sorted(seen_keys):
|
|
for display in sorted(target_map[key]):
|
|
seen_entries.append(display)
|
|
|
|
if output_path:
|
|
out = os.path.expanduser(output_path)
|
|
os.makedirs(os.path.dirname(os.path.abspath(out)) or ".", exist_ok=True)
|
|
with open(out, "w", encoding="utf-8") as fh:
|
|
for entry in missing_entries:
|
|
fh.write(f"{entry}\n")
|
|
|
|
if seen_output_path:
|
|
seen_out = os.path.expanduser(seen_output_path)
|
|
os.makedirs(os.path.dirname(os.path.abspath(seen_out)) or ".", exist_ok=True)
|
|
with open(seen_out, "w", encoding="utf-8") as fh:
|
|
for entry in seen_entries:
|
|
fh.write(f"{entry}\n")
|
|
|
|
return {
|
|
"missing": missing_entries,
|
|
"seen": seen_entries,
|
|
"total_missing": len(missing_entries),
|
|
}
|