"""Protected phrase extraction, storage, and runtime matching.""" from __future__ import annotations import logging import tomllib from functools import cache from pathlib import Path from python.ebook_search.protected_phrases.text_normalization import normalize_text logger = logging.getLogger(__name__) def _load_toml_string_set(path: Path, key: str) -> frozenset[str]: """Load and validate a TOML string list as a normalized immutable set.""" with path.open("rb") as file: body = tomllib.load(file) values = body.get(key) if not isinstance(values, list) or not all(isinstance(item, str) for item in values): msg = f"{path} must contain a {key!r} string list" raise ValueError(msg) return frozenset(normalize_text(value) for value in values if normalize_text(value)) @cache def _get_phrase_config_dir() -> Path: """Return the directory containing phrase configuration files.""" return Path(__file__).resolve().parent @cache def get_ignored_phrases() -> frozenset[str]: """Return ignored phrase strings loaded from TOML.""" return _load_toml_string_set(_get_phrase_config_dir() / "ignored_phrases.toml", "phrases") @cache def get_bad_ends() -> frozenset[str]: """Return bad phrase-ending tokens loaded from TOML.""" return _load_toml_string_set(_get_phrase_config_dir() / "bad_ends.toml", "tokens") @cache def get_bad_starts() -> frozenset[str]: """Return bad phrase-starting tokens loaded from TOML.""" return _load_toml_string_set(_get_phrase_config_dir() / "bad_starts.toml", "tokens") @cache def get_most_common_words() -> frozenset[str]: """Return the most common English words loaded from TOML.""" return _load_toml_string_set(_get_phrase_config_dir() / "most_common_words.toml", "words")