55 lines
1.8 KiB
Python
55 lines
1.8 KiB
Python
"""Protected phrase extraction, storage, and runtime matching."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import tomllib
|
|
from functools import cache
|
|
from pathlib import Path
|
|
|
|
from python.ebook_search.protected_phrases.text_normalization import normalize_text
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _load_toml_string_set(path: Path, key: str) -> frozenset[str]:
|
|
"""Load and validate a TOML string list as a normalized immutable set."""
|
|
with path.open("rb") as file:
|
|
body = tomllib.load(file)
|
|
|
|
values = body.get(key)
|
|
if not isinstance(values, list) or not all(isinstance(item, str) for item in values):
|
|
msg = f"{path} must contain a {key!r} string list"
|
|
raise ValueError(msg)
|
|
return frozenset(normalize_text(value) for value in values if normalize_text(value))
|
|
|
|
|
|
@cache
|
|
def _get_phrase_config_dir() -> Path:
|
|
"""Return the directory containing phrase configuration files."""
|
|
return Path(__file__).resolve().parent
|
|
|
|
|
|
@cache
|
|
def get_ignored_phrases() -> frozenset[str]:
|
|
"""Return ignored phrase strings loaded from TOML."""
|
|
return _load_toml_string_set(_get_phrase_config_dir() / "ignored_phrases.toml", "phrases")
|
|
|
|
|
|
@cache
|
|
def get_bad_ends() -> frozenset[str]:
|
|
"""Return bad phrase-ending tokens loaded from TOML."""
|
|
return _load_toml_string_set(_get_phrase_config_dir() / "bad_ends.toml", "tokens")
|
|
|
|
|
|
@cache
|
|
def get_bad_starts() -> frozenset[str]:
|
|
"""Return bad phrase-starting tokens loaded from TOML."""
|
|
return _load_toml_string_set(_get_phrase_config_dir() / "bad_starts.toml", "tokens")
|
|
|
|
|
|
@cache
|
|
def get_most_common_words() -> frozenset[str]:
|
|
"""Return the most common English words loaded from TOML."""
|
|
return _load_toml_string_set(_get_phrase_config_dir() / "most_common_words.toml", "words")
|