feat(ebook): add protected phrase extraction library with config-driven tuning
Refactor protected phrase handling from a single module into a python/ebook_search/protected_phrases package covering extraction, storage, and runtime matching. Phrase filtering is now data-driven via bundled TOML files: ignored_phrases, bad_starts, bad_ends, and most_common_words. Add phrase-tuning settings to EbookSearchConfig so candidate generation, scoring, LLM judging, and matching are configurable rather than hardcoded: token bounds, entity token limit, raw n-gram min count, frequency and chapter-spread score thresholds, candidate/LLM/target caps, confidence threshold, nesting defaults, and the phrase hit boost.
This commit is contained in:
@@ -86,6 +86,21 @@ class EbookSearchConfig(BaseSettings):
|
||||
validate_citations_enabled: bool = True
|
||||
bm25_index_dir: str = ".ebook_search_bm25"
|
||||
bm25_refresh_delay_seconds: int = 60
|
||||
protected_phrase_max_candidates_per_book: int = 5000
|
||||
protected_phrase_llm_candidates_per_book: int = 500
|
||||
protected_phrase_confidence_threshold: float = 0.80
|
||||
phrase_hit_boost: float = 0.25
|
||||
phrase_min_tokens: int = 2
|
||||
phrase_max_tokens: int = 5
|
||||
phrase_max_entity_tokens: int = 8
|
||||
phrase_raw_ngram_min_count: int = 2
|
||||
phrase_raw_count_score_threshold: int = 3
|
||||
phrase_raw_count_high_score_threshold: int = 10
|
||||
phrase_chapter_count_score_threshold: int = 2
|
||||
phrase_chapter_count_high_score_threshold: int = 5
|
||||
phrase_target_protected_per_book: int = 100
|
||||
phrase_default_allow_nested: bool = False
|
||||
phrase_default_suppress_children: bool = True
|
||||
|
||||
@field_validator("library_paths", mode="before")
|
||||
@classmethod
|
||||
|
||||
Reference in New Issue
Block a user