refactor(ebook): remove spaCy-ner attributes from PhraseCandidate and related functions

This commit is contained in:
2026-07-09 23:05:08 -04:00
parent 0cf05c52b9
commit ffdb93d352
5 changed files with 58 additions and 121 deletions
@@ -19,11 +19,8 @@ class PhraseCandidate:
token_count (int): Number of normalized tokens in the phrase.
source_raw_ngram (bool): Whether the raw n-gram extractor produced the phrase.
source_yake (bool): Whether YAKE keyword extraction produced the phrase.
source_spacy_ner (bool): Whether spaCy named-entity recognition produced the phrase.
source_spacy_noun_chunk (bool): Whether spaCy noun chunking produced the phrase.
source_capitalized (bool): Whether the capitalized-run extractor produced the phrase.
source_metadata (bool): Whether book metadata produced the phrase.
spacy_label (str | None): spaCy entity label when NER produced the phrase.
raw_count (int): Occurrences counted across the book text.
chapter_count (int): Number of chapters containing the phrase.
yake_score (float | None): Raw YAKE score when available; lower is better.
@@ -36,11 +33,8 @@ class PhraseCandidate:
token_count: int
source_raw_ngram: bool = False
source_yake: bool = False
source_spacy_ner: bool = False
source_spacy_noun_chunk: bool = False
source_capitalized: bool = False
source_metadata: bool = False
spacy_label: str | None = None
raw_count: int = 0
chapter_count: int = 0
yake_score: float | None = None