treefmt / nix fmt (pull_request) Failing after 5s
pytest / pytest (pull_request) Failing after 36s
build_systems / build-brain (pull_request) Successful in 51s
build_systems / build-bob (pull_request) Successful in 55s
build_systems / build-rhapsody-in-green (pull_request) Successful in 1m30s
build_systems / build-jeeves (pull_request) Successful in 3m9s
Run full-book candidate generation inside worker-owned sessions so each book commits independently during backfills. Abort recalculation when a book has no indexed chapters to preserve existing phrase data, and update admin/UI tests for the new generation flow.
206 lines
8.2 KiB
Python
206 lines
8.2 KiB
Python
"""Page routes for the EPUB search web UI."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import TYPE_CHECKING
|
|
|
|
from fastapi import APIRouter, BackgroundTasks, HTTPException, Request
|
|
from fastapi.responses import HTMLResponse, RedirectResponse
|
|
from sqlalchemy import func, select
|
|
|
|
from python.ebook_search.api.dependencies import (
|
|
AppConfig, # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
|
)
|
|
from python.ebook_search.api.judge_tasks import is_judging_book, pop_book_judgment_outcome, start_book_phrase_judgment
|
|
from python.ebook_search.api.web import templates
|
|
from python.ebook_search.protected_phrases.generate_ngrams import recalculate_candidate_phrases_for_book
|
|
from python.fastapi_tools import AsyncDbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
|
from python.orm.richie import EbookCandidatePhrase, EbookChapter, EbookChunk, EbookProtectedPhrase, EbookSource
|
|
|
|
if TYPE_CHECKING:
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
router = APIRouter()
|
|
|
|
|
|
@router.get("/", response_class=HTMLResponse)
|
|
async def index(request: Request, config: AppConfig) -> HTMLResponse:
|
|
"""Render the search page."""
|
|
return templates.TemplateResponse(request, "search.html", {"config": config})
|
|
|
|
|
|
@router.get("/books", response_class=HTMLResponse)
|
|
async def books(request: Request, session: AsyncDbSession) -> HTMLResponse:
|
|
"""Render the indexed books page."""
|
|
sources = list((await session.scalars(select(EbookSource).order_by(EbookSource.title))).all())
|
|
logger.info("ebook_books_page_loaded count=%s", len(sources))
|
|
return templates.TemplateResponse(request, "books.html", {"sources": sources})
|
|
|
|
|
|
async def get_chapter_count(session: AsyncSession, book_id: int) -> int:
|
|
"""Return the number of indexed chapters for one book."""
|
|
return await session.scalar(select(func.count(EbookChapter.id)).where(EbookChapter.source_id == book_id)) or 0
|
|
|
|
|
|
async def get_chunk_count(session: AsyncSession, book_id: int) -> int:
|
|
"""Return the number of indexed chunks for one book."""
|
|
return await session.scalar(select(func.count(EbookChunk.id)).where(EbookChunk.source_id == book_id)) or 0
|
|
|
|
|
|
async def get_candidate_count(session: AsyncSession, book_id: int) -> int:
|
|
"""Return the number of indexed candidates for one book."""
|
|
return (
|
|
await session.scalar(select(func.count(EbookCandidatePhrase.id)).where(EbookCandidatePhrase.book_id == book_id))
|
|
or 0
|
|
)
|
|
|
|
|
|
async def get_judged_candidate_count(session: AsyncSession, book_id: int) -> int:
|
|
"""Return the number of judged candidates for one book."""
|
|
return (
|
|
await session.scalar(
|
|
select(func.count(EbookCandidatePhrase.id)).where(
|
|
EbookCandidatePhrase.book_id == book_id,
|
|
EbookCandidatePhrase.llm_judged.is_(True),
|
|
)
|
|
)
|
|
or 0
|
|
)
|
|
|
|
|
|
async def get_protected_count(session: AsyncSession, book_id: int) -> int:
|
|
"""Return the number of protected phrases for one book."""
|
|
return (
|
|
await session.scalar(select(func.count(EbookProtectedPhrase.id)).where(EbookProtectedPhrase.book_id == book_id))
|
|
or 0
|
|
)
|
|
|
|
|
|
async def get_candidates(session: AsyncSession, book_id: int) -> list[EbookCandidatePhrase]:
|
|
"""Return the indexed candidates for one book."""
|
|
return list(
|
|
await session.scalars(
|
|
select(EbookCandidatePhrase)
|
|
.where(EbookCandidatePhrase.book_id == book_id)
|
|
.order_by(EbookCandidatePhrase.candidate_score.desc())
|
|
.limit(100)
|
|
)
|
|
)
|
|
|
|
|
|
async def get_protected_phrases(session: AsyncSession, book_id: int) -> list[EbookProtectedPhrase]:
|
|
"""Return the protected phrases for one book."""
|
|
return list(
|
|
await session.scalars(
|
|
select(EbookProtectedPhrase)
|
|
.where(EbookProtectedPhrase.book_id == book_id)
|
|
.order_by(EbookProtectedPhrase.importance.desc())
|
|
.limit(100)
|
|
)
|
|
)
|
|
|
|
|
|
@router.get("/books/{source_id}", response_class=HTMLResponse)
|
|
async def book_detail(source_id: int, request: Request, session: AsyncDbSession) -> HTMLResponse:
|
|
"""Render details for one indexed book."""
|
|
source = await session.get(EbookSource, source_id)
|
|
phrase_status_message = None
|
|
recalculated = request.query_params.get("phrases_recalculated")
|
|
if recalculated is not None:
|
|
phrase_status_message = f"Recalculated phrases; {recalculated} candidates generated"
|
|
judgment_outcome = pop_book_judgment_outcome(request.app, source_id)
|
|
if judgment_outcome is not None:
|
|
phrase_status_message = judgment_outcome
|
|
judging_in_progress = is_judging_book(request.app, source_id)
|
|
if judging_in_progress:
|
|
phrase_status_message = "Judging candidate phrases in the background; refresh to see progress"
|
|
if source is not None:
|
|
chapter_count = await get_chapter_count(session, source.id)
|
|
chunk_count = await get_chunk_count(session, source.id)
|
|
candidate_count = await get_candidate_count(session, source.id)
|
|
judged_candidate_count = await get_judged_candidate_count(session, source.id)
|
|
protected_count = await get_protected_count(session, source.id)
|
|
candidates = await get_candidates(session, source.id)
|
|
protected_phrases = await get_protected_phrases(session, source.id)
|
|
else:
|
|
chapter_count = 0
|
|
chunk_count = 0
|
|
candidate_count = 0
|
|
judged_candidate_count = 0
|
|
protected_count = 0
|
|
candidates = []
|
|
protected_phrases = []
|
|
logger.info(
|
|
"ebook_book_detail_loaded source_id=%s found=%s chapters=%s chunks=%s candidates=%s judged=%s protected=%s",
|
|
source_id,
|
|
source is not None,
|
|
chapter_count,
|
|
chunk_count,
|
|
candidate_count,
|
|
judged_candidate_count,
|
|
protected_count,
|
|
)
|
|
return templates.TemplateResponse(
|
|
request,
|
|
"book_detail.html",
|
|
{
|
|
"candidate_count": candidate_count,
|
|
"candidates": candidates,
|
|
"chapter_count": chapter_count,
|
|
"chunk_count": chunk_count,
|
|
"judged_candidate_count": judged_candidate_count,
|
|
"judging_in_progress": judging_in_progress,
|
|
"protected_count": protected_count,
|
|
"protected_phrases": protected_phrases,
|
|
"phrase_status_message": phrase_status_message,
|
|
"source": source,
|
|
},
|
|
)
|
|
|
|
|
|
@router.post("/books/{source_id}/recalculate-phrases")
|
|
async def recalculate_book_phrases(source_id: int, config: AppConfig, session: AsyncDbSession) -> RedirectResponse:
|
|
"""Clear and regenerate candidate phrases for one indexed book."""
|
|
source = await session.get(EbookSource, source_id)
|
|
if source is None:
|
|
raise HTTPException(status_code=404, detail="Book not found")
|
|
|
|
try:
|
|
result = await recalculate_candidate_phrases_for_book(session, source, config)
|
|
except ValueError as error:
|
|
raise HTTPException(status_code=409, detail=str(error)) from error
|
|
logger.info(
|
|
"ebook_book_phrase_recalculation_complete source_id=%s candidates=%s deleted_candidates=%s "
|
|
"deleted_protected=%s deleted_aliases=%s deleted_mentions=%s",
|
|
source_id,
|
|
result.candidate_phrases,
|
|
result.deleted_candidates,
|
|
result.deleted_protected_phrases,
|
|
result.deleted_aliases,
|
|
result.deleted_mentions,
|
|
)
|
|
return RedirectResponse(
|
|
url=f"/books/{source_id}?phrases_recalculated={result.candidate_phrases}",
|
|
status_code=303,
|
|
)
|
|
|
|
|
|
@router.post("/books/{source_id}/judge-phrases")
|
|
async def judge_book_phrases(
|
|
source_id: int,
|
|
request: Request,
|
|
background_tasks: BackgroundTasks,
|
|
session: AsyncDbSession,
|
|
) -> RedirectResponse:
|
|
"""Queue background judging of one book's candidate phrases and return immediately."""
|
|
source = await session.get(EbookSource, source_id)
|
|
if source is None:
|
|
raise HTTPException(status_code=404, detail="Book not found")
|
|
|
|
started = start_book_phrase_judgment(request.app, background_tasks, source.id)
|
|
logger.info("ebook_book_phrase_judgment_requested source_id=%s started=%s", source_id, started)
|
|
return RedirectResponse(url=f"/books/{source_id}", status_code=303)
|