Compare commits
164
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6d6226ed14 | ||
|
|
39ab7358bb | ||
|
|
4488441a84 | ||
|
|
681a2d8d12 | ||
|
|
f3e36f7ec3 | ||
|
|
d8693736d9 | ||
|
|
c7cd63f8e4 | ||
|
|
5c73fc9e9b | ||
|
|
e596652582 | ||
|
|
507c3f5406 | ||
|
|
7c1618cfd6 | ||
|
|
086224a4c9 | ||
|
|
c11895f144 | ||
|
|
dab18c1385 | ||
|
|
e34ed6c597 | ||
|
|
a19ace7959 | ||
|
|
0f84c7ce41 | ||
|
|
cc31483973 | ||
|
|
790fb6680d | ||
|
|
927c5cbc60 | ||
|
|
8db44916c8 | ||
|
|
f20c45bea9 | ||
|
|
6a0e3ebcdb | ||
|
|
d9115f7c91 | ||
|
|
1eecf7181d | ||
|
|
3e164831b5 | ||
|
|
872e55da1d | ||
|
|
47f77443dd | ||
|
|
8548f4ff64 | ||
|
|
682960b4c9 | ||
|
|
ddea13b93a | ||
|
|
b252b0bf6e | ||
|
|
fbf288649a | ||
|
|
db199f884d | ||
|
|
c6645eb37c | ||
|
|
3984081753 | ||
|
|
03f610feed | ||
|
|
8804a8df83 | ||
|
|
8daac4d98b | ||
|
|
7606ddc514 | ||
|
|
137bd8ed2f | ||
|
|
f936253ad5 | ||
|
|
fc01fb9d0b | ||
|
|
09d963ba34 | ||
|
|
855af0fdff | ||
|
|
fcb69cc68b | ||
|
|
6bc30115d9 | ||
|
|
6ae1ff1f5c | ||
|
|
dbc6b5b53b | ||
|
|
a9daa60c17 | ||
|
|
241a42b20d | ||
|
|
4640ebf8ce | ||
|
|
0c9583c1cc | ||
|
|
f71ae7d2c6 | ||
|
|
2e68c83021 | ||
|
|
b126987b63 | ||
|
|
68b3a38b81 | ||
|
|
a5d7c3be4f | ||
|
|
2995a75748 | ||
|
|
dce1838163 | ||
|
|
121eb979a4 | ||
|
|
c88315e9b6 | ||
|
|
50795ab7fc | ||
|
|
e9a80a0308 | ||
|
|
eb76edb740 | ||
|
|
c53afb3c70 | ||
|
|
8301db39e5 | ||
|
|
45a9e90524 | ||
|
|
dd67de3993 | ||
|
|
6e3635ca01 | ||
|
|
11cbe31152 | ||
|
|
5544dfa61c | ||
|
|
911df63513 | ||
|
|
07eb170b34 | ||
|
|
6715bbf0a5 | ||
|
|
26ff1f0fd3 | ||
|
|
666ea97754 | ||
|
|
e01c687625 | ||
|
|
82a367a2b6 | ||
|
|
ad1834c537 | ||
|
|
aed1e14d95 | ||
|
|
0a2d4c08cb | ||
|
|
db98bd3559 | ||
|
|
cdded5da12 | ||
|
|
d022251a58 | ||
|
|
e1ef4de6a3 | ||
|
|
5c230a267c | ||
|
|
5215d66d40 | ||
|
|
7ad198416b | ||
|
|
1461c2552a | ||
|
|
736717c2f8 | ||
|
|
ab2521867e | ||
|
|
8e0ab4190b | ||
|
|
734fd7641e | ||
|
|
e898e08c48 | ||
|
|
d916ea903c | ||
|
|
d8e916dbe6 | ||
|
|
48e9f0199d | ||
|
|
a526420c8d | ||
|
|
41e3e265af | ||
|
|
38a17f6146 | ||
|
|
fe48d4c1ad | ||
|
|
9290cb46ee | ||
|
|
acd3f2d3ac | ||
|
|
08e716f66a | ||
|
|
d197731af4 | ||
|
|
1ffc48bb02 | ||
|
|
b6395ef18f | ||
|
|
aff6f4e1bd | ||
|
|
a9a96db944 | ||
|
|
d34154541d | ||
|
|
5d3a851137 | ||
|
|
e05e5c77bc | ||
|
|
b0a2ebc052 | ||
|
|
f77c9657a3 | ||
|
|
f908f969d3 | ||
|
|
3cf49c5479 | ||
|
|
b34354f5e5 | ||
|
|
44826464de | ||
|
|
3de0ffccb0 | ||
|
|
c6c98b3e26 | ||
|
|
d459f3d675 | ||
|
|
33e4b37cce | ||
|
|
2a8e7e7f2b | ||
|
|
07759353be | ||
|
|
38fb14520e | ||
|
|
006ae6079a | ||
|
|
7d507fb7e1 | ||
|
|
0f69022e51 | ||
|
|
a260ae2470 | ||
|
|
820b4a53d2 | ||
|
|
ea77e83f06 | ||
|
|
a9da208bc3 | ||
|
|
739d7dd28c | ||
|
|
651599796e | ||
|
|
b9d440597c | ||
|
|
311cc5d7a7 | ||
|
|
fb2519046d | ||
|
|
bc6b1585ec | ||
|
|
d71330a85a | ||
|
|
df51aa5200 | ||
|
|
e93cc816db | ||
|
|
19050b4cf4 | ||
|
|
6676c15f75 | ||
|
|
27e487e322 | ||
|
|
4f28050eff | ||
|
|
b58ea60557 | ||
|
|
e95eedffe4 | ||
|
|
1abd53987c | ||
|
|
d1a3e7338a | ||
|
|
687ef0c167 | ||
|
|
3a86148352 | ||
|
|
fe9a2912e1 | ||
|
|
29a99fc210 | ||
|
|
d7651bf588 | ||
|
|
2865dcbe9c | ||
|
|
d920b77bab | ||
|
|
1b53167b53 | ||
|
|
9dabb9dc07 | ||
|
|
95630fe151 | ||
|
|
d3a889f100 | ||
|
|
6ce0671f51 | ||
|
|
25ab6b2ab6 | ||
|
|
374d7e8d38 |
@@ -0,0 +1,15 @@
|
|||||||
|
.git
|
||||||
|
.direnv
|
||||||
|
.mypy_cache
|
||||||
|
.pytest_cache
|
||||||
|
.ruff_cache
|
||||||
|
__pycache__
|
||||||
|
**/__pycache__
|
||||||
|
*.pyc
|
||||||
|
*.pyo
|
||||||
|
.ebook_search_bm25
|
||||||
|
result
|
||||||
|
result-*
|
||||||
|
*.egg-info
|
||||||
|
dist
|
||||||
|
build
|
||||||
@@ -17,12 +17,11 @@ jobs:
|
|||||||
- "bob"
|
- "bob"
|
||||||
- "brain"
|
- "brain"
|
||||||
- "jeeves"
|
- "jeeves"
|
||||||
- "leviathan"
|
|
||||||
- "rhapsody-in-green"
|
- "rhapsody-in-green"
|
||||||
continue-on-error: true
|
continue-on-error: true
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- name: Build default package
|
- name: Build default package
|
||||||
run: "nixos-rebuild build --flake ./#${{ matrix.system }}"
|
run: "nixos-rebuild build --accept-flake-config --flake ./#${{ matrix.system }}"
|
||||||
- name: copy to nix-cache
|
- name: copy to nix-cache
|
||||||
run: nix copy --accept-flake-config --to unix:///host-nix/var/nix/daemon-socket/socket .#nixosConfigurations.${{ matrix.system }}.config.system.build.toplevel
|
run: nix copy --accept-flake-config --to unix:///host-nix/var/nix/daemon-socket/socket .#nixosConfigurations.${{ matrix.system }}.config.system.build.toplevel
|
||||||
|
|||||||
@@ -1,30 +0,0 @@
|
|||||||
name: fix_eval_warnings
|
|
||||||
on:
|
|
||||||
workflow_run:
|
|
||||||
workflows: ["build_systems"]
|
|
||||||
types: [completed]
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
check-warnings:
|
|
||||||
if: >-
|
|
||||||
github.event.workflow_run.conclusion != 'cancelled' &&
|
|
||||||
github.event.workflow_run.head_branch == 'main' &&
|
|
||||||
(github.event.workflow_run.event == 'push' || github.event.workflow_run.event == 'schedule')
|
|
||||||
runs-on: self-hosted
|
|
||||||
permissions:
|
|
||||||
contents: write
|
|
||||||
pull-requests: write
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: Fix eval warnings
|
|
||||||
env:
|
|
||||||
GH_TOKEN: ${{ secrets.GH_TOKEN_FOR_UPDATES }}
|
|
||||||
run: >-
|
|
||||||
nix develop .#devShells.x86_64-linux.default -c
|
|
||||||
python -m python.eval_warnings.main
|
|
||||||
--run-id "${{ github.event.workflow_run.id }}"
|
|
||||||
--repo "${{ github.repository }}"
|
|
||||||
--ollama-url "${{ secrets.OLLAMA_URL }}"
|
|
||||||
--run-url "${{ github.event.workflow_run.html_url }}"
|
|
||||||
@@ -6,24 +6,18 @@ on:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
merge:
|
merge:
|
||||||
runs-on: ubuntu-latest
|
runs-on: self-hosted
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
pull-requests: write
|
pull-requests: write
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- name: merge_flake_lock_update
|
- name: merge_flake_lock_update
|
||||||
run: |
|
run: >-
|
||||||
pr_number=$(gh pr list --state open --author RichieCahill --label flake_lock_update --json number --jq '.[0].number')
|
nix develop .#devShells.x86_64-linux.default -c
|
||||||
echo "pr_number=$pr_number" >> $GITHUB_ENV
|
python -m python.gitea_flake_lock merge
|
||||||
if [ -n "$pr_number" ]; then
|
--repo "${{ github.repository }}"
|
||||||
gh pr merge "$pr_number" --rebase
|
|
||||||
else
|
|
||||||
echo "No open PR found with label flake_lock_update"
|
|
||||||
fi
|
|
||||||
env:
|
env:
|
||||||
GITHUB_TOKEN: ${{ secrets.GH_TOKEN_FOR_UPDATES }}
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
|
GITEA_URL: https://gitea.tmmworkshop.com
|
||||||
|
|||||||
@@ -1,13 +1,13 @@
|
|||||||
name: pytest
|
name: pytest
|
||||||
|
|
||||||
on:
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
push:
|
push:
|
||||||
branches:
|
branches:
|
||||||
- main
|
- main
|
||||||
pull_request:
|
pull_request:
|
||||||
branches:
|
branches:
|
||||||
- main
|
- main
|
||||||
merge_group:
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
pytest:
|
pytest:
|
||||||
|
|||||||
@@ -6,18 +6,21 @@ on:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
lockfile:
|
lockfile:
|
||||||
runs-on: ubuntu-latest
|
runs-on: self-hosted
|
||||||
|
permissions:
|
||||||
|
actions: write
|
||||||
|
contents: write
|
||||||
|
pull-requests: write
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
- name: Install Nix
|
|
||||||
uses: DeterminateSystems/nix-installer-action@main
|
|
||||||
- name: Update flake.lock
|
- name: Update flake.lock
|
||||||
uses: DeterminateSystems/update-flake-lock@main
|
run: nix flake update
|
||||||
with:
|
- name: Create or update flake.lock PR
|
||||||
token: ${{ secrets.GH_TOKEN_FOR_UPDATES }}
|
env:
|
||||||
pr-title: "Update flake.lock"
|
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||||
pr-labels: |
|
GITEA_URL: https://gitea.tmmworkshop.com
|
||||||
dependencies
|
run: >-
|
||||||
automated
|
nix develop .#devShells.x86_64-linux.default -c
|
||||||
flake_lock_update
|
python -m python.gitea_flake_lock update
|
||||||
|
--repo "${{ github.repository }}"
|
||||||
|
|||||||
@@ -172,3 +172,4 @@ frontend/node_modules/
|
|||||||
|
|
||||||
# data from testing llms
|
# data from testing llms
|
||||||
data/*
|
data/*
|
||||||
|
.ebook_search_bm25
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ keys:
|
|||||||
- &system_bob age1q47vup0tjhulkg7d6xwmdsgrw64h4ax3la3evzqpxyy4adsmk9fs56qz3y # cspell:disable-line
|
- &system_bob age1q47vup0tjhulkg7d6xwmdsgrw64h4ax3la3evzqpxyy4adsmk9fs56qz3y # cspell:disable-line
|
||||||
- &system_brain age1jhf7vm0005j60mjq63696frrmjhpy8kpc2d66mw044lqap5mjv4snmwvwm # cspell:disable-line
|
- &system_brain age1jhf7vm0005j60mjq63696frrmjhpy8kpc2d66mw044lqap5mjv4snmwvwm # cspell:disable-line
|
||||||
- &system_jeeves age13lmqgc3jvkyah5e3vcwmj4s5wsc2akctcga0lpc0x8v8du3fxprqp4ldkv # cspell:disable-line
|
- &system_jeeves age13lmqgc3jvkyah5e3vcwmj4s5wsc2akctcga0lpc0x8v8du3fxprqp4ldkv # cspell:disable-line
|
||||||
- &system_leviathan age1l272y8udvg60z7edgje42fu49uwt4x2gxn5zvywssnv9h2krms8s094m4k # cspell:disable-line
|
|
||||||
- &system_rhapsody age1ufnewppysaq2wwcl4ugngjz8pfzc5a35yg7luq0qmuqvctajcycs5lf6k4 # cspell:disable-line
|
- &system_rhapsody age1ufnewppysaq2wwcl4ugngjz8pfzc5a35yg7luq0qmuqvctajcycs5lf6k4 # cspell:disable-line
|
||||||
|
|
||||||
creation_rules:
|
creation_rules:
|
||||||
@@ -18,5 +17,4 @@ creation_rules:
|
|||||||
- *system_bob
|
- *system_bob
|
||||||
- *system_brain
|
- *system_brain
|
||||||
- *system_jeeves
|
- *system_jeeves
|
||||||
- *system_leviathan
|
|
||||||
- *system_rhapsody
|
- *system_rhapsody
|
||||||
|
|||||||
Vendored
+7
@@ -71,6 +71,7 @@
|
|||||||
"ehci",
|
"ehci",
|
||||||
"emerg",
|
"emerg",
|
||||||
"endlessh",
|
"endlessh",
|
||||||
|
"ents",
|
||||||
"errorlens",
|
"errorlens",
|
||||||
"esbenp",
|
"esbenp",
|
||||||
"esphome",
|
"esphome",
|
||||||
@@ -172,6 +173,8 @@
|
|||||||
"Networkd",
|
"Networkd",
|
||||||
"networkmanager",
|
"networkmanager",
|
||||||
"newtabpage",
|
"newtabpage",
|
||||||
|
"ngram",
|
||||||
|
"ngrams",
|
||||||
"nixfmt",
|
"nixfmt",
|
||||||
"nixos",
|
"nixos",
|
||||||
"nixpkgs",
|
"nixpkgs",
|
||||||
@@ -242,6 +245,7 @@
|
|||||||
"referer",
|
"referer",
|
||||||
"REFERERS",
|
"REFERERS",
|
||||||
"relatime",
|
"relatime",
|
||||||
|
"rerank",
|
||||||
"Rhosts",
|
"Rhosts",
|
||||||
"ripgrep",
|
"ripgrep",
|
||||||
"roboto",
|
"roboto",
|
||||||
@@ -297,7 +301,9 @@
|
|||||||
"uiprotect",
|
"uiprotect",
|
||||||
"uitour",
|
"uitour",
|
||||||
"unifi",
|
"unifi",
|
||||||
|
"unjudged",
|
||||||
"unrar",
|
"unrar",
|
||||||
|
"unstorable",
|
||||||
"unsubmitted",
|
"unsubmitted",
|
||||||
"uptimekuma",
|
"uptimekuma",
|
||||||
"urlbar",
|
"urlbar",
|
||||||
@@ -325,6 +331,7 @@
|
|||||||
"xcursorgen",
|
"xcursorgen",
|
||||||
"xdist",
|
"xdist",
|
||||||
"xhci",
|
"xhci",
|
||||||
|
"yake",
|
||||||
"yazi",
|
"yazi",
|
||||||
"yubikey",
|
"yubikey",
|
||||||
"yubioath",
|
"yubioath",
|
||||||
|
|||||||
@@ -1,12 +0,0 @@
|
|||||||
## Dev environment tips
|
|
||||||
|
|
||||||
- use treefmt to format all files
|
|
||||||
- make python code ruff compliant
|
|
||||||
- use pytest to test python code
|
|
||||||
- always use the minimum amount of complexity
|
|
||||||
- if judgment calls are easy to reverse make them. if not ask me first
|
|
||||||
- Match existing code style.
|
|
||||||
- Use builtin helpers getenv() over os.environ.get.
|
|
||||||
- Prefer single-purpose functions over “do everything” helpers.
|
|
||||||
- Avoid compatibility branches like PG_USER and POSTGRESQL_URL unless requested.
|
|
||||||
- Keep helpers only if reused or they simplify the code otherwise inline.
|
|
||||||
@@ -1 +1,51 @@
|
|||||||
# dotfiles
|
# dotfiles
|
||||||
|
|
||||||
|
## Installer ISO
|
||||||
|
|
||||||
|
Build a bootable NixOS ISO with the installer preinstalled:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
nix build .#iso
|
||||||
|
```
|
||||||
|
|
||||||
|
Write `result/iso/nixos-zfs-installer.iso` to a USB stick (for example with `dd`) or boot it in a VM. The image is the minimal NixOS installation CD with ZFS enabled and `nixos-installer` on `PATH`. SSH is enabled and the `nixos` and `root` accounts use the password `nixos`, so you can also run the installer remotely. Once booted:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo nixos-installer
|
||||||
|
```
|
||||||
|
|
||||||
|
The ISO bundles the `.#installer-nixos` package, a variant of the binary that keeps its Nix store linkage instead of being patched for foreign distributions.
|
||||||
|
|
||||||
|
## Installer binary
|
||||||
|
|
||||||
|
Build the self-contained installer executable with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
nix build .#installer
|
||||||
|
```
|
||||||
|
|
||||||
|
The flake package (defined in `python/installer/package.nix`) uses the Python builder in `python/installer/build.py`, which stages only the installer modules before running PyInstaller. You can also call it directly when `pyinstaller` and `patchelf` are on `PATH`:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
python -m python.installer.build --output ./nixos-installer
|
||||||
|
```
|
||||||
|
|
||||||
|
Copy `result/bin/nixos-installer` to the installer USB stick and run it as root from the NixOS live environment:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo ./nixos-installer
|
||||||
|
```
|
||||||
|
|
||||||
|
Validate the live environment first with:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
./nixos-installer --check
|
||||||
|
```
|
||||||
|
|
||||||
|
Paste a value into the TUI encryption password field to enable LUKS during install, or set `ENCRYPT_KEY`:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo env ENCRYPT_KEY='change-me' ./nixos-installer
|
||||||
|
```
|
||||||
|
|
||||||
|
The binary bundles the Python runtime and only the installer modules it imports. It still expects the NixOS installer environment to provide system install tools such as `parted`, `zfs`, `zpool`, `cryptsetup`, `nixos-generate-config`, and `nixos-install`.
|
||||||
|
|||||||
@@ -23,7 +23,10 @@
|
|||||||
boot = {
|
boot = {
|
||||||
tmp.useTmpfs = true;
|
tmp.useTmpfs = true;
|
||||||
kernelPackages = lib.mkDefault pkgs.linuxPackages_6_12;
|
kernelPackages = lib.mkDefault pkgs.linuxPackages_6_12;
|
||||||
zfs.package = lib.mkDefault pkgs.zfs_2_4;
|
zfs = {
|
||||||
|
package = lib.mkDefault pkgs.zfs_2_4;
|
||||||
|
forceImportRoot = lib.mkDefault false;
|
||||||
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
hardware.enableRedistributableFirmware = true;
|
hardware.enableRedistributableFirmware = true;
|
||||||
@@ -37,10 +40,17 @@
|
|||||||
|
|
||||||
nixpkgs = {
|
nixpkgs = {
|
||||||
overlays = builtins.attrValues outputs.overlays;
|
overlays = builtins.attrValues outputs.overlays;
|
||||||
config.allowUnfree = true;
|
config = {
|
||||||
|
allowUnfree = true;
|
||||||
|
permittedInsecurePackages = [
|
||||||
|
"openssl-1.1.1w" # This is for discord-canary
|
||||||
|
];
|
||||||
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
services = {
|
services = {
|
||||||
|
dbus.implementation = "dbus";
|
||||||
|
|
||||||
# firmware update
|
# firmware update
|
||||||
fwupd.enable = true;
|
fwupd.enable = true;
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,256 @@
|
|||||||
|
{
|
||||||
|
config,
|
||||||
|
lib,
|
||||||
|
pkgs,
|
||||||
|
...
|
||||||
|
}:
|
||||||
|
let
|
||||||
|
monitoringInterface = "ztwfunumly";
|
||||||
|
nodeTextfileDir = "/var/lib/prometheus-node-exporter-textfile";
|
||||||
|
|
||||||
|
mkProcessNameTemplate =
|
||||||
|
perPid: template: if perPid then "${template}:{{.PID}}:{{.StartTime}}" else template;
|
||||||
|
|
||||||
|
mkProcessMatchers = perPid: [
|
||||||
|
{
|
||||||
|
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Module}}";
|
||||||
|
cmdline = [ "^/nix/store[^ ]*/bin/python[^ ]* -m (?P<Module>[^ ]+)" ];
|
||||||
|
}
|
||||||
|
{
|
||||||
|
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||||
|
cmdline = [
|
||||||
|
"^/nix/store[^ ]*/bin/python[^ ]* /nix/store[^ ]*/bin/\\.?(?P<Wrapped>[^ /]+?)(?:-wrapped)?(?:\\s|$)"
|
||||||
|
];
|
||||||
|
}
|
||||||
|
{
|
||||||
|
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||||
|
cmdline = [
|
||||||
|
"^/nix/store[^ ]*/bin/node /nix/store[^ ]*-(?P<Wrapped>[A-Za-z0-9._+-]+)-[0-9][^ /]*/"
|
||||||
|
];
|
||||||
|
}
|
||||||
|
{
|
||||||
|
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||||
|
cmdline = [ "^/nix/store[^ ]*/(?:bin/|lib/[^ ]*/)?\\.?(?P<Wrapped>[^ /]+?)(?:-wrapped)?(?:\\s|$)" ];
|
||||||
|
}
|
||||||
|
{
|
||||||
|
name = mkProcessNameTemplate perPid "{{.Username}}:{{.ExeBase}}";
|
||||||
|
cmdline = [ ".+" ];
|
||||||
|
}
|
||||||
|
];
|
||||||
|
|
||||||
|
perPidConfig = pkgs.writeText "process-exporter-per-pid.yaml" (
|
||||||
|
builtins.toJSON {
|
||||||
|
process_names = mkProcessMatchers true;
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
zpoolLatencyScript = pkgs.writeShellScript "zpool-latency-exporter" ''
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
out_dir=${lib.escapeShellArg nodeTextfileDir}
|
||||||
|
host=${lib.escapeShellArg config.networking.hostName}
|
||||||
|
tmp_file="$(mktemp "$out_dir/zpool.prom.XXXXXX")"
|
||||||
|
trap 'rm -f "$tmp_file"' EXIT
|
||||||
|
|
||||||
|
pools="$(zpool list -H -o name | paste -sd, -)"
|
||||||
|
|
||||||
|
cat >"$tmp_file" <<'EOF'
|
||||||
|
# HELP zpool_iostat_total_wait_read_ns Average total read wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_total_wait_read_ns gauge
|
||||||
|
# HELP zpool_iostat_total_wait_write_ns Average total write wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_total_wait_write_ns gauge
|
||||||
|
# HELP zpool_iostat_disk_wait_read_ns Average disk read wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_disk_wait_read_ns gauge
|
||||||
|
# HELP zpool_iostat_disk_wait_write_ns Average disk write wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_disk_wait_write_ns gauge
|
||||||
|
# HELP zpool_iostat_syncq_wait_read_ns Average synchronous queue read wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_syncq_wait_read_ns gauge
|
||||||
|
# HELP zpool_iostat_syncq_wait_write_ns Average synchronous queue write wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_syncq_wait_write_ns gauge
|
||||||
|
# HELP zpool_iostat_asyncq_wait_read_ns Average asynchronous queue read wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_asyncq_wait_read_ns gauge
|
||||||
|
# HELP zpool_iostat_asyncq_wait_write_ns Average asynchronous queue write wait time reported by zpool iostat.
|
||||||
|
# TYPE zpool_iostat_asyncq_wait_write_ns gauge
|
||||||
|
EOF
|
||||||
|
|
||||||
|
zpool iostat -Hplvy -y 1 1 | awk -F '\t' -v host="$host" -v pools="$pools" '
|
||||||
|
function esc(str, out) {
|
||||||
|
out = str
|
||||||
|
gsub(/\\/, "\\\\", out)
|
||||||
|
gsub(/"/, "\\\"", out)
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
function emit(metric, pool, vdev, value) {
|
||||||
|
if (value == "" || value == "-") {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
printf "%s{host=\"%s\",pool=\"%s\",vdev=\"%s\"} %s\n",
|
||||||
|
metric,
|
||||||
|
esc(host),
|
||||||
|
esc(pool),
|
||||||
|
esc(vdev),
|
||||||
|
value
|
||||||
|
}
|
||||||
|
|
||||||
|
BEGIN {
|
||||||
|
split(pools, pool_names, ",")
|
||||||
|
for (idx in pool_names) {
|
||||||
|
if (pool_names[idx] != "") {
|
||||||
|
known_pools[pool_names[idx]] = 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
NF == 0 {
|
||||||
|
next
|
||||||
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
row_name = $1
|
||||||
|
|
||||||
|
if (row_name in known_pools) {
|
||||||
|
current_pool = row_name
|
||||||
|
current_vdev = "_pool"
|
||||||
|
} else if (current_pool == "") {
|
||||||
|
next
|
||||||
|
} else {
|
||||||
|
current_vdev = row_name
|
||||||
|
}
|
||||||
|
|
||||||
|
emit("zpool_iostat_total_wait_read_ns", current_pool, current_vdev, $8)
|
||||||
|
emit("zpool_iostat_total_wait_write_ns", current_pool, current_vdev, $9)
|
||||||
|
emit("zpool_iostat_disk_wait_read_ns", current_pool, current_vdev, $10)
|
||||||
|
emit("zpool_iostat_disk_wait_write_ns", current_pool, current_vdev, $11)
|
||||||
|
emit("zpool_iostat_syncq_wait_read_ns", current_pool, current_vdev, $12)
|
||||||
|
emit("zpool_iostat_syncq_wait_write_ns", current_pool, current_vdev, $13)
|
||||||
|
emit("zpool_iostat_asyncq_wait_read_ns", current_pool, current_vdev, $14)
|
||||||
|
emit("zpool_iostat_asyncq_wait_write_ns", current_pool, current_vdev, $15)
|
||||||
|
}
|
||||||
|
' >>"$tmp_file"
|
||||||
|
|
||||||
|
mv "$tmp_file" "$out_dir/zpool.prom"
|
||||||
|
trap - EXIT
|
||||||
|
'';
|
||||||
|
in
|
||||||
|
{
|
||||||
|
networking.firewall.interfaces.${monitoringInterface}.allowedTCPPorts = [
|
||||||
|
9100
|
||||||
|
9134
|
||||||
|
9256
|
||||||
|
9257
|
||||||
|
9633
|
||||||
|
];
|
||||||
|
|
||||||
|
services.prometheus.exporters = {
|
||||||
|
node = {
|
||||||
|
enable = true;
|
||||||
|
enabledCollectors = [
|
||||||
|
"pressure"
|
||||||
|
"processes"
|
||||||
|
"systemd"
|
||||||
|
];
|
||||||
|
extraFlags = [ "--collector.textfile.directory=${nodeTextfileDir}" ];
|
||||||
|
};
|
||||||
|
|
||||||
|
process = {
|
||||||
|
enable = true;
|
||||||
|
user = "root";
|
||||||
|
group = "root";
|
||||||
|
settings.process_names = mkProcessMatchers false;
|
||||||
|
extraFlags = [
|
||||||
|
"-gather-smaps=false"
|
||||||
|
"-remove-empty-groups=true"
|
||||||
|
"-threads=false"
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
smartctl.enable = true;
|
||||||
|
zfs.enable = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
programs.atop = {
|
||||||
|
enable = true;
|
||||||
|
atopService.enable = true;
|
||||||
|
atopRotateTimer.enable = true;
|
||||||
|
atopacctService.enable = true;
|
||||||
|
settings.interval = 30;
|
||||||
|
};
|
||||||
|
|
||||||
|
systemd = {
|
||||||
|
services = {
|
||||||
|
prometheus-process-pid-exporter = {
|
||||||
|
description = "Prometheus process exporter with per-PID naming";
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
after = [ "network.target" ];
|
||||||
|
serviceConfig = {
|
||||||
|
ExecStart = ''
|
||||||
|
${pkgs.prometheus-process-exporter}/bin/process-exporter \
|
||||||
|
--web.listen-address 0.0.0.0:9257 \
|
||||||
|
--config.path ${perPidConfig} \
|
||||||
|
-children=false \
|
||||||
|
-gather-smaps=false \
|
||||||
|
-remove-empty-groups=true \
|
||||||
|
-threads=false
|
||||||
|
'';
|
||||||
|
User = "root";
|
||||||
|
Group = "root";
|
||||||
|
Restart = "always";
|
||||||
|
WorkingDirectory = "/tmp";
|
||||||
|
CapabilityBoundingSet = [ "" ];
|
||||||
|
DeviceAllow = [ "" ];
|
||||||
|
LockPersonality = true;
|
||||||
|
MemoryDenyWriteExecute = true;
|
||||||
|
NoNewPrivileges = true;
|
||||||
|
PrivateDevices = true;
|
||||||
|
PrivateTmp = true;
|
||||||
|
ProtectClock = true;
|
||||||
|
ProtectControlGroups = true;
|
||||||
|
ProtectHome = true;
|
||||||
|
ProtectHostname = true;
|
||||||
|
ProtectKernelLogs = true;
|
||||||
|
ProtectKernelModules = true;
|
||||||
|
ProtectKernelTunables = true;
|
||||||
|
ProtectSystem = "strict";
|
||||||
|
RemoveIPC = true;
|
||||||
|
RestrictAddressFamilies = [
|
||||||
|
"AF_INET"
|
||||||
|
"AF_INET6"
|
||||||
|
];
|
||||||
|
RestrictNamespaces = true;
|
||||||
|
RestrictRealtime = true;
|
||||||
|
RestrictSUIDSGID = true;
|
||||||
|
SystemCallArchitectures = "native";
|
||||||
|
UMask = "0077";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
zpool-latency-exporter = {
|
||||||
|
description = "Exports ZFS latency metrics for node_exporter textfile collection";
|
||||||
|
after = [ "zfs-import.target" ];
|
||||||
|
requires = [ "zfs-import.target" ];
|
||||||
|
path = [
|
||||||
|
config.boot.zfs.package
|
||||||
|
pkgs.coreutils
|
||||||
|
pkgs.gawk
|
||||||
|
];
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
ExecStart = zpoolLatencyScript;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
timers.zpool-latency-exporter = {
|
||||||
|
wantedBy = [ "timers.target" ];
|
||||||
|
timerConfig = {
|
||||||
|
OnBootSec = "2m";
|
||||||
|
OnUnitActiveSec = "60s";
|
||||||
|
Unit = "zpool-latency-exporter.service";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
tmpfiles.rules = [ "d ${nodeTextfileDir} 0755 root root - -" ];
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -4,7 +4,7 @@
|
|||||||
flags = [ "--accept-flake-config" ];
|
flags = [ "--accept-flake-config" ];
|
||||||
randomizedDelaySec = "1h";
|
randomizedDelaySec = "1h";
|
||||||
persistent = true;
|
persistent = true;
|
||||||
flake = "github:RichieCahill/dotfiles";
|
flake = "git+https://gitea.tmmworkshop.com/richie/dotfiles?ref=main";
|
||||||
allowReboot = true;
|
allowReboot = true;
|
||||||
dates = "Sat *-*-* 06:00:00";
|
dates = "Sat *-*-* 06:00:00";
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -0,0 +1,76 @@
|
|||||||
|
# ZFS failed root import recovery
|
||||||
|
|
||||||
|
## Fast path
|
||||||
|
|
||||||
|
If the machine fails to boot because ZFS refuses to import `root_pool`:
|
||||||
|
|
||||||
|
### GRUB
|
||||||
|
|
||||||
|
1. At the bootloader menu, select the normal NixOS entry.
|
||||||
|
2. Press `e`.
|
||||||
|
3. Find the line that starts with `linux`.
|
||||||
|
4. Append this to the end of that line:
|
||||||
|
|
||||||
|
```text
|
||||||
|
zfs_force=1
|
||||||
|
```
|
||||||
|
|
||||||
|
5. Boot once with `Ctrl+x` or `F10`.
|
||||||
|
|
||||||
|
### systemd-boot
|
||||||
|
|
||||||
|
1. At the bootloader menu, highlight the normal NixOS entry.
|
||||||
|
2. Press `e`.
|
||||||
|
3. Append this to the end of the options line:
|
||||||
|
|
||||||
|
```text
|
||||||
|
zfs_force=1
|
||||||
|
```
|
||||||
|
|
||||||
|
4. Press `Enter` to boot once.
|
||||||
|
|
||||||
|
## After boot
|
||||||
|
|
||||||
|
Run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo zpool status
|
||||||
|
sudo zpool import
|
||||||
|
journalctl -b | rg "ZFS|zfs|import|root_pool"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Expected result
|
||||||
|
|
||||||
|
`sudo zpool status` should show `root_pool` as `ONLINE`.
|
||||||
|
|
||||||
|
## Reboot test
|
||||||
|
|
||||||
|
Run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo reboot
|
||||||
|
```
|
||||||
|
|
||||||
|
Do not add `zfs_force=1` the second time.
|
||||||
|
|
||||||
|
## If it still fails
|
||||||
|
|
||||||
|
Boot once more with:
|
||||||
|
|
||||||
|
```text
|
||||||
|
zfs_force=1
|
||||||
|
```
|
||||||
|
|
||||||
|
Then run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo zpool status -v
|
||||||
|
sudo zpool history | tail -n 50
|
||||||
|
journalctl -b | rg "ZFS|zfs|import|root_pool"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- Root pool name is `root_pool`.
|
||||||
|
- This is a one-time recovery path after disk moves, controller changes, dirty exports, or interrupted imports.
|
||||||
|
- Some hosts also need the LUKS unlock USB key inserted before boot.
|
||||||
File diff suppressed because one or more lines are too long
Generated
+42
-26
@@ -8,11 +8,11 @@
|
|||||||
},
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"dir": "pkgs/firefox-addons",
|
"dir": "pkgs/firefox-addons",
|
||||||
"lastModified": 1777435375,
|
"lastModified": 1782964936,
|
||||||
"narHash": "sha256-2WRfJbipnTz+EY3rHRnCoG4kWkzPczb/cLcWwhy/0QA=",
|
"narHash": "sha256-wXEBDr7/dFQYhVpDwCKc9fkrYQQE4x0bdirX1bsLBGA=",
|
||||||
"owner": "rycee",
|
"owner": "rycee",
|
||||||
"repo": "nur-expressions",
|
"repo": "nur-expressions",
|
||||||
"rev": "4d89e8e2c50711ee3fea3a25e662cfa5c6628e07",
|
"rev": "64feee871e0373dd6121e412c3fb12e372d1bfb5",
|
||||||
"type": "gitlab"
|
"type": "gitlab"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -29,11 +29,11 @@
|
|||||||
]
|
]
|
||||||
},
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1777434174,
|
"lastModified": 1783005591,
|
||||||
"narHash": "sha256-KwTyQ5g2qDhWIs/O6vH8HeF8n4JCzZIT/VYE7nYnukQ=",
|
"narHash": "sha256-NcLHV5uBAeggDUE2wPbKszjfyaSLsoqaYt7izOphkZw=",
|
||||||
"owner": "nix-community",
|
"owner": "nix-community",
|
||||||
"repo": "home-manager",
|
"repo": "home-manager",
|
||||||
"rev": "d3b4e4b1bd59aedd3d4eb0a8df7162edb6da4607",
|
"rev": "f469c79b955609d6a8fdd9e689be76a93b1621d7",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -43,12 +43,15 @@
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
"nixos-hardware": {
|
"nixos-hardware": {
|
||||||
|
"inputs": {
|
||||||
|
"nixpkgs": "nixpkgs"
|
||||||
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1776983936,
|
"lastModified": 1782562157,
|
||||||
"narHash": "sha256-ZOQyNqSvJ8UdrrqU1p7vaFcdL53idK+LOM8oRWEWh6o=",
|
"narHash": "sha256-a7+T6QSeowynwZ1ZJJbP8T8ntAytvrui8kFGJmIZt2c=",
|
||||||
"owner": "nixos",
|
"owner": "nixos",
|
||||||
"repo": "nixos-hardware",
|
"repo": "nixos-hardware",
|
||||||
"rev": "2096f3f411ce46e88a79ae4eafcfc9df8ed41c61",
|
"rev": "a9cf7546a938c737b079e738de73934a13de9784",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -60,27 +63,24 @@
|
|||||||
},
|
},
|
||||||
"nixpkgs": {
|
"nixpkgs": {
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1777268161,
|
"lastModified": 1767892417,
|
||||||
"narHash": "sha256-bxrdOn8SCOv8tN4JbTF/TXq7kjo9ag4M+C8yzzIRYbE=",
|
"narHash": "sha256-8bW3q88CEg2u4hSP66Vf4lpbLonHz7hqDNBMcCY7E9U=",
|
||||||
"owner": "nixos",
|
"rev": "3497aa5c9457a9d88d71fa93a4a8368816fbeeba",
|
||||||
"repo": "nixpkgs",
|
"type": "tarball",
|
||||||
"rev": "1c3fe55ad329cbcb28471bb30f05c9827f724c76",
|
"url": "https://releases.nixos.org/nixos/unstable/nixos-26.05pre924538.3497aa5c9457/nixexprs.tar.xz"
|
||||||
"type": "github"
|
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
"owner": "nixos",
|
"type": "tarball",
|
||||||
"ref": "nixos-unstable",
|
"url": "https://channels.nixos.org/nixos-unstable/nixexprs.tar.xz"
|
||||||
"repo": "nixpkgs",
|
|
||||||
"type": "github"
|
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"nixpkgs-master": {
|
"nixpkgs-master": {
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1777437048,
|
"lastModified": 1783021952,
|
||||||
"narHash": "sha256-Ca4jKXJuYp1D+DqiuQ/vGHRYKPlAZTn1vq7XDU9t18w=",
|
"narHash": "sha256-8PghAtSGGZ0umfVI8Qbd7ZbFrfZPiH1UwtVbgLeikDA=",
|
||||||
"owner": "nixos",
|
"owner": "nixos",
|
||||||
"repo": "nixpkgs",
|
"repo": "nixpkgs",
|
||||||
"rev": "1e1459dda883651ef85e23c7c6e2224cba195065",
|
"rev": "f136374c679c54171a3ace589d15e9e79a8bd086",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -106,12 +106,28 @@
|
|||||||
"type": "github"
|
"type": "github"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
"nixpkgs_2": {
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1782723713,
|
||||||
|
"narHash": "sha256-oPXCU/SSUokcGaJREHibG1CBX3+s/W7orDWQOZDsEeQ=",
|
||||||
|
"owner": "nixos",
|
||||||
|
"repo": "nixpkgs",
|
||||||
|
"rev": "b5aa0fbd538984f6e3d201be0005b4463d8b09f8",
|
||||||
|
"type": "github"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"owner": "nixos",
|
||||||
|
"ref": "nixos-unstable",
|
||||||
|
"repo": "nixpkgs",
|
||||||
|
"type": "github"
|
||||||
|
}
|
||||||
|
},
|
||||||
"root": {
|
"root": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"firefox-addons": "firefox-addons",
|
"firefox-addons": "firefox-addons",
|
||||||
"home-manager": "home-manager",
|
"home-manager": "home-manager",
|
||||||
"nixos-hardware": "nixos-hardware",
|
"nixos-hardware": "nixos-hardware",
|
||||||
"nixpkgs": "nixpkgs",
|
"nixpkgs": "nixpkgs_2",
|
||||||
"nixpkgs-master": "nixpkgs-master",
|
"nixpkgs-master": "nixpkgs-master",
|
||||||
"nixpkgs-stable": "nixpkgs-stable",
|
"nixpkgs-stable": "nixpkgs-stable",
|
||||||
"sops-nix": "sops-nix",
|
"sops-nix": "sops-nix",
|
||||||
@@ -125,11 +141,11 @@
|
|||||||
]
|
]
|
||||||
},
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1777338324,
|
"lastModified": 1782165805,
|
||||||
"narHash": "sha256-bc+ZZCmOTNq86/svGnw0tVpH7vJaLYvGLLKFYP08Q8E=",
|
"narHash": "sha256-478kKQBvK6SYTOdN2h9jhKJv94nbXRbFMfuL1WshErg=",
|
||||||
"owner": "Mic92",
|
"owner": "Mic92",
|
||||||
"repo": "sops-nix",
|
"repo": "sops-nix",
|
||||||
"rev": "8eaee5c45428b28b8c47a83e4c09dccec5f279b5",
|
"rev": "56b24064fdcaedca53553b1a6d607fd23b613a24",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
|
|||||||
@@ -65,38 +65,48 @@
|
|||||||
|
|
||||||
devShells = forEachSystem (pkgs: import ./shell.nix { inherit pkgs; });
|
devShells = forEachSystem (pkgs: import ./shell.nix { inherit pkgs; });
|
||||||
formatter = forEachSystem (pkgs: pkgs.treefmt);
|
formatter = forEachSystem (pkgs: pkgs.treefmt);
|
||||||
|
packages = forEachSystem (
|
||||||
|
pkgs:
|
||||||
|
let
|
||||||
|
installer = pkgs.callPackage ./python/installer/package.nix { };
|
||||||
|
installer-nixos = pkgs.callPackage ./python/installer/package.nix { patchElf = false; };
|
||||||
|
in
|
||||||
|
{
|
||||||
|
inherit installer installer-nixos;
|
||||||
|
default = installer;
|
||||||
|
}
|
||||||
|
// lib.optionalAttrs (pkgs.stdenv.hostPlatform.system == "x86_64-linux") {
|
||||||
|
iso = self.nixosConfigurations.iso.config.system.build.isoImage;
|
||||||
|
}
|
||||||
|
);
|
||||||
|
apps = forEachSystem (
|
||||||
|
pkgs:
|
||||||
|
let
|
||||||
|
system = pkgs.stdenv.hostPlatform.system;
|
||||||
|
installer = {
|
||||||
|
type = "app";
|
||||||
|
program = "${self.packages.${system}.installer}/bin/nixos-installer";
|
||||||
|
meta.description = "One-file NixOS ZFS installer.";
|
||||||
|
};
|
||||||
|
in
|
||||||
|
{
|
||||||
|
inherit installer;
|
||||||
|
default = installer;
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
nixosConfigurations = {
|
nixosConfigurations =
|
||||||
bob = lib.nixosSystem {
|
let
|
||||||
modules = [
|
hosts = builtins.attrNames (
|
||||||
./systems/bob
|
lib.filterAttrs (_: type: type == "directory") (builtins.readDir ./systems)
|
||||||
];
|
);
|
||||||
|
mkHost =
|
||||||
|
name:
|
||||||
|
lib.nixosSystem {
|
||||||
|
modules = [ ./systems/${name} ];
|
||||||
specialArgs = { inherit inputs outputs; };
|
specialArgs = { inherit inputs outputs; };
|
||||||
};
|
};
|
||||||
brain = lib.nixosSystem {
|
in
|
||||||
modules = [
|
lib.genAttrs hosts mkHost;
|
||||||
./systems/brain
|
|
||||||
];
|
|
||||||
specialArgs = { inherit inputs outputs; };
|
|
||||||
};
|
|
||||||
jeeves = lib.nixosSystem {
|
|
||||||
modules = [
|
|
||||||
./systems/jeeves
|
|
||||||
];
|
|
||||||
specialArgs = { inherit inputs outputs; };
|
|
||||||
};
|
|
||||||
rhapsody-in-green = lib.nixosSystem {
|
|
||||||
modules = [
|
|
||||||
./systems/rhapsody-in-green
|
|
||||||
];
|
|
||||||
specialArgs = { inherit inputs outputs; };
|
|
||||||
};
|
|
||||||
leviathan = lib.nixosSystem {
|
|
||||||
modules = [
|
|
||||||
./systems/leviathan
|
|
||||||
];
|
|
||||||
specialArgs = { inherit inputs outputs; };
|
|
||||||
};
|
|
||||||
};
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,24 +0,0 @@
|
|||||||
# Logs
|
|
||||||
logs
|
|
||||||
*.log
|
|
||||||
npm-debug.log*
|
|
||||||
yarn-debug.log*
|
|
||||||
yarn-error.log*
|
|
||||||
pnpm-debug.log*
|
|
||||||
lerna-debug.log*
|
|
||||||
|
|
||||||
node_modules
|
|
||||||
dist
|
|
||||||
dist-ssr
|
|
||||||
*.local
|
|
||||||
|
|
||||||
# Editor directories and files
|
|
||||||
.vscode/*
|
|
||||||
!.vscode/extensions.json
|
|
||||||
.idea
|
|
||||||
.DS_Store
|
|
||||||
*.suo
|
|
||||||
*.ntvs*
|
|
||||||
*.njsproj
|
|
||||||
*.sln
|
|
||||||
*.sw?
|
|
||||||
@@ -17,17 +17,17 @@
|
|||||||
|
|
||||||
python-env = final: _prev: {
|
python-env = final: _prev: {
|
||||||
my_python = final.python314.withPackages (
|
my_python = final.python314.withPackages (
|
||||||
ps: with ps; [
|
ps:
|
||||||
|
with ps;
|
||||||
|
[
|
||||||
alembic
|
alembic
|
||||||
apprise
|
apprise
|
||||||
apscheduler
|
apscheduler
|
||||||
fastapi
|
fastapi
|
||||||
fastapi-cli
|
fastapi-cli
|
||||||
faster-whisper
|
|
||||||
httpx
|
httpx
|
||||||
mypy
|
mypy
|
||||||
orjson
|
pgvector
|
||||||
polars
|
|
||||||
psycopg
|
psycopg
|
||||||
pydantic
|
pydantic
|
||||||
pyfakefs
|
pyfakefs
|
||||||
@@ -37,12 +37,8 @@
|
|||||||
pytest-xdist
|
pytest-xdist
|
||||||
python-multipart
|
python-multipart
|
||||||
ruff
|
ruff
|
||||||
scalene
|
|
||||||
sqlalchemy
|
|
||||||
sqlalchemy
|
sqlalchemy
|
||||||
tenacity
|
tenacity
|
||||||
textual
|
|
||||||
tiktoken
|
|
||||||
tinytuya
|
tinytuya
|
||||||
typer
|
typer
|
||||||
websockets
|
websockets
|
||||||
|
|||||||
+20
-9
@@ -3,7 +3,7 @@ name = "system_tools"
|
|||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
description = ""
|
description = ""
|
||||||
authors = [{ name = "Richie Cahill", email = "richie@tmmworkshop.com" }]
|
authors = [{ name = "Richie Cahill", email = "richie@tmmworkshop.com" }]
|
||||||
requires-python = "~=3.13.0"
|
requires-python = "~=3.14.0"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
# these dependencies are a best effort and aren't guaranteed to work
|
# these dependencies are a best effort and aren't guaranteed to work
|
||||||
@@ -12,26 +12,39 @@ dependencies = [
|
|||||||
"alembic",
|
"alembic",
|
||||||
"apprise",
|
"apprise",
|
||||||
"apscheduler",
|
"apscheduler",
|
||||||
|
"beautifulsoup4",
|
||||||
|
"bm25s",
|
||||||
|
"ebooklib",
|
||||||
|
"fastapi",
|
||||||
|
"fastapi-cli",
|
||||||
"httpx",
|
"httpx",
|
||||||
"python-multipart",
|
"jinja2",
|
||||||
|
"pgvector",
|
||||||
"polars",
|
"polars",
|
||||||
"psycopg[binary]",
|
"psycopg[binary]",
|
||||||
"pydantic",
|
"pydantic",
|
||||||
"pyyaml",
|
"pydantic-settings",
|
||||||
"sqlalchemy",
|
"python-multipart",
|
||||||
|
"sqlalchemy[asyncio]",
|
||||||
|
"tenacity",
|
||||||
|
"tiktoken",
|
||||||
|
"tinytuya",
|
||||||
"typer",
|
"typer",
|
||||||
|
"uvicorn",
|
||||||
"websockets",
|
"websockets",
|
||||||
|
"yake",
|
||||||
]
|
]
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
database = "python.database_cli:app"
|
database = "python.database_cli:app"
|
||||||
van-inventory = "python.van_inventory.main:serve"
|
|
||||||
whisper-transcribe = "python.tools.whisper.transcribe:main"
|
whisper-transcribe = "python.tools.whisper.transcribe:main"
|
||||||
|
|
||||||
[dependency-groups]
|
[dependency-groups]
|
||||||
dev = [
|
dev = [
|
||||||
|
"aiosqlite",
|
||||||
"mypy",
|
"mypy",
|
||||||
"pyfakefs",
|
"pyfakefs",
|
||||||
|
"pytest-asyncio",
|
||||||
"pytest-cov",
|
"pytest-cov",
|
||||||
"pytest-mock",
|
"pytest-mock",
|
||||||
"pytest-xdist",
|
"pytest-xdist",
|
||||||
@@ -41,7 +54,7 @@ dev = [
|
|||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
|
|
||||||
target-version = "py313"
|
target-version = "py314"
|
||||||
|
|
||||||
line-length = 120
|
line-length = 120
|
||||||
|
|
||||||
@@ -84,9 +97,6 @@ lint.ignore = [
|
|||||||
"python/alembic/**" = [
|
"python/alembic/**" = [
|
||||||
"INP001", # (perm) this creates LSP issues for alembic
|
"INP001", # (perm) this creates LSP issues for alembic
|
||||||
]
|
]
|
||||||
"python/signal_bot/**" = [
|
|
||||||
"D107", # (perm) class docstrings cover __init__
|
|
||||||
]
|
|
||||||
|
|
||||||
[tool.ruff.lint.pydocstyle]
|
[tool.ruff.lint.pydocstyle]
|
||||||
convention = "google"
|
convention = "google"
|
||||||
@@ -110,5 +120,6 @@ exclude_lines = [
|
|||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
addopts = "-n auto -ra"
|
addopts = "-n auto -ra"
|
||||||
|
asyncio_mode = "auto"
|
||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
# --cov=system_tools --cov-report=term-missing --cov-report=xml --cov-report=html --cov-branch
|
# --cov=system_tools --cov-report=term-missing --cov-report=xml --cov-report=html --cov-branch
|
||||||
|
|||||||
-1417
File diff suppressed because it is too large
Load Diff
-50
@@ -1,50 +0,0 @@
|
|||||||
"""adding FailedIngestion.
|
|
||||||
|
|
||||||
Revision ID: 2f43120e3ffc
|
|
||||||
Revises: f99be864fe69
|
|
||||||
Create Date: 2026-03-24 23:46:17.277897
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from alembic import op
|
|
||||||
|
|
||||||
from python.orm import DataScienceDevBase
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "2f43120e3ffc"
|
|
||||||
down_revision: str | None = "f99be864fe69"
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = DataScienceDevBase.schema_name
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Upgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.create_table(
|
|
||||||
"failed_ingestion",
|
|
||||||
sa.Column("raw_line", sa.Text(), nullable=False),
|
|
||||||
sa.Column("error", sa.Text(), nullable=False),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_failed_ingestion")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Downgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.drop_table("failed_ingestion", schema=schema)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
-2770
File diff suppressed because it is too large
Load Diff
-1391
File diff suppressed because it is too large
Load Diff
@@ -1,72 +0,0 @@
|
|||||||
"""Attach all partition tables to the posts parent table.
|
|
||||||
|
|
||||||
Alembic autogenerate creates partition tables as standalone tables but does not
|
|
||||||
emit the ALTER TABLE ... ATTACH PARTITION statements needed for PostgreSQL to
|
|
||||||
route inserts to the correct partition.
|
|
||||||
|
|
||||||
Revision ID: a1b2c3d4e5f6
|
|
||||||
Revises: 605b1794838f
|
|
||||||
Create Date: 2026-03-25 10:00:00.000000
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
from alembic import op
|
|
||||||
from sqlalchemy import text
|
|
||||||
|
|
||||||
from python.orm import DataScienceDevBase
|
|
||||||
from python.orm.data_science_dev.posts.partitions import (
|
|
||||||
PARTITION_END_YEAR,
|
|
||||||
PARTITION_START_YEAR,
|
|
||||||
iso_weeks_in_year,
|
|
||||||
week_bounds,
|
|
||||||
)
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "a1b2c3d4e5f6"
|
|
||||||
down_revision: str | None = "605b1794838f"
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = DataScienceDevBase.schema_name
|
|
||||||
|
|
||||||
ALREADY_ATTACHED_QUERY = text("""
|
|
||||||
SELECT inhrelid::regclass::text
|
|
||||||
FROM pg_inherits
|
|
||||||
WHERE inhparent = :parent::regclass
|
|
||||||
""")
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Attach all weekly partition tables to the posts parent table."""
|
|
||||||
connection = op.get_bind()
|
|
||||||
already_attached = {row[0] for row in connection.execute(ALREADY_ATTACHED_QUERY, {"parent": f"{schema}.posts"})}
|
|
||||||
|
|
||||||
for year in range(PARTITION_START_YEAR, PARTITION_END_YEAR + 1):
|
|
||||||
for week in range(1, iso_weeks_in_year(year) + 1):
|
|
||||||
table_name = f"posts_{year}_{week:02d}"
|
|
||||||
qualified_name = f"{schema}.{table_name}"
|
|
||||||
if qualified_name in already_attached:
|
|
||||||
continue
|
|
||||||
start, end = week_bounds(year, week)
|
|
||||||
start_str = start.strftime("%Y-%m-%d %H:%M:%S")
|
|
||||||
end_str = end.strftime("%Y-%m-%d %H:%M:%S")
|
|
||||||
op.execute(
|
|
||||||
f"ALTER TABLE {schema}.posts "
|
|
||||||
f"ATTACH PARTITION {qualified_name} "
|
|
||||||
f"FOR VALUES FROM ('{start_str}') TO ('{end_str}')"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Detach all weekly partition tables from the posts parent table."""
|
|
||||||
for year in range(PARTITION_START_YEAR, PARTITION_END_YEAR + 1):
|
|
||||||
for week in range(1, iso_weeks_in_year(year) + 1):
|
|
||||||
table_name = f"posts_{year}_{week:02d}"
|
|
||||||
op.execute(f"ALTER TABLE {schema}.posts DETACH PARTITION {schema}.{table_name}")
|
|
||||||
-153
@@ -1,153 +0,0 @@
|
|||||||
"""adding congress data.
|
|
||||||
|
|
||||||
Revision ID: 83bfc8af92d8
|
|
||||||
Revises: a1b2c3d4e5f6
|
|
||||||
Create Date: 2026-03-27 10:43:02.324510
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from alembic import op
|
|
||||||
|
|
||||||
from python.orm import DataScienceDevBase
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "83bfc8af92d8"
|
|
||||||
down_revision: str | None = "a1b2c3d4e5f6"
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = DataScienceDevBase.schema_name
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Upgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.create_table(
|
|
||||||
"bill",
|
|
||||||
sa.Column("congress", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("bill_type", sa.String(), nullable=False),
|
|
||||||
sa.Column("number", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("title", sa.String(), nullable=True),
|
|
||||||
sa.Column("title_short", sa.String(), nullable=True),
|
|
||||||
sa.Column("official_title", sa.String(), nullable=True),
|
|
||||||
sa.Column("status", sa.String(), nullable=True),
|
|
||||||
sa.Column("status_at", sa.Date(), nullable=True),
|
|
||||||
sa.Column("sponsor_bioguide_id", sa.String(), nullable=True),
|
|
||||||
sa.Column("subjects_top_term", sa.String(), nullable=True),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_bill")),
|
|
||||||
sa.UniqueConstraint("congress", "bill_type", "number", name="uq_bill_congress_type_number"),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_index("ix_bill_congress", "bill", ["congress"], unique=False, schema=schema)
|
|
||||||
op.create_table(
|
|
||||||
"legislator",
|
|
||||||
sa.Column("bioguide_id", sa.Text(), nullable=False),
|
|
||||||
sa.Column("thomas_id", sa.String(), nullable=True),
|
|
||||||
sa.Column("lis_id", sa.String(), nullable=True),
|
|
||||||
sa.Column("govtrack_id", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("opensecrets_id", sa.String(), nullable=True),
|
|
||||||
sa.Column("fec_ids", sa.String(), nullable=True),
|
|
||||||
sa.Column("first_name", sa.String(), nullable=False),
|
|
||||||
sa.Column("last_name", sa.String(), nullable=False),
|
|
||||||
sa.Column("official_full_name", sa.String(), nullable=True),
|
|
||||||
sa.Column("nickname", sa.String(), nullable=True),
|
|
||||||
sa.Column("birthday", sa.Date(), nullable=True),
|
|
||||||
sa.Column("gender", sa.String(), nullable=True),
|
|
||||||
sa.Column("current_party", sa.String(), nullable=True),
|
|
||||||
sa.Column("current_state", sa.String(), nullable=True),
|
|
||||||
sa.Column("current_district", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("current_chamber", sa.String(), nullable=True),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_legislator")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_index(op.f("ix_legislator_bioguide_id"), "legislator", ["bioguide_id"], unique=True, schema=schema)
|
|
||||||
op.create_table(
|
|
||||||
"bill_text",
|
|
||||||
sa.Column("bill_id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("version_code", sa.String(), nullable=False),
|
|
||||||
sa.Column("version_name", sa.String(), nullable=True),
|
|
||||||
sa.Column("text_content", sa.String(), nullable=True),
|
|
||||||
sa.Column("date", sa.Date(), nullable=True),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.ForeignKeyConstraint(
|
|
||||||
["bill_id"], [f"{schema}.bill.id"], name=op.f("fk_bill_text_bill_id_bill"), ondelete="CASCADE"
|
|
||||||
),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_bill_text")),
|
|
||||||
sa.UniqueConstraint("bill_id", "version_code", name="uq_bill_text_bill_id_version_code"),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_table(
|
|
||||||
"vote",
|
|
||||||
sa.Column("congress", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("chamber", sa.String(), nullable=False),
|
|
||||||
sa.Column("session", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("number", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("vote_type", sa.String(), nullable=True),
|
|
||||||
sa.Column("question", sa.String(), nullable=True),
|
|
||||||
sa.Column("result", sa.String(), nullable=True),
|
|
||||||
sa.Column("result_text", sa.String(), nullable=True),
|
|
||||||
sa.Column("vote_date", sa.Date(), nullable=False),
|
|
||||||
sa.Column("yea_count", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("nay_count", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("not_voting_count", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("present_count", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("bill_id", sa.Integer(), nullable=True),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.ForeignKeyConstraint(["bill_id"], [f"{schema}.bill.id"], name=op.f("fk_vote_bill_id_bill")),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_vote")),
|
|
||||||
sa.UniqueConstraint("congress", "chamber", "session", "number", name="uq_vote_congress_chamber_session_number"),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_index("ix_vote_congress_chamber", "vote", ["congress", "chamber"], unique=False, schema=schema)
|
|
||||||
op.create_index("ix_vote_date", "vote", ["vote_date"], unique=False, schema=schema)
|
|
||||||
op.create_table(
|
|
||||||
"vote_record",
|
|
||||||
sa.Column("vote_id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("legislator_id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("position", sa.String(), nullable=False),
|
|
||||||
sa.ForeignKeyConstraint(
|
|
||||||
["legislator_id"],
|
|
||||||
[f"{schema}.legislator.id"],
|
|
||||||
name=op.f("fk_vote_record_legislator_id_legislator"),
|
|
||||||
ondelete="CASCADE",
|
|
||||||
),
|
|
||||||
sa.ForeignKeyConstraint(
|
|
||||||
["vote_id"], [f"{schema}.vote.id"], name=op.f("fk_vote_record_vote_id_vote"), ondelete="CASCADE"
|
|
||||||
),
|
|
||||||
sa.PrimaryKeyConstraint("vote_id", "legislator_id", name=op.f("pk_vote_record")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Downgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.drop_table("vote_record", schema=schema)
|
|
||||||
op.drop_index("ix_vote_date", table_name="vote", schema=schema)
|
|
||||||
op.drop_index("ix_vote_congress_chamber", table_name="vote", schema=schema)
|
|
||||||
op.drop_table("vote", schema=schema)
|
|
||||||
op.drop_table("bill_text", schema=schema)
|
|
||||||
op.drop_index(op.f("ix_legislator_bioguide_id"), table_name="legislator", schema=schema)
|
|
||||||
op.drop_table("legislator", schema=schema)
|
|
||||||
op.drop_index("ix_bill_congress", table_name="bill", schema=schema)
|
|
||||||
op.drop_table("bill", schema=schema)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
-58
@@ -1,58 +0,0 @@
|
|||||||
"""adding LegislatorSocialMedia.
|
|
||||||
|
|
||||||
Revision ID: 5cd7eee3549d
|
|
||||||
Revises: 83bfc8af92d8
|
|
||||||
Create Date: 2026-03-29 11:53:44.224799
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from alembic import op
|
|
||||||
|
|
||||||
from python.orm import DataScienceDevBase
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "5cd7eee3549d"
|
|
||||||
down_revision: str | None = "83bfc8af92d8"
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = DataScienceDevBase.schema_name
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Upgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.create_table(
|
|
||||||
"legislator_social_media",
|
|
||||||
sa.Column("legislator_id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("platform", sa.String(), nullable=False),
|
|
||||||
sa.Column("account_name", sa.String(), nullable=False),
|
|
||||||
sa.Column("url", sa.String(), nullable=True),
|
|
||||||
sa.Column("source", sa.String(), nullable=False),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.ForeignKeyConstraint(
|
|
||||||
["legislator_id"],
|
|
||||||
[f"{schema}.legislator.id"],
|
|
||||||
name=op.f("fk_legislator_social_media_legislator_id_legislator"),
|
|
||||||
),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_legislator_social_media")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Downgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.drop_table("legislator_social_media", schema=schema)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
+93
@@ -0,0 +1,93 @@
|
|||||||
|
"""adding audiobook libreary metadata.
|
||||||
|
|
||||||
|
Revision ID: d7864d1ffc17
|
||||||
|
Revises: c8a794340928
|
||||||
|
Create Date: 2026-06-03 20:24:09.200837
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "d7864d1ffc17"
|
||||||
|
down_revision: str | None = "c8a794340928"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_table(
|
||||||
|
"audiobook_author",
|
||||||
|
sa.Column("name", sa.String(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook_author")),
|
||||||
|
sa.UniqueConstraint("name", name=op.f("uq_audiobook_author_name")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"audiobook_series",
|
||||||
|
sa.Column("name", sa.String(), nullable=False),
|
||||||
|
sa.Column("author_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["author_id"],
|
||||||
|
[f"{schema}.audiobook_author.id"],
|
||||||
|
name=op.f("fk_audiobook_series_author_id_audiobook_author"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook_series")),
|
||||||
|
sa.UniqueConstraint("author_id", "name", name=op.f("uq_audiobook_series_author_id")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"audiobook",
|
||||||
|
sa.Column("title", sa.String(), nullable=False),
|
||||||
|
sa.Column("author_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("series_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("series_index", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["author_id"],
|
||||||
|
[f"{schema}.audiobook_author.id"],
|
||||||
|
name=op.f("fk_audiobook_author_id_audiobook_author"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["series_id"],
|
||||||
|
[f"{schema}.audiobook_series.id"],
|
||||||
|
name=op.f("fk_audiobook_series_id_audiobook_series"),
|
||||||
|
ondelete="SET NULL",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_table("audiobook", schema=schema)
|
||||||
|
op.drop_table("audiobook_series", schema=schema)
|
||||||
|
op.drop_table("audiobook_author", schema=schema)
|
||||||
|
# ### end Alembic commands ###
|
||||||
@@ -0,0 +1,200 @@
|
|||||||
|
"""add ebook search tables.
|
||||||
|
|
||||||
|
Revision ID: 2db132cace1a
|
||||||
|
Revises: b3c60cc5beb5
|
||||||
|
Create Date: 2026-06-10 22:10:54.379159
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import pgvector
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "2db132cace1a"
|
||||||
|
down_revision: str | None = "b3c60cc5beb5"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_table(
|
||||||
|
"ebook_embedding_model",
|
||||||
|
sa.Column("name", sa.String(), nullable=False),
|
||||||
|
sa.Column("dimension", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("is_default", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_embedding_model")),
|
||||||
|
sa.UniqueConstraint("name", name=op.f("uq_ebook_embedding_model_name")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_source",
|
||||||
|
sa.Column("title", sa.String(), nullable=False),
|
||||||
|
sa.Column("author", sa.String(), nullable=True),
|
||||||
|
sa.Column("language", sa.String(), nullable=True),
|
||||||
|
sa.Column("publisher", sa.String(), nullable=True),
|
||||||
|
sa.Column("identifier", sa.String(), nullable=True),
|
||||||
|
sa.Column("file_path", sa.String(), nullable=False),
|
||||||
|
sa.Column("file_sha256", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("file_mtime", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("file_size", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_source")),
|
||||||
|
sa.UniqueConstraint("file_path", name=op.f("uq_ebook_source_file_path")),
|
||||||
|
sa.UniqueConstraint("file_sha256", name=op.f("uq_ebook_source_file_sha256")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_chapter",
|
||||||
|
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("spine_index", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("title", sa.String(), nullable=True),
|
||||||
|
sa.Column("href", sa.String(), nullable=True),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["source_id"],
|
||||||
|
[f"{schema}.ebook_source.id"],
|
||||||
|
name=op.f("fk_ebook_chapter_source_id_ebook_source"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chapter")),
|
||||||
|
sa.UniqueConstraint("source_id", "spine_index", name=op.f("uq_ebook_chapter_source_id")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_chunk",
|
||||||
|
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("chapter_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("chunk_index", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("text", sa.String(), nullable=False),
|
||||||
|
sa.Column("token_start", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("token_count", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("page_label", sa.String(), nullable=True),
|
||||||
|
sa.Column("content_sha256", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("search_text", sa.String(), nullable=False),
|
||||||
|
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["chapter_id"],
|
||||||
|
[f"{schema}.ebook_chapter.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_chapter_id_ebook_chapter"),
|
||||||
|
ondelete="SET NULL",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["source_id"],
|
||||||
|
[f"{schema}.ebook_source.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_source_id_ebook_source"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk")),
|
||||||
|
sa.UniqueConstraint("source_id", "chunk_index", name="uq_ebook_chunk_source_id_chunk_index"),
|
||||||
|
sa.UniqueConstraint("source_id", "content_sha256", name="uq_ebook_chunk_source_id_content_sha256"),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_chunk_embedding_1024",
|
||||||
|
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=1024), nullable=False),
|
||||||
|
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["chunk_id"],
|
||||||
|
[f"{schema}.ebook_chunk.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_1024_chunk_id_ebook_chunk"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["model_id"],
|
||||||
|
[f"{schema}.ebook_embedding_model.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_1024_model_id_ebook_embedding_model"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_1024")),
|
||||||
|
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_1024_chunk_id")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_chunk_embedding_2560",
|
||||||
|
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=2560), nullable=False),
|
||||||
|
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["chunk_id"],
|
||||||
|
[f"{schema}.ebook_chunk.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_2560_chunk_id_ebook_chunk"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["model_id"],
|
||||||
|
[f"{schema}.ebook_embedding_model.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_2560_model_id_ebook_embedding_model"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_2560")),
|
||||||
|
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_2560_chunk_id")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"ebook_chunk_embedding_4096",
|
||||||
|
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=4096), nullable=False),
|
||||||
|
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["chunk_id"],
|
||||||
|
[f"{schema}.ebook_chunk.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_4096_chunk_id_ebook_chunk"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["model_id"],
|
||||||
|
[f"{schema}.ebook_embedding_model.id"],
|
||||||
|
name=op.f("fk_ebook_chunk_embedding_4096_model_id_ebook_embedding_model"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_4096")),
|
||||||
|
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_4096_chunk_id")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_table("ebook_chunk_embedding_4096", schema=schema)
|
||||||
|
op.drop_table("ebook_chunk_embedding_2560", schema=schema)
|
||||||
|
op.drop_table("ebook_chunk_embedding_1024", schema=schema)
|
||||||
|
op.drop_table("ebook_chunk", schema=schema)
|
||||||
|
op.drop_table("ebook_chapter", schema=schema)
|
||||||
|
op.drop_table("ebook_source", schema=schema)
|
||||||
|
op.drop_table("ebook_embedding_model", schema=schema)
|
||||||
|
# ### end Alembic commands ###
|
||||||
+63
@@ -0,0 +1,63 @@
|
|||||||
|
"""updated series_index to float and added UniqueConstraint to audiobook and audiobook_author.
|
||||||
|
|
||||||
|
Revision ID: b3c60cc5beb5
|
||||||
|
Revises: d7864d1ffc17
|
||||||
|
Create Date: 2026-06-10 20:02:43.073725
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "b3c60cc5beb5"
|
||||||
|
down_revision: str | None = "d7864d1ffc17"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.alter_column(
|
||||||
|
"audiobook",
|
||||||
|
"series_index",
|
||||||
|
existing_type=sa.INTEGER(),
|
||||||
|
type_=sa.Float(),
|
||||||
|
existing_nullable=False,
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_unique_constraint(
|
||||||
|
op.f("uq_audiobook_author_id"),
|
||||||
|
"audiobook",
|
||||||
|
["author_id", "series_id", "title"],
|
||||||
|
schema=schema,
|
||||||
|
postgresql_nulls_not_distinct=True,
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_constraint(op.f("uq_audiobook_author_id"), "audiobook", schema=schema, type_="unique")
|
||||||
|
op.alter_column(
|
||||||
|
"audiobook",
|
||||||
|
"series_index",
|
||||||
|
existing_type=sa.Float(),
|
||||||
|
type_=sa.INTEGER(),
|
||||||
|
existing_nullable=False,
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
+54
@@ -0,0 +1,54 @@
|
|||||||
|
"""add 1024 ebook embedding cosine index.
|
||||||
|
|
||||||
|
Revision ID: c460105682d2
|
||||||
|
Revises: 2db132cace1a
|
||||||
|
Create Date: 2026-06-13 19:53:45.680289
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "c460105682d2"
|
||||||
|
down_revision: str | None = "2db132cace1a"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_index(
|
||||||
|
"ix_ebook_chunk_embedding_1024_embedding_cosine",
|
||||||
|
"ebook_chunk_embedding_1024",
|
||||||
|
["embedding"],
|
||||||
|
unique=False,
|
||||||
|
schema=schema,
|
||||||
|
postgresql_using="hnsw",
|
||||||
|
postgresql_ops={"embedding": "vector_cosine_ops"},
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_index(
|
||||||
|
"ix_ebook_chunk_embedding_1024_embedding_cosine",
|
||||||
|
table_name="ebook_chunk_embedding_1024",
|
||||||
|
schema=schema,
|
||||||
|
postgresql_using="hnsw",
|
||||||
|
postgresql_ops={"embedding": "vector_cosine_ops"},
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
"""adding haproxy data.
|
||||||
|
|
||||||
|
Revision ID: 96d72c748c24
|
||||||
|
Revises: c460105682d2
|
||||||
|
Create Date: 2026-06-23 16:37:17.768851
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "96d72c748c24"
|
||||||
|
down_revision: str | None = "c460105682d2"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_table(
|
||||||
|
"haproxy_request",
|
||||||
|
sa.Column("line_hash", sa.String(), nullable=False),
|
||||||
|
sa.Column("requested_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("client_ip", sa.String(), nullable=False),
|
||||||
|
sa.Column("client_port", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("frontend", sa.String(), nullable=False),
|
||||||
|
sa.Column("ssl", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("backend", sa.String(), nullable=False),
|
||||||
|
sa.Column("server", sa.String(), nullable=False),
|
||||||
|
sa.Column("time_request", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("time_queue", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("time_connect", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("time_response", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("time_total", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("status_code", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("bytes_read", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("termination_state", sa.String(), nullable=False),
|
||||||
|
sa.Column("active_connections", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("frontend_connections", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("backend_connections", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("server_connections", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("retries", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("server_queue", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("backend_queue", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("host", sa.String(), nullable=True),
|
||||||
|
sa.Column("user_agent", sa.String(), nullable=True),
|
||||||
|
sa.Column("method", sa.String(), nullable=False),
|
||||||
|
sa.Column("target", sa.String(), nullable=False),
|
||||||
|
sa.Column("path", sa.String(), nullable=False),
|
||||||
|
sa.Column("query", sa.String(), nullable=True),
|
||||||
|
sa.Column("http_version", sa.String(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_haproxy_request")),
|
||||||
|
sa.UniqueConstraint("line_hash", name=op.f("uq_haproxy_request_line_hash")),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_index(op.f("ix_haproxy_request_backend"), "haproxy_request", ["backend"], unique=False, schema=schema)
|
||||||
|
op.create_index(op.f("ix_haproxy_request_client_ip"), "haproxy_request", ["client_ip"], unique=False, schema=schema)
|
||||||
|
op.create_index(op.f("ix_haproxy_request_host"), "haproxy_request", ["host"], unique=False, schema=schema)
|
||||||
|
op.create_index(op.f("ix_haproxy_request_path"), "haproxy_request", ["path"], unique=False, schema=schema)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_haproxy_request_requested_at"), "haproxy_request", ["requested_at"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_haproxy_request_status_code"), "haproxy_request", ["status_code"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_haproxy_request_time_response"), "haproxy_request", ["time_response"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
op.f("ix_haproxy_request_user_agent"), "haproxy_request", ["user_agent"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_user_agent"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_time_response"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_status_code"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_requested_at"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_path"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_host"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_client_ip"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_index(op.f("ix_haproxy_request_backend"), table_name="haproxy_request", schema=schema)
|
||||||
|
op.drop_table("haproxy_request", schema=schema)
|
||||||
|
# ### end Alembic commands ###
|
||||||
+206
@@ -0,0 +1,206 @@
|
|||||||
|
"""adding Phrase metadata tables.
|
||||||
|
|
||||||
|
Revision ID: dddee09eddcc
|
||||||
|
Revises: 96d72c748c24
|
||||||
|
Create Date: 2026-06-29 00:49:07.344159
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from sqlalchemy.dialects import postgresql
|
||||||
|
|
||||||
|
from python.orm import RichieBase
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "dddee09eddcc"
|
||||||
|
down_revision: str | None = "96d72c748c24"
|
||||||
|
branch_labels: str | Sequence[str] | None = None
|
||||||
|
depends_on: str | Sequence[str] | None = None
|
||||||
|
|
||||||
|
schema = RichieBase.schema_name
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
"""Upgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_table(
|
||||||
|
"candidate_phrases",
|
||||||
|
sa.Column("book_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("series_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("phrase_text", sa.Text(), nullable=False),
|
||||||
|
sa.Column("phrase_norm", sa.Text(), nullable=False),
|
||||||
|
sa.Column("token_count", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("source_raw_ngram", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_yake", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_spacy_ner", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_spacy_noun_chunk", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_capitalized", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_metadata", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("spacy_label", sa.String(), nullable=True),
|
||||||
|
sa.Column("raw_count", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("chapter_count", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("yake_score", sa.Float(), nullable=True),
|
||||||
|
sa.Column("candidate_score", sa.Float(), nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"sample_contexts",
|
||||||
|
sa.JSON().with_variant(postgresql.JSONB(astext_type=sa.Text()), "postgresql"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("llm_judged", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("llm_keep", sa.Boolean(), nullable=True),
|
||||||
|
sa.Column("llm_confidence", sa.Float(), nullable=True),
|
||||||
|
sa.Column("llm_category", sa.String(), nullable=True),
|
||||||
|
sa.Column("llm_reason", sa.Text(), nullable=True),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["book_id"],
|
||||||
|
[f"{schema}.ebook_source.id"],
|
||||||
|
name=op.f("fk_candidate_phrases_book_id_ebook_source"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_candidate_phrases")),
|
||||||
|
sa.UniqueConstraint("book_id", "phrase_norm", name="uq_candidate_phrases_book_id_phrase_norm"),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"candidate_phrases_book_norm_idx", "candidate_phrases", ["book_id", "phrase_norm"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"candidate_phrases_book_score_idx",
|
||||||
|
"candidate_phrases",
|
||||||
|
["book_id", "candidate_score"],
|
||||||
|
unique=False,
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"protected_phrases",
|
||||||
|
sa.Column("book_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("series_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("phrase_text", sa.Text(), nullable=False),
|
||||||
|
sa.Column("phrase_norm", sa.Text(), nullable=False),
|
||||||
|
sa.Column("canonical_id", sa.String(), nullable=False),
|
||||||
|
sa.Column("phrase_type", sa.String(), nullable=True),
|
||||||
|
sa.Column("token_count", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("confidence", sa.Float(), nullable=False),
|
||||||
|
sa.Column("importance", sa.Float(), nullable=False),
|
||||||
|
sa.Column("allow_nested", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("suppress_children", sa.Boolean(), nullable=False),
|
||||||
|
sa.Column("source_candidate_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["book_id"],
|
||||||
|
[f"{schema}.ebook_source.id"],
|
||||||
|
name=op.f("fk_protected_phrases_book_id_ebook_source"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["source_candidate_id"],
|
||||||
|
[f"{schema}.candidate_phrases.id"],
|
||||||
|
name=op.f("fk_protected_phrases_source_candidate_id_candidate_phrases"),
|
||||||
|
ondelete="SET NULL",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_protected_phrases")),
|
||||||
|
sa.UniqueConstraint("book_id", "phrase_norm", name="uq_protected_phrases_book_id_phrase_norm"),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"protected_phrases_book_norm_idx", "protected_phrases", ["book_id", "phrase_norm"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index("protected_phrases_norm_idx", "protected_phrases", ["phrase_norm"], unique=False, schema=schema)
|
||||||
|
op.create_index(
|
||||||
|
"protected_phrases_series_norm_idx",
|
||||||
|
"protected_phrases",
|
||||||
|
["series_id", "phrase_norm"],
|
||||||
|
unique=False,
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"chunk_phrase_mentions",
|
||||||
|
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||||
|
sa.Column("phrase_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("book_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("series_id", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("start_char", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("end_char", sa.Integer(), nullable=True),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["book_id"],
|
||||||
|
[f"{schema}.ebook_source.id"],
|
||||||
|
name=op.f("fk_chunk_phrase_mentions_book_id_ebook_source"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["chunk_id"],
|
||||||
|
[f"{schema}.ebook_chunk.id"],
|
||||||
|
name=op.f("fk_chunk_phrase_mentions_chunk_id_ebook_chunk"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["phrase_id"],
|
||||||
|
[f"{schema}.protected_phrases.id"],
|
||||||
|
name=op.f("fk_chunk_phrase_mentions_phrase_id_protected_phrases"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_chunk_phrase_mentions")),
|
||||||
|
sa.UniqueConstraint("chunk_id", "phrase_id", "start_char", name="uq_chunk_phrase_mentions_chunk_phrase_start"),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"chunk_phrase_mentions_chunk_idx", "chunk_phrase_mentions", ["chunk_id"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"chunk_phrase_mentions_phrase_idx", "chunk_phrase_mentions", ["phrase_id"], unique=False, schema=schema
|
||||||
|
)
|
||||||
|
op.create_table(
|
||||||
|
"phrase_aliases",
|
||||||
|
sa.Column("phrase_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("alias_text", sa.Text(), nullable=False),
|
||||||
|
sa.Column("alias_norm", sa.Text(), nullable=False),
|
||||||
|
sa.Column("confidence", sa.Float(), nullable=False),
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(
|
||||||
|
["phrase_id"],
|
||||||
|
[f"{schema}.protected_phrases.id"],
|
||||||
|
name=op.f("fk_phrase_aliases_phrase_id_protected_phrases"),
|
||||||
|
ondelete="CASCADE",
|
||||||
|
),
|
||||||
|
sa.PrimaryKeyConstraint("id", name=op.f("pk_phrase_aliases")),
|
||||||
|
sa.UniqueConstraint("phrase_id", "alias_norm", name="uq_phrase_aliases_phrase_id_alias_norm"),
|
||||||
|
schema=schema,
|
||||||
|
)
|
||||||
|
op.create_index("phrase_aliases_norm_idx", "phrase_aliases", ["alias_norm"], unique=False, schema=schema)
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
"""Downgrade."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.drop_index("phrase_aliases_norm_idx", table_name="phrase_aliases", schema=schema)
|
||||||
|
op.drop_table("phrase_aliases", schema=schema)
|
||||||
|
op.drop_index("chunk_phrase_mentions_phrase_idx", table_name="chunk_phrase_mentions", schema=schema)
|
||||||
|
op.drop_index("chunk_phrase_mentions_chunk_idx", table_name="chunk_phrase_mentions", schema=schema)
|
||||||
|
op.drop_table("chunk_phrase_mentions", schema=schema)
|
||||||
|
op.drop_index("protected_phrases_series_norm_idx", table_name="protected_phrases", schema=schema)
|
||||||
|
op.drop_index("protected_phrases_norm_idx", table_name="protected_phrases", schema=schema)
|
||||||
|
op.drop_index("protected_phrases_book_norm_idx", table_name="protected_phrases", schema=schema)
|
||||||
|
op.drop_table("protected_phrases", schema=schema)
|
||||||
|
op.drop_index("candidate_phrases_book_score_idx", table_name="candidate_phrases", schema=schema)
|
||||||
|
op.drop_index("candidate_phrases_book_norm_idx", table_name="candidate_phrases", schema=schema)
|
||||||
|
op.drop_table("candidate_phrases", schema=schema)
|
||||||
|
# ### end Alembic commands ###
|
||||||
-100
@@ -1,100 +0,0 @@
|
|||||||
"""seprating signal_bot database.
|
|
||||||
|
|
||||||
Revision ID: 6eaf696e07a5
|
|
||||||
Revises:
|
|
||||||
Create Date: 2026-03-17 21:35:37.612672
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from alembic import op
|
|
||||||
from sqlalchemy.dialects import postgresql
|
|
||||||
|
|
||||||
from python.orm import SignalBotBase
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "6eaf696e07a5"
|
|
||||||
down_revision: str | None = None
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = SignalBotBase.schema_name
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Upgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.create_table(
|
|
||||||
"dead_letter_message",
|
|
||||||
sa.Column("source", sa.String(), nullable=False),
|
|
||||||
sa.Column("message", sa.Text(), nullable=False),
|
|
||||||
sa.Column("received_at", sa.DateTime(timezone=True), nullable=False),
|
|
||||||
sa.Column(
|
|
||||||
"status", postgresql.ENUM("UNPROCESSED", "PROCESSED", name="message_status", schema=schema), nullable=False
|
|
||||||
),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_dead_letter_message")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_table(
|
|
||||||
"role",
|
|
||||||
sa.Column("name", sa.String(length=50), nullable=False),
|
|
||||||
sa.Column("id", sa.SmallInteger(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_role")),
|
|
||||||
sa.UniqueConstraint("name", name=op.f("uq_role_name")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_table(
|
|
||||||
"signal_device",
|
|
||||||
sa.Column("phone_number", sa.String(length=50), nullable=False),
|
|
||||||
sa.Column("safety_number", sa.String(), nullable=True),
|
|
||||||
sa.Column(
|
|
||||||
"trust_level",
|
|
||||||
postgresql.ENUM("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", schema=schema),
|
|
||||||
nullable=False,
|
|
||||||
),
|
|
||||||
sa.Column("last_seen", sa.DateTime(timezone=True), nullable=False),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_signal_device")),
|
|
||||||
sa.UniqueConstraint("phone_number", name=op.f("uq_signal_device_phone_number")),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.create_table(
|
|
||||||
"device_role",
|
|
||||||
sa.Column("device_id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("role_id", sa.SmallInteger(), nullable=False),
|
|
||||||
sa.Column("id", sa.Integer(), nullable=False),
|
|
||||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
|
||||||
sa.ForeignKeyConstraint(
|
|
||||||
["device_id"], [f"{schema}.signal_device.id"], name=op.f("fk_device_role_device_id_signal_device")
|
|
||||||
),
|
|
||||||
sa.ForeignKeyConstraint(["role_id"], [f"{schema}.role.id"], name=op.f("fk_device_role_role_id_role")),
|
|
||||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_device_role")),
|
|
||||||
sa.UniqueConstraint("device_id", "role_id", name="uq_device_role_device_role"),
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Downgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.drop_table("device_role", schema=schema)
|
|
||||||
op.drop_table("signal_device", schema=schema)
|
|
||||||
op.drop_table("role", schema=schema)
|
|
||||||
op.drop_table("dead_letter_message", schema=schema)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
@@ -1,72 +0,0 @@
|
|||||||
"""test.
|
|
||||||
|
|
||||||
Revision ID: 66bdd532bcab
|
|
||||||
Revises: 6eaf696e07a5
|
|
||||||
Create Date: 2026-03-18 19:21:14.561568
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import TYPE_CHECKING
|
|
||||||
|
|
||||||
import sqlalchemy as sa
|
|
||||||
from alembic import op
|
|
||||||
from sqlalchemy.dialects import postgresql
|
|
||||||
|
|
||||||
from python.orm import SignalBotBase
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Sequence
|
|
||||||
|
|
||||||
# revision identifiers, used by Alembic.
|
|
||||||
revision: str = "66bdd532bcab"
|
|
||||||
down_revision: str | None = "6eaf696e07a5"
|
|
||||||
branch_labels: str | Sequence[str] | None = None
|
|
||||||
depends_on: str | Sequence[str] | None = None
|
|
||||||
|
|
||||||
schema = SignalBotBase.schema_name
|
|
||||||
|
|
||||||
|
|
||||||
def upgrade() -> None:
|
|
||||||
"""Upgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.alter_column(
|
|
||||||
"dead_letter_message",
|
|
||||||
"status",
|
|
||||||
existing_type=postgresql.ENUM("UNPROCESSED", "PROCESSED", name="message_status", schema=schema),
|
|
||||||
type_=sa.Enum("UNPROCESSED", "PROCESSED", name="message_status", native_enum=False),
|
|
||||||
existing_nullable=False,
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.alter_column(
|
|
||||||
"signal_device",
|
|
||||||
"trust_level",
|
|
||||||
existing_type=postgresql.ENUM("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", schema=schema),
|
|
||||||
type_=sa.Enum("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", native_enum=False),
|
|
||||||
existing_nullable=False,
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
|
|
||||||
|
|
||||||
def downgrade() -> None:
|
|
||||||
"""Downgrade."""
|
|
||||||
# ### commands auto generated by Alembic - please adjust! ###
|
|
||||||
op.alter_column(
|
|
||||||
"signal_device",
|
|
||||||
"trust_level",
|
|
||||||
existing_type=sa.Enum("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", native_enum=False),
|
|
||||||
type_=postgresql.ENUM("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", schema=schema),
|
|
||||||
existing_nullable=False,
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
op.alter_column(
|
|
||||||
"dead_letter_message",
|
|
||||||
"status",
|
|
||||||
existing_type=sa.Enum("UNPROCESSED", "PROCESSED", name="message_status", native_enum=False),
|
|
||||||
type_=postgresql.ENUM("UNPROCESSED", "PROCESSED", name="message_status", schema=schema),
|
|
||||||
existing_nullable=False,
|
|
||||||
schema=schema,
|
|
||||||
)
|
|
||||||
# ### end Alembic commands ###
|
|
||||||
@@ -1,16 +0,0 @@
|
|||||||
"""FastAPI dependencies."""
|
|
||||||
|
|
||||||
from collections.abc import Iterator
|
|
||||||
from typing import Annotated
|
|
||||||
|
|
||||||
from fastapi import Depends, Request
|
|
||||||
from sqlalchemy.orm import Session
|
|
||||||
|
|
||||||
|
|
||||||
def get_db(request: Request) -> Iterator[Session]:
|
|
||||||
"""Get database session from app state."""
|
|
||||||
with Session(request.app.state.engine) as session:
|
|
||||||
yield session
|
|
||||||
|
|
||||||
|
|
||||||
DbSession = Annotated[Session, Depends(get_db)]
|
|
||||||
+7
-3
@@ -1,19 +1,23 @@
|
|||||||
"""FastAPI interface for Contact database."""
|
"""FastAPI interface for Contact database."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
from collections.abc import AsyncIterator
|
|
||||||
from contextlib import asynccontextmanager
|
from contextlib import asynccontextmanager
|
||||||
from typing import Annotated
|
from typing import TYPE_CHECKING, Annotated
|
||||||
|
|
||||||
import typer
|
import typer
|
||||||
import uvicorn
|
import uvicorn
|
||||||
from fastapi import FastAPI
|
from fastapi import FastAPI
|
||||||
|
|
||||||
from python.api.middleware import ZstdMiddleware
|
|
||||||
from python.api.routers import contact_router, views_router
|
from python.api.routers import contact_router, views_router
|
||||||
from python.common import configure_logger
|
from python.common import configure_logger
|
||||||
|
from python.fastapi_tools import ZstdMiddleware
|
||||||
from python.orm.common import get_postgres_engine
|
from python.orm.common import get_postgres_engine
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import AsyncIterator
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from pydantic import BaseModel
|
|||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.orm import selectinload
|
from sqlalchemy.orm import selectinload
|
||||||
|
|
||||||
from python.api.dependencies import DbSession
|
from python.fastapi_tools.db import DbSession # noqa: TC001 this is a FastAPI needed at runtime
|
||||||
from python.orm.richie.contact import Contact, ContactRelationship, Need, RelationshipType
|
from python.orm.richie.contact import Contact, ContactRelationship, Need, RelationshipType
|
||||||
|
|
||||||
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from fastapi.templating import Jinja2Templates
|
|||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.orm import Session, selectinload
|
from sqlalchemy.orm import Session, selectinload
|
||||||
|
|
||||||
from python.api.dependencies import DbSession
|
from python.fastapi_tools.db import DbSession # noqa: TC001 this is a FastAPI needed at runtime
|
||||||
from python.orm.richie.contact import Contact, ContactRelationship, Need, RelationshipType
|
from python.orm.richie.contact import Contact, ContactRelationship, Need, RelationshipType
|
||||||
|
|
||||||
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
||||||
|
|||||||
+9
-34
@@ -3,28 +3,23 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
|
||||||
from datetime import UTC, datetime
|
from datetime import UTC, datetime
|
||||||
from os import getenv
|
from pathlib import Path
|
||||||
from subprocess import PIPE, Popen
|
from subprocess import PIPE, Popen
|
||||||
|
|
||||||
from apprise import Apprise
|
from python.logging_config import configure_logger as _configure_logger
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def configure_logger(level: str = "INFO") -> None:
|
def get_repo_dir() -> Path:
|
||||||
"""Configure the logger.
|
"""Return the repository root directory."""
|
||||||
|
return Path(__file__).resolve().parents[1]
|
||||||
|
|
||||||
Args:
|
|
||||||
level (str, optional): The logging level. Defaults to "INFO".
|
def configure_logger(level: str = "INFO") -> None:
|
||||||
"""
|
"""Configure the logger."""
|
||||||
logging.basicConfig(
|
_configure_logger(level)
|
||||||
level=level,
|
|
||||||
datefmt="%Y-%m-%dT%H:%M:%S%z",
|
|
||||||
format="%(asctime)s %(levelname)s %(filename)s:%(lineno)d - %(message)s",
|
|
||||||
handlers=[logging.StreamHandler(sys.stdout)],
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def bash_wrapper(command: str) -> tuple[str, int]:
|
def bash_wrapper(command: str) -> tuple[str, int]:
|
||||||
@@ -47,26 +42,6 @@ def bash_wrapper(command: str) -> tuple[str, int]:
|
|||||||
return output.decode(), process.returncode
|
return output.decode(), process.returncode
|
||||||
|
|
||||||
|
|
||||||
def signal_alert(body: str, title: str = "") -> None:
|
|
||||||
"""Send a signal alert.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
body (str): The body of the alert.
|
|
||||||
title (str, optional): The title of the alert. Defaults to "".
|
|
||||||
"""
|
|
||||||
apprise_client = Apprise()
|
|
||||||
|
|
||||||
from_phone = getenv("SIGNAL_ALERT_FROM_PHONE")
|
|
||||||
to_phone = getenv("SIGNAL_ALERT_TO_PHONE")
|
|
||||||
if not from_phone or not to_phone:
|
|
||||||
logger.info("SIGNAL_ALERT_FROM_PHONE or SIGNAL_ALERT_TO_PHONE not set")
|
|
||||||
return
|
|
||||||
|
|
||||||
apprise_client.add(f"signal://localhost:8989/{from_phone}/{to_phone}")
|
|
||||||
|
|
||||||
apprise_client.notify(title=title, body=body)
|
|
||||||
|
|
||||||
|
|
||||||
def utcnow() -> datetime:
|
def utcnow() -> datetime:
|
||||||
"""Get the current UTC time."""
|
"""Get the current UTC time."""
|
||||||
return datetime.now(tz=UTC)
|
return datetime.now(tz=UTC)
|
||||||
|
|||||||
@@ -1,3 +0,0 @@
|
|||||||
"""Data science CLI tools."""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
@@ -1,613 +0,0 @@
|
|||||||
"""Ingestion pipeline for loading congress data from unitedstates/congress JSON files.
|
|
||||||
|
|
||||||
Loads legislators, bills, votes, vote records, and bill text into the data_science_dev database.
|
|
||||||
Expects the parent directory to contain congress-tracker/ and congress-legislators/ as siblings.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
ingest-congress /path/to/parent/
|
|
||||||
ingest-congress /path/to/parent/ --congress 118
|
|
||||||
ingest-congress /path/to/parent/ --congress 118 --only bills
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
from pathlib import Path # noqa: TC003 needed at runtime for typer CLI argument
|
|
||||||
from typing import TYPE_CHECKING, Annotated
|
|
||||||
|
|
||||||
import orjson
|
|
||||||
import typer
|
|
||||||
import yaml
|
|
||||||
from sqlalchemy import select
|
|
||||||
from sqlalchemy.orm import Session
|
|
||||||
|
|
||||||
from python.common import configure_logger
|
|
||||||
from python.orm.common import get_postgres_engine
|
|
||||||
from python.orm.data_science_dev.congress import Bill, BillText, Legislator, LegislatorSocialMedia, Vote, VoteRecord
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Iterator
|
|
||||||
|
|
||||||
from sqlalchemy.engine import Engine
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
BATCH_SIZE = 10_000
|
|
||||||
|
|
||||||
app = typer.Typer(help="Ingest unitedstates/congress data into data_science_dev.")
|
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
|
||||||
def main(
|
|
||||||
parent_dir: Annotated[
|
|
||||||
Path,
|
|
||||||
typer.Argument(help="Parent directory containing congress-tracker/ and congress-legislators/"),
|
|
||||||
],
|
|
||||||
congress: Annotated[int | None, typer.Option(help="Only ingest a specific congress number")] = None,
|
|
||||||
only: Annotated[
|
|
||||||
str | None,
|
|
||||||
typer.Option(help="Only run a specific step: legislators, social-media, bills, votes, bill-text"),
|
|
||||||
] = None,
|
|
||||||
) -> None:
|
|
||||||
"""Ingest congress data from unitedstates/congress JSON files."""
|
|
||||||
configure_logger(level="INFO")
|
|
||||||
|
|
||||||
data_dir = parent_dir / "congress-tracker/congress/data/"
|
|
||||||
legislators_dir = parent_dir / "congress-legislators"
|
|
||||||
|
|
||||||
if not data_dir.is_dir():
|
|
||||||
typer.echo(f"Expected congress-tracker/ directory: {data_dir}", err=True)
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
|
|
||||||
if not legislators_dir.is_dir():
|
|
||||||
typer.echo(f"Expected congress-legislators/ directory: {legislators_dir}", err=True)
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
|
|
||||||
engine = get_postgres_engine(name="DATA_SCIENCE_DEV")
|
|
||||||
|
|
||||||
congress_dirs = _resolve_congress_dirs(data_dir, congress)
|
|
||||||
if not congress_dirs:
|
|
||||||
typer.echo("No congress directories found.", err=True)
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
|
|
||||||
logger.info("Found %d congress directories to process", len(congress_dirs))
|
|
||||||
|
|
||||||
steps: dict[str, tuple] = {
|
|
||||||
"legislators": (ingest_legislators, (engine, legislators_dir)),
|
|
||||||
"legislators-social-media": (ingest_social_media, (engine, legislators_dir)),
|
|
||||||
"bills": (ingest_bills, (engine, congress_dirs)),
|
|
||||||
"votes": (ingest_votes, (engine, congress_dirs)),
|
|
||||||
"bill-text": (ingest_bill_text, (engine, congress_dirs)),
|
|
||||||
}
|
|
||||||
|
|
||||||
if only:
|
|
||||||
if only not in steps:
|
|
||||||
typer.echo(f"Unknown step: {only}. Choose from: {', '.join(steps)}", err=True)
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
steps = {only: steps[only]}
|
|
||||||
|
|
||||||
for step_name, (step_func, step_args) in steps.items():
|
|
||||||
logger.info("=== Starting step: %s ===", step_name)
|
|
||||||
step_func(*step_args)
|
|
||||||
logger.info("=== Finished step: %s ===", step_name)
|
|
||||||
|
|
||||||
logger.info("ingest-congress done")
|
|
||||||
|
|
||||||
|
|
||||||
def _resolve_congress_dirs(data_dir: Path, congress: int | None) -> list[Path]:
|
|
||||||
"""Find congress number directories under data_dir."""
|
|
||||||
if congress is not None:
|
|
||||||
target = data_dir / str(congress)
|
|
||||||
return [target] if target.is_dir() else []
|
|
||||||
return sorted(path for path in data_dir.iterdir() if path.is_dir() and path.name.isdigit())
|
|
||||||
|
|
||||||
|
|
||||||
def _flush_batch(session: Session, batch: list[object], label: str) -> int:
|
|
||||||
"""Add a batch of ORM objects to the session and commit. Returns count added."""
|
|
||||||
if not batch:
|
|
||||||
return 0
|
|
||||||
session.add_all(batch)
|
|
||||||
session.commit()
|
|
||||||
count = len(batch)
|
|
||||||
logger.info("Committed %d %s", count, label)
|
|
||||||
batch.clear()
|
|
||||||
return count
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Legislators — loaded from congress-legislators YAML files
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_legislators(engine: Engine, legislators_dir: Path) -> None:
|
|
||||||
"""Load legislators from congress-legislators YAML files."""
|
|
||||||
legislators_data = _load_legislators_yaml(legislators_dir)
|
|
||||||
logger.info("Loaded %d legislators from YAML files", len(legislators_data))
|
|
||||||
|
|
||||||
with Session(engine) as session:
|
|
||||||
existing_legislators = {
|
|
||||||
legislator.bioguide_id: legislator for legislator in session.scalars(select(Legislator)).all()
|
|
||||||
}
|
|
||||||
logger.info("Found %d existing legislators in DB", len(existing_legislators))
|
|
||||||
|
|
||||||
total_inserted = 0
|
|
||||||
total_updated = 0
|
|
||||||
for entry in legislators_data:
|
|
||||||
bioguide_id = entry.get("id", {}).get("bioguide")
|
|
||||||
if not bioguide_id:
|
|
||||||
continue
|
|
||||||
|
|
||||||
fields = _parse_legislator(entry)
|
|
||||||
if existing := existing_legislators.get(bioguide_id):
|
|
||||||
changed = False
|
|
||||||
for field, value in fields.items():
|
|
||||||
if value is not None and getattr(existing, field) != value:
|
|
||||||
setattr(existing, field, value)
|
|
||||||
changed = True
|
|
||||||
if changed:
|
|
||||||
total_updated += 1
|
|
||||||
else:
|
|
||||||
session.add(Legislator(bioguide_id=bioguide_id, **fields))
|
|
||||||
total_inserted += 1
|
|
||||||
|
|
||||||
session.commit()
|
|
||||||
logger.info("Inserted %d new legislators, updated %d existing", total_inserted, total_updated)
|
|
||||||
|
|
||||||
|
|
||||||
def _load_legislators_yaml(legislators_dir: Path) -> list[dict]:
|
|
||||||
"""Load and combine legislators-current.yaml and legislators-historical.yaml."""
|
|
||||||
legislators: list[dict] = []
|
|
||||||
for filename in ("legislators-current.yaml", "legislators-historical.yaml"):
|
|
||||||
path = legislators_dir / filename
|
|
||||||
if not path.exists():
|
|
||||||
logger.warning("Legislators file not found: %s", path)
|
|
||||||
continue
|
|
||||||
with path.open() as file:
|
|
||||||
data = yaml.safe_load(file)
|
|
||||||
if isinstance(data, list):
|
|
||||||
legislators.extend(data)
|
|
||||||
return legislators
|
|
||||||
|
|
||||||
|
|
||||||
def _parse_legislator(entry: dict) -> dict:
|
|
||||||
"""Extract Legislator fields from a congress-legislators YAML entry."""
|
|
||||||
ids = entry.get("id", {})
|
|
||||||
name = entry.get("name", {})
|
|
||||||
bio = entry.get("bio", {})
|
|
||||||
terms = entry.get("terms", [])
|
|
||||||
latest_term = terms[-1] if terms else {}
|
|
||||||
|
|
||||||
fec_ids = ids.get("fec")
|
|
||||||
fec_ids_joined = ",".join(fec_ids) if isinstance(fec_ids, list) else fec_ids
|
|
||||||
|
|
||||||
chamber = latest_term.get("type")
|
|
||||||
chamber_normalized = {"rep": "House", "sen": "Senate"}.get(chamber, chamber)
|
|
||||||
|
|
||||||
return {
|
|
||||||
"thomas_id": ids.get("thomas"),
|
|
||||||
"lis_id": ids.get("lis"),
|
|
||||||
"govtrack_id": ids.get("govtrack"),
|
|
||||||
"opensecrets_id": ids.get("opensecrets"),
|
|
||||||
"fec_ids": fec_ids_joined,
|
|
||||||
"first_name": name.get("first"),
|
|
||||||
"last_name": name.get("last"),
|
|
||||||
"official_full_name": name.get("official_full"),
|
|
||||||
"nickname": name.get("nickname"),
|
|
||||||
"birthday": bio.get("birthday"),
|
|
||||||
"gender": bio.get("gender"),
|
|
||||||
"current_party": latest_term.get("party"),
|
|
||||||
"current_state": latest_term.get("state"),
|
|
||||||
"current_district": latest_term.get("district"),
|
|
||||||
"current_chamber": chamber_normalized,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Social Media — loaded from legislators-social-media.yaml
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
SOCIAL_MEDIA_PLATFORMS = {
|
|
||||||
"twitter": "https://twitter.com/{account}",
|
|
||||||
"facebook": "https://facebook.com/{account}",
|
|
||||||
"youtube": "https://youtube.com/{account}",
|
|
||||||
"instagram": "https://instagram.com/{account}",
|
|
||||||
"mastodon": None,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_social_media(engine: Engine, legislators_dir: Path) -> None:
|
|
||||||
"""Load social media accounts from legislators-social-media.yaml."""
|
|
||||||
social_media_path = legislators_dir / "legislators-social-media.yaml"
|
|
||||||
if not social_media_path.exists():
|
|
||||||
logger.warning("Social media file not found: %s", social_media_path)
|
|
||||||
return
|
|
||||||
|
|
||||||
with social_media_path.open() as file:
|
|
||||||
social_media_data = yaml.safe_load(file)
|
|
||||||
|
|
||||||
if not isinstance(social_media_data, list):
|
|
||||||
logger.warning("Unexpected format in %s", social_media_path)
|
|
||||||
return
|
|
||||||
|
|
||||||
logger.info("Loaded %d entries from legislators-social-media.yaml", len(social_media_data))
|
|
||||||
|
|
||||||
with Session(engine) as session:
|
|
||||||
legislator_map = _build_legislator_map(session)
|
|
||||||
existing_accounts = {
|
|
||||||
(account.legislator_id, account.platform)
|
|
||||||
for account in session.scalars(select(LegislatorSocialMedia)).all()
|
|
||||||
}
|
|
||||||
logger.info("Found %d existing social media accounts in DB", len(existing_accounts))
|
|
||||||
|
|
||||||
total_inserted = 0
|
|
||||||
total_updated = 0
|
|
||||||
for entry in social_media_data:
|
|
||||||
bioguide_id = entry.get("id", {}).get("bioguide")
|
|
||||||
if not bioguide_id:
|
|
||||||
continue
|
|
||||||
|
|
||||||
legislator_id = legislator_map.get(bioguide_id)
|
|
||||||
if legislator_id is None:
|
|
||||||
continue
|
|
||||||
|
|
||||||
social = entry.get("social", {})
|
|
||||||
for platform, url_template in SOCIAL_MEDIA_PLATFORMS.items():
|
|
||||||
account_name = social.get(platform)
|
|
||||||
if not account_name:
|
|
||||||
continue
|
|
||||||
|
|
||||||
url = url_template.format(account=account_name) if url_template else None
|
|
||||||
|
|
||||||
if (legislator_id, platform) in existing_accounts:
|
|
||||||
total_updated += 1
|
|
||||||
else:
|
|
||||||
session.add(
|
|
||||||
LegislatorSocialMedia(
|
|
||||||
legislator_id=legislator_id,
|
|
||||||
platform=platform,
|
|
||||||
account_name=str(account_name),
|
|
||||||
url=url,
|
|
||||||
source="https://github.com/unitedstates/congress-legislators",
|
|
||||||
)
|
|
||||||
)
|
|
||||||
existing_accounts.add((legislator_id, platform))
|
|
||||||
total_inserted += 1
|
|
||||||
|
|
||||||
session.commit()
|
|
||||||
logger.info("Inserted %d new social media accounts, updated %d existing", total_inserted, total_updated)
|
|
||||||
|
|
||||||
|
|
||||||
def _iter_voters(position_group: object) -> Iterator[dict]:
|
|
||||||
"""Yield voter dicts from a vote position group (handles list, single dict, or string)."""
|
|
||||||
if isinstance(position_group, dict):
|
|
||||||
yield position_group
|
|
||||||
elif isinstance(position_group, list):
|
|
||||||
for voter in position_group:
|
|
||||||
if isinstance(voter, dict):
|
|
||||||
yield voter
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Bills
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_bills(engine: Engine, congress_dirs: list[Path]) -> None:
|
|
||||||
"""Load bill data.json files."""
|
|
||||||
with Session(engine) as session:
|
|
||||||
existing_bills = {(bill.congress, bill.bill_type, bill.number) for bill in session.scalars(select(Bill)).all()}
|
|
||||||
logger.info("Found %d existing bills in DB", len(existing_bills))
|
|
||||||
|
|
||||||
total_inserted = 0
|
|
||||||
batch: list[Bill] = []
|
|
||||||
for congress_dir in congress_dirs:
|
|
||||||
bills_dir = congress_dir / "bills"
|
|
||||||
if not bills_dir.is_dir():
|
|
||||||
continue
|
|
||||||
logger.info("Scanning bills from %s", congress_dir.name)
|
|
||||||
for bill_file in bills_dir.rglob("data.json"):
|
|
||||||
data = _read_json(bill_file)
|
|
||||||
if data is None:
|
|
||||||
continue
|
|
||||||
bill = _parse_bill(data, existing_bills)
|
|
||||||
if bill is not None:
|
|
||||||
batch.append(bill)
|
|
||||||
if len(batch) >= BATCH_SIZE:
|
|
||||||
total_inserted += _flush_batch(session, batch, "bills")
|
|
||||||
|
|
||||||
total_inserted += _flush_batch(session, batch, "bills")
|
|
||||||
logger.info("Inserted %d new bills total", total_inserted)
|
|
||||||
|
|
||||||
|
|
||||||
def _parse_bill(data: dict, existing_bills: set[tuple[int, str, int]]) -> Bill | None:
|
|
||||||
"""Parse a bill data.json dict into a Bill ORM object, skipping existing."""
|
|
||||||
raw_congress = data.get("congress")
|
|
||||||
bill_type = data.get("bill_type")
|
|
||||||
raw_number = data.get("number")
|
|
||||||
if raw_congress is None or bill_type is None or raw_number is None:
|
|
||||||
return None
|
|
||||||
congress = int(raw_congress)
|
|
||||||
number = int(raw_number)
|
|
||||||
if (congress, bill_type, number) in existing_bills:
|
|
||||||
return None
|
|
||||||
|
|
||||||
sponsor_bioguide = None
|
|
||||||
sponsor = data.get("sponsor")
|
|
||||||
if sponsor:
|
|
||||||
sponsor_bioguide = sponsor.get("bioguide_id")
|
|
||||||
|
|
||||||
return Bill(
|
|
||||||
congress=congress,
|
|
||||||
bill_type=bill_type,
|
|
||||||
number=number,
|
|
||||||
title=data.get("short_title") or data.get("official_title"),
|
|
||||||
title_short=data.get("short_title"),
|
|
||||||
official_title=data.get("official_title"),
|
|
||||||
status=data.get("status"),
|
|
||||||
status_at=data.get("status_at"),
|
|
||||||
sponsor_bioguide_id=sponsor_bioguide,
|
|
||||||
subjects_top_term=data.get("subjects_top_term"),
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Votes (and vote records)
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_votes(engine: Engine, congress_dirs: list[Path]) -> None:
|
|
||||||
"""Load vote data.json files with their vote records."""
|
|
||||||
with Session(engine) as session:
|
|
||||||
legislator_map = _build_legislator_map(session)
|
|
||||||
logger.info("Loaded %d legislators into lookup map", len(legislator_map))
|
|
||||||
bill_map = _build_bill_map(session)
|
|
||||||
logger.info("Loaded %d bills into lookup map", len(bill_map))
|
|
||||||
existing_votes = {
|
|
||||||
(vote.congress, vote.chamber, vote.session, vote.number) for vote in session.scalars(select(Vote)).all()
|
|
||||||
}
|
|
||||||
logger.info("Found %d existing votes in DB", len(existing_votes))
|
|
||||||
|
|
||||||
total_inserted = 0
|
|
||||||
batch: list[Vote] = []
|
|
||||||
for congress_dir in congress_dirs:
|
|
||||||
votes_dir = congress_dir / "votes"
|
|
||||||
if not votes_dir.is_dir():
|
|
||||||
continue
|
|
||||||
logger.info("Scanning votes from %s", congress_dir.name)
|
|
||||||
for vote_file in votes_dir.rglob("data.json"):
|
|
||||||
data = _read_json(vote_file)
|
|
||||||
if data is None:
|
|
||||||
continue
|
|
||||||
vote = _parse_vote(data, legislator_map, bill_map, existing_votes)
|
|
||||||
if vote is not None:
|
|
||||||
batch.append(vote)
|
|
||||||
if len(batch) >= BATCH_SIZE:
|
|
||||||
total_inserted += _flush_batch(session, batch, "votes")
|
|
||||||
|
|
||||||
total_inserted += _flush_batch(session, batch, "votes")
|
|
||||||
logger.info("Inserted %d new votes total", total_inserted)
|
|
||||||
|
|
||||||
|
|
||||||
def _build_legislator_map(session: Session) -> dict[str, int]:
|
|
||||||
"""Build a mapping of bioguide_id -> legislator.id."""
|
|
||||||
return {legislator.bioguide_id: legislator.id for legislator in session.scalars(select(Legislator)).all()}
|
|
||||||
|
|
||||||
|
|
||||||
def _build_bill_map(session: Session) -> dict[tuple[int, str, int], int]:
|
|
||||||
"""Build a mapping of (congress, bill_type, number) -> bill.id."""
|
|
||||||
return {(bill.congress, bill.bill_type, bill.number): bill.id for bill in session.scalars(select(Bill)).all()}
|
|
||||||
|
|
||||||
|
|
||||||
def _parse_vote(
|
|
||||||
data: dict,
|
|
||||||
legislator_map: dict[str, int],
|
|
||||||
bill_map: dict[tuple[int, str, int], int],
|
|
||||||
existing_votes: set[tuple[int, str, int, int]],
|
|
||||||
) -> Vote | None:
|
|
||||||
"""Parse a vote data.json dict into a Vote ORM object with records."""
|
|
||||||
raw_congress = data.get("congress")
|
|
||||||
chamber = data.get("chamber")
|
|
||||||
raw_number = data.get("number")
|
|
||||||
vote_date = data.get("date")
|
|
||||||
if raw_congress is None or chamber is None or raw_number is None or vote_date is None:
|
|
||||||
return None
|
|
||||||
|
|
||||||
raw_session = data.get("session")
|
|
||||||
if raw_session is None:
|
|
||||||
return None
|
|
||||||
|
|
||||||
congress = int(raw_congress)
|
|
||||||
number = int(raw_number)
|
|
||||||
session_number = int(raw_session)
|
|
||||||
|
|
||||||
# Normalize chamber from "h"/"s" to "House"/"Senate"
|
|
||||||
chamber_normalized = {"h": "House", "s": "Senate"}.get(chamber, chamber)
|
|
||||||
|
|
||||||
if (congress, chamber_normalized, session_number, number) in existing_votes:
|
|
||||||
return None
|
|
||||||
|
|
||||||
# Resolve linked bill
|
|
||||||
bill_id = None
|
|
||||||
bill_ref = data.get("bill")
|
|
||||||
if bill_ref:
|
|
||||||
bill_key = (
|
|
||||||
int(bill_ref.get("congress", congress)),
|
|
||||||
bill_ref.get("type"),
|
|
||||||
int(bill_ref.get("number", 0)),
|
|
||||||
)
|
|
||||||
bill_id = bill_map.get(bill_key)
|
|
||||||
|
|
||||||
raw_votes = data.get("votes", {})
|
|
||||||
vote_counts = _count_votes(raw_votes)
|
|
||||||
vote_records = _build_vote_records(raw_votes, legislator_map)
|
|
||||||
|
|
||||||
return Vote(
|
|
||||||
congress=congress,
|
|
||||||
chamber=chamber_normalized,
|
|
||||||
session=session_number,
|
|
||||||
number=number,
|
|
||||||
vote_type=data.get("type"),
|
|
||||||
question=data.get("question"),
|
|
||||||
result=data.get("result"),
|
|
||||||
result_text=data.get("result_text"),
|
|
||||||
vote_date=vote_date[:10] if isinstance(vote_date, str) else vote_date,
|
|
||||||
bill_id=bill_id,
|
|
||||||
vote_records=vote_records,
|
|
||||||
**vote_counts,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _count_votes(raw_votes: dict) -> dict[str, int]:
|
|
||||||
"""Count voters per position category, correctly handling dict and list formats."""
|
|
||||||
yea_count = 0
|
|
||||||
nay_count = 0
|
|
||||||
not_voting_count = 0
|
|
||||||
present_count = 0
|
|
||||||
|
|
||||||
for position, position_group in raw_votes.items():
|
|
||||||
voter_count = sum(1 for _ in _iter_voters(position_group))
|
|
||||||
if position in ("Yea", "Aye"):
|
|
||||||
yea_count += voter_count
|
|
||||||
elif position in ("Nay", "No"):
|
|
||||||
nay_count += voter_count
|
|
||||||
elif position == "Not Voting":
|
|
||||||
not_voting_count += voter_count
|
|
||||||
elif position == "Present":
|
|
||||||
present_count += voter_count
|
|
||||||
|
|
||||||
return {
|
|
||||||
"yea_count": yea_count,
|
|
||||||
"nay_count": nay_count,
|
|
||||||
"not_voting_count": not_voting_count,
|
|
||||||
"present_count": present_count,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def _build_vote_records(raw_votes: dict, legislator_map: dict[str, int]) -> list[VoteRecord]:
|
|
||||||
"""Build VoteRecord objects from raw vote data."""
|
|
||||||
records: list[VoteRecord] = []
|
|
||||||
for position, position_group in raw_votes.items():
|
|
||||||
for voter in _iter_voters(position_group):
|
|
||||||
bioguide_id = voter.get("id")
|
|
||||||
if not bioguide_id:
|
|
||||||
continue
|
|
||||||
legislator_id = legislator_map.get(bioguide_id)
|
|
||||||
if legislator_id is None:
|
|
||||||
continue
|
|
||||||
records.append(
|
|
||||||
VoteRecord(
|
|
||||||
legislator_id=legislator_id,
|
|
||||||
position=position,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
return records
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Bill Text
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_bill_text(engine: Engine, congress_dirs: list[Path]) -> None:
|
|
||||||
"""Load bill text from text-versions directories."""
|
|
||||||
with Session(engine) as session:
|
|
||||||
bill_map = _build_bill_map(session)
|
|
||||||
logger.info("Loaded %d bills into lookup map", len(bill_map))
|
|
||||||
existing_bill_texts = {
|
|
||||||
(bill_text.bill_id, bill_text.version_code) for bill_text in session.scalars(select(BillText)).all()
|
|
||||||
}
|
|
||||||
logger.info("Found %d existing bill text versions in DB", len(existing_bill_texts))
|
|
||||||
|
|
||||||
total_inserted = 0
|
|
||||||
batch: list[BillText] = []
|
|
||||||
for congress_dir in congress_dirs:
|
|
||||||
logger.info("Scanning bill texts from %s", congress_dir.name)
|
|
||||||
for bill_text in _iter_bill_texts(congress_dir, bill_map, existing_bill_texts):
|
|
||||||
batch.append(bill_text)
|
|
||||||
if len(batch) >= BATCH_SIZE:
|
|
||||||
total_inserted += _flush_batch(session, batch, "bill texts")
|
|
||||||
|
|
||||||
total_inserted += _flush_batch(session, batch, "bill texts")
|
|
||||||
logger.info("Inserted %d new bill text versions total", total_inserted)
|
|
||||||
|
|
||||||
|
|
||||||
def _iter_bill_texts(
|
|
||||||
congress_dir: Path,
|
|
||||||
bill_map: dict[tuple[int, str, int], int],
|
|
||||||
existing_bill_texts: set[tuple[int, str]],
|
|
||||||
) -> Iterator[BillText]:
|
|
||||||
"""Yield BillText objects for a single congress directory, skipping existing."""
|
|
||||||
bills_dir = congress_dir / "bills"
|
|
||||||
if not bills_dir.is_dir():
|
|
||||||
return
|
|
||||||
|
|
||||||
for bill_dir in bills_dir.rglob("text-versions"):
|
|
||||||
if not bill_dir.is_dir():
|
|
||||||
continue
|
|
||||||
bill_key = _bill_key_from_dir(bill_dir.parent, congress_dir)
|
|
||||||
if bill_key is None:
|
|
||||||
continue
|
|
||||||
bill_id = bill_map.get(bill_key)
|
|
||||||
if bill_id is None:
|
|
||||||
continue
|
|
||||||
|
|
||||||
for version_dir in sorted(bill_dir.iterdir()):
|
|
||||||
if not version_dir.is_dir():
|
|
||||||
continue
|
|
||||||
if (bill_id, version_dir.name) in existing_bill_texts:
|
|
||||||
continue
|
|
||||||
text_content = _read_bill_text(version_dir)
|
|
||||||
version_data = _read_json(version_dir / "data.json")
|
|
||||||
yield BillText(
|
|
||||||
bill_id=bill_id,
|
|
||||||
version_code=version_dir.name,
|
|
||||||
version_name=version_data.get("version_name") if version_data else None,
|
|
||||||
date=version_data.get("issued_on") if version_data else None,
|
|
||||||
text_content=text_content,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def _bill_key_from_dir(bill_dir: Path, congress_dir: Path) -> tuple[int, str, int] | None:
|
|
||||||
"""Extract (congress, bill_type, number) from directory structure."""
|
|
||||||
congress = int(congress_dir.name)
|
|
||||||
bill_type = bill_dir.parent.name
|
|
||||||
name = bill_dir.name
|
|
||||||
# Directory name is like "hr3590" — strip the type prefix to get the number
|
|
||||||
number_str = name[len(bill_type) :]
|
|
||||||
if not number_str.isdigit():
|
|
||||||
return None
|
|
||||||
return (congress, bill_type, int(number_str))
|
|
||||||
|
|
||||||
|
|
||||||
def _read_bill_text(version_dir: Path) -> str | None:
|
|
||||||
"""Read bill text from a version directory, preferring .txt over .xml."""
|
|
||||||
for extension in ("txt", "htm", "html", "xml"):
|
|
||||||
candidates = list(version_dir.glob(f"document.{extension}"))
|
|
||||||
if not candidates:
|
|
||||||
candidates = list(version_dir.glob(f"*.{extension}"))
|
|
||||||
if candidates:
|
|
||||||
try:
|
|
||||||
return candidates[0].read_text(encoding="utf-8")
|
|
||||||
except Exception:
|
|
||||||
logger.exception("Failed to read %s", candidates[0])
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Helpers
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def _read_json(path: Path) -> dict | None:
|
|
||||||
"""Read and parse a JSON file, returning None on failure."""
|
|
||||||
try:
|
|
||||||
return orjson.loads(path.read_bytes())
|
|
||||||
except FileNotFoundError:
|
|
||||||
return None
|
|
||||||
except Exception:
|
|
||||||
logger.exception("Failed to parse %s", path)
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
app()
|
|
||||||
@@ -1,247 +0,0 @@
|
|||||||
"""Ingestion pipeline for loading JSONL post files into the weekly-partitioned posts table.
|
|
||||||
|
|
||||||
Usage:
|
|
||||||
ingest-posts /path/to/files/
|
|
||||||
ingest-posts /path/to/single_file.jsonl
|
|
||||||
ingest-posts /data/dir/ --workers 4 --batch-size 5000
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
from datetime import UTC, datetime
|
|
||||||
from pathlib import Path # noqa: TC003 this is needed for typer
|
|
||||||
from typing import TYPE_CHECKING, Annotated
|
|
||||||
|
|
||||||
import orjson
|
|
||||||
import psycopg
|
|
||||||
import typer
|
|
||||||
|
|
||||||
from python.common import configure_logger
|
|
||||||
from python.orm.common import get_connection_info
|
|
||||||
from python.parallelize import parallelize_process
|
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from collections.abc import Iterator
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
app = typer.Typer(help="Ingest JSONL post files into the partitioned posts table.")
|
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
|
||||||
def main(
|
|
||||||
path: Annotated[Path, typer.Argument(help="Directory containing JSONL files, or a single JSONL file")],
|
|
||||||
batch_size: Annotated[int, typer.Option(help="Rows per INSERT batch")] = 10000,
|
|
||||||
workers: Annotated[int, typer.Option(help="Parallel workers for multi-file ingestion")] = 4,
|
|
||||||
pattern: Annotated[str, typer.Option(help="Glob pattern for JSONL files")] = "*.jsonl",
|
|
||||||
) -> None:
|
|
||||||
"""Ingest JSONL post files into the weekly-partitioned posts table."""
|
|
||||||
configure_logger(level="INFO")
|
|
||||||
|
|
||||||
logger.info("starting ingest-posts")
|
|
||||||
logger.info("path=%s batch_size=%d workers=%d pattern=%s", path, batch_size, workers, pattern)
|
|
||||||
if path.is_file():
|
|
||||||
ingest_file(path, batch_size=batch_size)
|
|
||||||
elif path.is_dir():
|
|
||||||
ingest_directory(path, batch_size=batch_size, max_workers=workers, pattern=pattern)
|
|
||||||
else:
|
|
||||||
typer.echo(f"Path does not exist: {path}", err=True)
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
|
|
||||||
logger.info("ingest-posts done")
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_directory(
|
|
||||||
directory: Path,
|
|
||||||
*,
|
|
||||||
batch_size: int,
|
|
||||||
max_workers: int,
|
|
||||||
pattern: str = "*.jsonl",
|
|
||||||
) -> None:
|
|
||||||
"""Ingest all JSONL files in a directory using parallel workers."""
|
|
||||||
files = sorted(directory.glob(pattern))
|
|
||||||
if not files:
|
|
||||||
logger.warning("No JSONL files found in %s", directory)
|
|
||||||
return
|
|
||||||
|
|
||||||
logger.info("Found %d JSONL files to ingest", len(files))
|
|
||||||
|
|
||||||
kwargs_list = [{"path": fp, "batch_size": batch_size} for fp in files]
|
|
||||||
parallelize_process(ingest_file, kwargs_list, max_workers=max_workers)
|
|
||||||
|
|
||||||
|
|
||||||
SCHEMA = "main"
|
|
||||||
|
|
||||||
COLUMNS = (
|
|
||||||
"post_id",
|
|
||||||
"user_id",
|
|
||||||
"instance",
|
|
||||||
"date",
|
|
||||||
"text",
|
|
||||||
"langs",
|
|
||||||
"like_count",
|
|
||||||
"reply_count",
|
|
||||||
"repost_count",
|
|
||||||
"reply_to",
|
|
||||||
"replied_author",
|
|
||||||
"thread_root",
|
|
||||||
"thread_root_author",
|
|
||||||
"repost_from",
|
|
||||||
"reposted_author",
|
|
||||||
"quotes",
|
|
||||||
"quoted_author",
|
|
||||||
"labels",
|
|
||||||
"sent_label",
|
|
||||||
"sent_score",
|
|
||||||
)
|
|
||||||
|
|
||||||
INSERT_FROM_STAGING = f"""
|
|
||||||
INSERT INTO {SCHEMA}.posts ({", ".join(COLUMNS)})
|
|
||||||
SELECT {", ".join(COLUMNS)} FROM pg_temp.staging
|
|
||||||
ON CONFLICT (post_id, date) DO NOTHING
|
|
||||||
""" # noqa: S608
|
|
||||||
|
|
||||||
FAILED_INSERT = f"""
|
|
||||||
INSERT INTO {SCHEMA}.failed_ingestion (raw_line, error)
|
|
||||||
VALUES (%(raw_line)s, %(error)s)
|
|
||||||
""" # noqa: S608
|
|
||||||
|
|
||||||
|
|
||||||
def get_psycopg_connection() -> psycopg.Connection:
|
|
||||||
"""Create a raw psycopg3 connection from environment variables."""
|
|
||||||
database, host, port, username, password = get_connection_info("DATA_SCIENCE_DEV")
|
|
||||||
return psycopg.connect(
|
|
||||||
dbname=database,
|
|
||||||
host=host,
|
|
||||||
port=int(port),
|
|
||||||
user=username,
|
|
||||||
password=password,
|
|
||||||
autocommit=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_file(path: Path, *, batch_size: int) -> None:
|
|
||||||
"""Ingest a single JSONL file into the posts table."""
|
|
||||||
log_trigger = max(100_000 // batch_size, 1)
|
|
||||||
failed_lines: list[dict] = []
|
|
||||||
try:
|
|
||||||
with get_psycopg_connection() as connection:
|
|
||||||
for index, batch in enumerate(read_jsonl_batches(path, batch_size, failed_lines), 1):
|
|
||||||
ingest_batch(connection, batch)
|
|
||||||
if index % log_trigger == 0:
|
|
||||||
logger.info("Ingested %d batches (%d rows) from %s", index, index * batch_size, path)
|
|
||||||
|
|
||||||
if failed_lines:
|
|
||||||
logger.warning("Recording %d malformed lines from %s", len(failed_lines), path.name)
|
|
||||||
with connection.cursor() as cursor:
|
|
||||||
cursor.executemany(FAILED_INSERT, failed_lines)
|
|
||||||
connection.commit()
|
|
||||||
except Exception:
|
|
||||||
logger.exception("Failed to ingest file: %s", path)
|
|
||||||
raise
|
|
||||||
|
|
||||||
|
|
||||||
def ingest_batch(connection: psycopg.Connection, batch: list[dict]) -> None:
|
|
||||||
"""COPY batch into a temp staging table, then INSERT ... ON CONFLICT into posts."""
|
|
||||||
if not batch:
|
|
||||||
return
|
|
||||||
|
|
||||||
try:
|
|
||||||
with connection.cursor() as cursor:
|
|
||||||
cursor.execute(f"""
|
|
||||||
CREATE TEMP TABLE IF NOT EXISTS staging
|
|
||||||
(LIKE {SCHEMA}.posts INCLUDING DEFAULTS)
|
|
||||||
ON COMMIT DELETE ROWS
|
|
||||||
""")
|
|
||||||
cursor.execute("TRUNCATE pg_temp.staging")
|
|
||||||
|
|
||||||
with cursor.copy(f"COPY pg_temp.staging ({', '.join(COLUMNS)}) FROM STDIN") as copy:
|
|
||||||
for row in batch:
|
|
||||||
copy.write_row(tuple(row.get(column) for column in COLUMNS))
|
|
||||||
|
|
||||||
cursor.execute(INSERT_FROM_STAGING)
|
|
||||||
connection.commit()
|
|
||||||
except Exception as error:
|
|
||||||
connection.rollback()
|
|
||||||
|
|
||||||
if len(batch) == 1:
|
|
||||||
logger.exception("Skipping bad row post_id=%s", batch[0].get("post_id"))
|
|
||||||
with connection.cursor() as cursor:
|
|
||||||
cursor.execute(
|
|
||||||
FAILED_INSERT,
|
|
||||||
{
|
|
||||||
"raw_line": orjson.dumps(batch[0], default=str).decode(),
|
|
||||||
"error": str(error),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
connection.commit()
|
|
||||||
return
|
|
||||||
|
|
||||||
midpoint = len(batch) // 2
|
|
||||||
ingest_batch(connection, batch[:midpoint])
|
|
||||||
ingest_batch(connection, batch[midpoint:])
|
|
||||||
|
|
||||||
|
|
||||||
def read_jsonl_batches(file_path: Path, batch_size: int, failed_lines: list[dict]) -> Iterator[list[dict]]:
|
|
||||||
"""Stream a JSONL file and yield batches of transformed rows."""
|
|
||||||
batch: list[dict] = []
|
|
||||||
with file_path.open("r", encoding="utf-8") as handle:
|
|
||||||
for raw_line in handle:
|
|
||||||
line = raw_line.strip()
|
|
||||||
if not line:
|
|
||||||
continue
|
|
||||||
batch.extend(parse_line(line, file_path, failed_lines))
|
|
||||||
if len(batch) >= batch_size:
|
|
||||||
yield batch
|
|
||||||
batch = []
|
|
||||||
if batch:
|
|
||||||
yield batch
|
|
||||||
|
|
||||||
|
|
||||||
def parse_line(line: str, file_path: Path, failed_lines: list[dict]) -> Iterator[dict]:
|
|
||||||
"""Parse a JSONL line, handling concatenated JSON objects."""
|
|
||||||
try:
|
|
||||||
yield transform_row(orjson.loads(line))
|
|
||||||
except orjson.JSONDecodeError:
|
|
||||||
if "}{" not in line:
|
|
||||||
logger.warning("Skipping malformed line in %s: %s", file_path.name, line[:120])
|
|
||||||
failed_lines.append({"raw_line": line, "error": "malformed JSON"})
|
|
||||||
return
|
|
||||||
fragments = line.replace("}{", "}\n{").split("\n")
|
|
||||||
for fragment in fragments:
|
|
||||||
try:
|
|
||||||
yield transform_row(orjson.loads(fragment))
|
|
||||||
except (orjson.JSONDecodeError, KeyError, ValueError) as error:
|
|
||||||
logger.warning("Skipping malformed fragment in %s: %s", file_path.name, fragment[:120])
|
|
||||||
failed_lines.append({"raw_line": fragment, "error": str(error)})
|
|
||||||
except Exception as error:
|
|
||||||
logger.exception("Skipping bad row in %s: %s", file_path.name, line[:120])
|
|
||||||
failed_lines.append({"raw_line": line, "error": str(error)})
|
|
||||||
|
|
||||||
|
|
||||||
def transform_row(raw: dict) -> dict:
|
|
||||||
"""Transform a raw JSONL row into a dict matching the Posts table columns."""
|
|
||||||
raw["date"] = parse_date(raw["date"])
|
|
||||||
if raw.get("langs") is not None:
|
|
||||||
raw["langs"] = orjson.dumps(raw["langs"])
|
|
||||||
if raw.get("text") is not None:
|
|
||||||
raw["text"] = raw["text"].replace("\x00", "")
|
|
||||||
return raw
|
|
||||||
|
|
||||||
|
|
||||||
def parse_date(raw_date: int) -> datetime:
|
|
||||||
"""Parse compact YYYYMMDDHHmm integer into a naive datetime (input is UTC by spec)."""
|
|
||||||
return datetime(
|
|
||||||
raw_date // 100000000,
|
|
||||||
(raw_date // 1000000) % 100,
|
|
||||||
(raw_date // 10000) % 100,
|
|
||||||
(raw_date // 100) % 100,
|
|
||||||
raw_date % 100,
|
|
||||||
tzinfo=UTC,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
app()
|
|
||||||
+3
-29
@@ -4,12 +4,10 @@ Usage:
|
|||||||
database <db_name> <command> [args...]
|
database <db_name> <command> [args...]
|
||||||
|
|
||||||
Examples:
|
Examples:
|
||||||
database van_inventory upgrade head
|
|
||||||
database van_inventory downgrade head-1
|
|
||||||
database van_inventory revision --autogenerate -m "add meals table"
|
|
||||||
database van_inventory check
|
|
||||||
database richie check
|
database richie check
|
||||||
database richie upgrade head
|
database richie upgrade head
|
||||||
|
database richie downgrade head-1
|
||||||
|
database richie revision --autogenerate -m "add meals table"
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -48,10 +46,7 @@ class DatabaseConfig:
|
|||||||
|
|
||||||
def alembic_config(self) -> Config:
|
def alembic_config(self) -> Config:
|
||||||
"""Build an alembic Config for this database."""
|
"""Build an alembic Config for this database."""
|
||||||
# Runtime import needed — Config is in TYPE_CHECKING for the return type annotation
|
cfg = Config()
|
||||||
from alembic.config import Config as AlembicConfig # noqa: PLC0415
|
|
||||||
|
|
||||||
cfg = AlembicConfig()
|
|
||||||
cfg.set_main_option("script_location", self.script_location)
|
cfg.set_main_option("script_location", self.script_location)
|
||||||
cfg.set_main_option("file_template", self.file_template)
|
cfg.set_main_option("file_template", self.file_template)
|
||||||
cfg.set_main_option("prepend_sys_path", ".")
|
cfg.set_main_option("prepend_sys_path", ".")
|
||||||
@@ -76,27 +71,6 @@ DATABASES: dict[str, DatabaseConfig] = {
|
|||||||
base_class_name="RichieBase",
|
base_class_name="RichieBase",
|
||||||
models_module="python.orm.richie",
|
models_module="python.orm.richie",
|
||||||
),
|
),
|
||||||
"van_inventory": DatabaseConfig(
|
|
||||||
env_prefix="VAN_INVENTORY",
|
|
||||||
version_location="python/alembic/van_inventory/versions",
|
|
||||||
base_module="python.orm.van_inventory.base",
|
|
||||||
base_class_name="VanInventoryBase",
|
|
||||||
models_module="python.orm.van_inventory.models",
|
|
||||||
),
|
|
||||||
"signal_bot": DatabaseConfig(
|
|
||||||
env_prefix="SIGNALBOT",
|
|
||||||
version_location="python/alembic/signal_bot/versions",
|
|
||||||
base_module="python.orm.signal_bot.base",
|
|
||||||
base_class_name="SignalBotBase",
|
|
||||||
models_module="python.orm.signal_bot.models",
|
|
||||||
),
|
|
||||||
"data_science_dev": DatabaseConfig(
|
|
||||||
env_prefix="DATA_SCIENCE_DEV",
|
|
||||||
version_location="python/alembic/data_science_dev/versions",
|
|
||||||
base_module="python.orm.data_science_dev.base",
|
|
||||||
base_class_name="DataScienceDevBase",
|
|
||||||
models_module="python.orm.data_science_dev.models",
|
|
||||||
),
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
"""EPUB search package."""
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
"""Grounded answer generation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from python.ebook_search.llm_interface import request_chat_completion
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.search import SearchResult
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def answer_query(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
query: str,
|
||||||
|
results: list[SearchResult],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> str:
|
||||||
|
"""Answer a question using only retrieved chunks."""
|
||||||
|
if not config.answer_enabled:
|
||||||
|
logger.info("ebook_answer_skipped_disabled")
|
||||||
|
return "Answer generation is disabled. Source chunks are shown below."
|
||||||
|
|
||||||
|
if not results:
|
||||||
|
logger.info("ebook_answer_skipped_no_results")
|
||||||
|
return "No relevant sources were found."
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_answer_request_start base_url=%s model=%s sources=%s query_length=%s",
|
||||||
|
config.vllm_base_url,
|
||||||
|
config.chat_model,
|
||||||
|
len(results),
|
||||||
|
len(query),
|
||||||
|
)
|
||||||
|
context = "\n\n".join(
|
||||||
|
f"[{index}] {result.source_title}{' - ' + result.chapter_title if result.chapter_title else ''}\n{result.text}"
|
||||||
|
for index, result in enumerate(results, start=1)
|
||||||
|
)
|
||||||
|
content = await request_chat_completion(
|
||||||
|
client,
|
||||||
|
config,
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": (
|
||||||
|
"Answer only from the provided context. Cite sources with bracketed numbers like [1]. "
|
||||||
|
"If the context is insufficient, say so."
|
||||||
|
),
|
||||||
|
},
|
||||||
|
{"role": "user", "content": f"Question:\n{query}\n\nContext:\n{context}"},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_answer_request_complete model=%s answer_length=%s",
|
||||||
|
config.chat_model,
|
||||||
|
len(content),
|
||||||
|
)
|
||||||
|
return content or "The model returned an empty answer."
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Web and external API adapters for EPUB search."""
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
"""Background BM25 refresh tasks for the web app.
|
||||||
|
|
||||||
|
The refresh is scheduled on the event loop instead of a thread because the async psycopg
|
||||||
|
driver only works from the loop; a bare thread cannot open a session on the async engine.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.bm25_corpus import load_bm25_corpus, refresh_bm25_corpus
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncEngine
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def schedule_bm25_refresh(app: FastAPI) -> None:
|
||||||
|
"""Schedule a delayed BM25 corpus refresh, replacing any pending refresh.
|
||||||
|
|
||||||
|
Only called from route handlers, so a running event loop is guaranteed.
|
||||||
|
"""
|
||||||
|
cancel_bm25_refresh(app)
|
||||||
|
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
|
||||||
|
def start_refresh() -> None:
|
||||||
|
app.state.bm25_refresh_task = loop.create_task(refresh_bm25_for_app(app))
|
||||||
|
|
||||||
|
app.state.bm25_refresh_timer = loop.call_later(app.state.config.bm25_refresh_delay_seconds, start_refresh)
|
||||||
|
logger.info(
|
||||||
|
"ebook_bm25_refresh_scheduled delay_seconds=%s",
|
||||||
|
app.state.config.bm25_refresh_delay_seconds,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def cancel_bm25_refresh(app: FastAPI) -> None:
|
||||||
|
"""Cancel any pending BM25 corpus refresh timer and in-flight refresh task."""
|
||||||
|
existing_timer = getattr(app.state, "bm25_refresh_timer", None)
|
||||||
|
if existing_timer is not None:
|
||||||
|
existing_timer.cancel()
|
||||||
|
app.state.bm25_refresh_timer = None
|
||||||
|
logger.info("ebook_bm25_refresh_cancelled")
|
||||||
|
|
||||||
|
existing_task = getattr(app.state, "bm25_refresh_task", None)
|
||||||
|
if existing_task is not None:
|
||||||
|
if not existing_task.done():
|
||||||
|
existing_task.cancel()
|
||||||
|
app.state.bm25_refresh_task = None
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_bm25_for_app(app: FastAPI) -> None:
|
||||||
|
"""Refresh the BM25 corpus using the app engine and config."""
|
||||||
|
try:
|
||||||
|
await refresh_bm25_for_engine(app.state.engine, app.state.config)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("ebook_bm25_refresh_failed")
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_bm25_for_engine(engine: AsyncEngine, config: EbookSearchConfig) -> None:
|
||||||
|
"""Refresh the BM25 corpus using an async SQLAlchemy engine."""
|
||||||
|
async with AsyncSession(engine) as session:
|
||||||
|
await refresh_bm25_corpus(session, config)
|
||||||
|
load_bm25_corpus.cache_clear()
|
||||||
|
logger.info("ebook_bm25_corpus_cache_cleared_after_refresh")
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
"""FastAPI dependencies for the EPUB search app."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import Annotated
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from fastapi import Depends, Request
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncEngine
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
|
||||||
|
def get_config(request: Request) -> EbookSearchConfig:
|
||||||
|
"""Get the loaded search config from app state."""
|
||||||
|
return request.app.state.config
|
||||||
|
|
||||||
|
|
||||||
|
def get_engine(request: Request) -> AsyncEngine:
|
||||||
|
"""Get the database engine from app state."""
|
||||||
|
return request.app.state.engine
|
||||||
|
|
||||||
|
|
||||||
|
def get_http_client(request: Request) -> httpx.AsyncClient:
|
||||||
|
"""Get the shared LLM HTTP client from app state."""
|
||||||
|
return request.app.state.http_client
|
||||||
|
|
||||||
|
|
||||||
|
AppConfig = Annotated[EbookSearchConfig, Depends(get_config)]
|
||||||
|
AppEngine = Annotated[AsyncEngine, Depends(get_engine)]
|
||||||
|
AppHttpClient = Annotated[httpx.AsyncClient, Depends(get_http_client)]
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
"""Background phrase-judging tasks for the web app.
|
||||||
|
|
||||||
|
Judging a book sends one LLM request per candidate phrase, which can take minutes, so it must
|
||||||
|
not run inside the request where it would block the UI. Judgments run as async FastAPI
|
||||||
|
background tasks, awaited on the event loop after the response is sent, and are tracked per
|
||||||
|
book in app state so a second judge request for a book that is already being judged is
|
||||||
|
rejected instead of doubling the work.
|
||||||
|
|
||||||
|
State is loop-confined: every read and mutation happens on the event loop (async route
|
||||||
|
handlers and async background tasks) and no critical section contains an ``await``, so each
|
||||||
|
mutation is atomic per loop iteration and no locking is needed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.judge_ngrams import judge_candidate_phrases_for_books
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from fastapi import BackgroundTasks, FastAPI
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class JudgeTaskState:
|
||||||
|
"""Running book judgments and last outcome messages, keyed by book id."""
|
||||||
|
|
||||||
|
running_book_ids: set[int] = field(default_factory=set)
|
||||||
|
outcome_messages: dict[int, str] = field(default_factory=dict)
|
||||||
|
|
||||||
|
|
||||||
|
def get_judge_task_state(app: FastAPI) -> JudgeTaskState:
|
||||||
|
"""Return the app's judge task state, creating it on first use.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
app (FastAPI): App whose state holds the judge task registry.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
JudgeTaskState: The shared judge task state for this app.
|
||||||
|
"""
|
||||||
|
state = getattr(app.state, "judge_tasks", None)
|
||||||
|
if state is None:
|
||||||
|
state = JudgeTaskState()
|
||||||
|
app.state.judge_tasks = state
|
||||||
|
return state
|
||||||
|
|
||||||
|
|
||||||
|
def start_book_phrase_judgment(app: FastAPI, background_tasks: BackgroundTasks, source_id: int) -> bool:
|
||||||
|
"""Queue judging of one book's candidate phrases as a FastAPI background task.
|
||||||
|
|
||||||
|
The book is claimed before the response returns, so a repeated judge request cannot queue
|
||||||
|
a second run while one is pending or running.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
app (FastAPI): App supplying the engine, config, and judge task state.
|
||||||
|
background_tasks (BackgroundTasks): Request's background tasks to queue the judgment on.
|
||||||
|
source_id (int): Book to judge candidates for.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when a judgment was queued, False when one is already running for this book.
|
||||||
|
"""
|
||||||
|
state = get_judge_task_state(app)
|
||||||
|
if source_id in state.running_book_ids:
|
||||||
|
logger.info("ebook_book_phrase_judgment_already_running source_id=%s", source_id)
|
||||||
|
return False
|
||||||
|
state.running_book_ids.add(source_id)
|
||||||
|
state.outcome_messages.pop(source_id, None)
|
||||||
|
background_tasks.add_task(judge_book_phrases_for_app, app, source_id)
|
||||||
|
logger.info("ebook_book_phrase_judgment_queued source_id=%s", source_id)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
async def judge_book_phrases_for_app(app: FastAPI, source_id: int) -> None:
|
||||||
|
"""Judge one book using the app engine and config, recording the outcome message.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
app (FastAPI): App supplying the engine, config, and judge task state.
|
||||||
|
source_id (int): Book to judge candidates for.
|
||||||
|
"""
|
||||||
|
state = get_judge_task_state(app)
|
||||||
|
try:
|
||||||
|
result = await judge_candidate_phrases_for_books(app.state.engine, app.state.config, source_ids=[source_id])
|
||||||
|
logger.info(
|
||||||
|
"ebook_book_phrase_judgment_complete source_id=%s judged=%s protected=%s mentions=%s failed=%s",
|
||||||
|
source_id,
|
||||||
|
result.candidates_judged,
|
||||||
|
result.protected_phrases,
|
||||||
|
result.phrase_mentions,
|
||||||
|
result.books_failed,
|
||||||
|
)
|
||||||
|
if result.books_failed:
|
||||||
|
message = "Judging failed; see server logs for details"
|
||||||
|
else:
|
||||||
|
message = (
|
||||||
|
f"Judged {result.candidates_judged} candidates; {result.protected_phrases} protected phrases promoted"
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("ebook_book_phrase_judgment_task_failed source_id=%s", source_id)
|
||||||
|
message = "Judging failed; see server logs for details"
|
||||||
|
state.running_book_ids.discard(source_id)
|
||||||
|
state.outcome_messages[source_id] = message
|
||||||
|
|
||||||
|
|
||||||
|
def is_judging_book(app: FastAPI, source_id: int) -> bool:
|
||||||
|
"""Report whether a judgment is currently queued or running for one book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
app (FastAPI): App supplying the judge task state.
|
||||||
|
source_id (int): Book to check.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True while the book's judgment is pending or running.
|
||||||
|
"""
|
||||||
|
return source_id in get_judge_task_state(app).running_book_ids
|
||||||
|
|
||||||
|
|
||||||
|
def pop_book_judgment_outcome(app: FastAPI, source_id: int) -> str | None:
|
||||||
|
"""Return and clear the outcome message from one book's last finished judgment.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
app (FastAPI): App supplying the judge task state.
|
||||||
|
source_id (int): Book to fetch the outcome for.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str | None: The outcome message, or None when there is nothing new to report.
|
||||||
|
"""
|
||||||
|
return get_judge_task_state(app).outcome_messages.pop(source_id, None)
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
"""FastAPI HTMX app for EPUB search."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from contextlib import asynccontextmanager
|
||||||
|
from typing import TYPE_CHECKING, Annotated
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import typer
|
||||||
|
import uvicorn
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from fastapi.staticfiles import StaticFiles
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.common import configure_logger
|
||||||
|
from python.ebook_search.api.bm25_tasks import cancel_bm25_refresh
|
||||||
|
from python.ebook_search.api.routes import admin_router, health_router, page_router, search_router
|
||||||
|
from python.ebook_search.api.web import STATIC_DIR
|
||||||
|
from python.ebook_search.bm25_corpus import ensure_bm25_corpus
|
||||||
|
from python.ebook_search.config import load_config
|
||||||
|
from python.ebook_search.protected_phrases.pool import shutdown_extraction_pool
|
||||||
|
from python.fastapi_tools import ZstdMiddleware
|
||||||
|
from python.orm.common import get_async_postgres_engine
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import AsyncIterator
|
||||||
|
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@asynccontextmanager
|
||||||
|
async def lifespan(app: FastAPI) -> AsyncIterator[None]:
|
||||||
|
"""Manage application startup and shutdown resources."""
|
||||||
|
logger.info("ebook_search_startup")
|
||||||
|
config = load_config()
|
||||||
|
app.state.config = config
|
||||||
|
logger.info(
|
||||||
|
"ebook_search_config_loaded top_k=%s embedding_model=%s embedding_base_url=%s vllm_base_url=%s "
|
||||||
|
"rerank_enabled=%s phrase_matching_enabled=%s answer_enabled=%s library_paths=%s",
|
||||||
|
config.top_k,
|
||||||
|
config.embedding_model,
|
||||||
|
config.embedding_base_url,
|
||||||
|
config.vllm_base_url,
|
||||||
|
config.rerank.enabled,
|
||||||
|
config.phrase_matching_enabled,
|
||||||
|
config.answer_enabled,
|
||||||
|
len(config.library_paths),
|
||||||
|
)
|
||||||
|
if not config.library_paths:
|
||||||
|
logger.warning("ebook_search_no_library_paths_configured")
|
||||||
|
# Concurrent phrase judging opens one session per book worker on this engine, so size the pool
|
||||||
|
# to cover those plus headroom for ordinary web requests.
|
||||||
|
app.state.engine = get_async_postgres_engine(
|
||||||
|
name="RICHIE",
|
||||||
|
vector_engine=True,
|
||||||
|
pool_size=config.phrase_judge_book_workers + 10,
|
||||||
|
)
|
||||||
|
app.state.http_client = httpx.AsyncClient()
|
||||||
|
async with AsyncSession(app.state.engine, expire_on_commit=False) as session:
|
||||||
|
await ensure_bm25_corpus(session, config)
|
||||||
|
try:
|
||||||
|
yield
|
||||||
|
finally:
|
||||||
|
logger.info("ebook_search_shutdown")
|
||||||
|
cancel_bm25_refresh(app)
|
||||||
|
shutdown_extraction_pool()
|
||||||
|
await app.state.http_client.aclose()
|
||||||
|
await app.state.engine.dispose()
|
||||||
|
|
||||||
|
|
||||||
|
def create_app() -> FastAPI:
|
||||||
|
"""Create the EPUB search web app."""
|
||||||
|
app = FastAPI(title="EPUB Search", lifespan=lifespan)
|
||||||
|
app.add_middleware(ZstdMiddleware)
|
||||||
|
app.mount("/static", StaticFiles(directory=STATIC_DIR), name="static")
|
||||||
|
|
||||||
|
app.include_router(admin_router)
|
||||||
|
app.include_router(health_router)
|
||||||
|
app.include_router(page_router)
|
||||||
|
app.include_router(search_router)
|
||||||
|
|
||||||
|
return app
|
||||||
|
|
||||||
|
|
||||||
|
def serve(
|
||||||
|
host: Annotated[str, typer.Option("--host", "-h", help="Host to bind to")] = "127.0.0.1",
|
||||||
|
port: Annotated[int, typer.Option("--port", "-p", help="Port to bind to")] = 8070,
|
||||||
|
log_level: Annotated[str, typer.Option("--log-level", "-l", help="Log level")] = "INFO",
|
||||||
|
) -> None:
|
||||||
|
"""Start the EPUB search server."""
|
||||||
|
configure_logger(log_level)
|
||||||
|
uvicorn.run(create_app(), host=host, port=port)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
typer.run(serve)
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
"""EPUB search web route modules."""
|
||||||
|
|
||||||
|
from python.ebook_search.api.routes.admin import router as admin_router
|
||||||
|
from python.ebook_search.api.routes.health import router as health_router
|
||||||
|
from python.ebook_search.api.routes.page import router as page_router
|
||||||
|
from python.ebook_search.api.routes.search import router as search_router
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"admin_router",
|
||||||
|
"health_router",
|
||||||
|
"page_router",
|
||||||
|
"search_router",
|
||||||
|
]
|
||||||
@@ -0,0 +1,263 @@
|
|||||||
|
"""Admin routes for the EPUB search web UI."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from fastapi import APIRouter, Request
|
||||||
|
from fastapi.responses import HTMLResponse
|
||||||
|
|
||||||
|
from python.ebook_search.api.bm25_tasks import schedule_bm25_refresh
|
||||||
|
from python.ebook_search.api.dependencies import ( # noqa: TC001 FastAPI resolves these annotated dependencies at runtime
|
||||||
|
AppConfig,
|
||||||
|
AppEngine,
|
||||||
|
AppHttpClient,
|
||||||
|
)
|
||||||
|
from python.ebook_search.api.web import templates
|
||||||
|
from python.ebook_search.embeddings import embed_missing_chunks, embedding_model_stats
|
||||||
|
from python.ebook_search.ingest import ingest_configured_paths
|
||||||
|
from python.ebook_search.protected_phrases.generate_ngrams import generate_candidate_phrases_for_books
|
||||||
|
from python.ebook_search.protected_phrases.judge_ngrams import judge_candidate_phrases_for_books
|
||||||
|
from python.ebook_search.protected_phrases.store import book_ids_pending_first_judgment, corpus_phrase_stats
|
||||||
|
from python.fastapi_tools import AsyncDbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter(prefix="/admin")
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("", response_class=HTMLResponse)
|
||||||
|
async def admin(request: Request, config: AppConfig, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Render the admin page."""
|
||||||
|
stats = await embedding_model_stats(session)
|
||||||
|
phrase_stats = await corpus_phrase_stats(session)
|
||||||
|
logger.info(
|
||||||
|
"ebook_admin_page_loaded models=%s candidate_phrases=%s protected_phrases=%s",
|
||||||
|
len(stats),
|
||||||
|
phrase_stats.candidate_phrases,
|
||||||
|
phrase_stats.protected_phrases,
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"admin.html",
|
||||||
|
{"config": config, "stats": stats, "phrase_stats": phrase_stats},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/scan", response_class=HTMLResponse)
|
||||||
|
async def scan_library(request: Request, config: AppConfig, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Scan configured library paths for EPUB changes."""
|
||||||
|
try:
|
||||||
|
count = await ingest_configured_paths(session, config)
|
||||||
|
await session.commit()
|
||||||
|
except Exception as error:
|
||||||
|
logger.exception("ebook_admin_scan_failed")
|
||||||
|
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||||
|
|
||||||
|
logger.info("ebook_admin_scan_complete changed_files=%s", count)
|
||||||
|
if count > 0:
|
||||||
|
schedule_bm25_refresh(request.app)
|
||||||
|
return templates.TemplateResponse(request, "partials/admin_status.html", {"message": f"Indexed {count} EPUBs"})
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/phrases/generate-all", response_class=HTMLResponse)
|
||||||
|
async def generate_all_phrases(request: Request, config: AppConfig, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Regenerate candidate phrases for every indexed book without LLM judging."""
|
||||||
|
return await run_phrase_generation(request, config, session, only_missing=False)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/phrases/generate-missing", response_class=HTMLResponse)
|
||||||
|
async def generate_missing_phrases(request: Request, config: AppConfig, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Generate candidate phrases only for books that have none yet."""
|
||||||
|
return await run_phrase_generation(request, config, session, only_missing=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_phrase_generation(
|
||||||
|
request: Request,
|
||||||
|
config: AppConfig,
|
||||||
|
session: AsyncDbSession,
|
||||||
|
*,
|
||||||
|
only_missing: bool,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Run candidate phrase generation and render the outcome as an admin status partial.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
request (Request): Current request, for template rendering.
|
||||||
|
config (AppConfig): Runtime phrase-tuning settings.
|
||||||
|
session (AsyncDbSession): Active database session.
|
||||||
|
only_missing (bool): Only generate for books without candidates instead of every book.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
HTMLResponse: Status partial describing the generation outcome.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
result = await generate_candidate_phrases_for_books(session, config, only_missing=only_missing)
|
||||||
|
await session.commit()
|
||||||
|
except Exception as error:
|
||||||
|
await session.rollback()
|
||||||
|
logger.exception("ebook_admin_generate_phrases_failed only_missing=%s", only_missing)
|
||||||
|
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_admin_generate_phrases_complete only_missing=%s books_seen=%s books_built=%s candidates=%s",
|
||||||
|
only_missing,
|
||||||
|
result.books_seen,
|
||||||
|
result.books_built,
|
||||||
|
result.candidate_phrases,
|
||||||
|
)
|
||||||
|
if only_missing and result.books_seen == 0:
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{"message": "All books already have candidate phrases"},
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{
|
||||||
|
"message": (
|
||||||
|
f"Generated phrases for {result.books_built} of {result.books_seen} books; "
|
||||||
|
f"{result.candidate_phrases} candidates stored"
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/phrases/judge-all", response_class=HTMLResponse)
|
||||||
|
async def judge_all_phrases(request: Request, engine: AppEngine, config: AppConfig) -> HTMLResponse:
|
||||||
|
"""Judge unjudged candidate phrases across every indexed book."""
|
||||||
|
return await run_phrase_judgment(request, engine, config, source_ids=None)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/phrases/judge-missing", response_class=HTMLResponse)
|
||||||
|
async def judge_missing_phrases(
|
||||||
|
request: Request,
|
||||||
|
engine: AppEngine,
|
||||||
|
config: AppConfig,
|
||||||
|
session: AsyncDbSession,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Judge candidate phrases only for books where judging has never run."""
|
||||||
|
source_ids = await book_ids_pending_first_judgment(session)
|
||||||
|
if not source_ids:
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{"message": "All books with candidate phrases have been judged"},
|
||||||
|
)
|
||||||
|
return await run_phrase_judgment(request, engine, config, source_ids=source_ids)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_phrase_judgment(
|
||||||
|
request: Request,
|
||||||
|
engine: AppEngine,
|
||||||
|
config: AppConfig,
|
||||||
|
*,
|
||||||
|
source_ids: list[int] | None,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Run LLM judging for candidate phrases and render the outcome as an admin status partial.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
request (Request): Current request, for template rendering.
|
||||||
|
engine (AppEngine): Engine used to open per-book judging sessions.
|
||||||
|
config (AppConfig): Runtime phrase-tuning settings.
|
||||||
|
source_ids (list[int] | None): Books to judge; ``None`` judges every indexed book.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
HTMLResponse: Status partial describing the judging outcome.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
result = await judge_candidate_phrases_for_books(engine, config, source_ids=source_ids)
|
||||||
|
except Exception as error:
|
||||||
|
logger.exception("ebook_admin_judge_phrases_failed")
|
||||||
|
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_admin_judge_phrases_complete books_seen=%s books_judged=%s books_failed=%s candidates_judged=%s "
|
||||||
|
"protected=%s mentions=%s",
|
||||||
|
result.books_seen,
|
||||||
|
result.books_judged,
|
||||||
|
result.books_failed,
|
||||||
|
result.candidates_judged,
|
||||||
|
result.protected_phrases,
|
||||||
|
result.phrase_mentions,
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{
|
||||||
|
"message": (
|
||||||
|
f"Judged {result.candidates_judged} candidates across {result.books_judged} of "
|
||||||
|
f"{result.books_seen} books; {result.protected_phrases} protected phrases, "
|
||||||
|
f"{result.phrase_mentions} mentions"
|
||||||
|
+ (f"; {result.books_failed} books failed" if result.books_failed else "")
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/embed-missing", response_class=HTMLResponse)
|
||||||
|
async def embed_missing(
|
||||||
|
request: Request,
|
||||||
|
config: AppConfig,
|
||||||
|
session: AsyncDbSession,
|
||||||
|
client: AppHttpClient,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Embed chunks missing vectors for the configured model."""
|
||||||
|
try:
|
||||||
|
count = await embed_missing_chunks(session, client, config)
|
||||||
|
await session.commit()
|
||||||
|
except Exception as error:
|
||||||
|
logger.exception("ebook_admin_embed_missing_failed")
|
||||||
|
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||||
|
|
||||||
|
logger.info("ebook_admin_embed_missing_complete chunks=%s", count)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{"message": f"Embedded {count} chunks"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/embed-all", response_class=HTMLResponse)
|
||||||
|
async def embed_all(
|
||||||
|
request: Request,
|
||||||
|
config: AppConfig,
|
||||||
|
session: AsyncDbSession,
|
||||||
|
client: AppHttpClient,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Embed all chunks missing vectors in fixed-size batches."""
|
||||||
|
total = 0
|
||||||
|
batches = 0
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
count = await embed_missing_chunks(session, client, config)
|
||||||
|
if count == 0:
|
||||||
|
break
|
||||||
|
await session.commit()
|
||||||
|
total += count
|
||||||
|
batches += 1
|
||||||
|
logger.info(
|
||||||
|
"ebook_admin_embed_all_batch_complete batch=%s chunks=%s total_chunks=%s",
|
||||||
|
batches,
|
||||||
|
count,
|
||||||
|
total,
|
||||||
|
)
|
||||||
|
except Exception as error:
|
||||||
|
logger.exception(
|
||||||
|
"ebook_admin_embed_all_failed batches=%s chunks=%s",
|
||||||
|
batches,
|
||||||
|
total,
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/error.html",
|
||||||
|
{"message": f"Embed all failed after {total} chunks in {batches} batches: {error}"},
|
||||||
|
status_code=500,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("ebook_admin_embed_all_complete batches=%s chunks=%s", batches, total)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/admin_status.html",
|
||||||
|
{"message": f"Embedded {total} chunks in {batches} batches of {config.embedding_batch_size}"},
|
||||||
|
)
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
"""Liveness and readiness routes for the EPUB search service."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from http import HTTPStatus
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from fastapi import APIRouter
|
||||||
|
from fastapi.responses import JSONResponse
|
||||||
|
from sqlalchemy import literal, select
|
||||||
|
from sqlalchemy.exc import SQLAlchemyError
|
||||||
|
|
||||||
|
from python.ebook_search.api.dependencies import ( # noqa: TC001 FastAPI resolves these annotated dependencies at runtime
|
||||||
|
AppConfig,
|
||||||
|
AppHttpClient,
|
||||||
|
)
|
||||||
|
from python.ebook_search.bm25_corpus import bm25_index_exists, bm25_index_path, read_bm25_manifest
|
||||||
|
from python.ebook_search.llm_interface import check_chat_endpoint, check_embedding_endpoint
|
||||||
|
from python.fastapi_tools import AsyncDbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
import httpx
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter()
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/health")
|
||||||
|
async def health() -> dict[str, str]:
|
||||||
|
"""Liveness probe that returns ok without touching dependencies."""
|
||||||
|
return {"status": "ok"}
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/ready")
|
||||||
|
async def ready(config: AppConfig, session: AsyncDbSession, client: AppHttpClient) -> JSONResponse:
|
||||||
|
"""Readiness probe reporting database, embedding endpoint, and BM25 index status."""
|
||||||
|
database_ok = await check_database(session)
|
||||||
|
embedding_ok = await check_embedding_endpoint(client, config)
|
||||||
|
chat_status = await chat_endpoint_status(client, config)
|
||||||
|
bm25_status = check_bm25_status(config)
|
||||||
|
|
||||||
|
checks = {
|
||||||
|
"database": "ok" if database_ok else "fail",
|
||||||
|
"embedding": "ok" if embedding_ok else "fail",
|
||||||
|
"chat": chat_status,
|
||||||
|
"bm25": bm25_status,
|
||||||
|
}
|
||||||
|
if not database_ok:
|
||||||
|
status = "unavailable"
|
||||||
|
status_code = HTTPStatus.SERVICE_UNAVAILABLE
|
||||||
|
elif not embedding_ok or chat_status == "fail" or bm25_status == "missing":
|
||||||
|
status = "degraded"
|
||||||
|
status_code = HTTPStatus.OK
|
||||||
|
else:
|
||||||
|
status = "ready"
|
||||||
|
status_code = HTTPStatus.OK
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_ready_check status=%s database=%s embedding=%s chat=%s bm25=%s",
|
||||||
|
status,
|
||||||
|
database_ok,
|
||||||
|
embedding_ok,
|
||||||
|
chat_status,
|
||||||
|
bm25_status,
|
||||||
|
)
|
||||||
|
return JSONResponse(content={"status": status, "checks": checks}, status_code=status_code)
|
||||||
|
|
||||||
|
|
||||||
|
async def chat_endpoint_status(client: httpx.AsyncClient, config: EbookSearchConfig) -> str:
|
||||||
|
"""Return the answering chat endpoint status, or disabled when answers are off."""
|
||||||
|
if not config.answer_enabled:
|
||||||
|
return "disabled"
|
||||||
|
return "ok" if await check_chat_endpoint(client, config) else "fail"
|
||||||
|
|
||||||
|
|
||||||
|
async def check_database(session: AsyncSession) -> bool:
|
||||||
|
"""Return whether the database answers a trivial query."""
|
||||||
|
try:
|
||||||
|
await session.execute(select(literal(1)))
|
||||||
|
except SQLAlchemyError as error:
|
||||||
|
logger.warning("ebook_ready_database_unavailable error=%s", error)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def check_bm25_status(config: EbookSearchConfig) -> str:
|
||||||
|
"""Return the persisted BM25 index status without loading it into memory."""
|
||||||
|
index_path = bm25_index_path(config)
|
||||||
|
manifest = read_bm25_manifest(index_path)
|
||||||
|
if manifest is None or not bm25_index_exists(index_path, manifest):
|
||||||
|
return "missing"
|
||||||
|
if manifest.chunk_count == 0:
|
||||||
|
return "empty"
|
||||||
|
return "ok"
|
||||||
@@ -0,0 +1,202 @@
|
|||||||
|
"""Page routes for the EPUB search web UI."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from fastapi import APIRouter, BackgroundTasks, HTTPException, Request
|
||||||
|
from fastapi.responses import HTMLResponse, RedirectResponse
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
|
||||||
|
from python.ebook_search.api.dependencies import (
|
||||||
|
AppConfig, # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||||
|
)
|
||||||
|
from python.ebook_search.api.judge_tasks import is_judging_book, pop_book_judgment_outcome, start_book_phrase_judgment
|
||||||
|
from python.ebook_search.api.web import templates
|
||||||
|
from python.ebook_search.protected_phrases.generate_ngrams import recalculate_candidate_phrases_for_book
|
||||||
|
from python.fastapi_tools import AsyncDbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||||
|
from python.orm.richie import EbookCandidatePhrase, EbookChapter, EbookChunk, EbookProtectedPhrase, EbookSource
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter()
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/", response_class=HTMLResponse)
|
||||||
|
async def index(request: Request, config: AppConfig) -> HTMLResponse:
|
||||||
|
"""Render the search page."""
|
||||||
|
return templates.TemplateResponse(request, "search.html", {"config": config})
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/books", response_class=HTMLResponse)
|
||||||
|
async def books(request: Request, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Render the indexed books page."""
|
||||||
|
sources = list((await session.scalars(select(EbookSource).order_by(EbookSource.title))).all())
|
||||||
|
logger.info("ebook_books_page_loaded count=%s", len(sources))
|
||||||
|
return templates.TemplateResponse(request, "books.html", {"sources": sources})
|
||||||
|
|
||||||
|
|
||||||
|
async def get_chapter_count(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Return the number of indexed chapters for one book."""
|
||||||
|
return await session.scalar(select(func.count(EbookChapter.id)).where(EbookChapter.source_id == book_id)) or 0
|
||||||
|
|
||||||
|
|
||||||
|
async def get_chunk_count(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Return the number of indexed chunks for one book."""
|
||||||
|
return await session.scalar(select(func.count(EbookChunk.id)).where(EbookChunk.source_id == book_id)) or 0
|
||||||
|
|
||||||
|
|
||||||
|
async def get_candidate_count(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Return the number of indexed candidates for one book."""
|
||||||
|
return (
|
||||||
|
await session.scalar(select(func.count(EbookCandidatePhrase.id)).where(EbookCandidatePhrase.book_id == book_id))
|
||||||
|
or 0
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_judged_candidate_count(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Return the number of judged candidates for one book."""
|
||||||
|
return (
|
||||||
|
await session.scalar(
|
||||||
|
select(func.count(EbookCandidatePhrase.id)).where(
|
||||||
|
EbookCandidatePhrase.book_id == book_id,
|
||||||
|
EbookCandidatePhrase.llm_judged.is_(True),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
or 0
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_protected_count(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Return the number of protected phrases for one book."""
|
||||||
|
return (
|
||||||
|
await session.scalar(select(func.count(EbookProtectedPhrase.id)).where(EbookProtectedPhrase.book_id == book_id))
|
||||||
|
or 0
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_candidates(session: AsyncSession, book_id: int) -> list[EbookCandidatePhrase]:
|
||||||
|
"""Return the indexed candidates for one book."""
|
||||||
|
return list(
|
||||||
|
await session.scalars(
|
||||||
|
select(EbookCandidatePhrase)
|
||||||
|
.where(EbookCandidatePhrase.book_id == book_id)
|
||||||
|
.order_by(EbookCandidatePhrase.candidate_score.desc())
|
||||||
|
.limit(100)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_protected_phrases(session: AsyncSession, book_id: int) -> list[EbookProtectedPhrase]:
|
||||||
|
"""Return the protected phrases for one book."""
|
||||||
|
return list(
|
||||||
|
await session.scalars(
|
||||||
|
select(EbookProtectedPhrase)
|
||||||
|
.where(EbookProtectedPhrase.book_id == book_id)
|
||||||
|
.order_by(EbookProtectedPhrase.importance.desc())
|
||||||
|
.limit(100)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/books/{source_id}", response_class=HTMLResponse)
|
||||||
|
async def book_detail(source_id: int, request: Request, session: AsyncDbSession) -> HTMLResponse:
|
||||||
|
"""Render details for one indexed book."""
|
||||||
|
source = await session.get(EbookSource, source_id)
|
||||||
|
phrase_status_message = None
|
||||||
|
recalculated = request.query_params.get("phrases_recalculated")
|
||||||
|
if recalculated is not None:
|
||||||
|
phrase_status_message = f"Recalculated phrases; {recalculated} candidates generated"
|
||||||
|
judgment_outcome = pop_book_judgment_outcome(request.app, source_id)
|
||||||
|
if judgment_outcome is not None:
|
||||||
|
phrase_status_message = judgment_outcome
|
||||||
|
judging_in_progress = is_judging_book(request.app, source_id)
|
||||||
|
if judging_in_progress:
|
||||||
|
phrase_status_message = "Judging candidate phrases in the background; refresh to see progress"
|
||||||
|
if source is not None:
|
||||||
|
chapter_count = await get_chapter_count(session, source.id)
|
||||||
|
chunk_count = await get_chunk_count(session, source.id)
|
||||||
|
candidate_count = await get_candidate_count(session, source.id)
|
||||||
|
judged_candidate_count = await get_judged_candidate_count(session, source.id)
|
||||||
|
protected_count = await get_protected_count(session, source.id)
|
||||||
|
candidates = await get_candidates(session, source.id)
|
||||||
|
protected_phrases = await get_protected_phrases(session, source.id)
|
||||||
|
else:
|
||||||
|
chapter_count = 0
|
||||||
|
chunk_count = 0
|
||||||
|
candidate_count = 0
|
||||||
|
judged_candidate_count = 0
|
||||||
|
protected_count = 0
|
||||||
|
candidates = []
|
||||||
|
protected_phrases = []
|
||||||
|
logger.info(
|
||||||
|
"ebook_book_detail_loaded source_id=%s found=%s chapters=%s chunks=%s candidates=%s judged=%s protected=%s",
|
||||||
|
source_id,
|
||||||
|
source is not None,
|
||||||
|
chapter_count,
|
||||||
|
chunk_count,
|
||||||
|
candidate_count,
|
||||||
|
judged_candidate_count,
|
||||||
|
protected_count,
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"book_detail.html",
|
||||||
|
{
|
||||||
|
"candidate_count": candidate_count,
|
||||||
|
"candidates": candidates,
|
||||||
|
"chapter_count": chapter_count,
|
||||||
|
"chunk_count": chunk_count,
|
||||||
|
"judged_candidate_count": judged_candidate_count,
|
||||||
|
"judging_in_progress": judging_in_progress,
|
||||||
|
"protected_count": protected_count,
|
||||||
|
"protected_phrases": protected_phrases,
|
||||||
|
"phrase_status_message": phrase_status_message,
|
||||||
|
"source": source,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/books/{source_id}/recalculate-phrases")
|
||||||
|
async def recalculate_book_phrases(source_id: int, config: AppConfig, session: AsyncDbSession) -> RedirectResponse:
|
||||||
|
"""Clear and regenerate candidate phrases for one indexed book."""
|
||||||
|
source = await session.get(EbookSource, source_id)
|
||||||
|
if source is None:
|
||||||
|
raise HTTPException(status_code=404, detail="Book not found")
|
||||||
|
|
||||||
|
result = await recalculate_candidate_phrases_for_book(session, source, config, use_process_pool=True)
|
||||||
|
logger.info(
|
||||||
|
"ebook_book_phrase_recalculation_complete source_id=%s candidates=%s deleted_candidates=%s "
|
||||||
|
"deleted_protected=%s deleted_aliases=%s deleted_mentions=%s",
|
||||||
|
source_id,
|
||||||
|
result.candidate_phrases,
|
||||||
|
result.deleted_candidates,
|
||||||
|
result.deleted_protected_phrases,
|
||||||
|
result.deleted_aliases,
|
||||||
|
result.deleted_mentions,
|
||||||
|
)
|
||||||
|
return RedirectResponse(
|
||||||
|
url=f"/books/{source_id}?phrases_recalculated={result.candidate_phrases}",
|
||||||
|
status_code=303,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/books/{source_id}/judge-phrases")
|
||||||
|
async def judge_book_phrases(
|
||||||
|
source_id: int,
|
||||||
|
request: Request,
|
||||||
|
background_tasks: BackgroundTasks,
|
||||||
|
session: AsyncDbSession,
|
||||||
|
) -> RedirectResponse:
|
||||||
|
"""Queue background judging of one book's candidate phrases and return immediately."""
|
||||||
|
source = await session.get(EbookSource, source_id)
|
||||||
|
if source is None:
|
||||||
|
raise HTTPException(status_code=404, detail="Book not found")
|
||||||
|
|
||||||
|
started = start_book_phrase_judgment(request.app, background_tasks, source.id)
|
||||||
|
logger.info("ebook_book_phrase_judgment_requested source_id=%s started=%s", source_id, started)
|
||||||
|
return RedirectResponse(url=f"/books/{source_id}", status_code=303)
|
||||||
@@ -0,0 +1,129 @@
|
|||||||
|
"""Search routes for the EPUB search web UI."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import replace
|
||||||
|
from time import perf_counter
|
||||||
|
from typing import TYPE_CHECKING, Annotated
|
||||||
|
|
||||||
|
from fastapi import APIRouter, Form, Request
|
||||||
|
from fastapi.responses import HTMLResponse
|
||||||
|
|
||||||
|
from python.ebook_search.answer import answer_query
|
||||||
|
from python.ebook_search.api.dependencies import ( # noqa: TC001 FastAPI resolves these annotated dependencies at runtime
|
||||||
|
AppConfig,
|
||||||
|
AppEngine,
|
||||||
|
AppHttpClient,
|
||||||
|
)
|
||||||
|
from python.ebook_search.api.web import templates
|
||||||
|
from python.ebook_search.guardrails import (
|
||||||
|
CitationReport,
|
||||||
|
is_confident,
|
||||||
|
retrieval_confidence,
|
||||||
|
validate_citations,
|
||||||
|
)
|
||||||
|
from python.ebook_search.search import SearchResponse, search_ebooks
|
||||||
|
from python.ebook_search.timing import runtime_step_from_start
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
router = APIRouter()
|
||||||
|
|
||||||
|
|
||||||
|
async def build_answer(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
query: str,
|
||||||
|
response: SearchResponse,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> tuple[str, bool, CitationReport | None]:
|
||||||
|
"""Generate the answer for a search, returning ``(answer, low_confidence, citation_report)``."""
|
||||||
|
if not config.answer_enabled:
|
||||||
|
logger.info("ebook_answer_skipped_disabled")
|
||||||
|
return "Answer generation is disabled. Source chunks are shown below.", False, None
|
||||||
|
|
||||||
|
if not is_confident(response.results, config):
|
||||||
|
logger.info(
|
||||||
|
"ebook_answer_low_confidence confidence=%.4f threshold=%.4f",
|
||||||
|
retrieval_confidence(response.results),
|
||||||
|
config.min_retrieval_confidence,
|
||||||
|
)
|
||||||
|
answer = (
|
||||||
|
"Retrieval confidence is low for this query, so answer generation was skipped. "
|
||||||
|
"Source chunks are shown below."
|
||||||
|
)
|
||||||
|
return answer, True, None
|
||||||
|
|
||||||
|
try:
|
||||||
|
answer = await answer_query(client, query, response.results, config)
|
||||||
|
except RuntimeError as error:
|
||||||
|
logger.warning("ebook_answer_request_failed_falling_back error=%s", error)
|
||||||
|
return "Answer generation failed. Source chunks are still shown below.", False, None
|
||||||
|
|
||||||
|
citation_report = None
|
||||||
|
if config.validate_citations_enabled and response.results:
|
||||||
|
citation_report = validate_citations(answer, len(response.results))
|
||||||
|
if citation_report.invalid or not citation_report.grounded:
|
||||||
|
logger.warning(
|
||||||
|
"ebook_answer_citation_issue invalid=%s grounded=%s",
|
||||||
|
citation_report.invalid,
|
||||||
|
citation_report.grounded,
|
||||||
|
)
|
||||||
|
return answer, False, citation_report
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/search", response_class=HTMLResponse)
|
||||||
|
async def search(
|
||||||
|
request: Request,
|
||||||
|
config: AppConfig,
|
||||||
|
engine: AppEngine,
|
||||||
|
client: AppHttpClient,
|
||||||
|
query: Annotated[str, Form()],
|
||||||
|
rerank: Annotated[str | None, Form()] = None,
|
||||||
|
phrase_matching: Annotated[str | None, Form()] = None,
|
||||||
|
) -> HTMLResponse:
|
||||||
|
"""Run a search and render HTMX results."""
|
||||||
|
try:
|
||||||
|
response = await search_ebooks(
|
||||||
|
engine,
|
||||||
|
client,
|
||||||
|
query,
|
||||||
|
config,
|
||||||
|
rerank=rerank == "true",
|
||||||
|
phrase_matching=phrase_matching == "true",
|
||||||
|
)
|
||||||
|
except Exception as error:
|
||||||
|
logger.exception("ebook_search_request_failed")
|
||||||
|
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||||
|
|
||||||
|
answer_start = perf_counter()
|
||||||
|
answer, low_confidence, citation_report = await build_answer(client, query, response, config)
|
||||||
|
answer_step_name = "Answer generation" if config.answer_enabled else "Answer skipped"
|
||||||
|
response = replace(
|
||||||
|
response,
|
||||||
|
timings=(*response.timings, runtime_step_from_start(answer_step_name, answer_start)),
|
||||||
|
)
|
||||||
|
|
||||||
|
for step in response.timings:
|
||||||
|
logger.info("ebook_search_timing step=%r runtime_ms=%.1f", step.name, step.duration_ms)
|
||||||
|
logger.info(
|
||||||
|
"ebook_search_request_complete results=%s rank_label=%s runtime_ms=%.1f",
|
||||||
|
len(response.results),
|
||||||
|
response.rank_label,
|
||||||
|
response.total_runtime_ms,
|
||||||
|
)
|
||||||
|
return templates.TemplateResponse(
|
||||||
|
request,
|
||||||
|
"partials/results.html",
|
||||||
|
{
|
||||||
|
"answer": answer,
|
||||||
|
"response": response,
|
||||||
|
"low_confidence": low_confidence,
|
||||||
|
"citation_report": citation_report,
|
||||||
|
},
|
||||||
|
)
|
||||||
@@ -0,0 +1,447 @@
|
|||||||
|
:root {
|
||||||
|
--bg: #f4f5f7;
|
||||||
|
--surface: #ffffff;
|
||||||
|
--border: #e3e5ea;
|
||||||
|
--text: #1c1f24;
|
||||||
|
--muted: #6b7280;
|
||||||
|
--accent: #4f46e5;
|
||||||
|
--accent-soft: #eef0fe;
|
||||||
|
--danger: #b42318;
|
||||||
|
--warn-bg: #fff8eb;
|
||||||
|
--warn-border: #e0a92e;
|
||||||
|
--warn-text: #7a5008;
|
||||||
|
--radius: 12px;
|
||||||
|
--shadow: 0 1px 2px rgba(16, 24, 40, 0.04), 0 1px 3px rgba(16, 24, 40, 0.08);
|
||||||
|
}
|
||||||
|
|
||||||
|
html.theme-dark {
|
||||||
|
--bg: #0f1117;
|
||||||
|
--surface: #1a1d25;
|
||||||
|
--border: #2b303b;
|
||||||
|
--text: #e6e8ec;
|
||||||
|
--muted: #9aa1ad;
|
||||||
|
--accent: #818cf8;
|
||||||
|
--accent-soft: #262b45;
|
||||||
|
--danger: #f97066;
|
||||||
|
--warn-bg: #2a2410;
|
||||||
|
--warn-border: #b9881f;
|
||||||
|
--warn-text: #e8c97a;
|
||||||
|
--shadow: 0 1px 2px rgba(0, 0, 0, 0.3), 0 1px 3px rgba(0, 0, 0, 0.4);
|
||||||
|
color-scheme: dark;
|
||||||
|
}
|
||||||
|
|
||||||
|
* {
|
||||||
|
box-sizing: border-box;
|
||||||
|
}
|
||||||
|
|
||||||
|
body {
|
||||||
|
margin: 0;
|
||||||
|
background: var(--bg);
|
||||||
|
color: var(--text);
|
||||||
|
font-family: system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif;
|
||||||
|
line-height: 1.55;
|
||||||
|
}
|
||||||
|
|
||||||
|
main {
|
||||||
|
max-width: 820px;
|
||||||
|
margin: 0 auto;
|
||||||
|
padding: 32px 20px 64px;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Header / nav */
|
||||||
|
.site-header {
|
||||||
|
background: var(--surface);
|
||||||
|
border-bottom: 1px solid var(--border);
|
||||||
|
position: sticky;
|
||||||
|
top: 0;
|
||||||
|
z-index: 10;
|
||||||
|
}
|
||||||
|
|
||||||
|
.site-nav {
|
||||||
|
max-width: 820px;
|
||||||
|
margin: 0 auto;
|
||||||
|
padding: 12px 20px;
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 20px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.brand {
|
||||||
|
font-weight: 700;
|
||||||
|
font-size: 1.05rem;
|
||||||
|
color: var(--text);
|
||||||
|
text-decoration: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-links {
|
||||||
|
display: flex;
|
||||||
|
gap: 6px;
|
||||||
|
margin-right: auto;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-links a {
|
||||||
|
padding: 6px 12px;
|
||||||
|
border-radius: 8px;
|
||||||
|
color: var(--muted);
|
||||||
|
text-decoration: none;
|
||||||
|
font-size: 0.94rem;
|
||||||
|
transition: background 0.15s, color 0.15s;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-links a:hover {
|
||||||
|
background: var(--accent-soft);
|
||||||
|
color: var(--accent);
|
||||||
|
}
|
||||||
|
|
||||||
|
.dev-toggle {
|
||||||
|
display: inline-flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 6px;
|
||||||
|
font-size: 0.85rem;
|
||||||
|
color: var(--muted);
|
||||||
|
cursor: pointer;
|
||||||
|
user-select: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.theme-toggle {
|
||||||
|
display: inline-flex;
|
||||||
|
align-items: center;
|
||||||
|
justify-content: center;
|
||||||
|
width: 34px;
|
||||||
|
height: 34px;
|
||||||
|
padding: 0;
|
||||||
|
font-size: 1rem;
|
||||||
|
line-height: 1;
|
||||||
|
color: var(--text);
|
||||||
|
background: var(--bg);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: 8px;
|
||||||
|
cursor: pointer;
|
||||||
|
}
|
||||||
|
|
||||||
|
.theme-toggle:hover {
|
||||||
|
border-color: var(--accent);
|
||||||
|
filter: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
h1 {
|
||||||
|
font-size: 1.6rem;
|
||||||
|
margin: 0 0 20px;
|
||||||
|
}
|
||||||
|
|
||||||
|
h2 {
|
||||||
|
font-size: 1.15rem;
|
||||||
|
margin: 0 0 8px;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Cards */
|
||||||
|
.card {
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: var(--radius);
|
||||||
|
box-shadow: var(--shadow);
|
||||||
|
padding: 20px;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Search form */
|
||||||
|
form {
|
||||||
|
margin: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
label {
|
||||||
|
font-weight: 600;
|
||||||
|
font-size: 0.92rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
textarea {
|
||||||
|
display: block;
|
||||||
|
width: 100%;
|
||||||
|
margin: 8px 0 16px;
|
||||||
|
padding: 12px 14px;
|
||||||
|
font: inherit;
|
||||||
|
color: var(--text);
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: 10px;
|
||||||
|
resize: vertical;
|
||||||
|
transition: border-color 0.15s, box-shadow 0.15s;
|
||||||
|
}
|
||||||
|
|
||||||
|
textarea:focus {
|
||||||
|
outline: none;
|
||||||
|
border-color: var(--accent);
|
||||||
|
box-shadow: 0 0 0 3px var(--accent-soft);
|
||||||
|
}
|
||||||
|
|
||||||
|
.form-row {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 12px;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
}
|
||||||
|
|
||||||
|
.search-toggles {
|
||||||
|
display: flex;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 14px;
|
||||||
|
}
|
||||||
|
|
||||||
|
button {
|
||||||
|
padding: 10px 20px;
|
||||||
|
font: inherit;
|
||||||
|
font-weight: 600;
|
||||||
|
color: #fff;
|
||||||
|
background: var(--accent);
|
||||||
|
border: none;
|
||||||
|
border-radius: 10px;
|
||||||
|
cursor: pointer;
|
||||||
|
transition: filter 0.15s;
|
||||||
|
}
|
||||||
|
|
||||||
|
button:hover {
|
||||||
|
filter: brightness(1.08);
|
||||||
|
}
|
||||||
|
|
||||||
|
.check {
|
||||||
|
display: inline-flex;
|
||||||
|
gap: 8px;
|
||||||
|
align-items: center;
|
||||||
|
font-weight: 500;
|
||||||
|
color: var(--muted);
|
||||||
|
}
|
||||||
|
|
||||||
|
.actions {
|
||||||
|
display: flex;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 12px;
|
||||||
|
margin-bottom: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.actions-grid {
|
||||||
|
display: grid;
|
||||||
|
grid-template-columns: repeat(2, max-content);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Answer + results */
|
||||||
|
#results {
|
||||||
|
display: block;
|
||||||
|
margin-top: 28px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.rank-label {
|
||||||
|
font-size: 0.82rem;
|
||||||
|
font-weight: 600;
|
||||||
|
text-transform: uppercase;
|
||||||
|
letter-spacing: 0.04em;
|
||||||
|
color: var(--muted);
|
||||||
|
margin-bottom: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.answer {
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: var(--radius);
|
||||||
|
box-shadow: var(--shadow);
|
||||||
|
padding: 20px;
|
||||||
|
margin-bottom: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.answer p:last-child {
|
||||||
|
margin-bottom: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.results {
|
||||||
|
list-style: none;
|
||||||
|
padding: 0;
|
||||||
|
margin: 0;
|
||||||
|
display: grid;
|
||||||
|
gap: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.results > li {
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: var(--radius);
|
||||||
|
box-shadow: var(--shadow);
|
||||||
|
padding: 18px 20px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.results h2 {
|
||||||
|
font-size: 1.05rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.results h2 a {
|
||||||
|
color: var(--text);
|
||||||
|
text-decoration: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.results h2 a:hover {
|
||||||
|
color: var(--accent);
|
||||||
|
}
|
||||||
|
|
||||||
|
.meta {
|
||||||
|
color: var(--muted);
|
||||||
|
font-size: 0.88rem;
|
||||||
|
margin: 0 0 10px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.scores {
|
||||||
|
display: flex;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 8px;
|
||||||
|
margin: 14px 0 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.scores div {
|
||||||
|
display: inline-flex;
|
||||||
|
gap: 6px;
|
||||||
|
align-items: baseline;
|
||||||
|
padding: 3px 10px;
|
||||||
|
background: var(--bg);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: 999px;
|
||||||
|
font-size: 0.78rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.scores dt {
|
||||||
|
font-weight: 600;
|
||||||
|
color: var(--muted);
|
||||||
|
}
|
||||||
|
|
||||||
|
.scores dd {
|
||||||
|
margin: 0;
|
||||||
|
font-variant-numeric: tabular-nums;
|
||||||
|
}
|
||||||
|
|
||||||
|
.phrase-matches {
|
||||||
|
display: flex;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 8px;
|
||||||
|
align-items: baseline;
|
||||||
|
margin: 10px 0 0;
|
||||||
|
font-size: 0.78rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.phrase-matches-label {
|
||||||
|
color: var(--muted);
|
||||||
|
font-weight: 600;
|
||||||
|
}
|
||||||
|
|
||||||
|
.phrase-match {
|
||||||
|
padding: 3px 10px;
|
||||||
|
background: var(--bg);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: 999px;
|
||||||
|
color: var(--accent);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Runtime — developer diagnostics, hidden unless dev mode is on */
|
||||||
|
.runtime {
|
||||||
|
display: none;
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: var(--radius);
|
||||||
|
box-shadow: var(--shadow);
|
||||||
|
padding: 18px 20px;
|
||||||
|
margin-bottom: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
html.dev .runtime {
|
||||||
|
display: block;
|
||||||
|
}
|
||||||
|
|
||||||
|
.timing-chart {
|
||||||
|
display: grid;
|
||||||
|
gap: 8px;
|
||||||
|
padding: 0;
|
||||||
|
margin: 12px 0 0;
|
||||||
|
list-style: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.timing-chart li {
|
||||||
|
display: grid;
|
||||||
|
grid-template-columns: minmax(150px, 1fr) minmax(160px, 2fr) auto auto;
|
||||||
|
gap: 10px;
|
||||||
|
align-items: center;
|
||||||
|
font-size: 0.85rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.timing-bar {
|
||||||
|
height: 8px;
|
||||||
|
overflow: hidden;
|
||||||
|
background: var(--bg);
|
||||||
|
border-radius: 999px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.timing-bar span {
|
||||||
|
display: block;
|
||||||
|
height: 100%;
|
||||||
|
background: var(--accent);
|
||||||
|
border-radius: 999px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.timing-value,
|
||||||
|
.timing-remaining {
|
||||||
|
color: var(--muted);
|
||||||
|
font-variant-numeric: tabular-nums;
|
||||||
|
text-align: right;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Tables */
|
||||||
|
table {
|
||||||
|
width: 100%;
|
||||||
|
border-collapse: collapse;
|
||||||
|
background: var(--surface);
|
||||||
|
border: 1px solid var(--border);
|
||||||
|
border-radius: var(--radius);
|
||||||
|
overflow: hidden;
|
||||||
|
}
|
||||||
|
|
||||||
|
th,
|
||||||
|
td {
|
||||||
|
padding: 10px 14px;
|
||||||
|
border-bottom: 1px solid var(--border);
|
||||||
|
text-align: left;
|
||||||
|
font-size: 0.9rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
th {
|
||||||
|
font-weight: 600;
|
||||||
|
color: var(--muted);
|
||||||
|
background: var(--bg);
|
||||||
|
}
|
||||||
|
|
||||||
|
tbody tr:last-child td {
|
||||||
|
border-bottom: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
dl dt {
|
||||||
|
font-weight: 600;
|
||||||
|
color: var(--muted);
|
||||||
|
font-size: 0.85rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
dl dd {
|
||||||
|
margin: 0 0 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* States */
|
||||||
|
.error {
|
||||||
|
color: var(--danger);
|
||||||
|
font-weight: 600;
|
||||||
|
}
|
||||||
|
|
||||||
|
.notice {
|
||||||
|
margin: 12px 0;
|
||||||
|
padding: 10px 14px;
|
||||||
|
border-left: 3px solid var(--warn-border);
|
||||||
|
border-radius: 6px;
|
||||||
|
background: var(--warn-bg);
|
||||||
|
color: var(--warn-text);
|
||||||
|
font-weight: 500;
|
||||||
|
}
|
||||||
|
|
||||||
|
.status {
|
||||||
|
color: var(--muted);
|
||||||
|
}
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
{% extends "base.html" %} {% block title %}EPUB Admin{% endblock %} {% block
|
||||||
|
head %}
|
||||||
|
<script src="https://unpkg.com/htmx.org@2.0.4"></script>
|
||||||
|
{% endblock %} {% block content %}
|
||||||
|
<h1>Admin</h1>
|
||||||
|
<section id="admin-status"></section>
|
||||||
|
<section class="actions">
|
||||||
|
<form hx-post="/admin/scan" hx-target="#admin-status" hx-swap="innerHTML">
|
||||||
|
<button type="submit">Scan</button>
|
||||||
|
</form>
|
||||||
|
</section>
|
||||||
|
<section>
|
||||||
|
<h2>Embeddings</h2>
|
||||||
|
<section class="actions">
|
||||||
|
<form
|
||||||
|
hx-post="/admin/embed-missing"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Embed</button>
|
||||||
|
</form>
|
||||||
|
<form
|
||||||
|
hx-post="/admin/embed-all"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Embed all</button>
|
||||||
|
</form>
|
||||||
|
</section>
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Model</th>
|
||||||
|
<th>Dimensions</th>
|
||||||
|
<th>Embedded</th>
|
||||||
|
<th>Missing</th>
|
||||||
|
<th>Total chunks</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
{% for item in stats %}
|
||||||
|
<tr>
|
||||||
|
<td>{{ item.model_name }}</td>
|
||||||
|
<td>{{ item.dimension }}</td>
|
||||||
|
<td>{{ item.embedded_chunks }}</td>
|
||||||
|
<td>{{ item.missing_chunks }}</td>
|
||||||
|
<td>{{ item.total_chunks }}</td>
|
||||||
|
</tr>
|
||||||
|
{% endfor %}
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
</section>
|
||||||
|
<section>
|
||||||
|
<h2>Protected phrases</h2>
|
||||||
|
<section class="actions actions-grid">
|
||||||
|
<form
|
||||||
|
hx-post="/admin/phrases/generate-all"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Regenerate all phrases</button>
|
||||||
|
</form>
|
||||||
|
<form
|
||||||
|
hx-post="/admin/phrases/generate-missing"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Add missing phrases</button>
|
||||||
|
</form>
|
||||||
|
<form
|
||||||
|
hx-post="/admin/phrases/judge-all"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Judge all phrases</button>
|
||||||
|
</form>
|
||||||
|
<form
|
||||||
|
hx-post="/admin/phrases/judge-missing"
|
||||||
|
hx-target="#admin-status"
|
||||||
|
hx-swap="innerHTML"
|
||||||
|
>
|
||||||
|
<button type="submit">Judge missing phrases</button>
|
||||||
|
</form>
|
||||||
|
</section>
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Candidates</th>
|
||||||
|
<th>Judged</th>
|
||||||
|
<th>Unjudged</th>
|
||||||
|
<th>Protected</th>
|
||||||
|
<th>Books indexed</th>
|
||||||
|
<th>Books generated</th>
|
||||||
|
<th>Books fully judged</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
<tr>
|
||||||
|
<td>{{ phrase_stats.candidate_phrases }}</td>
|
||||||
|
<td>{{ phrase_stats.judged_candidates }}</td>
|
||||||
|
<td>{{ phrase_stats.unjudged_candidates }}</td>
|
||||||
|
<td>{{ phrase_stats.protected_phrases }}</td>
|
||||||
|
<td>{{ phrase_stats.total_books }}</td>
|
||||||
|
<td>{{ phrase_stats.books_with_candidates }}</td>
|
||||||
|
<td>{{ phrase_stats.books_fully_judged }}</td>
|
||||||
|
</tr>
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
</section>
|
||||||
|
{% endblock %}
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
<!DOCTYPE html>
|
||||||
|
<html lang="en">
|
||||||
|
<head>
|
||||||
|
<meta charset="UTF-8">
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||||
|
<title>{% block title %}EPUB Search{% endblock %}</title>
|
||||||
|
{% block head %}{% endblock %}
|
||||||
|
<link rel="stylesheet" href="/static/style.css?v={{ static_version('style.css') }}">
|
||||||
|
<script>
|
||||||
|
// Apply theme and dev mode before paint to avoid a flash of unstyled/wrong content.
|
||||||
|
(function () {
|
||||||
|
var stored = localStorage.getItem("ebook-theme");
|
||||||
|
var prefersDark = window.matchMedia("(prefers-color-scheme: dark)").matches;
|
||||||
|
var theme = stored || (prefersDark ? "dark" : "light");
|
||||||
|
document.documentElement.classList.add("theme-" + theme);
|
||||||
|
if (localStorage.getItem("ebook-dev-mode") === "on") {
|
||||||
|
document.documentElement.classList.add("dev");
|
||||||
|
}
|
||||||
|
})();
|
||||||
|
</script>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<header class="site-header">
|
||||||
|
<nav class="site-nav">
|
||||||
|
<a class="brand" href="/">EPUB Search</a>
|
||||||
|
<div class="nav-links">
|
||||||
|
<a href="/">Search</a>
|
||||||
|
<a href="/books">Books</a>
|
||||||
|
<a href="/admin">Admin</a>
|
||||||
|
</div>
|
||||||
|
<button type="button" id="theme-toggle" class="theme-toggle" title="Toggle light / dark theme" aria-label="Toggle theme"></button>
|
||||||
|
<label class="dev-toggle" title="Show developer diagnostics">
|
||||||
|
<input type="checkbox" id="dev-mode-toggle">
|
||||||
|
<span>Dev</span>
|
||||||
|
</label>
|
||||||
|
</nav>
|
||||||
|
</header>
|
||||||
|
<main>
|
||||||
|
{% block content %}{% endblock %}
|
||||||
|
</main>
|
||||||
|
<script>
|
||||||
|
(function () {
|
||||||
|
var toggle = document.getElementById("dev-mode-toggle");
|
||||||
|
if (toggle) {
|
||||||
|
toggle.checked = document.documentElement.classList.contains("dev");
|
||||||
|
toggle.addEventListener("change", function () {
|
||||||
|
document.documentElement.classList.toggle("dev", toggle.checked);
|
||||||
|
localStorage.setItem("ebook-dev-mode", toggle.checked ? "on" : "off");
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
var themeButton = document.getElementById("theme-toggle");
|
||||||
|
if (themeButton) {
|
||||||
|
var root = document.documentElement;
|
||||||
|
var sync = function () {
|
||||||
|
var isDark = root.classList.contains("theme-dark");
|
||||||
|
themeButton.textContent = isDark ? "☀️" : "🌙";
|
||||||
|
};
|
||||||
|
sync();
|
||||||
|
themeButton.addEventListener("click", function () {
|
||||||
|
var next = root.classList.contains("theme-dark") ? "light" : "dark";
|
||||||
|
root.classList.remove("theme-dark", "theme-light");
|
||||||
|
root.classList.add("theme-" + next);
|
||||||
|
localStorage.setItem("ebook-theme", next);
|
||||||
|
sync();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
})();
|
||||||
|
</script>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
@@ -0,0 +1,109 @@
|
|||||||
|
{% extends "base.html" %}
|
||||||
|
|
||||||
|
{% block title %}{% if source %}{{ source.title }}{% else %}Book not found{% endif %}{% endblock %}
|
||||||
|
|
||||||
|
{% block content %}
|
||||||
|
{% if source %}
|
||||||
|
<h1>{{ source.title }}</h1>
|
||||||
|
<p class="meta">{{ source.author or "Unknown author" }}</p>
|
||||||
|
{% if phrase_status_message %}
|
||||||
|
<p class="status">{{ phrase_status_message }}</p>
|
||||||
|
{% endif %}
|
||||||
|
<dl class="card">
|
||||||
|
<dt>File</dt>
|
||||||
|
<dd>{{ source.file_path }}</dd>
|
||||||
|
<dt>Chapters</dt>
|
||||||
|
<dd>{{ chapter_count }}</dd>
|
||||||
|
<dt>Chunks</dt>
|
||||||
|
<dd>{{ chunk_count }}</dd>
|
||||||
|
<dt>Candidates</dt>
|
||||||
|
<dd>{{ candidate_count }}</dd>
|
||||||
|
<dt>Judged</dt>
|
||||||
|
<dd>{{ judged_candidate_count }}</dd>
|
||||||
|
<dt>Protected</dt>
|
||||||
|
<dd>{{ protected_count }}</dd>
|
||||||
|
</dl>
|
||||||
|
<form
|
||||||
|
method="post"
|
||||||
|
action="/books/{{ source.id }}/recalculate-phrases"
|
||||||
|
onsubmit="return confirm('Remove old phrases for this book and generate new candidates?');"
|
||||||
|
>
|
||||||
|
<button type="submit">Recalculate phrases</button>
|
||||||
|
</form>
|
||||||
|
<form
|
||||||
|
method="post"
|
||||||
|
action="/books/{{ source.id }}/judge-phrases"
|
||||||
|
onsubmit="return confirm('Judge candidate phrases for this book with the LLM?');"
|
||||||
|
>
|
||||||
|
<button type="submit"{% if judging_in_progress %} disabled{% endif %}>
|
||||||
|
{% if judging_in_progress %}Judging…{% else %}Judge phrases{% endif %}
|
||||||
|
</button>
|
||||||
|
</form>
|
||||||
|
|
||||||
|
<section>
|
||||||
|
<h2>Candidate n-grams</h2>
|
||||||
|
{% if candidates %}
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Phrase</th>
|
||||||
|
<th>Status</th>
|
||||||
|
<th>Score</th>
|
||||||
|
<th>Count</th>
|
||||||
|
<th>Chapters</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
{% for candidate in candidates %}
|
||||||
|
<tr>
|
||||||
|
<td>{{ candidate.phrase_text }}</td>
|
||||||
|
<td>
|
||||||
|
{% if candidate.llm_judged %}
|
||||||
|
{% if candidate.llm_keep %}Kept{% else %}Rejected{% endif %}
|
||||||
|
{% else %}
|
||||||
|
Candidate
|
||||||
|
{% endif %}
|
||||||
|
</td>
|
||||||
|
<td>{{ "%.2f"|format(candidate.candidate_score) }}</td>
|
||||||
|
<td>{{ candidate.raw_count }}</td>
|
||||||
|
<td>{{ candidate.chapter_count }}</td>
|
||||||
|
</tr>
|
||||||
|
{% endfor %}
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
{% else %}
|
||||||
|
<p>No candidate n-grams.</p>
|
||||||
|
{% endif %}
|
||||||
|
</section>
|
||||||
|
|
||||||
|
<section>
|
||||||
|
<h2>Protected phrases</h2>
|
||||||
|
{% if protected_phrases %}
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Phrase</th>
|
||||||
|
<th>Type</th>
|
||||||
|
<th>Confidence</th>
|
||||||
|
<th>Importance</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
{% for phrase in protected_phrases %}
|
||||||
|
<tr>
|
||||||
|
<td>{{ phrase.phrase_text }}</td>
|
||||||
|
<td>{{ phrase.phrase_type or "phrase" }}</td>
|
||||||
|
<td>{{ "%.2f"|format(phrase.confidence) }}</td>
|
||||||
|
<td>{{ "%.2f"|format(phrase.importance) }}</td>
|
||||||
|
</tr>
|
||||||
|
{% endfor %}
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
{% else %}
|
||||||
|
<p>No protected phrases.</p>
|
||||||
|
{% endif %}
|
||||||
|
</section>
|
||||||
|
{% else %}
|
||||||
|
<h1>Book not found</h1>
|
||||||
|
{% endif %}
|
||||||
|
{% endblock %}
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
{% extends "base.html" %}
|
||||||
|
|
||||||
|
{% block title %}EPUB Books{% endblock %}
|
||||||
|
|
||||||
|
{% block content %}
|
||||||
|
<h1>Books</h1>
|
||||||
|
{% if sources %}
|
||||||
|
<ol class="results">
|
||||||
|
{% for source in sources %}
|
||||||
|
<li>
|
||||||
|
<h2><a href="/books/{{ source.id }}">{{ source.title }}</a></h2>
|
||||||
|
<p class="meta">{{ source.author or "Unknown author" }}</p>
|
||||||
|
</li>
|
||||||
|
{% endfor %}
|
||||||
|
</ol>
|
||||||
|
{% else %}
|
||||||
|
<p>No EPUBs indexed.</p>
|
||||||
|
{% endif %}
|
||||||
|
{% endblock %}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
<p class="status">{{ message }}</p>
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
<p class="error">{{ message }}</p>
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
<div class="rank-label">{{ response.rank_label }}</div>
|
||||||
|
{% if response.timings %}
|
||||||
|
<section class="runtime">
|
||||||
|
<h2>Runtime</h2>
|
||||||
|
<p class="meta">Total {{ "%.1f"|format(response.total_runtime_ms) }} ms</p>
|
||||||
|
<ol class="timing-chart">
|
||||||
|
{% set total = response.total_runtime_ms %}
|
||||||
|
{% set ns = namespace(remaining=total) %}
|
||||||
|
{% for step in response.timings %}
|
||||||
|
{% set width = (step.duration_ms / total * 100) if total else 0 %}
|
||||||
|
{% if step.counts_toward_total %}
|
||||||
|
{% set ns.remaining = ns.remaining - step.duration_ms %}
|
||||||
|
{% endif %}
|
||||||
|
<li>
|
||||||
|
<span class="timing-label">{{ step.name }}</span>
|
||||||
|
<span class="timing-bar"><span style="width: {{ "%.2f"|format(width) }}%"></span></span>
|
||||||
|
<span class="timing-value">{{ "%.1f"|format(step.duration_ms) }} ms</span>
|
||||||
|
<span class="timing-remaining">{{ "%.1f"|format([ns.remaining, 0]|max) }} ms left</span>
|
||||||
|
</li>
|
||||||
|
{% endfor %}
|
||||||
|
</ol>
|
||||||
|
</section>
|
||||||
|
{% endif %}
|
||||||
|
<section class="answer">
|
||||||
|
<h2>Answer</h2>
|
||||||
|
{% if low_confidence|default(false) %}
|
||||||
|
<p class="notice">Low retrieval confidence — answer generation was skipped.</p>
|
||||||
|
{% endif %}
|
||||||
|
{% set report = citation_report|default(none) %}
|
||||||
|
{% if report is not none and not report.grounded %}
|
||||||
|
<p class="notice">Unverified — no source citations were found in this answer.</p>
|
||||||
|
{% endif %}
|
||||||
|
{% if report is not none and report.invalid %}
|
||||||
|
<p class="notice">Invalid citations: {{ report.invalid|join(", ") }} (no matching source).</p>
|
||||||
|
{% endif %}
|
||||||
|
<p>{{ answer }}</p>
|
||||||
|
</section>
|
||||||
|
{% if response.results %}
|
||||||
|
<ol class="results">
|
||||||
|
{% for result in response.results %}
|
||||||
|
<li>
|
||||||
|
<h2>
|
||||||
|
{% if result.source_id %}
|
||||||
|
<a href="/books/{{ result.source_id }}">{{ result.source_title }}</a>
|
||||||
|
{% else %}
|
||||||
|
{{ result.source_title }}
|
||||||
|
{% endif %}
|
||||||
|
</h2>
|
||||||
|
<p class="meta">
|
||||||
|
{% if result.source_author %}{{ result.source_author }}{% endif %}
|
||||||
|
{% if result.chapter_title %} · {{ result.chapter_title }}{% endif %}
|
||||||
|
{% if result.page_label %} · page {{ result.page_label }}{% endif %}
|
||||||
|
</p>
|
||||||
|
<p>{{ result.text }}</p>
|
||||||
|
<dl class="scores">
|
||||||
|
<div>
|
||||||
|
<dt>final</dt>
|
||||||
|
<dd>{{ "%.3f"|format(result.score) }}</dd>
|
||||||
|
</div>
|
||||||
|
{% if result.rerank_score is not none %}
|
||||||
|
<div>
|
||||||
|
<dt>rerank</dt>
|
||||||
|
<dd>{{ "%.3f"|format(result.rerank_score) }}</dd>
|
||||||
|
</div>
|
||||||
|
{% endif %}
|
||||||
|
{% if result.vector_score is not none %}
|
||||||
|
<div>
|
||||||
|
<dt>vector cosine</dt>
|
||||||
|
<dd>{{ "%.3f"|format(result.vector_score) }}</dd>
|
||||||
|
</div>
|
||||||
|
{% endif %}
|
||||||
|
{% if result.bm25_score is not none %}
|
||||||
|
<div>
|
||||||
|
<dt>BM25</dt>
|
||||||
|
<dd>{{ "%.6f"|format(result.bm25_score) }}</dd>
|
||||||
|
</div>
|
||||||
|
{% endif %}
|
||||||
|
{% if result.fused_score is not none %}
|
||||||
|
<div>
|
||||||
|
<dt>RRF</dt>
|
||||||
|
<dd>{{ "%.3f"|format(result.fused_score) }}</dd>
|
||||||
|
</div>
|
||||||
|
{% endif %}
|
||||||
|
</dl>
|
||||||
|
{% if result.matched_phrases %}
|
||||||
|
<p class="phrase-matches">
|
||||||
|
<span class="phrase-matches-label">boosted by</span>
|
||||||
|
{% for phrase in result.matched_phrases %}
|
||||||
|
<span class="phrase-match">{{ phrase }}</span>
|
||||||
|
{% endfor %}
|
||||||
|
</p>
|
||||||
|
{% endif %}
|
||||||
|
</li>
|
||||||
|
{% endfor %}
|
||||||
|
</ol>
|
||||||
|
{% else %}
|
||||||
|
<p>No results.</p>
|
||||||
|
{% endif %}
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
{% extends "base.html" %}
|
||||||
|
|
||||||
|
{% block title %}EPUB Search{% endblock %}
|
||||||
|
{% block head %}<script src="https://unpkg.com/htmx.org@2.0.4"></script>{% endblock %}
|
||||||
|
|
||||||
|
{% block content %}
|
||||||
|
<h1>Search</h1>
|
||||||
|
<form class="card" hx-post="/search" hx-target="#results" hx-swap="innerHTML">
|
||||||
|
<label for="query">What are you looking for?</label>
|
||||||
|
<textarea id="query" name="query" rows="4" placeholder="Ask a question or paste a passage…" required
|
||||||
|
onkeydown="if (event.key === 'Enter' && !event.shiftKey) { event.preventDefault(); this.form.requestSubmit(); }"></textarea>
|
||||||
|
<div class="form-row">
|
||||||
|
<div class="search-toggles">
|
||||||
|
<label class="check">
|
||||||
|
<input type="checkbox" name="rerank" value="true" {% if config.rerank.enabled %}checked{% endif %}>
|
||||||
|
Rerank
|
||||||
|
</label>
|
||||||
|
<label class="check">
|
||||||
|
<input
|
||||||
|
type="checkbox"
|
||||||
|
name="phrase_matching"
|
||||||
|
value="true"
|
||||||
|
{% if config.phrase_matching_enabled %}checked{% endif %}
|
||||||
|
>
|
||||||
|
Phrase matching
|
||||||
|
</label>
|
||||||
|
</div>
|
||||||
|
<button type="submit">Search</button>
|
||||||
|
</div>
|
||||||
|
</form>
|
||||||
|
<section id="results"></section>
|
||||||
|
{% endblock %}
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
"""Shared web UI resources for EPUB search."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from fastapi.templating import Jinja2Templates
|
||||||
|
|
||||||
|
PACKAGE_DIR = Path(__file__).resolve().parent
|
||||||
|
TEMPLATE_DIR = PACKAGE_DIR / "templates"
|
||||||
|
STATIC_DIR = PACKAGE_DIR / "static"
|
||||||
|
|
||||||
|
|
||||||
|
def static_version(filename: str) -> int:
|
||||||
|
"""Return a cache-busting token for a static file based on its modification time."""
|
||||||
|
try:
|
||||||
|
return int((STATIC_DIR / filename).stat().st_mtime)
|
||||||
|
except OSError:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
templates = Jinja2Templates(directory=TEMPLATE_DIR)
|
||||||
|
templates.env.globals["static_version"] = static_version
|
||||||
@@ -0,0 +1,286 @@
|
|||||||
|
"""Persisted BM25 corpus management."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import shutil
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from functools import cache
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import bm25s
|
||||||
|
from sqlalchemy import func, select, union_all
|
||||||
|
|
||||||
|
from python.orm.richie import EbookChapter, EbookChunk, EbookSource
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
MANIFEST_NAME = "manifest.json"
|
||||||
|
REQUIRED_INDEX_FILES = frozenset(
|
||||||
|
{
|
||||||
|
"data.csc.index.npy",
|
||||||
|
"indices.csc.index.npy",
|
||||||
|
"indptr.csc.index.npy",
|
||||||
|
"params.index.json",
|
||||||
|
"vocab.index.json",
|
||||||
|
"corpus.jsonl",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BM25Manifest:
|
||||||
|
"""Metadata describing a persisted BM25 corpus."""
|
||||||
|
|
||||||
|
created_at: datetime
|
||||||
|
db_updated_at: datetime | None
|
||||||
|
chunk_count: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BM25Corpus:
|
||||||
|
"""Loaded persisted BM25 corpus and retriever."""
|
||||||
|
|
||||||
|
retriever: object | None
|
||||||
|
records: tuple[dict[str, object], ...]
|
||||||
|
manifest: BM25Manifest
|
||||||
|
|
||||||
|
|
||||||
|
class BM25CorpusUnavailableError(RuntimeError):
|
||||||
|
"""Raised when the persisted BM25 corpus cannot be loaded."""
|
||||||
|
|
||||||
|
|
||||||
|
def bm25_index_path(config: EbookSearchConfig) -> Path:
|
||||||
|
"""Return the configured BM25 index root path relative to the current working directory."""
|
||||||
|
path = Path(config.bm25_index_dir).expanduser()
|
||||||
|
if path.is_absolute():
|
||||||
|
return path
|
||||||
|
return Path.cwd() / path
|
||||||
|
|
||||||
|
|
||||||
|
def get_current_bm25_index(index_path: Path) -> Path:
|
||||||
|
"""Return the live BM25 index directory."""
|
||||||
|
current_path = index_path / "current"
|
||||||
|
if current_path.exists() or current_path.is_symlink():
|
||||||
|
return current_path
|
||||||
|
return index_path
|
||||||
|
|
||||||
|
|
||||||
|
async def ensure_bm25_corpus(session: AsyncSession, config: EbookSearchConfig) -> None:
|
||||||
|
"""Create or refresh the persisted BM25 corpus when it is missing or stale."""
|
||||||
|
index_path = bm25_index_path(config)
|
||||||
|
manifest = read_bm25_manifest(index_path)
|
||||||
|
db_updated_at = await corpus_last_updated_at(session)
|
||||||
|
if not bm25_index_exists(index_path, manifest):
|
||||||
|
logger.info("ebook_bm25_index_missing path=%s", index_path)
|
||||||
|
await refresh_bm25_corpus(session, config, db_updated_at=db_updated_at)
|
||||||
|
return
|
||||||
|
if db_updated_at is not None and manifest is not None and manifest.created_at < db_updated_at:
|
||||||
|
logger.info(
|
||||||
|
"ebook_bm25_index_stale path=%s created_at=%s db_updated_at=%s",
|
||||||
|
index_path,
|
||||||
|
manifest.created_at.isoformat(),
|
||||||
|
db_updated_at.isoformat(),
|
||||||
|
)
|
||||||
|
await refresh_bm25_corpus(session, config, db_updated_at=db_updated_at)
|
||||||
|
return
|
||||||
|
logger.info(
|
||||||
|
"ebook_bm25_index_current path=%s chunks=%s created_at=%s",
|
||||||
|
index_path,
|
||||||
|
manifest.chunk_count if manifest else 0,
|
||||||
|
manifest.created_at.isoformat() if manifest else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_bm25_corpus(
|
||||||
|
session: AsyncSession,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
db_updated_at: datetime | None = None,
|
||||||
|
) -> BM25Manifest:
|
||||||
|
"""Rebuild and persist the BM25 corpus from the current database chunks.
|
||||||
|
|
||||||
|
The index build is CPU and disk work, so it runs in a worker thread.
|
||||||
|
"""
|
||||||
|
index_path = bm25_index_path(config)
|
||||||
|
records, texts = await fetch_bm25_corpus_records(session)
|
||||||
|
manifest = BM25Manifest(
|
||||||
|
created_at=datetime.now(tz=UTC),
|
||||||
|
db_updated_at=db_updated_at if db_updated_at is not None else await corpus_last_updated_at(session),
|
||||||
|
chunk_count=len(records),
|
||||||
|
)
|
||||||
|
await asyncio.to_thread(write_bm25_corpus, index_path, records, texts, manifest)
|
||||||
|
logger.info(
|
||||||
|
"ebook_bm25_index_refreshed path=%s chunks=%s created_at=%s",
|
||||||
|
index_path,
|
||||||
|
manifest.chunk_count,
|
||||||
|
manifest.created_at.isoformat(),
|
||||||
|
)
|
||||||
|
return manifest
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def load_bm25_corpus(config: EbookSearchConfig) -> BM25Corpus:
|
||||||
|
"""Load the BM25 corpus into memory once per process.
|
||||||
|
|
||||||
|
Background refresh tasks clear this cache after rebuilding the on-disk corpus.
|
||||||
|
"""
|
||||||
|
index_path = bm25_index_path(config)
|
||||||
|
active_index_path = get_current_bm25_index(index_path)
|
||||||
|
logger.info("ebook_bm25_corpus_cache_load path=%s active_path=%s", index_path, active_index_path)
|
||||||
|
manifest = read_bm25_manifest(index_path)
|
||||||
|
if manifest is None or not bm25_index_exists(index_path, manifest):
|
||||||
|
msg = f"BM25 corpus is not available: {index_path}"
|
||||||
|
raise BM25CorpusUnavailableError(msg)
|
||||||
|
if manifest.chunk_count == 0:
|
||||||
|
return BM25Corpus(retriever=None, records=(), manifest=manifest)
|
||||||
|
|
||||||
|
retriever = bm25s.BM25.load(active_index_path, load_corpus=True, mmap=True)
|
||||||
|
records = tuple(dict(record) for record in retriever.corpus)
|
||||||
|
return BM25Corpus(retriever=retriever, records=records, manifest=manifest)
|
||||||
|
|
||||||
|
|
||||||
|
def score_bm25_corpus(query: str, corpus: BM25Corpus, *, limit: int) -> list[tuple[dict[str, object], float]]:
|
||||||
|
"""Score a query against a loaded BM25 corpus."""
|
||||||
|
if corpus.retriever is None or not corpus.records:
|
||||||
|
return []
|
||||||
|
k = min(limit, len(corpus.records))
|
||||||
|
documents, scores = corpus.retriever.retrieve(
|
||||||
|
bm25s.tokenize(query, show_progress=False),
|
||||||
|
corpus=list(corpus.records),
|
||||||
|
k=k,
|
||||||
|
show_progress=False,
|
||||||
|
)
|
||||||
|
results: list[tuple[dict[str, object], float]] = []
|
||||||
|
for document, score in zip(documents[0], scores[0], strict=True):
|
||||||
|
score_value = float(score)
|
||||||
|
if score_value <= 0:
|
||||||
|
continue
|
||||||
|
results.append((dict(document), score_value))
|
||||||
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
async def fetch_bm25_corpus_records(session: AsyncSession) -> tuple[list[dict[str, object]], list[str]]:
|
||||||
|
"""Fetch persistable BM25 corpus records and their matching index texts from the database.
|
||||||
|
|
||||||
|
search_text is only needed to build the index, so it is returned separately instead of
|
||||||
|
being persisted into the corpus records, which would double the corpus size.
|
||||||
|
"""
|
||||||
|
statement = (
|
||||||
|
select(
|
||||||
|
EbookChunk.id.label("chunk_id"),
|
||||||
|
EbookChunk.text.label("text"),
|
||||||
|
EbookSource.id.label("source_id"),
|
||||||
|
EbookSource.title.label("source_title"),
|
||||||
|
EbookSource.author.label("source_author"),
|
||||||
|
EbookChapter.title.label("chapter_title"),
|
||||||
|
EbookChunk.page_label.label("page_label"),
|
||||||
|
EbookChunk.search_text.label("bm25_text"),
|
||||||
|
)
|
||||||
|
.select_from(EbookChunk)
|
||||||
|
.join(EbookSource, EbookSource.id == EbookChunk.source_id)
|
||||||
|
.outerjoin(EbookChapter, EbookChapter.id == EbookChunk.chapter_id)
|
||||||
|
.order_by(EbookChunk.id)
|
||||||
|
)
|
||||||
|
records: list[dict[str, object]] = []
|
||||||
|
texts: list[str] = []
|
||||||
|
for row in (await session.execute(statement)).mappings():
|
||||||
|
record = dict(row)
|
||||||
|
texts.append(str(record.pop("bm25_text")))
|
||||||
|
records.append(record)
|
||||||
|
return records, texts
|
||||||
|
|
||||||
|
|
||||||
|
async def corpus_last_updated_at(session: AsyncSession) -> datetime | None:
|
||||||
|
"""Return the latest source/chapter/chunk update timestamp relevant to BM25 text."""
|
||||||
|
update_times = union_all(
|
||||||
|
select(func.max(EbookSource.updated).label("updated")),
|
||||||
|
select(func.max(EbookChapter.updated).label("updated")),
|
||||||
|
select(func.max(EbookChunk.updated).label("updated")),
|
||||||
|
).subquery()
|
||||||
|
return await session.scalar(select(func.max(update_times.c.updated)))
|
||||||
|
|
||||||
|
|
||||||
|
def write_bm25_corpus(
|
||||||
|
index_path: Path,
|
||||||
|
records: list[dict[str, object]],
|
||||||
|
texts: list[str],
|
||||||
|
manifest: BM25Manifest,
|
||||||
|
) -> None:
|
||||||
|
"""Write a BM25 corpus generation and publish it through the current symlink."""
|
||||||
|
index_path.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
generations_path = index_path / "generations"
|
||||||
|
generations_path.mkdir(exist_ok=True)
|
||||||
|
|
||||||
|
generation_path = next_bm25_generation_path(generations_path, manifest.created_at)
|
||||||
|
current_path = index_path / "current"
|
||||||
|
next_current_path = index_path / f".current.{generation_path.name}.tmp"
|
||||||
|
try:
|
||||||
|
generation_path.mkdir()
|
||||||
|
|
||||||
|
# Empty corpora publish a manifest-only generation so startup succeeds before any chunks exist.
|
||||||
|
if records:
|
||||||
|
retriever = bm25s.BM25()
|
||||||
|
retriever.index(bm25s.tokenize(texts, show_progress=False), show_progress=False)
|
||||||
|
retriever.save(generation_path, corpus=records, show_progress=False)
|
||||||
|
write_bm25_manifest(generation_path, manifest)
|
||||||
|
next_current_path.unlink(missing_ok=True)
|
||||||
|
next_current_path.symlink_to(generation_path, target_is_directory=True)
|
||||||
|
next_current_path.replace(current_path)
|
||||||
|
except Exception:
|
||||||
|
next_current_path.unlink(missing_ok=True)
|
||||||
|
shutil.rmtree(generation_path, ignore_errors=True)
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
def read_bm25_manifest(index_path: Path) -> BM25Manifest | None:
|
||||||
|
"""Read the BM25 manifest if it exists and is valid."""
|
||||||
|
manifest_path = get_current_bm25_index(index_path) / MANIFEST_NAME
|
||||||
|
if not manifest_path.exists():
|
||||||
|
return None
|
||||||
|
body = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||||
|
return BM25Manifest(
|
||||||
|
created_at=datetime.fromisoformat(str(body["created_at"])),
|
||||||
|
db_updated_at=datetime.fromisoformat(str(body["db_updated_at"])) if body.get("db_updated_at") else None,
|
||||||
|
chunk_count=int(body["chunk_count"]),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_bm25_manifest(index_path: Path, manifest: BM25Manifest) -> None:
|
||||||
|
"""Write the BM25 manifest to an index directory."""
|
||||||
|
body = {
|
||||||
|
"created_at": manifest.created_at.isoformat(),
|
||||||
|
"db_updated_at": manifest.db_updated_at.isoformat() if manifest.db_updated_at else None,
|
||||||
|
"chunk_count": manifest.chunk_count,
|
||||||
|
}
|
||||||
|
(index_path / MANIFEST_NAME).write_text(json.dumps(body, indent=2, sort_keys=True), encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def bm25_index_exists(index_path: Path, manifest: BM25Manifest | None) -> bool:
|
||||||
|
"""Return whether a usable persisted BM25 index exists."""
|
||||||
|
active_index_path = get_current_bm25_index(index_path)
|
||||||
|
if manifest is None or not active_index_path.is_dir():
|
||||||
|
return False
|
||||||
|
if manifest.chunk_count == 0:
|
||||||
|
return True
|
||||||
|
return all((active_index_path / file_name).exists() for file_name in REQUIRED_INDEX_FILES)
|
||||||
|
|
||||||
|
|
||||||
|
def next_bm25_generation_path(generations_path: Path, created_at: datetime) -> Path:
|
||||||
|
"""Return an unused dated BM25 generation path."""
|
||||||
|
base_name = created_at.astimezone(UTC).strftime("%Y%m%dT%H%M%S.%fZ")
|
||||||
|
generation_path = generations_path / base_name
|
||||||
|
suffix = 1
|
||||||
|
while generation_path.exists():
|
||||||
|
generation_path = generations_path / f"{base_name}.{suffix}"
|
||||||
|
suffix += 1
|
||||||
|
return generation_path
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""Configuration for the EPUB search app."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from os import getenv
|
||||||
|
from typing import Annotated, Self
|
||||||
|
|
||||||
|
from pydantic import AliasChoices, Field, field_validator, model_validator
|
||||||
|
from pydantic_settings import BaseSettings, NoDecode, SettingsConfigDict
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_embedding_alias(model: str) -> str:
|
||||||
|
"""Normalize a supported embedding alias to its provider model name."""
|
||||||
|
aliases = {
|
||||||
|
"Qwen3-Embedding-0.6B": "qwen3-embedding-0.6b",
|
||||||
|
"Qwen3-Embedding-4B": "qwen3-embedding-4b",
|
||||||
|
"Qwen3-Embedding-8B": "qwen3-embedding-8b",
|
||||||
|
"Qwen/Qwen3-Embedding-0.6B": "qwen3-embedding-0.6b",
|
||||||
|
"Qwen/Qwen3-Embedding-4B": "qwen3-embedding-4b",
|
||||||
|
"Qwen/Qwen3-Embedding-8B": "qwen3-embedding-8b",
|
||||||
|
"qwen3-embedding:0.6b": "qwen3-embedding-0.6b",
|
||||||
|
"qwen3-embedding:4b": "qwen3-embedding-4b",
|
||||||
|
"qwen3-embedding:8b": "qwen3-embedding-8b",
|
||||||
|
"qwen3-embedding-0.6b": "qwen3-embedding-0.6b",
|
||||||
|
"qwen3-embedding-4b": "qwen3-embedding-4b",
|
||||||
|
"qwen3-embedding-8b": "qwen3-embedding-8b",
|
||||||
|
}
|
||||||
|
standard_model = aliases.get(model)
|
||||||
|
if standard_model is None:
|
||||||
|
error = f"Embedding model {model} is not supported. Supported models are {aliases.keys()}"
|
||||||
|
raise ValueError(error)
|
||||||
|
return standard_model
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_embedding_model(default: str = "qwen3-embedding-0.6b") -> str:
|
||||||
|
"""Normalize the configured embedding alias to its provider model name."""
|
||||||
|
return normalize_embedding_alias(getenv("EBOOK_SEARCH_EMBEDDING_MODEL", default))
|
||||||
|
|
||||||
|
|
||||||
|
class RerankConfig(BaseSettings):
|
||||||
|
"""vLLM reranker settings."""
|
||||||
|
|
||||||
|
model_config = SettingsConfigDict(env_prefix="EBOOK_SEARCH_RERANK_", frozen=True, protected_namespaces=())
|
||||||
|
|
||||||
|
enabled: bool = True
|
||||||
|
base_url: str = "http://192.168.90.25:8001"
|
||||||
|
model: str = "qwen3-reranker-06b"
|
||||||
|
candidates: int = 24
|
||||||
|
timeout_seconds: float = 30.0
|
||||||
|
score_weight: float = 0.7
|
||||||
|
hybrid_weight: float = 0.3
|
||||||
|
|
||||||
|
|
||||||
|
class EbookSearchConfig(BaseSettings):
|
||||||
|
"""Runtime settings for EPUB search."""
|
||||||
|
|
||||||
|
model_config = SettingsConfigDict(
|
||||||
|
env_prefix="EBOOK_SEARCH_",
|
||||||
|
frozen=True,
|
||||||
|
populate_by_name=True,
|
||||||
|
protected_namespaces=(),
|
||||||
|
)
|
||||||
|
|
||||||
|
rerank: RerankConfig = Field(default_factory=RerankConfig)
|
||||||
|
top_k: int = 12
|
||||||
|
library_paths: Annotated[tuple[str, ...], NoDecode] = ()
|
||||||
|
chunk_tokens: int = 700
|
||||||
|
chunk_overlap: int = 100
|
||||||
|
vllm_base_url: str = "https://ollama.com/v1"
|
||||||
|
vllm_api_key: str = Field(
|
||||||
|
default="not-needed",
|
||||||
|
validation_alias=AliasChoices("EBOOK_SEARCH_VLLM_API_KEY", "OLLAMA_API_KEY"),
|
||||||
|
)
|
||||||
|
chat_model: str = "deepseek-v4-flash"
|
||||||
|
answer_enabled: bool = True
|
||||||
|
embedding_base_url: str = "http://192.168.90.25:8000/v1"
|
||||||
|
embedding_api_key: str = "not-needed"
|
||||||
|
embedding_model: str = "qwen3-embedding-0.6b"
|
||||||
|
embedding_batch_size: int = 32
|
||||||
|
embedding_timeout_seconds: float = 60.0
|
||||||
|
chat_timeout_seconds: float = 60.0
|
||||||
|
vector_candidate_multiplier: int = 4
|
||||||
|
bm25_candidate_limit: int = 120
|
||||||
|
rrf_rank_constant: int = 60
|
||||||
|
min_retrieval_confidence: float = 0.0
|
||||||
|
validate_citations_enabled: bool = True
|
||||||
|
bm25_index_dir: str = ".ebook_search_bm25"
|
||||||
|
bm25_refresh_delay_seconds: int = 60
|
||||||
|
protected_phrase_max_candidates_per_book: int = 5000
|
||||||
|
protected_phrase_llm_candidates_per_book: int = 500
|
||||||
|
protected_phrase_extraction_workers: int = 16
|
||||||
|
phrase_judge_book_workers: int = 20
|
||||||
|
phrase_judge_phrase_workers: int = 100
|
||||||
|
protected_phrase_confidence_threshold: float = 0.80
|
||||||
|
phrase_matching_enabled: bool = True
|
||||||
|
phrase_hit_boost: float = 0.25
|
||||||
|
phrase_min_tokens: int = 2
|
||||||
|
phrase_max_tokens: int = 5
|
||||||
|
phrase_max_entity_tokens: int = 8
|
||||||
|
phrase_raw_ngram_min_count: int = 2
|
||||||
|
phrase_raw_count_score_threshold: int = 3
|
||||||
|
phrase_raw_count_high_score_threshold: int = 10
|
||||||
|
phrase_chapter_count_score_threshold: int = 2
|
||||||
|
phrase_chapter_count_high_score_threshold: int = 5
|
||||||
|
phrase_target_protected_per_book: int = 100
|
||||||
|
phrase_default_allow_nested: bool = False
|
||||||
|
phrase_default_suppress_children: bool = True
|
||||||
|
|
||||||
|
@field_validator("library_paths", mode="before")
|
||||||
|
@classmethod
|
||||||
|
def split_library_paths(cls, value: object) -> object:
|
||||||
|
"""Split a colon-separated library path string into a tuple of paths."""
|
||||||
|
if isinstance(value, str):
|
||||||
|
return tuple(path for path in value.split(":") if path)
|
||||||
|
return value
|
||||||
|
|
||||||
|
@field_validator("embedding_model")
|
||||||
|
@classmethod
|
||||||
|
def normalize_embedding(cls, value: str) -> str:
|
||||||
|
"""Normalize the configured embedding alias to its provider model name."""
|
||||||
|
return normalize_embedding_alias(value)
|
||||||
|
|
||||||
|
@model_validator(mode="after")
|
||||||
|
def validate_runtime_consistency(self) -> Self:
|
||||||
|
"""Reject configurations that cannot serve the features they enable."""
|
||||||
|
if not self.embedding_base_url.strip():
|
||||||
|
msg = "embedding_base_url must be set"
|
||||||
|
raise ValueError(msg)
|
||||||
|
if self.answer_enabled and (not self.vllm_base_url.strip() or not self.chat_model.strip()):
|
||||||
|
msg = "answer_enabled requires vllm_base_url and chat_model to be set"
|
||||||
|
raise ValueError(msg)
|
||||||
|
if self.rerank.enabled and not self.rerank.base_url.strip():
|
||||||
|
msg = "rerank.enabled requires rerank.base_url to be set"
|
||||||
|
raise ValueError(msg)
|
||||||
|
return self
|
||||||
|
|
||||||
|
|
||||||
|
def load_rerank_config() -> RerankConfig:
|
||||||
|
"""Load reranker config from environment variables."""
|
||||||
|
return RerankConfig()
|
||||||
|
|
||||||
|
|
||||||
|
def load_config() -> EbookSearchConfig:
|
||||||
|
"""Load EPUB search config from environment variables."""
|
||||||
|
return EbookSearchConfig()
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
FROM python:3.14-slim
|
||||||
|
|
||||||
|
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||||
|
PYTHONUNBUFFERED=1 \
|
||||||
|
PIP_NO_CACHE_DIR=1 \
|
||||||
|
APP_DIR=/home/richie/dotfiles \
|
||||||
|
EBOOK_SEARCH_HOST=0.0.0.0 \
|
||||||
|
EBOOK_SEARCH_PORT=8070 \
|
||||||
|
EBOOK_SEARCH_BM25_INDEX_DIR=/data/bm25
|
||||||
|
|
||||||
|
WORKDIR ${APP_DIR}
|
||||||
|
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends build-essential curl \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
COPY pyproject.toml README.md LICENSE ./
|
||||||
|
COPY python ./python
|
||||||
|
|
||||||
|
RUN python -m pip install --upgrade pip \
|
||||||
|
&& python -m pip install \
|
||||||
|
"alembic" \
|
||||||
|
"beautifulsoup4" \
|
||||||
|
"bm25s" \
|
||||||
|
"ebooklib" \
|
||||||
|
"fastapi" \
|
||||||
|
"httpx" \
|
||||||
|
"jinja2" \
|
||||||
|
"pgvector" \
|
||||||
|
"psycopg[binary]" \
|
||||||
|
"pydantic" \
|
||||||
|
"pydantic-settings" \
|
||||||
|
"python-multipart" \
|
||||||
|
"sqlalchemy[asyncio]" \
|
||||||
|
"tiktoken" \
|
||||||
|
"typer" \
|
||||||
|
"uvicorn[standard]" \
|
||||||
|
"yake" \
|
||||||
|
&& python -m pip install --no-deps --editable "${APP_DIR}"
|
||||||
|
|
||||||
|
RUN useradd --create-home --uid 10001 app \
|
||||||
|
&& mkdir -p /data \
|
||||||
|
&& chown -R app:app /home/richie /data
|
||||||
|
|
||||||
|
USER app
|
||||||
|
|
||||||
|
EXPOSE 8070
|
||||||
|
|
||||||
|
CMD ["sh", "-c", "exec python -m python.ebook_search.api.main --host \"${EBOOK_SEARCH_HOST}\" --port \"${EBOOK_SEARCH_PORT}\" --log-level \"${EBOOK_SEARCH_LOG_LEVEL:-INFO}\""]
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
# Ebook Search Docker
|
||||||
|
|
||||||
|
Run the EPUB search app against the existing Postgres database on `jeeves`:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
ebook-search-containers start --library-path /path/to/epubs --build
|
||||||
|
```
|
||||||
|
|
||||||
|
All ebook-search Docker files live in this directory:
|
||||||
|
|
||||||
|
- `Dockerfile`
|
||||||
|
- `docker-compose.yml`
|
||||||
|
- `containers.py`
|
||||||
|
- `container.py`
|
||||||
|
|
||||||
|
The app listens on `http://localhost:8070`.
|
||||||
|
|
||||||
|
Useful lifecycle commands:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
ebook-search-containers build
|
||||||
|
ebook-search-containers start --library-path /path/to/epubs
|
||||||
|
ebook-search-containers logs
|
||||||
|
ebook-search-containers ps
|
||||||
|
ebook-search-containers stop
|
||||||
|
```
|
||||||
|
|
||||||
|
Direct compose usage from the repo root:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
docker compose -f python/ebook_search/docker/docker-compose.yml ps
|
||||||
|
```
|
||||||
|
|
||||||
|
The compose service also loads the repo root `.env` into the container via `env_file`.
|
||||||
|
|
||||||
|
Mount your EPUB directory by setting `EBOOK_LIBRARY_HOST_PATH` in an env file or on the command line. The container sees it as `/library`, and `EBOOK_SEARCH_LIBRARY_PATHS` is set to `/library` inside the container.
|
||||||
|
|
||||||
|
Database connection settings are controlled by `RICHIE_DB`, `RICHIE_HOST`, `RICHIE_PORT`, `RICHIE_USER`, and `RICHIE_PASSWORD`. The default host is `jeeves`.
|
||||||
|
|
||||||
|
Startup runs the Richie Alembic migrations automatically after creating the `main` schema and `vector` extension.
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Docker packaging and lifecycle tooling for ebook search."""
|
||||||
@@ -0,0 +1,229 @@
|
|||||||
|
"""Docker container lifecycle management for ebook search."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Annotated
|
||||||
|
|
||||||
|
import typer
|
||||||
|
|
||||||
|
from python.common import configure_logger, get_repo_dir
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def get_compose_file() -> Path:
|
||||||
|
"""Return the path to the docker-compose.yml file."""
|
||||||
|
return Path(__file__).resolve().with_name("docker-compose.yml")
|
||||||
|
|
||||||
|
|
||||||
|
def compose_base_args() -> list[str]:
|
||||||
|
"""Return the common docker compose arguments for the ebook search stack."""
|
||||||
|
return ["compose", "-f", str(get_compose_file())]
|
||||||
|
|
||||||
|
|
||||||
|
def docker_run(
|
||||||
|
arguments: list[str],
|
||||||
|
*,
|
||||||
|
env: dict[str, str] | None = None,
|
||||||
|
capture_output: bool = False,
|
||||||
|
) -> subprocess.CompletedProcess[str]:
|
||||||
|
"""Run docker with repo-root cwd and consistent error handling."""
|
||||||
|
logger.info("docker %s", " ".join(arguments))
|
||||||
|
return subprocess.run(
|
||||||
|
["docker", *arguments],
|
||||||
|
cwd=get_repo_dir(),
|
||||||
|
env=env,
|
||||||
|
text=True,
|
||||||
|
check=False,
|
||||||
|
capture_output=capture_output,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def compose_env(*, library_path: Path | None = None, port: int | None = None) -> dict[str, str]:
|
||||||
|
"""Return environment variables passed to docker compose."""
|
||||||
|
env = os.environ.copy()
|
||||||
|
if library_path is not None:
|
||||||
|
resolved_library = library_path.expanduser().resolve()
|
||||||
|
if not resolved_library.exists():
|
||||||
|
msg = f"EPUB library path does not exist: {resolved_library}"
|
||||||
|
raise FileNotFoundError(msg)
|
||||||
|
env["EBOOK_LIBRARY_HOST_PATH"] = str(resolved_library)
|
||||||
|
if port is not None:
|
||||||
|
env["EBOOK_SEARCH_PORT"] = str(port)
|
||||||
|
return env
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_compose_file() -> None:
|
||||||
|
"""Raise if the ebook search compose file is missing."""
|
||||||
|
if not get_compose_file().is_file():
|
||||||
|
msg = f"Compose file not found: {get_compose_file()}"
|
||||||
|
raise FileNotFoundError(msg)
|
||||||
|
|
||||||
|
|
||||||
|
def build_image() -> None:
|
||||||
|
"""Build the ebook search app image."""
|
||||||
|
ensure_compose_file()
|
||||||
|
result = docker_run([*compose_base_args(), "build"])
|
||||||
|
if result.returncode != 0:
|
||||||
|
msg = "Failed to build ebook search image"
|
||||||
|
raise RuntimeError(msg)
|
||||||
|
|
||||||
|
|
||||||
|
def start_stack(
|
||||||
|
*,
|
||||||
|
library_path: Path | None = None,
|
||||||
|
port: int | None = None,
|
||||||
|
build: bool = False,
|
||||||
|
) -> None:
|
||||||
|
"""Start the ebook search Docker compose stack."""
|
||||||
|
ensure_compose_file()
|
||||||
|
env = compose_env(library_path=library_path, port=port)
|
||||||
|
if build:
|
||||||
|
build_image()
|
||||||
|
result = docker_run(
|
||||||
|
[*compose_base_args(), "up", "-d"],
|
||||||
|
env=env,
|
||||||
|
)
|
||||||
|
if result.returncode != 0:
|
||||||
|
msg = f"Ebook search stack failed to start with code {result.returncode}"
|
||||||
|
raise RuntimeError(msg)
|
||||||
|
logger.info("Ebook search started.")
|
||||||
|
|
||||||
|
|
||||||
|
def stop_stack(
|
||||||
|
*,
|
||||||
|
volumes: bool = False,
|
||||||
|
) -> None:
|
||||||
|
"""Stop and remove ebook search containers."""
|
||||||
|
ensure_compose_file()
|
||||||
|
command = [*compose_base_args(), "down"]
|
||||||
|
if volumes:
|
||||||
|
command.append("-v")
|
||||||
|
result = docker_run(command)
|
||||||
|
if result.returncode != 0:
|
||||||
|
msg = f"Ebook search stack failed to stop with code {result.returncode}"
|
||||||
|
raise RuntimeError(msg)
|
||||||
|
|
||||||
|
|
||||||
|
def logs_stack(
|
||||||
|
*,
|
||||||
|
service: str | None = None,
|
||||||
|
tail: int = 100,
|
||||||
|
follow: bool = False,
|
||||||
|
) -> str | None:
|
||||||
|
"""Return recent logs from the ebook search stack."""
|
||||||
|
ensure_compose_file()
|
||||||
|
command = [*compose_base_args(), "logs", "--tail", str(tail)]
|
||||||
|
if follow:
|
||||||
|
command.append("--follow")
|
||||||
|
if service:
|
||||||
|
command.append(service)
|
||||||
|
result = docker_run(command, capture_output=not follow)
|
||||||
|
if result.returncode != 0:
|
||||||
|
return None
|
||||||
|
if follow:
|
||||||
|
return ""
|
||||||
|
return result.stdout + result.stderr
|
||||||
|
|
||||||
|
|
||||||
|
def ps_stack() -> str | None:
|
||||||
|
"""Return docker compose ps output for the ebook search stack."""
|
||||||
|
ensure_compose_file()
|
||||||
|
result = docker_run([*compose_base_args(), "ps"], capture_output=True)
|
||||||
|
if result.returncode != 0:
|
||||||
|
return None
|
||||||
|
return result.stdout + result.stderr
|
||||||
|
|
||||||
|
|
||||||
|
app = typer.Typer(help="Ebook search Docker container management.", no_args_is_help=True)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def build() -> None:
|
||||||
|
"""Build the ebook search Docker image."""
|
||||||
|
build_image()
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def start(
|
||||||
|
library_path: Annotated[Path | None, typer.Option(help="Override host path containing EPUB files.")] = None,
|
||||||
|
port: Annotated[int | None, typer.Option(help="Override host port for the web UI.")] = None,
|
||||||
|
*,
|
||||||
|
build: Annotated[bool, typer.Option("--build", help="Build the image before starting.")] = False,
|
||||||
|
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||||
|
) -> None:
|
||||||
|
"""Start the ebook search container."""
|
||||||
|
configure_logger(log_level)
|
||||||
|
start_stack(
|
||||||
|
library_path=library_path,
|
||||||
|
port=port,
|
||||||
|
build=build,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def stop(
|
||||||
|
*,
|
||||||
|
volumes: Annotated[bool, typer.Option("--volumes", help="Also remove ebook search data volumes.")] = False,
|
||||||
|
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||||
|
) -> None:
|
||||||
|
"""Stop and remove ebook search containers."""
|
||||||
|
configure_logger(log_level)
|
||||||
|
stop_stack(volumes=volumes)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def restart(
|
||||||
|
library_path: Annotated[Path | None, typer.Option(help="Override host path containing EPUB files.")] = None,
|
||||||
|
port: Annotated[int | None, typer.Option(help="Override host port for the web UI.")] = None,
|
||||||
|
*,
|
||||||
|
build: Annotated[bool, typer.Option("--build", help="Build the image before starting.")] = False,
|
||||||
|
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||||
|
) -> None:
|
||||||
|
"""Restart the ebook search stack."""
|
||||||
|
configure_logger(log_level)
|
||||||
|
stop_stack()
|
||||||
|
start_stack(
|
||||||
|
library_path=library_path,
|
||||||
|
port=port,
|
||||||
|
build=build,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command()
|
||||||
|
def logs(
|
||||||
|
service: Annotated[str | None, typer.Option(help="Service name, or omit for all services.")] = None,
|
||||||
|
tail: Annotated[int, typer.Option(help="Number of recent log lines.")] = 100,
|
||||||
|
*,
|
||||||
|
follow: Annotated[bool, typer.Option("--follow", "-f", help="Follow logs.")] = False,
|
||||||
|
) -> None:
|
||||||
|
"""Show recent ebook search container logs."""
|
||||||
|
output = logs_stack(service=service, tail=tail, follow=follow)
|
||||||
|
if output is None:
|
||||||
|
typer.echo("No ebook search containers found.")
|
||||||
|
raise typer.Exit(code=1)
|
||||||
|
if output:
|
||||||
|
typer.echo(output)
|
||||||
|
|
||||||
|
|
||||||
|
@app.command("ps")
|
||||||
|
def ps() -> None:
|
||||||
|
"""Show ebook search container status."""
|
||||||
|
output = ps_stack()
|
||||||
|
if output is None:
|
||||||
|
typer.echo("No ebook search containers found.")
|
||||||
|
raise typer.Exit(code=1)
|
||||||
|
typer.echo(output)
|
||||||
|
|
||||||
|
|
||||||
|
def cli() -> None:
|
||||||
|
"""Typer entry point."""
|
||||||
|
app()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
cli()
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
name: ebook-search
|
||||||
|
|
||||||
|
services:
|
||||||
|
ebook-search:
|
||||||
|
build:
|
||||||
|
context: ../../..
|
||||||
|
dockerfile: python/ebook_search/docker/Dockerfile
|
||||||
|
image: ebook-search:latest
|
||||||
|
restart: unless-stopped
|
||||||
|
ports:
|
||||||
|
- "${EBOOK_SEARCH_PORT:-8070}:8070"
|
||||||
|
extra_hosts:
|
||||||
|
- "jeeves:192.168.90.40"
|
||||||
|
env_file:
|
||||||
|
- ../../../.env
|
||||||
|
environment:
|
||||||
|
EBOOK_SEARCH_HOST: "0.0.0.0"
|
||||||
|
EBOOK_SEARCH_PORT: "8070"
|
||||||
|
EBOOK_SEARCH_LIBRARY_PATHS: "/library"
|
||||||
|
EBOOK_SEARCH_BM25_INDEX_DIR: "/data/bm25"
|
||||||
|
volumes:
|
||||||
|
- "${EBOOK_LIBRARY_HOST_PATH:-/home/richie/ebooks}:/library:ro"
|
||||||
|
- ebook-search-data:/data
|
||||||
|
healthcheck:
|
||||||
|
test:
|
||||||
|
[
|
||||||
|
"CMD-SHELL",
|
||||||
|
"curl -fsS http://127.0.0.1:8070/health >/dev/null || exit 1",
|
||||||
|
]
|
||||||
|
interval: 30s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 5
|
||||||
|
start_period: 30s
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
ebook-search-data:
|
||||||
@@ -0,0 +1,175 @@
|
|||||||
|
"""Embedding model helpers."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.dialects.postgresql import insert
|
||||||
|
|
||||||
|
from python.ebook_search.llm_interface import request_embeddings
|
||||||
|
from python.orm.richie import (
|
||||||
|
EbookChunk,
|
||||||
|
EbookChunkEmbedding1024,
|
||||||
|
EbookChunkEmbedding2560,
|
||||||
|
EbookChunkEmbedding4096,
|
||||||
|
EbookEmbeddingModel,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
MODEL_DIMENSIONS = {
|
||||||
|
"qwen3-embedding-0.6b": 1024,
|
||||||
|
"qwen3-embedding-4b": 2560,
|
||||||
|
"qwen3-embedding-8b": 4096,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def get_embedding_table(
|
||||||
|
dimension: int,
|
||||||
|
) -> type[EbookChunkEmbedding1024 | EbookChunkEmbedding2560 | EbookChunkEmbedding4096]:
|
||||||
|
"""Return the embedding table mapped to an embedding dimension."""
|
||||||
|
embedding_tables = {
|
||||||
|
1024: EbookChunkEmbedding1024,
|
||||||
|
2560: EbookChunkEmbedding2560,
|
||||||
|
4096: EbookChunkEmbedding4096,
|
||||||
|
}
|
||||||
|
table = embedding_tables.get(dimension)
|
||||||
|
if not table:
|
||||||
|
msg = f"Embedding dimension {dimension} is not supported"
|
||||||
|
raise ValueError(msg)
|
||||||
|
return table
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class EmbeddingModelStats:
|
||||||
|
"""Embedding coverage for one model."""
|
||||||
|
|
||||||
|
model_name: str
|
||||||
|
dimension: int
|
||||||
|
embedded_chunks: int
|
||||||
|
total_chunks: int
|
||||||
|
|
||||||
|
@property
|
||||||
|
def missing_chunks(self) -> int:
|
||||||
|
"""Return chunks missing this embedding model."""
|
||||||
|
return max(self.total_chunks - self.embedded_chunks, 0)
|
||||||
|
|
||||||
|
|
||||||
|
async def embed_texts(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
texts: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> list[list[float]]:
|
||||||
|
"""Embed text with the configured vLLM embedding model."""
|
||||||
|
logger.info(
|
||||||
|
"ebook_embed_request_start base_url=%s model=%s count=%s",
|
||||||
|
config.embedding_base_url,
|
||||||
|
config.embedding_model,
|
||||||
|
len(texts),
|
||||||
|
)
|
||||||
|
vectors = await request_embeddings(client, texts, config)
|
||||||
|
expected_dimension = MODEL_DIMENSIONS[config.embedding_model]
|
||||||
|
for vector in vectors:
|
||||||
|
if len(vector) != expected_dimension:
|
||||||
|
msg = f"Expected {expected_dimension} dimensions, got {len(vector)}"
|
||||||
|
raise ValueError(msg)
|
||||||
|
logger.info(
|
||||||
|
"ebook_embed_request_complete model=%s count=%s dimension=%s",
|
||||||
|
config.embedding_model,
|
||||||
|
len(vectors),
|
||||||
|
expected_dimension,
|
||||||
|
)
|
||||||
|
return vectors
|
||||||
|
|
||||||
|
|
||||||
|
async def embed_query(client: httpx.AsyncClient, query: str, config: EbookSearchConfig) -> list[float]:
|
||||||
|
"""Embed a search query with the Qwen retrieval instruction."""
|
||||||
|
instructed_query = f"Instruct: Retrieve relevant passages for the query.\nQuery: {query}"
|
||||||
|
return (await embed_texts(client, [instructed_query], config))[0]
|
||||||
|
|
||||||
|
|
||||||
|
async def ensure_embedding_models(session: AsyncSession) -> None:
|
||||||
|
"""Ensure supported embedding model rows exist."""
|
||||||
|
for name, dimension in MODEL_DIMENSIONS.items():
|
||||||
|
existing = await session.scalar(select(EbookEmbeddingModel).where(EbookEmbeddingModel.name == name))
|
||||||
|
if existing is None:
|
||||||
|
session.add(EbookEmbeddingModel(name=name, dimension=dimension, is_default=name == "qwen3-embedding-0.6b"))
|
||||||
|
logger.info("ebook_embedding_model_created model=%s dimension=%s", name, dimension)
|
||||||
|
await session.flush()
|
||||||
|
|
||||||
|
|
||||||
|
async def embedding_model_stats(session: AsyncSession) -> list[EmbeddingModelStats]:
|
||||||
|
"""Return embedding coverage counts for every supported model."""
|
||||||
|
total_chunks = await session.scalar(select(func.count(EbookChunk.id))) or 0
|
||||||
|
models = {
|
||||||
|
model.name: model
|
||||||
|
for model in await session.scalars(
|
||||||
|
select(EbookEmbeddingModel)
|
||||||
|
.where(EbookEmbeddingModel.name.in_(MODEL_DIMENSIONS))
|
||||||
|
.order_by(EbookEmbeddingModel.name)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
stats: list[EmbeddingModelStats] = []
|
||||||
|
for model_name, dimension in MODEL_DIMENSIONS.items():
|
||||||
|
model = models.get(model_name)
|
||||||
|
embedded_chunks = 0
|
||||||
|
if model is not None:
|
||||||
|
table = get_embedding_table(dimension)
|
||||||
|
embedded_chunks = await session.scalar(select(func.count(table.id)).where(table.model_id == model.id)) or 0
|
||||||
|
stats.append(
|
||||||
|
EmbeddingModelStats(
|
||||||
|
model_name=model_name,
|
||||||
|
dimension=dimension,
|
||||||
|
embedded_chunks=embedded_chunks,
|
||||||
|
total_chunks=total_chunks,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return stats
|
||||||
|
|
||||||
|
|
||||||
|
async def embed_missing_chunks(session: AsyncSession, client: httpx.AsyncClient, config: EbookSearchConfig) -> int:
|
||||||
|
"""Embed chunks missing embeddings for the configured model."""
|
||||||
|
await ensure_embedding_models(session)
|
||||||
|
model = await session.scalar(select(EbookEmbeddingModel).where(EbookEmbeddingModel.name == config.embedding_model))
|
||||||
|
if model is None:
|
||||||
|
supported_models = ", ".join(MODEL_DIMENSIONS)
|
||||||
|
msg = f"Unknown embedding model: {config.embedding_model}. Supported models: {supported_models}"
|
||||||
|
raise ValueError(msg)
|
||||||
|
|
||||||
|
table = get_embedding_table(model.dimension)
|
||||||
|
chunks = list(
|
||||||
|
await session.scalars(
|
||||||
|
select(EbookChunk)
|
||||||
|
.outerjoin(table, (table.chunk_id == EbookChunk.id) & (table.model_id == model.id))
|
||||||
|
.where(table.id.is_(None))
|
||||||
|
.order_by(EbookChunk.id)
|
||||||
|
.limit(config.embedding_batch_size)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if not chunks:
|
||||||
|
logger.info("ebook_embed_missing_none model=%s", config.embedding_model)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
logger.info("ebook_embed_missing_batch_start model=%s count=%s", config.embedding_model, len(chunks))
|
||||||
|
vectors = await embed_texts(client, [chunk.text for chunk in chunks], config)
|
||||||
|
rows = [
|
||||||
|
{"chunk_id": chunk.id, "model_id": model.id, "embedding": vector}
|
||||||
|
for chunk, vector in zip(chunks, vectors, strict=True)
|
||||||
|
]
|
||||||
|
statement = insert(table).values(rows).on_conflict_do_nothing(index_elements=["chunk_id", "model_id"])
|
||||||
|
await session.execute(statement)
|
||||||
|
await session.flush()
|
||||||
|
logger.info("ebook_embed_missing_batch_complete model=%s count=%s", config.embedding_model, len(rows))
|
||||||
|
return len(rows)
|
||||||
@@ -0,0 +1,95 @@
|
|||||||
|
"""EPUB parsing helpers."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
from ebooklib import ITEM_DOCUMENT, epub
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
WHITESPACE_RE = re.compile(r"\s+")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ParsedChapter:
|
||||||
|
"""Text extracted from one EPUB spine document."""
|
||||||
|
|
||||||
|
title: str | None
|
||||||
|
href: str | None
|
||||||
|
text: str
|
||||||
|
page_labels: tuple[str, ...]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ParsedEpub:
|
||||||
|
"""Parsed EPUB metadata and text."""
|
||||||
|
|
||||||
|
title: str
|
||||||
|
author: str | None
|
||||||
|
language: str | None
|
||||||
|
publisher: str | None
|
||||||
|
identifier: str | None
|
||||||
|
chapters: tuple[ParsedChapter, ...]
|
||||||
|
|
||||||
|
|
||||||
|
def parse_epub(path: Path) -> ParsedEpub:
|
||||||
|
"""Parse EPUB metadata and spine text."""
|
||||||
|
book = epub.read_epub(path)
|
||||||
|
chapters = []
|
||||||
|
for item in book.get_items_of_type(ITEM_DOCUMENT):
|
||||||
|
soup = BeautifulSoup(item.get_content(), "html.parser")
|
||||||
|
title = chapter_title(soup)
|
||||||
|
page_labels = tuple(extract_page_labels(soup))
|
||||||
|
text = clean_text(soup.get_text(" "))
|
||||||
|
if text:
|
||||||
|
chapters.append(ParsedChapter(title=title, href=item.get_name(), text=text, page_labels=page_labels))
|
||||||
|
|
||||||
|
return ParsedEpub(
|
||||||
|
title=metadata_value(book, "title") or path.stem,
|
||||||
|
author=metadata_value(book, "creator"),
|
||||||
|
language=metadata_value(book, "language"),
|
||||||
|
publisher=metadata_value(book, "publisher"),
|
||||||
|
identifier=metadata_value(book, "identifier"),
|
||||||
|
chapters=tuple(chapters),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def metadata_value(book: epub.EpubBook, name: str) -> str | None:
|
||||||
|
"""Return the first non-empty Dublin Core metadata value for a name."""
|
||||||
|
values = book.get_metadata("DC", name)
|
||||||
|
if not values:
|
||||||
|
return None
|
||||||
|
value = values[0][0]
|
||||||
|
return str(value).strip() or None
|
||||||
|
|
||||||
|
|
||||||
|
def chapter_title(soup: BeautifulSoup) -> str | None:
|
||||||
|
"""Extract the best available title from an EPUB document soup."""
|
||||||
|
heading = soup.find(["h1", "h2", "h3"])
|
||||||
|
if heading is None:
|
||||||
|
title = soup.find("title")
|
||||||
|
if title is None:
|
||||||
|
return None
|
||||||
|
return clean_text(title.get_text(" ")) or None
|
||||||
|
return clean_text(heading.get_text(" ")) or None
|
||||||
|
|
||||||
|
|
||||||
|
def extract_page_labels(soup: BeautifulSoup) -> list[str]:
|
||||||
|
"""Extract EPUB page-break labels from a document soup."""
|
||||||
|
labels: list[str] = []
|
||||||
|
for tag in soup.find_all(attrs={"epub:type": "pagebreak"}):
|
||||||
|
label = tag.get("title") or tag.get("aria-label") or tag.get_text(" ")
|
||||||
|
clean = clean_text(str(label))
|
||||||
|
if clean:
|
||||||
|
labels.append(clean)
|
||||||
|
return labels
|
||||||
|
|
||||||
|
|
||||||
|
def clean_text(text: str) -> str:
|
||||||
|
"""Normalize whitespace in extracted EPUB text."""
|
||||||
|
return WHITESPACE_RE.sub(" ", text).strip()
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Offline evaluation tooling for the ebook search pipeline."""
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
{"query": "Who is Damien Montgomery and how does he become a Jump Mage?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What is a Rune Wright and why is Damien so rare?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "How does jump magic let starships travel faster than light?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What is the role of the Mage-King of Mars in the Protectorate?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What happened aboard the Blue Jay in the first Starship's Mage book?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "Who is Captain David Rice?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "How are amplifiers and simulacrums used to power a ship's jump?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What duties does a Hand of the Mage-King carry out?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "Explain the structure of the Royal Martian Navy.", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "How do mages carve runes to enchant a starship?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What threat do the Legatan rebels pose to the Protectorate?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "How does Damien handle his first command?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What is the significance of the simulacrum on a jump ship?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "Describe a mage duel in the Starship's Mage series.", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What moral conflicts does Damien face as a Hand of the Mage-King?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "How does the Protectorate keep peace among its member worlds?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "Who is the Keeper of Oaths and how does Damien work with them?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||||
|
{"query": "What event is known as the Onset and how does it change the world?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "Who is the main character at the start of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How do survivors adapt after the Onset begins?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What new abilities emerge during the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "Describe the primary antagonist in the Onset series.", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How does society collapse and reorganize after the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What factions form in the aftermath of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How does the protagonist gain power throughout the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What is the cause or origin of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "Describe an early survival challenge faced after the Onset.", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How do the characters defend their stronghold during the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What relationships drive the protagonist's choices in the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How does the Onset escalate by the end of the first book?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What mysteries about the Onset remain unresolved?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How do the rules of the world change once the Onset takes hold?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "What weapons or tactics work best against the threats of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||||
|
{"query": "How does Bob Johansson become a von Neumann probe?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What is a replicant and why do Bob's copies have different personalities?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "Who are Riker, Homer, and Bill among the Bob clones?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What is GUPPI and how does Bob use it?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "Describe the threat posed by the Others.", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How does Bob protect and uplift the Deltans?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "Why do the replicants drift apart in personality over time?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What is the role of FAITH and the Brazilian Empire on Earth?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How does subspace communication work for the Bobs?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What happens to Bender after he goes missing?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How do the Bobs build self-replicating probes across the galaxy?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How does Bob evacuate humanity after Earth becomes uninhabitable?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "Describe the conflict between different factions of Bobs.", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What ethical dilemmas does Bob face when interfering with primitive species?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How does the original Bob differ from later generations of clones?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "How do the Bobs defeat the Others' system-harvesting fleets?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
{"query": "What role does Howard play in the human colonies?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||||
|
// querys not it the dataset
|
||||||
|
{"query": "How does Frodo destroy the One Ring in The Lord of the Rings?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "Who killed Dumbledore in Harry Potter and the Half-Blood Prince?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What house does Tyrion Lannister belong to in A Game of Thrones?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How does Paul Atreides control the spice on Arrakis in Dune?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What does the green light at the end of the dock mean in The Great Gatsby?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "Why does Hester Prynne wear a scarlet letter?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What does the white whale represent in Moby-Dick?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How does Elizabeth Bennet's view of Mr. Darcy change in Pride and Prejudice?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What crime does Raskolnikov commit in Crime and Punishment?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How does Katniss volunteer for the Hunger Games?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What is Winston Smith's job in Nineteen Eighty-Four?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "Who is Atticus Finch defending in To Kill a Mockingbird?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What is the capital of Australia?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How do I bake a sourdough loaf from scratch?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "Explain how photosynthesis converts sunlight into energy.", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What were the main causes of World War I?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How does compound interest work?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "How do I change a flat tire on a car?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What is the boiling point of water at sea level?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
|
{"query": "What is the recommended daily intake of vitamin D?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
"""Shared query set loading for evaluation and load testing.
|
||||||
|
|
||||||
|
Each JSONL record has a ``query`` and an optional reference ``answer``. ``answerable``
|
||||||
|
marks whether the query should be answerable from the library (false for out-of-corpus
|
||||||
|
"garbage" queries used to test the refusal path). Relevance for retrieval metrics is
|
||||||
|
labeled at source (book) granularity in ``relevant_sources``; source titles must match
|
||||||
|
``ebook_source.title`` values for the indexed corpus.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
DEFAULT_QUERIES_PATH = Path(__file__).parent / "data" / "queries.jsonl"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class GoldQuery:
|
||||||
|
"""One labeled query shared by the eval and load-test tools."""
|
||||||
|
|
||||||
|
query: str
|
||||||
|
answer: str | None
|
||||||
|
answerable: bool
|
||||||
|
relevant_sources: tuple[str, ...]
|
||||||
|
relevant_substrings: tuple[str, ...]
|
||||||
|
|
||||||
|
|
||||||
|
def load_gold_queries(path: Path = DEFAULT_QUERIES_PATH) -> list[GoldQuery]:
|
||||||
|
"""Load labeled queries from a JSONL file. Blank lines and ``//`` comment lines are skipped."""
|
||||||
|
queries: list[GoldQuery] = []
|
||||||
|
for line in path.read_text(encoding="utf-8").splitlines():
|
||||||
|
stripped = line.strip()
|
||||||
|
if not stripped or stripped.startswith("//"):
|
||||||
|
continue
|
||||||
|
record = json.loads(stripped)
|
||||||
|
queries.append(
|
||||||
|
GoldQuery(
|
||||||
|
query=str(record["query"]),
|
||||||
|
answer=record.get("answer"),
|
||||||
|
answerable=bool(record.get("answerable", True)),
|
||||||
|
relevant_sources=tuple(record.get("relevant_sources", ())),
|
||||||
|
relevant_substrings=tuple(record.get("relevant_substrings", ())),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return queries
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
"""Serve-time output guardrails for retrieval confidence and answer citations."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.search import SearchResult
|
||||||
|
|
||||||
|
CITATION_RE = re.compile(r"\[(\d+)\]")
|
||||||
|
|
||||||
|
|
||||||
|
def retrieval_confidence(results: list[SearchResult]) -> float:
|
||||||
|
"""Return the strongest interpretable relevance signal of the top result.
|
||||||
|
|
||||||
|
Reciprocal-rank-fusion scores are rank-based and not comparable across queries,
|
||||||
|
so the rerank relevance score is preferred, then vector cosine similarity, then
|
||||||
|
the final score.
|
||||||
|
"""
|
||||||
|
if not results:
|
||||||
|
return 0.0
|
||||||
|
top = results[0]
|
||||||
|
if top.rerank_score is not None:
|
||||||
|
return top.rerank_score
|
||||||
|
if top.vector_score is not None:
|
||||||
|
return top.vector_score
|
||||||
|
return top.score
|
||||||
|
|
||||||
|
|
||||||
|
def is_confident(results: list[SearchResult], config: EbookSearchConfig) -> bool:
|
||||||
|
"""Return whether top-result confidence meets the configured threshold."""
|
||||||
|
return retrieval_confidence(results) >= config.min_retrieval_confidence
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CitationReport:
|
||||||
|
"""Validation summary for bracketed citation markers in a generated answer."""
|
||||||
|
|
||||||
|
cited: tuple[int, ...]
|
||||||
|
invalid: tuple[int, ...]
|
||||||
|
grounded: bool
|
||||||
|
|
||||||
|
|
||||||
|
def validate_citations(answer: str, result_count: int) -> CitationReport:
|
||||||
|
"""Validate bracketed citation markers against the number of shown sources.
|
||||||
|
|
||||||
|
A marker is valid when it points to a returned source (``1..result_count``).
|
||||||
|
``grounded`` is true when the answer cites at least one valid source.
|
||||||
|
"""
|
||||||
|
markers = sorted({int(match.group(1)) for match in CITATION_RE.finditer(answer)})
|
||||||
|
valid = range(1, result_count + 1)
|
||||||
|
cited = tuple(marker for marker in markers if marker in valid)
|
||||||
|
invalid = tuple(marker for marker in markers if marker not in valid)
|
||||||
|
return CitationReport(cited=cited, invalid=invalid, grounded=bool(cited))
|
||||||
@@ -0,0 +1,222 @@
|
|||||||
|
"""EPUB ingestion into Richie DB."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import hashlib
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import tiktoken
|
||||||
|
from sqlalchemy import or_, select
|
||||||
|
|
||||||
|
from python.ebook_search.epub_parse import parse_epub
|
||||||
|
from python.ebook_search.protected_phrases.matching import index_chunk_phrase_mentions_for_book
|
||||||
|
from python.orm.richie import EbookChapter, EbookChunk, EbookSource
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
DEFAULT_CHUNK_TOKENS = 700
|
||||||
|
DEFAULT_CHUNK_OVERLAP = 100
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.epub_parse import ParsedChapter
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TextChunk:
|
||||||
|
"""A token-bounded chunk of text."""
|
||||||
|
|
||||||
|
text: str
|
||||||
|
token_start: int
|
||||||
|
token_count: int
|
||||||
|
|
||||||
|
|
||||||
|
def chunk_text(
|
||||||
|
text: str,
|
||||||
|
*,
|
||||||
|
chunk_tokens: int = DEFAULT_CHUNK_TOKENS,
|
||||||
|
overlap_tokens: int = DEFAULT_CHUNK_OVERLAP,
|
||||||
|
) -> list[TextChunk]:
|
||||||
|
"""Split text into overlapping token chunks."""
|
||||||
|
if chunk_tokens <= 0:
|
||||||
|
msg = "chunk_tokens must be positive"
|
||||||
|
raise ValueError(msg)
|
||||||
|
if overlap_tokens < 0 or overlap_tokens >= chunk_tokens:
|
||||||
|
msg = "overlap_tokens must be non-negative and smaller than chunk_tokens"
|
||||||
|
raise ValueError(msg)
|
||||||
|
|
||||||
|
encoding = tiktoken.get_encoding("cl100k_base")
|
||||||
|
tokens = encoding.encode(text)
|
||||||
|
if not tokens:
|
||||||
|
return []
|
||||||
|
|
||||||
|
chunks: list[TextChunk] = []
|
||||||
|
step = chunk_tokens - overlap_tokens
|
||||||
|
for start in range(0, len(tokens), step):
|
||||||
|
chunk = tokens[start : start + chunk_tokens]
|
||||||
|
if not chunk:
|
||||||
|
continue
|
||||||
|
chunks.append(
|
||||||
|
TextChunk(
|
||||||
|
text=encoding.decode(chunk).strip(),
|
||||||
|
token_start=start,
|
||||||
|
token_count=len(chunk),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if start + chunk_tokens >= len(tokens):
|
||||||
|
break
|
||||||
|
return [chunk for chunk in chunks if chunk.text]
|
||||||
|
|
||||||
|
|
||||||
|
def find_library_epubs(library_path: str) -> tuple[Path, list[Path] | None]:
|
||||||
|
"""Resolve one configured library path and collect its EPUB files (blocking filesystem walk).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[Path, list[Path] | None]: The expanded path and its EPUB files, or ``None`` when
|
||||||
|
the path is neither an EPUB file nor a directory.
|
||||||
|
"""
|
||||||
|
path = Path(library_path).expanduser()
|
||||||
|
if path.is_file() and path.suffix.lower() == ".epub":
|
||||||
|
return path, [path]
|
||||||
|
if path.is_dir():
|
||||||
|
return path, sorted(path.rglob("*.epub"))
|
||||||
|
return path, None
|
||||||
|
|
||||||
|
|
||||||
|
async def ingest_configured_paths(session: AsyncSession, config: EbookSearchConfig) -> int:
|
||||||
|
"""Ingest every EPUB found under configured library paths."""
|
||||||
|
count = 0
|
||||||
|
for library_path in config.library_paths:
|
||||||
|
path, epub_paths = await asyncio.to_thread(find_library_epubs, library_path)
|
||||||
|
logger.info("ebook_ingest_path_start path=%s", path)
|
||||||
|
if epub_paths is None:
|
||||||
|
logger.warning("ebook_ingest_path_missing path=%s", path)
|
||||||
|
continue
|
||||||
|
for epub_path in epub_paths:
|
||||||
|
count += int(await ingest_file(session, epub_path, config))
|
||||||
|
logger.info("ebook_ingest_paths_complete changed_files=%s configured_paths=%s", count, len(config.library_paths))
|
||||||
|
return count
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_ingest_path(path: Path) -> Path:
|
||||||
|
"""Expand and resolve an ingest path (blocking filesystem call)."""
|
||||||
|
return path.expanduser().resolve()
|
||||||
|
|
||||||
|
|
||||||
|
async def ingest_file(session: AsyncSession, path: Path, config: EbookSearchConfig) -> bool:
|
||||||
|
"""Ingest one EPUB file. Return True when the database changed."""
|
||||||
|
try:
|
||||||
|
resolved_path = await asyncio.to_thread(resolve_ingest_path, path)
|
||||||
|
logger.info("ebook_ingest_file_start path=%s", resolved_path)
|
||||||
|
file_hash = await asyncio.to_thread(sha256_file, resolved_path)
|
||||||
|
existing = await find_existing_source(session, resolved_path, file_hash)
|
||||||
|
if existing is not None and existing.file_sha256 == file_hash:
|
||||||
|
stat = resolved_path.stat()
|
||||||
|
existing.file_path = str(resolved_path)
|
||||||
|
existing.file_mtime = datetime.fromtimestamp(stat.st_mtime, tz=UTC)
|
||||||
|
existing.file_size = stat.st_size
|
||||||
|
await session.flush()
|
||||||
|
logger.info("ebook_ingest_file_unchanged source_id=%s path=%s", existing.id, resolved_path)
|
||||||
|
return False
|
||||||
|
if existing is not None:
|
||||||
|
logger.info("ebook_ingest_file_replacing source_id=%s path=%s", existing.id, resolved_path)
|
||||||
|
await session.delete(existing)
|
||||||
|
await session.flush()
|
||||||
|
|
||||||
|
stat = resolved_path.stat()
|
||||||
|
parsed = await asyncio.to_thread(parse_epub, resolved_path)
|
||||||
|
source = EbookSource(
|
||||||
|
title=parsed.title,
|
||||||
|
author=parsed.author,
|
||||||
|
language=parsed.language,
|
||||||
|
publisher=parsed.publisher,
|
||||||
|
identifier=parsed.identifier,
|
||||||
|
file_path=str(resolved_path),
|
||||||
|
file_sha256=file_hash,
|
||||||
|
file_mtime=datetime.fromtimestamp(stat.st_mtime, tz=UTC),
|
||||||
|
file_size=stat.st_size,
|
||||||
|
)
|
||||||
|
session.add(source)
|
||||||
|
await session.flush()
|
||||||
|
|
||||||
|
chunk_index = 0
|
||||||
|
for spine_index, parsed_chapter in enumerate(parsed.chapters):
|
||||||
|
chapter = EbookChapter(
|
||||||
|
source_id=source.id,
|
||||||
|
spine_index=spine_index,
|
||||||
|
title=parsed_chapter.title,
|
||||||
|
href=parsed_chapter.href,
|
||||||
|
)
|
||||||
|
session.add(chapter)
|
||||||
|
await session.flush()
|
||||||
|
chunk_index = add_chapter_chunks(session, source, chapter, parsed_chapter, chunk_index, config)
|
||||||
|
|
||||||
|
await session.commit()
|
||||||
|
mention_count = await index_chunk_phrase_mentions_for_book(session, source.id, config)
|
||||||
|
logger.info(
|
||||||
|
"ebook_ingest_file_complete source_id=%s path=%s chapters=%s chunks=%s phrase_mentions=%s",
|
||||||
|
source.id,
|
||||||
|
resolved_path,
|
||||||
|
len(parsed.chapters),
|
||||||
|
chunk_index,
|
||||||
|
mention_count,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logger.exception(f"ebook_ingest_file_error path={path}")
|
||||||
|
return False
|
||||||
|
else:
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
async def find_existing_source(session: AsyncSession, path: Path, file_hash: str) -> EbookSource | None:
|
||||||
|
"""Find an existing source by canonical path or file hash."""
|
||||||
|
return await session.scalar(
|
||||||
|
select(EbookSource).where(or_(EbookSource.file_path == str(path), EbookSource.file_sha256 == file_hash))
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def add_chapter_chunks(
|
||||||
|
session: AsyncSession,
|
||||||
|
source: EbookSource,
|
||||||
|
chapter: EbookChapter,
|
||||||
|
parsed_chapter: ParsedChapter,
|
||||||
|
chunk_index: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> int:
|
||||||
|
"""Add chunk rows for one parsed chapter and return the next chunk index."""
|
||||||
|
page_label = parsed_chapter.page_labels[0] if parsed_chapter.page_labels else None
|
||||||
|
for text_chunk in chunk_text(
|
||||||
|
parsed_chapter.text,
|
||||||
|
chunk_tokens=config.chunk_tokens,
|
||||||
|
overlap_tokens=config.chunk_overlap,
|
||||||
|
):
|
||||||
|
session.add(
|
||||||
|
EbookChunk(
|
||||||
|
source_id=source.id,
|
||||||
|
chapter_id=chapter.id,
|
||||||
|
chunk_index=chunk_index,
|
||||||
|
text=text_chunk.text,
|
||||||
|
token_start=text_chunk.token_start,
|
||||||
|
token_count=text_chunk.token_count,
|
||||||
|
page_label=page_label,
|
||||||
|
content_sha256=hashlib.sha256(text_chunk.text.encode()).hexdigest(),
|
||||||
|
search_text=f"{source.title} {source.author or ''} {chapter.title or ''} {text_chunk.text}",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
chunk_index += 1
|
||||||
|
return chunk_index
|
||||||
|
|
||||||
|
|
||||||
|
def sha256_file(path: Path) -> str:
|
||||||
|
"""Calculate the SHA-256 digest for a file."""
|
||||||
|
digest = hashlib.sha256()
|
||||||
|
with path.open("rb") as file:
|
||||||
|
for block in iter(lambda: file.read(1024 * 1024), b""):
|
||||||
|
digest.update(block)
|
||||||
|
return digest.hexdigest()
|
||||||
@@ -0,0 +1,223 @@
|
|||||||
|
"""LLM provider HTTP adapters."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig, RerankConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def auth_headers(api_key: str) -> dict[str, str]:
|
||||||
|
"""Build authorization headers when an API key is configured."""
|
||||||
|
if api_key == "not-needed":
|
||||||
|
return {}
|
||||||
|
return {"Authorization": f"Bearer {api_key}"}
|
||||||
|
|
||||||
|
|
||||||
|
async def request_embeddings(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
texts: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> list[list[float]]:
|
||||||
|
"""Request embeddings from the configured OpenAI-compatible endpoint.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
client (httpx.AsyncClient): Shared async client for LLM calls.
|
||||||
|
texts (Sequence[str]): Texts to embed.
|
||||||
|
config (EbookSearchConfig): Runtime settings supplying the endpoint, model, and auth.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[list[float]]: One embedding vector per input text.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
RuntimeError: If the request fails or the response cannot be parsed.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
response = await client.post(
|
||||||
|
f"{config.embedding_base_url.rstrip('/')}/embeddings",
|
||||||
|
headers=auth_headers(config.embedding_api_key),
|
||||||
|
json={"model": config.embedding_model, "input": list(texts)},
|
||||||
|
timeout=config.embedding_timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
return embedding_vectors_from_response(response.json())
|
||||||
|
except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
|
||||||
|
logger.exception(
|
||||||
|
"ebook_embed_request_failed base_url=%s model=%s count=%s",
|
||||||
|
config.embedding_base_url,
|
||||||
|
config.embedding_model,
|
||||||
|
len(texts),
|
||||||
|
)
|
||||||
|
msg = f"Embedding request failed. base_url={config.embedding_base_url} model={config.embedding_model}"
|
||||||
|
raise RuntimeError(msg) from error
|
||||||
|
|
||||||
|
|
||||||
|
async def check_embedding_endpoint(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
timeout_seconds: float = 5.0,
|
||||||
|
) -> bool:
|
||||||
|
"""Return whether the configured embedding endpoint answers a model listing."""
|
||||||
|
try:
|
||||||
|
response = await client.get(
|
||||||
|
f"{config.embedding_base_url.rstrip('/')}/models",
|
||||||
|
headers=auth_headers(config.embedding_api_key),
|
||||||
|
timeout=timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
except httpx.HTTPError as error:
|
||||||
|
logger.warning("ebook_embedding_endpoint_unreachable base_url=%s error=%s", config.embedding_base_url, error)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
async def check_chat_endpoint(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
timeout_seconds: float = 5.0,
|
||||||
|
) -> bool:
|
||||||
|
"""Return whether the configured chat (answering) endpoint answers a model listing."""
|
||||||
|
try:
|
||||||
|
response = await client.get(
|
||||||
|
f"{config.vllm_base_url.rstrip('/')}/models",
|
||||||
|
headers=auth_headers(config.vllm_api_key),
|
||||||
|
timeout=timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
except httpx.HTTPError as error:
|
||||||
|
logger.warning("ebook_chat_endpoint_unreachable base_url=%s error=%s", config.vllm_base_url, error)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def embedding_vectors_from_response(body: object) -> list[list[float]]:
|
||||||
|
"""Extract embedding vectors from an OpenAI-compatible embedding response."""
|
||||||
|
if not isinstance(body, dict):
|
||||||
|
msg = "Embedding response is not an object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
data = body["data"]
|
||||||
|
if not isinstance(data, list):
|
||||||
|
msg = "Embedding response data is not a list"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
vectors: list[list[float]] = []
|
||||||
|
for item in data:
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
msg = "Embedding item is not an object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
embedding = item["embedding"]
|
||||||
|
if not isinstance(embedding, list):
|
||||||
|
msg = "Embedding value is not a list"
|
||||||
|
raise TypeError(msg)
|
||||||
|
vectors.append([float(value) for value in embedding])
|
||||||
|
return vectors
|
||||||
|
|
||||||
|
|
||||||
|
async def request_rerank(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
query: str,
|
||||||
|
documents: Sequence[str],
|
||||||
|
config: RerankConfig,
|
||||||
|
) -> object | None:
|
||||||
|
"""Request rerank scores from the configured vLLM endpoint.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
client (httpx.AsyncClient): Shared async client for LLM calls.
|
||||||
|
query (str): Query the documents are scored against.
|
||||||
|
documents (Sequence[str]): Candidate documents to score.
|
||||||
|
config (RerankConfig): Rerank endpoint settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
object | None: The decoded response body, or ``None`` when it is not valid JSON.
|
||||||
|
"""
|
||||||
|
payload = {
|
||||||
|
"model": config.model,
|
||||||
|
"query": query,
|
||||||
|
"documents": list(documents),
|
||||||
|
}
|
||||||
|
response = await client.post(
|
||||||
|
f"{config.base_url.rstrip('/')}/rerank",
|
||||||
|
json=payload,
|
||||||
|
timeout=config.timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
try:
|
||||||
|
return response.json()
|
||||||
|
except ValueError:
|
||||||
|
logger.debug("ebook_rerank_response_invalid_json", extra={"response": response.text})
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def request_chat_completion(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
messages: Sequence[dict[str, str]],
|
||||||
|
) -> str:
|
||||||
|
"""Request a chat completion over a shared async client.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
client (httpx.AsyncClient): Shared async client whose connection pool bounds concurrency.
|
||||||
|
config (EbookSearchConfig): Runtime settings supplying the endpoint, model, and auth.
|
||||||
|
messages (Sequence[dict[str, str]]): OpenAI-style chat messages.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The assistant message text.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
RuntimeError: If the request fails or the response cannot be parsed.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
response = await client.post(
|
||||||
|
f"{config.vllm_base_url.rstrip('/')}/chat/completions",
|
||||||
|
headers=auth_headers(config.vllm_api_key),
|
||||||
|
json={
|
||||||
|
"model": config.chat_model,
|
||||||
|
"messages": list(messages),
|
||||||
|
"temperature": 0,
|
||||||
|
},
|
||||||
|
timeout=config.chat_timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
return chat_content_from_response(response.json())
|
||||||
|
except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
|
||||||
|
msg = f"Chat request failed. base_url={config.vllm_base_url} model={config.chat_model}"
|
||||||
|
raise RuntimeError(msg) from error
|
||||||
|
|
||||||
|
|
||||||
|
def chat_content_from_response(body: object) -> str:
|
||||||
|
"""Extract text content from an OpenAI-compatible chat response."""
|
||||||
|
if not isinstance(body, dict):
|
||||||
|
msg = "Chat response is not an object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
choices = body["choices"]
|
||||||
|
if not isinstance(choices, list) or not choices:
|
||||||
|
msg = "Chat response has no choices"
|
||||||
|
raise ValueError(msg)
|
||||||
|
|
||||||
|
first = choices[0]
|
||||||
|
if not isinstance(first, dict):
|
||||||
|
msg = "Chat choice is not an object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
message = first["message"]
|
||||||
|
if not isinstance(message, dict):
|
||||||
|
msg = "Chat message is not an object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
content = message.get("content") or ""
|
||||||
|
if not isinstance(content, str):
|
||||||
|
msg = "Chat content is not text"
|
||||||
|
raise TypeError(msg)
|
||||||
|
return content
|
||||||
@@ -0,0 +1,218 @@
|
|||||||
|
"""Load test for the EPUB search service.
|
||||||
|
|
||||||
|
Drives ``POST /search`` on a running server at a configurable concurrency and reports
|
||||||
|
latency percentiles, throughput, and HTTP status distribution. Queries are drawn from
|
||||||
|
the shared JSONL set (see ``eval/data/queries.jsonl``) that the eval also uses, so load
|
||||||
|
and evaluation exercise the same questions. Answer generation and reranking happen
|
||||||
|
server-side, so this exercises the full retrieval pipeline.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import math
|
||||||
|
import random
|
||||||
|
import statistics
|
||||||
|
import time
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Annotated
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import typer
|
||||||
|
|
||||||
|
from python.common import configure_logger
|
||||||
|
from python.ebook_search.eval.dataset import DEFAULT_QUERIES_PATH, load_gold_queries
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RequestResult:
|
||||||
|
"""Outcome of a single search request."""
|
||||||
|
|
||||||
|
status_code: int
|
||||||
|
latency_ms: float
|
||||||
|
ok: bool
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class LoadSummary:
|
||||||
|
"""Aggregate results of a load test run."""
|
||||||
|
|
||||||
|
total: int
|
||||||
|
successes: int
|
||||||
|
failures: int
|
||||||
|
wall_seconds: float
|
||||||
|
throughput_rps: float
|
||||||
|
latency_p50_ms: float
|
||||||
|
latency_p90_ms: float
|
||||||
|
latency_p95_ms: float
|
||||||
|
latency_p99_ms: float
|
||||||
|
latency_mean_ms: float
|
||||||
|
latency_max_ms: float
|
||||||
|
status_counts: dict[int, int]
|
||||||
|
|
||||||
|
|
||||||
|
def load_queries(queries_file: str | None) -> list[str]:
|
||||||
|
"""Return the query strings from the shared JSONL set (or a custom JSONL file)."""
|
||||||
|
path = Path(queries_file) if queries_file else DEFAULT_QUERIES_PATH
|
||||||
|
queries = [gold.query for gold in load_gold_queries(path)]
|
||||||
|
if not queries:
|
||||||
|
msg = f"No queries found in {path}"
|
||||||
|
raise typer.BadParameter(msg)
|
||||||
|
return queries
|
||||||
|
|
||||||
|
|
||||||
|
def pick_query(queries: list[str]) -> str:
|
||||||
|
"""Return a uniformly random query from the pool (not a security context)."""
|
||||||
|
return random.choice(queries) # noqa: S311 load-test query sampling is not security-sensitive
|
||||||
|
|
||||||
|
|
||||||
|
def percentile(values_sorted: list[float], pct: float) -> float:
|
||||||
|
"""Return the linearly-interpolated percentile of a sorted list."""
|
||||||
|
if not values_sorted:
|
||||||
|
return 0.0
|
||||||
|
rank = (pct / 100) * (len(values_sorted) - 1)
|
||||||
|
low = math.floor(rank)
|
||||||
|
high = math.ceil(rank)
|
||||||
|
if low == high:
|
||||||
|
return values_sorted[low]
|
||||||
|
return values_sorted[low] + (values_sorted[high] - values_sorted[low]) * (rank - low)
|
||||||
|
|
||||||
|
|
||||||
|
def summarize(results: list[RequestResult], wall_seconds: float) -> LoadSummary:
|
||||||
|
"""Aggregate per-request results into a load summary."""
|
||||||
|
latencies = sorted(result.latency_ms for result in results)
|
||||||
|
successes = sum(1 for result in results if result.ok)
|
||||||
|
status_counts: dict[int, int] = {}
|
||||||
|
for result in results:
|
||||||
|
status_counts[result.status_code] = status_counts.get(result.status_code, 0) + 1
|
||||||
|
return LoadSummary(
|
||||||
|
total=len(results),
|
||||||
|
successes=successes,
|
||||||
|
failures=len(results) - successes,
|
||||||
|
wall_seconds=wall_seconds,
|
||||||
|
throughput_rps=len(results) / wall_seconds if wall_seconds > 0 else 0.0,
|
||||||
|
latency_p50_ms=percentile(latencies, 50),
|
||||||
|
latency_p90_ms=percentile(latencies, 90),
|
||||||
|
latency_p95_ms=percentile(latencies, 95),
|
||||||
|
latency_p99_ms=percentile(latencies, 99),
|
||||||
|
latency_mean_ms=statistics.fmean(latencies) if latencies else 0.0,
|
||||||
|
latency_max_ms=latencies[-1] if latencies else 0.0,
|
||||||
|
status_counts=status_counts,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def send_search(client: httpx.AsyncClient, query: str, *, rerank: bool) -> RequestResult:
|
||||||
|
"""Send one search request and record its status and latency."""
|
||||||
|
data = {"query": query, "rerank": "true"} if rerank else {"query": query}
|
||||||
|
start = time.perf_counter()
|
||||||
|
try:
|
||||||
|
response = await client.post("/search", data=data)
|
||||||
|
except httpx.HTTPError as error:
|
||||||
|
logger.warning("ebook_loadtest_request_failed error=%s", error)
|
||||||
|
return RequestResult(status_code=0, latency_ms=(time.perf_counter() - start) * 1000, ok=False)
|
||||||
|
return RequestResult(
|
||||||
|
status_code=response.status_code,
|
||||||
|
latency_ms=(time.perf_counter() - start) * 1000,
|
||||||
|
ok=response.is_success,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def worker(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
queue: asyncio.Queue[str],
|
||||||
|
results: list[RequestResult],
|
||||||
|
*,
|
||||||
|
rerank: bool,
|
||||||
|
) -> None:
|
||||||
|
"""Pull queries off the queue and send requests until it is empty."""
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
query = queue.get_nowait()
|
||||||
|
except asyncio.QueueEmpty:
|
||||||
|
return
|
||||||
|
results.append(await send_search(client, query, rerank=rerank))
|
||||||
|
|
||||||
|
|
||||||
|
async def run_load(
|
||||||
|
*,
|
||||||
|
base_url: str,
|
||||||
|
queries: list[str],
|
||||||
|
request_count: int,
|
||||||
|
concurrency: int,
|
||||||
|
rerank: bool,
|
||||||
|
warmup: int,
|
||||||
|
timeout_seconds: float,
|
||||||
|
) -> LoadSummary:
|
||||||
|
"""Run the load test and return its aggregate summary."""
|
||||||
|
limits = httpx.Limits(max_connections=concurrency, max_keepalive_connections=concurrency)
|
||||||
|
async with httpx.AsyncClient(base_url=base_url, timeout=timeout_seconds, limits=limits) as client:
|
||||||
|
for _ in range(warmup):
|
||||||
|
await send_search(client, pick_query(queries), rerank=rerank)
|
||||||
|
|
||||||
|
queue: asyncio.Queue[str] = asyncio.Queue()
|
||||||
|
for _ in range(request_count):
|
||||||
|
queue.put_nowait(pick_query(queries))
|
||||||
|
|
||||||
|
results: list[RequestResult] = []
|
||||||
|
start = time.perf_counter()
|
||||||
|
workers = [asyncio.create_task(worker(client, queue, results, rerank=rerank)) for _ in range(concurrency)]
|
||||||
|
await asyncio.gather(*workers)
|
||||||
|
wall_seconds = time.perf_counter() - start
|
||||||
|
return summarize(results, wall_seconds)
|
||||||
|
|
||||||
|
|
||||||
|
def print_summary(summary: LoadSummary) -> None:
|
||||||
|
"""Print the load summary to stdout."""
|
||||||
|
typer.echo(f"requests={summary.total} successes={summary.successes} failures={summary.failures}")
|
||||||
|
typer.echo(f"wall={summary.wall_seconds:.2f}s throughput={summary.throughput_rps:.1f} req/s")
|
||||||
|
typer.echo(
|
||||||
|
f"latency_ms p50={summary.latency_p50_ms:.1f} p90={summary.latency_p90_ms:.1f} "
|
||||||
|
f"p95={summary.latency_p95_ms:.1f} p99={summary.latency_p99_ms:.1f} "
|
||||||
|
f"mean={summary.latency_mean_ms:.1f} max={summary.latency_max_ms:.1f}"
|
||||||
|
)
|
||||||
|
status_summary = " ".join(f"{code}={count}" for code, count in sorted(summary.status_counts.items()))
|
||||||
|
typer.echo(f"status {status_summary}")
|
||||||
|
|
||||||
|
|
||||||
|
def main(
|
||||||
|
*,
|
||||||
|
base_url: Annotated[str, typer.Option(help="Base URL of the running service")] = "http://127.0.0.1:8070",
|
||||||
|
request_count: Annotated[int, typer.Option("--requests", help="Total requests to send")] = 200,
|
||||||
|
concurrency: Annotated[int, typer.Option(help="Concurrent in-flight requests")] = 10,
|
||||||
|
rerank: Annotated[bool, typer.Option(help="Request server-side reranking")] = False,
|
||||||
|
warmup: Annotated[int, typer.Option(help="Warmup requests, not measured")] = 5,
|
||||||
|
timeout_seconds: Annotated[float, typer.Option("--timeout", help="Per-request timeout seconds")] = 120.0,
|
||||||
|
queries_file: Annotated[str | None, typer.Option(help="Query JSONL file (defaults to the shared set)")] = None,
|
||||||
|
log_level: Annotated[str, typer.Option(help="Log level")] = "WARNING",
|
||||||
|
) -> None:
|
||||||
|
"""Load test the search endpoint and report latency and throughput."""
|
||||||
|
configure_logger(log_level)
|
||||||
|
queries = load_queries(queries_file)
|
||||||
|
logger.info(
|
||||||
|
"ebook_loadtest_start base_url=%s requests=%s concurrency=%s rerank=%s queries=%s",
|
||||||
|
base_url,
|
||||||
|
request_count,
|
||||||
|
concurrency,
|
||||||
|
rerank,
|
||||||
|
len(queries),
|
||||||
|
)
|
||||||
|
summary = asyncio.run(
|
||||||
|
run_load(
|
||||||
|
base_url=base_url,
|
||||||
|
queries=queries,
|
||||||
|
request_count=request_count,
|
||||||
|
concurrency=concurrency,
|
||||||
|
rerank=rerank,
|
||||||
|
warmup=warmup,
|
||||||
|
timeout_seconds=timeout_seconds,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
print_summary(summary)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
typer.run(main)
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Init."""
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
"""Protected phrase extraction, storage, and runtime matching."""
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.config.lib import (
|
||||||
|
get_bad_ends,
|
||||||
|
get_bad_starts,
|
||||||
|
get_ignored_phrases,
|
||||||
|
get_junk_tokens,
|
||||||
|
get_most_common_words,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"get_bad_ends",
|
||||||
|
"get_bad_starts",
|
||||||
|
"get_ignored_phrases",
|
||||||
|
"get_junk_tokens",
|
||||||
|
"get_most_common_words",
|
||||||
|
]
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
tokens = [
|
||||||
|
"a",
|
||||||
|
"an",
|
||||||
|
"and",
|
||||||
|
"any",
|
||||||
|
"as",
|
||||||
|
"at",
|
||||||
|
"be",
|
||||||
|
"because",
|
||||||
|
"but",
|
||||||
|
"by",
|
||||||
|
"can",
|
||||||
|
"could",
|
||||||
|
"do",
|
||||||
|
"for",
|
||||||
|
"from",
|
||||||
|
"have",
|
||||||
|
"if",
|
||||||
|
"of",
|
||||||
|
"or",
|
||||||
|
"some",
|
||||||
|
"than",
|
||||||
|
"the",
|
||||||
|
"these",
|
||||||
|
"this",
|
||||||
|
"to",
|
||||||
|
"will",
|
||||||
|
"with",
|
||||||
|
"would",
|
||||||
|
"did",
|
||||||
|
]
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
tokens = [
|
||||||
|
"a",
|
||||||
|
"an",
|
||||||
|
"did",
|
||||||
|
"didn't",
|
||||||
|
"he",
|
||||||
|
"here",
|
||||||
|
"how",
|
||||||
|
"i",
|
||||||
|
"it",
|
||||||
|
"she",
|
||||||
|
"that",
|
||||||
|
"the",
|
||||||
|
"there",
|
||||||
|
"they",
|
||||||
|
"this",
|
||||||
|
"we",
|
||||||
|
"what",
|
||||||
|
"when",
|
||||||
|
"where",
|
||||||
|
"which",
|
||||||
|
"who",
|
||||||
|
"whom",
|
||||||
|
"whose",
|
||||||
|
"why",
|
||||||
|
"you",
|
||||||
|
]
|
||||||
@@ -0,0 +1,212 @@
|
|||||||
|
phrases = [
|
||||||
|
"a little",
|
||||||
|
"across the",
|
||||||
|
"and she",
|
||||||
|
"anyone in",
|
||||||
|
"are you",
|
||||||
|
"around him",
|
||||||
|
"around the",
|
||||||
|
"as much",
|
||||||
|
"as soon",
|
||||||
|
"at all",
|
||||||
|
"at least",
|
||||||
|
"before the",
|
||||||
|
"behind him",
|
||||||
|
"between the",
|
||||||
|
"but she",
|
||||||
|
"could not",
|
||||||
|
"did he",
|
||||||
|
"did i",
|
||||||
|
"did it",
|
||||||
|
"did not believe",
|
||||||
|
"did not care",
|
||||||
|
"did not even",
|
||||||
|
"did not know what",
|
||||||
|
"did not know",
|
||||||
|
"did not like",
|
||||||
|
"did not look",
|
||||||
|
"did not mean",
|
||||||
|
"did not move",
|
||||||
|
"did not need",
|
||||||
|
"did not see",
|
||||||
|
"did not seem",
|
||||||
|
"did not think",
|
||||||
|
"did not understand",
|
||||||
|
"did not want",
|
||||||
|
"did not",
|
||||||
|
"did she",
|
||||||
|
"did so",
|
||||||
|
"did that",
|
||||||
|
"did the",
|
||||||
|
"did they",
|
||||||
|
"did what",
|
||||||
|
"did you",
|
||||||
|
"didn't answer",
|
||||||
|
"didn't care",
|
||||||
|
"didn't even",
|
||||||
|
"didn't expect",
|
||||||
|
"didn't feel",
|
||||||
|
"didn't get",
|
||||||
|
"didn't i",
|
||||||
|
"didn't know",
|
||||||
|
"didn't like",
|
||||||
|
"didn't look",
|
||||||
|
"didn't make",
|
||||||
|
"didn't mean",
|
||||||
|
"didn't need",
|
||||||
|
"didn't really",
|
||||||
|
"didn't say",
|
||||||
|
"didn't see",
|
||||||
|
"didn't seem",
|
||||||
|
"didn't think",
|
||||||
|
"didn't want",
|
||||||
|
"didn't you",
|
||||||
|
"end up",
|
||||||
|
"ended up",
|
||||||
|
"had a",
|
||||||
|
"had been",
|
||||||
|
"have been",
|
||||||
|
"he asked",
|
||||||
|
"he concluded",
|
||||||
|
"he continued",
|
||||||
|
"he couldn't",
|
||||||
|
"he did",
|
||||||
|
"he didn't",
|
||||||
|
"he felt",
|
||||||
|
"he had",
|
||||||
|
"he hadn't",
|
||||||
|
"he knew",
|
||||||
|
"he noted",
|
||||||
|
"he pointed",
|
||||||
|
"he realized",
|
||||||
|
"he replied",
|
||||||
|
"he said",
|
||||||
|
"he saw",
|
||||||
|
"he tapped",
|
||||||
|
"he told",
|
||||||
|
"he was",
|
||||||
|
"he wasn't",
|
||||||
|
"his body",
|
||||||
|
"his chair",
|
||||||
|
"his feet",
|
||||||
|
"his hands",
|
||||||
|
"his head",
|
||||||
|
"his office",
|
||||||
|
"his own",
|
||||||
|
"his pc",
|
||||||
|
"his power",
|
||||||
|
"his shield",
|
||||||
|
"his sight",
|
||||||
|
"his voice",
|
||||||
|
"his wrist",
|
||||||
|
"how many",
|
||||||
|
"i am",
|
||||||
|
"i don't",
|
||||||
|
"i said",
|
||||||
|
"i was",
|
||||||
|
"i wouldn't",
|
||||||
|
"i'm not",
|
||||||
|
"if he",
|
||||||
|
"if they",
|
||||||
|
"is in",
|
||||||
|
"is not",
|
||||||
|
"is that",
|
||||||
|
"is the",
|
||||||
|
"it had",
|
||||||
|
"it had",
|
||||||
|
"it is",
|
||||||
|
"it was",
|
||||||
|
"it wasn't",
|
||||||
|
"it wasn't",
|
||||||
|
"no one",
|
||||||
|
"of course",
|
||||||
|
"of force",
|
||||||
|
"of it",
|
||||||
|
"of magic",
|
||||||
|
"of marines",
|
||||||
|
"of power",
|
||||||
|
"of those",
|
||||||
|
"old man",
|
||||||
|
"older man",
|
||||||
|
"one of",
|
||||||
|
"out of",
|
||||||
|
"set up",
|
||||||
|
"she admitted",
|
||||||
|
"she asked",
|
||||||
|
"she had",
|
||||||
|
"she replied",
|
||||||
|
"she said",
|
||||||
|
"she snapped",
|
||||||
|
"she told",
|
||||||
|
"she was",
|
||||||
|
"she'd been",
|
||||||
|
"shook his",
|
||||||
|
"sure he",
|
||||||
|
"tell you",
|
||||||
|
"that had",
|
||||||
|
"that is",
|
||||||
|
"that she",
|
||||||
|
"that was",
|
||||||
|
"the dark",
|
||||||
|
"the door",
|
||||||
|
"the first",
|
||||||
|
"the last",
|
||||||
|
"the man",
|
||||||
|
"the one",
|
||||||
|
"the only",
|
||||||
|
"the other",
|
||||||
|
"the rest",
|
||||||
|
"the room",
|
||||||
|
"the same",
|
||||||
|
"the two",
|
||||||
|
"the way",
|
||||||
|
"the world",
|
||||||
|
"there are",
|
||||||
|
"there was",
|
||||||
|
"there were",
|
||||||
|
"they are",
|
||||||
|
"they had",
|
||||||
|
"they were",
|
||||||
|
"they weren't",
|
||||||
|
"this is",
|
||||||
|
"this place",
|
||||||
|
"though he",
|
||||||
|
"through his",
|
||||||
|
"through the",
|
||||||
|
"to find",
|
||||||
|
"to get",
|
||||||
|
"to keep",
|
||||||
|
"to stay",
|
||||||
|
"to stop",
|
||||||
|
"to tell",
|
||||||
|
"to try",
|
||||||
|
"told her",
|
||||||
|
"told him",
|
||||||
|
"under his",
|
||||||
|
"was a",
|
||||||
|
"was enough",
|
||||||
|
"was going",
|
||||||
|
"was in",
|
||||||
|
"was no",
|
||||||
|
"was not",
|
||||||
|
"was now",
|
||||||
|
"was on",
|
||||||
|
"was one",
|
||||||
|
"was only",
|
||||||
|
"was still",
|
||||||
|
"was that",
|
||||||
|
"was the",
|
||||||
|
"was there",
|
||||||
|
"were in",
|
||||||
|
"what had",
|
||||||
|
"what happened",
|
||||||
|
"what was",
|
||||||
|
"where the",
|
||||||
|
"while i",
|
||||||
|
"you are",
|
||||||
|
"you can't",
|
||||||
|
"you don't",
|
||||||
|
"you know",
|
||||||
|
"you need",
|
||||||
|
"you were",
|
||||||
|
]
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
tokens = [
|
||||||
|
"said",
|
||||||
|
"asked",
|
||||||
|
"replied",
|
||||||
|
"answered",
|
||||||
|
"looked",
|
||||||
|
"nodded",
|
||||||
|
"turned",
|
||||||
|
"shook",
|
||||||
|
"smiled",
|
||||||
|
"shrugged",
|
||||||
|
"pointed",
|
||||||
|
"continued",
|
||||||
|
"repeated",
|
||||||
|
"stared",
|
||||||
|
"agreed",
|
||||||
|
"glanced",
|
||||||
|
"walked",
|
||||||
|
"told",
|
||||||
|
"thought",
|
||||||
|
"knew",
|
||||||
|
"wanted",
|
||||||
|
"muttered",
|
||||||
|
"whispered",
|
||||||
|
"laughed",
|
||||||
|
"sighed",
|
||||||
|
"paused",
|
||||||
|
"gestured",
|
||||||
|
"waved",
|
||||||
|
"frowned",
|
||||||
|
"grinned",
|
||||||
|
"admitted",
|
||||||
|
"found",
|
||||||
|
"noted",
|
||||||
|
"murmured",
|
||||||
|
"ordered",
|
||||||
|
"i'm",
|
||||||
|
"i've",
|
||||||
|
"i'd",
|
||||||
|
"i'll",
|
||||||
|
"it's",
|
||||||
|
"that's",
|
||||||
|
"don't",
|
||||||
|
"didn't",
|
||||||
|
"doesn't",
|
||||||
|
"can't",
|
||||||
|
"won't",
|
||||||
|
"wouldn't",
|
||||||
|
"couldn't",
|
||||||
|
"shouldn't",
|
||||||
|
"isn't",
|
||||||
|
"wasn't",
|
||||||
|
"aren't",
|
||||||
|
"weren't",
|
||||||
|
"you're",
|
||||||
|
"you've",
|
||||||
|
"you'll",
|
||||||
|
"we're",
|
||||||
|
"we've",
|
||||||
|
"we'll",
|
||||||
|
"they're",
|
||||||
|
"they've",
|
||||||
|
"he's",
|
||||||
|
"she's",
|
||||||
|
"there's",
|
||||||
|
"what's",
|
||||||
|
"let's",
|
||||||
|
"who's",
|
||||||
|
"he'd",
|
||||||
|
"she'd",
|
||||||
|
]
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
"""Protected phrase extraction, storage, and runtime matching."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import tomllib
|
||||||
|
from functools import cache
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import normalize_text
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_toml_string_set(path: Path, key: str) -> frozenset[str]:
|
||||||
|
"""Load and validate a TOML string list as a normalized immutable set."""
|
||||||
|
with path.open("rb") as file:
|
||||||
|
body = tomllib.load(file)
|
||||||
|
|
||||||
|
values = body.get(key)
|
||||||
|
if not isinstance(values, list) or not all(isinstance(item, str) for item in values):
|
||||||
|
msg = f"{path} must contain a {key!r} string list"
|
||||||
|
raise ValueError(msg)
|
||||||
|
return frozenset(normalize_text(value) for value in values if normalize_text(value))
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def _get_phrase_config_dir() -> Path:
|
||||||
|
"""Return the directory containing phrase configuration files."""
|
||||||
|
return Path(__file__).resolve().parent
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def get_ignored_phrases() -> frozenset[str]:
|
||||||
|
"""Return ignored phrase strings loaded from TOML."""
|
||||||
|
return _load_toml_string_set(_get_phrase_config_dir() / "ignored_phrases.toml", "phrases")
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def get_bad_ends() -> frozenset[str]:
|
||||||
|
"""Return bad phrase-ending tokens loaded from TOML."""
|
||||||
|
return _load_toml_string_set(_get_phrase_config_dir() / "bad_ends.toml", "tokens")
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def get_bad_starts() -> frozenset[str]:
|
||||||
|
"""Return bad phrase-starting tokens loaded from TOML."""
|
||||||
|
return _load_toml_string_set(_get_phrase_config_dir() / "bad_starts.toml", "tokens")
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def get_most_common_words() -> frozenset[str]:
|
||||||
|
"""Return the most common English words loaded from TOML."""
|
||||||
|
return _load_toml_string_set(_get_phrase_config_dir() / "most_common_words.toml", "words")
|
||||||
|
|
||||||
|
|
||||||
|
@cache
|
||||||
|
def get_junk_tokens() -> frozenset[str]:
|
||||||
|
"""Return junk tokens (dialogue verbs and pronoun contractions) loaded from TOML."""
|
||||||
|
return _load_toml_string_set(_get_phrase_config_dir() / "junk_tokens.toml", "tokens")
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
words = [
|
||||||
|
"the",
|
||||||
|
"be",
|
||||||
|
"to",
|
||||||
|
"of",
|
||||||
|
"and",
|
||||||
|
"a",
|
||||||
|
"in",
|
||||||
|
"that",
|
||||||
|
"have",
|
||||||
|
"I",
|
||||||
|
"it",
|
||||||
|
"for",
|
||||||
|
"not",
|
||||||
|
"on",
|
||||||
|
"with",
|
||||||
|
"he",
|
||||||
|
"as",
|
||||||
|
"you",
|
||||||
|
"do",
|
||||||
|
"at",
|
||||||
|
"this",
|
||||||
|
"but",
|
||||||
|
"his",
|
||||||
|
"by",
|
||||||
|
"from",
|
||||||
|
"they",
|
||||||
|
"we",
|
||||||
|
"say",
|
||||||
|
"her",
|
||||||
|
"she",
|
||||||
|
"or",
|
||||||
|
"an",
|
||||||
|
"will",
|
||||||
|
"my",
|
||||||
|
"one",
|
||||||
|
"all",
|
||||||
|
"would",
|
||||||
|
"there",
|
||||||
|
"their",
|
||||||
|
"what",
|
||||||
|
"so",
|
||||||
|
"up",
|
||||||
|
"out",
|
||||||
|
"if",
|
||||||
|
"about",
|
||||||
|
"who",
|
||||||
|
"get",
|
||||||
|
"which",
|
||||||
|
"go",
|
||||||
|
"me",
|
||||||
|
"when",
|
||||||
|
"make",
|
||||||
|
"can",
|
||||||
|
"like",
|
||||||
|
"time",
|
||||||
|
"no",
|
||||||
|
"just",
|
||||||
|
"him",
|
||||||
|
"know",
|
||||||
|
"take",
|
||||||
|
"people",
|
||||||
|
"into",
|
||||||
|
"year",
|
||||||
|
"your",
|
||||||
|
"good",
|
||||||
|
"some",
|
||||||
|
"could",
|
||||||
|
"them",
|
||||||
|
"see",
|
||||||
|
"other",
|
||||||
|
"than",
|
||||||
|
"then",
|
||||||
|
"now",
|
||||||
|
"look",
|
||||||
|
"only",
|
||||||
|
"come",
|
||||||
|
"its",
|
||||||
|
"over",
|
||||||
|
"think",
|
||||||
|
"also",
|
||||||
|
"back",
|
||||||
|
"after",
|
||||||
|
"use",
|
||||||
|
"two",
|
||||||
|
"how",
|
||||||
|
"our",
|
||||||
|
"work",
|
||||||
|
"first",
|
||||||
|
"well",
|
||||||
|
"way",
|
||||||
|
"even",
|
||||||
|
"new",
|
||||||
|
"want",
|
||||||
|
"because",
|
||||||
|
"any",
|
||||||
|
"these",
|
||||||
|
"give",
|
||||||
|
"day",
|
||||||
|
"most",
|
||||||
|
"us",
|
||||||
|
]
|
||||||
@@ -0,0 +1,853 @@
|
|||||||
|
"""Candidate phrase extraction and scoring for protected phrases."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from collections import Counter, defaultdict
|
||||||
|
from functools import lru_cache
|
||||||
|
from time import perf_counter
|
||||||
|
from typing import TYPE_CHECKING, Protocol
|
||||||
|
|
||||||
|
from yake import KeywordExtractor
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.config import (
|
||||||
|
get_bad_ends,
|
||||||
|
get_bad_starts,
|
||||||
|
get_ignored_phrases,
|
||||||
|
get_junk_tokens,
|
||||||
|
get_most_common_words,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.models import PhraseCandidate
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import tokenize, tokenize_with_offsets
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Iterable, Mapping, Sequence
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
BAD_START_SCORE_PENALTY = 10.0
|
||||||
|
BAD_END_SCORE_PENALTY = 10.0
|
||||||
|
MULTI_SOURCE_SCORE_BONUS = 2.0
|
||||||
|
MULTI_SOURCE_MIN_SOURCES = 2
|
||||||
|
CAPITALIZED_PHRASE_RE = re.compile(r"\b(?:[A-Z][a-zA-Z']+)(?:\s+(?:of|the|and|in|on|for|[A-Z][a-zA-Z']+)){0,6}")
|
||||||
|
|
||||||
|
|
||||||
|
class SpacySpan(Protocol):
|
||||||
|
"""Small protocol for the spaCy span attributes used by this module."""
|
||||||
|
|
||||||
|
text: str
|
||||||
|
|
||||||
|
|
||||||
|
class SpacyEntity(SpacySpan, Protocol):
|
||||||
|
"""Small protocol for the spaCy entity attributes used by this module."""
|
||||||
|
|
||||||
|
label_: str
|
||||||
|
|
||||||
|
|
||||||
|
class SpacyDoc(Protocol):
|
||||||
|
"""Small protocol for the spaCy doc attributes used by this module."""
|
||||||
|
|
||||||
|
ents: Iterable[SpacyEntity]
|
||||||
|
noun_chunks: Iterable[SpacySpan]
|
||||||
|
|
||||||
|
|
||||||
|
class SpacyLanguage(Protocol):
|
||||||
|
"""Small protocol for a callable spaCy language pipeline."""
|
||||||
|
|
||||||
|
def __call__(self, text: str) -> SpacyDoc:
|
||||||
|
"""Parse text into a spaCy-like doc."""
|
||||||
|
|
||||||
|
|
||||||
|
class YakeExtractor(Protocol):
|
||||||
|
"""Small protocol for the YAKE extractor used by this module."""
|
||||||
|
|
||||||
|
def extract_keywords(self, text: str) -> Iterable[tuple[str, float]]:
|
||||||
|
"""Return YAKE keyword tuples."""
|
||||||
|
|
||||||
|
|
||||||
|
class YakeExtractorFactory(Protocol):
|
||||||
|
"""Callable constructor protocol for YAKE keyword extractors."""
|
||||||
|
|
||||||
|
def __call__(self, *, lan: str, n: int, dedupLim: float, top: int) -> YakeExtractor: # noqa: N803
|
||||||
|
"""Create a YAKE keyword extractor.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
lan (str): Language code passed to YAKE.
|
||||||
|
n (int): Maximum n-gram size to extract.
|
||||||
|
dedupLim (float): Deduplication similarity threshold.
|
||||||
|
top (int): Maximum number of keyphrases to return.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
YakeExtractor: The constructed keyword extractor.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_candidate_phrase(
|
||||||
|
phrase_text: str,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
min_tokens: int | None = None,
|
||||||
|
max_tokens: int | None = None,
|
||||||
|
strip_leading_article: bool = False,
|
||||||
|
) -> tuple[str, str, int] | None:
|
||||||
|
"""Normalize a candidate phrase and validate token bounds.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
phrase_text (str): Raw phrase text to normalize.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
min_tokens (int | None): Minimum token count override; defaults to ``config.phrase_min_tokens``.
|
||||||
|
max_tokens (int | None): Maximum token count override; defaults to ``config.phrase_max_tokens``.
|
||||||
|
strip_leading_article (bool): Whether to drop a single leading English article.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[str, str, int] | None: Display text, normalized phrase, and token count, or ``None``
|
||||||
|
when the phrase falls outside the token bounds or is ignored.
|
||||||
|
"""
|
||||||
|
normalized_tokens = tokenize_with_offsets(phrase_text)
|
||||||
|
start = 0
|
||||||
|
if strip_leading_article and normalized_tokens and normalized_tokens[0].text in {"the", "a", "an"}:
|
||||||
|
start = 1
|
||||||
|
|
||||||
|
selected_tokens = normalized_tokens[start:]
|
||||||
|
min_count = config.phrase_min_tokens if min_tokens is None else min_tokens
|
||||||
|
max_count = config.phrase_max_tokens if max_tokens is None else max_tokens
|
||||||
|
if len(selected_tokens) < min_count or len(selected_tokens) > max_count:
|
||||||
|
return None
|
||||||
|
|
||||||
|
phrase_norm = " ".join(token.text for token in selected_tokens)
|
||||||
|
if phrase_norm in get_ignored_phrases():
|
||||||
|
return None
|
||||||
|
|
||||||
|
display_text = phrase_text[selected_tokens[0].start_char : selected_tokens[-1].end_char].strip()
|
||||||
|
return display_text or phrase_norm, phrase_norm, len(selected_tokens)
|
||||||
|
|
||||||
|
|
||||||
|
def count_raw_ngrams(tokens: Sequence[str], config: EbookSearchConfig) -> Counter[str]:
|
||||||
|
"""Count every n-gram window in one normalized token block.
|
||||||
|
|
||||||
|
``tokens`` are already normalized (see :func:`tokenize`), so each window's normalized form
|
||||||
|
is the joined tokens directly. Counting into a plain :class:`Counter` rather than
|
||||||
|
:class:`PhraseCandidate` objects keeps this hot loop cheap; callers filter ignored phrases
|
||||||
|
and materialize candidates per unique phrase afterwards, which is far fewer operations than
|
||||||
|
doing either per window.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tokens (Sequence[str]): Normalized tokens for one text block.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Counter[str]: Raw occurrence counts keyed by normalized phrase.
|
||||||
|
"""
|
||||||
|
return Counter(
|
||||||
|
" ".join(tokens[start : start + ngram_size])
|
||||||
|
for ngram_size in range(config.phrase_min_tokens, config.phrase_max_tokens + 1)
|
||||||
|
for start in range(len(tokens) - ngram_size + 1)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_raw_ngrams_by_chapter(
|
||||||
|
chapters: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Extract raw n-grams across chapters, tracking both raw counts and chapter spread.
|
||||||
|
|
||||||
|
Counting each chapter separately makes chapter spread fall out of dict membership: a phrase's
|
||||||
|
``chapter_count`` is simply how many per-chapter count maps contain it, so no per-window seen
|
||||||
|
tracking is needed. This also lets the enrichment step skip re-sliding the same n-gram sizes.
|
||||||
|
|
||||||
|
Phrases below the minimum raw count are dropped here rather than materialized: most unique
|
||||||
|
n-grams occur once, and :func:`filter_storable_candidates` would discard them as too rare
|
||||||
|
anyway, so building ``PhraseCandidate`` objects for them is wasted work.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks to slide n-gram windows over.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates meeting the minimum raw count, keyed by normalized
|
||||||
|
phrase, with raw and chapter counts.
|
||||||
|
"""
|
||||||
|
chapter_count_maps = [count_raw_ngrams(tokenize(chapter), config) for chapter in chapters]
|
||||||
|
total_counts: Counter[str] = Counter()
|
||||||
|
chapter_spread: Counter[str] = Counter()
|
||||||
|
for chapter_counts in chapter_count_maps:
|
||||||
|
total_counts.update(chapter_counts)
|
||||||
|
chapter_spread.update(chapter_counts.keys())
|
||||||
|
min_raw_count = minimum_candidate_raw_count(config)
|
||||||
|
ignored = get_ignored_phrases()
|
||||||
|
return {
|
||||||
|
phrase_norm: PhraseCandidate(
|
||||||
|
phrase_text=phrase_norm,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=phrase_norm.count(" ") + 1,
|
||||||
|
source_raw_ngram=True,
|
||||||
|
raw_count=raw_count,
|
||||||
|
chapter_count=chapter_spread[phrase_norm],
|
||||||
|
)
|
||||||
|
for phrase_norm, raw_count in total_counts.items()
|
||||||
|
if raw_count >= min_raw_count and phrase_norm not in ignored
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@lru_cache(maxsize=2)
|
||||||
|
def get_yake_extractor(max_ngram: int, top_k: int) -> KeywordExtractor:
|
||||||
|
"""Return a cached YAKE extractor for the given settings.
|
||||||
|
|
||||||
|
Constructing a ``KeywordExtractor`` loads the language's stopword list from disk, so it is
|
||||||
|
cached and reused across books rather than rebuilt on every call.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
max_ngram (int): Maximum n-gram size to extract.
|
||||||
|
top_k (int): Maximum number of keyphrases to request.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
KeywordExtractor: A shared extractor instance for the given settings.
|
||||||
|
"""
|
||||||
|
return KeywordExtractor(lan="en", n=max_ngram, dedupLim=0.85, top=top_k)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_yake_candidates(
|
||||||
|
book_text: str,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
top_k: int = 1000,
|
||||||
|
) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Extract YAKE keyphrases when the optional YAKE package is installed.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_text (str): Full book text to extract keyphrases from.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
top_k (int): Maximum number of YAKE keyphrases to request.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates keyed by normalized phrase, with YAKE scores.
|
||||||
|
"""
|
||||||
|
extractor = get_yake_extractor(config.phrase_max_tokens, top_k)
|
||||||
|
out: dict[str, PhraseCandidate] = {}
|
||||||
|
for phrase_text, yake_score in extractor.extract_keywords(book_text):
|
||||||
|
normalized = normalize_candidate_phrase(phrase_text, config)
|
||||||
|
if normalized is None:
|
||||||
|
continue
|
||||||
|
display_text, phrase_norm, token_count = normalized
|
||||||
|
out[phrase_norm] = PhraseCandidate(
|
||||||
|
phrase_text=display_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=token_count,
|
||||||
|
source_yake=True,
|
||||||
|
yake_score=float(yake_score),
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def extract_spacy_candidates(
|
||||||
|
book_text: str,
|
||||||
|
nlp: SpacyLanguage,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Extract spaCy named entities and noun chunks from one text block.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_text (str): Text block to parse with spaCy.
|
||||||
|
nlp (SpacyLanguage): Callable spaCy language pipeline.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates keyed by normalized phrase from entities and noun chunks.
|
||||||
|
"""
|
||||||
|
out: dict[str, PhraseCandidate] = {}
|
||||||
|
doc = nlp(book_text)
|
||||||
|
|
||||||
|
for ent in doc.ents:
|
||||||
|
normalized = normalize_candidate_phrase(
|
||||||
|
ent.text,
|
||||||
|
config,
|
||||||
|
max_tokens=config.phrase_max_entity_tokens,
|
||||||
|
)
|
||||||
|
if normalized is None:
|
||||||
|
continue
|
||||||
|
phrase_text, phrase_norm, token_count = normalized
|
||||||
|
out[phrase_norm] = PhraseCandidate(
|
||||||
|
phrase_text=phrase_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=token_count,
|
||||||
|
source_spacy_ner=True,
|
||||||
|
spacy_label=ent.label_,
|
||||||
|
)
|
||||||
|
|
||||||
|
for chunk in doc.noun_chunks:
|
||||||
|
normalized = normalize_candidate_phrase(chunk.text, config, strip_leading_article=True)
|
||||||
|
if normalized is None:
|
||||||
|
continue
|
||||||
|
phrase_text, phrase_norm, token_count = normalized
|
||||||
|
out[phrase_norm] = PhraseCandidate(
|
||||||
|
phrase_text=phrase_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=token_count,
|
||||||
|
source_spacy_noun_chunk=True,
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def extract_capitalized_phrases(original_text: str, config: EbookSearchConfig) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Extract capitalized phrase runs that often carry fictional terms.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
original_text (str): Original-case book text to scan for capitalized runs.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates keyed by normalized phrase from capitalized runs.
|
||||||
|
"""
|
||||||
|
out: dict[str, PhraseCandidate] = {}
|
||||||
|
for match in CAPITALIZED_PHRASE_RE.finditer(original_text):
|
||||||
|
phrase_text = match.group(0).strip()
|
||||||
|
normalized = normalize_candidate_phrase(
|
||||||
|
phrase_text,
|
||||||
|
config,
|
||||||
|
max_tokens=config.phrase_max_entity_tokens,
|
||||||
|
)
|
||||||
|
if normalized is None:
|
||||||
|
continue
|
||||||
|
display_text, phrase_norm, token_count = normalized
|
||||||
|
out[phrase_norm] = PhraseCandidate(
|
||||||
|
phrase_text=display_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=token_count,
|
||||||
|
source_capitalized=True,
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def extract_metadata_candidates(
|
||||||
|
metadata: Mapping[str, object] | None,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Extract phrases from book metadata values such as title, author, and series.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
metadata (Mapping[str, object] | None): Book metadata values, or ``None`` when unavailable.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates keyed by normalized phrase from metadata values.
|
||||||
|
"""
|
||||||
|
if metadata is None:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
out: dict[str, PhraseCandidate] = {}
|
||||||
|
for value in metadata.values():
|
||||||
|
if value is None:
|
||||||
|
continue
|
||||||
|
phrase_text = str(value).strip()
|
||||||
|
normalized = normalize_candidate_phrase(
|
||||||
|
phrase_text,
|
||||||
|
config,
|
||||||
|
max_tokens=config.phrase_max_entity_tokens,
|
||||||
|
)
|
||||||
|
if normalized is None:
|
||||||
|
continue
|
||||||
|
display_text, phrase_norm, token_count = normalized
|
||||||
|
out[phrase_norm] = PhraseCandidate(
|
||||||
|
phrase_text=display_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=token_count,
|
||||||
|
source_metadata=True,
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def merge_candidate_sources(*sources: Mapping[str, PhraseCandidate]) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Merge candidate dictionaries by normalized phrase.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
*sources (Mapping[str, PhraseCandidate]): Candidate maps to combine, keyed by normalized phrase.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: One merged candidate per normalized phrase.
|
||||||
|
"""
|
||||||
|
merged: dict[str, PhraseCandidate] = {}
|
||||||
|
for source in sources:
|
||||||
|
for phrase_norm, item in source.items():
|
||||||
|
existing = merged.setdefault(
|
||||||
|
phrase_norm,
|
||||||
|
PhraseCandidate(
|
||||||
|
phrase_text=item.phrase_text,
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
token_count=item.token_count,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
merge_candidate(existing, item)
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
def merge_candidate(existing: PhraseCandidate, item: PhraseCandidate) -> None:
|
||||||
|
"""Merge one candidate into an existing candidate object.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
existing (PhraseCandidate): Candidate mutated in place to absorb ``item``.
|
||||||
|
item (PhraseCandidate): Candidate whose sources, counts, and scores are merged in.
|
||||||
|
"""
|
||||||
|
existing.source_raw_ngram = existing.source_raw_ngram or item.source_raw_ngram
|
||||||
|
existing.source_yake = existing.source_yake or item.source_yake
|
||||||
|
existing.source_spacy_ner = existing.source_spacy_ner or item.source_spacy_ner
|
||||||
|
existing.source_spacy_noun_chunk = existing.source_spacy_noun_chunk or item.source_spacy_noun_chunk
|
||||||
|
existing.source_capitalized = existing.source_capitalized or item.source_capitalized
|
||||||
|
existing.source_metadata = existing.source_metadata or item.source_metadata
|
||||||
|
existing.raw_count += item.raw_count
|
||||||
|
existing.chapter_count = max(existing.chapter_count, item.chapter_count)
|
||||||
|
if item.yake_score is not None:
|
||||||
|
existing.yake_score = item.yake_score
|
||||||
|
if item.spacy_label:
|
||||||
|
existing.spacy_label = item.spacy_label
|
||||||
|
|
||||||
|
|
||||||
|
def enrich_with_frequency_and_chapter_counts(
|
||||||
|
candidates: Mapping[str, PhraseCandidate],
|
||||||
|
chapters: Sequence[str],
|
||||||
|
*,
|
||||||
|
counted_sizes: Iterable[int] = (),
|
||||||
|
) -> dict[str, PhraseCandidate]:
|
||||||
|
"""Add raw occurrence and chapter-spread counts to candidates.
|
||||||
|
|
||||||
|
Candidates whose ``token_count`` is in ``counted_sizes`` are left untouched: those counts
|
||||||
|
were already computed while sliding the chapters in :func:`extract_raw_ngrams_by_chapter`,
|
||||||
|
so re-sliding those n-gram sizes here would just duplicate that work.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidates (Mapping[str, PhraseCandidate]): Candidates to enrich, keyed by normalized phrase.
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks used to count occurrences and spread.
|
||||||
|
counted_sizes (Iterable[int]): Token counts whose counts are already populated and should be skipped.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, PhraseCandidate]: Candidates with updated ``raw_count`` and ``chapter_count`` values.
|
||||||
|
"""
|
||||||
|
if not candidates:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
already_counted = set(counted_sizes)
|
||||||
|
candidate_sets_by_size: dict[int, set[str]] = defaultdict(set)
|
||||||
|
for phrase_norm, candidate in candidates.items():
|
||||||
|
if candidate.token_count in already_counted:
|
||||||
|
continue
|
||||||
|
candidate_sets_by_size[candidate.token_count].add(phrase_norm)
|
||||||
|
|
||||||
|
enriched = dict(candidates)
|
||||||
|
if not candidate_sets_by_size:
|
||||||
|
return enriched
|
||||||
|
|
||||||
|
total_counts, chapter_counts = count_candidate_occurrences(candidate_sets_by_size, chapters)
|
||||||
|
for phrase_norm, candidate in enriched.items():
|
||||||
|
if candidate.token_count in already_counted:
|
||||||
|
continue
|
||||||
|
candidate.raw_count = max(candidate.raw_count, total_counts[phrase_norm])
|
||||||
|
candidate.chapter_count = chapter_counts[phrase_norm]
|
||||||
|
return enriched
|
||||||
|
|
||||||
|
|
||||||
|
def count_candidate_occurrences(
|
||||||
|
candidate_sets_by_size: Mapping[int, set[str]],
|
||||||
|
chapters: Sequence[str],
|
||||||
|
) -> tuple[dict[str, int], dict[str, int]]:
|
||||||
|
"""Count total occurrences and chapter spread for candidate phrases across chapters.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate_sets_by_size (Mapping[int, set[str]]): Candidate normalized phrases grouped by token count.
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks to slide n-gram windows over.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[dict[str, int], dict[str, int]]: Total occurrence counts and chapter-spread counts,
|
||||||
|
each keyed by normalized phrase.
|
||||||
|
"""
|
||||||
|
total_counts: defaultdict[str, int] = defaultdict(int)
|
||||||
|
chapter_counts: defaultdict[str, int] = defaultdict(int)
|
||||||
|
for chapter in chapters:
|
||||||
|
seen_in_chapter: set[str] = set()
|
||||||
|
chapter_tokens = tokenize(chapter)
|
||||||
|
for ngram_size, candidate_norms in candidate_sets_by_size.items():
|
||||||
|
for start in range(len(chapter_tokens) - ngram_size + 1):
|
||||||
|
phrase_norm = " ".join(chapter_tokens[start : start + ngram_size])
|
||||||
|
if phrase_norm not in candidate_norms:
|
||||||
|
continue
|
||||||
|
total_counts[phrase_norm] += 1
|
||||||
|
seen_in_chapter.add(phrase_norm)
|
||||||
|
for phrase_norm in seen_in_chapter:
|
||||||
|
chapter_counts[phrase_norm] += 1
|
||||||
|
return total_counts, chapter_counts
|
||||||
|
|
||||||
|
|
||||||
|
def filter_storable_candidates(
|
||||||
|
candidates: Mapping[str, PhraseCandidate],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> tuple[dict[str, PhraseCandidate], int, int, int, int]:
|
||||||
|
"""Remove candidates that should not be persisted.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidates (Mapping[str, PhraseCandidate]): Candidates to filter, keyed by normalized phrase.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[dict[str, PhraseCandidate], int, int, int, int]: The storable candidates followed by the
|
||||||
|
counts dropped for being too short, too rare, too common, and junk.
|
||||||
|
"""
|
||||||
|
min_raw_count = minimum_candidate_raw_count(config)
|
||||||
|
filtered: dict[str, PhraseCandidate] = {}
|
||||||
|
too_short = 0
|
||||||
|
too_rare = 0
|
||||||
|
too_common = 0
|
||||||
|
junk = 0
|
||||||
|
for phrase_norm, candidate in candidates.items():
|
||||||
|
if candidate.token_count < config.phrase_min_tokens:
|
||||||
|
too_short += 1
|
||||||
|
continue
|
||||||
|
if candidate.raw_count < min_raw_count:
|
||||||
|
too_rare += 1
|
||||||
|
continue
|
||||||
|
phrase_tokens = phrase_norm.split()
|
||||||
|
if is_most_common_word_phrase(phrase_tokens):
|
||||||
|
too_common += 1
|
||||||
|
continue
|
||||||
|
if is_junk_phrase(phrase_tokens):
|
||||||
|
junk += 1
|
||||||
|
continue
|
||||||
|
filtered[phrase_norm] = candidate
|
||||||
|
return filtered, too_short, too_rare, too_common, junk
|
||||||
|
|
||||||
|
|
||||||
|
def minimum_candidate_raw_count(config: EbookSearchConfig) -> int:
|
||||||
|
"""Return the minimum occurrence count required before storing a candidate.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: The minimum raw occurrence count, never less than 1.
|
||||||
|
"""
|
||||||
|
return max(config.phrase_raw_ngram_min_count, 1)
|
||||||
|
|
||||||
|
|
||||||
|
def is_most_common_word_phrase(phrase_tokens: list[str]) -> bool:
|
||||||
|
"""Return whether every token in a normalized phrase is a common word.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
phrase_tokens (list[str]): Normalized phrase tokens to inspect.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the phrase is non-empty and every token is a common word.
|
||||||
|
"""
|
||||||
|
common_words = get_most_common_words()
|
||||||
|
return bool(phrase_tokens) and all(token in common_words for token in phrase_tokens)
|
||||||
|
|
||||||
|
|
||||||
|
def is_junk_phrase(phrase_tokens: list[str]) -> bool:
|
||||||
|
"""Return whether a normalized phrase is lexical junk not worth LLM judging.
|
||||||
|
|
||||||
|
Judged data shows phrases containing a dialogue/action verb or a pronoun contraction are
|
||||||
|
never kept, and phrases whose tokens are mostly common words almost never are. Possessives
|
||||||
|
of proper nouns (``chapman's death``) pass because matching is by exact token, and
|
||||||
|
exactly-half-common bigrams (``data feed``) pass because the common-word rule is strict.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
phrase_tokens (list[str]): Normalized phrase tokens to inspect.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the phrase contains a junk token or is majority common words.
|
||||||
|
"""
|
||||||
|
if not phrase_tokens:
|
||||||
|
return False
|
||||||
|
junk_tokens = get_junk_tokens()
|
||||||
|
if any(token in junk_tokens for token in phrase_tokens):
|
||||||
|
return True
|
||||||
|
common_words = get_most_common_words()
|
||||||
|
half_phrase_len = len(phrase_tokens) // 2
|
||||||
|
return sum(token in common_words for token in phrase_tokens) > half_phrase_len
|
||||||
|
|
||||||
|
|
||||||
|
def score_candidate(candidate: PhraseCandidate, config: EbookSearchConfig) -> float:
|
||||||
|
"""Score a phrase candidate before LLM judging.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate to score.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
float: Combined score from sources, frequency, and length, less any penalties.
|
||||||
|
"""
|
||||||
|
score = source_score(candidate) + frequency_score(candidate, config) + token_count_score(candidate, config)
|
||||||
|
if non_raw_source_count(candidate) >= MULTI_SOURCE_MIN_SOURCES:
|
||||||
|
score += MULTI_SOURCE_SCORE_BONUS
|
||||||
|
if candidate.phrase_norm in get_ignored_phrases():
|
||||||
|
score -= 100.0
|
||||||
|
if has_bad_start(candidate.phrase_norm):
|
||||||
|
score -= BAD_START_SCORE_PENALTY
|
||||||
|
if has_bad_end(candidate.phrase_norm):
|
||||||
|
score -= BAD_END_SCORE_PENALTY
|
||||||
|
return score
|
||||||
|
|
||||||
|
|
||||||
|
def non_raw_source_count(candidate: PhraseCandidate) -> int:
|
||||||
|
"""Count the non-raw-ngram extraction sources that produced a candidate.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate whose enabled sources are counted.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of enabled sources other than the raw n-gram slide.
|
||||||
|
"""
|
||||||
|
return sum(
|
||||||
|
(
|
||||||
|
candidate.source_yake,
|
||||||
|
candidate.source_spacy_ner,
|
||||||
|
candidate.source_spacy_noun_chunk,
|
||||||
|
candidate.source_capitalized,
|
||||||
|
candidate.source_metadata,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def has_bad_start(phrase_norm: str) -> bool:
|
||||||
|
"""Return whether a normalized phrase starts with a bad starting token.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
phrase_norm (str): Normalized phrase text to inspect.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the first token is a known bad starting token.
|
||||||
|
"""
|
||||||
|
phrase_tokens = phrase_norm.split()
|
||||||
|
return bool(phrase_tokens and phrase_tokens[0] in get_bad_starts())
|
||||||
|
|
||||||
|
|
||||||
|
def has_bad_end(phrase_norm: str) -> bool:
|
||||||
|
"""Return whether a normalized phrase ends with a bad ending token.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
phrase_norm (str): Normalized phrase text to inspect.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the last token is a known bad ending token.
|
||||||
|
"""
|
||||||
|
phrase_tokens = phrase_norm.split()
|
||||||
|
return bool(phrase_tokens and phrase_tokens[-1] in get_bad_ends())
|
||||||
|
|
||||||
|
|
||||||
|
def source_score(candidate: PhraseCandidate) -> float:
|
||||||
|
"""Return the score contribution from extraction sources.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate whose enabled sources are weighted.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
float: Summed weight of the candidate's enabled extraction sources.
|
||||||
|
"""
|
||||||
|
return sum(
|
||||||
|
weight
|
||||||
|
for enabled, weight in (
|
||||||
|
(candidate.source_yake, 2.0),
|
||||||
|
(candidate.source_spacy_ner, 2.5),
|
||||||
|
(candidate.source_spacy_noun_chunk, 1.5),
|
||||||
|
(candidate.source_capitalized, 2.0),
|
||||||
|
(candidate.source_metadata, 2.0),
|
||||||
|
(candidate.source_raw_ngram, 0.5),
|
||||||
|
)
|
||||||
|
if enabled
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def frequency_score(candidate: PhraseCandidate, config: EbookSearchConfig) -> float:
|
||||||
|
"""Return the score contribution from frequency and chapter spread.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate whose counts are scored.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings holding score thresholds.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
float: Summed weight for each frequency and chapter-spread threshold the candidate meets.
|
||||||
|
"""
|
||||||
|
return sum(
|
||||||
|
weight
|
||||||
|
for count, threshold, weight in (
|
||||||
|
(candidate.raw_count, config.phrase_raw_count_score_threshold, 0.5),
|
||||||
|
(candidate.raw_count, config.phrase_raw_count_high_score_threshold, 0.5),
|
||||||
|
(candidate.chapter_count, config.phrase_chapter_count_score_threshold, 0.5),
|
||||||
|
(candidate.chapter_count, config.phrase_chapter_count_high_score_threshold, 0.5),
|
||||||
|
)
|
||||||
|
if count >= threshold
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def token_count_score(candidate: PhraseCandidate, config: EbookSearchConfig) -> float:
|
||||||
|
"""Return the score contribution from phrase length.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate whose token count is scored.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings holding the max token bound.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
float: Length-based score contribution, which may be negative for over- or under-length phrases.
|
||||||
|
"""
|
||||||
|
if candidate.token_count == 1:
|
||||||
|
return -0.5
|
||||||
|
if candidate.token_count in {2, 3, 4}:
|
||||||
|
return 0.5
|
||||||
|
if candidate.token_count > config.phrase_max_tokens:
|
||||||
|
return -1.0
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def get_sample_contexts(normalized_book_text: str, phrase_norm: str, max_contexts: int = 5) -> list[str]:
|
||||||
|
"""Return normalized context snippets containing a candidate phrase.
|
||||||
|
|
||||||
|
``normalized_book_text`` is expected to already be ``normalize_text``-ed by the caller
|
||||||
|
so the whole book is not re-normalized for every phrase.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
normalized_book_text (str): Whole book text, already normalized, to search.
|
||||||
|
phrase_norm (str): Normalized phrase to find contexts around.
|
||||||
|
max_contexts (int): Maximum number of context snippets to return.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[str]: Up to ``max_contexts`` normalized snippets surrounding the phrase.
|
||||||
|
"""
|
||||||
|
contexts: list[str] = []
|
||||||
|
start = 0
|
||||||
|
while len(contexts) < max_contexts:
|
||||||
|
index = normalized_book_text.find(phrase_norm, start)
|
||||||
|
if index == -1:
|
||||||
|
break
|
||||||
|
left = max(0, index - 300)
|
||||||
|
right = min(len(normalized_book_text), index + len(phrase_norm) + 300)
|
||||||
|
contexts.append(normalized_book_text[left:right])
|
||||||
|
start = index + len(phrase_norm)
|
||||||
|
return contexts
|
||||||
|
|
||||||
|
|
||||||
|
def candidate_source_names(candidate: PhraseCandidate) -> list[str]:
|
||||||
|
"""Return enabled source names for an extracted candidate.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate whose enabled sources are listed.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[str]: Names of the extraction sources that produced the candidate.
|
||||||
|
"""
|
||||||
|
names: list[str] = []
|
||||||
|
if candidate.source_raw_ngram:
|
||||||
|
names.append("raw_ngram")
|
||||||
|
if candidate.source_yake:
|
||||||
|
names.append("yake")
|
||||||
|
if candidate.source_spacy_ner:
|
||||||
|
names.append("spacy_ner")
|
||||||
|
if candidate.source_spacy_noun_chunk:
|
||||||
|
names.append("spacy_noun_chunk")
|
||||||
|
if candidate.source_capitalized:
|
||||||
|
names.append("capitalized")
|
||||||
|
if candidate.source_metadata:
|
||||||
|
names.append("metadata")
|
||||||
|
return names
|
||||||
|
|
||||||
|
|
||||||
|
def extract_phrase_candidates_for_book(
|
||||||
|
book_text: str,
|
||||||
|
chapters: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
nlp: SpacyLanguage | None = None,
|
||||||
|
metadata: Mapping[str, object] | None = None,
|
||||||
|
) -> list[PhraseCandidate]:
|
||||||
|
"""Extract, score, and limit phrase candidates for one book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_text (str): Full book text used for most extraction sources.
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks used for spaCy and frequency counts.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
nlp (SpacyLanguage | None): Optional spaCy pipeline for entity and noun-chunk sources.
|
||||||
|
metadata (Mapping[str, object] | None): Optional book metadata used as a candidate source.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[PhraseCandidate]: Scored candidates sorted best-first and capped per book.
|
||||||
|
"""
|
||||||
|
started_at = perf_counter()
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_start chapters=%s chars=%s min_tokens=%s max_tokens=%s max_candidates=%s",
|
||||||
|
len(chapters),
|
||||||
|
len(book_text),
|
||||||
|
config.phrase_min_tokens,
|
||||||
|
config.phrase_max_tokens,
|
||||||
|
config.protected_phrase_max_candidates_per_book,
|
||||||
|
)
|
||||||
|
raw_started_at = perf_counter()
|
||||||
|
raw = extract_raw_ngrams_by_chapter(chapters, config)
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_raw_complete candidates=%s duration_ms=%.1f",
|
||||||
|
len(raw),
|
||||||
|
(perf_counter() - raw_started_at) * 1000,
|
||||||
|
)
|
||||||
|
yake_started_at = perf_counter()
|
||||||
|
yake_candidates = extract_yake_candidates(book_text, config)
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_yake_complete candidates=%s duration_ms=%.1f",
|
||||||
|
len(yake_candidates),
|
||||||
|
(perf_counter() - yake_started_at) * 1000,
|
||||||
|
)
|
||||||
|
spacy_candidates: dict[str, PhraseCandidate] = {}
|
||||||
|
if nlp is not None:
|
||||||
|
spacy_started_at = perf_counter()
|
||||||
|
for chapter in chapters:
|
||||||
|
spacy_candidates = merge_candidate_sources(spacy_candidates, extract_spacy_candidates(chapter, nlp, config))
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_spacy_complete candidates=%s duration_ms=%.1f",
|
||||||
|
len(spacy_candidates),
|
||||||
|
(perf_counter() - spacy_started_at) * 1000,
|
||||||
|
)
|
||||||
|
capitalized_started_at = perf_counter()
|
||||||
|
capitalized = extract_capitalized_phrases(book_text, config)
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_capitalized_complete candidates=%s duration_ms=%.1f",
|
||||||
|
len(capitalized),
|
||||||
|
(perf_counter() - capitalized_started_at) * 1000,
|
||||||
|
)
|
||||||
|
metadata_candidates = extract_metadata_candidates(metadata, config)
|
||||||
|
|
||||||
|
candidates = merge_candidate_sources(raw, yake_candidates, spacy_candidates, capitalized, metadata_candidates)
|
||||||
|
enriched_started_at = perf_counter()
|
||||||
|
# Raw n-gram sizes were already counted per chapter above, so only enrich the remaining
|
||||||
|
# (entity-length) sizes here instead of re-sliding every size over the whole book.
|
||||||
|
candidates = enrich_with_frequency_and_chapter_counts(
|
||||||
|
candidates,
|
||||||
|
chapters,
|
||||||
|
counted_sizes=range(config.phrase_min_tokens, config.phrase_max_tokens + 1),
|
||||||
|
)
|
||||||
|
pre_filter_count = len(candidates)
|
||||||
|
candidates, filtered_too_short, filtered_too_rare, filtered_too_common, filtered_junk = filter_storable_candidates(
|
||||||
|
candidates, config
|
||||||
|
)
|
||||||
|
for candidate in candidates.values():
|
||||||
|
candidate.candidate_score = score_candidate(candidate, config)
|
||||||
|
|
||||||
|
limited = sorted(candidates.values(), key=lambda item: item.candidate_score, reverse=True)[
|
||||||
|
: config.protected_phrase_max_candidates_per_book
|
||||||
|
]
|
||||||
|
logger.info(
|
||||||
|
"ebook_phrase_candidate_extract_complete raw=%s yake=%s spacy=%s capitalized=%s metadata=%s "
|
||||||
|
"merged=%s filtered_too_short=%s filtered_too_rare=%s filtered_too_common=%s filtered_junk=%s "
|
||||||
|
"min_uses=%s storable=%s limited=%s enrich_score_ms=%.1f duration_ms=%.1f",
|
||||||
|
len(raw),
|
||||||
|
len(yake_candidates),
|
||||||
|
len(spacy_candidates),
|
||||||
|
len(capitalized),
|
||||||
|
len(metadata_candidates),
|
||||||
|
pre_filter_count,
|
||||||
|
filtered_too_short,
|
||||||
|
filtered_too_rare,
|
||||||
|
filtered_too_common,
|
||||||
|
filtered_junk,
|
||||||
|
minimum_candidate_raw_count(config),
|
||||||
|
len(candidates),
|
||||||
|
len(limited),
|
||||||
|
(perf_counter() - enriched_started_at) * 1000,
|
||||||
|
(perf_counter() - started_at) * 1000,
|
||||||
|
)
|
||||||
|
return limited
|
||||||
@@ -0,0 +1,370 @@
|
|||||||
|
"""Book-level orchestration for candidate n-gram generation and recalculation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
from collections import deque
|
||||||
|
from time import perf_counter
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from sqlalchemy import select
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.extraction import extract_phrase_candidates_for_book
|
||||||
|
from python.ebook_search.protected_phrases.models import (
|
||||||
|
BookCandidateResult,
|
||||||
|
PhraseCandidateGenerationResult,
|
||||||
|
PhraseRecalculationResult,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.pool import extract_phrase_candidates_in_pool, get_extraction_pool
|
||||||
|
from python.ebook_search.protected_phrases.store import (
|
||||||
|
bulk_upsert_unjudged_candidates,
|
||||||
|
delete_phrase_data_for_book,
|
||||||
|
load_book_chapter_texts,
|
||||||
|
metadata_for_source,
|
||||||
|
new_candidate_row,
|
||||||
|
prune_unstorable_unjudged_candidate_phrases,
|
||||||
|
)
|
||||||
|
from python.orm.richie import EbookCandidatePhrase, EbookSource
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Mapping, Sequence
|
||||||
|
from concurrent.futures import Future
|
||||||
|
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.protected_phrases.extraction import SpacyLanguage
|
||||||
|
from python.ebook_search.protected_phrases.models import PhraseCandidate
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def generate_candidate_phrases_for_books(
|
||||||
|
session: AsyncSession,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
only_missing: bool = False,
|
||||||
|
) -> PhraseCandidateGenerationResult:
|
||||||
|
"""Create or refresh candidate phrases for indexed books without calling the LLM judge.
|
||||||
|
|
||||||
|
Extraction always runs concurrently in the shared process pool so a full backfill uses
|
||||||
|
multiple cores.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
only_missing (bool): When True, only generate for books that have no candidate phrases
|
||||||
|
yet instead of refreshing every book.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseCandidateGenerationResult: Per-corpus counts of books seen, built, and candidates stored.
|
||||||
|
"""
|
||||||
|
source_query = select(EbookSource).order_by(EbookSource.id)
|
||||||
|
if only_missing:
|
||||||
|
has_candidates = select(EbookCandidatePhrase.id).where(EbookCandidatePhrase.book_id == EbookSource.id)
|
||||||
|
source_query = source_query.where(~has_candidates.exists())
|
||||||
|
sources = (await session.scalars(source_query)).all()
|
||||||
|
books_seen = len(sources)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_generation_start books_seen=%s min_tokens=%s max_tokens=%s max_candidates_per_book=%s",
|
||||||
|
books_seen,
|
||||||
|
config.phrase_min_tokens,
|
||||||
|
config.phrase_max_tokens,
|
||||||
|
config.protected_phrase_max_candidates_per_book,
|
||||||
|
)
|
||||||
|
|
||||||
|
outcomes = await generate_candidates_for_sources_pooled(session, sources, config)
|
||||||
|
|
||||||
|
result = PhraseCandidateGenerationResult(
|
||||||
|
books_seen=books_seen,
|
||||||
|
books_built=sum(1 for outcome in outcomes if outcome.built),
|
||||||
|
candidate_phrases=sum(outcome.candidates for outcome in outcomes),
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_generation_complete books_seen=%s books_built=%s candidate_total=%s",
|
||||||
|
result.books_seen,
|
||||||
|
result.books_built,
|
||||||
|
result.candidate_phrases,
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def generate_candidates_for_sources_pooled(
|
||||||
|
session: AsyncSession,
|
||||||
|
sources: Sequence[EbookSource],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> list[BookCandidateResult]:
|
||||||
|
"""Generate candidate phrases for many books, extracting them concurrently in worker processes.
|
||||||
|
|
||||||
|
Chapter loading and row persistence stay on the caller's session (serial), while the CPU-bound
|
||||||
|
extraction runs in the shared process pool. A bounded window of in-flight books overlaps
|
||||||
|
extraction across cores without loading every book's candidates into memory at once.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
sources (Sequence[EbookSource]): Indexed books to generate candidates for.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[BookCandidateResult]: One result per book.
|
||||||
|
"""
|
||||||
|
pool = get_extraction_pool(config.protected_phrase_extraction_workers)
|
||||||
|
max_in_flight = max(1, config.protected_phrase_extraction_workers) * 2
|
||||||
|
pending: deque[tuple[EbookSource, Future[list[PhraseCandidate]]]] = deque()
|
||||||
|
outcomes: list[BookCandidateResult] = []
|
||||||
|
|
||||||
|
async def drain_one() -> None:
|
||||||
|
source, future = pending.popleft()
|
||||||
|
extracted = await asyncio.wrap_future(future)
|
||||||
|
outcomes.append(await store_source_candidates(session, source, extracted, config))
|
||||||
|
|
||||||
|
try:
|
||||||
|
for source in sources:
|
||||||
|
chapters = await load_book_chapter_texts(session, source.id)
|
||||||
|
if not chapters:
|
||||||
|
logger.warning("ebook_candidate_phrase_generation_book_empty source_id=%s", source.id)
|
||||||
|
outcomes.append(BookCandidateResult())
|
||||||
|
continue
|
||||||
|
future = pool.submit(
|
||||||
|
extract_phrase_candidates_for_book,
|
||||||
|
"\n\n".join(chapters),
|
||||||
|
chapters,
|
||||||
|
config,
|
||||||
|
metadata=metadata_for_source(source),
|
||||||
|
)
|
||||||
|
pending.append((source, future))
|
||||||
|
if len(pending) >= max_in_flight:
|
||||||
|
await drain_one()
|
||||||
|
while pending:
|
||||||
|
await drain_one()
|
||||||
|
except Exception:
|
||||||
|
for _, future in pending:
|
||||||
|
future.cancel()
|
||||||
|
await session.rollback()
|
||||||
|
logger.exception("ebook_candidate_phrase_generation_pooled_failed")
|
||||||
|
raise
|
||||||
|
return outcomes
|
||||||
|
|
||||||
|
|
||||||
|
async def store_source_candidates(
|
||||||
|
session: AsyncSession,
|
||||||
|
source: EbookSource,
|
||||||
|
limited_candidates: list[PhraseCandidate],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> BookCandidateResult:
|
||||||
|
"""Persist and commit one book's already-extracted candidates.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
source (EbookSource): Book the candidates belong to.
|
||||||
|
limited_candidates (list[PhraseCandidate]): Scored candidates to persist.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
BookCandidateResult: Candidate count and that the book was committed.
|
||||||
|
"""
|
||||||
|
book_started_at = perf_counter()
|
||||||
|
saved_count = await store_candidate_phrases_for_book(session, source.id, None, limited_candidates, config)
|
||||||
|
await session.commit()
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_generation_book_committed source_id=%s candidates=%s duration_ms=%.1f",
|
||||||
|
source.id,
|
||||||
|
saved_count,
|
||||||
|
(perf_counter() - book_started_at) * 1000,
|
||||||
|
)
|
||||||
|
return BookCandidateResult(candidates=saved_count, built=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def recalculate_candidate_phrases_for_book(
|
||||||
|
session: AsyncSession,
|
||||||
|
source: EbookSource,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
nlp: SpacyLanguage | None = None,
|
||||||
|
use_process_pool: bool = False,
|
||||||
|
) -> PhraseRecalculationResult:
|
||||||
|
"""Remove all book phrase data, regenerate candidates, and commit the completed book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
source (EbookSource): Indexed book to recalculate.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
nlp (SpacyLanguage | None): Optional spaCy pipeline for entity and noun-chunk sources.
|
||||||
|
use_process_pool (bool): Run the CPU-bound extraction in a worker process so concurrent
|
||||||
|
recalculations do not serialize behind the GIL. Defaults to in-process for callers
|
||||||
|
(tests, backfills) that do not need it.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseRecalculationResult: Deleted-row counts and the number of candidates regenerated.
|
||||||
|
"""
|
||||||
|
started_at = perf_counter()
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_recalculation_start source_id=%s title=%r",
|
||||||
|
source.id,
|
||||||
|
source.title,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
deleted = await delete_phrase_data_for_book(session, source.id)
|
||||||
|
chapters = await load_book_chapter_texts(session, source.id)
|
||||||
|
if not chapters:
|
||||||
|
logger.warning("ebook_candidate_phrase_recalculation_book_empty source_id=%s", source.id)
|
||||||
|
await session.commit()
|
||||||
|
return PhraseRecalculationResult(
|
||||||
|
book_id=source.id,
|
||||||
|
deleted_candidates=deleted.deleted_candidates,
|
||||||
|
deleted_protected_phrases=deleted.deleted_protected_phrases,
|
||||||
|
deleted_aliases=deleted.deleted_aliases,
|
||||||
|
deleted_mentions=deleted.deleted_mentions,
|
||||||
|
candidate_phrases=0,
|
||||||
|
)
|
||||||
|
|
||||||
|
candidate_count = await generate_candidate_phrases_for_book(
|
||||||
|
session,
|
||||||
|
source.id,
|
||||||
|
series_id=None,
|
||||||
|
chapters=chapters,
|
||||||
|
config=config,
|
||||||
|
nlp=nlp,
|
||||||
|
metadata=metadata_for_source(source),
|
||||||
|
replace_all=True,
|
||||||
|
use_process_pool=use_process_pool,
|
||||||
|
)
|
||||||
|
await session.commit()
|
||||||
|
except Exception:
|
||||||
|
await session.rollback()
|
||||||
|
logger.exception("ebook_candidate_phrase_recalculation_failed source_id=%s", source.id)
|
||||||
|
raise
|
||||||
|
|
||||||
|
result = PhraseRecalculationResult(
|
||||||
|
book_id=source.id,
|
||||||
|
deleted_candidates=deleted.deleted_candidates,
|
||||||
|
deleted_protected_phrases=deleted.deleted_protected_phrases,
|
||||||
|
deleted_aliases=deleted.deleted_aliases,
|
||||||
|
deleted_mentions=deleted.deleted_mentions,
|
||||||
|
candidate_phrases=candidate_count,
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_recalculation_complete source_id=%s deleted_candidates=%s "
|
||||||
|
"deleted_protected=%s deleted_aliases=%s deleted_mentions=%s candidates=%s duration_ms=%.1f",
|
||||||
|
source.id,
|
||||||
|
result.deleted_candidates,
|
||||||
|
result.deleted_protected_phrases,
|
||||||
|
result.deleted_aliases,
|
||||||
|
result.deleted_mentions,
|
||||||
|
result.candidate_phrases,
|
||||||
|
(perf_counter() - started_at) * 1000,
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def generate_candidate_phrases_for_book(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
chapters: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
nlp: SpacyLanguage | None = None,
|
||||||
|
metadata: Mapping[str, object] | None = None,
|
||||||
|
replace_all: bool = False,
|
||||||
|
use_process_pool: bool = False,
|
||||||
|
) -> int:
|
||||||
|
"""Extract and store candidate phrases for one book without LLM judging.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
book_id (int): Book the candidates belong to.
|
||||||
|
series_id (int | None): Series scope for the stored candidates.
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks used for extraction and frequency counts.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
nlp (SpacyLanguage | None): Optional spaCy pipeline for entity and noun-chunk sources.
|
||||||
|
metadata (Mapping[str, object] | None): Optional book metadata used as a candidate source.
|
||||||
|
replace_all (bool): When the caller has already cleared this book's candidates (e.g. a
|
||||||
|
recalculation), skip the per-candidate existence lookup and bulk-insert new rows.
|
||||||
|
use_process_pool (bool): Run the CPU-bound extraction in a worker process to avoid
|
||||||
|
serializing concurrent requests behind the GIL. Ignored when ``nlp`` is set, since
|
||||||
|
the spaCy pipeline cannot be sent to a worker process.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of candidate phrase rows stored.
|
||||||
|
"""
|
||||||
|
started_at = perf_counter()
|
||||||
|
book_text = "\n\n".join(chapters)
|
||||||
|
if use_process_pool and nlp is None:
|
||||||
|
limited_candidates = await extract_phrase_candidates_in_pool(book_text, chapters, config, metadata=metadata)
|
||||||
|
else:
|
||||||
|
limited_candidates = extract_phrase_candidates_for_book(
|
||||||
|
book_text,
|
||||||
|
chapters,
|
||||||
|
config,
|
||||||
|
nlp=nlp,
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
saved_count = await store_candidate_phrases_for_book(
|
||||||
|
session,
|
||||||
|
book_id,
|
||||||
|
series_id,
|
||||||
|
limited_candidates,
|
||||||
|
config,
|
||||||
|
replace_all=replace_all,
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_generation_book_duration book_id=%s candidates=%s duration_ms=%.1f",
|
||||||
|
book_id,
|
||||||
|
saved_count,
|
||||||
|
(perf_counter() - started_at) * 1000,
|
||||||
|
)
|
||||||
|
return saved_count
|
||||||
|
|
||||||
|
|
||||||
|
async def store_candidate_phrases_for_book(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
limited_candidates: list[PhraseCandidate],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
replace_all: bool = False,
|
||||||
|
) -> int:
|
||||||
|
"""Persist already-extracted candidate phrase rows for one book without committing.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
book_id (int): Book the candidates belong to.
|
||||||
|
series_id (int | None): Series scope for the stored candidates.
|
||||||
|
limited_candidates (list[PhraseCandidate]): Scored candidates to persist.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
replace_all (bool): When the caller has already cleared this book's candidates, skip the
|
||||||
|
per-candidate existence lookup and bulk-insert new rows.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of candidate phrase rows stored.
|
||||||
|
"""
|
||||||
|
save_started_at = perf_counter()
|
||||||
|
if replace_all:
|
||||||
|
rows = [new_candidate_row(book_id, series_id, candidate) for candidate in limited_candidates]
|
||||||
|
session.add_all(rows)
|
||||||
|
await session.flush()
|
||||||
|
saved_count = len(rows)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_save_start book_id=%s candidates=%s mode=bulk_insert",
|
||||||
|
book_id,
|
||||||
|
len(limited_candidates),
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
pruned_count = await prune_unstorable_unjudged_candidate_phrases(session, book_id, config)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_save_start book_id=%s candidates=%s pruned_unstorable=%s",
|
||||||
|
book_id,
|
||||||
|
len(limited_candidates),
|
||||||
|
pruned_count,
|
||||||
|
)
|
||||||
|
saved_count = await bulk_upsert_unjudged_candidates(session, book_id, series_id, limited_candidates)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_save_complete book_id=%s candidates=%s save_ms=%.1f",
|
||||||
|
book_id,
|
||||||
|
saved_count,
|
||||||
|
(perf_counter() - save_started_at) * 1000,
|
||||||
|
)
|
||||||
|
return saved_count
|
||||||
@@ -0,0 +1,511 @@
|
|||||||
|
"""Book-level orchestration for LLM judging and promotion of candidate phrases."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from time import perf_counter
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.llm_interface import request_chat_completion
|
||||||
|
from python.ebook_search.protected_phrases.extraction import (
|
||||||
|
candidate_source_names,
|
||||||
|
get_sample_contexts,
|
||||||
|
is_junk_phrase,
|
||||||
|
is_most_common_word_phrase,
|
||||||
|
score_candidate,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.matching import index_chunk_phrase_mentions_for_book
|
||||||
|
from python.ebook_search.protected_phrases.models import BookJudgmentResult, LLMJudgment, PhraseJudgmentBackfillResult
|
||||||
|
from python.ebook_search.protected_phrases.store import (
|
||||||
|
count_protected_phrases,
|
||||||
|
count_unjudged_candidates,
|
||||||
|
load_book_text,
|
||||||
|
load_candidates_for_judgment,
|
||||||
|
phrase_candidate_from_row,
|
||||||
|
save_candidate_to_db,
|
||||||
|
upsert_protected_phrase,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import normalize_text
|
||||||
|
from python.orm.richie import EbookSource
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncEngine
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.protected_phrases.models import PhraseCandidate
|
||||||
|
from python.orm.richie import EbookProtectedPhrase
|
||||||
|
|
||||||
|
JSON_OBJECT_RE = re.compile(r"\{.*\}", re.DOTALL)
|
||||||
|
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def judge_candidate_phrases_for_books(
|
||||||
|
engine: AsyncEngine,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
source_ids: Sequence[int] | None = None,
|
||||||
|
) -> PhraseJudgmentBackfillResult:
|
||||||
|
"""Judge candidate phrases for books, fanning LLM calls out across books and phrases.
|
||||||
|
|
||||||
|
Up to ``phrase_judge_book_workers`` books are judged at once, and within each book candidates
|
||||||
|
are judged in concurrent chunks of ``phrase_judge_phrase_workers``. Each book uses its own
|
||||||
|
short-lived sessions for reads and writes; no database connection is held while LLM calls are
|
||||||
|
in flight. For a pseudo-single-threaded run (solo testing, debugging), set both worker
|
||||||
|
settings to 1.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
engine (AsyncEngine): Engine used to open one session per book.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings and chat configuration.
|
||||||
|
source_ids (Sequence[int] | None): Books to judge; ``None`` judges every indexed book.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseJudgmentBackfillResult: Per-corpus counts of books judged, failures, candidates,
|
||||||
|
protected phrases, and mentions.
|
||||||
|
"""
|
||||||
|
if source_ids is None:
|
||||||
|
async with AsyncSession(engine) as session:
|
||||||
|
source_ids = list((await session.scalars(select(EbookSource.id).order_by(EbookSource.id))).all())
|
||||||
|
books_seen = len(source_ids)
|
||||||
|
book_workers = max(1, config.phrase_judge_book_workers)
|
||||||
|
phrase_workers = max(1, config.phrase_judge_phrase_workers)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_start books_seen=%s book_workers=%s phrase_workers=%s "
|
||||||
|
"confidence_threshold=%.2f",
|
||||||
|
books_seen,
|
||||||
|
book_workers,
|
||||||
|
phrase_workers,
|
||||||
|
config.protected_phrase_confidence_threshold,
|
||||||
|
)
|
||||||
|
|
||||||
|
book_semaphore = asyncio.Semaphore(book_workers)
|
||||||
|
max_connections = book_workers * phrase_workers
|
||||||
|
limits = httpx.Limits(max_connections=max_connections, max_keepalive_connections=max_connections)
|
||||||
|
async with httpx.AsyncClient(limits=limits) as client:
|
||||||
|
outcomes = await asyncio.gather(
|
||||||
|
*(judge_one_book_async(engine, source_id, config, client, book_semaphore) for source_id in source_ids)
|
||||||
|
)
|
||||||
|
|
||||||
|
result = PhraseJudgmentBackfillResult(
|
||||||
|
books_seen=books_seen,
|
||||||
|
books_judged=sum(1 for outcome in outcomes if outcome.committed),
|
||||||
|
books_failed=sum(1 for outcome in outcomes if outcome.failed),
|
||||||
|
candidates_judged=sum(outcome.judged for outcome in outcomes),
|
||||||
|
protected_phrases=sum(outcome.protected for outcome in outcomes),
|
||||||
|
phrase_mentions=sum(outcome.mentions for outcome in outcomes),
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_complete books_seen=%s books_judged=%s books_failed=%s "
|
||||||
|
"candidates_judged=%s protected=%s mentions=%s",
|
||||||
|
result.books_seen,
|
||||||
|
result.books_judged,
|
||||||
|
result.books_failed,
|
||||||
|
result.candidates_judged,
|
||||||
|
result.protected_phrases,
|
||||||
|
result.phrase_mentions,
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def judge_one_book_async(
|
||||||
|
engine: AsyncEngine,
|
||||||
|
source_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
book_semaphore: asyncio.Semaphore,
|
||||||
|
) -> BookJudgmentResult:
|
||||||
|
"""Judge one book concurrently and persist the outcome, honoring the book-level limit.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
engine (AsyncEngine): Engine used to open the book's read and write sessions.
|
||||||
|
source_id (int): Book to judge candidates for.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
client (httpx.AsyncClient): Shared async client for LLM calls.
|
||||||
|
book_semaphore (asyncio.Semaphore): Caps how many books judge at once.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
BookJudgmentResult: The book's judgment outcome.
|
||||||
|
"""
|
||||||
|
async with book_semaphore:
|
||||||
|
try:
|
||||||
|
prepared = await prepare_book_judgment(engine, source_id, config)
|
||||||
|
if prepared is None:
|
||||||
|
return BookJudgmentResult()
|
||||||
|
work_items, target_remaining = prepared
|
||||||
|
judged = await judge_book_candidates_async(client, config, source_id, work_items, target_remaining)
|
||||||
|
if not judged:
|
||||||
|
return BookJudgmentResult()
|
||||||
|
return await persist_book_judgments(engine, source_id, config, judged)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("ebook_candidate_phrase_judgment_book_failed source_id=%s", source_id)
|
||||||
|
return BookJudgmentResult(failed=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def prepare_book_judgment(
|
||||||
|
engine: AsyncEngine,
|
||||||
|
source_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> tuple[list[tuple[int, PhraseCandidate]], int | None] | None:
|
||||||
|
"""Load one book's candidates to judge, with sample contexts, on a short-lived read session.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
engine (AsyncEngine): Engine used to open the read session.
|
||||||
|
source_id (int): Book to load candidates for.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[list[tuple[int, PhraseCandidate]], int | None] | None: Candidate rows paired with
|
||||||
|
in-memory candidates and the remaining protected-phrase target, or ``None`` when the book
|
||||||
|
has nothing to judge.
|
||||||
|
"""
|
||||||
|
judgment_limit = config.protected_phrase_llm_candidates_per_book
|
||||||
|
if judgment_limit <= 0:
|
||||||
|
return None
|
||||||
|
async with AsyncSession(engine) as session:
|
||||||
|
if not await count_unjudged_candidates(session, source_id, config):
|
||||||
|
logger.info("ebook_candidate_phrase_judgment_book_skip_no_unjudged source_id=%s", source_id)
|
||||||
|
return None
|
||||||
|
existing_protected = await count_protected_phrases(session, source_id)
|
||||||
|
target_remaining: int | None = None
|
||||||
|
if config.phrase_target_protected_per_book > 0:
|
||||||
|
target_remaining = max(config.phrase_target_protected_per_book - existing_protected, 0)
|
||||||
|
if target_remaining == 0:
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_skipped_target_met source_id=%s existing_protected=%s target=%s",
|
||||||
|
source_id,
|
||||||
|
existing_protected,
|
||||||
|
config.phrase_target_protected_per_book,
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
book_text = await load_book_text(session, source_id)
|
||||||
|
if not book_text:
|
||||||
|
logger.warning("ebook_candidate_phrase_judgment_book_empty source_id=%s", source_id)
|
||||||
|
return None
|
||||||
|
normalized_book_text = normalize_text(book_text)
|
||||||
|
# Stored rows may predate the current junk filters and score weights, so re-filter and
|
||||||
|
# rescore every unjudged row here instead of trusting the persisted candidate_score.
|
||||||
|
rows = await load_candidates_for_judgment(session, source_id, config)
|
||||||
|
scored_items: list[tuple[int, PhraseCandidate]] = []
|
||||||
|
skipped_junk = 0
|
||||||
|
for row in rows:
|
||||||
|
candidate = phrase_candidate_from_row(row)
|
||||||
|
if is_junk_phrase(candidate.phrase_norm.split()):
|
||||||
|
skipped_junk += 1
|
||||||
|
continue
|
||||||
|
candidate.candidate_score = score_candidate(candidate, config)
|
||||||
|
scored_items.append((row.id, candidate))
|
||||||
|
scored_items.sort(key=lambda item: item[1].candidate_score, reverse=True)
|
||||||
|
work_items = scored_items[:judgment_limit]
|
||||||
|
for _, candidate in work_items:
|
||||||
|
candidate.sample_contexts = candidate.sample_contexts or get_sample_contexts(
|
||||||
|
normalized_book_text, candidate.phrase_norm
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_candidates_loaded source_id=%s candidates=%s skipped_junk=%s "
|
||||||
|
"unjudged_rows=%s existing_protected=%s target_remaining=%s judgment_limit=%s",
|
||||||
|
source_id,
|
||||||
|
len(work_items),
|
||||||
|
skipped_junk,
|
||||||
|
len(rows),
|
||||||
|
existing_protected,
|
||||||
|
target_remaining,
|
||||||
|
judgment_limit,
|
||||||
|
)
|
||||||
|
return work_items, target_remaining
|
||||||
|
|
||||||
|
|
||||||
|
async def judge_book_candidates_async(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
source_id: int,
|
||||||
|
work_items: list[tuple[int, PhraseCandidate]],
|
||||||
|
target_remaining: int | None,
|
||||||
|
) -> list[tuple[int, PhraseCandidate, LLMJudgment, bool]]:
|
||||||
|
"""Judge a book's candidates in concurrent chunks, stopping once the target is reached.
|
||||||
|
|
||||||
|
Promotion decisions are made in memory so judging can stop early without any database writes.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
client (httpx.AsyncClient): Shared async client for LLM calls.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
source_id (int): Book being judged, for logging.
|
||||||
|
work_items (list[tuple[int, PhraseCandidate]]): Candidate row ids paired with candidates,
|
||||||
|
in best-first score order.
|
||||||
|
target_remaining (int | None): Remaining protected-phrase target, or ``None`` for no cap.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[tuple[int, PhraseCandidate, LLMJudgment, bool]]: Judged rows with their judgment and
|
||||||
|
whether each should be promoted.
|
||||||
|
"""
|
||||||
|
chunk_size = max(1, config.phrase_judge_phrase_workers)
|
||||||
|
judged: list[tuple[int, PhraseCandidate, LLMJudgment, bool]] = []
|
||||||
|
promoted = 0
|
||||||
|
for start in range(0, len(work_items), chunk_size):
|
||||||
|
chunk = work_items[start : start + chunk_size]
|
||||||
|
judgments = await asyncio.gather(*(judge_candidate_async(client, config, candidate) for _, candidate in chunk))
|
||||||
|
for (candidate_id, candidate), judgment in zip(chunk, judgments, strict=True):
|
||||||
|
promote = (target_remaining is None or promoted < target_remaining) and should_protect_judged_candidate(
|
||||||
|
candidate, judgment, source_id, config, candidate_id=candidate_id
|
||||||
|
)
|
||||||
|
if promote:
|
||||||
|
promoted += 1
|
||||||
|
judged.append((candidate_id, candidate, judgment, promote))
|
||||||
|
if target_remaining is not None and promoted >= target_remaining:
|
||||||
|
break
|
||||||
|
return judged
|
||||||
|
|
||||||
|
|
||||||
|
async def judge_candidate_async(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
candidate: PhraseCandidate,
|
||||||
|
) -> LLMJudgment:
|
||||||
|
"""Judge one candidate with the LLM over the shared async client.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
client (httpx.AsyncClient): Shared async client for LLM calls.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
candidate (PhraseCandidate): Candidate to judge.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
LLMJudgment: The parsed judgment.
|
||||||
|
"""
|
||||||
|
content = await request_chat_completion(client, config, build_judge_messages(candidate))
|
||||||
|
return parse_llm_judgment(content, config)
|
||||||
|
|
||||||
|
|
||||||
|
async def persist_book_judgments(
|
||||||
|
engine: AsyncEngine,
|
||||||
|
source_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
judged: list[tuple[int, PhraseCandidate, LLMJudgment, bool]],
|
||||||
|
) -> BookJudgmentResult:
|
||||||
|
"""Persist one book's judgments and promotions in a single committed transaction.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
engine (AsyncEngine): Engine used to open the write session.
|
||||||
|
source_id (int): Book being persisted.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
judged (list[tuple[int, PhraseCandidate, LLMJudgment, bool]]): Judged candidates with their
|
||||||
|
judgment and promotion flag.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
BookJudgmentResult: The book's committed counts, or a failed result on error.
|
||||||
|
"""
|
||||||
|
book_started_at = perf_counter()
|
||||||
|
async with AsyncSession(engine, expire_on_commit=False) as session:
|
||||||
|
try:
|
||||||
|
protected: list[EbookProtectedPhrase] = []
|
||||||
|
for candidate_id, candidate, judgment, promote in judged:
|
||||||
|
candidate_row = await save_candidate_to_db(session, source_id, None, candidate, judgment=judgment)
|
||||||
|
if promote:
|
||||||
|
protected.append(
|
||||||
|
await upsert_protected_phrase(session, source_id, None, candidate, judgment, candidate_row)
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_candidate_complete source_id=%s candidate_id=%s phrase=%r "
|
||||||
|
"keep=%s confidence=%.3f category=%r promoted=%s",
|
||||||
|
source_id,
|
||||||
|
candidate_id,
|
||||||
|
candidate.phrase_norm,
|
||||||
|
judgment.keep,
|
||||||
|
judgment.confidence,
|
||||||
|
judgment.category,
|
||||||
|
promote,
|
||||||
|
)
|
||||||
|
await session.flush()
|
||||||
|
mentions = await index_chunk_phrase_mentions_for_book(session, source_id, config) if protected else 0
|
||||||
|
await session.commit()
|
||||||
|
except Exception:
|
||||||
|
await session.rollback()
|
||||||
|
logger.exception("ebook_candidate_phrase_judgment_book_persist_failed source_id=%s", source_id)
|
||||||
|
return BookJudgmentResult(failed=True)
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_book_committed source_id=%s judged=%s protected=%s mentions=%s "
|
||||||
|
"duration_ms=%.1f",
|
||||||
|
source_id,
|
||||||
|
len(judged),
|
||||||
|
len(protected),
|
||||||
|
mentions,
|
||||||
|
(perf_counter() - book_started_at) * 1000,
|
||||||
|
)
|
||||||
|
return BookJudgmentResult(judged=len(judged), protected=len(protected), mentions=mentions, committed=True)
|
||||||
|
|
||||||
|
|
||||||
|
def should_protect_judged_candidate(
|
||||||
|
candidate: PhraseCandidate,
|
||||||
|
judgment: LLMJudgment,
|
||||||
|
book_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
candidate_id: int,
|
||||||
|
) -> bool:
|
||||||
|
"""Report whether a judged candidate qualifies to become a protected phrase.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): In-memory candidate that was judged.
|
||||||
|
judgment (LLMJudgment): Judge decision for the candidate.
|
||||||
|
book_id (int): Book the candidate belongs to, for logging.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
candidate_id (int): Stored candidate row id the judgment came from, for logging.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the judged candidate should be promoted to a protected phrase.
|
||||||
|
"""
|
||||||
|
if not judgment.keep or judgment.confidence < config.protected_phrase_confidence_threshold:
|
||||||
|
return False
|
||||||
|
accepted_norm = normalize_text(judgment.canonical or candidate.phrase_text)
|
||||||
|
accepted_tokens = accepted_norm.split()
|
||||||
|
accepted_token_count = len(accepted_tokens)
|
||||||
|
if accepted_token_count < config.phrase_min_tokens:
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_candidate_skip_short_canonical book_id=%s candidate_id=%s "
|
||||||
|
"phrase=%r canonical=%r token_count=%s min_tokens=%s",
|
||||||
|
book_id,
|
||||||
|
candidate_id,
|
||||||
|
candidate.phrase_norm,
|
||||||
|
accepted_norm,
|
||||||
|
accepted_token_count,
|
||||||
|
config.phrase_min_tokens,
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
if is_most_common_word_phrase(accepted_tokens):
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_judgment_candidate_skip_common_canonical book_id=%s candidate_id=%s "
|
||||||
|
"phrase=%r canonical=%r",
|
||||||
|
book_id,
|
||||||
|
candidate_id,
|
||||||
|
candidate.phrase_norm,
|
||||||
|
accepted_norm,
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def build_judge_messages(candidate: PhraseCandidate) -> list[dict[str, str]]:
|
||||||
|
"""Build the chat messages used to judge one candidate phrase.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (PhraseCandidate): Candidate to describe for the judge.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[dict[str, str]]: OpenAI-style system and user messages.
|
||||||
|
"""
|
||||||
|
payload = {
|
||||||
|
"phrase": candidate.phrase_norm,
|
||||||
|
"token_count": candidate.token_count,
|
||||||
|
"sources": candidate_source_names(candidate),
|
||||||
|
"raw_count": candidate.raw_count,
|
||||||
|
"chapter_count": candidate.chapter_count,
|
||||||
|
"contexts": candidate.sample_contexts,
|
||||||
|
}
|
||||||
|
return [
|
||||||
|
{
|
||||||
|
"role": "system",
|
||||||
|
"content": (
|
||||||
|
"Judge whether a candidate phrase from a book should be protected for RAG retrieval. "
|
||||||
|
"Do not extract new phrases. Reject common grammar fragments, ordinary nonspecific phrases, "
|
||||||
|
"unstable fragments, and phrases kept only because they are frequent. Keep people, places, "
|
||||||
|
"organizations, factions, events, technologies, fictional conditions, magic systems, formal titles, "
|
||||||
|
"named concepts, and recurring world-specific terms. Return only a JSON object with keys: keep, "
|
||||||
|
"canonical, category, aliases, confidence, importance, allow_nested, suppress_children, reason."
|
||||||
|
),
|
||||||
|
},
|
||||||
|
{"role": "user", "content": json.dumps(payload, ensure_ascii=True)},
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def parse_llm_judgment(content: str, config: EbookSearchConfig) -> LLMJudgment:
|
||||||
|
"""Parse and validate an LLM phrase-judge response.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content (str): Raw model response text.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings supplying nesting defaults.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
LLMJudgment: The parsed and validated judgment.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
TypeError: If the decoded JSON body is not an object.
|
||||||
|
"""
|
||||||
|
body = json.loads(extract_json_object(content))
|
||||||
|
if not isinstance(body, dict):
|
||||||
|
msg = "LLM phrase judge response is not a JSON object"
|
||||||
|
raise TypeError(msg)
|
||||||
|
|
||||||
|
aliases = body.get("aliases", ())
|
||||||
|
if not isinstance(aliases, list | tuple):
|
||||||
|
aliases = ()
|
||||||
|
return LLMJudgment(
|
||||||
|
keep=bool(body.get("keep", False)),
|
||||||
|
canonical=optional_text(body.get("canonical")),
|
||||||
|
category=optional_text(body.get("category")),
|
||||||
|
aliases=tuple(str(alias) for alias in aliases if isinstance(alias, str) and alias.strip()),
|
||||||
|
confidence=clamped_float(body.get("confidence"), default=0.0),
|
||||||
|
importance=clamped_float(body.get("importance"), default=0.5),
|
||||||
|
allow_nested=bool(body.get("allow_nested", config.phrase_default_allow_nested)),
|
||||||
|
suppress_children=bool(body.get("suppress_children", config.phrase_default_suppress_children)),
|
||||||
|
reason=optional_text(body.get("reason")),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_json_object(content: str) -> str:
|
||||||
|
"""Extract a JSON object from plain or fenced model output.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content (str): Raw model response text.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The substring spanning the first JSON object.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If no JSON object is found in the response.
|
||||||
|
"""
|
||||||
|
stripped = content.strip()
|
||||||
|
if stripped.startswith("{") and stripped.endswith("}"):
|
||||||
|
return stripped
|
||||||
|
match = JSON_OBJECT_RE.search(stripped)
|
||||||
|
if match is None:
|
||||||
|
msg = "LLM phrase judge response did not contain a JSON object"
|
||||||
|
raise ValueError(msg)
|
||||||
|
return match.group(0)
|
||||||
|
|
||||||
|
|
||||||
|
def optional_text(value: object) -> str | None:
|
||||||
|
"""Return stripped text for a nullable JSON value.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value (object): Decoded JSON value that may or may not be a string.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str | None: The stripped string, or ``None`` when it is not a non-empty string.
|
||||||
|
"""
|
||||||
|
if not isinstance(value, str):
|
||||||
|
return None
|
||||||
|
stripped = value.strip()
|
||||||
|
return stripped or None
|
||||||
|
|
||||||
|
|
||||||
|
def clamped_float(value: object, *, default: float) -> float:
|
||||||
|
"""Coerce a JSON number into the 0.0 to 1.0 range.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value (object): Decoded JSON value that may or may not be a number.
|
||||||
|
default (float): Fallback returned when ``value`` is not numeric.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
float: The value clamped to ``[0.0, 1.0]``, or ``default`` when non-numeric.
|
||||||
|
"""
|
||||||
|
if not isinstance(value, int | float):
|
||||||
|
return default
|
||||||
|
return min(max(float(value), 0.0), 1.0)
|
||||||
@@ -0,0 +1,491 @@
|
|||||||
|
"""Runtime protected-phrase matching and chunk mention indexing."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from collections import defaultdict
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from sqlalchemy import and_, delete, func, or_, select
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.config import get_ignored_phrases
|
||||||
|
from python.ebook_search.protected_phrases.models import (
|
||||||
|
ChunkPhraseHit,
|
||||||
|
HydratedPhraseMatch,
|
||||||
|
PhraseLookup,
|
||||||
|
PhraseMatch,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import tokenize_with_offsets
|
||||||
|
from python.orm.richie import (
|
||||||
|
EbookChunk,
|
||||||
|
EbookChunkPhraseMention,
|
||||||
|
EbookPhraseAlias,
|
||||||
|
EbookProtectedPhrase,
|
||||||
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Iterator, Sequence
|
||||||
|
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import NormalizedToken
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def load_phrase_lookup(
|
||||||
|
session: AsyncSession,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
book_id: int | None = None,
|
||||||
|
series_id: int | None = None,
|
||||||
|
) -> PhraseLookup:
|
||||||
|
"""Load protected phrases and aliases into RAM lookup maps.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
book_id (int | None): Optional book scope to restrict loaded phrases.
|
||||||
|
series_id (int | None): Optional series scope to restrict loaded phrases.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseLookup: Normalized phrase and alias maps with the token-window bounds to test.
|
||||||
|
"""
|
||||||
|
norm_to_ids: defaultdict[str, list[int]] = defaultdict(list)
|
||||||
|
alias_to_ids: defaultdict[str, list[int]] = defaultdict(list)
|
||||||
|
max_tokens = config.phrase_max_tokens
|
||||||
|
|
||||||
|
phrase_statement = select(
|
||||||
|
EbookProtectedPhrase.id,
|
||||||
|
EbookProtectedPhrase.phrase_norm,
|
||||||
|
EbookProtectedPhrase.token_count,
|
||||||
|
)
|
||||||
|
scope_filter = protected_phrase_scope_filter(book_id=book_id, series_id=series_id)
|
||||||
|
if scope_filter is not None:
|
||||||
|
phrase_statement = phrase_statement.where(scope_filter)
|
||||||
|
|
||||||
|
for row in await session.execute(phrase_statement):
|
||||||
|
phrase_id = int(row.id)
|
||||||
|
phrase_norm = str(row.phrase_norm)
|
||||||
|
norm_to_ids[phrase_norm].append(phrase_id)
|
||||||
|
max_tokens = max(max_tokens, int(row.token_count))
|
||||||
|
|
||||||
|
alias_statement = select(
|
||||||
|
EbookPhraseAlias.alias_norm,
|
||||||
|
EbookPhraseAlias.phrase_id,
|
||||||
|
).join(EbookProtectedPhrase, EbookProtectedPhrase.id == EbookPhraseAlias.phrase_id)
|
||||||
|
if scope_filter is not None:
|
||||||
|
alias_statement = alias_statement.where(scope_filter)
|
||||||
|
|
||||||
|
for row in await session.execute(alias_statement):
|
||||||
|
alias_norm = str(row.alias_norm)
|
||||||
|
alias_to_ids[alias_norm].append(int(row.phrase_id))
|
||||||
|
max_tokens = max(max_tokens, len(alias_norm.split()))
|
||||||
|
|
||||||
|
return PhraseLookup(
|
||||||
|
norm_to_phrase_ids={key: tuple(values) for key, values in norm_to_ids.items()},
|
||||||
|
alias_to_phrase_ids={key: tuple(values) for key, values in alias_to_ids.items()},
|
||||||
|
min_tokens=config.phrase_min_tokens,
|
||||||
|
max_tokens=max_tokens,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def protected_phrase_scope_filter(*, book_id: int | None, series_id: int | None) -> object | None:
|
||||||
|
"""Build a SQLAlchemy filter for optional phrase book and series scope.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_id (int | None): Optional book scope to include alongside global phrases.
|
||||||
|
series_id (int | None): Optional series scope to include alongside global phrases.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
object | None: A combined SQLAlchemy filter clause, or ``None`` when no scope is given.
|
||||||
|
"""
|
||||||
|
conditions = []
|
||||||
|
if book_id is not None:
|
||||||
|
conditions.append(or_(EbookProtectedPhrase.book_id.is_(None), EbookProtectedPhrase.book_id == book_id))
|
||||||
|
if series_id is not None:
|
||||||
|
conditions.append(or_(EbookProtectedPhrase.series_id.is_(None), EbookProtectedPhrase.series_id == series_id))
|
||||||
|
if not conditions:
|
||||||
|
return None
|
||||||
|
return and_(*conditions)
|
||||||
|
|
||||||
|
|
||||||
|
def generate_query_ngrams(
|
||||||
|
tokens_: Sequence[str],
|
||||||
|
min_n: int,
|
||||||
|
max_n: int,
|
||||||
|
) -> Iterator[tuple[str, int, int]]:
|
||||||
|
"""Generate normalized query windows from longest to shortest.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tokens_ (Sequence[str]): Normalized query tokens.
|
||||||
|
min_n (int): Smallest window size to yield.
|
||||||
|
max_n (int): Largest window size to yield, capped at the token count.
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
tuple[str, int, int]: Normalized window text with its start and end token indices.
|
||||||
|
"""
|
||||||
|
capped_max_n = min(max_n, len(tokens_))
|
||||||
|
for ngram_size in range(capped_max_n, min_n - 1, -1):
|
||||||
|
for start in range(len(tokens_) - ngram_size + 1):
|
||||||
|
end = start + ngram_size
|
||||||
|
phrase_norm = " ".join(tokens_[start:end])
|
||||||
|
if phrase_norm in get_ignored_phrases():
|
||||||
|
continue
|
||||||
|
yield phrase_norm, start, end
|
||||||
|
|
||||||
|
|
||||||
|
def detect_phrase_candidates(query_text: str, lookup: PhraseLookup) -> list[PhraseMatch]:
|
||||||
|
"""Detect protected phrase windows in a user query using RAM hash lookups.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
query_text (str): User query text to scan.
|
||||||
|
lookup (PhraseLookup): In-memory phrase and alias lookup maps.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[PhraseMatch]: Unhydrated phrase matches found in the query.
|
||||||
|
"""
|
||||||
|
return detect_phrase_candidates_from_tokens(tokenize_with_offsets(query_text), lookup)
|
||||||
|
|
||||||
|
|
||||||
|
def detect_phrase_candidates_in_text(text: str, lookup: PhraseLookup) -> list[PhraseMatch]:
|
||||||
|
"""Detect protected phrase windows in arbitrary text with character offsets.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text (str): Arbitrary text, such as a chunk, to scan.
|
||||||
|
lookup (PhraseLookup): In-memory phrase and alias lookup maps.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[PhraseMatch]: Unhydrated phrase matches found in the text.
|
||||||
|
"""
|
||||||
|
return detect_phrase_candidates_from_tokens(tokenize_with_offsets(text), lookup)
|
||||||
|
|
||||||
|
|
||||||
|
def detect_phrase_candidates_from_tokens(tokens_: Sequence[NormalizedToken], lookup: PhraseLookup) -> list[PhraseMatch]:
|
||||||
|
"""Detect protected phrase windows from already-normalized tokens.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tokens_ (Sequence[NormalizedToken]): Normalized tokens with character offsets.
|
||||||
|
lookup (PhraseLookup): In-memory phrase and alias lookup maps.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[PhraseMatch]: Deduplicated unhydrated phrase matches with token and character spans.
|
||||||
|
"""
|
||||||
|
matches: list[PhraseMatch] = []
|
||||||
|
seen: set[tuple[int | None, str, int, int]] = set()
|
||||||
|
token_texts = [token.text for token in tokens_]
|
||||||
|
for phrase_norm, start, end in generate_query_ngrams(token_texts, min_n=lookup.min_tokens, max_n=lookup.max_tokens):
|
||||||
|
phrase_ids = lookup.norm_to_phrase_ids.get(phrase_norm, ())
|
||||||
|
alias_ids = lookup.alias_to_phrase_ids.get(phrase_norm, ())
|
||||||
|
for phrase_id in (*phrase_ids, *alias_ids):
|
||||||
|
key = (phrase_id, phrase_norm, start, end)
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
matches.append(
|
||||||
|
PhraseMatch(
|
||||||
|
phrase_norm=phrase_norm,
|
||||||
|
phrase_id=phrase_id,
|
||||||
|
start_token=start,
|
||||||
|
end_token=end,
|
||||||
|
token_count=end - start,
|
||||||
|
start_char=tokens_[start].start_char,
|
||||||
|
end_char=tokens_[end - 1].end_char,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return matches
|
||||||
|
|
||||||
|
|
||||||
|
async def hydrate_matches(session: AsyncSession, matches: Sequence[PhraseMatch]) -> list[HydratedPhraseMatch]:
|
||||||
|
"""Fetch protected phrase metadata for raw phrase matches.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
matches (Sequence[PhraseMatch]): Unhydrated matches to enrich.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[HydratedPhraseMatch]: Matches with protected-phrase metadata attached.
|
||||||
|
"""
|
||||||
|
if not matches:
|
||||||
|
return []
|
||||||
|
|
||||||
|
phrase_ids = sorted({match.phrase_id for match in matches if match.phrase_id is not None})
|
||||||
|
if not phrase_ids:
|
||||||
|
return []
|
||||||
|
|
||||||
|
rows = {
|
||||||
|
row.id: row
|
||||||
|
for row in await session.scalars(select(EbookProtectedPhrase).where(EbookProtectedPhrase.id.in_(phrase_ids)))
|
||||||
|
}
|
||||||
|
hydrated: list[HydratedPhraseMatch] = []
|
||||||
|
for match in matches:
|
||||||
|
if match.phrase_id is None:
|
||||||
|
continue
|
||||||
|
phrase = rows.get(match.phrase_id)
|
||||||
|
if phrase is None:
|
||||||
|
continue
|
||||||
|
hydrated.append(
|
||||||
|
HydratedPhraseMatch(
|
||||||
|
phrase_id=phrase.id,
|
||||||
|
matched_norm=match.phrase_norm,
|
||||||
|
phrase_text=phrase.phrase_text,
|
||||||
|
phrase_norm=phrase.phrase_norm,
|
||||||
|
canonical_id=phrase.canonical_id,
|
||||||
|
phrase_type=phrase.phrase_type,
|
||||||
|
token_count=match.token_count,
|
||||||
|
confidence=phrase.confidence,
|
||||||
|
importance=phrase.importance,
|
||||||
|
allow_nested=phrase.allow_nested,
|
||||||
|
suppress_children=phrase.suppress_children,
|
||||||
|
start_token=match.start_token,
|
||||||
|
end_token=match.end_token,
|
||||||
|
start_char=match.start_char,
|
||||||
|
end_char=match.end_char,
|
||||||
|
book_id=phrase.book_id,
|
||||||
|
series_id=phrase.series_id,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return hydrated
|
||||||
|
|
||||||
|
|
||||||
|
def overlaps(first: HydratedPhraseMatch, second: HydratedPhraseMatch) -> bool:
|
||||||
|
"""Return whether two token spans overlap.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
first (HydratedPhraseMatch): First match to compare.
|
||||||
|
second (HydratedPhraseMatch): Second match to compare.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the two token spans share at least one token position.
|
||||||
|
"""
|
||||||
|
return not (first.end_token <= second.start_token or first.start_token >= second.end_token)
|
||||||
|
|
||||||
|
|
||||||
|
def is_inside(child: HydratedPhraseMatch, parent: HydratedPhraseMatch) -> bool:
|
||||||
|
"""Return whether one token span is strictly inside another.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
child (HydratedPhraseMatch): Candidate nested match.
|
||||||
|
parent (HydratedPhraseMatch): Candidate enclosing match.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when ``child`` lies within ``parent`` and is not the same span.
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
child.start_token >= parent.start_token
|
||||||
|
and child.end_token <= parent.end_token
|
||||||
|
and (child.start_token, child.end_token, child.phrase_id)
|
||||||
|
!= (parent.start_token, parent.end_token, parent.phrase_id)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def rank_match(match: HydratedPhraseMatch) -> tuple[float, float, int]:
|
||||||
|
"""Rank phrase matches by importance, confidence, then token count.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
match (HydratedPhraseMatch): Match to build a sort key for.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[float, float, int]: A comparable key of importance, confidence, and token count.
|
||||||
|
"""
|
||||||
|
return (match.importance, match.confidence, match.token_count)
|
||||||
|
|
||||||
|
|
||||||
|
def should_suppress(candidate: HydratedPhraseMatch, kept: HydratedPhraseMatch) -> bool:
|
||||||
|
"""Return whether an already-kept match should suppress a candidate.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
candidate (HydratedPhraseMatch): Match being considered for keeping.
|
||||||
|
kept (HydratedPhraseMatch): Match already kept that may suppress the candidate.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True when the candidate should be dropped in favor of the kept match.
|
||||||
|
"""
|
||||||
|
if not overlaps(candidate, kept):
|
||||||
|
return False
|
||||||
|
if candidate.canonical_id == kept.canonical_id:
|
||||||
|
return rank_match(kept) >= rank_match(candidate)
|
||||||
|
if is_inside(candidate, kept) and kept.suppress_children and not candidate.allow_nested:
|
||||||
|
return True
|
||||||
|
return not candidate.allow_nested and rank_match(kept) > rank_match(candidate)
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_overlaps(matches: Sequence[HydratedPhraseMatch]) -> list[HydratedPhraseMatch]:
|
||||||
|
"""Resolve overlapping phrase matches without relying only on longest match.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
matches (Sequence[HydratedPhraseMatch]): Hydrated matches that may overlap.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[HydratedPhraseMatch]: The kept, non-suppressed matches.
|
||||||
|
"""
|
||||||
|
sorted_matches = sorted(
|
||||||
|
matches,
|
||||||
|
key=lambda match: (match.start_token, -match.token_count, -match.importance, -match.confidence),
|
||||||
|
)
|
||||||
|
kept: list[HydratedPhraseMatch] = []
|
||||||
|
for candidate in sorted_matches:
|
||||||
|
if any(should_suppress(candidate, existing) for existing in kept):
|
||||||
|
continue
|
||||||
|
kept.append(candidate)
|
||||||
|
return kept
|
||||||
|
|
||||||
|
|
||||||
|
async def detect_protected_phrases_for_query(
|
||||||
|
session: AsyncSession,
|
||||||
|
query_text: str,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
lookup: PhraseLookup | None = None,
|
||||||
|
book_id: int | None = None,
|
||||||
|
series_id: int | None = None,
|
||||||
|
) -> list[HydratedPhraseMatch]:
|
||||||
|
"""Run the full online protected-phrase query-detection pipeline.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
query_text (str): User query text to detect phrases in.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
lookup (PhraseLookup | None): Optional preloaded lookup; loaded on demand when ``None``.
|
||||||
|
book_id (int | None): Optional book scope for lookup loading.
|
||||||
|
series_id (int | None): Optional series scope for lookup loading.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[HydratedPhraseMatch]: Hydrated, overlap-resolved phrase matches for the query.
|
||||||
|
"""
|
||||||
|
active_lookup = (
|
||||||
|
lookup
|
||||||
|
if lookup is not None
|
||||||
|
else await load_phrase_lookup(session, config, book_id=book_id, series_id=series_id)
|
||||||
|
)
|
||||||
|
return resolve_overlaps(await hydrate_matches(session, detect_phrase_candidates(query_text, active_lookup)))
|
||||||
|
|
||||||
|
|
||||||
|
async def index_chunk_phrase_mentions_for_book(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
series_id: int | None = None,
|
||||||
|
lookup: PhraseLookup | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Rebuild chunk phrase mentions for all chunks in one book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose chunk mentions are rebuilt.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
series_id (int | None): Optional series scope for lookup loading.
|
||||||
|
lookup (PhraseLookup | None): Optional preloaded lookup; loaded on demand when ``None``.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Total number of chunk phrase mentions indexed for the book.
|
||||||
|
"""
|
||||||
|
active_lookup = (
|
||||||
|
lookup
|
||||||
|
if lookup is not None
|
||||||
|
else await load_phrase_lookup(session, config, book_id=book_id, series_id=series_id)
|
||||||
|
)
|
||||||
|
await session.execute(delete(EbookChunkPhraseMention).where(EbookChunkPhraseMention.book_id == book_id))
|
||||||
|
chunks = await session.scalars(select(EbookChunk).where(EbookChunk.source_id == book_id).order_by(EbookChunk.id))
|
||||||
|
count = 0
|
||||||
|
for chunk in chunks:
|
||||||
|
count += await index_chunk_phrase_mentions(session, chunk, lookup=active_lookup)
|
||||||
|
await session.flush()
|
||||||
|
logger.info("ebook_chunk_phrase_mentions_indexed book_id=%s mentions=%s", book_id, count)
|
||||||
|
return count
|
||||||
|
|
||||||
|
|
||||||
|
async def index_chunk_phrase_mentions(session: AsyncSession, chunk: EbookChunk, *, lookup: PhraseLookup) -> int:
|
||||||
|
"""Store protected phrase mentions for one chunk.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
chunk (EbookChunk): Chunk whose text is scanned for phrase mentions.
|
||||||
|
lookup (PhraseLookup): In-memory phrase and alias lookup maps.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of phrase mentions stored for the chunk.
|
||||||
|
"""
|
||||||
|
raw_matches = detect_phrase_candidates_in_text(chunk.text, lookup)
|
||||||
|
hydrated = resolve_overlaps(await hydrate_matches(session, raw_matches))
|
||||||
|
for match in hydrated:
|
||||||
|
session.add(
|
||||||
|
EbookChunkPhraseMention(
|
||||||
|
chunk_id=chunk.id,
|
||||||
|
phrase_id=match.phrase_id,
|
||||||
|
book_id=match.book_id if match.book_id is not None else chunk.source_id,
|
||||||
|
series_id=match.series_id,
|
||||||
|
start_char=match.start_char if match.start_char is not None else 0,
|
||||||
|
end_char=match.end_char,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return len(hydrated)
|
||||||
|
|
||||||
|
|
||||||
|
async def phrase_hits_for_chunks(
|
||||||
|
session: AsyncSession,
|
||||||
|
*,
|
||||||
|
chunk_ids: Sequence[int],
|
||||||
|
phrase_ids: Sequence[int],
|
||||||
|
) -> dict[int, tuple[ChunkPhraseHit, ...]]:
|
||||||
|
"""Return matched protected phrases with mention counts by chunk id using indexed chunk mentions.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
chunk_ids (Sequence[int]): Chunk ids to look up mentions for.
|
||||||
|
phrase_ids (Sequence[int]): Protected phrase ids to restrict the results to.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[int, tuple[ChunkPhraseHit, ...]]: Phrase hits per chunk id, ordered by mention count.
|
||||||
|
"""
|
||||||
|
if not chunk_ids or not phrase_ids:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
mention_count = func.count(EbookChunkPhraseMention.phrase_id).label("mention_count")
|
||||||
|
statement = (
|
||||||
|
select(
|
||||||
|
EbookChunkPhraseMention.chunk_id,
|
||||||
|
EbookProtectedPhrase.id.label("phrase_id"),
|
||||||
|
EbookProtectedPhrase.phrase_text,
|
||||||
|
mention_count,
|
||||||
|
)
|
||||||
|
.join(EbookProtectedPhrase, EbookProtectedPhrase.id == EbookChunkPhraseMention.phrase_id)
|
||||||
|
.where(
|
||||||
|
EbookChunkPhraseMention.chunk_id.in_(chunk_ids),
|
||||||
|
EbookChunkPhraseMention.phrase_id.in_(phrase_ids),
|
||||||
|
)
|
||||||
|
.group_by(EbookChunkPhraseMention.chunk_id, EbookProtectedPhrase.id, EbookProtectedPhrase.phrase_text)
|
||||||
|
.order_by(EbookChunkPhraseMention.chunk_id, mention_count.desc(), EbookProtectedPhrase.phrase_text)
|
||||||
|
)
|
||||||
|
hits: defaultdict[int, list[ChunkPhraseHit]] = defaultdict(list)
|
||||||
|
for row in await session.execute(statement):
|
||||||
|
hits[int(row.chunk_id)].append(
|
||||||
|
ChunkPhraseHit(
|
||||||
|
phrase_id=int(row.phrase_id),
|
||||||
|
phrase_text=str(row.phrase_text),
|
||||||
|
mention_count=int(row.mention_count),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return {chunk_id: tuple(chunk_hits) for chunk_id, chunk_hits in hits.items()}
|
||||||
|
|
||||||
|
|
||||||
|
async def phrase_hit_counts_for_chunks(
|
||||||
|
session: AsyncSession,
|
||||||
|
*,
|
||||||
|
chunk_ids: Sequence[int],
|
||||||
|
phrase_ids: Sequence[int],
|
||||||
|
) -> dict[int, int]:
|
||||||
|
"""Return phrase-hit counts by chunk id using indexed chunk mentions.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
chunk_ids (Sequence[int]): Chunk ids to count mentions for.
|
||||||
|
phrase_ids (Sequence[int]): Protected phrase ids to restrict the counts to.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[int, int]: Total mention count per chunk id.
|
||||||
|
"""
|
||||||
|
hits = await phrase_hits_for_chunks(session, chunk_ids=chunk_ids, phrase_ids=phrase_ids)
|
||||||
|
return {chunk_id: sum(hit.mention_count for hit in chunk_hits) for chunk_id, chunk_hits in hits.items()}
|
||||||
@@ -0,0 +1,285 @@
|
|||||||
|
"""Dataclasses shared by protected phrase extraction, judging, matching, and backfills."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Mapping
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(slots=True)
|
||||||
|
class PhraseCandidate:
|
||||||
|
"""A phrase candidate with merged extraction-source metadata.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
phrase_text (str): Display text for the phrase.
|
||||||
|
phrase_norm (str): Normalized phrase used as the merge key.
|
||||||
|
token_count (int): Number of normalized tokens in the phrase.
|
||||||
|
source_raw_ngram (bool): Whether the raw n-gram extractor produced the phrase.
|
||||||
|
source_yake (bool): Whether YAKE keyword extraction produced the phrase.
|
||||||
|
source_spacy_ner (bool): Whether spaCy named-entity recognition produced the phrase.
|
||||||
|
source_spacy_noun_chunk (bool): Whether spaCy noun chunking produced the phrase.
|
||||||
|
source_capitalized (bool): Whether the capitalized-run extractor produced the phrase.
|
||||||
|
source_metadata (bool): Whether book metadata produced the phrase.
|
||||||
|
spacy_label (str | None): spaCy entity label when NER produced the phrase.
|
||||||
|
raw_count (int): Occurrences counted across the book text.
|
||||||
|
chapter_count (int): Number of chapters containing the phrase.
|
||||||
|
yake_score (float | None): Raw YAKE score when available; lower is better.
|
||||||
|
candidate_score (float): Combined pre-judging score.
|
||||||
|
sample_contexts (list[str]): Normalized context snippets around occurrences.
|
||||||
|
"""
|
||||||
|
|
||||||
|
phrase_text: str
|
||||||
|
phrase_norm: str
|
||||||
|
token_count: int
|
||||||
|
source_raw_ngram: bool = False
|
||||||
|
source_yake: bool = False
|
||||||
|
source_spacy_ner: bool = False
|
||||||
|
source_spacy_noun_chunk: bool = False
|
||||||
|
source_capitalized: bool = False
|
||||||
|
source_metadata: bool = False
|
||||||
|
spacy_label: str | None = None
|
||||||
|
raw_count: int = 0
|
||||||
|
chapter_count: int = 0
|
||||||
|
yake_score: float | None = None
|
||||||
|
candidate_score: float = 0.0
|
||||||
|
sample_contexts: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class LLMJudgment:
|
||||||
|
"""A structured phrase judgment returned by the LLM judge.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
keep (bool): Whether the judge accepted the phrase for protection.
|
||||||
|
canonical (str | None): Canonical phrase text chosen by the judge.
|
||||||
|
category (str | None): Phrase category such as person, place, or event.
|
||||||
|
aliases (tuple[str, ...]): Alternate surface forms for the phrase.
|
||||||
|
confidence (float): Judge confidence between 0.0 and 1.0.
|
||||||
|
importance (float): Judge importance between 0.0 and 1.0.
|
||||||
|
allow_nested (bool): Whether the phrase may match inside a larger kept match.
|
||||||
|
suppress_children (bool): Whether the phrase suppresses matches nested inside it.
|
||||||
|
reason (str | None): Free-text explanation from the judge.
|
||||||
|
"""
|
||||||
|
|
||||||
|
keep: bool
|
||||||
|
canonical: str | None
|
||||||
|
category: str | None
|
||||||
|
aliases: tuple[str, ...]
|
||||||
|
confidence: float
|
||||||
|
importance: float = 0.5
|
||||||
|
allow_nested: bool = False
|
||||||
|
suppress_children: bool = True
|
||||||
|
reason: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class PhraseLookup:
|
||||||
|
"""In-memory lookup maps used for constant-time phrase-window checks.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
norm_to_phrase_ids (Mapping[str, tuple[int, ...]]): Normalized phrase to protected phrase ids.
|
||||||
|
alias_to_phrase_ids (Mapping[str, tuple[int, ...]]): Normalized alias to protected phrase ids.
|
||||||
|
min_tokens (int): Smallest token-window size to test.
|
||||||
|
max_tokens (int): Largest token-window size to test.
|
||||||
|
"""
|
||||||
|
|
||||||
|
norm_to_phrase_ids: Mapping[str, tuple[int, ...]]
|
||||||
|
alias_to_phrase_ids: Mapping[str, tuple[int, ...]]
|
||||||
|
min_tokens: int
|
||||||
|
max_tokens: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class PhraseMatch:
|
||||||
|
"""An unhydrated query or chunk phrase match.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
phrase_norm (str): Normalized text of the matched window.
|
||||||
|
start_token (int): Index of the first matched token.
|
||||||
|
end_token (int): Index one past the last matched token.
|
||||||
|
token_count (int): Number of tokens in the match.
|
||||||
|
phrase_id (int | None): Matched protected phrase id when known.
|
||||||
|
start_char (int | None): Start character offset in the source text.
|
||||||
|
end_char (int | None): End character offset in the source text.
|
||||||
|
"""
|
||||||
|
|
||||||
|
phrase_norm: str
|
||||||
|
start_token: int
|
||||||
|
end_token: int
|
||||||
|
token_count: int
|
||||||
|
phrase_id: int | None = None
|
||||||
|
start_char: int | None = None
|
||||||
|
end_char: int | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class HydratedPhraseMatch:
|
||||||
|
"""A phrase match with protected-phrase metadata attached.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
phrase_id (int): Protected phrase id.
|
||||||
|
matched_norm (str): Normalized window text that matched.
|
||||||
|
phrase_text (str): Display text of the protected phrase.
|
||||||
|
phrase_norm (str): Normalized text of the protected phrase.
|
||||||
|
canonical_id (str): Deterministic ``category:slug`` identifier.
|
||||||
|
phrase_type (str | None): Phrase category.
|
||||||
|
token_count (int): Number of tokens in the match.
|
||||||
|
confidence (float): Stored judge confidence.
|
||||||
|
importance (float): Stored judge importance.
|
||||||
|
allow_nested (bool): Whether the phrase may match inside a larger kept match.
|
||||||
|
suppress_children (bool): Whether the phrase suppresses matches nested inside it.
|
||||||
|
start_token (int): Index of the first matched token.
|
||||||
|
end_token (int): Index one past the last matched token.
|
||||||
|
start_char (int | None): Start character offset in the source text.
|
||||||
|
end_char (int | None): End character offset in the source text.
|
||||||
|
book_id (int | None): Book scope of the phrase.
|
||||||
|
series_id (int | None): Series scope of the phrase.
|
||||||
|
"""
|
||||||
|
|
||||||
|
phrase_id: int
|
||||||
|
matched_norm: str
|
||||||
|
phrase_text: str
|
||||||
|
phrase_norm: str
|
||||||
|
canonical_id: str
|
||||||
|
phrase_type: str | None
|
||||||
|
token_count: int
|
||||||
|
confidence: float
|
||||||
|
importance: float
|
||||||
|
allow_nested: bool
|
||||||
|
suppress_children: bool
|
||||||
|
start_token: int
|
||||||
|
end_token: int
|
||||||
|
start_char: int | None = None
|
||||||
|
end_char: int | None = None
|
||||||
|
book_id: int | None = None
|
||||||
|
series_id: int | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class ChunkPhraseHit:
|
||||||
|
"""One protected phrase with its mention count inside one retrieved chunk.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
phrase_id (int): Protected phrase id.
|
||||||
|
phrase_text (str): Display text of the protected phrase.
|
||||||
|
mention_count (int): Indexed mentions of the phrase in the chunk.
|
||||||
|
"""
|
||||||
|
|
||||||
|
phrase_id: int
|
||||||
|
phrase_text: str
|
||||||
|
mention_count: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class PhraseCandidateGenerationResult:
|
||||||
|
"""Summary of candidate phrase extraction for indexed books.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
books_seen (int): Indexed books examined.
|
||||||
|
books_built (int): Books that had candidates generated and committed.
|
||||||
|
candidate_phrases (int): Candidate phrases stored across all books.
|
||||||
|
"""
|
||||||
|
|
||||||
|
books_seen: int
|
||||||
|
books_built: int
|
||||||
|
candidate_phrases: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class CorpusPhraseStats:
|
||||||
|
"""Corpus-wide candidate and protected phrase counts for the admin page.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
total_books (int): Indexed books in the corpus.
|
||||||
|
books_with_candidates (int): Books that have candidate phrases generated.
|
||||||
|
books_fully_judged (int): Books with candidates where every candidate has been judged.
|
||||||
|
candidate_phrases (int): Candidate phrases stored across all books.
|
||||||
|
judged_candidates (int): Candidate phrases that have been LLM judged.
|
||||||
|
unjudged_candidates (int): Candidate phrases still waiting for judgment.
|
||||||
|
protected_phrases (int): Protected phrases promoted across all books.
|
||||||
|
"""
|
||||||
|
|
||||||
|
total_books: int
|
||||||
|
books_with_candidates: int
|
||||||
|
books_fully_judged: int
|
||||||
|
candidate_phrases: int
|
||||||
|
judged_candidates: int
|
||||||
|
unjudged_candidates: int
|
||||||
|
protected_phrases: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class PhraseJudgmentBackfillResult:
|
||||||
|
"""Summary of LLM judging for stored candidate phrases.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
books_seen (int): Indexed books examined.
|
||||||
|
books_judged (int): Books with judgments committed.
|
||||||
|
books_failed (int): Books rolled back after an error.
|
||||||
|
candidates_judged (int): Candidate phrases sent to the LLM judge.
|
||||||
|
protected_phrases (int): Protected phrases promoted from candidates.
|
||||||
|
phrase_mentions (int): Chunk phrase mentions indexed across all books.
|
||||||
|
"""
|
||||||
|
|
||||||
|
books_seen: int
|
||||||
|
books_judged: int
|
||||||
|
books_failed: int
|
||||||
|
candidates_judged: int
|
||||||
|
protected_phrases: int
|
||||||
|
phrase_mentions: int
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class BookJudgmentResult:
|
||||||
|
"""Outcome of judging one book's candidate phrases.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
judged (int): Candidate phrases sent to the LLM judge.
|
||||||
|
protected (int): Protected phrases promoted from candidates.
|
||||||
|
mentions (int): Chunk phrase mentions indexed for the book.
|
||||||
|
committed (bool): Whether the book's judgments were committed.
|
||||||
|
failed (bool): Whether the book was rolled back after an error.
|
||||||
|
"""
|
||||||
|
|
||||||
|
judged: int = 0
|
||||||
|
protected: int = 0
|
||||||
|
mentions: int = 0
|
||||||
|
committed: bool = False
|
||||||
|
failed: bool = False
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class BookCandidateResult:
|
||||||
|
"""Outcome of generating one book's candidate phrases.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
candidates (int): Candidate phrases stored for the book.
|
||||||
|
built (bool): Whether candidate generation was committed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
candidates: int = 0
|
||||||
|
built: bool = False
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class PhraseRecalculationResult:
|
||||||
|
"""Summary of phrase cleanup and candidate regeneration for one book.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
book_id (int): Book the recalculation ran against.
|
||||||
|
deleted_candidates (int): Candidate phrase rows deleted.
|
||||||
|
deleted_protected_phrases (int): Protected phrase rows deleted.
|
||||||
|
deleted_aliases (int): Phrase alias rows deleted.
|
||||||
|
deleted_mentions (int): Chunk phrase mention rows deleted.
|
||||||
|
candidate_phrases (int): Candidate phrases regenerated after cleanup.
|
||||||
|
"""
|
||||||
|
|
||||||
|
book_id: int
|
||||||
|
deleted_candidates: int
|
||||||
|
deleted_protected_phrases: int
|
||||||
|
deleted_aliases: int
|
||||||
|
deleted_mentions: int
|
||||||
|
candidate_phrases: int
|
||||||
@@ -0,0 +1,101 @@
|
|||||||
|
"""Process pool for offloading CPU-bound phrase extraction off the request thread.
|
||||||
|
|
||||||
|
Phrase extraction is pure-Python CPU work (n-gram sliding, YAKE), so running it inline in a
|
||||||
|
sync request handler serializes concurrent recalculations behind the GIL. Submitting it to a
|
||||||
|
``ProcessPoolExecutor`` lets concurrent extractions run in parallel across cores instead. A
|
||||||
|
``spawn`` context is used so workers do not inherit the parent's database engine, connections,
|
||||||
|
or server threads.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import multiprocessing
|
||||||
|
import os
|
||||||
|
from concurrent.futures import ProcessPoolExecutor
|
||||||
|
from threading import Lock
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.extraction import extract_phrase_candidates_for_book
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Mapping, Sequence
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.protected_phrases.models import PhraseCandidate
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class _ExtractionPool:
|
||||||
|
"""Lazily created process-wide extraction pool and the lock guarding it."""
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self.lock = Lock()
|
||||||
|
self.pool: ProcessPoolExecutor | None = None
|
||||||
|
|
||||||
|
|
||||||
|
_extraction_pool = _ExtractionPool()
|
||||||
|
|
||||||
|
|
||||||
|
def get_extraction_pool(max_workers: int) -> ProcessPoolExecutor:
|
||||||
|
"""Return the shared extraction process pool, creating it on first use.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
max_workers (int): Desired worker count; values below 1 fall back to the CPU count.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
ProcessPoolExecutor: The shared pool for phrase extraction.
|
||||||
|
"""
|
||||||
|
with _extraction_pool.lock:
|
||||||
|
if _extraction_pool.pool is None:
|
||||||
|
workers = max_workers if max_workers > 0 else (os.cpu_count() or 1)
|
||||||
|
_extraction_pool.pool = ProcessPoolExecutor(
|
||||||
|
max_workers=workers,
|
||||||
|
mp_context=multiprocessing.get_context("spawn"),
|
||||||
|
)
|
||||||
|
logger.info("ebook_phrase_extraction_pool_started workers=%s", workers)
|
||||||
|
return _extraction_pool.pool
|
||||||
|
|
||||||
|
|
||||||
|
def shutdown_extraction_pool() -> None:
|
||||||
|
"""Shut down the shared extraction pool if it was started."""
|
||||||
|
with _extraction_pool.lock:
|
||||||
|
if _extraction_pool.pool is not None:
|
||||||
|
_extraction_pool.pool.shutdown(wait=False, cancel_futures=True)
|
||||||
|
_extraction_pool.pool = None
|
||||||
|
logger.info("ebook_phrase_extraction_pool_shutdown")
|
||||||
|
|
||||||
|
|
||||||
|
async def extract_phrase_candidates_in_pool(
|
||||||
|
book_text: str,
|
||||||
|
chapters: Sequence[str],
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
*,
|
||||||
|
metadata: Mapping[str, object] | None,
|
||||||
|
) -> list[PhraseCandidate]:
|
||||||
|
"""Run book phrase extraction in a worker process and await the result.
|
||||||
|
|
||||||
|
Only the CPU-bound extraction runs in the worker; the caller keeps all database work in the
|
||||||
|
request process. The spaCy pipeline is not supported here because it is not picklable, so
|
||||||
|
this always runs the non-spaCy extraction path.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_text (str): Full book text used for extraction.
|
||||||
|
chapters (Sequence[str]): Chapter-like text blocks used for frequency counts.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings.
|
||||||
|
metadata (Mapping[str, object] | None): Optional book metadata used as a candidate source.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[PhraseCandidate]: Scored candidates sorted best-first and capped per book.
|
||||||
|
"""
|
||||||
|
pool = get_extraction_pool(config.protected_phrase_extraction_workers)
|
||||||
|
future = pool.submit(
|
||||||
|
extract_phrase_candidates_for_book,
|
||||||
|
book_text,
|
||||||
|
list(chapters),
|
||||||
|
config,
|
||||||
|
metadata=dict(metadata) if metadata is not None else None,
|
||||||
|
)
|
||||||
|
return await asyncio.wrap_future(future)
|
||||||
@@ -0,0 +1,700 @@
|
|||||||
|
"""Database persistence for candidate and protected phrase rows."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from sqlalchemy import delete, func, or_, select
|
||||||
|
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
||||||
|
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||||
|
|
||||||
|
from python.ebook_search.protected_phrases.extraction import minimum_candidate_raw_count
|
||||||
|
from python.ebook_search.protected_phrases.models import (
|
||||||
|
CorpusPhraseStats,
|
||||||
|
PhraseCandidate,
|
||||||
|
PhraseRecalculationResult,
|
||||||
|
)
|
||||||
|
from python.ebook_search.protected_phrases.text_normalization import normalize_text
|
||||||
|
from python.orm.richie import (
|
||||||
|
EbookCandidatePhrase,
|
||||||
|
EbookChunk,
|
||||||
|
EbookChunkPhraseMention,
|
||||||
|
EbookPhraseAlias,
|
||||||
|
EbookProtectedPhrase,
|
||||||
|
EbookSource,
|
||||||
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
from sqlalchemy.dialects.postgresql.dml import Insert as PostgresInsert
|
||||||
|
from sqlalchemy.dialects.sqlite.dml import Insert as SqliteInsert
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from python.ebook_search.config import EbookSearchConfig
|
||||||
|
from python.ebook_search.protected_phrases.models import LLMJudgment
|
||||||
|
from python.orm.richie.base import TableBase
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def dialect_insert(session: AsyncSession, table: type[TableBase]) -> PostgresInsert | SqliteInsert:
|
||||||
|
"""Return a dialect-specific INSERT construct that supports ``ON CONFLICT DO UPDATE``.
|
||||||
|
|
||||||
|
Production runs on PostgreSQL while tests run on SQLite; both support upserts with
|
||||||
|
compatible SQLAlchemy constructs, so the correct one is chosen from the bound dialect.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session whose bind selects the dialect.
|
||||||
|
table (type[TableBase]): Mapped table to insert into.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PostgresInsert | SqliteInsert: A dialect insert exposing ``on_conflict_do_update``.
|
||||||
|
"""
|
||||||
|
if session.get_bind().dialect.name == "sqlite":
|
||||||
|
return sqlite_insert(table)
|
||||||
|
return pg_insert(table)
|
||||||
|
|
||||||
|
|
||||||
|
async def load_book_text(session: AsyncSession, book_id: int) -> str:
|
||||||
|
"""Load a book's indexed chunk text as one string for phrase extraction.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose chunk text is loaded.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The book's chunk text joined into a single string.
|
||||||
|
"""
|
||||||
|
texts = await session.scalars(
|
||||||
|
select(EbookChunk.text).where(EbookChunk.source_id == book_id).order_by(EbookChunk.chunk_index)
|
||||||
|
)
|
||||||
|
return "\n\n".join(stripped for text in texts if (stripped := text.strip()))
|
||||||
|
|
||||||
|
|
||||||
|
async def load_book_chapter_texts(session: AsyncSession, book_id: int) -> list[str]:
|
||||||
|
"""Reconstruct chapter-like text blocks from indexed chunks for phrase extraction.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose chunks are grouped into chapters.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[str]: Non-empty chapter-like text blocks in chunk order.
|
||||||
|
"""
|
||||||
|
rows = await session.execute(
|
||||||
|
select(EbookChunk.chapter_id, EbookChunk.text)
|
||||||
|
.where(EbookChunk.source_id == book_id)
|
||||||
|
.order_by(EbookChunk.chunk_index)
|
||||||
|
)
|
||||||
|
chapters: list[str] = []
|
||||||
|
current_chapter_id: int | None = None
|
||||||
|
current_parts: list[str] = []
|
||||||
|
have_current = False
|
||||||
|
|
||||||
|
for chapter_id, text in rows:
|
||||||
|
if have_current and chapter_id != current_chapter_id:
|
||||||
|
chapter_text = "\n\n".join(current_parts).strip()
|
||||||
|
if chapter_text:
|
||||||
|
chapters.append(chapter_text)
|
||||||
|
current_parts = []
|
||||||
|
current_chapter_id = chapter_id
|
||||||
|
current_parts.append(str(text))
|
||||||
|
have_current = True
|
||||||
|
|
||||||
|
if current_parts:
|
||||||
|
chapter_text = "\n\n".join(current_parts).strip()
|
||||||
|
if chapter_text:
|
||||||
|
chapters.append(chapter_text)
|
||||||
|
return chapters
|
||||||
|
|
||||||
|
|
||||||
|
def metadata_for_source(source: EbookSource) -> dict[str, object | None]:
|
||||||
|
"""Return phrase extraction metadata for one indexed source.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
source (EbookSource): Indexed source to read metadata from.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, object | None]: Title, author, language, publisher, and identifier values.
|
||||||
|
"""
|
||||||
|
return {
|
||||||
|
"title": source.title,
|
||||||
|
"author": source.author,
|
||||||
|
"language": source.language,
|
||||||
|
"publisher": source.publisher,
|
||||||
|
"identifier": source.identifier,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def metadata_for_source_id(session: AsyncSession, source_id: int) -> dict[str, object | None]:
|
||||||
|
"""Return phrase extraction metadata for one indexed source by id.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
source_id (int): Id of the indexed source to read metadata from.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, object | None]: Title, author, language, publisher, and identifier values.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If no source exists with the given id.
|
||||||
|
"""
|
||||||
|
source = await session.get(EbookSource, source_id)
|
||||||
|
if source is None:
|
||||||
|
msg = f"No indexed source with id {source_id}"
|
||||||
|
raise ValueError(msg)
|
||||||
|
return metadata_for_source(source)
|
||||||
|
|
||||||
|
|
||||||
|
async def count_protected_phrases(session: AsyncSession, book_id: int) -> int:
|
||||||
|
"""Count stored protected phrases for one book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose protected phrases are counted.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of protected phrases stored for the book.
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
await session.scalars(
|
||||||
|
select(func.count(EbookProtectedPhrase.id)).where(EbookProtectedPhrase.book_id == book_id)
|
||||||
|
)
|
||||||
|
).one()
|
||||||
|
|
||||||
|
|
||||||
|
async def count_unjudged_candidates(session: AsyncSession, book_id: int, config: EbookSearchConfig) -> int:
|
||||||
|
"""Count storable candidate rows for a book that have not yet been judged.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose unjudged candidates are counted.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings supplying storage thresholds.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of storable, unjudged candidate rows for the book.
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
await session.scalars(
|
||||||
|
select(func.count(EbookCandidatePhrase.id)).where(
|
||||||
|
EbookCandidatePhrase.book_id == book_id,
|
||||||
|
EbookCandidatePhrase.llm_judged.is_(False),
|
||||||
|
EbookCandidatePhrase.token_count >= config.phrase_min_tokens,
|
||||||
|
EbookCandidatePhrase.raw_count >= minimum_candidate_raw_count(config),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).one()
|
||||||
|
|
||||||
|
|
||||||
|
async def corpus_phrase_stats(session: AsyncSession) -> CorpusPhraseStats:
|
||||||
|
"""Summarize candidate and protected phrase coverage across the whole corpus.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
CorpusPhraseStats: Corpus-wide phrase counts and per-book coverage counts.
|
||||||
|
"""
|
||||||
|
total_books = (await session.scalars(select(func.count(EbookSource.id)))).one()
|
||||||
|
candidate_phrases, judged_candidates, books_with_candidates, books_with_unjudged = (
|
||||||
|
await session.execute(
|
||||||
|
select(
|
||||||
|
func.count(EbookCandidatePhrase.id),
|
||||||
|
func.count(EbookCandidatePhrase.id).filter(EbookCandidatePhrase.llm_judged.is_(True)),
|
||||||
|
func.count(func.distinct(EbookCandidatePhrase.book_id)),
|
||||||
|
func.count(func.distinct(EbookCandidatePhrase.book_id)).filter(
|
||||||
|
EbookCandidatePhrase.llm_judged.is_(False)
|
||||||
|
),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).one()
|
||||||
|
protected_phrases = (await session.scalars(select(func.count(EbookProtectedPhrase.id)))).one()
|
||||||
|
return CorpusPhraseStats(
|
||||||
|
total_books=total_books,
|
||||||
|
books_with_candidates=books_with_candidates,
|
||||||
|
books_fully_judged=books_with_candidates - books_with_unjudged,
|
||||||
|
candidate_phrases=candidate_phrases,
|
||||||
|
judged_candidates=judged_candidates,
|
||||||
|
unjudged_candidates=candidate_phrases - judged_candidates,
|
||||||
|
protected_phrases=protected_phrases,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def book_ids_pending_first_judgment(session: AsyncSession) -> list[int]:
|
||||||
|
"""Return books that have candidate phrases but no judged candidates yet.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[int]: Book ids with candidates where judging has never run, ordered by id.
|
||||||
|
"""
|
||||||
|
judged_books = select(EbookCandidatePhrase.book_id).where(EbookCandidatePhrase.llm_judged.is_(True)).distinct()
|
||||||
|
return list(
|
||||||
|
(
|
||||||
|
await session.scalars(
|
||||||
|
select(EbookCandidatePhrase.book_id)
|
||||||
|
.where(EbookCandidatePhrase.book_id.not_in(judged_books))
|
||||||
|
.distinct()
|
||||||
|
.order_by(EbookCandidatePhrase.book_id)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def load_candidates_for_judgment(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> Sequence[EbookCandidatePhrase]:
|
||||||
|
"""Load every storable unjudged candidate row for a book.
|
||||||
|
|
||||||
|
Rows may have been stored before the current junk filters and score weights existed, so
|
||||||
|
callers re-check :func:`is_junk_phrase` and rescore before selecting what to judge.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose candidates are loaded.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings supplying storage thresholds.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Sequence[EbookCandidatePhrase]: Storable, unjudged candidate rows ordered by stored score.
|
||||||
|
"""
|
||||||
|
query = (
|
||||||
|
select(EbookCandidatePhrase)
|
||||||
|
.where(
|
||||||
|
EbookCandidatePhrase.book_id == book_id,
|
||||||
|
EbookCandidatePhrase.llm_judged.is_(False),
|
||||||
|
EbookCandidatePhrase.token_count >= config.phrase_min_tokens,
|
||||||
|
EbookCandidatePhrase.raw_count >= minimum_candidate_raw_count(config),
|
||||||
|
)
|
||||||
|
.order_by(
|
||||||
|
EbookCandidatePhrase.candidate_score.desc(),
|
||||||
|
EbookCandidatePhrase.raw_count.desc(),
|
||||||
|
EbookCandidatePhrase.id,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return (await session.scalars(query)).all()
|
||||||
|
|
||||||
|
|
||||||
|
def phrase_candidate_from_row(row: EbookCandidatePhrase) -> PhraseCandidate:
|
||||||
|
"""Recreate an in-memory candidate from a persisted candidate row.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
row (EbookCandidatePhrase): Stored candidate row to convert.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseCandidate: An in-memory candidate mirroring the row's fields.
|
||||||
|
"""
|
||||||
|
return PhraseCandidate(
|
||||||
|
phrase_text=row.phrase_text,
|
||||||
|
phrase_norm=row.phrase_norm,
|
||||||
|
token_count=row.token_count,
|
||||||
|
source_raw_ngram=row.source_raw_ngram,
|
||||||
|
source_yake=row.source_yake,
|
||||||
|
source_spacy_ner=row.source_spacy_ner,
|
||||||
|
source_spacy_noun_chunk=row.source_spacy_noun_chunk,
|
||||||
|
source_capitalized=row.source_capitalized,
|
||||||
|
source_metadata=row.source_metadata,
|
||||||
|
spacy_label=row.spacy_label,
|
||||||
|
raw_count=row.raw_count,
|
||||||
|
chapter_count=row.chapter_count,
|
||||||
|
yake_score=row.yake_score,
|
||||||
|
candidate_score=row.candidate_score,
|
||||||
|
sample_contexts=row.sample_contexts or [],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def candidate_row_values(
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
candidate: PhraseCandidate,
|
||||||
|
*,
|
||||||
|
judgment: LLMJudgment | None,
|
||||||
|
) -> dict[str, object]:
|
||||||
|
"""Build the column values for one candidate phrase upsert.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_id (int): Book the candidate belongs to.
|
||||||
|
series_id (int | None): Series scope stored on the row.
|
||||||
|
candidate (PhraseCandidate): Candidate whose fields are written to the row.
|
||||||
|
judgment (LLMJudgment | None): Judgment to record, or ``None`` to leave the row unjudged.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict[str, object]: Column values keyed by column name.
|
||||||
|
"""
|
||||||
|
values: dict[str, object] = {
|
||||||
|
"book_id": book_id,
|
||||||
|
"phrase_norm": candidate.phrase_norm,
|
||||||
|
"series_id": series_id,
|
||||||
|
"phrase_text": candidate.phrase_text,
|
||||||
|
"token_count": candidate.token_count,
|
||||||
|
"source_raw_ngram": candidate.source_raw_ngram,
|
||||||
|
"source_yake": candidate.source_yake,
|
||||||
|
"source_spacy_ner": candidate.source_spacy_ner,
|
||||||
|
"source_spacy_noun_chunk": candidate.source_spacy_noun_chunk,
|
||||||
|
"source_capitalized": candidate.source_capitalized,
|
||||||
|
"source_metadata": candidate.source_metadata,
|
||||||
|
"spacy_label": candidate.spacy_label,
|
||||||
|
"raw_count": candidate.raw_count,
|
||||||
|
"chapter_count": candidate.chapter_count,
|
||||||
|
"yake_score": candidate.yake_score,
|
||||||
|
"candidate_score": candidate.candidate_score,
|
||||||
|
"llm_judged": judgment is not None,
|
||||||
|
}
|
||||||
|
if candidate.sample_contexts:
|
||||||
|
values["sample_contexts"] = list(candidate.sample_contexts)
|
||||||
|
if judgment is not None:
|
||||||
|
values.update(
|
||||||
|
llm_keep=judgment.keep,
|
||||||
|
llm_confidence=judgment.confidence,
|
||||||
|
llm_category=judgment.category,
|
||||||
|
llm_reason=judgment.reason,
|
||||||
|
)
|
||||||
|
return values
|
||||||
|
|
||||||
|
|
||||||
|
async def save_candidate_to_db(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
candidate: PhraseCandidate,
|
||||||
|
*,
|
||||||
|
judgment: LLMJudgment | None,
|
||||||
|
) -> EbookCandidatePhrase:
|
||||||
|
"""Insert or update one candidate phrase row.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book the candidate belongs to.
|
||||||
|
series_id (int | None): Series scope stored on the row.
|
||||||
|
candidate (PhraseCandidate): Candidate whose fields are written to the row.
|
||||||
|
judgment (LLMJudgment | None): Judgment to record, or ``None`` to leave the row unjudged.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
EbookCandidatePhrase: The inserted or updated candidate row.
|
||||||
|
"""
|
||||||
|
values = candidate_row_values(book_id, series_id, candidate, judgment=judgment)
|
||||||
|
|
||||||
|
# Preserve an existing judgment when this call is only refreshing candidate fields.
|
||||||
|
skip_update = {"book_id", "phrase_norm"}
|
||||||
|
if judgment is None:
|
||||||
|
skip_update.add("llm_judged")
|
||||||
|
insert_statement = dialect_insert(session, EbookCandidatePhrase).values(**values)
|
||||||
|
statement = insert_statement.on_conflict_do_update(
|
||||||
|
index_elements=["book_id", "phrase_norm"],
|
||||||
|
set_={column: insert_statement.excluded[column] for column in values if column not in skip_update},
|
||||||
|
).returning(EbookCandidatePhrase)
|
||||||
|
return (await session.scalars(statement, execution_options={"populate_existing": True})).one()
|
||||||
|
|
||||||
|
|
||||||
|
BULK_CANDIDATE_UPSERT_CHUNK = 1000
|
||||||
|
|
||||||
|
|
||||||
|
async def bulk_upsert_unjudged_candidates(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
candidates: Sequence[PhraseCandidate],
|
||||||
|
) -> int:
|
||||||
|
"""Insert or update many freshly extracted candidate rows in chunked multi-row upserts.
|
||||||
|
|
||||||
|
Saving one row per statement costs one database round trip per candidate, which dominated
|
||||||
|
generation time for full books, so candidates are written ``BULK_CANDIDATE_UPSERT_CHUNK``
|
||||||
|
rows per statement instead. Existing judgments and sample contexts are never overwritten:
|
||||||
|
fresh extractions carry no contexts, and ``llm_judged`` plus the ``llm_*`` columns are left
|
||||||
|
out of the conflict update. Candidates must have unique ``phrase_norm`` values, as produced
|
||||||
|
by extraction, since one multi-row upsert cannot touch the same row twice.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (Session): Active database session.
|
||||||
|
book_id (int): Book the candidates belong to.
|
||||||
|
series_id (int | None): Series scope stored on the rows.
|
||||||
|
candidates (Sequence[PhraseCandidate]): Freshly extracted candidates to persist.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of candidate rows written.
|
||||||
|
"""
|
||||||
|
values = [
|
||||||
|
candidate_row_values(book_id, series_id, candidate, judgment=None)
|
||||||
|
for candidate in candidates
|
||||||
|
if not candidate.sample_contexts
|
||||||
|
]
|
||||||
|
if len(values) != len(candidates):
|
||||||
|
msg = "bulk_upsert_unjudged_candidates only accepts freshly extracted candidates without sample contexts"
|
||||||
|
raise ValueError(msg)
|
||||||
|
skip_update = {"book_id", "phrase_norm", "llm_judged"}
|
||||||
|
for chunk_start in range(0, len(values), BULK_CANDIDATE_UPSERT_CHUNK):
|
||||||
|
chunk = values[chunk_start : chunk_start + BULK_CANDIDATE_UPSERT_CHUNK]
|
||||||
|
insert_statement = dialect_insert(session, EbookCandidatePhrase).values(chunk)
|
||||||
|
statement = insert_statement.on_conflict_do_update(
|
||||||
|
index_elements=["book_id", "phrase_norm"],
|
||||||
|
set_={column: insert_statement.excluded[column] for column in chunk[0] if column not in skip_update},
|
||||||
|
)
|
||||||
|
await session.execute(statement)
|
||||||
|
return len(values)
|
||||||
|
|
||||||
|
|
||||||
|
def new_candidate_row(book_id: int, series_id: int | None, candidate: PhraseCandidate) -> EbookCandidatePhrase:
|
||||||
|
"""Build a fresh unjudged candidate row without checking for an existing one.
|
||||||
|
|
||||||
|
Unlike :func:`save_candidate_to_db`, this does no lookup, so it is only safe when the caller
|
||||||
|
guarantees there is no existing row for ``(book_id, candidate.phrase_norm)`` — for example
|
||||||
|
right after :func:`delete_phrase_data_for_book` has cleared the book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
book_id (int): Book the candidate belongs to.
|
||||||
|
series_id (int | None): Series scope stored on the row.
|
||||||
|
candidate (PhraseCandidate): Candidate whose fields are written to the row.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
EbookCandidatePhrase: A new, unattached candidate row.
|
||||||
|
"""
|
||||||
|
row = EbookCandidatePhrase(book_id=book_id, phrase_norm=candidate.phrase_norm)
|
||||||
|
row.llm_judged = False
|
||||||
|
row.series_id = series_id
|
||||||
|
row.phrase_text = candidate.phrase_text
|
||||||
|
row.token_count = candidate.token_count
|
||||||
|
row.source_raw_ngram = candidate.source_raw_ngram
|
||||||
|
row.source_yake = candidate.source_yake
|
||||||
|
row.source_spacy_ner = candidate.source_spacy_ner
|
||||||
|
row.source_spacy_noun_chunk = candidate.source_spacy_noun_chunk
|
||||||
|
row.source_capitalized = candidate.source_capitalized
|
||||||
|
row.source_metadata = candidate.source_metadata
|
||||||
|
row.spacy_label = candidate.spacy_label
|
||||||
|
row.raw_count = candidate.raw_count
|
||||||
|
row.chapter_count = candidate.chapter_count
|
||||||
|
row.yake_score = candidate.yake_score
|
||||||
|
row.candidate_score = candidate.candidate_score
|
||||||
|
if candidate.sample_contexts:
|
||||||
|
row.sample_contexts = list(candidate.sample_contexts)
|
||||||
|
return row
|
||||||
|
|
||||||
|
|
||||||
|
async def upsert_protected_phrase(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
series_id: int | None,
|
||||||
|
candidate: PhraseCandidate,
|
||||||
|
judgment: LLMJudgment,
|
||||||
|
source_candidate: EbookCandidatePhrase,
|
||||||
|
) -> EbookProtectedPhrase:
|
||||||
|
"""Insert or update one accepted protected phrase and its aliases.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book the protected phrase belongs to.
|
||||||
|
series_id (int | None): Series scope stored on the phrase.
|
||||||
|
candidate (PhraseCandidate): Candidate the phrase was promoted from.
|
||||||
|
judgment (LLMJudgment): Accepted judgment supplying canonical text, category, and aliases.
|
||||||
|
source_candidate (EbookCandidatePhrase): Candidate row the phrase was promoted from.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
EbookProtectedPhrase: The inserted or updated protected phrase row.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: If the chosen phrase text normalizes to empty.
|
||||||
|
"""
|
||||||
|
phrase_text = judgment.canonical or candidate.phrase_text
|
||||||
|
phrase_norm = normalize_text(phrase_text)
|
||||||
|
if not phrase_norm:
|
||||||
|
msg = f"Protected phrase normalized to empty text: {phrase_text!r}"
|
||||||
|
raise ValueError(msg)
|
||||||
|
|
||||||
|
values = {
|
||||||
|
"book_id": book_id,
|
||||||
|
"phrase_norm": phrase_norm,
|
||||||
|
"series_id": series_id,
|
||||||
|
"phrase_text": phrase_text,
|
||||||
|
"canonical_id": make_canonical_id(judgment, phrase_norm),
|
||||||
|
"phrase_type": judgment.category,
|
||||||
|
"token_count": len(phrase_norm.split()),
|
||||||
|
"confidence": judgment.confidence,
|
||||||
|
"importance": judgment.importance,
|
||||||
|
"allow_nested": judgment.allow_nested,
|
||||||
|
"suppress_children": judgment.suppress_children,
|
||||||
|
"source_candidate_id": source_candidate.id,
|
||||||
|
}
|
||||||
|
insert_statement = dialect_insert(session, EbookProtectedPhrase).values(**values)
|
||||||
|
statement = insert_statement.on_conflict_do_update(
|
||||||
|
index_elements=["book_id", "phrase_norm"],
|
||||||
|
set_={
|
||||||
|
column: insert_statement.excluded[column] for column in values if column not in {"book_id", "phrase_norm"}
|
||||||
|
},
|
||||||
|
).returning(EbookProtectedPhrase)
|
||||||
|
row = (await session.scalars(statement, execution_options={"populate_existing": True})).one()
|
||||||
|
|
||||||
|
for alias_text in judgment.aliases:
|
||||||
|
await upsert_phrase_alias(session, row, alias_text)
|
||||||
|
return row
|
||||||
|
|
||||||
|
|
||||||
|
async def upsert_phrase_alias(
|
||||||
|
session: AsyncSession,
|
||||||
|
phrase: EbookProtectedPhrase,
|
||||||
|
alias_text: str,
|
||||||
|
) -> EbookPhraseAlias | None:
|
||||||
|
"""Insert or update one protected phrase alias.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
phrase (EbookProtectedPhrase): Protected phrase the alias points to.
|
||||||
|
alias_text (str): Alias surface form to store.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
EbookPhraseAlias | None: The alias row, or ``None`` when the alias is empty or equals the phrase.
|
||||||
|
"""
|
||||||
|
alias_norm = normalize_text(alias_text)
|
||||||
|
if not alias_norm or alias_norm == phrase.phrase_norm:
|
||||||
|
return None
|
||||||
|
|
||||||
|
insert_statement = dialect_insert(session, EbookPhraseAlias).values(
|
||||||
|
phrase_id=phrase.id,
|
||||||
|
alias_norm=alias_norm,
|
||||||
|
alias_text=alias_text,
|
||||||
|
confidence=1.0,
|
||||||
|
)
|
||||||
|
statement = insert_statement.on_conflict_do_update(
|
||||||
|
index_elements=["phrase_id", "alias_norm"],
|
||||||
|
set_={
|
||||||
|
"alias_text": insert_statement.excluded.alias_text,
|
||||||
|
"confidence": insert_statement.excluded.confidence,
|
||||||
|
},
|
||||||
|
).returning(EbookPhraseAlias)
|
||||||
|
return (await session.scalars(statement, execution_options={"populate_existing": True})).one()
|
||||||
|
|
||||||
|
|
||||||
|
def make_canonical_id(judgment: LLMJudgment, phrase_norm: str) -> str:
|
||||||
|
"""Create a deterministic canonical id from a judgment category and phrase.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
judgment (LLMJudgment): Judgment supplying the phrase category.
|
||||||
|
phrase_norm (str): Normalized phrase text to slugify.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: A ``category:slug`` canonical identifier.
|
||||||
|
"""
|
||||||
|
category = slugify_identifier(judgment.category or "phrase")
|
||||||
|
phrase_slug = slugify_identifier(phrase_norm)
|
||||||
|
return f"{category}:{phrase_slug}"
|
||||||
|
|
||||||
|
|
||||||
|
def slugify_identifier(value: str) -> str:
|
||||||
|
"""Normalize text for use inside a canonical id.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value (str): Text to slugify.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: A lowercase underscore slug, or ``"unknown"`` when empty.
|
||||||
|
"""
|
||||||
|
slug = re.sub(r"[^a-z0-9]+", "_", normalize_text(value).replace("'", ""))
|
||||||
|
return slug.strip("_") or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
async def prune_unstorable_unjudged_candidate_phrases(
|
||||||
|
session: AsyncSession,
|
||||||
|
book_id: int,
|
||||||
|
config: EbookSearchConfig,
|
||||||
|
) -> int:
|
||||||
|
"""Delete old unjudged candidate rows that no longer satisfy storage filters.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose stale candidates are pruned.
|
||||||
|
config (EbookSearchConfig): Runtime phrase-tuning settings supplying storage thresholds.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: Number of candidate rows deleted.
|
||||||
|
"""
|
||||||
|
deleted = rowcount(
|
||||||
|
await session.execute(
|
||||||
|
delete(EbookCandidatePhrase).where(
|
||||||
|
EbookCandidatePhrase.book_id == book_id,
|
||||||
|
EbookCandidatePhrase.llm_judged.is_(False),
|
||||||
|
or_(
|
||||||
|
EbookCandidatePhrase.token_count < config.phrase_min_tokens,
|
||||||
|
EbookCandidatePhrase.raw_count < minimum_candidate_raw_count(config),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if deleted:
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_unstorable_pruned book_id=%s deleted=%s min_tokens=%s min_uses=%s",
|
||||||
|
book_id,
|
||||||
|
deleted,
|
||||||
|
config.phrase_min_tokens,
|
||||||
|
minimum_candidate_raw_count(config),
|
||||||
|
)
|
||||||
|
return deleted
|
||||||
|
|
||||||
|
|
||||||
|
async def delete_phrase_data_for_book(session: AsyncSession, book_id: int) -> PhraseRecalculationResult:
|
||||||
|
"""Delete all candidate, protected, alias, and mention phrase data for one book.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
session (AsyncSession): Active database session.
|
||||||
|
book_id (int): Book whose phrase data is deleted.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
PhraseRecalculationResult: Deleted-row counts with ``candidate_phrases`` set to 0.
|
||||||
|
"""
|
||||||
|
protected_ids = (
|
||||||
|
await session.scalars(select(EbookProtectedPhrase.id).where(EbookProtectedPhrase.book_id == book_id))
|
||||||
|
).all()
|
||||||
|
deleted_aliases = 0
|
||||||
|
if protected_ids:
|
||||||
|
deleted_aliases = rowcount(
|
||||||
|
await session.execute(delete(EbookPhraseAlias).where(EbookPhraseAlias.phrase_id.in_(protected_ids)))
|
||||||
|
)
|
||||||
|
|
||||||
|
deleted_mentions = rowcount(
|
||||||
|
await session.execute(delete(EbookChunkPhraseMention).where(EbookChunkPhraseMention.book_id == book_id))
|
||||||
|
)
|
||||||
|
if protected_ids:
|
||||||
|
deleted_mentions += rowcount(
|
||||||
|
await session.execute(
|
||||||
|
delete(EbookChunkPhraseMention).where(EbookChunkPhraseMention.phrase_id.in_(protected_ids))
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
deleted_protected = rowcount(
|
||||||
|
await session.execute(delete(EbookProtectedPhrase).where(EbookProtectedPhrase.book_id == book_id))
|
||||||
|
)
|
||||||
|
deleted_candidates = rowcount(
|
||||||
|
await session.execute(delete(EbookCandidatePhrase).where(EbookCandidatePhrase.book_id == book_id))
|
||||||
|
)
|
||||||
|
await session.flush()
|
||||||
|
logger.info(
|
||||||
|
"ebook_candidate_phrase_data_deleted book_id=%s candidates=%s protected=%s aliases=%s mentions=%s",
|
||||||
|
book_id,
|
||||||
|
deleted_candidates,
|
||||||
|
deleted_protected,
|
||||||
|
deleted_aliases,
|
||||||
|
deleted_mentions,
|
||||||
|
)
|
||||||
|
return PhraseRecalculationResult(
|
||||||
|
book_id=book_id,
|
||||||
|
deleted_candidates=deleted_candidates,
|
||||||
|
deleted_protected_phrases=deleted_protected,
|
||||||
|
deleted_aliases=deleted_aliases,
|
||||||
|
deleted_mentions=deleted_mentions,
|
||||||
|
candidate_phrases=0,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def rowcount(result: object) -> int:
|
||||||
|
"""Return a safe integer rowcount from a SQLAlchemy execution result.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
result (object): SQLAlchemy execution result that may expose ``rowcount``.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
int: The result's rowcount, or 0 when it is missing or negative.
|
||||||
|
"""
|
||||||
|
count = getattr(result, "rowcount", 0)
|
||||||
|
return int(count if count is not None and count >= 0 else 0)
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
"""Protected phrase extraction, storage, and runtime matching."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
JSON_OBJECT_RE = re.compile(r"\{.*\}", re.DOTALL)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class NormalizedToken:
|
||||||
|
"""A normalized token plus its source character span."""
|
||||||
|
|
||||||
|
text: str
|
||||||
|
start_char: int
|
||||||
|
end_char: int
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_text(text: str) -> str:
|
||||||
|
"""Normalize text for phrase storage and lookup.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text (str): Raw text to normalize.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: Normalized tokens joined by single spaces.
|
||||||
|
"""
|
||||||
|
return " ".join(token.text for token in tokenize_with_offsets(text))
|
||||||
|
|
||||||
|
|
||||||
|
def tokenize(text: str) -> list[str]:
|
||||||
|
"""Normalize and split text into phrase-detection tokens.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text (str): Raw text to tokenize.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[str]: Normalized token strings.
|
||||||
|
"""
|
||||||
|
return [token.text for token in tokenize_with_offsets(text)]
|
||||||
|
|
||||||
|
|
||||||
|
def tokenize_with_offsets(text: str) -> list[NormalizedToken]:
|
||||||
|
"""Normalize text into tokens while preserving original character offsets.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text (str): Raw text to tokenize.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[NormalizedToken]: Normalized tokens with their source character spans.
|
||||||
|
"""
|
||||||
|
tokens: list[NormalizedToken] = []
|
||||||
|
current: list[str] = []
|
||||||
|
start_char: int | None = None
|
||||||
|
|
||||||
|
for index, char in enumerate(text):
|
||||||
|
normalized = normalize_char(char)
|
||||||
|
if normalized == " ":
|
||||||
|
if current and start_char is not None:
|
||||||
|
tokens.append(NormalizedToken(text="".join(current), start_char=start_char, end_char=index))
|
||||||
|
current = []
|
||||||
|
start_char = None
|
||||||
|
continue
|
||||||
|
if start_char is None:
|
||||||
|
start_char = index
|
||||||
|
current.append(normalized)
|
||||||
|
|
||||||
|
if current and start_char is not None:
|
||||||
|
tokens.append(NormalizedToken(text="".join(current), start_char=start_char, end_char=len(text)))
|
||||||
|
return tokens
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_char(char: str) -> str:
|
||||||
|
"""Normalize one character into a token character or a separator.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
char (str): Single source character to normalize.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The normalized token character, or a space acting as a separator.
|
||||||
|
"""
|
||||||
|
if char in {"\u2019", "\u2018"}:
|
||||||
|
return "'"
|
||||||
|
if char in {"-", "\u2013", "\u2014"}:
|
||||||
|
return " "
|
||||||
|
|
||||||
|
lowered = char.lower()
|
||||||
|
if lowered in "abcdefghijklmnopqrstuvwxyz0123456789'":
|
||||||
|
return lowered
|
||||||
|
return " "
|
||||||
@@ -0,0 +1,140 @@
|
|||||||
|
"""vLLM-backed optional reranking."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, replace
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from python.ebook_search.llm_interface import request_rerank
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from python.ebook_search.config import RerankConfig
|
||||||
|
from python.ebook_search.search import SearchResult
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RerankResult:
|
||||||
|
"""A relevance score for one candidate chunk."""
|
||||||
|
|
||||||
|
chunk_id: int
|
||||||
|
score: float
|
||||||
|
|
||||||
|
|
||||||
|
async def rerank_chunks(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
query: str,
|
||||||
|
candidates: list[SearchResult],
|
||||||
|
config: RerankConfig,
|
||||||
|
) -> list[SearchResult]:
|
||||||
|
"""Rerank candidates with a vLLM rerank endpoint."""
|
||||||
|
if not candidates:
|
||||||
|
return []
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"ebook_rerank_request_start base_url=%s model=%s candidates=%s",
|
||||||
|
config.base_url,
|
||||||
|
config.model,
|
||||||
|
len(candidates),
|
||||||
|
)
|
||||||
|
scores = await score_candidates(client, query, candidates, config)
|
||||||
|
results = sorted(
|
||||||
|
(
|
||||||
|
replace(
|
||||||
|
result,
|
||||||
|
score=final_rerank_score(result, scores[result.chunk_id].score, candidates, config),
|
||||||
|
rerank_score=scores[result.chunk_id].score,
|
||||||
|
)
|
||||||
|
for result in candidates
|
||||||
|
),
|
||||||
|
key=lambda result: result.score,
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"ebook_rerank_request_complete base_url=%s model=%s candidates=%s",
|
||||||
|
config.base_url,
|
||||||
|
config.model,
|
||||||
|
len(results),
|
||||||
|
)
|
||||||
|
return results
|
||||||
|
|
||||||
|
|
||||||
|
async def score_candidates(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
query: str,
|
||||||
|
candidates: list[SearchResult],
|
||||||
|
config: RerankConfig,
|
||||||
|
) -> dict[int, RerankResult]:
|
||||||
|
"""Score candidate chunks with the configured rerank API."""
|
||||||
|
body = await request_rerank(client, query, [candidate.text for candidate in candidates], config)
|
||||||
|
if body is None:
|
||||||
|
return zero_rerank_scores(candidates)
|
||||||
|
|
||||||
|
scores = parse_vllm_scores(body, candidates)
|
||||||
|
for result in scores.values():
|
||||||
|
logger.debug("ebook_rerank_candidate_scored chunk_id=%s score=%s", result.chunk_id, result.score)
|
||||||
|
return scores
|
||||||
|
|
||||||
|
|
||||||
|
def parse_vllm_scores(body: object, candidates: list[SearchResult]) -> dict[int, RerankResult]:
|
||||||
|
"""Parse vLLM rerank scores into chunk-id keyed results."""
|
||||||
|
if not isinstance(body, dict):
|
||||||
|
logger.debug("ebook_rerank_response_not_object", extra={"response": body})
|
||||||
|
return zero_rerank_scores(candidates)
|
||||||
|
|
||||||
|
results = body.get("results") or body.get("data")
|
||||||
|
if not isinstance(results, list):
|
||||||
|
logger.debug("ebook_rerank_response_missing_results", extra={"response": body})
|
||||||
|
return zero_rerank_scores(candidates)
|
||||||
|
|
||||||
|
scores = zero_rerank_scores(candidates)
|
||||||
|
for item in results:
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
continue
|
||||||
|
index = item.get("index")
|
||||||
|
score = item.get("relevance_score", item.get("score"))
|
||||||
|
if not isinstance(index, int) or index < 0 or index >= len(candidates):
|
||||||
|
continue
|
||||||
|
if not isinstance(score, int | float):
|
||||||
|
continue
|
||||||
|
chunk_id = candidates[index].chunk_id
|
||||||
|
scores[chunk_id] = RerankResult(chunk_id=chunk_id, score=clamp_score(float(score)))
|
||||||
|
return scores
|
||||||
|
|
||||||
|
|
||||||
|
def zero_rerank_scores(candidates: list[SearchResult]) -> dict[int, RerankResult]:
|
||||||
|
"""Return zero relevance scores for all candidate chunks."""
|
||||||
|
return {candidate.chunk_id: RerankResult(chunk_id=candidate.chunk_id, score=0.0) for candidate in candidates}
|
||||||
|
|
||||||
|
|
||||||
|
def clamp_score(score: float) -> float:
|
||||||
|
"""Clamp a rerank score into the supported 0.0 to 1.0 range."""
|
||||||
|
return min(max(score, 0.0), 1.0)
|
||||||
|
|
||||||
|
|
||||||
|
def final_rerank_score(
|
||||||
|
result: SearchResult,
|
||||||
|
rerank_score: float,
|
||||||
|
candidates: list[SearchResult],
|
||||||
|
config: RerankConfig,
|
||||||
|
) -> float:
|
||||||
|
"""Combine rerank relevance with normalized hybrid retrieval evidence."""
|
||||||
|
return (config.score_weight * rerank_score) + (config.hybrid_weight * normalized_hybrid_score(result, candidates))
|
||||||
|
|
||||||
|
|
||||||
|
def normalized_hybrid_score(result: SearchResult, candidates: list[SearchResult]) -> float:
|
||||||
|
"""Normalize a candidate hybrid score against the rerank candidate set."""
|
||||||
|
hybrid_scores = [
|
||||||
|
candidate.fused_score if candidate.fused_score is not None else candidate.score for candidate in candidates
|
||||||
|
]
|
||||||
|
low = min(hybrid_scores)
|
||||||
|
high = max(hybrid_scores)
|
||||||
|
if high == low:
|
||||||
|
return 1.0
|
||||||
|
|
||||||
|
score = result.fused_score if result.fused_score is not None else result.score
|
||||||
|
return (score - low) / (high - low)
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user