Compare commits
379
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
247e951a27 | ||
|
|
8c3de690c9 | ||
|
|
91575e3ab2 | ||
|
|
4799054601 | ||
|
|
e9636d19de | ||
|
|
2518ec8551 | ||
|
|
f5ec88f1e5 | ||
|
|
f4c4b11ff8 | ||
|
|
37f37c41ac | ||
|
|
80a521f297 | ||
|
|
4a410dbdf8 | ||
|
|
138b79bf97 | ||
|
|
47f77443dd | ||
|
|
8548f4ff64 | ||
|
|
682960b4c9 | ||
|
|
ddea13b93a | ||
|
|
b252b0bf6e | ||
|
|
fbf288649a | ||
|
|
db199f884d | ||
|
|
c6645eb37c | ||
|
|
3984081753 | ||
|
|
03f610feed | ||
|
|
8804a8df83 | ||
|
|
8daac4d98b | ||
|
|
7606ddc514 | ||
|
|
137bd8ed2f | ||
|
|
f936253ad5 | ||
|
|
fc01fb9d0b | ||
|
|
09d963ba34 | ||
|
|
855af0fdff | ||
|
|
fcb69cc68b | ||
|
|
6bc30115d9 | ||
|
|
6ae1ff1f5c | ||
|
|
dbc6b5b53b | ||
|
|
a9daa60c17 | ||
|
|
241a42b20d | ||
|
|
4640ebf8ce | ||
|
|
0c9583c1cc | ||
|
|
f71ae7d2c6 | ||
|
|
2e68c83021 | ||
|
|
b126987b63 | ||
|
|
68b3a38b81 | ||
|
|
a5d7c3be4f | ||
|
|
2995a75748 | ||
|
|
dce1838163 | ||
|
|
121eb979a4 | ||
|
|
c88315e9b6 | ||
|
|
50795ab7fc | ||
|
|
e9a80a0308 | ||
|
|
eb76edb740 | ||
|
|
c53afb3c70 | ||
|
|
8301db39e5 | ||
|
|
45a9e90524 | ||
|
|
dd67de3993 | ||
|
|
6e3635ca01 | ||
|
|
11cbe31152 | ||
|
|
5544dfa61c | ||
|
|
911df63513 | ||
|
|
07eb170b34 | ||
|
|
6715bbf0a5 | ||
|
|
26ff1f0fd3 | ||
|
|
666ea97754 | ||
|
|
e01c687625 | ||
|
|
82a367a2b6 | ||
|
|
ad1834c537 | ||
|
|
aed1e14d95 | ||
|
|
0a2d4c08cb | ||
|
|
db98bd3559 | ||
|
|
cdded5da12 | ||
|
|
d022251a58 | ||
|
|
e1ef4de6a3 | ||
|
|
5c230a267c | ||
|
|
5215d66d40 | ||
|
|
7ad198416b | ||
|
|
1461c2552a | ||
|
|
736717c2f8 | ||
|
|
ab2521867e | ||
|
|
8e0ab4190b | ||
|
|
734fd7641e | ||
|
|
e898e08c48 | ||
|
|
d916ea903c | ||
|
|
d8e916dbe6 | ||
|
|
48e9f0199d | ||
|
|
a526420c8d | ||
|
|
41e3e265af | ||
|
|
38a17f6146 | ||
|
|
fe48d4c1ad | ||
|
|
9290cb46ee | ||
|
|
acd3f2d3ac | ||
|
|
08e716f66a | ||
|
|
d197731af4 | ||
|
|
1ffc48bb02 | ||
|
|
b6395ef18f | ||
|
|
aff6f4e1bd | ||
|
|
a9a96db944 | ||
|
|
d34154541d | ||
|
|
5d3a851137 | ||
|
|
e05e5c77bc | ||
|
|
b0a2ebc052 | ||
|
|
f77c9657a3 | ||
|
|
f908f969d3 | ||
|
|
3cf49c5479 | ||
|
|
b34354f5e5 | ||
|
|
44826464de | ||
|
|
3de0ffccb0 | ||
|
|
c6c98b3e26 | ||
|
|
d459f3d675 | ||
|
|
33e4b37cce | ||
|
|
2a8e7e7f2b | ||
|
|
07759353be | ||
|
|
38fb14520e | ||
|
|
006ae6079a | ||
|
|
7d507fb7e1 | ||
|
|
0f69022e51 | ||
|
|
a260ae2470 | ||
|
|
820b4a53d2 | ||
|
|
ea77e83f06 | ||
|
|
a9da208bc3 | ||
|
|
739d7dd28c | ||
|
|
651599796e | ||
|
|
b9d440597c | ||
|
|
311cc5d7a7 | ||
|
|
fb2519046d | ||
|
|
bc6b1585ec | ||
|
|
d71330a85a | ||
|
|
df51aa5200 | ||
|
|
e93cc816db | ||
|
|
19050b4cf4 | ||
|
|
6676c15f75 | ||
|
|
27e487e322 | ||
|
|
4f28050eff | ||
|
|
b58ea60557 | ||
|
|
e95eedffe4 | ||
|
|
1abd53987c | ||
|
|
d1a3e7338a | ||
|
|
687ef0c167 | ||
|
|
3a86148352 | ||
|
|
fe9a2912e1 | ||
|
|
29a99fc210 | ||
|
|
d7651bf588 | ||
|
|
2865dcbe9c | ||
|
|
d920b77bab | ||
|
|
1b53167b53 | ||
|
|
9dabb9dc07 | ||
|
|
95630fe151 | ||
|
|
d3a889f100 | ||
|
|
6ce0671f51 | ||
|
|
25ab6b2ab6 | ||
|
|
374d7e8d38 | ||
|
|
957110b7e9 | ||
|
|
e7dc60f2c3 | ||
|
|
353a9d6787 | ||
|
|
9f2d3a3c89 | ||
|
|
73e221716f | ||
|
|
0d0ed5445a | ||
|
|
9e4c6f6f56 | ||
|
|
1cf4b99d18 | ||
|
|
b536fb9f09 | ||
|
|
c41a2ce3bd | ||
|
|
8ef776f859 | ||
|
|
d350c2d074 | ||
|
|
93d6914e9d | ||
|
|
7db063a240 | ||
|
|
dfe5997e0b | ||
|
|
68671a1e84 | ||
|
|
bcc2227cfd | ||
|
|
d6eec926e7 | ||
|
|
5ddf1c4cab | ||
|
|
5a2171b9c7 | ||
|
|
95c6ade154 | ||
|
|
a0bbc2896a | ||
|
|
736596c387 | ||
|
|
67622c0e51 | ||
|
|
d2f447a1af | ||
|
|
af365fce9a | ||
|
|
6430049e92 | ||
|
|
26e4620f8f | ||
|
|
93fc700fa2 | ||
|
|
8d1c1fc628 | ||
|
|
dda318753b | ||
|
|
261ff139f7 | ||
|
|
ba8ff35109 | ||
|
|
e368402eea | ||
|
|
dd9329d218 | ||
|
|
89f6627bed | ||
|
|
c5babf8bad | ||
|
|
dae38ffd9b | ||
|
|
ca62cc36a7 | ||
|
|
035410f39e | ||
|
|
e40ab757ca | ||
|
|
345ba94a59 | ||
|
|
f2084206b6 | ||
|
|
50e764146a | ||
|
|
ea97b5eb19 | ||
|
|
1ef2512daa | ||
|
|
f9a9e5395c | ||
|
|
d8e166a340 | ||
|
|
c266ba79f4 | ||
|
|
f627a5ac6e | ||
|
|
a5e7d97213 | ||
|
|
1419deb3c6 | ||
|
|
1f06692696 | ||
|
|
8f8177f36e | ||
|
|
8534edc285 | ||
|
|
73b28a855b | ||
|
|
0c0810a06b | ||
|
|
239bef975a | ||
|
|
2577b791f7 | ||
|
|
b4d9562591 | ||
|
|
66f972ac2b | ||
|
|
aca756f479 | ||
|
|
7f59f7f7ac | ||
|
|
70864c620f | ||
|
|
304f1c8433 | ||
|
|
1b5a036061 | ||
|
|
42330ec186 | ||
|
|
3f4373d1f6 | ||
|
|
cc73dfc467 | ||
|
|
976c3f9d3e | ||
|
|
2661127426 | ||
|
|
1b3e6725ea | ||
|
|
7d2fbaea43 | ||
|
|
a19b1c7e60 | ||
|
|
76da6cbc54 | ||
|
|
c83bbe2c24 | ||
|
|
7611a3b2df | ||
|
|
aec5e3e22b | ||
|
|
4e3273d5ec | ||
|
|
b5ee7c2dc2 | ||
|
|
958b06ecf0 | ||
|
|
71ad8ab29e | ||
|
|
852759c510 | ||
|
|
d684d5d62c | ||
|
|
f1e394565d | ||
|
|
754ced4822 | ||
|
|
5b054dfc8f | ||
|
|
663833d4fa | ||
|
|
433ec9a38e | ||
|
|
3a3267ee9a | ||
|
|
0497a50a43 | ||
|
|
6365dd8067 | ||
|
|
a6fbbd245f | ||
|
|
7ad321e5e2 | ||
|
|
14338e34df | ||
|
|
c73aa5c98a | ||
|
|
f762f12bd2 | ||
|
|
ab5df442c6 | ||
|
|
f11c9bed58 | ||
|
|
ab2d8dbd51 | ||
|
|
42ede19472 | ||
|
|
f4f33eacc4 | ||
|
|
51f6cd23ad | ||
|
|
3dadb145b7 | ||
|
|
75a67294ea | ||
|
|
58b25f2e89 | ||
|
|
568bf8dd38 | ||
|
|
82851eb287 | ||
|
|
b7bce0bcb9 | ||
|
|
583af965ad | ||
|
|
ec80bf1c5f | ||
|
|
bd490334f5 | ||
|
|
e893ea0f57 | ||
|
|
18f149b831 | ||
|
|
69f5b87e5f | ||
|
|
66acc010ca | ||
|
|
e8f3a563be | ||
|
|
8f1d765cad | ||
|
|
4f0ba687c4 | ||
|
|
27891c3903 | ||
|
|
ccdc61b4dd | ||
|
|
1d732bf41c | ||
|
|
13ba118cfc | ||
|
|
47c6f42d2f | ||
|
|
ff9dcde5d9 | ||
|
|
7de800b519 | ||
|
|
55767ad555 | ||
|
|
c262ff9048 | ||
|
|
9abac2978a | ||
|
|
70d20e55d2 | ||
|
|
f038f248a1 | ||
|
|
af828fc9c4 | ||
|
|
4d121ae9f9 | ||
|
|
959d599ff9 | ||
|
|
d470243fdd | ||
|
|
d96c93fa17 | ||
|
|
6bea380e3d | ||
|
|
56c933c8cb | ||
|
|
e7dae1eb4b | ||
|
|
17ebe50ac9 | ||
|
|
97b35ce27b | ||
|
|
595579fe8b | ||
|
|
fcfbce4e16 | ||
|
|
80af3377e6 | ||
|
|
557c1a4d5d | ||
|
|
89e37249af | ||
|
|
ccd523b4d0 | ||
|
|
606035432b | ||
|
|
4d2f6831e3 | ||
|
|
86e72d1da0 | ||
|
|
139727bf50 | ||
|
|
88c2f1b139 | ||
|
|
e75a3ef9c6 | ||
|
|
258f918794 | ||
|
|
cf4635922e | ||
|
|
0615ece46a | ||
|
|
8afa4fce6c | ||
|
|
8bbcd37933 | ||
|
|
037b2f9cf7 | ||
|
|
7dbc4c248f | ||
|
|
08dffc6f6d | ||
|
|
0109167b10 | ||
|
|
b87f6b0b34 | ||
|
|
35376c3fca | ||
|
|
0c218f2551 | ||
|
|
d0b66496a1 | ||
|
|
5101da4914 | ||
|
|
393545868f | ||
|
|
6bb7904782 | ||
|
|
59147834f7 | ||
|
|
52235239d0 | ||
|
|
9e43c3e8b8 | ||
|
|
156d624d81 | ||
|
|
9a7cf03a00 | ||
|
|
6299d42f75 | ||
|
|
e6472b2cf5 | ||
|
|
41d3a8fe1a | ||
|
|
e6ac8f8021 | ||
|
|
0f8f6f96d6 | ||
|
|
4cb4bd6f3d | ||
|
|
c046710258 | ||
|
|
7f9fbe3602 | ||
|
|
8ee3b4d6e5 | ||
|
|
18b7fb2d60 | ||
|
|
2f1fa5c750 | ||
|
|
164d0dd59e | ||
|
|
d4459643ab | ||
|
|
c09dba0c37 | ||
|
|
409f376166 | ||
|
|
a9a6e1f932 | ||
|
|
6472f07a88 | ||
|
|
51c79f6b40 | ||
|
|
b0d5147296 | ||
|
|
c56082b516 | ||
|
|
34b728c88f | ||
|
|
5697458bad | ||
|
|
276c2ac74b | ||
|
|
69e5aa20d5 | ||
|
|
3d1f773fa5 | ||
|
|
14dd1fe52e | ||
|
|
30fe41ea1b | ||
|
|
3a17c5514d | ||
|
|
c6586db91e | ||
|
|
81b199373e | ||
|
|
a957e23041 | ||
|
|
52389f729d | ||
|
|
cc2a609f52 | ||
|
|
ca4693a1ba | ||
|
|
90e5e0855d | ||
|
|
e339667c2b | ||
|
|
85540ee920 | ||
|
|
3be1b8aa8f | ||
|
|
7c56954cda | ||
|
|
290f972346 | ||
|
|
72c3ccfb6d | ||
|
|
9630633ff5 | ||
|
|
8c83f306b2 | ||
|
|
5b4609dc3b | ||
|
|
d1be25c6e8 | ||
|
|
31910586d2 | ||
|
|
b8dfd0852a | ||
|
|
6ce622e93e | ||
|
|
55e652a51d | ||
|
|
b5455a5483 | ||
|
|
8baf388061 | ||
|
|
7ffb7b4a37 | ||
|
|
eb04f4a56d | ||
|
|
5b8e543226 | ||
|
|
da48f62195 | ||
|
|
60f2ab1039 |
@@ -0,0 +1,28 @@
|
||||
.git
|
||||
.direnv
|
||||
.mypy_cache
|
||||
.pytest_cache
|
||||
.ruff_cache
|
||||
.venv
|
||||
**/.venv
|
||||
.env
|
||||
.cache
|
||||
.claude
|
||||
.coverage
|
||||
.vscode
|
||||
.stfolder
|
||||
.literotica_data
|
||||
esphome
|
||||
htmlcov
|
||||
data
|
||||
ebooks
|
||||
__pycache__
|
||||
**/__pycache__
|
||||
*.pyc
|
||||
*.pyo
|
||||
.ebook_search_bm25
|
||||
result
|
||||
result-*
|
||||
*.egg-info
|
||||
dist
|
||||
build
|
||||
@@ -17,12 +17,11 @@ jobs:
|
||||
- "bob"
|
||||
- "brain"
|
||||
- "jeeves"
|
||||
- "leviathan"
|
||||
- "rhapsody-in-green"
|
||||
continue-on-error: true
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Build default package
|
||||
run: "nixos-rebuild build --flake ./#${{ matrix.system }}"
|
||||
run: "nixos-rebuild build --accept-flake-config --flake ./#${{ matrix.system }}"
|
||||
- name: copy to nix-cache
|
||||
run: nix copy --to ssh://jeeves .#nixosConfigurations.${{ matrix.system }}.config.system.build.toplevel
|
||||
run: nix copy --accept-flake-config --to unix:///host-nix/var/nix/daemon-socket/socket .#nixosConfigurations.${{ matrix.system }}.config.system.build.toplevel
|
||||
|
||||
@@ -6,24 +6,18 @@ on:
|
||||
|
||||
jobs:
|
||||
merge:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: self-hosted
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: merge_flake_lock_update
|
||||
run: |
|
||||
pr_number=$(gh pr list --state open --author RichieCahill --label flake_lock_update --json number --jq '.[0].number')
|
||||
echo "pr_number=$pr_number" >> $GITHUB_ENV
|
||||
if [ -n "$pr_number" ]; then
|
||||
gh pr merge "$pr_number" --rebase
|
||||
else
|
||||
echo "No open PR found with label flake_lock_update"
|
||||
fi
|
||||
run: >-
|
||||
nix develop .#devShells.x86_64-linux.default -c
|
||||
python -m python.gitea_flake_lock merge
|
||||
--repo "${{ github.repository }}"
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GH_TOKEN_FOR_UPDATES }}
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_URL: https://gitea.tmmworkshop.com
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
name: pytest
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
merge_group:
|
||||
|
||||
jobs:
|
||||
pytest:
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
name: test ebook search
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
|
||||
env:
|
||||
UV_PYTHON_DOWNLOADS: never
|
||||
UV_CACHE_DIR: /var/cache/uv
|
||||
UV_LINK_MODE: copy
|
||||
|
||||
jobs:
|
||||
test-ebook-search:
|
||||
runs-on: self-hosted
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install dependencies
|
||||
run: nix develop .#devShells.x86_64-linux.ebook-search -c uv sync --locked --project python/ebook_search/docker
|
||||
- name: Run ebook search tests
|
||||
run: nix develop .#devShells.x86_64-linux.ebook-search -c uv run --project python/ebook_search/docker --no-sync pytest tests/ebook_search --override-ini addopts="-n auto -ra"
|
||||
@@ -6,18 +6,21 @@ on:
|
||||
|
||||
jobs:
|
||||
lockfile:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: self-hosted
|
||||
permissions:
|
||||
actions: write
|
||||
contents: write
|
||||
pull-requests: write
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
- name: Install Nix
|
||||
uses: DeterminateSystems/nix-installer-action@main
|
||||
- name: Update flake.lock
|
||||
uses: DeterminateSystems/update-flake-lock@main
|
||||
with:
|
||||
token: ${{ secrets.GH_TOKEN_FOR_UPDATES }}
|
||||
pr-title: "Update flake.lock"
|
||||
pr-labels: |
|
||||
dependencies
|
||||
automated
|
||||
flake_lock_update
|
||||
run: nix flake update
|
||||
- name: Create or update flake.lock PR
|
||||
env:
|
||||
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN }}
|
||||
GITEA_URL: https://gitea.tmmworkshop.com
|
||||
run: >-
|
||||
nix develop .#devShells.x86_64-linux.default -c
|
||||
python -m python.gitea_flake_lock update
|
||||
--repo "${{ github.repository }}"
|
||||
|
||||
@@ -165,3 +165,11 @@ test.*
|
||||
|
||||
# syncthing
|
||||
.stfolder
|
||||
|
||||
# Frontend build output
|
||||
frontend/dist/
|
||||
frontend/node_modules/
|
||||
|
||||
# data from testing llms
|
||||
data/*
|
||||
.ebook_search_bm25
|
||||
|
||||
@@ -7,7 +7,6 @@ keys:
|
||||
- &system_bob age1q47vup0tjhulkg7d6xwmdsgrw64h4ax3la3evzqpxyy4adsmk9fs56qz3y # cspell:disable-line
|
||||
- &system_brain age1jhf7vm0005j60mjq63696frrmjhpy8kpc2d66mw044lqap5mjv4snmwvwm # cspell:disable-line
|
||||
- &system_jeeves age13lmqgc3jvkyah5e3vcwmj4s5wsc2akctcga0lpc0x8v8du3fxprqp4ldkv # cspell:disable-line
|
||||
- &system_leviathan age1l272y8udvg60z7edgje42fu49uwt4x2gxn5zvywssnv9h2krms8s094m4k # cspell:disable-line
|
||||
- &system_rhapsody age1ufnewppysaq2wwcl4ugngjz8pfzc5a35yg7luq0qmuqvctajcycs5lf6k4 # cspell:disable-line
|
||||
|
||||
creation_rules:
|
||||
@@ -18,5 +17,4 @@ creation_rules:
|
||||
- *system_bob
|
||||
- *system_brain
|
||||
- *system_jeeves
|
||||
- *system_leviathan
|
||||
- *system_rhapsody
|
||||
|
||||
Vendored
+16
-7
@@ -40,7 +40,6 @@
|
||||
"cgroupdriver",
|
||||
"charliermarsh",
|
||||
"Checkpointing",
|
||||
"cloudflared",
|
||||
"codellama",
|
||||
"codezombiech",
|
||||
"compactmode",
|
||||
@@ -77,11 +76,11 @@
|
||||
"esphome",
|
||||
"extest",
|
||||
"fadvise",
|
||||
"fastfetch",
|
||||
"fastforwardteam",
|
||||
"FASTFOX",
|
||||
"ffmpegthumbnailer",
|
||||
"filebot",
|
||||
"filebrowser",
|
||||
"fileroller",
|
||||
"findbar",
|
||||
"Fira",
|
||||
@@ -98,6 +97,7 @@
|
||||
"getch",
|
||||
"getmaxyx",
|
||||
"ghdeploy",
|
||||
"gitea",
|
||||
"globalprivacycontrol",
|
||||
"gparted",
|
||||
"gtts",
|
||||
@@ -116,7 +116,9 @@
|
||||
"httpchk",
|
||||
"hurlenko",
|
||||
"hwloc",
|
||||
"ical",
|
||||
"ignorelist",
|
||||
"improv",
|
||||
"INITDB",
|
||||
"iocharset",
|
||||
"ioit",
|
||||
@@ -126,6 +128,8 @@
|
||||
"jnoortheen",
|
||||
"jsbc",
|
||||
"kagi",
|
||||
"keyformat",
|
||||
"keylocation",
|
||||
"kuma",
|
||||
"lazer",
|
||||
"levelname",
|
||||
@@ -162,7 +166,6 @@
|
||||
"mypy",
|
||||
"ncdu",
|
||||
"nemo",
|
||||
"neofetch",
|
||||
"nerdfonts",
|
||||
"netdev",
|
||||
"netdevs",
|
||||
@@ -200,6 +203,7 @@
|
||||
"peerconnection",
|
||||
"PESKYFOX",
|
||||
"PGID",
|
||||
"pgvector",
|
||||
"pipewire",
|
||||
"pkgs",
|
||||
"plugdev",
|
||||
@@ -225,12 +229,10 @@
|
||||
"pylint",
|
||||
"pymetno",
|
||||
"pymodbus",
|
||||
"pyopenweathermap",
|
||||
"pyownet",
|
||||
"pytest",
|
||||
"qbit",
|
||||
"qbittorrent",
|
||||
"qbittorrentvpn",
|
||||
"qbitvpn",
|
||||
"qalculate",
|
||||
"quicksuggest",
|
||||
"radarr",
|
||||
"readahead",
|
||||
@@ -240,6 +242,7 @@
|
||||
"referer",
|
||||
"REFERERS",
|
||||
"relatime",
|
||||
"rerank",
|
||||
"Rhosts",
|
||||
"ripgrep",
|
||||
"roboto",
|
||||
@@ -255,6 +258,7 @@
|
||||
"sessionmaker",
|
||||
"sessionstore",
|
||||
"shellcheck",
|
||||
"signalbot",
|
||||
"signon",
|
||||
"Signons",
|
||||
"skia",
|
||||
@@ -286,11 +290,14 @@
|
||||
"topstories",
|
||||
"treefmt",
|
||||
"twimg",
|
||||
"typedmonarchmoney",
|
||||
"typer",
|
||||
"uaccess",
|
||||
"ubiquiti",
|
||||
"ublock",
|
||||
"uiprotect",
|
||||
"uitour",
|
||||
"unifi",
|
||||
"unrar",
|
||||
"unsubmitted",
|
||||
"uptimekuma",
|
||||
@@ -301,6 +308,8 @@
|
||||
"useragent",
|
||||
"usernamehw",
|
||||
"userprefs",
|
||||
"vaninventory",
|
||||
"vdev",
|
||||
"vfat",
|
||||
"victron",
|
||||
"virt",
|
||||
|
||||
@@ -1 +1,51 @@
|
||||
# dotfiles
|
||||
|
||||
## Installer ISO
|
||||
|
||||
Build a bootable NixOS ISO with the installer preinstalled:
|
||||
|
||||
```sh
|
||||
nix build .#iso
|
||||
```
|
||||
|
||||
Write `result/iso/nixos-zfs-installer.iso` to a USB stick (for example with `dd`) or boot it in a VM. The image is the minimal NixOS installation CD with ZFS enabled and `nixos-installer` on `PATH`. SSH is enabled and the `nixos` and `root` accounts use the password `nixos`, so you can also run the installer remotely. Once booted:
|
||||
|
||||
```sh
|
||||
sudo nixos-installer
|
||||
```
|
||||
|
||||
The ISO bundles the `.#installer-nixos` package, a variant of the binary that keeps its Nix store linkage instead of being patched for foreign distributions.
|
||||
|
||||
## Installer binary
|
||||
|
||||
Build the self-contained installer executable with:
|
||||
|
||||
```sh
|
||||
nix build .#installer
|
||||
```
|
||||
|
||||
The flake package (defined in `python/installer/package.nix`) uses the Python builder in `python/installer/build.py`, which stages only the installer modules before running PyInstaller. You can also call it directly when `pyinstaller` and `patchelf` are on `PATH`:
|
||||
|
||||
```sh
|
||||
python -m python.installer.build --output ./nixos-installer
|
||||
```
|
||||
|
||||
Copy `result/bin/nixos-installer` to the installer USB stick and run it as root from the NixOS live environment:
|
||||
|
||||
```sh
|
||||
sudo ./nixos-installer
|
||||
```
|
||||
|
||||
Validate the live environment first with:
|
||||
|
||||
```sh
|
||||
./nixos-installer --check
|
||||
```
|
||||
|
||||
Paste a value into the TUI encryption password field to enable LUKS during install, or set `ENCRYPT_KEY`:
|
||||
|
||||
```sh
|
||||
sudo env ENCRYPT_KEY='change-me' ./nixos-installer
|
||||
```
|
||||
|
||||
The binary bundles the Python runtime and only the installer modules it imports. It still expects the NixOS installer environment to provide system install tools such as `parted`, `zfs`, `zpool`, `cryptsetup`, `nixos-generate-config`, and `nixos-install`.
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
./nh.nix
|
||||
./nix.nix
|
||||
./programs.nix
|
||||
./safe_reboot.nix
|
||||
./ssh.nix
|
||||
./snapshot_manager.nix
|
||||
];
|
||||
@@ -24,7 +23,10 @@
|
||||
boot = {
|
||||
tmp.useTmpfs = true;
|
||||
kernelPackages = lib.mkDefault pkgs.linuxPackages_6_12;
|
||||
zfs.package = lib.mkDefault pkgs.zfs_2_3;
|
||||
zfs = {
|
||||
package = lib.mkDefault pkgs.zfs_2_4;
|
||||
forceImportRoot = lib.mkDefault false;
|
||||
};
|
||||
};
|
||||
|
||||
hardware.enableRedistributableFirmware = true;
|
||||
@@ -38,10 +40,17 @@
|
||||
|
||||
nixpkgs = {
|
||||
overlays = builtins.attrValues outputs.overlays;
|
||||
config.allowUnfree = true;
|
||||
config = {
|
||||
allowUnfree = true;
|
||||
permittedInsecurePackages = [
|
||||
"openssl-1.1.1w" # This is for discord-canary
|
||||
];
|
||||
};
|
||||
};
|
||||
|
||||
services = {
|
||||
dbus.implementation = "dbus";
|
||||
|
||||
# firmware update
|
||||
fwupd.enable = true;
|
||||
|
||||
@@ -50,11 +59,6 @@
|
||||
PYTHONPATH = "${inputs.self}/";
|
||||
};
|
||||
|
||||
safe_reboot = {
|
||||
enable = lib.mkDefault true;
|
||||
datasetPrefix = "root_pool/";
|
||||
};
|
||||
|
||||
zfs = {
|
||||
trim.enable = lib.mkDefault true;
|
||||
autoScrub.enable = lib.mkDefault true;
|
||||
|
||||
@@ -33,6 +33,9 @@ in
|
||||
];
|
||||
warn-dirty = false;
|
||||
flake-registry = ""; # disable global flake registries
|
||||
connect-timeout = 10;
|
||||
download-buffer-size = 536870912;
|
||||
fallback = true;
|
||||
};
|
||||
|
||||
# Add each flake input as a registry and nix_path
|
||||
|
||||
@@ -1,56 +0,0 @@
|
||||
{
|
||||
config,
|
||||
inputs,
|
||||
lib,
|
||||
pkgs,
|
||||
...
|
||||
}:
|
||||
let
|
||||
cfg = config.services.safe_reboot;
|
||||
python_command =
|
||||
lib.escapeShellArgs (
|
||||
[
|
||||
"${pkgs.my_python}/bin/python"
|
||||
"-m"
|
||||
"python.tools.safe_reboot"
|
||||
]
|
||||
++ lib.optionals (cfg.drivePath != null) [ cfg.drivePath ]
|
||||
++ [
|
||||
"--dataset-prefix"
|
||||
cfg.datasetPrefix
|
||||
"--check-only"
|
||||
]
|
||||
);
|
||||
in
|
||||
{
|
||||
options.services.safe_reboot = {
|
||||
enable = lib.mkEnableOption "Safe reboot dataset/drive validation";
|
||||
datasetPrefix = lib.mkOption {
|
||||
type = lib.types.str;
|
||||
default = "root_pool/";
|
||||
description = "Dataset prefix that must have exec enabled before rebooting.";
|
||||
};
|
||||
drivePath = lib.mkOption {
|
||||
type = lib.types.nullOr lib.types.str;
|
||||
default = null;
|
||||
description = "Drive path that must exist before rebooting. Set to null to skip.";
|
||||
};
|
||||
};
|
||||
|
||||
config = lib.mkIf cfg.enable {
|
||||
systemd.services.safe-reboot-check = {
|
||||
description = "Safe reboot validation";
|
||||
before = [ "systemd-reboot.service" ];
|
||||
wantedBy = [ "reboot.target" ];
|
||||
partOf = [ "reboot.target" ];
|
||||
path = [ pkgs.zfs ];
|
||||
environment = {
|
||||
PYTHONPATH = "${inputs.self}/";
|
||||
};
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = python_command;
|
||||
};
|
||||
};
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
nix.settings = {
|
||||
trusted-substituters = [ "http://192.168.95.35:5000" ];
|
||||
substituters = [ "http://192.168.95.35:5000/?priority=1&want-mass-query=true" ];
|
||||
};
|
||||
}
|
||||
@@ -1,8 +1,8 @@
|
||||
{ pkgs, ... }:
|
||||
{
|
||||
boot = {
|
||||
kernelPackages = pkgs.linuxPackages_6_17;
|
||||
zfs.package = pkgs.zfs_unstable;
|
||||
kernelPackages = pkgs.linuxPackages_6_18;
|
||||
zfs.package = pkgs.zfs_2_4;
|
||||
};
|
||||
|
||||
hardware.bluetooth = {
|
||||
|
||||
@@ -0,0 +1,256 @@
|
||||
{
|
||||
config,
|
||||
lib,
|
||||
pkgs,
|
||||
...
|
||||
}:
|
||||
let
|
||||
monitoringInterface = "ztwfunumly";
|
||||
nodeTextfileDir = "/var/lib/prometheus-node-exporter-textfile";
|
||||
|
||||
mkProcessNameTemplate =
|
||||
perPid: template: if perPid then "${template}:{{.PID}}:{{.StartTime}}" else template;
|
||||
|
||||
mkProcessMatchers = perPid: [
|
||||
{
|
||||
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Module}}";
|
||||
cmdline = [ "^/nix/store[^ ]*/bin/python[^ ]* -m (?P<Module>[^ ]+)" ];
|
||||
}
|
||||
{
|
||||
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||
cmdline = [
|
||||
"^/nix/store[^ ]*/bin/python[^ ]* /nix/store[^ ]*/bin/\\.?(?P<Wrapped>[^ /]+?)(?:-wrapped)?(?:\\s|$)"
|
||||
];
|
||||
}
|
||||
{
|
||||
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||
cmdline = [
|
||||
"^/nix/store[^ ]*/bin/node /nix/store[^ ]*-(?P<Wrapped>[A-Za-z0-9._+-]+)-[0-9][^ /]*/"
|
||||
];
|
||||
}
|
||||
{
|
||||
name = mkProcessNameTemplate perPid "{{.Username}}:{{.Matches.Wrapped}}";
|
||||
cmdline = [ "^/nix/store[^ ]*/(?:bin/|lib/[^ ]*/)?\\.?(?P<Wrapped>[^ /]+?)(?:-wrapped)?(?:\\s|$)" ];
|
||||
}
|
||||
{
|
||||
name = mkProcessNameTemplate perPid "{{.Username}}:{{.ExeBase}}";
|
||||
cmdline = [ ".+" ];
|
||||
}
|
||||
];
|
||||
|
||||
perPidConfig = pkgs.writeText "process-exporter-per-pid.yaml" (
|
||||
builtins.toJSON {
|
||||
process_names = mkProcessMatchers true;
|
||||
}
|
||||
);
|
||||
|
||||
zpoolLatencyScript = pkgs.writeShellScript "zpool-latency-exporter" ''
|
||||
set -euo pipefail
|
||||
|
||||
out_dir=${lib.escapeShellArg nodeTextfileDir}
|
||||
host=${lib.escapeShellArg config.networking.hostName}
|
||||
tmp_file="$(mktemp "$out_dir/zpool.prom.XXXXXX")"
|
||||
trap 'rm -f "$tmp_file"' EXIT
|
||||
|
||||
pools="$(zpool list -H -o name | paste -sd, -)"
|
||||
|
||||
cat >"$tmp_file" <<'EOF'
|
||||
# HELP zpool_iostat_total_wait_read_ns Average total read wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_total_wait_read_ns gauge
|
||||
# HELP zpool_iostat_total_wait_write_ns Average total write wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_total_wait_write_ns gauge
|
||||
# HELP zpool_iostat_disk_wait_read_ns Average disk read wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_disk_wait_read_ns gauge
|
||||
# HELP zpool_iostat_disk_wait_write_ns Average disk write wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_disk_wait_write_ns gauge
|
||||
# HELP zpool_iostat_syncq_wait_read_ns Average synchronous queue read wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_syncq_wait_read_ns gauge
|
||||
# HELP zpool_iostat_syncq_wait_write_ns Average synchronous queue write wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_syncq_wait_write_ns gauge
|
||||
# HELP zpool_iostat_asyncq_wait_read_ns Average asynchronous queue read wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_asyncq_wait_read_ns gauge
|
||||
# HELP zpool_iostat_asyncq_wait_write_ns Average asynchronous queue write wait time reported by zpool iostat.
|
||||
# TYPE zpool_iostat_asyncq_wait_write_ns gauge
|
||||
EOF
|
||||
|
||||
zpool iostat -Hplvy -y 1 1 | awk -F '\t' -v host="$host" -v pools="$pools" '
|
||||
function esc(str, out) {
|
||||
out = str
|
||||
gsub(/\\/, "\\\\", out)
|
||||
gsub(/"/, "\\\"", out)
|
||||
return out
|
||||
}
|
||||
|
||||
function emit(metric, pool, vdev, value) {
|
||||
if (value == "" || value == "-") {
|
||||
return
|
||||
}
|
||||
|
||||
printf "%s{host=\"%s\",pool=\"%s\",vdev=\"%s\"} %s\n",
|
||||
metric,
|
||||
esc(host),
|
||||
esc(pool),
|
||||
esc(vdev),
|
||||
value
|
||||
}
|
||||
|
||||
BEGIN {
|
||||
split(pools, pool_names, ",")
|
||||
for (idx in pool_names) {
|
||||
if (pool_names[idx] != "") {
|
||||
known_pools[pool_names[idx]] = 1
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
NF == 0 {
|
||||
next
|
||||
}
|
||||
|
||||
{
|
||||
row_name = $1
|
||||
|
||||
if (row_name in known_pools) {
|
||||
current_pool = row_name
|
||||
current_vdev = "_pool"
|
||||
} else if (current_pool == "") {
|
||||
next
|
||||
} else {
|
||||
current_vdev = row_name
|
||||
}
|
||||
|
||||
emit("zpool_iostat_total_wait_read_ns", current_pool, current_vdev, $8)
|
||||
emit("zpool_iostat_total_wait_write_ns", current_pool, current_vdev, $9)
|
||||
emit("zpool_iostat_disk_wait_read_ns", current_pool, current_vdev, $10)
|
||||
emit("zpool_iostat_disk_wait_write_ns", current_pool, current_vdev, $11)
|
||||
emit("zpool_iostat_syncq_wait_read_ns", current_pool, current_vdev, $12)
|
||||
emit("zpool_iostat_syncq_wait_write_ns", current_pool, current_vdev, $13)
|
||||
emit("zpool_iostat_asyncq_wait_read_ns", current_pool, current_vdev, $14)
|
||||
emit("zpool_iostat_asyncq_wait_write_ns", current_pool, current_vdev, $15)
|
||||
}
|
||||
' >>"$tmp_file"
|
||||
|
||||
mv "$tmp_file" "$out_dir/zpool.prom"
|
||||
trap - EXIT
|
||||
'';
|
||||
in
|
||||
{
|
||||
networking.firewall.interfaces.${monitoringInterface}.allowedTCPPorts = [
|
||||
9100
|
||||
9134
|
||||
9256
|
||||
9257
|
||||
9633
|
||||
];
|
||||
|
||||
services.prometheus.exporters = {
|
||||
node = {
|
||||
enable = true;
|
||||
enabledCollectors = [
|
||||
"pressure"
|
||||
"processes"
|
||||
"systemd"
|
||||
];
|
||||
extraFlags = [ "--collector.textfile.directory=${nodeTextfileDir}" ];
|
||||
};
|
||||
|
||||
process = {
|
||||
enable = true;
|
||||
user = "root";
|
||||
group = "root";
|
||||
settings.process_names = mkProcessMatchers false;
|
||||
extraFlags = [
|
||||
"-gather-smaps=false"
|
||||
"-remove-empty-groups=true"
|
||||
"-threads=false"
|
||||
];
|
||||
};
|
||||
|
||||
smartctl.enable = true;
|
||||
zfs.enable = true;
|
||||
};
|
||||
|
||||
programs.atop = {
|
||||
enable = true;
|
||||
atopService.enable = true;
|
||||
atopRotateTimer.enable = true;
|
||||
atopacctService.enable = true;
|
||||
settings.interval = 30;
|
||||
};
|
||||
|
||||
systemd = {
|
||||
services = {
|
||||
prometheus-process-pid-exporter = {
|
||||
description = "Prometheus process exporter with per-PID naming";
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
after = [ "network.target" ];
|
||||
serviceConfig = {
|
||||
ExecStart = ''
|
||||
${pkgs.prometheus-process-exporter}/bin/process-exporter \
|
||||
--web.listen-address 0.0.0.0:9257 \
|
||||
--config.path ${perPidConfig} \
|
||||
-children=false \
|
||||
-gather-smaps=false \
|
||||
-remove-empty-groups=true \
|
||||
-threads=false
|
||||
'';
|
||||
User = "root";
|
||||
Group = "root";
|
||||
Restart = "always";
|
||||
WorkingDirectory = "/tmp";
|
||||
CapabilityBoundingSet = [ "" ];
|
||||
DeviceAllow = [ "" ];
|
||||
LockPersonality = true;
|
||||
MemoryDenyWriteExecute = true;
|
||||
NoNewPrivileges = true;
|
||||
PrivateDevices = true;
|
||||
PrivateTmp = true;
|
||||
ProtectClock = true;
|
||||
ProtectControlGroups = true;
|
||||
ProtectHome = true;
|
||||
ProtectHostname = true;
|
||||
ProtectKernelLogs = true;
|
||||
ProtectKernelModules = true;
|
||||
ProtectKernelTunables = true;
|
||||
ProtectSystem = "strict";
|
||||
RemoveIPC = true;
|
||||
RestrictAddressFamilies = [
|
||||
"AF_INET"
|
||||
"AF_INET6"
|
||||
];
|
||||
RestrictNamespaces = true;
|
||||
RestrictRealtime = true;
|
||||
RestrictSUIDSGID = true;
|
||||
SystemCallArchitectures = "native";
|
||||
UMask = "0077";
|
||||
};
|
||||
};
|
||||
|
||||
zpool-latency-exporter = {
|
||||
description = "Exports ZFS latency metrics for node_exporter textfile collection";
|
||||
after = [ "zfs-import.target" ];
|
||||
requires = [ "zfs-import.target" ];
|
||||
path = [
|
||||
config.boot.zfs.package
|
||||
pkgs.coreutils
|
||||
pkgs.gawk
|
||||
];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = zpoolLatencyScript;
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
timers.zpool-latency-exporter = {
|
||||
wantedBy = [ "timers.target" ];
|
||||
timerConfig = {
|
||||
OnBootSec = "2m";
|
||||
OnUnitActiveSec = "60s";
|
||||
Unit = "zpool-latency-exporter.service";
|
||||
};
|
||||
};
|
||||
|
||||
tmpfiles.rules = [ "d ${nodeTextfileDir} 0755 root root - -" ];
|
||||
};
|
||||
}
|
||||
@@ -12,7 +12,7 @@
|
||||
brain.id = "SSCGIPI-IV3VYKB-TRNIJE3-COV4T2H-CDBER7F-I2CGHYA-NWOEUDU-3T5QAAN"; # cspell:disable-line
|
||||
ipad.id = "KI76T3X-SFUGV2L-VSNYTKR-TSIUV5L-SHWD3HE-GQRGRCN-GY4UFMD-CW6Z6AX"; # cspell:disable-line
|
||||
jeeves.id = "ICRHXZW-ECYJCUZ-I4CZ64R-3XRK7CG-LL2HAAK-FGOHD22-BQA4AI6-5OAL6AG"; # cspell:disable-line
|
||||
phone.id = "TBRULKD-7DZPGGZ-F6LLB7J-MSO54AY-7KLPBIN-QOFK6PX-W2HBEWI-PHM2CQI"; # cspell:disable-line
|
||||
phone.id = "JPVQKQW-CFXOJXT-Q5G5F3H-QIDHDRE-GKHPTQB-GXZUQSP-U7FR7F7-INP3AAH"; # cspell:disable-line
|
||||
rhapsody-in-green.id = "ASL3KC4-3XEN6PA-7BQBRKE-A7JXLI6-DJT43BY-Q4WPOER-7UALUAZ-VTPQ6Q4"; # cspell:disable-line
|
||||
};
|
||||
};
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
flags = [ "--accept-flake-config" ];
|
||||
randomizedDelaySec = "1h";
|
||||
persistent = true;
|
||||
flake = "github:RichieCahill/dotfiles";
|
||||
flake = "git+https://gitea.tmmworkshop.com/richie/dotfiles?ref=main";
|
||||
allowReboot = true;
|
||||
dates = "Sat *-*-* 06:00:00";
|
||||
};
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
# ZFS failed root import recovery
|
||||
|
||||
## Fast path
|
||||
|
||||
If the machine fails to boot because ZFS refuses to import `root_pool`:
|
||||
|
||||
### GRUB
|
||||
|
||||
1. At the bootloader menu, select the normal NixOS entry.
|
||||
2. Press `e`.
|
||||
3. Find the line that starts with `linux`.
|
||||
4. Append this to the end of that line:
|
||||
|
||||
```text
|
||||
zfs_force=1
|
||||
```
|
||||
|
||||
5. Boot once with `Ctrl+x` or `F10`.
|
||||
|
||||
### systemd-boot
|
||||
|
||||
1. At the bootloader menu, highlight the normal NixOS entry.
|
||||
2. Press `e`.
|
||||
3. Append this to the end of the options line:
|
||||
|
||||
```text
|
||||
zfs_force=1
|
||||
```
|
||||
|
||||
4. Press `Enter` to boot once.
|
||||
|
||||
## After boot
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
sudo zpool status
|
||||
sudo zpool import
|
||||
journalctl -b | rg "ZFS|zfs|import|root_pool"
|
||||
```
|
||||
|
||||
## Expected result
|
||||
|
||||
`sudo zpool status` should show `root_pool` as `ONLINE`.
|
||||
|
||||
## Reboot test
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
sudo reboot
|
||||
```
|
||||
|
||||
Do not add `zfs_force=1` the second time.
|
||||
|
||||
## If it still fails
|
||||
|
||||
Boot once more with:
|
||||
|
||||
```text
|
||||
zfs_force=1
|
||||
```
|
||||
|
||||
Then run:
|
||||
|
||||
```bash
|
||||
sudo zpool status -v
|
||||
sudo zpool history | tail -n 50
|
||||
journalctl -b | rg "ZFS|zfs|import|root_pool"
|
||||
```
|
||||
|
||||
## Notes
|
||||
|
||||
- Root pool name is `root_pool`.
|
||||
- This is a one-time recovery path after disk moves, controller changes, dirty exports, or interrupted imports.
|
||||
- Some hosts also need the LUKS unlock USB key inserted before boot.
|
||||
@@ -1,129 +0,0 @@
|
||||
esphome:
|
||||
name: batteries
|
||||
friendly_name: batteries
|
||||
|
||||
esp32:
|
||||
board: esp32dev
|
||||
framework:
|
||||
type: arduino
|
||||
|
||||
logger:
|
||||
|
||||
api:
|
||||
encryption:
|
||||
key: !secret api_key
|
||||
|
||||
external_components:
|
||||
- source: github://syssi/esphome-jk-bms@main
|
||||
|
||||
ota:
|
||||
- platform: esphome
|
||||
password: !secret ota_password
|
||||
|
||||
wifi:
|
||||
ssid: !secret wifi_ssid
|
||||
password: !secret wifi_password
|
||||
|
||||
captive_portal:
|
||||
|
||||
esp32_ble_tracker:
|
||||
scan_parameters:
|
||||
interval: 1100ms
|
||||
window: 1100ms
|
||||
active: true
|
||||
|
||||
ble_client:
|
||||
- mac_address: "C8:47:80:29:0F:DB"
|
||||
id: jk_ble0
|
||||
- mac_address: "C8:47:80:37:9D:DD"
|
||||
id: jk_ble1
|
||||
|
||||
jk_bms_ble:
|
||||
- ble_client_id: jk_ble0
|
||||
protocol_version: JK02_32S
|
||||
throttle: 1s
|
||||
id: jk_bms0
|
||||
|
||||
- ble_client_id: jk_ble1
|
||||
protocol_version: JK02_32S
|
||||
throttle: 1s
|
||||
id: jk_bms1
|
||||
|
||||
sensor:
|
||||
# BMS1 sensors
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
total_voltage:
|
||||
name: "JK0 Total Voltage"
|
||||
current:
|
||||
name: "JK0 Current"
|
||||
state_of_charge:
|
||||
name: "JK0 SoC"
|
||||
power:
|
||||
name: "JK0 Power"
|
||||
temperature_sensor_1:
|
||||
name: "JK0 Temp 1"
|
||||
temperature_sensor_2:
|
||||
name: "JK0 Temp 2"
|
||||
balancing:
|
||||
name: "JK0 balancing"
|
||||
charging_cycles:
|
||||
name: "JK0 charging cycles"
|
||||
total_runtime:
|
||||
name: "JK0 total runtime"
|
||||
balancing_current:
|
||||
name: "JK0 balancing current"
|
||||
|
||||
# BMS2 sensors
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
total_voltage:
|
||||
name: "JK1 Total Voltage"
|
||||
current:
|
||||
name: "JK1 Current"
|
||||
state_of_charge:
|
||||
name: "JK1 SoC"
|
||||
power:
|
||||
name: "Jk1 Power"
|
||||
temperature_sensor_1:
|
||||
name: "JK1 Temp 1"
|
||||
temperature_sensor_2:
|
||||
name: "Jk1 Temp 2"
|
||||
balancing:
|
||||
name: "JK1 balancing"
|
||||
charging_cycles:
|
||||
name: "JK1 charging cycles"
|
||||
total_runtime:
|
||||
name: "JK1 total runtime"
|
||||
balancing_current:
|
||||
name: "JK1 balancing current"
|
||||
|
||||
text_sensor:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
errors:
|
||||
name: "JK0 Errors"
|
||||
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
errors:
|
||||
name: "JK1 Errors"
|
||||
|
||||
switch:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
charging:
|
||||
name: "JK0 Charging"
|
||||
discharging:
|
||||
name: "JK0 Discharging"
|
||||
balancer:
|
||||
name: "JK0 Balancing"
|
||||
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
charging:
|
||||
name: "JK1 Charging"
|
||||
discharging:
|
||||
name: "JK1 Discharging"
|
||||
balancer:
|
||||
name: "JK1 Balancing"
|
||||
@@ -0,0 +1,132 @@
|
||||
esphome:
|
||||
name: batteries
|
||||
friendly_name: batteries
|
||||
|
||||
esp32:
|
||||
board: esp32dev
|
||||
framework:
|
||||
type: arduino
|
||||
|
||||
logger:
|
||||
|
||||
api:
|
||||
encryption:
|
||||
key: !secret api_key
|
||||
|
||||
external_components:
|
||||
- source: github://syssi/esphome-jk-bms@main
|
||||
|
||||
ota:
|
||||
- platform: esphome
|
||||
password: !secret ota_password
|
||||
|
||||
wifi:
|
||||
ssid: !secret wifi_ssid
|
||||
password: !secret wifi_password
|
||||
fast_connect: on
|
||||
|
||||
captive_portal:
|
||||
|
||||
esp32_ble_tracker:
|
||||
scan_parameters:
|
||||
interval: 1100ms
|
||||
window: 1100ms
|
||||
active: true
|
||||
|
||||
ble_client:
|
||||
- mac_address: "C8:47:80:29:0F:DB"
|
||||
id: jk_ble0
|
||||
|
||||
jk_bms_ble:
|
||||
- ble_client_id: jk_ble0
|
||||
protocol_version: JK02_32S
|
||||
throttle: 1s
|
||||
id: jk_bms0
|
||||
|
||||
button:
|
||||
- platform: jk_bms_ble
|
||||
retrieve_settings:
|
||||
name: "JK0 retrieve settings"
|
||||
retrieve_device_info:
|
||||
name: "JK0 retrieve device info"
|
||||
|
||||
sensor:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
total_voltage:
|
||||
name: "JK0 Total Voltage"
|
||||
state_of_charge:
|
||||
name: "JK0 SoC"
|
||||
charging_power:
|
||||
name: "JK0 charging power"
|
||||
discharging_power:
|
||||
name: "JK0 discharging power"
|
||||
temperature_sensor_1:
|
||||
name: "JK0 Temp 1"
|
||||
temperature_sensor_2:
|
||||
name: "JK0 Temp 2"
|
||||
balancing:
|
||||
name: "JK0 balancing"
|
||||
total_runtime:
|
||||
name: "JK0 total runtime"
|
||||
balancing_current:
|
||||
name: "JK0 balancing current"
|
||||
delta_cell_voltage:
|
||||
name: "JK0 cell delta voltage"
|
||||
average_cell_voltage:
|
||||
name: "JK0 cell average voltage"
|
||||
cell_voltage_1:
|
||||
name: "JK0 cell voltage 1"
|
||||
cell_voltage_2:
|
||||
name: "JK0 cell voltage 2"
|
||||
cell_voltage_3:
|
||||
name: "JK0 cell voltage 3"
|
||||
cell_voltage_4:
|
||||
name: "JK0 cell voltage 4"
|
||||
cell_voltage_5:
|
||||
name: "JK0 cell voltage 5"
|
||||
cell_voltage_6:
|
||||
name: "JK0 cell voltage 6"
|
||||
cell_voltage_7:
|
||||
name: "JK0 cell voltage 7"
|
||||
cell_voltage_8:
|
||||
name: "JK0 cell voltage 8"
|
||||
cell_resistance_1:
|
||||
name: "JK0 cell resistance 1"
|
||||
cell_resistance_2:
|
||||
name: "JK0 cell resistance 2"
|
||||
cell_resistance_3:
|
||||
name: "JK0 cell resistance 3"
|
||||
cell_resistance_4:
|
||||
name: "JK0 cell resistance 4"
|
||||
cell_resistance_5:
|
||||
name: "JK0 cell resistance 5"
|
||||
cell_resistance_6:
|
||||
name: "JK0 cell resistance 6"
|
||||
cell_resistance_7:
|
||||
name: "JK0 cell resistance 7"
|
||||
cell_resistance_8:
|
||||
name: "JK0 cell resistance 8"
|
||||
total_charging_cycle_capacity:
|
||||
name: "JK0 total charging cycle capacity"
|
||||
|
||||
text_sensor:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
errors:
|
||||
name: "JK0 Errors"
|
||||
|
||||
switch:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms0
|
||||
charging:
|
||||
name: "JK0 Charging"
|
||||
discharging:
|
||||
name: "JK0 Discharging"
|
||||
balancer:
|
||||
name: "JK0 Balancing"
|
||||
|
||||
- platform: ble_client
|
||||
ble_client_id: jk_ble0
|
||||
name: "JK0 enable bluetooth connection"
|
||||
id: ble_client_switch0
|
||||
@@ -0,0 +1,132 @@
|
||||
esphome:
|
||||
name: battery1
|
||||
friendly_name: battery1
|
||||
|
||||
esp32:
|
||||
board: esp32dev
|
||||
framework:
|
||||
type: arduino
|
||||
|
||||
logger:
|
||||
|
||||
api:
|
||||
encryption:
|
||||
key: !secret api_key
|
||||
|
||||
external_components:
|
||||
- source: github://syssi/esphome-jk-bms@main
|
||||
|
||||
ota:
|
||||
- platform: esphome
|
||||
password: !secret ota_password
|
||||
|
||||
wifi:
|
||||
ssid: !secret wifi_ssid
|
||||
password: !secret wifi_password
|
||||
fast_connect: on
|
||||
|
||||
captive_portal:
|
||||
|
||||
esp32_ble_tracker:
|
||||
scan_parameters:
|
||||
interval: 1100ms
|
||||
window: 1100ms
|
||||
active: true
|
||||
|
||||
ble_client:
|
||||
- mac_address: "C8:47:80:37:9D:DD"
|
||||
id: jk_ble1
|
||||
|
||||
jk_bms_ble:
|
||||
- ble_client_id: jk_ble1
|
||||
protocol_version: JK02_32S
|
||||
throttle: 1s
|
||||
id: jk_bms1
|
||||
|
||||
button:
|
||||
- platform: jk_bms_ble
|
||||
retrieve_settings:
|
||||
name: "JK1 retrieve settings"
|
||||
retrieve_device_info:
|
||||
name: "JK1 retrieve device info"
|
||||
|
||||
sensor:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
total_voltage:
|
||||
name: "JK1 Total Voltage"
|
||||
state_of_charge:
|
||||
name: "JK1 SoC"
|
||||
charging_power:
|
||||
name: "JK1 charging power"
|
||||
discharging_power:
|
||||
name: "JK1 discharging power"
|
||||
temperature_sensor_1:
|
||||
name: "JK1 Temp 1"
|
||||
temperature_sensor_2:
|
||||
name: "JK1 Temp 2"
|
||||
balancing:
|
||||
name: "JK1 balancing"
|
||||
total_runtime:
|
||||
name: "JK1 total runtime"
|
||||
balancing_current:
|
||||
name: "JK1 balancing current"
|
||||
delta_cell_voltage:
|
||||
name: "JK1 cell delta voltage"
|
||||
average_cell_voltage:
|
||||
name: "JK1 cell average voltage"
|
||||
cell_voltage_1:
|
||||
name: "JK1 cell voltage 1"
|
||||
cell_voltage_2:
|
||||
name: "JK1 cell voltage 2"
|
||||
cell_voltage_3:
|
||||
name: "JK1 cell voltage 3"
|
||||
cell_voltage_4:
|
||||
name: "JK1 cell voltage 4"
|
||||
cell_voltage_5:
|
||||
name: "JK1 cell voltage 5"
|
||||
cell_voltage_6:
|
||||
name: "JK1 cell voltage 6"
|
||||
cell_voltage_7:
|
||||
name: "JK1 cell voltage 7"
|
||||
cell_voltage_8:
|
||||
name: "JK1 cell voltage 8"
|
||||
cell_resistance_1:
|
||||
name: "JK1 cell resistance 1"
|
||||
cell_resistance_2:
|
||||
name: "JK1 cell resistance 2"
|
||||
cell_resistance_3:
|
||||
name: "JK1 cell resistance 3"
|
||||
cell_resistance_4:
|
||||
name: "JK1 cell resistance 4"
|
||||
cell_resistance_5:
|
||||
name: "JK1 cell resistance 5"
|
||||
cell_resistance_6:
|
||||
name: "JK1 cell resistance 6"
|
||||
cell_resistance_7:
|
||||
name: "JK1 cell resistance 7"
|
||||
cell_resistance_8:
|
||||
name: "JK1 cell resistance 8"
|
||||
total_charging_cycle_capacity:
|
||||
name: "JK1 total charging cycle capacity"
|
||||
|
||||
text_sensor:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
errors:
|
||||
name: "JK1 Errors"
|
||||
|
||||
switch:
|
||||
- platform: jk_bms_ble
|
||||
jk_bms_ble_id: jk_bms1
|
||||
charging:
|
||||
name: "JK1 Charging"
|
||||
discharging:
|
||||
name: "JK1 Discharging"
|
||||
balancer:
|
||||
name: "JK1 Balancing"
|
||||
|
||||
- platform: ble_client
|
||||
ble_client_id: jk_ble1
|
||||
name: "JK1 enable bluetooth connection"
|
||||
id: ble_client_switch0
|
||||
@@ -0,0 +1,48 @@
|
||||
esphome:
|
||||
name: "environment"
|
||||
friendly_name: "environment"
|
||||
|
||||
esp32:
|
||||
board: esp32dev
|
||||
framework:
|
||||
type: arduino
|
||||
|
||||
i2c:
|
||||
sda: GPIO21
|
||||
scl: GPIO22
|
||||
scan: True
|
||||
id: bus_a
|
||||
|
||||
sensor:
|
||||
- platform: aht10
|
||||
i2c_id: bus_a
|
||||
address: 0x38
|
||||
variant: AHT20
|
||||
temperature:
|
||||
name: "environment Temperature"
|
||||
id: aht10_temperature
|
||||
humidity:
|
||||
name: "environment Humidity"
|
||||
id: aht10_humidity
|
||||
update_interval: 5s
|
||||
|
||||
web_server:
|
||||
port: 80
|
||||
|
||||
logger:
|
||||
level: DEBUG
|
||||
|
||||
api:
|
||||
encryption:
|
||||
key: !secret api_key
|
||||
|
||||
ota:
|
||||
- platform: esphome
|
||||
password: !secret ota_password
|
||||
|
||||
wifi:
|
||||
ssid: !secret wifi_ssid
|
||||
password: !secret wifi_password
|
||||
fast_connect: on
|
||||
|
||||
captive_portal:
|
||||
File diff suppressed because one or more lines are too long
Generated
+42
-26
@@ -8,11 +8,11 @@
|
||||
},
|
||||
"locked": {
|
||||
"dir": "pkgs/firefox-addons",
|
||||
"lastModified": 1763697825,
|
||||
"narHash": "sha256-AgCCcVPOi1tuzuW5/StlwqBjRWSX62oL97qWuxrq5UA=",
|
||||
"lastModified": 1783828963,
|
||||
"narHash": "sha256-eTytzcUJCaDUZ3/9EF0+V3fvlikQMQBwiX1Sx4Gy+No=",
|
||||
"owner": "rycee",
|
||||
"repo": "nur-expressions",
|
||||
"rev": "cefce78793603231be226fa77e7ad58e0e4899b8",
|
||||
"rev": "8d61e9afde605cd6c22dab68b83d7a71f0a6c5b2",
|
||||
"type": "gitlab"
|
||||
},
|
||||
"original": {
|
||||
@@ -29,11 +29,11 @@
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1763748372,
|
||||
"narHash": "sha256-AUc78Qv3sWir0hvbmfXoZ7Jzq9VVL97l+sP9Jgms+JU=",
|
||||
"lastModified": 1783823409,
|
||||
"narHash": "sha256-OI4IkRjRXa1e7hYmCGJDPDq5H/kPwhsyoS80cNUF9fI=",
|
||||
"owner": "nix-community",
|
||||
"repo": "home-manager",
|
||||
"rev": "d10a9b16b2a3ee28433f3d1c603f4e9f1fecb8e1",
|
||||
"rev": "7566825d4652a1b885bd4ce65bd9e8def432fec9",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
@@ -43,12 +43,15 @@
|
||||
}
|
||||
},
|
||||
"nixos-hardware": {
|
||||
"inputs": {
|
||||
"nixpkgs": "nixpkgs"
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1762847253,
|
||||
"narHash": "sha256-BWWnUUT01lPwCWUvS0p6Px5UOBFeXJ8jR+ZdLX8IbrU=",
|
||||
"lastModified": 1783792734,
|
||||
"narHash": "sha256-50rvY9GdFvpYDcMLcD/4cWSi0hVxArT5wsGlVsHy8eY=",
|
||||
"owner": "nixos",
|
||||
"repo": "nixos-hardware",
|
||||
"rev": "899dc449bc6428b9ee6b3b8f771ca2b0ef945ab9",
|
||||
"rev": "8efb4337e857949f4cfac86d12ef1066f417f31f",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
@@ -60,27 +63,24 @@
|
||||
},
|
||||
"nixpkgs": {
|
||||
"locked": {
|
||||
"lastModified": 1763421233,
|
||||
"narHash": "sha256-Stk9ZYRkGrnnpyJ4eqt9eQtdFWRRIvMxpNRf4sIegnw=",
|
||||
"owner": "nixos",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "89c2b2330e733d6cdb5eae7b899326930c2c0648",
|
||||
"type": "github"
|
||||
"lastModified": 1767892417,
|
||||
"narHash": "sha256-8bW3q88CEg2u4hSP66Vf4lpbLonHz7hqDNBMcCY7E9U=",
|
||||
"rev": "3497aa5c9457a9d88d71fa93a4a8368816fbeeba",
|
||||
"type": "tarball",
|
||||
"url": "https://releases.nixos.org/nixos/unstable/nixos-26.05pre924538.3497aa5c9457/nixexprs.tar.xz"
|
||||
},
|
||||
"original": {
|
||||
"owner": "nixos",
|
||||
"ref": "nixos-unstable",
|
||||
"repo": "nixpkgs",
|
||||
"type": "github"
|
||||
"type": "tarball",
|
||||
"url": "https://channels.nixos.org/nixos-unstable/nixexprs.tar.xz"
|
||||
}
|
||||
},
|
||||
"nixpkgs-master": {
|
||||
"locked": {
|
||||
"lastModified": 1763774007,
|
||||
"narHash": "sha256-PPeHfKA11P09kBkBD5pS3tIAFjnG5muHQnODQGTY87g=",
|
||||
"lastModified": 1783874024,
|
||||
"narHash": "sha256-Fd8rPvyBv6JjcO/nZxZiFQan6Fww/jAF4TYj0Th/Yfo=",
|
||||
"owner": "nixos",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "8a7cf7e9e18384533d9ecd0bfbcf475ac1dc497e",
|
||||
"rev": "0b4f03c64b236e4ba4252414274e92796c300124",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
@@ -106,12 +106,28 @@
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"nixpkgs_2": {
|
||||
"locked": {
|
||||
"lastModified": 1783776592,
|
||||
"narHash": "sha256-UgCQzxeWI75XM8G+hPrPh+MKzEPjG3SpAj7dtqSbksA=",
|
||||
"owner": "nixos",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "e7a3ca8092b61ff85b6a45bf863ea2b2d6a661b3",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
"owner": "nixos",
|
||||
"ref": "nixos-unstable",
|
||||
"repo": "nixpkgs",
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"root": {
|
||||
"inputs": {
|
||||
"firefox-addons": "firefox-addons",
|
||||
"home-manager": "home-manager",
|
||||
"nixos-hardware": "nixos-hardware",
|
||||
"nixpkgs": "nixpkgs",
|
||||
"nixpkgs": "nixpkgs_2",
|
||||
"nixpkgs-master": "nixpkgs-master",
|
||||
"nixpkgs-stable": "nixpkgs-stable",
|
||||
"sops-nix": "sops-nix",
|
||||
@@ -125,11 +141,11 @@
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1763607916,
|
||||
"narHash": "sha256-VefBA1JWRXM929mBAFohFUtQJLUnEwZ2vmYUNkFnSjE=",
|
||||
"lastModified": 1783174389,
|
||||
"narHash": "sha256-aCWC8ngycU7OdJrU2+Je3qf+1a2ykuBvpPhZT/9tXMc=",
|
||||
"owner": "Mic92",
|
||||
"repo": "sops-nix",
|
||||
"rev": "877bb495a6f8faf0d89fc10bd142c4b7ed2bcc0b",
|
||||
"rev": "f1406619a3884cd5c47992a70b8b35c9c0fcb4c9",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
|
||||
@@ -65,38 +65,48 @@
|
||||
|
||||
devShells = forEachSystem (pkgs: import ./shell.nix { inherit pkgs; });
|
||||
formatter = forEachSystem (pkgs: pkgs.treefmt);
|
||||
packages = forEachSystem (
|
||||
pkgs:
|
||||
let
|
||||
installer = pkgs.callPackage ./python/installer/package.nix { };
|
||||
installer-nixos = pkgs.callPackage ./python/installer/package.nix { patchElf = false; };
|
||||
in
|
||||
{
|
||||
inherit installer installer-nixos;
|
||||
default = installer;
|
||||
}
|
||||
// lib.optionalAttrs (pkgs.stdenv.hostPlatform.system == "x86_64-linux") {
|
||||
iso = self.nixosConfigurations.iso.config.system.build.isoImage;
|
||||
}
|
||||
);
|
||||
apps = forEachSystem (
|
||||
pkgs:
|
||||
let
|
||||
system = pkgs.stdenv.hostPlatform.system;
|
||||
installer = {
|
||||
type = "app";
|
||||
program = "${self.packages.${system}.installer}/bin/nixos-installer";
|
||||
meta.description = "One-file NixOS ZFS installer.";
|
||||
};
|
||||
in
|
||||
{
|
||||
inherit installer;
|
||||
default = installer;
|
||||
}
|
||||
);
|
||||
|
||||
nixosConfigurations = {
|
||||
bob = lib.nixosSystem {
|
||||
modules = [
|
||||
./systems/bob
|
||||
];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
brain = lib.nixosSystem {
|
||||
modules = [
|
||||
./systems/brain
|
||||
];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
jeeves = lib.nixosSystem {
|
||||
modules = [
|
||||
./systems/jeeves
|
||||
];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
rhapsody-in-green = lib.nixosSystem {
|
||||
modules = [
|
||||
./systems/rhapsody-in-green
|
||||
];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
leviathan = lib.nixosSystem {
|
||||
modules = [
|
||||
./systems/leviathan
|
||||
];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
};
|
||||
nixosConfigurations =
|
||||
let
|
||||
hosts = builtins.attrNames (
|
||||
lib.filterAttrs (_: type: type == "directory") (builtins.readDir ./systems)
|
||||
);
|
||||
mkHost =
|
||||
name:
|
||||
lib.nixosSystem {
|
||||
modules = [ ./systems/${name} ];
|
||||
specialArgs = { inherit inputs outputs; };
|
||||
};
|
||||
in
|
||||
lib.genAttrs hosts mkHost;
|
||||
};
|
||||
}
|
||||
|
||||
+13
-8
@@ -3,38 +3,43 @@
|
||||
# When applied, the stable nixpkgs set (declared in the flake inputs) will be accessible through 'pkgs.stable'
|
||||
stable = final: _prev: {
|
||||
stable = import inputs.nixpkgs-stable {
|
||||
system = final.system;
|
||||
system = final.stdenv.hostPlatform.system;
|
||||
config.allowUnfree = true;
|
||||
};
|
||||
};
|
||||
# When applied, the master nixpkgs set (declared in the flake inputs) will be accessible through 'pkgs.master'
|
||||
master = final: _prev: {
|
||||
master = import inputs.nixpkgs-master {
|
||||
system = final.system;
|
||||
system = final.stdenv.hostPlatform.system;
|
||||
config.allowUnfree = true;
|
||||
};
|
||||
};
|
||||
|
||||
python-env = final: _prev: {
|
||||
my_python = final.python313.withPackages (
|
||||
my_python = final.python314.withPackages (
|
||||
ps: with ps; [
|
||||
alembic
|
||||
apprise
|
||||
apscheduler
|
||||
fastapi
|
||||
fastapi-cli
|
||||
httpx
|
||||
mypy
|
||||
polars
|
||||
pgvector
|
||||
psycopg
|
||||
pydantic
|
||||
pyfakefs
|
||||
pytest
|
||||
pytest-cov
|
||||
pytest-mock
|
||||
pytest-xdist
|
||||
requests
|
||||
python-multipart
|
||||
ruff
|
||||
scalene
|
||||
sqlalchemy
|
||||
textual
|
||||
tenacity
|
||||
tinytuya
|
||||
typer
|
||||
types-requests
|
||||
websockets
|
||||
]
|
||||
);
|
||||
};
|
||||
|
||||
+51
-8
@@ -3,11 +3,41 @@ name = "system_tools"
|
||||
version = "0.1.0"
|
||||
description = ""
|
||||
authors = [{ name = "Richie Cahill", email = "richie@tmmworkshop.com" }]
|
||||
requires-python = "~=3.13.0"
|
||||
requires-python = "~=3.14.0"
|
||||
readme = "README.md"
|
||||
license = "MIT"
|
||||
# these dependencies are a best effort and aren't guaranteed to work
|
||||
dependencies = ["apprise", "apscheduler", "polars", "requests", "typer"]
|
||||
# for up-to-date dependencies, see overlays/default.nix
|
||||
dependencies = [
|
||||
"alembic",
|
||||
"apprise",
|
||||
"apscheduler",
|
||||
"beautifulsoup4",
|
||||
"bm25s",
|
||||
"ebooklib",
|
||||
"fastapi",
|
||||
"fastapi-cli",
|
||||
"httpx",
|
||||
"jinja2",
|
||||
"pgvector",
|
||||
"polars",
|
||||
"psycopg[binary]",
|
||||
"pydantic",
|
||||
"pydantic-settings",
|
||||
"python-multipart",
|
||||
"sqlalchemy",
|
||||
"tenacity",
|
||||
"tiktoken",
|
||||
"tinytuya",
|
||||
"typer",
|
||||
"uvicorn",
|
||||
"websockets",
|
||||
"yake",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
database = "python.database_cli:app"
|
||||
whisper-transcribe = "python.tools.whisper.transcribe:main"
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
@@ -18,12 +48,11 @@ dev = [
|
||||
"pytest-xdist",
|
||||
"pytest",
|
||||
"ruff",
|
||||
"types-requests",
|
||||
]
|
||||
|
||||
[tool.ruff]
|
||||
|
||||
target-version = "py313"
|
||||
target-version = "py314"
|
||||
|
||||
line-length = 120
|
||||
|
||||
@@ -33,26 +62,39 @@ lint.ignore = [
|
||||
"COM812", # (TEMP) conflicts when used with the formatter
|
||||
"ISC001", # (TEMP) conflicts when used with the formatter
|
||||
"S603", # (PERM) This is known to cause a false positive
|
||||
"S607", # (PERM) This is becoming a consistent annoyance
|
||||
]
|
||||
|
||||
[tool.ruff.lint.per-file-ignores]
|
||||
|
||||
"tests/**" = [
|
||||
"S101", # (perm) pytest needs asserts
|
||||
"ANN", # (perm) type annotations not needed in tests
|
||||
"D", # (perm) docstrings not needed in tests
|
||||
"PLR2004", # (perm) magic values are fine in test assertions
|
||||
"S101", # (perm) pytest needs asserts
|
||||
]
|
||||
"python/random/**" = [
|
||||
"python/stuff/**" = [
|
||||
"T201", # (perm) I don't care about print statements dir
|
||||
]
|
||||
"python/testing/**" = [
|
||||
"T201", # (perm) I don't care about print statements dir
|
||||
"ERA001", # (perm) I don't care about print statements dir
|
||||
]
|
||||
|
||||
"python/splendor/**" = [
|
||||
"S311", # (perm) there is no security issue here
|
||||
"T201", # (perm) I don't care about print statements dir
|
||||
"PLR2004", # (temps) need to think about this
|
||||
]
|
||||
"python/orm/**" = [
|
||||
"TC003", # (perm) this creates issues because sqlalchemy uses these at runtime
|
||||
]
|
||||
"python/congress_tracker/**" = [
|
||||
"TC003", # (perm) this creates issues because sqlalchemy uses these at runtime
|
||||
]
|
||||
|
||||
"python/alembic/**" = [
|
||||
"INP001", # (perm) this creates LSP issues for alembic
|
||||
]
|
||||
|
||||
[tool.ruff.lint.pydocstyle]
|
||||
convention = "google"
|
||||
@@ -75,5 +117,6 @@ exclude_lines = [
|
||||
]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
addopts = "-n auto -ra"
|
||||
addopts = "-n auto -ra --ignore=tests/ebook_search"
|
||||
testpaths = ["tests"]
|
||||
# --cov=system_tools --cov-report=term-missing --cov-report=xml --cov-report=html --cov-branch
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Alembic."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any, Literal
|
||||
|
||||
from alembic import context
|
||||
from alembic.script import write_hooks
|
||||
from sqlalchemy.schema import CreateSchema
|
||||
|
||||
from python.common import bash_wrapper
|
||||
from python.orm.common import get_postgres_engine
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import MutableMapping
|
||||
|
||||
from sqlalchemy.orm import DeclarativeBase
|
||||
|
||||
config = context.config
|
||||
|
||||
base_class: type[DeclarativeBase] = config.attributes.get("base")
|
||||
if base_class is None:
|
||||
error = "No base class provided. Use the database CLI to run alembic commands."
|
||||
raise RuntimeError(error)
|
||||
|
||||
target_metadata = base_class.metadata
|
||||
logging.basicConfig(
|
||||
level="DEBUG",
|
||||
datefmt="%Y-%m-%dT%H:%M:%S%z",
|
||||
format="%(asctime)s %(levelname)s %(filename)s:%(lineno)d - %(message)s",
|
||||
handlers=[logging.StreamHandler(sys.stdout)],
|
||||
)
|
||||
|
||||
|
||||
@write_hooks.register("dynamic_schema")
|
||||
def dynamic_schema(filename: str, _options: dict[Any, Any]) -> None:
|
||||
"""Dynamic schema."""
|
||||
original_file = Path(filename).read_text()
|
||||
schema_name = base_class.schema_name
|
||||
dynamic_schema_file_part1 = original_file.replace(f"schema='{schema_name}'", "schema=schema")
|
||||
dynamic_schema_file = dynamic_schema_file_part1.replace(f"'{schema_name}.", "f'{schema}.")
|
||||
Path(filename).write_text(dynamic_schema_file)
|
||||
|
||||
|
||||
@write_hooks.register("import_postgresql")
|
||||
def import_postgresql(filename: str, _options: dict[Any, Any]) -> None:
|
||||
"""Add postgresql dialect import when postgresql types are used."""
|
||||
content = Path(filename).read_text()
|
||||
if "postgresql." in content and "from sqlalchemy.dialects import postgresql" not in content:
|
||||
content = content.replace(
|
||||
"import sqlalchemy as sa\n",
|
||||
"import sqlalchemy as sa\nfrom sqlalchemy.dialects import postgresql\n",
|
||||
)
|
||||
Path(filename).write_text(content)
|
||||
|
||||
|
||||
@write_hooks.register("ruff")
|
||||
def ruff_check_and_format(filename: str, _options: dict[Any, Any]) -> None:
|
||||
"""Docstring for ruff_check_and_format."""
|
||||
bash_wrapper(f"ruff check --fix {filename}")
|
||||
bash_wrapper(f"ruff format {filename}")
|
||||
|
||||
|
||||
def include_name(
|
||||
name: str | None,
|
||||
type_: Literal["schema", "table", "column", "index", "unique_constraint", "foreign_key_constraint"],
|
||||
_parent_names: MutableMapping[Literal["schema_name", "table_name", "schema_qualified_table_name"], str | None],
|
||||
) -> bool:
|
||||
"""Filter tables to be included in the migration.
|
||||
|
||||
Args:
|
||||
name (str): The name of the table.
|
||||
type_ (str): The type of the table.
|
||||
_parent_names (MutableMapping): The names of the parent tables.
|
||||
|
||||
Returns:
|
||||
bool: True if the table should be included, False otherwise.
|
||||
|
||||
"""
|
||||
if type_ == "schema":
|
||||
# allows a database with multiple schemas to have separate alembic revisions
|
||||
return name == target_metadata.schema
|
||||
return True
|
||||
|
||||
|
||||
def run_migrations_online() -> None:
|
||||
"""Run migrations in 'online' mode.
|
||||
|
||||
In this scenario we need to create an Engine
|
||||
and associate a connection with the context.
|
||||
|
||||
"""
|
||||
env_prefix = config.attributes.get("env_prefix", "POSTGRES")
|
||||
connectable = get_postgres_engine(name=env_prefix)
|
||||
|
||||
with connectable.connect() as connection:
|
||||
schema = base_class.schema_name
|
||||
if not connectable.dialect.has_schema(connection, schema):
|
||||
answer = input(f"Schema {schema!r} does not exist. Create it? [y/N] ")
|
||||
if answer.lower() != "y":
|
||||
error = f"Schema {schema!r} does not exist. Exiting."
|
||||
raise SystemExit(error)
|
||||
connection.execute(CreateSchema(schema))
|
||||
connection.commit()
|
||||
|
||||
context.configure(
|
||||
connection=connection,
|
||||
target_metadata=target_metadata,
|
||||
include_schemas=True,
|
||||
version_table_schema=schema,
|
||||
include_name=include_name,
|
||||
)
|
||||
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
connection.commit()
|
||||
|
||||
|
||||
run_migrations_online()
|
||||
@@ -0,0 +1,113 @@
|
||||
"""created contact api.
|
||||
|
||||
Revision ID: edd7dd61a3d2
|
||||
Revises:
|
||||
Create Date: 2026-01-11 15:45:59.909266
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "edd7dd61a3d2"
|
||||
down_revision: str | None = None
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"contact",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("age", sa.Integer(), nullable=True),
|
||||
sa.Column("bio", sa.String(), nullable=True),
|
||||
sa.Column("current_job", sa.String(), nullable=True),
|
||||
sa.Column("gender", sa.String(), nullable=True),
|
||||
sa.Column("goals", sa.String(), nullable=True),
|
||||
sa.Column("legal_name", sa.String(), nullable=True),
|
||||
sa.Column("profile_pic", sa.String(), nullable=True),
|
||||
sa.Column("safe_conversation_starters", sa.String(), nullable=True),
|
||||
sa.Column("self_sufficiency_score", sa.Integer(), nullable=True),
|
||||
sa.Column("social_structure_style", sa.String(), nullable=True),
|
||||
sa.Column("ssn", sa.String(), nullable=True),
|
||||
sa.Column("suffix", sa.String(), nullable=True),
|
||||
sa.Column("timezone", sa.String(), nullable=True),
|
||||
sa.Column("topics_to_avoid", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_contact")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"need",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("description", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_need")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"contact_need",
|
||||
sa.Column("contact_id", sa.Integer(), nullable=False),
|
||||
sa.Column("need_id", sa.Integer(), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["contact_id"],
|
||||
[f"{schema}.contact.id"],
|
||||
name=op.f("fk_contact_need_contact_id_contact"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["need_id"], [f"{schema}.need.id"], name=op.f("fk_contact_need_need_id_need"), ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("contact_id", "need_id", name=op.f("pk_contact_need")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"contact_relationship",
|
||||
sa.Column("contact_id", sa.Integer(), nullable=False),
|
||||
sa.Column("related_contact_id", sa.Integer(), nullable=False),
|
||||
sa.Column("relationship_type", sa.String(length=100), nullable=False),
|
||||
sa.Column("closeness_weight", sa.Integer(), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["contact_id"],
|
||||
[f"{schema}.contact.id"],
|
||||
name=op.f("fk_contact_relationship_contact_id_contact"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["related_contact_id"],
|
||||
[f"{schema}.contact.id"],
|
||||
name=op.f("fk_contact_relationship_related_contact_id_contact"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("contact_id", "related_contact_id", name=op.f("pk_contact_relationship")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("contact_relationship", schema=schema)
|
||||
op.drop_table("contact_need", schema=schema)
|
||||
op.drop_table("need", schema=schema)
|
||||
op.drop_table("contact", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+135
@@ -0,0 +1,135 @@
|
||||
"""add congress tracker tables.
|
||||
|
||||
Revision ID: 3f71565e38de
|
||||
Revises: edd7dd61a3d2
|
||||
Create Date: 2026-02-12 16:36:09.457303
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "3f71565e38de"
|
||||
down_revision: str | None = "edd7dd61a3d2"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"bill",
|
||||
sa.Column("congress", sa.Integer(), nullable=False),
|
||||
sa.Column("bill_type", sa.String(), nullable=False),
|
||||
sa.Column("number", sa.Integer(), nullable=False),
|
||||
sa.Column("title", sa.String(), nullable=True),
|
||||
sa.Column("title_short", sa.String(), nullable=True),
|
||||
sa.Column("official_title", sa.String(), nullable=True),
|
||||
sa.Column("status", sa.String(), nullable=True),
|
||||
sa.Column("status_at", sa.Date(), nullable=True),
|
||||
sa.Column("sponsor_bioguide_id", sa.String(), nullable=True),
|
||||
sa.Column("subjects_top_term", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_bill")),
|
||||
sa.UniqueConstraint("congress", "bill_type", "number", name="uq_bill_congress_type_number"),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index("ix_bill_congress", "bill", ["congress"], unique=False, schema=schema)
|
||||
op.create_table(
|
||||
"legislator",
|
||||
sa.Column("bioguide_id", sa.Text(), nullable=False),
|
||||
sa.Column("thomas_id", sa.String(), nullable=True),
|
||||
sa.Column("lis_id", sa.String(), nullable=True),
|
||||
sa.Column("govtrack_id", sa.Integer(), nullable=True),
|
||||
sa.Column("opensecrets_id", sa.String(), nullable=True),
|
||||
sa.Column("fec_ids", sa.String(), nullable=True),
|
||||
sa.Column("first_name", sa.String(), nullable=False),
|
||||
sa.Column("last_name", sa.String(), nullable=False),
|
||||
sa.Column("official_full_name", sa.String(), nullable=True),
|
||||
sa.Column("nickname", sa.String(), nullable=True),
|
||||
sa.Column("birthday", sa.Date(), nullable=True),
|
||||
sa.Column("gender", sa.String(), nullable=True),
|
||||
sa.Column("current_party", sa.String(), nullable=True),
|
||||
sa.Column("current_state", sa.String(), nullable=True),
|
||||
sa.Column("current_district", sa.Integer(), nullable=True),
|
||||
sa.Column("current_chamber", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_legislator")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index(op.f("ix_legislator_bioguide_id"), "legislator", ["bioguide_id"], unique=True, schema=schema)
|
||||
op.create_table(
|
||||
"vote",
|
||||
sa.Column("congress", sa.Integer(), nullable=False),
|
||||
sa.Column("chamber", sa.String(), nullable=False),
|
||||
sa.Column("session", sa.Integer(), nullable=False),
|
||||
sa.Column("number", sa.Integer(), nullable=False),
|
||||
sa.Column("vote_type", sa.String(), nullable=True),
|
||||
sa.Column("question", sa.String(), nullable=True),
|
||||
sa.Column("result", sa.String(), nullable=True),
|
||||
sa.Column("result_text", sa.String(), nullable=True),
|
||||
sa.Column("vote_date", sa.Date(), nullable=False),
|
||||
sa.Column("yea_count", sa.Integer(), nullable=True),
|
||||
sa.Column("nay_count", sa.Integer(), nullable=True),
|
||||
sa.Column("not_voting_count", sa.Integer(), nullable=True),
|
||||
sa.Column("present_count", sa.Integer(), nullable=True),
|
||||
sa.Column("bill_id", sa.Integer(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(["bill_id"], [f"{schema}.bill.id"], name=op.f("fk_vote_bill_id_bill")),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_vote")),
|
||||
sa.UniqueConstraint("congress", "chamber", "session", "number", name="uq_vote_congress_chamber_session_number"),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index("ix_vote_congress_chamber", "vote", ["congress", "chamber"], unique=False, schema=schema)
|
||||
op.create_index("ix_vote_date", "vote", ["vote_date"], unique=False, schema=schema)
|
||||
op.create_table(
|
||||
"vote_record",
|
||||
sa.Column("vote_id", sa.Integer(), nullable=False),
|
||||
sa.Column("legislator_id", sa.Integer(), nullable=False),
|
||||
sa.Column("position", sa.String(), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["legislator_id"],
|
||||
[f"{schema}.legislator.id"],
|
||||
name=op.f("fk_vote_record_legislator_id_legislator"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["vote_id"], [f"{schema}.vote.id"], name=op.f("fk_vote_record_vote_id_vote"), ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("vote_id", "legislator_id", name=op.f("pk_vote_record")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("vote_record", schema=schema)
|
||||
op.drop_index("ix_vote_date", table_name="vote", schema=schema)
|
||||
op.drop_index("ix_vote_congress_chamber", table_name="vote", schema=schema)
|
||||
op.drop_table("vote", schema=schema)
|
||||
op.drop_index(op.f("ix_legislator_bioguide_id"), table_name="legislator", schema=schema)
|
||||
op.drop_table("legislator", schema=schema)
|
||||
op.drop_index("ix_bill_congress", table_name="bill", schema=schema)
|
||||
op.drop_table("bill", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+58
@@ -0,0 +1,58 @@
|
||||
"""adding SignalDevice for DeviceRegistry for signal bot.
|
||||
|
||||
Revision ID: 4c410c16e39c
|
||||
Revises: 3f71565e38de
|
||||
Create Date: 2026-03-09 14:51:24.228976
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "4c410c16e39c"
|
||||
down_revision: str | None = "3f71565e38de"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"signal_device",
|
||||
sa.Column("phone_number", sa.String(length=50), nullable=False),
|
||||
sa.Column("safety_number", sa.String(), nullable=False),
|
||||
sa.Column(
|
||||
"trust_level",
|
||||
postgresql.ENUM("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", schema=schema),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("last_seen", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_signal_device")),
|
||||
sa.UniqueConstraint("phone_number", name=op.f("uq_signal_device_phone_number")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("signal_device", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
@@ -0,0 +1,41 @@
|
||||
"""fixed safety number logic.
|
||||
|
||||
Revision ID: 99fec682516c
|
||||
Revises: 4c410c16e39c
|
||||
Create Date: 2026-03-09 16:25:25.085806
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "99fec682516c"
|
||||
down_revision: str | None = "4c410c16e39c"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.alter_column("signal_device", "safety_number", existing_type=sa.VARCHAR(), nullable=True, schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.alter_column("signal_device", "safety_number", existing_type=sa.VARCHAR(), nullable=False, schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
"""add dead_letter_message table.
|
||||
|
||||
Revision ID: a1b2c3d4e5f6
|
||||
Revises: 99fec682516c
|
||||
Create Date: 2026-03-10 12:00:00.000000
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "a1b2c3d4e5f6"
|
||||
down_revision: str | None = "99fec682516c"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
op.create_table(
|
||||
"dead_letter_message",
|
||||
sa.Column("source", sa.String(), nullable=False),
|
||||
sa.Column("message", sa.Text(), nullable=False),
|
||||
sa.Column("received_at", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column(
|
||||
"status",
|
||||
postgresql.ENUM("UNPROCESSED", "PROCESSED", name="message_status", schema=schema),
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_dead_letter_message")),
|
||||
schema=schema,
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
op.drop_table("dead_letter_message", schema=schema)
|
||||
op.execute(sa.text(f"DROP TYPE IF EXISTS {schema}.message_status"))
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
"""adding roles to signal devices.
|
||||
|
||||
Revision ID: 2ef7ba690159
|
||||
Revises: a1b2c3d4e5f6
|
||||
Create Date: 2026-03-16 19:22:38.020350
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "2ef7ba690159"
|
||||
down_revision: str | None = "a1b2c3d4e5f6"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"role",
|
||||
sa.Column("name", sa.String(length=50), nullable=False),
|
||||
sa.Column("id", sa.SmallInteger(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_role")),
|
||||
sa.UniqueConstraint("name", name=op.f("uq_role_name")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"device_role",
|
||||
sa.Column("device_id", sa.Integer(), nullable=False),
|
||||
sa.Column("role_id", sa.SmallInteger(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["device_id"], [f"{schema}.signal_device.id"], name=op.f("fk_device_role_device_id_signal_device")
|
||||
),
|
||||
sa.ForeignKeyConstraint(["role_id"], [f"{schema}.role.id"], name=op.f("fk_device_role_role_id_role")),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_device_role")),
|
||||
sa.UniqueConstraint("device_id", "role_id", name="uq_device_role_device_role"),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("device_role", schema=schema)
|
||||
op.drop_table("role", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+171
@@ -0,0 +1,171 @@
|
||||
"""seprating signal_bot database.
|
||||
|
||||
Revision ID: 6b275323f435
|
||||
Revises: 2ef7ba690159
|
||||
Create Date: 2026-03-18 08:34:28.785885
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "6b275323f435"
|
||||
down_revision: str | None = "2ef7ba690159"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("device_role", schema=schema)
|
||||
op.drop_table("signal_device", schema=schema)
|
||||
op.drop_table("role", schema=schema)
|
||||
op.drop_table("dead_letter_message", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"dead_letter_message",
|
||||
sa.Column("source", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.Column("message", sa.TEXT(), autoincrement=False, nullable=False),
|
||||
sa.Column("received_at", postgresql.TIMESTAMP(timezone=True), autoincrement=False, nullable=False),
|
||||
sa.Column(
|
||||
"status",
|
||||
postgresql.ENUM("UNPROCESSED", "PROCESSED", name="message_status", schema=schema),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_dead_letter_message")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"role",
|
||||
sa.Column("name", sa.VARCHAR(length=50), autoincrement=False, nullable=False),
|
||||
sa.Column(
|
||||
"id",
|
||||
sa.SMALLINT(),
|
||||
server_default=sa.text(f"nextval('{schema}.role_id_seq'::regclass)"),
|
||||
autoincrement=True,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_role")),
|
||||
sa.UniqueConstraint(
|
||||
"name", name=op.f("uq_role_name"), postgresql_include=[], postgresql_nulls_not_distinct=False
|
||||
),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"signal_device",
|
||||
sa.Column("phone_number", sa.VARCHAR(length=50), autoincrement=False, nullable=False),
|
||||
sa.Column("safety_number", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column(
|
||||
"trust_level",
|
||||
postgresql.ENUM("VERIFIED", "UNVERIFIED", "BLOCKED", name="trust_level", schema=schema),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column("last_seen", postgresql.TIMESTAMP(timezone=True), autoincrement=False, nullable=False),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_signal_device")),
|
||||
sa.UniqueConstraint(
|
||||
"phone_number",
|
||||
name=op.f("uq_signal_device_phone_number"),
|
||||
postgresql_include=[],
|
||||
postgresql_nulls_not_distinct=False,
|
||||
),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"device_role",
|
||||
sa.Column("device_id", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("role_id", sa.SMALLINT(), autoincrement=False, nullable=False),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["device_id"], [f"{schema}.signal_device.id"], name=op.f("fk_device_role_device_id_signal_device")
|
||||
),
|
||||
sa.ForeignKeyConstraint(["role_id"], [f"{schema}.role.id"], name=op.f("fk_device_role_role_id_role")),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_device_role")),
|
||||
sa.UniqueConstraint(
|
||||
"device_id",
|
||||
"role_id",
|
||||
name=op.f("uq_device_role_device_role"),
|
||||
postgresql_include=[],
|
||||
postgresql_nulls_not_distinct=False,
|
||||
),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
+187
@@ -0,0 +1,187 @@
|
||||
"""removed ds table from richie DB.
|
||||
|
||||
Revision ID: c8a794340928
|
||||
Revises: 6b275323f435
|
||||
Create Date: 2026-03-29 15:29:23.643146
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
from sqlalchemy.dialects import postgresql
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "c8a794340928"
|
||||
down_revision: str | None = "6b275323f435"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("vote_record", schema=schema)
|
||||
op.drop_index(op.f("ix_vote_congress_chamber"), table_name="vote", schema=schema)
|
||||
op.drop_index(op.f("ix_vote_date"), table_name="vote", schema=schema)
|
||||
op.drop_index(op.f("ix_legislator_bioguide_id"), table_name="legislator", schema=schema)
|
||||
op.drop_table("legislator", schema=schema)
|
||||
op.drop_table("vote", schema=schema)
|
||||
op.drop_index(op.f("ix_bill_congress"), table_name="bill", schema=schema)
|
||||
op.drop_table("bill", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"vote",
|
||||
sa.Column("congress", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("chamber", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.Column("session", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("number", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("vote_type", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("question", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("result", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("result_text", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("vote_date", sa.DATE(), autoincrement=False, nullable=False),
|
||||
sa.Column("yea_count", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("nay_count", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("not_voting_count", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("present_count", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("bill_id", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.ForeignKeyConstraint(["bill_id"], [f"{schema}.bill.id"], name=op.f("fk_vote_bill_id_bill")),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_vote")),
|
||||
sa.UniqueConstraint(
|
||||
"congress",
|
||||
"chamber",
|
||||
"session",
|
||||
"number",
|
||||
name=op.f("uq_vote_congress_chamber_session_number"),
|
||||
postgresql_include=[],
|
||||
postgresql_nulls_not_distinct=False,
|
||||
),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index(op.f("ix_vote_date"), "vote", ["vote_date"], unique=False, schema=schema)
|
||||
op.create_index(op.f("ix_vote_congress_chamber"), "vote", ["congress", "chamber"], unique=False, schema=schema)
|
||||
op.create_table(
|
||||
"vote_record",
|
||||
sa.Column("vote_id", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("legislator_id", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("position", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["legislator_id"],
|
||||
[f"{schema}.legislator.id"],
|
||||
name=op.f("fk_vote_record_legislator_id_legislator"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["vote_id"], [f"{schema}.vote.id"], name=op.f("fk_vote_record_vote_id_vote"), ondelete="CASCADE"
|
||||
),
|
||||
sa.PrimaryKeyConstraint("vote_id", "legislator_id", name=op.f("pk_vote_record")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"legislator",
|
||||
sa.Column("bioguide_id", sa.TEXT(), autoincrement=False, nullable=False),
|
||||
sa.Column("thomas_id", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("lis_id", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("govtrack_id", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("opensecrets_id", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("fec_ids", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("first_name", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.Column("last_name", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.Column("official_full_name", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("nickname", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("birthday", sa.DATE(), autoincrement=False, nullable=True),
|
||||
sa.Column("gender", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("current_party", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("current_state", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("current_district", sa.INTEGER(), autoincrement=False, nullable=True),
|
||||
sa.Column("current_chamber", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_legislator")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index(op.f("ix_legislator_bioguide_id"), "legislator", ["bioguide_id"], unique=True, schema=schema)
|
||||
op.create_table(
|
||||
"bill",
|
||||
sa.Column("congress", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("bill_type", sa.VARCHAR(), autoincrement=False, nullable=False),
|
||||
sa.Column("number", sa.INTEGER(), autoincrement=False, nullable=False),
|
||||
sa.Column("title", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("title_short", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("official_title", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("status", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("status_at", sa.DATE(), autoincrement=False, nullable=True),
|
||||
sa.Column("sponsor_bioguide_id", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("subjects_top_term", sa.VARCHAR(), autoincrement=False, nullable=True),
|
||||
sa.Column("id", sa.INTEGER(), autoincrement=True, nullable=False),
|
||||
sa.Column(
|
||||
"created",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.Column(
|
||||
"updated",
|
||||
postgresql.TIMESTAMP(timezone=True),
|
||||
server_default=sa.text("now()"),
|
||||
autoincrement=False,
|
||||
nullable=False,
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_bill")),
|
||||
sa.UniqueConstraint(
|
||||
"congress",
|
||||
"bill_type",
|
||||
"number",
|
||||
name=op.f("uq_bill_congress_type_number"),
|
||||
postgresql_include=[],
|
||||
postgresql_nulls_not_distinct=False,
|
||||
),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index(op.f("ix_bill_congress"), "bill", ["congress"], unique=False, schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+93
@@ -0,0 +1,93 @@
|
||||
"""adding audiobook libreary metadata.
|
||||
|
||||
Revision ID: d7864d1ffc17
|
||||
Revises: c8a794340928
|
||||
Create Date: 2026-06-03 20:24:09.200837
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "d7864d1ffc17"
|
||||
down_revision: str | None = "c8a794340928"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"audiobook_author",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook_author")),
|
||||
sa.UniqueConstraint("name", name=op.f("uq_audiobook_author_name")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"audiobook_series",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("author_id", sa.Integer(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["author_id"],
|
||||
[f"{schema}.audiobook_author.id"],
|
||||
name=op.f("fk_audiobook_series_author_id_audiobook_author"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook_series")),
|
||||
sa.UniqueConstraint("author_id", "name", name=op.f("uq_audiobook_series_author_id")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"audiobook",
|
||||
sa.Column("title", sa.String(), nullable=False),
|
||||
sa.Column("author_id", sa.Integer(), nullable=False),
|
||||
sa.Column("series_id", sa.Integer(), nullable=True),
|
||||
sa.Column("series_index", sa.Integer(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["author_id"],
|
||||
[f"{schema}.audiobook_author.id"],
|
||||
name=op.f("fk_audiobook_author_id_audiobook_author"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["series_id"],
|
||||
[f"{schema}.audiobook_series.id"],
|
||||
name=op.f("fk_audiobook_series_id_audiobook_series"),
|
||||
ondelete="SET NULL",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_audiobook")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("audiobook", schema=schema)
|
||||
op.drop_table("audiobook_series", schema=schema)
|
||||
op.drop_table("audiobook_author", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
@@ -0,0 +1,200 @@
|
||||
"""add ebook search tables.
|
||||
|
||||
Revision ID: 2db132cace1a
|
||||
Revises: b3c60cc5beb5
|
||||
Create Date: 2026-06-10 22:10:54.379159
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pgvector
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "2db132cace1a"
|
||||
down_revision: str | None = "b3c60cc5beb5"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"ebook_embedding_model",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("dimension", sa.Integer(), nullable=False),
|
||||
sa.Column("is_default", sa.Boolean(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_embedding_model")),
|
||||
sa.UniqueConstraint("name", name=op.f("uq_ebook_embedding_model_name")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_source",
|
||||
sa.Column("title", sa.String(), nullable=False),
|
||||
sa.Column("author", sa.String(), nullable=True),
|
||||
sa.Column("language", sa.String(), nullable=True),
|
||||
sa.Column("publisher", sa.String(), nullable=True),
|
||||
sa.Column("identifier", sa.String(), nullable=True),
|
||||
sa.Column("file_path", sa.String(), nullable=False),
|
||||
sa.Column("file_sha256", sa.String(length=64), nullable=False),
|
||||
sa.Column("file_mtime", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column("file_size", sa.BigInteger(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_source")),
|
||||
sa.UniqueConstraint("file_path", name=op.f("uq_ebook_source_file_path")),
|
||||
sa.UniqueConstraint("file_sha256", name=op.f("uq_ebook_source_file_sha256")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_chapter",
|
||||
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||
sa.Column("spine_index", sa.Integer(), nullable=False),
|
||||
sa.Column("title", sa.String(), nullable=True),
|
||||
sa.Column("href", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["source_id"],
|
||||
[f"{schema}.ebook_source.id"],
|
||||
name=op.f("fk_ebook_chapter_source_id_ebook_source"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chapter")),
|
||||
sa.UniqueConstraint("source_id", "spine_index", name=op.f("uq_ebook_chapter_source_id")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_chunk",
|
||||
sa.Column("source_id", sa.Integer(), nullable=False),
|
||||
sa.Column("chapter_id", sa.Integer(), nullable=True),
|
||||
sa.Column("chunk_index", sa.Integer(), nullable=False),
|
||||
sa.Column("text", sa.String(), nullable=False),
|
||||
sa.Column("token_start", sa.Integer(), nullable=False),
|
||||
sa.Column("token_count", sa.Integer(), nullable=False),
|
||||
sa.Column("page_label", sa.String(), nullable=True),
|
||||
sa.Column("content_sha256", sa.String(length=64), nullable=False),
|
||||
sa.Column("search_text", sa.String(), nullable=False),
|
||||
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["chapter_id"],
|
||||
[f"{schema}.ebook_chapter.id"],
|
||||
name=op.f("fk_ebook_chunk_chapter_id_ebook_chapter"),
|
||||
ondelete="SET NULL",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["source_id"],
|
||||
[f"{schema}.ebook_source.id"],
|
||||
name=op.f("fk_ebook_chunk_source_id_ebook_source"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk")),
|
||||
sa.UniqueConstraint("source_id", "chunk_index", name="uq_ebook_chunk_source_id_chunk_index"),
|
||||
sa.UniqueConstraint("source_id", "content_sha256", name="uq_ebook_chunk_source_id_content_sha256"),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_chunk_embedding_1024",
|
||||
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=1024), nullable=False),
|
||||
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["chunk_id"],
|
||||
[f"{schema}.ebook_chunk.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_1024_chunk_id_ebook_chunk"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["model_id"],
|
||||
[f"{schema}.ebook_embedding_model.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_1024_model_id_ebook_embedding_model"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_1024")),
|
||||
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_1024_chunk_id")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_chunk_embedding_2560",
|
||||
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=2560), nullable=False),
|
||||
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["chunk_id"],
|
||||
[f"{schema}.ebook_chunk.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_2560_chunk_id_ebook_chunk"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["model_id"],
|
||||
[f"{schema}.ebook_embedding_model.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_2560_model_id_ebook_embedding_model"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_2560")),
|
||||
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_2560_chunk_id")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"ebook_chunk_embedding_4096",
|
||||
sa.Column("chunk_id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("model_id", sa.Integer(), nullable=False),
|
||||
sa.Column("embedding", pgvector.sqlalchemy.vector.VECTOR(dim=4096), nullable=False),
|
||||
sa.Column("id", sa.BigInteger(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(
|
||||
["chunk_id"],
|
||||
[f"{schema}.ebook_chunk.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_4096_chunk_id_ebook_chunk"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.ForeignKeyConstraint(
|
||||
["model_id"],
|
||||
[f"{schema}.ebook_embedding_model.id"],
|
||||
name=op.f("fk_ebook_chunk_embedding_4096_model_id_ebook_embedding_model"),
|
||||
ondelete="CASCADE",
|
||||
),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_ebook_chunk_embedding_4096")),
|
||||
sa.UniqueConstraint("chunk_id", "model_id", name=op.f("uq_ebook_chunk_embedding_4096_chunk_id")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("ebook_chunk_embedding_4096", schema=schema)
|
||||
op.drop_table("ebook_chunk_embedding_2560", schema=schema)
|
||||
op.drop_table("ebook_chunk_embedding_1024", schema=schema)
|
||||
op.drop_table("ebook_chunk", schema=schema)
|
||||
op.drop_table("ebook_chapter", schema=schema)
|
||||
op.drop_table("ebook_source", schema=schema)
|
||||
op.drop_table("ebook_embedding_model", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+63
@@ -0,0 +1,63 @@
|
||||
"""updated series_index to float and added UniqueConstraint to audiobook and audiobook_author.
|
||||
|
||||
Revision ID: b3c60cc5beb5
|
||||
Revises: d7864d1ffc17
|
||||
Create Date: 2026-06-10 20:02:43.073725
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "b3c60cc5beb5"
|
||||
down_revision: str | None = "d7864d1ffc17"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.alter_column(
|
||||
"audiobook",
|
||||
"series_index",
|
||||
existing_type=sa.INTEGER(),
|
||||
type_=sa.Float(),
|
||||
existing_nullable=False,
|
||||
schema=schema,
|
||||
)
|
||||
op.create_unique_constraint(
|
||||
op.f("uq_audiobook_author_id"),
|
||||
"audiobook",
|
||||
["author_id", "series_id", "title"],
|
||||
schema=schema,
|
||||
postgresql_nulls_not_distinct=True,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_constraint(op.f("uq_audiobook_author_id"), "audiobook", schema=schema, type_="unique")
|
||||
op.alter_column(
|
||||
"audiobook",
|
||||
"series_index",
|
||||
existing_type=sa.Float(),
|
||||
type_=sa.INTEGER(),
|
||||
existing_nullable=False,
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
"""add 1024 ebook embedding cosine index.
|
||||
|
||||
Revision ID: c460105682d2
|
||||
Revises: 2db132cace1a
|
||||
Create Date: 2026-06-13 19:53:45.680289
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "c460105682d2"
|
||||
down_revision: str | None = "2db132cace1a"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_index(
|
||||
"ix_ebook_chunk_embedding_1024_embedding_cosine",
|
||||
"ebook_chunk_embedding_1024",
|
||||
["embedding"],
|
||||
unique=False,
|
||||
schema=schema,
|
||||
postgresql_using="hnsw",
|
||||
postgresql_ops={"embedding": "vector_cosine_ops"},
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_index(
|
||||
"ix_ebook_chunk_embedding_1024_embedding_cosine",
|
||||
table_name="ebook_chunk_embedding_1024",
|
||||
schema=schema,
|
||||
postgresql_using="hnsw",
|
||||
postgresql_ops={"embedding": "vector_cosine_ops"},
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
@@ -0,0 +1,103 @@
|
||||
"""adding haproxy data.
|
||||
|
||||
Revision ID: 96d72c748c24
|
||||
Revises: c460105682d2
|
||||
Create Date: 2026-06-23 16:37:17.768851
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import RichieBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "96d72c748c24"
|
||||
down_revision: str | None = "c460105682d2"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = RichieBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"haproxy_request",
|
||||
sa.Column("line_hash", sa.String(), nullable=False),
|
||||
sa.Column("requested_at", sa.DateTime(timezone=True), nullable=False),
|
||||
sa.Column("client_ip", sa.String(), nullable=False),
|
||||
sa.Column("client_port", sa.Integer(), nullable=False),
|
||||
sa.Column("frontend", sa.String(), nullable=False),
|
||||
sa.Column("ssl", sa.Boolean(), nullable=False),
|
||||
sa.Column("backend", sa.String(), nullable=False),
|
||||
sa.Column("server", sa.String(), nullable=False),
|
||||
sa.Column("time_request", sa.Integer(), nullable=False),
|
||||
sa.Column("time_queue", sa.Integer(), nullable=False),
|
||||
sa.Column("time_connect", sa.Integer(), nullable=False),
|
||||
sa.Column("time_response", sa.Integer(), nullable=False),
|
||||
sa.Column("time_total", sa.Integer(), nullable=False),
|
||||
sa.Column("status_code", sa.Integer(), nullable=False),
|
||||
sa.Column("bytes_read", sa.BigInteger(), nullable=False),
|
||||
sa.Column("termination_state", sa.String(), nullable=False),
|
||||
sa.Column("active_connections", sa.Integer(), nullable=False),
|
||||
sa.Column("frontend_connections", sa.Integer(), nullable=False),
|
||||
sa.Column("backend_connections", sa.Integer(), nullable=False),
|
||||
sa.Column("server_connections", sa.Integer(), nullable=False),
|
||||
sa.Column("retries", sa.Integer(), nullable=False),
|
||||
sa.Column("server_queue", sa.Integer(), nullable=False),
|
||||
sa.Column("backend_queue", sa.Integer(), nullable=False),
|
||||
sa.Column("host", sa.String(), nullable=True),
|
||||
sa.Column("user_agent", sa.String(), nullable=True),
|
||||
sa.Column("method", sa.String(), nullable=False),
|
||||
sa.Column("target", sa.String(), nullable=False),
|
||||
sa.Column("path", sa.String(), nullable=False),
|
||||
sa.Column("query", sa.String(), nullable=True),
|
||||
sa.Column("http_version", sa.String(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_haproxy_request")),
|
||||
sa.UniqueConstraint("line_hash", name=op.f("uq_haproxy_request_line_hash")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_index(op.f("ix_haproxy_request_backend"), "haproxy_request", ["backend"], unique=False, schema=schema)
|
||||
op.create_index(op.f("ix_haproxy_request_client_ip"), "haproxy_request", ["client_ip"], unique=False, schema=schema)
|
||||
op.create_index(op.f("ix_haproxy_request_host"), "haproxy_request", ["host"], unique=False, schema=schema)
|
||||
op.create_index(op.f("ix_haproxy_request_path"), "haproxy_request", ["path"], unique=False, schema=schema)
|
||||
op.create_index(
|
||||
op.f("ix_haproxy_request_requested_at"), "haproxy_request", ["requested_at"], unique=False, schema=schema
|
||||
)
|
||||
op.create_index(
|
||||
op.f("ix_haproxy_request_status_code"), "haproxy_request", ["status_code"], unique=False, schema=schema
|
||||
)
|
||||
op.create_index(
|
||||
op.f("ix_haproxy_request_time_response"), "haproxy_request", ["time_response"], unique=False, schema=schema
|
||||
)
|
||||
op.create_index(
|
||||
op.f("ix_haproxy_request_user_agent"), "haproxy_request", ["user_agent"], unique=False, schema=schema
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_index(op.f("ix_haproxy_request_user_agent"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_time_response"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_status_code"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_requested_at"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_path"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_host"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_client_ip"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_index(op.f("ix_haproxy_request_backend"), table_name="haproxy_request", schema=schema)
|
||||
op.drop_table("haproxy_request", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
@@ -0,0 +1,36 @@
|
||||
"""${message}.
|
||||
|
||||
Revision ID: ${up_revision}
|
||||
Revises: ${down_revision | comma,n}
|
||||
Create Date: ${create_date}
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
|
||||
from alembic import op
|
||||
from python.orm import ${config.attributes["base"].__name__}
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = ${repr(up_revision)}
|
||||
down_revision: str | None = ${repr(down_revision)}
|
||||
branch_labels: str | Sequence[str] | None = ${repr(branch_labels)}
|
||||
depends_on: str | Sequence[str] | None = ${repr(depends_on)}
|
||||
|
||||
schema=${config.attributes["base"].__name__}.schema_name
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
${upgrades if upgrades else "pass"}
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
${downgrades if downgrades else "pass"}
|
||||
+80
@@ -0,0 +1,80 @@
|
||||
"""starting van invintory.
|
||||
|
||||
Revision ID: 15e733499804
|
||||
Revises:
|
||||
Create Date: 2026-03-08 00:18:20.759720
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
from python.orm import VanInventoryBase
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "15e733499804"
|
||||
down_revision: str | None = None
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
schema = VanInventoryBase.schema_name
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Upgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.create_table(
|
||||
"items",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("quantity", sa.Float(), nullable=False),
|
||||
sa.Column("unit", sa.String(), nullable=False),
|
||||
sa.Column("category", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_items")),
|
||||
sa.UniqueConstraint("name", name=op.f("uq_items_name")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"meals",
|
||||
sa.Column("name", sa.String(), nullable=False),
|
||||
sa.Column("instructions", sa.String(), nullable=True),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_meals")),
|
||||
sa.UniqueConstraint("name", name=op.f("uq_meals_name")),
|
||||
schema=schema,
|
||||
)
|
||||
op.create_table(
|
||||
"meal_ingredients",
|
||||
sa.Column("meal_id", sa.Integer(), nullable=False),
|
||||
sa.Column("item_id", sa.Integer(), nullable=False),
|
||||
sa.Column("quantity_needed", sa.Float(), nullable=False),
|
||||
sa.Column("id", sa.Integer(), nullable=False),
|
||||
sa.Column("created", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.Column("updated", sa.DateTime(timezone=True), server_default=sa.text("now()"), nullable=False),
|
||||
sa.ForeignKeyConstraint(["item_id"], [f"{schema}.items.id"], name=op.f("fk_meal_ingredients_item_id_items")),
|
||||
sa.ForeignKeyConstraint(["meal_id"], [f"{schema}.meals.id"], name=op.f("fk_meal_ingredients_meal_id_meals")),
|
||||
sa.PrimaryKeyConstraint("id", name=op.f("pk_meal_ingredients")),
|
||||
sa.UniqueConstraint("meal_id", "item_id", name=op.f("uq_meal_ingredients_meal_id")),
|
||||
schema=schema,
|
||||
)
|
||||
# ### end Alembic commands ###
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Downgrade."""
|
||||
# ### commands auto generated by Alembic - please adjust! ###
|
||||
op.drop_table("meal_ingredients", schema=schema)
|
||||
op.drop_table("meals", schema=schema)
|
||||
op.drop_table("items", schema=schema)
|
||||
# ### end Alembic commands ###
|
||||
+9
-34
@@ -3,28 +3,23 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from datetime import UTC, datetime
|
||||
from os import getenv
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, Popen
|
||||
|
||||
from apprise import Apprise
|
||||
from python.logging_config import configure_logger as _configure_logger
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def configure_logger(level: str = "INFO") -> None:
|
||||
"""Configure the logger.
|
||||
def get_repo_dir() -> Path:
|
||||
"""Return the repository root directory."""
|
||||
return Path(__file__).resolve().parents[1]
|
||||
|
||||
Args:
|
||||
level (str, optional): The logging level. Defaults to "INFO".
|
||||
"""
|
||||
logging.basicConfig(
|
||||
level=level,
|
||||
datefmt="%Y-%m-%dT%H:%M:%S%z",
|
||||
format="%(asctime)s %(levelname)s %(filename)s:%(lineno)d - %(message)s",
|
||||
handlers=[logging.StreamHandler(sys.stdout)],
|
||||
)
|
||||
|
||||
def configure_logger(level: str = "INFO") -> None:
|
||||
"""Configure the logger."""
|
||||
_configure_logger(level)
|
||||
|
||||
|
||||
def bash_wrapper(command: str) -> tuple[str, int]:
|
||||
@@ -47,26 +42,6 @@ def bash_wrapper(command: str) -> tuple[str, int]:
|
||||
return output.decode(), process.returncode
|
||||
|
||||
|
||||
def signal_alert(body: str, title: str = "") -> None:
|
||||
"""Send a signal alert.
|
||||
|
||||
Args:
|
||||
body (str): The body of the alert.
|
||||
title (str, optional): The title of the alert. Defaults to "".
|
||||
"""
|
||||
apprise_client = Apprise()
|
||||
|
||||
from_phone = getenv("SIGNAL_ALERT_FROM_PHONE")
|
||||
to_phone = getenv("SIGNAL_ALERT_TO_PHONE")
|
||||
if not from_phone or not to_phone:
|
||||
logger.info("SIGNAL_ALERT_FROM_PHONE or SIGNAL_ALERT_TO_PHONE not set")
|
||||
return
|
||||
|
||||
apprise_client.add(f"signal://localhost:8989/{from_phone}/{to_phone}")
|
||||
|
||||
apprise_client.notify(title=title, body=body)
|
||||
|
||||
|
||||
def utcnow() -> datetime:
|
||||
"""Get the current UTC time."""
|
||||
return datetime.now(tz=UTC)
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
"""CLI wrapper around alembic for multi-database support.
|
||||
|
||||
Usage:
|
||||
database <db_name> <command> [args...]
|
||||
|
||||
Examples:
|
||||
database richie check
|
||||
database richie upgrade head
|
||||
database richie downgrade head-1
|
||||
database richie revision --autogenerate -m "add meals table"
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from importlib import import_module
|
||||
from typing import TYPE_CHECKING, Annotated
|
||||
|
||||
import typer
|
||||
from alembic.config import CommandLine, Config
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from sqlalchemy.orm import DeclarativeBase
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DatabaseConfig:
|
||||
"""Configuration for a database."""
|
||||
|
||||
env_prefix: str
|
||||
version_location: str
|
||||
base_module: str
|
||||
base_class_name: str
|
||||
models_module: str
|
||||
script_location: str = "python/alembic"
|
||||
file_template: str = "%%(year)d_%%(month).2d_%%(day).2d-%%(slug)s_%%(rev)s"
|
||||
|
||||
def get_base(self) -> type[DeclarativeBase]:
|
||||
"""Import and return the Base class."""
|
||||
module = import_module(self.base_module)
|
||||
return getattr(module, self.base_class_name)
|
||||
|
||||
def import_models(self) -> None:
|
||||
"""Import ORM models so alembic autogenerate can detect them."""
|
||||
import_module(self.models_module)
|
||||
|
||||
def alembic_config(self) -> Config:
|
||||
"""Build an alembic Config for this database."""
|
||||
cfg = Config()
|
||||
cfg.set_main_option("script_location", self.script_location)
|
||||
cfg.set_main_option("file_template", self.file_template)
|
||||
cfg.set_main_option("prepend_sys_path", ".")
|
||||
cfg.set_main_option("version_path_separator", "os")
|
||||
cfg.set_main_option("version_locations", self.version_location)
|
||||
cfg.set_main_option("revision_environment", "true")
|
||||
cfg.set_section_option("post_write_hooks", "hooks", "dynamic_schema,import_postgresql,ruff")
|
||||
cfg.set_section_option("post_write_hooks", "dynamic_schema.type", "dynamic_schema")
|
||||
cfg.set_section_option("post_write_hooks", "import_postgresql.type", "import_postgresql")
|
||||
cfg.set_section_option("post_write_hooks", "ruff.type", "ruff")
|
||||
cfg.attributes["base"] = self.get_base()
|
||||
cfg.attributes["env_prefix"] = self.env_prefix
|
||||
self.import_models()
|
||||
return cfg
|
||||
|
||||
|
||||
DATABASES: dict[str, DatabaseConfig] = {
|
||||
"richie": DatabaseConfig(
|
||||
env_prefix="RICHIE",
|
||||
version_location="python/alembic/richie/versions",
|
||||
base_module="python.orm.richie.base",
|
||||
base_class_name="RichieBase",
|
||||
models_module="python.orm.richie",
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
app = typer.Typer(help="Multi-database alembic wrapper.")
|
||||
|
||||
|
||||
@app.command(
|
||||
context_settings={"allow_extra_args": True, "ignore_unknown_options": True},
|
||||
)
|
||||
def main(
|
||||
ctx: typer.Context,
|
||||
db_name: Annotated[str, typer.Argument(help=f"Database name. Options: {', '.join(DATABASES)}")],
|
||||
command: Annotated[str, typer.Argument(help="Alembic command (upgrade, downgrade, revision, check, etc.)")],
|
||||
) -> None:
|
||||
"""Run an alembic command against the specified database."""
|
||||
db_config = DATABASES.get(db_name)
|
||||
if not db_config:
|
||||
typer.echo(f"Unknown database: {db_name!r}. Available: {', '.join(DATABASES)}", err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
alembic_cfg = db_config.alembic_config()
|
||||
|
||||
cmd_line = CommandLine()
|
||||
options = cmd_line.parser.parse_args([command, *ctx.args])
|
||||
cmd_line.run_cmd(alembic_cfg, options)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
"""EPUB search package."""
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Grounded answer generation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from python.ebook_search.llm_interface import request_chat_completion
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
from python.ebook_search.search import SearchResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def answer_query(query: str, results: list[SearchResult], config: EbookSearchConfig) -> str:
|
||||
"""Answer a question using only retrieved chunks."""
|
||||
if not config.answer_enabled:
|
||||
logger.info("ebook_answer_skipped_disabled")
|
||||
return "Answer generation is disabled. Source chunks are shown below."
|
||||
|
||||
if not results:
|
||||
logger.info("ebook_answer_skipped_no_results")
|
||||
return "No relevant sources were found."
|
||||
|
||||
logger.info(
|
||||
"ebook_answer_request_start base_url=%s model=%s sources=%s query_length=%s",
|
||||
config.vllm_base_url,
|
||||
config.chat_model,
|
||||
len(results),
|
||||
len(query),
|
||||
)
|
||||
context = "\n\n".join(
|
||||
f"[{index}] {result.source_title}{' - ' + result.chapter_title if result.chapter_title else ''}\n{result.text}"
|
||||
for index, result in enumerate(results, start=1)
|
||||
)
|
||||
content = request_chat_completion(
|
||||
config,
|
||||
[
|
||||
{
|
||||
"role": "system",
|
||||
"content": (
|
||||
"Answer only from the provided context. Cite sources with bracketed numbers like [1]. "
|
||||
"If the context is insufficient, say so."
|
||||
),
|
||||
},
|
||||
{"role": "user", "content": f"Question:\n{query}\n\nContext:\n{context}"},
|
||||
],
|
||||
)
|
||||
|
||||
logger.info(
|
||||
"ebook_answer_request_complete model=%s answer_length=%s",
|
||||
config.chat_model,
|
||||
len(content),
|
||||
)
|
||||
return content or "The model returned an empty answer."
|
||||
@@ -0,0 +1 @@
|
||||
"""Web and external API adapters for EPUB search."""
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Background BM25 refresh tasks for the web app."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from threading import Timer
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.bm25_corpus import load_bm25_corpus, refresh_bm25_corpus
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from fastapi import FastAPI
|
||||
from sqlalchemy.engine import Engine
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def schedule_bm25_refresh(app: FastAPI) -> None:
|
||||
"""Schedule a delayed BM25 corpus refresh, replacing any pending refresh."""
|
||||
existing_timer = getattr(app.state, "bm25_refresh_timer", None)
|
||||
if existing_timer is not None:
|
||||
existing_timer.cancel()
|
||||
|
||||
timer = Timer(app.state.config.bm25_refresh_delay_seconds, refresh_bm25_for_app, args=(app,))
|
||||
timer.daemon = True
|
||||
timer.start()
|
||||
app.state.bm25_refresh_timer = timer
|
||||
logger.info(
|
||||
"ebook_bm25_refresh_scheduled delay_seconds=%s",
|
||||
app.state.config.bm25_refresh_delay_seconds,
|
||||
)
|
||||
|
||||
|
||||
def cancel_bm25_refresh(app: FastAPI) -> None:
|
||||
"""Cancel any pending BM25 corpus refresh."""
|
||||
existing_timer = getattr(app.state, "bm25_refresh_timer", None)
|
||||
if existing_timer is not None:
|
||||
existing_timer.cancel()
|
||||
app.state.bm25_refresh_timer = None
|
||||
logger.info("ebook_bm25_refresh_cancelled")
|
||||
|
||||
|
||||
def refresh_bm25_for_app(app: FastAPI) -> None:
|
||||
"""Refresh the BM25 corpus using the app engine and config."""
|
||||
try:
|
||||
refresh_bm25_for_engine(app.state.engine, app.state.config)
|
||||
except Exception:
|
||||
logger.exception("ebook_bm25_refresh_failed")
|
||||
|
||||
|
||||
def refresh_bm25_for_engine(engine: Engine, config: EbookSearchConfig) -> None:
|
||||
"""Refresh the BM25 corpus using a SQLAlchemy engine."""
|
||||
with Session(engine) as session:
|
||||
refresh_bm25_corpus(session, config)
|
||||
load_bm25_corpus.cache_clear()
|
||||
logger.info("ebook_bm25_corpus_cache_cleared_after_refresh")
|
||||
@@ -0,0 +1,24 @@
|
||||
"""FastAPI dependencies for the EPUB search app."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Annotated
|
||||
|
||||
from fastapi import Depends, Request
|
||||
from sqlalchemy.engine import Engine
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
|
||||
def get_config(request: Request) -> EbookSearchConfig:
|
||||
"""Get the loaded search config from app state."""
|
||||
return request.app.state.config
|
||||
|
||||
|
||||
def get_engine(request: Request) -> Engine:
|
||||
"""Get the database engine from app state."""
|
||||
return request.app.state.engine
|
||||
|
||||
|
||||
AppConfig = Annotated[EbookSearchConfig, Depends(get_config)]
|
||||
AppEngine = Annotated[Engine, Depends(get_engine)]
|
||||
@@ -0,0 +1,86 @@
|
||||
"""FastAPI HTMX app for EPUB search."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import TYPE_CHECKING, Annotated
|
||||
|
||||
import typer
|
||||
import uvicorn
|
||||
from fastapi import FastAPI
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.common import configure_logger
|
||||
from python.ebook_search.api.bm25_tasks import cancel_bm25_refresh
|
||||
from python.ebook_search.api.routes import admin_router, health_router, page_router, search_router
|
||||
from python.ebook_search.api.web import STATIC_DIR
|
||||
from python.ebook_search.bm25_corpus import ensure_bm25_corpus
|
||||
from python.ebook_search.config import load_config
|
||||
from python.fastapi_tools import ZstdMiddleware
|
||||
from python.orm.common import get_postgres_engine
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI) -> AsyncIterator[None]:
|
||||
"""Manage application startup and shutdown resources."""
|
||||
logger.info("ebook_search_startup")
|
||||
config = load_config()
|
||||
app.state.config = config
|
||||
logger.info(
|
||||
"ebook_search_config_loaded top_k=%s embedding_model=%s embedding_base_url=%s vllm_base_url=%s "
|
||||
"rerank_enabled=%s answer_enabled=%s library_paths=%s",
|
||||
config.top_k,
|
||||
config.embedding_model,
|
||||
config.embedding_base_url,
|
||||
config.vllm_base_url,
|
||||
config.rerank.enabled,
|
||||
config.answer_enabled,
|
||||
len(config.library_paths),
|
||||
)
|
||||
if not config.library_paths:
|
||||
logger.warning("ebook_search_no_library_paths_configured")
|
||||
app.state.engine = get_postgres_engine(name="RICHIE", vector_engine=True)
|
||||
with Session(app.state.engine) as session:
|
||||
ensure_bm25_corpus(session, config)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
logger.info("ebook_search_shutdown")
|
||||
cancel_bm25_refresh(app)
|
||||
app.state.engine.dispose()
|
||||
|
||||
|
||||
def create_app() -> FastAPI:
|
||||
"""Create the EPUB search web app."""
|
||||
app = FastAPI(title="EPUB Search", lifespan=lifespan)
|
||||
app.add_middleware(ZstdMiddleware)
|
||||
app.mount("/static", StaticFiles(directory=STATIC_DIR), name="static")
|
||||
|
||||
app.include_router(admin_router)
|
||||
app.include_router(health_router)
|
||||
app.include_router(page_router)
|
||||
app.include_router(search_router)
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def serve(
|
||||
host: Annotated[str, typer.Option("--host", "-h", help="Host to bind to")] = "127.0.0.1",
|
||||
port: Annotated[int, typer.Option("--port", "-p", help="Port to bind to")] = 8070,
|
||||
log_level: Annotated[str, typer.Option("--log-level", "-l", help="Log level")] = "INFO",
|
||||
) -> None:
|
||||
"""Start the EPUB search server."""
|
||||
configure_logger(log_level)
|
||||
uvicorn.run(create_app(), host=host, port=port)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
typer.run(serve)
|
||||
@@ -0,0 +1,13 @@
|
||||
"""EPUB search web route modules."""
|
||||
|
||||
from python.ebook_search.api.routes.admin import router as admin_router
|
||||
from python.ebook_search.api.routes.health import router as health_router
|
||||
from python.ebook_search.api.routes.page import router as page_router
|
||||
from python.ebook_search.api.routes.search import router as search_router
|
||||
|
||||
__all__ = [
|
||||
"admin_router",
|
||||
"health_router",
|
||||
"page_router",
|
||||
"search_router",
|
||||
]
|
||||
@@ -0,0 +1,103 @@
|
||||
"""Admin routes for the EPUB search web UI."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Request
|
||||
from fastapi.responses import HTMLResponse
|
||||
|
||||
from python.ebook_search.api.bm25_tasks import schedule_bm25_refresh
|
||||
from python.ebook_search.api.dependencies import (
|
||||
AppConfig, # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
)
|
||||
from python.ebook_search.api.web import templates
|
||||
from python.ebook_search.embeddings import embed_missing_chunks, embedding_model_stats
|
||||
from python.ebook_search.ingest import ingest_configured_paths
|
||||
from python.fastapi_tools import DbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/admin")
|
||||
|
||||
|
||||
@router.get("", response_class=HTMLResponse)
|
||||
def admin(request: Request, config: AppConfig, session: DbSession) -> HTMLResponse:
|
||||
"""Render the admin page."""
|
||||
stats = embedding_model_stats(session)
|
||||
logger.info("ebook_admin_page_loaded models=%s", len(stats))
|
||||
return templates.TemplateResponse(request, "admin.html", {"config": config, "stats": stats})
|
||||
|
||||
|
||||
@router.post("/scan", response_class=HTMLResponse)
|
||||
def scan_library(request: Request, config: AppConfig, session: DbSession) -> HTMLResponse:
|
||||
"""Scan configured library paths for EPUB changes."""
|
||||
try:
|
||||
count = ingest_configured_paths(session, config)
|
||||
session.commit()
|
||||
except Exception as error:
|
||||
logger.exception("ebook_admin_scan_failed")
|
||||
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||
|
||||
logger.info("ebook_admin_scan_complete changed_files=%s", count)
|
||||
if count > 0:
|
||||
schedule_bm25_refresh(request.app)
|
||||
return templates.TemplateResponse(request, "partials/admin_status.html", {"message": f"Indexed {count} EPUBs"})
|
||||
|
||||
|
||||
@router.post("/embed-missing", response_class=HTMLResponse)
|
||||
def embed_missing(request: Request, config: AppConfig, session: DbSession) -> HTMLResponse:
|
||||
"""Embed chunks missing vectors for the configured model."""
|
||||
try:
|
||||
count = embed_missing_chunks(session, config)
|
||||
session.commit()
|
||||
except Exception as error:
|
||||
logger.exception("ebook_admin_embed_missing_failed")
|
||||
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||
|
||||
logger.info("ebook_admin_embed_missing_complete chunks=%s", count)
|
||||
return templates.TemplateResponse(
|
||||
request,
|
||||
"partials/admin_status.html",
|
||||
{"message": f"Embedded {count} chunks"},
|
||||
)
|
||||
|
||||
|
||||
@router.post("/embed-all", response_class=HTMLResponse)
|
||||
def embed_all(request: Request, config: AppConfig, session: DbSession) -> HTMLResponse:
|
||||
"""Embed all chunks missing vectors in fixed-size batches."""
|
||||
total = 0
|
||||
batches = 0
|
||||
try:
|
||||
while True:
|
||||
count = embed_missing_chunks(session, config)
|
||||
if count == 0:
|
||||
break
|
||||
session.commit()
|
||||
total += count
|
||||
batches += 1
|
||||
logger.info(
|
||||
"ebook_admin_embed_all_batch_complete batch=%s chunks=%s total_chunks=%s",
|
||||
batches,
|
||||
count,
|
||||
total,
|
||||
)
|
||||
except Exception as error:
|
||||
logger.exception(
|
||||
"ebook_admin_embed_all_failed batches=%s chunks=%s",
|
||||
batches,
|
||||
total,
|
||||
)
|
||||
return templates.TemplateResponse(
|
||||
request,
|
||||
"partials/error.html",
|
||||
{"message": f"Embed all failed after {total} chunks in {batches} batches: {error}"},
|
||||
status_code=500,
|
||||
)
|
||||
|
||||
logger.info("ebook_admin_embed_all_complete batches=%s chunks=%s", batches, total)
|
||||
return templates.TemplateResponse(
|
||||
request,
|
||||
"partials/admin_status.html",
|
||||
{"message": f"Embedded {total} chunks in {batches} batches of {config.embedding_batch_size}"},
|
||||
)
|
||||
@@ -0,0 +1,97 @@
|
||||
"""Liveness and readiness routes for the EPUB search service."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from http import HTTPStatus
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from fastapi import APIRouter
|
||||
from fastapi.responses import JSONResponse
|
||||
from sqlalchemy import literal, select
|
||||
from sqlalchemy.exc import SQLAlchemyError
|
||||
|
||||
from python.ebook_search.api.dependencies import (
|
||||
AppConfig, # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
)
|
||||
from python.ebook_search.bm25_corpus import bm25_index_exists, bm25_index_path, read_bm25_manifest
|
||||
from python.ebook_search.llm_interface import check_chat_endpoint, check_embedding_endpoint
|
||||
from python.fastapi_tools import DbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
@router.get("/health")
|
||||
def health() -> dict[str, str]:
|
||||
"""Liveness probe that returns ok without touching dependencies."""
|
||||
return {"status": "ok"}
|
||||
|
||||
|
||||
@router.get("/ready")
|
||||
def ready(config: AppConfig, session: DbSession) -> JSONResponse:
|
||||
"""Readiness probe reporting database, embedding endpoint, and BM25 index status."""
|
||||
database_ok = check_database(session)
|
||||
embedding_ok = check_embedding_endpoint(config)
|
||||
chat_status = chat_endpoint_status(config)
|
||||
bm25_status = check_bm25_status(config)
|
||||
|
||||
checks = {
|
||||
"database": "ok" if database_ok else "fail",
|
||||
"embedding": "ok" if embedding_ok else "fail",
|
||||
"chat": chat_status,
|
||||
"bm25": bm25_status,
|
||||
}
|
||||
if not database_ok:
|
||||
status = "unavailable"
|
||||
status_code = HTTPStatus.SERVICE_UNAVAILABLE
|
||||
elif not embedding_ok or chat_status == "fail" or bm25_status == "missing":
|
||||
status = "degraded"
|
||||
status_code = HTTPStatus.OK
|
||||
else:
|
||||
status = "ready"
|
||||
status_code = HTTPStatus.OK
|
||||
|
||||
logger.info(
|
||||
"ebook_ready_check status=%s database=%s embedding=%s chat=%s bm25=%s",
|
||||
status,
|
||||
database_ok,
|
||||
embedding_ok,
|
||||
chat_status,
|
||||
bm25_status,
|
||||
)
|
||||
return JSONResponse(content={"status": status, "checks": checks}, status_code=status_code)
|
||||
|
||||
|
||||
def chat_endpoint_status(config: EbookSearchConfig) -> str:
|
||||
"""Return the answering chat endpoint status, or disabled when answers are off."""
|
||||
if not config.answer_enabled:
|
||||
return "disabled"
|
||||
return "ok" if check_chat_endpoint(config) else "fail"
|
||||
|
||||
|
||||
def check_database(session: Session) -> bool:
|
||||
"""Return whether the database answers a trivial query."""
|
||||
try:
|
||||
session.execute(select(literal(1)))
|
||||
except SQLAlchemyError as error:
|
||||
logger.warning("ebook_ready_database_unavailable error=%s", error)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def check_bm25_status(config: EbookSearchConfig) -> str:
|
||||
"""Return the persisted BM25 index status without loading it into memory."""
|
||||
index_path = bm25_index_path(config)
|
||||
manifest = read_bm25_manifest(index_path)
|
||||
if manifest is None or not bm25_index_exists(index_path, manifest):
|
||||
return "missing"
|
||||
if manifest.chunk_count == 0:
|
||||
return "empty"
|
||||
return "ok"
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Page routes for the EPUB search web UI."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Request
|
||||
from fastapi.responses import HTMLResponse
|
||||
from sqlalchemy import select
|
||||
|
||||
from python.ebook_search.api.dependencies import (
|
||||
AppConfig, # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
)
|
||||
from python.ebook_search.api.web import templates
|
||||
from python.fastapi_tools import DbSession # noqa: TC001 FastAPI resolves this annotated dependency at runtime
|
||||
from python.orm.richie import EbookSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
@router.get("/", response_class=HTMLResponse)
|
||||
def index(request: Request, config: AppConfig) -> HTMLResponse:
|
||||
"""Render the search page."""
|
||||
return templates.TemplateResponse(request, "search.html", {"config": config})
|
||||
|
||||
|
||||
@router.get("/books", response_class=HTMLResponse)
|
||||
def books(request: Request, session: DbSession) -> HTMLResponse:
|
||||
"""Render the indexed books page."""
|
||||
sources = list(session.scalars(select(EbookSource).order_by(EbookSource.title)).all())
|
||||
logger.info("ebook_books_page_loaded count=%s", len(sources))
|
||||
return templates.TemplateResponse(request, "books.html", {"sources": sources})
|
||||
|
||||
|
||||
@router.get("/books/{source_id}", response_class=HTMLResponse)
|
||||
def book_detail(source_id: int, request: Request, session: DbSession) -> HTMLResponse:
|
||||
"""Render details for one indexed book."""
|
||||
source = session.get(EbookSource, source_id)
|
||||
if source is not None:
|
||||
chapter_count = len(source.chapters)
|
||||
chunk_count = len(source.chunks)
|
||||
else:
|
||||
chapter_count = 0
|
||||
chunk_count = 0
|
||||
logger.info(
|
||||
"ebook_book_detail_loaded source_id=%s found=%s chapters=%s chunks=%s",
|
||||
source_id,
|
||||
source is not None,
|
||||
chapter_count,
|
||||
chunk_count,
|
||||
)
|
||||
return templates.TemplateResponse(
|
||||
request,
|
||||
"book_detail.html",
|
||||
{"chapter_count": chapter_count, "chunk_count": chunk_count, "source": source},
|
||||
)
|
||||
@@ -0,0 +1,116 @@
|
||||
"""Search routes for the EPUB search web UI."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import replace
|
||||
from time import perf_counter
|
||||
from typing import TYPE_CHECKING, Annotated
|
||||
|
||||
from fastapi import APIRouter, Form, Request
|
||||
from fastapi.responses import HTMLResponse
|
||||
|
||||
from python.ebook_search.answer import answer_query
|
||||
from python.ebook_search.api.dependencies import ( # noqa: TC001 FastAPI resolves these annotated dependencies at runtime
|
||||
AppConfig,
|
||||
AppEngine,
|
||||
)
|
||||
from python.ebook_search.api.web import templates
|
||||
from python.ebook_search.guardrails import (
|
||||
CitationReport,
|
||||
is_confident,
|
||||
retrieval_confidence,
|
||||
validate_citations,
|
||||
)
|
||||
from python.ebook_search.search import SearchResponse, search_ebooks
|
||||
from python.ebook_search.timing import runtime_step_from_start
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
def build_answer(
|
||||
query: str,
|
||||
response: SearchResponse,
|
||||
config: EbookSearchConfig,
|
||||
) -> tuple[str, bool, CitationReport | None]:
|
||||
"""Generate the answer for a search, returning ``(answer, low_confidence, citation_report)``."""
|
||||
if not config.answer_enabled:
|
||||
logger.info("ebook_answer_skipped_disabled")
|
||||
return "Answer generation is disabled. Source chunks are shown below.", False, None
|
||||
|
||||
if not is_confident(response.results, config):
|
||||
logger.info(
|
||||
"ebook_answer_low_confidence confidence=%.4f threshold=%.4f",
|
||||
retrieval_confidence(response.results),
|
||||
config.min_retrieval_confidence,
|
||||
)
|
||||
answer = (
|
||||
"Retrieval confidence is low for this query, so answer generation was skipped. "
|
||||
"Source chunks are shown below."
|
||||
)
|
||||
return answer, True, None
|
||||
|
||||
try:
|
||||
answer = answer_query(query, response.results, config)
|
||||
except RuntimeError as error:
|
||||
logger.warning("ebook_answer_request_failed_falling_back error=%s", error)
|
||||
return "Answer generation failed. Source chunks are still shown below.", False, None
|
||||
|
||||
citation_report = None
|
||||
if config.validate_citations_enabled and response.results:
|
||||
citation_report = validate_citations(answer, len(response.results))
|
||||
if citation_report.invalid or not citation_report.grounded:
|
||||
logger.warning(
|
||||
"ebook_answer_citation_issue invalid=%s grounded=%s",
|
||||
citation_report.invalid,
|
||||
citation_report.grounded,
|
||||
)
|
||||
return answer, False, citation_report
|
||||
|
||||
|
||||
@router.post("/search", response_class=HTMLResponse)
|
||||
def search(
|
||||
request: Request,
|
||||
config: AppConfig,
|
||||
engine: AppEngine,
|
||||
query: Annotated[str, Form()],
|
||||
rerank: Annotated[str | None, Form()] = None,
|
||||
) -> HTMLResponse:
|
||||
"""Run a search and render HTMX results."""
|
||||
try:
|
||||
response = search_ebooks(engine, query, config, rerank=rerank == "true")
|
||||
except Exception as error:
|
||||
logger.exception("ebook_search_request_failed")
|
||||
return templates.TemplateResponse(request, "partials/error.html", {"message": str(error)}, status_code=500)
|
||||
|
||||
answer_start = perf_counter()
|
||||
answer, low_confidence, citation_report = build_answer(query, response, config)
|
||||
answer_step_name = "Answer generation" if config.answer_enabled else "Answer skipped"
|
||||
response = replace(
|
||||
response,
|
||||
timings=(*response.timings, runtime_step_from_start(answer_step_name, answer_start)),
|
||||
)
|
||||
|
||||
for step in response.timings:
|
||||
logger.info("ebook_search_timing step=%r runtime_ms=%.1f", step.name, step.duration_ms)
|
||||
logger.info(
|
||||
"ebook_search_request_complete results=%s rank_label=%s runtime_ms=%.1f",
|
||||
len(response.results),
|
||||
response.rank_label,
|
||||
response.total_runtime_ms,
|
||||
)
|
||||
return templates.TemplateResponse(
|
||||
request,
|
||||
"partials/results.html",
|
||||
{
|
||||
"answer": answer,
|
||||
"response": response,
|
||||
"low_confidence": low_confidence,
|
||||
"citation_report": citation_report,
|
||||
},
|
||||
)
|
||||
@@ -0,0 +1,414 @@
|
||||
:root {
|
||||
--bg: #f4f5f7;
|
||||
--surface: #ffffff;
|
||||
--border: #e3e5ea;
|
||||
--text: #1c1f24;
|
||||
--muted: #6b7280;
|
||||
--accent: #4f46e5;
|
||||
--accent-soft: #eef0fe;
|
||||
--danger: #b42318;
|
||||
--warn-bg: #fff8eb;
|
||||
--warn-border: #e0a92e;
|
||||
--warn-text: #7a5008;
|
||||
--radius: 12px;
|
||||
--shadow: 0 1px 2px rgba(16, 24, 40, 0.04), 0 1px 3px rgba(16, 24, 40, 0.08);
|
||||
}
|
||||
|
||||
html.theme-dark {
|
||||
--bg: #0f1117;
|
||||
--surface: #1a1d25;
|
||||
--border: #2b303b;
|
||||
--text: #e6e8ec;
|
||||
--muted: #9aa1ad;
|
||||
--accent: #818cf8;
|
||||
--accent-soft: #262b45;
|
||||
--danger: #f97066;
|
||||
--warn-bg: #2a2410;
|
||||
--warn-border: #b9881f;
|
||||
--warn-text: #e8c97a;
|
||||
--shadow: 0 1px 2px rgba(0, 0, 0, 0.3), 0 1px 3px rgba(0, 0, 0, 0.4);
|
||||
color-scheme: dark;
|
||||
}
|
||||
|
||||
* {
|
||||
box-sizing: border-box;
|
||||
}
|
||||
|
||||
body {
|
||||
margin: 0;
|
||||
background: var(--bg);
|
||||
color: var(--text);
|
||||
font-family: system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif;
|
||||
line-height: 1.55;
|
||||
}
|
||||
|
||||
main {
|
||||
max-width: 820px;
|
||||
margin: 0 auto;
|
||||
padding: 32px 20px 64px;
|
||||
}
|
||||
|
||||
/* Header / nav */
|
||||
.site-header {
|
||||
background: var(--surface);
|
||||
border-bottom: 1px solid var(--border);
|
||||
position: sticky;
|
||||
top: 0;
|
||||
z-index: 10;
|
||||
}
|
||||
|
||||
.site-nav {
|
||||
max-width: 820px;
|
||||
margin: 0 auto;
|
||||
padding: 12px 20px;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 20px;
|
||||
}
|
||||
|
||||
.brand {
|
||||
font-weight: 700;
|
||||
font-size: 1.05rem;
|
||||
color: var(--text);
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
.nav-links {
|
||||
display: flex;
|
||||
gap: 6px;
|
||||
margin-right: auto;
|
||||
}
|
||||
|
||||
.nav-links a {
|
||||
padding: 6px 12px;
|
||||
border-radius: 8px;
|
||||
color: var(--muted);
|
||||
text-decoration: none;
|
||||
font-size: 0.94rem;
|
||||
transition: background 0.15s, color 0.15s;
|
||||
}
|
||||
|
||||
.nav-links a:hover {
|
||||
background: var(--accent-soft);
|
||||
color: var(--accent);
|
||||
}
|
||||
|
||||
.dev-toggle {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 0.85rem;
|
||||
color: var(--muted);
|
||||
cursor: pointer;
|
||||
user-select: none;
|
||||
}
|
||||
|
||||
.theme-toggle {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
width: 34px;
|
||||
height: 34px;
|
||||
padding: 0;
|
||||
font-size: 1rem;
|
||||
line-height: 1;
|
||||
color: var(--text);
|
||||
background: var(--bg);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.theme-toggle:hover {
|
||||
border-color: var(--accent);
|
||||
filter: none;
|
||||
}
|
||||
|
||||
h1 {
|
||||
font-size: 1.6rem;
|
||||
margin: 0 0 20px;
|
||||
}
|
||||
|
||||
h2 {
|
||||
font-size: 1.15rem;
|
||||
margin: 0 0 8px;
|
||||
}
|
||||
|
||||
/* Cards */
|
||||
.card {
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius);
|
||||
box-shadow: var(--shadow);
|
||||
padding: 20px;
|
||||
}
|
||||
|
||||
/* Search form */
|
||||
form {
|
||||
margin: 0;
|
||||
}
|
||||
|
||||
label {
|
||||
font-weight: 600;
|
||||
font-size: 0.92rem;
|
||||
}
|
||||
|
||||
textarea {
|
||||
display: block;
|
||||
width: 100%;
|
||||
margin: 8px 0 16px;
|
||||
padding: 12px 14px;
|
||||
font: inherit;
|
||||
color: var(--text);
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 10px;
|
||||
resize: vertical;
|
||||
transition: border-color 0.15s, box-shadow 0.15s;
|
||||
}
|
||||
|
||||
textarea:focus {
|
||||
outline: none;
|
||||
border-color: var(--accent);
|
||||
box-shadow: 0 0 0 3px var(--accent-soft);
|
||||
}
|
||||
|
||||
.form-row {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
gap: 12px;
|
||||
flex-wrap: wrap;
|
||||
}
|
||||
|
||||
button {
|
||||
padding: 10px 20px;
|
||||
font: inherit;
|
||||
font-weight: 600;
|
||||
color: #fff;
|
||||
background: var(--accent);
|
||||
border: none;
|
||||
border-radius: 10px;
|
||||
cursor: pointer;
|
||||
transition: filter 0.15s;
|
||||
}
|
||||
|
||||
button:hover {
|
||||
filter: brightness(1.08);
|
||||
}
|
||||
|
||||
.check {
|
||||
display: inline-flex;
|
||||
gap: 8px;
|
||||
align-items: center;
|
||||
font-weight: 500;
|
||||
color: var(--muted);
|
||||
}
|
||||
|
||||
.actions {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 12px;
|
||||
margin-bottom: 24px;
|
||||
}
|
||||
|
||||
/* Answer + results */
|
||||
#results {
|
||||
display: block;
|
||||
margin-top: 28px;
|
||||
}
|
||||
|
||||
.rank-label {
|
||||
font-size: 0.82rem;
|
||||
font-weight: 600;
|
||||
text-transform: uppercase;
|
||||
letter-spacing: 0.04em;
|
||||
color: var(--muted);
|
||||
margin-bottom: 16px;
|
||||
}
|
||||
|
||||
.answer {
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius);
|
||||
box-shadow: var(--shadow);
|
||||
padding: 20px;
|
||||
margin-bottom: 24px;
|
||||
}
|
||||
|
||||
.answer p:last-child {
|
||||
margin-bottom: 0;
|
||||
}
|
||||
|
||||
.results {
|
||||
list-style: none;
|
||||
padding: 0;
|
||||
margin: 0;
|
||||
display: grid;
|
||||
gap: 16px;
|
||||
}
|
||||
|
||||
.results > li {
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius);
|
||||
box-shadow: var(--shadow);
|
||||
padding: 18px 20px;
|
||||
}
|
||||
|
||||
.results h2 {
|
||||
font-size: 1.05rem;
|
||||
}
|
||||
|
||||
.results h2 a {
|
||||
color: var(--text);
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
.results h2 a:hover {
|
||||
color: var(--accent);
|
||||
}
|
||||
|
||||
.meta {
|
||||
color: var(--muted);
|
||||
font-size: 0.88rem;
|
||||
margin: 0 0 10px;
|
||||
}
|
||||
|
||||
.scores {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px;
|
||||
margin: 14px 0 0;
|
||||
}
|
||||
|
||||
.scores div {
|
||||
display: inline-flex;
|
||||
gap: 6px;
|
||||
align-items: baseline;
|
||||
padding: 3px 10px;
|
||||
background: var(--bg);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 999px;
|
||||
font-size: 0.78rem;
|
||||
}
|
||||
|
||||
.scores dt {
|
||||
font-weight: 600;
|
||||
color: var(--muted);
|
||||
}
|
||||
|
||||
.scores dd {
|
||||
margin: 0;
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
/* Runtime — developer diagnostics, hidden unless dev mode is on */
|
||||
.runtime {
|
||||
display: none;
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius);
|
||||
box-shadow: var(--shadow);
|
||||
padding: 18px 20px;
|
||||
margin-bottom: 24px;
|
||||
}
|
||||
|
||||
html.dev .runtime {
|
||||
display: block;
|
||||
}
|
||||
|
||||
.timing-chart {
|
||||
display: grid;
|
||||
gap: 8px;
|
||||
padding: 0;
|
||||
margin: 12px 0 0;
|
||||
list-style: none;
|
||||
}
|
||||
|
||||
.timing-chart li {
|
||||
display: grid;
|
||||
grid-template-columns: minmax(150px, 1fr) minmax(160px, 2fr) auto auto;
|
||||
gap: 10px;
|
||||
align-items: center;
|
||||
font-size: 0.85rem;
|
||||
}
|
||||
|
||||
.timing-bar {
|
||||
height: 8px;
|
||||
overflow: hidden;
|
||||
background: var(--bg);
|
||||
border-radius: 999px;
|
||||
}
|
||||
|
||||
.timing-bar span {
|
||||
display: block;
|
||||
height: 100%;
|
||||
background: var(--accent);
|
||||
border-radius: 999px;
|
||||
}
|
||||
|
||||
.timing-value,
|
||||
.timing-remaining {
|
||||
color: var(--muted);
|
||||
font-variant-numeric: tabular-nums;
|
||||
text-align: right;
|
||||
}
|
||||
|
||||
/* Tables */
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
background: var(--surface);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius);
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
th,
|
||||
td {
|
||||
padding: 10px 14px;
|
||||
border-bottom: 1px solid var(--border);
|
||||
text-align: left;
|
||||
font-size: 0.9rem;
|
||||
}
|
||||
|
||||
th {
|
||||
font-weight: 600;
|
||||
color: var(--muted);
|
||||
background: var(--bg);
|
||||
}
|
||||
|
||||
tbody tr:last-child td {
|
||||
border-bottom: none;
|
||||
}
|
||||
|
||||
dl dt {
|
||||
font-weight: 600;
|
||||
color: var(--muted);
|
||||
font-size: 0.85rem;
|
||||
}
|
||||
|
||||
dl dd {
|
||||
margin: 0 0 12px;
|
||||
}
|
||||
|
||||
/* States */
|
||||
.error {
|
||||
color: var(--danger);
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.notice {
|
||||
margin: 12px 0;
|
||||
padding: 10px 14px;
|
||||
border-left: 3px solid var(--warn-border);
|
||||
border-radius: 6px;
|
||||
background: var(--warn-bg);
|
||||
color: var(--warn-text);
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.status {
|
||||
color: var(--muted);
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block title %}EPUB Admin{% endblock %}
|
||||
{% block head %}<script src="https://unpkg.com/htmx.org@2.0.4"></script>{% endblock %}
|
||||
|
||||
{% block content %}
|
||||
<h1>Admin</h1>
|
||||
<section id="admin-status"></section>
|
||||
<section class="actions">
|
||||
<form hx-post="/admin/scan" hx-target="#admin-status" hx-swap="innerHTML">
|
||||
<button type="submit">Scan</button>
|
||||
</form>
|
||||
<form hx-post="/admin/embed-missing" hx-target="#admin-status" hx-swap="innerHTML">
|
||||
<button type="submit">Embed</button>
|
||||
</form>
|
||||
<form hx-post="/admin/embed-all" hx-target="#admin-status" hx-swap="innerHTML">
|
||||
<button type="submit">Embed all</button>
|
||||
</form>
|
||||
</section>
|
||||
<section>
|
||||
<h2>Embeddings</h2>
|
||||
<table>
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Model</th>
|
||||
<th>Dimensions</th>
|
||||
<th>Embedded</th>
|
||||
<th>Missing</th>
|
||||
<th>Total chunks</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{% for item in stats %}
|
||||
<tr>
|
||||
<td>{{ item.model_name }}</td>
|
||||
<td>{{ item.dimension }}</td>
|
||||
<td>{{ item.embedded_chunks }}</td>
|
||||
<td>{{ item.missing_chunks }}</td>
|
||||
<td>{{ item.total_chunks }}</td>
|
||||
</tr>
|
||||
{% endfor %}
|
||||
</tbody>
|
||||
</table>
|
||||
</section>
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,71 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>{% block title %}EPUB Search{% endblock %}</title>
|
||||
{% block head %}{% endblock %}
|
||||
<link rel="stylesheet" href="/static/style.css?v={{ static_version('style.css') }}">
|
||||
<script>
|
||||
// Apply theme and dev mode before paint to avoid a flash of unstyled/wrong content.
|
||||
(function () {
|
||||
var stored = localStorage.getItem("ebook-theme");
|
||||
var prefersDark = window.matchMedia("(prefers-color-scheme: dark)").matches;
|
||||
var theme = stored || (prefersDark ? "dark" : "light");
|
||||
document.documentElement.classList.add("theme-" + theme);
|
||||
if (localStorage.getItem("ebook-dev-mode") === "on") {
|
||||
document.documentElement.classList.add("dev");
|
||||
}
|
||||
})();
|
||||
</script>
|
||||
</head>
|
||||
<body>
|
||||
<header class="site-header">
|
||||
<nav class="site-nav">
|
||||
<a class="brand" href="/">EPUB Search</a>
|
||||
<div class="nav-links">
|
||||
<a href="/">Search</a>
|
||||
<a href="/books">Books</a>
|
||||
<a href="/admin">Admin</a>
|
||||
</div>
|
||||
<button type="button" id="theme-toggle" class="theme-toggle" title="Toggle light / dark theme" aria-label="Toggle theme"></button>
|
||||
<label class="dev-toggle" title="Show developer diagnostics">
|
||||
<input type="checkbox" id="dev-mode-toggle">
|
||||
<span>Dev</span>
|
||||
</label>
|
||||
</nav>
|
||||
</header>
|
||||
<main>
|
||||
{% block content %}{% endblock %}
|
||||
</main>
|
||||
<script>
|
||||
(function () {
|
||||
var toggle = document.getElementById("dev-mode-toggle");
|
||||
if (toggle) {
|
||||
toggle.checked = document.documentElement.classList.contains("dev");
|
||||
toggle.addEventListener("change", function () {
|
||||
document.documentElement.classList.toggle("dev", toggle.checked);
|
||||
localStorage.setItem("ebook-dev-mode", toggle.checked ? "on" : "off");
|
||||
});
|
||||
}
|
||||
|
||||
var themeButton = document.getElementById("theme-toggle");
|
||||
if (themeButton) {
|
||||
var root = document.documentElement;
|
||||
var sync = function () {
|
||||
var isDark = root.classList.contains("theme-dark");
|
||||
themeButton.textContent = isDark ? "☀️" : "🌙";
|
||||
};
|
||||
sync();
|
||||
themeButton.addEventListener("click", function () {
|
||||
var next = root.classList.contains("theme-dark") ? "light" : "dark";
|
||||
root.classList.remove("theme-dark", "theme-light");
|
||||
root.classList.add("theme-" + next);
|
||||
localStorage.setItem("ebook-theme", next);
|
||||
sync();
|
||||
});
|
||||
}
|
||||
})();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,20 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block title %}{% if source %}{{ source.title }}{% else %}Book not found{% endif %}{% endblock %}
|
||||
|
||||
{% block content %}
|
||||
{% if source %}
|
||||
<h1>{{ source.title }}</h1>
|
||||
<p class="meta">{{ source.author or "Unknown author" }}</p>
|
||||
<dl class="card">
|
||||
<dt>File</dt>
|
||||
<dd>{{ source.file_path }}</dd>
|
||||
<dt>Chapters</dt>
|
||||
<dd>{{ chapter_count }}</dd>
|
||||
<dt>Chunks</dt>
|
||||
<dd>{{ chunk_count }}</dd>
|
||||
</dl>
|
||||
{% else %}
|
||||
<h1>Book not found</h1>
|
||||
{% endif %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,19 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block title %}EPUB Books{% endblock %}
|
||||
|
||||
{% block content %}
|
||||
<h1>Books</h1>
|
||||
{% if sources %}
|
||||
<ol class="results">
|
||||
{% for source in sources %}
|
||||
<li>
|
||||
<h2><a href="/books/{{ source.id }}">{{ source.title }}</a></h2>
|
||||
<p class="meta">{{ source.author or "Unknown author" }}</p>
|
||||
</li>
|
||||
{% endfor %}
|
||||
</ol>
|
||||
{% else %}
|
||||
<p>No EPUBs indexed.</p>
|
||||
{% endif %}
|
||||
{% endblock %}
|
||||
@@ -0,0 +1 @@
|
||||
<p class="status">{{ message }}</p>
|
||||
@@ -0,0 +1 @@
|
||||
<p class="error">{{ message }}</p>
|
||||
@@ -0,0 +1,90 @@
|
||||
<div class="rank-label">{{ response.rank_label }}</div>
|
||||
{% if response.timings %}
|
||||
<section class="runtime">
|
||||
<h2>Runtime</h2>
|
||||
<p class="meta">Total {{ "%.1f"|format(response.total_runtime_ms) }} ms</p>
|
||||
<ol class="timing-chart">
|
||||
{% set total = response.total_runtime_ms %}
|
||||
{% set ns = namespace(remaining=total) %}
|
||||
{% for step in response.timings %}
|
||||
{% set width = (step.duration_ms / total * 100) if total else 0 %}
|
||||
{% if step.counts_toward_total %}
|
||||
{% set ns.remaining = ns.remaining - step.duration_ms %}
|
||||
{% endif %}
|
||||
<li>
|
||||
<span class="timing-label">{{ step.name }}</span>
|
||||
<span class="timing-bar"><span style="width: {{ "%.2f"|format(width) }}%"></span></span>
|
||||
<span class="timing-value">{{ "%.1f"|format(step.duration_ms) }} ms</span>
|
||||
<span class="timing-remaining">{{ "%.1f"|format([ns.remaining, 0]|max) }} ms left</span>
|
||||
</li>
|
||||
{% endfor %}
|
||||
</ol>
|
||||
</section>
|
||||
{% endif %}
|
||||
<section class="answer">
|
||||
<h2>Answer</h2>
|
||||
{% if low_confidence|default(false) %}
|
||||
<p class="notice">Low retrieval confidence — answer generation was skipped.</p>
|
||||
{% endif %}
|
||||
{% set report = citation_report|default(none) %}
|
||||
{% if report is not none and not report.grounded %}
|
||||
<p class="notice">Unverified — no source citations were found in this answer.</p>
|
||||
{% endif %}
|
||||
{% if report is not none and report.invalid %}
|
||||
<p class="notice">Invalid citations: {{ report.invalid|join(", ") }} (no matching source).</p>
|
||||
{% endif %}
|
||||
<p>{{ answer }}</p>
|
||||
</section>
|
||||
{% if response.results %}
|
||||
<ol class="results">
|
||||
{% for result in response.results %}
|
||||
<li>
|
||||
<h2>
|
||||
{% if result.source_id %}
|
||||
<a href="/books/{{ result.source_id }}">{{ result.source_title }}</a>
|
||||
{% else %}
|
||||
{{ result.source_title }}
|
||||
{% endif %}
|
||||
</h2>
|
||||
<p class="meta">
|
||||
{% if result.source_author %}{{ result.source_author }}{% endif %}
|
||||
{% if result.chapter_title %} · {{ result.chapter_title }}{% endif %}
|
||||
{% if result.page_label %} · page {{ result.page_label }}{% endif %}
|
||||
</p>
|
||||
<p>{{ result.text }}</p>
|
||||
<dl class="scores">
|
||||
<div>
|
||||
<dt>final</dt>
|
||||
<dd>{{ "%.3f"|format(result.score) }}</dd>
|
||||
</div>
|
||||
{% if result.rerank_score is not none %}
|
||||
<div>
|
||||
<dt>rerank</dt>
|
||||
<dd>{{ "%.3f"|format(result.rerank_score) }}</dd>
|
||||
</div>
|
||||
{% endif %}
|
||||
{% if result.vector_score is not none %}
|
||||
<div>
|
||||
<dt>vector cosine</dt>
|
||||
<dd>{{ "%.3f"|format(result.vector_score) }}</dd>
|
||||
</div>
|
||||
{% endif %}
|
||||
{% if result.bm25_score is not none %}
|
||||
<div>
|
||||
<dt>BM25</dt>
|
||||
<dd>{{ "%.6f"|format(result.bm25_score) }}</dd>
|
||||
</div>
|
||||
{% endif %}
|
||||
{% if result.fused_score is not none %}
|
||||
<div>
|
||||
<dt>RRF</dt>
|
||||
<dd>{{ "%.3f"|format(result.fused_score) }}</dd>
|
||||
</div>
|
||||
{% endif %}
|
||||
</dl>
|
||||
</li>
|
||||
{% endfor %}
|
||||
</ol>
|
||||
{% else %}
|
||||
<p>No results.</p>
|
||||
{% endif %}
|
||||
@@ -0,0 +1,20 @@
|
||||
{% extends "base.html" %}
|
||||
|
||||
{% block title %}EPUB Search{% endblock %}
|
||||
{% block head %}<script src="https://unpkg.com/htmx.org@2.0.4"></script>{% endblock %}
|
||||
|
||||
{% block content %}
|
||||
<h1>Search</h1>
|
||||
<form class="card" hx-post="/search" hx-target="#results" hx-swap="innerHTML">
|
||||
<label for="query">What are you looking for?</label>
|
||||
<textarea id="query" name="query" rows="4" placeholder="Ask a question or paste a passage…" required></textarea>
|
||||
<div class="form-row">
|
||||
<label class="check">
|
||||
<input type="checkbox" name="rerank" value="true" {% if config.rerank.enabled %}checked{% endif %}>
|
||||
Rerank
|
||||
</label>
|
||||
<button type="submit">Search</button>
|
||||
</div>
|
||||
</form>
|
||||
<section id="results"></section>
|
||||
{% endblock %}
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Shared web UI resources for EPUB search."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi.templating import Jinja2Templates
|
||||
|
||||
PACKAGE_DIR = Path(__file__).resolve().parent
|
||||
TEMPLATE_DIR = PACKAGE_DIR / "templates"
|
||||
STATIC_DIR = PACKAGE_DIR / "static"
|
||||
|
||||
|
||||
def static_version(filename: str) -> int:
|
||||
"""Return a cache-busting token for a static file based on its modification time."""
|
||||
try:
|
||||
return int((STATIC_DIR / filename).stat().st_mtime)
|
||||
except OSError:
|
||||
return 0
|
||||
|
||||
|
||||
templates = Jinja2Templates(directory=TEMPLATE_DIR)
|
||||
templates.env.globals["static_version"] = static_version
|
||||
@@ -0,0 +1,281 @@
|
||||
"""Persisted BM25 corpus management."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import shutil
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from functools import cache
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import bm25s
|
||||
from sqlalchemy import func, select, union_all
|
||||
|
||||
from python.orm.richie import EbookChapter, EbookChunk, EbookSource
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
MANIFEST_NAME = "manifest.json"
|
||||
REQUIRED_INDEX_FILES = frozenset(
|
||||
{
|
||||
"data.csc.index.npy",
|
||||
"indices.csc.index.npy",
|
||||
"indptr.csc.index.npy",
|
||||
"params.index.json",
|
||||
"vocab.index.json",
|
||||
"corpus.jsonl",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BM25Manifest:
|
||||
"""Metadata describing a persisted BM25 corpus."""
|
||||
|
||||
created_at: datetime
|
||||
db_updated_at: datetime | None
|
||||
chunk_count: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BM25Corpus:
|
||||
"""Loaded persisted BM25 corpus and retriever."""
|
||||
|
||||
retriever: object | None
|
||||
records: tuple[dict[str, object], ...]
|
||||
manifest: BM25Manifest
|
||||
|
||||
|
||||
class BM25CorpusUnavailableError(RuntimeError):
|
||||
"""Raised when the persisted BM25 corpus cannot be loaded."""
|
||||
|
||||
|
||||
def bm25_index_path(config: EbookSearchConfig) -> Path:
|
||||
"""Return the configured BM25 index root path relative to the current working directory."""
|
||||
path = Path(config.bm25_index_dir).expanduser()
|
||||
if path.is_absolute():
|
||||
return path
|
||||
return Path.cwd() / path
|
||||
|
||||
|
||||
def get_current_bm25_index(index_path: Path) -> Path:
|
||||
"""Return the live BM25 index directory."""
|
||||
current_path = index_path / "current"
|
||||
if current_path.exists() or current_path.is_symlink():
|
||||
return current_path
|
||||
return index_path
|
||||
|
||||
|
||||
def ensure_bm25_corpus(session: Session, config: EbookSearchConfig) -> None:
|
||||
"""Create or refresh the persisted BM25 corpus when it is missing or stale."""
|
||||
index_path = bm25_index_path(config)
|
||||
manifest = read_bm25_manifest(index_path)
|
||||
db_updated_at = corpus_last_updated_at(session)
|
||||
if not bm25_index_exists(index_path, manifest):
|
||||
logger.info("ebook_bm25_index_missing path=%s", index_path)
|
||||
refresh_bm25_corpus(session, config, db_updated_at=db_updated_at)
|
||||
return
|
||||
if db_updated_at is not None and manifest is not None and manifest.created_at < db_updated_at:
|
||||
logger.info(
|
||||
"ebook_bm25_index_stale path=%s created_at=%s db_updated_at=%s",
|
||||
index_path,
|
||||
manifest.created_at.isoformat(),
|
||||
db_updated_at.isoformat(),
|
||||
)
|
||||
refresh_bm25_corpus(session, config, db_updated_at=db_updated_at)
|
||||
return
|
||||
logger.info(
|
||||
"ebook_bm25_index_current path=%s chunks=%s created_at=%s",
|
||||
index_path,
|
||||
manifest.chunk_count if manifest else 0,
|
||||
manifest.created_at.isoformat() if manifest else None,
|
||||
)
|
||||
|
||||
|
||||
def refresh_bm25_corpus(
|
||||
session: Session,
|
||||
config: EbookSearchConfig,
|
||||
*,
|
||||
db_updated_at: datetime | None = None,
|
||||
) -> BM25Manifest:
|
||||
"""Rebuild and persist the BM25 corpus from the current database chunks."""
|
||||
index_path = bm25_index_path(config)
|
||||
records, texts = fetch_bm25_corpus_records(session)
|
||||
manifest = BM25Manifest(
|
||||
created_at=datetime.now(tz=UTC),
|
||||
db_updated_at=db_updated_at if db_updated_at is not None else corpus_last_updated_at(session),
|
||||
chunk_count=len(records),
|
||||
)
|
||||
write_bm25_corpus(index_path, records, texts, manifest)
|
||||
logger.info(
|
||||
"ebook_bm25_index_refreshed path=%s chunks=%s created_at=%s",
|
||||
index_path,
|
||||
manifest.chunk_count,
|
||||
manifest.created_at.isoformat(),
|
||||
)
|
||||
return manifest
|
||||
|
||||
|
||||
@cache
|
||||
def load_bm25_corpus(config: EbookSearchConfig) -> BM25Corpus:
|
||||
"""Load the BM25 corpus into memory once per process.
|
||||
|
||||
Background refresh tasks clear this cache after rebuilding the on-disk corpus.
|
||||
"""
|
||||
index_path = bm25_index_path(config)
|
||||
active_index_path = get_current_bm25_index(index_path)
|
||||
logger.info("ebook_bm25_corpus_cache_load path=%s active_path=%s", index_path, active_index_path)
|
||||
manifest = read_bm25_manifest(index_path)
|
||||
if manifest is None or not bm25_index_exists(index_path, manifest):
|
||||
msg = f"BM25 corpus is not available: {index_path}"
|
||||
raise BM25CorpusUnavailableError(msg)
|
||||
if manifest.chunk_count == 0:
|
||||
return BM25Corpus(retriever=None, records=(), manifest=manifest)
|
||||
|
||||
retriever = bm25s.BM25.load(active_index_path, load_corpus=True, mmap=True)
|
||||
records = tuple(dict(record) for record in retriever.corpus)
|
||||
return BM25Corpus(retriever=retriever, records=records, manifest=manifest)
|
||||
|
||||
|
||||
def score_bm25_corpus(query: str, corpus: BM25Corpus, *, limit: int) -> list[tuple[dict[str, object], float]]:
|
||||
"""Score a query against a loaded BM25 corpus."""
|
||||
if corpus.retriever is None or not corpus.records:
|
||||
return []
|
||||
k = min(limit, len(corpus.records))
|
||||
documents, scores = corpus.retriever.retrieve(
|
||||
bm25s.tokenize(query, show_progress=False),
|
||||
corpus=list(corpus.records),
|
||||
k=k,
|
||||
show_progress=False,
|
||||
)
|
||||
results: list[tuple[dict[str, object], float]] = []
|
||||
for document, score in zip(documents[0], scores[0], strict=True):
|
||||
score_value = float(score)
|
||||
if score_value <= 0:
|
||||
continue
|
||||
results.append((dict(document), score_value))
|
||||
return results
|
||||
|
||||
|
||||
def fetch_bm25_corpus_records(session: Session) -> tuple[list[dict[str, object]], list[str]]:
|
||||
"""Fetch persistable BM25 corpus records and their matching index texts from the database.
|
||||
|
||||
search_text is only needed to build the index, so it is returned separately instead of
|
||||
being persisted into the corpus records, which would double the corpus size.
|
||||
"""
|
||||
statement = (
|
||||
select(
|
||||
EbookChunk.id.label("chunk_id"),
|
||||
EbookChunk.text.label("text"),
|
||||
EbookSource.title.label("source_title"),
|
||||
EbookSource.author.label("source_author"),
|
||||
EbookChapter.title.label("chapter_title"),
|
||||
EbookChunk.page_label.label("page_label"),
|
||||
EbookChunk.search_text.label("bm25_text"),
|
||||
)
|
||||
.select_from(EbookChunk)
|
||||
.join(EbookSource, EbookSource.id == EbookChunk.source_id)
|
||||
.outerjoin(EbookChapter, EbookChapter.id == EbookChunk.chapter_id)
|
||||
.order_by(EbookChunk.id)
|
||||
)
|
||||
records: list[dict[str, object]] = []
|
||||
texts: list[str] = []
|
||||
for row in session.execute(statement).mappings():
|
||||
record = dict(row)
|
||||
texts.append(str(record.pop("bm25_text")))
|
||||
records.append(record)
|
||||
return records, texts
|
||||
|
||||
|
||||
def corpus_last_updated_at(session: Session) -> datetime | None:
|
||||
"""Return the latest source/chapter/chunk update timestamp relevant to BM25 text."""
|
||||
update_times = union_all(
|
||||
select(func.max(EbookSource.updated).label("updated")),
|
||||
select(func.max(EbookChapter.updated).label("updated")),
|
||||
select(func.max(EbookChunk.updated).label("updated")),
|
||||
).subquery()
|
||||
return session.scalar(select(func.max(update_times.c.updated)))
|
||||
|
||||
|
||||
def write_bm25_corpus(
|
||||
index_path: Path,
|
||||
records: list[dict[str, object]],
|
||||
texts: list[str],
|
||||
manifest: BM25Manifest,
|
||||
) -> None:
|
||||
"""Write a BM25 corpus generation and publish it through the current symlink."""
|
||||
index_path.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
generations_path = index_path / "generations"
|
||||
generations_path.mkdir(exist_ok=True)
|
||||
|
||||
generation_path = next_bm25_generation_path(generations_path, manifest.created_at)
|
||||
current_path = index_path / "current"
|
||||
next_current_path = index_path / f".current.{generation_path.name}.tmp"
|
||||
try:
|
||||
generation_path.mkdir()
|
||||
|
||||
# Empty corpora publish a manifest-only generation so startup succeeds before any chunks exist.
|
||||
if records:
|
||||
retriever = bm25s.BM25()
|
||||
retriever.index(bm25s.tokenize(texts, show_progress=False), show_progress=False)
|
||||
retriever.save(generation_path, corpus=records, show_progress=False)
|
||||
write_bm25_manifest(generation_path, manifest)
|
||||
next_current_path.unlink(missing_ok=True)
|
||||
next_current_path.symlink_to(generation_path, target_is_directory=True)
|
||||
next_current_path.replace(current_path)
|
||||
except Exception:
|
||||
next_current_path.unlink(missing_ok=True)
|
||||
shutil.rmtree(generation_path, ignore_errors=True)
|
||||
raise
|
||||
|
||||
|
||||
def read_bm25_manifest(index_path: Path) -> BM25Manifest | None:
|
||||
"""Read the BM25 manifest if it exists and is valid."""
|
||||
manifest_path = get_current_bm25_index(index_path) / MANIFEST_NAME
|
||||
if not manifest_path.exists():
|
||||
return None
|
||||
body = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
return BM25Manifest(
|
||||
created_at=datetime.fromisoformat(str(body["created_at"])),
|
||||
db_updated_at=datetime.fromisoformat(str(body["db_updated_at"])) if body.get("db_updated_at") else None,
|
||||
chunk_count=int(body["chunk_count"]),
|
||||
)
|
||||
|
||||
|
||||
def write_bm25_manifest(index_path: Path, manifest: BM25Manifest) -> None:
|
||||
"""Write the BM25 manifest to an index directory."""
|
||||
body = {
|
||||
"created_at": manifest.created_at.isoformat(),
|
||||
"db_updated_at": manifest.db_updated_at.isoformat() if manifest.db_updated_at else None,
|
||||
"chunk_count": manifest.chunk_count,
|
||||
}
|
||||
(index_path / MANIFEST_NAME).write_text(json.dumps(body, indent=2, sort_keys=True), encoding="utf-8")
|
||||
|
||||
|
||||
def bm25_index_exists(index_path: Path, manifest: BM25Manifest | None) -> bool:
|
||||
"""Return whether a usable persisted BM25 index exists."""
|
||||
active_index_path = get_current_bm25_index(index_path)
|
||||
if manifest is None or not active_index_path.is_dir():
|
||||
return False
|
||||
if manifest.chunk_count == 0:
|
||||
return True
|
||||
return all((active_index_path / file_name).exists() for file_name in REQUIRED_INDEX_FILES)
|
||||
|
||||
|
||||
def next_bm25_generation_path(generations_path: Path, created_at: datetime) -> Path:
|
||||
"""Return an unused dated BM25 generation path."""
|
||||
base_name = created_at.astimezone(UTC).strftime("%Y%m%dT%H%M%S.%fZ")
|
||||
generation_path = generations_path / base_name
|
||||
suffix = 1
|
||||
while generation_path.exists():
|
||||
generation_path = generations_path / f"{base_name}.{suffix}"
|
||||
suffix += 1
|
||||
return generation_path
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Configuration for the EPUB search app."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from os import getenv
|
||||
from typing import Annotated, Self
|
||||
|
||||
from pydantic import AliasChoices, Field, field_validator, model_validator
|
||||
from pydantic_settings import BaseSettings, NoDecode, SettingsConfigDict
|
||||
|
||||
|
||||
def normalize_embedding_alias(model: str) -> str:
|
||||
"""Normalize a supported embedding alias to its provider model name."""
|
||||
aliases = {
|
||||
"Qwen3-Embedding-0.6B": "qwen3-embedding-0.6b",
|
||||
"Qwen3-Embedding-4B": "qwen3-embedding-4b",
|
||||
"Qwen3-Embedding-8B": "qwen3-embedding-8b",
|
||||
"Qwen/Qwen3-Embedding-0.6B": "qwen3-embedding-0.6b",
|
||||
"Qwen/Qwen3-Embedding-4B": "qwen3-embedding-4b",
|
||||
"Qwen/Qwen3-Embedding-8B": "qwen3-embedding-8b",
|
||||
"qwen3-embedding:0.6b": "qwen3-embedding-0.6b",
|
||||
"qwen3-embedding:4b": "qwen3-embedding-4b",
|
||||
"qwen3-embedding:8b": "qwen3-embedding-8b",
|
||||
"qwen3-embedding-0.6b": "qwen3-embedding-0.6b",
|
||||
"qwen3-embedding-4b": "qwen3-embedding-4b",
|
||||
"qwen3-embedding-8b": "qwen3-embedding-8b",
|
||||
}
|
||||
standard_model = aliases.get(model)
|
||||
if standard_model is None:
|
||||
error = f"Embedding model {model} is not supported. Supported models are {aliases.keys()}"
|
||||
raise ValueError(error)
|
||||
return standard_model
|
||||
|
||||
|
||||
def normalize_embedding_model(default: str = "qwen3-embedding-0.6b") -> str:
|
||||
"""Normalize the configured embedding alias to its provider model name."""
|
||||
return normalize_embedding_alias(getenv("EBOOK_SEARCH_EMBEDDING_MODEL", default))
|
||||
|
||||
|
||||
class RerankConfig(BaseSettings):
|
||||
"""vLLM reranker settings."""
|
||||
|
||||
model_config = SettingsConfigDict(env_prefix="EBOOK_SEARCH_RERANK_", frozen=True, protected_namespaces=())
|
||||
|
||||
enabled: bool = True
|
||||
base_url: str = "http://192.168.90.25:8001"
|
||||
model: str = "qwen3-reranker-06b"
|
||||
candidates: int = 24
|
||||
timeout_seconds: float = 30.0
|
||||
score_weight: float = 0.7
|
||||
hybrid_weight: float = 0.3
|
||||
|
||||
|
||||
class EbookSearchConfig(BaseSettings):
|
||||
"""Runtime settings for EPUB search."""
|
||||
|
||||
model_config = SettingsConfigDict(
|
||||
env_prefix="EBOOK_SEARCH_",
|
||||
frozen=True,
|
||||
populate_by_name=True,
|
||||
protected_namespaces=(),
|
||||
)
|
||||
|
||||
rerank: RerankConfig = Field(default_factory=RerankConfig)
|
||||
top_k: int = 12
|
||||
library_paths: Annotated[tuple[str, ...], NoDecode] = ()
|
||||
chunk_tokens: int = 700
|
||||
chunk_overlap: int = 100
|
||||
vllm_base_url: str = "https://ollama.com/v1"
|
||||
vllm_api_key: str = Field(
|
||||
default="not-needed",
|
||||
validation_alias=AliasChoices("EBOOK_SEARCH_VLLM_API_KEY", "OLLAMA_API_KEY"),
|
||||
)
|
||||
chat_model: str = "deepseek-v4-flash"
|
||||
answer_enabled: bool = True
|
||||
embedding_base_url: str = "http://192.168.90.25:8000/v1"
|
||||
embedding_api_key: str = "not-needed"
|
||||
embedding_model: str = "qwen3-embedding-0.6b"
|
||||
embedding_batch_size: int = 32
|
||||
embedding_timeout_seconds: float = 60.0
|
||||
chat_timeout_seconds: float = 60.0
|
||||
vector_candidate_multiplier: int = 4
|
||||
bm25_candidate_limit: int = 120
|
||||
rrf_rank_constant: int = 60
|
||||
min_retrieval_confidence: float = 0.0
|
||||
validate_citations_enabled: bool = True
|
||||
bm25_index_dir: str = ".ebook_search_bm25"
|
||||
bm25_refresh_delay_seconds: int = 60
|
||||
|
||||
@field_validator("library_paths", mode="before")
|
||||
@classmethod
|
||||
def split_library_paths(cls, value: object) -> object:
|
||||
"""Split a colon-separated library path string into a tuple of paths."""
|
||||
if isinstance(value, str):
|
||||
return tuple(path for path in value.split(":") if path)
|
||||
return value
|
||||
|
||||
@field_validator("embedding_model")
|
||||
@classmethod
|
||||
def normalize_embedding(cls, value: str) -> str:
|
||||
"""Normalize the configured embedding alias to its provider model name."""
|
||||
return normalize_embedding_alias(value)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_runtime_consistency(self) -> Self:
|
||||
"""Reject configurations that cannot serve the features they enable."""
|
||||
if not self.embedding_base_url.strip():
|
||||
msg = "embedding_base_url must be set"
|
||||
raise ValueError(msg)
|
||||
if self.answer_enabled and (not self.vllm_base_url.strip() or not self.chat_model.strip()):
|
||||
msg = "answer_enabled requires vllm_base_url and chat_model to be set"
|
||||
raise ValueError(msg)
|
||||
if self.rerank.enabled and not self.rerank.base_url.strip():
|
||||
msg = "rerank.enabled requires rerank.base_url to be set"
|
||||
raise ValueError(msg)
|
||||
return self
|
||||
|
||||
|
||||
def load_rerank_config() -> RerankConfig:
|
||||
"""Load reranker config from environment variables."""
|
||||
return RerankConfig()
|
||||
|
||||
|
||||
def load_config() -> EbookSearchConfig:
|
||||
"""Load EPUB search config from environment variables."""
|
||||
return EbookSearchConfig()
|
||||
@@ -0,0 +1,54 @@
|
||||
FROM python:3.14-slim AS base
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.26 /uv /uvx /bin/
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
APP_DIR=/home/richie/dotfiles \
|
||||
UV_PROJECT_ENVIRONMENT=/opt/venv \
|
||||
UV_PYTHON_DOWNLOADS=never \
|
||||
UV_NO_CACHE=1
|
||||
|
||||
# Separate ENV instruction so ${APP_DIR} and ${PATH} from above resolve.
|
||||
ENV PYTHONPATH=${APP_DIR} \
|
||||
PATH=/opt/venv/bin:${PATH}
|
||||
|
||||
WORKDIR ${APP_DIR}
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends build-essential curl \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY python/ebook_search/docker/pyproject.toml python/ebook_search/docker/uv.lock ./
|
||||
|
||||
RUN uv sync --locked --no-dev
|
||||
|
||||
|
||||
FROM base AS test
|
||||
|
||||
RUN uv sync --locked
|
||||
|
||||
COPY python ./python
|
||||
COPY tests/__init__.py ./tests/__init__.py
|
||||
COPY tests/ebook_search ./tests/ebook_search
|
||||
|
||||
CMD ["pytest"]
|
||||
|
||||
|
||||
FROM base AS runtime
|
||||
|
||||
ENV EBOOK_SEARCH_HOST=0.0.0.0 \
|
||||
EBOOK_SEARCH_PORT=8070 \
|
||||
EBOOK_SEARCH_BM25_INDEX_DIR=/data/bm25
|
||||
|
||||
COPY python ./python
|
||||
|
||||
RUN useradd --create-home --uid 10001 app \
|
||||
&& mkdir -p /data \
|
||||
&& chown -R app:app /home/richie /data
|
||||
|
||||
USER app
|
||||
|
||||
EXPOSE 8070
|
||||
|
||||
CMD ["sh", "-c", "exec python -m python.ebook_search.api.main --host \"${EBOOK_SEARCH_HOST}\" --port \"${EBOOK_SEARCH_PORT}\" --log-level \"${EBOOK_SEARCH_LOG_LEVEL:-INFO}\""]
|
||||
@@ -0,0 +1,77 @@
|
||||
# Ebook Search Docker
|
||||
|
||||
Run the EPUB search app against the existing Postgres database on `jeeves`:
|
||||
|
||||
```sh
|
||||
python -m python.ebook_search.docker.containers start --library-path /path/to/epubs --build
|
||||
```
|
||||
|
||||
All ebook-search Docker files live in this directory:
|
||||
|
||||
- `Dockerfile` — multi-stage: `test` (runs pytest) and `runtime` (default target, the app image)
|
||||
- `docker-compose.yml`
|
||||
- `containers.py` — Typer lifecycle CLI
|
||||
- `pyproject.toml` / `uv.lock` — the container's uv-locked dependencies
|
||||
|
||||
The app listens on `http://localhost:8070`.
|
||||
|
||||
Useful lifecycle commands:
|
||||
|
||||
```sh
|
||||
python -m python.ebook_search.docker.containers build
|
||||
python -m python.ebook_search.docker.containers start --library-path /path/to/epubs
|
||||
python -m python.ebook_search.docker.containers test
|
||||
python -m python.ebook_search.docker.containers logs
|
||||
python -m python.ebook_search.docker.containers ps
|
||||
python -m python.ebook_search.docker.containers stop
|
||||
```
|
||||
|
||||
Direct compose usage from the repo root:
|
||||
|
||||
```sh
|
||||
docker compose -f python/ebook_search/docker/docker-compose.yml ps
|
||||
```
|
||||
|
||||
## Dependencies
|
||||
|
||||
The image builds its environment with uv from `pyproject.toml` + `uv.lock` in this
|
||||
directory — this is the source of truth for the container's dependencies. To add or
|
||||
update a dependency, edit `pyproject.toml` here and regenerate the lock (uv is
|
||||
available in the `ebook-search` dev shell):
|
||||
|
||||
```sh
|
||||
nix develop .#ebook-search -c uv lock --project python/ebook_search/docker
|
||||
```
|
||||
|
||||
## Tests
|
||||
|
||||
The main pytest suite excludes `tests/ebook_search` (its dependencies are no longer
|
||||
in the nix dev shell). The `test ebook search` CI workflow runs them in a uv env
|
||||
built from the lockfile in this directory — same commands work locally from the
|
||||
repo root (the `--override-ini` drops the main suite's ignore):
|
||||
|
||||
```sh
|
||||
uv sync --locked --project python/ebook_search/docker
|
||||
uv run --project python/ebook_search/docker --no-sync pytest tests/ebook_search --override-ini addopts="-n auto -ra"
|
||||
```
|
||||
|
||||
They can also run inside the Docker `test` image, which validates the image itself:
|
||||
|
||||
```sh
|
||||
python -m python.ebook_search.docker.containers test
|
||||
```
|
||||
|
||||
or the raw docker equivalent:
|
||||
|
||||
```sh
|
||||
docker build --file python/ebook_search/docker/Dockerfile --target test --tag ebook-search:test .
|
||||
docker run --rm ebook-search:test
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
The compose service loads the repo root `.env` into the container via `env_file`.
|
||||
|
||||
Mount your EPUB directory by setting `EBOOK_LIBRARY_HOST_PATH` in an env file or on the command line. The container sees it as `/library`, and `EBOOK_SEARCH_LIBRARY_PATHS` is set to `/library` inside the container.
|
||||
|
||||
Database connection settings are controlled by `RICHIE_DB`, `RICHIE_HOST`, `RICHIE_PORT`, `RICHIE_USER`, and `RICHIE_PASSWORD`. The default host is `jeeves`.
|
||||
@@ -0,0 +1 @@
|
||||
"""Docker packaging and lifecycle tooling for ebook search."""
|
||||
@@ -0,0 +1,259 @@
|
||||
"""Docker container lifecycle management for ebook search."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
|
||||
import typer
|
||||
|
||||
from python.common import configure_logger, get_repo_dir
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def get_compose_file() -> Path:
|
||||
"""Return the path to the docker-compose.yml file."""
|
||||
return Path(__file__).resolve().with_name("docker-compose.yml")
|
||||
|
||||
|
||||
def compose_base_args() -> list[str]:
|
||||
"""Return the common docker compose arguments for the ebook search stack."""
|
||||
return ["compose", "-f", str(get_compose_file())]
|
||||
|
||||
|
||||
def docker_run(
|
||||
arguments: list[str],
|
||||
*,
|
||||
env: dict[str, str] | None = None,
|
||||
capture_output: bool = False,
|
||||
) -> subprocess.CompletedProcess[str]:
|
||||
"""Run docker with repo-root cwd and consistent error handling."""
|
||||
logger.info("docker %s", " ".join(arguments))
|
||||
return subprocess.run(
|
||||
["docker", *arguments],
|
||||
cwd=get_repo_dir(),
|
||||
env=env,
|
||||
text=True,
|
||||
check=False,
|
||||
capture_output=capture_output,
|
||||
)
|
||||
|
||||
|
||||
def compose_env(*, library_path: Path | None = None, port: int | None = None) -> dict[str, str]:
|
||||
"""Return environment variables passed to docker compose."""
|
||||
env = os.environ.copy()
|
||||
if library_path is not None:
|
||||
resolved_library = library_path.expanduser().resolve()
|
||||
if not resolved_library.exists():
|
||||
msg = f"EPUB library path does not exist: {resolved_library}"
|
||||
raise FileNotFoundError(msg)
|
||||
env["EBOOK_LIBRARY_HOST_PATH"] = str(resolved_library)
|
||||
if port is not None:
|
||||
env["EBOOK_SEARCH_PORT"] = str(port)
|
||||
return env
|
||||
|
||||
|
||||
def ensure_compose_file() -> None:
|
||||
"""Raise if the ebook search compose file is missing."""
|
||||
if not get_compose_file().is_file():
|
||||
msg = f"Compose file not found: {get_compose_file()}"
|
||||
raise FileNotFoundError(msg)
|
||||
|
||||
|
||||
def build_image() -> None:
|
||||
"""Build the ebook search app image."""
|
||||
ensure_compose_file()
|
||||
result = docker_run([*compose_base_args(), "build"])
|
||||
if result.returncode != 0:
|
||||
msg = "Failed to build ebook search image"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
|
||||
def build_test_image() -> None:
|
||||
"""Build the ebook search test Docker image."""
|
||||
dockerfile = Path(__file__).resolve().with_name("Dockerfile")
|
||||
result = docker_run(["build", "--file", str(dockerfile), "--target", "test", "--tag", "ebook-search:test", "."])
|
||||
if result.returncode != 0:
|
||||
msg = "Failed to build ebook search test image"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
|
||||
def run_test_image() -> None:
|
||||
"""Run the ebook search test suite inside Docker."""
|
||||
result = docker_run(["run", "--rm", "ebook-search:test"])
|
||||
if result.returncode != 0:
|
||||
msg = f"Ebook search tests failed with code {result.returncode}"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
|
||||
def start_stack(
|
||||
*,
|
||||
library_path: Path | None = None,
|
||||
port: int | None = None,
|
||||
build: bool = False,
|
||||
) -> None:
|
||||
"""Start the ebook search Docker compose stack."""
|
||||
ensure_compose_file()
|
||||
env = compose_env(library_path=library_path, port=port)
|
||||
if build:
|
||||
build_image()
|
||||
result = docker_run(
|
||||
[*compose_base_args(), "up", "-d"],
|
||||
env=env,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
msg = f"Ebook search stack failed to start with code {result.returncode}"
|
||||
raise RuntimeError(msg)
|
||||
logger.info("Ebook search started.")
|
||||
|
||||
|
||||
def stop_stack(
|
||||
*,
|
||||
volumes: bool = False,
|
||||
) -> None:
|
||||
"""Stop and remove ebook search containers."""
|
||||
ensure_compose_file()
|
||||
command = [*compose_base_args(), "down"]
|
||||
if volumes:
|
||||
command.append("-v")
|
||||
result = docker_run(command)
|
||||
if result.returncode != 0:
|
||||
msg = f"Ebook search stack failed to stop with code {result.returncode}"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
|
||||
def logs_stack(
|
||||
*,
|
||||
service: str | None = None,
|
||||
tail: int = 100,
|
||||
follow: bool = False,
|
||||
) -> str | None:
|
||||
"""Return recent logs from the ebook search stack."""
|
||||
ensure_compose_file()
|
||||
command = [*compose_base_args(), "logs", "--tail", str(tail)]
|
||||
if follow:
|
||||
command.append("--follow")
|
||||
if service:
|
||||
command.append(service)
|
||||
result = docker_run(command, capture_output=not follow)
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
if follow:
|
||||
return ""
|
||||
return result.stdout + result.stderr
|
||||
|
||||
|
||||
def ps_stack() -> str | None:
|
||||
"""Return docker compose ps output for the ebook search stack."""
|
||||
ensure_compose_file()
|
||||
result = docker_run([*compose_base_args(), "ps"], capture_output=True)
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
return result.stdout + result.stderr
|
||||
|
||||
|
||||
app = typer.Typer(help="Ebook search Docker container management.", no_args_is_help=True)
|
||||
|
||||
|
||||
@app.command()
|
||||
def build() -> None:
|
||||
"""Build the ebook search Docker image."""
|
||||
build_image()
|
||||
|
||||
|
||||
@app.command()
|
||||
def start(
|
||||
library_path: Annotated[Path | None, typer.Option(help="Override host path containing EPUB files.")] = None,
|
||||
port: Annotated[int | None, typer.Option(help="Override host port for the web UI.")] = None,
|
||||
*,
|
||||
build: Annotated[bool, typer.Option("--build", help="Build the image before starting.")] = False,
|
||||
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||
) -> None:
|
||||
"""Start the ebook search container."""
|
||||
configure_logger(log_level)
|
||||
start_stack(
|
||||
library_path=library_path,
|
||||
port=port,
|
||||
build=build,
|
||||
)
|
||||
|
||||
|
||||
@app.command()
|
||||
def stop(
|
||||
*,
|
||||
volumes: Annotated[bool, typer.Option("--volumes", help="Also remove ebook search data volumes.")] = False,
|
||||
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||
) -> None:
|
||||
"""Stop and remove ebook search containers."""
|
||||
configure_logger(log_level)
|
||||
stop_stack(volumes=volumes)
|
||||
|
||||
|
||||
@app.command()
|
||||
def restart(
|
||||
library_path: Annotated[Path | None, typer.Option(help="Override host path containing EPUB files.")] = None,
|
||||
port: Annotated[int | None, typer.Option(help="Override host port for the web UI.")] = None,
|
||||
*,
|
||||
build: Annotated[bool, typer.Option("--build", help="Build the image before starting.")] = False,
|
||||
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||
) -> None:
|
||||
"""Restart the ebook search stack."""
|
||||
configure_logger(log_level)
|
||||
stop_stack()
|
||||
start_stack(
|
||||
library_path=library_path,
|
||||
port=port,
|
||||
build=build,
|
||||
)
|
||||
|
||||
|
||||
@app.command()
|
||||
def logs(
|
||||
service: Annotated[str | None, typer.Option(help="Service name, or omit for all services.")] = None,
|
||||
tail: Annotated[int, typer.Option(help="Number of recent log lines.")] = 100,
|
||||
*,
|
||||
follow: Annotated[bool, typer.Option("--follow", "-f", help="Follow logs.")] = False,
|
||||
) -> None:
|
||||
"""Show recent ebook search container logs."""
|
||||
output = logs_stack(service=service, tail=tail, follow=follow)
|
||||
if output is None:
|
||||
typer.echo("No ebook search containers found.")
|
||||
raise typer.Exit(code=1)
|
||||
if output:
|
||||
typer.echo(output)
|
||||
|
||||
|
||||
@app.command("test")
|
||||
def run_tests(
|
||||
*,
|
||||
build: Annotated[bool, typer.Option("--build/--no-build", help="Build the test image before running.")] = True,
|
||||
log_level: Annotated[str, typer.Option(help="Log level.")] = "INFO",
|
||||
) -> None:
|
||||
"""Run ebook search tests inside the Docker test image."""
|
||||
configure_logger(log_level)
|
||||
if build:
|
||||
build_test_image()
|
||||
run_test_image()
|
||||
|
||||
|
||||
@app.command("ps")
|
||||
def ps() -> None:
|
||||
"""Show ebook search container status."""
|
||||
output = ps_stack()
|
||||
if output is None:
|
||||
typer.echo("No ebook search containers found.")
|
||||
raise typer.Exit(code=1)
|
||||
typer.echo(output)
|
||||
|
||||
|
||||
def cli() -> None:
|
||||
"""Typer entry point."""
|
||||
app()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
cli()
|
||||
@@ -0,0 +1,36 @@
|
||||
name: ebook-search
|
||||
|
||||
services:
|
||||
ebook-search:
|
||||
build:
|
||||
context: ../../..
|
||||
dockerfile: python/ebook_search/docker/Dockerfile
|
||||
image: ebook-search:latest
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "${EBOOK_SEARCH_PORT:-8070}:8070"
|
||||
extra_hosts:
|
||||
- "jeeves:192.168.90.40"
|
||||
env_file:
|
||||
- ../../../.env
|
||||
environment:
|
||||
EBOOK_SEARCH_HOST: "0.0.0.0"
|
||||
EBOOK_SEARCH_PORT: "8070"
|
||||
EBOOK_SEARCH_LIBRARY_PATHS: "/library"
|
||||
EBOOK_SEARCH_BM25_INDEX_DIR: "/data/bm25"
|
||||
volumes:
|
||||
- "${EBOOK_LIBRARY_HOST_PATH:-/home/richie/ebooks}:/library:ro"
|
||||
- ebook-search-data:/data
|
||||
healthcheck:
|
||||
test:
|
||||
[
|
||||
"CMD-SHELL",
|
||||
"curl -fsS http://127.0.0.1:8070/health >/dev/null || exit 1",
|
||||
]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 30s
|
||||
|
||||
volumes:
|
||||
ebook-search-data:
|
||||
@@ -0,0 +1,38 @@
|
||||
[project]
|
||||
name = "ebook-search"
|
||||
version = "0.1.0"
|
||||
description = "Locked runtime environment for the ebook search container."
|
||||
requires-python = "~=3.14.0"
|
||||
dependencies = [
|
||||
"alembic",
|
||||
"beautifulsoup4",
|
||||
"bm25s",
|
||||
"ebooklib",
|
||||
"fastapi",
|
||||
"httpx",
|
||||
"jinja2",
|
||||
"pgvector",
|
||||
"psycopg[binary]",
|
||||
"pydantic",
|
||||
"pydantic-settings",
|
||||
"python-multipart",
|
||||
"sqlalchemy",
|
||||
"tiktoken",
|
||||
"typer",
|
||||
"uvicorn[standard]",
|
||||
"yake",
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
"pytest",
|
||||
"pytest-mock",
|
||||
"pytest-xdist",
|
||||
]
|
||||
|
||||
[tool.uv]
|
||||
package = false
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
addopts = "-n auto -ra"
|
||||
testpaths = ["tests/ebook_search"]
|
||||
Generated
+1109
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,170 @@
|
||||
"""Embedding model helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from sqlalchemy import func, select
|
||||
from sqlalchemy.dialects.postgresql import insert
|
||||
|
||||
from python.ebook_search.llm_interface import request_embeddings
|
||||
from python.orm.richie import (
|
||||
EbookChunk,
|
||||
EbookChunkEmbedding1024,
|
||||
EbookChunkEmbedding2560,
|
||||
EbookChunkEmbedding4096,
|
||||
EbookEmbeddingModel,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
MODEL_DIMENSIONS = {
|
||||
"qwen3-embedding-0.6b": 1024,
|
||||
"qwen3-embedding-4b": 2560,
|
||||
"qwen3-embedding-8b": 4096,
|
||||
}
|
||||
|
||||
|
||||
def get_embedding_table(
|
||||
dimension: int,
|
||||
) -> type[EbookChunkEmbedding1024 | EbookChunkEmbedding2560 | EbookChunkEmbedding4096]:
|
||||
"""Return the embedding table mapped to an embedding dimension."""
|
||||
embedding_tables = {
|
||||
1024: EbookChunkEmbedding1024,
|
||||
2560: EbookChunkEmbedding2560,
|
||||
4096: EbookChunkEmbedding4096,
|
||||
}
|
||||
table = embedding_tables.get(dimension)
|
||||
if not table:
|
||||
msg = f"Embedding dimension {dimension} is not supported"
|
||||
raise ValueError(msg)
|
||||
return table
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EmbeddingModelStats:
|
||||
"""Embedding coverage for one model."""
|
||||
|
||||
model_name: str
|
||||
dimension: int
|
||||
embedded_chunks: int
|
||||
total_chunks: int
|
||||
|
||||
@property
|
||||
def missing_chunks(self) -> int:
|
||||
"""Return chunks missing this embedding model."""
|
||||
return max(self.total_chunks - self.embedded_chunks, 0)
|
||||
|
||||
|
||||
def embed_texts(texts: Sequence[str], config: EbookSearchConfig) -> list[list[float]]:
|
||||
"""Embed text with the configured vLLM embedding model."""
|
||||
logger.info(
|
||||
"ebook_embed_request_start base_url=%s model=%s count=%s",
|
||||
config.embedding_base_url,
|
||||
config.embedding_model,
|
||||
len(texts),
|
||||
)
|
||||
vectors = request_embeddings(texts, config)
|
||||
expected_dimension = MODEL_DIMENSIONS[config.embedding_model]
|
||||
for vector in vectors:
|
||||
if len(vector) != expected_dimension:
|
||||
msg = f"Expected {expected_dimension} dimensions, got {len(vector)}"
|
||||
raise ValueError(msg)
|
||||
logger.info(
|
||||
"ebook_embed_request_complete model=%s count=%s dimension=%s",
|
||||
config.embedding_model,
|
||||
len(vectors),
|
||||
expected_dimension,
|
||||
)
|
||||
return vectors
|
||||
|
||||
|
||||
def embed_query(query: str, config: EbookSearchConfig) -> list[float]:
|
||||
"""Embed a search query with the Qwen retrieval instruction."""
|
||||
instructed_query = f"Instruct: Retrieve relevant passages for the query.\nQuery: {query}"
|
||||
return embed_texts([instructed_query], config)[0]
|
||||
|
||||
|
||||
def ensure_embedding_models(session: Session) -> None:
|
||||
"""Ensure supported embedding model rows exist."""
|
||||
for name, dimension in MODEL_DIMENSIONS.items():
|
||||
existing = session.scalar(select(EbookEmbeddingModel).where(EbookEmbeddingModel.name == name))
|
||||
if existing is None:
|
||||
session.add(EbookEmbeddingModel(name=name, dimension=dimension, is_default=name == "qwen3-embedding-0.6b"))
|
||||
logger.info("ebook_embedding_model_created model=%s dimension=%s", name, dimension)
|
||||
session.flush()
|
||||
|
||||
|
||||
def embedding_model_stats(session: Session) -> list[EmbeddingModelStats]:
|
||||
"""Return embedding coverage counts for every supported model."""
|
||||
total_chunks = session.scalar(select(func.count(EbookChunk.id))) or 0
|
||||
models = {
|
||||
model.name: model
|
||||
for model in session.scalars(
|
||||
select(EbookEmbeddingModel)
|
||||
.where(EbookEmbeddingModel.name.in_(MODEL_DIMENSIONS))
|
||||
.order_by(EbookEmbeddingModel.name)
|
||||
)
|
||||
}
|
||||
|
||||
stats: list[EmbeddingModelStats] = []
|
||||
for model_name, dimension in MODEL_DIMENSIONS.items():
|
||||
model = models.get(model_name)
|
||||
embedded_chunks = 0
|
||||
if model is not None:
|
||||
table = get_embedding_table(dimension)
|
||||
embedded_chunks = session.scalar(select(func.count(table.id)).where(table.model_id == model.id)) or 0
|
||||
stats.append(
|
||||
EmbeddingModelStats(
|
||||
model_name=model_name,
|
||||
dimension=dimension,
|
||||
embedded_chunks=embedded_chunks,
|
||||
total_chunks=total_chunks,
|
||||
)
|
||||
)
|
||||
return stats
|
||||
|
||||
|
||||
def embed_missing_chunks(session: Session, config: EbookSearchConfig) -> int:
|
||||
"""Embed chunks missing embeddings for the configured model."""
|
||||
ensure_embedding_models(session)
|
||||
model = session.scalar(select(EbookEmbeddingModel).where(EbookEmbeddingModel.name == config.embedding_model))
|
||||
if model is None:
|
||||
supported_models = ", ".join(MODEL_DIMENSIONS)
|
||||
msg = f"Unknown embedding model: {config.embedding_model}. Supported models: {supported_models}"
|
||||
raise ValueError(msg)
|
||||
|
||||
table = get_embedding_table(model.dimension)
|
||||
chunks = list(
|
||||
session.scalars(
|
||||
select(EbookChunk)
|
||||
.outerjoin(table, (table.chunk_id == EbookChunk.id) & (table.model_id == model.id))
|
||||
.where(table.id.is_(None))
|
||||
.order_by(EbookChunk.id)
|
||||
.limit(config.embedding_batch_size)
|
||||
)
|
||||
)
|
||||
if not chunks:
|
||||
logger.info("ebook_embed_missing_none model=%s", config.embedding_model)
|
||||
return 0
|
||||
|
||||
logger.info("ebook_embed_missing_batch_start model=%s count=%s", config.embedding_model, len(chunks))
|
||||
vectors = embed_texts([chunk.text for chunk in chunks], config)
|
||||
rows = [
|
||||
{"chunk_id": chunk.id, "model_id": model.id, "embedding": vector}
|
||||
for chunk, vector in zip(chunks, vectors, strict=True)
|
||||
]
|
||||
statement = insert(table).values(rows).on_conflict_do_nothing(index_elements=["chunk_id", "model_id"])
|
||||
session.execute(statement)
|
||||
session.flush()
|
||||
logger.info("ebook_embed_missing_batch_complete model=%s count=%s", config.embedding_model, len(rows))
|
||||
return len(rows)
|
||||
@@ -0,0 +1,95 @@
|
||||
"""EPUB parsing helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from ebooklib import ITEM_DOCUMENT, epub
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from pathlib import Path
|
||||
|
||||
WHITESPACE_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ParsedChapter:
|
||||
"""Text extracted from one EPUB spine document."""
|
||||
|
||||
title: str | None
|
||||
href: str | None
|
||||
text: str
|
||||
page_labels: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ParsedEpub:
|
||||
"""Parsed EPUB metadata and text."""
|
||||
|
||||
title: str
|
||||
author: str | None
|
||||
language: str | None
|
||||
publisher: str | None
|
||||
identifier: str | None
|
||||
chapters: tuple[ParsedChapter, ...]
|
||||
|
||||
|
||||
def parse_epub(path: Path) -> ParsedEpub:
|
||||
"""Parse EPUB metadata and spine text."""
|
||||
book = epub.read_epub(path)
|
||||
chapters = []
|
||||
for item in book.get_items_of_type(ITEM_DOCUMENT):
|
||||
soup = BeautifulSoup(item.get_content(), "html.parser")
|
||||
title = chapter_title(soup)
|
||||
page_labels = tuple(extract_page_labels(soup))
|
||||
text = clean_text(soup.get_text(" "))
|
||||
if text:
|
||||
chapters.append(ParsedChapter(title=title, href=item.get_name(), text=text, page_labels=page_labels))
|
||||
|
||||
return ParsedEpub(
|
||||
title=metadata_value(book, "title") or path.stem,
|
||||
author=metadata_value(book, "creator"),
|
||||
language=metadata_value(book, "language"),
|
||||
publisher=metadata_value(book, "publisher"),
|
||||
identifier=metadata_value(book, "identifier"),
|
||||
chapters=tuple(chapters),
|
||||
)
|
||||
|
||||
|
||||
def metadata_value(book: epub.EpubBook, name: str) -> str | None:
|
||||
"""Return the first non-empty Dublin Core metadata value for a name."""
|
||||
values = book.get_metadata("DC", name)
|
||||
if not values:
|
||||
return None
|
||||
value = values[0][0]
|
||||
return str(value).strip() or None
|
||||
|
||||
|
||||
def chapter_title(soup: BeautifulSoup) -> str | None:
|
||||
"""Extract the best available title from an EPUB document soup."""
|
||||
heading = soup.find(["h1", "h2", "h3"])
|
||||
if heading is None:
|
||||
title = soup.find("title")
|
||||
if title is None:
|
||||
return None
|
||||
return clean_text(title.get_text(" ")) or None
|
||||
return clean_text(heading.get_text(" ")) or None
|
||||
|
||||
|
||||
def extract_page_labels(soup: BeautifulSoup) -> list[str]:
|
||||
"""Extract EPUB page-break labels from a document soup."""
|
||||
labels: list[str] = []
|
||||
for tag in soup.find_all(attrs={"epub:type": "pagebreak"}):
|
||||
label = tag.get("title") or tag.get("aria-label") or tag.get_text(" ")
|
||||
clean = clean_text(str(label))
|
||||
if clean:
|
||||
labels.append(clean)
|
||||
return labels
|
||||
|
||||
|
||||
def clean_text(text: str) -> str:
|
||||
"""Normalize whitespace in extracted EPUB text."""
|
||||
return WHITESPACE_RE.sub(" ", text).strip()
|
||||
@@ -0,0 +1 @@
|
||||
"""Offline evaluation tooling for the ebook search pipeline."""
|
||||
@@ -0,0 +1,71 @@
|
||||
{"query": "Who is Damien Montgomery and how does he become a Jump Mage?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What is a Rune Wright and why is Damien so rare?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "How does jump magic let starships travel faster than light?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What is the role of the Mage-King of Mars in the Protectorate?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What happened aboard the Blue Jay in the first Starship's Mage book?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "Who is Captain David Rice?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "How are amplifiers and simulacrums used to power a ship's jump?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What duties does a Hand of the Mage-King carry out?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "Explain the structure of the Royal Martian Navy.", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "How do mages carve runes to enchant a starship?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What threat do the Legatan rebels pose to the Protectorate?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "How does Damien handle his first command?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What is the significance of the simulacrum on a jump ship?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "Describe a mage duel in the Starship's Mage series.", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What moral conflicts does Damien face as a Hand of the Mage-King?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "How does the Protectorate keep peace among its member worlds?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "Who is the Keeper of Oaths and how does Damien work with them?", "answer": null, "answerable": true, "relevant_sources": ["Starship's Mage"]}
|
||||
{"query": "What event is known as the Onset and how does it change the world?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "Who is the main character at the start of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How do survivors adapt after the Onset begins?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What new abilities emerge during the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "Describe the primary antagonist in the Onset series.", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How does society collapse and reorganize after the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What factions form in the aftermath of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How does the protagonist gain power throughout the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What is the cause or origin of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "Describe an early survival challenge faced after the Onset.", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How do the characters defend their stronghold during the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What relationships drive the protagonist's choices in the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How does the Onset escalate by the end of the first book?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What mysteries about the Onset remain unresolved?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How do the rules of the world change once the Onset takes hold?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "What weapons or tactics work best against the threats of the Onset?", "answer": null, "answerable": true, "relevant_sources": ["The Onset"]}
|
||||
{"query": "How does Bob Johansson become a von Neumann probe?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What is a replicant and why do Bob's copies have different personalities?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "Who are Riker, Homer, and Bill among the Bob clones?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What is GUPPI and how does Bob use it?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "Describe the threat posed by the Others.", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How does Bob protect and uplift the Deltans?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "Why do the replicants drift apart in personality over time?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What is the role of FAITH and the Brazilian Empire on Earth?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How does subspace communication work for the Bobs?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What happens to Bender after he goes missing?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How do the Bobs build self-replicating probes across the galaxy?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How does Bob evacuate humanity after Earth becomes uninhabitable?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "Describe the conflict between different factions of Bobs.", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What ethical dilemmas does Bob face when interfering with primitive species?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How does the original Bob differ from later generations of clones?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "How do the Bobs defeat the Others' system-harvesting fleets?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
{"query": "What role does Howard play in the human colonies?", "answer": null, "answerable": true, "relevant_sources": ["We Are Legion (We Are Bob)"]}
|
||||
// querys not it the dataset
|
||||
{"query": "How does Frodo destroy the One Ring in The Lord of the Rings?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "Who killed Dumbledore in Harry Potter and the Half-Blood Prince?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What house does Tyrion Lannister belong to in A Game of Thrones?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How does Paul Atreides control the spice on Arrakis in Dune?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What does the green light at the end of the dock mean in The Great Gatsby?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "Why does Hester Prynne wear a scarlet letter?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What does the white whale represent in Moby-Dick?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How does Elizabeth Bennet's view of Mr. Darcy change in Pride and Prejudice?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What crime does Raskolnikov commit in Crime and Punishment?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How does Katniss volunteer for the Hunger Games?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What is Winston Smith's job in Nineteen Eighty-Four?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "Who is Atticus Finch defending in To Kill a Mockingbird?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What is the capital of Australia?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How do I bake a sourdough loaf from scratch?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "Explain how photosynthesis converts sunlight into energy.", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What were the main causes of World War I?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How does compound interest work?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "How do I change a flat tire on a car?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What is the boiling point of water at sea level?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
{"query": "What is the recommended daily intake of vitamin D?", "answer": null, "answerable": false, "relevant_sources": []}
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Shared query set loading for evaluation and load testing.
|
||||
|
||||
Each JSONL record has a ``query`` and an optional reference ``answer``. ``answerable``
|
||||
marks whether the query should be answerable from the library (false for out-of-corpus
|
||||
"garbage" queries used to test the refusal path). Relevance for retrieval metrics is
|
||||
labeled at source (book) granularity in ``relevant_sources``; source titles must match
|
||||
``ebook_source.title`` values for the indexed corpus.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
DEFAULT_QUERIES_PATH = Path(__file__).parent / "data" / "queries.jsonl"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class GoldQuery:
|
||||
"""One labeled query shared by the eval and load-test tools."""
|
||||
|
||||
query: str
|
||||
answer: str | None
|
||||
answerable: bool
|
||||
relevant_sources: tuple[str, ...]
|
||||
relevant_substrings: tuple[str, ...]
|
||||
|
||||
|
||||
def load_gold_queries(path: Path = DEFAULT_QUERIES_PATH) -> list[GoldQuery]:
|
||||
"""Load labeled queries from a JSONL file. Blank lines and ``//`` comment lines are skipped."""
|
||||
queries: list[GoldQuery] = []
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
stripped = line.strip()
|
||||
if not stripped or stripped.startswith("//"):
|
||||
continue
|
||||
record = json.loads(stripped)
|
||||
queries.append(
|
||||
GoldQuery(
|
||||
query=str(record["query"]),
|
||||
answer=record.get("answer"),
|
||||
answerable=bool(record.get("answerable", True)),
|
||||
relevant_sources=tuple(record.get("relevant_sources", ())),
|
||||
relevant_substrings=tuple(record.get("relevant_substrings", ())),
|
||||
)
|
||||
)
|
||||
return queries
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Serve-time output guardrails for retrieval confidence and answer citations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
from python.ebook_search.search import SearchResult
|
||||
|
||||
CITATION_RE = re.compile(r"\[(\d+)\]")
|
||||
|
||||
|
||||
def retrieval_confidence(results: list[SearchResult]) -> float:
|
||||
"""Return the strongest interpretable relevance signal of the top result.
|
||||
|
||||
Reciprocal-rank-fusion scores are rank-based and not comparable across queries,
|
||||
so the rerank relevance score is preferred, then vector cosine similarity, then
|
||||
the final score.
|
||||
"""
|
||||
if not results:
|
||||
return 0.0
|
||||
top = results[0]
|
||||
if top.rerank_score is not None:
|
||||
return top.rerank_score
|
||||
if top.vector_score is not None:
|
||||
return top.vector_score
|
||||
return top.score
|
||||
|
||||
|
||||
def is_confident(results: list[SearchResult], config: EbookSearchConfig) -> bool:
|
||||
"""Return whether top-result confidence meets the configured threshold."""
|
||||
return retrieval_confidence(results) >= config.min_retrieval_confidence
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CitationReport:
|
||||
"""Validation summary for bracketed citation markers in a generated answer."""
|
||||
|
||||
cited: tuple[int, ...]
|
||||
invalid: tuple[int, ...]
|
||||
grounded: bool
|
||||
|
||||
|
||||
def validate_citations(answer: str, result_count: int) -> CitationReport:
|
||||
"""Validate bracketed citation markers against the number of shown sources.
|
||||
|
||||
A marker is valid when it points to a returned source (``1..result_count``).
|
||||
``grounded`` is true when the answer cites at least one valid source.
|
||||
"""
|
||||
markers = sorted({int(match.group(1)) for match in CITATION_RE.finditer(answer)})
|
||||
valid = range(1, result_count + 1)
|
||||
cited = tuple(marker for marker in markers if marker in valid)
|
||||
invalid = tuple(marker for marker in markers if marker not in valid)
|
||||
return CitationReport(cited=cited, invalid=invalid, grounded=bool(cited))
|
||||
@@ -0,0 +1,195 @@
|
||||
"""EPUB ingestion into Richie DB."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import tiktoken
|
||||
from sqlalchemy import or_, select
|
||||
|
||||
from python.ebook_search.epub_parse import parse_epub
|
||||
from python.orm.richie import EbookChapter, EbookChunk, EbookSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
DEFAULT_CHUNK_TOKENS = 700
|
||||
DEFAULT_CHUNK_OVERLAP = 100
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
from python.ebook_search.epub_parse import ParsedChapter
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TextChunk:
|
||||
"""A token-bounded chunk of text."""
|
||||
|
||||
text: str
|
||||
token_start: int
|
||||
token_count: int
|
||||
|
||||
|
||||
def chunk_text(
|
||||
text: str,
|
||||
*,
|
||||
chunk_tokens: int = DEFAULT_CHUNK_TOKENS,
|
||||
overlap_tokens: int = DEFAULT_CHUNK_OVERLAP,
|
||||
) -> list[TextChunk]:
|
||||
"""Split text into overlapping token chunks."""
|
||||
if chunk_tokens <= 0:
|
||||
msg = "chunk_tokens must be positive"
|
||||
raise ValueError(msg)
|
||||
if overlap_tokens < 0 or overlap_tokens >= chunk_tokens:
|
||||
msg = "overlap_tokens must be non-negative and smaller than chunk_tokens"
|
||||
raise ValueError(msg)
|
||||
|
||||
encoding = tiktoken.get_encoding("cl100k_base")
|
||||
tokens = encoding.encode(text)
|
||||
if not tokens:
|
||||
return []
|
||||
|
||||
chunks: list[TextChunk] = []
|
||||
step = chunk_tokens - overlap_tokens
|
||||
for start in range(0, len(tokens), step):
|
||||
chunk = tokens[start : start + chunk_tokens]
|
||||
if not chunk:
|
||||
continue
|
||||
chunks.append(
|
||||
TextChunk(
|
||||
text=encoding.decode(chunk).strip(),
|
||||
token_start=start,
|
||||
token_count=len(chunk),
|
||||
)
|
||||
)
|
||||
if start + chunk_tokens >= len(tokens):
|
||||
break
|
||||
return [chunk for chunk in chunks if chunk.text]
|
||||
|
||||
|
||||
def ingest_configured_paths(session: Session, config: EbookSearchConfig) -> int:
|
||||
"""Ingest every EPUB found under configured library paths."""
|
||||
count = 0
|
||||
for library_path in config.library_paths:
|
||||
path = Path(library_path).expanduser()
|
||||
logger.info("ebook_ingest_path_start path=%s", path)
|
||||
if path.is_file() and path.suffix.lower() == ".epub":
|
||||
count += int(ingest_file(session, path, config))
|
||||
elif path.is_dir():
|
||||
for epub_path in sorted(path.rglob("*.epub")):
|
||||
count += int(ingest_file(session, epub_path, config))
|
||||
else:
|
||||
logger.warning("ebook_ingest_path_missing path=%s", path)
|
||||
logger.info("ebook_ingest_paths_complete changed_files=%s configured_paths=%s", count, len(config.library_paths))
|
||||
return count
|
||||
|
||||
|
||||
def ingest_file(session: Session, path: Path, config: EbookSearchConfig) -> bool:
|
||||
"""Ingest one EPUB file. Return True when the database changed."""
|
||||
resolved_path = path.expanduser().resolve()
|
||||
logger.info("ebook_ingest_file_start path=%s", resolved_path)
|
||||
file_hash = sha256_file(resolved_path)
|
||||
existing = find_existing_source(session, resolved_path, file_hash)
|
||||
if existing is not None and existing.file_sha256 == file_hash:
|
||||
stat = resolved_path.stat()
|
||||
existing.file_path = str(resolved_path)
|
||||
existing.file_mtime = datetime.fromtimestamp(stat.st_mtime, tz=UTC)
|
||||
existing.file_size = stat.st_size
|
||||
session.flush()
|
||||
logger.info("ebook_ingest_file_unchanged source_id=%s path=%s", existing.id, resolved_path)
|
||||
return False
|
||||
if existing is not None:
|
||||
logger.info("ebook_ingest_file_replacing source_id=%s path=%s", existing.id, resolved_path)
|
||||
session.delete(existing)
|
||||
session.flush()
|
||||
|
||||
stat = resolved_path.stat()
|
||||
parsed = parse_epub(resolved_path)
|
||||
source = EbookSource(
|
||||
title=parsed.title,
|
||||
author=parsed.author,
|
||||
language=parsed.language,
|
||||
publisher=parsed.publisher,
|
||||
identifier=parsed.identifier,
|
||||
file_path=str(resolved_path),
|
||||
file_sha256=file_hash,
|
||||
file_mtime=datetime.fromtimestamp(stat.st_mtime, tz=UTC),
|
||||
file_size=stat.st_size,
|
||||
)
|
||||
session.add(source)
|
||||
session.flush()
|
||||
|
||||
chunk_index = 0
|
||||
for spine_index, parsed_chapter in enumerate(parsed.chapters):
|
||||
chapter = EbookChapter(
|
||||
source_id=source.id,
|
||||
spine_index=spine_index,
|
||||
title=parsed_chapter.title,
|
||||
href=parsed_chapter.href,
|
||||
)
|
||||
session.add(chapter)
|
||||
session.flush()
|
||||
chunk_index = add_chapter_chunks(session, source, chapter, parsed_chapter, chunk_index, config)
|
||||
|
||||
session.flush()
|
||||
logger.info(
|
||||
"ebook_ingest_file_complete source_id=%s path=%s chapters=%s chunks=%s",
|
||||
source.id,
|
||||
resolved_path,
|
||||
len(parsed.chapters),
|
||||
chunk_index,
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def find_existing_source(session: Session, path: Path, file_hash: str) -> EbookSource | None:
|
||||
"""Find an existing source by canonical path or file hash."""
|
||||
return session.scalar(
|
||||
select(EbookSource).where(or_(EbookSource.file_path == str(path), EbookSource.file_sha256 == file_hash))
|
||||
)
|
||||
|
||||
|
||||
def add_chapter_chunks(
|
||||
session: Session,
|
||||
source: EbookSource,
|
||||
chapter: EbookChapter,
|
||||
parsed_chapter: ParsedChapter,
|
||||
chunk_index: int,
|
||||
config: EbookSearchConfig,
|
||||
) -> int:
|
||||
"""Add chunk rows for one parsed chapter and return the next chunk index."""
|
||||
page_label = parsed_chapter.page_labels[0] if parsed_chapter.page_labels else None
|
||||
for text_chunk in chunk_text(
|
||||
parsed_chapter.text,
|
||||
chunk_tokens=config.chunk_tokens,
|
||||
overlap_tokens=config.chunk_overlap,
|
||||
):
|
||||
session.add(
|
||||
EbookChunk(
|
||||
source_id=source.id,
|
||||
chapter_id=chapter.id,
|
||||
chunk_index=chunk_index,
|
||||
text=text_chunk.text,
|
||||
token_start=text_chunk.token_start,
|
||||
token_count=text_chunk.token_count,
|
||||
page_label=page_label,
|
||||
content_sha256=hashlib.sha256(text_chunk.text.encode()).hexdigest(),
|
||||
search_text=f"{source.title} {source.author or ''} {chapter.title or ''} {text_chunk.text}",
|
||||
)
|
||||
)
|
||||
chunk_index += 1
|
||||
return chunk_index
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
"""Calculate the SHA-256 digest for a file."""
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as file:
|
||||
for block in iter(lambda: file.read(1024 * 1024), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
@@ -0,0 +1,173 @@
|
||||
"""LLM provider HTTP adapters."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import httpx
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Sequence
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig, RerankConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def auth_headers(api_key: str) -> dict[str, str]:
|
||||
"""Build authorization headers when an API key is configured."""
|
||||
if api_key == "not-needed":
|
||||
return {}
|
||||
return {"Authorization": f"Bearer {api_key}"}
|
||||
|
||||
|
||||
def request_embeddings(texts: Sequence[str], config: EbookSearchConfig) -> list[list[float]]:
|
||||
"""Request embeddings from the configured OpenAI-compatible endpoint."""
|
||||
try:
|
||||
response = httpx.post(
|
||||
f"{config.embedding_base_url.rstrip('/')}/embeddings",
|
||||
headers=auth_headers(config.embedding_api_key),
|
||||
json={"model": config.embedding_model, "input": list(texts)},
|
||||
timeout=config.embedding_timeout_seconds,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return embedding_vectors_from_response(response.json())
|
||||
except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
|
||||
logger.exception(
|
||||
"ebook_embed_request_failed base_url=%s model=%s count=%s",
|
||||
config.embedding_base_url,
|
||||
config.embedding_model,
|
||||
len(texts),
|
||||
)
|
||||
msg = f"Embedding request failed. base_url={config.embedding_base_url} model={config.embedding_model}"
|
||||
raise RuntimeError(msg) from error
|
||||
|
||||
|
||||
def check_embedding_endpoint(config: EbookSearchConfig, *, timeout_seconds: float = 5.0) -> bool:
|
||||
"""Return whether the configured embedding endpoint answers a model listing."""
|
||||
try:
|
||||
response = httpx.get(
|
||||
f"{config.embedding_base_url.rstrip('/')}/models",
|
||||
headers=auth_headers(config.embedding_api_key),
|
||||
timeout=timeout_seconds,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except httpx.HTTPError as error:
|
||||
logger.warning("ebook_embedding_endpoint_unreachable base_url=%s error=%s", config.embedding_base_url, error)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def check_chat_endpoint(config: EbookSearchConfig, *, timeout_seconds: float = 5.0) -> bool:
|
||||
"""Return whether the configured chat (answering) endpoint answers a model listing."""
|
||||
try:
|
||||
response = httpx.get(
|
||||
f"{config.vllm_base_url.rstrip('/')}/models",
|
||||
headers=auth_headers(config.vllm_api_key),
|
||||
timeout=timeout_seconds,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except httpx.HTTPError as error:
|
||||
logger.warning("ebook_chat_endpoint_unreachable base_url=%s error=%s", config.vllm_base_url, error)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def embedding_vectors_from_response(body: object) -> list[list[float]]:
|
||||
"""Extract embedding vectors from an OpenAI-compatible embedding response."""
|
||||
if not isinstance(body, dict):
|
||||
msg = "Embedding response is not an object"
|
||||
raise TypeError(msg)
|
||||
|
||||
data = body["data"]
|
||||
if not isinstance(data, list):
|
||||
msg = "Embedding response data is not a list"
|
||||
raise TypeError(msg)
|
||||
|
||||
vectors: list[list[float]] = []
|
||||
for item in data:
|
||||
if not isinstance(item, dict):
|
||||
msg = "Embedding item is not an object"
|
||||
raise TypeError(msg)
|
||||
embedding = item["embedding"]
|
||||
if not isinstance(embedding, list):
|
||||
msg = "Embedding value is not a list"
|
||||
raise TypeError(msg)
|
||||
vectors.append([float(value) for value in embedding])
|
||||
return vectors
|
||||
|
||||
|
||||
def request_rerank(
|
||||
query: str,
|
||||
documents: Sequence[str],
|
||||
config: RerankConfig,
|
||||
) -> object | None:
|
||||
"""Request rerank scores from the configured vLLM endpoint."""
|
||||
payload = {
|
||||
"model": config.model,
|
||||
"query": query,
|
||||
"documents": list(documents),
|
||||
}
|
||||
response = httpx.post(
|
||||
f"{config.base_url.rstrip('/')}/rerank",
|
||||
json=payload,
|
||||
timeout=config.timeout_seconds,
|
||||
)
|
||||
response.raise_for_status()
|
||||
try:
|
||||
return response.json()
|
||||
except ValueError:
|
||||
logger.debug("ebook_rerank_response_invalid_json", extra={"response": response.text})
|
||||
return None
|
||||
|
||||
|
||||
def request_chat_completion(
|
||||
config: EbookSearchConfig,
|
||||
messages: Sequence[dict[str, str]],
|
||||
) -> str:
|
||||
"""Request a chat completion from the configured OpenAI-compatible endpoint."""
|
||||
try:
|
||||
response = httpx.post(
|
||||
f"{config.vllm_base_url.rstrip('/')}/chat/completions",
|
||||
headers=auth_headers(config.vllm_api_key),
|
||||
json={
|
||||
"model": config.chat_model,
|
||||
"messages": list(messages),
|
||||
"temperature": 0,
|
||||
},
|
||||
timeout=config.chat_timeout_seconds,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return chat_content_from_response(response.json())
|
||||
except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
|
||||
msg = f"Chat request failed. base_url={config.vllm_base_url} model={config.chat_model}"
|
||||
raise RuntimeError(msg) from error
|
||||
|
||||
|
||||
def chat_content_from_response(body: object) -> str:
|
||||
"""Extract text content from an OpenAI-compatible chat response."""
|
||||
if not isinstance(body, dict):
|
||||
msg = "Chat response is not an object"
|
||||
raise TypeError(msg)
|
||||
|
||||
choices = body["choices"]
|
||||
if not isinstance(choices, list) or not choices:
|
||||
msg = "Chat response has no choices"
|
||||
raise ValueError(msg)
|
||||
|
||||
first = choices[0]
|
||||
if not isinstance(first, dict):
|
||||
msg = "Chat choice is not an object"
|
||||
raise TypeError(msg)
|
||||
|
||||
message = first["message"]
|
||||
if not isinstance(message, dict):
|
||||
msg = "Chat message is not an object"
|
||||
raise TypeError(msg)
|
||||
|
||||
content = message.get("content") or ""
|
||||
if not isinstance(content, str):
|
||||
msg = "Chat content is not text"
|
||||
raise TypeError(msg)
|
||||
return content
|
||||
@@ -0,0 +1,218 @@
|
||||
"""Load test for the EPUB search service.
|
||||
|
||||
Drives ``POST /search`` on a running server at a configurable concurrency and reports
|
||||
latency percentiles, throughput, and HTTP status distribution. Queries are drawn from
|
||||
the shared JSONL set (see ``eval/data/queries.jsonl``) that the eval also uses, so load
|
||||
and evaluation exercise the same questions. Answer generation and reranking happen
|
||||
server-side, so this exercises the full retrieval pipeline.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import math
|
||||
import random
|
||||
import statistics
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
|
||||
import httpx
|
||||
import typer
|
||||
|
||||
from python.common import configure_logger
|
||||
from python.ebook_search.eval.dataset import DEFAULT_QUERIES_PATH, load_gold_queries
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RequestResult:
|
||||
"""Outcome of a single search request."""
|
||||
|
||||
status_code: int
|
||||
latency_ms: float
|
||||
ok: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LoadSummary:
|
||||
"""Aggregate results of a load test run."""
|
||||
|
||||
total: int
|
||||
successes: int
|
||||
failures: int
|
||||
wall_seconds: float
|
||||
throughput_rps: float
|
||||
latency_p50_ms: float
|
||||
latency_p90_ms: float
|
||||
latency_p95_ms: float
|
||||
latency_p99_ms: float
|
||||
latency_mean_ms: float
|
||||
latency_max_ms: float
|
||||
status_counts: dict[int, int]
|
||||
|
||||
|
||||
def load_queries(queries_file: str | None) -> list[str]:
|
||||
"""Return the query strings from the shared JSONL set (or a custom JSONL file)."""
|
||||
path = Path(queries_file) if queries_file else DEFAULT_QUERIES_PATH
|
||||
queries = [gold.query for gold in load_gold_queries(path)]
|
||||
if not queries:
|
||||
msg = f"No queries found in {path}"
|
||||
raise typer.BadParameter(msg)
|
||||
return queries
|
||||
|
||||
|
||||
def pick_query(queries: list[str]) -> str:
|
||||
"""Return a uniformly random query from the pool (not a security context)."""
|
||||
return random.choice(queries) # noqa: S311 load-test query sampling is not security-sensitive
|
||||
|
||||
|
||||
def percentile(values_sorted: list[float], pct: float) -> float:
|
||||
"""Return the linearly-interpolated percentile of a sorted list."""
|
||||
if not values_sorted:
|
||||
return 0.0
|
||||
rank = (pct / 100) * (len(values_sorted) - 1)
|
||||
low = math.floor(rank)
|
||||
high = math.ceil(rank)
|
||||
if low == high:
|
||||
return values_sorted[low]
|
||||
return values_sorted[low] + (values_sorted[high] - values_sorted[low]) * (rank - low)
|
||||
|
||||
|
||||
def summarize(results: list[RequestResult], wall_seconds: float) -> LoadSummary:
|
||||
"""Aggregate per-request results into a load summary."""
|
||||
latencies = sorted(result.latency_ms for result in results)
|
||||
successes = sum(1 for result in results if result.ok)
|
||||
status_counts: dict[int, int] = {}
|
||||
for result in results:
|
||||
status_counts[result.status_code] = status_counts.get(result.status_code, 0) + 1
|
||||
return LoadSummary(
|
||||
total=len(results),
|
||||
successes=successes,
|
||||
failures=len(results) - successes,
|
||||
wall_seconds=wall_seconds,
|
||||
throughput_rps=len(results) / wall_seconds if wall_seconds > 0 else 0.0,
|
||||
latency_p50_ms=percentile(latencies, 50),
|
||||
latency_p90_ms=percentile(latencies, 90),
|
||||
latency_p95_ms=percentile(latencies, 95),
|
||||
latency_p99_ms=percentile(latencies, 99),
|
||||
latency_mean_ms=statistics.fmean(latencies) if latencies else 0.0,
|
||||
latency_max_ms=latencies[-1] if latencies else 0.0,
|
||||
status_counts=status_counts,
|
||||
)
|
||||
|
||||
|
||||
async def send_search(client: httpx.AsyncClient, query: str, *, rerank: bool) -> RequestResult:
|
||||
"""Send one search request and record its status and latency."""
|
||||
data = {"query": query, "rerank": "true"} if rerank else {"query": query}
|
||||
start = time.perf_counter()
|
||||
try:
|
||||
response = await client.post("/search", data=data)
|
||||
except httpx.HTTPError as error:
|
||||
logger.warning("ebook_loadtest_request_failed error=%s", error)
|
||||
return RequestResult(status_code=0, latency_ms=(time.perf_counter() - start) * 1000, ok=False)
|
||||
return RequestResult(
|
||||
status_code=response.status_code,
|
||||
latency_ms=(time.perf_counter() - start) * 1000,
|
||||
ok=response.is_success,
|
||||
)
|
||||
|
||||
|
||||
async def worker(
|
||||
client: httpx.AsyncClient,
|
||||
queue: asyncio.Queue[str],
|
||||
results: list[RequestResult],
|
||||
*,
|
||||
rerank: bool,
|
||||
) -> None:
|
||||
"""Pull queries off the queue and send requests until it is empty."""
|
||||
while True:
|
||||
try:
|
||||
query = queue.get_nowait()
|
||||
except asyncio.QueueEmpty:
|
||||
return
|
||||
results.append(await send_search(client, query, rerank=rerank))
|
||||
|
||||
|
||||
async def run_load(
|
||||
*,
|
||||
base_url: str,
|
||||
queries: list[str],
|
||||
request_count: int,
|
||||
concurrency: int,
|
||||
rerank: bool,
|
||||
warmup: int,
|
||||
timeout_seconds: float,
|
||||
) -> LoadSummary:
|
||||
"""Run the load test and return its aggregate summary."""
|
||||
limits = httpx.Limits(max_connections=concurrency, max_keepalive_connections=concurrency)
|
||||
async with httpx.AsyncClient(base_url=base_url, timeout=timeout_seconds, limits=limits) as client:
|
||||
for _ in range(warmup):
|
||||
await send_search(client, pick_query(queries), rerank=rerank)
|
||||
|
||||
queue: asyncio.Queue[str] = asyncio.Queue()
|
||||
for _ in range(request_count):
|
||||
queue.put_nowait(pick_query(queries))
|
||||
|
||||
results: list[RequestResult] = []
|
||||
start = time.perf_counter()
|
||||
workers = [asyncio.create_task(worker(client, queue, results, rerank=rerank)) for _ in range(concurrency)]
|
||||
await asyncio.gather(*workers)
|
||||
wall_seconds = time.perf_counter() - start
|
||||
return summarize(results, wall_seconds)
|
||||
|
||||
|
||||
def print_summary(summary: LoadSummary) -> None:
|
||||
"""Print the load summary to stdout."""
|
||||
typer.echo(f"requests={summary.total} successes={summary.successes} failures={summary.failures}")
|
||||
typer.echo(f"wall={summary.wall_seconds:.2f}s throughput={summary.throughput_rps:.1f} req/s")
|
||||
typer.echo(
|
||||
f"latency_ms p50={summary.latency_p50_ms:.1f} p90={summary.latency_p90_ms:.1f} "
|
||||
f"p95={summary.latency_p95_ms:.1f} p99={summary.latency_p99_ms:.1f} "
|
||||
f"mean={summary.latency_mean_ms:.1f} max={summary.latency_max_ms:.1f}"
|
||||
)
|
||||
status_summary = " ".join(f"{code}={count}" for code, count in sorted(summary.status_counts.items()))
|
||||
typer.echo(f"status {status_summary}")
|
||||
|
||||
|
||||
def main(
|
||||
*,
|
||||
base_url: Annotated[str, typer.Option(help="Base URL of the running service")] = "http://127.0.0.1:8070",
|
||||
request_count: Annotated[int, typer.Option("--requests", help="Total requests to send")] = 200,
|
||||
concurrency: Annotated[int, typer.Option(help="Concurrent in-flight requests")] = 10,
|
||||
rerank: Annotated[bool, typer.Option(help="Request server-side reranking")] = False,
|
||||
warmup: Annotated[int, typer.Option(help="Warmup requests, not measured")] = 5,
|
||||
timeout_seconds: Annotated[float, typer.Option("--timeout", help="Per-request timeout seconds")] = 120.0,
|
||||
queries_file: Annotated[str | None, typer.Option(help="Query JSONL file (defaults to the shared set)")] = None,
|
||||
log_level: Annotated[str, typer.Option(help="Log level")] = "WARNING",
|
||||
) -> None:
|
||||
"""Load test the search endpoint and report latency and throughput."""
|
||||
configure_logger(log_level)
|
||||
queries = load_queries(queries_file)
|
||||
logger.info(
|
||||
"ebook_loadtest_start base_url=%s requests=%s concurrency=%s rerank=%s queries=%s",
|
||||
base_url,
|
||||
request_count,
|
||||
concurrency,
|
||||
rerank,
|
||||
len(queries),
|
||||
)
|
||||
summary = asyncio.run(
|
||||
run_load(
|
||||
base_url=base_url,
|
||||
queries=queries,
|
||||
request_count=request_count,
|
||||
concurrency=concurrency,
|
||||
rerank=rerank,
|
||||
warmup=warmup,
|
||||
timeout_seconds=timeout_seconds,
|
||||
)
|
||||
)
|
||||
print_summary(summary)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
typer.run(main)
|
||||
@@ -0,0 +1,132 @@
|
||||
"""vLLM-backed optional reranking."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from python.ebook_search.llm_interface import request_rerank
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from python.ebook_search.config import RerankConfig
|
||||
from python.ebook_search.search import SearchResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RerankResult:
|
||||
"""A relevance score for one candidate chunk."""
|
||||
|
||||
chunk_id: int
|
||||
score: float
|
||||
|
||||
|
||||
def rerank_chunks(query: str, candidates: list[SearchResult], config: RerankConfig) -> list[SearchResult]:
|
||||
"""Rerank candidates with a vLLM rerank endpoint."""
|
||||
if not candidates:
|
||||
return []
|
||||
|
||||
logger.info(
|
||||
"ebook_rerank_request_start base_url=%s model=%s candidates=%s",
|
||||
config.base_url,
|
||||
config.model,
|
||||
len(candidates),
|
||||
)
|
||||
scores = score_candidates(query, candidates, config)
|
||||
results = sorted(
|
||||
(
|
||||
replace(
|
||||
result,
|
||||
score=final_rerank_score(result, scores[result.chunk_id].score, candidates, config),
|
||||
rerank_score=scores[result.chunk_id].score,
|
||||
)
|
||||
for result in candidates
|
||||
),
|
||||
key=lambda result: result.score,
|
||||
reverse=True,
|
||||
)
|
||||
logger.info(
|
||||
"ebook_rerank_request_complete base_url=%s model=%s candidates=%s",
|
||||
config.base_url,
|
||||
config.model,
|
||||
len(results),
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
def score_candidates(
|
||||
query: str,
|
||||
candidates: list[SearchResult],
|
||||
config: RerankConfig,
|
||||
) -> dict[int, RerankResult]:
|
||||
"""Score candidate chunks with the configured rerank API."""
|
||||
body = request_rerank(query, [candidate.text for candidate in candidates], config)
|
||||
if body is None:
|
||||
return zero_rerank_scores(candidates)
|
||||
|
||||
scores = parse_vllm_scores(body, candidates)
|
||||
for result in scores.values():
|
||||
logger.debug("ebook_rerank_candidate_scored chunk_id=%s score=%s", result.chunk_id, result.score)
|
||||
return scores
|
||||
|
||||
|
||||
def parse_vllm_scores(body: object, candidates: list[SearchResult]) -> dict[int, RerankResult]:
|
||||
"""Parse vLLM rerank scores into chunk-id keyed results."""
|
||||
if not isinstance(body, dict):
|
||||
logger.debug("ebook_rerank_response_not_object", extra={"response": body})
|
||||
return zero_rerank_scores(candidates)
|
||||
|
||||
results = body.get("results") or body.get("data")
|
||||
if not isinstance(results, list):
|
||||
logger.debug("ebook_rerank_response_missing_results", extra={"response": body})
|
||||
return zero_rerank_scores(candidates)
|
||||
|
||||
scores = zero_rerank_scores(candidates)
|
||||
for item in results:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
index = item.get("index")
|
||||
score = item.get("relevance_score", item.get("score"))
|
||||
if not isinstance(index, int) or index < 0 or index >= len(candidates):
|
||||
continue
|
||||
if not isinstance(score, int | float):
|
||||
continue
|
||||
chunk_id = candidates[index].chunk_id
|
||||
scores[chunk_id] = RerankResult(chunk_id=chunk_id, score=clamp_score(float(score)))
|
||||
return scores
|
||||
|
||||
|
||||
def zero_rerank_scores(candidates: list[SearchResult]) -> dict[int, RerankResult]:
|
||||
"""Return zero relevance scores for all candidate chunks."""
|
||||
return {candidate.chunk_id: RerankResult(chunk_id=candidate.chunk_id, score=0.0) for candidate in candidates}
|
||||
|
||||
|
||||
def clamp_score(score: float) -> float:
|
||||
"""Clamp a rerank score into the supported 0.0 to 1.0 range."""
|
||||
return min(max(score, 0.0), 1.0)
|
||||
|
||||
|
||||
def final_rerank_score(
|
||||
result: SearchResult,
|
||||
rerank_score: float,
|
||||
candidates: list[SearchResult],
|
||||
config: RerankConfig,
|
||||
) -> float:
|
||||
"""Combine rerank relevance with normalized hybrid retrieval evidence."""
|
||||
return (config.score_weight * rerank_score) + (config.hybrid_weight * normalized_hybrid_score(result, candidates))
|
||||
|
||||
|
||||
def normalized_hybrid_score(result: SearchResult, candidates: list[SearchResult]) -> float:
|
||||
"""Normalize a candidate hybrid score against the rerank candidate set."""
|
||||
hybrid_scores = [
|
||||
candidate.fused_score if candidate.fused_score is not None else candidate.score for candidate in candidates
|
||||
]
|
||||
low = min(hybrid_scores)
|
||||
high = max(hybrid_scores)
|
||||
if high == low:
|
||||
return 1.0
|
||||
|
||||
score = result.fused_score if result.fused_score is not None else result.score
|
||||
return (score - low) / (high - low)
|
||||
@@ -0,0 +1,380 @@
|
||||
"""Hybrid search orchestration."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from pgvector.sqlalchemy import Vector
|
||||
from sqlalchemy import literal, select
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.ebook_search.bm25_corpus import (
|
||||
BM25CorpusUnavailableError,
|
||||
load_bm25_corpus,
|
||||
score_bm25_corpus,
|
||||
)
|
||||
from python.ebook_search.embeddings import MODEL_DIMENSIONS, embed_query, get_embedding_table
|
||||
from python.ebook_search.rerank import rerank_chunks
|
||||
from python.ebook_search.timing import RuntimeStep, timed_result
|
||||
from python.orm.richie import (
|
||||
EbookChapter,
|
||||
EbookChunk,
|
||||
EbookEmbeddingModel,
|
||||
EbookSource,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Mapping
|
||||
|
||||
from sqlalchemy.engine import Engine
|
||||
|
||||
from python.ebook_search.config import EbookSearchConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SearchResult:
|
||||
"""One source chunk returned by search."""
|
||||
|
||||
chunk_id: int
|
||||
text: str
|
||||
source_title: str
|
||||
score: float = 0.0
|
||||
vector_score: float | None = None
|
||||
bm25_score: float | None = None
|
||||
fused_score: float | None = None
|
||||
rerank_score: float | None = None
|
||||
source_author: str | None = None
|
||||
chapter_title: str | None = None
|
||||
page_label: str | None = None
|
||||
rank_source: str = "Hybrid"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SearchResponse:
|
||||
"""Search output for the UI."""
|
||||
|
||||
query: str
|
||||
results: list[SearchResult]
|
||||
rank_label: str
|
||||
timings: tuple[RuntimeStep, ...] = ()
|
||||
|
||||
@property
|
||||
def total_runtime_ms(self) -> float:
|
||||
"""Return total measured runtime for the response."""
|
||||
return sum(step.duration_ms for step in self.timings if step.counts_toward_total)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RetrievalResponse:
|
||||
"""Parallel retrieval output for vector and BM25 candidates."""
|
||||
|
||||
vector_results: list[SearchResult]
|
||||
lexical_results: list[SearchResult]
|
||||
timings: tuple[RuntimeStep, ...]
|
||||
|
||||
|
||||
def search_ebooks(
|
||||
engine: Engine,
|
||||
query: str,
|
||||
config: EbookSearchConfig,
|
||||
*,
|
||||
rerank: bool = False,
|
||||
) -> SearchResponse:
|
||||
"""Run hybrid vector/BM25 search and optional reranking."""
|
||||
if not query.strip():
|
||||
logger.info("ebook_search_empty_query")
|
||||
return SearchResponse(query=query, results=[], rank_label="Hybrid")
|
||||
|
||||
logger.info("ebook_search_start query_length=%s rerank=%s", len(query), rerank)
|
||||
timings: list[RuntimeStep] = []
|
||||
retrieval, timing = timed_result(
|
||||
"Hybrid retrieval",
|
||||
parallel_retrieval,
|
||||
engine,
|
||||
query,
|
||||
config,
|
||||
)
|
||||
timings.extend(retrieval.timings)
|
||||
timings.append(timing)
|
||||
fused, timing = timed_result(
|
||||
"Reciprocal rank fusion",
|
||||
reciprocal_rank_fusion,
|
||||
retrieval.vector_results,
|
||||
retrieval.lexical_results,
|
||||
rank_constant=config.rrf_rank_constant,
|
||||
)
|
||||
timings.append(timing)
|
||||
if config.rerank.enabled and rerank:
|
||||
response, timing = timed_result("Rerank", apply_rerank, query, fused, config)
|
||||
else:
|
||||
response, timing = timed_result("Rerank skipped", skip_rerank, query, fused, config)
|
||||
timings.append(timing)
|
||||
response = replace(response, timings=tuple(timings))
|
||||
logger.info(
|
||||
"ebook_search_complete vector_candidates=%s lexical_candidates=%s "
|
||||
"fused_candidates=%s returned=%s rank_label=%s runtime_ms=%.1f",
|
||||
len(retrieval.vector_results),
|
||||
len(retrieval.lexical_results),
|
||||
len(fused),
|
||||
len(response.results),
|
||||
response.rank_label,
|
||||
response.total_runtime_ms,
|
||||
)
|
||||
return response
|
||||
|
||||
|
||||
def parallel_retrieval(
|
||||
engine: Engine,
|
||||
query: str,
|
||||
config: EbookSearchConfig,
|
||||
) -> RetrievalResponse:
|
||||
"""Run vector and BM25 candidate retrieval concurrently with separate database sessions."""
|
||||
with ThreadPoolExecutor(max_workers=2, thread_name_prefix="ebook-search") as executor:
|
||||
vector_future = executor.submit(
|
||||
timed_result,
|
||||
"Embedding + vector search",
|
||||
vector_candidates,
|
||||
engine,
|
||||
query,
|
||||
config,
|
||||
)
|
||||
bm25_future = executor.submit(
|
||||
timed_result,
|
||||
"BM25 search",
|
||||
bm25_candidates,
|
||||
query,
|
||||
config,
|
||||
)
|
||||
vector_results, vector_timing = vector_future.result()
|
||||
lexical_results, lexical_timing = bm25_future.result()
|
||||
|
||||
logger.info(
|
||||
"ebook_parallel_retrieval_complete vector_candidates=%s lexical_candidates=%s",
|
||||
len(vector_results),
|
||||
len(lexical_results),
|
||||
)
|
||||
return RetrievalResponse(
|
||||
vector_results=vector_results,
|
||||
lexical_results=lexical_results,
|
||||
timings=(
|
||||
replace(vector_timing, counts_toward_total=False),
|
||||
replace(lexical_timing, counts_toward_total=False),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def skip_rerank(
|
||||
query: str,
|
||||
candidates: list[SearchResult],
|
||||
config: EbookSearchConfig,
|
||||
) -> SearchResponse:
|
||||
"""Return fused hybrid results without reranking."""
|
||||
logger.info("ebook_rerank_skipped candidates=%s", len(candidates))
|
||||
return SearchResponse(query=query, results=candidates[: config.top_k], rank_label="Hybrid")
|
||||
|
||||
|
||||
def apply_rerank(
|
||||
query: str,
|
||||
candidates: list[SearchResult],
|
||||
config: EbookSearchConfig,
|
||||
) -> SearchResponse:
|
||||
"""Rerank already-fused hybrid candidates."""
|
||||
reranked = rerank_chunks(query, candidates[: config.rerank.candidates], config.rerank)
|
||||
logger.info(
|
||||
"ebook_rerank_complete input_candidates=%s returned=%s",
|
||||
min(len(candidates), config.rerank.candidates),
|
||||
len(reranked),
|
||||
)
|
||||
return SearchResponse(
|
||||
query=query,
|
||||
results=[replace(result, rank_source="Hybrid + rerank") for result in reranked[: config.top_k]],
|
||||
rank_label="Hybrid + rerank",
|
||||
)
|
||||
|
||||
|
||||
def vector_candidates(engine: Engine, query: str, config: EbookSearchConfig) -> list[SearchResult]:
|
||||
"""Return pgvector cosine candidates for a natural-language query."""
|
||||
with Session(engine) as session:
|
||||
model = session.scalar(select(EbookEmbeddingModel).where(EbookEmbeddingModel.name == config.embedding_model))
|
||||
if model is None:
|
||||
msg = f"Embedding model is not registered: {config.embedding_model}"
|
||||
raise ValueError(msg)
|
||||
|
||||
expected_dimension = MODEL_DIMENSIONS[config.embedding_model]
|
||||
if model.dimension != expected_dimension:
|
||||
msg = f"Model row dimension {model.dimension} does not match configured dimension {expected_dimension}"
|
||||
raise ValueError(msg)
|
||||
|
||||
embedding = embed_query(query, config)
|
||||
limit = max(config.rerank.candidates, config.top_k) * config.vector_candidate_multiplier
|
||||
embedding_table = get_embedding_table(model.dimension)
|
||||
|
||||
embedding_param = literal(embedding, type_=Vector(model.dimension))
|
||||
distance = embedding_table.embedding.op("<=>")(embedding_param)
|
||||
score = (literal(1.0) - distance).label("score")
|
||||
statement = (
|
||||
select(
|
||||
EbookChunk.id.label("chunk_id"),
|
||||
EbookChunk.text.label("text"),
|
||||
EbookSource.title.label("source_title"),
|
||||
EbookSource.author.label("source_author"),
|
||||
EbookChapter.title.label("chapter_title"),
|
||||
EbookChunk.page_label.label("page_label"),
|
||||
score,
|
||||
)
|
||||
.select_from(embedding_table)
|
||||
.join(EbookChunk, EbookChunk.id == embedding_table.chunk_id)
|
||||
.join(EbookSource, EbookSource.id == EbookChunk.source_id)
|
||||
.outerjoin(EbookChapter, EbookChapter.id == EbookChunk.chapter_id)
|
||||
.where(embedding_table.model_id == model.id)
|
||||
.order_by(distance)
|
||||
.limit(limit)
|
||||
)
|
||||
rows = session.execute(statement).mappings()
|
||||
results = [search_result_from_row(row) for row in rows]
|
||||
logger.info(
|
||||
"ebook_vector_search_complete model=%s dimension=%s candidates=%s",
|
||||
config.embedding_model,
|
||||
model.dimension,
|
||||
len(results),
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
def bm25_candidates(query: str, config: EbookSearchConfig) -> list[SearchResult]:
|
||||
"""Return BM25-ranked lexical candidates using the persisted corpus."""
|
||||
try:
|
||||
corpus = load_bm25_corpus(config)
|
||||
except BM25CorpusUnavailableError as error:
|
||||
logger.warning("ebook_bm25_index_unavailable_skipping error=%s", error)
|
||||
return []
|
||||
|
||||
if not corpus.records:
|
||||
logger.info("ebook_bm25_search_complete corpus=0 candidates=0")
|
||||
return []
|
||||
|
||||
bm25_query = retrieval_query_from_text(query)
|
||||
scored_records = score_bm25_corpus(bm25_query, corpus, limit=config.bm25_candidate_limit)
|
||||
results = [
|
||||
replace(search_result_from_row(record), score=score, vector_score=None, bm25_score=score)
|
||||
for record, score in scored_records
|
||||
]
|
||||
|
||||
max_score = results[0].bm25_score if results else 0.0
|
||||
logger.info(
|
||||
"ebook_bm25_search_complete corpus=%s candidates=%s max_score=%.6f",
|
||||
len(corpus.records),
|
||||
len(results),
|
||||
max_score,
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
def reciprocal_rank_fusion(
|
||||
vector_results: list[SearchResult],
|
||||
lexical_results: list[SearchResult],
|
||||
rank_constant: int,
|
||||
) -> list[SearchResult]:
|
||||
"""Fuse vector and lexical rankings with Reciprocal Rank Fusion."""
|
||||
by_chunk: dict[int, SearchResult] = {}
|
||||
scores: defaultdict[int, float] = defaultdict(float)
|
||||
vector_scores: dict[int, float] = {}
|
||||
bm25_scores: dict[int, float] = {}
|
||||
|
||||
for rank, result in enumerate(vector_results, start=1):
|
||||
by_chunk.setdefault(result.chunk_id, result)
|
||||
vector_scores[result.chunk_id] = result.vector_score if result.vector_score is not None else result.score
|
||||
scores[result.chunk_id] += 1 / (rank_constant + rank)
|
||||
|
||||
for rank, result in enumerate(lexical_results, start=1):
|
||||
by_chunk.setdefault(result.chunk_id, result)
|
||||
bm25_scores[result.chunk_id] = result.bm25_score if result.bm25_score is not None else result.score
|
||||
scores[result.chunk_id] += 1 / (rank_constant + rank)
|
||||
|
||||
return sorted(
|
||||
(
|
||||
replace(
|
||||
result,
|
||||
score=scores[result.chunk_id],
|
||||
vector_score=vector_scores.get(result.chunk_id),
|
||||
bm25_score=bm25_scores.get(result.chunk_id),
|
||||
fused_score=scores[result.chunk_id],
|
||||
rank_source="Hybrid",
|
||||
)
|
||||
for result in by_chunk.values()
|
||||
),
|
||||
key=lambda result: result.score,
|
||||
reverse=True,
|
||||
)
|
||||
|
||||
|
||||
def search_result_from_row(row: Mapping[str, object]) -> SearchResult:
|
||||
"""Convert a database row mapping into a search result."""
|
||||
return SearchResult(
|
||||
chunk_id=int(row["chunk_id"]),
|
||||
text=str(row["text"]),
|
||||
source_title=str(row["source_title"]),
|
||||
source_author=optional_str(row["source_author"]),
|
||||
chapter_title=optional_str(row["chapter_title"]),
|
||||
page_label=optional_str(row["page_label"]),
|
||||
score=float(row["score"]) if "score" in row else 0.0,
|
||||
vector_score=float(row["score"]) if "score" in row else None,
|
||||
)
|
||||
|
||||
|
||||
def optional_str(value: object) -> str | None:
|
||||
"""Convert nullable database values to optional strings."""
|
||||
if value is None:
|
||||
return None
|
||||
return str(value)
|
||||
|
||||
|
||||
TOKEN_RE = re.compile(r"[A-Za-z0-9_]+")
|
||||
|
||||
|
||||
def tokens(text_value: str) -> list[str]:
|
||||
"""Extract tokens from a text value.
|
||||
|
||||
This is a simple approximation of the tokenization used by PostgreSQL's full-text search,
|
||||
which is sufficient for BM25 candidate retrieval. It lowercases tokens and includes alphanumeric characters and
|
||||
underscores.
|
||||
"""
|
||||
return [match.group(0).lower() for match in TOKEN_RE.finditer(text_value)]
|
||||
|
||||
|
||||
QUERY_STOP_WORDS = {
|
||||
"a",
|
||||
"an",
|
||||
"and",
|
||||
"are",
|
||||
"as",
|
||||
"at",
|
||||
"does",
|
||||
"for",
|
||||
"in",
|
||||
"is",
|
||||
"of",
|
||||
"the",
|
||||
"to",
|
||||
"what",
|
||||
"when",
|
||||
"where",
|
||||
"which",
|
||||
"who",
|
||||
"why",
|
||||
}
|
||||
|
||||
|
||||
def retrieval_query_from_text(query: str) -> str:
|
||||
"""Remove generic question words while preserving entity and series terms."""
|
||||
keywords = [token for token in tokens(query) if token not in QUERY_STOP_WORDS]
|
||||
if not keywords:
|
||||
return query
|
||||
return " ".join(keywords)
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Runtime timing helpers for EPUB search."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from time import perf_counter
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RuntimeStep:
|
||||
"""Elapsed runtime for one named search step."""
|
||||
|
||||
name: str
|
||||
duration_ms: float
|
||||
counts_toward_total: bool = True
|
||||
|
||||
|
||||
def runtime_step_from_start(name: str, start_seconds: float) -> RuntimeStep:
|
||||
"""Create a runtime step from a prior perf_counter timestamp."""
|
||||
return RuntimeStep(name=name, duration_ms=(perf_counter() - start_seconds) * 1000)
|
||||
|
||||
|
||||
def timed_result[T, **P](
|
||||
name: str,
|
||||
operation: Callable[P, T],
|
||||
*args: P.args,
|
||||
**kwargs: P.kwargs,
|
||||
) -> tuple[T, RuntimeStep]:
|
||||
"""Run an operation and return its result plus elapsed runtime."""
|
||||
start_seconds = perf_counter()
|
||||
result = operation(*args, **kwargs)
|
||||
return result, runtime_step_from_start(name, start_seconds)
|
||||
@@ -0,0 +1 @@
|
||||
"""Detect Nix evaluation warnings from build logs and create PRs with LLM-suggested fixes."""
|
||||
@@ -0,0 +1,449 @@
|
||||
"""Detect Nix evaluation warnings and create PRs with LLM-suggested fixes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import re
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
from zipfile import ZipFile
|
||||
|
||||
import typer
|
||||
from httpx import HTTPError, post
|
||||
|
||||
from python.common import configure_logger
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EvalWarning:
|
||||
"""A single Nix evaluation warning."""
|
||||
|
||||
system: str
|
||||
message: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class FileChange:
|
||||
"""A file change suggested by the LLM."""
|
||||
|
||||
file_path: str
|
||||
original: str
|
||||
fixed: str
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str], *, check: bool = True) -> subprocess.CompletedProcess[str]:
|
||||
"""Run a subprocess command and return the result.
|
||||
|
||||
Args:
|
||||
cmd: Command and arguments.
|
||||
check: Whether to raise on non-zero exit.
|
||||
|
||||
Returns:
|
||||
CompletedProcess with captured stdout/stderr.
|
||||
"""
|
||||
logger.debug("Running: %s", " ".join(cmd))
|
||||
return subprocess.run(cmd, capture_output=True, text=True, check=check)
|
||||
|
||||
|
||||
def download_logs(run_id: str, repo: str) -> dict[str, str]:
|
||||
"""Download build logs for a GitHub Actions run.
|
||||
|
||||
Args:
|
||||
run_id: The workflow run ID.
|
||||
repo: The GitHub repository (owner/repo).
|
||||
|
||||
Returns:
|
||||
Dict mapping zip entry names to their text content, filtered to build log files.
|
||||
|
||||
Raises:
|
||||
RuntimeError: If log download fails.
|
||||
"""
|
||||
result = subprocess.run(
|
||||
["gh", "api", f"repos/{repo}/actions/runs/{run_id}/logs"],
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
msg = f"Failed to download logs: {result.stderr.decode(errors='replace')}"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
logs: dict[str, str] = {}
|
||||
with ZipFile(BytesIO(result.stdout)) as zip_file:
|
||||
for name in zip_file.namelist():
|
||||
if name.startswith("build-") and name.endswith(".txt"):
|
||||
logs[name] = zip_file.read(name).decode(errors="replace")
|
||||
|
||||
return logs
|
||||
|
||||
|
||||
def parse_warnings(logs: dict[str, str]) -> set[EvalWarning]:
|
||||
"""Parse Nix evaluation warnings from build log contents.
|
||||
|
||||
Args:
|
||||
logs: Dict mapping zip entry names (e.g. "build-bob/2_Build.txt") to their text.
|
||||
|
||||
Returns:
|
||||
Deduplicated set of warnings.
|
||||
"""
|
||||
warnings: set[EvalWarning] = set()
|
||||
warning_pattern = re.compile(r"(?:^[\d\-T:.Z]+ )?(warning:|trace: warning:)")
|
||||
timestamp_prefix = re.compile(r"^[\d\-T:.Z]+ ")
|
||||
|
||||
for name, content in sorted(logs.items()):
|
||||
system = name.split("/")[0].removeprefix("build-")
|
||||
for line in content.splitlines():
|
||||
if warning_pattern.search(line):
|
||||
message = timestamp_prefix.sub("", line).strip()
|
||||
if message.startswith("warning: ignoring untrusted flake configuration setting"):
|
||||
continue
|
||||
logger.debug(f"Found warning: {line}")
|
||||
warnings.add(EvalWarning(system=system, message=message))
|
||||
|
||||
logger.info("Found %d unique warnings", len(warnings))
|
||||
return warnings
|
||||
|
||||
|
||||
def extract_referenced_files(warnings: set[EvalWarning]) -> dict[str, str]:
|
||||
"""Extract file paths referenced in warnings and read their contents.
|
||||
|
||||
Args:
|
||||
warnings: List of parsed warnings.
|
||||
|
||||
Returns:
|
||||
Dict mapping repo-relative file paths to their contents.
|
||||
"""
|
||||
paths: set[str] = set()
|
||||
warning_text = "\n".join(w.message for w in warnings)
|
||||
|
||||
nix_store_path = re.compile(r"/nix/store/[^/]+-source/([^:]+\.nix)")
|
||||
for match in nix_store_path.finditer(warning_text):
|
||||
paths.add(match.group(1))
|
||||
|
||||
repo_relative_path = re.compile(r"(?<![/\w])(systems|common|users|overlays)/[^:\s]+\.nix")
|
||||
for match in repo_relative_path.finditer(warning_text):
|
||||
paths.add(match.group(0))
|
||||
|
||||
files: dict[str, str] = {}
|
||||
for path_str in sorted(paths):
|
||||
path = Path(path_str)
|
||||
if path.is_file():
|
||||
files[path_str] = path.read_text()
|
||||
|
||||
if not files and Path("flake.nix").is_file():
|
||||
files["flake.nix"] = Path("flake.nix").read_text()
|
||||
|
||||
logger.info("Extracted %d referenced files", len(files))
|
||||
return files
|
||||
|
||||
|
||||
def compute_warning_hash(warnings: set[EvalWarning]) -> str:
|
||||
"""Compute a short hash of the warning set for deduplication.
|
||||
|
||||
Args:
|
||||
warnings: List of warnings.
|
||||
|
||||
Returns:
|
||||
8-character hex hash.
|
||||
"""
|
||||
text = "\n".join(sorted(f"[{w.system}] {w.message}" for w in warnings))
|
||||
return hashlib.sha256(text.encode()).hexdigest()[:8]
|
||||
|
||||
|
||||
def check_duplicate_pr(warning_hash: str) -> bool:
|
||||
"""Check if an open PR already exists for this warning hash.
|
||||
|
||||
Args:
|
||||
warning_hash: The hash to check.
|
||||
|
||||
Returns:
|
||||
True if a duplicate PR exists.
|
||||
|
||||
Raises:
|
||||
RuntimeError: If the gh CLI call fails.
|
||||
"""
|
||||
result = run_cmd(
|
||||
[
|
||||
"gh",
|
||||
"pr",
|
||||
"list",
|
||||
"--state",
|
||||
"open",
|
||||
"--label",
|
||||
"eval-warning-fix",
|
||||
"--json",
|
||||
"title",
|
||||
"--jq",
|
||||
".[].title",
|
||||
],
|
||||
check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
msg = f"Failed to check for duplicate PRs: {result.stderr}"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
for title in result.stdout.splitlines():
|
||||
if warning_hash in title:
|
||||
logger.info("Duplicate PR found for hash %s", warning_hash)
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def query_ollama(
|
||||
warnings: set[EvalWarning],
|
||||
files: dict[str, str],
|
||||
ollama_url: str,
|
||||
) -> str | None:
|
||||
"""Query Ollama for a fix suggestion.
|
||||
|
||||
Args:
|
||||
warnings: List of warnings.
|
||||
files: Referenced file contents.
|
||||
ollama_url: Ollama API base URL.
|
||||
|
||||
Returns:
|
||||
LLM response text, or None on failure.
|
||||
"""
|
||||
warning_text = "\n".join(f"[{w.system}] {w.message}" for w in warnings)
|
||||
file_context = "\n".join(f"--- FILE: {path} ---\n{content}\n--- END FILE ---" for path, content in files.items())
|
||||
|
||||
prompt = f"""You are a NixOS configuration expert. \
|
||||
Analyze the following Nix evaluation warnings and suggest fixes.
|
||||
|
||||
## Warnings
|
||||
{warning_text}
|
||||
|
||||
## Referenced Files
|
||||
{file_context}
|
||||
|
||||
## Instructions
|
||||
- Identify the root cause of each warning
|
||||
- Provide the exact file changes needed to fix the warnings
|
||||
- Output your response in two clearly separated sections:
|
||||
1. **REASONING**: Brief explanation of what causes each warning and how to fix it
|
||||
2. **CHANGES**: For each file that needs changes, output a block like:
|
||||
FILE: path/to/file.nix
|
||||
<<<<<<< ORIGINAL
|
||||
the original lines to replace
|
||||
=======
|
||||
the replacement lines
|
||||
>>>>>>> FIXED
|
||||
- Only suggest changes for files that exist in the repository
|
||||
- Do not add unnecessary complexity
|
||||
- Preserve the existing code style
|
||||
- If a warning comes from upstream nixpkgs and cannot be fixed in this repo, \
|
||||
say so in REASONING and do not suggest changes"""
|
||||
|
||||
try:
|
||||
response = post(
|
||||
f"{ollama_url}/api/generate",
|
||||
json={
|
||||
"model": "qwen3-coder:30b",
|
||||
"prompt": prompt,
|
||||
"stream": False,
|
||||
"options": {"num_predict": 4096},
|
||||
},
|
||||
timeout=300,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except HTTPError:
|
||||
logger.exception("Ollama request failed")
|
||||
return None
|
||||
|
||||
return response.json().get("response")
|
||||
|
||||
|
||||
def parse_changes(response: str) -> list[FileChange]:
|
||||
"""Parse file changes from the **CHANGES** section of the LLM response.
|
||||
|
||||
Expects blocks in the format:
|
||||
FILE: path/to/file.nix
|
||||
<<<<<<< ORIGINAL
|
||||
...
|
||||
=======
|
||||
...
|
||||
>>>>>>> FIXED
|
||||
|
||||
Args:
|
||||
response: Raw LLM response text.
|
||||
|
||||
Returns:
|
||||
List of parsed file changes.
|
||||
"""
|
||||
if "**CHANGES**" not in response:
|
||||
logger.warning("LLM response missing **CHANGES** section")
|
||||
return []
|
||||
|
||||
changes_section = response.split("**CHANGES**", 1)[1]
|
||||
|
||||
changes: list[FileChange] = []
|
||||
current_file = ""
|
||||
section: str | None = None
|
||||
original_lines: list[str] = []
|
||||
fixed_lines: list[str] = []
|
||||
|
||||
for line in changes_section.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("FILE:"):
|
||||
current_file = stripped.removeprefix("FILE:").strip()
|
||||
elif stripped == "<<<<<<< ORIGINAL":
|
||||
section = "original"
|
||||
original_lines = []
|
||||
elif stripped == "=======" and section == "original":
|
||||
section = "fixed"
|
||||
fixed_lines = []
|
||||
elif stripped == ">>>>>>> FIXED" and section == "fixed":
|
||||
section = None
|
||||
if current_file:
|
||||
changes.append(FileChange(current_file, "\n".join(original_lines), "\n".join(fixed_lines)))
|
||||
elif section == "original":
|
||||
original_lines.append(line)
|
||||
elif section == "fixed":
|
||||
fixed_lines.append(line)
|
||||
|
||||
logger.info("Parsed %d file changes", len(changes))
|
||||
return changes
|
||||
|
||||
|
||||
def apply_changes(changes: list[FileChange]) -> int:
|
||||
"""Apply file changes to the working directory.
|
||||
|
||||
Args:
|
||||
changes: List of changes to apply.
|
||||
|
||||
Returns:
|
||||
Number of changes successfully applied.
|
||||
"""
|
||||
applied = 0
|
||||
cwd = Path.cwd().resolve()
|
||||
for change in changes:
|
||||
path = Path(change.file_path).resolve()
|
||||
if not path.is_relative_to(cwd):
|
||||
logger.warning("Path traversal blocked: %s", change.file_path)
|
||||
continue
|
||||
if not path.is_file():
|
||||
logger.warning("File not found: %s", change.file_path)
|
||||
continue
|
||||
|
||||
content = path.read_text()
|
||||
if change.original not in content:
|
||||
logger.warning("Original text not found in %s", change.file_path)
|
||||
continue
|
||||
|
||||
path.write_text(content.replace(change.original, change.fixed, 1))
|
||||
logger.info("Applied fix to %s", change.file_path)
|
||||
applied += 1
|
||||
|
||||
return applied
|
||||
|
||||
|
||||
def create_pr(
|
||||
warning_hash: str,
|
||||
warnings: set[EvalWarning],
|
||||
llm_response: str,
|
||||
run_url: str,
|
||||
) -> None:
|
||||
"""Create a git branch and PR with the applied fixes.
|
||||
|
||||
Args:
|
||||
warning_hash: Short hash for branch naming and deduplication.
|
||||
warnings: Original warnings for the PR body.
|
||||
llm_response: Full LLM response for extracting reasoning.
|
||||
run_url: URL to the triggering build run.
|
||||
"""
|
||||
branch = f"fix/eval-warning-{warning_hash}"
|
||||
warning_text = "\n".join(f"[{w.system}] {w.message}" for w in warnings)
|
||||
|
||||
if "**REASONING**" not in llm_response:
|
||||
logger.warning("LLM response missing **REASONING** section")
|
||||
reasoning = ""
|
||||
else:
|
||||
_, after = llm_response.split("**REASONING**", 1)
|
||||
reasoning = "\n".join(after.split("**CHANGES**", 1)[0].strip().splitlines()[:50])
|
||||
|
||||
run_cmd(["git", "config", "user.name", "github-actions[bot]"])
|
||||
run_cmd(["git", "config", "user.email", "github-actions[bot]@users.noreply.github.com"])
|
||||
run_cmd(["git", "checkout", "-b", branch])
|
||||
run_cmd(["git", "add", "-A"])
|
||||
|
||||
diff_result = run_cmd(["git", "diff", "--cached", "--quiet"], check=False)
|
||||
if diff_result.returncode == 0:
|
||||
logger.info("No file changes to commit")
|
||||
return
|
||||
|
||||
run_cmd(["git", "commit", "-m", f"fix: resolve nix evaluation warnings ({warning_hash})"])
|
||||
run_cmd(["git", "push", "origin", branch, "--force"])
|
||||
|
||||
body = f"""## Nix Evaluation Warnings
|
||||
|
||||
Detected in [build_systems run]({run_url}):
|
||||
|
||||
```
|
||||
{warning_text}
|
||||
```
|
||||
|
||||
## LLM Analysis (qwen3-coder:30b)
|
||||
|
||||
{reasoning}
|
||||
|
||||
---
|
||||
*Auto-generated by fix_eval_warnings. Review carefully before merging.*"""
|
||||
|
||||
run_cmd(
|
||||
[
|
||||
"gh",
|
||||
"pr",
|
||||
"create",
|
||||
"--title",
|
||||
f"fix: resolve nix eval warnings ({warning_hash})",
|
||||
"--label",
|
||||
"automated",
|
||||
"--label",
|
||||
"eval-warning-fix",
|
||||
"--body",
|
||||
body,
|
||||
]
|
||||
)
|
||||
logger.info("PR created on branch %s", branch)
|
||||
|
||||
|
||||
def main(
|
||||
run_id: Annotated[str, typer.Option("--run-id", help="GitHub Actions run ID")],
|
||||
repo: Annotated[str, typer.Option("--repo", help="GitHub repository (owner/repo)")],
|
||||
ollama_url: Annotated[str, typer.Option("--ollama-url", help="Ollama API base URL")],
|
||||
run_url: Annotated[str, typer.Option("--run-url", help="URL to the triggering build run")],
|
||||
log_level: Annotated[str, typer.Option("--log-level", "-l", help="Log level")] = "INFO",
|
||||
) -> None:
|
||||
"""Detect Nix evaluation warnings and create PRs with LLM-suggested fixes."""
|
||||
configure_logger(log_level)
|
||||
|
||||
logs = download_logs(run_id, repo)
|
||||
warnings = parse_warnings(logs)
|
||||
if not warnings:
|
||||
return
|
||||
|
||||
warning_hash = compute_warning_hash(warnings)
|
||||
if check_duplicate_pr(warning_hash):
|
||||
return
|
||||
|
||||
files = extract_referenced_files(warnings)
|
||||
llm_response = query_ollama(warnings, files, ollama_url)
|
||||
if not llm_response:
|
||||
return
|
||||
|
||||
changes = parse_changes(llm_response)
|
||||
applied = apply_changes(changes)
|
||||
if applied == 0:
|
||||
logger.info("No changes could be applied")
|
||||
return
|
||||
|
||||
create_pr(warning_hash, warnings, llm_response, run_url)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
typer.run(main)
|
||||
@@ -0,0 +1,6 @@
|
||||
"""Reusable FastAPI tools."""
|
||||
|
||||
from python.fastapi_tools.db import DbSession, get_db
|
||||
from python.fastapi_tools.zstd_middleware import ZstdMiddleware
|
||||
|
||||
__all__ = ["DbSession", "ZstdMiddleware", "get_db"]
|
||||
@@ -0,0 +1,20 @@
|
||||
"""FastAPI dependencies."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Annotated
|
||||
|
||||
from fastapi import Depends, Request
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterator
|
||||
|
||||
|
||||
def get_db(request: Request) -> Iterator[Session]:
|
||||
"""Get database session from app state."""
|
||||
with Session(request.app.state.engine) as session:
|
||||
yield session
|
||||
|
||||
|
||||
DbSession = Annotated[Session, Depends(get_db)]
|
||||
@@ -0,0 +1,53 @@
|
||||
"""Zstd response compression middleware."""
|
||||
|
||||
from compression import zstd
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from starlette.middleware.base import BaseHTTPMiddleware, RequestResponseEndpoint
|
||||
from starlette.responses import Response
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from starlette.requests import Request
|
||||
|
||||
MINIMUM_RESPONSE_SIZE = 500
|
||||
|
||||
|
||||
class ZstdMiddleware(BaseHTTPMiddleware):
|
||||
"""Middleware that compresses responses with zstd when the client supports it."""
|
||||
|
||||
async def dispatch(self, request: Request, call_next: RequestResponseEndpoint) -> Response:
|
||||
"""Compress the response with zstd if the client accepts it."""
|
||||
accepted_encodings = request.headers.get("accept-encoding", "")
|
||||
if "zstd" not in accepted_encodings:
|
||||
return await call_next(request)
|
||||
|
||||
response = await call_next(request)
|
||||
|
||||
if response.headers.get("content-encoding") or "text/event-stream" in response.headers.get("content-type", ""):
|
||||
return response
|
||||
|
||||
body = b""
|
||||
async for chunk in response.body_iterator:
|
||||
body += chunk if isinstance(chunk, bytes) else chunk.encode()
|
||||
|
||||
if len(body) < MINIMUM_RESPONSE_SIZE:
|
||||
return Response(
|
||||
content=body,
|
||||
status_code=response.status_code,
|
||||
headers=dict(response.headers),
|
||||
media_type=response.media_type,
|
||||
)
|
||||
|
||||
compressed = zstd.compress(body)
|
||||
|
||||
headers = dict(response.headers)
|
||||
headers["content-encoding"] = "zstd"
|
||||
headers["content-length"] = str(len(compressed))
|
||||
headers.pop("transfer-encoding", None)
|
||||
|
||||
return Response(
|
||||
content=compressed,
|
||||
status_code=response.status_code,
|
||||
headers=headers,
|
||||
media_type=response.media_type,
|
||||
)
|
||||
+347
@@ -0,0 +1,347 @@
|
||||
"""Small Gitea API client for repository automation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Self
|
||||
from urllib.parse import quote
|
||||
|
||||
import httpx
|
||||
|
||||
DEFAULT_PAGE_SIZE = 100
|
||||
EXPECTED_NO_CONTENT = 204
|
||||
EXPECTED_CREATED = 201
|
||||
EXPECTED_OK = 200
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CreatedIssue:
|
||||
"""Issue data returned by Gitea."""
|
||||
|
||||
number: int | None
|
||||
html_url: str | None
|
||||
title: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PullRequest:
|
||||
"""Pull request data returned by Gitea."""
|
||||
|
||||
number: int
|
||||
title: str
|
||||
html_url: str | None
|
||||
labels: tuple[str, ...]
|
||||
head_branch: str | None
|
||||
base_branch: str | None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class WorkflowJob:
|
||||
"""Workflow job data returned by Gitea Actions."""
|
||||
|
||||
id: int
|
||||
name: str
|
||||
run_id: int | None
|
||||
status: str | None
|
||||
conclusion: str | None
|
||||
|
||||
|
||||
class GiteaError(RuntimeError):
|
||||
"""Raised when Gitea rejects an API request."""
|
||||
|
||||
|
||||
def split_repo_name(repo: str) -> tuple[str, str]:
|
||||
"""Split an owner/repo string into its parts."""
|
||||
owner, separator, repo_name = repo.partition("/")
|
||||
if not separator or not owner or not repo_name:
|
||||
msg = f"Invalid repository name: {repo}"
|
||||
raise ValueError(msg)
|
||||
return owner, repo_name
|
||||
|
||||
|
||||
class GiteaClient:
|
||||
"""HTTP client for the subset of Gitea APIs used in this repository."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
base_url: str,
|
||||
token: str,
|
||||
timeout: int = 30,
|
||||
transport: httpx.BaseTransport | None = None,
|
||||
) -> None:
|
||||
"""Initialize the Gitea client."""
|
||||
self._client = httpx.Client(
|
||||
base_url=base_url.rstrip("/"),
|
||||
timeout=timeout,
|
||||
headers={"Authorization": f"token {token}"},
|
||||
transport=transport,
|
||||
)
|
||||
|
||||
def create_issue(
|
||||
self,
|
||||
*,
|
||||
owner: str,
|
||||
repo: str,
|
||||
title: str,
|
||||
body: str,
|
||||
labels: list[int] | None = None,
|
||||
) -> CreatedIssue:
|
||||
"""Create a Gitea issue."""
|
||||
payload: dict[str, object] = {"title": title, "body": body, "labels": labels or []}
|
||||
response = self._request(
|
||||
"POST",
|
||||
f"/api/v1/repos/{owner}/{repo}/issues",
|
||||
expected_statuses={EXPECTED_CREATED},
|
||||
json=payload,
|
||||
)
|
||||
data = response.json()
|
||||
return CreatedIssue(
|
||||
number=_optional_int(data.get("number")),
|
||||
html_url=_optional_str(data.get("html_url")),
|
||||
title=str(data.get("title", title)),
|
||||
)
|
||||
|
||||
def resolve_label_ids(self, *, owner: str, repo: str, labels: list[str]) -> list[int]:
|
||||
"""Resolve label names to Gitea label IDs."""
|
||||
if not labels:
|
||||
return []
|
||||
|
||||
available_labels: dict[str, int] = {}
|
||||
page = 1
|
||||
while True:
|
||||
response = self._request(
|
||||
"GET",
|
||||
f"/api/v1/repos/{owner}/{repo}/labels",
|
||||
params={"page": page, "limit": DEFAULT_PAGE_SIZE},
|
||||
)
|
||||
batch = response.json()
|
||||
if not batch:
|
||||
break
|
||||
for label in batch:
|
||||
label_name = str(label.get("name", ""))
|
||||
label_id = _optional_int(label.get("id"))
|
||||
if label_name and label_id is not None:
|
||||
available_labels[label_name] = label_id
|
||||
if len(batch) < DEFAULT_PAGE_SIZE:
|
||||
break
|
||||
page += 1
|
||||
|
||||
missing = [label for label in labels if label not in available_labels]
|
||||
if missing:
|
||||
missing_names = ", ".join(sorted(missing))
|
||||
msg = f"Missing Gitea labels: {missing_names}"
|
||||
raise GiteaError(msg)
|
||||
|
||||
return [available_labels[label] for label in labels]
|
||||
|
||||
def list_open_pull_requests(
|
||||
self,
|
||||
*,
|
||||
owner: str,
|
||||
repo: str,
|
||||
labels: list[str] | None = None,
|
||||
head: str | None = None,
|
||||
) -> list[PullRequest]:
|
||||
"""List open pull requests for a repository."""
|
||||
expected_labels = set(labels or [])
|
||||
pull_requests: list[PullRequest] = []
|
||||
page = 1
|
||||
while True:
|
||||
response = self._request(
|
||||
"GET",
|
||||
f"/api/v1/repos/{owner}/{repo}/pulls",
|
||||
params={"state": "open", "page": page, "limit": DEFAULT_PAGE_SIZE},
|
||||
)
|
||||
batch = response.json()
|
||||
if not batch:
|
||||
break
|
||||
|
||||
for item in batch:
|
||||
pull_request = _pull_request_from_api(item)
|
||||
if head and pull_request.head_branch != head:
|
||||
continue
|
||||
if expected_labels and not expected_labels.issubset(set(pull_request.labels)):
|
||||
continue
|
||||
pull_requests.append(pull_request)
|
||||
|
||||
if len(batch) < DEFAULT_PAGE_SIZE:
|
||||
break
|
||||
page += 1
|
||||
|
||||
return pull_requests
|
||||
|
||||
def create_pull_request(
|
||||
self,
|
||||
*,
|
||||
owner: str,
|
||||
repo: str,
|
||||
title: str,
|
||||
body: str,
|
||||
head: str,
|
||||
base: str,
|
||||
labels: list[str] | None = None,
|
||||
) -> PullRequest:
|
||||
"""Create a pull request."""
|
||||
payload: dict[str, object] = {
|
||||
"title": title,
|
||||
"body": body,
|
||||
"head": head,
|
||||
"base": base,
|
||||
}
|
||||
if labels:
|
||||
payload["labels"] = self.resolve_label_ids(owner=owner, repo=repo, labels=labels)
|
||||
|
||||
response = self._request(
|
||||
"POST",
|
||||
f"/api/v1/repos/{owner}/{repo}/pulls",
|
||||
expected_statuses={EXPECTED_CREATED},
|
||||
json=payload,
|
||||
)
|
||||
return _pull_request_from_api(response.json())
|
||||
|
||||
def merge_pull_request(
|
||||
self,
|
||||
*,
|
||||
owner: str,
|
||||
repo: str,
|
||||
number: int,
|
||||
merge_method: str = "rebase",
|
||||
head_commit_id: str | None = None,
|
||||
delete_branch_after_merge: bool = False,
|
||||
) -> None:
|
||||
"""Merge a pull request."""
|
||||
payload: dict[str, object] = {
|
||||
"Do": merge_method,
|
||||
"delete_branch_after_merge": delete_branch_after_merge,
|
||||
}
|
||||
if head_commit_id:
|
||||
payload["head_commit_id"] = head_commit_id
|
||||
|
||||
self._request(
|
||||
"POST",
|
||||
f"/api/v1/repos/{owner}/{repo}/pulls/{number}/merge",
|
||||
json=payload,
|
||||
)
|
||||
|
||||
def dispatch_workflow(self, *, owner: str, repo: str, workflow_id: str, ref: str) -> None:
|
||||
"""Trigger a workflow_dispatch run."""
|
||||
workflow_path = quote(workflow_id, safe="")
|
||||
self._request(
|
||||
"POST",
|
||||
f"/api/v1/repos/{owner}/{repo}/actions/workflows/{workflow_path}/dispatches",
|
||||
expected_statuses={EXPECTED_OK, EXPECTED_NO_CONTENT},
|
||||
json={"ref": ref},
|
||||
)
|
||||
|
||||
def list_run_jobs(self, *, owner: str, repo: str, run_id: str | int) -> list[WorkflowJob]:
|
||||
"""List workflow jobs for a specific run."""
|
||||
jobs: list[WorkflowJob] = []
|
||||
page = 1
|
||||
while True:
|
||||
response = self._request(
|
||||
"GET",
|
||||
f"/api/v1/repos/{owner}/{repo}/actions/jobs",
|
||||
params={"page": page, "limit": DEFAULT_PAGE_SIZE},
|
||||
)
|
||||
payload = response.json()
|
||||
batch = payload.get("jobs", [])
|
||||
if not batch:
|
||||
break
|
||||
|
||||
for item in batch:
|
||||
if str(item.get("run_id")) != str(run_id):
|
||||
continue
|
||||
jobs.append(_workflow_job_from_api(item))
|
||||
|
||||
if len(batch) < DEFAULT_PAGE_SIZE:
|
||||
break
|
||||
page += 1
|
||||
|
||||
return jobs
|
||||
|
||||
def download_job_logs(self, *, owner: str, repo: str, job_id: int) -> str:
|
||||
"""Download logs for a workflow job."""
|
||||
response = self._request(
|
||||
"GET",
|
||||
f"/api/v1/repos/{owner}/{repo}/actions/jobs/{job_id}/logs",
|
||||
)
|
||||
return response.text
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close the underlying HTTP client."""
|
||||
self._client.close()
|
||||
|
||||
def __enter__(self) -> Self:
|
||||
"""Enter the context manager."""
|
||||
return self
|
||||
|
||||
def __exit__(self, *args: object) -> None:
|
||||
"""Close the HTTP client."""
|
||||
self.close()
|
||||
|
||||
def _request(
|
||||
self,
|
||||
method: str,
|
||||
path: str,
|
||||
*,
|
||||
expected_statuses: set[int] | None = None,
|
||||
**kwargs: object,
|
||||
) -> httpx.Response:
|
||||
"""Send an HTTP request and validate the response status."""
|
||||
response = self._client.request(method, path, **kwargs)
|
||||
statuses = expected_statuses or {EXPECTED_OK}
|
||||
if response.status_code not in statuses:
|
||||
msg = f"Gitea request failed ({response.status_code}): {response.text}"
|
||||
raise GiteaError(msg)
|
||||
return response
|
||||
|
||||
|
||||
def _pull_request_from_api(data: dict[str, object]) -> PullRequest:
|
||||
"""Convert Gitea API pull-request data into a dataclass."""
|
||||
number = _optional_int(data.get("number")) or _optional_int(data.get("index"))
|
||||
if number is None:
|
||||
msg = "Gitea pull request payload is missing a number"
|
||||
raise GiteaError(msg)
|
||||
|
||||
labels = tuple(str(label.get("name", "")) for label in data.get("labels", []))
|
||||
head = data.get("head", {})
|
||||
base = data.get("base", {})
|
||||
return PullRequest(
|
||||
number=number,
|
||||
title=str(data.get("title", "")),
|
||||
html_url=_optional_str(data.get("html_url")),
|
||||
labels=tuple(label for label in labels if label),
|
||||
head_branch=_optional_str(head.get("ref")) or _optional_str(data.get("head_branch")),
|
||||
base_branch=_optional_str(base.get("ref")) or _optional_str(data.get("base_branch")),
|
||||
)
|
||||
|
||||
|
||||
def _workflow_job_from_api(data: dict[str, object]) -> WorkflowJob:
|
||||
"""Convert Gitea API workflow-job data into a dataclass."""
|
||||
job_id = _optional_int(data.get("id"))
|
||||
if job_id is None:
|
||||
msg = "Gitea workflow job payload is missing an ID"
|
||||
raise GiteaError(msg)
|
||||
|
||||
return WorkflowJob(
|
||||
id=job_id,
|
||||
name=str(data.get("name", "")),
|
||||
run_id=_optional_int(data.get("run_id")),
|
||||
status=_optional_str(data.get("status")),
|
||||
conclusion=_optional_str(data.get("conclusion")),
|
||||
)
|
||||
|
||||
|
||||
def _optional_int(value: object) -> int | None:
|
||||
"""Convert an API value to an integer when present."""
|
||||
if value is None:
|
||||
return None
|
||||
return int(value)
|
||||
|
||||
|
||||
def _optional_str(value: object) -> str | None:
|
||||
"""Convert an API value to a string when present."""
|
||||
if value is None:
|
||||
return None
|
||||
return str(value)
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Automation helpers for flake.lock pull requests on Gitea."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
from os import getenv
|
||||
from typing import Annotated
|
||||
|
||||
import typer
|
||||
|
||||
from python.gitea import GiteaClient, PullRequest, split_repo_name
|
||||
|
||||
DEFAULT_BASE_BRANCH = "main"
|
||||
DEFAULT_BRANCH = "automation/update-flake-lock"
|
||||
DEFAULT_GITEA_URL = "https://gitea.tmmworkshop.com"
|
||||
PR_LABELS = ["dependencies", "automated", "flake_lock_update"]
|
||||
PR_CHECK_WORKFLOWS = ["build_systems.yml", "treefmt.yml", "pytest.yml"]
|
||||
PR_TITLE = "Update flake.lock"
|
||||
PR_BODY = "Automated flake.lock update."
|
||||
|
||||
app = typer.Typer(add_completion=False)
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str], *, check: bool = True) -> subprocess.CompletedProcess[str]:
|
||||
"""Run a subprocess command."""
|
||||
return subprocess.run(cmd, capture_output=True, text=True, check=check)
|
||||
|
||||
|
||||
def ensure_flake_lock_pull_request(
|
||||
client: GiteaClient,
|
||||
*,
|
||||
owner: str,
|
||||
repo: str,
|
||||
branch: str,
|
||||
base: str,
|
||||
) -> PullRequest:
|
||||
"""Return an existing flake.lock PR for the branch or create one."""
|
||||
pull_requests = client.list_open_pull_requests(owner=owner, repo=repo, head=branch)
|
||||
if pull_requests:
|
||||
return pull_requests[0]
|
||||
|
||||
return client.create_pull_request(
|
||||
owner=owner,
|
||||
repo=repo,
|
||||
title=PR_TITLE,
|
||||
body=PR_BODY,
|
||||
head=branch,
|
||||
base=base,
|
||||
labels=PR_LABELS,
|
||||
)
|
||||
|
||||
|
||||
def find_flake_lock_pull_request(client: GiteaClient, *, owner: str, repo: str) -> PullRequest | None:
|
||||
"""Find the first open flake.lock pull request."""
|
||||
pull_requests = client.list_open_pull_requests(owner=owner, repo=repo, labels=["flake_lock_update"])
|
||||
if not pull_requests:
|
||||
return None
|
||||
return pull_requests[0]
|
||||
|
||||
|
||||
def dispatch_pull_request_checks(client: GiteaClient, *, owner: str, repo: str, branch: str) -> None:
|
||||
"""Dispatch the workflows that normally run for pull requests."""
|
||||
for workflow in PR_CHECK_WORKFLOWS:
|
||||
client.dispatch_workflow(owner=owner, repo=repo, workflow_id=workflow, ref=branch)
|
||||
|
||||
|
||||
def has_worktree_changes() -> bool:
|
||||
"""Return whether `flake.lock` has worktree changes."""
|
||||
result = run_cmd(["git", "diff", "--quiet", "--", "flake.lock"], check=False)
|
||||
return result.returncode != 0
|
||||
|
||||
|
||||
def commit_flake_lock_update(*, branch: str) -> None:
|
||||
"""Commit the updated lock file to the automation branch."""
|
||||
run_cmd(["git", "config", "user.name", "gitea-actions[bot]"])
|
||||
run_cmd(["git", "config", "user.email", "gitea-actions@tmmworkshop.com"])
|
||||
run_cmd(["git", "checkout", "-B", branch])
|
||||
run_cmd(["git", "add", "flake.lock"])
|
||||
run_cmd(["git", "commit", "-m", "chore: update flake.lock"])
|
||||
|
||||
|
||||
def push_branch(*, branch: str) -> None:
|
||||
"""Push the automation branch to origin."""
|
||||
run_cmd(["git", "push", "origin", f"HEAD:{branch}", "--force"])
|
||||
|
||||
|
||||
def _required_gitea_token() -> str:
|
||||
"""Read the required Gitea token from the environment."""
|
||||
token = getenv("GITEA_TOKEN")
|
||||
if token:
|
||||
return token
|
||||
|
||||
msg = "GITEA_TOKEN environment variable is required"
|
||||
raise RuntimeError(msg)
|
||||
|
||||
|
||||
@app.command()
|
||||
def update(
|
||||
repo: Annotated[str, typer.Option("--repo", help="Gitea repository in owner/repo form")],
|
||||
base: Annotated[str, typer.Option("--base", help="Base branch")] = DEFAULT_BASE_BRANCH,
|
||||
branch: Annotated[str, typer.Option("--branch", help="Automation branch")] = DEFAULT_BRANCH,
|
||||
) -> None:
|
||||
"""Commit flake.lock changes and ensure a pull request exists."""
|
||||
if not has_worktree_changes():
|
||||
typer.echo("No flake.lock changes detected")
|
||||
return
|
||||
|
||||
commit_flake_lock_update(branch=branch)
|
||||
push_branch(branch=branch)
|
||||
|
||||
owner, repo_name = split_repo_name(repo)
|
||||
with GiteaClient(
|
||||
base_url=getenv("GITEA_URL", DEFAULT_GITEA_URL),
|
||||
token=_required_gitea_token(),
|
||||
) as client:
|
||||
pull_request = ensure_flake_lock_pull_request(
|
||||
client,
|
||||
owner=owner,
|
||||
repo=repo_name,
|
||||
branch=branch,
|
||||
base=base,
|
||||
)
|
||||
# We can remove this if Gitea fixes the following issue:
|
||||
# https://github.com/go-gitea/gitea/issues/33963
|
||||
dispatch_pull_request_checks(client, owner=owner, repo=repo_name, branch=branch)
|
||||
typer.echo(pull_request.html_url or f"Pull request #{pull_request.number}")
|
||||
|
||||
|
||||
@app.command()
|
||||
def merge(
|
||||
repo: Annotated[str, typer.Option("--repo", help="Gitea repository in owner/repo form")],
|
||||
) -> None:
|
||||
"""Merge the first open flake.lock pull request."""
|
||||
owner, repo_name = split_repo_name(repo)
|
||||
with GiteaClient(
|
||||
base_url=getenv("GITEA_URL", DEFAULT_GITEA_URL),
|
||||
token=_required_gitea_token(),
|
||||
) as client:
|
||||
pull_request = find_flake_lock_pull_request(client, owner=owner, repo=repo_name)
|
||||
if not pull_request:
|
||||
typer.echo("No open PR found with label flake_lock_update")
|
||||
return
|
||||
client.merge_pull_request(owner=owner, repo=repo_name, number=pull_request.number, merge_method="rebase")
|
||||
typer.echo(f"Merged PR #{pull_request.number}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
@@ -0,0 +1 @@
|
||||
"""Load HAProxy ``option httplog`` lines into SQLite and query them."""
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Command-line interface: load HAProxy logs into the Richie database.
|
||||
|
||||
The table schema is managed with alembic (``database richie upgrade head``); this
|
||||
command only inserts rows.
|
||||
|
||||
Examples:
|
||||
# stream the live log into the database (commit every line)
|
||||
journalctl -u haproxy -o cat -f | haproxy-logs ingest --batch-size 1
|
||||
|
||||
# backfill from a saved log file
|
||||
haproxy-logs ingest --file /tmp/haproxy.log
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Annotated
|
||||
|
||||
import typer
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from python.haproxy_logs.ingest import ingest_lines
|
||||
from python.orm.common import get_postgres_engine
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterable
|
||||
|
||||
app = typer.Typer(help="Load HAProxy logs into the Richie database.", no_args_is_help=True)
|
||||
|
||||
|
||||
@app.command()
|
||||
def ingest(
|
||||
file: Annotated[str | None, typer.Option(help="Read lines from a file instead of stdin.")] = None,
|
||||
batch_size: Annotated[int, typer.Option(help="Rows per commit; use 1 when tailing a live log.")] = 100,
|
||||
) -> None:
|
||||
"""Parse HAProxy log lines from stdin (or a file) and store them in the Richie DB."""
|
||||
engine = get_postgres_engine(name="RICHIE")
|
||||
with Session(engine) as session:
|
||||
result = ingest_lines(_read_lines(file), session, batch_size=batch_size)
|
||||
typer.echo(f"inserted={result.inserted} duplicates={result.duplicates} skipped={result.skipped}")
|
||||
|
||||
|
||||
def _read_lines(file: str | None) -> Iterable[str]:
|
||||
"""Yield log lines from a file, or from stdin when no file is given."""
|
||||
if file is None:
|
||||
yield from sys.stdin
|
||||
return
|
||||
with Path(file).open(encoding="utf-8", errors="replace") as handle:
|
||||
yield from handle
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Parse HAProxy log lines and persist them to the database.
|
||||
|
||||
Ingestion is idempotent: each row carries a ``line_hash`` (a unique column), and
|
||||
this module skips lines whose hash already exists, so the same logs can be
|
||||
re-ingested without creating duplicate records.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from sqlalchemy import select
|
||||
|
||||
from python.haproxy_logs.parser import parse_line
|
||||
from python.orm.richie.haproxy import HaproxyRequest
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Iterable
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class IngestResult:
|
||||
"""Summary counts for an ingest run."""
|
||||
|
||||
inserted: int
|
||||
skipped: int
|
||||
duplicates: int
|
||||
|
||||
|
||||
def ingest_lines(lines: Iterable[str], session: Session, *, batch_size: int = 100) -> IngestResult:
|
||||
"""Parse log lines and insert new ones into the database in batches.
|
||||
|
||||
Lines already present (same ``line_hash``) are dropped, as are duplicate
|
||||
lines within the same run. Unparseable non-blank lines are counted as
|
||||
``skipped`` rather than raising, so a stray startup message never aborts a
|
||||
streaming ingest.
|
||||
|
||||
Args:
|
||||
lines: Iterable of raw log lines (with or without a journald prefix).
|
||||
session: Database session to insert into.
|
||||
batch_size: Number of distinct rows to buffer before committing. Use
|
||||
``1`` when tailing a live log so rows land immediately.
|
||||
|
||||
Returns:
|
||||
Counts of inserted, skipped (unparseable) and duplicate lines.
|
||||
"""
|
||||
inserted = 0
|
||||
skipped = 0
|
||||
parsed_count = 0
|
||||
batch: dict[str, HaproxyRequest] = {}
|
||||
|
||||
for line in lines:
|
||||
parsed = parse_line(line)
|
||||
if parsed is None:
|
||||
if line.strip():
|
||||
skipped += 1
|
||||
continue
|
||||
parsed_count += 1
|
||||
batch[parsed["line_hash"]] = HaproxyRequest(**parsed)
|
||||
if len(batch) >= batch_size:
|
||||
inserted += _flush(batch, session)
|
||||
|
||||
inserted += _flush(batch, session)
|
||||
return IngestResult(inserted=inserted, skipped=skipped, duplicates=parsed_count - inserted)
|
||||
|
||||
|
||||
def _flush(batch: dict[str, HaproxyRequest], session: Session) -> int:
|
||||
"""Insert the rows whose hash is not already stored, then clear the batch.
|
||||
|
||||
Returns the number of rows actually written.
|
||||
"""
|
||||
if not batch:
|
||||
return 0
|
||||
|
||||
existing = set(
|
||||
session.scalars(select(HaproxyRequest.line_hash).where(HaproxyRequest.line_hash.in_(batch.keys()))),
|
||||
)
|
||||
new_rows = [row for line_hash, row in batch.items() if line_hash not in existing]
|
||||
batch.clear()
|
||||
|
||||
if not new_rows:
|
||||
return 0
|
||||
session.add_all(new_rows)
|
||||
session.commit()
|
||||
return len(new_rows)
|
||||
@@ -0,0 +1,132 @@
|
||||
"""Parse HAProxy ``option httplog`` lines into column mappings.
|
||||
|
||||
The expected format (with the request-header capture this project configures) is::
|
||||
|
||||
<client_ip>:<port> [<accept_date>] <frontend> <backend>/<server>
|
||||
<TR>/<Tw>/<Tc>/<Tr>/<Ta> <status> <bytes> <req_cookie> <resp_cookie>
|
||||
<term_state> <ac>/<fc>/<bc>/<sc>/<rc> <srv_q>/<back_q>
|
||||
{<host>|<user_agent>} "<method> <target> <version>"
|
||||
|
||||
Lines may still carry a systemd-journal prefix (``... haproxy[123]: ``); it is
|
||||
stripped before parsing. Lines that are not request logs (startup messages,
|
||||
health checks, ...) return ``None``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
|
||||
# Strips an optional journal prefix such as "Jun 22 20:17:46 jeeves haproxy[688739]: ".
|
||||
_JOURNAL_PREFIX = re.compile(r"^.*?haproxy(?:\[\d+\])?:\s+")
|
||||
|
||||
_LOG_LINE = re.compile(
|
||||
r"""
|
||||
^
|
||||
(?P<client_ip>[0-9a-fA-F:.]+):(?P<client_port>\d+)\s+
|
||||
\[(?P<accept_date>[^\]]+)\]\s+
|
||||
(?P<frontend>\S+)\s+
|
||||
(?P<backend>[^/\s]+)/(?P<server>\S+)\s+
|
||||
(?P<time_request>-?\d+)/(?P<time_queue>-?\d+)/(?P<time_connect>-?\d+)/
|
||||
(?P<time_response>-?\d+)/(?P<time_total>-?\d+)\s+
|
||||
(?P<status_code>-?\d+)\s+
|
||||
(?P<bytes_read>\d+)\s+
|
||||
\S+\s+\S+\s+ # captured request/response cookies
|
||||
(?P<termination_state>\S+)\s+
|
||||
(?P<active_connections>\d+)/(?P<frontend_connections>\d+)/
|
||||
(?P<backend_connections>\d+)/(?P<server_connections>\d+)/(?P<retries>\d+)\s+
|
||||
(?P<server_queue>\d+)/(?P<backend_queue>\d+)\s+
|
||||
(?P<captures>(?:\{[^}]*\}\s+)*)
|
||||
"(?P<request>[^"]*)"
|
||||
\s*$
|
||||
""",
|
||||
re.VERBOSE,
|
||||
)
|
||||
|
||||
_ACCEPT_DATE_FORMAT = "%d/%b/%Y:%H:%M:%S.%f"
|
||||
_HEADER_CAPTURE = re.compile(r"\{([^}]*)\}")
|
||||
|
||||
|
||||
def parse_line(line: str) -> dict[str, Any] | None:
|
||||
"""Parse one HAProxy http-log line into a ``HaproxyRequest`` column mapping.
|
||||
|
||||
Args:
|
||||
line: A raw log line, optionally still carrying a journald prefix.
|
||||
|
||||
Returns:
|
||||
A mapping of column name to value, or ``None`` if the line is not a
|
||||
request log (blank lines, startup messages, aborted health checks, ...).
|
||||
"""
|
||||
stripped = _JOURNAL_PREFIX.sub("", line.strip())
|
||||
match = _LOG_LINE.match(stripped)
|
||||
if match is None:
|
||||
return None
|
||||
|
||||
groups = match.groupdict()
|
||||
frontend = groups["frontend"]
|
||||
is_ssl = frontend.endswith("~")
|
||||
host, user_agent = _split_request_headers(groups["captures"])
|
||||
method, target, http_version = _split_request(groups["request"])
|
||||
path, _, query = target.partition("?")
|
||||
|
||||
return {
|
||||
"line_hash": hashlib.sha256(stripped.encode("utf-8")).hexdigest(),
|
||||
"requested_at": _parse_accept_date(groups["accept_date"]),
|
||||
"client_ip": groups["client_ip"],
|
||||
"client_port": int(groups["client_port"]),
|
||||
"frontend": frontend.rstrip("~"),
|
||||
"ssl": is_ssl,
|
||||
"backend": groups["backend"],
|
||||
"server": groups["server"],
|
||||
"time_request": int(groups["time_request"]),
|
||||
"time_queue": int(groups["time_queue"]),
|
||||
"time_connect": int(groups["time_connect"]),
|
||||
"time_response": int(groups["time_response"]),
|
||||
"time_total": int(groups["time_total"]),
|
||||
"status_code": int(groups["status_code"]),
|
||||
"bytes_read": int(groups["bytes_read"]),
|
||||
"termination_state": groups["termination_state"],
|
||||
"active_connections": int(groups["active_connections"]),
|
||||
"frontend_connections": int(groups["frontend_connections"]),
|
||||
"backend_connections": int(groups["backend_connections"]),
|
||||
"server_connections": int(groups["server_connections"]),
|
||||
"retries": int(groups["retries"]),
|
||||
"server_queue": int(groups["server_queue"]),
|
||||
"backend_queue": int(groups["backend_queue"]),
|
||||
"host": host,
|
||||
"user_agent": user_agent,
|
||||
"method": method,
|
||||
"target": target,
|
||||
"path": path,
|
||||
"query": query or None,
|
||||
"http_version": http_version,
|
||||
}
|
||||
|
||||
|
||||
def _parse_accept_date(value: str) -> datetime:
|
||||
"""Parse HAProxy's accept date and attach the host's local timezone."""
|
||||
# HAProxy logs naive local wall-clock time; astimezone() interprets it as the
|
||||
# host's local zone and returns a timezone-aware datetime.
|
||||
return datetime.strptime(value, _ACCEPT_DATE_FORMAT).astimezone()
|
||||
|
||||
|
||||
def _split_request_headers(captures: str) -> tuple[str | None, str | None]:
|
||||
"""Pull the Host and User-Agent out of the first ``{host|user-agent}`` capture."""
|
||||
match = _HEADER_CAPTURE.search(captures)
|
||||
if match is None:
|
||||
return None, None
|
||||
host, _, user_agent = match.group(1).partition("|")
|
||||
return (host or None), (user_agent or None)
|
||||
|
||||
|
||||
def _split_request(request: str) -> tuple[str, str, str]:
|
||||
"""Split a request line into method, target and HTTP version.
|
||||
|
||||
Tolerates malformed values such as ``<BADREQ>`` by returning empty strings
|
||||
for the missing parts.
|
||||
"""
|
||||
method, _, remainder = request.partition(" ")
|
||||
target, _, http_version = remainder.partition(" ")
|
||||
return method, target, http_version
|
||||
@@ -0,0 +1 @@
|
||||
"""Tuya heater control service."""
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user