Compare commits
60
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
412a73cedf | ||
|
|
5176dfcb77 | ||
|
|
b85ba92380 | ||
|
|
e9fe18266d | ||
|
|
5ec5a4d334 | ||
|
|
0c25f63600 | ||
|
|
64fed8671e | ||
|
|
3a29294f96 | ||
|
|
428168c07a | ||
|
|
fbe8a8bf21 | ||
|
|
157cc60e91 | ||
|
|
930b222b28 | ||
|
|
01cdd95bc7 | ||
|
|
e898069b03 | ||
|
|
b33f57d942 | ||
|
|
54043a97d2 | ||
|
|
45b03764d2 | ||
|
|
b70ed5cb51 | ||
|
|
0220f2d970 | ||
|
|
55e00ba960 | ||
|
|
85367daf7e | ||
|
|
7e33a8fe05 | ||
|
|
202cbed30e | ||
|
|
a3443d36f1 | ||
|
|
75fe41e266 | ||
|
|
d16a90f08f | ||
|
|
904d8fc76f | ||
|
|
315149e8a2 | ||
|
|
699a1a7fa2 | ||
|
|
22539644a9 | ||
|
|
a48be517e4 | ||
|
|
2e0d556c33 | ||
|
|
ff75f2c758 | ||
|
|
e967019964 | ||
|
|
f6bb06ac6e | ||
|
|
8ee406533c | ||
|
|
373c96d6d6 | ||
|
|
fc6b97bf38 | ||
|
|
5047277845 | ||
|
|
6d1890ce04 | ||
|
|
523117ec28 | ||
|
|
86e1d096ef | ||
|
|
3f39f1f8ae | ||
|
|
d321315acf | ||
|
|
b124a65d6e | ||
|
|
d78b21c27f | ||
|
|
968166c1fd | ||
|
|
8589adbd1b | ||
|
|
510306f24d | ||
|
|
bd8d68b9ba | ||
|
|
540522e828 | ||
|
|
5305d3bb5e | ||
|
|
7e04826fa5 | ||
|
|
96789620bb | ||
|
|
d4bdb5ed42 | ||
|
|
5f2d68694d | ||
|
|
d6207b5743 | ||
|
|
51c31e6bf6 | ||
|
|
961a63480e | ||
|
|
92ffa4f7c2 |
@@ -1,10 +1,28 @@
|
|||||||
# Change this secret to something random
|
# Change this secret to something random
|
||||||
CHITAI_TOKEN_SECRET=secret
|
CHITAI_TOKEN_SECRET=secret
|
||||||
|
|
||||||
|
# Which release to run. Pin this to a tag (e.g. 0.1.0) for a real deployment so a
|
||||||
|
# restart cannot silently move you to a newer image.
|
||||||
|
CHITAI_VERSION=latest
|
||||||
|
|
||||||
|
# The URL you reach the app on, exactly as it appears in the browser. Logging in fails
|
||||||
|
# with a 403 if this does not match, because SvelteKit rejects the form POST as
|
||||||
|
# cross-origin. Include the port unless it is the scheme default.
|
||||||
|
CHITAI_ORIGIN=http://localhost:3000
|
||||||
|
|
||||||
# Setup the path to your initial library (optional)
|
# Setup the path to your initial library (optional)
|
||||||
CHITAI_DEFAULT_LIBRARY_NAME=Books
|
CHITAI_DEFAULT_LIBRARY_NAME=Books
|
||||||
CHITAI_DEFAULT_LIBRARY_PATH="libraries/books"
|
CHITAI_DEFAULT_LIBRARY_PATH="libraries/books"
|
||||||
|
|
||||||
|
# Duplicate detection when importing files (optional).
|
||||||
|
# Scope: "library" compares against the library being uploaded to, "global" against
|
||||||
|
# every library, "off" disables the check.
|
||||||
|
CHITAI_DUPLICATE_SCOPE=library
|
||||||
|
|
||||||
|
# Where the consume watcher parks files it refused as duplicates. Keep it outside
|
||||||
|
# CHITAI_CONSUME_PATH, or the watcher picks them straight back up.
|
||||||
|
CHITAI_DUPLICATE_PATH="duplicates"
|
||||||
|
|
||||||
# You probably should not change these
|
# You probably should not change these
|
||||||
CHITAI_API_URL="http://backend:8000"
|
CHITAI_API_URL="http://backend:8000"
|
||||||
CHITAI_API_DEBUG=false
|
CHITAI_API_DEBUG=false
|
||||||
|
|||||||
@@ -0,0 +1,102 @@
|
|||||||
|
name: ci
|
||||||
|
|
||||||
|
# Formatting, linting, types and tests on every push and pull request.
|
||||||
|
#
|
||||||
|
# Blocking: ruff format, ruff check, pytest, prettier and svelte-check.
|
||||||
|
# Non-blocking: eslint, which reports two `{@html}` XSS findings in collapsible-text.svelte.
|
||||||
|
# Those are a real vulnerability rather than a lint nit — book descriptions from EPUB files
|
||||||
|
# are not sanitized — and fixing them is a backend change. Drop the `continue-on-error` once
|
||||||
|
# that lands, at which point every check blocks.
|
||||||
|
#
|
||||||
|
# The release workflow runs the blocking half again before it publishes anything.
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: ['**']
|
||||||
|
pull_request:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
backend:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
# pytest-databases starts PostgreSQL in a container and then connects to it, and those are
|
||||||
|
# two different addresses that have to be set separately:
|
||||||
|
#
|
||||||
|
# DOCKER_HOST which daemon to create the container on (_service.py get_docker_host)
|
||||||
|
# POSTGRES_HOST where the test then connects (docker/postgres.py, default
|
||||||
|
# 127.0.0.1 -- the job container's own loopback, where nothing listens,
|
||||||
|
# because the database is a sibling container on another namespace)
|
||||||
|
#
|
||||||
|
# Setting only the first leaves the tests dialling 127.0.0.1 and timing out with
|
||||||
|
# "Service 'pytest_databases_postgres' failed to come online".
|
||||||
|
#
|
||||||
|
# If the runner is ever given its own dind sidecar, drop this services block and keep the
|
||||||
|
# two env vars pointed at whatever host it exposes.
|
||||||
|
services:
|
||||||
|
docker:
|
||||||
|
image: docker:27-dind
|
||||||
|
options: --privileged
|
||||||
|
env:
|
||||||
|
DOCKER_TLS_CERTDIR: ''
|
||||||
|
|
||||||
|
env:
|
||||||
|
DOCKER_HOST: tcp://docker:2375
|
||||||
|
POSTGRES_HOST: docker
|
||||||
|
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
working-directory: backend
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
# astral-sh/setup-uv is not mirrored on gitea.com, unlike the actions used elsewhere
|
||||||
|
# here, so install uv directly rather than depending on DEFAULT_ACTIONS_URL.
|
||||||
|
- name: Install uv
|
||||||
|
run: |
|
||||||
|
curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: uv sync --locked
|
||||||
|
|
||||||
|
- name: Format
|
||||||
|
run: uv run ruff format --check src/ tests/
|
||||||
|
|
||||||
|
- name: Lint
|
||||||
|
run: uv run ruff check src/ tests/
|
||||||
|
|
||||||
|
- name: Tests
|
||||||
|
run: uv run pytest tests/ -q
|
||||||
|
|
||||||
|
frontend:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
working-directory: frontend
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: actions/setup-node@v4
|
||||||
|
with:
|
||||||
|
node-version: 24
|
||||||
|
|
||||||
|
- name: Enable pnpm
|
||||||
|
run: corepack enable
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: pnpm install --frozen-lockfile
|
||||||
|
|
||||||
|
# Split out of `pnpm lint` (which is prettier && eslint) so formatting can block
|
||||||
|
# while eslint does not.
|
||||||
|
- name: Format
|
||||||
|
run: pnpm exec prettier --check .
|
||||||
|
|
||||||
|
- name: Lint
|
||||||
|
run: pnpm exec eslint .
|
||||||
|
continue-on-error: true
|
||||||
|
|
||||||
|
- name: Types
|
||||||
|
run: pnpm check
|
||||||
@@ -0,0 +1,142 @@
|
|||||||
|
name: release
|
||||||
|
|
||||||
|
# Builds and publishes the two container images from a version tag.
|
||||||
|
#
|
||||||
|
# Tagging v1.2.3 publishes chitai-backend and chitai-frontend as 1.2.3, 1.2, 1 and latest.
|
||||||
|
# A prerelease tag (v1.2.3-rc.1) publishes only 1.2.3-rc.1 and leaves latest alone, which
|
||||||
|
# makes -rc tags a safe way to exercise this workflow.
|
||||||
|
#
|
||||||
|
# Note that `uses: docker/...` does not mean github.com here the way it would on GitHub. Gitea
|
||||||
|
# resolves a bare reference against the instance's DEFAULT_ACTIONS_URL, which defaults to
|
||||||
|
# gitea.com; all five actions below are mirrored there at these tags. If that setting is ever
|
||||||
|
# pointed somewhere without them, set it to `github` in app.ini rather than editing this file.
|
||||||
|
#
|
||||||
|
# Requires a runner with a working Docker daemon (a docker:dind sidecar, or host mode with
|
||||||
|
# the socket mounted) and, for the smoke job, the compose plugin.
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- 'v*'
|
||||||
|
|
||||||
|
env:
|
||||||
|
REGISTRY: git.jaroszew.ski
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
# The blocking half of ci.yml, run again on the tagged commit so a release cannot publish
|
||||||
|
# an image whose tests fail. Deliberately duplicated rather than shared: Gitea's support
|
||||||
|
# for reusable workflows is thinner than GitHub's, and this is a dozen lines.
|
||||||
|
# Keep in step with ci.yml.
|
||||||
|
quality:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Install uv
|
||||||
|
run: |
|
||||||
|
curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||||
|
|
||||||
|
- name: Backend format, lint and tests
|
||||||
|
working-directory: backend
|
||||||
|
run: |
|
||||||
|
uv sync --locked
|
||||||
|
uv run ruff format --check src/ tests/
|
||||||
|
uv run ruff check src/ tests/
|
||||||
|
uv run pytest tests/ -q
|
||||||
|
|
||||||
|
- uses: actions/setup-node@v4
|
||||||
|
with:
|
||||||
|
node-version: 24
|
||||||
|
|
||||||
|
- name: Frontend format and types
|
||||||
|
working-directory: frontend
|
||||||
|
run: |
|
||||||
|
corepack enable
|
||||||
|
pnpm install --frozen-lockfile
|
||||||
|
pnpm exec prettier --check .
|
||||||
|
pnpm check
|
||||||
|
|
||||||
|
build:
|
||||||
|
needs: quality
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
strategy:
|
||||||
|
# The two images are independent artifacts; don't cancel a good build for a bad one.
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
include:
|
||||||
|
- component: backend
|
||||||
|
context: ./backend
|
||||||
|
- component: frontend
|
||||||
|
context: ./frontend
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
|
- uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: ${{ env.REGISTRY }}
|
||||||
|
username: ${{ github.actor }}
|
||||||
|
password: ${{ secrets.REGISTRY_TOKEN || secrets.GITEA_TOKEN }}
|
||||||
|
|
||||||
|
- id: meta
|
||||||
|
uses: docker/metadata-action@v5
|
||||||
|
with:
|
||||||
|
images: ${{ env.REGISTRY }}/${{ github.repository }}-${{ matrix.component }}
|
||||||
|
tags: |
|
||||||
|
type=semver,pattern={{version}}
|
||||||
|
type=semver,pattern={{major}}.{{minor}}
|
||||||
|
type=semver,pattern={{major}}
|
||||||
|
labels: |
|
||||||
|
org.opencontainers.image.title=chitai-${{ matrix.component }}
|
||||||
|
org.opencontainers.image.source=https://git.jaroszew.ski/${{ github.repository }}
|
||||||
|
|
||||||
|
- uses: docker/build-push-action@v6
|
||||||
|
with:
|
||||||
|
context: ${{ matrix.context }}
|
||||||
|
platforms: linux/amd64
|
||||||
|
push: true
|
||||||
|
tags: ${{ steps.meta.outputs.tags }}
|
||||||
|
labels: ${{ steps.meta.outputs.labels }}
|
||||||
|
# Relies on the runner's cache server. If it is disabled, drop these two lines —
|
||||||
|
# a cold build of both images is only a few minutes.
|
||||||
|
cache-from: type=gha
|
||||||
|
cache-to: type=gha,mode=max
|
||||||
|
|
||||||
|
smoke:
|
||||||
|
needs: build
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: ${{ env.REGISTRY }}
|
||||||
|
username: ${{ github.actor }}
|
||||||
|
password: ${{ secrets.REGISTRY_TOKEN || secrets.GITEA_TOKEN }}
|
||||||
|
|
||||||
|
# Boots the stack from the images that were just pushed, rather than rebuilding them.
|
||||||
|
# This is what catches migrations failing from entrypoint.sh, the frontend being unable
|
||||||
|
# to reach the backend, and a missing runtime environment variable.
|
||||||
|
- name: Boot the published images
|
||||||
|
env:
|
||||||
|
TAG: ${{ github.ref_name }}
|
||||||
|
run: |
|
||||||
|
cp .env.prod-example .env
|
||||||
|
echo "CHITAI_VERSION=${TAG#v}" >> .env
|
||||||
|
mkdir -p libraries
|
||||||
|
docker compose pull backend frontend
|
||||||
|
docker compose up -d --wait --wait-timeout 180
|
||||||
|
curl -fsS http://localhost:8000/healthcheck
|
||||||
|
curl -fsS http://localhost:3000/login > /dev/null
|
||||||
|
|
||||||
|
- name: Tear down
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
docker compose logs --no-color || true
|
||||||
|
docker compose down -v || true
|
||||||
+2
-1
@@ -1,3 +1,4 @@
|
|||||||
.env
|
.env
|
||||||
.postgres/
|
.postgres/
|
||||||
.venv
|
.venv
|
||||||
|
tmp/
|
||||||
|
|||||||
@@ -9,13 +9,15 @@ a KOSync-compatible endpoint.
|
|||||||
|
|
||||||
## Layout
|
## Layout
|
||||||
|
|
||||||
| Path | What |
|
| Path | What |
|
||||||
| --- | --- |
|
| -------------------------- | ---------------------------------------------------------------------------------------- |
|
||||||
| `backend/` | Litestar REST API + PostgreSQL. See `backend/AGENTS.md`. |
|
| `backend/` | Litestar REST API + PostgreSQL. See `backend/AGENTS.md`. |
|
||||||
| `frontend/` | SvelteKit SSR web app. See `frontend/AGENTS.md`. |
|
| `frontend/` | SvelteKit SSR web app. See `frontend/AGENTS.md`. |
|
||||||
| `docker-compose.yml` | Production stack: `db` (postgres:17), `backend`, `frontend`. |
|
| `frontend/src/lib/vendor/` | Vendored `foliate-js` (the EPUB engine), copied by `frontend/scripts/vendor-foliate.sh`. |
|
||||||
| `docs/screenshots/` | Images used by `README.md`. |
|
| `frontend/static/pdfjs/` | Vendored pdf.js viewer, used by the PDF reader in an iframe. |
|
||||||
| `shell.nix` | Root dev shell; composes the two sub-shells. |
|
| `docker-compose.yml` | Production stack: `db` (postgres:17), `backend`, `frontend`. |
|
||||||
|
| `docs/screenshots/` | Images used by `README.md`. |
|
||||||
|
| `shell.nix` | Root dev shell; composes the two sub-shells. |
|
||||||
|
|
||||||
## Development environment
|
## Development environment
|
||||||
|
|
||||||
@@ -68,9 +70,12 @@ Backend (from `backend/`):
|
|||||||
```bash
|
```bash
|
||||||
uv run litestar --app-dir src/chitai/ run --reload # dev server on :8000
|
uv run litestar --app-dir src/chitai/ run --reload # dev server on :8000
|
||||||
pytest tests/ # needs Docker (pytest-databases)
|
pytest tests/ # needs Docker (pytest-databases)
|
||||||
ruff format src/
|
ruff format src/ tests/
|
||||||
alchemy --config chitai.database.config.config make-migrations
|
alchemy --config chitai.database.config.config make-migrations
|
||||||
alchemy --config chitai.database.config.config upgrade
|
alchemy --config chitai.database.config.config upgrade
|
||||||
|
|
||||||
|
# Import a Calibre library. Copies files; --dry-run reports without writing.
|
||||||
|
litestar --app-dir src/chitai/ calibre-import <path> --library <slug>
|
||||||
```
|
```
|
||||||
|
|
||||||
Frontend (from `frontend/`):
|
Frontend (from `frontend/`):
|
||||||
@@ -98,6 +103,13 @@ API docs are served by the running backend at `http://localhost:8000/schema/` (S
|
|||||||
manually against a live backend).
|
manually against a live backend).
|
||||||
- **Migrations are mandatory.** The app runs with `create_all=False`, so a model change without a
|
- **Migrations are mandatory.** The app runs with `create_all=False`, so a model change without a
|
||||||
matching Alembic revision will not reach the database.
|
matching Alembic revision will not reach the database.
|
||||||
|
- **CI gates formatting, linting, types and tests.** `.gitea/workflows/ci.yml` runs on every push
|
||||||
|
and pull request: `ruff format --check src/ tests/`, `ruff check src/ tests/`, `pytest`,
|
||||||
|
`prettier --check` and `pnpm check` all **block**; only `eslint` reports without failing, and
|
||||||
|
only until the two `{@html}` findings in `TODO.md` are fixed. Ruff covers `src/` and `tests/`
|
||||||
|
but deliberately not `migrations/`, whose alembic template emits imports it does not use. A
|
||||||
|
`v*` tag additionally builds and publishes both container images — see
|
||||||
|
`docs/ci-release-pipeline.md`.
|
||||||
- **Commit messages** follow `type: summary` — `feat:`, `fix:`, `refactor:`, `chore:`.
|
- **Commit messages** follow `type: summary` — `feat:`, `fix:`, `refactor:`, `chore:`.
|
||||||
- The three `README.md` files are user-facing. Agent-facing knowledge belongs in the `AGENTS.md`
|
- The three `README.md` files are user-facing. Agent-facing knowledge belongs in the `AGENTS.md`
|
||||||
files.
|
files.
|
||||||
|
|||||||
@@ -26,12 +26,18 @@ eBook library management application.
|
|||||||
|
|
||||||
### Installation (Production Deployment)
|
### Installation (Production Deployment)
|
||||||
|
|
||||||
1. Clone the repository: `git clone <repository_url>` (Replace `<repository_url>` with the actual URL)
|
1. Clone the repository: `git clone https://git.jaroszew.ski/patrick/chitai.git`
|
||||||
2. Navigate to the project root: `cd chitai`
|
2. Navigate to the project root: `cd chitai`
|
||||||
3. Build the Docker images: `docker compose build`
|
3. Copy the example environment file: `cp .env.prod-example .env`
|
||||||
4. Copy the example environment file and configure: `cp .env.prod-example .env`
|
4. Edit `.env`. At minimum set `CHITAI_TOKEN_SECRET` to something random, and set
|
||||||
5. Run the Docker containers in detached mode: `docker compose up -d`
|
`CHITAI_ORIGIN` to the URL you will reach the app on — logging in fails with a 403 if it
|
||||||
6. Access the frontend at: `http://localhost:3000/`
|
does not match. Pin `CHITAI_VERSION` to a release tag if you would rather not track `latest`.
|
||||||
|
5. Pull the released images: `docker compose pull`
|
||||||
|
6. Run the Docker containers in detached mode: `docker compose up -d`
|
||||||
|
7. Access the frontend at: `http://localhost:3000/`
|
||||||
|
|
||||||
|
To build the images from source instead of pulling them, run `docker compose build` in place of
|
||||||
|
step 5.
|
||||||
|
|
||||||
## Development
|
## Development
|
||||||
|
|
||||||
|
|||||||
@@ -49,8 +49,310 @@ Worth adding at the same time:
|
|||||||
- Deriving ISBN-10 from ISBN-13 when only the latter is present. It is a pure checksum
|
- Deriving ISBN-10 from ISBN-13 when only the latter is present. It is a pure checksum
|
||||||
conversion and doubles the chance of an external lookup matching.
|
conversion and doubles the chance of an external lookup matching.
|
||||||
|
|
||||||
|
### Any authenticated user can delete any library or book
|
||||||
|
|
||||||
|
`backend/src/chitai/database/models/library.py`, `backend/src/chitai/database/models/user.py`
|
||||||
|
|
||||||
|
There is no authorization tier. `Library` has no owner column, and `User` carries only
|
||||||
|
`email` and `password` — no role, no `is_active`. So every authenticated account can create
|
||||||
|
and delete libraries, and delete books along with their files on disk. Per-user scoping
|
||||||
|
exists only for reading progress and bookshelves, which `provide_book_service` restricts
|
||||||
|
correctly.
|
||||||
|
|
||||||
|
For a single-household deployment that may well be acceptable. The point is that it is
|
||||||
|
emergent rather than chosen. The cheapest meaningful step is an `is_admin` flag gating
|
||||||
|
library deletion and `delete_books` — a model change plus a migration.
|
||||||
|
|
||||||
|
### Basic auth answers a malformed header with a 500
|
||||||
|
|
||||||
|
`backend/src/chitai/middleware/basic_auth.py` — line 22
|
||||||
|
|
||||||
|
```python
|
||||||
|
username, password = b64decode(auth_header.split("Basic ")[1]).decode().split(":")
|
||||||
|
```
|
||||||
|
|
||||||
|
Nothing guards the parse. A `Bearer` token raises `IndexError`, non-base64 raises
|
||||||
|
`binascii.Error`, and a credential with no colon raises `ValueError` — as does a password
|
||||||
|
that *contains* one, since there is no `maxsplit=1`. Every case surfaces as a 500 on an
|
||||||
|
unauthenticated endpoint. All of them should be 401.
|
||||||
|
|
||||||
|
### An unknown KOSync API key returns 404
|
||||||
|
|
||||||
|
`backend/src/chitai/middleware/kosync_auth.py` — line 32
|
||||||
|
|
||||||
|
`KosyncDeviceService.get_by_api_key` uses `get_one`, which raises `NotFoundError`, but the
|
||||||
|
middleware catches only `PermissionDeniedException`. The global handler in
|
||||||
|
`exceptions/handlers.py` then renders it as a 404, so a device presenting a bad key is told
|
||||||
|
the route does not exist rather than that it is unauthorized. The same file still carries a
|
||||||
|
leftover `print(exc)`.
|
||||||
|
|
||||||
|
Worth doing at the same time: `KosyncDeviceService._generate_api_key` uses
|
||||||
|
`secrets.token_hex(8)`. 64 bits is thin for a long-lived bearer credential where 32 bytes
|
||||||
|
is the convention.
|
||||||
|
|
||||||
|
### The multi-book download cannot be driven from a test
|
||||||
|
|
||||||
|
`backend/src/chitai/services/book.py` — `BookService.get_files`
|
||||||
|
|
||||||
|
`/books/download` is the only handler returning a Litestar `Stream`, and it cannot be
|
||||||
|
exercised through `AsyncTestClient`. The request itself succeeds, then fixture teardown
|
||||||
|
hangs: the test transport never sends the `http.disconnect` that the streaming response
|
||||||
|
waits on, so the app's lifespan shutdown never completes. Coverage therefore sits at the
|
||||||
|
service level, on `get_files` directly.
|
||||||
|
|
||||||
|
Unresolved whether the endpoint also stalls behind a real ASGI server, where that
|
||||||
|
disconnect does arrive. Worth one manual check against `litestar run` before relying on it.
|
||||||
|
|
||||||
|
### The production image runs the development server
|
||||||
|
|
||||||
|
`backend/Dockerfile` — the final `CMD`
|
||||||
|
|
||||||
|
```
|
||||||
|
CMD ["litestar", "--app-dir", "chitai", "run", "--host", "0.0.0.0", "--port", "8000"]
|
||||||
|
```
|
||||||
|
|
||||||
|
`litestar run` is the CLI development runner. Production should invoke uvicorn or granian
|
||||||
|
directly, with a worker count.
|
||||||
|
|
||||||
|
### The delete_files flag is untested in both directions
|
||||||
|
|
||||||
|
`backend/tests/integration/test_book.py` — `test_remove_file_with_delete_files_false_keeps_filesystem_file`
|
||||||
|
and `test_remove_file_with_delete_files_true_removes_filesystem_file`
|
||||||
|
|
||||||
|
Both tests capture the file's path and then assert only `response.status_code == 204`.
|
||||||
|
Neither looks at the disk. So the flag that decides whether removing a file from a book
|
||||||
|
also **erases it from the filesystem** is covered in name only, in both directions.
|
||||||
|
|
||||||
|
Ruff surfaced this as two `F841` unused variables; the variables carry a `# noqa: F841`
|
||||||
|
and a comment rather than being deleted, so the gap stays visible. Remove the noqa when
|
||||||
|
the assertions land.
|
||||||
|
|
||||||
|
The reason it is not a two-line fix: `FileMetadata.path` is stored relative to `book.path`,
|
||||||
|
so the test has to resolve it against the library root to know what to stat. That
|
||||||
|
resolution is the same thing `BookService.get_files` is recorded as getting wrong (see
|
||||||
|
`backend/AGENTS.md`), so it is worth settling once and using in both places.
|
||||||
|
|
||||||
|
### No type checker on the backend
|
||||||
|
|
||||||
|
Formatting, linting and tests now run in CI (`.gitea/workflows/ci.yml`) and block, and the
|
||||||
|
tree is clean against them. What is still missing is a type checker: nothing runs one despite
|
||||||
|
`# type: ignore` comments in the tree.
|
||||||
|
|
||||||
|
`pyproject.toml` gained a `[tool.ruff.lint.per-file-ignores]` section for `__init__.py`
|
||||||
|
re-exports, but no rule selection — so only ruff's default `E4/E7/E9/F` rules run. Widening
|
||||||
|
that set is worthwhile and will surface a fresh batch of findings.
|
||||||
|
|
||||||
## Frontend
|
## Frontend
|
||||||
|
|
||||||
|
### Book descriptions are rendered as unsanitized HTML
|
||||||
|
|
||||||
|
`frontend/src/lib/components/ui/collapsible-text/collapsible-text.svelte` — lines 42 and 45
|
||||||
|
|
||||||
|
The component renders `{@html text}`, and its only caller is the book detail page:
|
||||||
|
`<CollapsibleText text={book.description} maxLength={500} />`. So whatever is in
|
||||||
|
`Book.description` reaches the DOM as markup.
|
||||||
|
|
||||||
|
The Calibre importer is fine — `services/calibre.py:401` passes comments through
|
||||||
|
`strip_html`, because Calibre stores HTML there. But `strip_html` is used **nowhere else in
|
||||||
|
the backend**, and `EpubExtractor._extract_description` returns
|
||||||
|
`epub.get_metadata("DC", "description")[0][0]` verbatim. EPUB `dc:description` routinely
|
||||||
|
carries markup, so an uploaded book with `<img src=x onerror=…>` in that field executes
|
||||||
|
script on the book page, with the session cookie in scope. Metadata edited through the UI
|
||||||
|
is stored unfiltered too.
|
||||||
|
|
||||||
|
This is the same class as the scripted-EPUB item below — untrusted file content reaching an
|
||||||
|
origin that holds a session — by a different route, and it does not need `allow-scripts` to
|
||||||
|
work.
|
||||||
|
|
||||||
|
Fix: sanitize at ingest, next to where Calibre already does. Reuse `strip_html` in
|
||||||
|
`_extract_description` if descriptions should be plain text, or run an allowlist sanitizer if
|
||||||
|
the formatting is worth keeping. Either way the stored rows need backfilling through the same
|
||||||
|
helper, since the validators only fire on write. Dropping `{@html}` to `{text}` in the
|
||||||
|
component fixes the display side but leaves the payload in the database.
|
||||||
|
|
||||||
|
These are the only two findings `pnpm exec eslint .` still reports; CI's eslint step stops
|
||||||
|
being `continue-on-error` once they are gone.
|
||||||
|
|
||||||
|
### Scripted EPUBs run against the app origin
|
||||||
|
|
||||||
|
**This is a regression from the foliate-js migration, not a pre-existing gap.**
|
||||||
|
|
||||||
|
The old epub.js reader never passed `allowScriptedContent`. epub.js defaults it to
|
||||||
|
`false`, which sets `iframe.sandbox = "allow-same-origin"` — no `allow-scripts` — so
|
||||||
|
script inside a book never ran. The vendored foliate-js sets, unconditionally:
|
||||||
|
|
||||||
|
```js
|
||||||
|
// paginator.js — and the same in fixed-layout.js
|
||||||
|
// `allow-scripts` is needed for events because of WebKit bug
|
||||||
|
this.#iframe.setAttribute("sandbox", "allow-same-origin allow-scripts");
|
||||||
|
```
|
||||||
|
|
||||||
|
`allow-same-origin` together with `allow-scripts` is the combination that makes the
|
||||||
|
sandbox attribute do nothing. Sections are served as same-origin `blob:` URLs, so script
|
||||||
|
in a book can reach `/api/*` with the session cookie attached. foliate's README says as
|
||||||
|
much and tells you to use a CSP instead; we have not added one.
|
||||||
|
|
||||||
|
This is not theoretical. Audiobookshelf shipped the same combination and got
|
||||||
|
**CVE-2024-35236** — scripted EPUB plus an unrestricted upload gave remote code
|
||||||
|
execution; fixed in 2.10.0 by making scripted content a per-library opt-in, off by
|
||||||
|
default. Kavita (CVE-2024-39307) and Jellyfin (fixed 10.9.8) are variations on it.
|
||||||
|
Write-up: <https://gebir.ge/blog/every-trick-in-the-book/>.
|
||||||
|
|
||||||
|
**An app-wide CSP is the wrong shape.** `kit.csp` with `script-src: ['self']` also blocks
|
||||||
|
`mode-watcher`'s inline `setInitialMode`, which sets the dark class before first paint —
|
||||||
|
SvelteKit only nonces the bootstrap script it injects itself, so every page load would
|
||||||
|
flash the light theme. Pinning a hash of a third-party inline script breaks silently on
|
||||||
|
upgrade.
|
||||||
|
|
||||||
|
**Grimmory solves it properly**, and it runs foliate-js too. Rather than handing foliate
|
||||||
|
the whole file, it serves each EPUB entry from its own endpoint and puts the strict
|
||||||
|
policy on that response:
|
||||||
|
|
||||||
|
```java
|
||||||
|
// EpubReaderController.java
|
||||||
|
response.setHeader("Content-Security-Policy", "script-src 'none'");
|
||||||
|
```
|
||||||
|
|
||||||
|
The app shell keeps its own, more permissive policy. That works because each section is
|
||||||
|
then a real same-origin document with its own header, rather than a `blob:` — and a
|
||||||
|
`blob:` inherits the CSP of the document that created it, which is exactly why a header
|
||||||
|
on `/api/books/download/…` would achieve nothing today.
|
||||||
|
|
||||||
|
Two ways forward:
|
||||||
|
|
||||||
|
1. **Cheap.** Patch the vendored `sandbox` attribute to drop `allow-scripts`, restoring
|
||||||
|
what epub.js gave us. Cost is the WebKit bug the upstream comment cites: events inside
|
||||||
|
the iframe get swallowed, which would likely break touch/swipe paging and possibly the
|
||||||
|
in-iframe keyboard handling in `foliate-view.svelte`. Needs testing before trusting.
|
||||||
|
2. **Right.** Follow grimmory: serve individual EPUB entries from the backend with
|
||||||
|
`script-src 'none'` on each response, and drive foliate through its loader hooks
|
||||||
|
instead of a whole-file blob. This is **net-new capability on both sides**, not a
|
||||||
|
rewiring of something that exists — see below.
|
||||||
|
|
||||||
|
Option 2 also fixes the memory cost below, which is why it is worth more than it looks.
|
||||||
|
|
||||||
|
#### What option 2 actually involves
|
||||||
|
|
||||||
|
Today the browser fetches the whole `.epub` from `download/{book_id}/{file_id}`, which
|
||||||
|
returns a Litestar `File` and knows nothing about the archive's contents. foliate then
|
||||||
|
opens the zip **in the browser** (`makeZipLoader` in `view.js`) and turns every chapter,
|
||||||
|
image and stylesheet into a `blob:` URL via `Loader.createURL` in `epub.js`. A `blob:`
|
||||||
|
carries no headers of its own — it inherits the CSP of the document that created it —
|
||||||
|
which is why there is nowhere to attach a policy except the app shell.
|
||||||
|
|
||||||
|
foliate's parser never touches the zip directly. `EPUB` is constructed with a loader:
|
||||||
|
|
||||||
|
```js
|
||||||
|
// view.js — makeZipLoader is one implementation; makeDirectoryLoader below is another
|
||||||
|
return { entries, loadText, loadBlob, getSize };
|
||||||
|
```
|
||||||
|
|
||||||
|
`name` is the **zip entry path**, because `makeZipLoader` keys its map on
|
||||||
|
`entry.filename` — so `OEBPS/Text/chapter01.xhtml`, `OEBPS/Images/cover.jpg`. The parser
|
||||||
|
resolves hrefs from the OPF manifest into those names and asks the loader for them,
|
||||||
|
without caring where the bytes come from. A third implementation that fetches over HTTP
|
||||||
|
is the same shape.
|
||||||
|
|
||||||
|
**Backend.** An endpoint taking a path inside the archive, e.g.
|
||||||
|
`GET books/{book_id}/files/{file_id}/entry/{path:path}`, returning a `Stream` over
|
||||||
|
`zipfile.ZipFile.open(name)` so an entry never lands in memory whole, with the content
|
||||||
|
type from the manifest and `Content-Security-Policy: script-src 'none'` on the response.
|
||||||
|
|
||||||
|
Two things to get right:
|
||||||
|
|
||||||
|
- **`path` is caller-supplied.** Resolve it against the archive's `namelist()` and reject
|
||||||
|
anything absent, rather than trusting the string — `../` traversal is the hazard.
|
||||||
|
- **`getSize` is synchronous** in foliate's loader contract, and it feeds `SectionProgress`,
|
||||||
|
which produces the reading percentage. So the endpoint needs a companion that returns
|
||||||
|
entry names and sizes up front — one extra call at open — because sizes cannot be
|
||||||
|
discovered per request.
|
||||||
|
|
||||||
|
Opening the zip per request costs a central-directory read each time. Probably fine for
|
||||||
|
chapter-sized reads, worth measuring rather than assuming.
|
||||||
|
|
||||||
|
Note the backend already reads inside EPUBs — `metadata_extractor.py` uses ebooklib at
|
||||||
|
ingest for title, authors, identifiers and the cover. What is missing is serving an
|
||||||
|
arbitrary entry by path, not the ability to open the archive.
|
||||||
|
|
||||||
|
**Frontend.** `foliate-view.svelte` stops calling `view.open(file)` and builds an `EPUB`
|
||||||
|
around a loader backed by that endpoint. The whole-file fetch in `epub-reader.svelte`
|
||||||
|
goes away with it.
|
||||||
|
|
||||||
|
### The proxy buffers whole files and drops range headers
|
||||||
|
|
||||||
|
Two separate problems that both live in `frontend/src/routes/api/[...path]/+server.ts`.
|
||||||
|
|
||||||
|
**Buffering.** Litestar already streams: `ASGIFileResponse` reads in 1 MB chunks
|
||||||
|
(`response/file.py`), so the backend never holds a file whole. The proxy then undoes it
|
||||||
|
with `await response.arrayBuffer()`, which does not resolve until the last byte arrives —
|
||||||
|
so the whole file sits in the node process, per concurrent reader, and the browser gets
|
||||||
|
nothing until it completes. Passing `response.body` straight through restores the stream
|
||||||
|
and is a small change.
|
||||||
|
|
||||||
|
This is now **responses only**. The request side was fixed for the Calibre archive upload,
|
||||||
|
which cannot be held in memory: POST and PATCH pass `request.body` through with
|
||||||
|
`duplex: 'half'` (`bodyOf` in the same file). The response side is the same shape of fix.
|
||||||
|
|
||||||
|
**Range.** The proxy forwards only `Content-Type`, `Content-Disposition` and
|
||||||
|
`Content-Length`. It never sends the client's `Range` upstream, and would drop
|
||||||
|
`Accept-Ranges` and `Content-Range` coming back — a 206 without `Content-Range` is
|
||||||
|
broken. So range support cannot work until the proxy is fixed, whatever the backend does.
|
||||||
|
|
||||||
|
**Litestar has no range support of its own.** In 2.21.1 the only mention of 206 in the
|
||||||
|
whole package is the `HTTP_206_PARTIAL_CONTENT` constant; there is no `Accept-Ranges` or
|
||||||
|
`Content-Range` handling anywhere. This has to be written.
|
||||||
|
|
||||||
|
#### What pdf.js actually needs
|
||||||
|
|
||||||
|
It decides from the **initial 200 response**, not from anything on a 206.
|
||||||
|
`validateRangeRequestCapabilities` in `frontend/static/pdfjs/build/pdf.mjs`:
|
||||||
|
|
||||||
|
```js
|
||||||
|
if (responseHeaders.get("Accept-Ranges") !== "bytes") {
|
||||||
|
return returnValues; // allowRangeRequests stays false
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
It also needs a parseable `Content-Length`, `Content-Encoding: identity`, and a length
|
||||||
|
greater than twice `rangeChunkSize`. Miss any of those and it downloads the whole file
|
||||||
|
however good the range support is.
|
||||||
|
|
||||||
|
So the single highest-value header is **`Accept-Ranges: bytes` on the ordinary 200** —
|
||||||
|
that is what makes pdf.js switch to fetching progressively at all.
|
||||||
|
|
||||||
|
#### Approach
|
||||||
|
|
||||||
|
Put it on the existing `get_file` handler in `controllers/book.py`, which already resolves
|
||||||
|
`book_id`/`file_id` through the service with library scoping and auth:
|
||||||
|
|
||||||
|
- No `Range` → `Stream` the file with `Accept-Ranges: bytes` and `Content-Length`.
|
||||||
|
- `Range` present → parse, seek, `Stream` with 206 and `Content-Range`.
|
||||||
|
- Proxy: forward `Range` up; pass `response.body` through; forward `Accept-Ranges`,
|
||||||
|
`Content-Range` and the status back.
|
||||||
|
|
||||||
|
There are `RangeRequestMiddleware` snippets circulating for Litestar that wrap
|
||||||
|
`create_static_files_router`. They are the wrong shape here — book files are served by an
|
||||||
|
authenticated handler resolving database ids, not by a directory mapping, and using one
|
||||||
|
would mean exposing disk paths as URLs and re-solving ownership checks that already
|
||||||
|
exist. The common version also only sets `Accept-Ranges` on the 206, so it would not
|
||||||
|
switch pdf.js over, and it derives its path with `str.lstrip(prefix)`, which strips a
|
||||||
|
character set rather than a prefix — `/static/castle.pdf` becomes `le.pdf`. Worth reading
|
||||||
|
its `parse_range_header` for the parsing rules and writing the rest fresh.
|
||||||
|
|
||||||
|
### Remove the epub.js locations-cache purge
|
||||||
|
|
||||||
|
`frontend/src/lib/reader/legacy-cache.ts` — `purgeLegacyLocationCache`
|
||||||
|
|
||||||
|
The epub.js reader cached generated locations in `localStorage` under
|
||||||
|
`${bookId}-locations`, a few hundred KB of JSON per long book. foliate computes
|
||||||
|
progress from section byte sizes at open time, so nothing writes those keys any
|
||||||
|
more, but existing browsers still hold them — and a reader near the 5–10 MB
|
||||||
|
origin quota would make the new reader-settings write throw `QuotaExceededError`.
|
||||||
|
|
||||||
|
The reader clears them once per browser, behind a `chitai:locations-purged` flag.
|
||||||
|
Delete the module, its call in `epub-reader.svelte` and the flag once deployments
|
||||||
|
have had a release or two to run it — after roughly 2026-12.
|
||||||
|
|
||||||
### Cover dimensions are unknown until load
|
### Cover dimensions are unknown until load
|
||||||
|
|
||||||
`book-cover.svelte` renders covers at a fixed height with natural width so nothing is
|
`book-cover.svelte` renders covers at a fixed height with natural width so nothing is
|
||||||
|
|||||||
@@ -88,6 +88,14 @@ The backend owns files on disk, not just rows:
|
|||||||
- **Layout** — `services/filesystem_library.py` (`BookPathGenerator`) renders a Jinja2 template
|
- **Layout** — `services/filesystem_library.py` (`BookPathGenerator`) renders a Jinja2 template
|
||||||
against book metadata to decide where a book lives under the library's `root_path`
|
against book metadata to decide where a book lives under the library's `root_path`
|
||||||
(default: `author/series/position - title/`).
|
(default: `author/series/position - title/`).
|
||||||
|
- **One directory per book, never shared.** The generated path is a pure function of the metadata,
|
||||||
|
so two books with the same author and title produce the same one — two editions, or an
|
||||||
|
`allow_duplicates` copy. `BookService._reserve_book_path` moves the later one to `title (2)`
|
||||||
|
before anything is written, and `update_book` reserves the same way so a rename cannot move a
|
||||||
|
book in on top of another. This matters because `book.path` is what deletes, moves and file
|
||||||
|
lookups act on: books sharing a directory means one overwrites the other's files, and deleting
|
||||||
|
either takes both. `_unused_path` does the same job for filenames within a directory.
|
||||||
|
A book that already has a `path` keeps it — `add_files` must follow the book, not the template.
|
||||||
- **Metadata extraction** — `services/metadata_extractor.py` reads EPUB (ebooklib) and PDF
|
- **Metadata extraction** — `services/metadata_extractor.py` reads EPUB (ebooklib) and PDF
|
||||||
(pypdfium2) files; extracted values fill only *empty* fields on the incoming payload.
|
(pypdfium2) files; extracted values fill only *empty* fields on the incoming payload.
|
||||||
- **Covers** — converted to WebP with a UUID filename under `settings.book_cover_path`, served by a
|
- **Covers** — converted to WebP with a UUID filename under `settings.book_cover_path`, served by a
|
||||||
@@ -100,6 +108,235 @@ The backend owns files on disk, not just rows:
|
|||||||
if it differs, moves the directory contents and prunes empty parents. Keep that in mind before
|
if it differs, moves the directory contents and prunes empty parents. Keep that in mind before
|
||||||
changing metadata handling.
|
changing metadata handling.
|
||||||
|
|
||||||
|
### Content types come from the extension, and null means null
|
||||||
|
|
||||||
|
Every ingest path names a file's format with `guess_content_type` (`services/utils.py`), never with
|
||||||
|
`mimetypes.guess_type` directly and never with what the client said. Python's built-in map answers
|
||||||
|
`None` for `.mobi`, `.azw`, `.prc`, `.fb2`, `.fbz`, `.lit`, `.lrf` and `.cb7` — most of what a
|
||||||
|
library imported from elsewhere carries — so `EBOOK_CONTENT_TYPES` fills those in. A browser's
|
||||||
|
`application/octet-stream` is discarded rather than used as a fallback: it is the client saying it
|
||||||
|
does not know, and storing it is indistinguishable from having determined a format.
|
||||||
|
|
||||||
|
When nothing can name the extension the column **stays null**. That is the honest answer, and only
|
||||||
|
one consumer cannot take it: OPDS `Link.type` is a required string, so
|
||||||
|
`services/opds/opds.py` substitutes `application/octet-stream` at that boundary. Litestar's
|
||||||
|
`ASGIFileResponse` already does its own fallback, so `get_file` can pass a null straight through.
|
||||||
|
`FileMetadataRead.content_type` is nullable for the same reason — it was once required, which
|
||||||
|
turned a stored null into a 500 on a book that was otherwise fine.
|
||||||
|
|
||||||
|
## Duplicate detection
|
||||||
|
|
||||||
|
Every ingest path screens incoming files against what is already stored, keyed on
|
||||||
|
**`(hash, size)`** — never the hash alone, because it samples 12 KiB (see below) and
|
||||||
|
EPUBs from one toolchain often share their first window. `FileMetadata.hash` carries a
|
||||||
|
plain, deliberately **non-unique** index: a collision must not be able to fail an import,
|
||||||
|
and older databases may already hold duplicates.
|
||||||
|
|
||||||
|
Scope comes from `CHITAI_DUPLICATE_SCOPE` (`library`, the default | `global` | `off`).
|
||||||
|
|
||||||
|
The policy differs by how deliberate the import is:
|
||||||
|
|
||||||
|
| Path | Behaviour |
|
||||||
|
| --- | --- |
|
||||||
|
| `create_many_from_files` (browser bulk) | Skip per file, skip a whole group whose files are all known, report everything skipped in `ImportResult.duplicates`. Re-dropping a folder to pick up what is new is the case this serves. |
|
||||||
|
| `create_book` (single, with metadata) | All-or-nothing: raises `DuplicateFilesError`, which `controllers/book.py` renders as a **409** carrying the refused files in `extra`. |
|
||||||
|
| `add_files` | A file the book already carries is a no-op; one stored under another book raises `DuplicateFilesError`. |
|
||||||
|
| `create_many_from_existing_files` (consume watcher) | Skips, and **moves the file to `CHITAI_DUPLICATE_PATH/<library slug>/`** — nothing is deleted, and it cannot stay put because `watchfiles` only reports additions. That path must stay outside `consume_path` or the watcher re-imports it and tries to read the directory name as a library slug. |
|
||||||
|
| `create_many_from_calibre` (Calibre import) | Skips per file, and skips a whole book whose files are all known. Nothing is moved — the source is somebody else's library. This is what makes a re-run a no-op and an interrupted import resumable by running it again. |
|
||||||
|
|
||||||
|
`allow_duplicates=true` overrides all of it, on every endpoint. Keep that working — the
|
||||||
|
hash is not proof of identity, so a wrong verdict has to be recoverable, and a scripted
|
||||||
|
import needs a way through. The **web UI deliberately does not offer it**: storing the
|
||||||
|
same bytes twice splits reading progress and shelf membership across two records that
|
||||||
|
can never converge, which is nothing anyone wants on purpose.
|
||||||
|
|
||||||
|
Three things to preserve when touching this code:
|
||||||
|
|
||||||
|
- **A match only counts while the file is on disk.** `find_duplicate_files` stats each
|
||||||
|
candidate, and `add_files` writes a missing file back into the row that already
|
||||||
|
describes it (`_restore_file`) instead of adding a second row beside it. The hash
|
||||||
|
lives in the database and the file does not, so without this a file deleted behind
|
||||||
|
the app's back would go on refusing its own replacement.
|
||||||
|
|
||||||
|
- **Screening runs before anything is written.** `fingerprint_upload` reads the spooled
|
||||||
|
upload and rewinds it; the resulting fingerprints are handed to `_save_book_files`,
|
||||||
|
which skips its own `StreamingHasher` when it already has the answer. Passing them
|
||||||
|
through is what keeps the file from being read twice.
|
||||||
|
- **`_screen_for_duplicates` extends the `known` dict as it goes**, so the same bytes
|
||||||
|
submitted twice in one request are caught. Those duplicates report `book_id: None` —
|
||||||
|
there is no row to point at yet.
|
||||||
|
|
||||||
|
`POST /books/duplicate-files` answers the same question from fingerprints alone, for
|
||||||
|
clients that want to ask before uploading anything.
|
||||||
|
|
||||||
|
### Book-level detection — a different question
|
||||||
|
|
||||||
|
The file check answers "are these the same bytes?". `find_duplicate_books` answers "is
|
||||||
|
this the same book?", which a re-scan, a re-zipped EPUB or another edition cannot be
|
||||||
|
asked with a hash. Two signals, either sufficient: a **shared identifier**, or a
|
||||||
|
**matching normalized title with at least one shared author**.
|
||||||
|
|
||||||
|
It **never blocks**. A metadata match is a guess — a work shares title and author with
|
||||||
|
its own translation, its own second edition and its own audiobook — so the book is
|
||||||
|
created and the candidates are reported alongside it in
|
||||||
|
`ImportResult.possible_duplicates`. File-level dedupe keeps its refuse/skip behaviour;
|
||||||
|
that one is near-certain and this one is not. Do not "improve" this into a refusal.
|
||||||
|
|
||||||
|
Two rules narrow it, both applied in Python over the small candidate set:
|
||||||
|
|
||||||
|
- **A shared author is required for a title match.** Without it every book the
|
||||||
|
extractors gave up on and titled `Unknown` is a duplicate of every other one. A book
|
||||||
|
with no authors can therefore only match on an identifier.
|
||||||
|
- **The same series at a different `series_position` disqualifies a match.** A trilogy
|
||||||
|
shares an author and often most of its title; the position is the library saying
|
||||||
|
outright that these are two books.
|
||||||
|
|
||||||
|
A book is compared under **several title keys, not one** (`_title_keys`). Ebook files
|
||||||
|
are overwhelmingly named `Title - Author.epub`, and wherever nothing inside the file
|
||||||
|
overrode that name the author ended up in the title column — so one copy is stored as
|
||||||
|
`Building Microservices` and another as `Building Microservices Sam Newman`. Both
|
||||||
|
directions are generated, the author stripped off and the author added on, which is why
|
||||||
|
the query is `normalized_title.in_(keys)` rather than `==`. This is still exact matching
|
||||||
|
on an indexed column: no similarity score, nothing to tune. It does not weaken the
|
||||||
|
shared-author requirement, which is a separate condition.
|
||||||
|
|
||||||
|
`find_duplicate_book_groups` is the library-wide pass behind
|
||||||
|
`GET /books/duplicate-books`, since the import-time check says nothing about a
|
||||||
|
collection someone already has. It buckets books by every key they carry and merges the
|
||||||
|
buckets with union-find, so A~B by ISBN and B~C by title land in one group. Pairs in
|
||||||
|
`duplicate_dismissals` are never merged — a reader disagreeing with one pairing must not
|
||||||
|
silently break a group that stands on other evidence.
|
||||||
|
|
||||||
|
### Author names have one stored form
|
||||||
|
|
||||||
|
`Author.name` is always the canonical form, produced by `format_author_name`. Extractors
|
||||||
|
hand over whatever the file said — `Newman, Sam;` from a `DC:creator` list, `Sam Newman`
|
||||||
|
from a PDF, `Sam Newman.epub` from a filename — and storing those verbatim is how one
|
||||||
|
person becomes four rows in the sidebar, four entries in the author filter, and four
|
||||||
|
books that never look like each other.
|
||||||
|
|
||||||
|
This is **display** canonicalization, distinct from `normalize_author`, which throws
|
||||||
|
away case, accents and spacing to build a comparison key nobody sees. Tidying only
|
||||||
|
removes what an extractor added: a trailing separator, a file extension, and the
|
||||||
|
`Surname, Given` ordering. It never touches case or accents — `Michał Płachta` and
|
||||||
|
`Steve McConnell` are the author's own spelling, not something to correct.
|
||||||
|
|
||||||
|
Three places have to agree, and `Author` keeps them together: the `@validates("name")`
|
||||||
|
hook, `unique_hash`, and `unique_filter`. `as_unique_async` looks a row up with the
|
||||||
|
filter and then constructs with the validator, so if the lookup used the raw name and
|
||||||
|
the insert used the tidy one, every variant spelling would miss the existing row and
|
||||||
|
then collide with it on the unique index.
|
||||||
|
|
||||||
|
A form the rule does not recognise is **left exactly as it was found** — `Dave Thomas,
|
||||||
|
Andy Hunt` is two people in one string, and flipping it would invent a third. Leaving a
|
||||||
|
mess visible beats rewriting it wrongly.
|
||||||
|
|
||||||
|
### The normalized columns are written by validators
|
||||||
|
|
||||||
|
`Book.normalized_title`, `Author.normalized_name` and `Identifier.normalized_value` are
|
||||||
|
derived from `services/matching.py` and kept current by **SQLAlchemy `@validates` hooks
|
||||||
|
on the models**, not by any service. Assigning them directly is always wrong.
|
||||||
|
|
||||||
|
This is deliberate and it is invisible at the call sites: `BookService` writes titles
|
||||||
|
through at least three paths (`to_model_on_create`, `to_model_on_update`, and the
|
||||||
|
`setattr` loop in `_populate_with_unique_relationships`), and a validator is the only
|
||||||
|
thing a fourth cannot bypass. The columns carry plain, **non-unique** btree indexes —
|
||||||
|
two spellings collapsing onto one value is the entire point.
|
||||||
|
|
||||||
|
`services/matching.py` is imported *inside* those validators rather than at module
|
||||||
|
scope: reaching it initialises the `chitai.services` package, which imports the
|
||||||
|
services, which import the models. Keep the local import.
|
||||||
|
|
||||||
|
The validators only fire on write, so a migration that adds one of these columns must
|
||||||
|
backfill existing rows through the same helpers — see the `data_upgrades()` hook in
|
||||||
|
`2026-08-15_add_book_matching_keys_and_duplicate__4358e7d4743a.py`.
|
||||||
|
|
||||||
|
**Changing anything in `services/matching.py` needs a revision that recomputes them.**
|
||||||
|
The keys are derived and already written, so a normalization change silently invalidates
|
||||||
|
every stored row: a book written under the old rules just stops matching one written
|
||||||
|
under the new rules, with nothing to show that anything is wrong. Copy
|
||||||
|
`2026-08-15_recompute_book_matching_keys_ed41acf21270.py`, which exists because
|
||||||
|
`normalize_title` learned to strip compact edition markers (`2E`, `5e`). It is
|
||||||
|
idempotent and safe to re-run.
|
||||||
|
|
||||||
|
## Importing from Calibre
|
||||||
|
|
||||||
|
Two pieces, deliberately separated:
|
||||||
|
|
||||||
|
- **`services/calibre.py`** reads `metadata.db` and the tree beside it. It knows nothing about
|
||||||
|
`Book`, `BookService` or a session, so it is testable without Postgres, and it reports what
|
||||||
|
Calibre wrote rather than what Chitai wants — identifiers come back keyed by `identifiers.type`
|
||||||
|
verbatim. It also unpacks a zipped library (`extract_calibre_archive`).
|
||||||
|
- **`BookService.create_many_from_calibre`** does the ingest, next to the two other ingest paths
|
||||||
|
because it needs the same privates they do (`_reserve_book_path`, `_save_cover_image`,
|
||||||
|
`_screen_for_duplicates`). The CLI in `cli.py` is a thin wrapper over it.
|
||||||
|
- **`services/calibre_import.py`** is lifecycle only — the job registry behind the endpoints:
|
||||||
|
state, progress, cancellation, and a session of its own.
|
||||||
|
|
||||||
|
`docs/calibre-import.md` is the full brief, including the phases not built yet. What matters here:
|
||||||
|
|
||||||
|
- **Files are copied, never moved.** `metadata.db` would go on pointing at files that are gone,
|
||||||
|
which quietly ruins a library somebody still uses. `copy_file` streams rather than using
|
||||||
|
`shutil.copy`, which would block the loop for a 40 MB read.
|
||||||
|
- **The extractors are not run.** This is the one ingest path that trusts its input: Calibre's
|
||||||
|
catalogue is curated and its filenames are truncated to ~42 characters, so `data.name` locates a
|
||||||
|
file and the database carries the metadata. The title is stored verbatim for the same reason —
|
||||||
|
no edition split out, no subtitle guessed.
|
||||||
|
- **The catalogue is copied before it is read**, and the copy is opened read-write. Calibre may be
|
||||||
|
running; opening the live file either sees a torn state or needs to recover a write-ahead log,
|
||||||
|
which read-only access cannot do. `close()` removes the copy in a `finally`, or a failure leaves
|
||||||
|
a catalogue-sized file in the temp directory.
|
||||||
|
- **`check_same_thread=False` plus an `asyncio.Lock`.** Every query runs through
|
||||||
|
`asyncio.to_thread`, which hands out whichever worker is free, so the connection outlives the
|
||||||
|
thread that opened it. The lock is what makes that safe. Removing either one reintroduces
|
||||||
|
`SQLite objects created in a thread can only be used in that same thread`, intermittently —
|
||||||
|
the pool often reuses one thread, so it passes until it does not.
|
||||||
|
- **One book never costs the run.** A failure is recorded in `CalibreImportResult.failed`, the
|
||||||
|
session is rolled back so the next book can use it, and the files that book had already copied
|
||||||
|
are deleted — an orphaned directory would make the next attempt reserve `title (2)` and look as
|
||||||
|
though it had worked. A cover that PIL cannot open costs the cover, not the book.
|
||||||
|
- **`calibre-uuid`, not `uuid`.** `books.uuid` is stable for the life of the row, so it is the
|
||||||
|
durable link back to the source and worth matching on. `uuid` is the name `normalize_identifier`
|
||||||
|
refuses, because an EPUB regenerates one per build.
|
||||||
|
|
||||||
|
### The API takes an uploaded archive; the CLI takes a path
|
||||||
|
|
||||||
|
**`POST /libraries/{id}/imports/calibre/upload`** is the only way in over HTTP. It takes a zipped
|
||||||
|
Calibre library, answers **202** with a job handle, and unpacks into a temp directory the job owns.
|
||||||
|
`GET /libraries/imports/{job_id}` is polled; `DELETE` on the same path stops it.
|
||||||
|
|
||||||
|
There is deliberately **no endpoint that imports from a server path**. A desktop Calibre install is
|
||||||
|
not on the server, and importing from a path the server can already see is a server-side operation
|
||||||
|
— which is what `litestar --app-dir src/chitai/ calibre-import <path> --library <slug>` is for,
|
||||||
|
including its `--dry-run`. Do not add the path endpoint back without being asked: it was built,
|
||||||
|
then removed on purpose.
|
||||||
|
|
||||||
|
- **The registry is in memory, so it assumes one worker process.** That holds today (`litestar run`
|
||||||
|
is single-process, and the consume watcher is already an in-process singleton), but the day
|
||||||
|
`TODO.md`'s "production image runs the development server" item is fixed with a worker count, a
|
||||||
|
poll can land on a worker that never heard of the job. `services/calibre_import.py` says so at
|
||||||
|
the top; an `import_jobs` table is the answer when that happens.
|
||||||
|
- **The job opens its own session.** The request that started it is long gone and its session
|
||||||
|
closed with it.
|
||||||
|
- **Cancelling is not aborting.** A flag is read between books, never during one, so a cancelled
|
||||||
|
import leaves whole books behind and never half of one. `task.cancel()` would abandon a book
|
||||||
|
mid-copy and leave files with no row describing them.
|
||||||
|
- **The job deletes its workspace** — the unpacked archive is a second copy of the whole library,
|
||||||
|
and the books worth keeping have been copied into the library proper by the time it ends. Removed
|
||||||
|
even when the run failed, since nothing will come back for it.
|
||||||
|
- **Extraction refuses** an entry pointing outside the archive (zip slip), an archive that will not
|
||||||
|
fit on disk, and one with no `metadata.db` within three levels. All three answer 400 before a job
|
||||||
|
exists, rather than as a job that reports FAILED a moment later.
|
||||||
|
- **The upload is streamed both sides.** The archive reaches disk in chunks rather than being read
|
||||||
|
whole, and the SvelteKit proxy passes `request.body` through instead of buffering it — see
|
||||||
|
`frontend/AGENTS.md`.
|
||||||
|
|
||||||
|
Things Calibre does that will produce wrong data if you forget them are documented at the top of
|
||||||
|
`services/calibre.py` — the `0101-01-01` date sentinel, `|` for a comma in an author name, the
|
||||||
|
REAL `series_index` that defaults to 1.0 for every book, HTML in `comments`, the views that need
|
||||||
|
SQLite functions Calibre registers from Python, and `books_pages_link` being both recent and
|
||||||
|
usually empty. `tests/calibre_fixtures.py` builds a library exercising all of them.
|
||||||
|
|
||||||
## KOReader hashing
|
## KOReader hashing
|
||||||
|
|
||||||
`services/utils.py` reimplements KOReader's partial-MD5 document identifier: 1 KiB samples at
|
`services/utils.py` reimplements KOReader's partial-MD5 document identifier: 1 KiB samples at
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
"""add file hash index
|
||||||
|
|
||||||
|
Revision ID: e9c2c7e875ae
|
||||||
|
Revises: 6d72d1bbc0ee
|
||||||
|
Create Date: 2026-08-13 14:52:09.341906
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend
|
||||||
|
from advanced_alchemy.types.encrypted_string import PGCryptoBackend
|
||||||
|
from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher
|
||||||
|
from advanced_alchemy.types.password_hash.passlib import PasslibHasher
|
||||||
|
from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher
|
||||||
|
from sqlalchemy import Text # noqa: F401
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"]
|
||||||
|
|
||||||
|
sa.GUID = GUID
|
||||||
|
sa.DateTimeUTC = DateTimeUTC
|
||||||
|
sa.ORA_JSONB = ORA_JSONB
|
||||||
|
sa.EncryptedString = EncryptedString
|
||||||
|
sa.EncryptedText = EncryptedText
|
||||||
|
sa.StoredObject = StoredObject
|
||||||
|
sa.PasswordHash = PasswordHash
|
||||||
|
sa.Argon2Hasher = Argon2Hasher
|
||||||
|
sa.PasslibHasher = PasslibHasher
|
||||||
|
sa.PwdlibHasher = PwdlibHasher
|
||||||
|
sa.FernetBackend = FernetBackend
|
||||||
|
sa.PGCryptoBackend = PGCryptoBackend
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision = 'e9c2c7e875ae'
|
||||||
|
down_revision = '6d72d1bbc0ee'
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
schema_upgrades()
|
||||||
|
data_upgrades()
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
data_downgrades()
|
||||||
|
schema_downgrades()
|
||||||
|
|
||||||
|
def schema_upgrades() -> None:
|
||||||
|
"""schema upgrade migrations go here."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
with op.batch_alter_table('file_metadata', schema=None) as batch_op:
|
||||||
|
batch_op.create_index('ix_file_metadata_hash', ['hash'], unique=False)
|
||||||
|
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
def schema_downgrades() -> None:
|
||||||
|
"""schema downgrade migrations go here."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
with op.batch_alter_table('file_metadata', schema=None) as batch_op:
|
||||||
|
batch_op.drop_index('ix_file_metadata_hash')
|
||||||
|
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
def data_upgrades() -> None:
|
||||||
|
"""Add any optional data upgrade migrations here!"""
|
||||||
|
|
||||||
|
def data_downgrades() -> None:
|
||||||
|
"""Add any optional data downgrade migrations here!"""
|
||||||
+163
@@ -0,0 +1,163 @@
|
|||||||
|
"""add book matching keys and duplicate dismissals
|
||||||
|
|
||||||
|
Revision ID: 4358e7d4743a
|
||||||
|
Revises: e9c2c7e875ae
|
||||||
|
Create Date: 2026-08-15 15:09:02.708914
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend
|
||||||
|
from advanced_alchemy.types.encrypted_string import PGCryptoBackend
|
||||||
|
from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher
|
||||||
|
from advanced_alchemy.types.password_hash.passlib import PasslibHasher
|
||||||
|
from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher
|
||||||
|
from sqlalchemy import Text # noqa: F401
|
||||||
|
|
||||||
|
__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"]
|
||||||
|
|
||||||
|
sa.GUID = GUID
|
||||||
|
sa.DateTimeUTC = DateTimeUTC
|
||||||
|
sa.ORA_JSONB = ORA_JSONB
|
||||||
|
sa.EncryptedString = EncryptedString
|
||||||
|
sa.EncryptedText = EncryptedText
|
||||||
|
sa.StoredObject = StoredObject
|
||||||
|
sa.PasswordHash = PasswordHash
|
||||||
|
sa.Argon2Hasher = Argon2Hasher
|
||||||
|
sa.PasslibHasher = PasslibHasher
|
||||||
|
sa.PwdlibHasher = PwdlibHasher
|
||||||
|
sa.FernetBackend = FernetBackend
|
||||||
|
sa.PGCryptoBackend = PGCryptoBackend
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision = '4358e7d4743a'
|
||||||
|
down_revision = 'e9c2c7e875ae'
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
schema_upgrades()
|
||||||
|
data_upgrades()
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
data_downgrades()
|
||||||
|
schema_downgrades()
|
||||||
|
|
||||||
|
def schema_upgrades() -> None:
|
||||||
|
"""schema upgrade migrations go here."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
op.create_table('duplicate_dismissals',
|
||||||
|
sa.Column('id', sa.BigInteger().with_variant(sa.Integer(), 'sqlite'), nullable=False),
|
||||||
|
sa.Column('book_a_id', sa.BigInteger().with_variant(sa.Integer(), 'sqlite'), nullable=False),
|
||||||
|
sa.Column('book_b_id', sa.BigInteger().with_variant(sa.Integer(), 'sqlite'), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(['book_a_id'], ['books.id'], name=op.f('fk_duplicate_dismissals_book_a_id_books'), ondelete='cascade'),
|
||||||
|
sa.ForeignKeyConstraint(['book_b_id'], ['books.id'], name=op.f('fk_duplicate_dismissals_book_b_id_books'), ondelete='cascade'),
|
||||||
|
sa.PrimaryKeyConstraint('id', name=op.f('pk_duplicate_dismissals')),
|
||||||
|
sa.UniqueConstraint('book_a_id', 'book_b_id', name=op.f('uq_duplicate_dismissals_book_a_id'))
|
||||||
|
)
|
||||||
|
with op.batch_alter_table('duplicate_dismissals', schema=None) as batch_op:
|
||||||
|
batch_op.create_index(batch_op.f('ix_duplicate_dismissals_book_a_id'), ['book_a_id'], unique=False)
|
||||||
|
batch_op.create_index(batch_op.f('ix_duplicate_dismissals_book_b_id'), ['book_b_id'], unique=False)
|
||||||
|
|
||||||
|
# `server_default` so the column can be added to a table that already has rows;
|
||||||
|
# `data_upgrades` fills in the real keys immediately afterwards.
|
||||||
|
with op.batch_alter_table('authors', schema=None) as batch_op:
|
||||||
|
batch_op.add_column(sa.Column('normalized_name', sa.String(), nullable=False, server_default=''))
|
||||||
|
batch_op.create_index(batch_op.f('ix_authors_normalized_name'), ['normalized_name'], unique=False)
|
||||||
|
|
||||||
|
with op.batch_alter_table('books', schema=None) as batch_op:
|
||||||
|
batch_op.add_column(sa.Column('normalized_title', sa.String(), nullable=False, server_default=''))
|
||||||
|
batch_op.create_index(batch_op.f('ix_books_normalized_title'), ['normalized_title'], unique=False)
|
||||||
|
|
||||||
|
with op.batch_alter_table('identifiers', schema=None) as batch_op:
|
||||||
|
batch_op.add_column(sa.Column('normalized_value', sa.String(), nullable=True))
|
||||||
|
batch_op.create_index(batch_op.f('ix_identifiers_normalized_value'), ['normalized_value'], unique=False)
|
||||||
|
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
def schema_downgrades() -> None:
|
||||||
|
"""schema downgrade migrations go here."""
|
||||||
|
# ### commands auto generated by Alembic - please adjust! ###
|
||||||
|
with op.batch_alter_table('identifiers', schema=None) as batch_op:
|
||||||
|
batch_op.drop_index(batch_op.f('ix_identifiers_normalized_value'))
|
||||||
|
batch_op.drop_column('normalized_value')
|
||||||
|
|
||||||
|
with op.batch_alter_table('books', schema=None) as batch_op:
|
||||||
|
batch_op.drop_index(batch_op.f('ix_books_normalized_title'))
|
||||||
|
batch_op.drop_column('normalized_title')
|
||||||
|
|
||||||
|
with op.batch_alter_table('authors', schema=None) as batch_op:
|
||||||
|
batch_op.drop_index(batch_op.f('ix_authors_normalized_name'))
|
||||||
|
batch_op.drop_column('normalized_name')
|
||||||
|
|
||||||
|
with op.batch_alter_table('duplicate_dismissals', schema=None) as batch_op:
|
||||||
|
batch_op.drop_index(batch_op.f('ix_duplicate_dismissals_book_b_id'))
|
||||||
|
batch_op.drop_index(batch_op.f('ix_duplicate_dismissals_book_a_id'))
|
||||||
|
|
||||||
|
op.drop_table('duplicate_dismissals')
|
||||||
|
# ### end Alembic commands ###
|
||||||
|
|
||||||
|
def data_upgrades() -> None:
|
||||||
|
"""
|
||||||
|
Fill the matching keys in for rows that already exist.
|
||||||
|
|
||||||
|
The validators on the models only fire when something is written, so without this
|
||||||
|
every book imported before today is invisible to duplicate detection. Run through
|
||||||
|
the same helpers the validators use, so a backfilled row and a freshly written one
|
||||||
|
are guaranteed to agree.
|
||||||
|
"""
|
||||||
|
from chitai.services.matching import (
|
||||||
|
normalize_author,
|
||||||
|
normalize_identifier,
|
||||||
|
normalize_title,
|
||||||
|
)
|
||||||
|
|
||||||
|
connection = op.get_bind()
|
||||||
|
|
||||||
|
books = connection.execute(sa.text("SELECT id, title FROM books")).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE books SET normalized_title = :key WHERE id = :id",
|
||||||
|
[{"id": id, "key": normalize_title(title)} for id, title in books],
|
||||||
|
)
|
||||||
|
|
||||||
|
authors = connection.execute(sa.text("SELECT id, name FROM authors")).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE authors SET normalized_name = :key WHERE id = :id",
|
||||||
|
[{"id": id, "key": normalize_author(name)} for id, name in authors],
|
||||||
|
)
|
||||||
|
|
||||||
|
identifiers = connection.execute(
|
||||||
|
sa.text("SELECT id, name, value FROM identifiers")
|
||||||
|
).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE identifiers SET normalized_value = :key WHERE id = :id",
|
||||||
|
[
|
||||||
|
{"id": id, "key": normalize_identifier(name, value)}
|
||||||
|
for id, name, value in identifiers
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _apply(connection, statement: str, parameters: list[dict]) -> None:
|
||||||
|
"""Run one update per row, in batches, skipping the work when there are none."""
|
||||||
|
batch_size = 1000
|
||||||
|
|
||||||
|
for start in range(0, len(parameters), batch_size):
|
||||||
|
connection.execute(sa.text(statement), parameters[start : start + batch_size])
|
||||||
|
|
||||||
|
|
||||||
|
def data_downgrades() -> None:
|
||||||
|
"""Add any optional data downgrade migrations here!"""
|
||||||
@@ -0,0 +1,130 @@
|
|||||||
|
"""canonicalize author names
|
||||||
|
|
||||||
|
Revision ID: 49a9e85a0ffc
|
||||||
|
Revises: ed41acf21270
|
||||||
|
Create Date: 2026-08-15 15:59:47.331545
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend
|
||||||
|
from advanced_alchemy.types.encrypted_string import PGCryptoBackend
|
||||||
|
from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher
|
||||||
|
from advanced_alchemy.types.password_hash.passlib import PasslibHasher
|
||||||
|
from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher
|
||||||
|
from sqlalchemy import Text # noqa: F401
|
||||||
|
|
||||||
|
__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"]
|
||||||
|
|
||||||
|
sa.GUID = GUID
|
||||||
|
sa.DateTimeUTC = DateTimeUTC
|
||||||
|
sa.ORA_JSONB = ORA_JSONB
|
||||||
|
sa.EncryptedString = EncryptedString
|
||||||
|
sa.EncryptedText = EncryptedText
|
||||||
|
sa.StoredObject = StoredObject
|
||||||
|
sa.PasswordHash = PasswordHash
|
||||||
|
sa.Argon2Hasher = Argon2Hasher
|
||||||
|
sa.PasslibHasher = PasslibHasher
|
||||||
|
sa.PwdlibHasher = PwdlibHasher
|
||||||
|
sa.FernetBackend = FernetBackend
|
||||||
|
sa.PGCryptoBackend = PGCryptoBackend
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision = '49a9e85a0ffc'
|
||||||
|
down_revision = 'ed41acf21270'
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
schema_upgrades()
|
||||||
|
data_upgrades()
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
data_downgrades()
|
||||||
|
schema_downgrades()
|
||||||
|
|
||||||
|
def schema_upgrades() -> None:
|
||||||
|
"""schema upgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def schema_downgrades() -> None:
|
||||||
|
"""schema downgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def data_upgrades() -> None:
|
||||||
|
"""
|
||||||
|
Rewrite every author into the canonical form, merging the rows that collide.
|
||||||
|
|
||||||
|
`Author.name` is only now guaranteed tidy — until this revision extractors wrote
|
||||||
|
whatever the file said, so one person could hold several rows: "Sam Newman" beside
|
||||||
|
"Newman, Sam;" beside "Sam Newman.epub" (the last from a filename whose extension
|
||||||
|
was never stripped). Each showed up as its own author in the sidebar and its own
|
||||||
|
filter, and no amount of fixing the extractors repairs a row already written.
|
||||||
|
|
||||||
|
Rows that canonicalize onto one name are merged into the lowest id, which keeps
|
||||||
|
whichever row the library has been referring to longest. `book.path` is stored, not
|
||||||
|
derived, so renaming an author moves nothing on disk.
|
||||||
|
"""
|
||||||
|
from chitai.services.matching import format_author_name, normalize_author
|
||||||
|
|
||||||
|
connection = op.get_bind()
|
||||||
|
authors = connection.execute(sa.text("SELECT id, name FROM authors")).fetchall()
|
||||||
|
|
||||||
|
groups: dict[str, list[int]] = {}
|
||||||
|
for id, name in sorted(authors):
|
||||||
|
# A name with nothing left of it after tidying is left exactly as it was:
|
||||||
|
# merging those together would invent one author out of several unrelated
|
||||||
|
# broken rows, which is worse than leaving the mess visible.
|
||||||
|
if canonical := format_author_name(name):
|
||||||
|
groups.setdefault(canonical, []).append(id)
|
||||||
|
|
||||||
|
for canonical, ids in groups.items():
|
||||||
|
winner, losers = ids[0], ids[1:]
|
||||||
|
|
||||||
|
for loser in losers:
|
||||||
|
# A book credited to both rows would otherwise breach the
|
||||||
|
# (book_id, author_id) unique constraint the moment the link is repointed.
|
||||||
|
connection.execute(
|
||||||
|
sa.text(
|
||||||
|
"DELETE FROM book_author_links WHERE author_id = :loser AND book_id IN"
|
||||||
|
" (SELECT book_id FROM book_author_links WHERE author_id = :winner)"
|
||||||
|
),
|
||||||
|
{"loser": loser, "winner": winner},
|
||||||
|
)
|
||||||
|
connection.execute(
|
||||||
|
sa.text(
|
||||||
|
"UPDATE book_author_links SET author_id = :winner"
|
||||||
|
" WHERE author_id = :loser"
|
||||||
|
),
|
||||||
|
{"loser": loser, "winner": winner},
|
||||||
|
)
|
||||||
|
connection.execute(
|
||||||
|
sa.text("DELETE FROM authors WHERE id = :loser"), {"loser": loser}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Only after the losers are gone, or this collides with the unique index.
|
||||||
|
connection.execute(
|
||||||
|
sa.text(
|
||||||
|
"UPDATE authors SET name = :name, normalized_name = :key WHERE id = :id"
|
||||||
|
),
|
||||||
|
{"id": winner, "name": canonical, "key": normalize_author(canonical)},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def data_downgrades() -> None:
|
||||||
|
"""
|
||||||
|
Nothing to undo.
|
||||||
|
|
||||||
|
The rows a merge removed are gone, and the spellings it replaced were never
|
||||||
|
recorded anywhere else — there is nothing to restore them from.
|
||||||
|
"""
|
||||||
@@ -0,0 +1,122 @@
|
|||||||
|
"""recompute book matching keys
|
||||||
|
|
||||||
|
Revision ID: ed41acf21270
|
||||||
|
Revises: 4358e7d4743a
|
||||||
|
Create Date: 2026-08-15 15:44:28.341020
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend
|
||||||
|
from advanced_alchemy.types.encrypted_string import PGCryptoBackend
|
||||||
|
from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher
|
||||||
|
from advanced_alchemy.types.password_hash.passlib import PasslibHasher
|
||||||
|
from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher
|
||||||
|
from sqlalchemy import Text # noqa: F401
|
||||||
|
|
||||||
|
__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"]
|
||||||
|
|
||||||
|
sa.GUID = GUID
|
||||||
|
sa.DateTimeUTC = DateTimeUTC
|
||||||
|
sa.ORA_JSONB = ORA_JSONB
|
||||||
|
sa.EncryptedString = EncryptedString
|
||||||
|
sa.EncryptedText = EncryptedText
|
||||||
|
sa.StoredObject = StoredObject
|
||||||
|
sa.PasswordHash = PasswordHash
|
||||||
|
sa.Argon2Hasher = Argon2Hasher
|
||||||
|
sa.PasslibHasher = PasslibHasher
|
||||||
|
sa.PwdlibHasher = PwdlibHasher
|
||||||
|
sa.FernetBackend = FernetBackend
|
||||||
|
sa.PGCryptoBackend = PGCryptoBackend
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision = 'ed41acf21270'
|
||||||
|
down_revision = '4358e7d4743a'
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
schema_upgrades()
|
||||||
|
data_upgrades()
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
data_downgrades()
|
||||||
|
schema_downgrades()
|
||||||
|
|
||||||
|
def schema_upgrades() -> None:
|
||||||
|
"""schema upgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def schema_downgrades() -> None:
|
||||||
|
"""schema downgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def data_upgrades() -> None:
|
||||||
|
"""
|
||||||
|
Recompute every matching key against the current normalization.
|
||||||
|
|
||||||
|
The keys are derived, so changing a helper in `services/matching.py` silently
|
||||||
|
invalidates every row already written — a book stored under the old rules simply
|
||||||
|
stops matching one stored under the new ones, with nothing to show that anything
|
||||||
|
is wrong. `normalize_title` learned to strip the compact edition markers a cover
|
||||||
|
actually carries ("2E", "5e"), which moved `Building Microservices, 2E` onto the
|
||||||
|
same key as `Building Microservices`.
|
||||||
|
|
||||||
|
Any later change to those helpers wants a revision that looks exactly like this
|
||||||
|
one. It is idempotent and safe to re-run.
|
||||||
|
"""
|
||||||
|
from chitai.services.matching import (
|
||||||
|
normalize_author,
|
||||||
|
normalize_identifier,
|
||||||
|
normalize_title,
|
||||||
|
)
|
||||||
|
|
||||||
|
connection = op.get_bind()
|
||||||
|
|
||||||
|
books = connection.execute(sa.text("SELECT id, title FROM books")).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE books SET normalized_title = :key WHERE id = :id",
|
||||||
|
[{"id": id, "key": normalize_title(title)} for id, title in books],
|
||||||
|
)
|
||||||
|
|
||||||
|
authors = connection.execute(sa.text("SELECT id, name FROM authors")).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE authors SET normalized_name = :key WHERE id = :id",
|
||||||
|
[{"id": id, "key": normalize_author(name)} for id, name in authors],
|
||||||
|
)
|
||||||
|
|
||||||
|
identifiers = connection.execute(
|
||||||
|
sa.text("SELECT id, name, value FROM identifiers")
|
||||||
|
).fetchall()
|
||||||
|
_apply(
|
||||||
|
connection,
|
||||||
|
"UPDATE identifiers SET normalized_value = :key WHERE id = :id",
|
||||||
|
[
|
||||||
|
{"id": id, "key": normalize_identifier(name, value)}
|
||||||
|
for id, name, value in identifiers
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _apply(connection, statement: str, parameters: list[dict]) -> None:
|
||||||
|
"""Run one update per row, in batches, skipping the work when there are none."""
|
||||||
|
batch_size = 1000
|
||||||
|
|
||||||
|
for start in range(0, len(parameters), batch_size):
|
||||||
|
connection.execute(sa.text(statement), parameters[start : start + batch_size])
|
||||||
|
|
||||||
|
|
||||||
|
def data_downgrades() -> None:
|
||||||
|
"""Add any optional data downgrade migrations here!"""
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
"""backfill file content types
|
||||||
|
|
||||||
|
Revision ID: d2d69065ede3
|
||||||
|
Revises: 49a9e85a0ffc
|
||||||
|
Create Date: 2026-08-17 11:37:58.959135
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import warnings
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import sqlalchemy as sa
|
||||||
|
from alembic import op
|
||||||
|
from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend
|
||||||
|
from advanced_alchemy.types.encrypted_string import PGCryptoBackend
|
||||||
|
from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher
|
||||||
|
from advanced_alchemy.types.password_hash.passlib import PasslibHasher
|
||||||
|
from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher
|
||||||
|
from sqlalchemy import Text # noqa: F401
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from collections.abc import Sequence
|
||||||
|
|
||||||
|
__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"]
|
||||||
|
|
||||||
|
sa.GUID = GUID
|
||||||
|
sa.DateTimeUTC = DateTimeUTC
|
||||||
|
sa.ORA_JSONB = ORA_JSONB
|
||||||
|
sa.EncryptedString = EncryptedString
|
||||||
|
sa.EncryptedText = EncryptedText
|
||||||
|
sa.StoredObject = StoredObject
|
||||||
|
sa.PasswordHash = PasswordHash
|
||||||
|
sa.Argon2Hasher = Argon2Hasher
|
||||||
|
sa.PasslibHasher = PasslibHasher
|
||||||
|
sa.PwdlibHasher = PwdlibHasher
|
||||||
|
sa.FernetBackend = FernetBackend
|
||||||
|
sa.PGCryptoBackend = PGCryptoBackend
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision = 'd2d69065ede3'
|
||||||
|
down_revision = '49a9e85a0ffc'
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
schema_upgrades()
|
||||||
|
data_upgrades()
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings("ignore", category=UserWarning)
|
||||||
|
with op.get_context().autocommit_block():
|
||||||
|
data_downgrades()
|
||||||
|
schema_downgrades()
|
||||||
|
|
||||||
|
def schema_upgrades() -> None:
|
||||||
|
"""schema upgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def schema_downgrades() -> None:
|
||||||
|
"""schema downgrade migrations go here."""
|
||||||
|
pass
|
||||||
|
|
||||||
|
def data_upgrades() -> None:
|
||||||
|
"""
|
||||||
|
Name the format of every file whose content type was never worked out.
|
||||||
|
|
||||||
|
`create_many_from_existing_files` filled the column from `mimetypes.guess_type`,
|
||||||
|
which answers None for `.mobi`, `.azw`, `.fb2` and `.lit` — so a consume-directory
|
||||||
|
import of any of those stored a null, and the OPDS acquisition link a reader app
|
||||||
|
uses to decide what it can open carried nothing.
|
||||||
|
|
||||||
|
Both write paths now go through `guess_content_type`, which this uses too, so the
|
||||||
|
formats in its table get named retroactively. A row it still cannot name is **left
|
||||||
|
null** rather than filled with a placeholder: null is the truth, the column is
|
||||||
|
nullable, and the one consumer that needs a string substitutes one itself.
|
||||||
|
Idempotent: it only looks at rows that carry nothing.
|
||||||
|
"""
|
||||||
|
from chitai.services.utils import guess_content_type
|
||||||
|
|
||||||
|
connection = op.get_bind()
|
||||||
|
|
||||||
|
files = connection.execute(
|
||||||
|
sa.text(
|
||||||
|
"SELECT id, path FROM file_metadata "
|
||||||
|
"WHERE content_type IS NULL OR content_type = ''"
|
||||||
|
)
|
||||||
|
).fetchall()
|
||||||
|
|
||||||
|
parameters = [
|
||||||
|
{"id": id, "content_type": content_type}
|
||||||
|
for id, path in files
|
||||||
|
if (content_type := guess_content_type(path)) is not None
|
||||||
|
]
|
||||||
|
|
||||||
|
if not parameters:
|
||||||
|
return
|
||||||
|
|
||||||
|
batch_size = 1000
|
||||||
|
for start in range(0, len(parameters), batch_size):
|
||||||
|
connection.execute(
|
||||||
|
sa.text(
|
||||||
|
"UPDATE file_metadata SET content_type = :content_type WHERE id = :id"
|
||||||
|
),
|
||||||
|
parameters[start : start + batch_size],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def data_downgrades() -> None:
|
||||||
|
"""Add any optional data downgrade migrations here!"""
|
||||||
@@ -36,11 +36,17 @@ dev = [
|
|||||||
"pytest>=8.4.2",
|
"pytest>=8.4.2",
|
||||||
"pytest-asyncio>=1.2.0",
|
"pytest-asyncio>=1.2.0",
|
||||||
"pytest-databases[postgres]>=0.15.0",
|
"pytest-databases[postgres]>=0.15.0",
|
||||||
|
"ruff==0.15.14",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[tool.ruff.lint.per-file-ignores]
|
||||||
|
# The package __init__.py files exist to re-export their modules' public names, so every
|
||||||
|
# import in them is "unused" as far as F401 is concerned.
|
||||||
|
"__init__.py" = ["F401"]
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
asyncio_mode = "auto"
|
asyncio_mode = "auto"
|
||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
filterwarnings = [
|
filterwarnings = [
|
||||||
"ignore::jwt.warnings.InsecureKeyLengthWarning",
|
"ignore::jwt.warnings.InsecureKeyLengthWarning",
|
||||||
]
|
]
|
||||||
|
|||||||
+2
-1
@@ -4,7 +4,8 @@ pkgs.mkShell {
|
|||||||
buildInputs = with pkgs; [
|
buildInputs = with pkgs; [
|
||||||
# Python development environment for Chitai
|
# Python development environment for Chitai
|
||||||
python313Packages.greenlet
|
python313Packages.greenlet
|
||||||
python313Packages.ruff
|
# ruff is a dev dependency in pyproject.toml, not a shell package: CI has no nix, so a
|
||||||
|
# nix-only formatter is one CI cannot run, and two copies could disagree on formatting.
|
||||||
uv
|
uv
|
||||||
|
|
||||||
# postgres database
|
# postgres database
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ from sqlalchemy.ext.asyncio import AsyncSession
|
|||||||
|
|
||||||
|
|
||||||
from chitai import controllers as c
|
from chitai import controllers as c
|
||||||
|
from chitai.cli import CalibreCLIPlugin
|
||||||
from chitai.config import settings
|
from chitai.config import settings
|
||||||
from chitai.database.config import alchemy
|
from chitai.database.config import alchemy
|
||||||
from chitai.database.models.user import User
|
from chitai.database.models.user import User
|
||||||
@@ -71,6 +72,7 @@ oauth2_auth = OAuth2PasswordBearerAuth[User](
|
|||||||
|
|
||||||
watcher_task: asyncio.Task
|
watcher_task: asyncio.Task
|
||||||
|
|
||||||
|
|
||||||
@asynccontextmanager
|
@asynccontextmanager
|
||||||
async def setup_db_connection(app: Litestar) -> AsyncGenerator[None, None]:
|
async def setup_db_connection(app: Litestar) -> AsyncGenerator[None, None]:
|
||||||
# Setup databse
|
# Setup databse
|
||||||
@@ -96,7 +98,7 @@ async def setup_db_connection(app: Litestar) -> AsyncGenerator[None, None]:
|
|||||||
|
|
||||||
@asynccontextmanager
|
@asynccontextmanager
|
||||||
async def setup_directory_watcher(app: Litestar) -> AsyncGenerator[None, None]:
|
async def setup_directory_watcher(app: Litestar) -> AsyncGenerator[None, None]:
|
||||||
|
|
||||||
# Create book covers directory if it does not exist
|
# Create book covers directory if it does not exist
|
||||||
await create_directory(settings.book_cover_path)
|
await create_directory(settings.book_cover_path)
|
||||||
# Create consume directory
|
# Create consume directory
|
||||||
@@ -106,14 +108,17 @@ async def setup_directory_watcher(app: Litestar) -> AsyncGenerator[None, None]:
|
|||||||
book_service = BookService(session=db_session)
|
book_service = BookService(session=db_session)
|
||||||
library_service = LibraryService(session=db_session)
|
library_service = LibraryService(session=db_session)
|
||||||
|
|
||||||
file_watcher = ConsumeDirectoryWatcher(settings.consume_path, library_service, book_service)
|
file_watcher = ConsumeDirectoryWatcher(
|
||||||
|
settings.consume_path, library_service, book_service
|
||||||
|
)
|
||||||
watcher_task = asyncio.create_task(file_watcher.init_watcher())
|
watcher_task = asyncio.create_task(file_watcher.init_watcher())
|
||||||
|
|
||||||
try:
|
try:
|
||||||
yield
|
yield
|
||||||
finally:
|
finally:
|
||||||
watcher_task.cancel()
|
watcher_task.cancel()
|
||||||
|
|
||||||
|
|
||||||
def create_app() -> Litestar:
|
def create_app() -> Litestar:
|
||||||
return Litestar(
|
return Litestar(
|
||||||
route_handlers=[
|
route_handlers=[
|
||||||
@@ -133,7 +138,7 @@ def create_app() -> Litestar:
|
|||||||
],
|
],
|
||||||
exception_handlers=exception_handlers,
|
exception_handlers=exception_handlers,
|
||||||
lifespan=[setup_db_connection, setup_directory_watcher],
|
lifespan=[setup_db_connection, setup_directory_watcher],
|
||||||
plugins=[alchemy],
|
plugins=[alchemy, CalibreCLIPlugin()],
|
||||||
on_app_init=[oauth2_auth.on_app_init],
|
on_app_init=[oauth2_auth.on_app_init],
|
||||||
openapi_config=OpenAPIConfig(
|
openapi_config=OpenAPIConfig(
|
||||||
title="Chitai",
|
title="Chitai",
|
||||||
|
|||||||
@@ -0,0 +1,186 @@
|
|||||||
|
# src/chitai/cli.py
|
||||||
|
|
||||||
|
"""
|
||||||
|
Extra commands on the `litestar` CLI.
|
||||||
|
|
||||||
|
Registered through `CalibreCLIPlugin` in `app.py`, so they run as
|
||||||
|
`litestar --app-dir src/chitai/ calibre-import …` and get the app's own configuration
|
||||||
|
without a second way to load it.
|
||||||
|
|
||||||
|
The Calibre import lives here as well as behind an endpoint because the case it exists
|
||||||
|
for is a one-time migration of a library that may be hundreds of gigabytes. That should
|
||||||
|
not depend on a browser tab staying open.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import click
|
||||||
|
from click import Group
|
||||||
|
from litestar.plugins import CLIPluginProtocol
|
||||||
|
|
||||||
|
from chitai.config import settings
|
||||||
|
from chitai.database.models import Library
|
||||||
|
from chitai.services.book import BookService, CalibreImportProgress, CalibreImportResult
|
||||||
|
from chitai.services.calibre import CalibreLibrary, CalibreLibraryError
|
||||||
|
from chitai.services.library import LibraryService
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreCLIPlugin(CLIPluginProtocol):
|
||||||
|
"""Adds `calibre-import` to the Litestar CLI."""
|
||||||
|
|
||||||
|
def on_cli_init(self, cli: Group) -> None:
|
||||||
|
cli.add_command(calibre_import)
|
||||||
|
|
||||||
|
|
||||||
|
@click.command(name="calibre-import")
|
||||||
|
@click.argument(
|
||||||
|
"source",
|
||||||
|
type=click.Path(exists=True, file_okay=False, path_type=Path),
|
||||||
|
)
|
||||||
|
@click.option(
|
||||||
|
"--library",
|
||||||
|
"library_slug",
|
||||||
|
required=True,
|
||||||
|
help="Slug of the Chitai library to import into.",
|
||||||
|
)
|
||||||
|
@click.option(
|
||||||
|
"--allow-duplicates",
|
||||||
|
is_flag=True,
|
||||||
|
help="Import books whose files the library already holds.",
|
||||||
|
)
|
||||||
|
@click.option(
|
||||||
|
"--dry-run",
|
||||||
|
is_flag=True,
|
||||||
|
help="Read the catalogue and report what it holds, without writing anything.",
|
||||||
|
)
|
||||||
|
def calibre_import(
|
||||||
|
source: Path, library_slug: str, allow_duplicates: bool, dry_run: bool
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Import a Calibre library from SOURCE, the directory holding its metadata.db.
|
||||||
|
|
||||||
|
Files are copied, never moved: the Calibre library is left exactly as it is, and
|
||||||
|
re-running skips whatever is already stored.
|
||||||
|
"""
|
||||||
|
asyncio.run(_import(source, library_slug, allow_duplicates, dry_run))
|
||||||
|
|
||||||
|
|
||||||
|
async def _import(
|
||||||
|
source: Path, library_slug: str, allow_duplicates: bool, dry_run: bool
|
||||||
|
) -> None:
|
||||||
|
try:
|
||||||
|
library_source = CalibreLibrary(source)
|
||||||
|
await library_source.open()
|
||||||
|
except CalibreLibraryError as exc:
|
||||||
|
raise click.ClickException(str(exc)) from exc
|
||||||
|
|
||||||
|
try:
|
||||||
|
if dry_run:
|
||||||
|
await _report(library_source)
|
||||||
|
return
|
||||||
|
|
||||||
|
async with settings.alchemy_config.get_session() as session:
|
||||||
|
library = await _library(session, library_slug)
|
||||||
|
|
||||||
|
result = await BookService(session=session).create_many_from_calibre(
|
||||||
|
library_source,
|
||||||
|
library,
|
||||||
|
allow_duplicates=allow_duplicates,
|
||||||
|
on_progress=_print_progress,
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
await library_source.close()
|
||||||
|
|
||||||
|
_print_summary(result)
|
||||||
|
|
||||||
|
if result.failed:
|
||||||
|
raise SystemExit(1)
|
||||||
|
|
||||||
|
|
||||||
|
async def _library(session: object, slug: str) -> Library:
|
||||||
|
"""Resolve the target library, or explain what the options were."""
|
||||||
|
service = LibraryService(session=session) # type: ignore[arg-type]
|
||||||
|
|
||||||
|
library = await service.get_one_or_none(Library.slug == slug)
|
||||||
|
|
||||||
|
if library is None:
|
||||||
|
available = ", ".join(sorted(item.slug for item in await service.list()))
|
||||||
|
raise click.ClickException(
|
||||||
|
f"No library with slug '{slug}'. Available: {available or 'none'}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# A read-only library is one pointing at a tree Chitai does not own. Copying books
|
||||||
|
# into it would write into somebody else's directory.
|
||||||
|
if library.read_only:
|
||||||
|
raise click.ClickException(
|
||||||
|
f"Library '{slug}' is read-only, so nothing can be imported into it"
|
||||||
|
)
|
||||||
|
|
||||||
|
return library
|
||||||
|
|
||||||
|
|
||||||
|
async def _report(source: CalibreLibrary) -> None:
|
||||||
|
"""Describe the catalogue without touching the database."""
|
||||||
|
books = await source.books()
|
||||||
|
|
||||||
|
click.echo(f"{len(books)} book(s) in {source.root}\n")
|
||||||
|
|
||||||
|
for book in books:
|
||||||
|
authors = ", ".join(book.authors) or "unknown author"
|
||||||
|
formats = ", ".join(file.format for file in book.files) or "no files"
|
||||||
|
click.echo(f" #{book.calibre_id:<6} {book.title}")
|
||||||
|
click.echo(f" {'':<7} {authors} · {formats}")
|
||||||
|
|
||||||
|
missing = [
|
||||||
|
book
|
||||||
|
for book in books
|
||||||
|
if any(not file.path.is_file() for file in book.files) or not book.files
|
||||||
|
]
|
||||||
|
|
||||||
|
if missing:
|
||||||
|
click.echo(
|
||||||
|
f"\n{len(missing)} book(s) have files the catalogue lists "
|
||||||
|
"but disk does not:"
|
||||||
|
)
|
||||||
|
for book in missing:
|
||||||
|
click.echo(f" #{book.calibre_id} {book.title}")
|
||||||
|
|
||||||
|
|
||||||
|
def _print_progress(progress: CalibreImportProgress) -> None:
|
||||||
|
marker = {"created": "+", "skipped": "-", "failed": "!"}.get(progress.outcome, " ")
|
||||||
|
detail = f" ({progress.detail})" if progress.detail else ""
|
||||||
|
|
||||||
|
click.echo(
|
||||||
|
f"[{progress.processed:>5}/{progress.total}] {marker} {progress.title}{detail}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _print_summary(result: CalibreImportResult) -> None:
|
||||||
|
click.echo(
|
||||||
|
f"\n{len(result.created)} created, {len(result.skipped)} skipped, "
|
||||||
|
f"{len(result.failed)} failed, of {result.total}."
|
||||||
|
)
|
||||||
|
|
||||||
|
if result.duplicate_files:
|
||||||
|
click.echo(
|
||||||
|
f"{len(result.duplicate_files)} individual file(s) were already stored and "
|
||||||
|
"were left out of books that imported otherwise."
|
||||||
|
)
|
||||||
|
|
||||||
|
for failure in result.failed:
|
||||||
|
click.echo(f" failed #{failure.calibre_id} {failure.title}: {failure.reason}")
|
||||||
|
|
||||||
|
# Imported all the same — a metadata match is a guess, and the duplicates screen is
|
||||||
|
# where these get decided.
|
||||||
|
for possible in result.possible_duplicates:
|
||||||
|
names = ", ".join(
|
||||||
|
f"{candidate.title} (#{candidate.book_id})"
|
||||||
|
for candidate in possible.candidates
|
||||||
|
)
|
||||||
|
click.echo(
|
||||||
|
f" possible duplicate {possible.title} (#{possible.book_id}) "
|
||||||
|
f"may already be in the library as: {names}"
|
||||||
|
)
|
||||||
@@ -1,9 +1,25 @@
|
|||||||
|
from enum import StrEnum
|
||||||
|
|
||||||
from pydantic import Field, PostgresDsn, computed_field
|
from pydantic import Field, PostgresDsn, computed_field
|
||||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||||
from advanced_alchemy.extensions.litestar import (
|
from advanced_alchemy.extensions.litestar import (
|
||||||
SQLAlchemyAsyncConfig,
|
SQLAlchemyAsyncConfig,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateScope(StrEnum):
|
||||||
|
"""How widely an incoming file is compared against what is already stored."""
|
||||||
|
|
||||||
|
LIBRARY = "library"
|
||||||
|
"""Only files in the library being uploaded to count as duplicates."""
|
||||||
|
|
||||||
|
GLOBAL = "global"
|
||||||
|
"""A file already held by any library counts as a duplicate."""
|
||||||
|
|
||||||
|
OFF = "off"
|
||||||
|
"""No duplicate detection at all."""
|
||||||
|
|
||||||
|
|
||||||
class Settings(BaseSettings):
|
class Settings(BaseSettings):
|
||||||
version: str = Field("0.0.1")
|
version: str = Field("0.0.1")
|
||||||
project_name: str = Field("chitai")
|
project_name: str = Field("chitai")
|
||||||
@@ -33,6 +49,14 @@ class Settings(BaseSettings):
|
|||||||
# Path to consume directory
|
# Path to consume directory
|
||||||
consume_path: str = Field("./consume")
|
consume_path: str = Field("./consume")
|
||||||
|
|
||||||
|
# Duplicate detection
|
||||||
|
duplicate_scope: DuplicateScope = Field(DuplicateScope.LIBRARY)
|
||||||
|
|
||||||
|
# Where the consume watcher parks files it refused as duplicates. Must sit
|
||||||
|
# outside `consume_path`, or the watcher picks them straight back up and
|
||||||
|
# tries to resolve the directory name as a library slug.
|
||||||
|
duplicate_path: str = Field("./duplicates")
|
||||||
|
|
||||||
@computed_field
|
@computed_field
|
||||||
@property
|
@property
|
||||||
def postgres_uri(self) -> PostgresDsn:
|
def postgres_uri(self) -> PostgresDsn:
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# src/chitai/controllers/access.py
|
# src/chitai/controllers/access.py
|
||||||
|
|
||||||
# Standard library
|
# Standard library
|
||||||
from typing import Annotated, Any
|
from typing import Annotated
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
|
|||||||
@@ -4,9 +4,8 @@
|
|||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from litestar import Controller, post, get, patch, delete
|
from litestar import Controller, get
|
||||||
from litestar.params import Dependency
|
from litestar.params import Dependency
|
||||||
from litestar.exceptions import HTTPException
|
|
||||||
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
||||||
from advanced_alchemy.service.pagination import OffsetPagination
|
from advanced_alchemy.service.pagination import OffsetPagination
|
||||||
from advanced_alchemy.service import FilterTypeT
|
from advanced_alchemy.service import FilterTypeT
|
||||||
|
|||||||
@@ -12,7 +12,12 @@ from litestar.params import Dependency, Body
|
|||||||
from litestar.enums import RequestEncodingType
|
from litestar.enums import RequestEncodingType
|
||||||
from litestar.response import File, Stream
|
from litestar.response import File, Stream
|
||||||
from litestar.exceptions import HTTPException
|
from litestar.exceptions import HTTPException
|
||||||
from litestar.status_codes import HTTP_400_BAD_REQUEST
|
from litestar.status_codes import (
|
||||||
|
HTTP_200_OK,
|
||||||
|
HTTP_204_NO_CONTENT,
|
||||||
|
HTTP_400_BAD_REQUEST,
|
||||||
|
HTTP_409_CONFLICT,
|
||||||
|
)
|
||||||
from litestar.datastructures import UploadFile
|
from litestar.datastructures import UploadFile
|
||||||
from advanced_alchemy.service.pagination import OffsetPagination
|
from advanced_alchemy.service.pagination import OffsetPagination
|
||||||
from advanced_alchemy.filters import CollectionFilter
|
from advanced_alchemy.filters import CollectionFilter
|
||||||
@@ -23,6 +28,24 @@ from chitai.services import dependencies as deps
|
|||||||
from chitai import schemas as s
|
from chitai import schemas as s
|
||||||
from chitai.database import models as m
|
from chitai.database import models as m
|
||||||
from chitai.services import BookService, BookProgressService
|
from chitai.services import BookService, BookProgressService
|
||||||
|
from chitai.services.book import DuplicateFilesError
|
||||||
|
|
||||||
|
|
||||||
|
def _duplicate_conflict(exc: DuplicateFilesError) -> HTTPException:
|
||||||
|
"""
|
||||||
|
Turn refused files into a 409 the caller can act on.
|
||||||
|
|
||||||
|
The files ride along in `extra` so the client can name them and offer to send them
|
||||||
|
again with `allow_duplicates`, rather than being told only that something clashed.
|
||||||
|
"""
|
||||||
|
return HTTPException(
|
||||||
|
status_code=HTTP_409_CONFLICT,
|
||||||
|
detail="These files are already in the library",
|
||||||
|
extra=[
|
||||||
|
s.DuplicateFileRead.model_validate(duplicate).model_dump(mode="json")
|
||||||
|
for duplicate in exc.duplicates
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class BookController(Controller):
|
class BookController(Controller):
|
||||||
@@ -63,6 +86,7 @@ class BookController(Controller):
|
|||||||
books_service: BookService,
|
books_service: BookService,
|
||||||
library: m.Library,
|
library: m.Library,
|
||||||
data: Annotated[s.BookCreate, Body(media_type=RequestEncodingType.MULTI_PART)],
|
data: Annotated[s.BookCreate, Body(media_type=RequestEncodingType.MULTI_PART)],
|
||||||
|
allow_duplicates: bool = False,
|
||||||
) -> s.BookRead:
|
) -> s.BookRead:
|
||||||
"""
|
"""
|
||||||
Create a new book with metadata and files.
|
Create a new book with metadata and files.
|
||||||
@@ -73,6 +97,10 @@ class BookController(Controller):
|
|||||||
Path Parameters:
|
Path Parameters:
|
||||||
library_id: The ID of the library the book belongs to.
|
library_id: The ID of the library the book belongs to.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
allow_duplicates: If True, store the files even if the library already
|
||||||
|
holds them.
|
||||||
|
|
||||||
Request Body:
|
Request Body:
|
||||||
data: Book creation data including metadata and files.
|
data: Book creation data including metadata and files.
|
||||||
|
|
||||||
@@ -83,9 +111,17 @@ class BookController(Controller):
|
|||||||
Returns:
|
Returns:
|
||||||
The created book as a BookRead schema.
|
The created book as a BookRead schema.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 409 if any of the files is already in the library.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
result = await books_service.create_book(data, library)
|
try:
|
||||||
|
result = await books_service.create_book(
|
||||||
|
data, library, screen_duplicates=not allow_duplicates
|
||||||
|
)
|
||||||
|
except DuplicateFilesError as exc:
|
||||||
|
raise _duplicate_conflict(exc)
|
||||||
|
|
||||||
book = await books_service.get(result.id)
|
book = await books_service.get(result.id)
|
||||||
return books_service.to_schema(book, schema_type=s.BookRead)
|
return books_service.to_schema(book, schema_type=s.BookRead)
|
||||||
|
|
||||||
@@ -97,13 +133,20 @@ class BookController(Controller):
|
|||||||
data: Annotated[
|
data: Annotated[
|
||||||
s.BooksCreateFromFiles, Body(media_type=RequestEncodingType.MULTI_PART)
|
s.BooksCreateFromFiles, Body(media_type=RequestEncodingType.MULTI_PART)
|
||||||
],
|
],
|
||||||
) -> OffsetPagination[s.BookRead]:
|
allow_duplicates: bool = False,
|
||||||
|
) -> s.BooksUploadResult:
|
||||||
"""
|
"""
|
||||||
Create multiple books from uploaded files.
|
Create multiple books from uploaded files.
|
||||||
|
|
||||||
Groups files by directory and creates separate books for each group.
|
Groups files by directory and creates separate books for each group.
|
||||||
Metadata is automatically extracted from the files.
|
Metadata is automatically extracted from the files.
|
||||||
|
|
||||||
|
Files the library already holds are skipped rather than refused, and reported
|
||||||
|
back so the caller can say which ones did not make it in and why.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
allow_duplicates: If True, store every file, even one already held.
|
||||||
|
|
||||||
Request Body:
|
Request Body:
|
||||||
data: Container with list of uploaded files.
|
data: Container with list of uploaded files.
|
||||||
|
|
||||||
@@ -112,19 +155,200 @@ class BookController(Controller):
|
|||||||
library: The library the books belong to.
|
library: The library the books belong to.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Paginated list of created books.
|
The books created, and the files skipped as duplicates.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
results = await books_service.create_many_from_files(data, library)
|
result = await books_service.create_many_from_files(
|
||||||
|
data, library, allow_duplicates=allow_duplicates
|
||||||
|
)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
raise HTTPException(
|
raise HTTPException(
|
||||||
status_code=HTTP_400_BAD_REQUEST, detail="Must upload at least one file"
|
status_code=HTTP_400_BAD_REQUEST, detail="Must upload at least one file"
|
||||||
)
|
)
|
||||||
|
|
||||||
books = await books_service.list(
|
books = (
|
||||||
CollectionFilter("id", [result.id for result in results])
|
await books_service.list(
|
||||||
|
CollectionFilter("id", [book.id for book in result.books])
|
||||||
|
)
|
||||||
|
if result.books
|
||||||
|
else []
|
||||||
)
|
)
|
||||||
return books_service.to_schema(books, schema_type=s.BookRead)
|
|
||||||
|
return s.BooksUploadResult(
|
||||||
|
created=[
|
||||||
|
books_service.to_schema(book, schema_type=s.BookRead) for book in books
|
||||||
|
],
|
||||||
|
skipped=[
|
||||||
|
s.DuplicateFileRead.model_validate(duplicate)
|
||||||
|
for duplicate in result.duplicates
|
||||||
|
],
|
||||||
|
possible_duplicates=[
|
||||||
|
s.PossibleDuplicateRead.model_validate(possible)
|
||||||
|
for possible in result.possible_duplicates
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
@get(path="duplicate-books")
|
||||||
|
async def list_duplicate_books(
|
||||||
|
self, books_service: BookService, library: m.Library
|
||||||
|
) -> list[s.DuplicateBookGroupRead]:
|
||||||
|
"""
|
||||||
|
Report books already in the library that look like copies of one another.
|
||||||
|
|
||||||
|
The import-time check only ever sees what is arriving, so this is what covers
|
||||||
|
a collection someone already has. Matching is on metadata and therefore a
|
||||||
|
guess: a group is a question for the reader, not a verdict.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
library_id: The library to review.
|
||||||
|
|
||||||
|
Injected Dependencies:
|
||||||
|
books_service: The book service for database operations.
|
||||||
|
library: The library to review.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
One entry per group of two or more books. Empty when there is nothing to
|
||||||
|
review, or when duplicate detection is switched off.
|
||||||
|
"""
|
||||||
|
groups = await books_service.find_duplicate_book_groups(library)
|
||||||
|
|
||||||
|
return [
|
||||||
|
s.DuplicateBookGroupRead(
|
||||||
|
books=[s.DuplicateBookRead.model_validate(book) for book in group]
|
||||||
|
)
|
||||||
|
for group in groups
|
||||||
|
]
|
||||||
|
|
||||||
|
@post(path="merge")
|
||||||
|
async def merge_books(
|
||||||
|
self, books_service: BookService, library: m.Library, data: s.BookMerge
|
||||||
|
) -> s.BookRead:
|
||||||
|
"""
|
||||||
|
Fold several books into one and delete the records folded in.
|
||||||
|
|
||||||
|
The survivor keeps its id, so links and bookmarks still resolve. Files, reading
|
||||||
|
progress, shelves, tags and unheld identifiers move onto it; metadata is only
|
||||||
|
changed by what `metadata` names, because choosing between two titles is the
|
||||||
|
reader's judgement rather than this endpoint's.
|
||||||
|
|
||||||
|
Nothing is removed from disk — a wrong merge should cost metadata that can be
|
||||||
|
retyped, not a book.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
library_id: The library the books belong to.
|
||||||
|
|
||||||
|
Request Body:
|
||||||
|
data: The survivor, the books to fold in, and the resolved metadata.
|
||||||
|
|
||||||
|
Injected Dependencies:
|
||||||
|
books_service: The book service for database operations.
|
||||||
|
library: The library the books belong to.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The surviving book.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 400 if fewer than two distinct books were named, one is
|
||||||
|
unknown, or they do not all belong to one library.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
book = await books_service.merge_books(
|
||||||
|
data.survivor_id,
|
||||||
|
data.merged_ids,
|
||||||
|
library,
|
||||||
|
metadata=data.metadata.model_dump(exclude_unset=True)
|
||||||
|
if data.metadata
|
||||||
|
else None,
|
||||||
|
)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise HTTPException(status_code=HTTP_400_BAD_REQUEST, detail=str(exc))
|
||||||
|
|
||||||
|
return books_service.to_schema(book, schema_type=s.BookRead)
|
||||||
|
|
||||||
|
@post(path="duplicate-books/dismissals", status_code=HTTP_204_NO_CONTENT)
|
||||||
|
async def dismiss_duplicate_books(
|
||||||
|
self, books_service: BookService, data: s.DuplicateDismissal
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Record that two books are not the same book.
|
||||||
|
|
||||||
|
Without this the review screen proposes the same wrong pair forever, which is
|
||||||
|
how a reader learns to stop looking at it.
|
||||||
|
|
||||||
|
Request Body:
|
||||||
|
data: The two book IDs. Order does not matter.
|
||||||
|
|
||||||
|
Injected Dependencies:
|
||||||
|
books_service: The book service for database operations.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 400 if the two IDs are the same or either book is unknown.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
await books_service.dismiss_duplicates(data.book_a_id, data.book_b_id)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise HTTPException(status_code=HTTP_400_BAD_REQUEST, detail=str(exc))
|
||||||
|
|
||||||
|
@delete(path="duplicate-books/dismissals")
|
||||||
|
async def restore_duplicate_books(
|
||||||
|
self, books_service: BookService, book_a_id: int, book_b_id: int
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Undo a dismissal, so the pair is proposed again.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
book_a_id: One of the two books.
|
||||||
|
book_b_id: The other. Order does not matter.
|
||||||
|
|
||||||
|
Injected Dependencies:
|
||||||
|
books_service: The book service for database operations.
|
||||||
|
"""
|
||||||
|
await books_service.restore_duplicates(book_a_id, book_b_id)
|
||||||
|
|
||||||
|
# A question, not a change: 200 rather than the 201 a POST would default to.
|
||||||
|
@post(path="duplicate-files", status_code=HTTP_200_OK)
|
||||||
|
async def check_duplicate_files(
|
||||||
|
self,
|
||||||
|
books_service: BookService,
|
||||||
|
library: m.Library,
|
||||||
|
data: list[s.FileFingerprint],
|
||||||
|
) -> list[s.DuplicateFileRead]:
|
||||||
|
"""
|
||||||
|
Report which of the given files the library already holds.
|
||||||
|
|
||||||
|
Lets a client ask before it uploads anything, which is the difference between
|
||||||
|
re-sending a folder of books and re-sending twelve kilobytes of hashes.
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
library_id: The library to check against.
|
||||||
|
|
||||||
|
Request Body:
|
||||||
|
data: Hash and size for each file, optionally with the name to echo back.
|
||||||
|
|
||||||
|
Injected Dependencies:
|
||||||
|
books_service: The book service for database operations.
|
||||||
|
library: The library to check against.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
One entry per submitted file that is already stored. Files that are not
|
||||||
|
are absent.
|
||||||
|
"""
|
||||||
|
matches = await books_service.find_duplicate_files(
|
||||||
|
((item.hash, item.size) for item in data), library
|
||||||
|
)
|
||||||
|
|
||||||
|
return [
|
||||||
|
s.DuplicateFileRead(
|
||||||
|
filename=item.filename or match.filename,
|
||||||
|
hash=match.hash,
|
||||||
|
size=match.size,
|
||||||
|
library_id=match.library_id,
|
||||||
|
book_id=match.book_id,
|
||||||
|
book_title=match.book_title,
|
||||||
|
)
|
||||||
|
for item in data
|
||||||
|
if (match := matches.get((item.hash, item.size))) is not None
|
||||||
|
]
|
||||||
|
|
||||||
@get(path="/{book_id:int}")
|
@get(path="/{book_id:int}")
|
||||||
async def get_book_by_id(
|
async def get_book_by_id(
|
||||||
@@ -303,13 +527,20 @@ class BookController(Controller):
|
|||||||
],
|
],
|
||||||
library: m.Library,
|
library: m.Library,
|
||||||
books_service: BookService,
|
books_service: BookService,
|
||||||
|
allow_duplicates: bool = False,
|
||||||
) -> s.BookRead:
|
) -> s.BookRead:
|
||||||
"""
|
"""
|
||||||
Add files to an existing book.
|
Add files to an existing book.
|
||||||
|
|
||||||
|
A file the book already carries is ignored, so re-sending one is harmless.
|
||||||
|
|
||||||
Path Parameters:
|
Path Parameters:
|
||||||
book_id: The ID of the book to modify
|
book_id: The ID of the book to modify
|
||||||
|
|
||||||
|
Query Parameters:
|
||||||
|
allow_duplicates: If True, store the files even if the library already
|
||||||
|
holds them.
|
||||||
|
|
||||||
Request Body:
|
Request Body:
|
||||||
files: The files to add to the book
|
files: The files to add to the book
|
||||||
|
|
||||||
@@ -320,9 +551,17 @@ class BookController(Controller):
|
|||||||
Returns:
|
Returns:
|
||||||
The modified book
|
The modified book
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 409 if a file is already stored under a different book.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
await books_service.add_files(book_id, data, library)
|
try:
|
||||||
|
await books_service.add_files(
|
||||||
|
book_id, data, library, allow_duplicates=allow_duplicates
|
||||||
|
)
|
||||||
|
except DuplicateFilesError as exc:
|
||||||
|
raise _duplicate_conflict(exc)
|
||||||
|
|
||||||
book = await books_service.get(book_id)
|
book = await books_service.get(book_id)
|
||||||
return books_service.to_schema(book, schema_type=s.BookRead)
|
return books_service.to_schema(book, schema_type=s.BookRead)
|
||||||
|
|
||||||
|
|||||||
@@ -8,47 +8,51 @@ from litestar import Controller, post, get, delete
|
|||||||
from litestar.di import Provide
|
from litestar.di import Provide
|
||||||
from chitai.services import dependencies as deps
|
from chitai.services import dependencies as deps
|
||||||
|
|
||||||
class DeviceController(Controller):
|
|
||||||
""" Controller for managing KOReader devices."""
|
|
||||||
|
|
||||||
dependencies = {
|
class DeviceController(Controller):
|
||||||
"device_service": Provide(deps.provide_kosync_device_service)
|
"""Controller for managing KOReader devices."""
|
||||||
}
|
|
||||||
|
dependencies = {"device_service": Provide(deps.provide_kosync_device_service)}
|
||||||
|
|
||||||
path = "/devices"
|
path = "/devices"
|
||||||
|
|
||||||
|
|
||||||
@get()
|
@get()
|
||||||
async def get_devices(self, device_service: KosyncDeviceService, current_user: User) -> OffsetPagination[KosyncDeviceRead]:
|
async def get_devices(
|
||||||
""" Return a list of all the user's devices."""
|
self, device_service: KosyncDeviceService, current_user: User
|
||||||
devices = await device_service.list(
|
) -> OffsetPagination[KosyncDeviceRead]:
|
||||||
KosyncDevice.user_id == current_user.id
|
"""Return a list of all the user's devices."""
|
||||||
)
|
devices = await device_service.list(KosyncDevice.user_id == current_user.id)
|
||||||
return device_service.to_schema(devices, schema_type=KosyncDeviceRead)
|
return device_service.to_schema(devices, schema_type=KosyncDeviceRead)
|
||||||
|
|
||||||
@post()
|
@post()
|
||||||
async def create_device(self, data: KosyncDeviceCreate, device_service: KosyncDeviceService, current_user: User) -> KosyncDeviceRead:
|
async def create_device(
|
||||||
device = await device_service.create({
|
self,
|
||||||
'name': data.name,
|
data: KosyncDeviceCreate,
|
||||||
'user_id': current_user.id
|
device_service: KosyncDeviceService,
|
||||||
})
|
current_user: User,
|
||||||
|
) -> KosyncDeviceRead:
|
||||||
|
device = await device_service.create(
|
||||||
|
{"name": data.name, "user_id": current_user.id}
|
||||||
|
)
|
||||||
return device_service.to_schema(device, schema_type=KosyncDeviceRead)
|
return device_service.to_schema(device, schema_type=KosyncDeviceRead)
|
||||||
|
|
||||||
@delete("/{device_id:int}")
|
@delete("/{device_id:int}")
|
||||||
async def delete_device(self, device_id: int, device_service: KosyncDeviceService, current_user: User) -> None:
|
async def delete_device(
|
||||||
|
self, device_id: int, device_service: KosyncDeviceService, current_user: User
|
||||||
|
) -> None:
|
||||||
# Ensure the device exists and is owned by the user
|
# Ensure the device exists and is owned by the user
|
||||||
device = await device_service.get_one(
|
device = await device_service.get_one(
|
||||||
KosyncDevice.id == device_id,
|
KosyncDevice.id == device_id, KosyncDevice.user_id == current_user.id
|
||||||
KosyncDevice.user_id == current_user.id
|
|
||||||
)
|
)
|
||||||
await device_service.delete(device.id)
|
await device_service.delete(device.id)
|
||||||
|
|
||||||
@get("/{device_id:int}/regenerate")
|
@get("/{device_id:int}/regenerate")
|
||||||
async def regenerate_device_api_key(self, device_id: int, device_service: KosyncDeviceService, current_user: User) -> KosyncDeviceRead:
|
async def regenerate_device_api_key(
|
||||||
|
self, device_id: int, device_service: KosyncDeviceService, current_user: User
|
||||||
|
) -> KosyncDeviceRead:
|
||||||
# Ensure the device exists and is owned by the user
|
# Ensure the device exists and is owned by the user
|
||||||
device = await device_service.get_one(
|
device = await device_service.get_one(
|
||||||
KosyncDevice.id == device_id,
|
KosyncDevice.id == device_id, KosyncDevice.user_id == current_user.id
|
||||||
KosyncDevice.user_id == current_user.id
|
|
||||||
)
|
)
|
||||||
updated_device = await device_service.regenerate_api_key(device.id)
|
updated_device = await device_service.regenerate_api_key(device.id)
|
||||||
return device_service.to_schema(updated_device, schema_type=KosyncDeviceRead)
|
return device_service.to_schema(updated_device, schema_type=KosyncDeviceRead)
|
||||||
|
|||||||
@@ -58,10 +58,14 @@ class KosyncController(Controller):
|
|||||||
user: m.User,
|
user: m.User,
|
||||||
) -> KosyncProgressRead:
|
) -> KosyncProgressRead:
|
||||||
"""Return the Kosync progress record associated with the given document."""
|
"""Return the Kosync progress record associated with the given document."""
|
||||||
progress = await kosync_progress_service.get_by_document_hash(user.id, document_id)
|
progress = await kosync_progress_service.get_by_document_hash(
|
||||||
|
user.id, document_id
|
||||||
|
)
|
||||||
|
|
||||||
if not progress:
|
if not progress:
|
||||||
raise HTTPException(status_code=404, detail="No progress found for document")
|
raise HTTPException(
|
||||||
|
status_code=404, detail="No progress found for document"
|
||||||
|
)
|
||||||
|
|
||||||
return KosyncProgressRead(
|
return KosyncProgressRead(
|
||||||
document=progress.document,
|
document=progress.document,
|
||||||
@@ -85,6 +89,3 @@ class KosyncController(Controller):
|
|||||||
detail="User accounts must be created via the main application",
|
detail="User accounts must be created via the main application",
|
||||||
status_code=HTTP_403_FORBIDDEN,
|
status_code=HTTP_403_FORBIDDEN,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,12 +1,20 @@
|
|||||||
# src/chitai/controllers/library.py
|
# src/chitai/controllers/library.py
|
||||||
|
|
||||||
# Standard library
|
# Standard library
|
||||||
|
import asyncio
|
||||||
|
import shutil
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from litestar import Controller, post, get, patch, delete
|
import aiofiles
|
||||||
from litestar.params import Dependency
|
from aiofiles import os as aios
|
||||||
|
from litestar import Controller, post, get, delete
|
||||||
|
from litestar.enums import RequestEncodingType
|
||||||
|
from litestar.params import Body, Dependency
|
||||||
from litestar.exceptions import HTTPException
|
from litestar.exceptions import HTTPException
|
||||||
|
from litestar.status_codes import HTTP_200_OK, HTTP_202_ACCEPTED
|
||||||
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
||||||
from advanced_alchemy.service.pagination import OffsetPagination
|
from advanced_alchemy.service.pagination import OffsetPagination
|
||||||
from advanced_alchemy.service import FilterTypeT
|
from advanced_alchemy.service import FilterTypeT
|
||||||
@@ -14,10 +22,21 @@ from advanced_alchemy.service import FilterTypeT
|
|||||||
# Local imports
|
# Local imports
|
||||||
from chitai.database import models as m
|
from chitai.database import models as m
|
||||||
from chitai.services import LibraryService
|
from chitai.services import LibraryService
|
||||||
from chitai.schemas.library import LibraryCreate, LibraryRead
|
from chitai.schemas.library import (
|
||||||
|
CalibreArchiveUpload,
|
||||||
|
CalibreImportRead,
|
||||||
|
LibraryCreate,
|
||||||
|
LibraryRead,
|
||||||
|
)
|
||||||
|
from chitai.services.calibre import CalibreLibraryError, extract_calibre_archive
|
||||||
|
from chitai.services.calibre_import import registry
|
||||||
from chitai.services.utils import DirectoryDoesNotExist
|
from chitai.services.utils import DirectoryDoesNotExist
|
||||||
|
|
||||||
|
|
||||||
|
# How much of an uploaded archive is held in memory at a time on its way to disk.
|
||||||
|
UPLOAD_CHUNK_SIZE = 262144 # 256 KiB
|
||||||
|
|
||||||
|
|
||||||
class LibraryController(Controller):
|
class LibraryController(Controller):
|
||||||
"""Controller for managing library operations."""
|
"""Controller for managing library operations."""
|
||||||
|
|
||||||
@@ -70,7 +89,153 @@ class LibraryController(Controller):
|
|||||||
Injected Dependencies:
|
Injected Dependencies:
|
||||||
library_service: The service used to query and return library data.
|
library_service: The service used to query and return library data.
|
||||||
"""
|
"""
|
||||||
results, total = await library_service.list_and_count(*filters, load=[m.Library.books])
|
results, total = await library_service.list_and_count(
|
||||||
|
*filters, load=[m.Library.books]
|
||||||
|
)
|
||||||
return library_service.to_schema(
|
return library_service.to_schema(
|
||||||
results, total, filters, schema_type=LibraryRead
|
results, total, filters, schema_type=LibraryRead
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@post(
|
||||||
|
path="{library_id:int}/imports/calibre/upload",
|
||||||
|
status_code=HTTP_202_ACCEPTED,
|
||||||
|
request_max_body_size=None,
|
||||||
|
)
|
||||||
|
async def upload_calibre_import(
|
||||||
|
self,
|
||||||
|
library_service: LibraryService,
|
||||||
|
library_id: int,
|
||||||
|
data: Annotated[
|
||||||
|
CalibreArchiveUpload, Body(media_type=RequestEncodingType.MULTI_PART)
|
||||||
|
],
|
||||||
|
) -> CalibreImportRead:
|
||||||
|
"""
|
||||||
|
Import a zipped Calibre library that was uploaded rather than named on disk.
|
||||||
|
|
||||||
|
For the case where the library is not on the server: zip the Calibre folder and
|
||||||
|
send it. Unpacked into a temp directory the job owns and deletes when it ends —
|
||||||
|
by which time the books worth keeping have been copied into the library proper.
|
||||||
|
|
||||||
|
The server-folder route stays the one for a very large library. This one has to
|
||||||
|
carry the whole archive over HTTP first.
|
||||||
|
|
||||||
|
Path Parameters:
|
||||||
|
library_id: The library to import into.
|
||||||
|
|
||||||
|
Request Body:
|
||||||
|
data: The `.zip` holding the Calibre library.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The job, already running.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 400 if the archive is not a zip, holds no `metadata.db`,
|
||||||
|
names an entry outside itself, or would not fit on disk; 409 if an
|
||||||
|
import into this library is already running.
|
||||||
|
"""
|
||||||
|
library = await self._importable(library_service, library_id)
|
||||||
|
|
||||||
|
if (running := registry.running_for(library_id)) is not None:
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=409,
|
||||||
|
detail="An import into this library is already running",
|
||||||
|
extra={"job_id": running.id},
|
||||||
|
)
|
||||||
|
|
||||||
|
workspace = Path(await asyncio.to_thread(tempfile.mkdtemp))
|
||||||
|
|
||||||
|
try:
|
||||||
|
archive = workspace / "upload.zip"
|
||||||
|
await data.archive.seek(0)
|
||||||
|
|
||||||
|
async with aiofiles.open(archive, "wb") as destination:
|
||||||
|
while chunk := await data.archive.read(UPLOAD_CHUNK_SIZE):
|
||||||
|
await destination.write(chunk)
|
||||||
|
|
||||||
|
unpacked = workspace / "library"
|
||||||
|
unpacked.mkdir()
|
||||||
|
|
||||||
|
catalogue = await extract_calibre_archive(archive, unpacked)
|
||||||
|
|
||||||
|
# The archive itself is dead weight once unpacked, and the library it
|
||||||
|
# unpacked to can be large.
|
||||||
|
await aios.remove(archive)
|
||||||
|
except CalibreLibraryError as exc:
|
||||||
|
await asyncio.to_thread(shutil.rmtree, workspace, True)
|
||||||
|
raise HTTPException(status_code=400, detail=str(exc))
|
||||||
|
except Exception:
|
||||||
|
await asyncio.to_thread(shutil.rmtree, workspace, True)
|
||||||
|
raise
|
||||||
|
|
||||||
|
job = registry.start(
|
||||||
|
library,
|
||||||
|
catalogue,
|
||||||
|
workspace=workspace,
|
||||||
|
label=data.archive.filename or "uploaded archive",
|
||||||
|
allow_duplicates=data.allow_duplicates,
|
||||||
|
)
|
||||||
|
|
||||||
|
return CalibreImportRead.model_validate(job)
|
||||||
|
|
||||||
|
@get(path="imports/{job_id:str}")
|
||||||
|
async def get_import(self, job_id: str) -> CalibreImportRead:
|
||||||
|
"""
|
||||||
|
Report on an import.
|
||||||
|
|
||||||
|
Polled by the client while a run is going. Jobs are held in memory, so this is
|
||||||
|
answered by the process that started it — see `services/calibre_import.py`.
|
||||||
|
|
||||||
|
Path Parameters:
|
||||||
|
job_id: The job to report on.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 404 if this process holds no such job.
|
||||||
|
"""
|
||||||
|
if (job := registry.get(job_id)) is None:
|
||||||
|
raise HTTPException(status_code=404, detail="No such import")
|
||||||
|
|
||||||
|
return CalibreImportRead.model_validate(job)
|
||||||
|
|
||||||
|
@delete(path="imports/{job_id:str}", status_code=HTTP_200_OK)
|
||||||
|
async def cancel_import(self, job_id: str) -> CalibreImportRead:
|
||||||
|
"""
|
||||||
|
Ask an import to stop after the book it is on.
|
||||||
|
|
||||||
|
Deliberately not an abort: a book abandoned mid-copy would leave files on disk
|
||||||
|
with no row describing them. Whatever it has imported stays imported.
|
||||||
|
|
||||||
|
Path Parameters:
|
||||||
|
job_id: The job to stop.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 404 if this process holds no such job.
|
||||||
|
"""
|
||||||
|
if (job := registry.cancel(job_id)) is None:
|
||||||
|
raise HTTPException(status_code=404, detail="No such import")
|
||||||
|
|
||||||
|
return CalibreImportRead.model_validate(job)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
async def _importable(
|
||||||
|
library_service: LibraryService, library_id: int
|
||||||
|
) -> m.Library:
|
||||||
|
"""
|
||||||
|
The library, if it can be imported into at all.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
HTTPException: 404 if there is no such library, 400 if it is read-only —
|
||||||
|
a read-only library points at a tree Chitai does not own, so copying
|
||||||
|
books into it would write into somebody else's directory.
|
||||||
|
"""
|
||||||
|
library = await library_service.get_one_or_none(m.Library.id == library_id)
|
||||||
|
|
||||||
|
if library is None:
|
||||||
|
raise HTTPException(status_code=404, detail="No such library")
|
||||||
|
|
||||||
|
if library.read_only:
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=400,
|
||||||
|
detail="This library is read-only, so nothing can be imported into it",
|
||||||
|
)
|
||||||
|
|
||||||
|
return library
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
|
|
||||||
from chitai.services import dependencies as deps
|
from chitai.services import dependencies as deps
|
||||||
from chitai.database import models as m
|
from chitai.database import models as m
|
||||||
from chitai.services.author import AuthorService
|
from chitai.services.author import AuthorService
|
||||||
@@ -6,7 +5,15 @@ from chitai.services.filters.author import AuthorLibraryFilter
|
|||||||
from chitai.services.filters.publisher import PublisherLibraryFilter
|
from chitai.services.filters.publisher import PublisherLibraryFilter
|
||||||
from chitai.services.filters.tags import TagLibraryFilter
|
from chitai.services.filters.tags import TagLibraryFilter
|
||||||
from chitai.services.opds.models import Entry, Link, LinkTypes, LinkRelations
|
from chitai.services.opds.models import Entry, Link, LinkTypes, LinkRelations
|
||||||
from chitai.services.opds.opds import create_acquisition_feed, create_navigation_feed, create_library_navigation_feed, create_collection_navigation_feed, create_pagination_links, create_search_link, get_opensearch_document
|
from chitai.services.opds.opds import (
|
||||||
|
create_acquisition_feed,
|
||||||
|
create_navigation_feed,
|
||||||
|
create_library_navigation_feed,
|
||||||
|
create_collection_navigation_feed,
|
||||||
|
create_pagination_links,
|
||||||
|
create_search_link,
|
||||||
|
get_opensearch_document,
|
||||||
|
)
|
||||||
from chitai.services import BookService, ShelfService, LibraryService
|
from chitai.services import BookService, ShelfService, LibraryService
|
||||||
from chitai.services.publisher import PublisherService
|
from chitai.services.publisher import PublisherService
|
||||||
from chitai.services.tag import TagService
|
from chitai.services.tag import TagService
|
||||||
@@ -14,7 +21,7 @@ from litestar import Controller, Request, Response, get
|
|||||||
from litestar.response import File
|
from litestar.response import File
|
||||||
from litestar.di import Provide
|
from litestar.di import Provide
|
||||||
from litestar.exceptions import HTTPException
|
from litestar.exceptions import HTTPException
|
||||||
from advanced_alchemy.filters import CollectionFilter, LimitOffset, OrderBy
|
from advanced_alchemy.filters import CollectionFilter, LimitOffset, OrderBy
|
||||||
from chitai.middleware.basic_auth import basic_auth_mw
|
from chitai.middleware.basic_auth import basic_auth_mw
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
@@ -22,11 +29,10 @@ from litestar.params import Dependency
|
|||||||
from advanced_alchemy.service import FilterTypeT
|
from advanced_alchemy.service import FilterTypeT
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
class OpdsController(Controller):
|
class OpdsController(Controller):
|
||||||
""" Controller for managing OPDS endpoints """
|
"""Controller for managing OPDS endpoints"""
|
||||||
|
|
||||||
middleware=[basic_auth_mw]
|
middleware = [basic_auth_mw]
|
||||||
|
|
||||||
dependencies = {
|
dependencies = {
|
||||||
"user": Provide(deps.provide_user_via_basic_auth),
|
"user": Provide(deps.provide_user_via_basic_auth),
|
||||||
@@ -76,32 +82,30 @@ class OpdsController(Controller):
|
|||||||
title=lib.name,
|
title=lib.name,
|
||||||
href=f"/opds/library/{lib.id}",
|
href=f"/opds/library/{lib.id}",
|
||||||
rel=LinkRelations.SUBSECTION,
|
rel=LinkRelations.SUBSECTION,
|
||||||
type=LinkTypes.NAVIGATION
|
type=LinkTypes.NAVIGATION,
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
) for lib in libraries
|
)
|
||||||
|
for lib in libraries
|
||||||
]
|
]
|
||||||
|
|
||||||
feed = create_navigation_feed(
|
feed = create_navigation_feed(
|
||||||
id="/opds",
|
id="/opds",
|
||||||
title="Root",
|
title="Root",
|
||||||
self_url="/opds",
|
self_url="/opds",
|
||||||
links=[
|
links=[
|
||||||
Link(
|
Link(
|
||||||
rel="search",
|
rel="search",
|
||||||
href="/opds/opensearch",
|
href="/opds/opensearch",
|
||||||
type="application/opensearchdescription+xml",
|
type="application/opensearchdescription+xml",
|
||||||
title="Search books",
|
title="Search books",
|
||||||
)
|
)
|
||||||
],
|
],
|
||||||
entries=entries
|
entries=entries,
|
||||||
)
|
|
||||||
|
|
||||||
return Response(
|
|
||||||
feed,
|
|
||||||
media_type="application/xml"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
return Response(feed, media_type="application/xml")
|
||||||
|
|
||||||
@get("/acquisition")
|
@get("/acquisition")
|
||||||
async def get_acquisition_feed(
|
async def get_acquisition_feed(
|
||||||
self,
|
self,
|
||||||
@@ -109,37 +113,41 @@ class OpdsController(Controller):
|
|||||||
feed_id: str,
|
feed_id: str,
|
||||||
feed_title: str,
|
feed_title: str,
|
||||||
books_service: BookService,
|
books_service: BookService,
|
||||||
book_filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = [],
|
book_filters: Annotated[
|
||||||
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = []
|
list[FilterTypeT], Dependency(skip_validation=True)
|
||||||
|
] = [],
|
||||||
|
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = [],
|
||||||
) -> Response:
|
) -> Response:
|
||||||
|
|
||||||
all_filters = [*filters, *book_filters]
|
all_filters = [*filters, *book_filters]
|
||||||
books, total = await books_service.list_and_count(*all_filters)
|
books, total = await books_service.list_and_count(*all_filters)
|
||||||
|
|
||||||
limit, offset = extract_limit_offset(all_filters)
|
limit, offset = extract_limit_offset(all_filters)
|
||||||
|
|
||||||
links = []
|
links = []
|
||||||
|
|
||||||
# Create pagination links if it is a paginated feed
|
# Create pagination links if it is a paginated feed
|
||||||
if request.query_params.get('paginated'):
|
if request.query_params.get("paginated"):
|
||||||
pagination = create_pagination_links(
|
pagination = create_pagination_links(
|
||||||
request=request,
|
request=request,
|
||||||
total=total,
|
total=total,
|
||||||
limit=limit,
|
limit=limit,
|
||||||
offset=offset,
|
offset=offset,
|
||||||
feed_title=feed_title,
|
feed_title=feed_title,
|
||||||
link_type=LinkTypes.ACQUISITION
|
link_type=LinkTypes.ACQUISITION,
|
||||||
)
|
)
|
||||||
|
|
||||||
links.extend([link for link in [pagination.next_link, pagination.prev_link] if link])
|
links.extend(
|
||||||
|
[link for link in [pagination.next_link, pagination.prev_link] if link]
|
||||||
|
)
|
||||||
|
|
||||||
# Add search link if this is a searchable feed
|
# Add search link if this is a searchable feed
|
||||||
if request.query_params.get('search'):
|
if request.query_params.get("search"):
|
||||||
links.append(create_search_link(request))
|
links.append(create_search_link(request))
|
||||||
|
|
||||||
# Create self URL
|
# Create self URL
|
||||||
self_url = f"{request.url.path}?{urlencode(list(request.query_params.items()), doseq=True)}"
|
self_url = f"{request.url.path}?{urlencode(list(request.query_params.items()), doseq=True)}"
|
||||||
|
|
||||||
feed = create_acquisition_feed(
|
feed = create_acquisition_feed(
|
||||||
id=feed_id,
|
id=feed_id,
|
||||||
title=feed_title,
|
title=feed_title,
|
||||||
@@ -147,28 +155,26 @@ class OpdsController(Controller):
|
|||||||
books=books,
|
books=books,
|
||||||
links=links,
|
links=links,
|
||||||
)
|
)
|
||||||
|
|
||||||
return Response(feed, media_type="application/xml")
|
|
||||||
|
|
||||||
|
return Response(feed, media_type="application/xml")
|
||||||
|
|
||||||
@get("/opensearch")
|
@get("/opensearch")
|
||||||
async def opensearch(self, user: m.User, request: Request) -> Response:
|
async def opensearch(self, user: m.User, request: Request) -> Response:
|
||||||
|
|
||||||
return Response(
|
return Response(
|
||||||
get_opensearch_document(
|
get_opensearch_document(
|
||||||
base_url=f'/opds/search?{urlencode(list(request.query_params.items()), doseq=True)}&'
|
base_url=f"/opds/search?{urlencode(list(request.query_params.items()), doseq=True)}&"
|
||||||
),
|
),
|
||||||
media_type="application/xml"
|
media_type="application/xml",
|
||||||
)
|
)
|
||||||
|
|
||||||
@get("/library/{library_id:int}")
|
@get("/library/{library_id:int}")
|
||||||
async def get_library_feed(self, library: m.Library) -> Response:
|
async def get_library_feed(self, library: m.Library) -> Response:
|
||||||
|
|
||||||
feed = create_library_navigation_feed(library)
|
feed = create_library_navigation_feed(library)
|
||||||
|
|
||||||
return Response(feed, media_type="application/xml")
|
return Response(feed, media_type="application/xml")
|
||||||
|
|
||||||
|
|
||||||
@get("/library/{library_id:int}/{collection_type:str}")
|
@get("/library/{library_id:int}/{collection_type:str}")
|
||||||
async def get_library_collection_feed(
|
async def get_library_collection_feed(
|
||||||
self,
|
self,
|
||||||
@@ -180,39 +186,51 @@ class OpdsController(Controller):
|
|||||||
tag_service: TagService,
|
tag_service: TagService,
|
||||||
publisher_service: PublisherService,
|
publisher_service: PublisherService,
|
||||||
request: Request,
|
request: Request,
|
||||||
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = []
|
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = [],
|
||||||
) -> Response:
|
) -> Response:
|
||||||
|
|
||||||
service_map = {
|
service_map = {
|
||||||
'shelves': (shelf_service, lambda: shelf_service.list_and_count(
|
"shelves": (
|
||||||
*filters,
|
shelf_service,
|
||||||
CollectionFilter("library_id", values=[library.id]),
|
lambda: shelf_service.list_and_count(
|
||||||
OrderBy("name", "asc"),
|
*filters,
|
||||||
m.BookList.user_id == user.id
|
CollectionFilter("library_id", values=[library.id]),
|
||||||
)),
|
OrderBy("name", "asc"),
|
||||||
'tags': (tag_service, lambda: tag_service.list_and_count(
|
m.BookList.user_id == user.id,
|
||||||
*filters,
|
),
|
||||||
TagLibraryFilter(libraries=[library.id]),
|
),
|
||||||
OrderBy("name", "asc"),
|
"tags": (
|
||||||
uniquify=True,
|
tag_service,
|
||||||
)),
|
lambda: tag_service.list_and_count(
|
||||||
'authors': (author_service, lambda: author_service.list_and_count(
|
*filters,
|
||||||
*filters,
|
TagLibraryFilter(libraries=[library.id]),
|
||||||
AuthorLibraryFilter(libraries=[library.id]),
|
OrderBy("name", "asc"),
|
||||||
OrderBy("name", "asc"),
|
uniquify=True,
|
||||||
uniquify=True
|
),
|
||||||
)),
|
),
|
||||||
'publishers': (publisher_service, lambda: publisher_service.list_and_count(
|
"authors": (
|
||||||
*filters,
|
author_service,
|
||||||
PublisherLibraryFilter(libraries=[library.id]),
|
lambda: author_service.list_and_count(
|
||||||
OrderBy("name", "asc"),
|
*filters,
|
||||||
uniquify=True
|
AuthorLibraryFilter(libraries=[library.id]),
|
||||||
))
|
OrderBy("name", "asc"),
|
||||||
|
uniquify=True,
|
||||||
|
),
|
||||||
|
),
|
||||||
|
"publishers": (
|
||||||
|
publisher_service,
|
||||||
|
lambda: publisher_service.list_and_count(
|
||||||
|
*filters,
|
||||||
|
PublisherLibraryFilter(libraries=[library.id]),
|
||||||
|
OrderBy("name", "asc"),
|
||||||
|
uniquify=True,
|
||||||
|
),
|
||||||
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
if collection_type not in service_map:
|
if collection_type not in service_map:
|
||||||
raise HTTPException(status_code=404, detail="Collection type not found")
|
raise HTTPException(status_code=404, detail="Collection type not found")
|
||||||
|
|
||||||
_, fetch_items = service_map[collection_type]
|
_, fetch_items = service_map[collection_type]
|
||||||
items, total = await fetch_items()
|
items, total = await fetch_items()
|
||||||
links = []
|
links = []
|
||||||
@@ -220,41 +238,40 @@ class OpdsController(Controller):
|
|||||||
# Create pagination links if it is a paginated feed
|
# Create pagination links if it is a paginated feed
|
||||||
limit, offset = extract_limit_offset(filters)
|
limit, offset = extract_limit_offset(filters)
|
||||||
|
|
||||||
if request.query_params.get('paginated'):
|
if request.query_params.get("paginated"):
|
||||||
pagination = create_pagination_links(
|
pagination = create_pagination_links(
|
||||||
request=request,
|
request=request,
|
||||||
total=total,
|
total=total,
|
||||||
limit=limit,
|
limit=limit,
|
||||||
offset=offset,
|
offset=offset,
|
||||||
feed_title=collection_type,
|
feed_title=collection_type,
|
||||||
link_type=LinkTypes.ACQUISITION
|
link_type=LinkTypes.ACQUISITION,
|
||||||
)
|
)
|
||||||
|
|
||||||
links.extend([link for link in [pagination.next_link, pagination.prev_link] if link])
|
links.extend(
|
||||||
|
[link for link in [pagination.next_link, pagination.prev_link] if link]
|
||||||
|
)
|
||||||
|
|
||||||
feed = create_collection_navigation_feed(library, collection_type, items, links)
|
feed = create_collection_navigation_feed(library, collection_type, items, links)
|
||||||
return Response(feed, media_type="application/xml")
|
return Response(feed, media_type="application/xml")
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@get("/search")
|
@get("/search")
|
||||||
async def search_books(
|
async def search_books(
|
||||||
self, books_service: BookService,
|
self,
|
||||||
|
books_service: BookService,
|
||||||
request: Request,
|
request: Request,
|
||||||
book_filters: Annotated[
|
book_filters: Annotated[
|
||||||
list[FilterTypeT], Dependency(skip_validation=True)
|
list[FilterTypeT], Dependency(skip_validation=True)
|
||||||
] = [],
|
] = [],
|
||||||
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = []
|
filters: Annotated[list[FilterTypeT], Dependency(skip_validation=True)] = [],
|
||||||
) -> Response:
|
) -> Response:
|
||||||
|
|
||||||
filters = [*filters, *book_filters]
|
filters = [*filters, *book_filters]
|
||||||
|
|
||||||
books, total = await books_service.list_and_count(
|
books, total = await books_service.list_and_count(*filters)
|
||||||
*filters
|
|
||||||
)
|
|
||||||
|
|
||||||
limit, offset = extract_limit_offset(filters)
|
limit, offset = extract_limit_offset(filters)
|
||||||
|
|
||||||
# Create pagination links
|
# Create pagination links
|
||||||
pagination = create_pagination_links(
|
pagination = create_pagination_links(
|
||||||
request=request,
|
request=request,
|
||||||
@@ -262,37 +279,34 @@ class OpdsController(Controller):
|
|||||||
limit=limit,
|
limit=limit,
|
||||||
offset=offset,
|
offset=offset,
|
||||||
feed_title="Search Results",
|
feed_title="Search Results",
|
||||||
link_type=LinkTypes.ACQUISITION
|
link_type=LinkTypes.ACQUISITION,
|
||||||
)
|
)
|
||||||
|
|
||||||
links = [link for link in [pagination.next_link, pagination.prev_link] if link]
|
links = [link for link in [pagination.next_link, pagination.prev_link] if link]
|
||||||
|
|
||||||
catalog_xml = create_acquisition_feed(
|
catalog_xml = create_acquisition_feed(
|
||||||
id=f"/opds/search?q=q",
|
id="/opds/search?q=q",
|
||||||
title="Search results",
|
title="Search results",
|
||||||
url=f"/opds/search?q=q",
|
url="/opds/search?q=q",
|
||||||
books=books,
|
books=books,
|
||||||
links=links
|
links=links,
|
||||||
)
|
)
|
||||||
|
|
||||||
return Response(catalog_xml, media_type="application/xml")
|
return Response(catalog_xml, media_type="application/xml")
|
||||||
|
|
||||||
@get(path="download/{book_id:int}/{file_id:int}")
|
@get(path="download/{book_id:int}/{file_id:int}")
|
||||||
async def get_file(
|
async def get_file(
|
||||||
self, book_id: int, file_id: int, books_service: BookService
|
self, book_id: int, file_id: int, books_service: BookService
|
||||||
) -> File:
|
) -> File:
|
||||||
|
|
||||||
return await books_service.get_file(book_id, file_id)
|
return await books_service.get_file(book_id, file_id)
|
||||||
|
|
||||||
|
|
||||||
def extract_limit_offset(filters: list[FilterTypeT]) -> tuple[int, int]:
|
def extract_limit_offset(filters: list[FilterTypeT]) -> tuple[int, int]:
|
||||||
"""Extract page size and offset from filters"""
|
"""Extract page size and offset from filters"""
|
||||||
limit_offset_filter = next(
|
limit_offset_filter = next((f for f in filters if isinstance(f, LimitOffset)), None)
|
||||||
(f for f in filters if isinstance(f, LimitOffset)),
|
|
||||||
None
|
|
||||||
)
|
|
||||||
|
|
||||||
if limit_offset_filter:
|
if limit_offset_filter:
|
||||||
return limit_offset_filter.limit, limit_offset_filter.offset
|
return limit_offset_filter.limit, limit_offset_filter.offset
|
||||||
|
|
||||||
raise ValueError("LimitOffset filter not found")
|
raise ValueError("LimitOffset filter not found")
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from litestar import Controller, post, get, patch, delete
|
from litestar import Controller, get
|
||||||
from litestar.params import Dependency
|
from litestar.params import Dependency
|
||||||
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
||||||
from advanced_alchemy.service.pagination import OffsetPagination
|
from advanced_alchemy.service.pagination import OffsetPagination
|
||||||
|
|||||||
@@ -70,7 +70,9 @@ class BookshelfController(Controller):
|
|||||||
filters.append(CollectionFilter("library_id", values=libraries))
|
filters.append(CollectionFilter("library_id", values=libraries))
|
||||||
filters.append(m.BookList.user_id == current_user.id)
|
filters.append(m.BookList.user_id == current_user.id)
|
||||||
|
|
||||||
results, total = await shelf_service.list_and_count(*filters, load=[selectinload(m.BookList.book_links)])
|
results, total = await shelf_service.list_and_count(
|
||||||
|
*filters, load=[selectinload(m.BookList.book_links)]
|
||||||
|
)
|
||||||
return shelf_service.to_schema(results, total, filters, schema_type=ShelfRead)
|
return shelf_service.to_schema(results, total, filters, schema_type=ShelfRead)
|
||||||
|
|
||||||
@post()
|
@post()
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from litestar import Controller, post, get, patch, delete
|
from litestar import Controller, get
|
||||||
from litestar.params import Dependency
|
from litestar.params import Dependency
|
||||||
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
from advanced_alchemy.extensions.litestar.providers import create_service_dependencies
|
||||||
from advanced_alchemy.service.pagination import OffsetPagination
|
from advanced_alchemy.service.pagination import OffsetPagination
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ from .book import Book, Identifier, FileMetadata
|
|||||||
from .book_list import BookList, BookListLink
|
from .book_list import BookList, BookListLink
|
||||||
from .book_progress import BookProgress
|
from .book_progress import BookProgress
|
||||||
from .book_series import BookSeries
|
from .book_series import BookSeries
|
||||||
|
from .duplicate_dismissal import DuplicateDismissal
|
||||||
from .kosync_device import KosyncDevice
|
from .kosync_device import KosyncDevice
|
||||||
from .kosync_progress import KosyncProgress
|
from .kosync_progress import KosyncProgress
|
||||||
from .library import Library
|
from .library import Library
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ from sqlalchemy import ColumnElement, ForeignKey, UniqueConstraint
|
|||||||
from sqlalchemy.orm import Mapped
|
from sqlalchemy.orm import Mapped
|
||||||
from sqlalchemy.orm import mapped_column
|
from sqlalchemy.orm import mapped_column
|
||||||
from sqlalchemy.orm import relationship
|
from sqlalchemy.orm import relationship
|
||||||
|
from sqlalchemy.orm import validates
|
||||||
|
|
||||||
from advanced_alchemy.base import BigIntAuditBase, BigIntBase
|
from advanced_alchemy.base import BigIntAuditBase, BigIntBase
|
||||||
from advanced_alchemy.mixins import UniqueMixin
|
from advanced_alchemy.mixins import UniqueMixin
|
||||||
@@ -16,18 +17,55 @@ if TYPE_CHECKING:
|
|||||||
class Author(BigIntAuditBase, UniqueMixin):
|
class Author(BigIntAuditBase, UniqueMixin):
|
||||||
__tablename__ = "authors"
|
__tablename__ = "authors"
|
||||||
|
|
||||||
|
# Always the canonical form — see `_canonicalize`. Extractors hand over whatever
|
||||||
|
# the file happened to say: "Newman, Sam;" from a `DC:creator` list, or
|
||||||
|
# "Sam Newman.epub" from a filename. Storing those verbatim is how one person ends
|
||||||
|
# up as several rows in the sidebar.
|
||||||
name: Mapped[str] = mapped_column(unique=True, index=True)
|
name: Mapped[str] = mapped_column(unique=True, index=True)
|
||||||
|
|
||||||
|
# Kept current by `_canonicalize` too — never assign it directly. Not unique: two
|
||||||
|
# spellings that survive canonicalization, "Steve Mcconnell" and "Steve McConnell",
|
||||||
|
# are still one person to a reader, which is what this column exists to express.
|
||||||
|
normalized_name: Mapped[str] = mapped_column(default="", index=True)
|
||||||
|
|
||||||
description: Mapped[Optional[str]]
|
description: Mapped[Optional[str]]
|
||||||
|
|
||||||
|
@validates("name")
|
||||||
|
def _canonicalize(self, _key: str, name: str) -> str:
|
||||||
|
"""
|
||||||
|
Store the tidied name, and derive the matching key from it.
|
||||||
|
|
||||||
|
A validator so no write can get around it, and `unique_hash` / `unique_filter`
|
||||||
|
below tidy the same way so `as_unique_async` looks the row up under the name it
|
||||||
|
would actually be stored as. All three have to agree: if the lookup used the
|
||||||
|
raw name and the insert used the tidy one, every variant spelling would miss
|
||||||
|
the existing row and then collide with it on the unique index.
|
||||||
|
"""
|
||||||
|
# Imported here rather than at module scope: `chitai.services.matching` cannot
|
||||||
|
# be reached without initialising the `chitai.services` package, which imports
|
||||||
|
# the services, which import this module.
|
||||||
|
from chitai.services.matching import format_author_name, normalize_author
|
||||||
|
|
||||||
|
name = format_author_name(name)
|
||||||
|
self.normalized_name = normalize_author(name)
|
||||||
|
return name
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _tidy(cls, name: str) -> str:
|
||||||
|
"""The name as `_canonicalize` would store it."""
|
||||||
|
from chitai.services.matching import format_author_name
|
||||||
|
|
||||||
|
return format_author_name(name)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def unique_hash(cls, name: str) -> Hashable:
|
def unique_hash(cls, name: str) -> Hashable:
|
||||||
"""Generate a unique hash for deduplication."""
|
"""Generate a unique hash for deduplication."""
|
||||||
return name
|
return cls._tidy(name)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def unique_filter(cls, name: str) -> ColumnElement[bool]:
|
def unique_filter(cls, name: str) -> ColumnElement[bool]:
|
||||||
"""SQL filter for finding existing records."""
|
"""SQL filter for finding existing records."""
|
||||||
return cls.name == name
|
return cls.name == cls._tidy(name)
|
||||||
|
|
||||||
def __repr__(self) -> str:
|
def __repr__(self) -> str:
|
||||||
return f"Author({self.name!r})"
|
return f"Author({self.name!r})"
|
||||||
|
|||||||
@@ -2,13 +2,10 @@ from datetime import date
|
|||||||
from typing import TYPE_CHECKING, Any, Optional
|
from typing import TYPE_CHECKING, Any, Optional
|
||||||
|
|
||||||
from sqlalchemy import Index, ForeignKey, UniqueConstraint
|
from sqlalchemy import Index, ForeignKey, UniqueConstraint
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship, validates
|
||||||
from sqlalchemy.orm import mapped_column
|
|
||||||
from sqlalchemy.orm import relationship
|
|
||||||
from sqlalchemy.ext.orderinglist import ordering_list
|
from sqlalchemy.ext.orderinglist import ordering_list
|
||||||
from sqlalchemy.ext.associationproxy import association_proxy
|
from sqlalchemy.ext.associationproxy import association_proxy
|
||||||
from sqlalchemy.ext.associationproxy import AssociationProxy
|
from sqlalchemy.ext.associationproxy import AssociationProxy
|
||||||
from sqlalchemy.orm.collections import attribute_keyed_dict
|
|
||||||
|
|
||||||
from advanced_alchemy.base import BigIntAuditBase, BigIntBase
|
from advanced_alchemy.base import BigIntAuditBase, BigIntBase
|
||||||
|
|
||||||
@@ -44,6 +41,13 @@ class Book(BigIntAuditBase):
|
|||||||
library: Mapped["Library"] = relationship(back_populates="books")
|
library: Mapped["Library"] = relationship(back_populates="books")
|
||||||
|
|
||||||
title: Mapped[str]
|
title: Mapped[str]
|
||||||
|
|
||||||
|
# Kept current by `_normalize_title` below — never assign it directly.
|
||||||
|
#
|
||||||
|
# Deliberately not unique: two spellings collapsing onto one value is the whole
|
||||||
|
# point of the column, and a second edition is allowed to exist.
|
||||||
|
normalized_title: Mapped[str] = mapped_column(default="", index=True)
|
||||||
|
|
||||||
subtitle: Mapped[Optional[str]]
|
subtitle: Mapped[Optional[str]]
|
||||||
description: Mapped[Optional[str]]
|
description: Mapped[Optional[str]]
|
||||||
published_date: Mapped[Optional[date]]
|
published_date: Mapped[Optional[date]]
|
||||||
@@ -111,6 +115,24 @@ class Book(BigIntAuditBase):
|
|||||||
def progress(self) -> Optional["BookProgress"]:
|
def progress(self) -> Optional["BookProgress"]:
|
||||||
return self.progress_records[0] if self.progress_records else None
|
return self.progress_records[0] if self.progress_records else None
|
||||||
|
|
||||||
|
@validates("title")
|
||||||
|
def _normalize_title(self, _key: str, title: str) -> str:
|
||||||
|
"""
|
||||||
|
Derive `normalized_title` from whatever writes the title.
|
||||||
|
|
||||||
|
A validator rather than a service call because `BookService` sets titles from
|
||||||
|
at least three places — `to_model_on_create`, `to_model_on_update` and the
|
||||||
|
`setattr` loop in `_populate_with_unique_relationships` — and a fourth would
|
||||||
|
otherwise leave the key silently stale.
|
||||||
|
"""
|
||||||
|
# Imported here rather than at module scope: `chitai.services.matching` cannot
|
||||||
|
# be reached without initialising the `chitai.services` package, which imports
|
||||||
|
# the services, which import this module.
|
||||||
|
from chitai.services.matching import normalize_title
|
||||||
|
|
||||||
|
self.normalized_title = normalize_title(title)
|
||||||
|
return title
|
||||||
|
|
||||||
def __repr__(self) -> str:
|
def __repr__(self) -> str:
|
||||||
return f"Book({self.title=!r})"
|
return f"Book({self.title=!r})"
|
||||||
|
|
||||||
@@ -132,6 +154,24 @@ class Identifier(BigIntBase):
|
|||||||
book_id: Mapped[int] = mapped_column(ForeignKey("books.id", ondelete="cascade"))
|
book_id: Mapped[int] = mapped_column(ForeignKey("books.id", ondelete="cascade"))
|
||||||
value: Mapped[str]
|
value: Mapped[str]
|
||||||
|
|
||||||
|
# Kept current by `_normalize` below — never assign it directly. Null for an
|
||||||
|
# identifier that cannot carry a match: a per-build UUID, or an ISBN that fails
|
||||||
|
# its own checksum.
|
||||||
|
normalized_value: Mapped[Optional[str]] = mapped_column(index=True)
|
||||||
|
|
||||||
|
@validates("name", "value")
|
||||||
|
def _normalize(self, key: str, value: str) -> str:
|
||||||
|
"""Recompute `normalized_value` whenever either half of the pair changes."""
|
||||||
|
from chitai.services.matching import normalize_identifier
|
||||||
|
|
||||||
|
name = value if key == "name" else self.name
|
||||||
|
raw = value if key == "value" else self.value
|
||||||
|
|
||||||
|
self.normalized_value = (
|
||||||
|
normalize_identifier(name, raw) if name and raw else None
|
||||||
|
)
|
||||||
|
return value
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
return f"Identifier({self.name!r} : {self.value!r})"
|
return f"Identifier({self.name!r} : {self.value!r})"
|
||||||
|
|
||||||
@@ -139,6 +179,15 @@ class Identifier(BigIntBase):
|
|||||||
class FileMetadata(BigIntBase):
|
class FileMetadata(BigIntBase):
|
||||||
__tablename__ = "file_metadata"
|
__tablename__ = "file_metadata"
|
||||||
|
|
||||||
|
__table_args__ = (
|
||||||
|
# Deliberately not unique. The hash is KOReader's partial MD5, which samples
|
||||||
|
# 12 KiB of the file, so two genuinely different files can collide — and an
|
||||||
|
# existing database may already hold duplicates, which a unique index would
|
||||||
|
# refuse to build over. Duplicate detection pairs it with `size` and treats a
|
||||||
|
# match as advisory, so this only has to make the lookup cheap.
|
||||||
|
Index("ix_file_metadata_hash", "hash"),
|
||||||
|
)
|
||||||
|
|
||||||
book_id: Mapped[int] = mapped_column(ForeignKey("books.id", ondelete="cascade"))
|
book_id: Mapped[int] = mapped_column(ForeignKey("books.id", ondelete="cascade"))
|
||||||
book: Mapped[Book] = relationship(back_populates="files")
|
book: Mapped[Book] = relationship(back_populates="files")
|
||||||
hash: Mapped[str]
|
hash: Mapped[str]
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ class BookList(BigIntAuditBase):
|
|||||||
return len(self.book_links) if self.book_links else 0
|
return len(self.book_links) if self.book_links else 0
|
||||||
except Exception:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
class BookListLink(BigIntBase):
|
class BookListLink(BigIntBase):
|
||||||
__tablename__ = "book_list_links"
|
__tablename__ = "book_list_links"
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
from typing import Optional
|
from typing import Optional
|
||||||
from sqlalchemy import ForeignKey
|
from sqlalchemy import ForeignKey
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
from sqlalchemy.orm import relationship
|
|
||||||
from advanced_alchemy.base import BigIntAuditBase
|
from advanced_alchemy.base import BigIntAuditBase
|
||||||
|
|
||||||
|
|
||||||
@@ -20,5 +19,5 @@ class BookProgress(BigIntAuditBase):
|
|||||||
pdf_page: Mapped[Optional[int]]
|
pdf_page: Mapped[Optional[int]]
|
||||||
percentage: Mapped[float]
|
percentage: Mapped[float]
|
||||||
completed: Mapped[Optional[bool]]
|
completed: Mapped[Optional[bool]]
|
||||||
device: Mapped[Optional[str]] # Device that updated the progress
|
device: Mapped[Optional[str]] # Device that updated the progress
|
||||||
device_id: Mapped[Optional[str]]
|
device_id: Mapped[Optional[str]]
|
||||||
|
|||||||
@@ -1,12 +1,10 @@
|
|||||||
from collections.abc import Hashable
|
from collections.abc import Hashable
|
||||||
from sqlalchemy.orm import Mapped, mapped_column
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
from sqlalchemy import ColumnElement, ForeignKey
|
from sqlalchemy import ColumnElement
|
||||||
|
|
||||||
from advanced_alchemy.base import BigIntAuditBase
|
from advanced_alchemy.base import BigIntAuditBase
|
||||||
from advanced_alchemy.mixins import UniqueMixin
|
from advanced_alchemy.mixins import UniqueMixin
|
||||||
|
|
||||||
from .book import Book
|
|
||||||
|
|
||||||
|
|
||||||
class BookSeries(BigIntAuditBase, UniqueMixin):
|
class BookSeries(BigIntAuditBase, UniqueMixin):
|
||||||
__tablename__ = "book_series"
|
__tablename__ = "book_series"
|
||||||
|
|||||||
@@ -0,0 +1,37 @@
|
|||||||
|
from sqlalchemy import ForeignKey, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from advanced_alchemy.base import BigIntBase
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateDismissal(BigIntBase):
|
||||||
|
"""
|
||||||
|
Two books a reader has said are not the same book.
|
||||||
|
|
||||||
|
Title and author matching is probabilistic, so it will keep proposing a second
|
||||||
|
edition, a translation and a sequel that shares its predecessor's name. A review
|
||||||
|
screen with no way to disagree with it nags forever, which is how people learn to
|
||||||
|
ignore a screen.
|
||||||
|
|
||||||
|
The pair is stored ordered — `book_a_id` is always the lower id — so "A and B" and
|
||||||
|
"B and A" are one row and the unique constraint can do its job. Use `pair()` rather
|
||||||
|
than assigning the columns directly.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "duplicate_dismissals"
|
||||||
|
__table_args__ = (UniqueConstraint("book_a_id", "book_b_id"),)
|
||||||
|
|
||||||
|
book_a_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("books.id", ondelete="cascade"), index=True
|
||||||
|
)
|
||||||
|
book_b_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("books.id", ondelete="cascade"), index=True
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def pair(first: int, second: int) -> tuple[int, int]:
|
||||||
|
"""The two book ids in the order this table stores them."""
|
||||||
|
return (first, second) if first <= second else (second, first)
|
||||||
|
|
||||||
|
def __repr__(self) -> str:
|
||||||
|
return f"DuplicateDismissal({self.book_a_id!r}, {self.book_b_id!r})"
|
||||||
@@ -1,9 +1,10 @@
|
|||||||
from sqlalchemy import ColumnElement, ForeignKey
|
from sqlalchemy import ForeignKey
|
||||||
from sqlalchemy.orm import Mapped
|
from sqlalchemy.orm import Mapped
|
||||||
from sqlalchemy.orm import mapped_column
|
from sqlalchemy.orm import mapped_column
|
||||||
|
|
||||||
from advanced_alchemy.base import BigIntAuditBase
|
from advanced_alchemy.base import BigIntAuditBase
|
||||||
|
|
||||||
|
|
||||||
class KosyncDevice(BigIntAuditBase):
|
class KosyncDevice(BigIntAuditBase):
|
||||||
__tablename__ = "devices"
|
__tablename__ = "devices"
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ class Library(BigIntAuditBase, SlugKey):
|
|||||||
name: Mapped[str] = mapped_column(unique=True)
|
name: Mapped[str] = mapped_column(unique=True)
|
||||||
root_path: Mapped[str]
|
root_path: Mapped[str]
|
||||||
# Which structure to save the files in the filesystem (i.e {author_name}/{title}.{ext})
|
# Which structure to save the files in the filesystem (i.e {author_name}/{title}.{ext})
|
||||||
path_template: Mapped[str]
|
path_template: Mapped[str]
|
||||||
description: Mapped[Optional[str]]
|
description: Mapped[Optional[str]]
|
||||||
icon: Mapped[str] = mapped_column(default="library")
|
icon: Mapped[str] = mapped_column(default="library")
|
||||||
read_only: Mapped[bool] = mapped_column(nullable=False, default=False)
|
read_only: Mapped[bool] = mapped_column(nullable=False, default=False)
|
||||||
|
|||||||
@@ -1,6 +1,4 @@
|
|||||||
from litestar import Request, Response, MediaType
|
from litestar import Request, Response, MediaType
|
||||||
from litestar.exceptions import HTTPException
|
|
||||||
from litestar.status_codes import HTTP_404_NOT_FOUND
|
|
||||||
|
|
||||||
from advanced_alchemy.exceptions import NotFoundError
|
from advanced_alchemy.exceptions import NotFoundError
|
||||||
|
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
from base64 import b64decode
|
from base64 import b64decode
|
||||||
from chitai.services.user import UserService
|
from chitai.services.user import UserService
|
||||||
from litestar.middleware import (
|
from litestar.middleware import (
|
||||||
AbstractAuthenticationMiddleware,
|
AbstractAuthenticationMiddleware,
|
||||||
AuthenticationResult,
|
AuthenticationResult,
|
||||||
DefineMiddleware
|
DefineMiddleware,
|
||||||
)
|
)
|
||||||
from litestar.connection import ASGIConnection
|
from litestar.connection import ASGIConnection
|
||||||
from litestar.exceptions import NotAuthorizedException, PermissionDeniedException
|
from litestar.exceptions import NotAuthorizedException, PermissionDeniedException
|
||||||
@@ -11,25 +11,30 @@ from chitai.config import settings
|
|||||||
|
|
||||||
|
|
||||||
class BasicAuthenticationMiddleware(AbstractAuthenticationMiddleware):
|
class BasicAuthenticationMiddleware(AbstractAuthenticationMiddleware):
|
||||||
async def authenticate_request(self, connection: ASGIConnection) -> AuthenticationResult:
|
async def authenticate_request(
|
||||||
"""Given a request, parse the header for Base64 encoded basic auth credentials. """
|
self, connection: ASGIConnection
|
||||||
|
) -> AuthenticationResult:
|
||||||
|
"""Given a request, parse the header for Base64 encoded basic auth credentials."""
|
||||||
|
|
||||||
# retrieve the auth header
|
# retrieve the auth header
|
||||||
auth_header = connection.headers.get("Authorization", None)
|
auth_header = connection.headers.get("Authorization", None)
|
||||||
if not auth_header:
|
if not auth_header:
|
||||||
raise NotAuthorizedException()
|
raise NotAuthorizedException()
|
||||||
|
|
||||||
username, password = b64decode(auth_header.split("Basic ")[1]).decode().split(":")
|
|
||||||
|
|
||||||
|
username, password = (
|
||||||
|
b64decode(auth_header.split("Basic ")[1]).decode().split(":")
|
||||||
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
db_session = settings.alchemy_config.provide_session(connection.app.state, connection.scope)
|
db_session = settings.alchemy_config.provide_session(
|
||||||
|
connection.app.state, connection.scope
|
||||||
|
)
|
||||||
user_service = UserService(db_session)
|
user_service = UserService(db_session)
|
||||||
user = await user_service.authenticate(username, password)
|
user = await user_service.authenticate(username, password)
|
||||||
return AuthenticationResult(user=user, auth=None)
|
return AuthenticationResult(user=user, auth=None)
|
||||||
|
|
||||||
except PermissionDeniedException:
|
except PermissionDeniedException:
|
||||||
raise NotAuthorizedException()
|
raise NotAuthorizedException()
|
||||||
|
|
||||||
|
|
||||||
basic_auth_mw = DefineMiddleware(BasicAuthenticationMiddleware)
|
basic_auth_mw = DefineMiddleware(BasicAuthenticationMiddleware)
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
from chitai.services.user import UserService
|
from chitai.services.user import UserService
|
||||||
from chitai.services.kosync_device import KosyncDeviceService
|
from chitai.services.kosync_device import KosyncDeviceService
|
||||||
from litestar.middleware import (
|
from litestar.middleware import (
|
||||||
AbstractAuthenticationMiddleware,
|
AbstractAuthenticationMiddleware,
|
||||||
AuthenticationResult,
|
AuthenticationResult,
|
||||||
DefineMiddleware
|
DefineMiddleware,
|
||||||
)
|
)
|
||||||
from litestar.connection import ASGIConnection
|
from litestar.connection import ASGIConnection
|
||||||
from litestar.exceptions import NotAuthorizedException, PermissionDeniedException
|
from litestar.exceptions import NotAuthorizedException, PermissionDeniedException
|
||||||
@@ -11,19 +11,23 @@ from chitai.config import settings
|
|||||||
|
|
||||||
|
|
||||||
class KosyncAuthenticationMiddleware(AbstractAuthenticationMiddleware):
|
class KosyncAuthenticationMiddleware(AbstractAuthenticationMiddleware):
|
||||||
async def authenticate_request(self, connection: ASGIConnection) -> AuthenticationResult:
|
async def authenticate_request(
|
||||||
"""Given a request, parse the header for Base64 encoded basic auth credentials. """
|
self, connection: ASGIConnection
|
||||||
|
) -> AuthenticationResult:
|
||||||
|
"""Given a request, parse the header for Base64 encoded basic auth credentials."""
|
||||||
|
|
||||||
# retrieve the auth header
|
# retrieve the auth header
|
||||||
api_key = connection.headers.get("X-AUTH-USER", None)
|
api_key = connection.headers.get("X-AUTH-USER", None)
|
||||||
if not api_key:
|
if not api_key:
|
||||||
raise NotAuthorizedException()
|
raise NotAuthorizedException()
|
||||||
|
|
||||||
try:
|
try:
|
||||||
db_session = settings.alchemy_config.provide_session(connection.app.state, connection.scope)
|
db_session = settings.alchemy_config.provide_session(
|
||||||
|
connection.app.state, connection.scope
|
||||||
|
)
|
||||||
user_service = UserService(db_session)
|
user_service = UserService(db_session)
|
||||||
device_service = KosyncDeviceService(db_session)
|
device_service = KosyncDeviceService(db_session)
|
||||||
|
|
||||||
device = await device_service.get_by_api_key(api_key)
|
device = await device_service.get_by_api_key(api_key)
|
||||||
user = await user_service.get(device.user_id)
|
user = await user_service.get(device.user_id)
|
||||||
|
|
||||||
@@ -32,6 +36,6 @@ class KosyncAuthenticationMiddleware(AbstractAuthenticationMiddleware):
|
|||||||
except PermissionDeniedException as exc:
|
except PermissionDeniedException as exc:
|
||||||
print(exc)
|
print(exc)
|
||||||
raise NotAuthorizedException()
|
raise NotAuthorizedException()
|
||||||
|
|
||||||
|
|
||||||
kosync_api_key_auth = DefineMiddleware(KosyncAuthenticationMiddleware)
|
kosync_api_key_auth = DefineMiddleware(KosyncAuthenticationMiddleware)
|
||||||
|
|||||||
@@ -4,7 +4,15 @@ from .book import (
|
|||||||
BookProgressCreate,
|
BookProgressCreate,
|
||||||
BookProgressRead,
|
BookProgressRead,
|
||||||
BooksCreateFromFiles,
|
BooksCreateFromFiles,
|
||||||
|
BooksUploadResult,
|
||||||
|
BookMerge,
|
||||||
BookMetadataUpdate,
|
BookMetadataUpdate,
|
||||||
|
DuplicateBookGroupRead,
|
||||||
|
DuplicateBookRead,
|
||||||
|
DuplicateDismissal,
|
||||||
|
DuplicateFileRead,
|
||||||
|
FileFingerprint,
|
||||||
|
PossibleDuplicateRead,
|
||||||
FileMetadataRead,
|
FileMetadataRead,
|
||||||
BookSeriesRead,
|
BookSeriesRead,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -30,7 +30,11 @@ class FileMetadataRead(BaseModel):
|
|||||||
path: str
|
path: str
|
||||||
hash: str
|
hash: str
|
||||||
size: int
|
size: int
|
||||||
content_type: str
|
|
||||||
|
# Nullable, though every ingest path now writes one through
|
||||||
|
# `guess_content_type`. Rows predating it can hold null, and a required field here
|
||||||
|
# turns one of those into a 500 on a book the reader can otherwise open.
|
||||||
|
content_type: str | None = None
|
||||||
|
|
||||||
@computed_field
|
@computed_field
|
||||||
@property
|
@property
|
||||||
@@ -129,6 +133,83 @@ class BooksCreateFromFiles(BaseModel):
|
|||||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||||
|
|
||||||
|
|
||||||
|
class FileFingerprint(BaseModel):
|
||||||
|
"""What a client can say about a file it has not uploaded yet."""
|
||||||
|
|
||||||
|
hash: str
|
||||||
|
size: int
|
||||||
|
filename: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateFileRead(BaseModel):
|
||||||
|
"""A file that was not stored because the library already holds its bytes."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(from_attributes=True)
|
||||||
|
|
||||||
|
filename: str
|
||||||
|
hash: str
|
||||||
|
size: int
|
||||||
|
library_id: int
|
||||||
|
|
||||||
|
# Null when the match was another file in the same upload, which has no row yet.
|
||||||
|
book_id: int | None = None
|
||||||
|
book_title: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateBookRead(BaseModel):
|
||||||
|
"""
|
||||||
|
A stored book that may be the same book as another one.
|
||||||
|
|
||||||
|
Unlike `DuplicateFileRead` this is a guess: the evidence is metadata two editions
|
||||||
|
of one work legitimately share. Nothing was refused on the strength of it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
model_config = ConfigDict(from_attributes=True)
|
||||||
|
|
||||||
|
book_id: int
|
||||||
|
title: str
|
||||||
|
authors: list[str]
|
||||||
|
library_id: int
|
||||||
|
cover_image: Path | None = None
|
||||||
|
|
||||||
|
# Why it matched: "identifier" and/or "title-author".
|
||||||
|
matched_on: list[str]
|
||||||
|
|
||||||
|
|
||||||
|
class PossibleDuplicateRead(BaseModel):
|
||||||
|
"""A book that was imported, together with what it might be a second copy of."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(from_attributes=True)
|
||||||
|
|
||||||
|
book_id: int
|
||||||
|
title: str
|
||||||
|
candidates: list[DuplicateBookRead]
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateBookGroupRead(BaseModel):
|
||||||
|
"""Books the library holds that all look like copies of one book."""
|
||||||
|
|
||||||
|
books: list[DuplicateBookRead]
|
||||||
|
|
||||||
|
|
||||||
|
class DuplicateDismissal(BaseModel):
|
||||||
|
"""Two books a reader is saying are not the same book."""
|
||||||
|
|
||||||
|
book_a_id: int
|
||||||
|
book_b_id: int
|
||||||
|
|
||||||
|
|
||||||
|
class BooksUploadResult(BaseModel):
|
||||||
|
"""The outcome of a multi-file upload: what was created, and what was skipped."""
|
||||||
|
|
||||||
|
created: list["BookRead"]
|
||||||
|
skipped: list[DuplicateFileRead]
|
||||||
|
|
||||||
|
# Created, not skipped — these are books that went in and look like something the
|
||||||
|
# library already had. The reader decides what to do about it.
|
||||||
|
possible_duplicates: list[PossibleDuplicateRead] = Field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
class BookMetadataUpdate(BaseModel):
|
class BookMetadataUpdate(BaseModel):
|
||||||
title: str | None = None
|
title: str | None = None
|
||||||
subtitle: str | None = None
|
subtitle: str | None = None
|
||||||
@@ -170,6 +251,20 @@ class BookMetadataUpdate(BaseModel):
|
|||||||
return v
|
return v
|
||||||
|
|
||||||
|
|
||||||
|
class BookMerge(BaseModel):
|
||||||
|
"""
|
||||||
|
Fold several books into one.
|
||||||
|
|
||||||
|
`metadata` is the reader's resolution of the fields the records disagreed on.
|
||||||
|
Anything it does not name keeps the survivor's value — merging metadata is a
|
||||||
|
judgement, so nothing is guessed on the caller's behalf.
|
||||||
|
"""
|
||||||
|
|
||||||
|
survivor_id: int
|
||||||
|
merged_ids: list[int]
|
||||||
|
metadata: Optional["BookMetadataUpdate"] = None
|
||||||
|
|
||||||
|
|
||||||
class BookProgressCreate(BaseModel):
|
class BookProgressCreate(BaseModel):
|
||||||
percentage: float
|
percentage: float
|
||||||
epub_cfi: str | None = None
|
epub_cfi: str | None = None
|
||||||
@@ -178,5 +273,3 @@ class BookProgressCreate(BaseModel):
|
|||||||
completed: bool | None = None
|
completed: bool | None = None
|
||||||
device_type: str | None = None
|
device_type: str | None = None
|
||||||
device_id: str | None = None
|
device_id: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,8 @@
|
|||||||
from pathlib import Path
|
|
||||||
from typing import Annotated
|
from typing import Annotated
|
||||||
from pydantic import BaseModel, Field, computed_field
|
from pydantic import BaseModel, ConfigDict, Field, SkipValidation, computed_field
|
||||||
from advanced_alchemy.utils.text import slugify
|
from litestar.datastructures import UploadFile
|
||||||
|
from advanced_alchemy.utils.text import slugify
|
||||||
|
|
||||||
|
|
||||||
class LibraryCreate(BaseModel):
|
class LibraryCreate(BaseModel):
|
||||||
name: Annotated[str, Field(min_length=1)]
|
name: Annotated[str, Field(min_length=1)]
|
||||||
@@ -16,6 +17,7 @@ class LibraryCreate(BaseModel):
|
|||||||
def slug(self) -> str:
|
def slug(self) -> str:
|
||||||
return slugify(self.name)
|
return slugify(self.name)
|
||||||
|
|
||||||
|
|
||||||
class LibraryRead(BaseModel):
|
class LibraryRead(BaseModel):
|
||||||
id: int
|
id: int
|
||||||
name: str
|
name: str
|
||||||
@@ -27,6 +29,57 @@ class LibraryRead(BaseModel):
|
|||||||
total: int | None = None
|
total: int | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreArchiveUpload(BaseModel):
|
||||||
|
"""
|
||||||
|
A zipped Calibre library.
|
||||||
|
|
||||||
|
The only way in through the API: a desktop Calibre install is usually not on the
|
||||||
|
server at all. Importing from a path the server can already see is a server-side
|
||||||
|
operation, and stays one — `litestar calibre-import` does that.
|
||||||
|
"""
|
||||||
|
|
||||||
|
archive: Annotated[UploadFile, SkipValidation]
|
||||||
|
allow_duplicates: bool = False
|
||||||
|
|
||||||
|
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||||
|
|
||||||
|
|
||||||
|
class ImportFailureRead(BaseModel):
|
||||||
|
"""A book the import could not store."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(from_attributes=True)
|
||||||
|
|
||||||
|
calibre_id: int
|
||||||
|
title: str
|
||||||
|
reason: str
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreImportRead(BaseModel):
|
||||||
|
"""A running or finished import."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(from_attributes=True)
|
||||||
|
|
||||||
|
id: str
|
||||||
|
library_id: int
|
||||||
|
source: str
|
||||||
|
state: str
|
||||||
|
|
||||||
|
total: int
|
||||||
|
processed: int
|
||||||
|
created: int
|
||||||
|
skipped: int
|
||||||
|
failed: int
|
||||||
|
|
||||||
|
current_title: str | None = None
|
||||||
|
failures: list[ImportFailureRead] = Field(default_factory=list)
|
||||||
|
|
||||||
|
# A count, not the records. The library's duplicates screen is what shows them.
|
||||||
|
possible_duplicates: int = 0
|
||||||
|
|
||||||
|
# Set when the run itself broke, as opposed to individual books failing.
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
|
||||||
class LibraryUpdate(BaseModel):
|
class LibraryUpdate(BaseModel):
|
||||||
name: str | None
|
name: str | None
|
||||||
root_path: str | None
|
root_path: str | None
|
||||||
|
|||||||
@@ -1,11 +1,11 @@
|
|||||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
from pydantic import BaseModel, Field, field_validator
|
||||||
|
|
||||||
|
|
||||||
class ShelfRead(BaseModel):
|
class ShelfRead(BaseModel):
|
||||||
id: int
|
id: int
|
||||||
title: str
|
title: str
|
||||||
library_id: int | None = None
|
library_id: int | None = None
|
||||||
total: int | None = None # Number of books in the shelf
|
total: int | None = None # Number of books in the shelf
|
||||||
|
|
||||||
|
|
||||||
class ShelfCreate(BaseModel):
|
class ShelfCreate(BaseModel):
|
||||||
|
|||||||
+1866
-90
File diff suppressed because it is too large
Load Diff
@@ -1,13 +1,8 @@
|
|||||||
# src/chitai/services/bookshelf.py
|
# src/chitai/services/bookshelf.py
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from typing import Any, Sequence
|
|
||||||
from advanced_alchemy.exceptions import ErrorMessages
|
|
||||||
from advanced_alchemy.filters import StatementFilter
|
|
||||||
from advanced_alchemy.service import SQLAlchemyAsyncRepositoryService
|
from advanced_alchemy.service import SQLAlchemyAsyncRepositoryService
|
||||||
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
||||||
from advanced_alchemy.utils.dataclass import Empty, EmptyType
|
|
||||||
from sqlalchemy import ColumnElement, Select, delete
|
|
||||||
|
|
||||||
# Local imports
|
# Local imports
|
||||||
from chitai.database.models.book_list import BookList, BookListLink
|
from chitai.database.models.book_list import BookList, BookListLink
|
||||||
|
|||||||
@@ -0,0 +1,585 @@
|
|||||||
|
# src/chitai/services/calibre.py
|
||||||
|
|
||||||
|
"""
|
||||||
|
Read a Calibre library.
|
||||||
|
|
||||||
|
This knows about `metadata.db` and the tree beside it, and nothing about `Book`,
|
||||||
|
`BookService` or a database session — it is a file-format reader, and it is testable
|
||||||
|
without Postgres or an app. Interpretation belongs to whoever imports what it returns:
|
||||||
|
identifiers come back exactly as Calibre wrote them, not folded onto Chitai's schemes.
|
||||||
|
|
||||||
|
Things about Calibre that are load-bearing here:
|
||||||
|
|
||||||
|
- **Never query the views.** `meta` and the `tag_browser_*` family call SQLite functions
|
||||||
|
Calibre registers from Python at connection time, so `SELECT * FROM meta` fails with
|
||||||
|
`no such function: sortconcat`. Only base tables are touched below.
|
||||||
|
- **An unknown date is a sentinel, not a null** — `0101-01-01`, Calibre's
|
||||||
|
`UNDEFINED_DATE`. It parses fine as a date, so nothing complains; it just makes every
|
||||||
|
book without a publication date look like it was published in the year 101.
|
||||||
|
- **`data.name` is not the title.** It is the on-disk stem, truncated to Calibre's
|
||||||
|
filename limit and sanitised, so the file is `The Project Gutenberg eBook #33283_
|
||||||
|
Calcul - Silvanus Phillips Thompson.pdf` for a book titled `The Project Gutenberg
|
||||||
|
eBook #33283: Calculus Made Easy, 2nd Edition`. Names locate files; the database
|
||||||
|
carries the metadata.
|
||||||
|
- **`authors.name` escapes a comma as `|`**, which Calibre reverses on read
|
||||||
|
(`AuthorsTable.unserialize` in its `db/tables.py`).
|
||||||
|
- **`series_index` defaults to 1.0 whether or not the book is in a series**, so a
|
||||||
|
position is only meaningful alongside a series.
|
||||||
|
- **`books_pages_link` is recent and often empty.** It is treated as optional both ways:
|
||||||
|
the table may not exist, and where it does the rows are frequently `pages = 0` with
|
||||||
|
`needs_scan = 1`.
|
||||||
|
|
||||||
|
Nothing walks the tree: every file is located through `books.path`, which is why
|
||||||
|
`.caltrash` — where Calibre keeps deleted books, still on disk — cannot be picked up by
|
||||||
|
accident.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import shutil
|
||||||
|
import sqlite3
|
||||||
|
import tempfile
|
||||||
|
import zipfile
|
||||||
|
from collections import defaultdict
|
||||||
|
from collections.abc import Callable
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime
|
||||||
|
from html.parser import HTMLParser
|
||||||
|
from pathlib import Path, PurePosixPath
|
||||||
|
|
||||||
|
|
||||||
|
METADATA_DB = "metadata.db"
|
||||||
|
COVER_FILENAME = "cover.jpg"
|
||||||
|
|
||||||
|
# Calibre writes `0101-01-01` for "no date". Any year this early is that sentinel rather
|
||||||
|
# than a publication date somebody meant.
|
||||||
|
EARLIEST_REAL_YEAR = 1000
|
||||||
|
|
||||||
|
# The sidecars a WAL-mode database keeps beside itself. Copied along with it so the
|
||||||
|
# snapshot can be recovered, since Calibre may be running while this reads.
|
||||||
|
_DATABASE_SIDECARS = ("-wal", "-shm")
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreLibraryError(Exception):
|
||||||
|
"""The library cannot be read at all — wrong directory, or no catalogue in it."""
|
||||||
|
|
||||||
|
|
||||||
|
# How far down an archive to look for `metadata.db`. Zipping a Calibre library gives
|
||||||
|
# either the directory itself or its contents, and a file manager may add a wrapper
|
||||||
|
# folder on top, so two levels of nesting is normal and more is somebody's backup tree.
|
||||||
|
_ARCHIVE_SEARCH_DEPTH = 3
|
||||||
|
|
||||||
|
# Extraction is refused unless the destination has the uncompressed size plus this
|
||||||
|
# much headroom. Filling the disk would take the whole application down, not just the
|
||||||
|
# import.
|
||||||
|
_DISK_HEADROOM = 256 * 1024 * 1024 # 256 MiB
|
||||||
|
|
||||||
|
|
||||||
|
def _archive_members(archive: zipfile.ZipFile) -> list[zipfile.ZipInfo]:
|
||||||
|
"""
|
||||||
|
The entries worth extracting, refusing any that would escape the destination.
|
||||||
|
|
||||||
|
`ZipFile.extract` does sanitise names, but relying on that silently is how the next
|
||||||
|
person to swap the extraction call reintroduces zip-slip. An archive naming
|
||||||
|
`../../etc/anything` is malformed or hostile, and either way there is nothing to
|
||||||
|
salvage by continuing.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
CalibreLibraryError: If any entry points outside the archive root.
|
||||||
|
"""
|
||||||
|
members = []
|
||||||
|
|
||||||
|
for member in archive.infolist():
|
||||||
|
if member.is_dir():
|
||||||
|
continue
|
||||||
|
|
||||||
|
name = PurePosixPath(member.filename)
|
||||||
|
|
||||||
|
if name.is_absolute() or ".." in name.parts:
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
f"The archive contains an entry outside itself: {member.filename!r}"
|
||||||
|
)
|
||||||
|
|
||||||
|
members.append(member)
|
||||||
|
|
||||||
|
return members
|
||||||
|
|
||||||
|
|
||||||
|
async def extract_calibre_archive(archive: Path, destination: Path) -> Path:
|
||||||
|
"""
|
||||||
|
Unpack a zipped Calibre library and find the catalogue inside it.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
archive: The `.zip` to unpack.
|
||||||
|
destination: An empty directory to unpack into. The caller owns it and is
|
||||||
|
responsible for removing it.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The directory holding `metadata.db`, which is what `CalibreLibrary` takes.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
CalibreLibraryError: If the file is not a zip, names an entry outside itself,
|
||||||
|
would not fit on disk, or holds no `metadata.db`.
|
||||||
|
"""
|
||||||
|
return await asyncio.to_thread(_extract_calibre_archive, archive, destination)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_calibre_archive(archive: Path, destination: Path) -> Path:
|
||||||
|
if not zipfile.is_zipfile(archive):
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
"That is not a zip file. A Calibre library has to be zipped, not tarred."
|
||||||
|
)
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive) as opened:
|
||||||
|
members = _archive_members(opened)
|
||||||
|
|
||||||
|
if not any(
|
||||||
|
PurePosixPath(member.filename).name == METADATA_DB for member in members
|
||||||
|
):
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
f"The archive holds no {METADATA_DB}, so it is not a Calibre library"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Checked before writing rather than discovered part-way through: a full disk
|
||||||
|
# takes the whole application down, and the number is in the archive already.
|
||||||
|
needed = sum(member.file_size for member in members)
|
||||||
|
free = shutil.disk_usage(destination).free
|
||||||
|
|
||||||
|
if needed + _DISK_HEADROOM > free:
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
f"Unpacking needs {needed // (1024 * 1024)} MiB and only "
|
||||||
|
f"{free // (1024 * 1024)} MiB is free"
|
||||||
|
)
|
||||||
|
|
||||||
|
opened.extractall(destination, members=members)
|
||||||
|
|
||||||
|
return _find_catalogue(destination)
|
||||||
|
|
||||||
|
|
||||||
|
def _find_catalogue(root: Path) -> Path:
|
||||||
|
"""The shallowest directory under `root` holding a `metadata.db`."""
|
||||||
|
candidates = sorted(
|
||||||
|
(path.parent for path in root.rglob(METADATA_DB) if path.is_file()),
|
||||||
|
key=lambda path: len(path.relative_to(root).parts),
|
||||||
|
)
|
||||||
|
|
||||||
|
for candidate in candidates:
|
||||||
|
if len(candidate.relative_to(root).parts) <= _ARCHIVE_SEARCH_DEPTH:
|
||||||
|
return candidate
|
||||||
|
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
f"No {METADATA_DB} within {_ARCHIVE_SEARCH_DEPTH} levels of the archive root"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CalibreFile:
|
||||||
|
"""One row of Calibre's `data` table: a book in one format."""
|
||||||
|
|
||||||
|
path: Path
|
||||||
|
"""Absolute path, resolved against the library root. Not checked for existence."""
|
||||||
|
|
||||||
|
format: str
|
||||||
|
"""As Calibre stores it, upper case: `EPUB`, `PDF`, `AZW3`."""
|
||||||
|
|
||||||
|
size: int
|
||||||
|
"""`data.uncompressed_size` — Calibre's claim, not a fresh stat."""
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CalibreBook:
|
||||||
|
"""One book, with everything Chitai has a column for and nothing it does not."""
|
||||||
|
|
||||||
|
calibre_id: int
|
||||||
|
uuid: str
|
||||||
|
title: str
|
||||||
|
authors: list[str] = field(default_factory=list)
|
||||||
|
description: str | None = None
|
||||||
|
published_date: date | None = None
|
||||||
|
series: str | None = None
|
||||||
|
series_position: str | None = None
|
||||||
|
tags: list[str] = field(default_factory=list)
|
||||||
|
publisher: str | None = None
|
||||||
|
language: str | None = None
|
||||||
|
|
||||||
|
identifiers: dict[str, str] = field(default_factory=dict)
|
||||||
|
"""Keyed by `identifiers.type` verbatim — `isbn`, `mobi-asin`, `amazon`."""
|
||||||
|
|
||||||
|
pages: int | None = None
|
||||||
|
cover: Path | None = None
|
||||||
|
files: list[CalibreFile] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreLibrary:
|
||||||
|
"""
|
||||||
|
A Calibre library on disk, opened for reading.
|
||||||
|
|
||||||
|
The catalogue is **copied** before it is read. Calibre may be running and writing,
|
||||||
|
and opening the live file either sees a torn state or needs to recover a write-ahead
|
||||||
|
log, which read-only access cannot do. The copy is small — hundreds of kilobytes for
|
||||||
|
a handful of books, single-digit megabytes for thousands — so this costs nothing and
|
||||||
|
removes the question. The original is never opened by SQLite at all.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, root: Path | str) -> None:
|
||||||
|
self.root = Path(root)
|
||||||
|
self._connection: sqlite3.Connection | None = None
|
||||||
|
self._workspace: Path | None = None
|
||||||
|
|
||||||
|
# Every query runs in a worker thread, and `asyncio.to_thread` hands out
|
||||||
|
# whichever one is free — so the connection outlives the thread that opened it
|
||||||
|
# and `check_same_thread` has to be off. The lock is what makes that safe: it
|
||||||
|
# keeps two queries off the connection at once, which is the thing that check
|
||||||
|
# was standing in for.
|
||||||
|
self._lock = asyncio.Lock()
|
||||||
|
|
||||||
|
@property
|
||||||
|
def database(self) -> Path:
|
||||||
|
return self.root / METADATA_DB
|
||||||
|
|
||||||
|
async def open(self) -> None:
|
||||||
|
"""
|
||||||
|
Copy the catalogue aside and connect to the copy.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
CalibreLibraryError: If there is no `metadata.db` under the root.
|
||||||
|
"""
|
||||||
|
if self._connection is not None:
|
||||||
|
return
|
||||||
|
|
||||||
|
if not await asyncio.to_thread(self.database.is_file):
|
||||||
|
raise CalibreLibraryError(
|
||||||
|
f"No {METADATA_DB} in '{self.root}' — that is not a Calibre library"
|
||||||
|
)
|
||||||
|
|
||||||
|
self._workspace = Path(await asyncio.to_thread(tempfile.mkdtemp))
|
||||||
|
copy = self._workspace / METADATA_DB
|
||||||
|
|
||||||
|
await asyncio.to_thread(self._copy_database, copy)
|
||||||
|
|
||||||
|
# Read-write on our own copy, deliberately: that is what lets SQLite recover a
|
||||||
|
# write-ahead log the source may have been mid-way through.
|
||||||
|
self._connection = sqlite3.connect(str(copy), check_same_thread=False)
|
||||||
|
|
||||||
|
def _copy_database(self, destination: Path) -> None:
|
||||||
|
shutil.copy2(self.database, destination)
|
||||||
|
|
||||||
|
for suffix in _DATABASE_SIDECARS:
|
||||||
|
sidecar = self.database.with_name(self.database.name + suffix)
|
||||||
|
if sidecar.is_file():
|
||||||
|
shutil.copy2(sidecar, destination.with_name(destination.name + suffix))
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
"""
|
||||||
|
Disconnect and remove the copy. Safe to call more than once.
|
||||||
|
|
||||||
|
The copy is removed even if closing the connection fails — otherwise a failure
|
||||||
|
here leaves a catalogue-sized file in the temp directory, and the caller that
|
||||||
|
failed is exactly the one that will not come back to tidy up.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
if self._connection is not None:
|
||||||
|
async with self._lock:
|
||||||
|
self._connection.close()
|
||||||
|
self._connection = None
|
||||||
|
finally:
|
||||||
|
if self._workspace is not None:
|
||||||
|
await asyncio.to_thread(shutil.rmtree, self._workspace, True)
|
||||||
|
self._workspace = None
|
||||||
|
|
||||||
|
async def __aenter__(self) -> CalibreLibrary:
|
||||||
|
await self.open()
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, *_exception: object) -> None:
|
||||||
|
await self.close()
|
||||||
|
|
||||||
|
async def count(self) -> int:
|
||||||
|
"""How many books the catalogue holds, without reading any of them."""
|
||||||
|
rows = await self._in_thread(
|
||||||
|
lambda: self._execute("SELECT count(*) FROM books")
|
||||||
|
)
|
||||||
|
return int(rows[0][0])
|
||||||
|
|
||||||
|
async def books(self) -> list[CalibreBook]:
|
||||||
|
"""
|
||||||
|
Read the whole catalogue.
|
||||||
|
|
||||||
|
One query per table and the joining done in Python, rather than a per-book query
|
||||||
|
across ten tables. Everything Chitai stores about a book is small, so a
|
||||||
|
self-hosted catalogue fits in memory comfortably.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Every book, in Calibre id order.
|
||||||
|
"""
|
||||||
|
return await self._in_thread(self._read_books)
|
||||||
|
|
||||||
|
async def _in_thread[T](self, work: Callable[[], T]) -> T:
|
||||||
|
"""Run one unit of SQLite work off the event loop, and only one at a time."""
|
||||||
|
async with self._lock:
|
||||||
|
return await asyncio.to_thread(work)
|
||||||
|
|
||||||
|
def _execute(self, statement: str) -> list[tuple]:
|
||||||
|
if self._connection is None:
|
||||||
|
raise CalibreLibraryError("The library is not open")
|
||||||
|
|
||||||
|
return self._connection.execute(statement).fetchall()
|
||||||
|
|
||||||
|
def _has_table(self, name: str) -> bool:
|
||||||
|
return bool(
|
||||||
|
self._execute(
|
||||||
|
f"SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = '{name}'"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def _read_books(self) -> list[CalibreBook]:
|
||||||
|
authors = self._grouped(
|
||||||
|
"SELECT bal.book, a.name FROM books_authors_link bal "
|
||||||
|
"JOIN authors a ON a.id = bal.author ORDER BY bal.id"
|
||||||
|
)
|
||||||
|
tags = self._grouped(
|
||||||
|
"SELECT btl.book, t.name FROM books_tags_link btl "
|
||||||
|
"JOIN tags t ON t.id = btl.tag ORDER BY t.name"
|
||||||
|
)
|
||||||
|
# Calibre's link tables are unique per book for these two, so the last write
|
||||||
|
# wins and there is nothing to choose between.
|
||||||
|
series = self._mapped(
|
||||||
|
"SELECT bsl.book, s.name FROM books_series_link bsl "
|
||||||
|
"JOIN series s ON s.id = bsl.series"
|
||||||
|
)
|
||||||
|
publishers = self._mapped(
|
||||||
|
"SELECT bpl.book, p.name FROM books_publishers_link bpl "
|
||||||
|
"JOIN publishers p ON p.id = bpl.publisher"
|
||||||
|
)
|
||||||
|
# A book can carry several languages; Chitai holds one, so the first wins.
|
||||||
|
languages = self._grouped(
|
||||||
|
"SELECT bll.book, l.lang_code FROM books_languages_link bll "
|
||||||
|
"JOIN languages l ON l.id = bll.lang_code ORDER BY bll.item_order"
|
||||||
|
)
|
||||||
|
descriptions = self._mapped("SELECT book, text FROM comments")
|
||||||
|
|
||||||
|
identifiers: dict[int, dict[str, str]] = defaultdict(dict)
|
||||||
|
for book_id, name, value in self._execute(
|
||||||
|
"SELECT book, type, val FROM identifiers"
|
||||||
|
):
|
||||||
|
if name and value:
|
||||||
|
identifiers[book_id][str(name)] = str(value)
|
||||||
|
|
||||||
|
files: dict[int, list[tuple[str, str, int]]] = defaultdict(list)
|
||||||
|
for book_id, format, name, size in self._execute(
|
||||||
|
"SELECT book, format, name, uncompressed_size FROM data ORDER BY id"
|
||||||
|
):
|
||||||
|
files[book_id].append((str(format), str(name), int(size or 0)))
|
||||||
|
|
||||||
|
pages: dict[int, int] = {}
|
||||||
|
if self._has_table("books_pages_link"):
|
||||||
|
pages = {
|
||||||
|
book_id: int(count)
|
||||||
|
for book_id, count in self._execute(
|
||||||
|
"SELECT book, pages FROM books_pages_link WHERE pages > 0"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
books = []
|
||||||
|
for row in self._execute(
|
||||||
|
"SELECT id, title, pubdate, series_index, path, uuid, has_cover "
|
||||||
|
"FROM books ORDER BY id"
|
||||||
|
):
|
||||||
|
book_id, title, pubdate, series_index, path, uuid, has_cover = row
|
||||||
|
directory = self.root / Path(str(path))
|
||||||
|
in_series = series.get(book_id)
|
||||||
|
|
||||||
|
books.append(
|
||||||
|
CalibreBook(
|
||||||
|
calibre_id=book_id,
|
||||||
|
uuid=str(uuid or ""),
|
||||||
|
title=str(title or ""),
|
||||||
|
authors=[
|
||||||
|
unescape_author(name) for name in authors.get(book_id, [])
|
||||||
|
],
|
||||||
|
description=strip_html(descriptions.get(book_id)),
|
||||||
|
published_date=parse_date(pubdate),
|
||||||
|
series=in_series,
|
||||||
|
# Meaningless without a series: Calibre defaults the index to 1.0 for
|
||||||
|
# every book, in a series or not.
|
||||||
|
series_position=(
|
||||||
|
format_series_index(series_index) if in_series else None
|
||||||
|
),
|
||||||
|
tags=tags.get(book_id, []),
|
||||||
|
publisher=publishers.get(book_id),
|
||||||
|
language=next(iter(languages.get(book_id, [])), None),
|
||||||
|
identifiers=dict(identifiers.get(book_id, {})),
|
||||||
|
pages=pages.get(book_id),
|
||||||
|
cover=directory / COVER_FILENAME if has_cover else None,
|
||||||
|
files=[
|
||||||
|
CalibreFile(
|
||||||
|
path=directory / f"{name}.{format.lower()}",
|
||||||
|
format=format,
|
||||||
|
size=size,
|
||||||
|
)
|
||||||
|
for format, name, size in files.get(book_id, [])
|
||||||
|
],
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return books
|
||||||
|
|
||||||
|
def _grouped(self, statement: str) -> dict[int, list[str]]:
|
||||||
|
"""Run a `(book, value)` query into one list per book, keeping row order."""
|
||||||
|
grouped: dict[int, list[str]] = defaultdict(list)
|
||||||
|
|
||||||
|
for book_id, value in self._execute(statement):
|
||||||
|
if value is not None:
|
||||||
|
grouped[book_id].append(str(value))
|
||||||
|
|
||||||
|
return grouped
|
||||||
|
|
||||||
|
def _mapped(self, statement: str) -> dict[int, str]:
|
||||||
|
"""Run a `(book, value)` query into one value per book."""
|
||||||
|
return {
|
||||||
|
book_id: str(value)
|
||||||
|
for book_id, value in self._execute(statement)
|
||||||
|
if value is not None
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def unescape_author(name: str) -> str:
|
||||||
|
"""
|
||||||
|
Undo Calibre's comma escaping.
|
||||||
|
|
||||||
|
`authors.name` stores a comma as `|`, and Calibre reverses it on the way out. Left
|
||||||
|
alone, `Doyle, Sir Arthur Conan` comes back as `Doyle| Sir Arthur Conan`.
|
||||||
|
"""
|
||||||
|
return name.replace("|", ",").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def parse_date(value: object) -> date | None:
|
||||||
|
"""
|
||||||
|
Read one of Calibre's timestamps, discarding its "unknown" sentinel.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value: The stored column, which is text in practice but need not be.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The date, or None for a null, an unparseable value, or Calibre's
|
||||||
|
`0101-01-01` placeholder.
|
||||||
|
"""
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if isinstance(value, datetime):
|
||||||
|
parsed = value.date()
|
||||||
|
elif isinstance(value, date):
|
||||||
|
parsed = value
|
||||||
|
else:
|
||||||
|
text = str(value).strip()
|
||||||
|
if not text:
|
||||||
|
return None
|
||||||
|
|
||||||
|
try:
|
||||||
|
parsed = datetime.fromisoformat(text).date()
|
||||||
|
except ValueError:
|
||||||
|
try:
|
||||||
|
parsed = date.fromisoformat(text[:10])
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
return parsed if parsed.year >= EARLIEST_REAL_YEAR else None
|
||||||
|
|
||||||
|
|
||||||
|
def format_series_index(index: object) -> str | None:
|
||||||
|
"""
|
||||||
|
Render `series_index` as the string `Book.series_position` holds.
|
||||||
|
|
||||||
|
Calibre stores a REAL, so volume seven arrives as `7.0` — which would be stored
|
||||||
|
verbatim and then compared as a string against the `7` everything else writes.
|
||||||
|
Fractional positions are real and are kept: `1.5` is a novella between two novels.
|
||||||
|
"""
|
||||||
|
if index is None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
try:
|
||||||
|
number = float(index)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
return str(int(number)) if number.is_integer() else f"{number:g}"
|
||||||
|
|
||||||
|
|
||||||
|
class _TextExtractor(HTMLParser):
|
||||||
|
"""Flatten markup to text, keeping the line breaks that carried meaning."""
|
||||||
|
|
||||||
|
# Tags whose boundaries are a line break rather than nothing at all. Without these
|
||||||
|
# a description of three paragraphs comes out as one run-on sentence.
|
||||||
|
_BREAKS = frozenset(
|
||||||
|
{
|
||||||
|
"p",
|
||||||
|
"br",
|
||||||
|
"div",
|
||||||
|
"li",
|
||||||
|
"tr",
|
||||||
|
"blockquote",
|
||||||
|
"hr",
|
||||||
|
"h1",
|
||||||
|
"h2",
|
||||||
|
"h3",
|
||||||
|
"h4",
|
||||||
|
"h5",
|
||||||
|
"h6",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
super().__init__(convert_charrefs=True)
|
||||||
|
self._parts: list[str] = []
|
||||||
|
|
||||||
|
def handle_data(self, data: str) -> None:
|
||||||
|
self._parts.append(data)
|
||||||
|
|
||||||
|
def handle_starttag(self, tag: str, _attrs: object) -> None:
|
||||||
|
self._break(tag)
|
||||||
|
|
||||||
|
def handle_endtag(self, tag: str) -> None:
|
||||||
|
self._break(tag)
|
||||||
|
|
||||||
|
def _break(self, tag: str) -> None:
|
||||||
|
"""
|
||||||
|
End the current line, once.
|
||||||
|
|
||||||
|
Both halves of `</p><p>` are a boundary, and the open tag of the very first
|
||||||
|
block is not one at all — so emitting a newline per tag turns two paragraphs
|
||||||
|
into two blank-line-separated ones with a leading gap. One break per boundary
|
||||||
|
is what the plain text wants.
|
||||||
|
"""
|
||||||
|
if tag in self._BREAKS and self._parts and not self._parts[-1].endswith("\n"):
|
||||||
|
self._parts.append("\n")
|
||||||
|
|
||||||
|
@property
|
||||||
|
def text(self) -> str:
|
||||||
|
lines = [line.strip() for line in "".join(self._parts).splitlines()]
|
||||||
|
|
||||||
|
return "\n".join(line for line in lines if line).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def strip_html(html: str | None) -> str | None:
|
||||||
|
"""
|
||||||
|
Turn Calibre's `comments` into plain text.
|
||||||
|
|
||||||
|
`comments.text` is always HTML, and `Book.description` is rendered as text — so the
|
||||||
|
tags would show literally on the book page.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
html: The stored comment, if there is one.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The text, or None when there was nothing or nothing survived.
|
||||||
|
"""
|
||||||
|
if not html:
|
||||||
|
return None
|
||||||
|
|
||||||
|
parser = _TextExtractor()
|
||||||
|
parser.feed(html)
|
||||||
|
parser.close()
|
||||||
|
|
||||||
|
return parser.text or None
|
||||||
@@ -0,0 +1,239 @@
|
|||||||
|
# src/chitai/services/calibre_import.py
|
||||||
|
|
||||||
|
"""
|
||||||
|
Run a Calibre import in the background and report on it.
|
||||||
|
|
||||||
|
The import outlives its request — a real library takes minutes to hours — so the handler
|
||||||
|
starts a task and hands back a handle to poll. The work itself is
|
||||||
|
`BookService.create_many_from_calibre`; everything here is lifecycle: state, progress,
|
||||||
|
cancellation, and a session of its own.
|
||||||
|
|
||||||
|
**This registry lives in memory, and therefore assumes one worker process.** That holds
|
||||||
|
today: the production `CMD` is `litestar run`, which is single-process, and the consume
|
||||||
|
watcher is already an in-process singleton with the same constraint. `TODO.md` records
|
||||||
|
that the production image should move to uvicorn with a worker count — the day that
|
||||||
|
happens, a poll can land on a worker that has never heard of the job, and this needs an
|
||||||
|
`import_jobs` table instead. It is written down here because nothing else will say so.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import shutil
|
||||||
|
import uuid
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from enum import StrEnum
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from chitai.config import settings
|
||||||
|
from chitai.database.models import Library
|
||||||
|
from chitai.services.book import (
|
||||||
|
BookService,
|
||||||
|
CalibreImportProgress,
|
||||||
|
CalibreImportResult,
|
||||||
|
UnimportedBook,
|
||||||
|
)
|
||||||
|
from chitai.services.calibre import CalibreLibrary
|
||||||
|
|
||||||
|
|
||||||
|
class ImportState(StrEnum):
|
||||||
|
"""Where a job has got to."""
|
||||||
|
|
||||||
|
RUNNING = "running"
|
||||||
|
FINISHED = "finished"
|
||||||
|
|
||||||
|
# Stopped on request. What it imported is complete.
|
||||||
|
CANCELLED = "cancelled"
|
||||||
|
|
||||||
|
# The run itself broke — an unreadable catalogue, a missing library. Distinct from
|
||||||
|
# individual books failing, which `failures` carries and which never stops the run.
|
||||||
|
FAILED = "failed"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ImportJob:
|
||||||
|
"""One import, running or finished."""
|
||||||
|
|
||||||
|
id: str
|
||||||
|
library_id: int
|
||||||
|
source: str
|
||||||
|
|
||||||
|
state: ImportState = ImportState.RUNNING
|
||||||
|
total: int = 0
|
||||||
|
processed: int = 0
|
||||||
|
created: int = 0
|
||||||
|
skipped: int = 0
|
||||||
|
failed: int = 0
|
||||||
|
|
||||||
|
current_title: str | None = None
|
||||||
|
failures: list[UnimportedBook] = field(default_factory=list)
|
||||||
|
|
||||||
|
# Books imported that look like something the library already had. A count, not the
|
||||||
|
# records: the duplicates screen is what shows them, and a big import would make
|
||||||
|
# this the largest thing in the response for no benefit.
|
||||||
|
possible_duplicates: int = 0
|
||||||
|
|
||||||
|
# Why the whole run stopped, when `state` is FAILED.
|
||||||
|
error: str | None = None
|
||||||
|
|
||||||
|
# The directory the job owns and must delete when it ends: what the uploaded archive
|
||||||
|
# was unpacked into.
|
||||||
|
workspace: Path | None = None
|
||||||
|
|
||||||
|
_stop: bool = False
|
||||||
|
|
||||||
|
@property
|
||||||
|
def finished(self) -> bool:
|
||||||
|
return self.state is not ImportState.RUNNING
|
||||||
|
|
||||||
|
def absorb(self, result: CalibreImportResult) -> None:
|
||||||
|
"""Take the final counts from a finished run."""
|
||||||
|
self.total = result.total
|
||||||
|
self.created = len(result.created)
|
||||||
|
self.skipped = len(result.skipped)
|
||||||
|
self.failed = len(result.failed)
|
||||||
|
self.failures = list(result.failed)
|
||||||
|
self.possible_duplicates = len(result.possible_duplicates)
|
||||||
|
self.current_title = None
|
||||||
|
|
||||||
|
self.state = ImportState.CANCELLED if result.stopped else ImportState.FINISHED
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreImportRegistry:
|
||||||
|
"""
|
||||||
|
The imports this process knows about.
|
||||||
|
|
||||||
|
One instance, held at module scope below. Jobs are kept after they finish so the
|
||||||
|
screen that started one can still read its result; nothing evicts them, which is
|
||||||
|
fine for a handful of one-time migrations and is the other reason a table would be
|
||||||
|
the answer if this ever needed to be durable.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self._jobs: dict[str, ImportJob] = {}
|
||||||
|
|
||||||
|
# Held only to keep the tasks referenced. Without this the event loop is free to
|
||||||
|
# garbage-collect a running task mid-import.
|
||||||
|
self._tasks: set[asyncio.Task] = set()
|
||||||
|
|
||||||
|
def get(self, job_id: str) -> ImportJob | None:
|
||||||
|
return self._jobs.get(job_id)
|
||||||
|
|
||||||
|
def running_for(self, library_id: int) -> ImportJob | None:
|
||||||
|
"""The unfinished import for a library, if it has one."""
|
||||||
|
return next(
|
||||||
|
(
|
||||||
|
job
|
||||||
|
for job in self._jobs.values()
|
||||||
|
if job.library_id == library_id and not job.finished
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
|
||||||
|
def cancel(self, job_id: str) -> ImportJob | None:
|
||||||
|
"""
|
||||||
|
Ask a job to stop after the book it is on.
|
||||||
|
|
||||||
|
Not `task.cancel()`: that would abandon a book mid-copy, leaving files on disk
|
||||||
|
with no row describing them. The flag is read between books.
|
||||||
|
"""
|
||||||
|
job = self._jobs.get(job_id)
|
||||||
|
|
||||||
|
if job is not None and not job.finished:
|
||||||
|
job._stop = True
|
||||||
|
|
||||||
|
return job
|
||||||
|
|
||||||
|
def start(
|
||||||
|
self,
|
||||||
|
library: Library,
|
||||||
|
source: Path,
|
||||||
|
workspace: Path,
|
||||||
|
label: str,
|
||||||
|
allow_duplicates: bool = False,
|
||||||
|
) -> ImportJob:
|
||||||
|
"""
|
||||||
|
Begin importing, and return the handle to poll.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
library: The library to import into.
|
||||||
|
source: The unpacked Calibre library's directory.
|
||||||
|
workspace: A directory the job owns and deletes when it ends — what the
|
||||||
|
uploaded archive was unpacked into. The books have been copied into the
|
||||||
|
library by then, so nothing is lost with it.
|
||||||
|
label: What to report as the source. `source` is a temp directory that would
|
||||||
|
mean nothing to the reader, so this is the archive's own name.
|
||||||
|
allow_duplicates: Import books whose files are already stored.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The job, already running.
|
||||||
|
"""
|
||||||
|
job = ImportJob(
|
||||||
|
id=str(uuid.uuid4()),
|
||||||
|
library_id=library.id,
|
||||||
|
source=label,
|
||||||
|
workspace=workspace,
|
||||||
|
)
|
||||||
|
self._jobs[job.id] = job
|
||||||
|
|
||||||
|
task = asyncio.create_task(self._run(job, library.id, source, allow_duplicates))
|
||||||
|
self._tasks.add(task)
|
||||||
|
task.add_done_callback(self._tasks.discard)
|
||||||
|
|
||||||
|
return job
|
||||||
|
|
||||||
|
async def _run(
|
||||||
|
self, job: ImportJob, library_id: int, source: Path, allow_duplicates: bool
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Do the import, recording everything on the job.
|
||||||
|
|
||||||
|
Opens a **session of its own**: the request that started this is long gone, and
|
||||||
|
its session was closed with it.
|
||||||
|
"""
|
||||||
|
from chitai.services.library import LibraryService
|
||||||
|
|
||||||
|
def on_progress(progress: CalibreImportProgress) -> None:
|
||||||
|
job.total = progress.total
|
||||||
|
job.processed = progress.processed
|
||||||
|
job.current_title = progress.title
|
||||||
|
|
||||||
|
if progress.outcome == "created":
|
||||||
|
job.created += 1
|
||||||
|
elif progress.outcome == "skipped":
|
||||||
|
job.skipped += 1
|
||||||
|
else:
|
||||||
|
job.failed += 1
|
||||||
|
|
||||||
|
catalogue = CalibreLibrary(source)
|
||||||
|
|
||||||
|
try:
|
||||||
|
await catalogue.open()
|
||||||
|
|
||||||
|
async with settings.alchemy_config.get_session() as session:
|
||||||
|
library = await LibraryService(session=session).get(library_id)
|
||||||
|
|
||||||
|
result = await BookService(session=session).create_many_from_calibre(
|
||||||
|
catalogue,
|
||||||
|
library,
|
||||||
|
allow_duplicates=allow_duplicates,
|
||||||
|
on_progress=on_progress,
|
||||||
|
should_stop=lambda: job._stop,
|
||||||
|
)
|
||||||
|
|
||||||
|
job.absorb(result)
|
||||||
|
except Exception as exc:
|
||||||
|
job.state = ImportState.FAILED
|
||||||
|
job.error = f"{type(exc).__name__}: {exc}"
|
||||||
|
finally:
|
||||||
|
await catalogue.close()
|
||||||
|
|
||||||
|
# An unpacked archive is a second copy of the whole library, and the books
|
||||||
|
# worth keeping have been copied into the library proper by now. Removed
|
||||||
|
# even when the run failed — especially then, since nothing will come back
|
||||||
|
# for it.
|
||||||
|
if job.workspace is not None:
|
||||||
|
await asyncio.to_thread(shutil.rmtree, job.workspace, True)
|
||||||
|
|
||||||
|
|
||||||
|
registry = CalibreImportRegistry()
|
||||||
@@ -1,19 +1,26 @@
|
|||||||
import asyncio
|
import asyncio
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
|
from chitai.config import settings
|
||||||
from chitai.database.models.library import Library
|
from chitai.database.models.library import Library
|
||||||
from chitai.services import BookService, LibraryService
|
from chitai.services import BookService, LibraryService
|
||||||
from chitai.services.metadata_extractor import Extractor
|
|
||||||
from chitai.services.utils import create_directory
|
from chitai.services.utils import create_directory
|
||||||
from watchfiles import awatch, Change
|
from watchfiles import awatch, Change
|
||||||
|
|
||||||
|
|
||||||
class ConsumeDirectoryWatcher:
|
class ConsumeDirectoryWatcher:
|
||||||
"""Watches a directory and batch processes files by their relative path."""
|
"""Watches a directory and batch processes files by their relative path."""
|
||||||
|
|
||||||
def __init__(self, watch_path: str, library_service: LibraryService, book_service: BookService, batch_delay: float = 3.0):
|
def __init__(
|
||||||
|
self,
|
||||||
|
watch_path: str,
|
||||||
|
library_service: LibraryService,
|
||||||
|
book_service: BookService,
|
||||||
|
batch_delay: float = 3.0,
|
||||||
|
):
|
||||||
"""
|
"""
|
||||||
Initialize the file watcher.
|
Initialize the file watcher.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
watch_path: Directory path to watch
|
watch_path: Directory path to watch
|
||||||
batch_delay: Seconds to wait before processing a batch
|
batch_delay: Seconds to wait before processing a batch
|
||||||
@@ -40,9 +47,9 @@ class ConsumeDirectoryWatcher:
|
|||||||
for change_type, file_path in changes:
|
for change_type, file_path in changes:
|
||||||
if change_type != Change.added:
|
if change_type != Change.added:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
file_path = Path(file_path)
|
file_path = Path(file_path)
|
||||||
|
|
||||||
# If a directory was added, scan it for existing files
|
# If a directory was added, scan it for existing files
|
||||||
if file_path.is_dir():
|
if file_path.is_dir():
|
||||||
print(f"Directory added: {file_path}")
|
print(f"Directory added: {file_path}")
|
||||||
@@ -50,28 +57,28 @@ class ConsumeDirectoryWatcher:
|
|||||||
else:
|
else:
|
||||||
print(f"File added: {file_path}")
|
print(f"File added: {file_path}")
|
||||||
await self._handle_file_added(file_path)
|
await self._handle_file_added(file_path)
|
||||||
|
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
print("File watcher stopped")
|
print("File watcher stopped")
|
||||||
# Wait for any pending processing tasks
|
# Wait for any pending processing tasks
|
||||||
if self._processing_tasks:
|
if self._processing_tasks:
|
||||||
await asyncio.gather(*self._processing_tasks, return_exceptions=True)
|
await asyncio.gather(*self._processing_tasks, return_exceptions=True)
|
||||||
raise
|
raise
|
||||||
|
|
||||||
async def _handle_file_added(self, file_path: Path):
|
async def _handle_file_added(self, file_path: Path):
|
||||||
"""Handle a single file being added."""
|
"""Handle a single file being added."""
|
||||||
# Get relative path from watch directory
|
# Get relative path from watch directory
|
||||||
|
|
||||||
rel_path = file_path.relative_to(self.watch_path)
|
rel_path = file_path.relative_to(self.watch_path)
|
||||||
parent_rel = rel_path.parent
|
parent_rel = rel_path.parent
|
||||||
library = parent_rel.parts[0]
|
library = parent_rel.parts[0]
|
||||||
|
|
||||||
# Add to appropriate group
|
# Add to appropriate group
|
||||||
self.file_groups[library].add(file_path)
|
self.file_groups[library].add(file_path)
|
||||||
|
|
||||||
# Schedule batch processing for this group
|
# Schedule batch processing for this group
|
||||||
self._schedule_batch_processing(library)
|
self._schedule_batch_processing(library)
|
||||||
|
|
||||||
async def _handle_directory_added(self, dir_path: Path):
|
async def _handle_directory_added(self, dir_path: Path):
|
||||||
"""Handle a directory being added - scan it for existing files."""
|
"""Handle a directory being added - scan it for existing files."""
|
||||||
try:
|
try:
|
||||||
@@ -82,46 +89,63 @@ class ConsumeDirectoryWatcher:
|
|||||||
await self._handle_file_added(file_path)
|
await self._handle_file_added(file_path)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Error scanning directory {dir_path}: {e}")
|
print(f"Error scanning directory {dir_path}: {e}")
|
||||||
|
|
||||||
def _schedule_batch_processing(self, library_slug: str):
|
def _schedule_batch_processing(self, library_slug: str):
|
||||||
"""Schedule batch processing for a specific path group."""
|
"""Schedule batch processing for a specific path group."""
|
||||||
# Create a task to process this group after a delay
|
# Create a task to process this group after a delay
|
||||||
task = asyncio.create_task(self._delayed_batch_process(library_slug))
|
task = asyncio.create_task(self._delayed_batch_process(library_slug))
|
||||||
self._processing_tasks.add(task)
|
self._processing_tasks.add(task)
|
||||||
task.add_done_callback(self._processing_tasks.discard)
|
task.add_done_callback(self._processing_tasks.discard)
|
||||||
|
|
||||||
async def _delayed_batch_process(self, library_slug: str):
|
async def _delayed_batch_process(self, library_slug: str):
|
||||||
"""Wait for batch delay, then process accumulated files."""
|
"""Wait for batch delay, then process accumulated files."""
|
||||||
await asyncio.sleep(self.batch_delay)
|
await asyncio.sleep(self.batch_delay)
|
||||||
|
|
||||||
# Get and clear the file list for this path
|
# Get and clear the file list for this path
|
||||||
if library_slug not in self.file_groups:
|
if library_slug not in self.file_groups:
|
||||||
return
|
return
|
||||||
|
|
||||||
files_to_process = self.file_groups[library_slug].copy()
|
files_to_process = self.file_groups[library_slug].copy()
|
||||||
self.file_groups[library_slug].clear()
|
self.file_groups[library_slug].clear()
|
||||||
|
|
||||||
if not files_to_process:
|
if not files_to_process:
|
||||||
return
|
return
|
||||||
|
|
||||||
print(f"Batch processing {len(files_to_process)} files from {library_slug}")
|
print(f"Batch processing {len(files_to_process)} files from {library_slug}")
|
||||||
await self._process_batch(files_to_process, library_slug)
|
await self._process_batch(files_to_process, library_slug)
|
||||||
|
|
||||||
async def _process_batch(self, file_paths: set[Path], library_slug: str):
|
async def _process_batch(self, file_paths: set[Path], library_slug: str):
|
||||||
"""Process a batch of files."""
|
"""Process a batch of files."""
|
||||||
try:
|
try:
|
||||||
|
result = await self.book_service.create_many_from_existing_files(
|
||||||
books = await self.book_service.create_many_from_existing_files(
|
|
||||||
list(file_paths),
|
list(file_paths),
|
||||||
self.watch_path / Path(library_slug),
|
self.watch_path / Path(library_slug),
|
||||||
library=await self._get_library(library_slug),
|
library=await self._get_library(library_slug),
|
||||||
)
|
)
|
||||||
|
|
||||||
print(f"Created {len(books)} books!")
|
print(f"Created {len(result.books)} books!")
|
||||||
|
|
||||||
|
if result.duplicates:
|
||||||
|
print(
|
||||||
|
f"Moved {len(result.duplicates)} already-stored file(s) "
|
||||||
|
f"to {settings.duplicate_path}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Imported all the same — a metadata match is a guess, and there is nobody
|
||||||
|
# here to ask. The library's duplicates screen is where these get decided.
|
||||||
|
for possible in result.possible_duplicates:
|
||||||
|
names = ", ".join(
|
||||||
|
f"{candidate.title} (#{candidate.book_id})"
|
||||||
|
for candidate in possible.candidates
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
f"Imported {possible.title!r} (#{possible.book_id}), which may "
|
||||||
|
f"already be in the library as: {names}"
|
||||||
|
)
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Error processing batch: {e}")
|
print(f"Error processing batch: {e}")
|
||||||
raise e
|
raise e
|
||||||
|
|
||||||
async def _get_library(self, slug: str) -> Library:
|
async def _get_library(self, slug: str) -> Library:
|
||||||
return await self.library_service.get_one(Library.slug == slug)
|
return await self.library_service.get_one(Library.slug == slug)
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
# Standard library
|
# Standard library
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
from typing import Any, AsyncGenerator, Callable, NotRequired, Optional
|
from typing import Any, AsyncGenerator, Optional
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
from advanced_alchemy.extensions.litestar.providers import (
|
from advanced_alchemy.extensions.litestar.providers import (
|
||||||
@@ -13,7 +13,6 @@ from advanced_alchemy.extensions.litestar.providers import (
|
|||||||
)
|
)
|
||||||
from advanced_alchemy.exceptions import NotFoundError
|
from advanced_alchemy.exceptions import NotFoundError
|
||||||
from advanced_alchemy.filters import CollectionFilter, StatementFilter
|
from advanced_alchemy.filters import CollectionFilter, StatementFilter
|
||||||
from advanced_alchemy.service import FilterTypeT
|
|
||||||
from sqlalchemy.orm import selectinload
|
from sqlalchemy.orm import selectinload
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
from litestar import Request
|
from litestar import Request
|
||||||
@@ -25,7 +24,6 @@ from litestar.di import Provide
|
|||||||
from advanced_alchemy.extensions.litestar.providers import create_filter_dependencies
|
from advanced_alchemy.extensions.litestar.providers import create_filter_dependencies
|
||||||
|
|
||||||
# Local imports
|
# Local imports
|
||||||
from chitai import schemas as s
|
|
||||||
from chitai.database import models as m
|
from chitai.database import models as m
|
||||||
from chitai.services import (
|
from chitai.services import (
|
||||||
UserService,
|
UserService,
|
||||||
@@ -127,9 +125,33 @@ def create_book_filter_dependencies(
|
|||||||
# Get base filters first
|
# Get base filters first
|
||||||
filters = create_filter_dependencies(config, dep_defaults)
|
filters = create_filter_dependencies(config, dep_defaults)
|
||||||
|
|
||||||
|
# OVERRIDE: id filter typed by the configured id type, not always `str`
|
||||||
|
#
|
||||||
|
# advanced_alchemy's `provide_id_filter` annotates `ids` as `list[str]` and
|
||||||
|
# ignores `config["id_filter"]` entirely, so `?ids=12` reaches the database as
|
||||||
|
# the string "12" and Postgres refuses `bigint = character varying`. Nothing
|
||||||
|
# called `?ids=` until the duplicates screen needed to fetch a handful of books
|
||||||
|
# by id, which is why it went unnoticed.
|
||||||
|
if id_type := config.get("id_filter"):
|
||||||
|
id_field = config.get("id_field", "id")
|
||||||
|
|
||||||
|
def provide_typed_id_filter(
|
||||||
|
ids=Parameter(query="ids", default=None, required=False),
|
||||||
|
) -> CollectionFilter:
|
||||||
|
return CollectionFilter(field_name=id_field, values=ids)
|
||||||
|
|
||||||
|
# Attached as a type object rather than written as an annotation: this module
|
||||||
|
# has `from __future__ import annotations`, so a written one is stored as the
|
||||||
|
# string "Optional[list[id_type]]" and resolved against module globals, where
|
||||||
|
# a local named `id_type` does not exist.
|
||||||
|
provide_typed_id_filter.__annotations__["ids"] = Optional[list[id_type]]
|
||||||
|
|
||||||
|
filters[dep_defaults.ID_FILTER_DEPENDENCY_KEY] = Provide(
|
||||||
|
provide_typed_id_filter, sync_to_thread=False
|
||||||
|
)
|
||||||
|
|
||||||
# OVERRIDE: Custom search filter with trigram search
|
# OVERRIDE: Custom search filter with trigram search
|
||||||
if config.get("search"):
|
if config.get("search"):
|
||||||
search_fields = config.get("search")
|
|
||||||
|
|
||||||
def provide_trigram_search_filter(
|
def provide_trigram_search_filter(
|
||||||
search_string: str | None = Parameter(
|
search_string: str | None = Parameter(
|
||||||
@@ -345,6 +367,7 @@ def provide_optional_user(request: Request[m.User, Token, Any]) -> m.User | None
|
|||||||
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
async def provide_user_via_basic_auth(request: Request[m.User, None, Any]) -> m.User:
|
async def provide_user_via_basic_auth(request: Request[m.User, None, Any]) -> m.User:
|
||||||
return request.user
|
return request.user
|
||||||
|
|
||||||
@@ -356,4 +379,3 @@ async def provide_user_via_kosync_auth(request: Request[m.User, None, Any]) -> m
|
|||||||
provide_kosync_device_service = create_service_provider(KosyncDeviceService)
|
provide_kosync_device_service = create_service_provider(KosyncDeviceService)
|
||||||
|
|
||||||
provide_kosync_progress_service = create_service_provider(KosyncProgressService)
|
provide_kosync_progress_service = create_service_provider(KosyncProgressService)
|
||||||
|
|
||||||
@@ -4,9 +4,7 @@ from pathlib import Path
|
|||||||
import re
|
import re
|
||||||
|
|
||||||
from jinja2 import Template
|
from jinja2 import Template
|
||||||
from advanced_alchemy.service import ModelDictT
|
|
||||||
|
|
||||||
import chitai.database.models as m
|
|
||||||
|
|
||||||
# TODO: Replace Jinja2 templates with a simpler custom templating system.
|
# TODO: Replace Jinja2 templates with a simpler custom templating system.
|
||||||
# Current Jinja2 implementation is overly complex for basic path generation.
|
# Current Jinja2 implementation is overly complex for basic path generation.
|
||||||
@@ -16,6 +14,43 @@ import chitai.database.models as m
|
|||||||
# - Auto-handle missing values (e.g., skip {series}/ if series is empty)
|
# - Auto-handle missing values (e.g., skip {series}/ if series is empty)
|
||||||
|
|
||||||
|
|
||||||
|
# Characters that cannot survive being interpolated into a path. The forward slash is
|
||||||
|
# the one that matters: titles legitimately contain it — "AC/DC", "Him/Her" — and the
|
||||||
|
# template writes the title straight into a directory name, so an unsanitised one
|
||||||
|
# silently adds a level and puts the book somewhere `book.path` does not describe.
|
||||||
|
# Calibre strips these from its own on-disk names and keeps the real title in its
|
||||||
|
# database, which is how an import surfaces them.
|
||||||
|
_UNSAFE_IN_PATH = re.compile(r"[/\\\x00-\x1f]")
|
||||||
|
|
||||||
|
|
||||||
|
def sanitize_path_component(value: str) -> str:
|
||||||
|
"""Make one metadata value safe to use as a single directory or file name."""
|
||||||
|
return _UNSAFE_IN_PATH.sub("_", value).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_components(book_data: dict) -> dict:
|
||||||
|
"""
|
||||||
|
A shallow copy of the metadata with the values a path is built from sanitised.
|
||||||
|
|
||||||
|
Only strings are touched, and only the fields the default template interpolates. A
|
||||||
|
caller's own template can reach anything else in the dict, which is a reason to keep
|
||||||
|
this conservative rather than to walk the whole structure.
|
||||||
|
"""
|
||||||
|
safe = dict(book_data)
|
||||||
|
|
||||||
|
for key in ("title", "series", "series_position"):
|
||||||
|
if isinstance(safe.get(key), str):
|
||||||
|
safe[key] = sanitize_path_component(safe[key])
|
||||||
|
|
||||||
|
if isinstance(safe.get("authors"), list):
|
||||||
|
safe["authors"] = [
|
||||||
|
sanitize_path_component(author) if isinstance(author, str) else author
|
||||||
|
for author in safe["authors"]
|
||||||
|
]
|
||||||
|
|
||||||
|
return safe
|
||||||
|
|
||||||
|
|
||||||
default_path_template = """
|
default_path_template = """
|
||||||
/{{book.authors[0] if book.authors else 'Unknown'}}
|
/{{book.authors[0] if book.authors else 'Unknown'}}
|
||||||
{%- if book.series -%}
|
{%- if book.series -%}
|
||||||
@@ -101,7 +136,12 @@ class BookPathGenerator:
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
result = self.root_path / Path(self.path_template.render(book=book_data))
|
# Sanitised per value, never on the rendered result: the separators the template
|
||||||
|
# puts *between* author, series and title are the whole point of it, and only the
|
||||||
|
# values interpolated into it must not contribute any of their own.
|
||||||
|
result = self.root_path / Path(
|
||||||
|
self.path_template.render(book=_safe_components(book_data))
|
||||||
|
)
|
||||||
|
|
||||||
# Clean up
|
# Clean up
|
||||||
result = re.sub(r"/+", "/", str(result)) # Remove consecutive backslashes
|
result = re.sub(r"/+", "/", str(result)) # Remove consecutive backslashes
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
from typing import Any, Optional
|
from typing import Optional
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
|
||||||
from advanced_alchemy.filters import (
|
from advanced_alchemy.filters import (
|
||||||
|
|||||||
@@ -3,11 +3,10 @@ from typing import Any, Optional
|
|||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
from sqlalchemy.orm import aliased
|
from sqlalchemy.orm import aliased
|
||||||
from sqlalchemy import Select, and_, desc, func, or_, text
|
from sqlalchemy import Select, and_, func, or_, text
|
||||||
from advanced_alchemy.filters import (
|
from advanced_alchemy.filters import (
|
||||||
StatementTypeT,
|
StatementTypeT,
|
||||||
StatementFilter,
|
StatementFilter,
|
||||||
CollectionFilter,
|
|
||||||
ModelT,
|
ModelT,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -135,7 +134,7 @@ class ProgressFilter(StatementFilter):
|
|||||||
status_conditions.append(
|
status_conditions.append(
|
||||||
and_(
|
and_(
|
||||||
or_(
|
or_(
|
||||||
m.BookProgress.completed == False,
|
m.BookProgress.completed.is_(False),
|
||||||
m.BookProgress.completed.is_(None),
|
m.BookProgress.completed.is_(None),
|
||||||
),
|
),
|
||||||
m.BookProgress.percentage > 0,
|
m.BookProgress.percentage > 0,
|
||||||
@@ -143,7 +142,7 @@ class ProgressFilter(StatementFilter):
|
|||||||
)
|
)
|
||||||
|
|
||||||
if ProgressStatus.READ in self.statuses:
|
if ProgressStatus.READ in self.statuses:
|
||||||
status_conditions.append(m.BookProgress.completed == True)
|
status_conditions.append(m.BookProgress.completed.is_(True))
|
||||||
|
|
||||||
if ProgressStatus.UNREAD in self.statuses:
|
if ProgressStatus.UNREAD in self.statuses:
|
||||||
status_conditions.append(m.BookProgress.id.is_(None))
|
status_conditions.append(m.BookProgress.id.is_(None))
|
||||||
@@ -154,6 +153,7 @@ class ProgressFilter(StatementFilter):
|
|||||||
@dataclass
|
@dataclass
|
||||||
class FileFilter(StatementFilter):
|
class FileFilter(StatementFilter):
|
||||||
"""Filter books that are related to the given files."""
|
"""Filter books that are related to the given files."""
|
||||||
|
|
||||||
file_ids: list[int]
|
file_ids: list[int]
|
||||||
|
|
||||||
def append_to_statement(
|
def append_to_statement(
|
||||||
@@ -165,17 +165,21 @@ class FileFilter(StatementFilter):
|
|||||||
|
|
||||||
return super().append_to_statement(statement, model, *args, **kwargs)
|
return super().append_to_statement(statement, model, *args, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class FileHashFilter(StatementFilter):
|
class FileHashFilter(StatementFilter):
|
||||||
file_hashes: list[str]
|
file_hashes: list[str]
|
||||||
|
|
||||||
def append_to_statement(self, statement: StatementTypeT, model: type[ModelT], *args, **kwargs) -> StatementTypeT:
|
def append_to_statement(
|
||||||
|
self, statement: StatementTypeT, model: type[ModelT], *args, **kwargs
|
||||||
|
) -> StatementTypeT:
|
||||||
statement = statement.where(
|
statement = statement.where(
|
||||||
m.Book.files.any(m.FileMetadata.hash.in_(self.file_hashes))
|
m.Book.files.any(m.FileMetadata.hash.in_(self.file_hashes))
|
||||||
)
|
)
|
||||||
|
|
||||||
return super().append_to_statement(statement, model, *args, **kwargs)
|
return super().append_to_statement(statement, model, *args, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class CustomOrderBy(StatementFilter):
|
class CustomOrderBy(StatementFilter):
|
||||||
"""Order by filter with support for 'random' and 'last accessed' orderings."""
|
"""Order by filter with support for 'random' and 'last accessed' orderings."""
|
||||||
|
|||||||
@@ -1,16 +1,21 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
import secrets
|
import secrets
|
||||||
from chitai.database.models.kosync_device import KosyncDevice
|
from chitai.database.models.kosync_device import KosyncDevice
|
||||||
from advanced_alchemy.service import SQLAlchemyAsyncRepositoryService, ModelDictT, schema_dump
|
from advanced_alchemy.service import (
|
||||||
|
SQLAlchemyAsyncRepositoryService,
|
||||||
|
ModelDictT,
|
||||||
|
schema_dump,
|
||||||
|
)
|
||||||
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
||||||
|
|
||||||
|
|
||||||
class KosyncDeviceService(SQLAlchemyAsyncRepositoryService[KosyncDevice]):
|
class KosyncDeviceService(SQLAlchemyAsyncRepositoryService[KosyncDevice]):
|
||||||
"""Service for managing KOReader devices."""
|
"""Service for managing KOReader devices."""
|
||||||
|
|
||||||
API_KEY_LENGTH_IN_BYTES = 8
|
API_KEY_LENGTH_IN_BYTES = 8
|
||||||
|
|
||||||
class Repo(SQLAlchemyAsyncRepository[KosyncDevice]):
|
class Repo(SQLAlchemyAsyncRepository[KosyncDevice]):
|
||||||
""" Repository for KosyncDevice entities."""
|
"""Repository for KosyncDevice entities."""
|
||||||
|
|
||||||
model_type = KosyncDevice
|
model_type = KosyncDevice
|
||||||
|
|
||||||
@@ -18,18 +23,17 @@ class KosyncDeviceService(SQLAlchemyAsyncRepositoryService[KosyncDevice]):
|
|||||||
|
|
||||||
async def create(self, data: ModelDictT[KosyncDevice], **kwargs) -> KosyncDevice:
|
async def create(self, data: ModelDictT[KosyncDevice], **kwargs) -> KosyncDevice:
|
||||||
data = schema_dump(data)
|
data = schema_dump(data)
|
||||||
data['api_key'] = self._generate_api_key()
|
data["api_key"] = self._generate_api_key()
|
||||||
return await super().create(data, **kwargs)
|
return await super().create(data, **kwargs)
|
||||||
|
|
||||||
async def get_by_api_key(self, api_key: str) -> KosyncDevice:
|
async def get_by_api_key(self, api_key: str) -> KosyncDevice:
|
||||||
return await self.get_one(KosyncDevice.api_key == api_key)
|
return await self.get_one(KosyncDevice.api_key == api_key)
|
||||||
|
|
||||||
async def regenerate_api_key(self, device_id: int) -> KosyncDevice:
|
async def regenerate_api_key(self, device_id: int) -> KosyncDevice:
|
||||||
device = await self.get(device_id)
|
device = await self.get(device_id)
|
||||||
api_key = self._generate_api_key()
|
api_key = self._generate_api_key()
|
||||||
device.api_key = api_key
|
device.api_key = api_key
|
||||||
return await self.update(device)
|
return await self.update(device)
|
||||||
|
|
||||||
|
|
||||||
def _generate_api_key(self) -> str:
|
def _generate_api_key(self) -> str:
|
||||||
return secrets.token_hex(self.API_KEY_LENGTH_IN_BYTES)
|
return secrets.token_hex(self.API_KEY_LENGTH_IN_BYTES)
|
||||||
|
|||||||
@@ -16,7 +16,9 @@ class KosyncProgressService(SQLAlchemyAsyncRepositoryService[KosyncProgress]):
|
|||||||
|
|
||||||
repository_type = Repo
|
repository_type = Repo
|
||||||
|
|
||||||
async def get_by_document_hash(self, user_id: int, document: str) -> KosyncProgress | None:
|
async def get_by_document_hash(
|
||||||
|
self, user_id: int, document: str
|
||||||
|
) -> KosyncProgress | None:
|
||||||
"""Get progress for a specific document and user."""
|
"""Get progress for a specific document and user."""
|
||||||
return await self.get_one_or_none(
|
return await self.get_one_or_none(
|
||||||
KosyncProgress.user_id == user_id,
|
KosyncProgress.user_id == user_id,
|
||||||
|
|||||||
@@ -5,7 +5,6 @@ from pathlib import Path
|
|||||||
from advanced_alchemy.service import SQLAlchemyAsyncRepositoryService
|
from advanced_alchemy.service import SQLAlchemyAsyncRepositoryService
|
||||||
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
from advanced_alchemy.repository import SQLAlchemyAsyncRepository
|
||||||
from advanced_alchemy import service
|
from advanced_alchemy import service
|
||||||
from advanced_alchemy.utils.text import slugify
|
|
||||||
|
|
||||||
# Local imports
|
# Local imports
|
||||||
from chitai.database.models.library import Library
|
from chitai.database.models.library import Library
|
||||||
@@ -18,6 +17,7 @@ from chitai.services.utils import (
|
|||||||
|
|
||||||
from chitai.config import settings
|
from chitai.config import settings
|
||||||
|
|
||||||
|
|
||||||
class LibraryService(SQLAlchemyAsyncRepositoryService[Library]):
|
class LibraryService(SQLAlchemyAsyncRepositoryService[Library]):
|
||||||
"""Service for managing libraries and their configuration."""
|
"""Service for managing libraries and their configuration."""
|
||||||
|
|
||||||
@@ -48,7 +48,7 @@ class LibraryService(SQLAlchemyAsyncRepositoryService[Library]):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
# TODO: What if a library root_path is a child of an existing library?
|
# TODO: What if a library root_path is a child of an existing library?
|
||||||
if existing := await self.list(Library.root_path == library.root_path):
|
if await self.list(Library.root_path == library.root_path):
|
||||||
raise ValueError(f"Library already exists at '{library.root_path}'")
|
raise ValueError(f"Library already exists at '{library.root_path}'")
|
||||||
|
|
||||||
if library.read_only:
|
if library.read_only:
|
||||||
@@ -57,8 +57,7 @@ class LibraryService(SQLAlchemyAsyncRepositoryService[Library]):
|
|||||||
raise DirectoryDoesNotExist(
|
raise DirectoryDoesNotExist(
|
||||||
f"Root directory '{library.root_path}' must exist for a read-only library"
|
f"Root directory '{library.root_path}' must exist for a read-only library"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# TODO: Verify the read-only library has read permissions
|
# TODO: Verify the read-only library has read permissions
|
||||||
created_library = await super().create(
|
created_library = await super().create(
|
||||||
service.schema_dump(library, exclude_unset=False), **kwargs
|
service.schema_dump(library, exclude_unset=False), **kwargs
|
||||||
@@ -71,9 +70,6 @@ class LibraryService(SQLAlchemyAsyncRepositoryService[Library]):
|
|||||||
await create_directory(Path(settings.consume_path) / Path(library.slug))
|
await create_directory(Path(settings.consume_path) / Path(library.slug))
|
||||||
|
|
||||||
return created_library
|
return created_library
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# TODO: Implement library deletion and optional file deletion
|
# TODO: Implement library deletion and optional file deletion
|
||||||
async def delete(
|
async def delete(
|
||||||
|
|||||||
@@ -0,0 +1,207 @@
|
|||||||
|
# src/chitai/services/matching.py
|
||||||
|
|
||||||
|
"""
|
||||||
|
Normalisation for book-level duplicate detection.
|
||||||
|
|
||||||
|
Two copies of one book rarely agree on how it is written down. One says
|
||||||
|
`The Metamorphosis`, the other `Metamorphosis`; one credits `Kafka, Franz`, the other
|
||||||
|
`Franz Kafka`; one carries the ISBN-10 and the other the ISBN-13 of the same edition.
|
||||||
|
These functions reduce each of those to a single key, so the comparison is an equality
|
||||||
|
test the database can index rather than a similarity score nobody can explain.
|
||||||
|
|
||||||
|
Everything here is pure: the keys are computed once and stored on the row (see
|
||||||
|
`Book.normalized_title`, `Author.normalized_name`, `Identifier.normalized_value`), so
|
||||||
|
no Postgres extension is needed at query time.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
|
|
||||||
|
from chitai.services.utils import is_valid_isbn, isbn10_to_isbn13
|
||||||
|
|
||||||
|
|
||||||
|
# Asides a title carries that say nothing about which book it is:
|
||||||
|
# "Frankenstein (Illustrated)", "Dune [Deluxe]".
|
||||||
|
_BRACKETED = re.compile(r"[(\[{][^)\]}]*[)\]}]")
|
||||||
|
|
||||||
|
# Edition and format qualifiers, matched only as a *trailing* run of words. Anchoring
|
||||||
|
# to the end is what keeps "The Illustrated Man" a book and "Moby Dick Illustrated" a
|
||||||
|
# format note — a qualifier trails the title, it is never the thing the title is about.
|
||||||
|
_EDITION_NOISE = re.compile(
|
||||||
|
r"\s+(?:"
|
||||||
|
# "2nd edition", but also the compact forms publishers actually print on a
|
||||||
|
# cover: "2E", "3 Ed", "5e". The number alone is never enough — "Catch 22" is
|
||||||
|
# a title and must survive.
|
||||||
|
r"\d+(?:st|nd|rd|th)?\s*(?:edition|edn|ed|e)"
|
||||||
|
r"|(?:first|second|third|fourth|fifth|sixth|new|revised|expanded|updated|"
|
||||||
|
r"annotated|illustrated|unabridged|abridged|complete|definitive|deluxe|"
|
||||||
|
r"anniversary|collectors|international|kindle|paperback|hardcover|hardback|"
|
||||||
|
r"ebook|audiobook)"
|
||||||
|
r"(?:\s+(?:and|&)\s+\w+)*"
|
||||||
|
r"(?:\s+ed(?:ition|n)?)?"
|
||||||
|
r")$"
|
||||||
|
)
|
||||||
|
|
||||||
|
_LEADING_ARTICLE = re.compile(r"^(?:the|a|an)\s+")
|
||||||
|
|
||||||
|
# Anything that is not a letter, a digit or a space, once accents are gone.
|
||||||
|
_PUNCTUATION = re.compile(r"[^0-9a-z ]+")
|
||||||
|
|
||||||
|
_WHITESPACE = re.compile(r"\s+")
|
||||||
|
|
||||||
|
# Junk an extractor leaves on the end of a name: the `;` from a `DC:creator` list, the
|
||||||
|
# `.epub` from a filename the author's name was read out of.
|
||||||
|
_TRAILING_SEPARATORS = " ;,&/"
|
||||||
|
_FILE_EXTENSION = re.compile(
|
||||||
|
r"\.(?:epub|pdf|mobi|azw3?|djvu|fb2|txt|cbz|cbr)$", re.IGNORECASE
|
||||||
|
)
|
||||||
|
|
||||||
|
# `J. R. R.` survives punctuation stripping as three one-letter words; `J.R.R.` as one.
|
||||||
|
# Joining any run of them makes both `jrr`.
|
||||||
|
_INITIAL_RUN = re.compile(r"\b(?:[a-z] )+[a-z]\b")
|
||||||
|
|
||||||
|
# ISBNs are the same number under several names; everything else keeps its own.
|
||||||
|
_ISBN_NAMES = {"isbn", "isbn-10", "isbn10", "isbn-13", "isbn13"}
|
||||||
|
|
||||||
|
# Generated fresh for every build of a file, so two copies of one book never share one.
|
||||||
|
# Matching on them would only re-find files the hash check already catches.
|
||||||
|
_PER_BUILD_NAMES = {"uuid", "urn:uuid"}
|
||||||
|
|
||||||
|
# Below this an identifier is not specific enough to be evidence: a Calibre `id` of
|
||||||
|
# "42" would otherwise pair two unrelated books.
|
||||||
|
_MIN_IDENTIFIER_LENGTH = 4
|
||||||
|
|
||||||
|
|
||||||
|
def _fold(text: str) -> str:
|
||||||
|
"""Casefolded, accent-free, punctuation-free, single-spaced."""
|
||||||
|
decomposed = unicodedata.normalize("NFKD", text.casefold())
|
||||||
|
unaccented = "".join(c for c in decomposed if not unicodedata.combining(c))
|
||||||
|
|
||||||
|
return _WHITESPACE.sub(" ", _PUNCTUATION.sub(" ", unaccented)).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_title(title: str | None) -> str:
|
||||||
|
"""
|
||||||
|
Reduce a title to the key two copies of one book should share.
|
||||||
|
|
||||||
|
`Book.subtitle` is already split off by `Extractor.format_book_title`, so only what
|
||||||
|
is left in the title column is considered here.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
title: The title as it was stored.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The comparison key, or an empty string if nothing survives normalisation —
|
||||||
|
which is the signal not to match on the title at all.
|
||||||
|
"""
|
||||||
|
if not title:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
folded = _fold(_BRACKETED.sub(" ", title).replace("&", " and "))
|
||||||
|
|
||||||
|
# Repeated because qualifiers stack: "Dune Deluxe Edition Illustrated".
|
||||||
|
while (trimmed := _EDITION_NOISE.sub("", folded)) != folded:
|
||||||
|
folded = trimmed
|
||||||
|
|
||||||
|
# An article says nothing, but a title that is only an article is not improved by
|
||||||
|
# having none, and neither is one that noise removal emptied out.
|
||||||
|
return _LEADING_ARTICLE.sub("", folded, count=1) or folded
|
||||||
|
|
||||||
|
|
||||||
|
def format_author_name(name: str | None) -> str:
|
||||||
|
"""
|
||||||
|
Tidy an author's name into the one form the library writes them in.
|
||||||
|
|
||||||
|
Distinct from `normalize_author`, which throws away case, accents and spacing to
|
||||||
|
build a comparison key. This one is what a reader sees, so it keeps everything
|
||||||
|
that belongs to the name and only removes what an extractor added: a trailing
|
||||||
|
separator left over from a creator list, a file extension carried in from a
|
||||||
|
filename, and the `Surname, Given` ordering that EPUBs file names under.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
name: The name as the file or filename gave it.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The name to store, or an empty string if there is nothing left of it.
|
||||||
|
"""
|
||||||
|
if not name:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
tidied = _FILE_EXTENSION.sub("", name.strip().strip(_TRAILING_SEPARATORS).strip())
|
||||||
|
|
||||||
|
if tidied.count(",") == 1:
|
||||||
|
surname, given = (part.strip() for part in tidied.split(","))
|
||||||
|
|
||||||
|
# Only when the part before the comma is a single word. "Dave Thomas, Andy
|
||||||
|
# Hunt" is two people in one string, and flipping it would invent a third
|
||||||
|
# person who does not exist. Leaving an unrecognised form alone is the safe
|
||||||
|
# failure; rewriting it wrongly is not.
|
||||||
|
if surname and given and " " not in surname:
|
||||||
|
tidied = f"{given} {surname}"
|
||||||
|
|
||||||
|
return _WHITESPACE.sub(" ", tidied).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_author(name: str | None) -> str:
|
||||||
|
"""
|
||||||
|
Reduce an author's name to the key their other books should share.
|
||||||
|
|
||||||
|
Deliberately not reduced to surname plus initial: that collides unrelated people,
|
||||||
|
and a wrong match here is a book pointed at a stranger's shelf.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
name: The name as it was stored, in either `Franz Kafka` or `Kafka, Franz` form.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The comparison key, or an empty string if nothing survives normalisation.
|
||||||
|
"""
|
||||||
|
if not name:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# `Kafka, Franz` is one name written backwards. More than one comma is a list, or a
|
||||||
|
# suffix, and guessing at either does more harm than leaving it alone.
|
||||||
|
if name.count(",") == 1:
|
||||||
|
surname, forename = name.split(",")
|
||||||
|
name = f"{forename.strip()} {surname.strip()}"
|
||||||
|
|
||||||
|
folded = _fold(name)
|
||||||
|
|
||||||
|
return _INITIAL_RUN.sub(lambda run: run.group().replace(" ", ""), folded)
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_identifier(name: str, value: str) -> str | None:
|
||||||
|
"""
|
||||||
|
Reduce one identifier to a `scheme:value` key, if it can carry a match at all.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
name: What kind of identifier it is, as stored.
|
||||||
|
value: The identifier itself.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The key, or None when the identifier is no use for matching: a per-build UUID,
|
||||||
|
something too short to be evidence, or an ISBN that fails its own checksum.
|
||||||
|
"""
|
||||||
|
name = (name or "").strip().casefold()
|
||||||
|
value = (value or "").strip()
|
||||||
|
|
||||||
|
if not name or not value or name in _PER_BUILD_NAMES:
|
||||||
|
return None
|
||||||
|
|
||||||
|
if name in _ISBN_NAMES:
|
||||||
|
digits = re.sub(r"[^0-9Xx]", "", value).upper()
|
||||||
|
|
||||||
|
if not is_valid_isbn(digits):
|
||||||
|
return None
|
||||||
|
|
||||||
|
# One scheme for both forms: a publisher prints whichever it likes, and the
|
||||||
|
# ISBN-10 and ISBN-13 of an edition are the same number written twice.
|
||||||
|
isbn = digits if len(digits) == 13 else isbn10_to_isbn13(digits)
|
||||||
|
return f"isbn:{isbn}" if isbn else None
|
||||||
|
|
||||||
|
folded = _fold(value) or value.casefold()
|
||||||
|
if len(folded) < _MIN_IDENTIFIER_LENGTH:
|
||||||
|
return None
|
||||||
|
|
||||||
|
return f"{name}:{folded}"
|
||||||
@@ -3,7 +3,6 @@
|
|||||||
# TODO: Code is a mess. Clean it up and add docstrings
|
# TODO: Code is a mess. Clean it up and add docstrings
|
||||||
|
|
||||||
# Standard library
|
# Standard library
|
||||||
from abc import ABC, abstractmethod
|
|
||||||
import datetime
|
import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
@@ -31,6 +30,159 @@ from chitai.services.utils import (
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
# Identifier schemes an EPUB can declare, mapped onto the names the rest of the app
|
||||||
|
# uses. A scheme arrives either as an `opf:scheme` attribute or as a prefix on the
|
||||||
|
# value itself (`urn:isbn:…`, `calibre:…`), and the two say the same thing.
|
||||||
|
_IDENTIFIER_SCHEMES = {
|
||||||
|
"isbn": "isbn",
|
||||||
|
"isbn10": "isbn-10",
|
||||||
|
"isbn-10": "isbn-10",
|
||||||
|
"isbn13": "isbn-13",
|
||||||
|
"isbn-13": "isbn-13",
|
||||||
|
"uuid": "uuid",
|
||||||
|
"calibre": "calibre",
|
||||||
|
"doi": "doi",
|
||||||
|
"asin": "asin",
|
||||||
|
"amazon": "asin",
|
||||||
|
"mobi-asin": "asin",
|
||||||
|
"google": "google",
|
||||||
|
"goodreads": "goodreads",
|
||||||
|
}
|
||||||
|
|
||||||
|
# `scheme:rest`, with an optional `urn:` in front of it.
|
||||||
|
_SCHEME_PREFIX = re.compile(r"^(?:urn:)?([A-Za-z][A-Za-z0-9.-]*):(.+)$")
|
||||||
|
|
||||||
|
_UUID = re.compile(r"^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$", re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_identifier(value: str, scheme: str | None = None) -> tuple[str, str] | None:
|
||||||
|
"""
|
||||||
|
Work out what one raw identifier is, and what it is worth storing as.
|
||||||
|
|
||||||
|
EPUBs write the same ISBN as `9780486282114`, `978-0-486-28211-4` and
|
||||||
|
`urn:isbn:978-0-486-28211-4`, and carry plenty of identifiers that are not ISBNs
|
||||||
|
at all. Validating the string verbatim keeps only the first form and throws the
|
||||||
|
rest away, so normalise first and name whatever survives.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value: The identifier as the file wrote it.
|
||||||
|
scheme: What the file said it is, if it said anything — an `opf:scheme`
|
||||||
|
attribute. A prefix on the value takes precedence over this.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The `(name, value)` to store, or `None` when there is nothing usable: an
|
||||||
|
empty value, or one declared to be an ISBN that fails its own checksum.
|
||||||
|
"""
|
||||||
|
value = (value or "").strip()
|
||||||
|
if not value:
|
||||||
|
return None
|
||||||
|
|
||||||
|
name = _IDENTIFIER_SCHEMES.get((scheme or "").strip().casefold())
|
||||||
|
|
||||||
|
# An unrecognised prefix is part of the value rather than a scheme —
|
||||||
|
# "http://example.com/book" is not an identifier called "http".
|
||||||
|
if (match := _SCHEME_PREFIX.match(value)) and (
|
||||||
|
prefixed := _IDENTIFIER_SCHEMES.get(match.group(1).casefold())
|
||||||
|
):
|
||||||
|
name = prefixed
|
||||||
|
value = match.group(2).strip()
|
||||||
|
|
||||||
|
if name is None or name.startswith("isbn"):
|
||||||
|
digits = re.sub(r"[^0-9Xx]", "", value).upper()
|
||||||
|
if is_valid_isbn(digits):
|
||||||
|
return f"isbn-{len(digits)}", digits
|
||||||
|
|
||||||
|
# Something that announced itself as an ISBN and is not one carries no
|
||||||
|
# information: storing it would link the reader to a page that does not exist.
|
||||||
|
if name is not None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
return name or ("uuid" if _UUID.match(value) else "id"), value
|
||||||
|
|
||||||
|
|
||||||
|
# Numbered editions, in the forms covers and catalogue records actually use:
|
||||||
|
# "3rd Edition", "2E", "4e", "8_e", "/6e", "(2nd edition)", "Third International Edition".
|
||||||
|
_ORDINAL_WORDS = {
|
||||||
|
"first": 1,
|
||||||
|
"second": 2,
|
||||||
|
"third": 3,
|
||||||
|
"fourth": 4,
|
||||||
|
"fifth": 5,
|
||||||
|
"sixth": 6,
|
||||||
|
"seventh": 7,
|
||||||
|
"eighth": 8,
|
||||||
|
"ninth": 9,
|
||||||
|
"tenth": 10,
|
||||||
|
"eleventh": 11,
|
||||||
|
"twelfth": 12,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Words that sit between the number and "Edition" and belong to the same statement.
|
||||||
|
_EDITION_QUALIFIER = (
|
||||||
|
r"(?:international|global|revised|updated|expanded|anniversary|deluxe|student|"
|
||||||
|
r"instructors?|annotated|illustrated|reprint)"
|
||||||
|
)
|
||||||
|
|
||||||
|
_EDITION = re.compile(
|
||||||
|
rf"""
|
||||||
|
[\s,;:/\-–—(\[]+ # the separator the statement hangs off
|
||||||
|
(?:
|
||||||
|
(?P<num>\d{{1,2}})\s*(?:st|nd|rd|th)?[\s_]*
|
||||||
|
(?:{_EDITION_QUALIFIER}\s+)*(?:edition\b|edn\b|ed\b\.?|e\b)
|
||||||
|
| (?P<word>{"|".join(_ORDINAL_WORDS)})\s+
|
||||||
|
(?:{_EDITION_QUALIFIER}\s+)*(?:edition\b|edn\b|ed\b\.?)
|
||||||
|
)
|
||||||
|
[\s)\]]* # and its closing bracket, if it had one
|
||||||
|
""",
|
||||||
|
re.IGNORECASE | re.VERBOSE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def split_edition(title: str | None) -> tuple[str | None, int | None]:
|
||||||
|
"""
|
||||||
|
Separate a numbered edition statement from the title it is written into.
|
||||||
|
|
||||||
|
"Fluent Python, 2nd Edition" is one book with a field for the edition, not a
|
||||||
|
title. Left in place it also splits the library: the second edition never looks
|
||||||
|
like the first, and neither matches the copy whose file simply did not mention it.
|
||||||
|
|
||||||
|
The number is what makes this safe. Nothing is stripped without one, so
|
||||||
|
"Catch 22" and "Blade Runner 2049" keep their numbers and "Global Edition" —
|
||||||
|
which is a variant, not a numbered edition, and has nowhere to go in an
|
||||||
|
integer column — is left in the title where it can still be read.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
title: The title as the file or filename gave it.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The title without the edition statement, and the edition number. The title
|
||||||
|
unchanged and None when there is no numbered edition in it, or when removing
|
||||||
|
it would leave nothing behind.
|
||||||
|
"""
|
||||||
|
if not title:
|
||||||
|
return title, None
|
||||||
|
|
||||||
|
if (match := _EDITION.search(title)) is None:
|
||||||
|
return title, None
|
||||||
|
|
||||||
|
edition = (
|
||||||
|
int(match["num"]) if match["num"] else _ORDINAL_WORDS[match["word"].casefold()]
|
||||||
|
)
|
||||||
|
|
||||||
|
stripped = _EDITION.sub(" ", title)
|
||||||
|
stripped = re.sub(r"\s{2,}", " ", stripped)
|
||||||
|
stripped = re.sub(
|
||||||
|
r"\s+([,;:.!?])", r"\1", stripped
|
||||||
|
) # "Works : What" → "Works: What"
|
||||||
|
stripped = stripped.strip(" ,;:-–—/")
|
||||||
|
|
||||||
|
# A title that is only an edition statement is not improved by having none.
|
||||||
|
if not stripped:
|
||||||
|
return title, None
|
||||||
|
|
||||||
|
return stripped, edition
|
||||||
|
|
||||||
|
|
||||||
class FileExtractor(Protocol):
|
class FileExtractor(Protocol):
|
||||||
@classmethod
|
@classmethod
|
||||||
async def extract_metadata(
|
async def extract_metadata(
|
||||||
@@ -38,7 +190,9 @@ class FileExtractor(Protocol):
|
|||||||
) -> dict[str, Any]: ...
|
) -> dict[str, Any]: ...
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
async def extract_text(cls, input: UploadFile | BinaryIO | bytes | Path | str) -> str: ...
|
async def extract_text(
|
||||||
|
cls, input: UploadFile | BinaryIO | bytes | Path | str
|
||||||
|
) -> str: ...
|
||||||
|
|
||||||
|
|
||||||
class Extractor:
|
class Extractor:
|
||||||
@@ -47,32 +201,65 @@ class Extractor:
|
|||||||
format_priorities = {"epub": 1, "pdf": 2}
|
format_priorities = {"epub": 1, "pdf": 2}
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
async def extract_metadata(cls, files: list[UploadFile] | list[Path], root_path: Path | None = None) -> dict[str, Any]:
|
async def extract_metadata(
|
||||||
|
cls, files: list[UploadFile] | list[Path], root_path: Path | None = None
|
||||||
|
) -> dict[str, Any]:
|
||||||
metadata = {}
|
metadata = {}
|
||||||
|
|
||||||
# Sort based on file priority
|
# Sort based on file priority
|
||||||
# EPUB tends to give better metadata results over pdf
|
# EPUB tends to give better metadata results over pdf
|
||||||
sorted_files = sorted(files, key=lambda f: Extractor._get_file_priority(f))
|
sorted_files = sorted(files, key=lambda f: Extractor._get_file_priority(f))
|
||||||
|
|
||||||
|
# Identifiers accumulate across formats instead of replacing each other. Every
|
||||||
|
# other field is a single value where the later, better-trusted format simply
|
||||||
|
# wins, but identifiers are a *collection*: a book holding an EPUB and a PDF
|
||||||
|
# genuinely carries what both of them declare, and merging the dict wholesale
|
||||||
|
# threw away everything the earlier format found. An EPUB that declares an
|
||||||
|
# ASIN, a Google volume id and a Calibre id kept none of them once a PDF
|
||||||
|
# contributed a single ISBN.
|
||||||
|
identifiers: dict[str, str] = {}
|
||||||
|
|
||||||
for file in sorted_files:
|
for file in sorted_files:
|
||||||
match get_file_extension(file):
|
match get_file_extension(file):
|
||||||
case "epub":
|
case "epub":
|
||||||
metadata = metadata | await EpubExtractor.extract_metadata(file)
|
extracted = await EpubExtractor.extract_metadata(file)
|
||||||
case "pdf":
|
case "pdf":
|
||||||
metadata = metadata | await PdfExtractor.extract_metadata(file)
|
extracted = await PdfExtractor.extract_metadata(file)
|
||||||
case _:
|
case _:
|
||||||
break
|
break
|
||||||
|
|
||||||
|
# First writer wins per name, and the files are already ordered by how
|
||||||
|
# far their metadata can be trusted. A `dc:identifier` the publisher
|
||||||
|
# declared outranks an ISBN scraped out of a PDF's copyright page, which
|
||||||
|
# routinely prints the ISBNs of other formats and older editions too.
|
||||||
|
for name, value in (extracted.pop("identifiers", None) or {}).items():
|
||||||
|
identifiers.setdefault(name, value)
|
||||||
|
|
||||||
|
metadata = metadata | extracted
|
||||||
|
|
||||||
|
if identifiers:
|
||||||
|
metadata["identifiers"] = identifiers
|
||||||
|
|
||||||
# Get metadata from file names
|
# Get metadata from file names
|
||||||
for file in files:
|
for file in files:
|
||||||
metadata = FilenameExtractor.extract_metadata(file) | metadata
|
metadata = FilenameExtractor.extract_metadata(file) | metadata
|
||||||
|
|
||||||
# Get metadata from filepath
|
# Get metadata from filepath. Kept on the left so that anything the file
|
||||||
metadata = metadata | FilepathExtractor.extract_metadata(files[0], root_path)
|
# itself declared outranks a guess made from its directory names — a folder
|
||||||
|
# called "Fluent Python - Luciano Ramalho" must not overwrite the title the
|
||||||
|
# EPUB already carries.
|
||||||
|
metadata = FilepathExtractor.extract_metadata(files[0], root_path) | metadata
|
||||||
|
|
||||||
# format the title
|
# format the title
|
||||||
if metadata.get('title', None):
|
if metadata.get("title", None):
|
||||||
title, subtitle = Extractor.format_book_title(metadata["title"])
|
# Before the subtitle split, so the edition cannot be mistaken for one:
|
||||||
|
# "How Linux Works, 3rd Edition: What Every Superuser Should Know" has to
|
||||||
|
# lose the edition first for the colon count to mean anything.
|
||||||
|
title, edition = split_edition(metadata["title"])
|
||||||
|
if edition is not None:
|
||||||
|
metadata.setdefault("edition", edition)
|
||||||
|
|
||||||
|
title, subtitle = Extractor.format_book_title(title)
|
||||||
metadata["title"] = title
|
metadata["title"] = title
|
||||||
metadata["subtitle"] = subtitle
|
metadata["subtitle"] = subtitle
|
||||||
|
|
||||||
@@ -111,8 +298,8 @@ class Extractor:
|
|||||||
file_ext = get_file_extension(filename)
|
file_ext = get_file_extension(filename)
|
||||||
|
|
||||||
if file_ext is None:
|
if file_ext is None:
|
||||||
return float('inf')
|
return float("inf")
|
||||||
|
|
||||||
return Extractor.format_priorities.get(file_ext, float("inf"))
|
return Extractor.format_priorities.get(file_ext, float("inf"))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -233,7 +420,7 @@ class PdfExtractor(FileExtractor):
|
|||||||
try:
|
try:
|
||||||
return datetime.datetime.strptime(date_portion, "%Y%m%d").date()
|
return datetime.datetime.strptime(date_portion, "%Y%m%d").date()
|
||||||
|
|
||||||
except Exception as e:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -360,15 +547,23 @@ class EpubExtractor(FileExtractor):
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def _extract_identifiers(cls, epub: epub.EpubBook) -> dict[str, str]:
|
def _extract_identifiers(cls, epub: epub.EpubBook) -> dict[str, str]:
|
||||||
|
"""
|
||||||
|
Every `DC:identifier` the file carries, keyed by what kind of thing it is.
|
||||||
|
|
||||||
|
Non-ISBN identifiers are kept: `Identifier` is a free-form name/value pair, so
|
||||||
|
a Calibre id or an ASIN costs nothing to store and is one more thing two copies
|
||||||
|
of a book can be recognised by.
|
||||||
|
"""
|
||||||
identifiers = {}
|
identifiers = {}
|
||||||
|
|
||||||
for id in epub.get_metadata("DC", "identifier"):
|
for value, attributes in epub.get_metadata("DC", "identifier"):
|
||||||
if is_valid_isbn(id[0]):
|
scheme = None
|
||||||
if len(id[0]) == 13:
|
if isinstance(attributes, dict):
|
||||||
identifiers.update({"isbn-13": id[0]})
|
scheme = attributes.get("opf:scheme") or attributes.get("scheme")
|
||||||
|
|
||||||
elif len(id[0]) == 10:
|
if (parsed := parse_identifier(value, scheme)) is not None:
|
||||||
identifiers.update({"isbn-10": id[0]})
|
name, parsed_value = parsed
|
||||||
|
identifiers[name] = parsed_value
|
||||||
|
|
||||||
return identifiers
|
return identifiers
|
||||||
|
|
||||||
@@ -377,7 +572,7 @@ class EpubExtractor(FileExtractor):
|
|||||||
try:
|
try:
|
||||||
return epub.get_metadata("DC", "description")[0][0]
|
return epub.get_metadata("DC", "description")[0][0]
|
||||||
|
|
||||||
except:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -386,15 +581,15 @@ class EpubExtractor(FileExtractor):
|
|||||||
date_str = epub.get_metadata("DC", "date")[0][0].split("T")[0]
|
date_str = epub.get_metadata("DC", "date")[0][0].split("T")[0]
|
||||||
return datetime.date.fromisoformat(date_str)
|
return datetime.date.fromisoformat(date_str)
|
||||||
|
|
||||||
except:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def _extract_publisher(cls, epub: epub.EpubBook) -> str | None:
|
def _extract_publisher(cls, epub: epub.EpubBook) -> str | None:
|
||||||
try:
|
try:
|
||||||
epub.get_metadata("DC", "publisher")[0][0]
|
return epub.get_metadata("DC", "publisher")[0][0]
|
||||||
|
|
||||||
except:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -418,7 +613,7 @@ class EpubExtractor(FileExtractor):
|
|||||||
cover_item = epub.get_item_with_id(cover_id)
|
cover_item = epub.get_item_with_id(cover_id)
|
||||||
if cover_item:
|
if cover_item:
|
||||||
return PIL.Image.open(BytesIO(cover_item.content))
|
return PIL.Image.open(BytesIO(cover_item.content))
|
||||||
except Exception as e:
|
except Exception:
|
||||||
pass # Fallback to next strategy
|
pass # Fallback to next strategy
|
||||||
|
|
||||||
# Strategy 2: Search image filenames for "cover" keyword
|
# Strategy 2: Search image filenames for "cover" keyword
|
||||||
@@ -439,8 +634,10 @@ class FilepathExtractor(FileExtractor):
|
|||||||
"""Extracts metadata from the filepath."""
|
"""Extracts metadata from the filepath."""
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def extract_metadata(cls, input: UploadFile | Path | str, root_path: Path | None = None) -> dict[str, Any]:
|
def extract_metadata(
|
||||||
|
cls, input: UploadFile | Path | str, root_path: Path | None = None
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
|
||||||
if isinstance(input, UploadFile):
|
if isinstance(input, UploadFile):
|
||||||
path = Path(input.filename).parent
|
path = Path(input.filename).parent
|
||||||
else:
|
else:
|
||||||
@@ -448,33 +645,34 @@ class FilepathExtractor(FileExtractor):
|
|||||||
|
|
||||||
if root_path:
|
if root_path:
|
||||||
path = path.relative_to(root_path)
|
path = path.relative_to(root_path)
|
||||||
|
|
||||||
parts = path.parts
|
parts = path.parts
|
||||||
metadata: dict[str, str | None] = {}
|
metadata: dict[str, str | None] = {}
|
||||||
|
|
||||||
if len(parts) == 3:
|
if len(parts) == 3:
|
||||||
# Format: Author/Series/Part - Title/filename
|
# Format: Author/Series/Part - Title/filename
|
||||||
metadata['author'] = parts[0]
|
metadata["author"] = parts[0]
|
||||||
|
|
||||||
# Extract part number and title from directory name (parts[2])
|
# Extract part number and title from directory name (parts[2])
|
||||||
dirname = parts[2]
|
dirname = parts[2]
|
||||||
match = re.match(r'^([\d.]+)\s*-\s*(.+)$', dirname)
|
match = re.match(r"^([\d.]+)\s*-\s*(.+)$", dirname)
|
||||||
|
|
||||||
if match:
|
if match:
|
||||||
metadata['series_position'] = match.group(1) # Keep as string
|
metadata["series_position"] = match.group(1) # Keep as string
|
||||||
metadata['series'] = parts[1]
|
metadata["series"] = parts[1]
|
||||||
metadata['title'] = match.group(2).strip()
|
metadata["title"] = match.group(2).strip()
|
||||||
else:
|
else:
|
||||||
metadata['series'] = parts[1]
|
metadata["series"] = parts[1]
|
||||||
metadata['title'] = path.stem
|
metadata["title"] = path.stem
|
||||||
|
|
||||||
elif len(parts) == 2:
|
elif len(parts) == 2:
|
||||||
# Format: Author/Title
|
# Format: Author/Title
|
||||||
metadata['author'] = parts[0]
|
metadata["author"] = parts[0]
|
||||||
metadata['title'] = path.stem # Remove extension
|
metadata["title"] = path.stem # Remove extension
|
||||||
|
|
||||||
return metadata
|
return metadata
|
||||||
|
|
||||||
|
|
||||||
class FilenameExtractor(FileExtractor):
|
class FilenameExtractor(FileExtractor):
|
||||||
"""Extracts metadata from the filename."""
|
"""Extracts metadata from the filename."""
|
||||||
|
|
||||||
@@ -486,7 +684,11 @@ class FilenameExtractor(FileExtractor):
|
|||||||
elif isinstance(input, Path):
|
elif isinstance(input, Path):
|
||||||
filename = get_filename(input, ext=False)
|
filename = get_filename(input, ext=False)
|
||||||
elif isinstance(input, UploadFile):
|
elif isinstance(input, UploadFile):
|
||||||
filename = Path(input.filename).name
|
# `.stem`, not `.name`: the extension is not part of the metadata, and
|
||||||
|
# this is the browser upload path, so keeping it is how a library fills
|
||||||
|
# up with authors called "Sam Newman.epub". The other two branches have
|
||||||
|
# always stripped it.
|
||||||
|
filename = Path(input.filename).stem
|
||||||
else:
|
else:
|
||||||
raise ValueError("Input type not supported")
|
raise ValueError("Input type not supported")
|
||||||
|
|
||||||
|
|||||||
@@ -4,43 +4,57 @@ from typing import Any, Literal, Optional, Sequence
|
|||||||
from pydantic import BaseModel, Field
|
from pydantic import BaseModel, Field
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
|
|
||||||
class LinkTypes(StrEnum):
|
class LinkTypes(StrEnum):
|
||||||
NAVIGATION = "application/atom+xml;profile=opds-catalog;kind=navigation"
|
NAVIGATION = "application/atom+xml;profile=opds-catalog;kind=navigation"
|
||||||
ACQUISITION = "application/atom+xml;profile=opds-catalog;kind=acquisition"
|
ACQUISITION = "application/atom+xml;profile=opds-catalog;kind=acquisition"
|
||||||
OPEN_SEARCH = "application/opensearchdescription+xml"
|
OPEN_SEARCH = "application/opensearchdescription+xml"
|
||||||
|
|
||||||
|
|
||||||
class AcquisitionRelations(StrEnum):
|
class AcquisitionRelations(StrEnum):
|
||||||
_BASE = "http://opds-spec.org/acquisition"
|
_BASE = "http://opds-spec.org/acquisition"
|
||||||
|
|
||||||
ACQUISITION = _BASE # A generic relation that indicates that the entry may be retrieved
|
ACQUISITION = (
|
||||||
OPEN_ACCESS = f"{_BASE}/open-access" # Entry may be retrieved without any requirement
|
_BASE # A generic relation that indicates that the entry may be retrieved
|
||||||
BORROW = f"{_BASE}/borrow" # Entry may be retrieved as part of a lending transaction
|
)
|
||||||
BUY = f"{_BASE}/buy" # Entry may be retrieved as part of a purchase
|
OPEN_ACCESS = (
|
||||||
SAMPLE = f"{_BASE}/sample" # Subset of the entry may be retrieved
|
f"{_BASE}/open-access" # Entry may be retrieved without any requirement
|
||||||
PREVIEW = f"{_BASE}/preview" # Subset of the entry may be retrieved
|
)
|
||||||
SUBSCRIBE = f"{_BASE}/subscribe" # Entry my be retrieved as a part of a subscription
|
BORROW = (
|
||||||
|
f"{_BASE}/borrow" # Entry may be retrieved as part of a lending transaction
|
||||||
|
)
|
||||||
|
BUY = f"{_BASE}/buy" # Entry may be retrieved as part of a purchase
|
||||||
|
SAMPLE = f"{_BASE}/sample" # Subset of the entry may be retrieved
|
||||||
|
PREVIEW = f"{_BASE}/preview" # Subset of the entry may be retrieved
|
||||||
|
SUBSCRIBE = (
|
||||||
|
f"{_BASE}/subscribe" # Entry my be retrieved as a part of a subscription
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class NavigationRelations(StrEnum):
|
class NavigationRelations(StrEnum):
|
||||||
_BASE = ""
|
_BASE = ""
|
||||||
|
|
||||||
|
|
||||||
class LinkRelations(StrEnum):
|
class LinkRelations(StrEnum):
|
||||||
""" Link types for OPDSv1.2 related resources
|
"""Link types for OPDSv1.2 related resources
|
||||||
|
|
||||||
https://specs.opds.io/opds-1.2.html#6-additional-link-relations
|
https://specs.opds.io/opds-1.2.html#6-additional-link-relations
|
||||||
"""
|
"""
|
||||||
|
|
||||||
_BASE = "http://opds-spec.org"
|
_BASE = "http://opds-spec.org"
|
||||||
|
|
||||||
START = "start" # The OPDS catalog root
|
START = "start" # The OPDS catalog root
|
||||||
SUBSECTION = "subsection" # an OPDS feed not better described by any of the below relations
|
SUBSECTION = (
|
||||||
SHELF = f"{_BASE}/shelf" # Entries acquired by the euser
|
"subsection" # an OPDS feed not better described by any of the below relations
|
||||||
SUBSCRIPTIONS = f"{_BASE}/subscriptions" # Entries available with users's subscription
|
)
|
||||||
NEW = f"{_BASE}/sort/new" # Newest entries
|
SHELF = f"{_BASE}/shelf" # Entries acquired by the euser
|
||||||
POPULAR = f"{_BASE}/sort/popular" # Most popular entries
|
SUBSCRIPTIONS = (
|
||||||
FEATURED = f"{_BASE}/featured" # Entries selected for promotion
|
f"{_BASE}/subscriptions" # Entries available with users's subscription
|
||||||
RECOMMENDED = f"{_BASE}/recommended" # Entries recommended to the specific user
|
)
|
||||||
|
NEW = f"{_BASE}/sort/new" # Newest entries
|
||||||
|
POPULAR = f"{_BASE}/sort/popular" # Most popular entries
|
||||||
|
FEATURED = f"{_BASE}/featured" # Entries selected for promotion
|
||||||
|
RECOMMENDED = f"{_BASE}/recommended" # Entries recommended to the specific user
|
||||||
|
|
||||||
|
|
||||||
class Feed(BaseModel): # OPDS Catalog root element
|
class Feed(BaseModel): # OPDS Catalog root element
|
||||||
@@ -117,9 +131,8 @@ class AcquisitionFeedLink(Link):
|
|||||||
|
|
||||||
|
|
||||||
class NavigationFeedLink(Link):
|
class NavigationFeedLink(Link):
|
||||||
type: str = Field(
|
type: str = Field(default=LinkTypes.NAVIGATION, serialization_alias="@type")
|
||||||
default=LinkTypes.NAVIGATION, serialization_alias="@type"
|
|
||||||
)
|
|
||||||
|
|
||||||
class Content(BaseModel):
|
class Content(BaseModel):
|
||||||
type: Literal["text"] = Field(default="text", serialization_alias="@type")
|
type: Literal["text"] = Field(default="text", serialization_alias="@type")
|
||||||
@@ -170,9 +183,10 @@ class Entry(BaseModel):
|
|||||||
data = super().model_dump(**kwargs)
|
data = super().model_dump(**kwargs)
|
||||||
return {"entry": data}
|
return {"entry": data}
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class PaginationResult:
|
class PaginationResult:
|
||||||
next_link: Optional[Link]
|
next_link: Optional[Link]
|
||||||
prev_link: Optional[Link]
|
prev_link: Optional[Link]
|
||||||
current_offset: int
|
current_offset: int
|
||||||
total_count: int
|
total_count: int
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
|
|
||||||
from typing import Any, Callable, Sequence
|
from typing import Any, Callable, Sequence
|
||||||
from urllib.parse import quote_plus, urlencode
|
from urllib.parse import quote_plus, urlencode
|
||||||
from litestar import Request
|
from litestar import Request
|
||||||
@@ -16,9 +15,10 @@ from .models import (
|
|||||||
AcquisitionFeedLink,
|
AcquisitionFeedLink,
|
||||||
NavigationFeed,
|
NavigationFeed,
|
||||||
NavigationFeedLink,
|
NavigationFeedLink,
|
||||||
PaginationResult
|
PaginationResult,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def get_opensearch_document(base_url: str = "/opds/search?") -> str:
|
def get_opensearch_document(base_url: str = "/opds/search?") -> str:
|
||||||
search = {
|
search = {
|
||||||
"OpenSearchDescription": {
|
"OpenSearchDescription": {
|
||||||
@@ -47,8 +47,13 @@ def convert_book_to_entry(book: m.Book) -> Entry:
|
|||||||
link=[
|
link=[
|
||||||
ImageLink(href=f"/{book.cover_image}", type="image/webp"),
|
ImageLink(href=f"/{book.cover_image}", type="image/webp"),
|
||||||
*[
|
*[
|
||||||
|
# The only place a content type has to be a string: `Link.type` is
|
||||||
|
# required, and a null fails the whole feed rather than one entry.
|
||||||
|
# `application/octet-stream` is the registered way to say "opaque
|
||||||
|
# bytes", which is exactly what an unnamed format is.
|
||||||
AcquisitionLink(
|
AcquisitionLink(
|
||||||
href=f"/opds/download/{book.id}/{file.id}", type=file.content_type
|
href=f"/opds/download/{book.id}/{file.id}",
|
||||||
|
type=file.content_type or "application/octet-stream",
|
||||||
)
|
)
|
||||||
for file in book.files
|
for file in book.files
|
||||||
],
|
],
|
||||||
@@ -115,19 +120,20 @@ def create_navigation_feed(
|
|||||||
pretty=True,
|
pretty=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def create_library_navigation_feed(library: m.Library) -> str:
|
def create_library_navigation_feed(library: m.Library) -> str:
|
||||||
|
|
||||||
entries = [
|
entries = [
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/all-books",
|
id=f"/opds/library/{library.id}/all-books",
|
||||||
title='All Books',
|
title="All Books",
|
||||||
link=[
|
link=[
|
||||||
AcquisitionFeedLink(
|
AcquisitionFeedLink(
|
||||||
rel="subsection",
|
rel="subsection",
|
||||||
href=f"/opds/acquisition?libraries={library.id}&paginated=1&pageSize=50&feed_title=AllBooks&feed_id=/opds/library/{library.id}/all-books",
|
href=f"/opds/acquisition?libraries={library.id}&paginated=1&pageSize=50&feed_title=AllBooks&feed_id=/opds/library/{library.id}/all-books",
|
||||||
title="All Books",
|
title="All Books",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/recently-added",
|
id=f"/opds/library/{library.id}/recently-added",
|
||||||
@@ -136,9 +142,9 @@ def create_library_navigation_feed(library: m.Library) -> str:
|
|||||||
NavigationFeedLink(
|
NavigationFeedLink(
|
||||||
rel="http://opds-spec.org/sort/new",
|
rel="http://opds-spec.org/sort/new",
|
||||||
href=f"/opds/acquisition?libraries={library.id}&orderBy=created_at&pageSize=50&feed_title=RecentlyAdded&feed_id=/opds/library/{library.id}/recently-added",
|
href=f"/opds/acquisition?libraries={library.id}&orderBy=created_at&pageSize=50&feed_title=RecentlyAdded&feed_id=/opds/library/{library.id}/recently-added",
|
||||||
title="Recently Added"
|
title="Recently Added",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/shelves",
|
id=f"/opds/library/{library.id}/shelves",
|
||||||
@@ -147,9 +153,9 @@ def create_library_navigation_feed(library: m.Library) -> str:
|
|||||||
NavigationFeedLink(
|
NavigationFeedLink(
|
||||||
rel="subsection",
|
rel="subsection",
|
||||||
href=f"/opds/library/{library.id}/shelves?paginated=1&pageSize=10",
|
href=f"/opds/library/{library.id}/shelves?paginated=1&pageSize=10",
|
||||||
title="Bookshelves"
|
title="Bookshelves",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/tags",
|
id=f"/opds/library/{library.id}/tags",
|
||||||
@@ -158,9 +164,9 @@ def create_library_navigation_feed(library: m.Library) -> str:
|
|||||||
NavigationFeedLink(
|
NavigationFeedLink(
|
||||||
rel="subsection",
|
rel="subsection",
|
||||||
href=f"/opds/library/{library.id}/tags?paginated=1&pageSize=10",
|
href=f"/opds/library/{library.id}/tags?paginated=1&pageSize=10",
|
||||||
title="Tags"
|
title="Tags",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/authors",
|
id=f"/opds/library/{library.id}/authors",
|
||||||
@@ -169,9 +175,9 @@ def create_library_navigation_feed(library: m.Library) -> str:
|
|||||||
NavigationFeedLink(
|
NavigationFeedLink(
|
||||||
rel="subsection",
|
rel="subsection",
|
||||||
href=f"/opds/library/{library.id}/authors?paginated=1&pageSize=10",
|
href=f"/opds/library/{library.id}/authors?paginated=1&pageSize=10",
|
||||||
title="Authors"
|
title="Authors",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/publishers",
|
id=f"/opds/library/{library.id}/publishers",
|
||||||
@@ -180,34 +186,34 @@ def create_library_navigation_feed(library: m.Library) -> str:
|
|||||||
NavigationFeedLink(
|
NavigationFeedLink(
|
||||||
rel="subsection",
|
rel="subsection",
|
||||||
href=f"/opds/library/{library.id}/publishers?paginated=1&pageSize=10",
|
href=f"/opds/library/{library.id}/publishers?paginated=1&pageSize=10",
|
||||||
title="Publishers"
|
title="Publishers",
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
),
|
),
|
||||||
|
|
||||||
]
|
]
|
||||||
|
|
||||||
feed = create_navigation_feed(
|
feed = create_navigation_feed(
|
||||||
id=f'/library/{library.id}',
|
id=f"/library/{library.id}",
|
||||||
title=library.name,
|
title=library.name,
|
||||||
self_url=f'/opds/library/{library.id}',
|
self_url=f"/opds/library/{library.id}",
|
||||||
links=[
|
links=[],
|
||||||
|
entries=entries,
|
||||||
],
|
|
||||||
entries=entries
|
|
||||||
)
|
)
|
||||||
|
|
||||||
return feed
|
return feed
|
||||||
|
|
||||||
|
|
||||||
def create_collection_navigation_feed(
|
def create_collection_navigation_feed(
|
||||||
library: m.Library,
|
library: m.Library,
|
||||||
collection_type: str,
|
collection_type: str,
|
||||||
items: Sequence[m.BookList | m.Tag | m.Author | m.Publisher | m.BookSeries],
|
items: Sequence[m.BookList | m.Tag | m.Author | m.Publisher | m.BookSeries],
|
||||||
links: list[Link] = list(),
|
links: list[Link] = list(),
|
||||||
# Title is usually derived from the model's name or title
|
# Title is usually derived from the model's name or title
|
||||||
get_title: Callable[[Any], str] = lambda x: getattr(x, 'title', getattr(x, 'name', str(x)))
|
get_title: Callable[[Any], str] = lambda x: getattr(
|
||||||
|
x, "title", getattr(x, "name", str(x))
|
||||||
|
),
|
||||||
) -> str:
|
) -> str:
|
||||||
|
|
||||||
entries = [
|
entries = [
|
||||||
Entry(
|
Entry(
|
||||||
id=f"/opds/library/{library.id}/{collection_type}/{item.id}",
|
id=f"/opds/library/{library.id}/{collection_type}/{item.id}",
|
||||||
@@ -217,53 +223,43 @@ def create_collection_navigation_feed(
|
|||||||
href=f"/opds/acquisition?{collection_type}={item.id}&pageSize=50&paginated=1&feed_title={quote_plus(get_title(item))}&feed_id=/opds/library/{library.id}/{collection_type}/{item.id}&search=True",
|
href=f"/opds/acquisition?{collection_type}={item.id}&pageSize=50&paginated=1&feed_title={quote_plus(get_title(item))}&feed_id=/opds/library/{library.id}/{collection_type}/{item.id}&search=True",
|
||||||
title=get_title(item),
|
title=get_title(item),
|
||||||
)
|
)
|
||||||
]
|
],
|
||||||
) for item in items
|
)
|
||||||
|
for item in items
|
||||||
]
|
]
|
||||||
|
|
||||||
return create_navigation_feed(
|
return create_navigation_feed(
|
||||||
id=f"/opds/library/{library.id}/{collection_type}",
|
id=f"/opds/library/{library.id}/{collection_type}",
|
||||||
title=collection_type.title(),
|
title=collection_type.title(),
|
||||||
self_url=f'/opds/library/{library.id}/{collection_type}',
|
self_url=f"/opds/library/{library.id}/{collection_type}",
|
||||||
entries=entries,
|
entries=entries,
|
||||||
links=links
|
links=links,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def create_next_paginated_link(
|
def create_next_paginated_link(
|
||||||
request: Request,
|
request: Request, total: int, current_count: int, offset: int, feed_title: str
|
||||||
total: int,
|
) -> Link | None:
|
||||||
current_count: int,
|
if total <= current_count + offset:
|
||||||
offset: int,
|
return None
|
||||||
feed_title: str
|
|
||||||
) -> Link | None:
|
params = dict(request.query_params)
|
||||||
if total <= current_count + offset:
|
params["currentPage"] = params.get("currentPage", 1) + 1
|
||||||
return None
|
|
||||||
|
next_url = f"{request.url.path}?{urlencode(list(params.items()), doseq=True)}"
|
||||||
params = dict(request.query_params)
|
|
||||||
params['currentPage'] = params.get('currentPage', 1) + 1
|
return Link(rel="next", href=next_url, title=feed_title, type=LinkTypes.NAVIGATION)
|
||||||
|
|
||||||
next_url = f"{request.url.path}?{urlencode(list(params.items()), doseq=True)}"
|
|
||||||
|
|
||||||
return Link(
|
|
||||||
rel="next",
|
|
||||||
href=next_url,
|
|
||||||
title=feed_title,
|
|
||||||
type=LinkTypes.NAVIGATION
|
|
||||||
)
|
|
||||||
|
|
||||||
def create_search_link(
|
def create_search_link(
|
||||||
request: Request,
|
request: Request, exclude_params: set[str] | None = None
|
||||||
exclude_params: set[str] | None = None
|
|
||||||
) -> Link:
|
) -> Link:
|
||||||
"""Create search link with current filters applied"""
|
"""Create search link with current filters applied"""
|
||||||
if exclude_params is None:
|
if exclude_params is None:
|
||||||
exclude_params = {'currentPage', 'feed_title', 'feed_id', 'search', 'paginated'}
|
exclude_params = {"currentPage", "feed_title", "feed_id", "search", "paginated"}
|
||||||
|
|
||||||
params = {
|
params = {k: v for k, v in request.query_params.items() if k not in exclude_params}
|
||||||
k: v for k, v in request.query_params.items()
|
|
||||||
if k not in exclude_params
|
|
||||||
}
|
|
||||||
|
|
||||||
return Link(
|
return Link(
|
||||||
rel="search",
|
rel="search",
|
||||||
href=f"/opds/opensearch?{urlencode(list(params.items()), doseq=True)}",
|
href=f"/opds/opensearch?{urlencode(list(params.items()), doseq=True)}",
|
||||||
@@ -272,7 +268,6 @@ def create_search_link(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def create_pagination_links(
|
def create_pagination_links(
|
||||||
request: Request,
|
request: Request,
|
||||||
total: int,
|
total: int,
|
||||||
@@ -282,38 +277,35 @@ def create_pagination_links(
|
|||||||
link_type: str = LinkTypes.ACQUISITION,
|
link_type: str = LinkTypes.ACQUISITION,
|
||||||
) -> PaginationResult:
|
) -> PaginationResult:
|
||||||
"""Create next/prev pagination links using limit/offset"""
|
"""Create next/prev pagination links using limit/offset"""
|
||||||
|
|
||||||
next_link = None
|
next_link = None
|
||||||
prev_link = None
|
prev_link = None
|
||||||
|
|
||||||
# Create next link if there are more items
|
# Create next link if there are more items
|
||||||
if offset + limit < total:
|
if offset + limit < total:
|
||||||
params = dict(request.query_params)
|
params = dict(request.query_params)
|
||||||
# Calculate next page number
|
# Calculate next page number
|
||||||
current_page = (offset // limit) + 1
|
current_page = (offset // limit) + 1
|
||||||
params['currentPage'] = current_page + 1
|
params["currentPage"] = current_page + 1
|
||||||
|
|
||||||
next_url = f"{request.url.path}?{urlencode(params, doseq=True)}"
|
next_url = f"{request.url.path}?{urlencode(params, doseq=True)}"
|
||||||
next_link = Link(
|
next_link = Link(
|
||||||
rel="next",
|
rel="next", href=next_url, title=f"{feed_title} - Next", type=link_type
|
||||||
href=next_url,
|
|
||||||
title=f"{feed_title} - Next",
|
|
||||||
type=link_type
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Create previous link if not on first page
|
# Create previous link if not on first page
|
||||||
if offset > 0:
|
if offset > 0:
|
||||||
params = dict(request.query_params)
|
params = dict(request.query_params)
|
||||||
# Calculate previous page number
|
# Calculate previous page number
|
||||||
current_page = (offset // limit) + 1
|
current_page = (offset // limit) + 1
|
||||||
params['currentPage'] = max(1, current_page - 1)
|
params["currentPage"] = max(1, current_page - 1)
|
||||||
|
|
||||||
prev_url = f"{request.url.path}?{urlencode(params, doseq=True)}"
|
prev_url = f"{request.url.path}?{urlencode(params, doseq=True)}"
|
||||||
prev_link = Link(
|
prev_link = Link(
|
||||||
rel="previous",
|
rel="previous",
|
||||||
href=prev_url,
|
href=prev_url,
|
||||||
title=f"{feed_title} - Previous",
|
title=f"{feed_title} - Previous",
|
||||||
type=link_type
|
type=link_type,
|
||||||
)
|
)
|
||||||
|
|
||||||
return PaginationResult(next_link, prev_link, offset, total)
|
return PaginationResult(next_link, prev_link, offset, total)
|
||||||
|
|||||||
@@ -3,13 +3,12 @@
|
|||||||
# Standard library
|
# Standard library
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import errno
|
||||||
import hashlib
|
import hashlib
|
||||||
|
import mimetypes
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import shutil
|
import shutil
|
||||||
from typing import TYPE_CHECKING, BinaryIO
|
from typing import BinaryIO
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
from hashlib import _Hash
|
|
||||||
|
|
||||||
# Third-party libraries
|
# Third-party libraries
|
||||||
import PIL
|
import PIL
|
||||||
@@ -32,6 +31,9 @@ KO_STEP = 1024
|
|||||||
KO_SAMPLE_SIZE = 1024
|
KO_SAMPLE_SIZE = 1024
|
||||||
KO_INDICES = range(-1, 11) # -1 to 10 inclusive
|
KO_INDICES = range(-1, 11) # -1 to 10 inclusive
|
||||||
|
|
||||||
|
# How much is read at a time while hashing.
|
||||||
|
HASH_CHUNK_SIZE = 262144 # 256 KiB
|
||||||
|
|
||||||
|
|
||||||
def _lshift32(val: int, shift: int) -> int:
|
def _lshift32(val: int, shift: int) -> int:
|
||||||
"""
|
"""
|
||||||
@@ -100,10 +102,9 @@ async def calculate_koreader_hash(file_path: Path) -> str:
|
|||||||
offsets = _get_koreader_offsets()
|
offsets = _get_koreader_offsets()
|
||||||
|
|
||||||
file_pos = 0
|
file_pos = 0
|
||||||
chunk_size = 262144 # 256 KiB
|
|
||||||
|
|
||||||
async with aiofiles.open(file_path, "rb") as f:
|
async with aiofiles.open(file_path, "rb") as f:
|
||||||
while chunk := await f.read(chunk_size):
|
while chunk := await f.read(HASH_CHUNK_SIZE):
|
||||||
_partial_md5_from_chunk(chunk, hasher, offsets, file_pos)
|
_partial_md5_from_chunk(chunk, hasher, offsets, file_pos)
|
||||||
file_pos += len(chunk)
|
file_pos += len(chunk)
|
||||||
|
|
||||||
@@ -132,6 +133,49 @@ class StreamingHasher:
|
|||||||
"""Return the final hash."""
|
"""Return the final hash."""
|
||||||
return self.hasher.hexdigest()
|
return self.hasher.hexdigest()
|
||||||
|
|
||||||
|
@property
|
||||||
|
def size(self) -> int:
|
||||||
|
"""Total number of bytes fed in so far."""
|
||||||
|
return self.position
|
||||||
|
|
||||||
|
|
||||||
|
async def fingerprint_upload(file: UploadFile) -> tuple[str, int]:
|
||||||
|
"""
|
||||||
|
Calculate the hash and byte size of an uploaded file without storing it.
|
||||||
|
|
||||||
|
Duplicate detection has to answer before anything is written to the library, so
|
||||||
|
the file is read here and rewound for whoever writes it afterwards.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file: The uploaded file to read.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The file's `(hash, size)` pair.
|
||||||
|
"""
|
||||||
|
hasher = StreamingHasher()
|
||||||
|
|
||||||
|
await file.seek(0)
|
||||||
|
while chunk := await file.read(HASH_CHUNK_SIZE):
|
||||||
|
hasher.update(chunk)
|
||||||
|
await file.seek(0)
|
||||||
|
|
||||||
|
return hasher.hexdigest(), hasher.size
|
||||||
|
|
||||||
|
|
||||||
|
async def fingerprint_file(file_path: Path) -> tuple[str, int]:
|
||||||
|
"""
|
||||||
|
Calculate the hash and byte size of a file already on disk.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_path: Path to the file to read.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The file's `(hash, size)` pair.
|
||||||
|
"""
|
||||||
|
stats = await aios.stat(file_path)
|
||||||
|
return await calculate_koreader_hash(file_path), stats.st_size
|
||||||
|
|
||||||
|
|
||||||
##################################
|
##################################
|
||||||
# Filesystem related utilities #
|
# Filesystem related utilities #
|
||||||
##################################
|
##################################
|
||||||
@@ -169,10 +213,11 @@ async def create_directory(dir_path: Path | str) -> None:
|
|||||||
|
|
||||||
await aios.makedirs(dir_path, exist_ok=True)
|
await aios.makedirs(dir_path, exist_ok=True)
|
||||||
|
|
||||||
|
|
||||||
async def move_file(src_path: Path, dest_path: Path, create_dirs=True) -> None:
|
async def move_file(src_path: Path, dest_path: Path, create_dirs=True) -> None:
|
||||||
"""
|
"""
|
||||||
Move a file from source to destination asynchronously.
|
Move a file from source to destination asynchronously.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
source_path: Path to the source file
|
source_path: Path to the source file
|
||||||
destination_path: Path to the destination file
|
destination_path: Path to the destination file
|
||||||
@@ -184,7 +229,40 @@ async def move_file(src_path: Path, dest_path: Path, create_dirs=True) -> None:
|
|||||||
if dest_dir: # Only create if there's a directory path
|
if dest_dir: # Only create if there's a directory path
|
||||||
await aios.makedirs(dest_dir, exist_ok=True)
|
await aios.makedirs(dest_dir, exist_ok=True)
|
||||||
|
|
||||||
await aios.rename(src_path, dest_path)
|
try:
|
||||||
|
await aios.rename(src_path, dest_path)
|
||||||
|
except OSError as exc:
|
||||||
|
if exc.errno != errno.EXDEV:
|
||||||
|
raise
|
||||||
|
|
||||||
|
# Source and destination are on different filesystems, which rename cannot
|
||||||
|
# cross. Libraries, the consume directory and the duplicates directory are
|
||||||
|
# all configured separately, so they can easily be separate mounts.
|
||||||
|
shutil.move(str(src_path), str(dest_path))
|
||||||
|
|
||||||
|
|
||||||
|
async def copy_file(src_path: Path, dest_path: Path, create_dirs: bool = True) -> None:
|
||||||
|
"""
|
||||||
|
Copy a file, streaming it rather than reading it whole.
|
||||||
|
|
||||||
|
`shutil.copy` would block the event loop for as long as the read takes, which for a
|
||||||
|
40 MB ebook — and a few thousand of them in a row — is not acceptable.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
src_path: The file to copy. Left exactly as it is.
|
||||||
|
dest_path: Where the copy goes.
|
||||||
|
create_dirs: Create the destination's parent directories first.
|
||||||
|
"""
|
||||||
|
if create_dirs and dest_path.parent:
|
||||||
|
await aios.makedirs(dest_path.parent, exist_ok=True)
|
||||||
|
|
||||||
|
async with (
|
||||||
|
aiofiles.open(src_path, "rb") as source,
|
||||||
|
aiofiles.open(dest_path, "wb") as destination,
|
||||||
|
):
|
||||||
|
while chunk := await source.read(HASH_CHUNK_SIZE):
|
||||||
|
await destination.write(chunk)
|
||||||
|
|
||||||
|
|
||||||
async def move_dir_contents(source_dir: Path | str, target_dir: Path | str) -> None:
|
async def move_dir_contents(source_dir: Path | str, target_dir: Path | str) -> None:
|
||||||
"""
|
"""
|
||||||
@@ -375,7 +453,7 @@ def get_file_extension(file: Path | str | UploadFile) -> str | None:
|
|||||||
|
|
||||||
elif isinstance(file, UploadFile):
|
elif isinstance(file, UploadFile):
|
||||||
return Path(file.filename).suffix.lower()[1:]
|
return Path(file.filename).suffix.lower()[1:]
|
||||||
|
|
||||||
raise ValueError("file object type is not supported")
|
raise ValueError("file object type is not supported")
|
||||||
|
|
||||||
|
|
||||||
@@ -400,13 +478,73 @@ def get_filename(file: Path | str, ext: bool = True) -> str:
|
|||||||
filename = Path(file.name)
|
filename = Path(file.name)
|
||||||
else:
|
else:
|
||||||
raise ValueError("file object type is not supported")
|
raise ValueError("file object type is not supported")
|
||||||
|
|
||||||
if ext:
|
if ext:
|
||||||
return str(filename)
|
return str(filename)
|
||||||
|
|
||||||
return filename.stem
|
return filename.stem
|
||||||
|
|
||||||
|
|
||||||
|
# Content types for the ebook formats `mimetypes` does not know. Python's built-in map
|
||||||
|
# covers `.epub`, `.pdf`, `.azw3`, `.cbz`, `.cbr` and `.djvu`, and answers `None` for
|
||||||
|
# every format below — so a library imported from elsewhere, which is where MOBI and
|
||||||
|
# AZW files come from, stores nothing for them.
|
||||||
|
#
|
||||||
|
# That matters downstream because an OPDS acquisition link is how a reader app decides
|
||||||
|
# whether it can open a file at all.
|
||||||
|
|
||||||
|
# What a client sends when it does not know either. Treated as an absence rather than
|
||||||
|
# an answer: storing it would be indistinguishable from having determined a format, and
|
||||||
|
# it is the value browsers post for every extension they do not recognise.
|
||||||
|
_UNSPECIFIED = "application/octet-stream"
|
||||||
|
|
||||||
|
EBOOK_CONTENT_TYPES = {
|
||||||
|
"mobi": "application/x-mobipocket-ebook",
|
||||||
|
"prc": "application/x-mobipocket-ebook",
|
||||||
|
"azw": "application/vnd.amazon.ebook",
|
||||||
|
"fb2": "application/x-fictionbook+xml",
|
||||||
|
"fbz": "application/x-zip-compressed-fb2",
|
||||||
|
"lit": "application/x-ms-reader",
|
||||||
|
"lrf": "application/x-sony-bbeb",
|
||||||
|
"cb7": "application/x-cb7",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def guess_content_type(
|
||||||
|
file: Path | str | UploadFile, fallback: str | None = None
|
||||||
|
) -> str | None:
|
||||||
|
"""
|
||||||
|
Name a file's format from its extension.
|
||||||
|
|
||||||
|
The extension is trusted ahead of anything a client said: a browser posts
|
||||||
|
`application/octet-stream` for every format it does not recognise, which is most
|
||||||
|
ebook formats, and that answer is worth less than the `.mobi` on the end of the
|
||||||
|
name.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file: The file to name, as a path or an upload.
|
||||||
|
fallback: What to use when neither table knows the extension — a client-supplied
|
||||||
|
content type, if there is one. `application/octet-stream` is discarded: it
|
||||||
|
is the client saying it does not know, which is not information.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The content type, or None when nothing can name it. Null is the honest answer
|
||||||
|
and the column is nullable: a caller that structurally needs a string should
|
||||||
|
substitute one where it needs it, rather than have an invented value stored.
|
||||||
|
"""
|
||||||
|
extension = get_file_extension(file)
|
||||||
|
|
||||||
|
if known := EBOOK_CONTENT_TYPES.get(extension):
|
||||||
|
return known
|
||||||
|
|
||||||
|
guessed, _ = mimetypes.guess_type(get_filename(file))
|
||||||
|
|
||||||
|
if fallback == _UNSPECIFIED:
|
||||||
|
fallback = None
|
||||||
|
|
||||||
|
return guessed or fallback
|
||||||
|
|
||||||
|
|
||||||
###############################
|
###############################
|
||||||
# ISBN Validation utilities #
|
# ISBN Validation utilities #
|
||||||
###############################
|
###############################
|
||||||
@@ -432,7 +570,7 @@ def is_valid_isbn(isbn: str) -> bool:
|
|||||||
return is_valid_isbn13(isbn)
|
return is_valid_isbn13(isbn)
|
||||||
else:
|
else:
|
||||||
return False
|
return False
|
||||||
except:
|
except Exception:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
@@ -457,6 +595,29 @@ def is_valid_isbn10(isbn: str) -> bool:
|
|||||||
return str(check_digit) == isbn[-1] or (check_digit == 10 and isbn[-1] in "Xx")
|
return str(check_digit) == isbn[-1] or (check_digit == 10 and isbn[-1] in "Xx")
|
||||||
|
|
||||||
|
|
||||||
|
def isbn10_to_isbn13(isbn: str) -> str | None:
|
||||||
|
"""
|
||||||
|
Convert an ISBN-10 to the ISBN-13 naming the same edition.
|
||||||
|
|
||||||
|
The two are the same number written twice: prefix `978`, drop the ISBN-10 check
|
||||||
|
digit, recompute the check digit under the ISBN-13 rule. Matching only works if
|
||||||
|
both forms collapse onto one, since a publisher may print either.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
isbn: A 10-character ISBN, digits and an optional trailing `X` only.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The equivalent ISBN-13, or None if the input is not a valid ISBN-10.
|
||||||
|
"""
|
||||||
|
if not is_valid_isbn(isbn) or len(isbn) != 10:
|
||||||
|
return None
|
||||||
|
|
||||||
|
digits = f"978{isbn[:9]}"
|
||||||
|
total = sum(int(digit) * (1 if i % 2 == 0 else 3) for i, digit in enumerate(digits))
|
||||||
|
|
||||||
|
return f"{digits}{(10 - total % 10) % 10}"
|
||||||
|
|
||||||
|
|
||||||
def is_valid_isbn13(isbn: str) -> bool:
|
def is_valid_isbn13(isbn: str) -> bool:
|
||||||
"""
|
"""
|
||||||
Validate an ISBN-13 number using its check digit.
|
Validate an ISBN-13 number using its check digit.
|
||||||
|
|||||||
@@ -0,0 +1,282 @@
|
|||||||
|
"""
|
||||||
|
Build a Calibre library on disk, for tests to read.
|
||||||
|
|
||||||
|
Generated rather than committed as a binary `metadata.db`, because the rows worth
|
||||||
|
testing are the awkward ones — the year-101 pubdate, a `|` in an author name, a REAL
|
||||||
|
series index, HTML in a comment — and those are clearer written out in Python than
|
||||||
|
hidden inside a blob.
|
||||||
|
|
||||||
|
The schema below is Calibre's own, copied from a real library's `sqlite_master`, reduced
|
||||||
|
to the tables the reader touches. `books_pages_link` is created separately by
|
||||||
|
`add_pages`: it is recent, and a library made by an older Calibre will not have it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import shutil
|
||||||
|
import sqlite3
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
SCHEMA = """
|
||||||
|
CREATE TABLE books (
|
||||||
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||||
|
title TEXT NOT NULL DEFAULT 'Unknown' COLLATE NOCASE,
|
||||||
|
sort TEXT COLLATE NOCASE,
|
||||||
|
timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||||
|
pubdate TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||||
|
series_index REAL NOT NULL DEFAULT 1.0,
|
||||||
|
author_sort TEXT COLLATE NOCASE,
|
||||||
|
path TEXT NOT NULL DEFAULT '',
|
||||||
|
uuid TEXT,
|
||||||
|
has_cover BOOL DEFAULT 0,
|
||||||
|
last_modified TIMESTAMP NOT NULL DEFAULT '2000-01-01 00:00:00+00:00'
|
||||||
|
);
|
||||||
|
CREATE TABLE authors (
|
||||||
|
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
|
||||||
|
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name)
|
||||||
|
);
|
||||||
|
CREATE TABLE books_authors_link (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, author INTEGER NOT NULL,
|
||||||
|
UNIQUE(book, author)
|
||||||
|
);
|
||||||
|
CREATE TABLE publishers (
|
||||||
|
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
|
||||||
|
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name)
|
||||||
|
);
|
||||||
|
CREATE TABLE books_publishers_link (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, publisher INTEGER NOT NULL,
|
||||||
|
UNIQUE(book)
|
||||||
|
);
|
||||||
|
CREATE TABLE tags (
|
||||||
|
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
|
||||||
|
link TEXT NOT NULL DEFAULT '', UNIQUE (name)
|
||||||
|
);
|
||||||
|
CREATE TABLE books_tags_link (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, tag INTEGER NOT NULL,
|
||||||
|
UNIQUE(book, tag)
|
||||||
|
);
|
||||||
|
CREATE TABLE series (
|
||||||
|
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
|
||||||
|
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE (name)
|
||||||
|
);
|
||||||
|
CREATE TABLE books_series_link (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, series INTEGER NOT NULL,
|
||||||
|
UNIQUE(book)
|
||||||
|
);
|
||||||
|
CREATE TABLE languages (
|
||||||
|
id INTEGER PRIMARY KEY, lang_code TEXT NOT NULL COLLATE NOCASE,
|
||||||
|
link TEXT NOT NULL DEFAULT '', UNIQUE(lang_code)
|
||||||
|
);
|
||||||
|
CREATE TABLE books_languages_link (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, lang_code INTEGER NOT NULL,
|
||||||
|
item_order INTEGER NOT NULL DEFAULT 0, UNIQUE(book, lang_code)
|
||||||
|
);
|
||||||
|
CREATE TABLE comments (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
|
||||||
|
text TEXT NOT NULL COLLATE NOCASE, UNIQUE(book)
|
||||||
|
);
|
||||||
|
CREATE TABLE identifiers (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
|
||||||
|
type TEXT NOT NULL DEFAULT 'isbn' COLLATE NOCASE,
|
||||||
|
val TEXT NOT NULL COLLATE NOCASE, UNIQUE(book, type)
|
||||||
|
);
|
||||||
|
CREATE TABLE data (
|
||||||
|
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
|
||||||
|
format TEXT NOT NULL COLLATE NOCASE, uncompressed_size INTEGER NOT NULL,
|
||||||
|
name TEXT NOT NULL, UNIQUE(book, format)
|
||||||
|
);
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Calibre's own "no date". Stored, never null, and a valid date — which is exactly why
|
||||||
|
# it has to be recognised rather than parsed.
|
||||||
|
UNDEFINED_DATE = "0101-01-01 00:00:00+00:00"
|
||||||
|
|
||||||
|
# What a `cover.jpg` that PIL cannot read looks like. Real libraries hold these, from
|
||||||
|
# an interrupted download or a failed conversion.
|
||||||
|
CORRUPT_COVER = b"\xff\xd8\xff\xe0 not really a jpeg"
|
||||||
|
|
||||||
|
|
||||||
|
def write_cover(path: Path) -> None:
|
||||||
|
"""
|
||||||
|
Write a real, readable JPEG.
|
||||||
|
|
||||||
|
Generated with PIL rather than embedded as a hex blob: a hand-rolled JPEG that is
|
||||||
|
subtly malformed fails inside the import as an unrelated error, which is exactly the
|
||||||
|
confusion this avoids.
|
||||||
|
"""
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
Image.new("RGB", (2, 3), (10, 20, 30)).save(path, "JPEG")
|
||||||
|
|
||||||
|
|
||||||
|
class CalibreFixture:
|
||||||
|
"""A Calibre library being assembled under `root`."""
|
||||||
|
|
||||||
|
def __init__(self, root: Path) -> None:
|
||||||
|
self.root = root
|
||||||
|
self.root.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
self.connection = sqlite3.connect(self.root / "metadata.db")
|
||||||
|
self.connection.executescript(SCHEMA)
|
||||||
|
|
||||||
|
def add_pages_table(self) -> None:
|
||||||
|
"""Add `books_pages_link`, which only a recent Calibre creates."""
|
||||||
|
self.connection.executescript(
|
||||||
|
"""
|
||||||
|
CREATE TABLE books_pages_link (
|
||||||
|
book INTEGER PRIMARY KEY,
|
||||||
|
pages INTEGER DEFAULT 0 NOT NULL,
|
||||||
|
algorithm INTEGER DEFAULT 0 NOT NULL,
|
||||||
|
format TEXT DEFAULT '' NOT NULL COLLATE NOCASE,
|
||||||
|
format_size INTEGER DEFAULT 0 NOT NULL,
|
||||||
|
timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||||
|
needs_scan INTEGER NOT NULL DEFAULT 0
|
||||||
|
);
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
def add_book(
|
||||||
|
self,
|
||||||
|
book_id: int,
|
||||||
|
title: str,
|
||||||
|
*,
|
||||||
|
authors: list[str] | None = None,
|
||||||
|
pubdate: str = UNDEFINED_DATE,
|
||||||
|
series: str | None = None,
|
||||||
|
series_index: float = 1.0,
|
||||||
|
tags: list[str] | None = None,
|
||||||
|
publisher: str | None = None,
|
||||||
|
languages: list[str] | None = None,
|
||||||
|
comment: str | None = None,
|
||||||
|
identifiers: dict[str, str] | None = None,
|
||||||
|
uuid: str | None = None,
|
||||||
|
pages: int | None = None,
|
||||||
|
cover: bool = False,
|
||||||
|
corrupt_cover: bool = False,
|
||||||
|
formats: dict[str, Path] | None = None,
|
||||||
|
directory: str | None = None,
|
||||||
|
) -> Path:
|
||||||
|
"""
|
||||||
|
Add one book, with its files laid out the way Calibre lays them out.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
formats: Format name (`EPUB`) to a real file to copy in. Its on-disk stem is
|
||||||
|
Calibre's, not the title — that is the point of the `data` table.
|
||||||
|
directory: Override the `books.path` value, for testing a row whose
|
||||||
|
directory is not where the convention would put it.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The book's directory.
|
||||||
|
"""
|
||||||
|
author_names = authors or ["Unknown"]
|
||||||
|
relative = directory or f"{author_names[0]}/{title} ({book_id})"
|
||||||
|
book_directory = self.root / relative
|
||||||
|
book_directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO books (id, title, pubdate, series_index, path, uuid, has_cover) "
|
||||||
|
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||||
|
(
|
||||||
|
book_id,
|
||||||
|
title,
|
||||||
|
pubdate,
|
||||||
|
series_index,
|
||||||
|
relative,
|
||||||
|
uuid or f"uuid-{book_id}",
|
||||||
|
int(cover or corrupt_cover),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
for name in author_names:
|
||||||
|
self._link("authors", "books_authors_link", "author", book_id, name)
|
||||||
|
|
||||||
|
for name in tags or []:
|
||||||
|
self._link("tags", "books_tags_link", "tag", book_id, name)
|
||||||
|
|
||||||
|
if series:
|
||||||
|
self._link("series", "books_series_link", "series", book_id, series)
|
||||||
|
|
||||||
|
if publisher:
|
||||||
|
self._link(
|
||||||
|
"publishers", "books_publishers_link", "publisher", book_id, publisher
|
||||||
|
)
|
||||||
|
|
||||||
|
for order, code in enumerate(languages or []):
|
||||||
|
language_id = self._lookup("languages", "lang_code", code)
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO books_languages_link (book, lang_code, item_order) "
|
||||||
|
"VALUES (?, ?, ?)",
|
||||||
|
(book_id, language_id, order),
|
||||||
|
)
|
||||||
|
|
||||||
|
if comment is not None:
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO comments (book, text) VALUES (?, ?)", (book_id, comment)
|
||||||
|
)
|
||||||
|
|
||||||
|
for name, value in (identifiers or {}).items():
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO identifiers (book, type, val) VALUES (?, ?, ?)",
|
||||||
|
(book_id, name, value),
|
||||||
|
)
|
||||||
|
|
||||||
|
if pages is not None:
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO books_pages_link (book, pages) VALUES (?, ?)",
|
||||||
|
(book_id, pages),
|
||||||
|
)
|
||||||
|
|
||||||
|
if corrupt_cover:
|
||||||
|
(book_directory / "cover.jpg").write_bytes(CORRUPT_COVER)
|
||||||
|
elif cover:
|
||||||
|
write_cover(book_directory / "cover.jpg")
|
||||||
|
|
||||||
|
for format, origin in (formats or {}).items():
|
||||||
|
# Calibre's on-disk stem: sanitised, truncated, and not the title.
|
||||||
|
stem = f"{title[:40]} - {author_names[0]}".replace(":", "_")
|
||||||
|
destination = book_directory / f"{stem}.{format.lower()}"
|
||||||
|
shutil.copy(origin, destination)
|
||||||
|
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO data (book, format, uncompressed_size, name) "
|
||||||
|
"VALUES (?, ?, ?, ?)",
|
||||||
|
(book_id, format, destination.stat().st_size, stem),
|
||||||
|
)
|
||||||
|
|
||||||
|
return book_directory
|
||||||
|
|
||||||
|
def add_missing_format(self, book_id: int, format: str, stem: str) -> None:
|
||||||
|
"""Record a file in the catalogue without putting one on disk."""
|
||||||
|
self.connection.execute(
|
||||||
|
"INSERT INTO data (book, format, uncompressed_size, name) VALUES (?, ?, ?, ?)",
|
||||||
|
(book_id, format, 1234, stem),
|
||||||
|
)
|
||||||
|
|
||||||
|
def _link(
|
||||||
|
self, table: str, link_table: str, column: str, book_id: int, name: str
|
||||||
|
) -> None:
|
||||||
|
item_id = self._lookup(table, "name", name)
|
||||||
|
self.connection.execute(
|
||||||
|
f"INSERT INTO {link_table} (book, {column}) VALUES (?, ?)",
|
||||||
|
(book_id, item_id),
|
||||||
|
)
|
||||||
|
|
||||||
|
def _lookup(self, table: str, column: str, value: str) -> int:
|
||||||
|
row = self.connection.execute(
|
||||||
|
f"SELECT id FROM {table} WHERE {column} = ?", (value,)
|
||||||
|
).fetchone()
|
||||||
|
|
||||||
|
if row:
|
||||||
|
return int(row[0])
|
||||||
|
|
||||||
|
cursor = self.connection.execute(
|
||||||
|
f"INSERT INTO {table} ({column}) VALUES (?)", (value,)
|
||||||
|
)
|
||||||
|
return int(cursor.lastrowid or 0)
|
||||||
|
|
||||||
|
def commit(self) -> Path:
|
||||||
|
"""Finish writing and return the library root."""
|
||||||
|
self.connection.commit()
|
||||||
|
self.connection.close()
|
||||||
|
return self.root
|
||||||
@@ -2,6 +2,8 @@ from pathlib import Path
|
|||||||
from uuid import uuid4
|
from uuid import uuid4
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
|
from chitai import services
|
||||||
|
|
||||||
from advanced_alchemy.base import UUIDAuditBase
|
from advanced_alchemy.base import UUIDAuditBase
|
||||||
from litestar.testing import AsyncTestClient
|
from litestar.testing import AsyncTestClient
|
||||||
from sqlalchemy import text
|
from sqlalchemy import text
|
||||||
@@ -40,7 +42,13 @@ pytest_plugins = [
|
|||||||
|
|
||||||
@pytest.fixture(autouse=True)
|
@pytest.fixture(autouse=True)
|
||||||
def _patch_settings(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
def _patch_settings(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
monkeypatch.setattr(settings, "book_cover_path", f"{tmp_path}/covers")
|
# PIL will not create the directory it is asked to save into, so anything that
|
||||||
|
# imports a file carrying a cover needs it to exist first.
|
||||||
|
covers = tmp_path / "covers"
|
||||||
|
covers.mkdir()
|
||||||
|
|
||||||
|
monkeypatch.setattr(settings, "book_cover_path", str(covers))
|
||||||
|
monkeypatch.setattr(settings, "duplicate_path", str(tmp_path / "duplicates"))
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture(name="engine")
|
@pytest.fixture(name="engine")
|
||||||
@@ -149,7 +157,6 @@ async def other_authenticated_client(
|
|||||||
|
|
||||||
|
|
||||||
# Service fixtures
|
# Service fixtures
|
||||||
from chitai import services
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
import pytest
|
import pytest
|
||||||
from httpx import AsyncClient
|
from httpx import AsyncClient
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from litestar.status_codes import HTTP_400_BAD_REQUEST
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
@@ -37,7 +38,8 @@ from pathlib import Path
|
|||||||
(
|
(
|
||||||
Path("tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf"),
|
Path("tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf"),
|
||||||
2,
|
2,
|
||||||
"The Project Gutenberg eBook #33283: Calculus Made Easy, 2nd Edition",
|
# The ", 2nd Edition" is split off into `edition`, not kept in the title.
|
||||||
|
"The Project Gutenberg eBook #33283: Calculus Made Easy",
|
||||||
["Silvanus Phillips Thompson"],
|
["Silvanus Phillips Thompson"],
|
||||||
),
|
),
|
||||||
],
|
],
|
||||||
@@ -208,11 +210,28 @@ async def test_get_book_file(
|
|||||||
assert downloaded_content == file_content
|
assert downloaded_content == file_content
|
||||||
|
|
||||||
|
|
||||||
|
async def test_list_books_by_id(populated_authenticated_client: AsyncClient) -> None:
|
||||||
|
"""
|
||||||
|
`?ids=` has to reach the database as integers.
|
||||||
|
|
||||||
|
advanced_alchemy's stock id filter annotates the parameter as `list[str]` whatever
|
||||||
|
the configured id type, so the ids arrived as strings and Postgres refused to
|
||||||
|
compare a bigint primary key against them. Nothing called it until a screen needed
|
||||||
|
to fetch a handful of books by id.
|
||||||
|
"""
|
||||||
|
response = await populated_authenticated_client.get(
|
||||||
|
"/books?ids=1&ids=2&pageSize=10"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 200
|
||||||
|
assert sorted(book["id"] for book in response.json()["items"]) == [1, 2]
|
||||||
|
|
||||||
|
|
||||||
async def test_get_book_by_id(populated_authenticated_client: AsyncClient) -> None:
|
async def test_get_book_by_id(populated_authenticated_client: AsyncClient) -> None:
|
||||||
"""Test retrieving a specific book by ID."""
|
"""Test retrieving a specific book by ID."""
|
||||||
|
|
||||||
# Retrieve the book
|
# Retrieve the book
|
||||||
response = await populated_authenticated_client.get(f"/books/1")
|
response = await populated_authenticated_client.get("/books/1")
|
||||||
|
|
||||||
assert response.status_code == 200
|
assert response.status_code == 200
|
||||||
book_data = response.json()
|
book_data = response.json()
|
||||||
@@ -284,13 +303,13 @@ async def test_delete_book_metadata_only(
|
|||||||
|
|
||||||
# Delete book without deleting files
|
# Delete book without deleting files
|
||||||
response = await populated_authenticated_client.delete(
|
response = await populated_authenticated_client.delete(
|
||||||
f"/books?book_ids=3&delete_files=false&library_id=1"
|
"/books?book_ids=3&delete_files=false&library_id=1"
|
||||||
)
|
)
|
||||||
|
|
||||||
assert response.status_code == 204
|
assert response.status_code == 204
|
||||||
|
|
||||||
# Verify book is deleted
|
# Verify book is deleted
|
||||||
get_response = await populated_authenticated_client.get(f"/books/3")
|
get_response = await populated_authenticated_client.get("/books/3")
|
||||||
assert get_response.status_code == 404
|
assert get_response.status_code == 404
|
||||||
|
|
||||||
|
|
||||||
@@ -301,7 +320,7 @@ async def test_delete_book_with_files(
|
|||||||
|
|
||||||
# Delete book and files
|
# Delete book and files
|
||||||
response = await populated_authenticated_client.delete(
|
response = await populated_authenticated_client.delete(
|
||||||
f"/books?book_ids=3&delete_files=true&library_id=1"
|
"/books?book_ids=3&delete_files=true&library_id=1"
|
||||||
)
|
)
|
||||||
|
|
||||||
assert response.status_code == 204
|
assert response.status_code == 204
|
||||||
@@ -314,7 +333,7 @@ async def test_delete_specific_book_files(
|
|||||||
|
|
||||||
# Delete specific file
|
# Delete specific file
|
||||||
response = await populated_authenticated_client.delete(
|
response = await populated_authenticated_client.delete(
|
||||||
f"/books/1/files?file_ids=1",
|
"/books/1/files?file_ids=1",
|
||||||
)
|
)
|
||||||
|
|
||||||
assert response.status_code == 204
|
assert response.status_code == 204
|
||||||
@@ -331,7 +350,7 @@ async def test_update_reading_progress(
|
|||||||
}
|
}
|
||||||
|
|
||||||
response = await populated_authenticated_client.post(
|
response = await populated_authenticated_client.post(
|
||||||
f"/books/progress/1",
|
"/books/progress/1",
|
||||||
json=progress_data,
|
json=progress_data,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -366,7 +385,403 @@ async def test_create_multiple_books_from_directory(
|
|||||||
|
|
||||||
assert response.status_code == 201
|
assert response.status_code == 201
|
||||||
data = response.json()
|
data = response.json()
|
||||||
assert len(data.get("items") or data.get("data")) >= 1
|
assert len(data["created"]) == 2
|
||||||
|
assert data["skipped"] == []
|
||||||
|
|
||||||
|
|
||||||
|
async def test_create_books_from_parent_directory_keeps_embedded_title(
|
||||||
|
authenticated_client: AsyncClient,
|
||||||
|
) -> None:
|
||||||
|
"""A folder name in the upload path must not override the file's own metadata.
|
||||||
|
|
||||||
|
The browser sends webkitRelativePath, so picking the shelf above a book's folder
|
||||||
|
submits one more path component than picking the folder itself. That extra level
|
||||||
|
used to make the directory name win over the title inside the EPUB.
|
||||||
|
"""
|
||||||
|
source = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
files = [
|
||||||
|
(
|
||||||
|
"files",
|
||||||
|
(
|
||||||
|
"Shelf/Metamorphosis - Franz Kafka/Metamorphosis.epub",
|
||||||
|
source.read_bytes(),
|
||||||
|
"application/epub+zip",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=files, data={"library_id": 1}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
|
||||||
|
books = response.json()["created"]
|
||||||
|
assert len(books) == 1
|
||||||
|
assert books[0]["title"] == "Metamorphosis"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_create_books_groups_formats_within_one_folder(
|
||||||
|
authenticated_client: AsyncClient,
|
||||||
|
) -> None:
|
||||||
|
"""Picking a book's own folder yields one book with both formats, not two books."""
|
||||||
|
epub = Path("tests/data_files/Metamorphosis - Franz Kafka.epub").read_bytes()
|
||||||
|
pdf = Path(
|
||||||
|
"tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf"
|
||||||
|
).read_bytes()
|
||||||
|
|
||||||
|
files = [
|
||||||
|
("files", ("Metamorphosis/Metamorphosis.epub", epub, "application/epub+zip")),
|
||||||
|
("files", ("Metamorphosis/Metamorphosis.pdf", pdf, "application/pdf")),
|
||||||
|
]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=files, data={"library_id": 1}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
|
||||||
|
books = response.json()["created"]
|
||||||
|
assert len(books) == 1
|
||||||
|
assert len(books[0]["files"]) == 2
|
||||||
|
|
||||||
|
|
||||||
|
class TestDuplicateHandling:
|
||||||
|
"""A file the library already holds must not be stored a second time."""
|
||||||
|
|
||||||
|
epub_path = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
|
||||||
|
def upload(self, name: str | None = None) -> list[tuple[str, tuple]]:
|
||||||
|
return [
|
||||||
|
(
|
||||||
|
"files",
|
||||||
|
(
|
||||||
|
name or self.epub_path.name,
|
||||||
|
self.epub_path.read_bytes(),
|
||||||
|
"application/epub+zip",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
|
async def test_bulk_upload_reports_skipped_files(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""Re-dropping a folder must import what is new and name what was not."""
|
||||||
|
first = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
assert first.status_code == 201
|
||||||
|
created = first.json()["created"][0]
|
||||||
|
|
||||||
|
second = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
|
||||||
|
assert second.status_code == 201
|
||||||
|
result = second.json()
|
||||||
|
assert result["created"] == []
|
||||||
|
assert len(result["skipped"]) == 1
|
||||||
|
|
||||||
|
skipped = result["skipped"][0]
|
||||||
|
assert skipped["filename"] == self.epub_path.name
|
||||||
|
assert skipped["book_id"] == created["id"]
|
||||||
|
assert skipped["book_title"] == created["title"]
|
||||||
|
|
||||||
|
async def test_bulk_upload_can_be_forced(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1&allow_duplicates=true", files=self.upload()
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
assert len(response.json()["created"]) == 1
|
||||||
|
assert response.json()["skipped"] == []
|
||||||
|
|
||||||
|
async def test_single_book_create_conflicts(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""Naming files deliberately earns a refusal rather than a silent drop."""
|
||||||
|
await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books?library_id=1", files=self.upload(), data={"library_id": 1}
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 409
|
||||||
|
assert response.json()["extra"][0]["filename"] == self.epub_path.name
|
||||||
|
|
||||||
|
forced = await authenticated_client.post(
|
||||||
|
"/books?library_id=1&allow_duplicates=true",
|
||||||
|
files=self.upload(),
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
assert forced.status_code == 201
|
||||||
|
|
||||||
|
async def test_adding_another_books_file_conflicts(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
book_id = created.json()["created"][0]["id"]
|
||||||
|
|
||||||
|
other = await authenticated_client.post(
|
||||||
|
"/books?library_id=1",
|
||||||
|
files=[
|
||||||
|
(
|
||||||
|
"files",
|
||||||
|
(
|
||||||
|
"war.epub",
|
||||||
|
Path(
|
||||||
|
"tests/data_files/The Art of War - Sun Tzu.epub"
|
||||||
|
).read_bytes(),
|
||||||
|
"application/epub+zip",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
],
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
other_id = other.json()["id"]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
f"/books/{book_id}/files",
|
||||||
|
files=[
|
||||||
|
(
|
||||||
|
"files",
|
||||||
|
(
|
||||||
|
"war.epub",
|
||||||
|
Path(
|
||||||
|
"tests/data_files/The Art of War - Sun Tzu.epub"
|
||||||
|
).read_bytes(),
|
||||||
|
"application/epub+zip",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 409
|
||||||
|
assert response.json()["extra"][0]["book_id"] == other_id
|
||||||
|
|
||||||
|
async def test_resending_a_books_own_file_changes_nothing(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
book_id = created.json()["created"][0]["id"]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
f"/books/{book_id}/files", files=self.upload()
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
assert len(response.json()["files"]) == 1
|
||||||
|
|
||||||
|
async def test_duplicates_can_be_checked_before_uploading(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""The pre-flight check answers from hashes alone, with no file sent."""
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload()
|
||||||
|
)
|
||||||
|
book = created.json()["created"][0]
|
||||||
|
stored = book["files"][0]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/duplicate-files?library_id=1",
|
||||||
|
json=[
|
||||||
|
{
|
||||||
|
"hash": stored["hash"],
|
||||||
|
"size": stored["size"],
|
||||||
|
"filename": "local-copy.epub",
|
||||||
|
},
|
||||||
|
{"hash": stored["hash"], "size": stored["size"] + 1},
|
||||||
|
{"hash": "0" * 32, "size": 1234},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 200
|
||||||
|
|
||||||
|
matches = response.json()
|
||||||
|
assert len(matches) == 1
|
||||||
|
assert matches[0]["filename"] == "local-copy.epub"
|
||||||
|
assert matches[0]["book_id"] == book["id"]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
class TestDuplicateBooks:
|
||||||
|
"""A second copy of a book is imported and reported, never refused."""
|
||||||
|
|
||||||
|
epub_path = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
|
||||||
|
def upload(self, name: str, pad: bool = False) -> list[tuple[str, tuple]]:
|
||||||
|
"""
|
||||||
|
The fixture, optionally padded so it is a different file and the same book.
|
||||||
|
|
||||||
|
Padding the archive changes its size and its sampled hash without disturbing
|
||||||
|
the metadata, which is exactly the case the file-level check cannot see.
|
||||||
|
"""
|
||||||
|
data = self.epub_path.read_bytes() + (b"\0" * 64 if pad else b"")
|
||||||
|
|
||||||
|
return [("files", (name, data, "application/epub+zip"))]
|
||||||
|
|
||||||
|
async def test_a_second_edition_is_created_and_reported(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
first = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("first.epub")
|
||||||
|
)
|
||||||
|
original = first.json()["created"][0]
|
||||||
|
|
||||||
|
second = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("second.epub", pad=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
assert second.status_code == 201
|
||||||
|
body = second.json()
|
||||||
|
|
||||||
|
# Created, not skipped: a metadata match is a guess, and refusing a legitimate
|
||||||
|
# second edition costs more than a note does.
|
||||||
|
assert len(body["created"]) == 1
|
||||||
|
assert body["skipped"] == []
|
||||||
|
|
||||||
|
assert len(body["possible_duplicates"]) == 1
|
||||||
|
possible = body["possible_duplicates"][0]
|
||||||
|
assert possible["book_id"] == body["created"][0]["id"]
|
||||||
|
|
||||||
|
candidate = possible["candidates"][0]
|
||||||
|
assert candidate["book_id"] == original["id"]
|
||||||
|
assert candidate["title"] == original["title"]
|
||||||
|
assert candidate["authors"] == ["Franz Kafka"]
|
||||||
|
assert "title-author" in candidate["matched_on"]
|
||||||
|
|
||||||
|
async def test_the_review_screen_groups_them(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
first = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("first.epub")
|
||||||
|
)
|
||||||
|
second = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("second.epub", pad=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
book_ids = sorted(
|
||||||
|
[first.json()["created"][0]["id"], second.json()["created"][0]["id"]]
|
||||||
|
)
|
||||||
|
|
||||||
|
response = await authenticated_client.get("/books/duplicate-books?library_id=1")
|
||||||
|
|
||||||
|
assert response.status_code == 200
|
||||||
|
groups = response.json()
|
||||||
|
assert len(groups) == 1
|
||||||
|
assert [book["book_id"] for book in groups[0]["books"]] == book_ids
|
||||||
|
|
||||||
|
async def test_a_dismissed_group_stays_dismissed(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
first = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("first.epub")
|
||||||
|
)
|
||||||
|
second = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("second.epub", pad=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
pair = {
|
||||||
|
"book_a_id": first.json()["created"][0]["id"],
|
||||||
|
"book_b_id": second.json()["created"][0]["id"],
|
||||||
|
}
|
||||||
|
|
||||||
|
dismissed = await authenticated_client.post(
|
||||||
|
"/books/duplicate-books/dismissals", json=pair
|
||||||
|
)
|
||||||
|
assert dismissed.status_code == 204
|
||||||
|
|
||||||
|
response = await authenticated_client.get("/books/duplicate-books?library_id=1")
|
||||||
|
assert response.json() == []
|
||||||
|
|
||||||
|
restored = await authenticated_client.delete(
|
||||||
|
"/books/duplicate-books/dismissals"
|
||||||
|
f"?book_a_id={pair['book_b_id']}&book_b_id={pair['book_a_id']}"
|
||||||
|
)
|
||||||
|
assert restored.status_code == 204
|
||||||
|
|
||||||
|
response = await authenticated_client.get("/books/duplicate-books?library_id=1")
|
||||||
|
assert len(response.json()) == 1
|
||||||
|
|
||||||
|
async def test_two_books_merge_into_one(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""The survivor keeps its id and gains the other's file; the other is gone."""
|
||||||
|
first = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("first.epub")
|
||||||
|
)
|
||||||
|
second = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("second.epub", pad=True)
|
||||||
|
)
|
||||||
|
keep = first.json()["created"][0]
|
||||||
|
fold = second.json()["created"][0]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/merge?library_id=1",
|
||||||
|
json={
|
||||||
|
"survivor_id": keep["id"],
|
||||||
|
"merged_ids": [fold["id"]],
|
||||||
|
"metadata": {"title": "Metamorphosis", "edition": 2},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
merged = response.json()
|
||||||
|
|
||||||
|
assert merged["id"] == keep["id"]
|
||||||
|
assert merged["edition"] == 2
|
||||||
|
assert len(merged["files"]) == 2
|
||||||
|
|
||||||
|
# The folded record is gone, and the group it formed with it.
|
||||||
|
assert (
|
||||||
|
await authenticated_client.get(f"/books/{fold['id']}")
|
||||||
|
).status_code == 404
|
||||||
|
groups = await authenticated_client.get("/books/duplicate-books?library_id=1")
|
||||||
|
assert groups.json() == []
|
||||||
|
|
||||||
|
async def test_merging_an_unknown_book_is_refused(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/books/fromFiles?library_id=1", files=self.upload("first.epub")
|
||||||
|
)
|
||||||
|
keep = created.json()["created"][0]["id"]
|
||||||
|
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/merge?library_id=1",
|
||||||
|
json={"survivor_id": keep, "merged_ids": [9999]},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 400
|
||||||
|
|
||||||
|
async def test_dismissing_an_unknown_book_is_refused(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books/duplicate-books/dismissals",
|
||||||
|
json={"book_a_id": 1, "book_b_id": 9999},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 400
|
||||||
|
|
||||||
|
|
||||||
|
# NOTE: the multi-book ZIP download is covered at the service level, in
|
||||||
|
# tests/unit/test_services/test_book_service.py. Driving `/books/download` through
|
||||||
|
# AsyncTestClient hangs in fixture teardown: it is the only `Stream` endpoint in the
|
||||||
|
# app, and the test transport never sends the `http.disconnect` that Litestar's
|
||||||
|
# streaming response waits on, so the app's lifespan shutdown never completes.
|
||||||
|
|
||||||
|
|
||||||
# async def test_delete_book_metadata(authenticated_client: AsyncClient) -> None:
|
# async def test_delete_book_metadata(authenticated_client: AsyncClient) -> None:
|
||||||
@@ -382,16 +797,6 @@ async def test_create_multiple_books_from_directory(
|
|||||||
# async def test_edit_book_metadata(authenticated_client: AsyncClient) -> None:
|
# async def test_edit_book_metadata(authenticated_client: AsyncClient) -> None:
|
||||||
# raise NotImplementedError()
|
# raise NotImplementedError()
|
||||||
|
|
||||||
import pytest
|
|
||||||
import aiofiles
|
|
||||||
from httpx import AsyncClient
|
|
||||||
from pathlib import Path
|
|
||||||
from litestar.status_codes import HTTP_400_BAD_REQUEST
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
from httpx import AsyncClient
|
|
||||||
from datetime import date
|
|
||||||
|
|
||||||
|
|
||||||
class TestMetadataUpdates:
|
class TestMetadataUpdates:
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
@@ -445,8 +850,10 @@ class TestMetadataUpdates:
|
|||||||
(
|
(
|
||||||
"authors", # Update with new authors
|
"authors", # Update with new authors
|
||||||
["New Author 1", "New Author 2"],
|
["New Author 1", "New Author 2"],
|
||||||
lambda data: {a["name"] for a in data["authors"]}
|
lambda data: (
|
||||||
== {"New Author 1", "New Author 2"},
|
{a["name"] for a in data["authors"]}
|
||||||
|
== {"New Author 1", "New Author 2"}
|
||||||
|
),
|
||||||
),
|
),
|
||||||
(
|
(
|
||||||
"authors", # Clear authors
|
"authors", # Clear authors
|
||||||
@@ -456,8 +863,9 @@ class TestMetadataUpdates:
|
|||||||
(
|
(
|
||||||
"tags", # Update with new tags
|
"tags", # Update with new tags
|
||||||
["Tag 1", "Tag 2", "Tag 3"],
|
["Tag 1", "Tag 2", "Tag 3"],
|
||||||
lambda data: {t["name"] for t in data["tags"]}
|
lambda data: (
|
||||||
== {"Tag 1", "Tag 2", "Tag 3"},
|
{t["name"] for t in data["tags"]} == {"Tag 1", "Tag 2", "Tag 3"}
|
||||||
|
),
|
||||||
),
|
),
|
||||||
(
|
(
|
||||||
"tags", # Clear tags
|
"tags", # Clear tags
|
||||||
@@ -477,8 +885,10 @@ class TestMetadataUpdates:
|
|||||||
(
|
(
|
||||||
"identifiers", # Update with new identifiers
|
"identifiers", # Update with new identifiers
|
||||||
{"isbn-13": "978-1234567890", "doi": "10.example/id"},
|
{"isbn-13": "978-1234567890", "doi": "10.example/id"},
|
||||||
lambda data: data["identifiers"]
|
lambda data: (
|
||||||
== {"isbn-13": "978-1234567890", "doi": "10.example/id"},
|
data["identifiers"]
|
||||||
|
== {"isbn-13": "978-1234567890", "doi": "10.example/id"}
|
||||||
|
),
|
||||||
),
|
),
|
||||||
(
|
(
|
||||||
"identifiers", # Clear identifiers
|
"identifiers", # Clear identifiers
|
||||||
@@ -649,7 +1059,7 @@ class TestMetadataUpdates:
|
|||||||
|
|
||||||
result = response.json()
|
result = response.json()
|
||||||
|
|
||||||
assert result[updated_field] == None
|
assert result[updated_field] is None
|
||||||
|
|
||||||
@pytest.mark.parametrize(
|
@pytest.mark.parametrize(
|
||||||
("updated_field"),
|
("updated_field"),
|
||||||
@@ -811,7 +1221,9 @@ class TestFileManagement:
|
|||||||
pytest.skip("Book has no files")
|
pytest.skip("Book has no files")
|
||||||
|
|
||||||
file_id = book_data["files"][0]["id"]
|
file_id = book_data["files"][0]["id"]
|
||||||
filename = book_data["files"][0].get("path")
|
# TODO: this test asserts only the 204 and never checks the disk, so the flag it is
|
||||||
|
# named for is untested. See TODO.md; drop the noqa when the assertion lands.
|
||||||
|
filename = book_data["files"][0].get("path") # noqa: F841
|
||||||
|
|
||||||
# Remove file without deleting from filesystem
|
# Remove file without deleting from filesystem
|
||||||
response = await populated_authenticated_client.delete(
|
response = await populated_authenticated_client.delete(
|
||||||
@@ -836,7 +1248,8 @@ class TestFileManagement:
|
|||||||
|
|
||||||
book_data = add_response.json()
|
book_data = add_response.json()
|
||||||
file_id = book_data["files"][-1]["id"]
|
file_id = book_data["files"][-1]["id"]
|
||||||
file_path = book_data["files"][-1].get("path")
|
# TODO: as above -- the file is never checked for removal from disk.
|
||||||
|
file_path = book_data["files"][-1].get("path") # noqa: F841
|
||||||
|
|
||||||
# Remove file with deletion from filesystem
|
# Remove file with deletion from filesystem
|
||||||
response = await populated_authenticated_client.delete(
|
response = await populated_authenticated_client.delete(
|
||||||
@@ -868,3 +1281,94 @@ class TestFileManagement:
|
|||||||
)
|
)
|
||||||
# Should succeed (idempotent)
|
# Should succeed (idempotent)
|
||||||
assert response2.status_code == 204
|
assert response2.status_code == 204
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
class TestUnnameableFormats:
|
||||||
|
"""
|
||||||
|
A file whose format nothing can name must still round-trip.
|
||||||
|
|
||||||
|
`mimetypes.guess_type` answers None for `.mobi`, `.azw`, `.fb2` and `.lit`, which
|
||||||
|
is most of what a library imported from elsewhere carries alongside its EPUBs.
|
||||||
|
`FileMetadataRead.content_type` used to be a required string, so such a book was
|
||||||
|
created and then failed serialisation on its way back out — a 500 on a book the
|
||||||
|
reader can otherwise download.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def upload(self, name: str) -> list[tuple[str, tuple]]:
|
||||||
|
# `application/octet-stream` is what a browser posts for these, and it is not
|
||||||
|
# an answer — the extension is what names the format.
|
||||||
|
return [("files", (name, b"BOOKMOBI\x00 payload", "application/octet-stream"))]
|
||||||
|
|
||||||
|
async def test_a_mobi_is_named_from_its_extension(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books?library_id=1",
|
||||||
|
files=self.upload("Dune.mobi"),
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
book = response.json()
|
||||||
|
assert book["files"][0]["content_type"] == "application/x-mobipocket-ebook"
|
||||||
|
|
||||||
|
detail = await authenticated_client.get(f"/books/{book['id']}")
|
||||||
|
assert detail.status_code == 200
|
||||||
|
|
||||||
|
async def test_an_unknown_extension_stores_no_content_type(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""Null, not a placeholder — and the book still serialises either way."""
|
||||||
|
response = await authenticated_client.post(
|
||||||
|
"/books?library_id=1",
|
||||||
|
files=self.upload("Notes.xyzzy"),
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 201
|
||||||
|
book = response.json()
|
||||||
|
assert book["files"][0]["content_type"] is None
|
||||||
|
|
||||||
|
detail = await authenticated_client.get(f"/books/{book['id']}")
|
||||||
|
assert detail.status_code == 200
|
||||||
|
assert detail.json()["files"][0]["content_type"] is None
|
||||||
|
|
||||||
|
async def test_the_file_downloads(self, authenticated_client: AsyncClient) -> None:
|
||||||
|
"""Litestar supplies its own media type when the row carries none."""
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/books?library_id=1",
|
||||||
|
files=self.upload("Notes.xyzzy"),
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
book = created.json()
|
||||||
|
|
||||||
|
response = await authenticated_client.get(
|
||||||
|
f"/books/download/{book['id']}/{book['files'][0]['id']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert response.status_code == 200
|
||||||
|
assert response.headers["content-type"] == "application/octet-stream"
|
||||||
|
|
||||||
|
async def test_the_opds_feed_survives_a_null_content_type(
|
||||||
|
self, authenticated_client: AsyncClient
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
The one place the type has to be a string.
|
||||||
|
|
||||||
|
`Link.type` is required, so a null fails the whole feed rather than one entry.
|
||||||
|
OPDS clients speak Basic, not the JWT the rest of the API uses.
|
||||||
|
"""
|
||||||
|
await authenticated_client.post(
|
||||||
|
"/books?library_id=1",
|
||||||
|
files=self.upload("Notes.xyzzy"),
|
||||||
|
data={"library_id": 1},
|
||||||
|
)
|
||||||
|
|
||||||
|
feed = await authenticated_client.get(
|
||||||
|
"/opds/acquisition?feed_id=all&feed_title=All+Books",
|
||||||
|
auth=("user1@example.com", "password123"),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert feed.status_code == 200
|
||||||
|
assert 'type="application/octet-stream"' in feed.text
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
import pytest
|
|
||||||
from httpx import AsyncClient
|
from httpx import AsyncClient
|
||||||
|
|
||||||
|
|
||||||
@@ -200,7 +199,6 @@ async def test_remove_books_from_shelf(
|
|||||||
"/books", params={"shelves": shelf_id}
|
"/books", params={"shelves": shelf_id}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
assert books_response.status_code == 200
|
assert books_response.status_code == 200
|
||||||
assert books_response.json()["total"] == 2
|
assert books_response.json()["total"] == 2
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,405 @@
|
|||||||
|
"""
|
||||||
|
Tests for the Calibre import endpoints.
|
||||||
|
|
||||||
|
The API takes a zipped library and nothing else — a desktop Calibre install is usually
|
||||||
|
not on the server, and importing from a path the server can already see stays a
|
||||||
|
server-side operation (`litestar calibre-import`).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import tempfile
|
||||||
|
import zipfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from httpx import AsyncClient
|
||||||
|
|
||||||
|
from chitai.services.calibre_import import registry
|
||||||
|
|
||||||
|
from tests.calibre_fixtures import CalibreFixture
|
||||||
|
|
||||||
|
|
||||||
|
EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
OTHER_EPUB = Path("tests/data_files/The Art of War - Sun Tzu.epub")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _clear_registry():
|
||||||
|
"""The registry is a module-level singleton, so it leaks between tests."""
|
||||||
|
registry._jobs.clear()
|
||||||
|
yield
|
||||||
|
registry._jobs.clear()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(name="source")
|
||||||
|
def fx_source(tmp_path: Path) -> Path:
|
||||||
|
fixture = CalibreFixture(tmp_path / "calibre")
|
||||||
|
fixture.add_book(
|
||||||
|
1,
|
||||||
|
"The Metamorphosis",
|
||||||
|
authors=["Franz Kafka"],
|
||||||
|
tags=["Fiction"],
|
||||||
|
cover=True,
|
||||||
|
formats={"EPUB": EPUB},
|
||||||
|
)
|
||||||
|
fixture.add_book(
|
||||||
|
2, "The Art of War", authors=["Sun Tzu"], formats={"EPUB": OTHER_EPUB}
|
||||||
|
)
|
||||||
|
fixture.add_book(3, "Metadata Only", authors=["Nobody"])
|
||||||
|
|
||||||
|
return fixture.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def zip_of(root: Path, into: Path, prefix: str = "") -> Path:
|
||||||
|
"""Zip a directory the way a file manager would."""
|
||||||
|
into.mkdir(parents=True, exist_ok=True)
|
||||||
|
archive = into / "library.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
for path in sorted(root.rglob("*")):
|
||||||
|
if path.is_file():
|
||||||
|
writing.write(path, f"{prefix}{path.relative_to(root)}")
|
||||||
|
|
||||||
|
return archive
|
||||||
|
|
||||||
|
|
||||||
|
async def upload(
|
||||||
|
client: AsyncClient,
|
||||||
|
archive: Path,
|
||||||
|
library_id: int = 1,
|
||||||
|
allow_duplicates: bool = False,
|
||||||
|
) -> tuple[int, dict]:
|
||||||
|
response = await client.post(
|
||||||
|
f"/libraries/{library_id}/imports/calibre/upload",
|
||||||
|
files=[("archive", (archive.name, archive.read_bytes(), "application/zip"))],
|
||||||
|
data={"allow_duplicates": str(allow_duplicates).lower()},
|
||||||
|
)
|
||||||
|
|
||||||
|
return response.status_code, response.json()
|
||||||
|
|
||||||
|
|
||||||
|
async def wait_for(client: AsyncClient, job_id: str) -> dict:
|
||||||
|
"""Poll until the job is no longer running, the way the screen does."""
|
||||||
|
for _ in range(200):
|
||||||
|
response = await client.get(f"/libraries/imports/{job_id}")
|
||||||
|
assert response.status_code == 200
|
||||||
|
|
||||||
|
job = response.json()
|
||||||
|
if job["state"] != "running":
|
||||||
|
return job
|
||||||
|
|
||||||
|
await asyncio.sleep(0.05)
|
||||||
|
|
||||||
|
raise AssertionError("the import never finished")
|
||||||
|
|
||||||
|
|
||||||
|
async def test_an_uploaded_library_imports(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
status, job = await upload(
|
||||||
|
authenticated_client,
|
||||||
|
zip_of(source, tmp_path / "out", prefix="Calibre Library/"),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert status == 202
|
||||||
|
assert job["state"] == "running"
|
||||||
|
assert job["library_id"] == 1
|
||||||
|
|
||||||
|
# The archive's name, not the temp directory it was unpacked into.
|
||||||
|
assert job["source"] == "library.zip"
|
||||||
|
|
||||||
|
finished = await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert finished["state"] == "finished"
|
||||||
|
assert finished["total"] == 3
|
||||||
|
assert finished["created"] == 2
|
||||||
|
assert finished["skipped"] == 1
|
||||||
|
assert finished["failed"] == 0
|
||||||
|
assert finished["error"] is None
|
||||||
|
assert finished["current_title"] is None
|
||||||
|
|
||||||
|
listed = await authenticated_client.get("/books?library_id=1")
|
||||||
|
titles = [book["title"] for book in listed.json()["items"]]
|
||||||
|
assert "The Metamorphosis" in titles
|
||||||
|
assert "The Art of War" in titles
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_library_zipped_without_a_wrapping_folder(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Zipping the contents is as common as zipping the folder."""
|
||||||
|
status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
|
||||||
|
|
||||||
|
assert status == 202
|
||||||
|
finished = await wait_for(authenticated_client, job["id"])
|
||||||
|
assert finished["created"] == 2
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_unpacked_copy_is_cleaned_up(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
An unpacked archive is a second copy of the whole library.
|
||||||
|
|
||||||
|
The books worth keeping have been copied into the library by the time the job ends,
|
||||||
|
so nothing is lost with it — and nothing will come back for it.
|
||||||
|
"""
|
||||||
|
status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
|
||||||
|
assert status == 202
|
||||||
|
|
||||||
|
workspace = registry.get(job["id"]).workspace
|
||||||
|
assert workspace is not None
|
||||||
|
|
||||||
|
await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert not workspace.exists()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_uploading_the_same_library_twice_imports_nothing_new(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Re-running is safe, which is what makes an interrupted import resumable."""
|
||||||
|
archive = zip_of(source, tmp_path / "out")
|
||||||
|
|
||||||
|
for _ in range(2):
|
||||||
|
_, job = await upload(authenticated_client, archive)
|
||||||
|
finished = await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert finished["created"] == 0
|
||||||
|
assert finished["skipped"] == 3
|
||||||
|
|
||||||
|
|
||||||
|
async def test_allow_duplicates_stores_the_files_again(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
The one option the screen offers, and it has to reach the import.
|
||||||
|
|
||||||
|
Without it the second pass skips everything, which is the previous test.
|
||||||
|
"""
|
||||||
|
archive = zip_of(source, tmp_path / "out")
|
||||||
|
|
||||||
|
_, first = await upload(authenticated_client, archive)
|
||||||
|
await wait_for(authenticated_client, first["id"])
|
||||||
|
|
||||||
|
_, second = await upload(authenticated_client, archive, allow_duplicates=True)
|
||||||
|
finished = await wait_for(authenticated_client, second["id"])
|
||||||
|
|
||||||
|
assert finished["created"] == 2
|
||||||
|
assert finished["skipped"] == 1 # still the book with no files
|
||||||
|
|
||||||
|
|
||||||
|
async def test_two_imports_into_one_library_conflict(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
archive = zip_of(source, tmp_path / "out")
|
||||||
|
|
||||||
|
first_status, first = await upload(authenticated_client, archive)
|
||||||
|
assert first_status == 202
|
||||||
|
|
||||||
|
second_status, second = await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert second_status == 409
|
||||||
|
assert second["extra"]["job_id"] == first["id"]
|
||||||
|
|
||||||
|
await wait_for(authenticated_client, first["id"])
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_finished_import_does_not_block_the_next_one(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
archive = zip_of(source, tmp_path / "out")
|
||||||
|
|
||||||
|
_, first = await upload(authenticated_client, archive)
|
||||||
|
await wait_for(authenticated_client, first["id"])
|
||||||
|
|
||||||
|
status, second = await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert status == 202
|
||||||
|
await wait_for(authenticated_client, second["id"])
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cancelling_stops_after_the_current_book(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Cancelling is not aborting: a book abandoned mid-copy would leave files with no row.
|
||||||
|
|
||||||
|
Whether this cancels before any book, after one, or after the lot is a race — the
|
||||||
|
catalogue is three books long. What must hold either way is that the state is
|
||||||
|
terminal and every book it did import is complete.
|
||||||
|
"""
|
||||||
|
_, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
|
||||||
|
|
||||||
|
cancelled = await authenticated_client.delete(f"/libraries/imports/{job['id']}")
|
||||||
|
assert cancelled.status_code == 200
|
||||||
|
|
||||||
|
final = await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert final["state"] in {"cancelled", "finished"}
|
||||||
|
|
||||||
|
listed = await authenticated_client.get("/books?library_id=1")
|
||||||
|
for book in listed.json()["items"]:
|
||||||
|
assert book["files"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_failures_are_reported_on_the_job(
|
||||||
|
authenticated_client: AsyncClient, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||||
|
) -> None:
|
||||||
|
"""One book failing is recorded and does not stop the run."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "calibre")
|
||||||
|
fixture.add_book(1, "Fine", authors=["A"], formats={"EPUB": EPUB})
|
||||||
|
fixture.add_book(2, "Doomed", authors=["B"], formats={"EPUB": OTHER_EPUB})
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
from chitai.services.book import BookService
|
||||||
|
|
||||||
|
original = BookService.create
|
||||||
|
|
||||||
|
async def fail_on_the_second(self, data, *args, **kwargs):
|
||||||
|
if isinstance(data, dict) and data.get("title") == "Doomed":
|
||||||
|
raise RuntimeError("no room on the shelf")
|
||||||
|
return await original(self, data, *args, **kwargs)
|
||||||
|
|
||||||
|
monkeypatch.setattr(BookService, "create", fail_on_the_second)
|
||||||
|
|
||||||
|
_, job = await upload(authenticated_client, zip_of(root, tmp_path / "out"))
|
||||||
|
finished = await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert finished["state"] == "finished"
|
||||||
|
assert finished["created"] == 1
|
||||||
|
assert finished["failed"] == 1
|
||||||
|
assert finished["failures"][0]["calibre_id"] == 2
|
||||||
|
assert "no room on the shelf" in finished["failures"][0]["reason"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_second_copy_is_counted_as_a_possible_duplicate(
|
||||||
|
authenticated_client: AsyncClient, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""A count, not the records — the duplicates screen is what shows them."""
|
||||||
|
padded = tmp_path / "padded.epub"
|
||||||
|
padded.write_bytes(EPUB.read_bytes() + b"\0" * 64)
|
||||||
|
|
||||||
|
fixture = CalibreFixture(tmp_path / "calibre")
|
||||||
|
fixture.add_book(
|
||||||
|
1, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": EPUB}
|
||||||
|
)
|
||||||
|
fixture.add_book(
|
||||||
|
2, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": padded}
|
||||||
|
)
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
_, job = await upload(authenticated_client, zip_of(root, tmp_path / "out"))
|
||||||
|
finished = await wait_for(authenticated_client, job["id"])
|
||||||
|
|
||||||
|
assert finished["created"] == 2
|
||||||
|
assert finished["possible_duplicates"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
async def test_polling_an_unknown_job(authenticated_client: AsyncClient) -> None:
|
||||||
|
response = await authenticated_client.get("/libraries/imports/not-a-job")
|
||||||
|
assert response.status_code == 404
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cancelling_an_unknown_job(authenticated_client: AsyncClient) -> None:
|
||||||
|
response = await authenticated_client.delete("/libraries/imports/not-a-job")
|
||||||
|
assert response.status_code == 404
|
||||||
|
|
||||||
|
|
||||||
|
async def test_importing_into_a_library_that_does_not_exist(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
status, _ = await upload(
|
||||||
|
authenticated_client, zip_of(source, tmp_path / "out"), library_id=999
|
||||||
|
)
|
||||||
|
|
||||||
|
assert status == 404
|
||||||
|
|
||||||
|
|
||||||
|
async def test_importing_into_a_read_only_library(
|
||||||
|
authenticated_client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""A read-only library points at a tree Chitai does not own."""
|
||||||
|
root = tmp_path / "read-only"
|
||||||
|
root.mkdir()
|
||||||
|
|
||||||
|
created = await authenticated_client.post(
|
||||||
|
"/libraries",
|
||||||
|
json={"name": "Read Only", "root_path": str(root), "read_only": True},
|
||||||
|
)
|
||||||
|
assert created.status_code == 201
|
||||||
|
|
||||||
|
status, body = await upload(
|
||||||
|
authenticated_client,
|
||||||
|
zip_of(source, tmp_path / "out"),
|
||||||
|
library_id=created.json()["id"],
|
||||||
|
)
|
||||||
|
|
||||||
|
assert status == 400
|
||||||
|
assert "read-only" in body["detail"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_an_import_needs_authentication(
|
||||||
|
client: AsyncClient, source: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
status, _ = await upload(client, zip_of(source, tmp_path / "out"))
|
||||||
|
|
||||||
|
assert status == 401
|
||||||
|
|
||||||
|
|
||||||
|
class TestRefusedArchives:
|
||||||
|
"""Everything wrong with an archive is answered now, not as a job that fails later."""
|
||||||
|
|
||||||
|
async def test_a_hostile_archive(
|
||||||
|
self, authenticated_client: AsyncClient, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Zip slip."""
|
||||||
|
archive = tmp_path / "hostile.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("metadata.db", "not really")
|
||||||
|
writing.writestr("../../escaped.txt", "gotcha")
|
||||||
|
|
||||||
|
status, body = await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert status == 400
|
||||||
|
assert "outside itself" in body["detail"]
|
||||||
|
assert registry._jobs == {}
|
||||||
|
|
||||||
|
async def test_something_that_is_not_a_zip(
|
||||||
|
self, authenticated_client: AsyncClient, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
archive = tmp_path / "notes.txt"
|
||||||
|
archive.write_bytes(b"just some text")
|
||||||
|
|
||||||
|
status, body = await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert status == 400
|
||||||
|
assert "not a zip" in body["detail"]
|
||||||
|
|
||||||
|
async def test_an_archive_with_no_catalogue(
|
||||||
|
self, authenticated_client: AsyncClient, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
archive = tmp_path / "books.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("Some Book.epub", "content")
|
||||||
|
|
||||||
|
status, body = await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert status == 400
|
||||||
|
assert "no metadata.db" in body["detail"]
|
||||||
|
|
||||||
|
async def test_a_refusal_leaves_no_temp_files(
|
||||||
|
self, authenticated_client: AsyncClient, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Every refusal path removes the workspace it had already made."""
|
||||||
|
before = set(Path(tempfile.gettempdir()).glob("tmp*"))
|
||||||
|
|
||||||
|
archive = tmp_path / "books.zip"
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("Some Book.epub", "content")
|
||||||
|
|
||||||
|
await upload(authenticated_client, archive)
|
||||||
|
|
||||||
|
assert set(Path(tempfile.gettempdir()).glob("tmp*")) == before
|
||||||
@@ -8,7 +8,9 @@ from pathlib import Path
|
|||||||
# Known KOReader hashes for test files
|
# Known KOReader hashes for test files
|
||||||
TEST_FILES = {
|
TEST_FILES = {
|
||||||
"Moby Dick; Or, The Whale - Herman Melville.epub": {
|
"Moby Dick; Or, The Whale - Herman Melville.epub": {
|
||||||
"path": Path("tests/data_files/Moby Dick; Or, The Whale - Herman Melville.epub"),
|
"path": Path(
|
||||||
|
"tests/data_files/Moby Dick; Or, The Whale - Herman Melville.epub"
|
||||||
|
),
|
||||||
"hash": "ceeef909ec65653ba77e1380dff998fb",
|
"hash": "ceeef909ec65653ba77e1380dff998fb",
|
||||||
"content_type": "application/epub+zip",
|
"content_type": "application/epub+zip",
|
||||||
},
|
},
|
||||||
@@ -59,7 +61,9 @@ async def test_add_file_to_book_generates_correct_hash(
|
|||||||
first_book = TEST_FILES["Moby Dick; Or, The Whale - Herman Melville.epub"]
|
first_book = TEST_FILES["Moby Dick; Or, The Whale - Herman Melville.epub"]
|
||||||
first_content = first_book["path"].read_bytes()
|
first_content = first_book["path"].read_bytes()
|
||||||
|
|
||||||
files = [("files", (first_book["path"].name, first_content, first_book["content_type"]))]
|
files = [
|
||||||
|
("files", (first_book["path"].name, first_content, first_book["content_type"]))
|
||||||
|
]
|
||||||
data = {"library_id": "1"}
|
data = {"library_id": "1"}
|
||||||
|
|
||||||
create_response = await authenticated_client.post(
|
create_response = await authenticated_client.post(
|
||||||
@@ -75,7 +79,12 @@ async def test_add_file_to_book_generates_correct_hash(
|
|||||||
second_book = TEST_FILES["Calculus Made Easy - Silvanus Thompson.pdf"]
|
second_book = TEST_FILES["Calculus Made Easy - Silvanus Thompson.pdf"]
|
||||||
second_content = second_book["path"].read_bytes()
|
second_content = second_book["path"].read_bytes()
|
||||||
|
|
||||||
add_files = [("data", (second_book["path"].name, second_content, second_book["content_type"]))]
|
add_files = [
|
||||||
|
(
|
||||||
|
"data",
|
||||||
|
(second_book["path"].name, second_content, second_book["content_type"]),
|
||||||
|
)
|
||||||
|
]
|
||||||
|
|
||||||
add_response = await authenticated_client.post(
|
add_response = await authenticated_client.post(
|
||||||
f"/books/{book_id}/files",
|
f"/books/{book_id}/files",
|
||||||
|
|||||||
@@ -40,5 +40,5 @@ async def test_create_library(
|
|||||||
assert result["name"] == "Test Library"
|
assert result["name"] == "Test Library"
|
||||||
assert result["root_path"] == f"{tmp_path}/books"
|
assert result["root_path"] == f"{tmp_path}/books"
|
||||||
assert result["path_template"] == "{author}/{title}"
|
assert result["path_template"] == "{author}/{title}"
|
||||||
assert result["read_only"] == False
|
assert result["read_only"] is False
|
||||||
assert result["description"] is None
|
assert result["description"] is None
|
||||||
|
|||||||
@@ -0,0 +1,392 @@
|
|||||||
|
import zipfile
|
||||||
|
from datetime import date
|
||||||
|
from pathlib import Path
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from chitai.services.calibre import (
|
||||||
|
CalibreLibrary,
|
||||||
|
CalibreLibraryError,
|
||||||
|
extract_calibre_archive,
|
||||||
|
format_series_index,
|
||||||
|
parse_date,
|
||||||
|
strip_html,
|
||||||
|
unescape_author,
|
||||||
|
)
|
||||||
|
|
||||||
|
from tests.calibre_fixtures import UNDEFINED_DATE, CalibreFixture
|
||||||
|
|
||||||
|
|
||||||
|
EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
PDF = Path("tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(name="library_root")
|
||||||
|
def fx_library_root(tmp_path: Path) -> Path:
|
||||||
|
"""A small Calibre library covering the rows that are easy to read wrongly."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "Calibre Library")
|
||||||
|
fixture.add_pages_table()
|
||||||
|
|
||||||
|
fixture.add_book(
|
||||||
|
1,
|
||||||
|
"The Metamorphosis",
|
||||||
|
authors=["Franz Kafka"],
|
||||||
|
pubdate="1915-10-15 00:00:00+00:00",
|
||||||
|
tags=["Fiction", "Absurdist"],
|
||||||
|
publisher="Kurt Wolff Verlag",
|
||||||
|
languages=["deu", "eng"],
|
||||||
|
comment="<p>A travelling salesman.</p><p>He wakes up <i>changed</i>.</p>",
|
||||||
|
identifiers={"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"},
|
||||||
|
uuid="11111111-2222-3333-4444-555555555555",
|
||||||
|
pages=201,
|
||||||
|
cover=True,
|
||||||
|
formats={"EPUB": EPUB},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Volume seven of a series, and no publication date — the two values most likely to
|
||||||
|
# be carried through verbatim when they should not be.
|
||||||
|
fixture.add_book(
|
||||||
|
2,
|
||||||
|
"Persepolis Rising",
|
||||||
|
# Calibre escapes the comma and nothing else, so the space after it is stored
|
||||||
|
# as-is: `Corey, Jr.` is written `Corey| Jr.`.
|
||||||
|
authors=["Corey| Jr., James S. A."],
|
||||||
|
series="The Expanse",
|
||||||
|
series_index=7.0,
|
||||||
|
formats={"EPUB": EPUB},
|
||||||
|
)
|
||||||
|
|
||||||
|
# A novella between two novels: a fractional position is real and must survive.
|
||||||
|
fixture.add_book(
|
||||||
|
3, "Strange Dogs", series="The Expanse", series_index=6.5, formats={"PDF": PDF}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Every row Calibre will happily hold and Chitai cannot use: no files at all.
|
||||||
|
fixture.add_book(4, "Metadata Only")
|
||||||
|
|
||||||
|
# A catalogue row whose file is not on disk.
|
||||||
|
fixture.add_book(5, "Lost Book")
|
||||||
|
fixture.add_missing_format(5, "EPUB", "Lost Book - Unknown")
|
||||||
|
|
||||||
|
return fixture.commit()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_reads_a_book_whole(library_root: Path) -> None:
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
assert await library.count() == 5
|
||||||
|
books = await library.books()
|
||||||
|
|
||||||
|
book = books[0]
|
||||||
|
|
||||||
|
assert book.calibre_id == 1
|
||||||
|
assert book.title == "The Metamorphosis"
|
||||||
|
assert book.authors == ["Franz Kafka"]
|
||||||
|
assert book.published_date == date(1915, 10, 15)
|
||||||
|
assert book.tags == ["Absurdist", "Fiction"]
|
||||||
|
assert book.publisher == "Kurt Wolff Verlag"
|
||||||
|
assert book.pages == 201
|
||||||
|
assert book.uuid == "11111111-2222-3333-4444-555555555555"
|
||||||
|
|
||||||
|
# One language, and the one Calibre put first.
|
||||||
|
assert book.language == "deu"
|
||||||
|
|
||||||
|
# Reported as Calibre wrote them: folding `amazon` onto `asin` is the importer's
|
||||||
|
# job, not the reader's.
|
||||||
|
assert book.identifiers == {"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"}
|
||||||
|
|
||||||
|
assert book.cover is not None
|
||||||
|
assert book.cover.is_file()
|
||||||
|
|
||||||
|
assert len(book.files) == 1
|
||||||
|
assert book.files[0].format == "EPUB"
|
||||||
|
assert book.files[0].path.is_file()
|
||||||
|
# The stem is Calibre's, truncated and sanitised — never the title.
|
||||||
|
assert book.files[0].path.name != f"{book.title}.epub"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_undefined_date_is_not_a_date(library_root: Path) -> None:
|
||||||
|
"""`0101-01-01` parses fine, which is exactly the problem."""
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
books = {book.calibre_id: book for book in await library.books()}
|
||||||
|
|
||||||
|
assert books[2].published_date is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_series_position_is_a_plain_string(library_root: Path) -> None:
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
books = {book.calibre_id: book for book in await library.books()}
|
||||||
|
|
||||||
|
assert books[2].series == "The Expanse"
|
||||||
|
assert books[2].series_position == "7"
|
||||||
|
|
||||||
|
assert books[3].series_position == "6.5"
|
||||||
|
|
||||||
|
# `series_index` defaults to 1.0 for every book, so a position without a series
|
||||||
|
# would invent a volume one out of nothing.
|
||||||
|
assert books[1].series is None
|
||||||
|
assert books[1].series_position is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_author_commas_are_unescaped(library_root: Path) -> None:
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
books = {book.calibre_id: book for book in await library.books()}
|
||||||
|
|
||||||
|
assert books[2].authors == ["Corey, Jr., James S. A."]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_comments_come_back_as_text(library_root: Path) -> None:
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
books = {book.calibre_id: book for book in await library.books()}
|
||||||
|
|
||||||
|
assert books[1].description == "A travelling salesman.\nHe wakes up changed."
|
||||||
|
assert books[2].description is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_files_are_reported_whether_or_not_they_exist(library_root: Path) -> None:
|
||||||
|
"""
|
||||||
|
The reader says what the catalogue says. Whether the bytes are there is a question
|
||||||
|
for whoever is about to copy them, which stats them anyway.
|
||||||
|
"""
|
||||||
|
async with CalibreLibrary(library_root) as library:
|
||||||
|
books = {book.calibre_id: book for book in await library.books()}
|
||||||
|
|
||||||
|
assert books[4].files == []
|
||||||
|
|
||||||
|
assert len(books[5].files) == 1
|
||||||
|
assert not books[5].files[0].path.exists()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_library_without_the_pages_table_still_reads(tmp_path: Path) -> None:
|
||||||
|
"""`books_pages_link` is recent; an older library simply does not have it."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "Old Library")
|
||||||
|
fixture.add_book(1, "Old Book", formats={"EPUB": EPUB})
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
async with CalibreLibrary(root) as library:
|
||||||
|
books = await library.books()
|
||||||
|
|
||||||
|
assert books[0].pages is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_original_is_never_opened(library_root: Path) -> None:
|
||||||
|
"""
|
||||||
|
The catalogue is copied before it is read, and the copy goes away afterwards.
|
||||||
|
|
||||||
|
Calibre may be running and writing; this is what keeps a live library out of it.
|
||||||
|
"""
|
||||||
|
before = (library_root / "metadata.db").read_bytes()
|
||||||
|
|
||||||
|
library = CalibreLibrary(library_root)
|
||||||
|
await library.open()
|
||||||
|
workspace = library._workspace
|
||||||
|
|
||||||
|
assert workspace is not None and (workspace / "metadata.db").is_file()
|
||||||
|
|
||||||
|
await library.close()
|
||||||
|
|
||||||
|
assert not workspace.exists()
|
||||||
|
assert (library_root / "metadata.db").read_bytes() == before
|
||||||
|
|
||||||
|
|
||||||
|
async def test_closing_twice_is_harmless(library_root: Path) -> None:
|
||||||
|
library = CalibreLibrary(library_root)
|
||||||
|
await library.open()
|
||||||
|
await library.close()
|
||||||
|
await library.close()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_directory_that_is_not_a_calibre_library(tmp_path: Path) -> None:
|
||||||
|
with pytest.raises(CalibreLibraryError, match="not a Calibre library"):
|
||||||
|
await CalibreLibrary(tmp_path).open()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("stored", "expected"),
|
||||||
|
[
|
||||||
|
("2017-12-04 04:00:00+00:00", date(2017, 12, 4)),
|
||||||
|
("2001-07-02 00:00:00+00:00", date(2001, 7, 2)),
|
||||||
|
("1999-01-31", date(1999, 1, 31)),
|
||||||
|
# Calibre's sentinel, and anything else implausibly early.
|
||||||
|
(UNDEFINED_DATE, None),
|
||||||
|
("0101-01-01", None),
|
||||||
|
(None, None),
|
||||||
|
("", None),
|
||||||
|
("not a date", None),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_parse_date(stored: str | None, expected: date | None) -> None:
|
||||||
|
assert parse_date(stored) == expected
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("index", "expected"),
|
||||||
|
[
|
||||||
|
(7.0, "7"),
|
||||||
|
(1.0, "1"),
|
||||||
|
(6.5, "6.5"),
|
||||||
|
(0.0, "0"),
|
||||||
|
(12.25, "12.25"),
|
||||||
|
(None, None),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_format_series_index(index: float | None, expected: str | None) -> None:
|
||||||
|
assert format_series_index(index) == expected
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("stored", "expected"),
|
||||||
|
[
|
||||||
|
("Doyle| Sir Arthur Conan", "Doyle, Sir Arthur Conan"),
|
||||||
|
("Franz Kafka", "Franz Kafka"),
|
||||||
|
(" Herman Melville ", "Herman Melville"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_unescape_author(stored: str, expected: str) -> None:
|
||||||
|
assert unescape_author(stored) == expected
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("html", "expected"),
|
||||||
|
[
|
||||||
|
("<p>One.</p><p>Two.</p>", "One.\nTwo."),
|
||||||
|
("Plain text", "Plain text"),
|
||||||
|
("<div>A<br>B</div>", "A\nB"),
|
||||||
|
("<p>Café & bar</p>", "Café & bar"),
|
||||||
|
("<ul><li>One</li><li>Two</li></ul>", "One\nTwo"),
|
||||||
|
# Markup carrying no text at all is nothing, not an empty description.
|
||||||
|
("<p></p>", None),
|
||||||
|
("", None),
|
||||||
|
(None, None),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_strip_html(html: str | None, expected: str | None) -> None:
|
||||||
|
assert strip_html(html) == expected
|
||||||
|
|
||||||
|
|
||||||
|
class TestArchives:
|
||||||
|
"""A Calibre library that arrives zipped rather than as a path."""
|
||||||
|
|
||||||
|
def zipped(self, root: Path, into: Path, prefix: str = "") -> Path:
|
||||||
|
"""Zip a directory the way a file manager would."""
|
||||||
|
archive = into / "library.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
for path in sorted(root.rglob("*")):
|
||||||
|
if path.is_file():
|
||||||
|
writing.write(path, f"{prefix}{path.relative_to(root)}")
|
||||||
|
|
||||||
|
return archive
|
||||||
|
|
||||||
|
async def test_a_library_zipped_at_its_root(
|
||||||
|
self, library_root: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
archive = self.zipped(library_root, tmp_path)
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
catalogue = await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
assert catalogue == destination
|
||||||
|
async with CalibreLibrary(catalogue) as library:
|
||||||
|
assert await library.count() == 5
|
||||||
|
|
||||||
|
async def test_a_library_zipped_inside_a_folder(
|
||||||
|
self, library_root: Path, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Zipping the folder itself is at least as common as zipping its contents."""
|
||||||
|
archive = self.zipped(library_root, tmp_path, prefix="Calibre Library/")
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
catalogue = await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
assert catalogue == destination / "Calibre Library"
|
||||||
|
async with CalibreLibrary(catalogue) as library:
|
||||||
|
assert await library.count() == 5
|
||||||
|
|
||||||
|
async def test_an_entry_pointing_outside_the_archive_is_refused(
|
||||||
|
self, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Zip slip. `ZipFile.extract` sanitises names itself, but relying on that silently
|
||||||
|
is how the next person to change the extraction call reintroduces it.
|
||||||
|
"""
|
||||||
|
archive = tmp_path / "hostile.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("metadata.db", "not really")
|
||||||
|
writing.writestr("../../escaped.txt", "gotcha")
|
||||||
|
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
with pytest.raises(CalibreLibraryError, match="outside itself"):
|
||||||
|
await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
assert not (tmp_path.parent / "escaped.txt").exists()
|
||||||
|
|
||||||
|
async def test_something_that_is_not_a_zip(self, tmp_path: Path) -> None:
|
||||||
|
archive = tmp_path / "not.zip"
|
||||||
|
archive.write_bytes(b"PK-ish, but no")
|
||||||
|
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
with pytest.raises(CalibreLibraryError, match="not a zip file"):
|
||||||
|
await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
async def test_an_archive_with_no_catalogue(self, tmp_path: Path) -> None:
|
||||||
|
archive = tmp_path / "books.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("Some Book.epub", "content")
|
||||||
|
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
with pytest.raises(CalibreLibraryError, match="no metadata.db"):
|
||||||
|
await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
# Refused before anything was written.
|
||||||
|
assert list(destination.iterdir()) == []
|
||||||
|
|
||||||
|
async def test_a_catalogue_buried_too_deep(self, tmp_path: Path) -> None:
|
||||||
|
"""Somebody's whole backup tree is not a library, however much it contains one."""
|
||||||
|
archive = tmp_path / "backup.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("backups/2026/january/library/metadata.db", "not really")
|
||||||
|
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
with pytest.raises(CalibreLibraryError, match="within 3 levels"):
|
||||||
|
await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
async def test_an_archive_too_big_for_the_disk(
|
||||||
|
self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Checked before writing, not discovered part-way through.
|
||||||
|
|
||||||
|
A full disk takes the whole application down, and the size is in the archive
|
||||||
|
already.
|
||||||
|
"""
|
||||||
|
archive = tmp_path / "huge.zip"
|
||||||
|
|
||||||
|
with zipfile.ZipFile(archive, "w") as writing:
|
||||||
|
writing.writestr("metadata.db", "not really")
|
||||||
|
|
||||||
|
destination = tmp_path / "unpacked"
|
||||||
|
destination.mkdir()
|
||||||
|
|
||||||
|
monkeypatch.setattr(
|
||||||
|
"chitai.services.calibre.shutil.disk_usage",
|
||||||
|
lambda _path: SimpleNamespace(total=1024, used=1024, free=0),
|
||||||
|
)
|
||||||
|
|
||||||
|
with pytest.raises(CalibreLibraryError, match="only"):
|
||||||
|
await extract_calibre_archive(archive, destination)
|
||||||
|
|
||||||
|
assert list(destination.iterdir()) == []
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from chitai.services.utils import guess_content_type
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("filename", "expected"),
|
||||||
|
[
|
||||||
|
# What `mimetypes` already knows, kept here so a host with a thin
|
||||||
|
# /etc/mime.types cannot change the answer without a test noticing.
|
||||||
|
("Frankenstein.epub", "application/epub+zip"),
|
||||||
|
("Calculus.pdf", "application/pdf"),
|
||||||
|
("Persepolis.azw3", "application/vnd.amazon.mobi8-ebook"),
|
||||||
|
("Watchmen.cbz", "application/vnd.comicbook+zip"),
|
||||||
|
# What it does not, and where a Calibre library's older formats live.
|
||||||
|
("Dune.mobi", "application/x-mobipocket-ebook"),
|
||||||
|
("Dune.prc", "application/x-mobipocket-ebook"),
|
||||||
|
("Dune.azw", "application/vnd.amazon.ebook"),
|
||||||
|
("Voyna i Mir.fb2", "application/x-fictionbook+xml"),
|
||||||
|
("Voyna i Mir.fbz", "application/x-zip-compressed-fb2"),
|
||||||
|
("Reader.lit", "application/x-ms-reader"),
|
||||||
|
("Reader.lrf", "application/x-sony-bbeb"),
|
||||||
|
("Watchmen.cb7", "application/x-cb7"),
|
||||||
|
# Case is not part of the answer, and Calibre writes formats uppercase.
|
||||||
|
("Dune.MOBI", "application/x-mobipocket-ebook"),
|
||||||
|
# Nothing can name these, and None is the answer rather than a placeholder.
|
||||||
|
("Notes.xyzzy", None),
|
||||||
|
("README", None),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_guess_content_type(filename: str, expected: str | None) -> None:
|
||||||
|
assert guess_content_type(Path(filename)) == expected
|
||||||
|
# A str and a Path must agree, and an upload's `filename` carries its relative
|
||||||
|
# path, so a name with directories in front of it has to resolve the same way.
|
||||||
|
assert guess_content_type(filename) == expected
|
||||||
|
assert guess_content_type(f"Some Author/Some Book/{filename}") == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_fallback_is_used_only_when_the_extension_says_nothing() -> None:
|
||||||
|
"""A client's claim fills a gap; it never overrides the name."""
|
||||||
|
assert (
|
||||||
|
guess_content_type(Path("Dune.mobi"), fallback="application/pdf")
|
||||||
|
== "application/x-mobipocket-ebook"
|
||||||
|
)
|
||||||
|
assert (
|
||||||
|
guess_content_type(Path("Notes.xyzzy"), fallback="application/epub+zip")
|
||||||
|
== "application/epub+zip"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_an_unspecified_fallback_is_not_an_answer() -> None:
|
||||||
|
"""
|
||||||
|
`application/octet-stream` from a client is it saying it does not know.
|
||||||
|
|
||||||
|
Browsers post exactly that for every extension they do not recognise, which is most
|
||||||
|
ebook formats. Storing it would be indistinguishable from having determined a
|
||||||
|
format, so it is discarded and the column keeps its null.
|
||||||
|
"""
|
||||||
|
assert (
|
||||||
|
guess_content_type(Path("Notes.xyzzy"), fallback="application/octet-stream")
|
||||||
|
is None
|
||||||
|
)
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
"""Tests for BookPathGenerator."""
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from chitai.services.filesystem_library import (
|
||||||
|
BookPathGenerator,
|
||||||
|
sanitize_path_component,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
ROOT = Path("/library")
|
||||||
|
|
||||||
|
|
||||||
|
def path_for(**book) -> Path:
|
||||||
|
return BookPathGenerator(ROOT).generate_path(book)
|
||||||
|
|
||||||
|
|
||||||
|
def test_author_and_title() -> None:
|
||||||
|
assert path_for(title="Dune", authors=["Frank Herbert"]) == (
|
||||||
|
ROOT / "Frank Herbert" / "Dune"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_book_with_no_authors() -> None:
|
||||||
|
assert path_for(title="Beowulf", authors=[]) == ROOT / "Unknown" / "Beowulf"
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_series_adds_a_level_and_pads_the_position() -> None:
|
||||||
|
assert (
|
||||||
|
path_for(
|
||||||
|
title="Persepolis Rising",
|
||||||
|
authors=["James S. A. Corey"],
|
||||||
|
series="The Expanse",
|
||||||
|
series_position="7",
|
||||||
|
)
|
||||||
|
== ROOT / "James S. A. Corey" / "The Expanse" / "07 - Persepolis Rising"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_slash_in_a_title_does_not_add_a_directory() -> None:
|
||||||
|
"""
|
||||||
|
The separators in the path come from the template, never from the metadata.
|
||||||
|
|
||||||
|
A title with a slash in it — "AC/DC", "Him/Her" — would otherwise put the book one
|
||||||
|
level below where `book.path` says it is, which is what deletes, moves and file
|
||||||
|
lookups all act on. Calibre keeps the real title in its database and strips this
|
||||||
|
from its own directory names, so an import is where they surface.
|
||||||
|
"""
|
||||||
|
generated = path_for(title="Back in Black: AC/DC", authors=["Murray Engleheart"])
|
||||||
|
|
||||||
|
assert generated == ROOT / "Murray Engleheart" / "Back in Black: AC_DC"
|
||||||
|
assert generated.relative_to(ROOT).parts == (
|
||||||
|
"Murray Engleheart",
|
||||||
|
"Back in Black: AC_DC",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_slash_in_an_author_or_series_is_handled_too() -> None:
|
||||||
|
assert path_for(title="Split", authors=["A/B Collective"]) == (
|
||||||
|
ROOT / "A_B Collective" / "Split"
|
||||||
|
)
|
||||||
|
assert (
|
||||||
|
path_for(
|
||||||
|
title="Volume One",
|
||||||
|
authors=["Someone"],
|
||||||
|
series="Either/Or",
|
||||||
|
series_position="1",
|
||||||
|
)
|
||||||
|
== ROOT / "Someone" / "Either_Or" / "01 - Volume One"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_control_characters_are_removed() -> None:
|
||||||
|
assert path_for(title="Line\nBreak", authors=["Someone"]) == (
|
||||||
|
ROOT / "Someone" / "Line_Break"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_sanitize_path_component() -> None:
|
||||||
|
assert sanitize_path_component("AC/DC") == "AC_DC"
|
||||||
|
assert sanitize_path_component("back\\slash") == "back_slash"
|
||||||
|
assert sanitize_path_component(" padded ") == "padded"
|
||||||
|
# Colons and other punctuation are legal in a path and are left alone.
|
||||||
|
assert sanitize_path_component("Title: Subtitle") == "Title: Subtitle"
|
||||||
@@ -0,0 +1,241 @@
|
|||||||
|
"""Tests for the normalization behind book-level duplicate detection."""
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from chitai.services.matching import (
|
||||||
|
format_author_name,
|
||||||
|
normalize_author,
|
||||||
|
normalize_identifier,
|
||||||
|
normalize_title,
|
||||||
|
)
|
||||||
|
from chitai.services.metadata_extractor import parse_identifier
|
||||||
|
from chitai.services.utils import isbn10_to_isbn13
|
||||||
|
|
||||||
|
|
||||||
|
class TestNormalizeTitle:
|
||||||
|
"""Two copies of one book rarely agree on how the title is written."""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("title", "expected"),
|
||||||
|
[
|
||||||
|
("The Metamorphosis", "metamorphosis"),
|
||||||
|
("Metamorphosis", "metamorphosis"),
|
||||||
|
("METAMORPHOSIS", "metamorphosis"),
|
||||||
|
("A Tale of Two Cities", "tale of two cities"),
|
||||||
|
("An Enquiry", "enquiry"),
|
||||||
|
# Accents, punctuation and ampersands are spelling, not identity.
|
||||||
|
("Les Misérables", "les miserables"),
|
||||||
|
("Moby Dick; Or, The Whale", "moby dick or the whale"),
|
||||||
|
("Sense & Sensibility", "sense and sensibility"),
|
||||||
|
# Bracketed asides and trailing edition noise say nothing about the book.
|
||||||
|
("Frankenstein (Illustrated)", "frankenstein"),
|
||||||
|
("Frankenstein [Kindle Edition]", "frankenstein"),
|
||||||
|
("Frankenstein, 2nd Edition", "frankenstein"),
|
||||||
|
# The compact forms a cover actually carries.
|
||||||
|
("Building Microservices, 2E", "building microservices"),
|
||||||
|
("Building Microservices 2e", "building microservices"),
|
||||||
|
("Frankenstein 3 Ed", "frankenstein"),
|
||||||
|
("Dungeons & Dragons 5e", "dungeons and dragons"),
|
||||||
|
("Frankenstein Revised Edition", "frankenstein"),
|
||||||
|
("Dune Deluxe Edition Illustrated", "dune"),
|
||||||
|
("", ""),
|
||||||
|
(None, ""),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_titles_that_should_agree(self, title: str | None, expected: str) -> None:
|
||||||
|
assert normalize_title(title) == expected
|
||||||
|
|
||||||
|
def test_a_qualifier_that_is_the_title_survives(self) -> None:
|
||||||
|
"""A trailing qualifier is noise; the same word at the front is the book."""
|
||||||
|
assert normalize_title("The Illustrated Man") == "illustrated man"
|
||||||
|
|
||||||
|
def test_normalization_never_empties_a_title(self) -> None:
|
||||||
|
"""An article-only title is not improved by having no article left."""
|
||||||
|
assert normalize_title("The") == "the"
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"title",
|
||||||
|
["Catch 22", "Fahrenheit 451", "Blade Runner 2049", "1984", "Apollo 13"],
|
||||||
|
)
|
||||||
|
def test_a_number_is_not_an_edition(self, title: str) -> None:
|
||||||
|
"""Edition stripping keys on the `e`; a bare number is part of the title."""
|
||||||
|
assert normalize_title(title) == title.casefold()
|
||||||
|
|
||||||
|
|
||||||
|
class TestNormalizeAuthor:
|
||||||
|
"""One person, written down several ways."""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("name", "expected"),
|
||||||
|
[
|
||||||
|
("Franz Kafka", "franz kafka"),
|
||||||
|
("Kafka, Franz", "franz kafka"),
|
||||||
|
("KAFKA, FRANZ", "franz kafka"),
|
||||||
|
("Émile Zola", "emile zola"),
|
||||||
|
("Doyle, Arthur Conan", "arthur conan doyle"),
|
||||||
|
# Runs of initials are joined, so spacing them out changes nothing.
|
||||||
|
("J.R.R. Tolkien", "jrr tolkien"),
|
||||||
|
("J. R. R. Tolkien", "jrr tolkien"),
|
||||||
|
("JRR Tolkien", "jrr tolkien"),
|
||||||
|
("", ""),
|
||||||
|
(None, ""),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_names_that_should_agree(self, name: str | None, expected: str) -> None:
|
||||||
|
assert normalize_author(name) == expected
|
||||||
|
|
||||||
|
def test_two_people_are_not_reduced_together(self) -> None:
|
||||||
|
"""Surname plus initial would collide unrelated writers; it is not used."""
|
||||||
|
assert normalize_author("Charles Dickens") != normalize_author("Colin Dexter")
|
||||||
|
|
||||||
|
def test_a_list_is_left_alone(self) -> None:
|
||||||
|
"""More than one comma is a list or a suffix, and guessing does more harm."""
|
||||||
|
assert normalize_author("Smith, John, Jr.") == "smith john jr"
|
||||||
|
|
||||||
|
|
||||||
|
class TestFormatAuthorName:
|
||||||
|
"""What gets stored and shown, as opposed to what gets compared."""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("written", "expected"),
|
||||||
|
[
|
||||||
|
# A leftover separator from a `DC:creator` list.
|
||||||
|
("Newman, Sam;", "Sam Newman"),
|
||||||
|
("Sam Newman ", "Sam Newman"),
|
||||||
|
(" Dan Vanderkam ", "Dan Vanderkam"),
|
||||||
|
# `Surname, Given` is how EPUBs file a name, not how anyone reads it.
|
||||||
|
("Kleppmann, Martin", "Martin Kleppmann"),
|
||||||
|
("Huxley, Aldous", "Aldous Huxley"),
|
||||||
|
("Liu, Cixin", "Cixin Liu"),
|
||||||
|
# An extension carried in from the filename the name was read out of.
|
||||||
|
("Sam Newman.epub", "Sam Newman"),
|
||||||
|
("Franz Kafka.mobi", "Franz Kafka"),
|
||||||
|
("Brian W. Kernighan.epub", "Brian W. Kernighan"),
|
||||||
|
("", ""),
|
||||||
|
(None, ""),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_names_are_tidied(self, written: str | None, expected: str) -> None:
|
||||||
|
assert format_author_name(written) == expected
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"written",
|
||||||
|
[
|
||||||
|
# Two people in one string. Flipping it would invent a third.
|
||||||
|
"Dave Thomas, Andy Hunt",
|
||||||
|
"Mark Richards, Neal Ford",
|
||||||
|
# A compound surname is not recognised, and is left alone rather than
|
||||||
|
# rearranged wrongly.
|
||||||
|
"García Márquez, Gabriel",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_an_unrecognised_form_is_left_alone(self, written: str) -> None:
|
||||||
|
assert format_author_name(written) == written
|
||||||
|
|
||||||
|
def test_case_and_accents_belong_to_the_author(self) -> None:
|
||||||
|
"""Tidying removes what an extractor added; it does not correct spelling."""
|
||||||
|
assert format_author_name("Michał Płachta.epub") == "Michał Płachta"
|
||||||
|
assert format_author_name("Steve McConnell") == "Steve McConnell"
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"written", ["Newman, Sam;", "Sam Newman.epub", "Kleppmann, Martin"]
|
||||||
|
)
|
||||||
|
def test_tidying_is_idempotent(self, written: str) -> None:
|
||||||
|
"""`unique_filter` tidies a name that may already be tidy; it must not drift."""
|
||||||
|
once = format_author_name(written)
|
||||||
|
assert format_author_name(once) == once
|
||||||
|
|
||||||
|
|
||||||
|
class TestNormalizeIdentifier:
|
||||||
|
"""Identifiers only help if the same edition produces the same key."""
|
||||||
|
|
||||||
|
def test_isbn_10_and_isbn_13_are_one_key(self) -> None:
|
||||||
|
assert normalize_identifier("isbn-10", "0486282112") == "isbn:9780486282114"
|
||||||
|
assert normalize_identifier("isbn-13", "9780486282114") == "isbn:9780486282114"
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"written", ["978-0-486-28211-4", "978 0 486 28211 4", "9780486282114"]
|
||||||
|
)
|
||||||
|
def test_formatting_is_not_part_of_an_isbn(self, written: str) -> None:
|
||||||
|
assert normalize_identifier("isbn", written) == "isbn:9780486282114"
|
||||||
|
|
||||||
|
def test_an_isbn_that_fails_its_checksum_is_no_evidence(self) -> None:
|
||||||
|
assert normalize_identifier("isbn-13", "9780486282115") is None
|
||||||
|
|
||||||
|
def test_uuids_are_refused(self) -> None:
|
||||||
|
"""Generated per build, so they only re-find what the hash check catches."""
|
||||||
|
assert (
|
||||||
|
normalize_identifier("uuid", "3f2b1c4e-1111-2222-3333-444455556666") is None
|
||||||
|
)
|
||||||
|
assert (
|
||||||
|
normalize_identifier("urn:uuid", "3f2b1c4e-1111-2222-3333-444455556666")
|
||||||
|
is None
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_other_schemes_keep_their_own_key(self) -> None:
|
||||||
|
assert normalize_identifier("asin", "B000FC0PDA") == "asin:b000fc0pda"
|
||||||
|
assert normalize_identifier("ASIN", "b000fc0pda") == "asin:b000fc0pda"
|
||||||
|
|
||||||
|
def test_something_too_short_is_not_evidence(self) -> None:
|
||||||
|
"""A Calibre id of "42" would otherwise pair two unrelated books."""
|
||||||
|
assert normalize_identifier("calibre", "42") is None
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(("name", "value"), [("", "1234567"), ("asin", "")])
|
||||||
|
def test_half_an_identifier_is_no_identifier(self, name: str, value: str) -> None:
|
||||||
|
assert normalize_identifier(name, value) is None
|
||||||
|
|
||||||
|
|
||||||
|
class TestIsbnConversion:
|
||||||
|
def test_isbn_10_converts_to_its_isbn_13(self) -> None:
|
||||||
|
assert isbn10_to_isbn13("0486282112") == "9780486282114"
|
||||||
|
|
||||||
|
def test_a_trailing_x_is_a_digit(self) -> None:
|
||||||
|
assert isbn10_to_isbn13("043942089X") == "9780439420891"
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("isbn", ["0486282113", "9780486282114", "nonsense"])
|
||||||
|
def test_anything_that_is_not_an_isbn_10_converts_to_nothing(
|
||||||
|
self, isbn: str
|
||||||
|
) -> None:
|
||||||
|
assert isbn10_to_isbn13(isbn) is None
|
||||||
|
|
||||||
|
|
||||||
|
class TestParseIdentifier:
|
||||||
|
"""What an EPUB writes, and what is worth storing for it."""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"written",
|
||||||
|
[
|
||||||
|
"9780486282114",
|
||||||
|
"978-0-486-28211-4",
|
||||||
|
"urn:isbn:9780486282114",
|
||||||
|
"urn:isbn:978-0-486-28211-4",
|
||||||
|
"ISBN:978-0-486-28211-4",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_isbns_survive_however_they_are_written(self, written: str) -> None:
|
||||||
|
assert parse_identifier(written) == ("isbn-13", "9780486282114")
|
||||||
|
|
||||||
|
def test_the_scheme_attribute_is_read_too(self) -> None:
|
||||||
|
assert parse_identifier("0-486-28211-2", "ISBN") == ("isbn-10", "0486282112")
|
||||||
|
|
||||||
|
def test_non_isbn_identifiers_are_kept(self) -> None:
|
||||||
|
assert parse_identifier("urn:uuid:3f2b1c4e-1111-2222-3333-444455556666") == (
|
||||||
|
"uuid",
|
||||||
|
"3f2b1c4e-1111-2222-3333-444455556666",
|
||||||
|
)
|
||||||
|
assert parse_identifier("calibre:1234") == ("calibre", "1234")
|
||||||
|
assert parse_identifier("B000FC0PDA", "mobi-asin") == ("asin", "B000FC0PDA")
|
||||||
|
|
||||||
|
def test_an_unrecognised_prefix_is_part_of_the_value(self) -> None:
|
||||||
|
""" "http://example.com/book" is not an identifier called "http"."""
|
||||||
|
assert parse_identifier("http://www.gutenberg.org/5200") == (
|
||||||
|
"id",
|
||||||
|
"http://www.gutenberg.org/5200",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_a_declared_isbn_that_is_not_one_is_dropped(self) -> None:
|
||||||
|
assert parse_identifier("urn:isbn:not-an-isbn") is None
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("written", ["", " ", None])
|
||||||
|
def test_nothing_yields_nothing(self, written: str | None) -> None:
|
||||||
|
assert parse_identifier(written) is None
|
||||||
@@ -1,7 +1,13 @@
|
|||||||
import pytest
|
import pytest
|
||||||
|
from ebooklib import epub
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from datetime import date
|
from datetime import date
|
||||||
from chitai.services.metadata_extractor import EpubExtractor
|
from chitai.services.metadata_extractor import (
|
||||||
|
EpubExtractor,
|
||||||
|
Extractor,
|
||||||
|
PdfExtractor,
|
||||||
|
split_edition,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio()
|
@pytest.mark.asyncio()
|
||||||
@@ -15,3 +21,180 @@ class TestEpubExtractor:
|
|||||||
assert metadata["authors"] == ["Herman Melville"]
|
assert metadata["authors"] == ["Herman Melville"]
|
||||||
assert metadata["published_date"] == date(year=2001, month=7, day=1)
|
assert metadata["published_date"] == date(year=2001, month=7, day=1)
|
||||||
|
|
||||||
|
|
||||||
|
EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
|
||||||
|
PDF = Path("tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio()
|
||||||
|
class TestIdentifierMerging:
|
||||||
|
"""A book's formats each contribute identifiers; none of them replaces the rest."""
|
||||||
|
|
||||||
|
async def test_every_format_contributes(
|
||||||
|
self, monkeypatch: pytest.MonkeyPatch
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Identifiers are a collection, not a single value.
|
||||||
|
|
||||||
|
Merging the whole dict meant the last format to report won outright: an EPUB
|
||||||
|
declaring an ASIN, a Google volume id and a Calibre id kept none of them once
|
||||||
|
a PDF contributed one ISBN.
|
||||||
|
"""
|
||||||
|
|
||||||
|
async def epub(_file):
|
||||||
|
return {
|
||||||
|
"title": "How Linux Works",
|
||||||
|
"identifiers": {"isbn-13": "9781718500419", "asin": "1718500408"},
|
||||||
|
}
|
||||||
|
|
||||||
|
async def pdf(_file):
|
||||||
|
return {
|
||||||
|
"identifiers": {"isbn-13": "9781718500402", "isbn-10": "1593270356"}
|
||||||
|
}
|
||||||
|
|
||||||
|
monkeypatch.setattr(EpubExtractor, "extract_metadata", epub)
|
||||||
|
monkeypatch.setattr(PdfExtractor, "extract_metadata", pdf)
|
||||||
|
|
||||||
|
metadata = await Extractor.extract_metadata(
|
||||||
|
[Path("How Linux Works.epub"), Path("How Linux Works.pdf")]
|
||||||
|
)
|
||||||
|
|
||||||
|
assert metadata["identifiers"] == {
|
||||||
|
# Declared by the publisher's toolchain, so it outranks the PDF's, which
|
||||||
|
# was scraped off a copyright page that also prints the print edition's.
|
||||||
|
"isbn-13": "9781718500419",
|
||||||
|
"asin": "1718500408",
|
||||||
|
"isbn-10": "1593270356",
|
||||||
|
}
|
||||||
|
|
||||||
|
async def test_a_second_format_that_finds_nothing_erases_nothing(self) -> None:
|
||||||
|
"""The PDF fixture carries no ISBN, so it must leave the EPUB's alone."""
|
||||||
|
metadata = await Extractor.extract_metadata([EPUB, PDF])
|
||||||
|
|
||||||
|
assert metadata["identifiers"] == {"id": "http://www.gutenberg.org/5200"}
|
||||||
|
|
||||||
|
async def test_one_format_on_its_own_is_unaffected(self) -> None:
|
||||||
|
metadata = await Extractor.extract_metadata([EPUB])
|
||||||
|
|
||||||
|
assert metadata["identifiers"] == {"id": "http://www.gutenberg.org/5200"}
|
||||||
|
|
||||||
|
async def test_no_identifiers_anywhere_leaves_the_field_absent(self) -> None:
|
||||||
|
"""An empty dict would count as extracted metadata and overwrite nothing."""
|
||||||
|
metadata = await Extractor.extract_metadata([PDF])
|
||||||
|
|
||||||
|
assert "identifiers" not in metadata
|
||||||
|
|
||||||
|
|
||||||
|
class TestSplitEdition:
|
||||||
|
"""An edition is a field on the book, not part of what the book is called."""
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("title", "stripped", "edition"),
|
||||||
|
[
|
||||||
|
# Every form below is one that turned up in a real library.
|
||||||
|
("Fluent Python, 2nd Edition", "Fluent Python", 2),
|
||||||
|
("Building Microservices, 2E", "Building Microservices", 2),
|
||||||
|
("Digital Image Processing, 4e", "Digital Image Processing", 4),
|
||||||
|
(
|
||||||
|
"Network Security Essentials: Applications and Standards/6e",
|
||||||
|
"Network Security Essentials: Applications and Standards",
|
||||||
|
6,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Refactoring: Improving the Design of Existing Code (2nd edition)",
|
||||||
|
"Refactoring: Improving the Design of Existing Code",
|
||||||
|
2,
|
||||||
|
),
|
||||||
|
# Ordinal words, including a qualifier sitting inside the statement.
|
||||||
|
(
|
||||||
|
"The Art of Computer Programming: Volume 1 / Fundamental Algorithms, Third Edition",
|
||||||
|
"The Art of Computer Programming: Volume 1 / Fundamental Algorithms",
|
||||||
|
3,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Introduction to the Theory of Computation, Third International Edition",
|
||||||
|
"Introduction to the Theory of Computation",
|
||||||
|
3,
|
||||||
|
),
|
||||||
|
# Mid-title, before a subtitle and before a trailing author.
|
||||||
|
(
|
||||||
|
"How Linux Works, 3rd Edition: What Every Superuser Should Know",
|
||||||
|
"How Linux Works: What Every Superuser Should Know",
|
||||||
|
3,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"Code Complete, 2nd Edition - Steve McConnell",
|
||||||
|
"Code Complete - Steve McConnell",
|
||||||
|
2,
|
||||||
|
),
|
||||||
|
# An underscore between the number and the "e", beside an unnumbered
|
||||||
|
# qualifier that has nowhere to go in an integer column and so stays put.
|
||||||
|
(
|
||||||
|
"Cryptography and Network Security, Global Edition, 8_e - Stallings",
|
||||||
|
"Cryptography and Network Security, Global Edition - Stallings",
|
||||||
|
8,
|
||||||
|
),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_editions_are_split_out(
|
||||||
|
self, title: str, stripped: str, edition: int
|
||||||
|
) -> None:
|
||||||
|
assert split_edition(title) == (stripped, edition)
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"title",
|
||||||
|
[
|
||||||
|
# A number alone is never an edition — these are titles.
|
||||||
|
"Catch 22",
|
||||||
|
"Fahrenheit 451",
|
||||||
|
"Blade Runner 2049",
|
||||||
|
"1984",
|
||||||
|
"Apollo 13",
|
||||||
|
"Slaughterhouse 5",
|
||||||
|
"The Art of Computer Programming: Volume 1",
|
||||||
|
# "Edition" with no number cannot be stored, so it stays where it can
|
||||||
|
# still be read.
|
||||||
|
"Cryptography and Network Security: Principles and Practice, Global Edition",
|
||||||
|
"Building Microservices",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_titles_are_left_alone(self, title: str) -> None:
|
||||||
|
assert split_edition(title) == (title, None)
|
||||||
|
|
||||||
|
def test_a_title_that_is_only_an_edition_is_kept(self) -> None:
|
||||||
|
"""Stripping must never leave a book with no title at all."""
|
||||||
|
assert split_edition("2nd Edition") == ("2nd Edition", None)
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("title", ["", None])
|
||||||
|
def test_nothing_yields_nothing(self, title: str | None) -> None:
|
||||||
|
assert split_edition(title) == (title, None)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio()
|
||||||
|
class TestEditionFromFiles:
|
||||||
|
async def test_extraction_moves_the_edition_off_the_title(self) -> None:
|
||||||
|
"""The PDF fixture calls itself a 2nd edition in its own metadata title."""
|
||||||
|
metadata = await Extractor.extract_metadata([PDF])
|
||||||
|
|
||||||
|
assert (
|
||||||
|
metadata["title"]
|
||||||
|
== "The Project Gutenberg eBook #33283: Calculus Made Easy"
|
||||||
|
)
|
||||||
|
assert metadata["edition"] == 2
|
||||||
|
|
||||||
|
|
||||||
|
class TestEpubPublisher:
|
||||||
|
"""The publisher was looked up and then dropped on the floor."""
|
||||||
|
|
||||||
|
def test_a_declared_publisher_is_returned(self) -> None:
|
||||||
|
"""
|
||||||
|
The lookup discarded its own result and fell off the end of the function, so
|
||||||
|
every EPUB reported no publisher no matter what it said.
|
||||||
|
"""
|
||||||
|
book = epub.EpubBook()
|
||||||
|
book.add_metadata("DC", "publisher", "No Starch Press")
|
||||||
|
|
||||||
|
assert EpubExtractor._extract_publisher(book) == "No Starch Press"
|
||||||
|
|
||||||
|
def test_no_publisher_is_none(self) -> None:
|
||||||
|
assert EpubExtractor._extract_publisher(epub.EpubBook()) is None
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -2,19 +2,12 @@
|
|||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
|
from sqlalchemy import select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from chitai.services import ShelfService
|
from chitai.services import ShelfService
|
||||||
from chitai.database import models as m
|
from chitai.database import models as m
|
||||||
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
from sqlalchemy import select
|
|
||||||
|
|
||||||
from chitai.services.bookshelf import ShelfService
|
|
||||||
from chitai.services import BookService
|
|
||||||
from chitai.database.models.book_list import BookList, BookListLink
|
from chitai.database.models.book_list import BookList, BookListLink
|
||||||
from chitai.database import models as m
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
|
|||||||
@@ -0,0 +1,423 @@
|
|||||||
|
"""Tests for importing a Calibre library through BookService."""
|
||||||
|
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from chitai.config import settings
|
||||||
|
from chitai.database import models as m
|
||||||
|
from chitai.services import BookService
|
||||||
|
from chitai.services.calibre import CalibreLibrary
|
||||||
|
|
||||||
|
from tests.calibre_fixtures import CalibreFixture
|
||||||
|
|
||||||
|
|
||||||
|
DATA_FILES = Path("tests/data_files")
|
||||||
|
EPUB = DATA_FILES / "Metamorphosis - Franz Kafka.epub"
|
||||||
|
OTHER_EPUB = DATA_FILES / "The Art of War - Sun Tzu.epub"
|
||||||
|
PDF = DATA_FILES / "Calculus Made Easy - Silvanus Thompson.pdf"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(name="calibre_root")
|
||||||
|
def fx_calibre_root(tmp_path: Path) -> Path:
|
||||||
|
"""Three books: one plain, one in two formats, one Chitai cannot use."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_pages_table()
|
||||||
|
|
||||||
|
fixture.add_book(
|
||||||
|
1,
|
||||||
|
"The Metamorphosis",
|
||||||
|
authors=["Franz Kafka"],
|
||||||
|
pubdate="1915-10-15 00:00:00+00:00",
|
||||||
|
tags=["Fiction", "Absurdist"],
|
||||||
|
publisher="Kurt Wolff Verlag",
|
||||||
|
languages=["deu"],
|
||||||
|
comment="<p>He wakes up <i>changed</i>.</p>",
|
||||||
|
identifiers={"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"},
|
||||||
|
uuid="11111111-2222-3333-4444-555555555555",
|
||||||
|
pages=201,
|
||||||
|
cover=True,
|
||||||
|
formats={"EPUB": EPUB},
|
||||||
|
)
|
||||||
|
|
||||||
|
fixture.add_book(
|
||||||
|
2,
|
||||||
|
"The Art of War",
|
||||||
|
authors=["Sun Tzu"],
|
||||||
|
series="Classics",
|
||||||
|
series_index=3.0,
|
||||||
|
formats={"EPUB": OTHER_EPUB, "PDF": PDF},
|
||||||
|
)
|
||||||
|
|
||||||
|
fixture.add_book(3, "Metadata Only", authors=["Nobody"])
|
||||||
|
|
||||||
|
return fixture.commit()
|
||||||
|
|
||||||
|
|
||||||
|
async def library_of(root: Path) -> CalibreLibrary:
|
||||||
|
source = CalibreLibrary(root)
|
||||||
|
await source.open()
|
||||||
|
return source
|
||||||
|
|
||||||
|
|
||||||
|
async def test_imports_a_catalogue(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert result.total == 3
|
||||||
|
assert len(result.created) == 2
|
||||||
|
|
||||||
|
# The book with no files is left out: a record with nothing to read, and a directory
|
||||||
|
# to match, is worse than not importing it.
|
||||||
|
assert [skipped.calibre_id for skipped in result.skipped] == [3]
|
||||||
|
assert result.skipped[0].reason == "no files in the catalogue"
|
||||||
|
assert result.failed == []
|
||||||
|
|
||||||
|
book = await books_service.get(result.created[0])
|
||||||
|
|
||||||
|
assert book.title == "The Metamorphosis"
|
||||||
|
assert [author.name for author in book.authors] == ["Franz Kafka"]
|
||||||
|
assert sorted(tag.name for tag in book.tags) == ["Absurdist", "Fiction"]
|
||||||
|
assert book.publisher is not None and book.publisher.name == "Kurt Wolff Verlag"
|
||||||
|
assert book.published_date is not None and book.published_date.year == 1915
|
||||||
|
assert book.language == "deu"
|
||||||
|
assert book.pages == 201
|
||||||
|
assert book.cover_image is not None
|
||||||
|
|
||||||
|
# The HTML is gone; `Book.description` is rendered as text.
|
||||||
|
assert book.description == "He wakes up changed."
|
||||||
|
|
||||||
|
|
||||||
|
async def test_two_formats_are_one_book(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
book = await books_service.get(result.created[1])
|
||||||
|
|
||||||
|
assert book.title == "The Art of War"
|
||||||
|
assert sorted(Path(file.path).suffix for file in book.files) == [".epub", ".pdf"]
|
||||||
|
|
||||||
|
# A REAL series index reaches the column as the string everything else writes.
|
||||||
|
assert book.series is not None and book.series.title == "Classics"
|
||||||
|
assert book.series_position == "3"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_identifiers_are_folded_onto_chitai_schemes(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
`amazon` becomes `asin`, a hyphenated ISBN survives, and the Calibre uuid is kept.
|
||||||
|
|
||||||
|
The uuid is deliberately not stored under `uuid`, which duplicate matching ignores
|
||||||
|
because an EPUB regenerates one per build. Calibre's is stable, so it is the durable
|
||||||
|
link back to the row it came from.
|
||||||
|
"""
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
book = await books_service.get(result.created[0])
|
||||||
|
identifiers = {identifier.name: identifier.value for identifier in book.identifiers}
|
||||||
|
|
||||||
|
assert identifiers["asin"] == "B01N5IB20Q"
|
||||||
|
assert identifiers["isbn-13"] == "9780486290300"
|
||||||
|
assert identifiers["calibre-uuid"] == "11111111-2222-3333-4444-555555555555"
|
||||||
|
|
||||||
|
matching = {
|
||||||
|
identifier.name: identifier.normalized_value for identifier in book.identifiers
|
||||||
|
}
|
||||||
|
|
||||||
|
# Stored under its own name, matched under one scheme for both ISBN forms.
|
||||||
|
assert matching["isbn-13"] == "isbn:9780486290300"
|
||||||
|
|
||||||
|
# And the uuid carries a real matching key, which is the whole reason it is not
|
||||||
|
# filed under `uuid`.
|
||||||
|
assert matching["calibre-uuid"] is not None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_the_source_library_is_left_alone(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
"""Files are copied. Moving them would leave `metadata.db` pointing at nothing."""
|
||||||
|
before = {
|
||||||
|
path: path.stat().st_mtime_ns
|
||||||
|
for path in sorted(calibre_root.rglob("*"))
|
||||||
|
if path.is_file()
|
||||||
|
}
|
||||||
|
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
after = {
|
||||||
|
path: path.stat().st_mtime_ns
|
||||||
|
for path in sorted(calibre_root.rglob("*"))
|
||||||
|
if path.is_file()
|
||||||
|
}
|
||||||
|
|
||||||
|
assert after == before
|
||||||
|
|
||||||
|
# And the copies are really there, under the library's own layout.
|
||||||
|
for book_id in result.created:
|
||||||
|
book = await books_service.get(book_id)
|
||||||
|
for file in book.files:
|
||||||
|
assert (Path(book.path or "") / file.path).is_file()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_importing_twice_creates_nothing(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Re-running is safe with no bookkeeping: the bytes are recognised wherever they sit.
|
||||||
|
|
||||||
|
This is what makes an interrupted import resumable by simply running it again.
|
||||||
|
"""
|
||||||
|
for _ in range(2):
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert result.created == []
|
||||||
|
assert sorted(skipped.reason for skipped in result.skipped) == [
|
||||||
|
"already stored",
|
||||||
|
"already stored",
|
||||||
|
"no files in the catalogue",
|
||||||
|
]
|
||||||
|
|
||||||
|
held_by = [
|
||||||
|
skipped.book_id
|
||||||
|
for skipped in result.skipped
|
||||||
|
if skipped.reason == "already stored"
|
||||||
|
]
|
||||||
|
assert all(book_id is not None for book_id in held_by)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_file_the_catalogue_lists_but_disk_does_not(
|
||||||
|
books_service: BookService, test_library: m.Library, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""Calibre keeps the row when a file is moved away behind its back."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_book(1, "Present", authors=["A"], formats={"EPUB": EPUB})
|
||||||
|
fixture.add_book(2, "Absent", authors=["B"])
|
||||||
|
fixture.add_missing_format(2, "EPUB", "Absent - B")
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
source = await library_of(root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert len(result.created) == 1
|
||||||
|
assert [(s.calibre_id, s.reason) for s in result.skipped] == [
|
||||||
|
(2, "no files on disk")
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_one_broken_book_does_not_stop_the_import(
|
||||||
|
books_service: BookService, test_library: m.Library, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
A failure is recorded and the run continues, leaving no files behind for it.
|
||||||
|
|
||||||
|
An orphaned directory would make the next attempt reserve `title (2)` and look as
|
||||||
|
though it had worked.
|
||||||
|
"""
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_book(1, "First", authors=["A"], formats={"EPUB": EPUB})
|
||||||
|
fixture.add_book(2, "Doomed", authors=["B"], formats={"EPUB": OTHER_EPUB})
|
||||||
|
fixture.add_book(3, "Third", authors=["C"], formats={"PDF": PDF})
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
original = books_service.create
|
||||||
|
|
||||||
|
async def fail_on_the_second(data, *args, **kwargs):
|
||||||
|
if isinstance(data, dict) and data.get("title") == "Doomed":
|
||||||
|
raise RuntimeError("no room on the shelf")
|
||||||
|
return await original(data, *args, **kwargs)
|
||||||
|
|
||||||
|
books_service.create = fail_on_the_second # type: ignore[method-assign]
|
||||||
|
|
||||||
|
source = await library_of(root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
books_service.create = original # type: ignore[method-assign]
|
||||||
|
|
||||||
|
assert len(result.created) == 2
|
||||||
|
assert len(result.failed) == 1
|
||||||
|
assert result.failed[0].calibre_id == 2
|
||||||
|
assert "no room on the shelf" in result.failed[0].reason
|
||||||
|
|
||||||
|
# Nothing of the failed book was left in the library. Checked against the path the
|
||||||
|
# template would have produced, rather than by walking the root — the Calibre source
|
||||||
|
# sits under it in these tests, and its own files are meant to still be there.
|
||||||
|
assert not (Path(test_library.root_path) / "B").exists()
|
||||||
|
|
||||||
|
# And the books either side of it are where they should be.
|
||||||
|
for book_id in result.created:
|
||||||
|
book = await books_service.get(book_id)
|
||||||
|
for file in book.files:
|
||||||
|
assert (Path(book.path or "") / file.path).is_file()
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_cover_that_cannot_be_read_is_not_fatal(
|
||||||
|
books_service: BookService, test_library: m.Library, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
A truncated `cover.jpg` costs the cover, not the book.
|
||||||
|
|
||||||
|
Real libraries hold them, from an interrupted download or a failed conversion, and
|
||||||
|
the cover is the one thing in the directory that can be replaced from the book page.
|
||||||
|
"""
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_book(
|
||||||
|
1, "Unreadable Cover", authors=["A"], corrupt_cover=True, formats={"EPUB": EPUB}
|
||||||
|
)
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
source = await library_of(root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert result.failed == []
|
||||||
|
assert len(result.created) == 1
|
||||||
|
|
||||||
|
book = await books_service.get(result.created[0])
|
||||||
|
assert book.cover_image is None
|
||||||
|
assert len(book.files) == 1
|
||||||
|
|
||||||
|
|
||||||
|
async def test_shared_authors_and_tags_are_one_row_each(
|
||||||
|
books_service: BookService, test_library: m.Library, tmp_path: Path, session
|
||||||
|
) -> None:
|
||||||
|
"""Two books by one author must not produce two `Author` rows."""
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_book(
|
||||||
|
1, "One", authors=["Franz Kafka"], tags=["Fiction"], formats={"EPUB": EPUB}
|
||||||
|
)
|
||||||
|
fixture.add_book(
|
||||||
|
2,
|
||||||
|
"Two",
|
||||||
|
authors=["Franz Kafka"],
|
||||||
|
tags=["Fiction"],
|
||||||
|
formats={"EPUB": OTHER_EPUB},
|
||||||
|
)
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
source = await library_of(root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert len(result.created) == 2
|
||||||
|
|
||||||
|
first, second = [await books_service.get(book_id) for book_id in result.created]
|
||||||
|
|
||||||
|
assert first.authors[0].id == second.authors[0].id
|
||||||
|
assert first.tags[0].id == second.tags[0].id
|
||||||
|
|
||||||
|
|
||||||
|
async def test_a_second_copy_is_reported_not_refused(
|
||||||
|
books_service: BookService, test_library: m.Library, tmp_path: Path
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Two catalogue rows for one book, with different bytes, both import.
|
||||||
|
|
||||||
|
File-level dedupe cannot see it — the archives differ — so book-level detection
|
||||||
|
reports the pair and leaves the decision to the reader.
|
||||||
|
"""
|
||||||
|
padded = tmp_path / "padded.epub"
|
||||||
|
padded.write_bytes(EPUB.read_bytes() + b"\0" * 64)
|
||||||
|
|
||||||
|
fixture = CalibreFixture(tmp_path / "source")
|
||||||
|
fixture.add_book(
|
||||||
|
1, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": EPUB}
|
||||||
|
)
|
||||||
|
fixture.add_book(
|
||||||
|
2, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": padded}
|
||||||
|
)
|
||||||
|
root = fixture.commit()
|
||||||
|
|
||||||
|
source = await library_of(root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert len(result.created) == 2
|
||||||
|
assert len(result.possible_duplicates) == 1
|
||||||
|
assert result.possible_duplicates[0].candidates[0].book_id == result.created[0]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_allow_duplicates_stores_the_same_bytes_again(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
for allow in (False, True):
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(
|
||||||
|
source, test_library, allow_duplicates=allow
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert len(result.created) == 2
|
||||||
|
|
||||||
|
|
||||||
|
async def test_duplicate_scope_off_imports_everything(
|
||||||
|
books_service: BookService,
|
||||||
|
test_library: m.Library,
|
||||||
|
calibre_root: Path,
|
||||||
|
monkeypatch: pytest.MonkeyPatch,
|
||||||
|
) -> None:
|
||||||
|
monkeypatch.setattr(settings, "duplicate_scope", "off")
|
||||||
|
|
||||||
|
for _ in range(2):
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
result = await books_service.create_many_from_calibre(source, test_library)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert len(result.created) == 2
|
||||||
|
|
||||||
|
|
||||||
|
async def test_progress_is_reported_per_book(
|
||||||
|
books_service: BookService, test_library: m.Library, calibre_root: Path
|
||||||
|
) -> None:
|
||||||
|
"""The import is long enough that its progress is the only thing worth watching."""
|
||||||
|
seen = []
|
||||||
|
|
||||||
|
source = await library_of(calibre_root)
|
||||||
|
try:
|
||||||
|
await books_service.create_many_from_calibre(
|
||||||
|
source, test_library, on_progress=seen.append
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
await source.close()
|
||||||
|
|
||||||
|
assert [progress.processed for progress in seen] == [1, 2, 3]
|
||||||
|
assert all(progress.total == 3 for progress in seen)
|
||||||
|
assert [progress.outcome for progress in seen] == ["created", "created", "skipped"]
|
||||||
|
assert seen[0].title == "The Metamorphosis"
|
||||||
@@ -34,8 +34,8 @@ class TestLibraryServiceCRUD:
|
|||||||
assert library.name == "Test Library"
|
assert library.name == "Test Library"
|
||||||
assert library.root_path == library_path
|
assert library.root_path == library_path
|
||||||
assert library.path_template == "{author}/{title}"
|
assert library.path_template == "{author}/{title}"
|
||||||
assert library.description == None
|
assert library.description is None
|
||||||
assert library.read_only == False
|
assert library.read_only is False
|
||||||
|
|
||||||
# Check if directory was created
|
# Check if directory was created
|
||||||
assert Path(library.root_path).is_dir()
|
assert Path(library.root_path).is_dir()
|
||||||
@@ -56,8 +56,8 @@ class TestLibraryServiceCRUD:
|
|||||||
read_only=False,
|
read_only=False,
|
||||||
)
|
)
|
||||||
|
|
||||||
with pytest.raises(PermissionError) as exc_info:
|
with pytest.raises(PermissionError):
|
||||||
library = await library_service.create(library_data)
|
await library_service.create(library_data)
|
||||||
|
|
||||||
# Check if directory was created
|
# Check if directory was created
|
||||||
assert not Path(library_path).exists()
|
assert not Path(library_path).exists()
|
||||||
@@ -86,8 +86,8 @@ class TestLibraryServiceCRUD:
|
|||||||
assert library.name == "Test Library"
|
assert library.name == "Test Library"
|
||||||
assert library.root_path == library_path
|
assert library.root_path == library_path
|
||||||
assert library.path_template == "{author}/{title}"
|
assert library.path_template == "{author}/{title}"
|
||||||
assert library.description == None
|
assert library.description is None
|
||||||
assert library.read_only == True
|
assert library.read_only is True
|
||||||
|
|
||||||
async def test_create_library_read_only_nonexistent_path(
|
async def test_create_library_read_only_nonexistent_path(
|
||||||
self, library_service: LibraryService, tmp_path: Path
|
self, library_service: LibraryService, tmp_path: Path
|
||||||
@@ -138,7 +138,7 @@ class TestLibraryServiceCRUD:
|
|||||||
assert library.root_path == "./books"
|
assert library.root_path == "./books"
|
||||||
assert library.path_template == "{author}/{title}"
|
assert library.path_template == "{author}/{title}"
|
||||||
assert library.description is None
|
assert library.description is None
|
||||||
assert library.read_only == False
|
assert library.read_only is False
|
||||||
|
|
||||||
# async def test_delete_library_keep_files(
|
# async def test_delete_library_keep_files(
|
||||||
# self, session: AsyncSession, library_service: LibraryService
|
# self, session: AsyncSession, library_service: LibraryService
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ class TestUserServiceAuthentication:
|
|||||||
|
|
||||||
# Create a user with a known password
|
# Create a user with a known password
|
||||||
password = "password123"
|
password = "password123"
|
||||||
user = m.User(email=f"test@example.com", password=password)
|
user = m.User(email="test@example.com", password=password)
|
||||||
|
|
||||||
session.add(user)
|
session.add(user)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
@@ -52,7 +52,7 @@ class TestUserServiceAuthentication:
|
|||||||
|
|
||||||
# Create user
|
# Create user
|
||||||
password = "password123"
|
password = "password123"
|
||||||
user = m.User(email=f"test@example.com", password=password)
|
user = m.User(email="test@example.com", password=password)
|
||||||
|
|
||||||
session.add(user)
|
session.add(user)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
@@ -85,7 +85,7 @@ class TestUserServiceCRUD:
|
|||||||
) -> None:
|
) -> None:
|
||||||
"""Test getting user by email."""
|
"""Test getting user by email."""
|
||||||
|
|
||||||
user = m.User(email=f"test@example.com", password="password123")
|
user = m.User(email="test@example.com", password="password123")
|
||||||
|
|
||||||
session.add(user)
|
session.add(user)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
@@ -102,12 +102,12 @@ class TestUserServiceCRUD:
|
|||||||
"""Test creating a new user with a duplicate email."""
|
"""Test creating a new user with a duplicate email."""
|
||||||
|
|
||||||
# Create first user
|
# Create first user
|
||||||
user = m.User(email=f"test@example.com", password="password123")
|
user = m.User(email="test@example.com", password="password123")
|
||||||
session.add(user)
|
session.add(user)
|
||||||
await session.commit()
|
await session.commit()
|
||||||
|
|
||||||
# Create second user
|
# Create second user
|
||||||
user = m.User(email=f"test@example.com", password="password12345")
|
user = m.User(email="test@example.com", password="password12345")
|
||||||
|
|
||||||
with pytest.raises(IntegrityError) as exc_info:
|
with pytest.raises(IntegrityError) as exc_info:
|
||||||
session.add(user)
|
session.add(user)
|
||||||
|
|||||||
Generated
+27
@@ -261,6 +261,7 @@ dev = [
|
|||||||
{ name = "pytest" },
|
{ name = "pytest" },
|
||||||
{ name = "pytest-asyncio" },
|
{ name = "pytest-asyncio" },
|
||||||
{ name = "pytest-databases", extra = ["postgres"] },
|
{ name = "pytest-databases", extra = ["postgres"] },
|
||||||
|
{ name = "ruff" },
|
||||||
]
|
]
|
||||||
|
|
||||||
[package.metadata]
|
[package.metadata]
|
||||||
@@ -286,6 +287,7 @@ dev = [
|
|||||||
{ name = "pytest", specifier = ">=8.4.2" },
|
{ name = "pytest", specifier = ">=8.4.2" },
|
||||||
{ name = "pytest-asyncio", specifier = ">=1.2.0" },
|
{ name = "pytest-asyncio", specifier = ">=1.2.0" },
|
||||||
{ name = "pytest-databases", extras = ["postgres"], specifier = ">=0.15.0" },
|
{ name = "pytest-databases", extras = ["postgres"], specifier = ">=0.15.0" },
|
||||||
|
{ name = "ruff", specifier = "==0.15.14" },
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -1277,6 +1279,31 @@ wheels = [
|
|||||||
{ url = "https://files.pythonhosted.org/packages/ca/e5/d708d262b600a352abe01c2ae360d8ff75b0af819b78e9af293191d928e6/rich_click-1.9.7-py3-none-any.whl", hash = "sha256:2f99120fca78f536e07b114d3b60333bc4bb2a0969053b1250869bcdc1b5351b", size = 71491, upload-time = "2026-01-31T04:29:26.777Z" },
|
{ url = "https://files.pythonhosted.org/packages/ca/e5/d708d262b600a352abe01c2ae360d8ff75b0af819b78e9af293191d928e6/rich_click-1.9.7-py3-none-any.whl", hash = "sha256:2f99120fca78f536e07b114d3b60333bc4bb2a0969053b1250869bcdc1b5351b", size = 71491, upload-time = "2026-01-31T04:29:26.777Z" },
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "ruff"
|
||||||
|
version = "0.15.14"
|
||||||
|
source = { registry = "https://pypi.org/simple" }
|
||||||
|
sdist = { url = "https://files.pythonhosted.org/packages/dc/8a/8bce2894573e9dae6ff4d77fe34ad727d79b9e6238ad288c5638990d90f6/ruff-0.15.14.tar.gz", hash = "sha256:48e866b165be4a9bdbf310f7d3c9a07edef2fe8cd63ffeb4e00bb590506ebf9f", size = 4700910, upload-time = "2026-05-21T14:34:55.177Z" }
|
||||||
|
wheels = [
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/b9/c8/74a92c6ff9fcfb4f1f947126d3ebee8389276e161ecc85de5bda7cda51bd/ruff-0.15.14-py3-none-linux_armv6l.whl", hash = "sha256:8dd2db9416e487c8d4b01fa7056bb02c4d05969d4f8d17a08c229c2f4ff3c108", size = 10739177, upload-time = "2026-05-21T14:34:37.332Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/45/91/254a35c20acc38a7223c9d2d594af12e794432464f2cdeb52af1dc4a892d/ruff-0.15.14-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:be4ff55af755bd71a00ab3dc6bd7ffc467bd76e0df6881e286c2e3d23e8fb43b", size = 11144969, upload-time = "2026-05-21T14:34:43.978Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/56/9e/d13e40f83b8d0a94430e6778ce1d94a43b38cf2efe63278bdd2b4c65abbf/ruff-0.15.14-py3-none-macosx_11_0_arm64.whl", hash = "sha256:48d5909d7d06276ce7dde6d32bfa4b0d4cb2651145cd8ee4b440722cbc77832f", size = 10478207, upload-time = "2026-05-21T14:34:48.378Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/8d/f1/b15a7839fa4f332f8acec78e20564f26bb2d866e3d21710b877fd0263000/ruff-0.15.14-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ca8cbfa94c4f90984a67561978602746d4cd27103568f745fa90eee3f0d4107d", size = 10818459, upload-time = "2026-05-21T14:34:22.318Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/45/33/53d651177f84f94b400a0e27f8824eeada3dddc9d5ee8aeb048f4352a520/ruff-0.15.14-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:9a6bbc0333f1ab053423bcbf6226477d266ca7cec7738c4c8e3f55647803f3c4", size = 10541800, upload-time = "2026-05-21T14:34:20.209Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/b8/a6/868f87e0bf9786ed24b5d0d0ad8676b8a94fd1912f42cddf9cfc7857818a/ruff-0.15.14-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:8a24a4f7605d7003a6674d4387651effd939dead3fddd0f36561eb77a9a2e542", size = 11342149, upload-time = "2026-05-21T14:34:46.365Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/a7/8b/38cd5c19faffdcc05a408d2b78edccc69492ab9720eadb49ea15ef80d768/ruff-0.15.14-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:049b5326e53ed80978f2fc041a280603f69dd6b0c95464342a2bb4572d9d9e2f", size = 12212563, upload-time = "2026-05-21T14:34:28.579Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/3e/4d/a3c5b874a556d5731e3e657aaf04311bb76f0a5c3ec220ed43051be6b64b/ruff-0.15.14-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:d4ed42e6696c8dfa5f06728e6441993901f548eb92d73bc472cb5a38d1395fbf", size = 11493299, upload-time = "2026-05-21T14:34:41.836Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/1e/c0/56472c251d09858a53e51efbd485b09e1995d8731668b76d52e5dd6ee0f1/ruff-0.15.14-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:715c543cf450c4888251f91c52f1942a800541d9bddd7ac060aa4e6b77ae7cba", size = 11455931, upload-time = "2026-05-21T14:34:57.276Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/2c/4a/e2e7b4d8dbf233d4eace59c75bc3435fa6d8bd3bae82d351d4e4300c0fd1/ruff-0.15.14-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:72ebab6013ec887d439d8b7593737a0a4ffb06d45d209d4e4bf2e92813082d3f", size = 11400794, upload-time = "2026-05-21T14:34:39.773Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/97/c7/83c0539fe34c3e09136204d1e75d6052492364e0b3cb05e9465423f567d7/ruff-0.15.14-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:49072d36abdbe97a8dd7f480afe9c675699c0c495d4c84076e2c1203c4550581", size = 10804759, upload-time = "2026-05-21T14:34:31.045Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/86/a6/18f2bfc095a2ab4a78745644e428205532ce6653a5d0fa8501572891534d/ruff-0.15.14-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:958522aee105068640c2c2ceae08f413ae44d922f52a1374ac13d6a96032fc93", size = 10539517, upload-time = "2026-05-21T14:34:53.064Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/54/3a/5a8b3b69c654d4e4bf1d246ac5b49cbcdac6eaab6905925f8915f31e3b80/ruff-0.15.14-py3-none-musllinux_1_2_i686.whl", hash = "sha256:f3707da619a143a2e8830e2abab8224478d69ace2d28cb6c20543ae97c36bf61", size = 11065169, upload-time = "2026-05-21T14:34:24.484Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/ed/c5/8864e4e7925b836ea354b31d57641ec03830564e281a8b6f061f8c3e0ec1/ruff-0.15.14-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:bb01d645694e3ec0102105d07ef2d53703970407d59c04e59d3ba0b7a1d53553", size = 11560214, upload-time = "2026-05-21T14:34:50.975Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/36/38/012bf76752e1f89ed50b77b99532d90f3a3e287bc7918e1fc0948ac866ac/ruff-0.15.14-py3-none-win32.whl", hash = "sha256:6d0c1ad2a0ab718d39b6d8fd2217981ce4d625cd96a720095f798fb47d8b13e6", size = 10805548, upload-time = "2026-05-21T14:34:33.453Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/d1/b7/4ea2c170f10ad760fff2a5250beb18897719dc8b52b53a24cddbb9dd3f19/ruff-0.15.14-py3-none-win_amd64.whl", hash = "sha256:802342981e056db3851a7836e5b070f8f15f67d4a685ae2a6160939d364b2902", size = 11939523, upload-time = "2026-05-21T14:34:18.077Z" },
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/62/d5/bc97ff895ec35cf3925d4bd60f3b39d822f377a446906ec9bcc87405e59b/ruff-0.15.14-py3-none-win_arm64.whl", hash = "sha256:ff47b90a9ef6a40c9e2f3b479c1fb78531adf055b94c1eba0a7ba04b31951826", size = 11208607, upload-time = "2026-05-21T14:34:26.525Z" },
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "six"
|
name = "six"
|
||||||
version = "1.17.0"
|
version = "1.17.0"
|
||||||
|
|||||||
@@ -24,6 +24,10 @@ services:
|
|||||||
|
|
||||||
|
|
||||||
backend:
|
backend:
|
||||||
|
image: git.jaroszew.ski/patrick/chitai-backend:${CHITAI_VERSION:-latest}
|
||||||
|
|
||||||
|
# Only used when the image above is not present locally. Run `docker compose build` to
|
||||||
|
# build from source instead of pulling a release.
|
||||||
build: ./backend
|
build: ./backend
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
@@ -53,6 +57,8 @@ services:
|
|||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
|
||||||
frontend:
|
frontend:
|
||||||
|
image: git.jaroszew.ski/patrick/chitai-frontend:${CHITAI_VERSION:-latest}
|
||||||
|
|
||||||
build: ./frontend
|
build: ./frontend
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
@@ -66,6 +72,9 @@ services:
|
|||||||
environment:
|
environment:
|
||||||
VITE_BACKEND_API_URL: ${CHITAI_API_URL}
|
VITE_BACKEND_API_URL: ${CHITAI_API_URL}
|
||||||
|
|
||||||
|
# Must match the URL the browser uses, or SvelteKit rejects form POSTs as cross-origin.
|
||||||
|
ORIGIN: ${CHITAI_ORIGIN:-http://localhost:3000}
|
||||||
|
|
||||||
depends_on:
|
depends_on:
|
||||||
backend:
|
backend:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
|||||||
@@ -0,0 +1,407 @@
|
|||||||
|
# Implementation brief: importing a Calibre library
|
||||||
|
|
||||||
|
Written for an agent picking this up cold. Read the repo-root `AGENTS.md` and
|
||||||
|
`backend/AGENTS.md` first — this brief assumes both, particularly the **Filesystem behaviour**
|
||||||
|
and **Duplicate detection** sections.
|
||||||
|
|
||||||
|
Every claim about Calibre's schema and on-disk layout below was checked against a real library at
|
||||||
|
`~/Documents/Calibre Library` (6 books, current Calibre). Where a fact came from Calibre's source
|
||||||
|
rather than that library, it says so.
|
||||||
|
|
||||||
|
## Feasibility: high, and most of the machinery already exists
|
||||||
|
|
||||||
|
`metadata.db` is plain SQLite with a schema that has been stable for a decade, and the files sit
|
||||||
|
beside it in a predictable tree. Chitai already has every piece an import needs:
|
||||||
|
|
||||||
|
| Needed | Already in the tree |
|
||||||
|
| --- | --- |
|
||||||
|
| Ingest files that are already on disk | `BookService.create_many_from_existing_files` (`services/book.py:1280`) |
|
||||||
|
| Deduplicate authors/tags/publishers/series | `_populate_with_unique_relationships` (`services/book.py:1752`), via `as_unique_async` |
|
||||||
|
| Generic identifiers with a scheme map | `Identifier`, and `parse_identifier` (`services/metadata_extractor.py:58`) — whose map already covers `isbn`, `amazon`, `mobi-asin`, `google`, `goodreads`, `doi`, `calibre` |
|
||||||
|
| Decide where a book lives on disk | `BookPathGenerator`, `_reserve_book_path` (`services/book.py:1066`) |
|
||||||
|
| Not import the same book twice | `find_duplicate_files` (`:461`) and `find_duplicate_books` (`:535`) |
|
||||||
|
| Store a cover | `_save_cover_image` (`:1994`) |
|
||||||
|
|
||||||
|
So this is **a reader, not new ingest machinery**: turn Calibre rows into the metadata dict
|
||||||
|
`BookService` already accepts, and hand it to a slightly generalised version of the consume-directory
|
||||||
|
path. The parsing is the easy half.
|
||||||
|
|
||||||
|
The hard parts are elsewhere, and all three are addressed below:
|
||||||
|
|
||||||
|
1. Calibre libraries hold formats Chitai cannot describe, let alone read — and one of them
|
||||||
|
**currently 500s the book detail endpoint** (see prerequisites).
|
||||||
|
2. A 5,000-book import is a long-running job, and the app has no job/progress concept.
|
||||||
|
3. Whether files are **copied** into the library or **referenced in place** — which is a
|
||||||
|
product decision with a large blast radius, because in-place means Chitai's write paths point
|
||||||
|
at somebody's Calibre library.
|
||||||
|
|
||||||
|
## What a Calibre library actually is
|
||||||
|
|
||||||
|
```
|
||||||
|
Calibre Library/
|
||||||
|
├── metadata.db the whole catalogue
|
||||||
|
├── metadata_db_prefs_backup.json ignore
|
||||||
|
├── .caltrash/ .calnotes/ ignore — deleted books still live in .caltrash
|
||||||
|
└── <Author Name>/
|
||||||
|
└── <Title> (<book id>)/ == books.path
|
||||||
|
├── cover.jpg iff books.has_cover
|
||||||
|
├── metadata.opf ignore; the db is authoritative
|
||||||
|
└── <data.name>.<format> one per row in `data`
|
||||||
|
```
|
||||||
|
|
||||||
|
The tables that matter, and nothing else: `books`, `authors` + `books_authors_link`,
|
||||||
|
`publishers` + `books_publishers_link`, `tags` + `books_tags_link`, `series` +
|
||||||
|
`books_series_link`, `languages` + `books_languages_link`, `comments`, `identifiers`, `data`,
|
||||||
|
`books_pages_link`, `last_read_positions`.
|
||||||
|
|
||||||
|
Ten things that will produce wrong data if you do not know them:
|
||||||
|
|
||||||
|
- **Never query the views.** `meta`, `tag_browser_*` and friends call SQLite functions Calibre
|
||||||
|
registers from Python at connection time. Verified: `SELECT * FROM meta` fails with
|
||||||
|
`no such function: sortconcat`. Query base tables only.
|
||||||
|
|
||||||
|
- **`pubdate` has a sentinel, not a null.** An unknown publication date is stored as
|
||||||
|
`0101-01-01 00:00:00+00:00` (Calibre's `UNDEFINED_DATE`, year 101). It parses fine as a
|
||||||
|
`date`, so nothing will complain — two of the six books in the reference library carry it. Drop
|
||||||
|
any `pubdate` with year < 1000. The same sentinel appears in `timestamp`.
|
||||||
|
|
||||||
|
- **`data.name` is lossy and is not the title.** It is the on-disk stem, truncated to Calibre's
|
||||||
|
filename limit and sanitised. Verified in the reference library: the book titled
|
||||||
|
`The Project Gutenberg eBook #33283: Calculus Made Easy, 2nd Edition` is stored as
|
||||||
|
`The Project Gutenberg eBook #33283_ Calcul - Silvanus Phillips Thompson.pdf`. So the file
|
||||||
|
extractors must not be consulted for metadata (see decision 2), and `books.path`/`data.name`
|
||||||
|
are for *locating* files only.
|
||||||
|
|
||||||
|
- **`books.title` may contain characters Calibre strips from its own paths** — `:` became `_`
|
||||||
|
above, and titles legitimately contain `/` (`AC/DC`). `BookPathGenerator` interpolates the
|
||||||
|
title straight into a path and only collapses repeated slashes
|
||||||
|
(`services/filesystem_library.py`), so an unsanitised Calibre title can silently add a
|
||||||
|
directory level. Sanitise `/` and control characters out of `title` before path generation.
|
||||||
|
|
||||||
|
- **`authors.name` escapes commas as `|`.** Calibre's `AuthorsTable` unserialises with
|
||||||
|
`name.replace('|', ',')` (from Calibre's `db/tables.py`; the reference library has no such
|
||||||
|
name, so this one is unverified locally). Do the same replacement, and pass `authors.name` —
|
||||||
|
**not** `authors.sort`, which is `Melville, Herman`. `format_author_name` would flip the sort
|
||||||
|
form correctly anyway, but there is no reason to hand it the worse input.
|
||||||
|
|
||||||
|
- **`series_index` is a REAL.** `7.0` must become `"7"`, not `"7.0"` — `Book.series_position` is
|
||||||
|
a string, and `find_duplicate_books`'s series-position disqualifier compares it as one.
|
||||||
|
|
||||||
|
- **`languages.lang_code` is ISO 639-2/B** (`eng`), while `EpubExtractor` stores raw
|
||||||
|
`DC:language` (`en`). Both will coexist in the column. `Book.language` is free text and the
|
||||||
|
edit form is a plain `<input>`, so nothing breaks; normalising to two letters is optional
|
||||||
|
polish, not part of this work.
|
||||||
|
|
||||||
|
- **`comments.text` is HTML.** `Book.description` is rendered as plain text by
|
||||||
|
`CollapsibleText`, so `<p>` tags will show literally. Strip to text on import.
|
||||||
|
|
||||||
|
- **`books_pages_link` is usually empty of real data.** It carries `needs_scan` and, in the
|
||||||
|
reference library, `pages = 0` for all six books. Only use it when `pages > 0`.
|
||||||
|
|
||||||
|
- **`identifiers.type` is free text.** The reference library holds `isbn`, `amazon` and
|
||||||
|
`mobi-asin`, all of which `parse_identifier` already maps. Feed every identifier through it and
|
||||||
|
keep whatever survives; do not filter to a known list.
|
||||||
|
|
||||||
|
## Decisions to settle before writing code
|
||||||
|
|
||||||
|
1. **Copy files into the library. Do not move, do not reference in place** — for the first
|
||||||
|
version. Moving leaves `metadata.db` pointing at files that are gone, which quietly destroys a
|
||||||
|
library the user still uses. Referencing in place is genuinely desirable (nobody wants two
|
||||||
|
copies of 80 GB) but it points `book.path` at the Calibre tree, and `update_book` **moves
|
||||||
|
directories** while `delete_books` **deletes files** — so a metadata edit in Chitai would
|
||||||
|
rearrange somebody's Calibre library. `Library.read_only` exists but is enforced in exactly one
|
||||||
|
place (`services/library.py:54`, at creation). See phase 3.
|
||||||
|
|
||||||
|
2. **Trust Calibre's metadata; do not run the extractors.** Calibre's catalogue is curated, its
|
||||||
|
filenames are truncated garbage, and running `Extractor.extract_metadata` over thousands of
|
||||||
|
files means opening every EPUB and rendering a cover page from every PDF. Take the cover from
|
||||||
|
`cover.jpg` directly. The one exception worth allowing: fill `pages` from the file when
|
||||||
|
Calibre has no useful value, behind a flag, off by default.
|
||||||
|
|
||||||
|
3. ~~**The source is a server-side path, not an upload.**~~ **Reversed in review, and the reason
|
||||||
|
this brief was wrong is worth keeping.** The premise — "the library lives on the same host as
|
||||||
|
the backend in every realistic deployment" — is false for the common case: Calibre is a desktop
|
||||||
|
application, and its library is on the desktop. So the split is by *surface*, not by preference:
|
||||||
|
|
||||||
|
- **Over HTTP: an uploaded zip only.** `…/imports/calibre/upload`. There is no endpoint taking a
|
||||||
|
server path; one was built and then removed deliberately.
|
||||||
|
- **On the server: a path only.** `litestar calibre-import <path>`, which is where a very large
|
||||||
|
library or a headless migration belongs — an upload has to carry the whole archive across
|
||||||
|
first.
|
||||||
|
|
||||||
|
A CLI taking a path needs no justification. An *endpoint* taking one would have: it would let any
|
||||||
|
authenticated caller read any directory the backend can, and `TODO.md` records there is no
|
||||||
|
authorization tier at all. Not adding it is one less thing to gate later.
|
||||||
|
|
||||||
|
4. **Import into an existing Chitai library**, chosen by the caller. Creating a library is
|
||||||
|
already one action, and the Calibre tree is rewritten by `BookPathGenerator` regardless.
|
||||||
|
|
||||||
|
5. **Re-running an import must be safe, and file-level dedupe already makes it so.** The same
|
||||||
|
bytes are recognised by `(hash, size)` whatever their path, so a second run over the same
|
||||||
|
library skips everything. No import bookkeeping is needed for idempotency.
|
||||||
|
|
||||||
|
## Prerequisite: a `.mobi` file breaks the book endpoint — **done**
|
||||||
|
|
||||||
|
> Landed ahead of the import itself. `guess_content_type` in `services/utils.py` now names every
|
||||||
|
> format from its extension, the column keeps a null when nothing can name one, and the OPDS feed
|
||||||
|
> substitutes `application/octet-stream` at the one place a string is required. The Read control is
|
||||||
|
> driven by `isReadable` rather than by the file count. See the section below for why it mattered,
|
||||||
|
> and `backend/AGENTS.md` for the rule as it now stands.
|
||||||
|
|
||||||
|
|
||||||
|
`FileMetadataRead.content_type` is a required `str` (`schemas/book.py:33`), but every ingest path
|
||||||
|
fills it from `mimetypes.guess_type`, which returns `None` for `.mobi`, `.azw`, `.fb2`, `.lit`
|
||||||
|
and `.htmlz` (verified). `FileMetadata.content_type` is nullable in the model, so the row stores
|
||||||
|
fine and then fails response validation on the way out — a book whose only file is a MOBI would
|
||||||
|
be unreadable through the API.
|
||||||
|
|
||||||
|
Nothing in the tree hits this today because the browser upload path is used with EPUBs and PDFs.
|
||||||
|
A Calibre library is full of MOBI and AZW3. Fix it first, either way round:
|
||||||
|
|
||||||
|
- make the schema field `str | None`, and/or
|
||||||
|
- add a small extension→MIME table for the ebook formats `mimetypes` does not know
|
||||||
|
(`application/x-mobipocket-ebook`, `application/vnd.amazon.ebook`, `application/x-fictionbook+xml`).
|
||||||
|
|
||||||
|
Do both, in fact: the table is the right answer for OPDS clients, which choose an acquisition link
|
||||||
|
by MIME type, and the nullable field is the safety net.
|
||||||
|
|
||||||
|
**Related, but not a blocker:** Chitai reads EPUB and PDF only. `openBookInReader`
|
||||||
|
(`book/[bookId]/+page.svelte:66`) branches on `getFileType(...) === 'EPUB' | 'PDF'` and does
|
||||||
|
nothing for anything else, so an AZW3-only book gets a Read button that silently fails. Importing
|
||||||
|
those files is still right — they are downloadable and they are the user's — but the button
|
||||||
|
should be disabled for a book with no readable file. One `$derived` on the page, worth doing in
|
||||||
|
the same branch.
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
### 1. `services/calibre.py` — a pure reader, no Chitai types
|
||||||
|
|
||||||
|
```python
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CalibreFile:
|
||||||
|
path: Path # absolute, resolved against the library root
|
||||||
|
format: str # "EPUB", as stored
|
||||||
|
size: int # data.uncompressed_size, for a cheap sanity check
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CalibreBook:
|
||||||
|
calibre_id: int
|
||||||
|
uuid: str
|
||||||
|
title: str
|
||||||
|
authors: list[str]
|
||||||
|
... # one field per row of the mapping table below
|
||||||
|
cover: Path | None
|
||||||
|
files: list[CalibreFile]
|
||||||
|
|
||||||
|
class CalibreLibrary:
|
||||||
|
def __init__(self, root: Path) -> None: ...
|
||||||
|
async def open(self) -> None: ... # copy + connect, see below
|
||||||
|
async def books(self) -> AsyncIterator[CalibreBook]: ...
|
||||||
|
async def close(self) -> None: ...
|
||||||
|
```
|
||||||
|
|
||||||
|
Deliberately knows nothing about `Book`, `BookService` or the session — it is a file-format
|
||||||
|
reader, unit-testable against a fixture database with no Postgres and no app.
|
||||||
|
|
||||||
|
Two implementation notes:
|
||||||
|
|
||||||
|
- **Copy `metadata.db` to a temp file and read the copy.** Calibre may be running and writing;
|
||||||
|
opening the live file read-only either sees a torn state or needs the `-wal` sidecar. The
|
||||||
|
database is small (438 KB for six books, single-digit MB for thousands), so a copy costs
|
||||||
|
nothing and removes the whole problem.
|
||||||
|
- **`sqlite3` inside `asyncio.to_thread`, not a new dependency.** The connection is used for a
|
||||||
|
handful of queries. Do not add `aiosqlite` for this.
|
||||||
|
|
||||||
|
Read the whole catalogue in **one query per table** and join in Python — six or so `SELECT`s and
|
||||||
|
a few dicts, versus a per-book N+1 across ten tables. At self-hosted scale the entire catalogue
|
||||||
|
minus descriptions fits in memory comfortably; if `comments.text` for 20k books is a concern,
|
||||||
|
fetch that one table per batch.
|
||||||
|
|
||||||
|
### 2. The mapping
|
||||||
|
|
||||||
|
| Calibre | Chitai | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `books.title` | `title` | Sanitise `/` and control chars for path generation. `Extractor.format_book_title` may still be worth applying to split a subtitle at the second colon — but **do not** run `split_edition`, Calibre's title is the curated one. |
|
||||||
|
| `authors.name` via `books_authors_link` | `authors` | `\|` → `,`. Order by `books_authors_link.id`; `Book.author_links` is an `ordering_list`, so insertion order is the displayed order. |
|
||||||
|
| `comments.text` | `description` | Strip HTML to text. |
|
||||||
|
| `books.pubdate` | `published_date` | Drop the year-101 sentinel. |
|
||||||
|
| `series.name`, `books.series_index` | `series`, `series_position` | `7.0` → `"7"`. |
|
||||||
|
| `tags.name` | `tags` | |
|
||||||
|
| `publishers.name` | `publisher` | `books_publishers_link` is unique per book. |
|
||||||
|
| `languages.lang_code` (lowest `item_order`) | `language` | Chitai holds one. |
|
||||||
|
| `identifiers.type` / `.val` | `identifiers` | Through `parse_identifier`; keep what survives. |
|
||||||
|
| `books.uuid` | `identifiers["calibre-uuid"]` | The one durable link back to the source row. `normalize_identifier` returns a key for it (it is not in `_PER_BUILD_NAMES`), which is *desirable*: a book re-imported from the same Calibre library matches on it exactly. |
|
||||||
|
| `books_pages_link.pages` | `pages` | Only when `> 0`. |
|
||||||
|
| `cover.jpg` when `has_cover` | `cover_image` | |
|
||||||
|
| `data` rows | `files` | |
|
||||||
|
| `books.timestamp` | — | `Book.created_at` is audit-managed; do not fight it. |
|
||||||
|
| `ratings`, `annotations`, `custom_columns` | — | No home in the model. Out of scope. |
|
||||||
|
| `last_read_positions` | `BookProgress` | Phase 2. |
|
||||||
|
|
||||||
|
### 3. The ingest
|
||||||
|
|
||||||
|
> **As built, this went on `BookService` as `create_many_from_calibre`, not into a separate
|
||||||
|
> `services/calibre_import.py`.** The orchestration needs `_reserve_book_path`,
|
||||||
|
> `_save_cover_image`, `_screen_for_duplicates` and `_record_possible_duplicates`, and reaching
|
||||||
|
> into four privates from another module is worse than one more method in the file where the other
|
||||||
|
> two ingest paths already live. `_record_possible_duplicates` was changed to take the list it
|
||||||
|
> appends to rather than an `ImportResult`, so every ingest path can share it whatever its own
|
||||||
|
> result type is. Two other deviations: `CalibreLibrary.books()` returns a list rather than an
|
||||||
|
> async iterator, because the caller needs the total up front anyway; and the reader reports
|
||||||
|
> identifiers exactly as Calibre keyed them, with the fold onto Chitai's schemes done by the
|
||||||
|
> importer, which keeps the reader free of Chitai imports.
|
||||||
|
|
||||||
|
Per book, in this order — it mirrors `create_many_from_existing_files`, which is the closest
|
||||||
|
existing shape:
|
||||||
|
|
||||||
|
1. `fingerprint_file` each source file (`services/utils.py:164`).
|
||||||
|
2. `find_duplicate_files` against the target library. All files known → skip the book entirely,
|
||||||
|
recording it. Some known → import the rest.
|
||||||
|
3. Build the metadata dict from the `CalibreBook`.
|
||||||
|
4. `_reserve_book_path(path_gen.generate_path(data))`.
|
||||||
|
5. **Copy** each file to `parent / _unused_path(...)`, building `FileMetadata` from the
|
||||||
|
fingerprint already computed. `services/utils.py` has `move_file` but no copy — add
|
||||||
|
`copy_file` beside it, streaming through `aiofiles` in `CHUNK_SIZE` blocks like
|
||||||
|
`_save_book_files` does, not `shutil.copy` (a 40 MB blocking read inside the event loop).
|
||||||
|
6. Cover: open `cover.jpg` with PIL and hand the `Image` to `_save_cover_image`, which already
|
||||||
|
accepts one and converts to WebP.
|
||||||
|
7. `super().create(data)` through `BookService`, then `find_duplicate_books` and record
|
||||||
|
candidates — same as `_record_possible_duplicates` (`services/book.py:1253`).
|
||||||
|
8. **Commit per book.** A 5,000-book import inside one transaction is one failure away from
|
||||||
|
nothing, and per-book commits are what lets the library page show books arriving — which is
|
||||||
|
the behaviour commit `85367da` deliberately built.
|
||||||
|
|
||||||
|
Report an `ImportResult`-shaped outcome; reuse `ImportResult` itself if it fits, extending it with
|
||||||
|
a `failures: list[tuple[int, str]]` keyed by Calibre id. **One book must never fail the run** —
|
||||||
|
a missing file, an unreadable cover or a `NOT NULL` violation gets recorded and skipped.
|
||||||
|
|
||||||
|
### 4. Progress, and where the import runs
|
||||||
|
|
||||||
|
> **As built, the HTTP surface is an uploaded archive and nothing else** — see decision 3.
|
||||||
|
>
|
||||||
|
> `POST …/imports/calibre/upload` takes a zipped library. The job owns the temp directory it is
|
||||||
|
> unpacked into and deletes it when it ends. Extraction refuses zip slip, an archive too big for the
|
||||||
|
> disk, and one with no `metadata.db` within three levels — all answered 400 before a job exists.
|
||||||
|
>
|
||||||
|
> This forced a fix to the SvelteKit proxy, which buffered request bodies with `arrayBuffer()`:
|
||||||
|
> survivable for one book, not for a multi-gigabyte archive. POST and PATCH now stream
|
||||||
|
> `request.body` through with `duplex: 'half'`.
|
||||||
|
>
|
||||||
|
> **A preview endpoint was built and then removed with the path route.** It read a server-side
|
||||||
|
> catalogue and reported its size before writing anything, which is only useful when the caller
|
||||||
|
> named a directory. An upload has already been carried across by the time anything can be read, so
|
||||||
|
> unpacking it *is* the validation step — an archive that is not a Calibre library is refused there.
|
||||||
|
|
||||||
|
The import outlives its request, so the handler starts it and returns a handle:
|
||||||
|
|
||||||
|
- `POST /libraries/{library_id:int}/imports/calibre` — body `{path, copy_files: true}`, returns
|
||||||
|
`{job_id, total}`. 202.
|
||||||
|
- `GET /libraries/imports/{job_id}` — `{state, total, processed, created, skipped, failed,
|
||||||
|
current_title, errors}`.
|
||||||
|
- `DELETE /libraries/imports/{job_id}` — cancel; the task checks a flag between books.
|
||||||
|
|
||||||
|
Keep the registry **in memory**, a `dict[str, ImportJob]` on a module-level singleton, with the
|
||||||
|
task created by `asyncio.create_task`. This matches what the app already does — the consume
|
||||||
|
watcher is an in-process singleton started from a lifespan hook — and it is roughly thirty lines
|
||||||
|
against a model, a migration and a service for the alternative.
|
||||||
|
|
||||||
|
State that limitation explicitly in the docstring: **it assumes one worker process.** The
|
||||||
|
production `CMD` is `litestar run`, which is single-process, so this holds today; `TODO.md`
|
||||||
|
already records that the production image should move to uvicorn with a worker count, and doing
|
||||||
|
that would mean a poll landing on a worker that has never heard of the job. The consume watcher
|
||||||
|
has the same problem, so this is not a new constraint — but the next person to add workers needs
|
||||||
|
to find it written down. If import *history* is ever wanted, that is when an `import_jobs` table
|
||||||
|
earns its migration.
|
||||||
|
|
||||||
|
**Also add a CLI entry point.** A 200 GB library imported through a browser tab that must stay
|
||||||
|
open is a bad experience, and `pyproject.toml` already declares a `chitai` script. A Litestar CLI
|
||||||
|
command (`litestar --app-dir src/chitai/ calibre-import <path> --library <slug>`) is ~20 lines
|
||||||
|
over the same service and is the right tool for the initial migration, which is the case this
|
||||||
|
whole feature exists for. The endpoint is for people who would rather click.
|
||||||
|
|
||||||
|
### 5. Frontend
|
||||||
|
|
||||||
|
Model it on the duplicates screen, which is the closest precedent in shape and placement:
|
||||||
|
|
||||||
|
- Route `(root)/settings/libraries/[libraryId]/import` — beside
|
||||||
|
`settings/libraries/[libraryId]/duplicates`, reached from the library settings page.
|
||||||
|
- `getCalibreImport` (a `query`) and `cancelCalibreImport` (a `command`) in
|
||||||
|
`src/lib/api/calibre-import.remote.ts`, re-exported from `src/lib/api/index.ts`. **Starting an
|
||||||
|
import is not a remote function**: the archive goes to the backend through the proxy so the
|
||||||
|
browser streams straight through, where a remote function would put the whole thing through the
|
||||||
|
SvelteKit process first. The screen uses `XMLHttpRequest` for it, which is the only way to get
|
||||||
|
upload progress.
|
||||||
|
- A file input, an upload progress bar, then a progress bar polling `getCalibreImport` every second
|
||||||
|
or two, a running count, and the failures listed at the end with their Calibre ids.
|
||||||
|
- Finish with a link to the library's duplicates screen. An import into a non-empty library is
|
||||||
|
the single most likely way to produce duplicate books, and that screen already handles them.
|
||||||
|
|
||||||
|
Do **not** route this through the upload tray. The tray reports on a client-driven queue it owns
|
||||||
|
(`upload-queue.svelte.ts`); this is server-side work whose state survives a page reload, and
|
||||||
|
conflating the two would mean teaching the tray to poll.
|
||||||
|
|
||||||
|
Regenerate `src/lib/schema/openapi/schema.d.ts` against a backend running **your** branch —
|
||||||
|
a stale server silently writes a stale file.
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
The fixture is the interesting part. Build a Calibre library in a `tmp_path` fixture rather than
|
||||||
|
committing a binary `metadata.db`: a helper that executes the subset of Calibre's `CREATE TABLE`
|
||||||
|
statements (they are in this document's shape, and in any real library's `sqlite_master`), inserts
|
||||||
|
a handful of books, and lays out `<Author>/<Title> (id)/` directories containing the existing
|
||||||
|
EPUB and PDF fixtures from `backend/tests/data_files/` plus a copy of `cover.jpg`. Generated
|
||||||
|
beats committed here because the tests need to assert on *odd* rows — the pubdate sentinel, a
|
||||||
|
`|` in an author name, a title with a colon — and those are clearer written in Python than hidden
|
||||||
|
in a blob.
|
||||||
|
|
||||||
|
- **Unit** (`tests/unit/test_calibre.py`) — the reader alone: field mapping; the year-101 pubdate
|
||||||
|
dropped; `series_index` 7.0 → `"7"`; `|` unescaped in an author name; HTML stripped from
|
||||||
|
`comments`; `pages = 0` ignored; identifiers passed through `parse_identifier`; a `data` row
|
||||||
|
whose file is missing from disk reported rather than raised; `.caltrash` never walked.
|
||||||
|
- **Service** (`tests/unit/test_services/test_calibre_import.py`) — a book with two formats lands
|
||||||
|
as one record with two files; the source files still exist afterwards; a second run over the
|
||||||
|
same library creates nothing; a library with one broken book imports the rest; authors and tags
|
||||||
|
shared between two books produce one `Author` / `Tag` row each; `possible_duplicates` reported
|
||||||
|
when the target library already holds the same book.
|
||||||
|
- **Integration** (`tests/integration/test_calibre_import.py`) — `POST` returns 202 with a job id,
|
||||||
|
polling reaches a terminal state, and the books are then listable through `GET /books`. A
|
||||||
|
`.mobi`-only book must come back from `GET /books/{id}` without a 500 — that is the
|
||||||
|
prerequisite's regression test.
|
||||||
|
|
||||||
|
`pytest` needs Docker (`pytest-databases`).
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
```bash
|
||||||
|
nix-shell # postgres + migrations applied
|
||||||
|
cd backend
|
||||||
|
pytest tests/ # take your own baseline first
|
||||||
|
uv run litestar --app-dir src/chitai/ run --port 8001 # for the OpenAPI regeneration
|
||||||
|
cd ../frontend && pnpm check # baseline: 30 errors, 1 warning, 8 files
|
||||||
|
```
|
||||||
|
|
||||||
|
Neither `pnpm check` nor `pnpm lint` is clean on this repo — baseline before assuming an error is
|
||||||
|
yours.
|
||||||
|
|
||||||
|
End to end, against a real library (`~/Documents/Calibre Library` will do): import into an empty
|
||||||
|
Chitai library and confirm the six books arrive with their covers, authors, tags, series
|
||||||
|
positions and identifiers intact; that the AZW3 book is listed and downloadable; that the source
|
||||||
|
library is byte-for-byte untouched (`diff -r` a copy taken beforehand); and that re-running the
|
||||||
|
import creates nothing and reports six skipped books. Then import the same library into a
|
||||||
|
library that already holds one of those books by upload, and confirm it lands on the duplicates
|
||||||
|
screen rather than as a second copy.
|
||||||
|
|
||||||
|
## Phasing
|
||||||
|
|
||||||
|
| Phase | Scope |
|
||||||
|
| --- | --- |
|
||||||
|
| **0** ✅ | The `content_type` prerequisite, plus the Read button. **Done** — see below. |
|
||||||
|
| **1** ✅ | `services/calibre.py`, the ingest, the CLI command, copy-only. **Done** — a headless one-time migration works today. |
|
||||||
|
| **2** ◐ | The upload endpoint, the job registry and the settings screen — **done**. The HTTP surface is an uploaded zip only; see decision 3, which this reversed. `last_read_positions` → `BookProgress` is **not** done: it needs a Calibre-user → Chitai-user mapping, and the answer differs between the endpoint (which has a `current_user`) and the CLI (which has none). That decision is the next thing to make. |
|
||||||
|
| **3** | Reference-in-place import. Its real content is **enforcing `Library.read_only`** across `update_book`, `delete_books`, `add_files` and `remove_files` — which is a feature of its own and should not be smuggled in under an import. |
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
Annotations and highlights (no model to put them in), custom columns, ratings, virtual libraries
|
||||||
|
and saved searches → bookshelves, format conversion, writing anything back to Calibre, and any
|
||||||
|
form of continuing two-way sync. This is a one-way migration.
|
||||||
@@ -0,0 +1,302 @@
|
|||||||
|
# Building release images in CI
|
||||||
|
|
||||||
|
Design for publishing `chitai-backend` and `chitai-frontend` container images from a tagged
|
||||||
|
release, on Gitea Actions, to the Gitea package registry, for `linux/amd64`.
|
||||||
|
|
||||||
|
## What exists today
|
||||||
|
|
||||||
|
- `origin` is Gitea 1.27 at `git.jaroszew.ski`. The repo is **public**, with `has_actions` and
|
||||||
|
`has_packages` both true. There is no `.gitea/` or `.github/` directory and no CI of any kind.
|
||||||
|
- There are **no tags** in the repository. `backend/pyproject.toml` says `0.1.0`,
|
||||||
|
`frontend/package.json` says `0.0.1`; nothing reads either.
|
||||||
|
- `backend/Dockerfile` and `frontend/Dockerfile` are both multi-stage and both already build a
|
||||||
|
runnable image. `docker-compose.yml` builds them from source with `build: ./backend` and
|
||||||
|
`build: ./frontend` — there is no `image:` key, so there is nothing for a user to pull.
|
||||||
|
|
||||||
|
So the work is not "make the images build" — they build. It is "make an image built on one machine
|
||||||
|
correct on another", then automate producing one per tag.
|
||||||
|
|
||||||
|
## The blocker: the frontend image hardcodes the backend URL
|
||||||
|
|
||||||
|
This has to be fixed before publishing an image is meaningful.
|
||||||
|
|
||||||
|
`frontend/src/lib/server/config.ts` reads the backend URL through `import.meta.env`:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
export const BACKEND_API_URL = import.meta.env.VITE_BACKEND_API_URL || 'http://localhost:8000';
|
||||||
|
```
|
||||||
|
|
||||||
|
Vite replaces `import.meta.env.VITE_*` **at build time**, including in the SSR bundle. The current
|
||||||
|
build output in `frontend/build/` shows exactly what that produces:
|
||||||
|
|
||||||
|
```js
|
||||||
|
// frontend/build/server/chunks/config-BvKh7uym.js
|
||||||
|
const BACKEND_API_URL = "http://localhost:8000";
|
||||||
|
```
|
||||||
|
|
||||||
|
`frontend/Dockerfile` sets no `VITE_BACKEND_API_URL` before `pnpm run build`, so the string baked
|
||||||
|
into any image built from it is `http://localhost:8000`. The `VITE_BACKEND_API_URL: ${CHITAI_API_URL}`
|
||||||
|
entry under the compose `frontend` service is therefore **dead** — it sets a process environment
|
||||||
|
variable that nothing reads, in a process whose value was decided at build time. Inside the
|
||||||
|
container `localhost:8000` is the frontend's own port, not the backend.
|
||||||
|
|
||||||
|
Two ways out, and only one of them is right for a published image:
|
||||||
|
|
||||||
|
- **Build arg.** `ARG VITE_BACKEND_API_URL` in the build stage. This works, but it makes the image
|
||||||
|
specific to one deployment's topology — CI would bake `http://backend:8000` and anyone whose
|
||||||
|
service is named differently gets an image that cannot be repointed. Wrong for a release artifact.
|
||||||
|
- **Runtime env, recommended.** Read it through SvelteKit's dynamic env, which is `process.env` at
|
||||||
|
request time:
|
||||||
|
|
||||||
|
```ts
|
||||||
|
import { env } from '$env/dynamic/private';
|
||||||
|
export const BACKEND_API_URL = env.VITE_BACKEND_API_URL || 'http://localhost:8000';
|
||||||
|
```
|
||||||
|
|
||||||
|
`$lib/server/config.ts` is server-only and its one consumer (`$lib/server/api.ts`) is too, so
|
||||||
|
`$env/dynamic/private` is available everywhere it is used. The compose entry then starts working
|
||||||
|
as written, and the same image serves any deployment.
|
||||||
|
|
||||||
|
Worth renaming the variable to `CHITAI_API_URL` at the same time — the `VITE_` prefix now means
|
||||||
|
the opposite of what it does — but that is a follow-up, not a prerequisite. If you do rename it,
|
||||||
|
the fallback in `.env.prod-example` and the compose `environment:` block move with it.
|
||||||
|
|
||||||
|
Same class of problem, same fix window: `frontend/Dockerfile` sets `ENV ORIGIN=http://localhost:3000`.
|
||||||
|
That one *is* read at runtime by `adapter-node`, so it can be overridden — but nothing overrode it,
|
||||||
|
and adapter-node rejects cross-origin form POSTs when `ORIGIN` does not match the browser's, so
|
||||||
|
every deployment behind a real domain 403s on its first login. Now set as
|
||||||
|
`ORIGIN: ${CHITAI_ORIGIN:-http://localhost:3000}` on the compose `frontend` service, with a
|
||||||
|
documented `CHITAI_ORIGIN` in `.env.prod-example`.
|
||||||
|
|
||||||
|
**Both are done.** The built module now reads `private_env.VITE_BACKEND_API_URL`, and the published
|
||||||
|
image was verified by running it with `VITE_BACKEND_API_URL` pointed at a throwaway listener: the
|
||||||
|
container's `GET /access/me` arrived there rather than at `localhost:8000`.
|
||||||
|
|
||||||
|
## Toolchain pinning
|
||||||
|
|
||||||
|
Two reproducibility gaps that a release pipeline turns from cosmetic into real, because CI builds
|
||||||
|
from a clean container every time and your laptop does not.
|
||||||
|
|
||||||
|
- **pnpm has no pin.** `frontend/package.json` has no `packageManager` field, so `corepack enable`
|
||||||
|
followed by `pnpm install` resolves to whatever version corepack considers current on the day the
|
||||||
|
build runs. `pnpm-lock.yaml` is `lockfileVersion: 9.0`, i.e. pnpm 9/10; a future pnpm 11 could
|
||||||
|
refuse it, and `--frozen-lockfile` would fail a release for reasons unrelated to the release. Add
|
||||||
|
`"packageManager": "pnpm@<version from your nix shell>"` and let corepack honour it.
|
||||||
|
- **`pnpm-workspace.yaml` is never copied into the image.** It carries `onlyBuiltDependencies`
|
||||||
|
(`esbuild`, `@tailwindcss/oxide`), and pnpm 10 blocks postinstall scripts that are not listed
|
||||||
|
there. Both install stages in `frontend/Dockerfile` copy only `pnpm-lock.yaml` and `package.json`,
|
||||||
|
so the container install runs under different rules than the local one. Copy it alongside
|
||||||
|
`package.json` in the `prod-deps` and `build` stages so the two agree.
|
||||||
|
|
||||||
|
Both are one-line changes and both belong before the first tag, not after. **Both are done** —
|
||||||
|
`packageManager` is pinned to the dev shell's `pnpm@11.20.0`, and `pnpm-workspace.yaml` is copied
|
||||||
|
into the `prod-deps` and `build` stages. `docker compose build` was re-run against the change.
|
||||||
|
|
||||||
|
## Prerequisites on the Gitea instance
|
||||||
|
|
||||||
|
Verify these before writing the workflow; each one fails the job in a way that looks like a bug in
|
||||||
|
the workflow.
|
||||||
|
|
||||||
|
1. **A registered `act_runner`.** Gitea Actions is enabled instance-side but does nothing without a
|
||||||
|
runner. Register one against the repo or the instance with the label `ubuntu-latest`.
|
||||||
|
2. **The runner needs a Docker daemon, at an address the job can reach.** `act_runner` in docker
|
||||||
|
mode runs each job inside a container with no daemon of its own. Either run a `docker:dind`
|
||||||
|
sidecar next to the runner and set `DOCKER_HOST=tcp://docker:2376` (with TLS certs shared over a
|
||||||
|
volume), or run the runner in host mode. The dind sidecar is the safer of the two — mounting the
|
||||||
|
host socket into job containers gives any workflow root on the runner host.
|
||||||
|
|
||||||
|
**Reachability is a separate question from availability, and it bit us.** A containerised job
|
||||||
|
with a working daemon still fails `pytest tests/` with
|
||||||
|
`Service 'pytest_databases_postgres' failed to come online`: the database container starts
|
||||||
|
fine, but its published port lands on the daemon's network namespace while the test process
|
||||||
|
looks for it on the job container's loopback.
|
||||||
|
|
||||||
|
Two variables control two different things, and both must be set:
|
||||||
|
|
||||||
|
| Variable | Decides | Read by |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `DOCKER_HOST` | which daemon the container is created on | `_service.py` `get_docker_host()` |
|
||||||
|
| `POSTGRES_HOST` | the address the test then connects to | `docker/postgres.py` `postgres_host`, default `127.0.0.1` |
|
||||||
|
|
||||||
|
Setting only `DOCKER_HOST` is not enough — `DockerService.run()` takes `container_host` as a
|
||||||
|
plain argument defaulting to `127.0.0.1`, and the postgres fixture fills it from
|
||||||
|
`POSTGRES_HOST`. (There *is* a `DOCKER_HOST`-parsing helper in `pytest_databases`, but it is
|
||||||
|
`_get_docker_ip()` on the docker-compose class in `docker/__init__.py` and no part of this
|
||||||
|
path uses it. Do not be misled by it, as I was.)
|
||||||
|
|
||||||
|
Until the runner grows its own sidecar, `ci.yml`'s backend job carries a `docker:dind` service
|
||||||
|
of its own with `DOCKER_HOST: tcp://docker:2375`. That needs the runner to permit
|
||||||
|
`--privileged`. The same treatment is still owed to `release.yml` — its `quality` job runs the
|
||||||
|
same tests, and its `smoke` job talks to compose, where the published ports would move to the
|
||||||
|
dind host too, so `curl http://localhost:3000` becomes `curl http://docker:3000`. Configuring
|
||||||
|
the runner once (option 1) avoids all of that.
|
||||||
|
3. **Action resolution.** A bare `uses: docker/build-push-action@v6` does not mean github.com here.
|
||||||
|
Gitea resolves it against `[actions] DEFAULT_ACTIONS_URL`, which defaults to `https://gitea.com`.
|
||||||
|
That is fine as it stands — `actions/checkout@v4`, `docker/setup-buildx-action@v3`,
|
||||||
|
`docker/login-action@v3`, `docker/metadata-action@v5` and `docker/build-push-action@v6` are all
|
||||||
|
mirrored on gitea.com at those tags (verified 2026-08-17). Worth knowing because it is where an
|
||||||
|
action reference resolves from if that setting is ever changed; `DEFAULT_ACTIONS_URL = github`
|
||||||
|
in `app.ini` is the fix if so. Writing full `https://` URLs in `uses:` also works on Gitea but
|
||||||
|
is invalid syntax on GitHub Actions, so it would cost portability for no gain.
|
||||||
|
4. **A registry credential.** Gitea auto-injects `secrets.GITEA_TOKEN`, but whether it carries
|
||||||
|
package-write scope has varied across versions. Try it first; if the push 401s, create a personal
|
||||||
|
access token with `write:package` and store it as the repo secret `REGISTRY_TOKEN`. The workflow
|
||||||
|
below reads `REGISTRY_TOKEN` with a fallback to the automatic token.
|
||||||
|
5. **Package visibility.** Gitea ties package visibility to the owner rather than offering a
|
||||||
|
per-package toggle. The repo is public, so anonymous pulls should work — confirm with a
|
||||||
|
`docker pull` from a logged-out machine after the first release, because the README will tell
|
||||||
|
people to do exactly that.
|
||||||
|
|
||||||
|
## The workflow
|
||||||
|
|
||||||
|
Written as **`.gitea/workflows/release.yml`**, triggered by tags matching `v*`. Notes on the choices
|
||||||
|
made there:
|
||||||
|
|
||||||
|
- **`github.*` context, not `gitea.*`.** Both exist on Gitea; the third-party actions read the
|
||||||
|
`GITHUB_*` environment anyway, and using it keeps the file portable if the repo is ever mirrored.
|
||||||
|
`github.repository` is `patrick/chitai`, so the images are
|
||||||
|
`git.jaroszew.ski/patrick/chitai-backend` and `…/chitai-frontend`.
|
||||||
|
- **Tags produced from `v1.2.3`:** `1.2.3`, `1.2`, `1`, and `latest`. `metadata-action`'s default
|
||||||
|
`latest=auto` flavour adds `latest` only for a non-prerelease semver, so `v0.2.0-rc.1` publishes
|
||||||
|
`0.2.0-rc.1` and leaves `latest` where it was. That is what makes release-candidate tags a safe
|
||||||
|
way to exercise the pipeline.
|
||||||
|
- **`fail-fast: false`** so a frontend failure does not cancel a backend build that was going to
|
||||||
|
succeed. The two images are independent artifacts; a half-published release is easier to reason
|
||||||
|
about than a cancelled one.
|
||||||
|
- **`type=gha` cache** relies on `act_runner`'s built-in cache server exporting
|
||||||
|
`ACTIONS_CACHE_URL` / `ACTIONS_RUNTIME_TOKEN`. If your runner has caching disabled, buildx warns
|
||||||
|
and continues, or errors depending on version — just delete the two `cache-*` lines. The
|
||||||
|
Dockerfiles' `--mount=type=cache` blocks do nothing across ephemeral runners either way, and a
|
||||||
|
cold build of both images is a few minutes.
|
||||||
|
|
||||||
|
## Gating: formatting, linting, types and tests
|
||||||
|
|
||||||
|
`ci.yml` runs these on every push and pull request, and `release.yml`'s `quality` job runs the
|
||||||
|
blocking half again on the tagged commit, so a release cannot publish an image whose tests fail.
|
||||||
|
|
||||||
|
My first draft of this document argued for keeping all of it out of the release path, on the
|
||||||
|
grounds that everything was red. Measuring rather than trusting `TODO.md` showed that was mostly
|
||||||
|
wrong:
|
||||||
|
|
||||||
|
| Check | Before | After the sweep | Gate |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `pytest tests/` | **411 passed** in 3 min | unchanged | blocking |
|
||||||
|
| `ruff format --check src/` | 27 of 66 files | clean | blocking |
|
||||||
|
| `ruff check src/` | 110 errors | clean | blocking |
|
||||||
|
| `prettier --check .` | 48 files | clean | blocking |
|
||||||
|
| `eslint .` | 1804 → **87 real** | 87 | non-blocking |
|
||||||
|
| `pnpm check` | 30 errors | 30 | non-blocking |
|
||||||
|
|
||||||
|
The test suite was green the whole time, so gating on it costs nothing. `ruff check`'s 110 were
|
||||||
|
104 unused imports, and **all 64 that survived `--fix` were in `__init__.py`** — deliberate
|
||||||
|
re-exports, so the fix is `per-file-ignores` in `pyproject.toml`, not deleting them. Of eslint's
|
||||||
|
1804, **1717 were the vendored pdf.js under `static/`**, which the config never ignored the way it
|
||||||
|
ignores `src/lib/vendor/`; adding it leaves 87 real ones.
|
||||||
|
|
||||||
|
Two things the sweep turned up that are worth remembering:
|
||||||
|
|
||||||
|
- **ruff's suggested fix for `E712` would have broken the query.** `services/filters/book.py` had
|
||||||
|
`m.BookProgress.completed == True`, and ruff proposes `if m.BookProgress.completed:` — Python
|
||||||
|
truthiness on a SQLAlchemy Column, which does not generate SQL at all. The correct idiom is
|
||||||
|
`.is_(True)`, which the adjacent line already used. Never run `ruff check --fix --unsafe-fixes`
|
||||||
|
over query-building code without reading every hunk.
|
||||||
|
- **Prettier reformatted the generated `schema.d.ts`**, which was 3109 of the sweep's 3946 changed
|
||||||
|
lines. `openapi-typescript` writes its own style, so that file would have churned by thousands of
|
||||||
|
lines on every regeneration — and then failed the very gate being added. It is in
|
||||||
|
`.prettierignore` now.
|
||||||
|
|
||||||
|
`eslint` and `pnpm check` run with `continue-on-error: true`. That is an honest weak gate: it
|
||||||
|
reports without failing, so the counts stay visible and cannot grow silently unnoticed, but nobody
|
||||||
|
is blocked by 117 pre-existing problems they did not create. Drop the line from each step as its
|
||||||
|
count reaches zero.
|
||||||
|
|
||||||
|
The other thing worth gating on is that the images actually start, which is the failure mode a release
|
||||||
|
introduces and which nothing else catches. That is the `smoke` job in the same workflow: it copies
|
||||||
|
`.env.prod-example`, pins `CHITAI_VERSION` to the tag, `docker compose pull`s the images that were
|
||||||
|
just pushed and brings the stack up with `--wait`, then curls both healthchecks.
|
||||||
|
|
||||||
|
This is cheap and it covers the three things that break a release image: migrations failing to apply
|
||||||
|
from `entrypoint.sh`, the frontend being unable to reach the backend (the baked-URL bug above —
|
||||||
|
`--wait` fails because the frontend never goes healthy), and a missing runtime env var. It tests the
|
||||||
|
artifact that was pushed, not a rebuild of it.
|
||||||
|
|
||||||
|
`--wait` depends on healthchecks. `db` declares one inline in `docker-compose.yml`, and the backend
|
||||||
|
image carries a `HEALTHCHECK` in its Dockerfile which compose inherits. The frontend had neither, so
|
||||||
|
one was added to `frontend/Dockerfile` — in the image rather than in compose, matching the backend
|
||||||
|
and covering anyone running the image without compose. It probes `/login` with **node's global
|
||||||
|
`fetch`, not curl**, because `node:24-slim` ships no curl and adding one for a healthcheck is a
|
||||||
|
package and a CVE surface for nothing. It reads `PORT` so it keeps working if the port is
|
||||||
|
overridden.
|
||||||
|
|
||||||
|
Note the smoke job runs `cp .env.prod-example .env`, which is destructive on a developer machine —
|
||||||
|
it is safe only because a CI checkout has no `.env`. Don't run those lines locally.
|
||||||
|
|
||||||
|
## Making the images consumable
|
||||||
|
|
||||||
|
The images are pointless if `docker-compose.yml` still builds from source. Both services now carry
|
||||||
|
`image:` **and** keep `build:` — compose pulls when the image is absent, and `docker compose build`
|
||||||
|
still builds from source and tags the result under the same name:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
backend:
|
||||||
|
image: git.jaroszew.ski/patrick/chitai-backend:${CHITAI_VERSION:-latest}
|
||||||
|
build: ./backend
|
||||||
|
```
|
||||||
|
|
||||||
|
`CHITAI_VERSION` and `CHITAI_ORIGIN` are documented in `.env.prod-example`, and the README's install
|
||||||
|
steps now copy and edit `.env` *before* pulling — the pull cannot resolve `${CHITAI_VERSION}` until
|
||||||
|
that file exists, so the old ordering would have silently fetched `latest`.
|
||||||
|
|
||||||
|
## Versioning
|
||||||
|
|
||||||
|
Adopt `vMAJOR.MINOR.PATCH`. The tag is the source of truth; `metadata-action` derives everything
|
||||||
|
from it and the OCI labels record it. There is no tag today, so the first release is a decision —
|
||||||
|
`v0.1.0` matches `backend/pyproject.toml` and is the honest number for an app with an open
|
||||||
|
"any authenticated user can delete any library" item.
|
||||||
|
|
||||||
|
Do not automate bumping `pyproject.toml` and `package.json` from the tag. It requires CI to commit
|
||||||
|
back to the repo, and neither version is read by anything. Either keep a two-line manual checklist
|
||||||
|
(bump both, commit, tag) or delete `frontend/package.json`'s version field and treat the backend's
|
||||||
|
as the product version. Bumping by hand is the smaller cost.
|
||||||
|
|
||||||
|
## Order of work
|
||||||
|
|
||||||
|
- [x] **Runtime config.** `frontend/src/lib/server/config.ts` → `$env/dynamic/private`; `ORIGIN` on
|
||||||
|
the compose frontend service; `CHITAI_ORIGIN` in `.env.prod-example`.
|
||||||
|
- [x] **Toolchain pins.** `packageManager` in `frontend/package.json`; `pnpm-workspace.yaml` copied
|
||||||
|
in both `frontend/Dockerfile` install stages.
|
||||||
|
- [x] **The workflow.** `.gitea/workflows/release.yml`, both jobs.
|
||||||
|
- [x] **Consumable images.** `image:` keys in `docker-compose.yml`, frontend `HEALTHCHECK`,
|
||||||
|
`CHITAI_VERSION`, README install steps.
|
||||||
|
- [ ] **Stand up `act_runner` with a working Docker daemon.** Not something the repo can carry.
|
||||||
|
Prove it with a throwaway workflow running `docker version` before trusting the release one.
|
||||||
|
- [ ] **Tag `v0.1.0-rc.1`.** Confirm two images land in the registry, that `latest` was *not* moved,
|
||||||
|
and that the smoke job goes green.
|
||||||
|
- [ ] **Confirm an anonymous `docker pull`** from a logged-out machine, since the README tells people
|
||||||
|
to do exactly that.
|
||||||
|
- [ ] **Tag `v0.1.0`.**
|
||||||
|
|
||||||
|
Everything the repository can hold is in place; what remains is instance-side and needs a real tag
|
||||||
|
to exercise. Verified locally along the way: `docker compose build` succeeds against both changed
|
||||||
|
Dockerfiles, `docker compose config` resolves the image names, and the built frontend image
|
||||||
|
honours `VITE_BACKEND_API_URL` at runtime and reports `healthy` to Docker.
|
||||||
|
|
||||||
|
## Deferred, deliberately
|
||||||
|
|
||||||
|
- **Multi-arch.** amd64 only for now. Adding `linux/arm64` under QEMU roughly triples the job and the
|
||||||
|
Python and pnpm installs are exactly the workload emulation is worst at; a native arm runner and a
|
||||||
|
fan-out/merge workflow is the answer if it is ever needed.
|
||||||
|
- **Building on `main`.** An `edge` tag from every push to `main` is a two-line addition to the same
|
||||||
|
workflow (`on: push: branches: [main]` plus `type=raw,value=edge,enable={{is_default_branch}}`).
|
||||||
|
Left out because it doubles registry churn for a deployment target that does not exist yet.
|
||||||
|
- **PR CI.** Formatting, lint, type and test gates are a separate workflow on a separate trigger and
|
||||||
|
should not be mixed into the release path — see the gating section.
|
||||||
|
- **Gitea Releases.** A job creating a release object with generated notes is nice-to-have and can
|
||||||
|
be added once the tags mean something.
|
||||||
|
- **Signing / SBOM / provenance.** `build-push-action` can emit provenance and an SBOM, and cosign
|
||||||
|
can sign the digests. Worth it if the images are ever consumed by anyone other than you; not worth
|
||||||
|
the key management before then.
|
||||||
|
- **`backend/.dockerignore` is thin.** It excludes `.git`, `__pycache__` and the ruff cache, but not
|
||||||
|
`.venv/`, `.postgres/`, `libraries/`, `covers/` or `tests/`. None of those reach the image — the
|
||||||
|
Dockerfile copies specific paths — and none exist in a CI checkout since they are gitignored, so
|
||||||
|
this does not affect the pipeline. It does make local builds slower than they need to be.
|
||||||
@@ -0,0 +1,146 @@
|
|||||||
|
# Implementation brief: move Duplicates into library settings
|
||||||
|
|
||||||
|
Written for an agent picking this up cold. Read the repo-root `AGENTS.md` and
|
||||||
|
`frontend/AGENTS.md` first — this brief assumes both.
|
||||||
|
|
||||||
|
**This is a frontend-only change.** The backend already scopes everything by library
|
||||||
|
(`GET /books/duplicate-books?library_id=`), so no endpoint, schema or migration is
|
||||||
|
involved.
|
||||||
|
|
||||||
|
## Where this starts from
|
||||||
|
|
||||||
|
The duplicates review screen exists and works. It currently lives at
|
||||||
|
|
||||||
|
```
|
||||||
|
frontend/src/routes/(root)/(library)/library/[libraryId]/duplicates/
|
||||||
|
+page.server.ts loads the groups, plus the full Book records merge needs
|
||||||
|
+page.svelte group cards, "Not duplicates", "Merge…"
|
||||||
|
```
|
||||||
|
|
||||||
|
and is reached from a **Duplicates entry in the main sidebar**
|
||||||
|
(`frontend/src/lib/components/layout/nav-main.svelte`), which is what this change
|
||||||
|
removes.
|
||||||
|
|
||||||
|
Settings today is a flat, entirely global four-item nav
|
||||||
|
(`frontend/src/routes/(root)/settings/+layout.svelte`): Account, Appearance, Libraries,
|
||||||
|
Devices. `settings/libraries/+page.svelte` is a single table of every library whose rows
|
||||||
|
link *out* to the library itself. **There is nowhere that means "settings for this
|
||||||
|
library"** — that is the gap this change fills.
|
||||||
|
|
||||||
|
## What to build — option B
|
||||||
|
|
||||||
|
Libraries expands in the settings nav. Every library is a sub-item; selecting one swaps
|
||||||
|
the pane; Duplicates is a section inside that pane. All libraries and all their sections
|
||||||
|
end up one click apart.
|
||||||
|
|
||||||
|
```
|
||||||
|
/settings/libraries the existing table (leave it as the index)
|
||||||
|
/settings/libraries/[libraryId] redirects to the first section
|
||||||
|
/settings/libraries/[libraryId]/duplicates the review screen, moved
|
||||||
|
```
|
||||||
|
|
||||||
|
Suggested files:
|
||||||
|
|
||||||
|
| Path | What |
|
||||||
|
| --- | --- |
|
||||||
|
| `settings/libraries/[libraryId]/+layout.svelte` | Library name, and the section tabs |
|
||||||
|
| `settings/libraries/[libraryId]/+page.ts` | `redirect(303, …/duplicates)` |
|
||||||
|
| `settings/libraries/[libraryId]/duplicates/+page.server.ts` | Moved verbatim |
|
||||||
|
| `settings/libraries/[libraryId]/duplicates/+page.svelte` | Moved verbatim |
|
||||||
|
|
||||||
|
Duplicates is the **only** real section today. Build the tab strip so General and Danger
|
||||||
|
zone have somewhere obvious to land, but do not invent them now — an empty tab is worse
|
||||||
|
than no tab.
|
||||||
|
|
||||||
|
## The nav
|
||||||
|
|
||||||
|
In `settings/+layout.svelte`, `items` is a flat `as const` array matched on
|
||||||
|
`page.route.id`. Libraries needs to render its children beneath it:
|
||||||
|
|
||||||
|
```svelte
|
||||||
|
{#each libraryState.libraries as library (library.id)}
|
||||||
|
<a href={resolve('/(root)/settings/libraries/[libraryId]/duplicates', {
|
||||||
|
libraryId: String(library.id) })}> … </a>
|
||||||
|
{/each}
|
||||||
|
```
|
||||||
|
|
||||||
|
`getLibraryState()` **is** available under `/settings` — it is set in
|
||||||
|
`(root)/+layout.svelte`, above the settings group, and `settings/libraries/+page.svelte`
|
||||||
|
already uses it. No new load function is needed to list the libraries.
|
||||||
|
|
||||||
|
**Active state is matched on route id, not pathname.** There is a comment in
|
||||||
|
`settings/+layout.svelte` explaining why: `resolve()` returns an absolute path on the
|
||||||
|
client and a relative one during SSR, so a pathname comparison is false on the server and
|
||||||
|
true after hydration, and the highlight flashes in. A nested library item is active when
|
||||||
|
the route id matches **and** `page.params.libraryId === String(library.id)` — both, or
|
||||||
|
every library lights up at once.
|
||||||
|
|
||||||
|
## Things that will bite
|
||||||
|
|
||||||
|
1. **Remove the sidebar entry in the same change.** `nav-main.svelte` gained a
|
||||||
|
`Duplicates` item and a `CopyCheck` import when the screen was built. Delete both, and
|
||||||
|
delete the old route directory. Doing the removal and the move together is the point —
|
||||||
|
split across two commits the screen is unreachable in between.
|
||||||
|
|
||||||
|
2. **Delete the old route, do not leave it.** Two live copies of a screen that both write
|
||||||
|
is how they drift.
|
||||||
|
|
||||||
|
3. **`setBookSelectionState` is not available under `/settings`.** It is set in
|
||||||
|
`(root)/(library)/+layout.svelte`, which the settings group is not inside. This is
|
||||||
|
fine — the duplicates page uses `BookImage` directly, not `book-thumbnail.svelte`, and
|
||||||
|
`MergeBooks` takes its `libraryId` as a prop. **Verify this stays true** if you touch
|
||||||
|
either component; a `getBookSelectionState()` under settings returns `undefined` and
|
||||||
|
fails at the first access, not at import.
|
||||||
|
|
||||||
|
4. **Keep `depends('app:duplicate-books')`.** Both the dismiss action and `MergeBooks`
|
||||||
|
call `invalidate('app:duplicate-books')` to make a resolved group leave the screen.
|
||||||
|
Drop it and the page silently stops refreshing. `MergeBooks` also invalidates
|
||||||
|
`app:books`, which is a no-op under settings and should stay that way.
|
||||||
|
|
||||||
|
5. **The settings shell is height-constrained.** `settings/+layout.svelte` is
|
||||||
|
`h-[calc(100vh-var(--header-height)-2rem)]` with `overflow-auto` on the content pane.
|
||||||
|
The review screen is a long list of cards — it must scroll *inside* that pane. Its
|
||||||
|
current `mx-auto max-w-5xl` wrapper will want revisiting.
|
||||||
|
|
||||||
|
6. **Three levels of nav is option B's known cost.** Nav → library → section, and the
|
||||||
|
pane is narrower than the full-width route the screen was designed against. The group
|
||||||
|
cards are `w-36` covers in a wrapping flex row, so they reflow, but check a group of
|
||||||
|
four at a narrow window before calling it done.
|
||||||
|
|
||||||
|
7. **`resolve()` must be a direct call in markup** for `svelte/no-navigation-without-resolve`.
|
||||||
|
Where `nav-main.svelte` computes a url through a variable it carries an
|
||||||
|
`eslint-disable-next-line`; prefer the direct call over inheriting that.
|
||||||
|
|
||||||
|
8. **The loader depends on the `?ids=` fix.** `+page.server.ts` fetches full `Book`
|
||||||
|
records with `listBooks({ ids, pageSize })` because the merge workbench needs
|
||||||
|
identifiers, description and publisher, which `DuplicateBookRead` does not carry.
|
||||||
|
advanced_alchemy's stock id filter types that parameter as `list[str]` regardless of
|
||||||
|
config, which made Postgres refuse `bigint = character varying`; the override lives in
|
||||||
|
`backend/src/chitai/services/dependencies.py` (`create_book_filter_dependencies`).
|
||||||
|
If `GET /books?ids=1&ids=2` 500s, that override is missing — do not work around it in
|
||||||
|
the loader.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
The **General** and **Danger zone** sections (rename, path template, read-only, consume
|
||||||
|
directory, delete), and any change to the merge workbench itself. The toolbar entry point
|
||||||
|
for merge — select 2+ books in the library view — is unrelated and stays where it is.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd frontend
|
||||||
|
pnpm check # baseline: 30 errors, 1 warning, 8 files — none of them yours
|
||||||
|
pnpm lint # not clean either; check the files you touched, not the tree
|
||||||
|
pnpm build
|
||||||
|
```
|
||||||
|
|
||||||
|
By hand, with a library that has a duplicate group:
|
||||||
|
|
||||||
|
- Settings → Libraries lists every library beneath it; clicking one opens its pane.
|
||||||
|
- Duplicates shows the same groups the old route did, and scrolls inside the settings pane.
|
||||||
|
- **Not duplicates** removes the group and it stays gone after a reload.
|
||||||
|
- **Merge…** opens the workbench, merges, and the group leaves the screen.
|
||||||
|
- The main sidebar no longer has a Duplicates entry, and
|
||||||
|
`/library/<id>/duplicates` no longer resolves.
|
||||||
|
- A library with no duplicates shows the empty state, not a blank pane.
|
||||||
@@ -8,9 +8,14 @@ bun.lockb
|
|||||||
# Ignore artifacts:
|
# Ignore artifacts:
|
||||||
build
|
build
|
||||||
coverage
|
coverage
|
||||||
|
.pytest_cache
|
||||||
|
|
||||||
# Miscellaneous
|
# Miscellaneous
|
||||||
/static/
|
/static/
|
||||||
|
|
||||||
# Vendored third-party source, copied verbatim by scripts/vendor-foliate.sh
|
# Vendored third-party source, copied verbatim by scripts/vendor-foliate.sh
|
||||||
/src/lib/vendor/
|
/src/lib/vendor/
|
||||||
|
|
||||||
|
# Generated by openapi-typescript, which has its own formatting. Reformatting it here would
|
||||||
|
# make every regeneration a several-thousand-line diff.
|
||||||
|
/src/lib/schema/openapi/schema.d.ts
|
||||||
|
|||||||
+87
-17
@@ -4,13 +4,16 @@ SvelteKit web app for the eBook library. See the repo-root `AGENTS.md` for the o
|
|||||||
dev-environment setup.
|
dev-environment setup.
|
||||||
|
|
||||||
**Stack:** SvelteKit 2 with `adapter-node` · Svelte 5 (runes) · Tailwind v4 · Zod v4 ·
|
**Stack:** SvelteKit 2 with `adapter-node` · Svelte 5 (runes) · Tailwind v4 · Zod v4 ·
|
||||||
`epubjs` · `mode-watcher` (dark mode) · `svelte-sonner` (toasts) · pnpm.
|
vendored `foliate-js` (EPUB) · vendored `pdf.js` (PDF) · `mode-watcher` (dark mode) ·
|
||||||
|
`svelte-sonner` (toasts) · pnpm.
|
||||||
|
|
||||||
Two experimental flags are on in `svelte.config.js` and the codebase depends on both:
|
Two experimental flags are on in `svelte.config.js` and the codebase depends on both:
|
||||||
`kit.experimental.remoteFunctions` and `compilerOptions.experimental.async` (`await` in components).
|
`kit.experimental.remoteFunctions` and `compilerOptions.experimental.async` (`await` in components).
|
||||||
|
|
||||||
Tailwind v4 has **no config file** — the theme, oklch colour tokens and `@custom-variant dark` all
|
Tailwind v4 has **no config file** — the theme, colour tokens (hex, not oklch) and
|
||||||
live in `src/app.css`.
|
`@custom-variant dark` all live in `src/app.css`. `dark` is the _only_ custom variant defined, so
|
||||||
|
generated components that assume others — shadcn's slider ships `data-horizontal:` / `data-vertical:`
|
||||||
|
classes — silently produce no styles. Use the `data-[orientation=…]` form instead.
|
||||||
|
|
||||||
## Talking to the backend
|
## Talking to the backend
|
||||||
|
|
||||||
@@ -65,8 +68,12 @@ Svelte context with a module-level `Symbol` key and a `setXState` / `getXState`
|
|||||||
|
|
||||||
```ts
|
```ts
|
||||||
const LIBRARY_KEY = Symbol('LIBRARY');
|
const LIBRARY_KEY = Symbol('LIBRARY');
|
||||||
export function setLibraryState(libraries: Library[]) { return setContext(LIBRARY_KEY, new LibraryState(libraries)); }
|
export function setLibraryState(libraries: Library[]) {
|
||||||
export function getLibraryState() { return getContext<ReturnType<typeof setLibraryState>>(LIBRARY_KEY); }
|
return setContext(LIBRARY_KEY, new LibraryState(libraries));
|
||||||
|
}
|
||||||
|
export function getLibraryState() {
|
||||||
|
return getContext<ReturnType<typeof setLibraryState>>(LIBRARY_KEY);
|
||||||
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Follow that pattern rather than introducing stores. `library.svelte.ts` is the reference — including
|
Follow that pattern rather than introducing stores. `library.svelte.ts` is the reference — including
|
||||||
@@ -79,10 +86,56 @@ its optimistic-delete-with-rollback and toast handling. `bookCollection` / `book
|
|||||||
`@ieedan/shadcn-svelte-extras` (`jsrepo.json`). Treat as generated: add components with the CLIs
|
`@ieedan/shadcn-svelte-extras` (`jsrepo.json`). Treat as generated: add components with the CLIs
|
||||||
rather than hand-writing them, and prefer wrapping over editing.
|
rather than hand-writing them, and prefer wrapping over editing.
|
||||||
- App components live in `forms/`, `layout/`, `view/` (browser, grid/list/table, filters, sort) and
|
- App components live in `forms/`, `layout/`, `view/` (browser, grid/list/table, filters, sort) and
|
||||||
`reader/` (epub reader + chapter sidebar).
|
`reader/` (see [The readers](#the-readers)).
|
||||||
- `cn()` from `$lib/utils` merges Tailwind classes; the `WithElementRef` / `WithoutChild` helpers
|
- `cn()` from `$lib/utils` merges Tailwind classes; the `WithElementRef` / `WithoutChild` helpers
|
||||||
there are the shadcn prop-typing conventions.
|
there are the shadcn prop-typing conventions.
|
||||||
|
|
||||||
|
## The readers
|
||||||
|
|
||||||
|
**PDF** is the pdf.js viewer vendored under `static/pdfjs/`, pointed at by an iframe. Untouched by
|
||||||
|
the EPUB work; leave it alone unless the task is about PDFs.
|
||||||
|
|
||||||
|
**EPUB** is built on `foliate-js`, copied verbatim into `src/lib/vendor/foliate-js/` by
|
||||||
|
`scripts/vendor-foliate.sh` (pinned commit; see `src/lib/vendor/foliate-js/README.chitai.md`).
|
||||||
|
Upstream has no npm release and recommends a submodule; this repo has none and already vendors
|
||||||
|
pdf.js the same way, so it is copied instead. Only the import closure reachable from `view.js` is
|
||||||
|
vendored, and **`pdf.js` in that directory is our stub, not upstream's** — the real one imports a
|
||||||
|
bare `@pdfjs/pdf.min.mjs` that Rollup resolves at build time even though the path never runs.
|
||||||
|
|
||||||
|
Layout:
|
||||||
|
|
||||||
|
| Path | What |
|
||||||
|
| ------------------------------------------ | ---------------------------------------------------------------------------- |
|
||||||
|
| `lib/vendor/foliate-js/` | The engine. Do not edit — `vendor-foliate.sh` overwrites it. |
|
||||||
|
| `lib/reader/foliate.ts` | Lazy loader for the custom elements. The only thing that imports `$foliate`. |
|
||||||
|
| `lib/reader/settings.ts` · `stylesheet.ts` | Defaults/bounds, and the CSS injected into the book. |
|
||||||
|
| `lib/reader/progress.ts` | Debounced progress writer with a `sendBeacon` flush. |
|
||||||
|
| `lib/state/reader-settings.svelte.ts` | Settings state, persisted to `localStorage`. |
|
||||||
|
| `components/reader/foliate-view.svelte` | Wraps `<foliate-view>`; owns the imperative lifecycle. |
|
||||||
|
| `components/reader/epub-reader.svelte` | The shell: chrome, TOC, errors, progress. |
|
||||||
|
|
||||||
|
Things that will bite:
|
||||||
|
|
||||||
|
- **`$foliate` is a Vite-only alias.** It is deliberately absent from `kit.alias` and tsconfig
|
||||||
|
`paths` so TypeScript cannot resolve it and falls back to the ambient declaration in
|
||||||
|
`lib/reader/foliate-js.d.ts`; `src/lib/vendor` is also in tsconfig `exclude`. Without both,
|
||||||
|
`checkJs` walks ~11k lines of untyped JS. The declaration file must **not** be named `foliate.d.ts`
|
||||||
|
— beside `foliate.ts`, TypeScript takes it for that file's emitted declaration and drops it.
|
||||||
|
- **Never import the vendored code at module scope.** `view.js` calls `customElements.define` and
|
||||||
|
subclasses `HTMLElement` on import, so it must stay behind `loadFoliate()` inside `onMount`. SSR is
|
||||||
|
otherwise on for the reader route.
|
||||||
|
- **Sections render in iframes, which swallow key events.** Keyboard handlers are bound per section
|
||||||
|
document on the `load` event, and modifier combinations are replayed onto the host window so app
|
||||||
|
shortcuts (the sidebar's ctrl+B) still work while reading.
|
||||||
|
- **Renderer settings split two ways.** Flow, gap, margins, column count and line width are
|
||||||
|
_attributes_ set with `setAttribute` (there is no JS property API, no `margin` shorthand and no
|
||||||
|
`spread` — a spread is `max-column-count: 2`). Typography is CSS passed to `renderer.setStyles`,
|
||||||
|
which takes a `[before, after]` pair: the first is prepended to the section head so the book
|
||||||
|
overrides it, the second appended so it wins. User settings belong in the second, with
|
||||||
|
`!important`, or the book's own CSS beats them.
|
||||||
|
- **Progress needs no locations pre-pass.** `relocate` carries both a CFI and an overall `fraction`,
|
||||||
|
which map straight onto `epub_cfi` and `percentage`.
|
||||||
|
|
||||||
## Routing
|
## Routing
|
||||||
|
|
||||||
Route groups carry the layout structure:
|
Route groups carry the layout structure:
|
||||||
@@ -95,21 +148,38 @@ Route groups carry the layout structure:
|
|||||||
## Conventions
|
## Conventions
|
||||||
|
|
||||||
Prettier (`.prettierrc`): tabs, single quotes, no trailing commas, 100 columns, with the Svelte and
|
Prettier (`.prettierrc`): tabs, single quotes, no trailing commas, 100 columns, with the Svelte and
|
||||||
Tailwind plugins. Run `pnpm check` (svelte-check) and `pnpm lint` before considering work done.
|
Tailwind plugins. Run `pnpm check` (svelte-check) and `pnpm lint` before considering work done —
|
||||||
|
but take a baseline first, because neither is clean (see below).
|
||||||
|
|
||||||
|
`src/lib/vendor/` is excluded from Prettier, ESLint and svelte-check. Don't reformat vendored code.
|
||||||
|
|
||||||
## Known rough edges
|
## Known rough edges
|
||||||
|
|
||||||
Observed in the current tree — don't mistake these for intentional patterns to copy:
|
Observed in the current tree — don't mistake these for intentional patterns to copy:
|
||||||
|
|
||||||
- `src/lib/schema/openapi/schema.d.ts` is **stale** — it predates the `BookProgress` rework and has
|
- **`pnpm check` is clean as of 2026-08-17 and CI blocks on it** — 0 errors, 0 warnings. Any error
|
||||||
no `BookProgressRead`, so `book.progress` types as `{}`. This is the source of most of the ~104
|
you see is yours. `src/lib/schema/openapi/schema.d.ts` is
|
||||||
errors `pnpm check` reports on a clean tree; regenerating it should clear them. Get a baseline
|
current; regenerate it after any backend API change, with
|
||||||
before assuming an error is yours.
|
`pnpm exec openapi-typescript http://localhost:8000/schema/openapi.json -o src/lib/schema/openapi/schema.d.ts`
|
||||||
|
against a backend running **your** branch — a stale server silently writes a stale file.
|
||||||
|
- **Prettier is clean and CI blocks on it** — run `pnpm format` before finishing. Two things it
|
||||||
|
must not touch are in `.prettierignore`: the vendored foliate-js, and
|
||||||
|
`src/lib/schema/openapi/schema.d.ts`, which `openapi-typescript` regenerates in its own style.
|
||||||
|
- `pnpm exec eslint .` reports **2 errors as of 2026-08-17**, both `svelte/no-at-html-tags` in
|
||||||
|
`collapsible-text.svelte`. They are a genuine XSS hole, not a lint nit — see `TODO.md`. CI runs
|
||||||
|
eslint non-blocking (`continue-on-error`) only until that is fixed. `static/pdfjs/` is ignored
|
||||||
|
alongside `src/lib/vendor/` — it is vendored too, and linting it produced 1717 further errors.
|
||||||
|
- Two rules are off for `**/*.svelte` in `eslint.config.js` because they predate runes and
|
||||||
|
misread them: `no-useless-assignment` (every `$bindable()` default) and
|
||||||
|
`@typescript-eslint/no-unused-expressions` (a bare `book;` declaring an `$effect` dependency).
|
||||||
|
A leading underscore marks an intentionally unused binding.
|
||||||
- `src/routes/api/[...path]/+server.ts` — all four handlers are annotated `RequestHandler` while the
|
- `src/routes/api/[...path]/+server.ts` — all four handlers are annotated `RequestHandler` while the
|
||||||
import of that type is commented out at line 4.
|
import of that type is commented out at line 4. It also buffers whole **responses** with
|
||||||
- `src/app.d.ts` — `App.Locals["user"]` is typed from `lucide-svelte`'s `User` *icon* component
|
`arrayBuffer()` and forwards no `Range` header, so book downloads are not streamed. **Requests**
|
||||||
|
are streamed — POST and PATCH pass `request.body` through with `duplex: 'half'` (see `bodyOf`),
|
||||||
|
because a zipped Calibre library upload cannot be held in this process. The response side is
|
||||||
|
still buffered; see `TODO.md`.
|
||||||
|
- `src/app.d.ts` — `App.Locals["user"]` is typed from `lucide-svelte`'s `User` _icon_ component
|
||||||
rather than the `User` interface in `$lib/server/auth`.
|
rather than the `User` interface in `$lib/server/auth`.
|
||||||
- Uncommitted work in progress (as of 2026-08-10): library icons, spanning
|
- No CSP, which foliate's README asks for because EPUBs can carry scripts. See `TODO.md` for why it
|
||||||
`components/ui/icon-picker/`, the newly vendored `components/ui/popover/`,
|
is not enabled yet.
|
||||||
`forms/library-create-form.svelte`, `layout/library-switcher.svelte` and `schema/library.ts`.
|
|
||||||
Prefer not to refactor those files mid-flight.
|
|
||||||
|
|||||||
+7
-2
@@ -11,7 +11,9 @@ WORKDIR /app
|
|||||||
|
|
||||||
FROM base AS prod-deps
|
FROM base AS prod-deps
|
||||||
|
|
||||||
COPY pnpm-lock.yaml ./
|
# pnpm-workspace.yaml carries onlyBuiltDependencies; without it pnpm blocks the postinstall
|
||||||
|
# scripts it lists, so the install here would not match a local one.
|
||||||
|
COPY pnpm-lock.yaml pnpm-workspace.yaml ./
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/pnpm/store \
|
RUN --mount=type=cache,target=/pnpm/store \
|
||||||
pnpm fetch --frozen-lockfile
|
pnpm fetch --frozen-lockfile
|
||||||
@@ -25,7 +27,7 @@ RUN --mount=type=cache,target=/pnpm/store \
|
|||||||
|
|
||||||
FROM base AS build
|
FROM base AS build
|
||||||
|
|
||||||
COPY pnpm-lock.yaml package.json ./
|
COPY pnpm-lock.yaml pnpm-workspace.yaml package.json ./
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/pnpm/store \
|
RUN --mount=type=cache,target=/pnpm/store \
|
||||||
pnpm install --frozen-lockfile
|
pnpm install --frozen-lockfile
|
||||||
@@ -54,4 +56,7 @@ EXPOSE 3000
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
|
HEALTHCHECK --interval=10s --timeout=3s --retries=5 --start-period=10s \
|
||||||
|
CMD ["node", "-e", "fetch(`http://127.0.0.1:${process.env.PORT || 3000}/login`).then(r => process.exit(r.ok ? 0 : 1)).catch(() => process.exit(1))"]
|
||||||
|
|
||||||
CMD [ "node", "build" ]
|
CMD [ "node", "build" ]
|
||||||
|
|||||||
@@ -13,7 +13,9 @@ const gitignorePath = fileURLToPath(new URL('./.gitignore', import.meta.url));
|
|||||||
export default defineConfig(
|
export default defineConfig(
|
||||||
includeIgnoreFile(gitignorePath),
|
includeIgnoreFile(gitignorePath),
|
||||||
// Vendored third-party source. Tracked, so .gitignore does not cover it.
|
// Vendored third-party source. Tracked, so .gitignore does not cover it.
|
||||||
{ ignores: ['src/lib/vendor/**'] },
|
// static/pdfjs is the pdf.js viewer, vendored the same way as src/lib/vendor/foliate-js;
|
||||||
|
// linting it produced 1717 of the 1804 errors this config used to report.
|
||||||
|
{ ignores: ['src/lib/vendor/**', 'static/pdfjs/**'] },
|
||||||
js.configs.recommended,
|
js.configs.recommended,
|
||||||
...ts.configs.recommended,
|
...ts.configs.recommended,
|
||||||
...svelte.configs.recommended,
|
...svelte.configs.recommended,
|
||||||
@@ -26,7 +28,38 @@ export default defineConfig(
|
|||||||
rules: {
|
rules: {
|
||||||
// typescript-eslint strongly recommend that you do not use the no-undef lint rule on TypeScript projects.
|
// typescript-eslint strongly recommend that you do not use the no-undef lint rule on TypeScript projects.
|
||||||
// see: https://typescript-eslint.io/troubleshooting/faqs/eslint/#i-get-errors-from-the-no-undef-rule-about-global-variables-not-being-defined-even-though-there-are-no-typescript-errors
|
// see: https://typescript-eslint.io/troubleshooting/faqs/eslint/#i-get-errors-from-the-no-undef-rule-about-global-variables-not-being-defined-even-though-there-are-no-typescript-errors
|
||||||
'no-undef': 'off'
|
'no-undef': 'off',
|
||||||
|
// A leading underscore marks a binding that exists to hold a position — a callback
|
||||||
|
// parameter the signature requires, or the discarded half of a destructure.
|
||||||
|
'@typescript-eslint/no-unused-vars': [
|
||||||
|
'error',
|
||||||
|
{
|
||||||
|
argsIgnorePattern: '^_',
|
||||||
|
varsIgnorePattern: '^_',
|
||||||
|
caughtErrorsIgnorePattern: '^_',
|
||||||
|
destructuredArrayIgnorePattern: '^_'
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
// Generated shadcn components take href as a prop and cannot resolve it —
|
||||||
|
// that is the caller's job. Editing them here would be lost on regeneration.
|
||||||
|
files: ['src/lib/components/ui/**'],
|
||||||
|
rules: { 'svelte/no-navigation-without-resolve': 'off' }
|
||||||
|
},
|
||||||
|
{
|
||||||
|
// no-useless-assignment joined eslint:recommended in ESLint 10, and its flow analysis
|
||||||
|
// does not model runes: it reads `let { ref = $bindable(null) } = $props()` as a value
|
||||||
|
// that is never read. Deleting the default, as it suggests, breaks the binding.
|
||||||
|
//
|
||||||
|
// no-unused-expressions is off for the same reason: a bare `book;` inside an $effect is
|
||||||
|
// how a reactive dependency is declared when the read would otherwise be untracked.
|
||||||
|
// Removing the statement stops the effect re-running.
|
||||||
|
files: ['**/*.svelte'],
|
||||||
|
rules: {
|
||||||
|
'no-useless-assignment': 'off',
|
||||||
|
'@typescript-eslint/no-unused-expressions': 'off'
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
|
|||||||
+28
-28
@@ -2,6 +2,7 @@
|
|||||||
"name": "chitai-web",
|
"name": "chitai-web",
|
||||||
"private": true,
|
"private": true,
|
||||||
"version": "0.0.1",
|
"version": "0.0.1",
|
||||||
|
"packageManager": "pnpm@11.20.0",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"dev": "vite dev",
|
"dev": "vite dev",
|
||||||
@@ -14,43 +15,42 @@
|
|||||||
"lint": "prettier --check . && eslint ."
|
"lint": "prettier --check . && eslint ."
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
"@eslint/compat": "^1.4.1",
|
"@eslint/compat": "^2.1.0",
|
||||||
"@eslint/js": "^9.39.4",
|
"@eslint/js": "^10.0.1",
|
||||||
"@iconify/svelte": "^5.2.1",
|
"@iconify/svelte": "^5.2.2",
|
||||||
"@internationalized/date": "^3.12.0",
|
"@internationalized/date": "^3.12.3",
|
||||||
"@lucide/svelte": "^0.544.0",
|
"@lucide/svelte": "^1.31.0",
|
||||||
"@sveltejs/adapter-node": "^5.5.4",
|
"@sveltejs/adapter-node": "^5.5.7",
|
||||||
"@sveltejs/kit": "^2.53.4",
|
"@sveltejs/kit": "^2.70.2",
|
||||||
"@sveltejs/vite-plugin-svelte": "^6.2.4",
|
"@sveltejs/vite-plugin-svelte": "^7.3.0",
|
||||||
"@tailwindcss/vite": "^4.2.1",
|
"@tailwindcss/vite": "^4.3.3",
|
||||||
"@types/node": "^22.19.15",
|
"@types/node": "^26.2.0",
|
||||||
"bits-ui": "^2.16.3",
|
"bits-ui": "^2.18.1",
|
||||||
"clsx": "^2.1.1",
|
"clsx": "^2.1.1",
|
||||||
"eslint": "^9.39.4",
|
"eslint": "^10.8.1",
|
||||||
"eslint-config-prettier": "^10.1.8",
|
"eslint-config-prettier": "^10.1.8",
|
||||||
"eslint-plugin-svelte": "^3.15.0",
|
"eslint-plugin-svelte": "^3.23.0",
|
||||||
"globals": "^16.5.0",
|
"globals": "^17.11.0",
|
||||||
"jsrepo": "^2.5.2",
|
"jsrepo": "^3.8.1",
|
||||||
"openapi-typescript": "^7.13.0",
|
"openapi-typescript": "^7.13.0",
|
||||||
"prettier": "^3.8.1",
|
"prettier": "^3.9.6",
|
||||||
"prettier-plugin-svelte": "^3.5.2",
|
"prettier-plugin-svelte": "^4.1.1",
|
||||||
"prettier-plugin-tailwindcss": "^0.8.1",
|
"prettier-plugin-tailwindcss": "^0.8.1",
|
||||||
"svelte": "^5.53.7",
|
"svelte": "^5.56.9",
|
||||||
"svelte-check": "^4.4.5",
|
"svelte-check": "^4.7.6",
|
||||||
"tailwind-merge": "^3.5.0",
|
"tailwind-merge": "^3.6.0",
|
||||||
"tailwind-scrollbar": "^4.0.2",
|
"tailwind-scrollbar": "^4.0.2",
|
||||||
"tailwind-variants": "^3.2.2",
|
"tailwind-variants": "^3.3.1",
|
||||||
"tailwindcss": "^4.2.1",
|
"tailwindcss": "^4.3.3",
|
||||||
"tw-animate-css": "^1.4.0",
|
"tw-animate-css": "^1.4.0",
|
||||||
"typescript": "^5.9.3",
|
"typescript": "^6.0.3",
|
||||||
"typescript-eslint": "^8.56.1",
|
"typescript-eslint": "^8.67.0",
|
||||||
"vite": "^7.3.1"
|
"vite": "^8.2.1"
|
||||||
},
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"construct-style-sheets-polyfill": "^3.1.0",
|
"construct-style-sheets-polyfill": "^3.1.0",
|
||||||
"epubjs": "^0.3.93",
|
|
||||||
"mode-watcher": "^1.1.0",
|
"mode-watcher": "^1.1.0",
|
||||||
"svelte-sonner": "^1.0.8",
|
"svelte-sonner": "^1.2.1",
|
||||||
"zod": "^4.3.6"
|
"zod": "^4.4.3"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Generated
+1592
-2251
File diff suppressed because it is too large
Load Diff
@@ -10,7 +10,8 @@
|
|||||||
--radius: 0.625rem;
|
--radius: 0.625rem;
|
||||||
|
|
||||||
/* Typography — system stacks, so nothing depends on a CDN or a webfont build. */
|
/* Typography — system stacks, so nothing depends on a CDN or a webfont build. */
|
||||||
--app-font-sans: system-ui, -apple-system, 'Segoe UI', Roboto, 'Helvetica Neue', Arial, sans-serif;
|
--app-font-sans:
|
||||||
|
system-ui, -apple-system, 'Segoe UI', Roboto, 'Helvetica Neue', Arial, sans-serif;
|
||||||
--app-font-serif: Georgia, 'Iowan Old Style', 'Times New Roman', serif;
|
--app-font-serif: Georgia, 'Iowan Old Style', 'Times New Roman', serif;
|
||||||
--app-font-mono: ui-monospace, 'SF Mono', 'Cascadia Mono', Menlo, Consolas, monospace;
|
--app-font-mono: ui-monospace, 'SF Mono', 'Cascadia Mono', Menlo, Consolas, monospace;
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ import { BACKEND_API_URL } from '$lib/server/config';
|
|||||||
import { invalid, redirect } from '@sveltejs/kit';
|
import { invalid, redirect } from '@sveltejs/kit';
|
||||||
|
|
||||||
export const login = form(loginSchema, async (data, issue) => {
|
export const login = form(loginSchema, async (data, issue) => {
|
||||||
const { cookies, locals } = getRequestEvent();
|
const { cookies } = getRequestEvent();
|
||||||
|
|
||||||
// Create URL-encoded form data
|
// Create URL-encoded form data
|
||||||
const formData = new URLSearchParams();
|
const formData = new URLSearchParams();
|
||||||
|
|||||||
@@ -8,12 +8,31 @@ import {
|
|||||||
deleteBooksSchema,
|
deleteBooksSchema,
|
||||||
editBookMetadataSchema,
|
editBookMetadataSchema,
|
||||||
updateBookProgressSchema,
|
updateBookProgressSchema,
|
||||||
|
duplicateDismissalSchema,
|
||||||
|
bookMergeSchema,
|
||||||
type Book,
|
type Book,
|
||||||
|
type BooksUploadResult,
|
||||||
|
type DuplicateBookGroup,
|
||||||
bookFilesUpload
|
bookFilesUpload
|
||||||
} from '$lib/schema/index';
|
} from '$lib/schema/index';
|
||||||
import { stringCoerce, type PaginatedResponse } from '$lib/schema/common';
|
import { stringCoerce, type PaginatedResponse } from '$lib/schema/common';
|
||||||
import { createQueryParams } from '$lib/utils';
|
import { createQueryParams } from '$lib/utils';
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The backend's own message for a failed response, rather than its JSON envelope.
|
||||||
|
*
|
||||||
|
* A refused duplicate answers 409 with a `detail` worth reading and the offending
|
||||||
|
* files in `extra`; passing the body through whole puts JSON in front of the reader.
|
||||||
|
*/
|
||||||
|
function detailOf(body: string): string {
|
||||||
|
try {
|
||||||
|
const parsed = JSON.parse(body);
|
||||||
|
return typeof parsed?.detail === 'string' ? parsed.detail : body;
|
||||||
|
} catch {
|
||||||
|
return body;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
export const getBook = query(stringCoerce, async (id): Promise<Book> => {
|
export const getBook = query(stringCoerce, async (id): Promise<Book> => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
@@ -68,26 +87,29 @@ export const updateBookCover = form(bookCoverUpload, async ({ book_id, file }) =
|
|||||||
return await response.json();
|
return await response.json();
|
||||||
});
|
});
|
||||||
|
|
||||||
export const uploadBooks = form(booksUpload, async ({ library_id, files }) => {
|
export const uploadBooks = form(
|
||||||
const { locals } = getRequestEvent();
|
booksUpload,
|
||||||
|
async ({ library_id, files }): Promise<BooksUploadResult> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const formData = new FormData();
|
const formData = new FormData();
|
||||||
files.forEach((file) => {
|
files.forEach((file) => {
|
||||||
formData.append('files', file);
|
formData.append('files', file);
|
||||||
});
|
});
|
||||||
|
|
||||||
const response = await locals.api.postMultipart(
|
const response = await locals.api.postMultipart(
|
||||||
`/books/fromFiles?library_id=${library_id}`,
|
`/books/fromFiles?library_id=${library_id}`,
|
||||||
formData
|
formData
|
||||||
);
|
);
|
||||||
|
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
const message = await response.text();
|
const message = await response.text();
|
||||||
error(response.status, message);
|
error(response.status, message);
|
||||||
|
}
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
}
|
}
|
||||||
|
);
|
||||||
return await response.json();
|
|
||||||
});
|
|
||||||
|
|
||||||
export const uploadBookFiles = form(bookFilesUpload, async ({ book_id, files }) => {
|
export const uploadBookFiles = form(bookFilesUpload, async ({ book_id, files }) => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
@@ -100,8 +122,9 @@ export const uploadBookFiles = form(bookFilesUpload, async ({ book_id, files })
|
|||||||
const response = await locals.api.postMultipart(`/books/${book_id}/files`, formData);
|
const response = await locals.api.postMultipart(`/books/${book_id}/files`, formData);
|
||||||
|
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
const message = await response.text();
|
// 409 here means the file is already stored under a different book, which is
|
||||||
error(response.status, message);
|
// something the reader can act on — so the message has to survive the trip.
|
||||||
|
error(response.status, detailOf(await response.text()));
|
||||||
}
|
}
|
||||||
|
|
||||||
return await response.json();
|
return await response.json();
|
||||||
@@ -133,6 +156,65 @@ export const deleteBookFiles = command(deleteBookFilesSchema, async ({ book_id,
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Books already in the library that look like copies of one another.
|
||||||
|
*
|
||||||
|
* Metadata only, so every group is a question rather than a verdict — which is why
|
||||||
|
* the screen it feeds offers a way to disagree.
|
||||||
|
*/
|
||||||
|
export const listDuplicateBooks = query(
|
||||||
|
stringCoerce,
|
||||||
|
async (libraryId): Promise<DuplicateBookGroup[]> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const response = await locals.api.get(`/books/duplicate-books?library_id=${libraryId}`);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fold several books into one and delete the records folded in.
|
||||||
|
*
|
||||||
|
* Irreversible, so the caller is expected to have shown what is about to happen.
|
||||||
|
*/
|
||||||
|
export const mergeBooks = command(
|
||||||
|
bookMergeSchema,
|
||||||
|
async ({ library_id, ...data }): Promise<Book> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const response = await locals.api.post(`/books/merge?library_id=${library_id}`, data);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
|
}
|
||||||
|
);
|
||||||
|
|
||||||
|
/** Record that two books are not the same book, so the pair stops being proposed. */
|
||||||
|
export const dismissDuplicateBooks = command(duplicateDismissalSchema, async (data) => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const response = await locals.api.post('/books/duplicate-books/dismissals', data);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
});
|
||||||
|
|
||||||
|
/** Undo a dismissal, so the pair is proposed again. */
|
||||||
|
export const restoreDuplicateBooks = command(duplicateDismissalSchema, async (data) => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const params = createQueryParams(data);
|
||||||
|
|
||||||
|
const response = await locals.api.delete(
|
||||||
|
`/books/duplicate-books/dismissals?${params.toString()}`
|
||||||
|
);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
});
|
||||||
|
|
||||||
export const updateBookProgress = command(
|
export const updateBookProgress = command(
|
||||||
updateBookProgressSchema,
|
updateBookProgressSchema,
|
||||||
async ({ book_ids, ...data }) => {
|
async ({ book_ids, ...data }) => {
|
||||||
|
|||||||
@@ -1,5 +1,10 @@
|
|||||||
import { command, getRequestEvent, query } from '$app/server';
|
import { command, getRequestEvent, query } from '$app/server';
|
||||||
import { bookshelfCreate, bookshelfQuerySchema, modifyBooksInShelf, type Bookshelf } from '$lib/schema/bookshelf';
|
import {
|
||||||
|
bookshelfCreate,
|
||||||
|
bookshelfQuerySchema,
|
||||||
|
modifyBooksInShelf,
|
||||||
|
type Bookshelf
|
||||||
|
} from '$lib/schema/bookshelf';
|
||||||
import { createQueryParams } from '$lib/utils';
|
import { createQueryParams } from '$lib/utils';
|
||||||
import { error } from '@sveltejs/kit';
|
import { error } from '@sveltejs/kit';
|
||||||
|
|
||||||
@@ -19,46 +24,51 @@ export const listBookshelves = query(bookshelfQuerySchema, async (data) => {
|
|||||||
return await response.json();
|
return await response.json();
|
||||||
});
|
});
|
||||||
|
|
||||||
export const addBooksToShelf = command(modifyBooksInShelf, async ({ shelf_id, ...data }): Promise<Bookshelf> => {
|
export const addBooksToShelf = command(
|
||||||
const { locals } = getRequestEvent();
|
modifyBooksInShelf,
|
||||||
|
async ({ shelf_id, ...data }): Promise<Bookshelf> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const params = createQueryParams(data);
|
const params = createQueryParams(data);
|
||||||
|
|
||||||
const response = await locals.api.post(`/shelves/${shelf_id}/books?${params.toString()}`, {});
|
const response = await locals.api.post(`/shelves/${shelf_id}/books?${params.toString()}`, {});
|
||||||
|
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
const message = await response.text();
|
const message = await response.text();
|
||||||
error(response.status, message);
|
error(response.status, message);
|
||||||
|
}
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
}
|
}
|
||||||
|
);
|
||||||
|
|
||||||
return await response.json()
|
export const removeBooksFromShelf = command(
|
||||||
});
|
modifyBooksInShelf,
|
||||||
|
async ({ shelf_id, ...data }): Promise<Bookshelf> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
export const removeBooksFromShelf = command(modifyBooksInShelf, async ({ shelf_id, ...data }): Promise<Bookshelf> => {
|
const params = createQueryParams(data);
|
||||||
const { locals } = getRequestEvent();
|
|
||||||
|
|
||||||
const params = createQueryParams(data);
|
const response = await locals.api.delete(`/shelves/${shelf_id}/books?${params.toString()}`);
|
||||||
|
|
||||||
const response = await locals.api.delete(`/shelves/${shelf_id}/books?${params.toString()}`);
|
if (!response.ok) {
|
||||||
|
const message = await response.text();
|
||||||
|
error(response.status, message);
|
||||||
|
}
|
||||||
|
|
||||||
if (!response.ok) {
|
return await response.json();
|
||||||
const message = await response.text();
|
|
||||||
error(response.status, message);
|
|
||||||
}
|
}
|
||||||
|
);
|
||||||
return await response.json()
|
|
||||||
});
|
|
||||||
|
|
||||||
|
|
||||||
export const createBookshelf = command(bookshelfCreate, async (data) => {
|
export const createBookshelf = command(bookshelfCreate, async (data) => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const response = await locals.api.post(`/shelves`, data)
|
const response = await locals.api.post(`/shelves`, data);
|
||||||
|
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
const message = await response.text();
|
const message = await response.text();
|
||||||
error(response.status, message);
|
error(response.status, message);
|
||||||
}
|
}
|
||||||
|
|
||||||
return await response.json()
|
return await response.json();
|
||||||
})
|
});
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
import { command, getRequestEvent, query } from '$app/server';
|
||||||
|
import { error } from '@sveltejs/kit';
|
||||||
|
|
||||||
|
import { stringCoerce } from '$lib/schema/common';
|
||||||
|
import type { CalibreImport } from '$lib/schema/library';
|
||||||
|
|
||||||
|
/** The backend's own message for a failed response, rather than its JSON envelope. */
|
||||||
|
function detailOf(body: string): string {
|
||||||
|
try {
|
||||||
|
const parsed = JSON.parse(body);
|
||||||
|
return typeof parsed?.detail === 'string' ? parsed.detail : body;
|
||||||
|
} catch {
|
||||||
|
return body;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Where an import has got to.
|
||||||
|
*
|
||||||
|
* A `query` rather than a `command` so it can be refreshed, but it is polled on a timer
|
||||||
|
* rather than cached — the answer changes on its own.
|
||||||
|
*
|
||||||
|
* Starting an import is deliberately **not** here: the archive goes straight to the
|
||||||
|
* backend through the proxy, so it never passes through this process. See the import
|
||||||
|
* screen's `upload`.
|
||||||
|
*/
|
||||||
|
export const getCalibreImport = query(stringCoerce, async (jobId): Promise<CalibreImport> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const response = await locals.api.get(`/libraries/imports/${jobId}`);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
|
});
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Ask an import to stop after the book it is on.
|
||||||
|
*
|
||||||
|
* Not an abort: a book abandoned mid-copy would leave files on disk with no row
|
||||||
|
* describing them. Whatever it has imported stays imported.
|
||||||
|
*/
|
||||||
|
export const cancelCalibreImport = command(stringCoerce, async (jobId): Promise<CalibreImport> => {
|
||||||
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
|
const response = await locals.api.delete(`/libraries/imports/${jobId}`);
|
||||||
|
|
||||||
|
if (!response.ok) error(response.status, detailOf(await response.text()));
|
||||||
|
|
||||||
|
return await response.json();
|
||||||
|
});
|
||||||
@@ -4,40 +4,40 @@ import { error } from '@sveltejs/kit';
|
|||||||
import z from 'zod';
|
import z from 'zod';
|
||||||
|
|
||||||
export const listDevices = query(async (): Promise<Device[]> => {
|
export const listDevices = query(async (): Promise<Device[]> => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const response = await locals.api.get(`/devices`);
|
const response = await locals.api.get(`/devices`);
|
||||||
|
|
||||||
if (!response.ok) error(500, 'An unkown error occurred');
|
if (!response.ok) error(500, 'An unkown error occurred');
|
||||||
|
|
||||||
const deviceResult = await response.json();
|
const deviceResult = await response.json();
|
||||||
return deviceResult.items
|
return deviceResult.items;
|
||||||
});
|
});
|
||||||
|
|
||||||
export const createDevice = form(createDeviceSchema, async (data): Promise<Device> => {
|
export const createDevice = form(createDeviceSchema, async (data): Promise<Device> => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const response = await locals.api.post(`/devices`, data);
|
const response = await locals.api.post(`/devices`, data);
|
||||||
|
|
||||||
if (!response.ok) error(500, 'An unknown error occurred');
|
if (!response.ok) error(500, 'An unknown error occurred');
|
||||||
|
|
||||||
return await response.json();
|
return await response.json();
|
||||||
});
|
});
|
||||||
|
|
||||||
export const regenerateDeviceApiKey = command(z.string(), async (deviceId): Promise<Device> => {
|
export const regenerateDeviceApiKey = command(z.string(), async (deviceId): Promise<Device> => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const response = await locals.api.get(`/devices/${deviceId}/regenerate`);
|
const response = await locals.api.get(`/devices/${deviceId}/regenerate`);
|
||||||
|
|
||||||
if (!response.ok) error(500, 'An unknown error occurred');
|
if (!response.ok) error(500, 'An unknown error occurred');
|
||||||
|
|
||||||
return await response.json();
|
return await response.json();
|
||||||
})
|
});
|
||||||
|
|
||||||
export const deleteDevice = command(z.string(), async (deviceId): Promise<void> => {
|
export const deleteDevice = command(z.string(), async (deviceId): Promise<void> => {
|
||||||
const { locals } = getRequestEvent();
|
const { locals } = getRequestEvent();
|
||||||
|
|
||||||
const response = await locals.api.delete(`/devices/${deviceId}`);
|
const response = await locals.api.delete(`/devices/${deviceId}`);
|
||||||
|
|
||||||
if (!response.ok) error(500, 'An unknown error occurred');
|
if (!response.ok) error(500, 'An unknown error occurred');
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ export * from './auth.remote';
|
|||||||
export * from './author.remote';
|
export * from './author.remote';
|
||||||
export * from './book.remote';
|
export * from './book.remote';
|
||||||
export * from './bookshelf.remote';
|
export * from './bookshelf.remote';
|
||||||
|
export * from './calibre-import.remote';
|
||||||
export * from './library.remote';
|
export * from './library.remote';
|
||||||
export * from './publisher.remote';
|
export * from './publisher.remote';
|
||||||
export * from './tag.remote';
|
export * from './tag.remote';
|
||||||
|
|||||||
@@ -25,6 +25,6 @@ export const createLibrary = form(libraryCreateSchema, async (data) => {
|
|||||||
return await response.json();
|
return await response.json();
|
||||||
});
|
});
|
||||||
|
|
||||||
export const deleteLibrary = query('unchecked', async (data) => {
|
export const deleteLibrary = query('unchecked', async (_data) => {
|
||||||
throw new Error('Not implemented');
|
throw new Error('Not implemented');
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -3,8 +3,6 @@
|
|||||||
import { Checkbox } from '$lib/components/ui/checkbox/index.js';
|
import { Checkbox } from '$lib/components/ui/checkbox/index.js';
|
||||||
import { Button } from '$lib/components/ui/button/index.js';
|
import { Button } from '$lib/components/ui/button/index.js';
|
||||||
import * as Dialog from '$lib/components/ui/dialog/index.js';
|
import * as Dialog from '$lib/components/ui/dialog/index.js';
|
||||||
import { getBookSelectionState } from '$lib/state/bookSelection.svelte';
|
|
||||||
import { getBookOperationsState } from '$lib/state/bookOperations.svelte';
|
|
||||||
|
|
||||||
let {
|
let {
|
||||||
open = $bindable(),
|
open = $bindable(),
|
||||||
@@ -13,12 +11,9 @@
|
|||||||
}: {
|
}: {
|
||||||
open: boolean;
|
open: boolean;
|
||||||
title?: string;
|
title?: string;
|
||||||
deleteFn: (deleteFiles: boolean) => {};
|
deleteFn: (deleteFiles: boolean) => void | Promise<void>;
|
||||||
} = $props();
|
} = $props();
|
||||||
|
|
||||||
const selectedState = getBookSelectionState();
|
|
||||||
const bookOps = getBookOperationsState();
|
|
||||||
|
|
||||||
let deleteFiles = $state(false);
|
let deleteFiles = $state(false);
|
||||||
</script>
|
</script>
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,223 @@
|
|||||||
|
<script lang="ts">
|
||||||
|
import { FileUp, Folder, FileText } from '@lucide/svelte';
|
||||||
|
|
||||||
|
import { Button } from '$lib/components/ui/button/index.js';
|
||||||
|
import type { FileRejectedReason } from '$lib/components/ui/file-drop-zone';
|
||||||
|
import { cn } from '$lib/utils';
|
||||||
|
|
||||||
|
let {
|
||||||
|
onUpload,
|
||||||
|
onFileRejected,
|
||||||
|
accept,
|
||||||
|
maxFileSize,
|
||||||
|
disabled = false,
|
||||||
|
class: className
|
||||||
|
}: {
|
||||||
|
onUpload: (files: File[]) => Promise<void> | void;
|
||||||
|
onFileRejected?: (opts: { reason: FileRejectedReason; file: File }) => void;
|
||||||
|
/** Comma separated extensions and/or MIME types, as the `accept` attribute takes. */
|
||||||
|
accept?: string;
|
||||||
|
/** Bytes. */
|
||||||
|
maxFileSize?: number;
|
||||||
|
disabled?: boolean;
|
||||||
|
class?: string;
|
||||||
|
} = $props();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Two inputs rather than one.
|
||||||
|
*
|
||||||
|
* `webkitdirectory` is not a filter — it switches the picker into folder mode,
|
||||||
|
* so a single input can offer files or folders but never both. The drop target
|
||||||
|
* has no such constraint and stays one area.
|
||||||
|
*/
|
||||||
|
let fileInput = $state<HTMLInputElement>();
|
||||||
|
let folderInput = $state<HTMLInputElement>();
|
||||||
|
|
||||||
|
let dragging = $state(false);
|
||||||
|
let busy = $state(false);
|
||||||
|
|
||||||
|
const active = $derived(!disabled && !busy);
|
||||||
|
|
||||||
|
function accepts(file: File): FileRejectedReason | undefined {
|
||||||
|
if (maxFileSize !== undefined && file.size > maxFileSize) return 'Maximum file size exceeded';
|
||||||
|
if (!accept) return undefined;
|
||||||
|
|
||||||
|
const name = file.name.toLowerCase();
|
||||||
|
const type = file.type.toLowerCase();
|
||||||
|
|
||||||
|
const ok = accept
|
||||||
|
.split(',')
|
||||||
|
.map((pattern) => pattern.trim().toLowerCase())
|
||||||
|
.some((pattern) => {
|
||||||
|
// Match on the pattern, not the file's type. Testing `type` here is
|
||||||
|
// what makes MOBI fail: browsers report no MIME type for it, so a
|
||||||
|
// ".mobi" rule never gets compared against the filename.
|
||||||
|
if (pattern.startsWith('.')) return name.endsWith(pattern);
|
||||||
|
if (pattern.endsWith('/*')) return type.startsWith(pattern.slice(0, -1));
|
||||||
|
return type === pattern;
|
||||||
|
});
|
||||||
|
|
||||||
|
return ok ? undefined : 'File type not allowed';
|
||||||
|
}
|
||||||
|
|
||||||
|
/** readEntries hands back at most 100 at a time and signals the end with an empty batch. */
|
||||||
|
function readAll(reader: FileSystemDirectoryReader): Promise<FileSystemEntry[]> {
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
const entries: FileSystemEntry[] = [];
|
||||||
|
|
||||||
|
const next = () =>
|
||||||
|
reader.readEntries((batch) => {
|
||||||
|
if (batch.length === 0) return resolve(entries);
|
||||||
|
entries.push(...batch);
|
||||||
|
next();
|
||||||
|
}, reject);
|
||||||
|
|
||||||
|
next();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Flattens a dropped entry into files, naming each one with its path inside the
|
||||||
|
* dropped folder so it matches what the folder picker puts in
|
||||||
|
* `webkitRelativePath` — which is what the upload form reads to keep structure.
|
||||||
|
*/
|
||||||
|
async function walk(entry: FileSystemEntry, prefix = ''): Promise<File[]> {
|
||||||
|
if (entry.isFile) {
|
||||||
|
const file = await new Promise<File>((resolve, reject) =>
|
||||||
|
(entry as FileSystemFileEntry).file(resolve, reject)
|
||||||
|
);
|
||||||
|
return [new File([file], `${prefix}${file.name}`, { type: file.type })];
|
||||||
|
}
|
||||||
|
|
||||||
|
if (entry.isDirectory) {
|
||||||
|
const entries = await readAll((entry as FileSystemDirectoryEntry).createReader());
|
||||||
|
const nested = await Promise.all(entries.map((e) => walk(e, `${prefix}${entry.name}/`)));
|
||||||
|
return nested.flat();
|
||||||
|
}
|
||||||
|
|
||||||
|
return [];
|
||||||
|
}
|
||||||
|
|
||||||
|
async function handleDrop(event: DragEvent) {
|
||||||
|
event.preventDefault();
|
||||||
|
dragging = false;
|
||||||
|
if (!active) return;
|
||||||
|
|
||||||
|
// Read the entries synchronously: DataTransfer is emptied as soon as this
|
||||||
|
// handler yields, so awaiting first loses everything that was dropped.
|
||||||
|
const entries = Array.from(event.dataTransfer?.items ?? [])
|
||||||
|
.filter((item) => item.kind === 'file')
|
||||||
|
.map((item) => item.webkitGetAsEntry())
|
||||||
|
.filter((entry): entry is FileSystemEntry => entry !== null);
|
||||||
|
|
||||||
|
// Older engines expose no entries; fall back to the flat list, which cannot
|
||||||
|
// contain folders anyway.
|
||||||
|
if (entries.length === 0) {
|
||||||
|
await submit(Array.from(event.dataTransfer?.files ?? []));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
busy = true;
|
||||||
|
try {
|
||||||
|
const nested = await Promise.all(entries.map((entry) => walk(entry)));
|
||||||
|
await submit(nested.flat());
|
||||||
|
} finally {
|
||||||
|
busy = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function handleChange(event: Event) {
|
||||||
|
const input = event.currentTarget as HTMLInputElement;
|
||||||
|
const chosen = Array.from(input.files ?? []);
|
||||||
|
// Reset first so picking the same file twice still fires a change event.
|
||||||
|
input.value = '';
|
||||||
|
await submit(chosen);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function submit(candidates: File[]) {
|
||||||
|
const accepted: File[] = [];
|
||||||
|
|
||||||
|
for (const file of candidates) {
|
||||||
|
const reason = accepts(file);
|
||||||
|
if (reason) onFileRejected?.({ file, reason });
|
||||||
|
else accepted.push(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (accepted.length > 0) await onUpload(accepted);
|
||||||
|
}
|
||||||
|
</script>
|
||||||
|
|
||||||
|
<div
|
||||||
|
role="group"
|
||||||
|
aria-label="Add books"
|
||||||
|
aria-disabled={!active}
|
||||||
|
ondragover={(e) => {
|
||||||
|
e.preventDefault();
|
||||||
|
if (active) dragging = true;
|
||||||
|
}}
|
||||||
|
ondragleave={() => (dragging = false)}
|
||||||
|
ondrop={handleDrop}
|
||||||
|
class={cn(
|
||||||
|
'flex flex-col items-center gap-3 rounded-lg border-2 border-dashed border-border bg-accent/20 p-6 text-center transition-colors',
|
||||||
|
dragging && 'border-primary bg-accent/50',
|
||||||
|
!active && 'opacity-50',
|
||||||
|
className
|
||||||
|
)}
|
||||||
|
>
|
||||||
|
<div
|
||||||
|
class="flex size-12 place-items-center justify-center rounded-full border border-dashed border-border text-muted-foreground"
|
||||||
|
>
|
||||||
|
<FileUp class="size-5" />
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="flex flex-col gap-0.5">
|
||||||
|
<span class="font-medium">
|
||||||
|
{busy ? 'Reading folder…' : 'Drop books here'}
|
||||||
|
</span>
|
||||||
|
<span class="text-sm text-muted-foreground">A folder keeps its structure</span>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<span class="font-mono text-[10px] tracking-widest text-muted-foreground uppercase">or</span>
|
||||||
|
|
||||||
|
<div class="flex flex-wrap justify-center gap-2">
|
||||||
|
<Button
|
||||||
|
type="button"
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
disabled={!active}
|
||||||
|
onclick={() => fileInput?.click()}
|
||||||
|
>
|
||||||
|
<FileText class="size-4" />
|
||||||
|
Choose files
|
||||||
|
</Button>
|
||||||
|
<Button
|
||||||
|
type="button"
|
||||||
|
variant="outline"
|
||||||
|
size="sm"
|
||||||
|
disabled={!active}
|
||||||
|
onclick={() => folderInput?.click()}
|
||||||
|
>
|
||||||
|
<Folder class="size-4" />
|
||||||
|
Choose folder
|
||||||
|
</Button>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<input
|
||||||
|
bind:this={fileInput}
|
||||||
|
type="file"
|
||||||
|
multiple
|
||||||
|
{accept}
|
||||||
|
class="hidden"
|
||||||
|
onchange={handleChange}
|
||||||
|
/>
|
||||||
|
<!-- webkitdirectory is why this needs to be a second input: it turns the
|
||||||
|
picker into a folder chooser rather than filtering what it accepts. -->
|
||||||
|
<input
|
||||||
|
bind:this={folderInput}
|
||||||
|
type="file"
|
||||||
|
multiple
|
||||||
|
webkitdirectory
|
||||||
|
class="hidden"
|
||||||
|
onchange={handleChange}
|
||||||
|
/>
|
||||||
|
</div>
|
||||||
@@ -1,165 +1,231 @@
|
|||||||
<script lang="ts">
|
<script lang="ts">
|
||||||
|
import { untrack } from 'svelte';
|
||||||
|
import { goto } from '$app/navigation';
|
||||||
|
import { resolve } from '$app/paths';
|
||||||
|
import { X } from '@lucide/svelte';
|
||||||
|
|
||||||
import * as Dialog from '$lib/components/ui/dialog/index.js';
|
import * as Dialog from '$lib/components/ui/dialog/index.js';
|
||||||
import * as Field from '$lib/components/ui/field/index.js';
|
import * as Field from '$lib/components/ui/field/index.js';
|
||||||
import { Button } from '$lib/components/ui/button';
|
|
||||||
import * as NativeSelect from '$lib/components/ui/native-select/index.js';
|
import * as NativeSelect from '$lib/components/ui/native-select/index.js';
|
||||||
import {
|
import { Button } from '$lib/components/ui/button';
|
||||||
displaySize,
|
|
||||||
FileDropZone,
|
|
||||||
type FileDropZoneProps
|
|
||||||
} from '$lib/components/ui/file-drop-zone';
|
|
||||||
import { X } from '@lucide/svelte';
|
|
||||||
import { toast } from 'svelte-sonner';
|
|
||||||
|
|
||||||
import { Switch } from '$lib/components/ui/switch/index';
|
import { Switch } from '$lib/components/ui/switch/index';
|
||||||
|
import { displaySize, type FileRejectedReason } from '$lib/components/ui/file-drop-zone';
|
||||||
|
import BookDropZone from './book-drop-zone.svelte';
|
||||||
|
|
||||||
import { tick } from 'svelte';
|
|
||||||
import { Spinner } from '$lib/components/ui/spinner/index';
|
|
||||||
import { getLibraryState } from '$lib/state/library.svelte';
|
import { getLibraryState } from '$lib/state/library.svelte';
|
||||||
import { uploadBooks } from '$lib/api';
|
import { getUploadQueueState } from '$lib/state/upload-queue.svelte';
|
||||||
import type { Book, PaginatedResponse } from '$lib/schema';
|
|
||||||
import { goto } from '$app/navigation';
|
|
||||||
|
|
||||||
let { open = $bindable() }: { open?: boolean } = $props();
|
let { open = $bindable() }: { open?: boolean } = $props();
|
||||||
|
|
||||||
let libraryState = getLibraryState();
|
const libraryState = getLibraryState();
|
||||||
|
const queue = getUploadQueueState();
|
||||||
$effect(() => {
|
|
||||||
uploadBooks.fields.library_id.set(libraryState.activeLibrary!.id);
|
|
||||||
});
|
|
||||||
|
|
||||||
let files = $derived(uploadBooks.fields.files.value() ?? []);
|
|
||||||
|
|
||||||
|
let libraryId = $state<number>(untrack(() => libraryState.activeLibrary!.id));
|
||||||
|
let files = $state<File[]>([]);
|
||||||
let autoUploadOnDrop = $state(true);
|
let autoUploadOnDrop = $state(true);
|
||||||
let navigateOnUpload = $state(true);
|
let navigateOnUpload = $state(true);
|
||||||
let formEl = $state<HTMLFormElement>();
|
|
||||||
|
|
||||||
const onUpload: FileDropZoneProps['onUpload'] = async (uploadedFiles) => {
|
const totalSize = $derived(files.reduce((sum, file) => sum + file.size, 0));
|
||||||
// Rename files to use webkitRelativePath so directory structure is preserved through form submission
|
|
||||||
const renamedFiles = uploadedFiles.map(f =>
|
/**
|
||||||
new File([f], f.webkitRelativePath || f.name, { type: f.type })
|
* Collected rather than raised one at a time. A folder of a few hundred books
|
||||||
|
* carries covers and notes alongside them, and a toast per rejected file
|
||||||
|
* buries the screen.
|
||||||
|
*/
|
||||||
|
let rejected = $state<{ name: string; reason: FileRejectedReason }[]>([]);
|
||||||
|
let showRejected = $state(false);
|
||||||
|
|
||||||
|
// Start each visit clean — a list left over from last time reads as though it
|
||||||
|
// applies to what was just chosen.
|
||||||
|
$effect(() => {
|
||||||
|
if (open) {
|
||||||
|
untrack(() => {
|
||||||
|
files = [];
|
||||||
|
rejected = [];
|
||||||
|
showRejected = false;
|
||||||
|
libraryId = libraryState.activeLibrary!.id;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
const onUpload = async (uploadedFiles: File[]) => {
|
||||||
|
// Rename to the path relative to the chosen folder, which is what decides
|
||||||
|
// how the API groups files into books. Dropped folders already arrive named
|
||||||
|
// this way, since the drop zone builds the path while walking them.
|
||||||
|
const named = uploadedFiles.map(
|
||||||
|
(file) => new File([file], file.webkitRelativePath || file.name, { type: file.type })
|
||||||
);
|
);
|
||||||
uploadBooks.fields.files.set([...Array.from(files), ...renamedFiles]);
|
|
||||||
if (autoUploadOnDrop && files.length > 0) {
|
// Same relative path twice is the same file — dropping a folder a second
|
||||||
await tick();
|
// time should not queue everything again.
|
||||||
formEl?.requestSubmit();
|
const seen = new Set(files.map((file) => file.name));
|
||||||
}
|
files = [...files, ...named.filter((file) => !seen.has(file.name))];
|
||||||
|
|
||||||
|
if (autoUploadOnDrop) await startUpload();
|
||||||
};
|
};
|
||||||
|
|
||||||
const onFileRejected: FileDropZoneProps['onFileRejected'] = async ({ reason, file }) => {
|
const onFileRejected = ({ reason, file }: { reason: FileRejectedReason; file: File }) => {
|
||||||
toast.error(`${file.name} failed to upload!`, { description: reason });
|
rejected = [...rejected, { name: file.webkitRelativePath || file.name, reason }];
|
||||||
};
|
};
|
||||||
|
|
||||||
function navigateToBooks(books: PaginatedResponse<Book>) {
|
/**
|
||||||
|
* Files picked from a folder carry their whole relative path as the name, so
|
||||||
|
* truncating the end would cut off the filename — the only part worth reading.
|
||||||
|
* Split it and let the folder sit on its own, quieter line.
|
||||||
|
*/
|
||||||
|
function splitPath(path: string) {
|
||||||
|
const cut = path.lastIndexOf('/');
|
||||||
|
return cut === -1
|
||||||
|
? { dir: '', name: path }
|
||||||
|
: { dir: path.slice(0, cut), name: path.slice(cut + 1) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Hands the books to the queue and closes.
|
||||||
|
*
|
||||||
|
* Nothing is awaited here: the queue lives in the root layout and reports
|
||||||
|
* through the tray, so the import carries on while the library stays usable.
|
||||||
|
*/
|
||||||
|
function startUpload() {
|
||||||
|
if (files.length === 0) return;
|
||||||
|
|
||||||
|
const queued = files;
|
||||||
|
const target = libraryId;
|
||||||
|
|
||||||
|
files = [];
|
||||||
|
rejected = [];
|
||||||
open = false;
|
open = false;
|
||||||
let libraryId = books.items[0].library_id;
|
|
||||||
libraryState.setActive(libraryId);
|
queue.enqueue(target, queued, ({ created, firstBook }) => {
|
||||||
if (books.items.length === 1) {
|
if (created === 0) return;
|
||||||
goto(`/book/${books.items[0].id}`);
|
|
||||||
} else {
|
const library = libraryState.libraries.find((lib) => lib.id === target);
|
||||||
goto(`/library/${libraryId}/view?orderBy=created_at&sortOrder=desc`);
|
if (library) library.total = (library.total ?? 0) + created;
|
||||||
}
|
|
||||||
|
// Only for a single book. Jumping somewhere after a bulk import would
|
||||||
|
// land minutes after the reader moved on.
|
||||||
|
if (navigateOnUpload && created === 1 && firstBook) {
|
||||||
|
libraryState.setActive(target);
|
||||||
|
void goto(resolve('/(root)/(library)/book/[bookId]', { bookId: String(firstBook.id) }));
|
||||||
|
}
|
||||||
|
});
|
||||||
}
|
}
|
||||||
</script>
|
</script>
|
||||||
|
|
||||||
<Dialog.Root bind:open>
|
<Dialog.Root bind:open>
|
||||||
<Dialog.Content>
|
<!--
|
||||||
{#if uploadBooks.pending}
|
Wider than the default lg: a folder's worth of rows needs the room.
|
||||||
<div class="flex flex-col items-center gap-4">
|
|
||||||
<span class="text-lg font-semibold"
|
|
||||||
>Uploading {uploadBooks.fields.files.value().length} files...</span
|
|
||||||
>
|
|
||||||
<Spinner class="scale-150" />
|
|
||||||
</div>
|
|
||||||
{:else}
|
|
||||||
<Dialog.Header>
|
|
||||||
<Dialog.Title>Upload Books</Dialog.Title>
|
|
||||||
</Dialog.Header>
|
|
||||||
|
|
||||||
<form
|
overflow-hidden and the min-w-0 on the body below are what keep a long
|
||||||
{...uploadBooks.enhance(async ({ submit, form }) => {
|
filename inside the dialog. Dialog.Content is a grid, and grid and flex
|
||||||
try {
|
items default to min-width:auto — they refuse to shrink below their
|
||||||
await submit();
|
content's intrinsic width, so one long name widened the body and pushed it
|
||||||
|
straight through the dialog's edge regardless of any truncate further down.
|
||||||
|
-->
|
||||||
|
<Dialog.Content class="overflow-hidden sm:max-w-2xl">
|
||||||
|
<Dialog.Header>
|
||||||
|
<Dialog.Title>Add books</Dialog.Title>
|
||||||
|
<Dialog.Description>
|
||||||
|
Files or a folder. A folder becomes one book per directory.
|
||||||
|
</Dialog.Description>
|
||||||
|
</Dialog.Header>
|
||||||
|
|
||||||
// Check if there are any validation issues
|
<div class="flex w-full min-w-0 flex-col gap-3">
|
||||||
const issues = uploadBooks.fields.allIssues();
|
<div class="flex flex-col gap-1.5">
|
||||||
if (issues && issues.length > 0) {
|
<Field.Label for="library_id">Library</Field.Label>
|
||||||
return;
|
<NativeSelect.Root id="library_id" bind:value={libraryId} class="w-48">
|
||||||
}
|
{#each libraryState.libraries as library (library.id)}
|
||||||
|
<NativeSelect.Option value={library.id}>{library.name}</NativeSelect.Option>
|
||||||
// Update library book count
|
|
||||||
const count = uploadBooks.result.total
|
|
||||||
libraryState.libraries.find(lib => uploadBooks.fields.library_id.value() == lib.id.toString())!.total += count
|
|
||||||
|
|
||||||
// Reset the files field
|
|
||||||
uploadBooks.fields.files.set([]);
|
|
||||||
toast.success('Books successfully uploaded!');
|
|
||||||
|
|
||||||
if (navigateOnUpload) {
|
|
||||||
navigateToBooks(uploadBooks.result);
|
|
||||||
}
|
|
||||||
} catch (error) {
|
|
||||||
console.error('Failed to upload book: ', error);
|
|
||||||
toast.error('Failed to upload books');
|
|
||||||
}
|
|
||||||
})}
|
|
||||||
bind:this={formEl}
|
|
||||||
enctype="multipart/form-data"
|
|
||||||
class="flex w-full flex-col gap-2 p-4"
|
|
||||||
>
|
|
||||||
<!-- Library select field -->
|
|
||||||
<Field.Label for="library_id">Select Library</Field.Label>
|
|
||||||
<NativeSelect.Root {...uploadBooks.fields.library_id.as('select')} class="w-36">
|
|
||||||
{#each libraryState.libraries as library}
|
|
||||||
<NativeSelect.Option value={library.id}>
|
|
||||||
{library.name}
|
|
||||||
</NativeSelect.Option>
|
|
||||||
{/each}
|
{/each}
|
||||||
</NativeSelect.Root>
|
</NativeSelect.Root>
|
||||||
|
</div>
|
||||||
|
|
||||||
<FileDropZone
|
<BookDropZone
|
||||||
{onUpload}
|
{onUpload}
|
||||||
{onFileRejected}
|
{onFileRejected}
|
||||||
directory={true}
|
accept=".pdf,.epub,.mobi,application/pdf,application/epub+zip,application/x-mobipocket-ebook"
|
||||||
accept=".pdf,.epub,.mobi,application/pdf,application/epub+zip,application/x-mobipocket-ebook"
|
/>
|
||||||
sublabel="Only PDF, EPUB, and MOBI files supported"
|
|
||||||
/>
|
{#if files.length > 0}
|
||||||
<input class="hidden" {...uploadBooks.fields.files.as('file multiple')} />
|
<div class="flex items-baseline justify-between border-b pb-1 text-sm">
|
||||||
<div class="flex max-h-[300px] flex-col gap-2 overflow-y-auto">
|
<span><strong class="tabular-nums">{files.length}</strong> ready to upload</span>
|
||||||
{#each files as file, idx}
|
<span class="font-mono text-xs text-muted-foreground tabular-nums">
|
||||||
<div class="flex place-items-center justify-between gap-2">
|
{displaySize(totalSize)}
|
||||||
<div class="flex flex-col">
|
</span>
|
||||||
<span>{file.name}</span>
|
</div>
|
||||||
<span class="text-xs text-muted-foreground">{displaySize(file.size)}</span>
|
{/if}
|
||||||
</div>
|
|
||||||
<Button
|
<div class="flex max-h-[300px] min-w-0 flex-col gap-2 overflow-y-auto">
|
||||||
variant="outline"
|
{#each files as file, idx (file.name)}
|
||||||
size="icon"
|
{@const location = splitPath(file.name)}
|
||||||
onclick={() => {
|
<div class="flex min-w-0 items-center justify-between gap-2">
|
||||||
uploadBooks.fields.files.set([
|
<!-- flex-1 as well as min-w-0: without a constrained width there is
|
||||||
...Array.from(files).slice(0, idx),
|
nothing for truncate to act against and the row pushes the dialog wide -->
|
||||||
...Array.from(files).slice(idx + 1)
|
<div class="flex min-w-0 flex-1 flex-col">
|
||||||
]);
|
<span class="truncate text-sm" title={file.name}>{location.name}</span>
|
||||||
}}
|
<span class="flex min-w-0 items-baseline gap-2 text-xs text-muted-foreground">
|
||||||
>
|
{#if location.dir}
|
||||||
<X />
|
<span class="truncate font-mono" title={location.dir}>{location.dir}</span>
|
||||||
</Button>
|
{/if}
|
||||||
|
<span class="shrink-0 tabular-nums">{displaySize(file.size)}</span>
|
||||||
|
</span>
|
||||||
</div>
|
</div>
|
||||||
{/each}
|
<Button
|
||||||
</div>
|
variant="outline"
|
||||||
|
size="icon"
|
||||||
|
class="shrink-0"
|
||||||
|
onclick={() => (files = files.filter((_, i) => i !== idx))}
|
||||||
|
>
|
||||||
|
<X />
|
||||||
|
<span class="sr-only">Remove {file.name}</span>
|
||||||
|
</Button>
|
||||||
|
</div>
|
||||||
|
{/each}
|
||||||
|
</div>
|
||||||
|
|
||||||
<div class="flex flex-col gap-2">
|
{#if rejected.length > 0}
|
||||||
<div class="flex items-center gap-2">
|
<div class="rounded-md border border-star/50 bg-star/10 p-2 text-sm">
|
||||||
<Switch bind:checked={autoUploadOnDrop} />
|
<div class="flex items-center justify-between gap-2">
|
||||||
<Field.Label for="auto-upload-on-drop">Auto upload on file drop</Field.Label>
|
<span>
|
||||||
<Button type="submit" class="ml-auto w-fit">Upload</Button>
|
{rejected.length}
|
||||||
</div>
|
{rejected.length === 1 ? 'file was' : 'files were'} skipped
|
||||||
<div class="flex items-center gap-2">
|
</span>
|
||||||
<Switch bind:checked={navigateOnUpload} />
|
<Button variant="ghost" size="sm" onclick={() => (showRejected = !showRejected)}>
|
||||||
<Field.Label for="navigate-to-book">Navigate to book on upload</Field.Label>
|
{showRejected ? 'Hide' : 'Show'}
|
||||||
|
</Button>
|
||||||
</div>
|
</div>
|
||||||
|
{#if showRejected}
|
||||||
|
<ul class="mt-2 flex max-h-32 min-w-0 flex-col gap-1 overflow-y-auto">
|
||||||
|
{#each rejected as entry (entry.name)}
|
||||||
|
<li class="flex min-w-0 justify-between gap-2 text-xs text-muted-foreground">
|
||||||
|
<span class="min-w-0 flex-1 truncate" title={entry.name}>
|
||||||
|
{splitPath(entry.name).name}
|
||||||
|
</span>
|
||||||
|
<span class="shrink-0">{entry.reason}</span>
|
||||||
|
</li>
|
||||||
|
{/each}
|
||||||
|
</ul>
|
||||||
|
{/if}
|
||||||
</div>
|
</div>
|
||||||
</form>
|
{/if}
|
||||||
{/if}
|
|
||||||
|
<div class="flex flex-col gap-2 border-t pt-3">
|
||||||
|
<div class="flex items-center gap-2">
|
||||||
|
<Switch id="auto-upload-on-drop" bind:checked={autoUploadOnDrop} />
|
||||||
|
<Field.Label for="auto-upload-on-drop" class="font-normal">
|
||||||
|
Start as soon as books are added
|
||||||
|
</Field.Label>
|
||||||
|
<Button class="ml-auto w-fit" disabled={files.length === 0} onclick={startUpload}>
|
||||||
|
Upload
|
||||||
|
</Button>
|
||||||
|
</div>
|
||||||
|
<div class="flex items-center gap-2">
|
||||||
|
<Switch id="navigate-to-book" bind:checked={navigateOnUpload} />
|
||||||
|
<Field.Label for="navigate-to-book" class="font-normal">
|
||||||
|
Open the book when a single one is added
|
||||||
|
</Field.Label>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
</Dialog.Content>
|
</Dialog.Content>
|
||||||
</Dialog.Root>
|
</Dialog.Root>
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user