diff --git a/.gitignore b/.gitignore index e36d596..9c9e0e8 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,4 @@ .env .postgres/ -.venv \ No newline at end of file +.venv +tmp/ diff --git a/AGENTS.md b/AGENTS.md index 46473a6..61a0be2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -73,6 +73,9 @@ pytest tests/ # needs Docker (pytest-data ruff format src/ alchemy --config chitai.database.config.config make-migrations alchemy --config chitai.database.config.config upgrade + +# Import a Calibre library. Copies files; --dry-run reports without writing. +litestar --app-dir src/chitai/ calibre-import --library ``` Frontend (from `frontend/`): diff --git a/TODO.md b/TODO.md index 5cf2712..b7ed6cb 100644 --- a/TODO.md +++ b/TODO.md @@ -242,6 +242,10 @@ so the whole file sits in the node process, per concurrent reader, and the brows nothing until it completes. Passing `response.body` straight through restores the stream and is a small change. +This is now **responses only**. The request side was fixed for the Calibre archive upload, +which cannot be held in memory: POST and PATCH pass `request.body` through with +`duplex: 'half'` (`bodyOf` in the same file). The response side is the same shape of fix. + **Range.** The proxy forwards only `Content-Type`, `Content-Disposition` and `Content-Length`. It never sends the client's `Range` upstream, and would drop `Accept-Ranges` and `Content-Range` coming back — a 206 without `Content-Range` is diff --git a/backend/AGENTS.md b/backend/AGENTS.md index 8133b78..5297222 100644 --- a/backend/AGENTS.md +++ b/backend/AGENTS.md @@ -108,6 +108,22 @@ The backend owns files on disk, not just rows: if it differs, moves the directory contents and prunes empty parents. Keep that in mind before changing metadata handling. +### Content types come from the extension, and null means null + +Every ingest path names a file's format with `guess_content_type` (`services/utils.py`), never with +`mimetypes.guess_type` directly and never with what the client said. Python's built-in map answers +`None` for `.mobi`, `.azw`, `.prc`, `.fb2`, `.fbz`, `.lit`, `.lrf` and `.cb7` — most of what a +library imported from elsewhere carries — so `EBOOK_CONTENT_TYPES` fills those in. A browser's +`application/octet-stream` is discarded rather than used as a fallback: it is the client saying it +does not know, and storing it is indistinguishable from having determined a format. + +When nothing can name the extension the column **stays null**. That is the honest answer, and only +one consumer cannot take it: OPDS `Link.type` is a required string, so +`services/opds/opds.py` substitutes `application/octet-stream` at that boundary. Litestar's +`ASGIFileResponse` already does its own fallback, so `get_file` can pass a null straight through. +`FileMetadataRead.content_type` is nullable for the same reason — it was once required, which +turned a stored null into a 500 on a book that was otherwise fine. + ## Duplicate detection Every ingest path screens incoming files against what is already stored, keyed on @@ -126,6 +142,7 @@ The policy differs by how deliberate the import is: | `create_book` (single, with metadata) | All-or-nothing: raises `DuplicateFilesError`, which `controllers/book.py` renders as a **409** carrying the refused files in `extra`. | | `add_files` | A file the book already carries is a no-op; one stored under another book raises `DuplicateFilesError`. | | `create_many_from_existing_files` (consume watcher) | Skips, and **moves the file to `CHITAI_DUPLICATE_PATH//`** — nothing is deleted, and it cannot stay put because `watchfiles` only reports additions. That path must stay outside `consume_path` or the watcher re-imports it and tries to read the directory name as a library slug. | +| `create_many_from_calibre` (Calibre import) | Skips per file, and skips a whole book whose files are all known. Nothing is moved — the source is somebody else's library. This is what makes a re-run a no-op and an interrupted import resumable by running it again. | `allow_duplicates=true` overrides all of it, on every endpoint. Keep that working — the hash is not proof of identity, so a wrong verdict has to be recoverable, and a scripted @@ -242,6 +259,84 @@ under the new rules, with nothing to show that anything is wrong. Copy `normalize_title` learned to strip compact edition markers (`2E`, `5e`). It is idempotent and safe to re-run. +## Importing from Calibre + +Two pieces, deliberately separated: + +- **`services/calibre.py`** reads `metadata.db` and the tree beside it. It knows nothing about + `Book`, `BookService` or a session, so it is testable without Postgres, and it reports what + Calibre wrote rather than what Chitai wants — identifiers come back keyed by `identifiers.type` + verbatim. It also unpacks a zipped library (`extract_calibre_archive`). +- **`BookService.create_many_from_calibre`** does the ingest, next to the two other ingest paths + because it needs the same privates they do (`_reserve_book_path`, `_save_cover_image`, + `_screen_for_duplicates`). The CLI in `cli.py` is a thin wrapper over it. +- **`services/calibre_import.py`** is lifecycle only — the job registry behind the endpoints: + state, progress, cancellation, and a session of its own. + +`docs/calibre-import.md` is the full brief, including the phases not built yet. What matters here: + +- **Files are copied, never moved.** `metadata.db` would go on pointing at files that are gone, + which quietly ruins a library somebody still uses. `copy_file` streams rather than using + `shutil.copy`, which would block the loop for a 40 MB read. +- **The extractors are not run.** This is the one ingest path that trusts its input: Calibre's + catalogue is curated and its filenames are truncated to ~42 characters, so `data.name` locates a + file and the database carries the metadata. The title is stored verbatim for the same reason — + no edition split out, no subtitle guessed. +- **The catalogue is copied before it is read**, and the copy is opened read-write. Calibre may be + running; opening the live file either sees a torn state or needs to recover a write-ahead log, + which read-only access cannot do. `close()` removes the copy in a `finally`, or a failure leaves + a catalogue-sized file in the temp directory. +- **`check_same_thread=False` plus an `asyncio.Lock`.** Every query runs through + `asyncio.to_thread`, which hands out whichever worker is free, so the connection outlives the + thread that opened it. The lock is what makes that safe. Removing either one reintroduces + `SQLite objects created in a thread can only be used in that same thread`, intermittently — + the pool often reuses one thread, so it passes until it does not. +- **One book never costs the run.** A failure is recorded in `CalibreImportResult.failed`, the + session is rolled back so the next book can use it, and the files that book had already copied + are deleted — an orphaned directory would make the next attempt reserve `title (2)` and look as + though it had worked. A cover that PIL cannot open costs the cover, not the book. +- **`calibre-uuid`, not `uuid`.** `books.uuid` is stable for the life of the row, so it is the + durable link back to the source and worth matching on. `uuid` is the name `normalize_identifier` + refuses, because an EPUB regenerates one per build. + +### The API takes an uploaded archive; the CLI takes a path + +**`POST /libraries/{id}/imports/calibre/upload`** is the only way in over HTTP. It takes a zipped +Calibre library, answers **202** with a job handle, and unpacks into a temp directory the job owns. +`GET /libraries/imports/{job_id}` is polled; `DELETE` on the same path stops it. + +There is deliberately **no endpoint that imports from a server path**. A desktop Calibre install is +not on the server, and importing from a path the server can already see is a server-side operation +— which is what `litestar --app-dir src/chitai/ calibre-import --library ` is for, +including its `--dry-run`. Do not add the path endpoint back without being asked: it was built, +then removed on purpose. + +- **The registry is in memory, so it assumes one worker process.** That holds today (`litestar run` + is single-process, and the consume watcher is already an in-process singleton), but the day + `TODO.md`'s "production image runs the development server" item is fixed with a worker count, a + poll can land on a worker that never heard of the job. `services/calibre_import.py` says so at + the top; an `import_jobs` table is the answer when that happens. +- **The job opens its own session.** The request that started it is long gone and its session + closed with it. +- **Cancelling is not aborting.** A flag is read between books, never during one, so a cancelled + import leaves whole books behind and never half of one. `task.cancel()` would abandon a book + mid-copy and leave files with no row describing them. +- **The job deletes its workspace** — the unpacked archive is a second copy of the whole library, + and the books worth keeping have been copied into the library proper by the time it ends. Removed + even when the run failed, since nothing will come back for it. +- **Extraction refuses** an entry pointing outside the archive (zip slip), an archive that will not + fit on disk, and one with no `metadata.db` within three levels. All three answer 400 before a job + exists, rather than as a job that reports FAILED a moment later. +- **The upload is streamed both sides.** The archive reaches disk in chunks rather than being read + whole, and the SvelteKit proxy passes `request.body` through instead of buffering it — see + `frontend/AGENTS.md`. + +Things Calibre does that will produce wrong data if you forget them are documented at the top of +`services/calibre.py` — the `0101-01-01` date sentinel, `|` for a comma in an author name, the +REAL `series_index` that defaults to 1.0 for every book, HTML in `comments`, the views that need +SQLite functions Calibre registers from Python, and `books_pages_link` being both recent and +usually empty. `tests/calibre_fixtures.py` builds a library exercising all of them. + ## KOReader hashing `services/utils.py` reimplements KOReader's partial-MD5 document identifier: 1 KiB samples at diff --git a/backend/migrations/versions/2026-08-17_backfill_file_content_types_d2d69065ede3.py b/backend/migrations/versions/2026-08-17_backfill_file_content_types_d2d69065ede3.py new file mode 100644 index 0000000..2d1a095 --- /dev/null +++ b/backend/migrations/versions/2026-08-17_backfill_file_content_types_d2d69065ede3.py @@ -0,0 +1,114 @@ +"""backfill file content types + +Revision ID: d2d69065ede3 +Revises: 49a9e85a0ffc +Create Date: 2026-08-17 11:37:58.959135 + +""" + +import warnings +from typing import TYPE_CHECKING + +import sqlalchemy as sa +from alembic import op +from advanced_alchemy.types import EncryptedString, EncryptedText, GUID, ORA_JSONB, DateTimeUTC, StoredObject, PasswordHash, FernetBackend +from advanced_alchemy.types.encrypted_string import PGCryptoBackend +from advanced_alchemy.types.password_hash.argon2 import Argon2Hasher +from advanced_alchemy.types.password_hash.passlib import PasslibHasher +from advanced_alchemy.types.password_hash.pwdlib import PwdlibHasher +from sqlalchemy import Text # noqa: F401 + +if TYPE_CHECKING: + from collections.abc import Sequence + +__all__ = ["downgrade", "upgrade", "schema_upgrades", "schema_downgrades", "data_upgrades", "data_downgrades"] + +sa.GUID = GUID +sa.DateTimeUTC = DateTimeUTC +sa.ORA_JSONB = ORA_JSONB +sa.EncryptedString = EncryptedString +sa.EncryptedText = EncryptedText +sa.StoredObject = StoredObject +sa.PasswordHash = PasswordHash +sa.Argon2Hasher = Argon2Hasher +sa.PasslibHasher = PasslibHasher +sa.PwdlibHasher = PwdlibHasher +sa.FernetBackend = FernetBackend +sa.PGCryptoBackend = PGCryptoBackend + +# revision identifiers, used by Alembic. +revision = 'd2d69065ede3' +down_revision = '49a9e85a0ffc' +branch_labels = None +depends_on = None + + +def upgrade() -> None: + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", category=UserWarning) + with op.get_context().autocommit_block(): + schema_upgrades() + data_upgrades() + +def downgrade() -> None: + with warnings.catch_warnings(): + warnings.filterwarnings("ignore", category=UserWarning) + with op.get_context().autocommit_block(): + data_downgrades() + schema_downgrades() + +def schema_upgrades() -> None: + """schema upgrade migrations go here.""" + pass + +def schema_downgrades() -> None: + """schema downgrade migrations go here.""" + pass + +def data_upgrades() -> None: + """ + Name the format of every file whose content type was never worked out. + + `create_many_from_existing_files` filled the column from `mimetypes.guess_type`, + which answers None for `.mobi`, `.azw`, `.fb2` and `.lit` — so a consume-directory + import of any of those stored a null, and the OPDS acquisition link a reader app + uses to decide what it can open carried nothing. + + Both write paths now go through `guess_content_type`, which this uses too, so the + formats in its table get named retroactively. A row it still cannot name is **left + null** rather than filled with a placeholder: null is the truth, the column is + nullable, and the one consumer that needs a string substitutes one itself. + Idempotent: it only looks at rows that carry nothing. + """ + from chitai.services.utils import guess_content_type + + connection = op.get_bind() + + files = connection.execute( + sa.text( + "SELECT id, path FROM file_metadata " + "WHERE content_type IS NULL OR content_type = ''" + ) + ).fetchall() + + parameters = [ + {"id": id, "content_type": content_type} + for id, path in files + if (content_type := guess_content_type(path)) is not None + ] + + if not parameters: + return + + batch_size = 1000 + for start in range(0, len(parameters), batch_size): + connection.execute( + sa.text( + "UPDATE file_metadata SET content_type = :content_type WHERE id = :id" + ), + parameters[start : start + batch_size], + ) + + +def data_downgrades() -> None: + """Add any optional data downgrade migrations here!""" diff --git a/backend/src/chitai/app.py b/backend/src/chitai/app.py index 102a957..9dfb72a 100644 --- a/backend/src/chitai/app.py +++ b/backend/src/chitai/app.py @@ -16,6 +16,7 @@ from sqlalchemy.ext.asyncio import AsyncSession from chitai import controllers as c +from chitai.cli import CalibreCLIPlugin from chitai.config import settings from chitai.database.config import alchemy from chitai.database.models.user import User @@ -133,7 +134,7 @@ def create_app() -> Litestar: ], exception_handlers=exception_handlers, lifespan=[setup_db_connection, setup_directory_watcher], - plugins=[alchemy], + plugins=[alchemy, CalibreCLIPlugin()], on_app_init=[oauth2_auth.on_app_init], openapi_config=OpenAPIConfig( title="Chitai", diff --git a/backend/src/chitai/cli.py b/backend/src/chitai/cli.py new file mode 100644 index 0000000..e8f5bca --- /dev/null +++ b/backend/src/chitai/cli.py @@ -0,0 +1,186 @@ +# src/chitai/cli.py + +""" +Extra commands on the `litestar` CLI. + +Registered through `CalibreCLIPlugin` in `app.py`, so they run as +`litestar --app-dir src/chitai/ calibre-import …` and get the app's own configuration +without a second way to load it. + +The Calibre import lives here as well as behind an endpoint because the case it exists +for is a one-time migration of a library that may be hundreds of gigabytes. That should +not depend on a browser tab staying open. +""" + +from __future__ import annotations + +import asyncio +from pathlib import Path + +import click +from click import Group +from litestar.plugins import CLIPluginProtocol + +from chitai.config import settings +from chitai.database.models import Library +from chitai.services.book import BookService, CalibreImportProgress, CalibreImportResult +from chitai.services.calibre import CalibreLibrary, CalibreLibraryError +from chitai.services.library import LibraryService + + +class CalibreCLIPlugin(CLIPluginProtocol): + """Adds `calibre-import` to the Litestar CLI.""" + + def on_cli_init(self, cli: Group) -> None: + cli.add_command(calibre_import) + + +@click.command(name="calibre-import") +@click.argument( + "source", + type=click.Path(exists=True, file_okay=False, path_type=Path), +) +@click.option( + "--library", + "library_slug", + required=True, + help="Slug of the Chitai library to import into.", +) +@click.option( + "--allow-duplicates", + is_flag=True, + help="Import books whose files the library already holds.", +) +@click.option( + "--dry-run", + is_flag=True, + help="Read the catalogue and report what it holds, without writing anything.", +) +def calibre_import( + source: Path, library_slug: str, allow_duplicates: bool, dry_run: bool +) -> None: + """ + Import a Calibre library from SOURCE, the directory holding its metadata.db. + + Files are copied, never moved: the Calibre library is left exactly as it is, and + re-running skips whatever is already stored. + """ + asyncio.run(_import(source, library_slug, allow_duplicates, dry_run)) + + +async def _import( + source: Path, library_slug: str, allow_duplicates: bool, dry_run: bool +) -> None: + try: + library_source = CalibreLibrary(source) + await library_source.open() + except CalibreLibraryError as exc: + raise click.ClickException(str(exc)) from exc + + try: + if dry_run: + await _report(library_source) + return + + async with settings.alchemy_config.get_session() as session: + library = await _library(session, library_slug) + + result = await BookService(session=session).create_many_from_calibre( + library_source, + library, + allow_duplicates=allow_duplicates, + on_progress=_print_progress, + ) + finally: + await library_source.close() + + _print_summary(result) + + if result.failed: + raise SystemExit(1) + + +async def _library(session: object, slug: str) -> Library: + """Resolve the target library, or explain what the options were.""" + service = LibraryService(session=session) # type: ignore[arg-type] + + library = await service.get_one_or_none(Library.slug == slug) + + if library is None: + available = ", ".join(sorted(item.slug for item in await service.list())) + raise click.ClickException( + f"No library with slug '{slug}'. Available: {available or 'none'}" + ) + + # A read-only library is one pointing at a tree Chitai does not own. Copying books + # into it would write into somebody else's directory. + if library.read_only: + raise click.ClickException( + f"Library '{slug}' is read-only, so nothing can be imported into it" + ) + + return library + + +async def _report(source: CalibreLibrary) -> None: + """Describe the catalogue without touching the database.""" + books = await source.books() + + click.echo(f"{len(books)} book(s) in {source.root}\n") + + for book in books: + authors = ", ".join(book.authors) or "unknown author" + formats = ", ".join(file.format for file in book.files) or "no files" + click.echo(f" #{book.calibre_id:<6} {book.title}") + click.echo(f" {'':<7} {authors} · {formats}") + + missing = [ + book + for book in books + if any(not file.path.is_file() for file in book.files) or not book.files + ] + + if missing: + click.echo( + f"\n{len(missing)} book(s) have files the catalogue lists " + "but disk does not:" + ) + for book in missing: + click.echo(f" #{book.calibre_id} {book.title}") + + +def _print_progress(progress: CalibreImportProgress) -> None: + marker = {"created": "+", "skipped": "-", "failed": "!"}.get(progress.outcome, " ") + detail = f" ({progress.detail})" if progress.detail else "" + + click.echo( + f"[{progress.processed:>5}/{progress.total}] {marker} {progress.title}{detail}" + ) + + +def _print_summary(result: CalibreImportResult) -> None: + click.echo( + f"\n{len(result.created)} created, {len(result.skipped)} skipped, " + f"{len(result.failed)} failed, of {result.total}." + ) + + if result.duplicate_files: + click.echo( + f"{len(result.duplicate_files)} individual file(s) were already stored and " + "were left out of books that imported otherwise." + ) + + for failure in result.failed: + click.echo(f" failed #{failure.calibre_id} {failure.title}: {failure.reason}") + + # Imported all the same — a metadata match is a guess, and the duplicates screen is + # where these get decided. + for possible in result.possible_duplicates: + names = ", ".join( + f"{candidate.title} (#{candidate.book_id})" + for candidate in possible.candidates + ) + click.echo( + f" possible duplicate {possible.title} (#{possible.book_id}) " + f"may already be in the library as: {names}" + ) diff --git a/backend/src/chitai/controllers/library.py b/backend/src/chitai/controllers/library.py index c8dfdfa..f73dd50 100644 --- a/backend/src/chitai/controllers/library.py +++ b/backend/src/chitai/controllers/library.py @@ -1,12 +1,20 @@ # src/chitai/controllers/library.py # Standard library +import asyncio +import shutil +import tempfile +from pathlib import Path from typing import Annotated # Third-party libraries +import aiofiles +from aiofiles import os as aios from litestar import Controller, post, get, patch, delete -from litestar.params import Dependency +from litestar.enums import RequestEncodingType +from litestar.params import Body, Dependency from litestar.exceptions import HTTPException +from litestar.status_codes import HTTP_200_OK, HTTP_202_ACCEPTED from advanced_alchemy.extensions.litestar.providers import create_service_dependencies from advanced_alchemy.service.pagination import OffsetPagination from advanced_alchemy.service import FilterTypeT @@ -14,10 +22,21 @@ from advanced_alchemy.service import FilterTypeT # Local imports from chitai.database import models as m from chitai.services import LibraryService -from chitai.schemas.library import LibraryCreate, LibraryRead +from chitai.schemas.library import ( + CalibreArchiveUpload, + CalibreImportRead, + LibraryCreate, + LibraryRead, +) +from chitai.services.calibre import CalibreLibraryError, extract_calibre_archive +from chitai.services.calibre_import import registry from chitai.services.utils import DirectoryDoesNotExist +# How much of an uploaded archive is held in memory at a time on its way to disk. +UPLOAD_CHUNK_SIZE = 262144 # 256 KiB + + class LibraryController(Controller): """Controller for managing library operations.""" @@ -74,3 +93,147 @@ class LibraryController(Controller): return library_service.to_schema( results, total, filters, schema_type=LibraryRead ) + + @post( + path="{library_id:int}/imports/calibre/upload", + status_code=HTTP_202_ACCEPTED, + request_max_body_size=None, + ) + async def upload_calibre_import( + self, + library_service: LibraryService, + library_id: int, + data: Annotated[ + CalibreArchiveUpload, Body(media_type=RequestEncodingType.MULTI_PART) + ], + ) -> CalibreImportRead: + """ + Import a zipped Calibre library that was uploaded rather than named on disk. + + For the case where the library is not on the server: zip the Calibre folder and + send it. Unpacked into a temp directory the job owns and deletes when it ends — + by which time the books worth keeping have been copied into the library proper. + + The server-folder route stays the one for a very large library. This one has to + carry the whole archive over HTTP first. + + Path Parameters: + library_id: The library to import into. + + Request Body: + data: The `.zip` holding the Calibre library. + + Returns: + The job, already running. + + Raises: + HTTPException: 400 if the archive is not a zip, holds no `metadata.db`, + names an entry outside itself, or would not fit on disk; 409 if an + import into this library is already running. + """ + library = await self._importable(library_service, library_id) + + if (running := registry.running_for(library_id)) is not None: + raise HTTPException( + status_code=409, + detail="An import into this library is already running", + extra={"job_id": running.id}, + ) + + workspace = Path(await asyncio.to_thread(tempfile.mkdtemp)) + + try: + archive = workspace / "upload.zip" + await data.archive.seek(0) + + async with aiofiles.open(archive, "wb") as destination: + while chunk := await data.archive.read(UPLOAD_CHUNK_SIZE): + await destination.write(chunk) + + unpacked = workspace / "library" + unpacked.mkdir() + + catalogue = await extract_calibre_archive(archive, unpacked) + + # The archive itself is dead weight once unpacked, and the library it + # unpacked to can be large. + await aios.remove(archive) + except CalibreLibraryError as exc: + await asyncio.to_thread(shutil.rmtree, workspace, True) + raise HTTPException(status_code=400, detail=str(exc)) + except Exception: + await asyncio.to_thread(shutil.rmtree, workspace, True) + raise + + job = registry.start( + library, + catalogue, + workspace=workspace, + label=data.archive.filename or "uploaded archive", + allow_duplicates=data.allow_duplicates, + ) + + return CalibreImportRead.model_validate(job) + + @get(path="imports/{job_id:str}") + async def get_import(self, job_id: str) -> CalibreImportRead: + """ + Report on an import. + + Polled by the client while a run is going. Jobs are held in memory, so this is + answered by the process that started it — see `services/calibre_import.py`. + + Path Parameters: + job_id: The job to report on. + + Raises: + HTTPException: 404 if this process holds no such job. + """ + if (job := registry.get(job_id)) is None: + raise HTTPException(status_code=404, detail="No such import") + + return CalibreImportRead.model_validate(job) + + @delete(path="imports/{job_id:str}", status_code=HTTP_200_OK) + async def cancel_import(self, job_id: str) -> CalibreImportRead: + """ + Ask an import to stop after the book it is on. + + Deliberately not an abort: a book abandoned mid-copy would leave files on disk + with no row describing them. Whatever it has imported stays imported. + + Path Parameters: + job_id: The job to stop. + + Raises: + HTTPException: 404 if this process holds no such job. + """ + if (job := registry.cancel(job_id)) is None: + raise HTTPException(status_code=404, detail="No such import") + + return CalibreImportRead.model_validate(job) + + @staticmethod + async def _importable( + library_service: LibraryService, library_id: int + ) -> m.Library: + """ + The library, if it can be imported into at all. + + Raises: + HTTPException: 404 if there is no such library, 400 if it is read-only — + a read-only library points at a tree Chitai does not own, so copying + books into it would write into somebody else's directory. + """ + library = await library_service.get_one_or_none(m.Library.id == library_id) + + if library is None: + raise HTTPException(status_code=404, detail="No such library") + + if library.read_only: + raise HTTPException( + status_code=400, + detail="This library is read-only, so nothing can be imported into it", + ) + + return library diff --git a/backend/src/chitai/schemas/book.py b/backend/src/chitai/schemas/book.py index de511e0..17ea732 100644 --- a/backend/src/chitai/schemas/book.py +++ b/backend/src/chitai/schemas/book.py @@ -30,7 +30,11 @@ class FileMetadataRead(BaseModel): path: str hash: str size: int - content_type: str + + # Nullable, though every ingest path now writes one through + # `guess_content_type`. Rows predating it can hold null, and a required field here + # turns one of those into a 500 on a book the reader can otherwise open. + content_type: str | None = None @computed_field @property diff --git a/backend/src/chitai/schemas/library.py b/backend/src/chitai/schemas/library.py index 36232f5..55225f5 100644 --- a/backend/src/chitai/schemas/library.py +++ b/backend/src/chitai/schemas/library.py @@ -1,6 +1,7 @@ from pathlib import Path from typing import Annotated -from pydantic import BaseModel, Field, computed_field +from pydantic import BaseModel, ConfigDict, Field, SkipValidation, computed_field +from litestar.datastructures import UploadFile from advanced_alchemy.utils.text import slugify class LibraryCreate(BaseModel): @@ -27,6 +28,57 @@ class LibraryRead(BaseModel): total: int | None = None +class CalibreArchiveUpload(BaseModel): + """ + A zipped Calibre library. + + The only way in through the API: a desktop Calibre install is usually not on the + server at all. Importing from a path the server can already see is a server-side + operation, and stays one — `litestar calibre-import` does that. + """ + + archive: Annotated[UploadFile, SkipValidation] + allow_duplicates: bool = False + + model_config = ConfigDict(arbitrary_types_allowed=True) + + +class ImportFailureRead(BaseModel): + """A book the import could not store.""" + + model_config = ConfigDict(from_attributes=True) + + calibre_id: int + title: str + reason: str + + +class CalibreImportRead(BaseModel): + """A running or finished import.""" + + model_config = ConfigDict(from_attributes=True) + + id: str + library_id: int + source: str + state: str + + total: int + processed: int + created: int + skipped: int + failed: int + + current_title: str | None = None + failures: list[ImportFailureRead] = Field(default_factory=list) + + # A count, not the records. The library's duplicates screen is what shows them. + possible_duplicates: int = 0 + + # Set when the run itself broke, as opposed to individual books failing. + error: str | None = None + + class LibraryUpdate(BaseModel): name: str | None root_path: str | None diff --git a/backend/src/chitai/services/book.py b/backend/src/chitai/services/book.py index 3399b41..b0f24ac 100644 --- a/backend/src/chitai/services/book.py +++ b/backend/src/chitai/services/book.py @@ -3,7 +3,6 @@ # Standard library from __future__ import annotations from collections import defaultdict -import mimetypes from collections.abc import Callable, Iterable, Sequence from dataclasses import dataclass, field, replace @@ -51,6 +50,7 @@ from chitai.database.models import ( Library, ) from chitai.schemas.book import BooksCreateFromFiles +from chitai.services.calibre import CalibreBook, CalibreLibrary from chitai.services.filesystem_library import BookPathGenerator from chitai.services.matching import ( normalize_author, @@ -58,11 +58,14 @@ from chitai.services.matching import ( normalize_title, ) from chitai.services.metadata_extractor import Extractor as MetadataExtractor +from chitai.services.metadata_extractor import parse_identifier from chitai.services.utils import ( cleanup_empty_parent_directories, + copy_file, delete_file, fingerprint_file, fingerprint_upload, + guess_content_type, move_dir_contents, move_file, save_image, @@ -85,6 +88,10 @@ Fingerprint = tuple[str, int] # How much of a file is held in memory at a time while it is written to disk. CHUNK_SIZE = 262144 # 256 KiB +# What Calibre's own `books.uuid` is stored as. Not `uuid`: that name is reserved for +# the per-build identifier an EPUB carries, which duplicate matching ignores. +CALIBRE_UUID = "calibre-uuid" + @dataclass(frozen=True) class DuplicateFile: @@ -147,6 +154,58 @@ class ImportResult: possible_duplicates: list[PossibleDuplicate] = field(default_factory=list) +@dataclass(frozen=True) +class UnimportedBook: + """A book in a source catalogue that did not become a Chitai book, and why.""" + + calibre_id: int + title: str + reason: str + + # The book already holding its files, when that is why it was left out. + book_id: int | None = None + + +@dataclass(frozen=True) +class CalibreImportProgress: + """One book's outcome, as it happens. + + Reported per book rather than at the end: an import of a real library takes long + enough that its progress is the only thing worth looking at while it runs. + """ + + processed: int + total: int + title: str + outcome: str + """`created`, `skipped` or `failed`.""" + + detail: str | None = None + + +@dataclass +class CalibreImportResult: + """What importing a Calibre library produced. + + Holds ids rather than `Book` objects: a real catalogue runs to thousands of books, + and nothing downstream needs them all in memory at once. + """ + + total: int = 0 + created: list[int] = field(default_factory=list) + skipped: list[UnimportedBook] = field(default_factory=list) + failed: list[UnimportedBook] = field(default_factory=list) + + # Individual files left out of books that were otherwise imported. + duplicate_files: list[DuplicateFile] = field(default_factory=list) + + possible_duplicates: list[PossibleDuplicate] = field(default_factory=list) + + # The run gave up before reaching the end of the catalogue, on request. What it did + # get through is complete; `processed` is simply short of `total`. + stopped: bool = False + + class DuplicateFilesError(Exception): """Raised when an import would store files that are already in the library.""" @@ -1246,12 +1305,12 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): **kwargs, ) result.books.append(book) - await self._record_possible_duplicates(result, book, library) + await self._record_possible_duplicates(result.possible_duplicates, book, library) return result async def _record_possible_duplicates( - self, result: ImportResult, book: Book, library: Library + self, into: list[PossibleDuplicate], book: Book, library: Library ) -> None: """ Note anything the library already holds that this book might be a copy of. @@ -1262,7 +1321,9 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): the same import is a candidate too. Args: - result: The import being assembled, appended to in place. + into: The collection to append to, in place. Takes the list rather than the + whole result so that every ingest path can use it whatever its own + result type is. book: The book that was just created. library: The library it went into. """ @@ -1271,7 +1332,7 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): ) if candidates: - result.possible_duplicates.append( + into.append( PossibleDuplicate( book_id=book.id, title=book.title, candidates=candidates ) @@ -1357,7 +1418,6 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): for file in accepted: file_hash, file_size = fingerprints[file.name] - content_type, _ = mimetypes.guess_type(file) filename = path_gen.generate_filename(data, Path(file.name)) @@ -1366,7 +1426,7 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): path=str(filename), size=file_size, hash=file_hash, - content_type=content_type, + content_type=guess_content_type(file), ) ) @@ -1386,12 +1446,280 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): book = await super().create(data) result.books.append(book) - await self._record_possible_duplicates(result, book, library) + await self._record_possible_duplicates(result.possible_duplicates, book, library) await self.repository.session.commit() return result + async def create_many_from_calibre( + self, + source: CalibreLibrary, + library: Library, + allow_duplicates: bool = False, + on_progress: Callable[[CalibreImportProgress], None] | None = None, + should_stop: Callable[[], bool] | None = None, + ) -> CalibreImportResult: + """ + Import a Calibre library, **copying** its files rather than taking them. + + Calibre's catalogue is curated and its filenames are truncated, so the metadata + is taken from `metadata.db` and the extractors are not run at all — the one + ingest path here that trusts what it is given. The title is stored verbatim for + the same reason: no edition is split out of it and no subtitle is guessed, since + the field is one somebody maintained by hand. + + Files are **copied**, never moved. `metadata.db` would go on pointing at files + that were gone, which quietly ruins a library the reader still uses; the source + tree is left byte-for-byte alone. + + Re-running is safe and needs no bookkeeping: the same bytes are recognised + wherever they sit, so a second pass over one library skips all of it. + + Args: + source: An open `CalibreLibrary`. + library: The Chitai library to import into. + allow_duplicates: Import books whose files are already stored. + on_progress: Called once per book, as it is decided. + should_stop: Asked before each book whether to give up. Checked between + books rather than during one, so a cancelled import leaves whole books + behind and never half of one. + + Returns: + What was created, left out and flagged. A cancelled run returns what it got + through — every book in `created` is committed and complete. + """ + books = await source.books() + result = CalibreImportResult(total=len(books)) + + for processed, entry in enumerate(books, start=1): + if should_stop is not None and should_stop(): + result.stopped = True + break + + try: + outcome, detail = await self._import_calibre_book( + entry, library, result, allow_duplicates + ) + except Exception as exc: + # One book must never cost the rest of the import. The session is left + # unusable by a failed flush, so it is rolled back before the next book + # touches it. + await self.repository.session.rollback() + + outcome, detail = "failed", f"{type(exc).__name__}: {exc}" + result.failed.append( + UnimportedBook( + calibre_id=entry.calibre_id, title=entry.title, reason=detail + ) + ) + + if on_progress is not None: + on_progress( + CalibreImportProgress( + processed=processed, + total=result.total, + title=entry.title, + outcome=outcome, + detail=detail, + ) + ) + + return result + + async def _import_calibre_book( + self, + entry: CalibreBook, + library: Library, + result: CalibreImportResult, + allow_duplicates: bool, + ) -> tuple[str, str | None]: + """ + Import one Calibre book, or say why it was left out. + + Args: + entry: The book as Calibre describes it. + library: The library to import into. + result: The import being assembled, appended to in place. + allow_duplicates: Store files the library already holds. + + Returns: + The outcome — `created` or `skipped` — and a detail for the progress report. + + Raises: + Exception: Anything the write path raises, after the files this book had + already copied are removed again. An orphaned directory would make the + next attempt reserve `title (2)` and look like it had succeeded. + """ + # What the catalogue claims and what is on disk can disagree: Calibre keeps the + # row when a file is moved away behind its back. + present = [ + file.path + for file in entry.files + if await aios.path.isfile(file.path) + ] + + if not present: + reason = ( + "no files on disk" + if entry.files + else "no files in the catalogue" + ) + result.skipped.append( + UnimportedBook( + calibre_id=entry.calibre_id, title=entry.title, reason=reason + ) + ) + return "skipped", reason + + fingerprints = {path.name: await fingerprint_file(path) for path in present} + + if allow_duplicates: + accepted, rejected = present, [] + else: + known = await self.find_duplicate_files(fingerprints.values(), library) + accepted, rejected = self._screen_for_duplicates( + present, fingerprints, known, library + ) + + if not accepted: + held_by = next( + (duplicate.book_id for _, duplicate in rejected if duplicate.book_id), + None, + ) + result.skipped.append( + UnimportedBook( + calibre_id=entry.calibre_id, + title=entry.title, + reason="already stored", + book_id=held_by, + ) + ) + return "skipped", "already stored" + + # Files this book already has under another name are reported but do not stop + # the formats that are new from being imported. + result.duplicate_files.extend(duplicate for _, duplicate in rejected) + + data = self._calibre_metadata(entry, library) + + path_gen = BookPathGenerator(library.root_path) + parent = await self._reserve_book_path(path_gen.generate_path(data)) + data["path"] = str(parent) + + copied: list[Path] = [] + try: + await self._attach_calibre_cover(entry, data) + + file_metadata = [] + for path in accepted: + filename = path_gen.generate_filename(data, Path(path.name)) + destination = _unused_path(parent / filename) + + await copy_file(path, destination) + copied.append(destination) + + file_hash, file_size = fingerprints[path.name] + file_metadata.append( + FileMetadata( + path=destination.name, + size=file_size, + hash=file_hash, + content_type=guess_content_type(path), + ) + ) + + data["files"] = file_metadata + + book = await self.create(data) + result.created.append(book.id) + + await self._record_possible_duplicates( + result.possible_duplicates, book, library + ) + + await self.repository.session.commit() + except Exception: + for path in copied: + await delete_file(path) + cleanup_empty_parent_directories(parent, Path(library.root_path)) + raise + + return "created", None + + def _calibre_metadata(self, entry: CalibreBook, library: Library) -> dict[str, Any]: + """ + Turn one `CalibreBook` into the payload `create` takes. + + Args: + entry: The book as Calibre describes it. + library: The library it is going into. + + Returns: + The metadata dict, with no files or path in it yet. + """ + # Folded onto Chitai's schemes here rather than in the reader, which reports what + # Calibre wrote. `parse_identifier` maps `amazon` and `mobi-asin` both onto + # `asin`, and a book carrying both would otherwise breach the unique + # `(name, book_id)` constraint — so the first one wins, as it does everywhere + # else identifiers are merged. + identifiers: dict[str, str] = {} + for name, value in entry.identifiers.items(): + if parsed := parse_identifier(value, scheme=name): + identifiers.setdefault(*parsed) + + # Deliberately not passed through `parse_identifier`, which would recognise the + # shape and file it under `uuid` — a name `normalize_identifier` refuses, since + # an EPUB's uuid is regenerated per build. Calibre's is stable for the life of + # the row, so it is the one durable link back to the source and worth matching + # on when the same library is imported twice. + if entry.uuid: + identifiers.setdefault(CALIBRE_UUID, entry.uuid) + + return { + "library_id": library.id, + "title": entry.title or "Unknown", + "authors": list(entry.authors), + "description": entry.description, + "published_date": entry.published_date, + "series": entry.series, + "series_position": entry.series_position, + "tags": list(entry.tags), + "publisher": entry.publisher, + "language": entry.language, + "identifiers": identifiers, + "pages": entry.pages, + } + + async def _attach_calibre_cover(self, entry: CalibreBook, data: dict) -> None: + """ + Put Calibre's `cover.jpg` on the payload, for `_save_cover_image` to convert. + + Taken from the file Calibre already rendered rather than extracted again from + the book: it is the cover the reader chose, and opening every EPUB and rendering + the first page of every PDF is the slowest thing an import could do. + + Args: + entry: The book as Calibre describes it. + data: The payload, modified in place. + """ + if entry.cover is None or not await aios.path.isfile(entry.cover): + return + + try: + # Copied out of the context manager: the image is saved later, by which time + # the file this was opened from is closed. + with Image.open(entry.cover) as cover: + data["cover_image"] = cover.copy() + except OSError: + # A truncated or malformed cover.jpg is not a reason to refuse the book. It + # is the least important thing in the directory, and one can be added later + # from the book page; the files cannot. + data.pop("cover_image", None) + return + + await self._save_cover_image(data) + @staticmethod async def _quarantine_duplicate( file: Path, consume_path: Path, library: Library @@ -1479,7 +1807,12 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): if not await aios.path.isfile(path): raise ValueError("The file is missing") - return File(path, media_type=file.content_type) + # Rows written before `guess_content_type` existed can hold null here, and + # so can `guess_content_type` itself. Litestar fills a null in from the + # filename, falling back to `application/octet-stream` on the response. + return File( + path, media_type=file.content_type or guess_content_type(file.path) + ) raise ValueError("No such file for the given book") @@ -1964,7 +2297,10 @@ class BookService(SQLAlchemyAsyncRepositoryService[Book]): path=str(filename), size=file_size, hash=fingerprint[0] if fingerprint else hasher.hexdigest(), - content_type=file.content_type, + # The name outranks what the browser said: it posts + # `application/octet-stream` for every format it does not know, + # which is most of them. + content_type=guess_content_type(file, fallback=file.content_type), ) ) diff --git a/backend/src/chitai/services/calibre.py b/backend/src/chitai/services/calibre.py new file mode 100644 index 0000000..37620e5 --- /dev/null +++ b/backend/src/chitai/services/calibre.py @@ -0,0 +1,570 @@ +# src/chitai/services/calibre.py + +""" +Read a Calibre library. + +This knows about `metadata.db` and the tree beside it, and nothing about `Book`, +`BookService` or a database session — it is a file-format reader, and it is testable +without Postgres or an app. Interpretation belongs to whoever imports what it returns: +identifiers come back exactly as Calibre wrote them, not folded onto Chitai's schemes. + +Things about Calibre that are load-bearing here: + +- **Never query the views.** `meta` and the `tag_browser_*` family call SQLite functions + Calibre registers from Python at connection time, so `SELECT * FROM meta` fails with + `no such function: sortconcat`. Only base tables are touched below. +- **An unknown date is a sentinel, not a null** — `0101-01-01`, Calibre's + `UNDEFINED_DATE`. It parses fine as a date, so nothing complains; it just makes every + book without a publication date look like it was published in the year 101. +- **`data.name` is not the title.** It is the on-disk stem, truncated to Calibre's + filename limit and sanitised, so the file is `The Project Gutenberg eBook #33283_ + Calcul - Silvanus Phillips Thompson.pdf` for a book titled `The Project Gutenberg + eBook #33283: Calculus Made Easy, 2nd Edition`. Names locate files; the database + carries the metadata. +- **`authors.name` escapes a comma as `|`**, which Calibre reverses on read + (`AuthorsTable.unserialize` in its `db/tables.py`). +- **`series_index` defaults to 1.0 whether or not the book is in a series**, so a + position is only meaningful alongside a series. +- **`books_pages_link` is recent and often empty.** It is treated as optional both ways: + the table may not exist, and where it does the rows are frequently `pages = 0` with + `needs_scan = 1`. + +Nothing walks the tree: every file is located through `books.path`, which is why +`.caltrash` — where Calibre keeps deleted books, still on disk — cannot be picked up by +accident. +""" + +from __future__ import annotations + +import asyncio +import shutil +import sqlite3 +import tempfile +import zipfile +from collections import defaultdict +from collections.abc import Callable +from dataclasses import dataclass, field +from datetime import date, datetime +from html.parser import HTMLParser +from pathlib import Path, PurePosixPath + + +METADATA_DB = "metadata.db" +COVER_FILENAME = "cover.jpg" + +# Calibre writes `0101-01-01` for "no date". Any year this early is that sentinel rather +# than a publication date somebody meant. +EARLIEST_REAL_YEAR = 1000 + +# The sidecars a WAL-mode database keeps beside itself. Copied along with it so the +# snapshot can be recovered, since Calibre may be running while this reads. +_DATABASE_SIDECARS = ("-wal", "-shm") + + +class CalibreLibraryError(Exception): + """The library cannot be read at all — wrong directory, or no catalogue in it.""" + + +# How far down an archive to look for `metadata.db`. Zipping a Calibre library gives +# either the directory itself or its contents, and a file manager may add a wrapper +# folder on top, so two levels of nesting is normal and more is somebody's backup tree. +_ARCHIVE_SEARCH_DEPTH = 3 + +# Extraction is refused unless the destination has the uncompressed size plus this +# much headroom. Filling the disk would take the whole application down, not just the +# import. +_DISK_HEADROOM = 256 * 1024 * 1024 # 256 MiB + + +def _archive_members(archive: zipfile.ZipFile) -> list[zipfile.ZipInfo]: + """ + The entries worth extracting, refusing any that would escape the destination. + + `ZipFile.extract` does sanitise names, but relying on that silently is how the next + person to swap the extraction call reintroduces zip-slip. An archive naming + `../../etc/anything` is malformed or hostile, and either way there is nothing to + salvage by continuing. + + Raises: + CalibreLibraryError: If any entry points outside the archive root. + """ + members = [] + + for member in archive.infolist(): + if member.is_dir(): + continue + + name = PurePosixPath(member.filename) + + if name.is_absolute() or ".." in name.parts: + raise CalibreLibraryError( + f"The archive contains an entry outside itself: {member.filename!r}" + ) + + members.append(member) + + return members + + +async def extract_calibre_archive(archive: Path, destination: Path) -> Path: + """ + Unpack a zipped Calibre library and find the catalogue inside it. + + Args: + archive: The `.zip` to unpack. + destination: An empty directory to unpack into. The caller owns it and is + responsible for removing it. + + Returns: + The directory holding `metadata.db`, which is what `CalibreLibrary` takes. + + Raises: + CalibreLibraryError: If the file is not a zip, names an entry outside itself, + would not fit on disk, or holds no `metadata.db`. + """ + return await asyncio.to_thread(_extract_calibre_archive, archive, destination) + + +def _extract_calibre_archive(archive: Path, destination: Path) -> Path: + if not zipfile.is_zipfile(archive): + raise CalibreLibraryError( + "That is not a zip file. A Calibre library has to be zipped, not tarred." + ) + + with zipfile.ZipFile(archive) as opened: + members = _archive_members(opened) + + if not any( + PurePosixPath(member.filename).name == METADATA_DB for member in members + ): + raise CalibreLibraryError( + f"The archive holds no {METADATA_DB}, so it is not a Calibre library" + ) + + # Checked before writing rather than discovered part-way through: a full disk + # takes the whole application down, and the number is in the archive already. + needed = sum(member.file_size for member in members) + free = shutil.disk_usage(destination).free + + if needed + _DISK_HEADROOM > free: + raise CalibreLibraryError( + f"Unpacking needs {needed // (1024 * 1024)} MiB and only " + f"{free // (1024 * 1024)} MiB is free" + ) + + opened.extractall(destination, members=members) + + return _find_catalogue(destination) + + +def _find_catalogue(root: Path) -> Path: + """The shallowest directory under `root` holding a `metadata.db`.""" + candidates = sorted( + (path.parent for path in root.rglob(METADATA_DB) if path.is_file()), + key=lambda path: len(path.relative_to(root).parts), + ) + + for candidate in candidates: + if len(candidate.relative_to(root).parts) <= _ARCHIVE_SEARCH_DEPTH: + return candidate + + raise CalibreLibraryError( + f"No {METADATA_DB} within {_ARCHIVE_SEARCH_DEPTH} levels of the archive root" + ) + + +@dataclass(frozen=True) +class CalibreFile: + """One row of Calibre's `data` table: a book in one format.""" + + path: Path + """Absolute path, resolved against the library root. Not checked for existence.""" + + format: str + """As Calibre stores it, upper case: `EPUB`, `PDF`, `AZW3`.""" + + size: int + """`data.uncompressed_size` — Calibre's claim, not a fresh stat.""" + + +@dataclass(frozen=True) +class CalibreBook: + """One book, with everything Chitai has a column for and nothing it does not.""" + + calibre_id: int + uuid: str + title: str + authors: list[str] = field(default_factory=list) + description: str | None = None + published_date: date | None = None + series: str | None = None + series_position: str | None = None + tags: list[str] = field(default_factory=list) + publisher: str | None = None + language: str | None = None + + identifiers: dict[str, str] = field(default_factory=dict) + """Keyed by `identifiers.type` verbatim — `isbn`, `mobi-asin`, `amazon`.""" + + pages: int | None = None + cover: Path | None = None + files: list[CalibreFile] = field(default_factory=list) + + +class CalibreLibrary: + """ + A Calibre library on disk, opened for reading. + + The catalogue is **copied** before it is read. Calibre may be running and writing, + and opening the live file either sees a torn state or needs to recover a write-ahead + log, which read-only access cannot do. The copy is small — hundreds of kilobytes for + a handful of books, single-digit megabytes for thousands — so this costs nothing and + removes the question. The original is never opened by SQLite at all. + """ + + def __init__(self, root: Path | str) -> None: + self.root = Path(root) + self._connection: sqlite3.Connection | None = None + self._workspace: Path | None = None + + # Every query runs in a worker thread, and `asyncio.to_thread` hands out + # whichever one is free — so the connection outlives the thread that opened it + # and `check_same_thread` has to be off. The lock is what makes that safe: it + # keeps two queries off the connection at once, which is the thing that check + # was standing in for. + self._lock = asyncio.Lock() + + @property + def database(self) -> Path: + return self.root / METADATA_DB + + async def open(self) -> None: + """ + Copy the catalogue aside and connect to the copy. + + Raises: + CalibreLibraryError: If there is no `metadata.db` under the root. + """ + if self._connection is not None: + return + + if not await asyncio.to_thread(self.database.is_file): + raise CalibreLibraryError( + f"No {METADATA_DB} in '{self.root}' — that is not a Calibre library" + ) + + self._workspace = Path(await asyncio.to_thread(tempfile.mkdtemp)) + copy = self._workspace / METADATA_DB + + await asyncio.to_thread(self._copy_database, copy) + + # Read-write on our own copy, deliberately: that is what lets SQLite recover a + # write-ahead log the source may have been mid-way through. + self._connection = sqlite3.connect(str(copy), check_same_thread=False) + + def _copy_database(self, destination: Path) -> None: + shutil.copy2(self.database, destination) + + for suffix in _DATABASE_SIDECARS: + sidecar = self.database.with_name(self.database.name + suffix) + if sidecar.is_file(): + shutil.copy2(sidecar, destination.with_name(destination.name + suffix)) + + async def close(self) -> None: + """ + Disconnect and remove the copy. Safe to call more than once. + + The copy is removed even if closing the connection fails — otherwise a failure + here leaves a catalogue-sized file in the temp directory, and the caller that + failed is exactly the one that will not come back to tidy up. + """ + try: + if self._connection is not None: + async with self._lock: + self._connection.close() + self._connection = None + finally: + if self._workspace is not None: + await asyncio.to_thread(shutil.rmtree, self._workspace, True) + self._workspace = None + + async def __aenter__(self) -> CalibreLibrary: + await self.open() + return self + + async def __aexit__(self, *_exception: object) -> None: + await self.close() + + async def count(self) -> int: + """How many books the catalogue holds, without reading any of them.""" + rows = await self._in_thread(lambda: self._execute("SELECT count(*) FROM books")) + return int(rows[0][0]) + + async def books(self) -> list[CalibreBook]: + """ + Read the whole catalogue. + + One query per table and the joining done in Python, rather than a per-book query + across ten tables. Everything Chitai stores about a book is small, so a + self-hosted catalogue fits in memory comfortably. + + Returns: + Every book, in Calibre id order. + """ + return await self._in_thread(self._read_books) + + async def _in_thread[T](self, work: Callable[[], T]) -> T: + """Run one unit of SQLite work off the event loop, and only one at a time.""" + async with self._lock: + return await asyncio.to_thread(work) + + def _execute(self, statement: str) -> list[tuple]: + if self._connection is None: + raise CalibreLibraryError("The library is not open") + + return self._connection.execute(statement).fetchall() + + def _has_table(self, name: str) -> bool: + return bool( + self._execute( + f"SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = '{name}'" + ) + ) + + def _read_books(self) -> list[CalibreBook]: + authors = self._grouped( + "SELECT bal.book, a.name FROM books_authors_link bal " + "JOIN authors a ON a.id = bal.author ORDER BY bal.id" + ) + tags = self._grouped( + "SELECT btl.book, t.name FROM books_tags_link btl " + "JOIN tags t ON t.id = btl.tag ORDER BY t.name" + ) + # Calibre's link tables are unique per book for these two, so the last write + # wins and there is nothing to choose between. + series = self._mapped( + "SELECT bsl.book, s.name FROM books_series_link bsl " + "JOIN series s ON s.id = bsl.series" + ) + publishers = self._mapped( + "SELECT bpl.book, p.name FROM books_publishers_link bpl " + "JOIN publishers p ON p.id = bpl.publisher" + ) + # A book can carry several languages; Chitai holds one, so the first wins. + languages = self._grouped( + "SELECT bll.book, l.lang_code FROM books_languages_link bll " + "JOIN languages l ON l.id = bll.lang_code ORDER BY bll.item_order" + ) + descriptions = self._mapped("SELECT book, text FROM comments") + + identifiers: dict[int, dict[str, str]] = defaultdict(dict) + for book_id, name, value in self._execute( + "SELECT book, type, val FROM identifiers" + ): + if name and value: + identifiers[book_id][str(name)] = str(value) + + files: dict[int, list[tuple[str, str, int]]] = defaultdict(list) + for book_id, format, name, size in self._execute( + "SELECT book, format, name, uncompressed_size FROM data ORDER BY id" + ): + files[book_id].append((str(format), str(name), int(size or 0))) + + pages: dict[int, int] = {} + if self._has_table("books_pages_link"): + pages = { + book_id: int(count) + for book_id, count in self._execute( + "SELECT book, pages FROM books_pages_link WHERE pages > 0" + ) + } + + books = [] + for row in self._execute( + "SELECT id, title, pubdate, series_index, path, uuid, has_cover " + "FROM books ORDER BY id" + ): + book_id, title, pubdate, series_index, path, uuid, has_cover = row + directory = self.root / Path(str(path)) + in_series = series.get(book_id) + + books.append( + CalibreBook( + calibre_id=book_id, + uuid=str(uuid or ""), + title=str(title or ""), + authors=[unescape_author(name) for name in authors.get(book_id, [])], + description=strip_html(descriptions.get(book_id)), + published_date=parse_date(pubdate), + series=in_series, + # Meaningless without a series: Calibre defaults the index to 1.0 for + # every book, in a series or not. + series_position=( + format_series_index(series_index) if in_series else None + ), + tags=tags.get(book_id, []), + publisher=publishers.get(book_id), + language=next(iter(languages.get(book_id, [])), None), + identifiers=dict(identifiers.get(book_id, {})), + pages=pages.get(book_id), + cover=directory / COVER_FILENAME if has_cover else None, + files=[ + CalibreFile( + path=directory / f"{name}.{format.lower()}", + format=format, + size=size, + ) + for format, name, size in files.get(book_id, []) + ], + ) + ) + + return books + + def _grouped(self, statement: str) -> dict[int, list[str]]: + """Run a `(book, value)` query into one list per book, keeping row order.""" + grouped: dict[int, list[str]] = defaultdict(list) + + for book_id, value in self._execute(statement): + if value is not None: + grouped[book_id].append(str(value)) + + return grouped + + def _mapped(self, statement: str) -> dict[int, str]: + """Run a `(book, value)` query into one value per book.""" + return { + book_id: str(value) + for book_id, value in self._execute(statement) + if value is not None + } + + +def unescape_author(name: str) -> str: + """ + Undo Calibre's comma escaping. + + `authors.name` stores a comma as `|`, and Calibre reverses it on the way out. Left + alone, `Doyle, Sir Arthur Conan` comes back as `Doyle| Sir Arthur Conan`. + """ + return name.replace("|", ",").strip() + + +def parse_date(value: object) -> date | None: + """ + Read one of Calibre's timestamps, discarding its "unknown" sentinel. + + Args: + value: The stored column, which is text in practice but need not be. + + Returns: + The date, or None for a null, an unparseable value, or Calibre's + `0101-01-01` placeholder. + """ + if value is None: + return None + + if isinstance(value, datetime): + parsed = value.date() + elif isinstance(value, date): + parsed = value + else: + text = str(value).strip() + if not text: + return None + + try: + parsed = datetime.fromisoformat(text).date() + except ValueError: + try: + parsed = date.fromisoformat(text[:10]) + except ValueError: + return None + + return parsed if parsed.year >= EARLIEST_REAL_YEAR else None + + +def format_series_index(index: object) -> str | None: + """ + Render `series_index` as the string `Book.series_position` holds. + + Calibre stores a REAL, so volume seven arrives as `7.0` — which would be stored + verbatim and then compared as a string against the `7` everything else writes. + Fractional positions are real and are kept: `1.5` is a novella between two novels. + """ + if index is None: + return None + + try: + number = float(index) + except (TypeError, ValueError): + return None + + return str(int(number)) if number.is_integer() else f"{number:g}" + + +class _TextExtractor(HTMLParser): + """Flatten markup to text, keeping the line breaks that carried meaning.""" + + # Tags whose boundaries are a line break rather than nothing at all. Without these + # a description of three paragraphs comes out as one run-on sentence. + _BREAKS = frozenset( + { + "p", "br", "div", "li", "tr", "blockquote", "hr", + "h1", "h2", "h3", "h4", "h5", "h6", + } + ) + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self._parts: list[str] = [] + + def handle_data(self, data: str) -> None: + self._parts.append(data) + + def handle_starttag(self, tag: str, _attrs: object) -> None: + self._break(tag) + + def handle_endtag(self, tag: str) -> None: + self._break(tag) + + def _break(self, tag: str) -> None: + """ + End the current line, once. + + Both halves of `

` are a boundary, and the open tag of the very first + block is not one at all — so emitting a newline per tag turns two paragraphs + into two blank-line-separated ones with a leading gap. One break per boundary + is what the plain text wants. + """ + if tag in self._BREAKS and self._parts and not self._parts[-1].endswith("\n"): + self._parts.append("\n") + + @property + def text(self) -> str: + lines = [line.strip() for line in "".join(self._parts).splitlines()] + + return "\n".join(line for line in lines if line).strip() + + +def strip_html(html: str | None) -> str | None: + """ + Turn Calibre's `comments` into plain text. + + `comments.text` is always HTML, and `Book.description` is rendered as text — so the + tags would show literally on the book page. + + Args: + html: The stored comment, if there is one. + + Returns: + The text, or None when there was nothing or nothing survived. + """ + if not html: + return None + + parser = _TextExtractor() + parser.feed(html) + parser.close() + + return parser.text or None diff --git a/backend/src/chitai/services/calibre_import.py b/backend/src/chitai/services/calibre_import.py new file mode 100644 index 0000000..04d7d4d --- /dev/null +++ b/backend/src/chitai/services/calibre_import.py @@ -0,0 +1,239 @@ +# src/chitai/services/calibre_import.py + +""" +Run a Calibre import in the background and report on it. + +The import outlives its request — a real library takes minutes to hours — so the handler +starts a task and hands back a handle to poll. The work itself is +`BookService.create_many_from_calibre`; everything here is lifecycle: state, progress, +cancellation, and a session of its own. + +**This registry lives in memory, and therefore assumes one worker process.** That holds +today: the production `CMD` is `litestar run`, which is single-process, and the consume +watcher is already an in-process singleton with the same constraint. `TODO.md` records +that the production image should move to uvicorn with a worker count — the day that +happens, a poll can land on a worker that has never heard of the job, and this needs an +`import_jobs` table instead. It is written down here because nothing else will say so. +""" + +from __future__ import annotations + +import asyncio +import shutil +import uuid +from dataclasses import dataclass, field +from enum import StrEnum +from pathlib import Path + +from chitai.config import settings +from chitai.database.models import Library +from chitai.services.book import ( + BookService, + CalibreImportProgress, + CalibreImportResult, + UnimportedBook, +) +from chitai.services.calibre import CalibreLibrary + + +class ImportState(StrEnum): + """Where a job has got to.""" + + RUNNING = "running" + FINISHED = "finished" + + # Stopped on request. What it imported is complete. + CANCELLED = "cancelled" + + # The run itself broke — an unreadable catalogue, a missing library. Distinct from + # individual books failing, which `failures` carries and which never stops the run. + FAILED = "failed" + + +@dataclass +class ImportJob: + """One import, running or finished.""" + + id: str + library_id: int + source: str + + state: ImportState = ImportState.RUNNING + total: int = 0 + processed: int = 0 + created: int = 0 + skipped: int = 0 + failed: int = 0 + + current_title: str | None = None + failures: list[UnimportedBook] = field(default_factory=list) + + # Books imported that look like something the library already had. A count, not the + # records: the duplicates screen is what shows them, and a big import would make + # this the largest thing in the response for no benefit. + possible_duplicates: int = 0 + + # Why the whole run stopped, when `state` is FAILED. + error: str | None = None + + # The directory the job owns and must delete when it ends: what the uploaded archive + # was unpacked into. + workspace: Path | None = None + + _stop: bool = False + + @property + def finished(self) -> bool: + return self.state is not ImportState.RUNNING + + def absorb(self, result: CalibreImportResult) -> None: + """Take the final counts from a finished run.""" + self.total = result.total + self.created = len(result.created) + self.skipped = len(result.skipped) + self.failed = len(result.failed) + self.failures = list(result.failed) + self.possible_duplicates = len(result.possible_duplicates) + self.current_title = None + + self.state = ImportState.CANCELLED if result.stopped else ImportState.FINISHED + + +class CalibreImportRegistry: + """ + The imports this process knows about. + + One instance, held at module scope below. Jobs are kept after they finish so the + screen that started one can still read its result; nothing evicts them, which is + fine for a handful of one-time migrations and is the other reason a table would be + the answer if this ever needed to be durable. + """ + + def __init__(self) -> None: + self._jobs: dict[str, ImportJob] = {} + + # Held only to keep the tasks referenced. Without this the event loop is free to + # garbage-collect a running task mid-import. + self._tasks: set[asyncio.Task] = set() + + def get(self, job_id: str) -> ImportJob | None: + return self._jobs.get(job_id) + + def running_for(self, library_id: int) -> ImportJob | None: + """The unfinished import for a library, if it has one.""" + return next( + ( + job + for job in self._jobs.values() + if job.library_id == library_id and not job.finished + ), + None, + ) + + def cancel(self, job_id: str) -> ImportJob | None: + """ + Ask a job to stop after the book it is on. + + Not `task.cancel()`: that would abandon a book mid-copy, leaving files on disk + with no row describing them. The flag is read between books. + """ + job = self._jobs.get(job_id) + + if job is not None and not job.finished: + job._stop = True + + return job + + def start( + self, + library: Library, + source: Path, + workspace: Path, + label: str, + allow_duplicates: bool = False, + ) -> ImportJob: + """ + Begin importing, and return the handle to poll. + + Args: + library: The library to import into. + source: The unpacked Calibre library's directory. + workspace: A directory the job owns and deletes when it ends — what the + uploaded archive was unpacked into. The books have been copied into the + library by then, so nothing is lost with it. + label: What to report as the source. `source` is a temp directory that would + mean nothing to the reader, so this is the archive's own name. + allow_duplicates: Import books whose files are already stored. + + Returns: + The job, already running. + """ + job = ImportJob( + id=str(uuid.uuid4()), + library_id=library.id, + source=label, + workspace=workspace, + ) + self._jobs[job.id] = job + + task = asyncio.create_task(self._run(job, library.id, source, allow_duplicates)) + self._tasks.add(task) + task.add_done_callback(self._tasks.discard) + + return job + + async def _run( + self, job: ImportJob, library_id: int, source: Path, allow_duplicates: bool + ) -> None: + """ + Do the import, recording everything on the job. + + Opens a **session of its own**: the request that started this is long gone, and + its session was closed with it. + """ + from chitai.services.library import LibraryService + + def on_progress(progress: CalibreImportProgress) -> None: + job.total = progress.total + job.processed = progress.processed + job.current_title = progress.title + + if progress.outcome == "created": + job.created += 1 + elif progress.outcome == "skipped": + job.skipped += 1 + else: + job.failed += 1 + + catalogue = CalibreLibrary(source) + + try: + await catalogue.open() + + async with settings.alchemy_config.get_session() as session: + library = await LibraryService(session=session).get(library_id) + + result = await BookService(session=session).create_many_from_calibre( + catalogue, + library, + allow_duplicates=allow_duplicates, + on_progress=on_progress, + should_stop=lambda: job._stop, + ) + + job.absorb(result) + except Exception as exc: + job.state = ImportState.FAILED + job.error = f"{type(exc).__name__}: {exc}" + finally: + await catalogue.close() + + # An unpacked archive is a second copy of the whole library, and the books + # worth keeping have been copied into the library proper by now. Removed + # even when the run failed — especially then, since nothing will come back + # for it. + if job.workspace is not None: + await asyncio.to_thread(shutil.rmtree, job.workspace, True) + + +registry = CalibreImportRegistry() diff --git a/backend/src/chitai/services/filesystem_library.py b/backend/src/chitai/services/filesystem_library.py index dfde73d..a3a2551 100644 --- a/backend/src/chitai/services/filesystem_library.py +++ b/backend/src/chitai/services/filesystem_library.py @@ -16,6 +16,43 @@ import chitai.database.models as m # - Auto-handle missing values (e.g., skip {series}/ if series is empty) +# Characters that cannot survive being interpolated into a path. The forward slash is +# the one that matters: titles legitimately contain it — "AC/DC", "Him/Her" — and the +# template writes the title straight into a directory name, so an unsanitised one +# silently adds a level and puts the book somewhere `book.path` does not describe. +# Calibre strips these from its own on-disk names and keeps the real title in its +# database, which is how an import surfaces them. +_UNSAFE_IN_PATH = re.compile(r"[/\\\x00-\x1f]") + + +def sanitize_path_component(value: str) -> str: + """Make one metadata value safe to use as a single directory or file name.""" + return _UNSAFE_IN_PATH.sub("_", value).strip() + + +def _safe_components(book_data: dict) -> dict: + """ + A shallow copy of the metadata with the values a path is built from sanitised. + + Only strings are touched, and only the fields the default template interpolates. A + caller's own template can reach anything else in the dict, which is a reason to keep + this conservative rather than to walk the whole structure. + """ + safe = dict(book_data) + + for key in ("title", "series", "series_position"): + if isinstance(safe.get(key), str): + safe[key] = sanitize_path_component(safe[key]) + + if isinstance(safe.get("authors"), list): + safe["authors"] = [ + sanitize_path_component(author) if isinstance(author, str) else author + for author in safe["authors"] + ] + + return safe + + default_path_template = """ /{{book.authors[0] if book.authors else 'Unknown'}} {%- if book.series -%} @@ -101,7 +138,12 @@ class BookPathGenerator: """ - result = self.root_path / Path(self.path_template.render(book=book_data)) + # Sanitised per value, never on the rendered result: the separators the template + # puts *between* author, series and title are the whole point of it, and only the + # values interpolated into it must not contribute any of their own. + result = self.root_path / Path( + self.path_template.render(book=_safe_components(book_data)) + ) # Clean up result = re.sub(r"/+", "/", str(result)) # Remove consecutive backslashes diff --git a/backend/src/chitai/services/opds/opds.py b/backend/src/chitai/services/opds/opds.py index 5b1b01b..f70ad49 100644 --- a/backend/src/chitai/services/opds/opds.py +++ b/backend/src/chitai/services/opds/opds.py @@ -47,8 +47,13 @@ def convert_book_to_entry(book: m.Book) -> Entry: link=[ ImageLink(href=f"/{book.cover_image}", type="image/webp"), *[ + # The only place a content type has to be a string: `Link.type` is + # required, and a null fails the whole feed rather than one entry. + # `application/octet-stream` is the registered way to say "opaque + # bytes", which is exactly what an unnamed format is. AcquisitionLink( - href=f"/opds/download/{book.id}/{file.id}", type=file.content_type + href=f"/opds/download/{book.id}/{file.id}", + type=file.content_type or "application/octet-stream", ) for file in book.files ], diff --git a/backend/src/chitai/services/utils.py b/backend/src/chitai/services/utils.py index 4d1d0d9..bd6b907 100644 --- a/backend/src/chitai/services/utils.py +++ b/backend/src/chitai/services/utils.py @@ -5,6 +5,7 @@ from __future__ import annotations import errno import hashlib +import mimetypes from pathlib import Path import shutil from typing import BinaryIO @@ -238,6 +239,29 @@ async def move_file(src_path: Path, dest_path: Path, create_dirs=True) -> None: # all configured separately, so they can easily be separate mounts. shutil.move(str(src_path), str(dest_path)) +async def copy_file(src_path: Path, dest_path: Path, create_dirs: bool = True) -> None: + """ + Copy a file, streaming it rather than reading it whole. + + `shutil.copy` would block the event loop for as long as the read takes, which for a + 40 MB ebook — and a few thousand of them in a row — is not acceptable. + + Args: + src_path: The file to copy. Left exactly as it is. + dest_path: Where the copy goes. + create_dirs: Create the destination's parent directories first. + """ + if create_dirs and dest_path.parent: + await aios.makedirs(dest_path.parent, exist_ok=True) + + async with ( + aiofiles.open(src_path, "rb") as source, + aiofiles.open(dest_path, "wb") as destination, + ): + while chunk := await source.read(HASH_CHUNK_SIZE): + await destination.write(chunk) + + async def move_dir_contents(source_dir: Path | str, target_dir: Path | str) -> None: """ Move all contents from source directory to target directory. @@ -459,6 +483,66 @@ def get_filename(file: Path | str, ext: bool = True) -> str: return filename.stem +# Content types for the ebook formats `mimetypes` does not know. Python's built-in map +# covers `.epub`, `.pdf`, `.azw3`, `.cbz`, `.cbr` and `.djvu`, and answers `None` for +# every format below — so a library imported from elsewhere, which is where MOBI and +# AZW files come from, stores nothing for them. +# +# That matters downstream because an OPDS acquisition link is how a reader app decides +# whether it can open a file at all. + +# What a client sends when it does not know either. Treated as an absence rather than +# an answer: storing it would be indistinguishable from having determined a format, and +# it is the value browsers post for every extension they do not recognise. +_UNSPECIFIED = "application/octet-stream" + +EBOOK_CONTENT_TYPES = { + "mobi": "application/x-mobipocket-ebook", + "prc": "application/x-mobipocket-ebook", + "azw": "application/vnd.amazon.ebook", + "fb2": "application/x-fictionbook+xml", + "fbz": "application/x-zip-compressed-fb2", + "lit": "application/x-ms-reader", + "lrf": "application/x-sony-bbeb", + "cb7": "application/x-cb7", +} + + +def guess_content_type( + file: Path | str | UploadFile, fallback: str | None = None +) -> str | None: + """ + Name a file's format from its extension. + + The extension is trusted ahead of anything a client said: a browser posts + `application/octet-stream` for every format it does not recognise, which is most + ebook formats, and that answer is worth less than the `.mobi` on the end of the + name. + + Args: + file: The file to name, as a path or an upload. + fallback: What to use when neither table knows the extension — a client-supplied + content type, if there is one. `application/octet-stream` is discarded: it + is the client saying it does not know, which is not information. + + Returns: + The content type, or None when nothing can name it. Null is the honest answer + and the column is nullable: a caller that structurally needs a string should + substitute one where it needs it, rather than have an invented value stored. + """ + extension = get_file_extension(file) + + if known := EBOOK_CONTENT_TYPES.get(extension): + return known + + guessed, _ = mimetypes.guess_type(get_filename(file)) + + if fallback == _UNSPECIFIED: + fallback = None + + return guessed or fallback + + ############################### # ISBN Validation utilities # ############################### diff --git a/backend/tests/calibre_fixtures.py b/backend/tests/calibre_fixtures.py new file mode 100644 index 0000000..bf5327e --- /dev/null +++ b/backend/tests/calibre_fixtures.py @@ -0,0 +1,282 @@ +""" +Build a Calibre library on disk, for tests to read. + +Generated rather than committed as a binary `metadata.db`, because the rows worth +testing are the awkward ones — the year-101 pubdate, a `|` in an author name, a REAL +series index, HTML in a comment — and those are clearer written out in Python than +hidden inside a blob. + +The schema below is Calibre's own, copied from a real library's `sqlite_master`, reduced +to the tables the reader touches. `books_pages_link` is created separately by +`add_pages`: it is recent, and a library made by an older Calibre will not have it. +""" + +from __future__ import annotations + +import shutil +import sqlite3 +from pathlib import Path + + +SCHEMA = """ +CREATE TABLE books ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + title TEXT NOT NULL DEFAULT 'Unknown' COLLATE NOCASE, + sort TEXT COLLATE NOCASE, + timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP, + pubdate TIMESTAMP DEFAULT CURRENT_TIMESTAMP, + series_index REAL NOT NULL DEFAULT 1.0, + author_sort TEXT COLLATE NOCASE, + path TEXT NOT NULL DEFAULT '', + uuid TEXT, + has_cover BOOL DEFAULT 0, + last_modified TIMESTAMP NOT NULL DEFAULT '2000-01-01 00:00:00+00:00' +); +CREATE TABLE authors ( + id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE, + sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name) +); +CREATE TABLE books_authors_link ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, author INTEGER NOT NULL, + UNIQUE(book, author) +); +CREATE TABLE publishers ( + id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE, + sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name) +); +CREATE TABLE books_publishers_link ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, publisher INTEGER NOT NULL, + UNIQUE(book) +); +CREATE TABLE tags ( + id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE, + link TEXT NOT NULL DEFAULT '', UNIQUE (name) +); +CREATE TABLE books_tags_link ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, tag INTEGER NOT NULL, + UNIQUE(book, tag) +); +CREATE TABLE series ( + id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE, + sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE (name) +); +CREATE TABLE books_series_link ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, series INTEGER NOT NULL, + UNIQUE(book) +); +CREATE TABLE languages ( + id INTEGER PRIMARY KEY, lang_code TEXT NOT NULL COLLATE NOCASE, + link TEXT NOT NULL DEFAULT '', UNIQUE(lang_code) +); +CREATE TABLE books_languages_link ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, lang_code INTEGER NOT NULL, + item_order INTEGER NOT NULL DEFAULT 0, UNIQUE(book, lang_code) +); +CREATE TABLE comments ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, + text TEXT NOT NULL COLLATE NOCASE, UNIQUE(book) +); +CREATE TABLE identifiers ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, + type TEXT NOT NULL DEFAULT 'isbn' COLLATE NOCASE, + val TEXT NOT NULL COLLATE NOCASE, UNIQUE(book, type) +); +CREATE TABLE data ( + id INTEGER PRIMARY KEY, book INTEGER NOT NULL, + format TEXT NOT NULL COLLATE NOCASE, uncompressed_size INTEGER NOT NULL, + name TEXT NOT NULL, UNIQUE(book, format) +); +""" + +# Calibre's own "no date". Stored, never null, and a valid date — which is exactly why +# it has to be recognised rather than parsed. +UNDEFINED_DATE = "0101-01-01 00:00:00+00:00" + +# What a `cover.jpg` that PIL cannot read looks like. Real libraries hold these, from +# an interrupted download or a failed conversion. +CORRUPT_COVER = b"\xff\xd8\xff\xe0 not really a jpeg" + + +def write_cover(path: Path) -> None: + """ + Write a real, readable JPEG. + + Generated with PIL rather than embedded as a hex blob: a hand-rolled JPEG that is + subtly malformed fails inside the import as an unrelated error, which is exactly the + confusion this avoids. + """ + from PIL import Image + + Image.new("RGB", (2, 3), (10, 20, 30)).save(path, "JPEG") + + +class CalibreFixture: + """A Calibre library being assembled under `root`.""" + + def __init__(self, root: Path) -> None: + self.root = root + self.root.mkdir(parents=True, exist_ok=True) + + self.connection = sqlite3.connect(self.root / "metadata.db") + self.connection.executescript(SCHEMA) + + def add_pages_table(self) -> None: + """Add `books_pages_link`, which only a recent Calibre creates.""" + self.connection.executescript( + """ + CREATE TABLE books_pages_link ( + book INTEGER PRIMARY KEY, + pages INTEGER DEFAULT 0 NOT NULL, + algorithm INTEGER DEFAULT 0 NOT NULL, + format TEXT DEFAULT '' NOT NULL COLLATE NOCASE, + format_size INTEGER DEFAULT 0 NOT NULL, + timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP, + needs_scan INTEGER NOT NULL DEFAULT 0 + ); + """ + ) + + def add_book( + self, + book_id: int, + title: str, + *, + authors: list[str] | None = None, + pubdate: str = UNDEFINED_DATE, + series: str | None = None, + series_index: float = 1.0, + tags: list[str] | None = None, + publisher: str | None = None, + languages: list[str] | None = None, + comment: str | None = None, + identifiers: dict[str, str] | None = None, + uuid: str | None = None, + pages: int | None = None, + cover: bool = False, + corrupt_cover: bool = False, + formats: dict[str, Path] | None = None, + directory: str | None = None, + ) -> Path: + """ + Add one book, with its files laid out the way Calibre lays them out. + + Args: + formats: Format name (`EPUB`) to a real file to copy in. Its on-disk stem is + Calibre's, not the title — that is the point of the `data` table. + directory: Override the `books.path` value, for testing a row whose + directory is not where the convention would put it. + + Returns: + The book's directory. + """ + author_names = authors or ["Unknown"] + relative = directory or f"{author_names[0]}/{title} ({book_id})" + book_directory = self.root / relative + book_directory.mkdir(parents=True, exist_ok=True) + + self.connection.execute( + "INSERT INTO books (id, title, pubdate, series_index, path, uuid, has_cover) " + "VALUES (?, ?, ?, ?, ?, ?, ?)", + ( + book_id, + title, + pubdate, + series_index, + relative, + uuid or f"uuid-{book_id}", + int(cover or corrupt_cover), + ), + ) + + for name in author_names: + self._link("authors", "books_authors_link", "author", book_id, name) + + for name in tags or []: + self._link("tags", "books_tags_link", "tag", book_id, name) + + if series: + self._link("series", "books_series_link", "series", book_id, series) + + if publisher: + self._link( + "publishers", "books_publishers_link", "publisher", book_id, publisher + ) + + for order, code in enumerate(languages or []): + language_id = self._lookup("languages", "lang_code", code) + self.connection.execute( + "INSERT INTO books_languages_link (book, lang_code, item_order) " + "VALUES (?, ?, ?)", + (book_id, language_id, order), + ) + + if comment is not None: + self.connection.execute( + "INSERT INTO comments (book, text) VALUES (?, ?)", (book_id, comment) + ) + + for name, value in (identifiers or {}).items(): + self.connection.execute( + "INSERT INTO identifiers (book, type, val) VALUES (?, ?, ?)", + (book_id, name, value), + ) + + if pages is not None: + self.connection.execute( + "INSERT INTO books_pages_link (book, pages) VALUES (?, ?)", + (book_id, pages), + ) + + if corrupt_cover: + (book_directory / "cover.jpg").write_bytes(CORRUPT_COVER) + elif cover: + write_cover(book_directory / "cover.jpg") + + for format, origin in (formats or {}).items(): + # Calibre's on-disk stem: sanitised, truncated, and not the title. + stem = f"{title[:40]} - {author_names[0]}".replace(":", "_") + destination = book_directory / f"{stem}.{format.lower()}" + shutil.copy(origin, destination) + + self.connection.execute( + "INSERT INTO data (book, format, uncompressed_size, name) " + "VALUES (?, ?, ?, ?)", + (book_id, format, destination.stat().st_size, stem), + ) + + return book_directory + + def add_missing_format(self, book_id: int, format: str, stem: str) -> None: + """Record a file in the catalogue without putting one on disk.""" + self.connection.execute( + "INSERT INTO data (book, format, uncompressed_size, name) VALUES (?, ?, ?, ?)", + (book_id, format, 1234, stem), + ) + + def _link( + self, table: str, link_table: str, column: str, book_id: int, name: str + ) -> None: + item_id = self._lookup(table, "name", name) + self.connection.execute( + f"INSERT INTO {link_table} (book, {column}) VALUES (?, ?)", + (book_id, item_id), + ) + + def _lookup(self, table: str, column: str, value: str) -> int: + row = self.connection.execute( + f"SELECT id FROM {table} WHERE {column} = ?", (value,) + ).fetchone() + + if row: + return int(row[0]) + + cursor = self.connection.execute( + f"INSERT INTO {table} ({column}) VALUES (?)", (value,) + ) + return int(cursor.lastrowid or 0) + + def commit(self) -> Path: + """Finish writing and return the library root.""" + self.connection.commit() + self.connection.close() + return self.root diff --git a/backend/tests/integration/test_book.py b/backend/tests/integration/test_book.py index 62139dd..574300b 100644 --- a/backend/tests/integration/test_book.py +++ b/backend/tests/integration/test_book.py @@ -1272,3 +1272,94 @@ class TestFileManagement: ) # Should succeed (idempotent) assert response2.status_code == 204 + + +@pytest.mark.asyncio +class TestUnnameableFormats: + """ + A file whose format nothing can name must still round-trip. + + `mimetypes.guess_type` answers None for `.mobi`, `.azw`, `.fb2` and `.lit`, which + is most of what a library imported from elsewhere carries alongside its EPUBs. + `FileMetadataRead.content_type` used to be a required string, so such a book was + created and then failed serialisation on its way back out — a 500 on a book the + reader can otherwise download. + """ + + def upload(self, name: str) -> list[tuple[str, tuple]]: + # `application/octet-stream` is what a browser posts for these, and it is not + # an answer — the extension is what names the format. + return [("files", (name, b"BOOKMOBI\x00 payload", "application/octet-stream"))] + + async def test_a_mobi_is_named_from_its_extension( + self, authenticated_client: AsyncClient + ) -> None: + response = await authenticated_client.post( + "/books?library_id=1", files=self.upload("Dune.mobi"), data={"library_id": 1} + ) + + assert response.status_code == 201 + book = response.json() + assert book["files"][0]["content_type"] == "application/x-mobipocket-ebook" + + detail = await authenticated_client.get(f"/books/{book['id']}") + assert detail.status_code == 200 + + async def test_an_unknown_extension_stores_no_content_type( + self, authenticated_client: AsyncClient + ) -> None: + """Null, not a placeholder — and the book still serialises either way.""" + response = await authenticated_client.post( + "/books?library_id=1", + files=self.upload("Notes.xyzzy"), + data={"library_id": 1}, + ) + + assert response.status_code == 201 + book = response.json() + assert book["files"][0]["content_type"] is None + + detail = await authenticated_client.get(f"/books/{book['id']}") + assert detail.status_code == 200 + assert detail.json()["files"][0]["content_type"] is None + + async def test_the_file_downloads( + self, authenticated_client: AsyncClient + ) -> None: + """Litestar supplies its own media type when the row carries none.""" + created = await authenticated_client.post( + "/books?library_id=1", + files=self.upload("Notes.xyzzy"), + data={"library_id": 1}, + ) + book = created.json() + + response = await authenticated_client.get( + f"/books/download/{book['id']}/{book['files'][0]['id']}" + ) + + assert response.status_code == 200 + assert response.headers["content-type"] == "application/octet-stream" + + async def test_the_opds_feed_survives_a_null_content_type( + self, authenticated_client: AsyncClient + ) -> None: + """ + The one place the type has to be a string. + + `Link.type` is required, so a null fails the whole feed rather than one entry. + OPDS clients speak Basic, not the JWT the rest of the API uses. + """ + await authenticated_client.post( + "/books?library_id=1", + files=self.upload("Notes.xyzzy"), + data={"library_id": 1}, + ) + + feed = await authenticated_client.get( + "/opds/acquisition?feed_id=all&feed_title=All+Books", + auth=("user1@example.com", "password123"), + ) + + assert feed.status_code == 200 + assert 'type="application/octet-stream"' in feed.text diff --git a/backend/tests/integration/test_calibre_import.py b/backend/tests/integration/test_calibre_import.py new file mode 100644 index 0000000..6e46f61 --- /dev/null +++ b/backend/tests/integration/test_calibre_import.py @@ -0,0 +1,400 @@ +""" +Tests for the Calibre import endpoints. + +The API takes a zipped library and nothing else — a desktop Calibre install is usually +not on the server, and importing from a path the server can already see stays a +server-side operation (`litestar calibre-import`). +""" + +import asyncio +import tempfile +import zipfile +from pathlib import Path + +import pytest +from httpx import AsyncClient + +from chitai.services.calibre_import import registry + +from tests.calibre_fixtures import CalibreFixture + + +EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub") +OTHER_EPUB = Path("tests/data_files/The Art of War - Sun Tzu.epub") + + +@pytest.fixture(autouse=True) +def _clear_registry(): + """The registry is a module-level singleton, so it leaks between tests.""" + registry._jobs.clear() + yield + registry._jobs.clear() + + +@pytest.fixture(name="source") +def fx_source(tmp_path: Path) -> Path: + fixture = CalibreFixture(tmp_path / "calibre") + fixture.add_book( + 1, + "The Metamorphosis", + authors=["Franz Kafka"], + tags=["Fiction"], + cover=True, + formats={"EPUB": EPUB}, + ) + fixture.add_book(2, "The Art of War", authors=["Sun Tzu"], formats={"EPUB": OTHER_EPUB}) + fixture.add_book(3, "Metadata Only", authors=["Nobody"]) + + return fixture.commit() + + +def zip_of(root: Path, into: Path, prefix: str = "") -> Path: + """Zip a directory the way a file manager would.""" + into.mkdir(parents=True, exist_ok=True) + archive = into / "library.zip" + + with zipfile.ZipFile(archive, "w") as writing: + for path in sorted(root.rglob("*")): + if path.is_file(): + writing.write(path, f"{prefix}{path.relative_to(root)}") + + return archive + + +async def upload( + client: AsyncClient, + archive: Path, + library_id: int = 1, + allow_duplicates: bool = False, +) -> tuple[int, dict]: + response = await client.post( + f"/libraries/{library_id}/imports/calibre/upload", + files=[("archive", (archive.name, archive.read_bytes(), "application/zip"))], + data={"allow_duplicates": str(allow_duplicates).lower()}, + ) + + return response.status_code, response.json() + + +async def wait_for(client: AsyncClient, job_id: str) -> dict: + """Poll until the job is no longer running, the way the screen does.""" + for _ in range(200): + response = await client.get(f"/libraries/imports/{job_id}") + assert response.status_code == 200 + + job = response.json() + if job["state"] != "running": + return job + + await asyncio.sleep(0.05) + + raise AssertionError("the import never finished") + + +async def test_an_uploaded_library_imports( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + status, job = await upload( + authenticated_client, zip_of(source, tmp_path / "out", prefix="Calibre Library/") + ) + + assert status == 202 + assert job["state"] == "running" + assert job["library_id"] == 1 + + # The archive's name, not the temp directory it was unpacked into. + assert job["source"] == "library.zip" + + finished = await wait_for(authenticated_client, job["id"]) + + assert finished["state"] == "finished" + assert finished["total"] == 3 + assert finished["created"] == 2 + assert finished["skipped"] == 1 + assert finished["failed"] == 0 + assert finished["error"] is None + assert finished["current_title"] is None + + listed = await authenticated_client.get("/books?library_id=1") + titles = [book["title"] for book in listed.json()["items"]] + assert "The Metamorphosis" in titles + assert "The Art of War" in titles + + +async def test_a_library_zipped_without_a_wrapping_folder( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """Zipping the contents is as common as zipping the folder.""" + status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out")) + + assert status == 202 + finished = await wait_for(authenticated_client, job["id"]) + assert finished["created"] == 2 + + +async def test_the_unpacked_copy_is_cleaned_up( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """ + An unpacked archive is a second copy of the whole library. + + The books worth keeping have been copied into the library by the time the job ends, + so nothing is lost with it — and nothing will come back for it. + """ + status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out")) + assert status == 202 + + workspace = registry.get(job["id"]).workspace + assert workspace is not None + + await wait_for(authenticated_client, job["id"]) + + assert not workspace.exists() + + +async def test_uploading_the_same_library_twice_imports_nothing_new( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """Re-running is safe, which is what makes an interrupted import resumable.""" + archive = zip_of(source, tmp_path / "out") + + for _ in range(2): + _, job = await upload(authenticated_client, archive) + finished = await wait_for(authenticated_client, job["id"]) + + assert finished["created"] == 0 + assert finished["skipped"] == 3 + + +async def test_allow_duplicates_stores_the_files_again( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """ + The one option the screen offers, and it has to reach the import. + + Without it the second pass skips everything, which is the previous test. + """ + archive = zip_of(source, tmp_path / "out") + + _, first = await upload(authenticated_client, archive) + await wait_for(authenticated_client, first["id"]) + + _, second = await upload(authenticated_client, archive, allow_duplicates=True) + finished = await wait_for(authenticated_client, second["id"]) + + assert finished["created"] == 2 + assert finished["skipped"] == 1 # still the book with no files + + +async def test_two_imports_into_one_library_conflict( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + archive = zip_of(source, tmp_path / "out") + + first_status, first = await upload(authenticated_client, archive) + assert first_status == 202 + + second_status, second = await upload(authenticated_client, archive) + + assert second_status == 409 + assert second["extra"]["job_id"] == first["id"] + + await wait_for(authenticated_client, first["id"]) + + +async def test_a_finished_import_does_not_block_the_next_one( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + archive = zip_of(source, tmp_path / "out") + + _, first = await upload(authenticated_client, archive) + await wait_for(authenticated_client, first["id"]) + + status, second = await upload(authenticated_client, archive) + + assert status == 202 + await wait_for(authenticated_client, second["id"]) + + +async def test_cancelling_stops_after_the_current_book( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """ + Cancelling is not aborting: a book abandoned mid-copy would leave files with no row. + + Whether this cancels before any book, after one, or after the lot is a race — the + catalogue is three books long. What must hold either way is that the state is + terminal and every book it did import is complete. + """ + _, job = await upload(authenticated_client, zip_of(source, tmp_path / "out")) + + cancelled = await authenticated_client.delete(f"/libraries/imports/{job['id']}") + assert cancelled.status_code == 200 + + final = await wait_for(authenticated_client, job["id"]) + + assert final["state"] in {"cancelled", "finished"} + + listed = await authenticated_client.get("/books?library_id=1") + for book in listed.json()["items"]: + assert book["files"] + + +async def test_failures_are_reported_on_the_job( + authenticated_client: AsyncClient, tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """One book failing is recorded and does not stop the run.""" + fixture = CalibreFixture(tmp_path / "calibre") + fixture.add_book(1, "Fine", authors=["A"], formats={"EPUB": EPUB}) + fixture.add_book(2, "Doomed", authors=["B"], formats={"EPUB": OTHER_EPUB}) + root = fixture.commit() + + from chitai.services.book import BookService + + original = BookService.create + + async def fail_on_the_second(self, data, *args, **kwargs): + if isinstance(data, dict) and data.get("title") == "Doomed": + raise RuntimeError("no room on the shelf") + return await original(self, data, *args, **kwargs) + + monkeypatch.setattr(BookService, "create", fail_on_the_second) + + _, job = await upload(authenticated_client, zip_of(root, tmp_path / "out")) + finished = await wait_for(authenticated_client, job["id"]) + + assert finished["state"] == "finished" + assert finished["created"] == 1 + assert finished["failed"] == 1 + assert finished["failures"][0]["calibre_id"] == 2 + assert "no room on the shelf" in finished["failures"][0]["reason"] + + +async def test_a_second_copy_is_counted_as_a_possible_duplicate( + authenticated_client: AsyncClient, tmp_path: Path +) -> None: + """A count, not the records — the duplicates screen is what shows them.""" + padded = tmp_path / "padded.epub" + padded.write_bytes(EPUB.read_bytes() + b"\0" * 64) + + fixture = CalibreFixture(tmp_path / "calibre") + fixture.add_book(1, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": EPUB}) + fixture.add_book( + 2, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": padded} + ) + root = fixture.commit() + + _, job = await upload(authenticated_client, zip_of(root, tmp_path / "out")) + finished = await wait_for(authenticated_client, job["id"]) + + assert finished["created"] == 2 + assert finished["possible_duplicates"] == 1 + + +async def test_polling_an_unknown_job(authenticated_client: AsyncClient) -> None: + response = await authenticated_client.get("/libraries/imports/not-a-job") + assert response.status_code == 404 + + +async def test_cancelling_an_unknown_job(authenticated_client: AsyncClient) -> None: + response = await authenticated_client.delete("/libraries/imports/not-a-job") + assert response.status_code == 404 + + +async def test_importing_into_a_library_that_does_not_exist( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + status, _ = await upload( + authenticated_client, zip_of(source, tmp_path / "out"), library_id=999 + ) + + assert status == 404 + + +async def test_importing_into_a_read_only_library( + authenticated_client: AsyncClient, source: Path, tmp_path: Path +) -> None: + """A read-only library points at a tree Chitai does not own.""" + root = tmp_path / "read-only" + root.mkdir() + + created = await authenticated_client.post( + "/libraries", + json={"name": "Read Only", "root_path": str(root), "read_only": True}, + ) + assert created.status_code == 201 + + status, body = await upload( + authenticated_client, + zip_of(source, tmp_path / "out"), + library_id=created.json()["id"], + ) + + assert status == 400 + assert "read-only" in body["detail"] + + +async def test_an_import_needs_authentication( + client: AsyncClient, source: Path, tmp_path: Path +) -> None: + status, _ = await upload(client, zip_of(source, tmp_path / "out")) + + assert status == 401 + + +class TestRefusedArchives: + """Everything wrong with an archive is answered now, not as a job that fails later.""" + + async def test_a_hostile_archive( + self, authenticated_client: AsyncClient, tmp_path: Path + ) -> None: + """Zip slip.""" + archive = tmp_path / "hostile.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("metadata.db", "not really") + writing.writestr("../../escaped.txt", "gotcha") + + status, body = await upload(authenticated_client, archive) + + assert status == 400 + assert "outside itself" in body["detail"] + assert registry._jobs == {} + + async def test_something_that_is_not_a_zip( + self, authenticated_client: AsyncClient, tmp_path: Path + ) -> None: + archive = tmp_path / "notes.txt" + archive.write_bytes(b"just some text") + + status, body = await upload(authenticated_client, archive) + + assert status == 400 + assert "not a zip" in body["detail"] + + async def test_an_archive_with_no_catalogue( + self, authenticated_client: AsyncClient, tmp_path: Path + ) -> None: + archive = tmp_path / "books.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("Some Book.epub", "content") + + status, body = await upload(authenticated_client, archive) + + assert status == 400 + assert "no metadata.db" in body["detail"] + + async def test_a_refusal_leaves_no_temp_files( + self, authenticated_client: AsyncClient, tmp_path: Path + ) -> None: + """Every refusal path removes the workspace it had already made.""" + before = set(Path(tempfile.gettempdir()).glob("tmp*")) + + archive = tmp_path / "books.zip" + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("Some Book.epub", "content") + + await upload(authenticated_client, archive) + + assert set(Path(tempfile.gettempdir()).glob("tmp*")) == before diff --git a/backend/tests/unit/test_calibre.py b/backend/tests/unit/test_calibre.py new file mode 100644 index 0000000..fce5d97 --- /dev/null +++ b/backend/tests/unit/test_calibre.py @@ -0,0 +1,392 @@ +import zipfile +from datetime import date +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from chitai.services.calibre import ( + CalibreLibrary, + CalibreLibraryError, + extract_calibre_archive, + format_series_index, + parse_date, + strip_html, + unescape_author, +) + +from tests.calibre_fixtures import UNDEFINED_DATE, CalibreFixture + + +EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub") +PDF = Path("tests/data_files/Calculus Made Easy - Silvanus Thompson.pdf") + + +@pytest.fixture(name="library_root") +def fx_library_root(tmp_path: Path) -> Path: + """A small Calibre library covering the rows that are easy to read wrongly.""" + fixture = CalibreFixture(tmp_path / "Calibre Library") + fixture.add_pages_table() + + fixture.add_book( + 1, + "The Metamorphosis", + authors=["Franz Kafka"], + pubdate="1915-10-15 00:00:00+00:00", + tags=["Fiction", "Absurdist"], + publisher="Kurt Wolff Verlag", + languages=["deu", "eng"], + comment="

A travelling salesman.

He wakes up changed.

", + identifiers={"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"}, + uuid="11111111-2222-3333-4444-555555555555", + pages=201, + cover=True, + formats={"EPUB": EPUB}, + ) + + # Volume seven of a series, and no publication date — the two values most likely to + # be carried through verbatim when they should not be. + fixture.add_book( + 2, + "Persepolis Rising", + # Calibre escapes the comma and nothing else, so the space after it is stored + # as-is: `Corey, Jr.` is written `Corey| Jr.`. + authors=["Corey| Jr., James S. A."], + series="The Expanse", + series_index=7.0, + formats={"EPUB": EPUB}, + ) + + # A novella between two novels: a fractional position is real and must survive. + fixture.add_book( + 3, "Strange Dogs", series="The Expanse", series_index=6.5, formats={"PDF": PDF} + ) + + # Every row Calibre will happily hold and Chitai cannot use: no files at all. + fixture.add_book(4, "Metadata Only") + + # A catalogue row whose file is not on disk. + fixture.add_book(5, "Lost Book") + fixture.add_missing_format(5, "EPUB", "Lost Book - Unknown") + + return fixture.commit() + + +async def test_reads_a_book_whole(library_root: Path) -> None: + async with CalibreLibrary(library_root) as library: + assert await library.count() == 5 + books = await library.books() + + book = books[0] + + assert book.calibre_id == 1 + assert book.title == "The Metamorphosis" + assert book.authors == ["Franz Kafka"] + assert book.published_date == date(1915, 10, 15) + assert book.tags == ["Absurdist", "Fiction"] + assert book.publisher == "Kurt Wolff Verlag" + assert book.pages == 201 + assert book.uuid == "11111111-2222-3333-4444-555555555555" + + # One language, and the one Calibre put first. + assert book.language == "deu" + + # Reported as Calibre wrote them: folding `amazon` onto `asin` is the importer's + # job, not the reader's. + assert book.identifiers == {"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"} + + assert book.cover is not None + assert book.cover.is_file() + + assert len(book.files) == 1 + assert book.files[0].format == "EPUB" + assert book.files[0].path.is_file() + # The stem is Calibre's, truncated and sanitised — never the title. + assert book.files[0].path.name != f"{book.title}.epub" + + +async def test_the_undefined_date_is_not_a_date(library_root: Path) -> None: + """`0101-01-01` parses fine, which is exactly the problem.""" + async with CalibreLibrary(library_root) as library: + books = {book.calibre_id: book for book in await library.books()} + + assert books[2].published_date is None + + +async def test_series_position_is_a_plain_string(library_root: Path) -> None: + async with CalibreLibrary(library_root) as library: + books = {book.calibre_id: book for book in await library.books()} + + assert books[2].series == "The Expanse" + assert books[2].series_position == "7" + + assert books[3].series_position == "6.5" + + # `series_index` defaults to 1.0 for every book, so a position without a series + # would invent a volume one out of nothing. + assert books[1].series is None + assert books[1].series_position is None + + +async def test_author_commas_are_unescaped(library_root: Path) -> None: + async with CalibreLibrary(library_root) as library: + books = {book.calibre_id: book for book in await library.books()} + + assert books[2].authors == ["Corey, Jr., James S. A."] + + +async def test_comments_come_back_as_text(library_root: Path) -> None: + async with CalibreLibrary(library_root) as library: + books = {book.calibre_id: book for book in await library.books()} + + assert books[1].description == "A travelling salesman.\nHe wakes up changed." + assert books[2].description is None + + +async def test_files_are_reported_whether_or_not_they_exist(library_root: Path) -> None: + """ + The reader says what the catalogue says. Whether the bytes are there is a question + for whoever is about to copy them, which stats them anyway. + """ + async with CalibreLibrary(library_root) as library: + books = {book.calibre_id: book for book in await library.books()} + + assert books[4].files == [] + + assert len(books[5].files) == 1 + assert not books[5].files[0].path.exists() + + +async def test_a_library_without_the_pages_table_still_reads(tmp_path: Path) -> None: + """`books_pages_link` is recent; an older library simply does not have it.""" + fixture = CalibreFixture(tmp_path / "Old Library") + fixture.add_book(1, "Old Book", formats={"EPUB": EPUB}) + root = fixture.commit() + + async with CalibreLibrary(root) as library: + books = await library.books() + + assert books[0].pages is None + + +async def test_the_original_is_never_opened(library_root: Path) -> None: + """ + The catalogue is copied before it is read, and the copy goes away afterwards. + + Calibre may be running and writing; this is what keeps a live library out of it. + """ + before = (library_root / "metadata.db").read_bytes() + + library = CalibreLibrary(library_root) + await library.open() + workspace = library._workspace + + assert workspace is not None and (workspace / "metadata.db").is_file() + + await library.close() + + assert not workspace.exists() + assert (library_root / "metadata.db").read_bytes() == before + + +async def test_closing_twice_is_harmless(library_root: Path) -> None: + library = CalibreLibrary(library_root) + await library.open() + await library.close() + await library.close() + + +async def test_a_directory_that_is_not_a_calibre_library(tmp_path: Path) -> None: + with pytest.raises(CalibreLibraryError, match="not a Calibre library"): + await CalibreLibrary(tmp_path).open() + + +@pytest.mark.parametrize( + ("stored", "expected"), + [ + ("2017-12-04 04:00:00+00:00", date(2017, 12, 4)), + ("2001-07-02 00:00:00+00:00", date(2001, 7, 2)), + ("1999-01-31", date(1999, 1, 31)), + # Calibre's sentinel, and anything else implausibly early. + (UNDEFINED_DATE, None), + ("0101-01-01", None), + (None, None), + ("", None), + ("not a date", None), + ], +) +def test_parse_date(stored: str | None, expected: date | None) -> None: + assert parse_date(stored) == expected + + +@pytest.mark.parametrize( + ("index", "expected"), + [ + (7.0, "7"), + (1.0, "1"), + (6.5, "6.5"), + (0.0, "0"), + (12.25, "12.25"), + (None, None), + ], +) +def test_format_series_index(index: float | None, expected: str | None) -> None: + assert format_series_index(index) == expected + + +@pytest.mark.parametrize( + ("stored", "expected"), + [ + ("Doyle| Sir Arthur Conan", "Doyle, Sir Arthur Conan"), + ("Franz Kafka", "Franz Kafka"), + (" Herman Melville ", "Herman Melville"), + ], +) +def test_unescape_author(stored: str, expected: str) -> None: + assert unescape_author(stored) == expected + + +@pytest.mark.parametrize( + ("html", "expected"), + [ + ("

One.

Two.

", "One.\nTwo."), + ("Plain text", "Plain text"), + ("
A
B
", "A\nB"), + ("

Café & bar

", "Café & bar"), + ("
  • One
  • Two
", "One\nTwo"), + # Markup carrying no text at all is nothing, not an empty description. + ("

", None), + ("", None), + (None, None), + ], +) +def test_strip_html(html: str | None, expected: str | None) -> None: + assert strip_html(html) == expected + + +class TestArchives: + """A Calibre library that arrives zipped rather than as a path.""" + + def zipped(self, root: Path, into: Path, prefix: str = "") -> Path: + """Zip a directory the way a file manager would.""" + archive = into / "library.zip" + + with zipfile.ZipFile(archive, "w") as writing: + for path in sorted(root.rglob("*")): + if path.is_file(): + writing.write(path, f"{prefix}{path.relative_to(root)}") + + return archive + + async def test_a_library_zipped_at_its_root( + self, library_root: Path, tmp_path: Path + ) -> None: + archive = self.zipped(library_root, tmp_path) + destination = tmp_path / "unpacked" + destination.mkdir() + + catalogue = await extract_calibre_archive(archive, destination) + + assert catalogue == destination + async with CalibreLibrary(catalogue) as library: + assert await library.count() == 5 + + async def test_a_library_zipped_inside_a_folder( + self, library_root: Path, tmp_path: Path + ) -> None: + """Zipping the folder itself is at least as common as zipping its contents.""" + archive = self.zipped(library_root, tmp_path, prefix="Calibre Library/") + destination = tmp_path / "unpacked" + destination.mkdir() + + catalogue = await extract_calibre_archive(archive, destination) + + assert catalogue == destination / "Calibre Library" + async with CalibreLibrary(catalogue) as library: + assert await library.count() == 5 + + async def test_an_entry_pointing_outside_the_archive_is_refused( + self, tmp_path: Path + ) -> None: + """ + Zip slip. `ZipFile.extract` sanitises names itself, but relying on that silently + is how the next person to change the extraction call reintroduces it. + """ + archive = tmp_path / "hostile.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("metadata.db", "not really") + writing.writestr("../../escaped.txt", "gotcha") + + destination = tmp_path / "unpacked" + destination.mkdir() + + with pytest.raises(CalibreLibraryError, match="outside itself"): + await extract_calibre_archive(archive, destination) + + assert not (tmp_path.parent / "escaped.txt").exists() + + async def test_something_that_is_not_a_zip(self, tmp_path: Path) -> None: + archive = tmp_path / "not.zip" + archive.write_bytes(b"PK-ish, but no") + + destination = tmp_path / "unpacked" + destination.mkdir() + + with pytest.raises(CalibreLibraryError, match="not a zip file"): + await extract_calibre_archive(archive, destination) + + async def test_an_archive_with_no_catalogue(self, tmp_path: Path) -> None: + archive = tmp_path / "books.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("Some Book.epub", "content") + + destination = tmp_path / "unpacked" + destination.mkdir() + + with pytest.raises(CalibreLibraryError, match="no metadata.db"): + await extract_calibre_archive(archive, destination) + + # Refused before anything was written. + assert list(destination.iterdir()) == [] + + async def test_a_catalogue_buried_too_deep(self, tmp_path: Path) -> None: + """Somebody's whole backup tree is not a library, however much it contains one.""" + archive = tmp_path / "backup.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("backups/2026/january/library/metadata.db", "not really") + + destination = tmp_path / "unpacked" + destination.mkdir() + + with pytest.raises(CalibreLibraryError, match="within 3 levels"): + await extract_calibre_archive(archive, destination) + + async def test_an_archive_too_big_for_the_disk( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + """ + Checked before writing, not discovered part-way through. + + A full disk takes the whole application down, and the size is in the archive + already. + """ + archive = tmp_path / "huge.zip" + + with zipfile.ZipFile(archive, "w") as writing: + writing.writestr("metadata.db", "not really") + + destination = tmp_path / "unpacked" + destination.mkdir() + + monkeypatch.setattr( + "chitai.services.calibre.shutil.disk_usage", + lambda _path: SimpleNamespace(total=1024, used=1024, free=0), + ) + + with pytest.raises(CalibreLibraryError, match="only"): + await extract_calibre_archive(archive, destination) + + assert list(destination.iterdir()) == [] diff --git a/backend/tests/unit/test_content_type.py b/backend/tests/unit/test_content_type.py new file mode 100644 index 0000000..f740c6e --- /dev/null +++ b/backend/tests/unit/test_content_type.py @@ -0,0 +1,64 @@ +from pathlib import Path + +import pytest + +from chitai.services.utils import guess_content_type + + +@pytest.mark.parametrize( + ("filename", "expected"), + [ + # What `mimetypes` already knows, kept here so a host with a thin + # /etc/mime.types cannot change the answer without a test noticing. + ("Frankenstein.epub", "application/epub+zip"), + ("Calculus.pdf", "application/pdf"), + ("Persepolis.azw3", "application/vnd.amazon.mobi8-ebook"), + ("Watchmen.cbz", "application/vnd.comicbook+zip"), + # What it does not, and where a Calibre library's older formats live. + ("Dune.mobi", "application/x-mobipocket-ebook"), + ("Dune.prc", "application/x-mobipocket-ebook"), + ("Dune.azw", "application/vnd.amazon.ebook"), + ("Voyna i Mir.fb2", "application/x-fictionbook+xml"), + ("Voyna i Mir.fbz", "application/x-zip-compressed-fb2"), + ("Reader.lit", "application/x-ms-reader"), + ("Reader.lrf", "application/x-sony-bbeb"), + ("Watchmen.cb7", "application/x-cb7"), + # Case is not part of the answer, and Calibre writes formats uppercase. + ("Dune.MOBI", "application/x-mobipocket-ebook"), + # Nothing can name these, and None is the answer rather than a placeholder. + ("Notes.xyzzy", None), + ("README", None), + ], +) +def test_guess_content_type(filename: str, expected: str | None) -> None: + assert guess_content_type(Path(filename)) == expected + # A str and a Path must agree, and an upload's `filename` carries its relative + # path, so a name with directories in front of it has to resolve the same way. + assert guess_content_type(filename) == expected + assert guess_content_type(f"Some Author/Some Book/{filename}") == expected + + +def test_fallback_is_used_only_when_the_extension_says_nothing() -> None: + """A client's claim fills a gap; it never overrides the name.""" + assert ( + guess_content_type(Path("Dune.mobi"), fallback="application/pdf") + == "application/x-mobipocket-ebook" + ) + assert ( + guess_content_type(Path("Notes.xyzzy"), fallback="application/epub+zip") + == "application/epub+zip" + ) + + +def test_an_unspecified_fallback_is_not_an_answer() -> None: + """ + `application/octet-stream` from a client is it saying it does not know. + + Browsers post exactly that for every extension they do not recognise, which is most + ebook formats. Storing it would be indistinguishable from having determined a + format, so it is discarded and the column keeps its null. + """ + assert ( + guess_content_type(Path("Notes.xyzzy"), fallback="application/octet-stream") + is None + ) diff --git a/backend/tests/unit/test_filesystem_library.py b/backend/tests/unit/test_filesystem_library.py new file mode 100644 index 0000000..8e6edca --- /dev/null +++ b/backend/tests/unit/test_filesystem_library.py @@ -0,0 +1,69 @@ +"""Tests for BookPathGenerator.""" + +from pathlib import Path + +from chitai.services.filesystem_library import BookPathGenerator, sanitize_path_component + + +ROOT = Path("/library") + + +def path_for(**book) -> Path: + return BookPathGenerator(ROOT).generate_path(book) + + +def test_author_and_title() -> None: + assert path_for(title="Dune", authors=["Frank Herbert"]) == ( + ROOT / "Frank Herbert" / "Dune" + ) + + +def test_a_book_with_no_authors() -> None: + assert path_for(title="Beowulf", authors=[]) == ROOT / "Unknown" / "Beowulf" + + +def test_a_series_adds_a_level_and_pads_the_position() -> None: + assert path_for( + title="Persepolis Rising", + authors=["James S. A. Corey"], + series="The Expanse", + series_position="7", + ) == ROOT / "James S. A. Corey" / "The Expanse" / "07 - Persepolis Rising" + + +def test_a_slash_in_a_title_does_not_add_a_directory() -> None: + """ + The separators in the path come from the template, never from the metadata. + + A title with a slash in it — "AC/DC", "Him/Her" — would otherwise put the book one + level below where `book.path` says it is, which is what deletes, moves and file + lookups all act on. Calibre keeps the real title in its database and strips this + from its own directory names, so an import is where they surface. + """ + generated = path_for(title="Back in Black: AC/DC", authors=["Murray Engleheart"]) + + assert generated == ROOT / "Murray Engleheart" / "Back in Black: AC_DC" + assert generated.relative_to(ROOT).parts == ("Murray Engleheart", "Back in Black: AC_DC") + + +def test_a_slash_in_an_author_or_series_is_handled_too() -> None: + assert path_for(title="Split", authors=["A/B Collective"]) == ( + ROOT / "A_B Collective" / "Split" + ) + assert path_for( + title="Volume One", authors=["Someone"], series="Either/Or", series_position="1" + ) == ROOT / "Someone" / "Either_Or" / "01 - Volume One" + + +def test_control_characters_are_removed() -> None: + assert path_for(title="Line\nBreak", authors=["Someone"]) == ( + ROOT / "Someone" / "Line_Break" + ) + + +def test_sanitize_path_component() -> None: + assert sanitize_path_component("AC/DC") == "AC_DC" + assert sanitize_path_component("back\\slash") == "back_slash" + assert sanitize_path_component(" padded ") == "padded" + # Colons and other punctuation are legal in a path and are left alone. + assert sanitize_path_component("Title: Subtitle") == "Title: Subtitle" diff --git a/backend/tests/unit/test_services/test_calibre_import_service.py b/backend/tests/unit/test_services/test_calibre_import_service.py new file mode 100644 index 0000000..87f2617 --- /dev/null +++ b/backend/tests/unit/test_services/test_calibre_import_service.py @@ -0,0 +1,415 @@ +"""Tests for importing a Calibre library through BookService.""" + +from pathlib import Path + +import pytest + +from chitai.config import settings +from chitai.database import models as m +from chitai.services import BookService +from chitai.services.calibre import CalibreLibrary + +from tests.calibre_fixtures import CalibreFixture + + +DATA_FILES = Path("tests/data_files") +EPUB = DATA_FILES / "Metamorphosis - Franz Kafka.epub" +OTHER_EPUB = DATA_FILES / "The Art of War - Sun Tzu.epub" +PDF = DATA_FILES / "Calculus Made Easy - Silvanus Thompson.pdf" + + +@pytest.fixture(name="calibre_root") +def fx_calibre_root(tmp_path: Path) -> Path: + """Three books: one plain, one in two formats, one Chitai cannot use.""" + fixture = CalibreFixture(tmp_path / "source") + fixture.add_pages_table() + + fixture.add_book( + 1, + "The Metamorphosis", + authors=["Franz Kafka"], + pubdate="1915-10-15 00:00:00+00:00", + tags=["Fiction", "Absurdist"], + publisher="Kurt Wolff Verlag", + languages=["deu"], + comment="

He wakes up changed.

", + identifiers={"isbn": "978-0-486-29030-0", "amazon": "B01N5IB20Q"}, + uuid="11111111-2222-3333-4444-555555555555", + pages=201, + cover=True, + formats={"EPUB": EPUB}, + ) + + fixture.add_book( + 2, + "The Art of War", + authors=["Sun Tzu"], + series="Classics", + series_index=3.0, + formats={"EPUB": OTHER_EPUB, "PDF": PDF}, + ) + + fixture.add_book(3, "Metadata Only", authors=["Nobody"]) + + return fixture.commit() + + +async def library_of(root: Path) -> CalibreLibrary: + source = CalibreLibrary(root) + await source.open() + return source + + +async def test_imports_a_catalogue( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert result.total == 3 + assert len(result.created) == 2 + + # The book with no files is left out: a record with nothing to read, and a directory + # to match, is worse than not importing it. + assert [skipped.calibre_id for skipped in result.skipped] == [3] + assert result.skipped[0].reason == "no files in the catalogue" + assert result.failed == [] + + book = await books_service.get(result.created[0]) + + assert book.title == "The Metamorphosis" + assert [author.name for author in book.authors] == ["Franz Kafka"] + assert sorted(tag.name for tag in book.tags) == ["Absurdist", "Fiction"] + assert book.publisher is not None and book.publisher.name == "Kurt Wolff Verlag" + assert book.published_date is not None and book.published_date.year == 1915 + assert book.language == "deu" + assert book.pages == 201 + assert book.cover_image is not None + + # The HTML is gone; `Book.description` is rendered as text. + assert book.description == "He wakes up changed." + + +async def test_two_formats_are_one_book( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + book = await books_service.get(result.created[1]) + + assert book.title == "The Art of War" + assert sorted(Path(file.path).suffix for file in book.files) == [".epub", ".pdf"] + + # A REAL series index reaches the column as the string everything else writes. + assert book.series is not None and book.series.title == "Classics" + assert book.series_position == "3" + + +async def test_identifiers_are_folded_onto_chitai_schemes( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + """ + `amazon` becomes `asin`, a hyphenated ISBN survives, and the Calibre uuid is kept. + + The uuid is deliberately not stored under `uuid`, which duplicate matching ignores + because an EPUB regenerates one per build. Calibre's is stable, so it is the durable + link back to the row it came from. + """ + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + book = await books_service.get(result.created[0]) + identifiers = {identifier.name: identifier.value for identifier in book.identifiers} + + assert identifiers["asin"] == "B01N5IB20Q" + assert identifiers["isbn-13"] == "9780486290300" + assert identifiers["calibre-uuid"] == "11111111-2222-3333-4444-555555555555" + + matching = { + identifier.name: identifier.normalized_value for identifier in book.identifiers + } + + # Stored under its own name, matched under one scheme for both ISBN forms. + assert matching["isbn-13"] == "isbn:9780486290300" + + # And the uuid carries a real matching key, which is the whole reason it is not + # filed under `uuid`. + assert matching["calibre-uuid"] is not None + + +async def test_the_source_library_is_left_alone( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + """Files are copied. Moving them would leave `metadata.db` pointing at nothing.""" + before = { + path: path.stat().st_mtime_ns + for path in sorted(calibre_root.rglob("*")) + if path.is_file() + } + + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + after = { + path: path.stat().st_mtime_ns + for path in sorted(calibre_root.rglob("*")) + if path.is_file() + } + + assert after == before + + # And the copies are really there, under the library's own layout. + for book_id in result.created: + book = await books_service.get(book_id) + for file in book.files: + assert (Path(book.path or "") / file.path).is_file() + + +async def test_importing_twice_creates_nothing( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + """ + Re-running is safe with no bookkeeping: the bytes are recognised wherever they sit. + + This is what makes an interrupted import resumable by simply running it again. + """ + for _ in range(2): + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert result.created == [] + assert sorted(skipped.reason for skipped in result.skipped) == [ + "already stored", + "already stored", + "no files in the catalogue", + ] + + held_by = [ + skipped.book_id for skipped in result.skipped if skipped.reason == "already stored" + ] + assert all(book_id is not None for book_id in held_by) + + +async def test_a_file_the_catalogue_lists_but_disk_does_not( + books_service: BookService, test_library: m.Library, tmp_path: Path +) -> None: + """Calibre keeps the row when a file is moved away behind its back.""" + fixture = CalibreFixture(tmp_path / "source") + fixture.add_book(1, "Present", authors=["A"], formats={"EPUB": EPUB}) + fixture.add_book(2, "Absent", authors=["B"]) + fixture.add_missing_format(2, "EPUB", "Absent - B") + root = fixture.commit() + + source = await library_of(root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert len(result.created) == 1 + assert [(s.calibre_id, s.reason) for s in result.skipped] == [(2, "no files on disk")] + + +async def test_one_broken_book_does_not_stop_the_import( + books_service: BookService, test_library: m.Library, tmp_path: Path +) -> None: + """ + A failure is recorded and the run continues, leaving no files behind for it. + + An orphaned directory would make the next attempt reserve `title (2)` and look as + though it had worked. + """ + fixture = CalibreFixture(tmp_path / "source") + fixture.add_book(1, "First", authors=["A"], formats={"EPUB": EPUB}) + fixture.add_book(2, "Doomed", authors=["B"], formats={"EPUB": OTHER_EPUB}) + fixture.add_book(3, "Third", authors=["C"], formats={"PDF": PDF}) + root = fixture.commit() + + original = books_service.create + + async def fail_on_the_second(data, *args, **kwargs): + if isinstance(data, dict) and data.get("title") == "Doomed": + raise RuntimeError("no room on the shelf") + return await original(data, *args, **kwargs) + + books_service.create = fail_on_the_second # type: ignore[method-assign] + + source = await library_of(root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + books_service.create = original # type: ignore[method-assign] + + assert len(result.created) == 2 + assert len(result.failed) == 1 + assert result.failed[0].calibre_id == 2 + assert "no room on the shelf" in result.failed[0].reason + + # Nothing of the failed book was left in the library. Checked against the path the + # template would have produced, rather than by walking the root — the Calibre source + # sits under it in these tests, and its own files are meant to still be there. + assert not (Path(test_library.root_path) / "B").exists() + + # And the books either side of it are where they should be. + for book_id in result.created: + book = await books_service.get(book_id) + for file in book.files: + assert (Path(book.path or "") / file.path).is_file() + + +async def test_a_cover_that_cannot_be_read_is_not_fatal( + books_service: BookService, test_library: m.Library, tmp_path: Path +) -> None: + """ + A truncated `cover.jpg` costs the cover, not the book. + + Real libraries hold them, from an interrupted download or a failed conversion, and + the cover is the one thing in the directory that can be replaced from the book page. + """ + fixture = CalibreFixture(tmp_path / "source") + fixture.add_book( + 1, "Unreadable Cover", authors=["A"], corrupt_cover=True, formats={"EPUB": EPUB} + ) + root = fixture.commit() + + source = await library_of(root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert result.failed == [] + assert len(result.created) == 1 + + book = await books_service.get(result.created[0]) + assert book.cover_image is None + assert len(book.files) == 1 + + +async def test_shared_authors_and_tags_are_one_row_each( + books_service: BookService, test_library: m.Library, tmp_path: Path, session +) -> None: + """Two books by one author must not produce two `Author` rows.""" + fixture = CalibreFixture(tmp_path / "source") + fixture.add_book( + 1, "One", authors=["Franz Kafka"], tags=["Fiction"], formats={"EPUB": EPUB} + ) + fixture.add_book( + 2, "Two", authors=["Franz Kafka"], tags=["Fiction"], formats={"EPUB": OTHER_EPUB} + ) + root = fixture.commit() + + source = await library_of(root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert len(result.created) == 2 + + first, second = [await books_service.get(book_id) for book_id in result.created] + + assert first.authors[0].id == second.authors[0].id + assert first.tags[0].id == second.tags[0].id + + +async def test_a_second_copy_is_reported_not_refused( + books_service: BookService, test_library: m.Library, tmp_path: Path +) -> None: + """ + Two catalogue rows for one book, with different bytes, both import. + + File-level dedupe cannot see it — the archives differ — so book-level detection + reports the pair and leaves the decision to the reader. + """ + padded = tmp_path / "padded.epub" + padded.write_bytes(EPUB.read_bytes() + b"\0" * 64) + + fixture = CalibreFixture(tmp_path / "source") + fixture.add_book( + 1, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": EPUB} + ) + fixture.add_book( + 2, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": padded} + ) + root = fixture.commit() + + source = await library_of(root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert len(result.created) == 2 + assert len(result.possible_duplicates) == 1 + assert result.possible_duplicates[0].candidates[0].book_id == result.created[0] + + +async def test_allow_duplicates_stores_the_same_bytes_again( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + for allow in (False, True): + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre( + source, test_library, allow_duplicates=allow + ) + finally: + await source.close() + + assert len(result.created) == 2 + + +async def test_duplicate_scope_off_imports_everything( + books_service: BookService, + test_library: m.Library, + calibre_root: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(settings, "duplicate_scope", "off") + + for _ in range(2): + source = await library_of(calibre_root) + try: + result = await books_service.create_many_from_calibre(source, test_library) + finally: + await source.close() + + assert len(result.created) == 2 + + +async def test_progress_is_reported_per_book( + books_service: BookService, test_library: m.Library, calibre_root: Path +) -> None: + """The import is long enough that its progress is the only thing worth watching.""" + seen = [] + + source = await library_of(calibre_root) + try: + await books_service.create_many_from_calibre( + source, test_library, on_progress=seen.append + ) + finally: + await source.close() + + assert [progress.processed for progress in seen] == [1, 2, 3] + assert all(progress.total == 3 for progress in seen) + assert [progress.outcome for progress in seen] == ["created", "created", "skipped"] + assert seen[0].title == "The Metamorphosis" diff --git a/docs/calibre-import.md b/docs/calibre-import.md new file mode 100644 index 0000000..7d47d0d --- /dev/null +++ b/docs/calibre-import.md @@ -0,0 +1,407 @@ +# Implementation brief: importing a Calibre library + +Written for an agent picking this up cold. Read the repo-root `AGENTS.md` and +`backend/AGENTS.md` first — this brief assumes both, particularly the **Filesystem behaviour** +and **Duplicate detection** sections. + +Every claim about Calibre's schema and on-disk layout below was checked against a real library at +`~/Documents/Calibre Library` (6 books, current Calibre). Where a fact came from Calibre's source +rather than that library, it says so. + +## Feasibility: high, and most of the machinery already exists + +`metadata.db` is plain SQLite with a schema that has been stable for a decade, and the files sit +beside it in a predictable tree. Chitai already has every piece an import needs: + +| Needed | Already in the tree | +| --- | --- | +| Ingest files that are already on disk | `BookService.create_many_from_existing_files` (`services/book.py:1280`) | +| Deduplicate authors/tags/publishers/series | `_populate_with_unique_relationships` (`services/book.py:1752`), via `as_unique_async` | +| Generic identifiers with a scheme map | `Identifier`, and `parse_identifier` (`services/metadata_extractor.py:58`) — whose map already covers `isbn`, `amazon`, `mobi-asin`, `google`, `goodreads`, `doi`, `calibre` | +| Decide where a book lives on disk | `BookPathGenerator`, `_reserve_book_path` (`services/book.py:1066`) | +| Not import the same book twice | `find_duplicate_files` (`:461`) and `find_duplicate_books` (`:535`) | +| Store a cover | `_save_cover_image` (`:1994`) | + +So this is **a reader, not new ingest machinery**: turn Calibre rows into the metadata dict +`BookService` already accepts, and hand it to a slightly generalised version of the consume-directory +path. The parsing is the easy half. + +The hard parts are elsewhere, and all three are addressed below: + +1. Calibre libraries hold formats Chitai cannot describe, let alone read — and one of them + **currently 500s the book detail endpoint** (see prerequisites). +2. A 5,000-book import is a long-running job, and the app has no job/progress concept. +3. Whether files are **copied** into the library or **referenced in place** — which is a + product decision with a large blast radius, because in-place means Chitai's write paths point + at somebody's Calibre library. + +## What a Calibre library actually is + +``` +Calibre Library/ +├── metadata.db the whole catalogue +├── metadata_db_prefs_backup.json ignore +├── .caltrash/ .calnotes/ ignore — deleted books still live in .caltrash +└── / + └── (<book id>)/ == books.path + ├── cover.jpg iff books.has_cover + ├── metadata.opf ignore; the db is authoritative + └── <data.name>.<format> one per row in `data` +``` + +The tables that matter, and nothing else: `books`, `authors` + `books_authors_link`, +`publishers` + `books_publishers_link`, `tags` + `books_tags_link`, `series` + +`books_series_link`, `languages` + `books_languages_link`, `comments`, `identifiers`, `data`, +`books_pages_link`, `last_read_positions`. + +Ten things that will produce wrong data if you do not know them: + +- **Never query the views.** `meta`, `tag_browser_*` and friends call SQLite functions Calibre + registers from Python at connection time. Verified: `SELECT * FROM meta` fails with + `no such function: sortconcat`. Query base tables only. + +- **`pubdate` has a sentinel, not a null.** An unknown publication date is stored as + `0101-01-01 00:00:00+00:00` (Calibre's `UNDEFINED_DATE`, year 101). It parses fine as a + `date`, so nothing will complain — two of the six books in the reference library carry it. Drop + any `pubdate` with year < 1000. The same sentinel appears in `timestamp`. + +- **`data.name` is lossy and is not the title.** It is the on-disk stem, truncated to Calibre's + filename limit and sanitised. Verified in the reference library: the book titled + `The Project Gutenberg eBook #33283: Calculus Made Easy, 2nd Edition` is stored as + `The Project Gutenberg eBook #33283_ Calcul - Silvanus Phillips Thompson.pdf`. So the file + extractors must not be consulted for metadata (see decision 2), and `books.path`/`data.name` + are for *locating* files only. + +- **`books.title` may contain characters Calibre strips from its own paths** — `:` became `_` + above, and titles legitimately contain `/` (`AC/DC`). `BookPathGenerator` interpolates the + title straight into a path and only collapses repeated slashes + (`services/filesystem_library.py`), so an unsanitised Calibre title can silently add a + directory level. Sanitise `/` and control characters out of `title` before path generation. + +- **`authors.name` escapes commas as `|`.** Calibre's `AuthorsTable` unserialises with + `name.replace('|', ',')` (from Calibre's `db/tables.py`; the reference library has no such + name, so this one is unverified locally). Do the same replacement, and pass `authors.name` — + **not** `authors.sort`, which is `Melville, Herman`. `format_author_name` would flip the sort + form correctly anyway, but there is no reason to hand it the worse input. + +- **`series_index` is a REAL.** `7.0` must become `"7"`, not `"7.0"` — `Book.series_position` is + a string, and `find_duplicate_books`'s series-position disqualifier compares it as one. + +- **`languages.lang_code` is ISO 639-2/B** (`eng`), while `EpubExtractor` stores raw + `DC:language` (`en`). Both will coexist in the column. `Book.language` is free text and the + edit form is a plain `<input>`, so nothing breaks; normalising to two letters is optional + polish, not part of this work. + +- **`comments.text` is HTML.** `Book.description` is rendered as plain text by + `CollapsibleText`, so `<p>` tags will show literally. Strip to text on import. + +- **`books_pages_link` is usually empty of real data.** It carries `needs_scan` and, in the + reference library, `pages = 0` for all six books. Only use it when `pages > 0`. + +- **`identifiers.type` is free text.** The reference library holds `isbn`, `amazon` and + `mobi-asin`, all of which `parse_identifier` already maps. Feed every identifier through it and + keep whatever survives; do not filter to a known list. + +## Decisions to settle before writing code + +1. **Copy files into the library. Do not move, do not reference in place** — for the first + version. Moving leaves `metadata.db` pointing at files that are gone, which quietly destroys a + library the user still uses. Referencing in place is genuinely desirable (nobody wants two + copies of 80 GB) but it points `book.path` at the Calibre tree, and `update_book` **moves + directories** while `delete_books` **deletes files** — so a metadata edit in Chitai would + rearrange somebody's Calibre library. `Library.read_only` exists but is enforced in exactly one + place (`services/library.py:54`, at creation). See phase 3. + +2. **Trust Calibre's metadata; do not run the extractors.** Calibre's catalogue is curated, its + filenames are truncated garbage, and running `Extractor.extract_metadata` over thousands of + files means opening every EPUB and rendering a cover page from every PDF. Take the cover from + `cover.jpg` directly. The one exception worth allowing: fill `pages` from the file when + Calibre has no useful value, behind a flag, off by default. + +3. ~~**The source is a server-side path, not an upload.**~~ **Reversed in review, and the reason + this brief was wrong is worth keeping.** The premise — "the library lives on the same host as + the backend in every realistic deployment" — is false for the common case: Calibre is a desktop + application, and its library is on the desktop. So the split is by *surface*, not by preference: + + - **Over HTTP: an uploaded zip only.** `…/imports/calibre/upload`. There is no endpoint taking a + server path; one was built and then removed deliberately. + - **On the server: a path only.** `litestar calibre-import <path>`, which is where a very large + library or a headless migration belongs — an upload has to carry the whole archive across + first. + + A CLI taking a path needs no justification. An *endpoint* taking one would have: it would let any + authenticated caller read any directory the backend can, and `TODO.md` records there is no + authorization tier at all. Not adding it is one less thing to gate later. + +4. **Import into an existing Chitai library**, chosen by the caller. Creating a library is + already one action, and the Calibre tree is rewritten by `BookPathGenerator` regardless. + +5. **Re-running an import must be safe, and file-level dedupe already makes it so.** The same + bytes are recognised by `(hash, size)` whatever their path, so a second run over the same + library skips everything. No import bookkeeping is needed for idempotency. + +## Prerequisite: a `.mobi` file breaks the book endpoint — **done** + +> Landed ahead of the import itself. `guess_content_type` in `services/utils.py` now names every +> format from its extension, the column keeps a null when nothing can name one, and the OPDS feed +> substitutes `application/octet-stream` at the one place a string is required. The Read control is +> driven by `isReadable` rather than by the file count. See the section below for why it mattered, +> and `backend/AGENTS.md` for the rule as it now stands. + + +`FileMetadataRead.content_type` is a required `str` (`schemas/book.py:33`), but every ingest path +fills it from `mimetypes.guess_type`, which returns `None` for `.mobi`, `.azw`, `.fb2`, `.lit` +and `.htmlz` (verified). `FileMetadata.content_type` is nullable in the model, so the row stores +fine and then fails response validation on the way out — a book whose only file is a MOBI would +be unreadable through the API. + +Nothing in the tree hits this today because the browser upload path is used with EPUBs and PDFs. +A Calibre library is full of MOBI and AZW3. Fix it first, either way round: + +- make the schema field `str | None`, and/or +- add a small extension→MIME table for the ebook formats `mimetypes` does not know + (`application/x-mobipocket-ebook`, `application/vnd.amazon.ebook`, `application/x-fictionbook+xml`). + +Do both, in fact: the table is the right answer for OPDS clients, which choose an acquisition link +by MIME type, and the nullable field is the safety net. + +**Related, but not a blocker:** Chitai reads EPUB and PDF only. `openBookInReader` +(`book/[bookId]/+page.svelte:66`) branches on `getFileType(...) === 'EPUB' | 'PDF'` and does +nothing for anything else, so an AZW3-only book gets a Read button that silently fails. Importing +those files is still right — they are downloadable and they are the user's — but the button +should be disabled for a book with no readable file. One `$derived` on the page, worth doing in +the same branch. + +## Design + +### 1. `services/calibre.py` — a pure reader, no Chitai types + +```python +@dataclass(frozen=True) +class CalibreFile: + path: Path # absolute, resolved against the library root + format: str # "EPUB", as stored + size: int # data.uncompressed_size, for a cheap sanity check + +@dataclass(frozen=True) +class CalibreBook: + calibre_id: int + uuid: str + title: str + authors: list[str] + ... # one field per row of the mapping table below + cover: Path | None + files: list[CalibreFile] + +class CalibreLibrary: + def __init__(self, root: Path) -> None: ... + async def open(self) -> None: ... # copy + connect, see below + async def books(self) -> AsyncIterator[CalibreBook]: ... + async def close(self) -> None: ... +``` + +Deliberately knows nothing about `Book`, `BookService` or the session — it is a file-format +reader, unit-testable against a fixture database with no Postgres and no app. + +Two implementation notes: + +- **Copy `metadata.db` to a temp file and read the copy.** Calibre may be running and writing; + opening the live file read-only either sees a torn state or needs the `-wal` sidecar. The + database is small (438 KB for six books, single-digit MB for thousands), so a copy costs + nothing and removes the whole problem. +- **`sqlite3` inside `asyncio.to_thread`, not a new dependency.** The connection is used for a + handful of queries. Do not add `aiosqlite` for this. + +Read the whole catalogue in **one query per table** and join in Python — six or so `SELECT`s and +a few dicts, versus a per-book N+1 across ten tables. At self-hosted scale the entire catalogue +minus descriptions fits in memory comfortably; if `comments.text` for 20k books is a concern, +fetch that one table per batch. + +### 2. The mapping + +| Calibre | Chitai | Notes | +| --- | --- | --- | +| `books.title` | `title` | Sanitise `/` and control chars for path generation. `Extractor.format_book_title` may still be worth applying to split a subtitle at the second colon — but **do not** run `split_edition`, Calibre's title is the curated one. | +| `authors.name` via `books_authors_link` | `authors` | `\|` → `,`. Order by `books_authors_link.id`; `Book.author_links` is an `ordering_list`, so insertion order is the displayed order. | +| `comments.text` | `description` | Strip HTML to text. | +| `books.pubdate` | `published_date` | Drop the year-101 sentinel. | +| `series.name`, `books.series_index` | `series`, `series_position` | `7.0` → `"7"`. | +| `tags.name` | `tags` | | +| `publishers.name` | `publisher` | `books_publishers_link` is unique per book. | +| `languages.lang_code` (lowest `item_order`) | `language` | Chitai holds one. | +| `identifiers.type` / `.val` | `identifiers` | Through `parse_identifier`; keep what survives. | +| `books.uuid` | `identifiers["calibre-uuid"]` | The one durable link back to the source row. `normalize_identifier` returns a key for it (it is not in `_PER_BUILD_NAMES`), which is *desirable*: a book re-imported from the same Calibre library matches on it exactly. | +| `books_pages_link.pages` | `pages` | Only when `> 0`. | +| `cover.jpg` when `has_cover` | `cover_image` | | +| `data` rows | `files` | | +| `books.timestamp` | — | `Book.created_at` is audit-managed; do not fight it. | +| `ratings`, `annotations`, `custom_columns` | — | No home in the model. Out of scope. | +| `last_read_positions` | `BookProgress` | Phase 2. | + +### 3. The ingest + +> **As built, this went on `BookService` as `create_many_from_calibre`, not into a separate +> `services/calibre_import.py`.** The orchestration needs `_reserve_book_path`, +> `_save_cover_image`, `_screen_for_duplicates` and `_record_possible_duplicates`, and reaching +> into four privates from another module is worse than one more method in the file where the other +> two ingest paths already live. `_record_possible_duplicates` was changed to take the list it +> appends to rather than an `ImportResult`, so every ingest path can share it whatever its own +> result type is. Two other deviations: `CalibreLibrary.books()` returns a list rather than an +> async iterator, because the caller needs the total up front anyway; and the reader reports +> identifiers exactly as Calibre keyed them, with the fold onto Chitai's schemes done by the +> importer, which keeps the reader free of Chitai imports. + +Per book, in this order — it mirrors `create_many_from_existing_files`, which is the closest +existing shape: + +1. `fingerprint_file` each source file (`services/utils.py:164`). +2. `find_duplicate_files` against the target library. All files known → skip the book entirely, + recording it. Some known → import the rest. +3. Build the metadata dict from the `CalibreBook`. +4. `_reserve_book_path(path_gen.generate_path(data))`. +5. **Copy** each file to `parent / _unused_path(...)`, building `FileMetadata` from the + fingerprint already computed. `services/utils.py` has `move_file` but no copy — add + `copy_file` beside it, streaming through `aiofiles` in `CHUNK_SIZE` blocks like + `_save_book_files` does, not `shutil.copy` (a 40 MB blocking read inside the event loop). +6. Cover: open `cover.jpg` with PIL and hand the `Image` to `_save_cover_image`, which already + accepts one and converts to WebP. +7. `super().create(data)` through `BookService`, then `find_duplicate_books` and record + candidates — same as `_record_possible_duplicates` (`services/book.py:1253`). +8. **Commit per book.** A 5,000-book import inside one transaction is one failure away from + nothing, and per-book commits are what lets the library page show books arriving — which is + the behaviour commit `85367da` deliberately built. + +Report an `ImportResult`-shaped outcome; reuse `ImportResult` itself if it fits, extending it with +a `failures: list[tuple[int, str]]` keyed by Calibre id. **One book must never fail the run** — +a missing file, an unreadable cover or a `NOT NULL` violation gets recorded and skipped. + +### 4. Progress, and where the import runs + +> **As built, the HTTP surface is an uploaded archive and nothing else** — see decision 3. +> +> `POST …/imports/calibre/upload` takes a zipped library. The job owns the temp directory it is +> unpacked into and deletes it when it ends. Extraction refuses zip slip, an archive too big for the +> disk, and one with no `metadata.db` within three levels — all answered 400 before a job exists. +> +> This forced a fix to the SvelteKit proxy, which buffered request bodies with `arrayBuffer()`: +> survivable for one book, not for a multi-gigabyte archive. POST and PATCH now stream +> `request.body` through with `duplex: 'half'`. +> +> **A preview endpoint was built and then removed with the path route.** It read a server-side +> catalogue and reported its size before writing anything, which is only useful when the caller +> named a directory. An upload has already been carried across by the time anything can be read, so +> unpacking it *is* the validation step — an archive that is not a Calibre library is refused there. + +The import outlives its request, so the handler starts it and returns a handle: + +- `POST /libraries/{library_id:int}/imports/calibre` — body `{path, copy_files: true}`, returns + `{job_id, total}`. 202. +- `GET /libraries/imports/{job_id}` — `{state, total, processed, created, skipped, failed, + current_title, errors}`. +- `DELETE /libraries/imports/{job_id}` — cancel; the task checks a flag between books. + +Keep the registry **in memory**, a `dict[str, ImportJob]` on a module-level singleton, with the +task created by `asyncio.create_task`. This matches what the app already does — the consume +watcher is an in-process singleton started from a lifespan hook — and it is roughly thirty lines +against a model, a migration and a service for the alternative. + +State that limitation explicitly in the docstring: **it assumes one worker process.** The +production `CMD` is `litestar run`, which is single-process, so this holds today; `TODO.md` +already records that the production image should move to uvicorn with a worker count, and doing +that would mean a poll landing on a worker that has never heard of the job. The consume watcher +has the same problem, so this is not a new constraint — but the next person to add workers needs +to find it written down. If import *history* is ever wanted, that is when an `import_jobs` table +earns its migration. + +**Also add a CLI entry point.** A 200 GB library imported through a browser tab that must stay +open is a bad experience, and `pyproject.toml` already declares a `chitai` script. A Litestar CLI +command (`litestar --app-dir src/chitai/ calibre-import <path> --library <slug>`) is ~20 lines +over the same service and is the right tool for the initial migration, which is the case this +whole feature exists for. The endpoint is for people who would rather click. + +### 5. Frontend + +Model it on the duplicates screen, which is the closest precedent in shape and placement: + +- Route `(root)/settings/libraries/[libraryId]/import` — beside + `settings/libraries/[libraryId]/duplicates`, reached from the library settings page. +- `getCalibreImport` (a `query`) and `cancelCalibreImport` (a `command`) in + `src/lib/api/calibre-import.remote.ts`, re-exported from `src/lib/api/index.ts`. **Starting an + import is not a remote function**: the archive goes to the backend through the proxy so the + browser streams straight through, where a remote function would put the whole thing through the + SvelteKit process first. The screen uses `XMLHttpRequest` for it, which is the only way to get + upload progress. +- A file input, an upload progress bar, then a progress bar polling `getCalibreImport` every second + or two, a running count, and the failures listed at the end with their Calibre ids. +- Finish with a link to the library's duplicates screen. An import into a non-empty library is + the single most likely way to produce duplicate books, and that screen already handles them. + +Do **not** route this through the upload tray. The tray reports on a client-driven queue it owns +(`upload-queue.svelte.ts`); this is server-side work whose state survives a page reload, and +conflating the two would mean teaching the tray to poll. + +Regenerate `src/lib/schema/openapi/schema.d.ts` against a backend running **your** branch — +a stale server silently writes a stale file. + +## Testing + +The fixture is the interesting part. Build a Calibre library in a `tmp_path` fixture rather than +committing a binary `metadata.db`: a helper that executes the subset of Calibre's `CREATE TABLE` +statements (they are in this document's shape, and in any real library's `sqlite_master`), inserts +a handful of books, and lays out `<Author>/<Title> (id)/` directories containing the existing +EPUB and PDF fixtures from `backend/tests/data_files/` plus a copy of `cover.jpg`. Generated +beats committed here because the tests need to assert on *odd* rows — the pubdate sentinel, a +`|` in an author name, a title with a colon — and those are clearer written in Python than hidden +in a blob. + +- **Unit** (`tests/unit/test_calibre.py`) — the reader alone: field mapping; the year-101 pubdate + dropped; `series_index` 7.0 → `"7"`; `|` unescaped in an author name; HTML stripped from + `comments`; `pages = 0` ignored; identifiers passed through `parse_identifier`; a `data` row + whose file is missing from disk reported rather than raised; `.caltrash` never walked. +- **Service** (`tests/unit/test_services/test_calibre_import.py`) — a book with two formats lands + as one record with two files; the source files still exist afterwards; a second run over the + same library creates nothing; a library with one broken book imports the rest; authors and tags + shared between two books produce one `Author` / `Tag` row each; `possible_duplicates` reported + when the target library already holds the same book. +- **Integration** (`tests/integration/test_calibre_import.py`) — `POST` returns 202 with a job id, + polling reaches a terminal state, and the books are then listable through `GET /books`. A + `.mobi`-only book must come back from `GET /books/{id}` without a 500 — that is the + prerequisite's regression test. + +`pytest` needs Docker (`pytest-databases`). + +## Verification + +```bash +nix-shell # postgres + migrations applied +cd backend +pytest tests/ # take your own baseline first +uv run litestar --app-dir src/chitai/ run --port 8001 # for the OpenAPI regeneration +cd ../frontend && pnpm check # baseline: 30 errors, 1 warning, 8 files +``` + +Neither `pnpm check` nor `pnpm lint` is clean on this repo — baseline before assuming an error is +yours. + +End to end, against a real library (`~/Documents/Calibre Library` will do): import into an empty +Chitai library and confirm the six books arrive with their covers, authors, tags, series +positions and identifiers intact; that the AZW3 book is listed and downloadable; that the source +library is byte-for-byte untouched (`diff -r` a copy taken beforehand); and that re-running the +import creates nothing and reports six skipped books. Then import the same library into a +library that already holds one of those books by upload, and confirm it lands on the duplicates +screen rather than as a second copy. + +## Phasing + +| Phase | Scope | +| --- | --- | +| **0** ✅ | The `content_type` prerequisite, plus the Read button. **Done** — see below. | +| **1** ✅ | `services/calibre.py`, the ingest, the CLI command, copy-only. **Done** — a headless one-time migration works today. | +| **2** ◐ | The upload endpoint, the job registry and the settings screen — **done**. The HTTP surface is an uploaded zip only; see decision 3, which this reversed. `last_read_positions` → `BookProgress` is **not** done: it needs a Calibre-user → Chitai-user mapping, and the answer differs between the endpoint (which has a `current_user`) and the CLI (which has none). That decision is the next thing to make. | +| **3** | Reference-in-place import. Its real content is **enforcing `Library.read_only`** across `update_book`, `delete_books`, `add_files` and `remove_files` — which is a feature of its own and should not be smuggled in under an import. | + +## Out of scope + +Annotations and highlights (no model to put them in), custom columns, ratings, virtual libraries +and saved searches → bookshelves, format conversion, writing anything back to Calibre, and any +form of continuing two-way sync. This is a one-way migration. diff --git a/frontend/AGENTS.md b/frontend/AGENTS.md index 14a2897..60f01ab 100644 --- a/frontend/AGENTS.md +++ b/frontend/AGENTS.md @@ -167,8 +167,11 @@ Observed in the current tree — don't mistake these for intentional patterns to `svelte/no-navigation-without-resolve` on plain `href`s, plus ~59 files Prettier would rewrite (mostly vendored shadcn components). Check the files you touched, not the whole tree. - `src/routes/api/[...path]/+server.ts` — all four handlers are annotated `RequestHandler` while the - import of that type is commented out at line 4. It also buffers whole responses with - `arrayBuffer()` and forwards no `Range` header, so book downloads are not streamed. + import of that type is commented out at line 4. It also buffers whole **responses** with + `arrayBuffer()` and forwards no `Range` header, so book downloads are not streamed. **Requests** + are streamed — POST and PATCH pass `request.body` through with `duplex: 'half'` (see `bodyOf`), + because a zipped Calibre library upload cannot be held in this process. The response side is + still buffered; see `TODO.md`. - `src/app.d.ts` — `App.Locals["user"]` is typed from `lucide-svelte`'s `User` _icon_ component rather than the `User` interface in `$lib/server/auth`. - No CSP, which foliate's README asks for because EPUBs can carry scripts. See `TODO.md` for why it diff --git a/frontend/src/lib/api/calibre-import.remote.ts b/frontend/src/lib/api/calibre-import.remote.ts new file mode 100644 index 0000000..fcf2c7f --- /dev/null +++ b/frontend/src/lib/api/calibre-import.remote.ts @@ -0,0 +1,51 @@ +import { command, getRequestEvent, query } from '$app/server'; +import { error } from '@sveltejs/kit'; + +import { stringCoerce } from '$lib/schema/common'; +import type { CalibreImport } from '$lib/schema/library'; + +/** The backend's own message for a failed response, rather than its JSON envelope. */ +function detailOf(body: string): string { + try { + const parsed = JSON.parse(body); + return typeof parsed?.detail === 'string' ? parsed.detail : body; + } catch { + return body; + } +} + +/** + * Where an import has got to. + * + * A `query` rather than a `command` so it can be refreshed, but it is polled on a timer + * rather than cached — the answer changes on its own. + * + * Starting an import is deliberately **not** here: the archive goes straight to the + * backend through the proxy, so it never passes through this process. See the import + * screen's `upload`. + */ +export const getCalibreImport = query(stringCoerce, async (jobId): Promise<CalibreImport> => { + const { locals } = getRequestEvent(); + + const response = await locals.api.get(`/libraries/imports/${jobId}`); + + if (!response.ok) error(response.status, detailOf(await response.text())); + + return await response.json(); +}); + +/** + * Ask an import to stop after the book it is on. + * + * Not an abort: a book abandoned mid-copy would leave files on disk with no row + * describing them. Whatever it has imported stays imported. + */ +export const cancelCalibreImport = command(stringCoerce, async (jobId): Promise<CalibreImport> => { + const { locals } = getRequestEvent(); + + const response = await locals.api.delete(`/libraries/imports/${jobId}`); + + if (!response.ok) error(response.status, detailOf(await response.text())); + + return await response.json(); +}); diff --git a/frontend/src/lib/api/index.ts b/frontend/src/lib/api/index.ts index 7b9d134..4946561 100644 --- a/frontend/src/lib/api/index.ts +++ b/frontend/src/lib/api/index.ts @@ -2,6 +2,7 @@ export * from './auth.remote'; export * from './author.remote'; export * from './book.remote'; export * from './bookshelf.remote'; +export * from './calibre-import.remote'; export * from './library.remote'; export * from './publisher.remote'; export * from './tag.remote'; diff --git a/frontend/src/lib/schema/library.ts b/frontend/src/lib/schema/library.ts index 72b6463..d5e197f 100644 --- a/frontend/src/lib/schema/library.ts +++ b/frontend/src/lib/schema/library.ts @@ -19,3 +19,6 @@ export const libraryCreateSchema = z.object({ export type LibraryQuerySchema = typeof libraryQuerySchema; export type LibraryCreateSchema = typeof libraryCreateSchema; + +export type CalibreImport = components['schemas']['CalibreImportRead']; +export type ImportFailure = components['schemas']['ImportFailureRead']; diff --git a/frontend/src/lib/schema/openapi/schema.d.ts b/frontend/src/lib/schema/openapi/schema.d.ts index b0d809c..6bae1d1 100644 --- a/frontend/src/lib/schema/openapi/schema.d.ts +++ b/frontend/src/lib/schema/openapi/schema.d.ts @@ -230,6 +230,24 @@ export interface paths { patch?: never; trace?: never; }; + "/libraries/imports/{job_id}": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** GetImport */ + get: operations["LibrariesImportsJobIdGetImport"]; + put?: never; + post?: never; + /** CancelImport */ + delete: operations["LibrariesImportsJobIdCancelImport"]; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/libraries": { parameters: { query?: never; @@ -248,6 +266,23 @@ export interface paths { patch?: never; trace?: never; }; + "/libraries/{library_id}/imports/calibre/upload": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** UploadCalibreImport */ + post: operations["LibrariesLibraryIdImportsCalibreUploadUploadCalibreImport"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/access/me": { parameters: { query?: never; @@ -791,6 +826,30 @@ export interface components { skipped: components["schemas"]["DuplicateFileRead"][]; possible_duplicates?: components["schemas"]["PossibleDuplicateRead"][]; }; + /** CalibreArchiveUpload */ + CalibreArchiveUpload: { + /** Format: binary */ + archive: string; + /** @default false */ + allow_duplicates: boolean; + }; + /** CalibreImportRead */ + CalibreImportRead: { + id: string; + library_id: number; + source: string; + state: string; + total: number; + processed: number; + created: number; + skipped: number; + failed: number; + current_title?: string | null; + failures?: components["schemas"]["ImportFailureRead"][]; + /** @default 0 */ + possible_duplicates: number; + error?: string | null; + }; /** DuplicateBookGroupRead */ DuplicateBookGroupRead: { books: components["schemas"]["DuplicateBookRead"][]; @@ -831,9 +890,15 @@ export interface components { path: string; hash: string; size: number; - content_type: string; + content_type?: string | null; readonly filename: string; }; + /** ImportFailureRead */ + ImportFailureRead: { + calibre_id: number; + title: string; + reason: string; + }; /** KosyncDeviceCreate */ KosyncDeviceCreate: { name: string; @@ -1686,6 +1751,80 @@ export interface operations { }; }; }; + LibrariesImportsJobIdGetImport: { + parameters: { + query?: never; + header?: never; + path: { + job_id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Request fulfilled, document follows */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["CalibreImportRead"]; + }; + }; + /** @description Bad request syntax or unsupported method */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": { + status_code: number; + detail: string; + extra?: null | { + [key: string]: unknown; + } | unknown[]; + }; + }; + }; + }; + }; + LibrariesImportsJobIdCancelImport: { + parameters: { + query?: never; + header?: never; + path: { + job_id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Request fulfilled, document follows */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["CalibreImportRead"]; + }; + }; + /** @description Bad request syntax or unsupported method */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": { + status_code: number; + detail: string; + extra?: null | { + [key: string]: unknown; + } | unknown[]; + }; + }; + }; + }; + }; LibrariesListLibraries: { parameters: { query?: { @@ -1772,6 +1911,47 @@ export interface operations { }; }; }; + LibrariesLibraryIdImportsCalibreUploadUploadCalibreImport: { + parameters: { + query?: never; + header?: never; + path: { + library_id: number; + }; + cookie?: never; + }; + requestBody: { + content: { + "multipart/form-data": components["schemas"]["CalibreArchiveUpload"]; + }; + }; + responses: { + /** @description Request accepted, processing continues off-line */ + 202: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["CalibreImportRead"]; + }; + }; + /** @description Bad request syntax or unsupported method */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": { + status_code: number; + detail: string; + extra?: null | { + [key: string]: unknown; + } | unknown[]; + }; + }; + }; + }; + }; AccessMeGetUserInfo: { parameters: { query?: never; diff --git a/frontend/src/lib/utils.ts b/frontend/src/lib/utils.ts index e200420..9d4e8f8 100644 --- a/frontend/src/lib/utils.ts +++ b/frontend/src/lib/utils.ts @@ -81,7 +81,10 @@ function isbnDigits(value: string) { } function identifierKey(name: string) { - return name.trim().toLowerCase().replace(/[\s_]+/g, '-'); + return name + .trim() + .toLowerCase() + .replace(/[\s_]+/g, '-'); } export function describeIdentifier(name: string, value: string) { @@ -111,4 +114,17 @@ export function getFileType(filename: string) { return extension.toUpperCase(); } +/** + * The formats there is a reader for. Everything else is download-only. + * + * A library imported from elsewhere carries MOBI, AZW3 and CBZ files, which are + * legitimate to store and to download but have nowhere to open — the reader routes + * are `read/epub` and `read/pdf` and there is no third one. + */ +const READABLE_FILE_TYPES = ['EPUB', 'PDF']; + +export function isReadable(filename: string) { + return READABLE_FILE_TYPES.includes(getFileType(filename)); +} + export const pluck = (array: [], key: string) => array.map((obj) => obj?.[key]); diff --git a/frontend/src/routes/(root)/(library)/book/[bookId]/+page.svelte b/frontend/src/routes/(root)/(library)/book/[bookId]/+page.svelte index 2c2e1ed..b5f6bb4 100644 --- a/frontend/src/routes/(root)/(library)/book/[bookId]/+page.svelte +++ b/frontend/src/routes/(root)/(library)/book/[bookId]/+page.svelte @@ -18,7 +18,13 @@ PlusIcon, Trash2 } from '@lucide/svelte'; - import { describeIdentifier, formatFileSize, getFileType, sortIdentifiers } from '$lib/utils.js'; + import { + describeIdentifier, + formatFileSize, + getFileType, + isReadable, + sortIdentifiers + } from '$lib/utils.js'; import { getBookOperationsState } from '$lib/state/bookOperations.svelte.js'; import { getBookshelfState } from '$lib/state/bookshelf.svelte.js'; import { getLibraryState } from '$lib/state/library.svelte.js'; @@ -52,6 +58,11 @@ book.progress?.percentage ? Math.round(book.progress.percentage * 100) : 0 ); + // Only the files there is a reader for. A book can be stored in a format Chitai + // cannot open — a Calibre library is full of MOBI and AZW3 — and offering to read + // one opened a window that did nothing at all. + const readableFiles = $derived(book.files.filter((file) => isReadable(file.filename))); + // The primary action states what it will actually do. const readLabel = $derived( book.progress?.completed @@ -86,6 +97,10 @@ 'inline-flex items-center gap-2 rounded-lg bg-primary-foreground px-4 py-2 text-sm font-semibold text-primary tabular-nums transition-colors hover:bg-primary-foreground/90'; const bandGhost = 'inline-flex items-center gap-2 rounded-lg border border-primary-foreground/35 px-4 py-2 text-sm font-medium transition-colors hover:bg-primary-foreground/10'; + + // With nothing to read, downloading is the only thing left to do — so it takes the + // primary slot rather than leaving the band with no filled button on it. + const bandDownload = $derived(readableFiles.length ? bandGhost : bandPrimary); </script> <div class="flex h-full flex-col overflow-y-auto"> @@ -132,9 +147,13 @@ </p> {/if} - <!-- Primary actions live on the band, not in a side rail --> + <!-- + Primary actions live on the band, not in a side rail. Read and Download + are counted separately: everything can be downloaded, only EPUB and PDF + can be opened, so a book can have two files and one way to read it. + --> <div class="mt-5 flex flex-wrap items-center gap-2"> - {#if book.files.length > 1} + {#if readableFiles.length > 1} <DropdownMenu.Root> <DropdownMenu.Trigger class={bandPrimary}> <BookOpenText class="size-4" /> @@ -144,7 +163,7 @@ <DropdownMenu.Group> <DropdownMenu.GroupHeading>File formats:</DropdownMenu.GroupHeading> <DropdownMenu.Separator /> - {#each book.files as file (file.id)} + {#each readableFiles as file (file.id)} <DropdownMenu.Item onclick={() => openBookInReader(file)}> {getFileType(file.filename)} </DropdownMenu.Item> @@ -152,9 +171,20 @@ </DropdownMenu.Group> </DropdownMenu.Content> </DropdownMenu.Root> + {:else if readableFiles.length === 1} + <button + type="button" + class={bandPrimary} + onclick={() => openBookInReader(readableFiles[0])} + > + <BookOpenText class="size-4" /> + {readLabel} + </button> + {/if} + {#if book.files.length > 1} <DropdownMenu.Root> - <DropdownMenu.Trigger class={bandGhost}> + <DropdownMenu.Trigger class={bandDownload}> <Download class="size-4" /> Download </DropdownMenu.Trigger> @@ -181,16 +211,7 @@ {:else if book.files.length === 1} <button type="button" - class={bandPrimary} - onclick={() => openBookInReader(book.files[0])} - > - <BookOpenText class="size-4" /> - {readLabel} - </button> - - <button - type="button" - class={bandGhost} + class={bandDownload} onclick={() => bookOps.downloadBookFile(book.id, book.files[0].id, book.files[0].filename)} > @@ -491,7 +512,7 @@ > <Table.Cell class="text-right"> <div class="flex justify-end gap-1"> - {#if getFileType(file.filename) === 'EPUB' || getFileType(file.filename) === 'PDF'} + {#if isReadable(file.filename)} <Tooltip.Provider> <Tooltip.Root ignoreNonKeyboardFocus> <Tooltip.Trigger diff --git a/frontend/src/routes/(root)/settings/libraries/[libraryId]/+layout.svelte b/frontend/src/routes/(root)/settings/libraries/[libraryId]/+layout.svelte index de81529..511d4a5 100644 --- a/frontend/src/routes/(root)/settings/libraries/[libraryId]/+layout.svelte +++ b/frontend/src/routes/(root)/settings/libraries/[libraryId]/+layout.svelte @@ -13,14 +13,14 @@ libraryState.libraries.find((lib) => String(lib.id) === page.params.libraryId) ); - // Duplicates is the only section today. The strip exists so General and a - // danger zone have an obvious place to land; neither is built yet, and an - // empty tab is worse than no tab. + // The strip exists so General and a danger zone have an obvious place to land; + // neither is built yet, and an empty tab is worse than no tab. // // Active state is matched on route id, not pathname, for the reason spelled // out in settings/+layout.svelte. const sections = [ - { title: 'Duplicates', routeId: '/(root)/settings/libraries/[libraryId]/duplicates' } + { title: 'Duplicates', routeId: '/(root)/settings/libraries/[libraryId]/duplicates' }, + { title: 'Import', routeId: '/(root)/settings/libraries/[libraryId]/import' } ] as const; </script> diff --git a/frontend/src/routes/(root)/settings/libraries/[libraryId]/import/+page.svelte b/frontend/src/routes/(root)/settings/libraries/[libraryId]/import/+page.svelte new file mode 100644 index 0000000..37c8940 --- /dev/null +++ b/frontend/src/routes/(root)/settings/libraries/[libraryId]/import/+page.svelte @@ -0,0 +1,367 @@ +<script lang="ts"> + import { page } from '$app/state'; + import { resolve } from '$app/paths'; + import { onDestroy } from 'svelte'; + import { toast } from 'svelte-sonner'; + import { BookOpen, Loader2, TriangleAlert, Upload } from '@lucide/svelte'; + + import * as Card from '$lib/components/ui/card/index.js'; + import * as Field from '$lib/components/ui/field/index.js'; + import * as Table from '$lib/components/ui/table/index.js'; + import { Button } from '$lib/components/ui/button/index.js'; + import { Input } from '$lib/components/ui/input/index.js'; + import { Checkbox } from '$lib/components/ui/checkbox/index.js'; + import { Progress } from '$lib/components/ui/progress/index.js'; + import { cancelCalibreImport, getCalibreImport } from '$lib/api/calibre-import.remote'; + import { formatFileSize } from '$lib/utils'; + import type { CalibreImport } from '$lib/schema/library'; + + const libraryId = $derived(page.params.libraryId!); + + let archive = $state<File | null>(null); + let allowDuplicates = $state(false); + + let job = $state<CalibreImport | null>(null); + + let busy = $state(false); + let problem = $state<string | null>(null); + + /** How much of the archive has reached the server, 0–1, while it is going up. */ + let uploaded = $state<number | null>(null); + + /** How often a running import is asked where it has got to. */ + const POLL_MS = 1000; + let poll: ReturnType<typeof setInterval> | undefined; + + const running = $derived(job?.state === 'running'); + + const percent = $derived( + job && job.total > 0 ? Math.round((job.processed / job.total) * 100) : 0 + ); + + function stopPolling() { + clearInterval(poll); + poll = undefined; + } + + onDestroy(stopPolling); + + function pick(event: Event) { + const input = event.currentTarget as HTMLInputElement; + archive = input.files?.[0] ?? null; + problem = null; + } + + /** + * The backend's own message for a failed proxied response. + * + * The proxy wraps the upstream body in SvelteKit's error envelope, so the useful + * `detail` is one or two layers down. + */ + function detailOf(body: string): string { + try { + const parsed = JSON.parse(body); + + if (typeof parsed?.detail === 'string') return parsed.detail; + if (typeof parsed?.message === 'string') return detailOf(parsed.message); + } catch { + // Not JSON — the raw text is the best there is. + } + + return body; + } + + /** + * Follow a running import until it lands. + * + * Polled rather than pushed: the job lives on the server, so this survives a reload + * and does not depend on the tab that started it staying open. + */ + function follow(jobId: string) { + stopPolling(); + + poll = setInterval(async () => { + try { + job = await getCalibreImport(jobId); + } catch (error) { + problem = error instanceof Error ? error.message : 'Lost track of the import'; + stopPolling(); + return; + } + + if (job.state !== 'running') { + stopPolling(); + announce(job); + } + }, POLL_MS); + } + + function announce(finished: CalibreImport) { + const created = `${finished.created} book${finished.created === 1 ? '' : 's'} imported`; + + if (finished.state === 'failed') toast.error(finished.error ?? 'The import failed'); + else if (finished.state === 'cancelled') toast.info(`Import stopped — ${created}`); + else if (finished.failed > 0) toast.warning(`${created}, ${finished.failed} failed`); + else toast.success(created); + } + + /** + * Send the archive and start the import it becomes. + * + * `XMLHttpRequest` rather than `fetch` for the one thing fetch cannot do: report how + * much of the body has gone up. On a large archive that is the only progress there is + * for minutes at a time. + * + * It goes through the proxy so the browser streams straight to the backend — a remote + * function would put the whole archive through the SvelteKit process first. + */ + function send(file: File): Promise<CalibreImport> { + return new Promise((resolve, reject) => { + const request = new XMLHttpRequest(); + + request.open('POST', `/api/libraries/${libraryId}/imports/calibre/upload`); + + request.upload.addEventListener('progress', (event) => { + if (event.lengthComputable) uploaded = event.loaded / event.total; + }); + + request.addEventListener('load', () => { + if (request.status >= 200 && request.status < 300) { + resolve(JSON.parse(request.responseText)); + } else { + reject(new Error(detailOf(request.responseText))); + } + }); + + request.addEventListener('error', () => reject(new Error('The upload failed'))); + request.addEventListener('abort', () => reject(new Error('The upload was stopped'))); + + const body = new FormData(); + body.append('archive', file); + body.append('allow_duplicates', String(allowDuplicates)); + + request.send(body); + }); + } + + async function start() { + if (!archive) return; + + busy = true; + problem = null; + uploaded = 0; + + try { + job = await send(archive); + follow(job.id); + } catch (error) { + problem = error instanceof Error ? error.message : 'Could not upload the archive'; + } finally { + busy = false; + uploaded = null; + } + } + + async function stop() { + if (!job) return; + + try { + job = await cancelCalibreImport(job.id); + } catch (error) { + problem = error instanceof Error ? error.message : 'Could not stop the import'; + } + } + + function reset() { + stopPolling(); + job = null; + problem = null; + archive = null; + uploaded = null; + } +</script> + +<div class="flex flex-col gap-4"> + <Card.Root> + <Card.Header> + <Card.Title>Import from Calibre</Card.Title> + <Card.Description> + Zip your Calibre library folder and upload it here. Nothing is taken from the original — the + books are copied in, and uploading the same library again only picks up what is new. + </Card.Description> + </Card.Header> + + <Card.Content class="flex flex-col gap-4"> + <Field.Field> + <Field.Label for="archive">Zipped Calibre library</Field.Label> + <!-- + The native file button inherits the input's own text styling, which leaves + "Browse…" looking like the first half of the sentence "Browse… No file + selected." The `file:` variants target ::file-selector-button, so it can be + made to read as a button without replacing the input with a custom one. + Styled here rather than in `ui/input`, which is generated. + --> + <Input + id="archive" + type="file" + accept=".zip,application/zip" + onchange={pick} + disabled={running || busy} + class="py-1.5 file:mr-3 file:cursor-pointer file:rounded-sm file:border file:border-input file:bg-secondary file:px-2 file:py-0.5 file:text-secondary-foreground file:hover:bg-secondary/80" + /> + <Field.Description> + The zip has to contain <code class="font-mono text-xs">metadata.db</code> — zip the whole Calibre + folder rather than just the books. + </Field.Description> + </Field.Field> + + {#if archive} + <p class="text-sm text-muted-foreground"> + {archive.name} · {formatFileSize(archive.size)} + </p> + {/if} + + <div class="flex items-center gap-2"> + <Checkbox id="allow-duplicates" bind:checked={allowDuplicates} disabled={running || busy} /> + <label for="allow-duplicates" class="text-sm"> + Import books this library already holds + </label> + </div> + + {#if uploaded !== null} + <div class="flex flex-col gap-1"> + <Progress value={Math.round(uploaded * 100)} /> + <p class="text-xs text-muted-foreground"> + Uploading · {Math.round(uploaded * 100)}% + </p> + </div> + {/if} + + {#if problem} + <p class="flex items-center gap-2 text-sm text-destructive"> + <TriangleAlert class="size-4 shrink-0" /> + {problem} + </p> + {/if} + </Card.Content> + + <Card.Footer class="gap-2"> + {#if running} + <Button variant="outline" onclick={stop}>Stop after this book</Button> + {:else if job} + <Button onclick={reset}>Import another</Button> + {:else} + <!-- + No preview step: the catalogue cannot be read until the archive is on the + server, and by then it has been carried across anyway. Unpacking it is what + refuses an archive that is not a Calibre library. + --> + <Button onclick={start} disabled={busy || !archive}> + {#if busy} + <Loader2 class="size-4 animate-spin" /> + {:else} + <Upload class="size-4" /> + {/if} + Upload and import + </Button> + {/if} + </Card.Footer> + </Card.Root> + + {#if job} + <Card.Root> + <Card.Header> + <Card.Title class="flex items-center gap-2"> + {#if running}<Loader2 class="size-4 animate-spin" />{/if} + {running ? 'Importing' : job.state === 'failed' ? 'Import failed' : 'Import finished'} + </Card.Title> + <Card.Description> + {#if running && job.current_title} + {job.processed} of {job.total} · {job.current_title} + {:else if job.state === 'cancelled'} + Stopped after {job.processed} of {job.total}. Everything imported is complete. + {:else if job.state === 'failed'} + {job.error ?? 'The import stopped before it finished.'} + {:else} + {job.processed} of {job.total} books considered. + {/if} + </Card.Description> + </Card.Header> + + <Card.Content class="flex flex-col gap-4"> + <Progress value={percent} /> + + <div class="grid grid-cols-3 gap-3 text-sm"> + <div> + <p class="font-mono text-lg tabular-nums">{job.created}</p> + <p class="text-muted-foreground">imported</p> + </div> + <div> + <p class="font-mono text-lg tabular-nums">{job.skipped}</p> + <p class="text-muted-foreground">already here</p> + </div> + <div> + <p class="font-mono text-lg tabular-nums">{job.failed}</p> + <p class="text-muted-foreground">failed</p> + </div> + </div> + + {#if job.created > 0} + <a + href={resolve('/(root)/(library)/library/[libraryId]/view', { libraryId })} + class="inline-flex w-fit items-center gap-2 text-sm hover:underline" + > + <BookOpen class="size-4" /> + See the books + </a> + {/if} + + <!-- + An import into a library that already holds books is the likeliest way to + end up with two records for one book, and that screen already handles them. + --> + {#if job.possible_duplicates > 0} + <div class="rounded-lg border p-3 text-sm"> + <p> + {job.possible_duplicates} imported book{job.possible_duplicates === 1 ? '' : 's'} + look{job.possible_duplicates === 1 ? 's' : ''} like something this library already had. + They were imported all the same — a metadata match is a guess. + </p> + <a + href={resolve('/(root)/settings/libraries/[libraryId]/duplicates', { libraryId })} + class="mt-2 inline-block font-medium hover:underline" + > + Review duplicates + </a> + </div> + {/if} + + {#if job.failures?.length} + <div class="flex flex-col gap-2"> + <p class="text-sm font-medium">Books that could not be imported</p> + <div class="overflow-x-auto rounded-lg border"> + <Table.Root> + <Table.Header> + <Table.Row> + <Table.Head class="w-[80px]">Calibre</Table.Head> + <Table.Head>Title</Table.Head> + <Table.Head>Reason</Table.Head> + </Table.Row> + </Table.Header> + <Table.Body> + {#each job.failures as failure (failure.calibre_id)} + <Table.Row> + <Table.Cell class="font-mono text-xs">#{failure.calibre_id}</Table.Cell> + <Table.Cell>{failure.title}</Table.Cell> + <Table.Cell class="font-mono text-xs">{failure.reason}</Table.Cell> + </Table.Row> + {/each} + </Table.Body> + </Table.Root> + </div> + </div> + {/if} + </Card.Content> + </Card.Root> + {/if} +</div> diff --git a/frontend/src/routes/api/[...path]/+server.ts b/frontend/src/routes/api/[...path]/+server.ts index 35125d3..ea9eaae 100644 --- a/frontend/src/routes/api/[...path]/+server.ts +++ b/frontend/src/routes/api/[...path]/+server.ts @@ -30,6 +30,20 @@ async function handleResponse(response: Response) { } } +/** + * The body to forward, as fetch init options. + * + * Passed through as a stream rather than read into memory first: this process should + * never hold a whole upload, which for a zipped Calibre library could be many + * gigabytes. A request with no body contributes nothing, since `duplex` without a body + * is rejected. + */ +function bodyOf(request: Request) { + if (!request.body) return {}; + + return { body: request.body, duplex: 'half' } as RequestInit; +} + // Shared function to prepare the request with authentication function prepareRequest(locals: App.Locals, request: Request) { const token = locals.authToken || 'server-default-token'; @@ -76,13 +90,14 @@ export const POST: RequestHandler = async ({ params, locals, fetch, request, url const backendUrl = `${BACKEND_API_URL}/${path}${queryString}`; const headers = prepareRequest(locals, request); - // Get the request body - const body = await request.arrayBuffer(); - const response = await fetch(backendUrl, { method: 'POST', headers, - body + // Streamed, not buffered. `arrayBuffer()` held the whole upload in this + // process before forwarding a byte of it, which is survivable for one book + // and not for a zipped Calibre library. `duplex: 'half'` is required by the + // fetch spec whenever the body is a stream. + ...bodyOf(request) }); return handleResponse(response); @@ -100,13 +115,10 @@ export const PATCH: RequestHandler = async ({ params, locals, fetch, request, ur const backendUrl = `${BACKEND_API_URL}/${path}${queryString}`; const headers = prepareRequest(locals, request); - // Get the request body - const body = await request.arrayBuffer(); - const response = await fetch(backendUrl, { method: 'PATCH', headers, - body + ...bodyOf(request) }); return handleResponse(response);