Files
chitai/backend/tests/integration/test_calibre_import.py
T
patrick 55e00ba960 feat: import a Calibre library
Reads metadata.db and copies the books into a library — from a zip uploaded on
the library settings page, or from a path with `litestar calibre-import`. The
source is never touched, and re-running only picks up what is new.

Also names the formats mimetypes does not know: a Calibre library is full of
MOBI and AZW3, and a null content type used to fail the book endpoint.
2026-08-17 13:38:44 -04:00

401 lines
13 KiB
Python

"""
Tests for the Calibre import endpoints.
The API takes a zipped library and nothing else — a desktop Calibre install is usually
not on the server, and importing from a path the server can already see stays a
server-side operation (`litestar calibre-import`).
"""
import asyncio
import tempfile
import zipfile
from pathlib import Path
import pytest
from httpx import AsyncClient
from chitai.services.calibre_import import registry
from tests.calibre_fixtures import CalibreFixture
EPUB = Path("tests/data_files/Metamorphosis - Franz Kafka.epub")
OTHER_EPUB = Path("tests/data_files/The Art of War - Sun Tzu.epub")
@pytest.fixture(autouse=True)
def _clear_registry():
"""The registry is a module-level singleton, so it leaks between tests."""
registry._jobs.clear()
yield
registry._jobs.clear()
@pytest.fixture(name="source")
def fx_source(tmp_path: Path) -> Path:
fixture = CalibreFixture(tmp_path / "calibre")
fixture.add_book(
1,
"The Metamorphosis",
authors=["Franz Kafka"],
tags=["Fiction"],
cover=True,
formats={"EPUB": EPUB},
)
fixture.add_book(2, "The Art of War", authors=["Sun Tzu"], formats={"EPUB": OTHER_EPUB})
fixture.add_book(3, "Metadata Only", authors=["Nobody"])
return fixture.commit()
def zip_of(root: Path, into: Path, prefix: str = "") -> Path:
"""Zip a directory the way a file manager would."""
into.mkdir(parents=True, exist_ok=True)
archive = into / "library.zip"
with zipfile.ZipFile(archive, "w") as writing:
for path in sorted(root.rglob("*")):
if path.is_file():
writing.write(path, f"{prefix}{path.relative_to(root)}")
return archive
async def upload(
client: AsyncClient,
archive: Path,
library_id: int = 1,
allow_duplicates: bool = False,
) -> tuple[int, dict]:
response = await client.post(
f"/libraries/{library_id}/imports/calibre/upload",
files=[("archive", (archive.name, archive.read_bytes(), "application/zip"))],
data={"allow_duplicates": str(allow_duplicates).lower()},
)
return response.status_code, response.json()
async def wait_for(client: AsyncClient, job_id: str) -> dict:
"""Poll until the job is no longer running, the way the screen does."""
for _ in range(200):
response = await client.get(f"/libraries/imports/{job_id}")
assert response.status_code == 200
job = response.json()
if job["state"] != "running":
return job
await asyncio.sleep(0.05)
raise AssertionError("the import never finished")
async def test_an_uploaded_library_imports(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
status, job = await upload(
authenticated_client, zip_of(source, tmp_path / "out", prefix="Calibre Library/")
)
assert status == 202
assert job["state"] == "running"
assert job["library_id"] == 1
# The archive's name, not the temp directory it was unpacked into.
assert job["source"] == "library.zip"
finished = await wait_for(authenticated_client, job["id"])
assert finished["state"] == "finished"
assert finished["total"] == 3
assert finished["created"] == 2
assert finished["skipped"] == 1
assert finished["failed"] == 0
assert finished["error"] is None
assert finished["current_title"] is None
listed = await authenticated_client.get("/books?library_id=1")
titles = [book["title"] for book in listed.json()["items"]]
assert "The Metamorphosis" in titles
assert "The Art of War" in titles
async def test_a_library_zipped_without_a_wrapping_folder(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""Zipping the contents is as common as zipping the folder."""
status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
assert status == 202
finished = await wait_for(authenticated_client, job["id"])
assert finished["created"] == 2
async def test_the_unpacked_copy_is_cleaned_up(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""
An unpacked archive is a second copy of the whole library.
The books worth keeping have been copied into the library by the time the job ends,
so nothing is lost with it — and nothing will come back for it.
"""
status, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
assert status == 202
workspace = registry.get(job["id"]).workspace
assert workspace is not None
await wait_for(authenticated_client, job["id"])
assert not workspace.exists()
async def test_uploading_the_same_library_twice_imports_nothing_new(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""Re-running is safe, which is what makes an interrupted import resumable."""
archive = zip_of(source, tmp_path / "out")
for _ in range(2):
_, job = await upload(authenticated_client, archive)
finished = await wait_for(authenticated_client, job["id"])
assert finished["created"] == 0
assert finished["skipped"] == 3
async def test_allow_duplicates_stores_the_files_again(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""
The one option the screen offers, and it has to reach the import.
Without it the second pass skips everything, which is the previous test.
"""
archive = zip_of(source, tmp_path / "out")
_, first = await upload(authenticated_client, archive)
await wait_for(authenticated_client, first["id"])
_, second = await upload(authenticated_client, archive, allow_duplicates=True)
finished = await wait_for(authenticated_client, second["id"])
assert finished["created"] == 2
assert finished["skipped"] == 1 # still the book with no files
async def test_two_imports_into_one_library_conflict(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
archive = zip_of(source, tmp_path / "out")
first_status, first = await upload(authenticated_client, archive)
assert first_status == 202
second_status, second = await upload(authenticated_client, archive)
assert second_status == 409
assert second["extra"]["job_id"] == first["id"]
await wait_for(authenticated_client, first["id"])
async def test_a_finished_import_does_not_block_the_next_one(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
archive = zip_of(source, tmp_path / "out")
_, first = await upload(authenticated_client, archive)
await wait_for(authenticated_client, first["id"])
status, second = await upload(authenticated_client, archive)
assert status == 202
await wait_for(authenticated_client, second["id"])
async def test_cancelling_stops_after_the_current_book(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""
Cancelling is not aborting: a book abandoned mid-copy would leave files with no row.
Whether this cancels before any book, after one, or after the lot is a race — the
catalogue is three books long. What must hold either way is that the state is
terminal and every book it did import is complete.
"""
_, job = await upload(authenticated_client, zip_of(source, tmp_path / "out"))
cancelled = await authenticated_client.delete(f"/libraries/imports/{job['id']}")
assert cancelled.status_code == 200
final = await wait_for(authenticated_client, job["id"])
assert final["state"] in {"cancelled", "finished"}
listed = await authenticated_client.get("/books?library_id=1")
for book in listed.json()["items"]:
assert book["files"]
async def test_failures_are_reported_on_the_job(
authenticated_client: AsyncClient, tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""One book failing is recorded and does not stop the run."""
fixture = CalibreFixture(tmp_path / "calibre")
fixture.add_book(1, "Fine", authors=["A"], formats={"EPUB": EPUB})
fixture.add_book(2, "Doomed", authors=["B"], formats={"EPUB": OTHER_EPUB})
root = fixture.commit()
from chitai.services.book import BookService
original = BookService.create
async def fail_on_the_second(self, data, *args, **kwargs):
if isinstance(data, dict) and data.get("title") == "Doomed":
raise RuntimeError("no room on the shelf")
return await original(self, data, *args, **kwargs)
monkeypatch.setattr(BookService, "create", fail_on_the_second)
_, job = await upload(authenticated_client, zip_of(root, tmp_path / "out"))
finished = await wait_for(authenticated_client, job["id"])
assert finished["state"] == "finished"
assert finished["created"] == 1
assert finished["failed"] == 1
assert finished["failures"][0]["calibre_id"] == 2
assert "no room on the shelf" in finished["failures"][0]["reason"]
async def test_a_second_copy_is_counted_as_a_possible_duplicate(
authenticated_client: AsyncClient, tmp_path: Path
) -> None:
"""A count, not the records — the duplicates screen is what shows them."""
padded = tmp_path / "padded.epub"
padded.write_bytes(EPUB.read_bytes() + b"\0" * 64)
fixture = CalibreFixture(tmp_path / "calibre")
fixture.add_book(1, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": EPUB})
fixture.add_book(
2, "The Metamorphosis", authors=["Franz Kafka"], formats={"EPUB": padded}
)
root = fixture.commit()
_, job = await upload(authenticated_client, zip_of(root, tmp_path / "out"))
finished = await wait_for(authenticated_client, job["id"])
assert finished["created"] == 2
assert finished["possible_duplicates"] == 1
async def test_polling_an_unknown_job(authenticated_client: AsyncClient) -> None:
response = await authenticated_client.get("/libraries/imports/not-a-job")
assert response.status_code == 404
async def test_cancelling_an_unknown_job(authenticated_client: AsyncClient) -> None:
response = await authenticated_client.delete("/libraries/imports/not-a-job")
assert response.status_code == 404
async def test_importing_into_a_library_that_does_not_exist(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
status, _ = await upload(
authenticated_client, zip_of(source, tmp_path / "out"), library_id=999
)
assert status == 404
async def test_importing_into_a_read_only_library(
authenticated_client: AsyncClient, source: Path, tmp_path: Path
) -> None:
"""A read-only library points at a tree Chitai does not own."""
root = tmp_path / "read-only"
root.mkdir()
created = await authenticated_client.post(
"/libraries",
json={"name": "Read Only", "root_path": str(root), "read_only": True},
)
assert created.status_code == 201
status, body = await upload(
authenticated_client,
zip_of(source, tmp_path / "out"),
library_id=created.json()["id"],
)
assert status == 400
assert "read-only" in body["detail"]
async def test_an_import_needs_authentication(
client: AsyncClient, source: Path, tmp_path: Path
) -> None:
status, _ = await upload(client, zip_of(source, tmp_path / "out"))
assert status == 401
class TestRefusedArchives:
"""Everything wrong with an archive is answered now, not as a job that fails later."""
async def test_a_hostile_archive(
self, authenticated_client: AsyncClient, tmp_path: Path
) -> None:
"""Zip slip."""
archive = tmp_path / "hostile.zip"
with zipfile.ZipFile(archive, "w") as writing:
writing.writestr("metadata.db", "not really")
writing.writestr("../../escaped.txt", "gotcha")
status, body = await upload(authenticated_client, archive)
assert status == 400
assert "outside itself" in body["detail"]
assert registry._jobs == {}
async def test_something_that_is_not_a_zip(
self, authenticated_client: AsyncClient, tmp_path: Path
) -> None:
archive = tmp_path / "notes.txt"
archive.write_bytes(b"just some text")
status, body = await upload(authenticated_client, archive)
assert status == 400
assert "not a zip" in body["detail"]
async def test_an_archive_with_no_catalogue(
self, authenticated_client: AsyncClient, tmp_path: Path
) -> None:
archive = tmp_path / "books.zip"
with zipfile.ZipFile(archive, "w") as writing:
writing.writestr("Some Book.epub", "content")
status, body = await upload(authenticated_client, archive)
assert status == 400
assert "no metadata.db" in body["detail"]
async def test_a_refusal_leaves_no_temp_files(
self, authenticated_client: AsyncClient, tmp_path: Path
) -> None:
"""Every refusal path removes the workspace it had already made."""
before = set(Path(tempfile.gettempdir()).glob("tmp*"))
archive = tmp_path / "books.zip"
with zipfile.ZipFile(archive, "w") as writing:
writing.writestr("Some Book.epub", "content")
await upload(authenticated_client, archive)
assert set(Path(tempfile.gettempdir()).glob("tmp*")) == before