Files
chitai/backend/tests/calibre_fixtures.py
T
patrick 55e00ba960 feat: import a Calibre library
Reads metadata.db and copies the books into a library — from a zip uploaded on
the library settings page, or from a path with `litestar calibre-import`. The
source is never touched, and re-running only picks up what is new.

Also names the formats mimetypes does not know: a Calibre library is full of
MOBI and AZW3, and a null content type used to fail the book endpoint.
2026-08-17 13:38:44 -04:00

283 lines
9.9 KiB
Python

"""
Build a Calibre library on disk, for tests to read.
Generated rather than committed as a binary `metadata.db`, because the rows worth
testing are the awkward ones — the year-101 pubdate, a `|` in an author name, a REAL
series index, HTML in a comment — and those are clearer written out in Python than
hidden inside a blob.
The schema below is Calibre's own, copied from a real library's `sqlite_master`, reduced
to the tables the reader touches. `books_pages_link` is created separately by
`add_pages`: it is recent, and a library made by an older Calibre will not have it.
"""
from __future__ import annotations
import shutil
import sqlite3
from pathlib import Path
SCHEMA = """
CREATE TABLE books (
id INTEGER PRIMARY KEY AUTOINCREMENT,
title TEXT NOT NULL DEFAULT 'Unknown' COLLATE NOCASE,
sort TEXT COLLATE NOCASE,
timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
pubdate TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
series_index REAL NOT NULL DEFAULT 1.0,
author_sort TEXT COLLATE NOCASE,
path TEXT NOT NULL DEFAULT '',
uuid TEXT,
has_cover BOOL DEFAULT 0,
last_modified TIMESTAMP NOT NULL DEFAULT '2000-01-01 00:00:00+00:00'
);
CREATE TABLE authors (
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name)
);
CREATE TABLE books_authors_link (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, author INTEGER NOT NULL,
UNIQUE(book, author)
);
CREATE TABLE publishers (
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE(name)
);
CREATE TABLE books_publishers_link (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, publisher INTEGER NOT NULL,
UNIQUE(book)
);
CREATE TABLE tags (
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
link TEXT NOT NULL DEFAULT '', UNIQUE (name)
);
CREATE TABLE books_tags_link (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, tag INTEGER NOT NULL,
UNIQUE(book, tag)
);
CREATE TABLE series (
id INTEGER PRIMARY KEY, name TEXT NOT NULL COLLATE NOCASE,
sort TEXT COLLATE NOCASE, link TEXT NOT NULL DEFAULT '', UNIQUE (name)
);
CREATE TABLE books_series_link (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, series INTEGER NOT NULL,
UNIQUE(book)
);
CREATE TABLE languages (
id INTEGER PRIMARY KEY, lang_code TEXT NOT NULL COLLATE NOCASE,
link TEXT NOT NULL DEFAULT '', UNIQUE(lang_code)
);
CREATE TABLE books_languages_link (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL, lang_code INTEGER NOT NULL,
item_order INTEGER NOT NULL DEFAULT 0, UNIQUE(book, lang_code)
);
CREATE TABLE comments (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
text TEXT NOT NULL COLLATE NOCASE, UNIQUE(book)
);
CREATE TABLE identifiers (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
type TEXT NOT NULL DEFAULT 'isbn' COLLATE NOCASE,
val TEXT NOT NULL COLLATE NOCASE, UNIQUE(book, type)
);
CREATE TABLE data (
id INTEGER PRIMARY KEY, book INTEGER NOT NULL,
format TEXT NOT NULL COLLATE NOCASE, uncompressed_size INTEGER NOT NULL,
name TEXT NOT NULL, UNIQUE(book, format)
);
"""
# Calibre's own "no date". Stored, never null, and a valid date — which is exactly why
# it has to be recognised rather than parsed.
UNDEFINED_DATE = "0101-01-01 00:00:00+00:00"
# What a `cover.jpg` that PIL cannot read looks like. Real libraries hold these, from
# an interrupted download or a failed conversion.
CORRUPT_COVER = b"\xff\xd8\xff\xe0 not really a jpeg"
def write_cover(path: Path) -> None:
"""
Write a real, readable JPEG.
Generated with PIL rather than embedded as a hex blob: a hand-rolled JPEG that is
subtly malformed fails inside the import as an unrelated error, which is exactly the
confusion this avoids.
"""
from PIL import Image
Image.new("RGB", (2, 3), (10, 20, 30)).save(path, "JPEG")
class CalibreFixture:
"""A Calibre library being assembled under `root`."""
def __init__(self, root: Path) -> None:
self.root = root
self.root.mkdir(parents=True, exist_ok=True)
self.connection = sqlite3.connect(self.root / "metadata.db")
self.connection.executescript(SCHEMA)
def add_pages_table(self) -> None:
"""Add `books_pages_link`, which only a recent Calibre creates."""
self.connection.executescript(
"""
CREATE TABLE books_pages_link (
book INTEGER PRIMARY KEY,
pages INTEGER DEFAULT 0 NOT NULL,
algorithm INTEGER DEFAULT 0 NOT NULL,
format TEXT DEFAULT '' NOT NULL COLLATE NOCASE,
format_size INTEGER DEFAULT 0 NOT NULL,
timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
needs_scan INTEGER NOT NULL DEFAULT 0
);
"""
)
def add_book(
self,
book_id: int,
title: str,
*,
authors: list[str] | None = None,
pubdate: str = UNDEFINED_DATE,
series: str | None = None,
series_index: float = 1.0,
tags: list[str] | None = None,
publisher: str | None = None,
languages: list[str] | None = None,
comment: str | None = None,
identifiers: dict[str, str] | None = None,
uuid: str | None = None,
pages: int | None = None,
cover: bool = False,
corrupt_cover: bool = False,
formats: dict[str, Path] | None = None,
directory: str | None = None,
) -> Path:
"""
Add one book, with its files laid out the way Calibre lays them out.
Args:
formats: Format name (`EPUB`) to a real file to copy in. Its on-disk stem is
Calibre's, not the title — that is the point of the `data` table.
directory: Override the `books.path` value, for testing a row whose
directory is not where the convention would put it.
Returns:
The book's directory.
"""
author_names = authors or ["Unknown"]
relative = directory or f"{author_names[0]}/{title} ({book_id})"
book_directory = self.root / relative
book_directory.mkdir(parents=True, exist_ok=True)
self.connection.execute(
"INSERT INTO books (id, title, pubdate, series_index, path, uuid, has_cover) "
"VALUES (?, ?, ?, ?, ?, ?, ?)",
(
book_id,
title,
pubdate,
series_index,
relative,
uuid or f"uuid-{book_id}",
int(cover or corrupt_cover),
),
)
for name in author_names:
self._link("authors", "books_authors_link", "author", book_id, name)
for name in tags or []:
self._link("tags", "books_tags_link", "tag", book_id, name)
if series:
self._link("series", "books_series_link", "series", book_id, series)
if publisher:
self._link(
"publishers", "books_publishers_link", "publisher", book_id, publisher
)
for order, code in enumerate(languages or []):
language_id = self._lookup("languages", "lang_code", code)
self.connection.execute(
"INSERT INTO books_languages_link (book, lang_code, item_order) "
"VALUES (?, ?, ?)",
(book_id, language_id, order),
)
if comment is not None:
self.connection.execute(
"INSERT INTO comments (book, text) VALUES (?, ?)", (book_id, comment)
)
for name, value in (identifiers or {}).items():
self.connection.execute(
"INSERT INTO identifiers (book, type, val) VALUES (?, ?, ?)",
(book_id, name, value),
)
if pages is not None:
self.connection.execute(
"INSERT INTO books_pages_link (book, pages) VALUES (?, ?)",
(book_id, pages),
)
if corrupt_cover:
(book_directory / "cover.jpg").write_bytes(CORRUPT_COVER)
elif cover:
write_cover(book_directory / "cover.jpg")
for format, origin in (formats or {}).items():
# Calibre's on-disk stem: sanitised, truncated, and not the title.
stem = f"{title[:40]} - {author_names[0]}".replace(":", "_")
destination = book_directory / f"{stem}.{format.lower()}"
shutil.copy(origin, destination)
self.connection.execute(
"INSERT INTO data (book, format, uncompressed_size, name) "
"VALUES (?, ?, ?, ?)",
(book_id, format, destination.stat().st_size, stem),
)
return book_directory
def add_missing_format(self, book_id: int, format: str, stem: str) -> None:
"""Record a file in the catalogue without putting one on disk."""
self.connection.execute(
"INSERT INTO data (book, format, uncompressed_size, name) VALUES (?, ?, ?, ?)",
(book_id, format, 1234, stem),
)
def _link(
self, table: str, link_table: str, column: str, book_id: int, name: str
) -> None:
item_id = self._lookup(table, "name", name)
self.connection.execute(
f"INSERT INTO {link_table} (book, {column}) VALUES (?, ?)",
(book_id, item_id),
)
def _lookup(self, table: str, column: str, value: str) -> int:
row = self.connection.execute(
f"SELECT id FROM {table} WHERE {column} = ?", (value,)
).fetchone()
if row:
return int(row[0])
cursor = self.connection.execute(
f"INSERT INTO {table} ({column}) VALUES (?)", (value,)
)
return int(cursor.lastrowid or 0)
def commit(self) -> Path:
"""Finish writing and return the library root."""
self.connection.commit()
self.connection.close()
return self.root