Files
chitai/backend/tests/unit/test_services/test_book_service.py
T
patrick 22539644a9 feat: merge books into one record
Files move on disk before any row moves, then everything repoints in one
transaction and the folded records are deleted. Progress keeps whichever is
furthest, shelves and tags union, and identifiers move only under a name the
survivor lacks. Metadata is left alone unless the caller resolves it, since
choosing between two titles is a judgement this endpoint cannot make. Nothing
is removed from disk.
2026-08-16 20:21:10 -04:00

1303 lines
48 KiB
Python

"""Tests for BookService"""
import zipfile
from io import BytesIO
from pathlib import Path
import pytest
import aiofiles.os as aios
from litestar.datastructures import UploadFile
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from chitai.config import DuplicateScope, settings
from chitai.schemas import BookCreate, BooksCreateFromFiles
from chitai.services import BookService
from chitai.services.book import DuplicateFilesError
from chitai.database import models as m
DATA_FILES = Path("tests/data_files")
EPUB = DATA_FILES / "Metamorphosis - Franz Kafka.epub"
OTHER_EPUB = DATA_FILES / "The Art of War - Sun Tzu.epub"
PDF = DATA_FILES / "Calculus Made Easy - Silvanus Thompson.pdf"
def upload(path: Path, name: str | None = None) -> UploadFile:
"""An uploaded file carrying the bytes of one of the test fixtures."""
return UploadFile(
content_type="application/pdf" if path.suffix == ".pdf" else "application/epub+zip",
filename=name or path.name,
file_data=path.read_bytes(),
)
def edition(path: Path, name: str) -> UploadFile:
"""
Another edition of one of the fixtures: same book, different bytes.
Padding the archive changes the size and the sampled hash without disturbing
anything a reader or a metadata extractor sees, which is the case file-level
dedupe cannot answer and book-level detection exists for.
"""
return UploadFile(
content_type="application/epub+zip",
filename=name,
file_data=path.read_bytes() + b"\0" * 64,
)
@pytest.mark.asyncio
class TestBookServiceCRUD:
"""Test CRUD operation for libraries."""
async def test_update_book(
self, books_service: BookService, test_library: m.Library
) -> None:
book_data = BookCreate(
library_id=1,
title="Fellowship of the Ring",
authors=["J.R.R Tolkien"],
tags=["Fantasy"],
identifiers={"isbn-13": "9780261102354"},
pages=427,
)
book = await books_service.to_model_on_create(book_data.model_dump())
assert isinstance(book, m.Book)
# Add path manually as it won't be generated (not using the create function, but manually inserting into db)
book.path = f"{test_library.root_path}/J.R.R Tolkien/The Fellowship of the Ring"
await aios.makedirs(book.path) # type: ignore[arg-type]
books_service.repository.session.add(book)
await books_service.repository.session.commit()
await books_service.repository.session.refresh(book)
await books_service.update_book(
book.id,
{
"title": "The Fellowship of the Ring",
"identifiers": {"isbn-10": "9780261102354"},
"edition": 3,
"publisher": "Tolkien Estate",
"series": "The Lord of the Rings",
"series_position": "1",
"tags": ["Fantasy", "Adventure"],
},
test_library,
)
updated_book = await books_service.get(book.id)
# Assert updated information is correct
assert updated_book.title == "The Fellowship of the Ring"
assert (
updated_book.path
== f"{test_library.root_path}/J.R.R Tolkien/The Lord of the Rings/01 - The Fellowship of the Ring"
)
assert len(updated_book.identifiers)
assert updated_book.identifiers[0].value == "9780261102354"
assert updated_book.edition == 3
assert updated_book.publisher is not None
assert updated_book.publisher.name == "Tolkien Estate"
assert len(updated_book.tags) == 2
async def test_update_book_reuses_existing_links(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Re-submitting a relationship a book already has must not duplicate its link.
The link tables are unique on (book_id, tag_id) / (book_id, author_id), and
identifiers on (name, book_id), so replacing a collection wholesale used to
insert a row that collided with the one it was replacing.
"""
book_data = BookCreate(
library_id=1,
title="The Two Towers",
authors=["J.R.R Tolkien"],
tags=["Fantasy"],
identifiers={"isbn-13": "9780261102358"},
pages=352,
)
book = await books_service.to_model_on_create(book_data.model_dump())
assert isinstance(book, m.Book)
# Matches what the default template generates, so these updates move nothing.
book.path = f"{test_library.root_path}/J.R.R Tolkien/The Two Towers"
await aios.makedirs(book.path) # type: ignore[arg-type]
books_service.repository.session.add(book)
await books_service.repository.session.commit()
await books_service.repository.session.refresh(book)
# Through the service, so the link collections come back eagerly loaded.
created_book = await books_service.get(book.id)
original_tag_link_id = created_book.tag_links[0].id
# Every collection resubmitted unchanged.
await books_service.update_book(
book.id,
{
"authors": ["J.R.R Tolkien"],
"tags": ["Fantasy"],
"identifiers": {"isbn-13": "9780261102358"},
},
test_library,
)
updated_book = await books_service.get(book.id)
assert [tag.name for tag in updated_book.tags] == ["Fantasy"]
assert [author.name for author in updated_book.authors] == ["J.R.R Tolkien"]
assert len(updated_book.identifiers) == 1
# The existing link is reused, not deleted and reinserted.
assert updated_book.tag_links[0].id == original_tag_link_id
# Keeping one tag while adding another.
await books_service.update_book(
book.id, {"tags": ["Fantasy", "Adventure"]}, test_library
)
updated_book = await books_service.get(book.id)
assert [tag.name for tag in updated_book.tags] == ["Fantasy", "Adventure"]
# Dropping one while keeping the other.
await books_service.update_book(book.id, {"tags": ["Adventure"]}, test_library)
updated_book = await books_service.get(book.id)
assert [tag.name for tag in updated_book.tags] == ["Adventure"]
# A name the book already carries has its value updated in place.
await books_service.update_book(
book.id, {"identifiers": {"isbn-13": "9780261102999"}}, test_library
)
updated_book = await books_service.get(book.id)
assert len(updated_book.identifiers) == 1
assert updated_book.identifiers[0].value == "9780261102999"
async def test_get_files_zips_every_book_file(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The multi-book download must contain the real bytes of each file.
`file.path` holds a bare filename relative to `book.path`, so resolving it on
its own matched nothing on disk and yielded a valid but empty archive.
"""
contents = {
"Dracula.epub": b"epub-payload-" * 512,
"Dracula.pdf": b"pdf-payload-" * 512,
}
book_data = BookCreate(
library_id=test_library.id,
title="Dracula",
authors=["Bram Stoker"],
pages=418,
)
book = await books_service.to_model_on_create(book_data.model_dump())
assert isinstance(book, m.Book)
book.path = f"{test_library.root_path}/Bram Stoker/Dracula"
await aios.makedirs(book.path) # type: ignore[arg-type]
for name, payload in contents.items():
Path(book.path, name).write_bytes(payload)
book.files.append(
m.FileMetadata(
path=name,
size=len(payload),
hash=f"hash-{name}",
content_type=None,
)
)
books_service.repository.session.add(book)
await books_service.repository.session.commit()
archive_bytes = b"".join(
[
chunk
async for chunk in books_service.get_files([book.id], test_library.id)
]
)
with zipfile.ZipFile(BytesIO(archive_bytes)) as archive:
assert archive.testzip() is None
names = archive.namelist()
assert len(names) == len(contents)
# Entries are namespaced by the book's directory, so two books sharing a
# filename cannot overwrite one another.
assert {Path(name).parent.name for name in names} == {"Dracula"}
for name, payload in contents.items():
entry = next(n for n in names if Path(n).name == name)
assert archive.read(entry) == payload
@pytest.mark.asyncio
class TestBookServiceDuplicates:
"""Files that are already stored must not be stored again."""
async def test_duplicate_upload_is_skipped_and_reported(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The second import of a file creates nothing and names where it already is."""
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
assert len(first.books) == 1
assert first.duplicates == []
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
assert second.books == []
assert len(second.duplicates) == 1
duplicate = second.duplicates[0]
assert duplicate.filename == EPUB.name
assert duplicate.book_id == first.books[0].id
assert duplicate.book_title == first.books[0].title
async def test_new_format_beside_a_duplicate_is_still_imported(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A folder is not all-or-nothing: the file that is new must still land."""
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB, "Metamorphosis/book.epub")]),
test_library,
)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(
files=[
upload(EPUB, "Metamorphosis/book.epub"),
upload(PDF, "Metamorphosis/book.pdf"),
]
),
test_library,
)
assert len(result.books) == 1
assert [d.filename for d in result.duplicates] == ["Metamorphosis/book.epub"]
book = await books_service.get(result.books[0].id)
assert [file.path for file in book.files] == ["book.pdf"]
async def test_duplicate_within_one_upload_is_caught(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The same bytes twice in one request has no row to match against yet."""
result = await books_service.create_many_from_files(
BooksCreateFromFiles(
files=[upload(EPUB, "first.epub"), upload(EPUB, "second.epub")]
),
test_library,
)
assert len(result.books) == 1
assert len(result.duplicates) == 1
assert result.duplicates[0].filename == "second.epub"
# Nothing in the database holds it yet, so there is no book to point at.
assert result.duplicates[0].book_id is None
async def test_allow_duplicates_stores_the_file_anyway(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The escape hatch has to work: the hash is not proof of identity."""
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]),
test_library,
allow_duplicates=True,
)
assert len(result.books) == 1
assert result.duplicates == []
async def test_a_different_file_is_not_a_duplicate(
self, books_service: BookService, test_library: m.Library
) -> None:
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB), upload(OTHER_EPUB)]), test_library
)
assert len(result.books) == 2
assert result.duplicates == []
async def test_matching_hash_with_a_different_size_is_not_a_duplicate(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The hash samples 12 KiB, so the size is what makes a match trustworthy."""
created = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
stored = (await books_service.get(created.books[0].id)).files[0]
matches = await books_service.find_duplicate_files(
[(stored.hash, stored.size), (stored.hash, stored.size + 1)], test_library
)
assert (stored.hash, stored.size) in matches
assert (stored.hash, stored.size + 1) not in matches
async def test_create_book_refuses_and_writes_nothing(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A single-book create names its files, so it is refused rather than trimmed."""
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
before = sorted(Path(test_library.root_path).rglob("*"))
with pytest.raises(DuplicateFilesError) as excinfo:
await books_service.create_book(
BookCreate(
library_id=test_library.id,
title="Metamorphosis",
authors=["Franz Kafka"],
files=[upload(EPUB)],
).model_dump(),
test_library,
)
assert len(excinfo.value.duplicates) == 1
assert sorted(Path(test_library.root_path).rglob("*")) == before
async def test_scope_decides_whether_libraries_share(
self,
books_service: BookService,
test_library: m.Library,
session: AsyncSession,
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A second library is a separate collection by default, and not under `global`."""
other = m.Library(
name="Second Library",
slug="second-library",
root_path=str(tmp_path / "second"),
path_template=test_library.path_template,
read_only=False,
)
session.add(other)
await session.commit()
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), other
)
assert len(result.books) == 1
assert result.duplicates == []
monkeypatch.setattr(settings, "duplicate_scope", DuplicateScope.GLOBAL)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), other
)
assert result.books == []
assert len(result.duplicates) == 1
async def test_scope_off_disables_detection(
self,
books_service: BookService,
test_library: m.Library,
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(settings, "duplicate_scope", DuplicateScope.OFF)
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB), upload(EPUB, "copy.epub")]),
test_library,
)
assert len(result.books) == 2
assert result.duplicates == []
async def test_re_adding_a_file_to_its_own_book_does_nothing(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Asking for a state that already holds is not a conflict."""
created = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
book_id = created.books[0].id
await books_service.add_files(book_id, [upload(EPUB)], test_library)
book = await books_service.get(book_id)
assert len(book.files) == 1
async def test_adding_another_books_file_is_refused(
self, books_service: BookService, test_library: m.Library
) -> None:
created = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB), upload(OTHER_EPUB)]), test_library
)
first, second = created.books
with pytest.raises(DuplicateFilesError) as excinfo:
await books_service.add_files(first.id, [upload(OTHER_EPUB)], test_library)
assert excinfo.value.duplicates[0].book_id == second.id
assert len((await books_service.get(first.id)).files) == 1
async def test_consume_duplicate_is_moved_aside(
self,
books_service: BookService,
test_library: m.Library,
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The watcher cannot ask, so a refused file is parked rather than dropped."""
quarantine = tmp_path / "duplicates"
monkeypatch.setattr(settings, "duplicate_path", str(quarantine))
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
consume = tmp_path / "consume"
await aios.makedirs(consume)
dropped = consume / EPUB.name
dropped.write_bytes(EPUB.read_bytes())
result = await books_service.create_many_from_existing_files(
[dropped], consume, test_library
)
assert result.books == []
assert len(result.duplicates) == 1
assert not dropped.exists()
assert (quarantine / test_library.slug / EPUB.name).is_file()
async def test_consume_imports_what_is_new(
self,
books_service: BookService,
test_library: m.Library,
tmp_path: Path,
) -> None:
"""The screening must not disturb the ordinary consume import."""
consume = tmp_path / "consume"
await aios.makedirs(consume / "Metamorphosis")
dropped = consume / "Metamorphosis" / EPUB.name
dropped.write_bytes(EPUB.read_bytes())
result = await books_service.create_many_from_existing_files(
[dropped], consume, test_library
)
assert len(result.books) == 1
assert result.duplicates == []
book = await books_service.get(result.books[0].id)
assert (Path(book.path) / book.files[0].path).is_file()
@pytest.mark.asyncio
class TestBookPathCollisions:
"""Two books must never share a directory, whatever their metadata says."""
async def test_forced_duplicate_gets_its_own_copy(
self, books_service: BookService, test_library: m.Library
) -> None:
"""`allow_duplicates` must add a book, not overwrite the one already there.
The path comes from the metadata alone, so a forced duplicate generates the
same directory and the same filename. Writing it lands on top of the original:
one file on disk, two books pointing at it, and deleting either takes both.
"""
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]),
test_library,
allow_duplicates=True,
)
original = await books_service.get(first.books[0].id)
forced = await books_service.get(second.books[0].id)
assert original.path != forced.path
paths = {
Path(book.path) / book.files[0].path for book in (original, forced)
}
assert len(paths) == 2
assert all(path.is_file() for path in paths)
async def test_deleting_a_forced_duplicate_keeps_the_original(
self, books_service: BookService, test_library: m.Library
) -> None:
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]),
test_library,
allow_duplicates=True,
)
original = await books_service.get(first.books[0].id)
kept = Path(original.path) / original.files[0].path
await books_service.delete_books(
[second.books[0].id], test_library, delete_files=True
)
assert kept.is_file()
async def test_editing_metadata_cannot_merge_into_another_book(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A rename that collides must step aside rather than move in on top."""
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(OTHER_EPUB)]), test_library
)
original = await books_service.get(first.books[0].id)
# Renamed onto the first book's author and title.
await books_service.update_book(
second.books[0].id,
{"title": original.title, "authors": [author.name for author in original.authors]},
test_library,
)
moved = await books_service.get(second.books[0].id)
assert moved.path != original.path
assert (Path(original.path) / original.files[0].path).is_file()
assert (Path(moved.path) / moved.files[0].path).is_file()
async def test_adding_a_file_lands_in_the_books_own_directory(
self, books_service: BookService, test_library: m.Library
) -> None:
"""`add_files` must follow the book, not the template."""
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
forced = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]),
test_library,
allow_duplicates=True,
)
book_id = forced.books[0].id
await books_service.add_files(book_id, [upload(PDF)], test_library)
book = await books_service.get(book_id)
assert len(book.files) == 2
for file in book.files:
assert (Path(book.path) / file.path).is_file()
@pytest.mark.asyncio
class TestMissingFiles:
"""A row whose bytes are gone must not stand in for the file itself."""
async def test_a_missing_file_is_not_a_duplicate(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Otherwise the library refuses to take back a file it can no longer open."""
created = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
book = await books_service.get(created.books[0].id)
(Path(book.path) / book.files[0].path).unlink()
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
assert len(result.books) == 1
assert result.duplicates == []
restored = await books_service.get(result.books[0].id)
assert (Path(restored.path) / restored.files[0].path).is_file()
async def test_re_adding_a_missing_file_puts_it_back(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The book already has a row for it, so the bytes go back where it says."""
created = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
book_id = created.books[0].id
book = await books_service.get(book_id)
path = Path(book.path) / book.files[0].path
path.unlink()
await books_service.add_files(book_id, [upload(EPUB)], test_library)
book = await books_service.get(book_id)
assert len(book.files) == 1
assert path.is_file()
assert path.read_bytes() == EPUB.read_bytes()
async def store_book(
books_service: BookService, library: m.Library, **metadata
) -> m.Book:
"""
A book in the database carrying exactly the metadata given, and no files.
Book-level matching reads metadata only, so going through the file pipeline would
just make every case depend on what a fixture EPUB happens to declare.
"""
book = await books_service.to_model_on_create(
BookCreate(library_id=library.id, **metadata).model_dump()
)
books_service.repository.session.add(book)
await books_service.repository.session.commit()
return book
@pytest.mark.asyncio
class TestDuplicateBooks:
"""The same book arriving as different bytes: reported, never refused."""
async def test_an_identifier_alone_is_enough(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Two printings of one edition agree on the ISBN and nothing else."""
stored = await store_book(
books_service,
test_library,
title="Metamorphosis",
authors=["Franz Kafka"],
identifiers={"isbn-13": "9780486282114"},
)
matches = await books_service.find_duplicate_books(
{
"title": "Die Verwandlung",
"authors": ["Someone Else"],
# The ISBN-10 of the same edition, written with its hyphens.
"identifiers": {"isbn-10": "0-486-28211-2"},
},
test_library,
)
assert [match.book_id for match in matches] == [stored.id]
assert matches[0].matched_on == ["identifier"]
assert matches[0].authors == ["Franz Kafka"]
async def test_a_title_and_a_shared_author_are_enough(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A re-scan carries no identifier at all, only what is on the cover."""
stored = await store_book(
books_service,
test_library,
title="The Metamorphosis",
authors=["Kafka, Franz"],
)
matches = await books_service.find_duplicate_books(
{"title": "Metamorphosis", "authors": ["Franz Kafka"]}, test_library
)
assert [match.book_id for match in matches] == [stored.id]
assert matches[0].matched_on == ["title-author"]
async def test_an_author_left_in_the_title_still_matches(
self, books_service: BookService, test_library: m.Library
) -> None:
"""
Files are named `Title - Author.epub`, and that name often became the title.
One copy stored as "Building Microservices" and another as "Building
Microservices Sam Newman" are the same book, and comparing the title columns
as they stand would never say so.
"""
stored = await store_book(
books_service,
test_library,
title="Building Microservices - Sam Newman",
authors=["Sam Newman"],
)
matches = await books_service.find_duplicate_books(
{"title": "Building Microservices", "authors": ["Newman, Sam;"]}, test_library
)
assert [match.book_id for match in matches] == [stored.id]
assert matches[0].matched_on == ["title-author"]
async def test_the_author_in_the_title_works_the_other_way_round(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Whichever copy arrived first, the comparison has to reach the other."""
stored = await store_book(
books_service,
test_library,
title="Building Microservices",
authors=["Newman, Sam;"],
)
matches = await books_service.find_duplicate_books(
{"title": "Building Microservices - Sam Newman", "authors": ["Sam Newman"]},
test_library,
)
assert [match.book_id for match in matches] == [stored.id]
async def test_a_shared_author_is_still_required(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The title variants must not become a way around the author requirement."""
await store_book(
books_service,
test_library,
title="Building Microservices - Sam Newman",
authors=["Sam Newman"],
)
assert (
await books_service.find_duplicate_books(
{"title": "Building Microservices", "authors": ["Martin Fowler"]},
test_library,
)
== []
)
async def test_a_title_without_a_shared_author_is_not_enough(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Otherwise every book the extractors gave up on matches every other one."""
await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
assert (
await books_service.find_duplicate_books(
{"title": "Metamorphosis", "authors": ["Peter Kuper"]}, test_library
)
== []
)
async def test_a_book_with_no_authors_can_only_match_on_an_identifier(
self, books_service: BookService, test_library: m.Library
) -> None:
await store_book(
books_service, test_library, title="Unknown", authors=["Franz Kafka"]
)
assert (
await books_service.find_duplicate_books(
{"title": "Unknown", "authors": []}, test_library
)
== []
)
async def test_another_volume_of_a_series_is_not_a_duplicate(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The position is the library saying outright that these are two books."""
await store_book(
books_service,
test_library,
title="Foundation",
authors=["Isaac Asimov"],
series="Foundation",
series_position="1",
)
incoming = {
"title": "Foundation",
"authors": ["Isaac Asimov"],
"series": "Foundation",
}
assert await books_service.find_duplicate_books(
incoming | {"series_position": "2"}, test_library
) == []
# The same volume, written a little differently, still matches.
assert len(
await books_service.find_duplicate_books(
incoming | {"series_position": "1.0"}, test_library
)
) == 1
async def test_a_book_is_not_its_own_duplicate(
self, books_service: BookService, test_library: m.Library
) -> None:
stored = await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
assert (
await books_service.find_duplicate_books(
stored, test_library, exclude_book_id=stored.id
)
== []
)
async def test_scope_off_reports_nothing(
self,
books_service: BookService,
test_library: m.Library,
monkeypatch: pytest.MonkeyPatch,
) -> None:
await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
monkeypatch.setattr(settings, "duplicate_scope", DuplicateScope.OFF)
incoming = {"title": "Metamorphosis", "authors": ["Franz Kafka"]}
assert await books_service.find_duplicate_books(incoming, test_library) == []
assert await books_service.find_duplicate_book_groups(test_library) == []
async def test_an_import_reports_the_copy_it_just_made(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Different bytes, same book — created, and said so."""
await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB)]), test_library
)
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[edition(EPUB, "second.epub")]), test_library
)
assert len(result.books) == 1
assert result.duplicates == []
assert len(result.possible_duplicates) == 1
possible = result.possible_duplicates[0]
assert possible.book_id == result.books[0].id
assert "title-author" in possible.candidates[0].matched_on
@pytest.mark.asyncio
class TestAuthorNames:
"""One person is one row, however the file happened to spell them."""
async def test_a_variant_spelling_reuses_the_existing_author(
self, books_service: BookService, test_library: m.Library
) -> None:
"""
`as_unique_async` looks a name up before inserting it, so the lookup and the
insert have to tidy identically. If they disagree, every variant misses the
existing row and then collides with it on the unique index.
"""
first = await store_book(
books_service, test_library, title="Building Microservices", authors=["Sam Newman"]
)
second = await store_book(
books_service,
test_library,
title="Monolith to Microservices",
authors=["Newman, Sam;"],
)
assert [author.name for author in first.authors] == ["Sam Newman"]
assert [author.name for author in second.authors] == ["Sam Newman"]
assert first.authors[0].id == second.authors[0].id
@pytest.mark.parametrize(
("written", "stored"),
[
("Newman, Sam;", "Sam Newman"),
("Sam Newman.epub", "Sam Newman"),
(" Sam Newman ", "Sam Newman"),
# Two people in one string is left exactly as it was found.
("Dave Thomas, Andy Hunt", "Dave Thomas, Andy Hunt"),
],
)
async def test_the_stored_name_is_the_tidy_one(
self,
books_service: BookService,
test_library: m.Library,
written: str,
stored: str,
) -> None:
book = await store_book(
books_service, test_library, title="Some Book", authors=[written]
)
assert [author.name for author in book.authors] == [stored]
async def test_an_upload_does_not_keep_the_file_extension(
self, books_service: BookService, test_library: m.Library
) -> None:
"""
A filename is the fallback when the file declares no author of its own, and
"Franz Kafka.epub" is not a person.
"""
result = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB, "Some Unknown Book - Ada Lovelace.epub")]),
test_library,
)
book = await books_service.get(result.books[0].id)
for author in book.authors:
assert not author.name.endswith((".epub", ".pdf", ".mobi"))
@pytest.mark.asyncio
class TestMergeBooks:
"""Folding several books into one, and what has to come with them."""
async def test_files_move_onto_the_survivor(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The rows move and so do the bytes — `file.path` is relative to `book.path`."""
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB, "one.epub")]), test_library
)
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(PDF, "two.pdf")]), test_library
)
keep, fold = first.books[0], second.books[0]
gone = Path(fold.path)
merged = await books_service.merge_books(keep.id, [fold.id], test_library)
assert len(merged.files) == 2
for file in merged.files:
assert (Path(merged.path) / file.path).is_file()
assert not gone.exists()
with pytest.raises(Exception):
await books_service.get(fold.id)
async def test_a_filename_collision_is_given_its_own_name(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Two books' files can share a name; one must not overwrite the other."""
first = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[upload(EPUB, "book.epub")]), test_library
)
second = await books_service.create_many_from_files(
BooksCreateFromFiles(files=[edition(EPUB, "book.epub")]), test_library
)
merged = await books_service.merge_books(
first.books[0].id, [second.books[0].id], test_library
)
paths = sorted(file.path for file in merged.files)
assert len(paths) == 2 and paths[0] != paths[1]
for file in merged.files:
assert (Path(merged.path) / file.path).is_file()
async def test_metadata_is_only_what_the_caller_resolved(
self, books_service: BookService, test_library: m.Library
) -> None:
"""The survivor keeps its own fields unless the caller says otherwise."""
keep = await store_book(
books_service, test_library, title="Building Microservices", authors=["Sam Newman"]
)
fold = await store_book(
books_service,
test_library,
title="Building Microservices",
authors=["Sam Newman"],
publisher="O'Reilly",
edition=2,
)
merged = await books_service.merge_books(keep.id, [fold.id], test_library)
assert merged.edition is None
assert merged.publisher is None
other = await store_book(
books_service, test_library, title="Monolith", authors=["Sam Newman"], edition=3
)
merged = await books_service.merge_books(
keep.id, [other.id], test_library, metadata={"edition": 3}
)
assert merged.edition == 3
async def test_resolved_collections_do_not_collide_with_what_moved(
self, books_service: BookService, test_library: m.Library
) -> None:
"""
Tags and identifiers move onto the survivor *and* appear in the resolution.
The survivor's loaded collections do not learn about a row repointed by a bulk
statement, so reconciling them against stale state builds a second link and
breaches the unique constraint. Scalar-only metadata never sees it.
"""
keep = await store_book(
books_service, test_library, title="A", authors=["X"], tags=["Kept"]
)
fold = await store_book(
books_service,
test_library,
title="B",
authors=["X"],
tags=["Moved"],
identifiers={"asin": "B000FC0PDA"},
)
merged = await books_service.merge_books(
keep.id,
[fold.id],
test_library,
metadata={
"tags": ["Kept", "Moved"],
"identifiers": {"asin": "B000FC0PDA"},
"publisher": "O'Reilly",
},
)
assert sorted(tag.name for tag in merged.tags) == ["Kept", "Moved"]
assert {i.name: i.value for i in merged.identifiers} == {"asin": "B000FC0PDA"}
assert merged.publisher is not None and merged.publisher.name == "O'Reilly"
async def test_the_furthest_progress_survives(
self,
books_service: BookService,
test_library: m.Library,
test_user: m.User,
session: AsyncSession,
) -> None:
"""A reader who got 60% through the PDF has not gone back to page one."""
keep = await store_book(books_service, test_library, title="A", authors=["X"])
fold = await store_book(books_service, test_library, title="B", authors=["X"])
session.add_all(
[
m.BookProgress(user_id=test_user.id, book_id=keep.id, percentage=0.1),
m.BookProgress(user_id=test_user.id, book_id=fold.id, percentage=0.6),
]
)
await session.commit()
await books_service.merge_books(keep.id, [fold.id], test_library)
rows = (
await books_service.repository.session.execute(
select(m.BookProgress).where(m.BookProgress.book_id == keep.id)
)
).scalars().all()
assert [row.percentage for row in rows] == [0.6]
async def test_shelves_and_tags_come_across_without_duplicating(
self,
books_service: BookService,
test_library: m.Library,
test_user: m.User,
session: AsyncSession,
) -> None:
"""Both books on one shelf must not leave the survivor linked to it twice."""
keep = await store_book(
books_service, test_library, title="A", authors=["X"], tags=["Shared", "Only Keep"]
)
fold = await store_book(
books_service, test_library, title="B", authors=["X"], tags=["Shared", "Only Fold"]
)
shelf = m.BookList(title="Later", user_id=test_user.id, library_id=test_library.id)
session.add(shelf)
await session.commit()
session.add_all(
[
m.BookListLink(book_id=keep.id, list_id=shelf.id, position=0),
m.BookListLink(book_id=fold.id, list_id=shelf.id, position=0),
]
)
await session.commit()
merged = await books_service.merge_books(keep.id, [fold.id], test_library)
assert sorted(tag.name for tag in merged.tags) == ["Only Fold", "Only Keep", "Shared"]
links = (
await books_service.repository.session.execute(
select(m.BookListLink).where(m.BookListLink.book_id == keep.id)
)
).scalars().all()
assert len(links) == 1
async def test_an_identifier_moves_only_under_a_name_the_survivor_lacks(
self, books_service: BookService, test_library: m.Library
) -> None:
"""`(name, book_id)` is unique, so a competing isbn-13 cannot come across."""
keep = await store_book(
books_service,
test_library,
title="A",
authors=["X"],
identifiers={"isbn-13": "9780486282114"},
)
fold = await store_book(
books_service,
test_library,
title="B",
authors=["X"],
identifiers={"isbn-13": "9781492034025", "asin": "B000FC0PDA"},
)
merged = await books_service.merge_books(keep.id, [fold.id], test_library)
stored = {i.name: i.value for i in merged.identifiers}
assert stored == {"isbn-13": "9780486282114", "asin": "B000FC0PDA"}
async def test_dismissals_follow_the_survivor(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A verdict about some third book still holds after its partner is merged."""
keep = await store_book(books_service, test_library, title="A", authors=["X"])
fold = await store_book(books_service, test_library, title="B", authors=["X"])
third = await store_book(books_service, test_library, title="C", authors=["X"])
# One pair between the two being merged, one pointing outside the merge.
await books_service.dismiss_duplicates(keep.id, fold.id)
await books_service.dismiss_duplicates(fold.id, third.id)
await books_service.merge_books(keep.id, [fold.id], test_library)
pairs = await books_service._dismissed_pairs()
assert pairs == {m.DuplicateDismissal.pair(keep.id, third.id)}
@pytest.mark.parametrize("merged", [[], [1]])
async def test_a_merge_needs_two_different_books(
self, books_service: BookService, test_library: m.Library, merged: list[int]
) -> None:
keep = await store_book(books_service, test_library, title="A", authors=["X"])
ids = [keep.id] if merged else []
with pytest.raises(ValueError):
await books_service.merge_books(keep.id, ids, test_library)
async def test_an_unknown_book_is_refused(
self, books_service: BookService, test_library: m.Library
) -> None:
keep = await store_book(books_service, test_library, title="A", authors=["X"])
with pytest.raises(ValueError):
await books_service.merge_books(keep.id, [keep.id + 10_000], test_library)
@pytest.mark.asyncio
class TestDuplicateBookGroups:
"""The pass over a library someone already has."""
async def test_groups_join_transitively(
self, books_service: BookService, test_library: m.Library
) -> None:
"""A and B by ISBN, B and C by title and author, all three in one group."""
first = await store_book(
books_service,
test_library,
title="Metamorphosis",
authors=["Franz Kafka"],
identifiers={"isbn-13": "9780486282114"},
)
second = await store_book(
books_service,
test_library,
title="The Trial",
authors=["Franz Kafka"],
identifiers={"isbn-10": "0486282112"},
)
third = await store_book(
books_service, test_library, title="Trial", authors=["Kafka, Franz"]
)
groups = await books_service.find_duplicate_book_groups(test_library)
assert len(groups) == 1
assert [book.book_id for book in groups[0]] == [first.id, second.id, third.id]
reasons = {book.book_id: book.matched_on for book in groups[0]}
assert reasons[first.id] == ["identifier"]
assert reasons[second.id] == ["identifier", "title-author"]
assert reasons[third.id] == ["title-author"]
async def test_a_lone_book_is_not_a_group(
self, books_service: BookService, test_library: m.Library
) -> None:
await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
assert await books_service.find_duplicate_book_groups(test_library) == []
async def test_a_dismissed_pair_is_left_out(
self, books_service: BookService, test_library: m.Library
) -> None:
"""Disagreeing with one pairing must not break a group standing on another."""
first = await store_book(
books_service,
test_library,
title="Metamorphosis",
authors=["Franz Kafka"],
identifiers={"isbn-13": "9780486282114"},
)
second = await store_book(
books_service,
test_library,
title="The Trial",
authors=["Franz Kafka"],
identifiers={"isbn-10": "0486282112"},
)
third = await store_book(
books_service, test_library, title="Trial", authors=["Kafka, Franz"]
)
await books_service.dismiss_duplicates(second.id, first.id)
groups = await books_service.find_duplicate_book_groups(test_library)
assert [[book.book_id for book in group] for group in groups] == [
[second.id, third.id]
]
await books_service.restore_duplicates(first.id, second.id)
groups = await books_service.find_duplicate_book_groups(test_library)
assert [[book.book_id for book in group] for group in groups] == [
[first.id, second.id, third.id]
]
async def test_a_book_cannot_be_dismissed_against_itself(
self, books_service: BookService, test_library: m.Library
) -> None:
stored = await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
with pytest.raises(ValueError):
await books_service.dismiss_duplicates(stored.id, stored.id)
async def test_dismissing_an_unknown_book_is_refused(
self, books_service: BookService, test_library: m.Library
) -> None:
stored = await store_book(
books_service, test_library, title="Metamorphosis", authors=["Franz Kafka"]
)
with pytest.raises(ValueError):
await books_service.dismiss_duplicates(stored.id, stored.id + 10_000)