Files
chitai/backend/tests/unit/test_matching.py
T
patrick 5ec5a4d334 chore: bring the test suite up to ruff's standards
Mostly automatic: empty f-strings, unused imports, formatting. The manual half
was duplicated import blocks stranded mid-file and `== None` assertions, which
become `is None` here because they compare plain attributes -- unlike the
identical rule in filters/book.py, where they build SQL.

Two unused variables are the tests never checking the disk in either
delete_files direction. Left in place under a noqa so the gap stays visible.
2026-08-17 20:56:24 -04:00

242 lines
9.9 KiB
Python

"""Tests for the normalization behind book-level duplicate detection."""
import pytest
from chitai.services.matching import (
format_author_name,
normalize_author,
normalize_identifier,
normalize_title,
)
from chitai.services.metadata_extractor import parse_identifier
from chitai.services.utils import isbn10_to_isbn13
class TestNormalizeTitle:
"""Two copies of one book rarely agree on how the title is written."""
@pytest.mark.parametrize(
("title", "expected"),
[
("The Metamorphosis", "metamorphosis"),
("Metamorphosis", "metamorphosis"),
("METAMORPHOSIS", "metamorphosis"),
("A Tale of Two Cities", "tale of two cities"),
("An Enquiry", "enquiry"),
# Accents, punctuation and ampersands are spelling, not identity.
("Les Misérables", "les miserables"),
("Moby Dick; Or, The Whale", "moby dick or the whale"),
("Sense & Sensibility", "sense and sensibility"),
# Bracketed asides and trailing edition noise say nothing about the book.
("Frankenstein (Illustrated)", "frankenstein"),
("Frankenstein [Kindle Edition]", "frankenstein"),
("Frankenstein, 2nd Edition", "frankenstein"),
# The compact forms a cover actually carries.
("Building Microservices, 2E", "building microservices"),
("Building Microservices 2e", "building microservices"),
("Frankenstein 3 Ed", "frankenstein"),
("Dungeons & Dragons 5e", "dungeons and dragons"),
("Frankenstein Revised Edition", "frankenstein"),
("Dune Deluxe Edition Illustrated", "dune"),
("", ""),
(None, ""),
],
)
def test_titles_that_should_agree(self, title: str | None, expected: str) -> None:
assert normalize_title(title) == expected
def test_a_qualifier_that_is_the_title_survives(self) -> None:
"""A trailing qualifier is noise; the same word at the front is the book."""
assert normalize_title("The Illustrated Man") == "illustrated man"
def test_normalization_never_empties_a_title(self) -> None:
"""An article-only title is not improved by having no article left."""
assert normalize_title("The") == "the"
@pytest.mark.parametrize(
"title",
["Catch 22", "Fahrenheit 451", "Blade Runner 2049", "1984", "Apollo 13"],
)
def test_a_number_is_not_an_edition(self, title: str) -> None:
"""Edition stripping keys on the `e`; a bare number is part of the title."""
assert normalize_title(title) == title.casefold()
class TestNormalizeAuthor:
"""One person, written down several ways."""
@pytest.mark.parametrize(
("name", "expected"),
[
("Franz Kafka", "franz kafka"),
("Kafka, Franz", "franz kafka"),
("KAFKA, FRANZ", "franz kafka"),
("Émile Zola", "emile zola"),
("Doyle, Arthur Conan", "arthur conan doyle"),
# Runs of initials are joined, so spacing them out changes nothing.
("J.R.R. Tolkien", "jrr tolkien"),
("J. R. R. Tolkien", "jrr tolkien"),
("JRR Tolkien", "jrr tolkien"),
("", ""),
(None, ""),
],
)
def test_names_that_should_agree(self, name: str | None, expected: str) -> None:
assert normalize_author(name) == expected
def test_two_people_are_not_reduced_together(self) -> None:
"""Surname plus initial would collide unrelated writers; it is not used."""
assert normalize_author("Charles Dickens") != normalize_author("Colin Dexter")
def test_a_list_is_left_alone(self) -> None:
"""More than one comma is a list or a suffix, and guessing does more harm."""
assert normalize_author("Smith, John, Jr.") == "smith john jr"
class TestFormatAuthorName:
"""What gets stored and shown, as opposed to what gets compared."""
@pytest.mark.parametrize(
("written", "expected"),
[
# A leftover separator from a `DC:creator` list.
("Newman, Sam;", "Sam Newman"),
("Sam Newman ", "Sam Newman"),
(" Dan Vanderkam ", "Dan Vanderkam"),
# `Surname, Given` is how EPUBs file a name, not how anyone reads it.
("Kleppmann, Martin", "Martin Kleppmann"),
("Huxley, Aldous", "Aldous Huxley"),
("Liu, Cixin", "Cixin Liu"),
# An extension carried in from the filename the name was read out of.
("Sam Newman.epub", "Sam Newman"),
("Franz Kafka.mobi", "Franz Kafka"),
("Brian W. Kernighan.epub", "Brian W. Kernighan"),
("", ""),
(None, ""),
],
)
def test_names_are_tidied(self, written: str | None, expected: str) -> None:
assert format_author_name(written) == expected
@pytest.mark.parametrize(
"written",
[
# Two people in one string. Flipping it would invent a third.
"Dave Thomas, Andy Hunt",
"Mark Richards, Neal Ford",
# A compound surname is not recognised, and is left alone rather than
# rearranged wrongly.
"García Márquez, Gabriel",
],
)
def test_an_unrecognised_form_is_left_alone(self, written: str) -> None:
assert format_author_name(written) == written
def test_case_and_accents_belong_to_the_author(self) -> None:
"""Tidying removes what an extractor added; it does not correct spelling."""
assert format_author_name("Michał Płachta.epub") == "Michał Płachta"
assert format_author_name("Steve McConnell") == "Steve McConnell"
@pytest.mark.parametrize(
"written", ["Newman, Sam;", "Sam Newman.epub", "Kleppmann, Martin"]
)
def test_tidying_is_idempotent(self, written: str) -> None:
"""`unique_filter` tidies a name that may already be tidy; it must not drift."""
once = format_author_name(written)
assert format_author_name(once) == once
class TestNormalizeIdentifier:
"""Identifiers only help if the same edition produces the same key."""
def test_isbn_10_and_isbn_13_are_one_key(self) -> None:
assert normalize_identifier("isbn-10", "0486282112") == "isbn:9780486282114"
assert normalize_identifier("isbn-13", "9780486282114") == "isbn:9780486282114"
@pytest.mark.parametrize(
"written", ["978-0-486-28211-4", "978 0 486 28211 4", "9780486282114"]
)
def test_formatting_is_not_part_of_an_isbn(self, written: str) -> None:
assert normalize_identifier("isbn", written) == "isbn:9780486282114"
def test_an_isbn_that_fails_its_checksum_is_no_evidence(self) -> None:
assert normalize_identifier("isbn-13", "9780486282115") is None
def test_uuids_are_refused(self) -> None:
"""Generated per build, so they only re-find what the hash check catches."""
assert (
normalize_identifier("uuid", "3f2b1c4e-1111-2222-3333-444455556666") is None
)
assert (
normalize_identifier("urn:uuid", "3f2b1c4e-1111-2222-3333-444455556666")
is None
)
def test_other_schemes_keep_their_own_key(self) -> None:
assert normalize_identifier("asin", "B000FC0PDA") == "asin:b000fc0pda"
assert normalize_identifier("ASIN", "b000fc0pda") == "asin:b000fc0pda"
def test_something_too_short_is_not_evidence(self) -> None:
"""A Calibre id of "42" would otherwise pair two unrelated books."""
assert normalize_identifier("calibre", "42") is None
@pytest.mark.parametrize(("name", "value"), [("", "1234567"), ("asin", "")])
def test_half_an_identifier_is_no_identifier(self, name: str, value: str) -> None:
assert normalize_identifier(name, value) is None
class TestIsbnConversion:
def test_isbn_10_converts_to_its_isbn_13(self) -> None:
assert isbn10_to_isbn13("0486282112") == "9780486282114"
def test_a_trailing_x_is_a_digit(self) -> None:
assert isbn10_to_isbn13("043942089X") == "9780439420891"
@pytest.mark.parametrize("isbn", ["0486282113", "9780486282114", "nonsense"])
def test_anything_that_is_not_an_isbn_10_converts_to_nothing(
self, isbn: str
) -> None:
assert isbn10_to_isbn13(isbn) is None
class TestParseIdentifier:
"""What an EPUB writes, and what is worth storing for it."""
@pytest.mark.parametrize(
"written",
[
"9780486282114",
"978-0-486-28211-4",
"urn:isbn:9780486282114",
"urn:isbn:978-0-486-28211-4",
"ISBN:978-0-486-28211-4",
],
)
def test_isbns_survive_however_they_are_written(self, written: str) -> None:
assert parse_identifier(written) == ("isbn-13", "9780486282114")
def test_the_scheme_attribute_is_read_too(self) -> None:
assert parse_identifier("0-486-28211-2", "ISBN") == ("isbn-10", "0486282112")
def test_non_isbn_identifiers_are_kept(self) -> None:
assert parse_identifier("urn:uuid:3f2b1c4e-1111-2222-3333-444455556666") == (
"uuid",
"3f2b1c4e-1111-2222-3333-444455556666",
)
assert parse_identifier("calibre:1234") == ("calibre", "1234")
assert parse_identifier("B000FC0PDA", "mobi-asin") == ("asin", "B000FC0PDA")
def test_an_unrecognised_prefix_is_part_of_the_value(self) -> None:
""" "http://example.com/book" is not an identifier called "http"."""
assert parse_identifier("http://www.gutenberg.org/5200") == (
"id",
"http://www.gutenberg.org/5200",
)
def test_a_declared_isbn_that_is_not_one_is_dropped(self) -> None:
assert parse_identifier("urn:isbn:not-an-isbn") is None
@pytest.mark.parametrize("written", ["", " ", None])
def test_nothing_yields_nothing(self, written: str | None) -> None:
assert parse_identifier(written) is None