import pytest from epubmaker.models import BuildContext from epubmaker.parser import extract, identify, parse_book @pytest.fixture def ctx(): return BuildContext() def test_extract_returns_inner_content(): assert extract("") == "Title: Foo Bar" def test_extract_with_newline(): assert extract("\n") == "Title: Foo Bar" def test_extract_returns_none_for_non_comment(): assert extract("
Not a comment
") is None def test_identify_directory(ctx): identify("", ctx) assert ctx.book.directory == "mybook" assert ctx.in_section is False def test_identify_title_plain(ctx): identify("", ctx) assert ctx.book.title == "My Book" assert ctx.book.titlesort == "" def test_identify_title_the(ctx): identify("", ctx) assert ctx.book.title == "The Great Book" assert ctx.book.titlesort == "Great Book, The" def test_identify_title_a(ctx): identify("", ctx) assert ctx.book.title == "A Good Story" assert ctx.book.titlesort == "Good Story, A" def test_identify_title_an(ctx): identify("", ctx) assert ctx.book.title == "An Old Tale" assert ctx.book.titlesort == "Old Tale, An" def test_identify_author(ctx): identify("", ctx) assert len(ctx.book.author) == 1 assert ctx.book.author[0].last == "Smith" assert ctx.book.author[0].first == "John" def test_identify_multiple_authors(ctx): identify("", ctx) identify("", ctx) assert len(ctx.book.author) == 2 assert ctx.book.author[1].last == "Jones" def test_identify_editor(ctx): identify("", ctx) assert ctx.book.editor.last == "Brown" assert ctx.book.editor.first == "Alice" def test_identify_publisher(ctx): identify("", ctx) assert ctx.book.publisher == "Acme Press" def test_identify_copyright(ctx): identify("", ctx) assert ctx.book.copyright == "2024 Author" def test_identify_isbn(ctx): identify("", ctx) assert ctx.book.isbn == "978-0-000-00000-0" def test_identify_uuid_explicit(ctx): identify("", ctx) assert ctx.book.uuid == "abc-123" def test_identify_uuid_empty_generates_uuid(ctx): identify("", ctx) assert ctx.book.uuid != "" assert ctx.book.uuid is not None def test_identify_source(ctx): identify("", ctx) assert ctx.book.source == "Some Original Work" def test_identify_subjects(ctx): identify("", ctx) assert ctx.book.subjects == "Fiction, Mystery" def test_identify_series(ctx): identify("", ctx) assert ctx.book.series.number == "1" assert ctx.book.series.name == "My Series Name" def test_identify_cover(ctx): identify("", ctx) assert ctx.book.cover.name == "cover" assert ctx.book.cover.type == "jpeg" def test_identify_text_section(ctx): identify("", ctx) assert ctx.in_section is True assert len(ctx.book.sections) == 1 assert ctx.book.sections[0].type == "text" assert ctx.book.sections[0].name == "chapter1" assert ctx.book.sections[0].title == "Chapter One" def test_identify_toc_section(ctx): identify("", ctx) assert ctx.in_section is True assert ctx.book.sections[0].type == "toc" def test_identify_title_page(ctx): identify("", ctx) assert ctx.book.sections[0].type == "title" def test_identify_copyright_page(ctx): identify("", ctx) assert ctx.book.sections[0].type == "copyright-page" def test_identify_dedication(ctx): identify("", ctx) assert ctx.book.sections[0].type == "dedication" def test_identify_foreward(ctx): identify("", ctx) assert ctx.book.sections[0].type == "forward" def test_identify_notes(ctx): identify("", ctx) assert ctx.book.sections[0].type == "notes" def test_identify_acknowledgement(ctx): identify("", ctx) assert ctx.book.sections[0].type == "acknowledgement" def test_identify_metadata_clears_in_section(ctx): identify("", ctx) assert ctx.in_section is True identify("", ctx) assert ctx.in_section is False def test_identify_image_jpeg(ctx): identify("", ctx) identify("", ctx) assert len(ctx.book.images) == 1 assert ctx.book.images[0].name == "photo.jpg" assert ctx.book.images[0].type == "jpeg" assert ctx.book.images[0].alt == "Alt text" assert ctx.book.images[0].caption == "Caption text" def test_identify_image_png(ctx): identify("", ctx) identify("", ctx) assert ctx.book.images[0].type == "png" def test_identify_image_appends_to_section_body(ctx): identify("", ctx) identify("", ctx) assert "photo.jpg" in ctx.book.sections[-1].body def test_identify_image_with_caption_uses_div(ctx): identify("", ctx) identify("", ctx) assert 'Body content
\n" ) ctx = BuildContext() ctx.book.filename = str(source) parse_book(ctx) assert ctx.book.title == "Test Book" assert ctx.book.directory == "mybook" assert len(ctx.book.author) == 1 assert ctx.book.author[0].last == "Doe" assert len(ctx.book.sections) == 1 assert "Body content
\n" in ctx.book.sections[0].body def test_parse_book_body_not_captured_outside_section(tmp_path): source = tmp_path / "book.html" source.write_text( "\n" "This is outside any section
\n" "\n" "Inside section
\n" ) ctx = BuildContext() ctx.book.filename = str(source) parse_book(ctx) assert len(ctx.book.sections) == 1 assert "outside any section" not in ctx.book.sections[0].body assert "Inside section" in ctx.book.sections[0].body