diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..442bb94 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,3 @@ +[pytest] +pythonpath = src +testpaths = tests diff --git a/requirements.txt b/requirements.txt index e69de29..e079f8a 100644 --- a/requirements.txt +++ b/requirements.txt @@ -0,0 +1 @@ +pytest diff --git a/src/epubmaker/cli.py b/src/epubmaker/cli.py index 1ef61da..bcbafc1 100644 --- a/src/epubmaker/cli.py +++ b/src/epubmaker/cli.py @@ -1,50 +1,45 @@ import os import argparse from datetime import datetime -from . import state -from .models import Book -from .parser import parseBook +from .models import BuildContext +from .parser import parse_book from .writer import ( - writeStructure, writeMisc, writeStyles, writeNCX, - writeCover, writeContents, writeSections, writeIllustrations, writeEpub, + write_structure, write_misc, write_styles, write_ncx, + write_cover, write_contents, write_sections, write_illustrations, write_epub, ) -def parseArgs(): +def parse_args(ctx: BuildContext) -> bool: parser = argparse.ArgumentParser() parser.add_argument("fileName", help="File to process") parser.add_argument("-f", "--force", help="Force overwrite an existing book directory", action="store_true") parser.add_argument("-v", "--verbose", help="Provide updates while processing", action="store_true") args = parser.parse_args() - state.book.filename = str(args.fileName) - if not os.path.isfile(state.book.filename): - print("File '{}' is not found. Exiting...".format(state.book.filename)) + ctx.book.filename = str(args.fileName) + if not os.path.isfile(ctx.book.filename): + print("File '{}' is not found. Exiting...".format(ctx.book.filename)) exit() - state.verbose = args.verbose + ctx.verbose = args.verbose return args.force -def main(): - state.book = Book() - state.book.publishDate = ( - str(datetime.now().year) + "-" + - str(datetime.now().month) + "-" + - str(datetime.now().day) +def main() -> None: + ctx = BuildContext() + ctx.book.publish_date = "{}-{}-{}".format( + datetime.now().year, + datetime.now().month, + datetime.now().day, ) - state.inSection = False - state.inStyle = False - state.verbose = False - - overwrite = parseArgs() - parseBook() - if overwrite and state.verbose: + overwrite = parse_args(ctx) + parse_book(ctx) + if overwrite and ctx.verbose: print("Force flag, overwriting existing directory") - writeStructure(overwrite) - writeMisc() - writeStyles() - writeNCX() - writeCover() - writeContents() - writeSections() - writeIllustrations() - writeEpub() + write_structure(ctx, overwrite) + write_misc(ctx) + write_styles(ctx) + write_ncx(ctx) + write_cover(ctx) + write_contents(ctx) + write_sections(ctx) + write_illustrations(ctx) + write_epub(ctx) diff --git a/src/epubmaker/models.py b/src/epubmaker/models.py index 2d63a9e..c81a0c8 100644 --- a/src/epubmaker/models.py +++ b/src/epubmaker/models.py @@ -1,49 +1,60 @@ +from dataclasses import dataclass, field +from typing import List + + +@dataclass class Name: - def __init__(self): - self.first = "" - self.last = "" + first: str = "" + last: str = "" +@dataclass class Series: - def __init__(self): - self.number = "" - self.name = "" + number: str = "" + name: str = "" +@dataclass class Image: - def __init__(self): - self.name = "" - self.type = "" - self.alt = "" - self.caption = "" + name: str = "" + type: str = "" + alt: str = "" + caption: str = "" +@dataclass class Section: - def __init__(self): - self.type = "" - self.name = "" - self.title = "" - self.body = "" + type: str = "" + name: str = "" + title: str = "" + body: str = "" +@dataclass class Book: - def __init__(self): - self.filename = "" - self.title = "" - self.titlesort = "" - self.series = Series() - self.author = [] - self.artist = Name() - self.editor = Name() - self.publisher = "" - self.copyright = "" - self.subjects = "" - self.style = "" - self.directory = "" - self.isbn = "" - self.uuid = "" - self.source = "" - self.images = [] - self.sections = [] - self.cover = Image() - self.publishDate = "" + filename: str = "" + title: str = "" + titlesort: str = "" + series: Series = field(default_factory=Series) + author: List[Name] = field(default_factory=list) + artist: Name = field(default_factory=Name) + editor: Name = field(default_factory=Name) + publisher: str = "" + copyright: str = "" + subjects: str = "" + style: str = "" + directory: str = "" + isbn: str = "" + uuid: str = "" + source: str = "" + images: List[Image] = field(default_factory=list) + sections: List[Section] = field(default_factory=list) + cover: Image = field(default_factory=Image) + publish_date: str = "" + + +@dataclass +class BuildContext: + book: Book = field(default_factory=Book) + verbose: bool = False + in_section: bool = False diff --git a/src/epubmaker/parser.py b/src/epubmaker/parser.py index db0d227..815e182 100644 --- a/src/epubmaker/parser.py +++ b/src/epubmaker/parser.py @@ -1,174 +1,156 @@ from uuid import uuid4 -from . import state -from .models import Name, Image, Section +from .models import Name, Image, Section, BuildContext -def extract(line): +def extract(line: str) -> str | None: if line.startswith("\n' + image_code = '' + image.alt + '\n' + image_code += ' \n' else: - imageCode = '
\n' - imageCode += ' ' + image.alt + '\n' - imageCode += ' \n' - imageCode += '

' + image.caption + '

\n' - imageCode += '
\n
\n' - state.book.sections[-1].body += imageCode + image_code = '
\n' + image_code += ' ' + image.alt + '\n' + image_code += ' \n' + image_code += '

' + image.caption + "

\n" + image_code += "
\n
\n" + ctx.book.sections[-1].body += image_code case _: pass -def findStyle(line): - stylesheetFile = line.split(' ')[2].split('=')[1].replace('"', '').replace('>', '').strip() +def find_style(line: str, ctx: BuildContext) -> None: + stylesheet_file = line.split(" ")[2].split("=")[1].replace('"', "").replace(">", "").strip() contents = "" - with open(stylesheetFile) as file: - for line in file: - if line != '' and not line.strip().startswith("/*"): + with open(stylesheet_file) as f: + for line in f: + if line != "" and not line.strip().startswith("/*"): contents += line - state.book.style = contents + ctx.book.style = contents print("found style") -def parseBook(): - filename = state.book.filename - if state.verbose: - print("Parsing {}".format(filename)) - with open(filename) as source: +def parse_book(ctx: BuildContext) -> None: + if ctx.verbose: + print("Parsing {}".format(ctx.book.filename)) + with open(ctx.book.filename) as source: for line in source: - if line.startswith('") == "Title: Foo Bar" + + +def test_extract_with_newline(): + assert extract("\n") == "Title: Foo Bar" + + +def test_extract_returns_none_for_non_comment(): + assert extract("

Not a comment

") is None + + +def test_identify_directory(ctx): + identify("", ctx) + assert ctx.book.directory == "mybook" + assert ctx.in_section is False + + +def test_identify_title_plain(ctx): + identify("", ctx) + assert ctx.book.title == "My Book" + assert ctx.book.titlesort == "" + + +def test_identify_title_the(ctx): + identify("", ctx) + assert ctx.book.title == "The Great Book" + assert ctx.book.titlesort == "Great Book, The" + + +def test_identify_title_a(ctx): + identify("", ctx) + assert ctx.book.title == "A Good Story" + assert ctx.book.titlesort == "Good Story, A" + + +def test_identify_title_an(ctx): + identify("", ctx) + assert ctx.book.title == "An Old Tale" + assert ctx.book.titlesort == "Old Tale, An" + + +def test_identify_author(ctx): + identify("", ctx) + assert len(ctx.book.author) == 1 + assert ctx.book.author[0].last == "Smith" + assert ctx.book.author[0].first == "John" + + +def test_identify_multiple_authors(ctx): + identify("", ctx) + identify("", ctx) + assert len(ctx.book.author) == 2 + assert ctx.book.author[1].last == "Jones" + + +def test_identify_editor(ctx): + identify("", ctx) + assert ctx.book.editor.last == "Brown" + assert ctx.book.editor.first == "Alice" + + +def test_identify_publisher(ctx): + identify("", ctx) + assert ctx.book.publisher == "Acme Press" + + +def test_identify_copyright(ctx): + identify("", ctx) + assert ctx.book.copyright == "2024 Author" + + +def test_identify_isbn(ctx): + identify("", ctx) + assert ctx.book.isbn == "978-0-000-00000-0" + + +def test_identify_uuid_explicit(ctx): + identify("", ctx) + assert ctx.book.uuid == "abc-123" + + +def test_identify_uuid_empty_generates_uuid(ctx): + identify("", ctx) + assert ctx.book.uuid != "" + assert ctx.book.uuid is not None + + +def test_identify_source(ctx): + identify("", ctx) + assert ctx.book.source == "Some Original Work" + + +def test_identify_subjects(ctx): + identify("", ctx) + assert ctx.book.subjects == "Fiction, Mystery" + + +def test_identify_series(ctx): + identify("", ctx) + assert ctx.book.series.number == "1" + assert ctx.book.series.name == "My Series Name" + + +def test_identify_cover(ctx): + identify("", ctx) + assert ctx.book.cover.name == "cover" + assert ctx.book.cover.type == "jpeg" + + +def test_identify_text_section(ctx): + identify("", ctx) + assert ctx.in_section is True + assert len(ctx.book.sections) == 1 + assert ctx.book.sections[0].type == "text" + assert ctx.book.sections[0].name == "chapter1" + assert ctx.book.sections[0].title == "Chapter One" + + +def test_identify_toc_section(ctx): + identify("", ctx) + assert ctx.in_section is True + assert ctx.book.sections[0].type == "toc" + + +def test_identify_title_page(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "title" + + +def test_identify_copyright_page(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "copyright-page" + + +def test_identify_dedication(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "dedication" + + +def test_identify_foreward(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "forward" + + +def test_identify_notes(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "notes" + + +def test_identify_acknowledgement(ctx): + identify("", ctx) + assert ctx.book.sections[0].type == "acknowledgement" + + +def test_identify_metadata_clears_in_section(ctx): + identify("", ctx) + assert ctx.in_section is True + identify("", ctx) + assert ctx.in_section is False + + +def test_identify_image_jpeg(ctx): + identify("", ctx) + identify("", ctx) + assert len(ctx.book.images) == 1 + assert ctx.book.images[0].name == "photo.jpg" + assert ctx.book.images[0].type == "jpeg" + assert ctx.book.images[0].alt == "Alt text" + assert ctx.book.images[0].caption == "Caption text" + + +def test_identify_image_png(ctx): + identify("", ctx) + identify("", ctx) + assert ctx.book.images[0].type == "png" + + +def test_identify_image_appends_to_section_body(ctx): + identify("", ctx) + identify("", ctx) + assert "photo.jpg" in ctx.book.sections[-1].body + + +def test_identify_image_with_caption_uses_div(ctx): + identify("", ctx) + identify("", ctx) + assert '
' in ctx.book.sections[-1].body + assert "My Caption" in ctx.book.sections[-1].body + + +def test_identify_image_without_caption_uses_img(ctx): + identify("", ctx) + identify("", ctx) + body = ctx.book.sections[-1].body + assert "' not in body + + +def test_identify_unknown_key_ignored(ctx): + identify("", ctx) + assert ctx.book.title == "" + assert ctx.book.directory == "" + + +def test_parse_book_populates_book(tmp_path): + source = tmp_path / "book.html" + source.write_text( + "\n" + "\n" + "\n" + "\n" + "

Body content

\n" + ) + ctx = BuildContext() + ctx.book.filename = str(source) + parse_book(ctx) + assert ctx.book.title == "Test Book" + assert ctx.book.directory == "mybook" + assert len(ctx.book.author) == 1 + assert ctx.book.author[0].last == "Doe" + assert len(ctx.book.sections) == 1 + assert "

Body content

\n" in ctx.book.sections[0].body + + +def test_parse_book_body_not_captured_outside_section(tmp_path): + source = tmp_path / "book.html" + source.write_text( + "\n" + "

This is outside any section

\n" + "\n" + "

Inside section

\n" + ) + ctx = BuildContext() + ctx.book.filename = str(source) + parse_book(ctx) + assert len(ctx.book.sections) == 1 + assert "outside any section" not in ctx.book.sections[0].body + assert "Inside section" in ctx.book.sections[0].body diff --git a/tests/test_writer.py b/tests/test_writer.py new file mode 100644 index 0000000..2afb8eb --- /dev/null +++ b/tests/test_writer.py @@ -0,0 +1,243 @@ +import os +import pytest +from unittest.mock import patch +from epubmaker.models import BuildContext, Book, Section, Image, Name + + +@pytest.fixture +def ctx(tmp_path): + context = BuildContext() + context.book.title = "Test Book" + context.book.uuid = "test-uuid-1234" + context.book.isbn = "978-0-000-00000-0" + context.book.publisher = "Test Publisher" + context.book.copyright = "2024 Test" + context.book.subjects = "Fiction" + context.book.source = "Original Source" + context.book.publish_date = "2024-1-1" + context.book.directory = str(tmp_path / "testbook") + context.book.filename = str(tmp_path / "source" / "book.html") + context.book.author.append(Name(first="Jane", last="Doe")) + return context + + +from epubmaker.writer import ( + write_structure, write_misc, write_ncx, write_toc, + write_cover, write_contents, write_sections, write_illustrations, +) + + +# --- write_structure --- + +def test_write_structure_creates_epub_directories(ctx): + write_structure(ctx, overwrite=False) + for subdir in ["META-INF", "OEBPS", "OEBPS/Images", "OEBPS/Styles", "OEBPS/Text"]: + assert os.path.isdir(os.path.join(ctx.book.directory, subdir)) + + +def test_write_structure_exits_if_exists(ctx): + os.makedirs(ctx.book.directory) + with pytest.raises(SystemExit): + write_structure(ctx, overwrite=False) + + +def test_write_structure_overwrite_replaces_existing(ctx): + os.makedirs(ctx.book.directory) + sentinel = os.path.join(ctx.book.directory, "old_file.txt") + open(sentinel, "w").close() + write_structure(ctx, overwrite=True) + assert os.path.isdir(ctx.book.directory) + assert not os.path.exists(sentinel) + + +# --- write_misc --- + +def test_write_misc_mimetype_content(ctx): + write_structure(ctx, overwrite=False) + write_misc(ctx) + with open(os.path.join(ctx.book.directory, "mimetype")) as f: + assert f.read() == "application/epub+zip" + + +def test_write_misc_container_xml_references_opf(ctx): + write_structure(ctx, overwrite=False) + write_misc(ctx) + with open(os.path.join(ctx.book.directory, "META-INF", "container.xml")) as f: + assert 'full-path="OEBPS/content.opf"' in f.read() + + +def test_write_misc_ibooks_display_options_created(ctx): + write_structure(ctx, overwrite=False) + write_misc(ctx) + path = os.path.join(ctx.book.directory, "META-INF", "com.apple.ibooks.display-options.xml") + assert os.path.isfile(path) + + +# --- write_ncx --- + +def test_write_ncx_contains_title_and_uuid(ctx): + write_structure(ctx, overwrite=False) + write_ncx(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "toc.ncx")) as f: + content = f.read() + assert ctx.book.title in content + assert ctx.book.uuid in content + + +def test_write_ncx_includes_non_toc_sections(ctx): + ctx.book.sections.append(Section(type="text", name="chapter1", title="Chapter One")) + write_structure(ctx, overwrite=False) + write_ncx(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "toc.ncx")) as f: + content = f.read() + assert "chapter1" in content + assert "Chapter One" in content + + +def test_write_ncx_excludes_toc_type_sections(ctx): + ctx.book.sections.append(Section(type="toc", name="toc", title="Table of Contents")) + write_structure(ctx, overwrite=False) + write_ncx(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "toc.ncx")) as f: + content = f.read() + assert 'id ="toc"' not in content + + +def test_write_ncx_play_order_increments(ctx): + for i in range(3): + ctx.book.sections.append(Section(type="text", name=f"ch{i}", title=f"Chapter {i}")) + write_structure(ctx, overwrite=False) + write_ncx(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "toc.ncx")) as f: + content = f.read() + assert 'playOrder="2"' in content + assert 'playOrder="3"' in content + assert 'playOrder="4"' in content + + +# --- write_toc --- + +def test_write_toc_writes_section_body(ctx): + section = Section(body="

Contents here

\n") + write_structure(ctx, overwrite=False) + write_toc(ctx, section) + with open(os.path.join(ctx.book.directory, "OEBPS", "toc.xhtml")) as f: + content = f.read() + assert "

Contents here

" in content + assert "Table of Contents" in content + + +# --- write_cover --- + +def test_write_cover_creates_xhtml_with_title(ctx, tmp_path): + source_images = tmp_path / "source" / "Images" + source_images.mkdir(parents=True) + (source_images / "cover.jpeg").write_bytes(b"fake") + write_structure(ctx, overwrite=False) + write_cover(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "Text", "cover.xhtml")) as f: + content = f.read() + assert ctx.book.title in content + assert 'src="../Images/cover.jpeg"' in content + + +def test_write_cover_copies_image_to_output(ctx, tmp_path): + source_images = tmp_path / "source" / "Images" + source_images.mkdir(parents=True) + (source_images / "cover.jpeg").write_bytes(b"image-data") + write_structure(ctx, overwrite=False) + write_cover(ctx) + assert os.path.isfile(os.path.join(ctx.book.directory, "OEBPS", "Images", "cover.jpeg")) + + +# --- write_sections --- + +def test_write_sections_creates_xhtml_per_section(ctx): + ctx.book.sections.append(Section(type="text", name="chapter1", title="Chapter One", body="

Content

\n")) + write_structure(ctx, overwrite=False) + write_sections(ctx) + path = os.path.join(ctx.book.directory, "OEBPS", "Text", "chapter1.xhtml") + assert os.path.isfile(path) + with open(path) as f: + content = f.read() + assert "Chapter One" in content + assert "

Content

" in content + + +def test_write_sections_routes_toc_to_toc_xhtml(ctx): + ctx.book.sections.append(Section(type="toc", name="toc", title="Table of Contents", body="

toc

\n")) + write_structure(ctx, overwrite=False) + write_sections(ctx) + assert os.path.isfile(os.path.join(ctx.book.directory, "OEBPS", "toc.xhtml")) + assert not os.path.isfile(os.path.join(ctx.book.directory, "OEBPS", "Text", "toc.xhtml")) + + +def test_write_sections_multiple_sections(ctx): + for i in range(3): + ctx.book.sections.append(Section(type="text", name=f"ch{i}", title=f"Chapter {i}", body=f"

ch{i}

\n")) + write_structure(ctx, overwrite=False) + write_sections(ctx) + for i in range(3): + assert os.path.isfile(os.path.join(ctx.book.directory, "OEBPS", "Text", f"ch{i}.xhtml")) + + +# --- write_illustrations --- + +def test_write_illustrations_skips_when_no_images(ctx): + write_structure(ctx, overwrite=False) + write_illustrations(ctx) + assert not os.path.isfile(os.path.join(ctx.book.directory, "OEBPS", "Text", "illustrations.xhtml")) + + +def test_write_illustrations_creates_file_with_image_links(ctx): + ctx.book.images.append(Image(name="photo.jpg", caption="A scenic photo")) + write_structure(ctx, overwrite=False) + write_illustrations(ctx) + path = os.path.join(ctx.book.directory, "OEBPS", "Text", "illustrations.xhtml") + assert os.path.isfile(path) + with open(path) as f: + content = f.read() + assert "photo.jpg" in content + assert "A scenic photo" in content + + +# --- write_contents --- + +def test_write_contents_creates_opf_with_metadata(ctx): + write_structure(ctx, overwrite=False) + write_contents(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "content.opf")) as f: + content = f.read() + assert ctx.book.title in content + assert ctx.book.publisher in content + assert ctx.book.isbn in content + + +def test_write_contents_includes_section_in_manifest(ctx): + ctx.book.sections.append(Section(type="text", name="chapter1", title="Chapter One")) + write_structure(ctx, overwrite=False) + write_contents(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "content.opf")) as f: + content = f.read() + assert "chapter1.xhtml" in content + + +def test_write_contents_includes_series_metadata(ctx): + ctx.book.series.name = "My Series" + ctx.book.series.number = "1" + write_structure(ctx, overwrite=False) + write_contents(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "content.opf")) as f: + content = f.read() + assert "My Series" in content + assert "calibre:series" in content + + +def test_write_contents_includes_illustrations_when_images_present(ctx): + ctx.book.images.append(Image(name="photo.jpg", type="jpeg", caption="caption")) + write_structure(ctx, overwrite=False) + with patch("epubmaker.writer.copy"): + write_contents(ctx) + with open(os.path.join(ctx.book.directory, "OEBPS", "content.opf")) as f: + content = f.read() + assert "illustrations.xhtml" in content