kinda works

This commit is contained in:
Rory Hinnen
2026-05-31 11:03:34 -05:00
parent bfb496641b
commit 935908b786
2 changed files with 605 additions and 1 deletions
+2 -1
View File
@@ -1 +1,2 @@
.DS_Store
*.DS_Store
*__pycache__
+603
View File
@@ -0,0 +1,603 @@
import sys, os, argparse
from uuid import uuid4
from shutil import rmtree, copy
from datetime import datetime
import subprocess
class Name:
def __init__(self):
self.first = ""
self.last = ""
class Series:
def __init__(self):
self.number = ""
self.name = ""
class Image:
def __init__(self):
self.name = ""
self.type = ""
self.alt = ""
self.caption = ""
class Section:
def __init__(self):
self.type = ""
# types:
# cover
# toc
# copyright-page
# dedication
# text
# foreward
# notes
# acknowledments
# style
# title
self.name = ""
self.title = ""
self.body = ""
class Book:
def __init__(self):
self.filename = ""
self.title = ""
self.titlesort = ""
self.series = Series()
self.author = []
self.artist = Name()
self.editor = Name()
self.publisher = ""
self.copyright = ""
self.subjects = ""
self.style = ""
self.directory = ""
self.isbn = ""
self.uuid = ""
self.source = ""
self.images = []
self.sections = []
self.cover = Image()
self.publishDate = ""
def extract(line):
if line.startswith("<!-- "):
return line[5:-4]
def identify(line):
global book, inSection
key, value = extract(line).split(" ",1)
match key:
case "Directory:":
inSection = False
book.directory = value.strip()
case "Title:":
inSection = False
book.title = value.strip()
if book.title.startswith("A "):
book.titlesort = book.title[2:] + ", " + book.title[:1]
if book.title.startswith("An "):
book.titlesort = book.title[3:] + ", " + book.title[:2]
if book.title.startswith("The "):
book.titlesort = book.title[4:] + ", " + book.title[:3]
case "Titlesort":
inSection = False
book.titlesort = value.strip()
case "Author:":
inSection = False
last, first = value.strip().split(', ')
name = Name()
name.first = first
name.last = last
book.author.append(name)
case "Editor:":
inSection = False
book.editor.last, book.editor.first = value.strip().split(', ')
case "Artist:":
inSection = False
book.artist.last, book.artist.first = value.strip().split(', ')
case "Publisher:":
inSection = False
book.publisher = value.strip()
case "Copyright:":
inSection = False
book.copyright = value.strip()
case "ISBN:":
inSection = False
book.isbn = value.strip()
case "UUID:":
inSection = False
if value.strip() == "":
book.uuid = uuid4()
else:
book.uuid = value.strip()
case "Source:":
inSection = False
book.source = value.strip()
case "Subjects:":
inSection = False
book.subjects = value.strip()
case "Series:":
inSection = False
book.series.number, book.series.name = value.strip().split(' ', 1)
case "Cover:":
inSection = False
book.cover.name, book.cover.type = value.strip().split('.')
case "TitlePage:":
inSection = True
section = Section()
section.type = "title"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "CopyrightPage:":
inSection = True
section = Section()
section.type = "copyright-page"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Dedication:":
inSection = True
section = Section()
section.type = "dedication"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Text:":
inSection = True
section = Section()
section.type = "text"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Foreward:":
inSection = True
section = Section()
section.type = "forward"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Notes:":
inSection = True
section = Section()
section.type = "notes"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "License:":
inSection = True
section = Section()
section.type = "license"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Acknowledgement:":
inSection = True
section = Section()
section.type = "acknowledgement"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "TOC:":
inSection = True
section = Section()
section.type = "toc"
section.name, section.title = value.strip().split(' ', 1)
book.sections.append(section)
case "Image:":
image = Image()
image.name, imageText = value.strip().split(' ',1)
extension = image.name.strip().split('.')[1]
match extension:
case "jpg":
image.type = "jpeg"
case "jpeg":
image.type = "jpeg"
case "png":
image.type = "png"
image.alt, image.caption = imageText.strip().split('::')
book.images.append(image)
if not image.caption:
imageCode = '<img src="../Images/' + image.name +'" alt="' + image.alt + '"/>\n'
imageCode += ' <a id="' + image.name.split('.')[0] + '"><!-- Anchor --></a>\n'
else:
imageCode = '<div class="center group">\n'
imageCode += ' <img src="../Images/' + image.name + '" alt="' + image.alt + '" />\n'
imageCode += ' <a id="' + image.name.split('.')[0] + '"><!-- Anchor --></a>\n'
imageCode += ' <p class="caption">' + image.caption + '</p>\n'
imageCode += '</div>\n<br />\n'
book.sections[-1].body += imageCode
case _:
pass
def findStyle(line):
global book
stylesheetFile = line.split(' ')[2].split('=')[1].replace('"','').replace('>','').strip()
contents = ""
with open(stylesheetFile) as file:
for line in file:
if line != '<style>\n' and line != '</style>' and not line.strip().startswith("/*"):
contents += line
book.style = contents
print("found style")
def parseBook():
global book, inSection
filename = book.filename
if verbose:
print("Parsing {}".format(filename))
with open(filename) as source:
for line in source:
if line.startswith('<!-- '):
identify(line)
elif line.startswith('<link rel="stylesheet"'):
findStyle(line)
elif inSection:
book.sections[-1].body += line
if verbose:
print("Parse complete for {}".format(book.title))
def writeStructure(overwrite):
global verbose
mode = 0o775
if os.path.isdir(book.directory):
if overwrite:
rmtree(book.directory)
else:
print("writeStructure: Directory exists.")
exit()
os.mkdir(book.directory, mode)
for directory in ["META-INF", "OEBPS", "OEBPS/Images", "OEBPS/Styles", "OEBPS/Text"]:
path = os.path.join(book.directory, directory)
os.mkdir(path, mode)
if verbose:
print("Finished writing ePub structure")
def writeMisc():
global verbose
access = 'w'
directory = book.directory
file = os.path.join(book.directory, "mimetype")
mimetype = open(file, access)
mimetype.write("application/epub+zip")
mimetype.close()
file = os.path.join(directory, "META-INF", "com.apple.ibooks.display-options.xml")
meta1 = open(file, access)
meta1.write('<?xml version="1.0" encoding="UTF-8"?>\n<display_options>\n <platform name="*">\n <option name="specified-fonts">false</option>\n </platform>\n</display_options>')
meta1.close()
file = os.path.join(directory, "META-INF", "container.xml")
meta2 = open(file, access)
meta2.write('<?xml version="1.0" encoding="UTF-8"?>\n<container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">\n <rootfiles>\n <rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/>\n </rootfiles>\n</container>')
meta2.close()
if verbose:
print("Finished writing metadata")
def writeStyles():
access = 'w'
file = os.path.join(book.directory, "OEBPS", "Styles", "style.css")
style = open(file, access)
style.write(book.style)
style.close()
new = os.path.join(book.directory, "OEBPS", "Styles", "Inconsolata-Regular.otf")
old = os.path.join(os.path.split(book.filename)[0], "Images", "Inconsolata-Regular.otf")
copy(old, new)
new = os.path.join(book.directory, "OEBPS", "Styles", "invisible1.ttf")
old = os.path.join(os.path.split(book.filename)[0], "Images", "invisible1.ttf")
copy(old, new)
new = os.path.join(book.directory, "OEBPS", "Styles", "MgOpenModataRegular.ttf")
old = os.path.join(os.path.split(book.filename)[0], "Images", "MgOpenModataRegular.ttf")
copy(old, new)
if verbose:
print("Finished writing style")
def writeNCX():
directory = book.directory
access = 'w'
file = os.path.join(directory, 'OEBPS', 'toc.ncx')
navpoints = []
contentsstart = """<?xml version="1.0" encoding="UTF-8" standalone="no" ?>
<!DOCTYPE ncx PUBLIC "-//NISO//DTD ncx 2005-1//EN" "http://www.daisy.org/z3986/2005/ncx-2005-1.dtd">
<ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1">
<head>
<meta content="urn:isbn:""" + str(book.uuid) + """" name="dtb:uid"/>
<meta content="1" name="dtb:depth"/>
<meta content="-1" name="dtb:totalPageCount"/>
<meta content="-1" name="dtb:maxPageNumber"/>
</head>
<docTitle>
<text>""" + book.title.strip() + """</text>
</docTitle>
<navMap>
<navPoint id="cover" playOrder="1">
<navLabel>
<text>Cover</text>
</navLabel>
<content src="Text/cover.xhtml"/>
</navPoint>""" + '\n'
contentsend = """ </navMap>
</ncx>"""
number = 2
for section in book.sections:
if section.type != "toc":
navpoints.append("\t<navPoint id =\"" + section.name.strip() + "\" playOrder=\"" + str(number) + "\">\n" +
"\t\t<navLabel>\n\t\t\t<text>" + section.title.strip() + "</text>\n\t\t</navLabel>\n" +
"\t\t<content src=\"Text/" + section.name.strip() + ".xhtml\"/>\n\t</navPoint>\n")
number += 1
contents = contentsstart
for navpoint in navpoints:
contents += navpoint
contents += contentsend
ncx = open(file, access)
ncx.write(contents)
ncx.close()
if verbose:
print("Finished writing ncx")
def writeTOC(section):
access = 'w'
file = os.path.join(book.directory,'OEBPS', 'toc.xhtml')
contents = """<?xml version="1.0" encoding="utf-8"?>
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
<head>
<title>Table of Contents</title>
<link href="Styles/style.css" rel="stylesheet" type="text/css" />
</head>
<body>
"""
contents += section.body
contents += """
</body>
</html>"""
toc = open(file, access)
toc.write(contents)
toc.close()
if verbose:
print("Finished writing toc")
def writeCover():
access = 'w'
file = os.path.join(book.directory, 'OEBPS', 'Text', 'cover.xhtml')
contents = """<?xml version="1.0" encoding="utf-8" standalone="no"?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN"
"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">
<html xmlns="http://www.w3.org/1999/xhtml">
<head>
<title>Cover</title>
<style type="text/css">
/* <![CDATA[ */
img { max-width: 100%; }
body { oeb-column-number: 1; }
#cover-image { text-align: center; }
/* ]]> */
</style>
</head>
<body>
<div id="cover-image">
<img alt= """ + '"' + book.title.strip() + '"' + """ src="../Images/cover.jpeg" />
</div>
</body>
</html>
"""
new = os.path.join(book.directory, "OEBPS", "Images", "cover.jpeg")
old = os.path.join(os.path.split(book.filename)[0], "Images", "cover.jpeg")
cover = open(file, access)
cover.write(contents)
cover.close()
copy(old, new)
if verbose:
print("Finished writing cover")
def writeContents():
contents = '<?xml version="1.0" encoding="UTF-8"?>\n'
contents += '<package xmlns="http://www.idpf.org/2007/opf" version="3.0" xml:lang="en" unique-identifier="bookid" prefix="rendition: http://www.idpf.org/vocab/rendition/# cc: http://creativecommons.org/ns#">\n'
contents += ' <metadata xmlns:dc="http://purl.org/dc/elements/1.1/">\n'
contents += ' <dc:identifier id="bookid">urn:uuid' + book.isbn + '</dc:identifier>\n'
contents += ' <dc:title id="pub-title">' + book.title.strip() + '</dc:title>\n'
contents += ' <meta refines="#pub-title" property="title-type">main</meta>\n'
if book.titlesort != "":
contents += ' <meta refines="#pub-title" property="file-as">' + book.titlesort + '</meta>\n'
if book.series.name != "":
print("found a book series")
contents += ' <dc:title id="pub-title2">' + book.series.name.strip() + '</dc:title>\n'
contents += ' <meta refines="#pub-title2" property="title-type">collection</meta>\n'
contents += ' <dc:title id="pub-title3">' + book.series.number + '</dc:title>\n'
contents += ' <meta refines="#pub-title3" property="title-type">display-seq</meta>\n'
contents += ' <meta name="calibre:series" content="' + book.series.name.strip() + '" />\n'
contents += ' <meta name="calibre:series_index" content="' + book.series.number.strip() + '" />\n'
contents += ' <dc:language id="pub-language">en</dc:language>\n'
contents += ' <dc:date>' + book.publishDate + '</dc:date>\n'
for name in book.author:
contents += ' <dc:creator id="author">' + name.first.strip() + ' ' + name.last.strip() + '</dc:creator>\n'
contents += ' <meta refines="#author" scheme="marc:relators" property="role">aut</meta>\n'
for name in book.author:
contents += ' <meta refines="#author" property="file-as">' + name.last.strip() + ", " + name.first.strip() + '</meta>\n'
contents += ' <dc:publisher>' + book.publisher.strip() + '</dc:publisher>\n'
contents += ' <dc:rights>' + book.copyright.strip() + '</dc:rights>\n'
contents += ' <dc:subject>' + book.subjects.strip() + '</dc:subject>\n'
contents += ' <dc:source>' + book.source.strip() + '</dc:source>\n'
contents += ' <meta name="cover" content="cover.jpeg" />\n'
contents += """ </metadata>
<manifest>
<item href="Images/cover.jpeg" id="cover.jpeg" media-type="image/jpeg" properties="cover-image" />
<item href="Text/cover.xhtml" id="cover" media-type="application/xhtml+xml" />
<item href="toc.xhtml" id="toc" media-type="application/xhtml+xml" properties="nav"/>
<item href="toc.ncx" id="ncx" media-type="application/x-dtbncx+xml" />
<item href="Styles/style.css" id="css" media-type="text/css" />
<item href="Styles/MgOpenModataRegular.ttf" id="mgopenmodataregular" media-type="application/vnd.ms-opentype" />
<item href="Styles/invisible1.ttf" id="invisible1" media-type="application/vnd.ms-opentype" />
<item href="Styles/Inconsolata-Regular.otf" id="inconsolate" media-type="application/vnd.ms-opentype" />
"""
if len(book.images) > 0:
contents += ' <item href="Text/illustrations.xhtml" id="illustrations" media-type="application/xhtml+xml" />\n'
for section in book.sections:
if section.type != 'toc':
contents += " <item href=" + '"' + "Text/" + section.name + '.xhtml" id="' + section.name + '" media-type="application/xhtml+xml" />\n'
for image in book.images:
contents += " <item href=" + '"' + "Images/" + image.name + '" id="' + image.name.split('.')[0] + '" media-type="image/' + image.type + '" />\n'
new = os.path.join(book.directory, "OEBPS", "Images", image.name)
old = os.path.join(os.path.split(book.filename)[0], "Images", image.name)
copy(old, new)
contents += """ </manifest>
<spine toc="ncx">
<itemref idref="cover" />
"""
if len(book.images) > 0:
contents += ' <itemref idref="illustrations" />\n'
for section in book.sections:
contents += ' <itemref idref="' + section.name + '" />\n'
contents += """ </spine>
<guide>
<reference href="Text/cover.xhtml" title="Cover" type="cover" />
<reference href="toc.xhtml" title="Table of Contents" type="toc" />
"""
if len(book.images) > 0:
contents += ' <reference href="Text/illustrations.xhtml" title="Illustrations" type="text" />\n'
for section in book.sections:
if section.type != 'toc':
contents += ' <reference href="Text/' + section.name.strip() + '.xhtml" title="' + section.title.strip() + '" type="' + section.type.strip() + '" />\n'
contents += """ </guide>
</package>
"""
access = 'w'
file = os.path.join(book.directory, 'OEBPS', 'content.opf')
content = open(file, access)
content.write(contents)
content.close()
if verbose:
print("Finished writing content")
def writeSections():
access = 'w'
for section in book.sections:
if section.type.strip() == "toc":
writeTOC(section)
continue
contents = """<?xml version="1.0" encoding="utf-8" standalone="no"?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN"
"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
<head>
"""
contents += ' <title>' + section.title.strip() + '</title>\n <link href="../Styles/style.css" rel="stylesheet" type="text/css" />\n </head>\n <body>\n'
contents += ' <div id="' + section.name.strip() + '" xml:lang="en-US">\n'
contents += section.body
contents += " </div>\n </body>\n</html>\n"
file = os.path.join(book.directory, 'OEBPS', 'Text', section.name.strip() + ".xhtml")
content = open(file, access)
content.write(contents)
content.close()
if verbose:
print(f"Finished writing section {section.name}.xhtml")
def writeIllustrations():
access = 'w'
contents = """<?xml version="1.0" encoding="utf-8" standalone="no"?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.1//EN"
"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops">
<head>
<title>Illustrations</title>
<link href="../Styles/style.css" rel="stylesheet" type="text/css" />
</head>
<body>
<h1 class="title">Illustrations</h1>
<table>
"""
for image in book.images:
contents += ' <tr><td><a href="../Images/' + image.name + '">' + image.caption + '</a></td></tr>\n'
contents += ' </table>\n</body>\n</html>'
if len(book.images) > 0:
file = os.path.join(book.directory, 'OEBPS', 'Text', "illustrations.xhtml")
content = open(file, access)
content.write(contents)
content.close()
if verbose:
print(f"Finished writing section illustrations.xhtml")
def writeEpub():
cmd = f'cd {book.directory}; zip -X0 "{book.directory}.epub" mimetype;zip -X9Dr "{book.directory}.epub" META-INF OEBPS'
subprocess.run(cmd, shell=True)
if verbose:
print(f"Finished writing {book.directory}.epub")
def parseArgs():
global book, verbose
parser = argparse.ArgumentParser()
parser.add_argument("fileName", help="File to process")
parser.add_argument("-f", "--force", help="Force overwrite an existing book directory", action="store_true")
parser.add_argument("-v", "--verbose", help="Provide updates while processing", action="store_true")
args = parser.parse_args()
book.filename = str(args.fileName)
if not os.path.isfile(book.filename):
print("File '{}' is not found. Exiting...".format(book.filename))
exit()
verbose = args.verbose
return args.force
def main():
global book, inStyle, inSection, verbose
book = Book()
book.publishDate = str(datetime.now().year) + "-" + str(datetime.now().month) + "-" + str(datetime.now().day)
inSection = False
inStyle = False
force = False
verbose = False
overwrite = parseArgs()
parseBook()
if (overwrite and verbose):
print("Force flag, overwriting existing directory")
writeStructure(overwrite)
writeMisc()
writeStyles()
writeNCX()
writeCover()
writeContents()
writeSections()
writeIllustrations()
writeEpub()
global book, inSection, inStyle
main() # run the program
if __name__ == "__main__":
main()