#!/usr/bin/env python3
"""
generate_epub.py
Converts the same category-coded .txt files used by generate_latex.py
(CODE-CCC.VV.txt, one occasion per file) directly into a single valid
EPUB3 file, suitable for upload to KDP as the Kindle edition.
This does NOT go through the LaTeX/PDF pipeline -- print-specific
formatting (drop caps, lettrine, fixed page breaks, hard \\ line breaks
mid-stanza) doesn't translate to Kindle's reflowable text, where the
reader controls font size and line spacing. Instead, this script parses
the .txt files directly (same parser shape as generate_latex.py) and
emits clean, semantic XHTML -- one file per category chapter, containing
all of that chapter's prayers -- with a proper EPUB3 nav document AND a
legacy NCX (older Kindle "Go To" menus still expect NCX; including both
costs nothing and avoids navigation complaints).
Within each prayer, stanza breaks (blank lines in the source) become
boundaries; single line breaks WITHIN a stanza become , so the
poem's line structure is preserved without relying on print-only
line-break macros.
Usage:
python generate_epub.py txt output.epub \\
--title "Prayers for Daily Life" \\
--subtitle "101 Devotional Prayers for Every Occasion" \\
--author "George Marin" \\
--publisher "Zechariah Press" \\
--cover cover.jpg \\
--uuid urn:uuid:xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx
If --uuid isn't given, a random one is generated each run -- fine for
testing, but pick ONE fixed UUID for the real book and pass it every
time (this is the file's permanent identity; changing it between
uploads can confuse KDP/retailers into treating updates as a new book).
"""
import argparse
import html
import re
import sys
import uuid
import zipfile
from pathlib import Path
from categories import CATEGORIES, VALID_CODES, category_name
# --- Parsing (same shape as generate_latex.py) ------------------------------
HEADER_RE = re.compile(
r"^(?P##\s*\S+)[ \t]*\n"
r">(?!>)[ \t]*(?P[^\n]*)\n"
r">>[ \t]*(?P[^\n]*)\n"
)
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P.+)$")
FILENAME_RE = re.compile(r"^(?P[A-Z]+)-(?P\d{3})\.(?P\d{2,3})(?:-[a-z])?\.txt$")
REF_RE = re.compile(r"^(?P\d+):(?P\d+)")
def strip_comment_lines(text: str) -> str:
lines = text.split("\n")
out = []
in_block = False
for ln in lines:
stripped = ln.strip()
if not in_block and stripped == "%v":
in_block = True
continue
if in_block:
if stripped == "%^":
in_block = False
continue
if ln.lstrip().startswith("%"):
continue
out.append(ln)
return "\n".join(out)
def parse_ref(ref_line: str) -> str:
return ref_line.lstrip("#").strip()
def ref_sort_key(ref: str):
m = REF_RE.match(ref)
if not m:
return (9999, 9999)
return (int(m.group("chapter")), int(m.group("verse")))
def parse_file(text: str):
text = strip_comment_lines(text)
stripped = text.lstrip("\n")
m = HEADER_RE.match(stripped)
if not m:
return None
ref = parse_ref(m.group("ref"))
title = m.group("title").strip()
quote = m.group("quote").strip()
body = text[m.end():]
matches = list(OCCASION_LINE_RE.finditer(body))
if len(matches) != 1:
return None
mo = matches[0]
occasion = mo.group("text").strip()
prayer_text = body[mo.end():].strip()
if not prayer_text:
return None
return {"ref": ref, "title": title, "quote": quote, "occasion": occasion, "text": prayer_text}
_MINOR_WORDS = {
"a", "an", "and", "as", "at", "but", "by", "en", "for", "if", "in",
"nor", "of", "on", "or", "per", "the", "to", "v", "via", "vs",
}
def _title_case_word(word: str) -> str:
i = 0
while i < len(word) and not word[i].isalpha():
i += 1
if i >= len(word):
return word
return word[:i] + word[i].upper() + word[i + 1:].lower()
def title_case(text: str) -> str:
words = text.split(" ")
out = []
for idx, w in enumerate(words):
bare = re.sub(r"[^a-zA-Z']", "", w).lower()
if 0 < idx < len(words) - 1 and bare in _MINOR_WORDS:
out.append(w.lower())
else:
out.append(_title_case_word(w))
return " ".join(out)
# --- HTML rendering ----------------------------------------------------------
def esc(text: str) -> str:
return html.escape(text, quote=False)
def convert_latex_typography(text: str) -> str:
"""Converts this project's LaTeX-only typographic conventions into
real Unicode characters, since none of them mean anything to an
EPUB reader: "---" (em dash ligature) -> em dash; ``...'' (curly
quote ligature) -> real curly quotes."""
text = text.replace("---", "\u2014")
text = text.replace("``", "\u201c").replace("''", "\u201d")
return text
def render_foreword_from_latex(tex_text: str) -> str:
"""Parses an ACTUAL template_forward.tex file (not a separate plain-text
copy -- single source of truth, same file that feeds the print
pipeline) into foreword HTML. Handles this project's specific
foreword conventions:
- \\chapter*{...} and \\addcontentsline{...} lines are dropped (the
EPUB foreword gets its own
here instead)
- \\medskip and \\bigskip mark paragraph breaks
- \\noindent is stripped
- wrapped lines within one paragraph are joined with spaces (the
LaTeX source hard-wraps for readability; those line breaks
aren't meaningful the way a prayer's stanza line breaks are)
- \\textit{...} becomes ...
- "---" / ``...'' typography converted via convert_latex_typography()
"""
text = tex_text
text = re.sub(r"\\chapter\*\{[^}]*\}", "", text)
text = re.sub(r"\\addcontentsline\{[^}]*\}\{[^}]*\}\{[^}]*\}", "", text)
# \medskip and \bigskip both mark a paragraph break
chunks = re.split(r"\\(?:med|big)skip", text)
paragraphs = []
for chunk in chunks:
chunk = chunk.replace("\\noindent", "")
# collapse hard-wrapped source lines into one flowing paragraph
joined = " ".join(ln.strip() for ln in chunk.split("\n") if ln.strip())
if not joined:
continue
joined = convert_latex_typography(joined)
# \textit{...} -> ..., done via sentinel markers so the
# em-tags themselves survive the HTML-escape step untouched
joined = re.sub(r"\\textit\{([^}]*)\}", lambda m: "\uE000" + m.group(1) + "\uE001", joined)
joined = esc(joined)
joined = joined.replace("\uE000", "").replace("\uE001", "")
paragraphs.append(f"
, in-stanza line break = ).
A dagger (†) follows the final line (always "Amen.") of every
prayer, matching the print edition's "Amen.\\enspace$\\dagger$"
convention.
See convert_latex_typography() -- the source files use LaTeX-only
ligature conventions for em dashes and curly quotes that need
converting to real Unicode characters outside the LaTeX pipeline."""
stanzas = [b.strip() for b in re.split(r"\n\s*\n", entry["text"].strip()) if b.strip()]
para_html = []
for i, stanza in enumerate(stanzas):
lines = [esc(convert_latex_typography(ln.strip())) for ln in stanza.split("\n") if ln.strip()]
joined = " \n".join(lines)
if i == len(stanzas) - 1:
joined += " †"
para_html.append(f"
{joined}
")
body = "\n".join(para_html)
title = esc(title_case(entry["occasion"]))
quote = esc(convert_latex_typography(entry["quote"]))
ref = esc(entry["ref"])
return f"""
{title}
{quote} (Ps. {ref})
{body}
"""
def render_chapter_xhtml(code: str, entries: list, title_prefix: str, flourish_img: str | None = None):
"""Returns (html, toc_items) where toc_items is a list of
(anchor_id, prayer_title) pairs -- used by the caller to build the
nested chapter->prayer TOC (nav.xhtml, toc.ncx, AND the visible
in-book toc.xhtml page all share this same structure).
flourish_img, if given, is the filename (already placed under
OEBPS/images/) of a decorative divider shown under the chapter
title -- same graphic as the print edition's chapter openers."""
name = esc(category_name(code))
sections = []
toc_items = []
for i, e in enumerate(entries):
anchor = f"{code.lower()}-{i+1:03d}"
sections.append(render_prayer_html(e, anchor))
toc_items.append((anchor, title_case(e["occasion"])))
body = "\n\n".join(sections)
flourish_html = ""
if flourish_img:
flourish_html = f'
'
html_out = f"""
{name}
{name}
{flourish_html}
{body}
"""
return html_out, toc_items
def render_simple_xhtml(title: str, body_html: str) -> str:
return f"""
{esc(title)}
{body_html}
"""
DEFAULT_CSS = """
body { font-family: serif; margin: 1em; line-height: 1.4; }
h1 { text-align: center; margin-top: 1.5em; margin-bottom: 1em; page-break-before: always; }
h2 { text-align: center; }
h3 { margin-top: 2em; page-break-before: always; }
.prayer { margin-bottom: 2em; }
.verse { font-style: italic; margin: 1em 2em; }
.verse .ref { font-style: normal; }
.titlepage { text-align: center; margin-top: 3em; }
.titlepage h1 { page-break-before: avoid; font-size: 1.8em; margin-bottom: 0.2em; }
.titlepage .subtitle { font-size: 1.1em; margin-bottom: 2em; }
.titlepage .author { font-size: 1.2em; margin-top: 2em; }
.copyright { font-size: 0.85em; }
.foreword p { margin-bottom: 1em; }
.center { text-align: center; }
/* Chapter-opening divider graphic, sized relative to text width so it
scales with the reader's font size / screen, not a fixed pixel size. */
p.flourish { text-align: center; margin: 0.5em 0 1.5em 0; }
p.flourish img { width: 30%; max-width: 220px; height: auto; }
"""
# --- EPUB packaging -----------------------------------------------------------
CONTAINER_XML = """
"""
def _manifest_item_line(iid: str, href: str, mtype: str, is_nav: bool) -> str:
# Kept as a plain function (not inlined in the f-string below) because
# a backslash-escaped quote inside an f-string's {} expression part is
# only legal on Python 3.12+ (PEP 701) -- this needs to run on older
# Python too.
nav_attr = ' properties="nav"' if is_nav else ""
return f' '
def build_opf(book_uuid: str, title: str, subtitle: str, author: str, publisher: str,
manifest_items: list, spine_ids: list, cover_id: str | None,
toc_page_href: str | None) -> str:
meta_cover = f'' if cover_id else ""
manifest = "\n".join(_manifest_item_line(iid, href, mtype, is_nav) for iid, href, mtype, is_nav in manifest_items)
spine = "\n".join(f' ' for sid in spine_ids)
full_title = f"{title}: {subtitle}" if subtitle else title
# is EPUB2-era, but Kindle's conversion pipeline and older
# reading apps still use it to find the table of contents -- costs
# nothing to include alongside the EPUB3 nav doc.
guide = ""
if toc_page_href:
guide = f"""
"""
return f"""
{esc(book_uuid)}{esc(full_title)}{esc(author)}{esc(publisher)}en
2026-01-01T00:00:00Z
{meta_cover}
{manifest}
{spine}
{guide}
"""
def build_nav(title: str, toc_entries: list) -> str:
"""toc_entries: list of dicts {href, label, children: [(href, label), ...]}."""
items = []
for entry in toc_entries:
li = f'
' for h, l in entry["children"])
li += f'\n \n{sub}\n \n '
li += ""
items.append(li)
items_str = "\n".join(items)
return f"""
{esc(title)}
"""
def render_psalm_index_xhtml(title: str, all_entries: list) -> str:
"""all_entries: list of (ref, href_with_anchor, prayer_title), already
sorted by ref_sort_key. Unlike the print index (ref -> page number,
meaningless in reflowable text with no fixed pages), each entry here
is a direct clickable link to that prayer -- arguably MORE useful
than the print version, since it's one tap instead of a page flip.
href_with_anchor is relative to OEBPS/text/ (where this file lives),
same convention as render_toc_page_xhtml."""
items = "\n".join(
f'
'
for ref, href, prayer_title in all_entries
)
body = f'
Index of Psalm References
\n{items}\n
'
return render_simple_xhtml(title, body)
def render_toc_page_xhtml(title: str, toc_entries: list) -> str:
"""A real, visible in-book Contents page (unlike nav.xhtml, which
most reading apps treat as sidebar-menu metadata, not a page you
actually read through). This is what was missing -- KDP/Kindle
conversion and many EPUB readers expect an actual TOC page in the
reading order, not just the EPUB3 nav doc.
IMPORTANT: this file itself lives at OEBPS/text/toc.xhtml, alongside
the chapter files -- so unlike nav.xhtml (which lives at the OEBPS/
root and needs "text/..." prefixed hrefs), links here must be
relative to the text/ directory, i.e. with that prefix stripped."""
def strip_prefix(href: str) -> str:
return href[len("text/"):] if href.startswith("text/") else href
items = []
for entry in toc_entries:
href = strip_prefix(entry["href"])
li = f'
' for h, l in entry["children"])
li += f'\n \n{sub}\n \n '
li += ""
items.append(li)
items_str = "\n".join(items)
body = f'
Contents
\n{items_str}\n
'
return render_simple_xhtml(title, body)
def build_ncx(book_uuid: str, title: str, toc_entries: list) -> str:
points = []
counter = [0]
def next_order():
counter[0] += 1
return counter[0]
for entry in toc_entries:
parent_order = next_order()
child_points = []
for h, l in entry["children"]:
child_order = next_order()
child_points.append(f""" {esc(l)}""")
child_str = "\n".join(child_points)
points.append(f""" {esc(entry['label'])}
{child_str}
""")
points_str = "\n".join(points)
return f"""
{esc(title)}
{points_str}
"""
def write_epub(out_path: Path, files: dict, image_paths: list):
"""files: dict of archive-path -> str content (text files) OR bytes.
image_paths: list of local Path objects to copy in under
OEBPS/images/ (cover, flourish, etc.) -- their manifest entries are
added by the caller; this just physically copies the bytes in.
mimetype MUST be first and MUST be stored (not deflated)."""
if out_path.exists():
out_path.unlink()
with zipfile.ZipFile(out_path, "w") as z:
z.writestr("mimetype", "application/epub+zip", compress_type=zipfile.ZIP_STORED)
for path, content in files.items():
data = content if isinstance(content, bytes) else content.encode("utf-8")
z.writestr(path, data, compress_type=zipfile.ZIP_DEFLATED)
for img_path in image_paths:
z.write(img_path, f"OEBPS/images/{img_path.name}", compress_type=zipfile.ZIP_DEFLATED)
# --- Main ---------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("txt_dir", type=Path)
ap.add_argument("output_epub", type=Path)
ap.add_argument("--title", default="Prayers for Daily Life")
ap.add_argument("--subtitle", default="101 Devotional Prayers for Every Occasion")
ap.add_argument("--author", default="George Marin")
ap.add_argument("--publisher", default="Zechariah Press")
ap.add_argument("--cover", type=Path, default=None, help="Cover image (jpg/png). Strongly recommended for KDP.")
ap.add_argument("--flourish", type=Path, default=None, help="Optional decorative divider image (jpg/png) shown under each chapter title -- same graphic as the print edition's chapter openers, if you want it.")
ap.add_argument("--foreword", type=Path, default=None, help="Your actual template_forward.tex file -- parsed directly (paragraph breaks on \\medskip/\\bigskip, \\textit{} to italics, etc.), not a separate plain-text copy.")
ap.add_argument("--review-url", default=None, help="If given, adds a back-matter review-request page with a clickable link to this URL (NOT a QR code -- pointless in an ebook, since the reader's already on the device).")
ap.add_argument("--no-psalm-index", action="store_true", help="Omit the back-matter Index of Psalm References (on by default). Unlike the print version, this is a page of CLICKABLE LINKS, not page numbers -- reflowable text has no fixed pages to index.")
ap.add_argument("--uuid", default=None, help="Fixed book UUID (e.g. urn:uuid:...). Random if omitted -- fine for testing, but pin one for the real release.")
ap.add_argument("--category-order", default=None, help="Comma-separated category codes, e.g. WON,WWF,CRI,...")
args = ap.parse_args()
# Fail loudly and immediately if an optional file path was actually
# GIVEN but doesn't resolve -- silently skipping it (the old
# behavior) is exactly the kind of mistake that ships a KDP EPUB
# with no embedded cover and nobody notices until much later.
for flag_name, path in [
("--cover", args.cover),
("--flourish", args.flourish),
("--foreword", args.foreword),
]:
if path is not None and not path.exists():
sys.exit(f"{flag_name} was given as {path!s}, but that file doesn't exist. Fix the path or drop the flag.")
book_uuid = args.uuid or f"urn:uuid:{uuid.uuid4()}"
order = args.category_order.split(",") if args.category_order else list(CATEGORIES.keys())
for c in order:
if c not in VALID_CODES:
sys.exit(f"Unknown category code in --category-order: {c}")
txt_files = sorted(args.txt_dir.glob("*.txt"))
by_category: dict[str, list] = {c: [] for c in order}
skipped = []
for path in txt_files:
m = FILENAME_RE.match(path.name)
if not m or m.group("code") not in VALID_CODES:
continue
parsed = parse_file(path.read_text(encoding="utf-8"))
if parsed is None:
skipped.append(path.name)
continue
code = m.group("code")
if code not in by_category:
by_category[code] = []
by_category[code].append(parsed)
if skipped:
print(f"Skipped {len(skipped)} file(s) that didn't parse:")
for s in skipped:
print(f" {s}")
for code in by_category:
by_category[code].sort(key=lambda e: ref_sort_key(e["ref"]))
total = sum(len(v) for v in by_category.values())
print(f"Building EPUB from {total} entries across {sum(1 for v in by_category.values() if v)} categories.")
# --- Assemble files ---
files: dict[str, object] = {}
files["META-INF/container.xml"] = CONTAINER_XML
files["OEBPS/css/style.css"] = DEFAULT_CSS
manifest_items = [] # (id, href, media-type, is_nav)
spine_ids = []
toc_entries = [] # list of {href, label, children: [(href#anchor, label), ...]}
# Title page
title_html = f"""
"""
files["OEBPS/text/copyright.xhtml"] = render_simple_xhtml("Copyright", copyright_html)
manifest_items.append(("copyright", "text/copyright.xhtml", "application/xhtml+xml", False))
spine_ids.append("copyright")
# Table of Contents -- a REAL visible page in the reading order, not just
# the EPUB3 nav.xhtml (which most reading apps only use for their own
# sidebar menu, not as an actual page). Placed here as a spine_ids
# placeholder; the id "toc-page" is inserted once its file is written
# below, once we know the full nested structure -- so reserve its
# position in the spine now, fill in content after the loop.
TOC_PAGE_POS = len(spine_ids)
spine_ids.append("toc-page") # placeholder, replaced below once content is built
# Foreword (optional) -- parsed directly from the real template_forward.tex
if args.foreword: # existence already validated above
fw_html = "
Foreword
\n" + render_foreword_from_latex(args.foreword.read_text(encoding="utf-8"))
files["OEBPS/text/foreword.xhtml"] = render_simple_xhtml("Foreword", f'
{fw_html}
')
manifest_items.append(("foreword", "text/foreword.xhtml", "application/xhtml+xml", False))
spine_ids.append("foreword")
toc_entries.append({"href": "text/foreword.xhtml", "label": "Foreword", "children": []})
# One XHTML file per category chapter, plus its nested prayer-level
# TOC entries (anchor links within that chapter's file)
for code in order:
entries = by_category.get(code, [])
if not entries:
continue
chapter_html, prayer_toc_items = render_chapter_xhtml(
code, entries, args.title,
flourish_img=args.flourish.name if args.flourish else None, # existence already validated above
)
fname = f"{code.lower()}.xhtml"
files[f"OEBPS/text/{fname}"] = chapter_html
cid = f"chap-{code.lower()}"
manifest_items.append((cid, f"text/{fname}", "application/xhtml+xml", False))
spine_ids.append(cid)
children = [(f"text/{fname}#{anchor}", label) for anchor, label in prayer_toc_items]
toc_entries.append({"href": f"text/{fname}", "label": category_name(code), "children": children})
# Review request (optional) -- a real hyperlink, not a QR code; see
# --review-url help text for why a QR code doesn't make sense here.
if args.review_url:
review_html = f"""
Did this book meet you somewhere?
If it did, a review helps the next reader find their way here too.
"""
files["OEBPS/text/review.xhtml"] = render_simple_xhtml("A Request", review_html)
manifest_items.append(("review", "text/review.xhtml", "application/xhtml+xml", False))
spine_ids.append("review")
# Index of Psalm References (optional, on by default) -- flat list
# across ALL categories, sorted by verse reference, each entry
# linking directly to its prayer. Built here (after every chapter's
# anchors are known) rather than during the per-category loop above,
# since it needs the complete cross-category picture.
if not args.no_psalm_index:
all_entries = []
for code in order:
for i, e in enumerate(by_category.get(code, [])):
anchor = f"{code.lower()}-{i+1:03d}"
href = f"{code.lower()}.xhtml#{anchor}"
all_entries.append((e["ref"], href, title_case(e["occasion"])))
all_entries.sort(key=lambda t: ref_sort_key(t[0]))
files["OEBPS/text/psalm-index.xhtml"] = render_psalm_index_xhtml("Index of Psalm References", all_entries)
manifest_items.append(("psalm-index", "text/psalm-index.xhtml", "application/xhtml+xml", False))
spine_ids.append("psalm-index")
toc_entries.append({"href": "text/psalm-index.xhtml", "label": "Index of Psalm References", "children": []})
# Now that toc_entries is fully built, write the actual TOC page and
# fill in the spine placeholder reserved above.
files["OEBPS/text/toc.xhtml"] = render_toc_page_xhtml("Contents", toc_entries)
manifest_items.append(("toc-page", "text/toc.xhtml", "application/xhtml+xml", False))
spine_ids[TOC_PAGE_POS] = "toc-page"
# Cover
cover_id = None
image_paths = [] # local files to physically copy into OEBPS/images/
if args.cover: # existence already validated above
image_paths.append(args.cover)
media_type = "image/jpeg" if args.cover.suffix.lower() in (".jpg", ".jpeg") else "image/png"
cover_id = "cover-image"
manifest_items.append((cover_id, f"images/{args.cover.name}", media_type, False))
cover_xhtml = f'
'
files["OEBPS/text/cover.xhtml"] = render_simple_xhtml("Cover", cover_xhtml)
manifest_items.append(("cover-page", "text/cover.xhtml", "application/xhtml+xml", False))
spine_ids.insert(0, "cover-page")
if args.flourish: # existence already validated above
image_paths.append(args.flourish)
media_type = "image/jpeg" if args.flourish.suffix.lower() in (".jpg", ".jpeg") else "image/png"
manifest_items.append(("flourish-image", f"images/{args.flourish.name}", media_type, False))
# Nav + NCX -- both reference the SAME nested toc_entries as the
# visible toc.xhtml page above, so the reader-UI sidebar and the
# in-book Contents page always agree.
manifest_items.append(("nav", "nav.xhtml", "application/xhtml+xml", True))
manifest_items.append(("ncx", "toc.ncx", "application/x-dtbncx+xml", False))
files["OEBPS/nav.xhtml"] = build_nav(args.title, toc_entries)
files["OEBPS/toc.ncx"] = build_ncx(book_uuid, args.title, toc_entries)
files["OEBPS/content.opf"] = build_opf(
book_uuid, args.title, args.subtitle, args.author, args.publisher,
manifest_items, spine_ids, cover_id, toc_page_href="text/toc.xhtml",
)
write_epub(args.output_epub, files, image_paths)
print(f"Wrote {args.output_epub} ({total} prayers, {sum(1 for v in by_category.values() if v)} chapters)")
if __name__ == "__main__":
main()