#!/usr/bin/env python3 """ generate_epub.py Converts the same category-coded .txt files used by generate_latex.py (CODE-CCC.VV.txt, one occasion per file) directly into a single valid EPUB3 file, suitable for upload to KDP as the Kindle edition. This does NOT go through the LaTeX/PDF pipeline -- print-specific formatting (drop caps, lettrine, fixed page breaks, hard \\ line breaks mid-stanza) doesn't translate to Kindle's reflowable text, where the reader controls font size and line spacing. Instead, this script parses the .txt files directly (same parser shape as generate_latex.py) and emits clean, semantic XHTML -- one file per category chapter, containing all of that chapter's prayers -- with a proper EPUB3 nav document AND a legacy NCX (older Kindle "Go To" menus still expect NCX; including both costs nothing and avoids navigation complaints). Within each prayer, stanza breaks (blank lines in the source) become

boundaries; single line breaks WITHIN a stanza become
, so the poem's line structure is preserved without relying on print-only line-break macros. Usage: python generate_epub.py txt output.epub \\ --title "Prayers for Daily Life" \\ --subtitle "101 Devotional Prayers for Every Occasion" \\ --author "George Marin" \\ --publisher "Zechariah Press" \\ --cover cover.jpg \\ --uuid urn:uuid:xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx If --uuid isn't given, a random one is generated each run -- fine for testing, but pick ONE fixed UUID for the real book and pass it every time (this is the file's permanent identity; changing it between uploads can confuse KDP/retailers into treating updates as a new book). """ import argparse import html import re import sys import uuid import zipfile from pathlib import Path from categories import CATEGORIES, VALID_CODES, category_name # --- Parsing (same shape as generate_latex.py) ------------------------------ HEADER_RE = re.compile( r"^(?P##\s*\S+)[ \t]*\n" r">(?!>)[ \t]*(?P[^\n]*)\n" r">>[ \t]*(?P<quote>[^\n]*)\n" ) OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$") FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<chapter>\d{3})\.(?P<verse>\d{2,3})(?:-[a-z])?\.txt$") REF_RE = re.compile(r"^(?P<chapter>\d+):(?P<verse>\d+)") def strip_comment_lines(text: str) -> str: lines = text.split("\n") out = [] in_block = False for ln in lines: stripped = ln.strip() if not in_block and stripped == "%v": in_block = True continue if in_block: if stripped == "%^": in_block = False continue if ln.lstrip().startswith("%"): continue out.append(ln) return "\n".join(out) def parse_ref(ref_line: str) -> str: return ref_line.lstrip("#").strip() def ref_sort_key(ref: str): m = REF_RE.match(ref) if not m: return (9999, 9999) return (int(m.group("chapter")), int(m.group("verse"))) def parse_file(text: str): text = strip_comment_lines(text) stripped = text.lstrip("\n") m = HEADER_RE.match(stripped) if not m: return None ref = parse_ref(m.group("ref")) title = m.group("title").strip() quote = m.group("quote").strip() body = text[m.end():] matches = list(OCCASION_LINE_RE.finditer(body)) if len(matches) != 1: return None mo = matches[0] occasion = mo.group("text").strip() prayer_text = body[mo.end():].strip() if not prayer_text: return None return {"ref": ref, "title": title, "quote": quote, "occasion": occasion, "text": prayer_text} _MINOR_WORDS = { "a", "an", "and", "as", "at", "but", "by", "en", "for", "if", "in", "nor", "of", "on", "or", "per", "the", "to", "v", "via", "vs", } def _title_case_word(word: str) -> str: i = 0 while i < len(word) and not word[i].isalpha(): i += 1 if i >= len(word): return word return word[:i] + word[i].upper() + word[i + 1:].lower() def title_case(text: str) -> str: words = text.split(" ") out = [] for idx, w in enumerate(words): bare = re.sub(r"[^a-zA-Z']", "", w).lower() if 0 < idx < len(words) - 1 and bare in _MINOR_WORDS: out.append(w.lower()) else: out.append(_title_case_word(w)) return " ".join(out) # --- HTML rendering ---------------------------------------------------------- def esc(text: str) -> str: return html.escape(text, quote=False) def convert_latex_typography(text: str) -> str: """Converts this project's LaTeX-only typographic conventions into real Unicode characters, since none of them mean anything to an EPUB reader: "---" (em dash ligature) -> em dash; ``...'' (curly quote ligature) -> real curly quotes.""" text = text.replace("---", "\u2014") text = text.replace("``", "\u201c").replace("''", "\u201d") return text def render_foreword_from_latex(tex_text: str) -> str: """Parses an ACTUAL template_forward.tex file (not a separate plain-text copy -- single source of truth, same file that feeds the print pipeline) into foreword HTML. Handles this project's specific foreword conventions: - \\chapter*{...} and \\addcontentsline{...} lines are dropped (the EPUB foreword gets its own <h1> here instead) - \\medskip and \\bigskip mark paragraph breaks - \\noindent is stripped - wrapped lines within one paragraph are joined with spaces (the LaTeX source hard-wraps for readability; those line breaks aren't meaningful the way a prayer's stanza line breaks are) - \\textit{...} becomes <em>...</em> - "---" / ``...'' typography converted via convert_latex_typography() """ text = tex_text text = re.sub(r"\\chapter\*\{[^}]*\}", "", text) text = re.sub(r"\\addcontentsline\{[^}]*\}\{[^}]*\}\{[^}]*\}", "", text) # \medskip and \bigskip both mark a paragraph break chunks = re.split(r"\\(?:med|big)skip", text) paragraphs = [] for chunk in chunks: chunk = chunk.replace("\\noindent", "") # collapse hard-wrapped source lines into one flowing paragraph joined = " ".join(ln.strip() for ln in chunk.split("\n") if ln.strip()) if not joined: continue joined = convert_latex_typography(joined) # \textit{...} -> <em>...</em>, done via sentinel markers so the # em-tags themselves survive the HTML-escape step untouched joined = re.sub(r"\\textit\{([^}]*)\}", lambda m: "\uE000" + m.group(1) + "\uE001", joined) joined = esc(joined) joined = joined.replace("\uE000", "<em>").replace("\uE001", "</em>") paragraphs.append(f"<p>{joined}</p>") return "\n".join(paragraphs) def render_prayer_html(entry: dict, anchor_id: str) -> str: """One prayer as semantic XHTML: heading, verse blockquote, body paragraphs (stanza = <p>, in-stanza line break = <br/>). A dagger (†) follows the final line (always "Amen.") of every prayer, matching the print edition's "Amen.\\enspace$\\dagger$" convention. See convert_latex_typography() -- the source files use LaTeX-only ligature conventions for em dashes and curly quotes that need converting to real Unicode characters outside the LaTeX pipeline.""" stanzas = [b.strip() for b in re.split(r"\n\s*\n", entry["text"].strip()) if b.strip()] para_html = [] for i, stanza in enumerate(stanzas): lines = [esc(convert_latex_typography(ln.strip())) for ln in stanza.split("\n") if ln.strip()] joined = "<br/>\n".join(lines) if i == len(stanzas) - 1: joined += " †" para_html.append(f"<p>{joined}</p>") body = "\n".join(para_html) title = esc(title_case(entry["occasion"])) quote = esc(convert_latex_typography(entry["quote"])) ref = esc(entry["ref"]) return f"""<section class="prayer" id="{anchor_id}"> <h3>{title}</h3> <blockquote class="verse"><p>{quote} <span class="ref">(Ps. {ref})</span></p></blockquote> {body} </section>""" def render_chapter_xhtml(code: str, entries: list, title_prefix: str, flourish_img: str | None = None): """Returns (html, toc_items) where toc_items is a list of (anchor_id, prayer_title) pairs -- used by the caller to build the nested chapter->prayer TOC (nav.xhtml, toc.ncx, AND the visible in-book toc.xhtml page all share this same structure). flourish_img, if given, is the filename (already placed under OEBPS/images/) of a decorative divider shown under the chapter title -- same graphic as the print edition's chapter openers.""" name = esc(category_name(code)) sections = [] toc_items = [] for i, e in enumerate(entries): anchor = f"{code.lower()}-{i+1:03d}" sections.append(render_prayer_html(e, anchor)) toc_items.append((anchor, title_case(e["occasion"]))) body = "\n\n".join(sections) flourish_html = "" if flourish_img: flourish_html = f'<p class="flourish"><img src="../images/{esc(flourish_img)}" alt="" role="presentation"/></p>' html_out = f"""<?xml version="1.0" encoding="utf-8"?> <!DOCTYPE html> <html xmlns="http://www.w3.org/1999/xhtml" xmlns:epub="http://www.idpf.org/2007/ops" lang="en"> <head> <title>{name}

{name}

{flourish_html} {body} """ return html_out, toc_items def render_simple_xhtml(title: str, body_html: str) -> str: return f""" {esc(title)} {body_html} """ DEFAULT_CSS = """ body { font-family: serif; margin: 1em; line-height: 1.4; } h1 { text-align: center; margin-top: 1.5em; margin-bottom: 1em; page-break-before: always; } h2 { text-align: center; } h3 { margin-top: 2em; page-break-before: always; } .prayer { margin-bottom: 2em; } .verse { font-style: italic; margin: 1em 2em; } .verse .ref { font-style: normal; } .titlepage { text-align: center; margin-top: 3em; } .titlepage h1 { page-break-before: avoid; font-size: 1.8em; margin-bottom: 0.2em; } .titlepage .subtitle { font-size: 1.1em; margin-bottom: 2em; } .titlepage .author { font-size: 1.2em; margin-top: 2em; } .copyright { font-size: 0.85em; } .foreword p { margin-bottom: 1em; } .center { text-align: center; } /* Chapter-opening divider graphic, sized relative to text width so it scales with the reader's font size / screen, not a fixed pixel size. */ p.flourish { text-align: center; margin: 0.5em 0 1.5em 0; } p.flourish img { width: 30%; max-width: 220px; height: auto; } """ # --- EPUB packaging ----------------------------------------------------------- CONTAINER_XML = """ """ def _manifest_item_line(iid: str, href: str, mtype: str, is_nav: bool) -> str: # Kept as a plain function (not inlined in the f-string below) because # a backslash-escaped quote inside an f-string's {} expression part is # only legal on Python 3.12+ (PEP 701) -- this needs to run on older # Python too. nav_attr = ' properties="nav"' if is_nav else "" return f' ' def build_opf(book_uuid: str, title: str, subtitle: str, author: str, publisher: str, manifest_items: list, spine_ids: list, cover_id: str | None, toc_page_href: str | None) -> str: meta_cover = f'' if cover_id else "" manifest = "\n".join(_manifest_item_line(iid, href, mtype, is_nav) for iid, href, mtype, is_nav in manifest_items) spine = "\n".join(f' ' for sid in spine_ids) full_title = f"{title}: {subtitle}" if subtitle else title # is EPUB2-era, but Kindle's conversion pipeline and older # reading apps still use it to find the table of contents -- costs # nothing to include alongside the EPUB3 nav doc. guide = "" if toc_page_href: guide = f""" """ return f""" {esc(book_uuid)} {esc(full_title)} {esc(author)} {esc(publisher)} en 2026-01-01T00:00:00Z {meta_cover} {manifest} {spine} {guide} """ def build_nav(title: str, toc_entries: list) -> str: """toc_entries: list of dicts {href, label, children: [(href, label), ...]}.""" items = [] for entry in toc_entries: li = f'
  • {esc(entry["label"])}' if entry["children"]: sub = "\n".join(f'
  • {esc(l)}
  • ' for h, l in entry["children"]) li += f'\n
      \n{sub}\n
    \n ' li += "" items.append(li) items_str = "\n".join(items) return f""" {esc(title)} """ def render_psalm_index_xhtml(title: str, all_entries: list) -> str: """all_entries: list of (ref, href_with_anchor, prayer_title), already sorted by ref_sort_key. Unlike the print index (ref -> page number, meaningless in reflowable text with no fixed pages), each entry here is a direct clickable link to that prayer -- arguably MORE useful than the print version, since it's one tap instead of a page flip. href_with_anchor is relative to OEBPS/text/ (where this file lives), same convention as render_toc_page_xhtml.""" items = "\n".join( f'
  • {esc(ref)} — {esc(prayer_title)}
  • ' for ref, href, prayer_title in all_entries ) body = f'

    Index of Psalm References

      \n{items}\n
    ' return render_simple_xhtml(title, body) def render_toc_page_xhtml(title: str, toc_entries: list) -> str: """A real, visible in-book Contents page (unlike nav.xhtml, which most reading apps treat as sidebar-menu metadata, not a page you actually read through). This is what was missing -- KDP/Kindle conversion and many EPUB readers expect an actual TOC page in the reading order, not just the EPUB3 nav doc. IMPORTANT: this file itself lives at OEBPS/text/toc.xhtml, alongside the chapter files -- so unlike nav.xhtml (which lives at the OEBPS/ root and needs "text/..." prefixed hrefs), links here must be relative to the text/ directory, i.e. with that prefix stripped.""" def strip_prefix(href: str) -> str: return href[len("text/"):] if href.startswith("text/") else href items = [] for entry in toc_entries: href = strip_prefix(entry["href"]) li = f'
  • {esc(entry["label"])}' if entry["children"]: sub = "\n".join(f'
  • {esc(l)}
  • ' for h, l in entry["children"]) li += f'\n
      \n{sub}\n
    \n ' li += "" items.append(li) items_str = "\n".join(items) body = f'

    Contents

      \n{items_str}\n
    ' return render_simple_xhtml(title, body) def build_ncx(book_uuid: str, title: str, toc_entries: list) -> str: points = [] counter = [0] def next_order(): counter[0] += 1 return counter[0] for entry in toc_entries: parent_order = next_order() child_points = [] for h, l in entry["children"]: child_order = next_order() child_points.append(f""" {esc(l)} """) child_str = "\n".join(child_points) points.append(f""" {esc(entry['label'])} {child_str} """) points_str = "\n".join(points) return f""" {esc(title)} {points_str} """ def write_epub(out_path: Path, files: dict, image_paths: list): """files: dict of archive-path -> str content (text files) OR bytes. image_paths: list of local Path objects to copy in under OEBPS/images/ (cover, flourish, etc.) -- their manifest entries are added by the caller; this just physically copies the bytes in. mimetype MUST be first and MUST be stored (not deflated).""" if out_path.exists(): out_path.unlink() with zipfile.ZipFile(out_path, "w") as z: z.writestr("mimetype", "application/epub+zip", compress_type=zipfile.ZIP_STORED) for path, content in files.items(): data = content if isinstance(content, bytes) else content.encode("utf-8") z.writestr(path, data, compress_type=zipfile.ZIP_DEFLATED) for img_path in image_paths: z.write(img_path, f"OEBPS/images/{img_path.name}", compress_type=zipfile.ZIP_DEFLATED) # --- Main --------------------------------------------------------------------- def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("txt_dir", type=Path) ap.add_argument("output_epub", type=Path) ap.add_argument("--title", default="Prayers for Daily Life") ap.add_argument("--subtitle", default="101 Devotional Prayers for Every Occasion") ap.add_argument("--author", default="George Marin") ap.add_argument("--publisher", default="Zechariah Press") ap.add_argument("--cover", type=Path, default=None, help="Cover image (jpg/png). Strongly recommended for KDP.") ap.add_argument("--flourish", type=Path, default=None, help="Optional decorative divider image (jpg/png) shown under each chapter title -- same graphic as the print edition's chapter openers, if you want it.") ap.add_argument("--foreword", type=Path, default=None, help="Your actual template_forward.tex file -- parsed directly (paragraph breaks on \\medskip/\\bigskip, \\textit{} to italics, etc.), not a separate plain-text copy.") ap.add_argument("--review-url", default=None, help="If given, adds a back-matter review-request page with a clickable link to this URL (NOT a QR code -- pointless in an ebook, since the reader's already on the device).") ap.add_argument("--no-psalm-index", action="store_true", help="Omit the back-matter Index of Psalm References (on by default). Unlike the print version, this is a page of CLICKABLE LINKS, not page numbers -- reflowable text has no fixed pages to index.") ap.add_argument("--uuid", default=None, help="Fixed book UUID (e.g. urn:uuid:...). Random if omitted -- fine for testing, but pin one for the real release.") ap.add_argument("--category-order", default=None, help="Comma-separated category codes, e.g. WON,WWF,CRI,...") args = ap.parse_args() # Fail loudly and immediately if an optional file path was actually # GIVEN but doesn't resolve -- silently skipping it (the old # behavior) is exactly the kind of mistake that ships a KDP EPUB # with no embedded cover and nobody notices until much later. for flag_name, path in [ ("--cover", args.cover), ("--flourish", args.flourish), ("--foreword", args.foreword), ]: if path is not None and not path.exists(): sys.exit(f"{flag_name} was given as {path!s}, but that file doesn't exist. Fix the path or drop the flag.") book_uuid = args.uuid or f"urn:uuid:{uuid.uuid4()}" order = args.category_order.split(",") if args.category_order else list(CATEGORIES.keys()) for c in order: if c not in VALID_CODES: sys.exit(f"Unknown category code in --category-order: {c}") txt_files = sorted(args.txt_dir.glob("*.txt")) by_category: dict[str, list] = {c: [] for c in order} skipped = [] for path in txt_files: m = FILENAME_RE.match(path.name) if not m or m.group("code") not in VALID_CODES: continue parsed = parse_file(path.read_text(encoding="utf-8")) if parsed is None: skipped.append(path.name) continue code = m.group("code") if code not in by_category: by_category[code] = [] by_category[code].append(parsed) if skipped: print(f"Skipped {len(skipped)} file(s) that didn't parse:") for s in skipped: print(f" {s}") for code in by_category: by_category[code].sort(key=lambda e: ref_sort_key(e["ref"])) total = sum(len(v) for v in by_category.values()) print(f"Building EPUB from {total} entries across {sum(1 for v in by_category.values() if v)} categories.") # --- Assemble files --- files: dict[str, object] = {} files["META-INF/container.xml"] = CONTAINER_XML files["OEBPS/css/style.css"] = DEFAULT_CSS manifest_items = [] # (id, href, media-type, is_nav) spine_ids = [] toc_entries = [] # list of {href, label, children: [(href#anchor, label), ...]} # Title page title_html = f"""

    {esc(args.title)}

    {esc(args.subtitle)}

    {esc(args.author)}

    """ files["OEBPS/text/titlepage.xhtml"] = render_simple_xhtml(args.title, title_html) manifest_items.append(("titlepage", "text/titlepage.xhtml", "application/xhtml+xml", False)) spine_ids.append("titlepage") # Copyright page copyright_html = f"""""" files["OEBPS/text/copyright.xhtml"] = render_simple_xhtml("Copyright", copyright_html) manifest_items.append(("copyright", "text/copyright.xhtml", "application/xhtml+xml", False)) spine_ids.append("copyright") # Table of Contents -- a REAL visible page in the reading order, not just # the EPUB3 nav.xhtml (which most reading apps only use for their own # sidebar menu, not as an actual page). Placed here as a spine_ids # placeholder; the id "toc-page" is inserted once its file is written # below, once we know the full nested structure -- so reserve its # position in the spine now, fill in content after the loop. TOC_PAGE_POS = len(spine_ids) spine_ids.append("toc-page") # placeholder, replaced below once content is built # Foreword (optional) -- parsed directly from the real template_forward.tex if args.foreword: # existence already validated above fw_html = "

    Foreword

    \n" + render_foreword_from_latex(args.foreword.read_text(encoding="utf-8")) files["OEBPS/text/foreword.xhtml"] = render_simple_xhtml("Foreword", f'
    {fw_html}
    ') manifest_items.append(("foreword", "text/foreword.xhtml", "application/xhtml+xml", False)) spine_ids.append("foreword") toc_entries.append({"href": "text/foreword.xhtml", "label": "Foreword", "children": []}) # One XHTML file per category chapter, plus its nested prayer-level # TOC entries (anchor links within that chapter's file) for code in order: entries = by_category.get(code, []) if not entries: continue chapter_html, prayer_toc_items = render_chapter_xhtml( code, entries, args.title, flourish_img=args.flourish.name if args.flourish else None, # existence already validated above ) fname = f"{code.lower()}.xhtml" files[f"OEBPS/text/{fname}"] = chapter_html cid = f"chap-{code.lower()}" manifest_items.append((cid, f"text/{fname}", "application/xhtml+xml", False)) spine_ids.append(cid) children = [(f"text/{fname}#{anchor}", label) for anchor, label in prayer_toc_items] toc_entries.append({"href": f"text/{fname}", "label": category_name(code), "children": children}) # Review request (optional) -- a real hyperlink, not a QR code; see # --review-url help text for why a QR code doesn't make sense here. if args.review_url: review_html = f"""

    Did this book meet you somewhere?

    If it did, a review helps the next reader find their way here too.

    {esc(args.review_url)}

    """ files["OEBPS/text/review.xhtml"] = render_simple_xhtml("A Request", review_html) manifest_items.append(("review", "text/review.xhtml", "application/xhtml+xml", False)) spine_ids.append("review") # Index of Psalm References (optional, on by default) -- flat list # across ALL categories, sorted by verse reference, each entry # linking directly to its prayer. Built here (after every chapter's # anchors are known) rather than during the per-category loop above, # since it needs the complete cross-category picture. if not args.no_psalm_index: all_entries = [] for code in order: for i, e in enumerate(by_category.get(code, [])): anchor = f"{code.lower()}-{i+1:03d}" href = f"{code.lower()}.xhtml#{anchor}" all_entries.append((e["ref"], href, title_case(e["occasion"]))) all_entries.sort(key=lambda t: ref_sort_key(t[0])) files["OEBPS/text/psalm-index.xhtml"] = render_psalm_index_xhtml("Index of Psalm References", all_entries) manifest_items.append(("psalm-index", "text/psalm-index.xhtml", "application/xhtml+xml", False)) spine_ids.append("psalm-index") toc_entries.append({"href": "text/psalm-index.xhtml", "label": "Index of Psalm References", "children": []}) # Now that toc_entries is fully built, write the actual TOC page and # fill in the spine placeholder reserved above. files["OEBPS/text/toc.xhtml"] = render_toc_page_xhtml("Contents", toc_entries) manifest_items.append(("toc-page", "text/toc.xhtml", "application/xhtml+xml", False)) spine_ids[TOC_PAGE_POS] = "toc-page" # Cover cover_id = None image_paths = [] # local files to physically copy into OEBPS/images/ if args.cover: # existence already validated above image_paths.append(args.cover) media_type = "image/jpeg" if args.cover.suffix.lower() in (".jpg", ".jpeg") else "image/png" cover_id = "cover-image" manifest_items.append((cover_id, f"images/{args.cover.name}", media_type, False)) cover_xhtml = f'
    Cover
    ' files["OEBPS/text/cover.xhtml"] = render_simple_xhtml("Cover", cover_xhtml) manifest_items.append(("cover-page", "text/cover.xhtml", "application/xhtml+xml", False)) spine_ids.insert(0, "cover-page") if args.flourish: # existence already validated above image_paths.append(args.flourish) media_type = "image/jpeg" if args.flourish.suffix.lower() in (".jpg", ".jpeg") else "image/png" manifest_items.append(("flourish-image", f"images/{args.flourish.name}", media_type, False)) # Nav + NCX -- both reference the SAME nested toc_entries as the # visible toc.xhtml page above, so the reader-UI sidebar and the # in-book Contents page always agree. manifest_items.append(("nav", "nav.xhtml", "application/xhtml+xml", True)) manifest_items.append(("ncx", "toc.ncx", "application/x-dtbncx+xml", False)) files["OEBPS/nav.xhtml"] = build_nav(args.title, toc_entries) files["OEBPS/toc.ncx"] = build_ncx(book_uuid, args.title, toc_entries) files["OEBPS/content.opf"] = build_opf( book_uuid, args.title, args.subtitle, args.author, args.publisher, manifest_items, spine_ids, cover_id, toc_page_href="text/toc.xhtml", ) write_epub(args.output_epub, files, image_paths) print(f"Wrote {args.output_epub} ({total} prayers, {sum(1 for v in by_category.values() if v)} chapters)") if __name__ == "__main__": main()