#!/usr/bin/env python3 """ generate_latex.py Converts the finished, category-coded .txt files (one occasion per file, named CODE-CCC.VV.txt / CODE-CCC.VV-b.txt per the classify_occasions.py / rename_to_code_first.sh pipeline) into LaTeX. Writes one .tex file per .txt file, plus a parent main.tex that groups them into chapters by category (in the fixed order from categories.py, overridable with --category-order) and \\input{}s them in Psalm-reference order within each chapter. Expected per-file format -- a three-line header (reference, hand-written title, full verse quote), then exactly one "Occasion: ..." entry followed by its prayer: ## 23:6 > Goodness and Loving Kindness Follow Me >> Surely goodness and loving kindness shall follow me... Occasion: Grieving a loved one Lord, their place at the table is empty -- ... Amen. (A file with more than one Occasion entry -- e.g. a leftover BONEYARD holding-pen file -- is skipped; those aren't meant to reach the book.) Only files whose name starts with one of the 7 valid category codes (see categories.py) are processed. BONEYARD-*.txt files, and anything else that doesn't match the CODE-CCC.VV(-suffix).txt pattern, are skipped automatically -- no need to move them out of the directory first. INDEXING: each entry is indexed by its Psalm reference (e.g. "23:6"), not by occasion topic -- so the back-of-book index answers "where does this verse appear", which is what matters now that the book itself is organized by category rather than by Psalm order. The index key uses a zero-padded numeric sort ("023.006@23:6") so makeindex collates verses in Psalm order rather than alphabetically; a verse cited in more than one category naturally collects multiple page numbers under one entry. The \\Occasion{} macro (define it in your LaTeX preamble) is unchanged from before -- a visible label above each prayer: \\newcommand{\\Occasion}[1]{\\textit{Occasion: #1}\\\\[\\smallskipamount]} Two external LaTeX templates control the output: template.tex (--template) -- one devotion/page. Placeholders: [[NUM]] sequential devotion number, in FINAL BOOK ORDER (grouped by category, then by Psalm reference) -- NOT file/directory order. [[TITLE]] the OCCASION text, title-cased (e.g. "When I'm Simply Tired") -- this drives the page heading and the nested TOC entry via \\psalmentry. The hand-written devotion title (the '>' header line) is no longer used anywhere in the rendered book. [[REF]] Psalm chapter:verse [[VERSE]] the verse quote [[OCCASIONS]] carries ONLY the (invisible) Psalm-reference \\index{} call for this entry -- no visible text. (This placeholder's name predates several rounds of repurposing; it no longer has anything to do with "occasions" or a visible title caption.) The template must include [[OCCASIONS]] somewhere or the verse won't appear in the index at all. [[PRAYER]] the prayer body only (no label, no index) main-template.tex (--main-template) -- the book shell. Copied verbatim to latex/main.tex on every run -- it has no [[INPUTS]] placeholder anymore. Instead, it should contain a literal \\input{inputs} where the prayers belong; this script manages the SEPARATE file latex/inputs.tex: - If inputs.tex does not exist yet, it's generated fresh: chapter headings (one per category with entries, in the order from categories.py, overridable with --category-order) with \\input{} lines beneath each, in Psalm-reference order. The chapter heading macro defaults to \\chapter{...} -- override with --category-macro if your document class wants \\part{...} or a custom heading command instead. - If inputs.tex already exists, it is left completely untouched -- edit it by hand any time to fine-tune the order (within or across categories); reruns of this script will never overwrite your edits. Numbering ([[NUM]] below) follows whatever order is actually in inputs.tex, not the script's own default sort. Entries found in txt_dir but not yet mentioned in inputs.tex still get their .tex file generated (so it's ready to use) and are appended at the end with a provisional number, but are reported clearly so you know to place them by hand. \\input{} lines pointing at files that no longer exist in txt_dir are reported as stale (would break the build) but never silently removed. Usage: python generate_latex.py txt latex python generate_latex.py txt latex --category-macro part python generate_latex.py txt latex --category-order WWF,CRI,LOS,GUI,TMP,REL,WON """ import argparse import re import sys from pathlib import Path from categories import CATEGORIES, VALID_CODES, category_name # --- Parsing ---------------------------------------------------------------- HEADER_RE = re.compile( r"^(?P##\s*\S+)[ \t]*\n" r">(?!>)[ \t]*(?P[^\n]*)\n" r">>[ \t]*(?P<quote>[^\n]*)\n" ) OLD_HEADER_RE = re.compile(r"^(?P<ref>##\s*\S+)[ \t]*\n>(?!>)[ \t]*(?P<quote>[^\n]*)\n") OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$") # CODE-CCC.VV.txt or CODE-CCC.VV-b.txt (verse may be 2 or 3 digits, for # Psalm 119). Anything not matching this, or whose code isn't one of the # 7 valid categories, is skipped. FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<chapter>\d{3})\.(?P<verse>\d{2,3})(?:-[a-z])?\.txt$") # Parses the ref as WRITTEN IN THE FILE HEADER (e.g. "23:6", "139:23-24") # for sorting and index-key purposes -- more authoritative than the # filename, and handles verse ranges by sorting on the first number. REF_RE = re.compile(r"^(?P<chapter>\d+):(?P<verse>\d+)") def strip_comment_lines(text: str) -> str: lines = text.split("\n") out = [] in_block = False for ln in lines: stripped = ln.strip() if not in_block and stripped == "%v": in_block = True continue if in_block: if stripped == "%^": in_block = False continue if ln.lstrip().startswith("%"): continue out.append(ln) return "\n".join(out) def parse_ref(ref_line: str) -> str: return ref_line.lstrip("#").strip() def ref_sort_key(ref: str): """'23:6' -> (23, 6). '139:23-24' -> (139, 23). Falls back to (9999, 9999) (sorts last) if the ref doesn't parse, rather than crashing the whole run.""" m = REF_RE.match(ref) if not m: return (9999, 9999) return (int(m.group("chapter")), int(m.group("verse"))) def parse_file(text: str): """Returns one of: {"style": "ok", ref, title, quote, entries: [{"occasion", "text"}]} {"style": "needs_title", ref} {"style": "no_occasions", ref} {"style": "incomplete", ref, missing} {"style": "multi_occasion", ref} -- more than one Occasion line; not expected in the final per-category file layout None """ text = strip_comment_lines(text) stripped = text.lstrip("\n") m = HEADER_RE.match(stripped) if not m: old_m = OLD_HEADER_RE.match(stripped) if old_m: return {"style": "needs_title", "ref": parse_ref(old_m.group("ref"))} return None ref = parse_ref(m.group("ref")) title = m.group("title").strip() quote = m.group("quote").strip() body = text[m.end():] matches = list(OCCASION_LINE_RE.finditer(body)) if not matches: return {"style": "no_occasions", "ref": ref} if len(matches) > 1: return {"style": "multi_occasion", "ref": ref} entries = [] missing = [] for i, mo in enumerate(matches): occasion = mo.group("text").strip() start = mo.end() end = matches[i + 1].start() if i + 1 < len(matches) else len(body) prayer_text = body[start:end].strip() if not prayer_text: missing.append(occasion) entries.append({"occasion": occasion, "text": prayer_text}) if missing: return {"style": "incomplete", "ref": ref, "missing": missing} return {"style": "ok", "ref": ref, "title": title, "quote": quote, "entries": entries} # --- Title casing -------------------------------------------------------- _MINOR_WORDS = { "a", "an", "and", "as", "at", "but", "by", "en", "for", "if", "in", "nor", "of", "on", "or", "per", "the", "to", "v", "via", "vs", } def _title_case_word(word: str) -> str: """Capitalize a word's first letter, lowercase the rest -- crucially WITHOUT capitalizing a letter after an apostrophe, so "i'm" becomes "I'm", not "I'M" (which str.title() gets wrong).""" i = 0 while i < len(word) and not word[i].isalpha(): i += 1 if i >= len(word): return word return word[:i] + word[i].upper() + word[i + 1:].lower() def title_case(text: str) -> str: """Standard title-case rules: capitalize every word except a small set of articles/conjunctions/short prepositions, which stay lowercase UNLESS they're the first or last word. Handles contractions and possessives correctly (see _title_case_word).""" words = text.split(" ") out = [] for idx, w in enumerate(words): bare = re.sub(r"[^a-zA-Z']", "", w).lower() if 0 < idx < len(words) - 1 and bare in _MINOR_WORDS: out.append(w.lower()) else: out.append(_title_case_word(w)) return " ".join(out) # --- LaTeX escaping ----------------------------------------------------- _LATEX_ESCAPES = [ ("\\", r"\textbackslash{}"), ("&", r"\&"), ("%", r"\%"), ("$", r"\$"), ("#", r"\#"), ("_", r"\_"), ("{", r"\{"), ("}", r"\}"), ("~", r"\textasciitilde{}"), ("^", r"\textasciicircum{}"), ] def escape_latex(s: str) -> str: out = [] for ch in s: replaced = False for orig, repl in _LATEX_ESCAPES: if ch == orig: out.append(repl) replaced = True break if not replaced: out.append(ch) return "".join(out) def split_blocks(text: str): return [b.strip() for b in re.split(r"\n\s*\n", text.strip()) if b.strip()] # --- LaTeX rendering ---------------------------------------------------- def render_prayer(entry: dict) -> str: """Prayer body only. No occasion label, no index -- those live in render_occasion_label, which fills the separate [[OCCASIONS]] placeholder (shown above the verse quote, per the template's layout).""" stanzas = split_blocks(entry["text"]) stanza_texts = [] for stanza in stanzas: raw_lines = [ln.strip() for ln in stanza.split("\n") if ln.strip()] esc_lines = [escape_latex(ln) for ln in raw_lines] stanza_texts.append(" \\\\\n".join(esc_lines)) body = " \\\\[\\medskipamount]\n".join(stanza_texts) return "\\noindent " + body def render_ref_index(ref: str) -> str: """\\index{023.006@23:6} -- padded numeric sort key so makeindex collates in Psalm order, '@' display text is the ref as written.""" chapter, verse = ref_sort_key(ref) sort_key = f"{chapter:03d}.{verse:03d}" disp = escape_latex(ref) return f"\\index{{{sort_key}@{disp}}}" PRAYER_DIVIDER = "\n\\begin{center}\\textasteriskcentered\\end{center}\n" PLACEHOLDERS = ("[[NUM]]", "[[TITLE]]", "[[REF]]", "[[VERSE]]", "[[OCCASIONS]]", "[[PRAYER]]") def render_entry_tex(number: int, parsed: dict, template: str) -> str: # The page heading and TOC entry (fed via [[TITLE]] into \psalmentry's # second arg) are the OCCASION, title-cased -- the hand-written # devotion title (the '>' header line) is no longer used anywhere in # the rendered book at all. If a file somehow has more than one # occasion (not expected in the current one-occasion-per-file # layout), the first one's occasion becomes the page heading, since a # page can only have one heading. heading_esc = escape_latex(title_case(parsed["entries"][0]["occasion"])) ref_esc = escape_latex(parsed["ref"]) quote_esc = escape_latex(parsed["quote"]) prayer_tex = PRAYER_DIVIDER.join(render_prayer(e) for e in parsed["entries"]) # [[OCCASIONS]] now carries ONLY the invisible Psalm-reference index # call -- no visible text -- so the template doesn't need editing, # but the placeholder must still be present somewhere in it. ref_index_tex = render_ref_index(parsed["ref"]) out = template out = out.replace("[[NUM]]", str(number)) out = out.replace("[[TITLE]]", heading_esc) out = out.replace("[[REF]]", ref_esc) out = out.replace("[[VERSE]]", quote_esc) out = out.replace("[[OCCASIONS]]", ref_index_tex) out = out.replace("[[PRAYER]]", prayer_tex) leftover = re.findall(r"\[\[[A-Z]+\]\]", out) if leftover: print(f" warning: unrecognized placeholder(s) left in template: {sorted(set(leftover))}", file=sys.stderr) if not out.endswith("\n"): out += "\n" return out def render_inputs_block(grouped: "OrderedDictType", category_macro: str) -> str: """grouped: ordered mapping of code -> list of basenames (already in Psalm-reference order), for categories that have at least one entry. Returns the raw chapter+\\input{} block -- this is what gets written to inputs.tex when it doesn't exist yet, NOT injected into main.tex directly (main.tex now \\input{}s inputs.tex itself, per the template).""" blocks = [] for code, basenames in grouped.items(): if not basenames: continue name = category_name(code) heading = f"\\{category_macro}{{{escape_latex(name)}}}" inputs = "\n".join(f"\\input{{{name_}}}" for name_ in basenames) blocks.append(f"{heading}\n{inputs}") text = "\n\n".join(blocks) if not text.endswith("\n"): text += "\n" return text INPUT_LINE_RE = re.compile(r"\\input\{(?P<name>[^}]+)\}") CHAPTER_LINE_RE = re.compile(r"\\(?:chapter|part|section)\*?\{(?P<name>[^}]*)\}") def parse_inputs_tex(text: str, code_by_name: dict) -> tuple: """Parses an EXISTING (hand-edited) inputs.tex to recover the actual final order. Returns (ordered_basenames, mismatches) where mismatches is a list of (basename, expected_code, chapter_context) for any \\input{} that appears under a chapter heading whose name doesn't match that file's own category code -- a likely sign of an accidental drag-and-drop during reordering, reported but never auto-fixed.""" ordered = [] mismatches = [] current_chapter_code = None for line in text.split("\n"): cm = CHAPTER_LINE_RE.search(line) if cm: current_chapter_code = code_by_name.get(cm.group("name").strip()) continue im = INPUT_LINE_RE.search(line) if im: name = im.group("name").strip() ordered.append(name) file_code = name.split("-", 1)[0] if current_chapter_code and file_code != current_chapter_code: mismatches.append((name, file_code, current_chapter_code)) return ordered, mismatches # --- Main ----------------------------------------------------------------- def main(): from collections import OrderedDict parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("txt_dir", nargs="?", default="txt", help="Input directory of .txt files (default: txt)") parser.add_argument("latex_dir", nargs="?", default="latex", help="Output directory for .tex files (default: latex)") parser.add_argument("--pattern", default="*.txt", help="Glob pattern for input files (default: *.txt)") parser.add_argument("--template", default="template-prayer.tex", help="Path to the per-entry LaTeX template (default: template-prayer.tex)") parser.add_argument("--main-template", default="template-main.tex", help="Path to the main.tex template (default: template-main.tex)") parser.add_argument("--category-macro", default="chapter", help="LaTeX heading macro for each category, without backslash (default: chapter)") parser.add_argument( "--category-order", default=None, help="Comma-separated category codes for book order, e.g. WWF,CRI,LOS,GUI,TMP,REL,WON " "(default: the order defined in categories.py)", ) args = parser.parse_args() txt_dir = Path(args.txt_dir) latex_dir = Path(args.latex_dir) template_path = Path(args.template) main_template_path = Path(args.main_template) if not txt_dir.is_dir(): sys.exit(f"Not a directory: {txt_dir}") if not template_path.is_file(): sys.exit(f"Template not found: {template_path}") if not main_template_path.is_file(): sys.exit(f"Main template not found: {main_template_path}") latex_dir.mkdir(parents=True, exist_ok=True) if args.category_order: order = [c.strip().upper() for c in args.category_order.split(",")] bad = [c for c in order if c not in VALID_CODES] if bad: sys.exit(f"Unknown category code(s) in --category-order: {bad}") else: order = list(CATEGORIES.keys()) template = template_path.read_text(encoding="utf-8") missing = [p for p in PLACEHOLDERS if p not in template] if missing: print(f"Note: entry template doesn't use these placeholders: {missing}", file=sys.stderr) main_template = main_template_path.read_text(encoding="utf-8") files = sorted(txt_dir.glob(args.pattern)) if not files: sys.exit(f"No files matching {args.pattern} in {txt_dir}") print(f"Found {len(files)} files in {txt_dir}") # code -> list of {"path", "parsed", "ref_key"} by_category = OrderedDict((c, []) for c in order) skipped = [] ignored_filename = [] for path in files: m = FILENAME_RE.match(path.name) if not m or m.group("code") not in VALID_CODES: ignored_filename.append(path.name) continue code = m.group("code") if code not in by_category: # Valid code, but not included in --category-order -- treat # like any other skip rather than silently dropping it. print(f"SKIP (category {code} not in --category-order): {path.name}") skipped.append(path.name) continue text = path.read_text(encoding="utf-8") parsed = parse_file(text) if parsed is None: print(f"SKIP (no recognizable header found): {path.name}") skipped.append(path.name) continue if parsed["style"] == "needs_title": print(f"SKIP (needs a title line added -- old two-line header) [{parsed['ref']}]: {path.name}") skipped.append(path.name) continue if parsed["style"] == "no_occasions": print(f"SKIP (no Occasion line) [{parsed['ref']}]: {path.name}") skipped.append(path.name) continue if parsed["style"] == "multi_occasion": print(f"SKIP (more than one Occasion line -- looks like a BONEYARD-style file) [{parsed['ref']}]: {path.name}") skipped.append(path.name) continue if parsed["style"] == "incomplete": print(f"SKIP (occasion with no prayer yet) [{parsed['ref']}]: {path.name}") for o in parsed["missing"]: print(f" missing: {o}") skipped.append(path.name) continue by_category[code].append({"path": path, "parsed": parsed, "ref_key": ref_sort_key(parsed["ref"])}) if ignored_filename: print(f"\nIgnored {len(ignored_filename)} file(s) not matching CODE-CCC.VV(-suffix).txt " f"with a valid category (e.g. BONEYARD-*.txt):") for name in ignored_filename[:10]: print(f" - {name}") if len(ignored_filename) > 10: print(f" ... and {len(ignored_filename) - 10} more") # Sort each category's entries by Psalm reference. for code in by_category: by_category[code].sort(key=lambda e: e["ref_key"]) print() print("Category breakdown (all valid entries found in txt/):") total = 0 for code in order: n = len(by_category[code]) total += n print(f" {code:4s} {category_name(code):32s} {n}") print(f" {'':4s} {'TOTAL':32s} {total}") print() # Flat lookup of every valid entry, keyed by basename, plus the # default fallback order (category order, then Psalm ref) -- used # either to seed a fresh inputs.tex, or to provisionally place any # entry the hand-edited inputs.tex doesn't mention yet. entry_by_basename = {} default_order = [] for code in order: for item in by_category[code]: basename = item["path"].stem entry_by_basename[basename] = item default_order.append(basename) inputs_path = latex_dir / "inputs.tex" code_by_name = {category_name(c): c for c in VALID_CODES} if not inputs_path.exists(): grouped = OrderedDict((c, [item["path"].stem for item in by_category[c]]) for c in order) inputs_path.write_text(render_inputs_block(grouped, args.category_macro), encoding="utf-8") final_order = list(default_order) print(f"No inputs.tex found -- generated a fresh one at {inputs_path} " f"from the default category/Psalm order.") print("Edit it by hand any time to fine-tune ordering -- future runs will") print("leave your edits alone as long as the file exists.") else: existing_text = inputs_path.read_text(encoding="utf-8") parsed_order, mismatches = parse_inputs_tex(existing_text, code_by_name) print(f"Found existing {inputs_path} -- honoring its order, NOT regenerating it.") print("(delete it if you want a fresh default ordering)") seen = set() final_order = [] stale = [] for basename in parsed_order: if basename in seen: print(f" note: {basename} is \\input{{}} more than once in inputs.tex -- using its first occurrence") continue seen.add(basename) if basename not in entry_by_basename: stale.append(basename) continue final_order.append(basename) missing_from_inputs = [b for b in default_order if b not in seen] if stale: print(f"\n WARNING -- {len(stale)} \\input{{}} line(s) in inputs.tex point to files " f"that no longer exist in {txt_dir}/ (would break the build if left in):") for name in stale: print(f" - {name}") if mismatches: print(f"\n note -- {len(mismatches)} entr{'y' if len(mismatches) == 1 else 'ies'} in inputs.tex " f"sit under a chapter that doesn't match their own category code:") for name, file_code, chapter_code in mismatches: print(f" - {name} (file is {file_code}, but listed under {chapter_code}'s chapter)") if missing_from_inputs: print(f"\n {len(missing_from_inputs)} entr{'y' if len(missing_from_inputs) == 1 else 'ies'} in " f"{txt_dir}/ not yet placed in inputs.tex -- added at the end, provisionally numbered " f"(add a \\input{{}} line by hand to place properly):") for name in missing_from_inputs: print(f" - {name}") final_order.extend(missing_from_inputs) print() number = 0 for basename in final_order: item = entry_by_basename[basename] number += 1 tex = render_entry_tex(number, item["parsed"], template) out_path = latex_dir / (basename + ".tex") out_path.write_text(tex, encoding="utf-8") main_tex = main_template leftover = re.findall(r"\[\[[A-Z]+\]\]", main_tex) if leftover: print(f"warning: unrecognized placeholder(s) left in main template: {sorted(set(leftover))}", file=sys.stderr) (latex_dir / "main.tex").write_text(main_tex, encoding="utf-8") print(f"Wrote {number} entry files to {latex_dir}/ (numbered per inputs.tex order), plus main.tex") if skipped: print(f"Skipped {len(skipped)} file(s) with parse problems:") for name in skipped: print(f" - {name}") print() print("To build: cd into the latex dir and run pdflatex twice, then") print("makeindex main, then pdflatex twice more, to resolve the TOC") print("and the Psalm-reference index.") if __name__ == "__main__": main()