710 lines
29 KiB
Python
710 lines
29 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
generate_latex.py
|
|
|
|
Converts the finished, category-coded .txt files (one occasion per file,
|
|
named CODE-CCC.VV.txt / CODE-CCC.VV-b.txt per the classify_occasions.py /
|
|
rename_to_code_first.sh pipeline) into LaTeX. Writes one .tex file per
|
|
.txt file, plus a parent main.tex that groups them into chapters by
|
|
category (in the fixed order from categories.py, overridable with
|
|
--category-order) and \\input{}s them in Psalm-reference order within
|
|
each chapter.
|
|
|
|
Expected per-file format -- a three-line header (reference, hand-written
|
|
title, full verse quote), then exactly one "Occasion: ..." entry
|
|
followed by its prayer:
|
|
|
|
## 23:6
|
|
> Goodness and Loving Kindness Follow Me
|
|
>> Surely goodness and loving kindness shall follow me...
|
|
|
|
Occasion: Grieving a loved one
|
|
Lord, their place at the table is empty --
|
|
...
|
|
Amen.
|
|
|
|
(A file with more than one Occasion entry -- e.g. a leftover BONEYARD
|
|
holding-pen file -- is skipped; those aren't meant to reach the book.)
|
|
|
|
Only files whose name starts with one of the 7 valid category codes
|
|
(see categories.py) are processed. BONEYARD-*.txt files, and anything
|
|
else that doesn't match the CODE-CCC.VV(-suffix).txt pattern, are
|
|
skipped automatically -- no need to move them out of the directory
|
|
first.
|
|
|
|
INDEXING: each entry is indexed by its Psalm reference (e.g. "23:6"),
|
|
not by occasion topic -- so the back-of-book index answers "where does
|
|
this verse appear", which is what matters now that the book itself is
|
|
organized by category rather than by Psalm order. The index key uses a
|
|
zero-padded numeric sort ("023.006@23:6") so makeindex collates verses
|
|
in Psalm order rather than alphabetically; a verse cited in more than
|
|
one category naturally collects multiple page numbers under one entry.
|
|
|
|
The \\Occasion{} macro (define it in your LaTeX preamble) is unchanged
|
|
from before -- a visible label above each prayer:
|
|
|
|
\\newcommand{\\Occasion}[1]{\\textit{Occasion: #1}\\\\[\\smallskipamount]}
|
|
|
|
Two external LaTeX templates control the output:
|
|
|
|
template.tex (--template) -- one devotion/page. Placeholders:
|
|
[[NUM]] sequential devotion number, in FINAL BOOK ORDER
|
|
(grouped by category, then by Psalm reference) --
|
|
NOT file/directory order.
|
|
[[TITLE]] the OCCASION text, title-cased (e.g. "When I'm Simply
|
|
Tired") -- this drives the page heading and the
|
|
nested TOC entry via \\psalmentry. The hand-written
|
|
devotion title (the '>' header line) is no longer
|
|
used anywhere in the rendered book.
|
|
[[REF]] Psalm chapter:verse
|
|
[[VERSE]] the verse quote
|
|
[[OCCASIONS]] carries ONLY the (invisible) Psalm-reference \\index{}
|
|
call for this entry -- no visible text. (This
|
|
placeholder's name predates several rounds of
|
|
repurposing; it no longer has anything to do with
|
|
"occasions" or a visible title caption.) The template
|
|
must include [[OCCASIONS]] somewhere or the verse
|
|
won't appear in the index at all.
|
|
[[PRAYER]] the prayer body only (no label, no index)
|
|
|
|
main-template.tex (--main-template) -- the book shell. Copied verbatim
|
|
to latex/main.tex on every run -- it has no [[INPUTS]] placeholder
|
|
anymore. Instead, it should contain a literal \\input{inputs} where
|
|
the prayers belong; this script manages the SEPARATE file
|
|
latex/inputs.tex:
|
|
|
|
- If inputs.tex does not exist yet, it's generated fresh: chapter
|
|
headings (one per category with entries, in the order from
|
|
categories.py, overridable with --category-order) with
|
|
\\input{} lines beneath each, in Psalm-reference order. The
|
|
chapter heading macro defaults to \\chapter{...} -- override
|
|
with --category-macro if your document class wants \\part{...}
|
|
or a custom heading command instead.
|
|
|
|
- If inputs.tex already exists, it is left completely untouched --
|
|
edit it by hand any time to fine-tune the order (within or
|
|
across categories); reruns of this script will never overwrite
|
|
your edits. Numbering ([[NUM]] below) follows whatever order is
|
|
actually in inputs.tex, not the script's own default sort.
|
|
Entries found in txt_dir but not yet mentioned in inputs.tex
|
|
still get their .tex file generated (so it's ready to use) and
|
|
are appended at the end with a provisional number, but are
|
|
reported clearly so you know to place them by hand.
|
|
\\input{} lines pointing at files that no longer exist in
|
|
txt_dir are reported as stale (would break the build) but never
|
|
silently removed.
|
|
|
|
Usage:
|
|
python generate_latex.py txt latex
|
|
python generate_latex.py txt latex --category-macro part
|
|
python generate_latex.py txt latex --category-order WWF,CRI,LOS,GUI,TMP,REL,WON
|
|
"""
|
|
|
|
import argparse
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
from categories import CATEGORIES, VALID_CODES, category_name
|
|
|
|
|
|
class _Tee:
|
|
"""Duplicates writes to multiple streams -- used so every print()
|
|
already in this script (stdout AND stderr) also lands in a log
|
|
file, without having to touch each individual print() call. Set up
|
|
once args are parsed and latex_dir exists; see main()."""
|
|
|
|
def __init__(self, *streams):
|
|
self._streams = streams
|
|
|
|
def write(self, data):
|
|
for s in self._streams:
|
|
s.write(data)
|
|
|
|
def flush(self):
|
|
for s in self._streams:
|
|
s.flush()
|
|
|
|
# --- Parsing ----------------------------------------------------------------
|
|
|
|
HEADER_RE = re.compile(
|
|
r"^(?P<ref>##\s*\S+)[ \t]*\n"
|
|
r">(?!>)[ \t]*(?P<title>[^\n]*)\n"
|
|
r">>[ \t]*(?P<quote>[^\n]*)\n"
|
|
)
|
|
OLD_HEADER_RE = re.compile(r"^(?P<ref>##\s*\S+)[ \t]*\n>(?!>)[ \t]*(?P<quote>[^\n]*)\n")
|
|
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$")
|
|
|
|
# CODE-CCC.VV.txt or CODE-CCC.VV-b.txt (verse may be 2 or 3 digits, for
|
|
# Psalm 119). Anything not matching this, or whose code isn't one of the
|
|
# 7 valid categories, is skipped.
|
|
FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<chapter>\d{3})\.(?P<verse>\d{2,3})(?:-[a-z])?\.txt$")
|
|
|
|
# Parses the ref as WRITTEN IN THE FILE HEADER (e.g. "23:6", "139:23-24")
|
|
# for sorting and index-key purposes -- more authoritative than the
|
|
# filename, and handles verse ranges by sorting on the first number.
|
|
REF_RE = re.compile(r"^(?P<chapter>\d+):(?P<verse>\d+)")
|
|
|
|
|
|
def strip_comment_lines(text: str) -> str:
|
|
lines = text.split("\n")
|
|
out = []
|
|
in_block = False
|
|
for ln in lines:
|
|
stripped = ln.strip()
|
|
if not in_block and stripped == "%v":
|
|
in_block = True
|
|
continue
|
|
if in_block:
|
|
if stripped == "%^":
|
|
in_block = False
|
|
continue
|
|
if ln.lstrip().startswith("%"):
|
|
continue
|
|
out.append(ln)
|
|
return "\n".join(out)
|
|
|
|
|
|
def parse_ref(ref_line: str) -> str:
|
|
return ref_line.lstrip("#").strip()
|
|
|
|
|
|
def ref_sort_key(ref: str):
|
|
"""'23:6' -> (23, 6). '139:23-24' -> (139, 23). Falls back to (9999, 9999)
|
|
(sorts last) if the ref doesn't parse, rather than crashing the whole run."""
|
|
m = REF_RE.match(ref)
|
|
if not m:
|
|
return (9999, 9999)
|
|
return (int(m.group("chapter")), int(m.group("verse")))
|
|
|
|
|
|
def parse_file(text: str):
|
|
"""Returns one of:
|
|
{"style": "ok", ref, title, quote, entries: [{"occasion", "text"}]}
|
|
{"style": "needs_title", ref}
|
|
{"style": "no_occasions", ref}
|
|
{"style": "incomplete", ref, missing}
|
|
{"style": "multi_occasion", ref} -- more than one Occasion line;
|
|
not expected in the final
|
|
per-category file layout
|
|
None
|
|
"""
|
|
text = strip_comment_lines(text)
|
|
stripped = text.lstrip("\n")
|
|
|
|
m = HEADER_RE.match(stripped)
|
|
if not m:
|
|
old_m = OLD_HEADER_RE.match(stripped)
|
|
if old_m:
|
|
return {"style": "needs_title", "ref": parse_ref(old_m.group("ref"))}
|
|
return None
|
|
|
|
ref = parse_ref(m.group("ref"))
|
|
title = m.group("title").strip()
|
|
quote = m.group("quote").strip()
|
|
body = text[m.end():]
|
|
|
|
matches = list(OCCASION_LINE_RE.finditer(body))
|
|
if not matches:
|
|
return {"style": "no_occasions", "ref": ref}
|
|
|
|
if len(matches) > 1:
|
|
return {"style": "multi_occasion", "ref": ref}
|
|
|
|
entries = []
|
|
missing = []
|
|
for i, mo in enumerate(matches):
|
|
occasion = mo.group("text").strip()
|
|
start = mo.end()
|
|
end = matches[i + 1].start() if i + 1 < len(matches) else len(body)
|
|
prayer_text = body[start:end].strip()
|
|
if not prayer_text:
|
|
missing.append(occasion)
|
|
entries.append({"occasion": occasion, "text": prayer_text})
|
|
|
|
if missing:
|
|
return {"style": "incomplete", "ref": ref, "missing": missing}
|
|
|
|
return {"style": "ok", "ref": ref, "title": title, "quote": quote, "entries": entries}
|
|
|
|
|
|
# --- Title casing --------------------------------------------------------
|
|
|
|
_MINOR_WORDS = {
|
|
"a", "an", "and", "as", "at", "but", "by", "en", "for", "if", "in",
|
|
"nor", "of", "on", "or", "per", "the", "to", "v", "via", "vs",
|
|
}
|
|
|
|
|
|
def _title_case_word(word: str) -> str:
|
|
"""Capitalize a word's first letter, lowercase the rest -- crucially
|
|
WITHOUT capitalizing a letter after an apostrophe, so "i'm" becomes
|
|
"I'm", not "I'M" (which str.title() gets wrong)."""
|
|
i = 0
|
|
while i < len(word) and not word[i].isalpha():
|
|
i += 1
|
|
if i >= len(word):
|
|
return word
|
|
return word[:i] + word[i].upper() + word[i + 1:].lower()
|
|
|
|
|
|
def title_case(text: str) -> str:
|
|
"""Standard title-case rules: capitalize every word except a small
|
|
set of articles/conjunctions/short prepositions, which stay
|
|
lowercase UNLESS they're the first or last word. Handles
|
|
contractions and possessives correctly (see _title_case_word)."""
|
|
words = text.split(" ")
|
|
out = []
|
|
for idx, w in enumerate(words):
|
|
bare = re.sub(r"[^a-zA-Z']", "", w).lower()
|
|
if 0 < idx < len(words) - 1 and bare in _MINOR_WORDS:
|
|
out.append(w.lower())
|
|
else:
|
|
out.append(_title_case_word(w))
|
|
return " ".join(out)
|
|
|
|
|
|
# --- LaTeX escaping -----------------------------------------------------
|
|
|
|
_LATEX_ESCAPES = [
|
|
("\\", r"\textbackslash{}"),
|
|
("&", r"\&"),
|
|
("%", r"\%"),
|
|
("$", r"\$"),
|
|
("#", r"\#"),
|
|
("_", r"\_"),
|
|
("{", r"\{"),
|
|
("}", r"\}"),
|
|
("~", r"\textasciitilde{}"),
|
|
("^", r"\textasciicircum{}"),
|
|
]
|
|
|
|
|
|
def escape_latex(s: str) -> str:
|
|
out = []
|
|
for ch in s:
|
|
replaced = False
|
|
for orig, repl in _LATEX_ESCAPES:
|
|
if ch == orig:
|
|
out.append(repl)
|
|
replaced = True
|
|
break
|
|
if not replaced:
|
|
out.append(ch)
|
|
return "".join(out)
|
|
|
|
|
|
def split_blocks(text: str):
|
|
return [b.strip() for b in re.split(r"\n\s*\n", text.strip()) if b.strip()]
|
|
|
|
|
|
# --- LaTeX rendering ----------------------------------------------------
|
|
|
|
# Matches a line that starts DIRECTLY with a letter -- the case where a
|
|
# real lettrine-style drop cap (which must be the literal first token of
|
|
# the paragraph) can be used cleanly. Captures the first letter and the
|
|
# rest of that same word separately, since \lettrine{first}{restofword}
|
|
# wants them split (lettrine.cfg renders the rest of the word in small
|
|
# caps by default).
|
|
LETTRINE_RE = re.compile(r"^([A-Za-z])([A-Za-z']*)(.*)$", re.DOTALL)
|
|
|
|
# Fallback for a line that opens with punctuation (a quote mark, an em
|
|
# dash, etc.) -- tested against real lettrine output and confirmed it
|
|
# looks cramped/wrong when the leading punctuation is folded into
|
|
# \lettrine's argument, so these get the simpler inline \dropcap{}
|
|
# treatment instead, with the punctuation left on the normal baseline
|
|
# in front of it.
|
|
DROPCAP_RE = re.compile(r"^(\W*)([A-Za-z])(.*)$", re.DOTALL)
|
|
|
|
|
|
# Drop cap height, in lines -- fixed at 2 regardless of how long the
|
|
# first stanza actually runs. Note this means the margin WILL snap back
|
|
# for any stanza longer than 2 lines (most of them), same artifact
|
|
# flagged and fixed-away earlier in this file's history -- kept here
|
|
# anyway because the shorter, more compact cap was explicitly preferred
|
|
# over avoiding that artifact. min() with the stanza's own line count
|
|
# just guards the rare 1-line-stanza case, where reserving 2 lines of
|
|
# indent for a stanza that only has 1 would leave dead space.
|
|
DROPCAP_LINES = 2
|
|
|
|
|
|
def apply_dropcap(line: str, stanza_line_count: int) -> str:
|
|
"""Wraps the first letter of a prayer's opening line in a drop cap.
|
|
Uses a real multi-line \\lettrine{}{} when the line starts directly
|
|
with a letter -- lines= is fixed at DROPCAP_LINES (see above).
|
|
|
|
Falls back to a simple enlarged inline \\dropcap{} only when the
|
|
line opens with punctuation (e.g. a quotation mark), since lettrine
|
|
can't cleanly absorb a leading non-letter character into its own
|
|
argument.
|
|
|
|
Both macros need defining in your preamble, e.g.:
|
|
\\usepackage{lettrine}
|
|
\\newcommand{\\dropcap}[1]{{\\bfseries\\Large #1}}
|
|
"""
|
|
lm = LETTRINE_RE.match(line)
|
|
if lm and not line[:1].isspace():
|
|
first, restword, rest = lm.groups()
|
|
lines_opt = min(stanza_line_count, DROPCAP_LINES)
|
|
return (
|
|
f"\\lettrine[lines={lines_opt},nindent=0pt]{{" + escape_latex(first) + "}{" + escape_latex(restword) + "}"
|
|
+ escape_latex(rest)
|
|
)
|
|
dm = DROPCAP_RE.match(line)
|
|
if not dm:
|
|
return escape_latex(line)
|
|
lead, letter, rest = dm.groups()
|
|
return escape_latex(lead) + "\\dropcap{" + escape_latex(letter) + "}" + escape_latex(rest)
|
|
|
|
|
|
def render_prayer(entry: dict) -> str:
|
|
"""Prayer body only. No occasion label, no index -- those live in
|
|
render_occasion_label, which fills the separate [[OCCASIONS]]
|
|
placeholder (shown above the verse quote, per the template's layout).
|
|
|
|
The very first letter of the prayer's opening line gets a drop cap --
|
|
see apply_dropcap() for the lettrine/fallback split and how its
|
|
height is chosen."""
|
|
stanzas = split_blocks(entry["text"])
|
|
stanza_texts = []
|
|
for s_idx, stanza in enumerate(stanzas):
|
|
raw_lines = [ln.strip() for ln in stanza.split("\n") if ln.strip()]
|
|
esc_lines = []
|
|
for l_idx, ln in enumerate(raw_lines):
|
|
if s_idx == 0 and l_idx == 0:
|
|
esc_lines.append(apply_dropcap(ln, len(raw_lines)))
|
|
else:
|
|
esc_lines.append(escape_latex(ln))
|
|
stanza_texts.append(" \\\\\n".join(esc_lines))
|
|
body = " \\\\[\\medskipamount]\n".join(stanza_texts)
|
|
return "\\noindent " + body
|
|
|
|
|
|
def render_ref_index(ref: str) -> str:
|
|
"""\\index{023.006@23:6} -- padded numeric sort key so makeindex
|
|
collates in Psalm order, '@' display text is the ref as written."""
|
|
chapter, verse = ref_sort_key(ref)
|
|
sort_key = f"{chapter:03d}.{verse:03d}"
|
|
disp = escape_latex(ref)
|
|
return f"\\index{{{sort_key}@{disp}}}"
|
|
|
|
|
|
PRAYER_DIVIDER = "\n\\begin{center}\\textasteriskcentered\\end{center}\n"
|
|
|
|
PLACEHOLDERS = ("[[NUM]]", "[[TITLE]]", "[[REF]]", "[[VERSE]]", "[[OCCASIONS]]", "[[PRAYER]]")
|
|
|
|
|
|
def render_entry_tex(number: int, parsed: dict, template: str) -> str:
|
|
# The page heading and TOC entry (fed via [[TITLE]] into \psalmentry's
|
|
# second arg) are the OCCASION, title-cased -- the hand-written
|
|
# devotion title (the '>' header line) is no longer used anywhere in
|
|
# the rendered book at all. If a file somehow has more than one
|
|
# occasion (not expected in the current one-occasion-per-file
|
|
# layout), the first one's occasion becomes the page heading, since a
|
|
# page can only have one heading.
|
|
heading_esc = escape_latex(title_case(parsed["entries"][0]["occasion"]))
|
|
ref_esc = escape_latex(parsed["ref"])
|
|
quote_esc = escape_latex(parsed["quote"])
|
|
prayer_tex = PRAYER_DIVIDER.join(render_prayer(e) for e in parsed["entries"])
|
|
# [[OCCASIONS]] now carries ONLY the invisible Psalm-reference index
|
|
# call -- no visible text -- so the template doesn't need editing,
|
|
# but the placeholder must still be present somewhere in it.
|
|
ref_index_tex = render_ref_index(parsed["ref"])
|
|
|
|
out = template
|
|
out = out.replace("[[NUM]]", str(number))
|
|
out = out.replace("[[TITLE]]", heading_esc)
|
|
out = out.replace("[[REF]]", ref_esc)
|
|
out = out.replace("[[VERSE]]", quote_esc)
|
|
out = out.replace("[[OCCASIONS]]", ref_index_tex)
|
|
out = out.replace("[[PRAYER]]", prayer_tex)
|
|
|
|
leftover = re.findall(r"\[\[[A-Z]+\]\]", out)
|
|
if leftover:
|
|
print(f" warning: unrecognized placeholder(s) left in template: {sorted(set(leftover))}", file=sys.stderr)
|
|
|
|
if not out.endswith("\n"):
|
|
out += "\n"
|
|
return out
|
|
|
|
|
|
def render_inputs_block(grouped: "OrderedDictType", category_macro: str) -> str:
|
|
"""grouped: ordered mapping of code -> list of (basename, occasion)
|
|
pairs (already in Psalm-reference order), for categories that have
|
|
at least one entry. Each \\input{} line gets a trailing comment with
|
|
that entry's occasion text, purely to make a hand-edited inputs.tex
|
|
easier to skim/reorder -- comments are inert to LaTeX, so this is
|
|
just a courtesy to whoever's editing the file by hand. Returns the
|
|
raw chapter+\\input{} block -- this is what gets written to
|
|
inputs.tex when it doesn't exist yet, NOT injected into main.tex
|
|
directly (main.tex now \\input{}s inputs.tex itself, per the
|
|
template)."""
|
|
blocks = []
|
|
for code, entries in grouped.items():
|
|
if not entries:
|
|
continue
|
|
name = category_name(code)
|
|
heading = f"\\{category_macro}{{{escape_latex(name)}}}"
|
|
inputs = "\n".join(f"\\input{{{basename}}} % {occasion}" for basename, occasion in entries)
|
|
blocks.append(f"{heading}\n{inputs}")
|
|
text = "\n\n".join(blocks)
|
|
if not text.endswith("\n"):
|
|
text += "\n"
|
|
return text
|
|
|
|
|
|
INPUT_LINE_RE = re.compile(r"\\input\{(?P<name>[^}]+)\}")
|
|
CHAPTER_LINE_RE = re.compile(r"\\(?:chapter|part|section)\*?\{(?P<name>[^}]*)\}")
|
|
|
|
|
|
def parse_inputs_tex(text: str, code_by_name: dict) -> tuple:
|
|
"""Parses an EXISTING (hand-edited) inputs.tex to recover the actual
|
|
final order. Returns (ordered_basenames, mismatches) where
|
|
mismatches is a list of (basename, expected_code, chapter_context)
|
|
for any \\input{} that appears under a chapter heading whose name
|
|
doesn't match that file's own category code -- a likely sign of an
|
|
accidental drag-and-drop during reordering, reported but never
|
|
auto-fixed."""
|
|
ordered = []
|
|
mismatches = []
|
|
current_chapter_code = None
|
|
for line in text.split("\n"):
|
|
cm = CHAPTER_LINE_RE.search(line)
|
|
if cm:
|
|
current_chapter_code = code_by_name.get(cm.group("name").strip())
|
|
continue
|
|
im = INPUT_LINE_RE.search(line)
|
|
if im:
|
|
name = im.group("name").strip()
|
|
ordered.append(name)
|
|
file_code = name.split("-", 1)[0]
|
|
if current_chapter_code and file_code != current_chapter_code:
|
|
mismatches.append((name, file_code, current_chapter_code))
|
|
return ordered, mismatches
|
|
|
|
|
|
# --- Main -----------------------------------------------------------------
|
|
|
|
def main():
|
|
from collections import OrderedDict
|
|
|
|
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
parser.add_argument("txt_dir", nargs="?", default="txt", help="Input directory of .txt files (default: txt)")
|
|
parser.add_argument("latex_dir", nargs="?", default="latex", help="Output directory for .tex files (default: latex)")
|
|
parser.add_argument("--pattern", default="*.txt", help="Glob pattern for input files (default: *.txt)")
|
|
parser.add_argument("--template", default="template-prayer.tex", help="Path to the per-entry LaTeX template (default: template-prayer.tex)")
|
|
parser.add_argument("--main-template", default="template-main.tex", help="Path to the main.tex template (default: template-main.tex)")
|
|
parser.add_argument("--category-macro", default="chapter", help="LaTeX heading macro for each category, without backslash (default: chapter)")
|
|
parser.add_argument(
|
|
"--category-order", default=None,
|
|
help="Comma-separated category codes for book order, e.g. WWF,CRI,LOS,GUI,TMP,REL,WON "
|
|
"(default: the order defined in categories.py)",
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
txt_dir = Path(args.txt_dir)
|
|
latex_dir = Path(args.latex_dir)
|
|
template_path = Path(args.template)
|
|
main_template_path = Path(args.main_template)
|
|
|
|
if not txt_dir.is_dir():
|
|
sys.exit(f"Not a directory: {txt_dir}")
|
|
if not template_path.is_file():
|
|
sys.exit(f"Template not found: {template_path}")
|
|
if not main_template_path.is_file():
|
|
sys.exit(f"Main template not found: {main_template_path}")
|
|
latex_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
# From here on, everything printed (stdout and stderr both) also goes
|
|
# to generate_latex.log in latex_dir -- so skip/warning messages
|
|
# survive even when this script's own output gets buried in the
|
|
# surrounding pdflatex/makeindex noise of a full build pipeline.
|
|
log_path = latex_dir / "generate_latex.log"
|
|
log_file = log_path.open("w", encoding="utf-8")
|
|
sys.stdout = _Tee(sys.stdout, log_file)
|
|
sys.stderr = _Tee(sys.stderr, log_file)
|
|
print(f"(full log also being written to {log_path})")
|
|
print()
|
|
|
|
if args.category_order:
|
|
order = [c.strip().upper() for c in args.category_order.split(",")]
|
|
bad = [c for c in order if c not in VALID_CODES]
|
|
if bad:
|
|
sys.exit(f"Unknown category code(s) in --category-order: {bad}")
|
|
else:
|
|
order = list(CATEGORIES.keys())
|
|
|
|
template = template_path.read_text(encoding="utf-8")
|
|
missing = [p for p in PLACEHOLDERS if p not in template]
|
|
if missing:
|
|
print(f"Note: entry template doesn't use these placeholders: {missing}", file=sys.stderr)
|
|
|
|
main_template = main_template_path.read_text(encoding="utf-8")
|
|
|
|
files = sorted(txt_dir.glob(args.pattern))
|
|
if not files:
|
|
sys.exit(f"No files matching {args.pattern} in {txt_dir}")
|
|
|
|
print(f"Found {len(files)} files in {txt_dir}")
|
|
|
|
# code -> list of {"path", "parsed", "ref_key"}
|
|
by_category = OrderedDict((c, []) for c in order)
|
|
skipped = []
|
|
ignored_filename = []
|
|
|
|
for path in files:
|
|
m = FILENAME_RE.match(path.name)
|
|
if not m or m.group("code") not in VALID_CODES:
|
|
ignored_filename.append(path.name)
|
|
continue
|
|
code = m.group("code")
|
|
if code not in by_category:
|
|
# Valid code, but not included in --category-order -- treat
|
|
# like any other skip rather than silently dropping it.
|
|
print(f"SKIP (category {code} not in --category-order): {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
|
|
text = path.read_text(encoding="utf-8")
|
|
parsed = parse_file(text)
|
|
|
|
if parsed is None:
|
|
print(f"SKIP (no recognizable header found): {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
if parsed["style"] == "needs_title":
|
|
print(f"SKIP (needs a title line added -- old two-line header) [{parsed['ref']}]: {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
if parsed["style"] == "no_occasions":
|
|
print(f"SKIP (no Occasion line) [{parsed['ref']}]: {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
if parsed["style"] == "multi_occasion":
|
|
print(f"SKIP (more than one Occasion line -- looks like a BONEYARD-style file) [{parsed['ref']}]: {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
if parsed["style"] == "incomplete":
|
|
print(f"SKIP (occasion with no prayer yet) [{parsed['ref']}]: {path.name}")
|
|
for o in parsed["missing"]:
|
|
print(f" missing: {o}")
|
|
skipped.append(path.name)
|
|
continue
|
|
|
|
by_category[code].append({"path": path, "parsed": parsed, "ref_key": ref_sort_key(parsed["ref"])})
|
|
|
|
if ignored_filename:
|
|
print(f"\nIgnored {len(ignored_filename)} file(s) not matching CODE-CCC.VV(-suffix).txt "
|
|
f"with a valid category (e.g. BONEYARD-*.txt):")
|
|
for name in ignored_filename[:10]:
|
|
print(f" - {name}")
|
|
if len(ignored_filename) > 10:
|
|
print(f" ... and {len(ignored_filename) - 10} more")
|
|
|
|
# Sort each category's entries by Psalm reference.
|
|
for code in by_category:
|
|
by_category[code].sort(key=lambda e: e["ref_key"])
|
|
|
|
print()
|
|
print("Category breakdown (all valid entries found in txt/):")
|
|
total = 0
|
|
for code in order:
|
|
n = len(by_category[code])
|
|
total += n
|
|
print(f" {code:4s} {category_name(code):32s} {n}")
|
|
print(f" {'':4s} {'TOTAL':32s} {total}")
|
|
print()
|
|
|
|
# Flat lookup of every valid entry, keyed by basename, plus the
|
|
# default fallback order (category order, then Psalm ref) -- used
|
|
# either to seed a fresh inputs.tex, or to provisionally place any
|
|
# entry the hand-edited inputs.tex doesn't mention yet.
|
|
entry_by_basename = {}
|
|
default_order = []
|
|
for code in order:
|
|
for item in by_category[code]:
|
|
basename = item["path"].stem
|
|
entry_by_basename[basename] = item
|
|
default_order.append(basename)
|
|
|
|
inputs_path = latex_dir / "inputs.tex"
|
|
code_by_name = {category_name(c): c for c in VALID_CODES}
|
|
|
|
if not inputs_path.exists():
|
|
grouped = OrderedDict(
|
|
(c, [(item["path"].stem, item["parsed"]["entries"][0]["occasion"]) for item in by_category[c]])
|
|
for c in order
|
|
)
|
|
inputs_path.write_text(render_inputs_block(grouped, args.category_macro), encoding="utf-8")
|
|
final_order = list(default_order)
|
|
print(f"No inputs.tex found -- generated a fresh one at {inputs_path} "
|
|
f"from the default category/Psalm order.")
|
|
print("Edit it by hand any time to fine-tune ordering -- future runs will")
|
|
print("leave your edits alone as long as the file exists.")
|
|
else:
|
|
existing_text = inputs_path.read_text(encoding="utf-8")
|
|
parsed_order, mismatches = parse_inputs_tex(existing_text, code_by_name)
|
|
print(f"Found existing {inputs_path} -- honoring its order, NOT regenerating it.")
|
|
print("(delete it if you want a fresh default ordering)")
|
|
|
|
seen = set()
|
|
final_order = []
|
|
stale = []
|
|
for basename in parsed_order:
|
|
if basename in seen:
|
|
print(f" note: {basename} is \\input{{}} more than once in inputs.tex -- using its first occurrence")
|
|
continue
|
|
seen.add(basename)
|
|
if basename not in entry_by_basename:
|
|
stale.append(basename)
|
|
continue
|
|
final_order.append(basename)
|
|
|
|
missing_from_inputs = [b for b in default_order if b not in seen]
|
|
|
|
if stale:
|
|
print(f"\n WARNING -- {len(stale)} \\input{{}} line(s) in inputs.tex point to files "
|
|
f"that no longer exist in {txt_dir}/ (would break the build if left in):")
|
|
for name in stale:
|
|
print(f" - {name}")
|
|
if mismatches:
|
|
print(f"\n note -- {len(mismatches)} entr{'y' if len(mismatches) == 1 else 'ies'} in inputs.tex "
|
|
f"sit under a chapter that doesn't match their own category code:")
|
|
for name, file_code, chapter_code in mismatches:
|
|
print(f" - {name} (file is {file_code}, but listed under {chapter_code}'s chapter)")
|
|
if missing_from_inputs:
|
|
print(f"\n {len(missing_from_inputs)} entr{'y' if len(missing_from_inputs) == 1 else 'ies'} in "
|
|
f"{txt_dir}/ not yet placed in inputs.tex -- added at the end, provisionally numbered "
|
|
f"(add a \\input{{}} line by hand to place properly):")
|
|
for name in missing_from_inputs:
|
|
print(f" - {name}")
|
|
final_order.extend(missing_from_inputs)
|
|
print()
|
|
|
|
number = 0
|
|
for basename in final_order:
|
|
item = entry_by_basename[basename]
|
|
number += 1
|
|
tex = render_entry_tex(number, item["parsed"], template)
|
|
out_path = latex_dir / (basename + ".tex")
|
|
out_path.write_text(tex, encoding="utf-8")
|
|
|
|
main_tex = main_template
|
|
leftover = re.findall(r"\[\[[A-Z]+\]\]", main_tex)
|
|
if leftover:
|
|
print(f"warning: unrecognized placeholder(s) left in main template: {sorted(set(leftover))}", file=sys.stderr)
|
|
(latex_dir / "main.tex").write_text(main_tex, encoding="utf-8")
|
|
|
|
print(f"Wrote {number} entry files to {latex_dir}/ (numbered per inputs.tex order), plus main.tex")
|
|
if skipped:
|
|
print(f"Skipped {len(skipped)} file(s) with parse problems:")
|
|
for name in skipped:
|
|
print(f" - {name}")
|
|
print()
|
|
print("To build: cd into the latex dir and run pdflatex twice, then")
|
|
print("makeindex main, then pdflatex twice more, to resolve the TOC")
|
|
print("and the Psalm-reference index.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |