Books/2026.08-PrayersForDailyLife/classify_occasions.py
2026-09-09 19:59:36 -07:00

422 lines
16 KiB
Python

#!/usr/bin/env python3
"""
classify_occasions.py
One-time (well -- rerunnable) migration script. Reads the old-format
.txt files (each verse with a full header and one or more "Occasion:"
lines, some possibly hand-culled with '%') from an input directory
-- your renamed ./old -- and, for every occasion found, asks the model
which of the 7 fixed categories (see categories.py) it honestly
belongs to. An occasion may land in more than one category; if none
fit, it goes to BONEYARD instead.
IMPORTANT: this script deliberately IGNORES all prior hand-culling.
Every '%Occasion: ...' line is un-commented and treated as live, and
every '%v ... %^' block is unwrapped. The idea is to start the new
category-based structure from the full original set of drafted
occasions, not from whatever survived your old cull pass -- you'll
cull again, fresh, once occasions land in their new category files.
For each (verse, category) assignment, writes a bare target file to
the output directory -- your ./txt -- in the new naming convention:
CCC.VV-CODE.txt e.g. 023.06-WWF.txt
CCC.VV-CODE-b.txt on collision (same verse+category, another
occasion), then -c, -d, ...
containing the new-format header (ref/title/quote) plus ONE
"Occasion: ..." line and no prayer text -- ready for an
updated generate_petitions2.py to write into. The old prayer text is
NOT carried over; these prayers are meant to be regenerated with the
strengthened Pass 2 prompt, not migrated verbatim.
Occasions the model can't honestly place in any of the 7 categories
are written instead to one BONEYARD file per verse:
CCC.VV-BONEYARD.txt
in the OLD multi-occasion style (header + all its boneyard occasions
listed together, no prayers) so you can review, delete, or manually
recategorize them at your leisure. Boneyard files are never picked up
by Pass 2.
Usage:
pip install openai python-dotenv
python classify_occasions.py old txt
python classify_occasions.py old txt --dry-run
python classify_occasions.py old txt --model gpt-5.4
"""
import argparse
import json
import os
import re
import sys
import time
from pathlib import Path
from dotenv import load_dotenv
from categories import CATEGORIES, VALID_CODES, BONEYARD, scope_block
load_dotenv()
API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY")
if not API_KEY:
sys.exit(
"No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) "
"to a .env file in the working directory, or export it in your shell."
)
try:
from openai import OpenAI
except ImportError:
sys.exit("Missing dependency. Run: pip install openai")
client = OpenAI(api_key=API_KEY)
# --- Parsing (old format) ---------------------------------------------------
FULL_HEADER_RE = re.compile(
r"^(?P<ref>##\s*\S+)[ \t]*\n"
r">(?!>)[ \t]*(?P<title>[^\n]*)\n"
r">>[ \t]*(?P<quote>[^\n]*)\n"
)
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$")
def uncull(text: str) -> str:
"""The inverse of strip_comment_lines() elsewhere in this project:
restores everything that was hand-culled instead of dropping it.
- A line '%Occasion: ...' becomes a live 'Occasion: ...' line.
- Standalone '%v' / '%^' marker lines are removed, but everything
between them is kept as-is (it was never itself prefixed with
'%', so no further unwrapping is needed there).
- Any OTHER line starting with '%' is treated as a genuine author
note (not a cull) and is dropped -- it was never content.
"""
lines = text.split("\n")
out = []
for ln in lines:
stripped = ln.strip()
if stripped in ("%v", "%^"):
continue
if stripped.startswith("%Occasion:"):
out.append(ln.lstrip()[1:]) # drop the leading '%' only
continue
if ln.lstrip().startswith("%"):
continue # a genuine note, not content -- drop it
out.append(ln)
return "\n".join(out)
def parse_ref(ref_line: str) -> str:
return ref_line.lstrip("#").strip()
def parse_file(text: str):
"""Returns {"ref", "title", "quote", "occasions": [{"text", "content"}, ...]}
or None. "content" is the prayer text already written under that
occasion (possibly "" if Pass 2 never got to it) -- carried through
so hand-polished prayers survive the category split instead of being
discarded and regenerated."""
text = uncull(text)
stripped = text.lstrip("\n")
m = FULL_HEADER_RE.match(stripped)
if not m:
return None
ref = parse_ref(m.group("ref"))
title = m.group("title").strip()
quote = m.group("quote").strip()
body = text[m.end():]
matches = list(OCCASION_LINE_RE.finditer(body))
occasions = []
seen = set()
for i, mo in enumerate(matches):
occasion_text = mo.group("text").strip()
start = mo.end()
end = matches[i + 1].start() if i + 1 < len(matches) else len(body)
content = body[start:end].strip()
# De-dup exact repeat occasion phrases (can happen if an occasion
# was both live and separately re-added by hand at some point) --
# keep the first occurrence's content.
key = occasion_text.lower()
if key in seen:
continue
seen.add(key)
occasions.append({"text": occasion_text, "content": content})
return {"ref": ref, "title": title, "quote": quote, "occasions": occasions}
def ref_to_filestem(ref: str) -> str:
"""'23:6' -> '023.06'. Verse padded to at least 2 digits (Psalm 119
runs past 99 -- str.zfill only pads up, so 176 stays '176')."""
chapter, verse = ref.split(":")
return f"{int(chapter):03d}.{verse.zfill(2)}"
# --- Classification call ----------------------------------------------------
SYSTEM_PROMPT = (
"You sort draft devotional-prayer occasions into a fixed set of 7 "
"categories for a Christian devotional book, one verse at a time.\n\n"
"The 7 categories, with their scope:\n\n"
f"{scope_block()}\n\n"
"For EACH occasion given, decide which of the 7 category codes it "
"honestly belongs to. Be generous: if an occasion plausibly fits a "
"category, include it, even if it could also fit another -- an "
"occasion may be assigned to more than one category when it "
"genuinely serves both (e.g. \"feeling under attack\" can honestly "
"be both WWF and REL). It is far better to over-include than to "
"force a stretch or drop something useful -- a human will cull by "
"hand later.\n\n"
"Only use \"BONEYARD\" -- and ONLY that, with no other code -- for "
"an occasion that does not honestly fit any of the 7. Do not combine "
"BONEYARD with a real code.\n\n"
"Do not edit, rephrase, or improve the occasion text. Echo it back "
"exactly as given, for matching."
)
USER_PROMPT_TEMPLATE = """\
Verse ({ref}): {quote}
Classify each of the following {count} occasions:
{occasions_list}
Respond ONLY with valid JSON in this exact shape, no other text, one \
entry per occasion above, in the same order:
{{
"classifications": [
{{"occasion": "...", "codes": ["WWF", "REL"]}},
...
]
}}
"""
def build_user_prompt(ref: str, quote: str, occasions: list) -> str:
occasions_list = "\n".join(f"{i}. {o}" for i, o in enumerate(occasions, start=1))
return USER_PROMPT_TEMPLATE.format(
ref=ref, quote=quote, count=len(occasions), occasions_list=occasions_list
)
def classify(ref: str, quote: str, occasions: list, model: str, retries: int = 3):
prompt = build_user_prompt(ref, quote, occasions)
last_err = None
for attempt in range(1, retries + 1):
try:
resp = client.chat.completions.create(
model=model,
messages=[
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": prompt},
],
temperature=0.2,
response_format={"type": "json_object"},
)
data = json.loads(resp.choices[0].message.content)
results = data["classifications"]
if len(results) != len(occasions):
raise ValueError(f"Expected {len(occasions)} results, got {len(results)}")
out = []
for expected, r in zip(occasions, results):
if r.get("occasion", "").strip().lower() != expected.strip().lower():
raise ValueError(
f"Occasion mismatch: expected {expected!r}, got {r.get('occasion')!r}"
)
codes = r.get("codes")
if not isinstance(codes, list) or not codes:
raise ValueError(f"Bad codes for {expected!r}: {codes!r}")
if codes == [BONEYARD] or codes == [BONEYARD.lower()]:
out.append((expected, [BONEYARD]))
continue
bad = [c for c in codes if c not in VALID_CODES]
if bad:
raise ValueError(f"Unknown code(s) {bad} for {expected!r}")
out.append((expected, codes))
return out
except Exception as e: # noqa: BLE001
last_err = e
print(f" attempt {attempt}/{retries} failed: {e}", file=sys.stderr)
time.sleep(1.5 * attempt)
raise RuntimeError(f"Giving up classifying {ref}: {last_err}")
# --- Writing new-format files ------------------------------------------------
def next_free_path(out_dir: Path, base_stem: str) -> Path:
"""base_stem e.g. '023.06-WWF' -> first free of 023.06-WWF.txt,
023.06-WWF-b.txt, 023.06-WWF-c.txt, ..."""
candidate = out_dir / f"{base_stem}.txt"
if not candidate.exists():
return candidate
for suffix in "bcdefghijklmnopqrstuvwxyz":
candidate = out_dir / f"{base_stem}-{suffix}.txt"
if not candidate.exists():
return candidate
raise RuntimeError(f"Too many collisions for {base_stem} -- resolve by hand")
def write_category_file(out_dir: Path, stem: str, code: str, ref: str, title: str,
quote: str, occasion: str, prayer_content: str, dry_run: bool) -> Path:
"""prayer_content is the existing prayer text, if any (carried over
verbatim -- this script never rewrites prayer text). Empty string
means no prayer was ever written for this occasion; the file is left
"pending" in the same sense generate_petitions2.py already expects,
so it will still pick this one up."""
base_stem = f"{stem}-{code}"
path = next_free_path(out_dir, base_stem)
body = f"Occasion: {occasion}"
if prayer_content:
body += f"\n{prayer_content}"
content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n"
if not dry_run:
path.write_text(content, encoding="utf-8")
return path
def write_boneyard_file(out_dir: Path, stem: str, ref: str, title: str, quote: str,
occasions: list, dry_run: bool) -> Path:
"""occasions is a list of {"text", "content"} dicts. Prayer content
(if any) is carried over so nothing hand-written is lost even for
occasions that didn't fit a category."""
path = out_dir / f"{stem}-{BONEYARD}.txt"
# If it already exists (e.g. rerun), append rather than clobber.
existing = []
if path.exists():
old = parse_file(path.read_text(encoding="utf-8"))
if old:
existing = old["occasions"]
existing_keys = {e["text"].lower() for e in existing}
all_occasions = existing + [o for o in occasions if o["text"].lower() not in existing_keys]
blocks = []
for o in all_occasions:
block = f"Occasion: {o['text']}"
if o["content"]:
block += f"\n{o['content']}"
blocks.append(block)
body = "\n\n".join(blocks)
content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n"
if not dry_run:
path.write_text(content, encoding="utf-8")
return path
# --- Main -------------------------------------------------------------------
def main():
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument("in_dir", help="Input directory of old-format .txt files (e.g. old)")
parser.add_argument("out_dir", help="Output directory for new per-category files (e.g. txt)")
parser.add_argument("--pattern", default="*.txt", help="Glob pattern for input files (default: *.txt)")
parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)")
parser.add_argument("--dry-run", action="store_true", help="Show what would happen without writing anything")
args = parser.parse_args()
in_dir = Path(args.in_dir)
out_dir = Path(args.out_dir)
if not in_dir.is_dir():
sys.exit(f"Not a directory: {in_dir}")
if not args.dry_run:
out_dir.mkdir(parents=True, exist_ok=True)
files = sorted(in_dir.glob(args.pattern))
if not files:
sys.exit(f"No files matching {args.pattern} in {in_dir}")
print(f"Found {len(files)} files in {in_dir}")
if args.dry_run:
print("(dry run -- nothing will be written or called)")
print()
total_occasions = 0
total_category_files = 0
total_boneyard = 0
skipped = []
failed = []
for path in files:
text = path.read_text(encoding="utf-8")
parsed = parse_file(text)
if parsed is None:
print(f"SKIP (no recognizable full header): {path.name}")
skipped.append(path.name)
continue
if not parsed["occasions"]:
print(f"SKIP (no occasions found): {path.name}")
skipped.append(path.name)
continue
ref, title, quote = parsed["ref"], parsed["title"], parsed["quote"]
stem = ref_to_filestem(ref)
print(f"{path.name} [Psalm {ref}] ({len(parsed['occasions'])} occasion(s))")
if args.dry_run:
print(" (dry run -- skipping classification call)")
continue
occasion_texts = [o["text"] for o in parsed["occasions"]]
content_by_text = {o["text"]: o["content"] for o in parsed["occasions"]}
try:
results = classify(ref, quote, occasion_texts, args.model)
except Exception as e: # noqa: BLE001
print(f" FAILED: {e}", file=sys.stderr)
failed.append(path.name)
continue
boneyard_occasions = []
for occasion, codes in results:
total_occasions += 1
prayer_content = content_by_text.get(occasion, "")
if codes == [BONEYARD]:
boneyard_occasions.append({"text": occasion, "content": prayer_content})
continue
for code in codes:
out_path = write_category_file(
out_dir, stem, code, ref, title, quote, occasion, prayer_content, args.dry_run
)
total_category_files += 1
has_prayer = " (has prayer)" if prayer_content else " (bare)"
print(f" {code}: {out_path.name}{has_prayer}")
if boneyard_occasions:
out_path = write_boneyard_file(
out_dir, stem, ref, title, quote, boneyard_occasions, args.dry_run
)
total_boneyard += len(boneyard_occasions)
print(f" BONEYARD ({len(boneyard_occasions)}): {out_path.name}")
print()
print(
f"Done. Verses processed: {len(files) - len(skipped) - len(failed)}, "
f"occasions classified: {total_occasions}, "
f"category files written: {total_category_files}, "
f"boneyard occasions: {total_boneyard}"
)
if skipped:
print(f"Skipped {len(skipped)} file(s) (no header/occasions):")
for name in skipped:
print(f" - {name}")
if failed:
print(f"Failed {len(failed)} file(s):")
for name in failed:
print(f" - {name}")
print()
print("Next: hand-review the BONEYARD files, then run the updated")
print("generate_petitions2.py over the new category files in", out_dir)
if __name__ == "__main__":
main()