#!/usr/bin/env python3 """ classify_occasions.py One-time (well -- rerunnable) migration script. Reads the old-format .txt files (each verse with a full header and one or more "Occasion:" lines, some possibly hand-culled with '%') from an input directory -- your renamed ./old -- and, for every occasion found, asks the model which of the 7 fixed categories (see categories.py) it honestly belongs to. An occasion may land in more than one category; if none fit, it goes to BONEYARD instead. IMPORTANT: this script deliberately IGNORES all prior hand-culling. Every '%Occasion: ...' line is un-commented and treated as live, and every '%v ... %^' block is unwrapped. The idea is to start the new category-based structure from the full original set of drafted occasions, not from whatever survived your old cull pass -- you'll cull again, fresh, once occasions land in their new category files. For each (verse, category) assignment, writes a bare target file to the output directory -- your ./txt -- in the new naming convention: CCC.VV-CODE.txt e.g. 023.06-WWF.txt CCC.VV-CODE-b.txt on collision (same verse+category, another occasion), then -c, -d, ... containing the new-format header (ref/title/quote) plus ONE "Occasion: ..." line and no prayer text -- ready for an updated generate_petitions2.py to write into. The old prayer text is NOT carried over; these prayers are meant to be regenerated with the strengthened Pass 2 prompt, not migrated verbatim. Occasions the model can't honestly place in any of the 7 categories are written instead to one BONEYARD file per verse: CCC.VV-BONEYARD.txt in the OLD multi-occasion style (header + all its boneyard occasions listed together, no prayers) so you can review, delete, or manually recategorize them at your leisure. Boneyard files are never picked up by Pass 2. Usage: pip install openai python-dotenv python classify_occasions.py old txt python classify_occasions.py old txt --dry-run python classify_occasions.py old txt --model gpt-5.4 """ import argparse import json import os import re import sys import time from pathlib import Path from dotenv import load_dotenv from categories import CATEGORIES, VALID_CODES, BONEYARD, scope_block load_dotenv() API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY") if not API_KEY: sys.exit( "No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) " "to a .env file in the working directory, or export it in your shell." ) try: from openai import OpenAI except ImportError: sys.exit("Missing dependency. Run: pip install openai") client = OpenAI(api_key=API_KEY) # --- Parsing (old format) --------------------------------------------------- FULL_HEADER_RE = re.compile( r"^(?P##\s*\S+)[ \t]*\n" r">(?!>)[ \t]*(?P[^\n]*)\n" r">>[ \t]*(?P<quote>[^\n]*)\n" ) OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$") def uncull(text: str) -> str: """The inverse of strip_comment_lines() elsewhere in this project: restores everything that was hand-culled instead of dropping it. - A line '%Occasion: ...' becomes a live 'Occasion: ...' line. - Standalone '%v' / '%^' marker lines are removed, but everything between them is kept as-is (it was never itself prefixed with '%', so no further unwrapping is needed there). - Any OTHER line starting with '%' is treated as a genuine author note (not a cull) and is dropped -- it was never content. """ lines = text.split("\n") out = [] for ln in lines: stripped = ln.strip() if stripped in ("%v", "%^"): continue if stripped.startswith("%Occasion:"): out.append(ln.lstrip()[1:]) # drop the leading '%' only continue if ln.lstrip().startswith("%"): continue # a genuine note, not content -- drop it out.append(ln) return "\n".join(out) def parse_ref(ref_line: str) -> str: return ref_line.lstrip("#").strip() def parse_file(text: str): """Returns {"ref", "title", "quote", "occasions": [{"text", "content"}, ...]} or None. "content" is the prayer text already written under that occasion (possibly "" if Pass 2 never got to it) -- carried through so hand-polished prayers survive the category split instead of being discarded and regenerated.""" text = uncull(text) stripped = text.lstrip("\n") m = FULL_HEADER_RE.match(stripped) if not m: return None ref = parse_ref(m.group("ref")) title = m.group("title").strip() quote = m.group("quote").strip() body = text[m.end():] matches = list(OCCASION_LINE_RE.finditer(body)) occasions = [] seen = set() for i, mo in enumerate(matches): occasion_text = mo.group("text").strip() start = mo.end() end = matches[i + 1].start() if i + 1 < len(matches) else len(body) content = body[start:end].strip() # De-dup exact repeat occasion phrases (can happen if an occasion # was both live and separately re-added by hand at some point) -- # keep the first occurrence's content. key = occasion_text.lower() if key in seen: continue seen.add(key) occasions.append({"text": occasion_text, "content": content}) return {"ref": ref, "title": title, "quote": quote, "occasions": occasions} def ref_to_filestem(ref: str) -> str: """'23:6' -> '023.06'. Verse padded to at least 2 digits (Psalm 119 runs past 99 -- str.zfill only pads up, so 176 stays '176').""" chapter, verse = ref.split(":") return f"{int(chapter):03d}.{verse.zfill(2)}" # --- Classification call ---------------------------------------------------- SYSTEM_PROMPT = ( "You sort draft devotional-prayer occasions into a fixed set of 7 " "categories for a Christian devotional book, one verse at a time.\n\n" "The 7 categories, with their scope:\n\n" f"{scope_block()}\n\n" "For EACH occasion given, decide which of the 7 category codes it " "honestly belongs to. Be generous: if an occasion plausibly fits a " "category, include it, even if it could also fit another -- an " "occasion may be assigned to more than one category when it " "genuinely serves both (e.g. \"feeling under attack\" can honestly " "be both WWF and REL). It is far better to over-include than to " "force a stretch or drop something useful -- a human will cull by " "hand later.\n\n" "Only use \"BONEYARD\" -- and ONLY that, with no other code -- for " "an occasion that does not honestly fit any of the 7. Do not combine " "BONEYARD with a real code.\n\n" "Do not edit, rephrase, or improve the occasion text. Echo it back " "exactly as given, for matching." ) USER_PROMPT_TEMPLATE = """\ Verse ({ref}): {quote} Classify each of the following {count} occasions: {occasions_list} Respond ONLY with valid JSON in this exact shape, no other text, one \ entry per occasion above, in the same order: {{ "classifications": [ {{"occasion": "...", "codes": ["WWF", "REL"]}}, ... ] }} """ def build_user_prompt(ref: str, quote: str, occasions: list) -> str: occasions_list = "\n".join(f"{i}. {o}" for i, o in enumerate(occasions, start=1)) return USER_PROMPT_TEMPLATE.format( ref=ref, quote=quote, count=len(occasions), occasions_list=occasions_list ) def classify(ref: str, quote: str, occasions: list, model: str, retries: int = 3): prompt = build_user_prompt(ref, quote, occasions) last_err = None for attempt in range(1, retries + 1): try: resp = client.chat.completions.create( model=model, messages=[ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": prompt}, ], temperature=0.2, response_format={"type": "json_object"}, ) data = json.loads(resp.choices[0].message.content) results = data["classifications"] if len(results) != len(occasions): raise ValueError(f"Expected {len(occasions)} results, got {len(results)}") out = [] for expected, r in zip(occasions, results): if r.get("occasion", "").strip().lower() != expected.strip().lower(): raise ValueError( f"Occasion mismatch: expected {expected!r}, got {r.get('occasion')!r}" ) codes = r.get("codes") if not isinstance(codes, list) or not codes: raise ValueError(f"Bad codes for {expected!r}: {codes!r}") if codes == [BONEYARD] or codes == [BONEYARD.lower()]: out.append((expected, [BONEYARD])) continue bad = [c for c in codes if c not in VALID_CODES] if bad: raise ValueError(f"Unknown code(s) {bad} for {expected!r}") out.append((expected, codes)) return out except Exception as e: # noqa: BLE001 last_err = e print(f" attempt {attempt}/{retries} failed: {e}", file=sys.stderr) time.sleep(1.5 * attempt) raise RuntimeError(f"Giving up classifying {ref}: {last_err}") # --- Writing new-format files ------------------------------------------------ def next_free_path(out_dir: Path, base_stem: str) -> Path: """base_stem e.g. '023.06-WWF' -> first free of 023.06-WWF.txt, 023.06-WWF-b.txt, 023.06-WWF-c.txt, ...""" candidate = out_dir / f"{base_stem}.txt" if not candidate.exists(): return candidate for suffix in "bcdefghijklmnopqrstuvwxyz": candidate = out_dir / f"{base_stem}-{suffix}.txt" if not candidate.exists(): return candidate raise RuntimeError(f"Too many collisions for {base_stem} -- resolve by hand") def write_category_file(out_dir: Path, stem: str, code: str, ref: str, title: str, quote: str, occasion: str, prayer_content: str, dry_run: bool) -> Path: """prayer_content is the existing prayer text, if any (carried over verbatim -- this script never rewrites prayer text). Empty string means no prayer was ever written for this occasion; the file is left "pending" in the same sense generate_petitions2.py already expects, so it will still pick this one up.""" base_stem = f"{stem}-{code}" path = next_free_path(out_dir, base_stem) body = f"Occasion: {occasion}" if prayer_content: body += f"\n{prayer_content}" content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n" if not dry_run: path.write_text(content, encoding="utf-8") return path def write_boneyard_file(out_dir: Path, stem: str, ref: str, title: str, quote: str, occasions: list, dry_run: bool) -> Path: """occasions is a list of {"text", "content"} dicts. Prayer content (if any) is carried over so nothing hand-written is lost even for occasions that didn't fit a category.""" path = out_dir / f"{stem}-{BONEYARD}.txt" # If it already exists (e.g. rerun), append rather than clobber. existing = [] if path.exists(): old = parse_file(path.read_text(encoding="utf-8")) if old: existing = old["occasions"] existing_keys = {e["text"].lower() for e in existing} all_occasions = existing + [o for o in occasions if o["text"].lower() not in existing_keys] blocks = [] for o in all_occasions: block = f"Occasion: {o['text']}" if o["content"]: block += f"\n{o['content']}" blocks.append(block) body = "\n\n".join(blocks) content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n" if not dry_run: path.write_text(content, encoding="utf-8") return path # --- Main ------------------------------------------------------------------- def main(): parser = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument("in_dir", help="Input directory of old-format .txt files (e.g. old)") parser.add_argument("out_dir", help="Output directory for new per-category files (e.g. txt)") parser.add_argument("--pattern", default="*.txt", help="Glob pattern for input files (default: *.txt)") parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)") parser.add_argument("--dry-run", action="store_true", help="Show what would happen without writing anything") args = parser.parse_args() in_dir = Path(args.in_dir) out_dir = Path(args.out_dir) if not in_dir.is_dir(): sys.exit(f"Not a directory: {in_dir}") if not args.dry_run: out_dir.mkdir(parents=True, exist_ok=True) files = sorted(in_dir.glob(args.pattern)) if not files: sys.exit(f"No files matching {args.pattern} in {in_dir}") print(f"Found {len(files)} files in {in_dir}") if args.dry_run: print("(dry run -- nothing will be written or called)") print() total_occasions = 0 total_category_files = 0 total_boneyard = 0 skipped = [] failed = [] for path in files: text = path.read_text(encoding="utf-8") parsed = parse_file(text) if parsed is None: print(f"SKIP (no recognizable full header): {path.name}") skipped.append(path.name) continue if not parsed["occasions"]: print(f"SKIP (no occasions found): {path.name}") skipped.append(path.name) continue ref, title, quote = parsed["ref"], parsed["title"], parsed["quote"] stem = ref_to_filestem(ref) print(f"{path.name} [Psalm {ref}] ({len(parsed['occasions'])} occasion(s))") if args.dry_run: print(" (dry run -- skipping classification call)") continue occasion_texts = [o["text"] for o in parsed["occasions"]] content_by_text = {o["text"]: o["content"] for o in parsed["occasions"]} try: results = classify(ref, quote, occasion_texts, args.model) except Exception as e: # noqa: BLE001 print(f" FAILED: {e}", file=sys.stderr) failed.append(path.name) continue boneyard_occasions = [] for occasion, codes in results: total_occasions += 1 prayer_content = content_by_text.get(occasion, "") if codes == [BONEYARD]: boneyard_occasions.append({"text": occasion, "content": prayer_content}) continue for code in codes: out_path = write_category_file( out_dir, stem, code, ref, title, quote, occasion, prayer_content, args.dry_run ) total_category_files += 1 has_prayer = " (has prayer)" if prayer_content else " (bare)" print(f" {code}: {out_path.name}{has_prayer}") if boneyard_occasions: out_path = write_boneyard_file( out_dir, stem, ref, title, quote, boneyard_occasions, args.dry_run ) total_boneyard += len(boneyard_occasions) print(f" BONEYARD ({len(boneyard_occasions)}): {out_path.name}") print() print( f"Done. Verses processed: {len(files) - len(skipped) - len(failed)}, " f"occasions classified: {total_occasions}, " f"category files written: {total_category_files}, " f"boneyard occasions: {total_boneyard}" ) if skipped: print(f"Skipped {len(skipped)} file(s) (no header/occasions):") for name in skipped: print(f" - {name}") if failed: print(f"Failed {len(failed)} file(s):") for name in failed: print(f" - {name}") print() print("Next: hand-review the BONEYARD files, then run the updated") print("generate_petitions2.py over the new category files in", out_dir) if __name__ == "__main__": main()