422 lines
16 KiB
Python
422 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
classify_occasions.py
|
|
|
|
One-time (well -- rerunnable) migration script. Reads the old-format
|
|
.txt files (each verse with a full header and one or more "Occasion:"
|
|
lines, some possibly hand-culled with '%') from an input directory
|
|
-- your renamed ./old -- and, for every occasion found, asks the model
|
|
which of the 7 fixed categories (see categories.py) it honestly
|
|
belongs to. An occasion may land in more than one category; if none
|
|
fit, it goes to BONEYARD instead.
|
|
|
|
IMPORTANT: this script deliberately IGNORES all prior hand-culling.
|
|
Every '%Occasion: ...' line is un-commented and treated as live, and
|
|
every '%v ... %^' block is unwrapped. The idea is to start the new
|
|
category-based structure from the full original set of drafted
|
|
occasions, not from whatever survived your old cull pass -- you'll
|
|
cull again, fresh, once occasions land in their new category files.
|
|
|
|
For each (verse, category) assignment, writes a bare target file to
|
|
the output directory -- your ./txt -- in the new naming convention:
|
|
|
|
CCC.VV-CODE.txt e.g. 023.06-WWF.txt
|
|
CCC.VV-CODE-b.txt on collision (same verse+category, another
|
|
occasion), then -c, -d, ...
|
|
|
|
containing the new-format header (ref/title/quote) plus ONE
|
|
"Occasion: ..." line and no prayer text -- ready for an
|
|
updated generate_petitions2.py to write into. The old prayer text is
|
|
NOT carried over; these prayers are meant to be regenerated with the
|
|
strengthened Pass 2 prompt, not migrated verbatim.
|
|
|
|
Occasions the model can't honestly place in any of the 7 categories
|
|
are written instead to one BONEYARD file per verse:
|
|
|
|
CCC.VV-BONEYARD.txt
|
|
|
|
in the OLD multi-occasion style (header + all its boneyard occasions
|
|
listed together, no prayers) so you can review, delete, or manually
|
|
recategorize them at your leisure. Boneyard files are never picked up
|
|
by Pass 2.
|
|
|
|
Usage:
|
|
pip install openai python-dotenv
|
|
python classify_occasions.py old txt
|
|
python classify_occasions.py old txt --dry-run
|
|
python classify_occasions.py old txt --model gpt-5.4
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
from categories import CATEGORIES, VALID_CODES, BONEYARD, scope_block
|
|
|
|
load_dotenv()
|
|
|
|
API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY")
|
|
if not API_KEY:
|
|
sys.exit(
|
|
"No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) "
|
|
"to a .env file in the working directory, or export it in your shell."
|
|
)
|
|
|
|
try:
|
|
from openai import OpenAI
|
|
except ImportError:
|
|
sys.exit("Missing dependency. Run: pip install openai")
|
|
|
|
client = OpenAI(api_key=API_KEY)
|
|
|
|
# --- Parsing (old format) ---------------------------------------------------
|
|
|
|
FULL_HEADER_RE = re.compile(
|
|
r"^(?P<ref>##\s*\S+)[ \t]*\n"
|
|
r">(?!>)[ \t]*(?P<title>[^\n]*)\n"
|
|
r">>[ \t]*(?P<quote>[^\n]*)\n"
|
|
)
|
|
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$")
|
|
|
|
|
|
def uncull(text: str) -> str:
|
|
"""The inverse of strip_comment_lines() elsewhere in this project:
|
|
restores everything that was hand-culled instead of dropping it.
|
|
|
|
- A line '%Occasion: ...' becomes a live 'Occasion: ...' line.
|
|
- Standalone '%v' / '%^' marker lines are removed, but everything
|
|
between them is kept as-is (it was never itself prefixed with
|
|
'%', so no further unwrapping is needed there).
|
|
- Any OTHER line starting with '%' is treated as a genuine author
|
|
note (not a cull) and is dropped -- it was never content.
|
|
"""
|
|
lines = text.split("\n")
|
|
out = []
|
|
for ln in lines:
|
|
stripped = ln.strip()
|
|
if stripped in ("%v", "%^"):
|
|
continue
|
|
if stripped.startswith("%Occasion:"):
|
|
out.append(ln.lstrip()[1:]) # drop the leading '%' only
|
|
continue
|
|
if ln.lstrip().startswith("%"):
|
|
continue # a genuine note, not content -- drop it
|
|
out.append(ln)
|
|
return "\n".join(out)
|
|
|
|
|
|
def parse_ref(ref_line: str) -> str:
|
|
return ref_line.lstrip("#").strip()
|
|
|
|
|
|
def parse_file(text: str):
|
|
"""Returns {"ref", "title", "quote", "occasions": [{"text", "content"}, ...]}
|
|
or None. "content" is the prayer text already written under that
|
|
occasion (possibly "" if Pass 2 never got to it) -- carried through
|
|
so hand-polished prayers survive the category split instead of being
|
|
discarded and regenerated."""
|
|
text = uncull(text)
|
|
stripped = text.lstrip("\n")
|
|
m = FULL_HEADER_RE.match(stripped)
|
|
if not m:
|
|
return None
|
|
|
|
ref = parse_ref(m.group("ref"))
|
|
title = m.group("title").strip()
|
|
quote = m.group("quote").strip()
|
|
body = text[m.end():]
|
|
|
|
matches = list(OCCASION_LINE_RE.finditer(body))
|
|
occasions = []
|
|
seen = set()
|
|
for i, mo in enumerate(matches):
|
|
occasion_text = mo.group("text").strip()
|
|
start = mo.end()
|
|
end = matches[i + 1].start() if i + 1 < len(matches) else len(body)
|
|
content = body[start:end].strip()
|
|
# De-dup exact repeat occasion phrases (can happen if an occasion
|
|
# was both live and separately re-added by hand at some point) --
|
|
# keep the first occurrence's content.
|
|
key = occasion_text.lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
occasions.append({"text": occasion_text, "content": content})
|
|
|
|
return {"ref": ref, "title": title, "quote": quote, "occasions": occasions}
|
|
|
|
|
|
def ref_to_filestem(ref: str) -> str:
|
|
"""'23:6' -> '023.06'. Verse padded to at least 2 digits (Psalm 119
|
|
runs past 99 -- str.zfill only pads up, so 176 stays '176')."""
|
|
chapter, verse = ref.split(":")
|
|
return f"{int(chapter):03d}.{verse.zfill(2)}"
|
|
|
|
|
|
# --- Classification call ----------------------------------------------------
|
|
|
|
SYSTEM_PROMPT = (
|
|
"You sort draft devotional-prayer occasions into a fixed set of 7 "
|
|
"categories for a Christian devotional book, one verse at a time.\n\n"
|
|
"The 7 categories, with their scope:\n\n"
|
|
f"{scope_block()}\n\n"
|
|
"For EACH occasion given, decide which of the 7 category codes it "
|
|
"honestly belongs to. Be generous: if an occasion plausibly fits a "
|
|
"category, include it, even if it could also fit another -- an "
|
|
"occasion may be assigned to more than one category when it "
|
|
"genuinely serves both (e.g. \"feeling under attack\" can honestly "
|
|
"be both WWF and REL). It is far better to over-include than to "
|
|
"force a stretch or drop something useful -- a human will cull by "
|
|
"hand later.\n\n"
|
|
"Only use \"BONEYARD\" -- and ONLY that, with no other code -- for "
|
|
"an occasion that does not honestly fit any of the 7. Do not combine "
|
|
"BONEYARD with a real code.\n\n"
|
|
"Do not edit, rephrase, or improve the occasion text. Echo it back "
|
|
"exactly as given, for matching."
|
|
)
|
|
|
|
USER_PROMPT_TEMPLATE = """\
|
|
Verse ({ref}): {quote}
|
|
|
|
Classify each of the following {count} occasions:
|
|
|
|
{occasions_list}
|
|
|
|
Respond ONLY with valid JSON in this exact shape, no other text, one \
|
|
entry per occasion above, in the same order:
|
|
{{
|
|
"classifications": [
|
|
{{"occasion": "...", "codes": ["WWF", "REL"]}},
|
|
...
|
|
]
|
|
}}
|
|
"""
|
|
|
|
|
|
def build_user_prompt(ref: str, quote: str, occasions: list) -> str:
|
|
occasions_list = "\n".join(f"{i}. {o}" for i, o in enumerate(occasions, start=1))
|
|
return USER_PROMPT_TEMPLATE.format(
|
|
ref=ref, quote=quote, count=len(occasions), occasions_list=occasions_list
|
|
)
|
|
|
|
|
|
def classify(ref: str, quote: str, occasions: list, model: str, retries: int = 3):
|
|
prompt = build_user_prompt(ref, quote, occasions)
|
|
last_err = None
|
|
for attempt in range(1, retries + 1):
|
|
try:
|
|
resp = client.chat.completions.create(
|
|
model=model,
|
|
messages=[
|
|
{"role": "system", "content": SYSTEM_PROMPT},
|
|
{"role": "user", "content": prompt},
|
|
],
|
|
temperature=0.2,
|
|
response_format={"type": "json_object"},
|
|
)
|
|
data = json.loads(resp.choices[0].message.content)
|
|
results = data["classifications"]
|
|
if len(results) != len(occasions):
|
|
raise ValueError(f"Expected {len(occasions)} results, got {len(results)}")
|
|
|
|
out = []
|
|
for expected, r in zip(occasions, results):
|
|
if r.get("occasion", "").strip().lower() != expected.strip().lower():
|
|
raise ValueError(
|
|
f"Occasion mismatch: expected {expected!r}, got {r.get('occasion')!r}"
|
|
)
|
|
codes = r.get("codes")
|
|
if not isinstance(codes, list) or not codes:
|
|
raise ValueError(f"Bad codes for {expected!r}: {codes!r}")
|
|
if codes == [BONEYARD] or codes == [BONEYARD.lower()]:
|
|
out.append((expected, [BONEYARD]))
|
|
continue
|
|
bad = [c for c in codes if c not in VALID_CODES]
|
|
if bad:
|
|
raise ValueError(f"Unknown code(s) {bad} for {expected!r}")
|
|
out.append((expected, codes))
|
|
return out
|
|
except Exception as e: # noqa: BLE001
|
|
last_err = e
|
|
print(f" attempt {attempt}/{retries} failed: {e}", file=sys.stderr)
|
|
time.sleep(1.5 * attempt)
|
|
raise RuntimeError(f"Giving up classifying {ref}: {last_err}")
|
|
|
|
|
|
# --- Writing new-format files ------------------------------------------------
|
|
|
|
def next_free_path(out_dir: Path, base_stem: str) -> Path:
|
|
"""base_stem e.g. '023.06-WWF' -> first free of 023.06-WWF.txt,
|
|
023.06-WWF-b.txt, 023.06-WWF-c.txt, ..."""
|
|
candidate = out_dir / f"{base_stem}.txt"
|
|
if not candidate.exists():
|
|
return candidate
|
|
for suffix in "bcdefghijklmnopqrstuvwxyz":
|
|
candidate = out_dir / f"{base_stem}-{suffix}.txt"
|
|
if not candidate.exists():
|
|
return candidate
|
|
raise RuntimeError(f"Too many collisions for {base_stem} -- resolve by hand")
|
|
|
|
|
|
def write_category_file(out_dir: Path, stem: str, code: str, ref: str, title: str,
|
|
quote: str, occasion: str, prayer_content: str, dry_run: bool) -> Path:
|
|
"""prayer_content is the existing prayer text, if any (carried over
|
|
verbatim -- this script never rewrites prayer text). Empty string
|
|
means no prayer was ever written for this occasion; the file is left
|
|
"pending" in the same sense generate_petitions2.py already expects,
|
|
so it will still pick this one up."""
|
|
base_stem = f"{stem}-{code}"
|
|
path = next_free_path(out_dir, base_stem)
|
|
body = f"Occasion: {occasion}"
|
|
if prayer_content:
|
|
body += f"\n{prayer_content}"
|
|
content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n"
|
|
if not dry_run:
|
|
path.write_text(content, encoding="utf-8")
|
|
return path
|
|
|
|
|
|
def write_boneyard_file(out_dir: Path, stem: str, ref: str, title: str, quote: str,
|
|
occasions: list, dry_run: bool) -> Path:
|
|
"""occasions is a list of {"text", "content"} dicts. Prayer content
|
|
(if any) is carried over so nothing hand-written is lost even for
|
|
occasions that didn't fit a category."""
|
|
path = out_dir / f"{stem}-{BONEYARD}.txt"
|
|
# If it already exists (e.g. rerun), append rather than clobber.
|
|
existing = []
|
|
if path.exists():
|
|
old = parse_file(path.read_text(encoding="utf-8"))
|
|
if old:
|
|
existing = old["occasions"]
|
|
existing_keys = {e["text"].lower() for e in existing}
|
|
all_occasions = existing + [o for o in occasions if o["text"].lower() not in existing_keys]
|
|
|
|
blocks = []
|
|
for o in all_occasions:
|
|
block = f"Occasion: {o['text']}"
|
|
if o["content"]:
|
|
block += f"\n{o['content']}"
|
|
blocks.append(block)
|
|
body = "\n\n".join(blocks)
|
|
content = f"## {ref}\n> {title}\n>> {quote}\n\n{body}\n"
|
|
if not dry_run:
|
|
path.write_text(content, encoding="utf-8")
|
|
return path
|
|
|
|
|
|
# --- Main -------------------------------------------------------------------
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
|
)
|
|
parser.add_argument("in_dir", help="Input directory of old-format .txt files (e.g. old)")
|
|
parser.add_argument("out_dir", help="Output directory for new per-category files (e.g. txt)")
|
|
parser.add_argument("--pattern", default="*.txt", help="Glob pattern for input files (default: *.txt)")
|
|
parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)")
|
|
parser.add_argument("--dry-run", action="store_true", help="Show what would happen without writing anything")
|
|
args = parser.parse_args()
|
|
|
|
in_dir = Path(args.in_dir)
|
|
out_dir = Path(args.out_dir)
|
|
if not in_dir.is_dir():
|
|
sys.exit(f"Not a directory: {in_dir}")
|
|
if not args.dry_run:
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
files = sorted(in_dir.glob(args.pattern))
|
|
if not files:
|
|
sys.exit(f"No files matching {args.pattern} in {in_dir}")
|
|
|
|
print(f"Found {len(files)} files in {in_dir}")
|
|
if args.dry_run:
|
|
print("(dry run -- nothing will be written or called)")
|
|
print()
|
|
|
|
total_occasions = 0
|
|
total_category_files = 0
|
|
total_boneyard = 0
|
|
skipped = []
|
|
failed = []
|
|
|
|
for path in files:
|
|
text = path.read_text(encoding="utf-8")
|
|
parsed = parse_file(text)
|
|
if parsed is None:
|
|
print(f"SKIP (no recognizable full header): {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
if not parsed["occasions"]:
|
|
print(f"SKIP (no occasions found): {path.name}")
|
|
skipped.append(path.name)
|
|
continue
|
|
|
|
ref, title, quote = parsed["ref"], parsed["title"], parsed["quote"]
|
|
stem = ref_to_filestem(ref)
|
|
print(f"{path.name} [Psalm {ref}] ({len(parsed['occasions'])} occasion(s))")
|
|
|
|
if args.dry_run:
|
|
print(" (dry run -- skipping classification call)")
|
|
continue
|
|
|
|
occasion_texts = [o["text"] for o in parsed["occasions"]]
|
|
content_by_text = {o["text"]: o["content"] for o in parsed["occasions"]}
|
|
|
|
try:
|
|
results = classify(ref, quote, occasion_texts, args.model)
|
|
except Exception as e: # noqa: BLE001
|
|
print(f" FAILED: {e}", file=sys.stderr)
|
|
failed.append(path.name)
|
|
continue
|
|
|
|
boneyard_occasions = []
|
|
for occasion, codes in results:
|
|
total_occasions += 1
|
|
prayer_content = content_by_text.get(occasion, "")
|
|
if codes == [BONEYARD]:
|
|
boneyard_occasions.append({"text": occasion, "content": prayer_content})
|
|
continue
|
|
for code in codes:
|
|
out_path = write_category_file(
|
|
out_dir, stem, code, ref, title, quote, occasion, prayer_content, args.dry_run
|
|
)
|
|
total_category_files += 1
|
|
has_prayer = " (has prayer)" if prayer_content else " (bare)"
|
|
print(f" {code}: {out_path.name}{has_prayer}")
|
|
|
|
if boneyard_occasions:
|
|
out_path = write_boneyard_file(
|
|
out_dir, stem, ref, title, quote, boneyard_occasions, args.dry_run
|
|
)
|
|
total_boneyard += len(boneyard_occasions)
|
|
print(f" BONEYARD ({len(boneyard_occasions)}): {out_path.name}")
|
|
|
|
print()
|
|
print(
|
|
f"Done. Verses processed: {len(files) - len(skipped) - len(failed)}, "
|
|
f"occasions classified: {total_occasions}, "
|
|
f"category files written: {total_category_files}, "
|
|
f"boneyard occasions: {total_boneyard}"
|
|
)
|
|
if skipped:
|
|
print(f"Skipped {len(skipped)} file(s) (no header/occasions):")
|
|
for name in skipped:
|
|
print(f" - {name}")
|
|
if failed:
|
|
print(f"Failed {len(failed)} file(s):")
|
|
for name in failed:
|
|
print(f" - {name}")
|
|
print()
|
|
print("Next: hand-review the BONEYARD files, then run the updated")
|
|
print("generate_petitions2.py over the new category files in", out_dir)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|