#!/usr/bin/env python3 """ score_and_select.py Run this after classify_occasions.py (and, ideally, revise_prayers.py) have produced a large pool of finished, per-category prayer files in ./txt (named CODE-CCC.VV.txt / CODE-CCC.VV-b.txt per rename_to_code_first.sh). This script narrows that pool down to a manageable shortlist per category for final human selection. Steps, in order: 1. DELETE any file with NO prayer content at all (bare occasion, never written). These are backlog, not candidates. 2. SCORE every remaining candidate, batched per category, against a fixed rubric (concreteness/imagery, single governing image, absence of known stock-phrase problems, genuine category fit). 3. DE-DUPLICATE near-identical occasions within a category (e.g. five different verses all drafted as "fear of the future") -- among a near-duplicate group, only the highest scorer is kept as a candidate for selection; the rest are treated as excluded. 4. SELECT the top --n-per-category (default 25) per category, with a verse cap (default 1 prayer per verse) so the shortlist doesn't cluster on a few popular verses. The cap is relaxed automatically, one step at a time, if a category can't otherwise reach quota. 5. MOVE every file that wasn't selected -- whether cut for score, redundancy, or the verse cap -- into ./txt/boneyard/ (created if needed). Nothing scored is ever deleted; only step 1's truly-empty files are. 6. WRITE selection_report.txt: for each category, the selected files ranked by score with their ref and occasion, plus a summary of how many candidates existed and whether the category fell short of quota (a signal to boost that category via generate_petitions1.py later). Files whose code prefix isn't one of the 7 valid categories (i.e. the BONEYARD-CCC.VV.txt holding-pen files from classify_occasions.py) are left alone entirely -- this script only touches classified files. Usage: pip install openai python-dotenv python score_and_select.py txt python score_and_select.py txt --dry-run python score_and_select.py txt --n-per-category 25 --verse-cap 1 """ import argparse import difflib import json import os import re import shutil import sys import time from collections import defaultdict from pathlib import Path from dotenv import load_dotenv from categories import CATEGORIES, VALID_CODES, scope_block load_dotenv() API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY") if not API_KEY: sys.exit( "No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) " "to a .env file in the working directory, or export it in your shell." ) try: from openai import OpenAI except ImportError: sys.exit("Missing dependency. Run: pip install openai") client = OpenAI(api_key=API_KEY) # --- Parsing ----------------------------------------------------------- FULL_HEADER_RE = re.compile( r"^(?P##\s*\S+)[ \t]*\n" r">(?!>)[ \t]*(?P[^\n]*)\n" r">>[ \t]*(?P<quote>[^\n]*)\n" ) OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$") FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<verse>\d{3}\.\d{2,3})(?:-[a-z])?\.txt$") def parse_file(text: str): """Single-occasion new-format file -> {"ref","title","quote","occasion","content"} or None. Multi-occasion (boneyard-style) or malformed files return None.""" stripped = text.lstrip("\n") m = FULL_HEADER_RE.match(stripped) if not m: return None ref = m.group("ref").lstrip("#").strip() title = m.group("title").strip() quote = m.group("quote").strip() body = text[m.end():] matches = list(OCCASION_LINE_RE.finditer(body)) if len(matches) != 1: return None mo = matches[0] occasion = mo.group("text").strip() content = body[mo.end():].strip() return {"ref": ref, "title": title, "quote": quote, "occasion": occasion, "content": content} # --- Scoring rubric & call ---------------------------------------------- RUBRIC = ( "Score each prayer 0-10 (integer) on overall quality as a candidate " "for a finished devotional book, weighing:\n" " - Concreteness: real images and specifics, not generic abstraction.\n" " - A single governing image system, held all the way through, not " "mixed metaphors.\n" " - Freedom from tired stock phrasing (e.g. 'I know how fast/quickly...', " "'the phone keeps ringing', vague summarizing lines like 'that thin life').\n" " - Genuine fit to its stated category -- not just technically related, " "but actually about that occasion.\n" " - Craft: does the ending land, is the voice consistent, is it free of " "typos/grammar errors.\n" "A 9-10 is publication-ready as-is. A 5-6 has a real idea but needs " "editing. Below 4 is generic filler or a poor category fit." ) SYSTEM_PROMPT = ( "You are scoring a batch of devotional prayers, all written for the same " "category, to help an author pick the strongest ~25 out of a much larger " "pool for a finished book.\n\n" f"Category: {{category_name}} -- {{category_scope}}\n\n" f"{RUBRIC}\n\n" "Echo the ref and occasion back exactly as given, for matching." ) USER_PROMPT_TEMPLATE = """\ Score each of the following {count} prayers: {items} Respond ONLY with valid JSON in this exact shape, one entry per prayer \ above, in the same order: {{ "scores": [ {{"ref": "...", "occasion": "...", "score": 7, "note": "one short phrase"}}, ... ] }} """ def build_batch_prompt(items: list) -> str: blocks = [] for i, it in enumerate(items, start=1): blocks.append( f"{i}. Ref: {it['ref']}\n Occasion: {it['occasion']}\n " f"Text:\n {it['content'].replace(chr(10), chr(10) + ' ')}" ) return USER_PROMPT_TEMPLATE.format(count=len(items), items="\n\n".join(blocks)) def score_batch(code: str, items: list, model: str, retries: int = 3): """items: list of {ref, occasion, content, ...}. Returns list of (score:int, note:str) aligned to items, in order.""" name, scope = CATEGORIES[code] system = SYSTEM_PROMPT.format(category_name=name, category_scope=scope) prompt = build_batch_prompt(items) last_err = None for attempt in range(1, retries + 1): try: resp = client.chat.completions.create( model=model, messages=[ {"role": "system", "content": system}, {"role": "user", "content": prompt}, ], temperature=0.2, response_format={"type": "json_object"}, ) data = json.loads(resp.choices[0].message.content) results = data["scores"] if len(results) != len(items): raise ValueError(f"Expected {len(items)} scores, got {len(results)}") out = [] for expected, r in zip(items, results): if r.get("ref", "").strip() != expected["ref"].strip(): raise ValueError(f"Ref mismatch: expected {expected['ref']!r}, got {r.get('ref')!r}") score = r.get("score") if not isinstance(score, (int, float)): raise ValueError(f"Bad score for {expected['ref']}: {score!r}") out.append((int(round(score)), r.get("note", ""))) return out except Exception as e: # noqa: BLE001 last_err = e print(f" batch attempt {attempt}/{retries} failed: {e}", file=sys.stderr) time.sleep(1.5 * attempt) raise RuntimeError(f"Giving up scoring a batch for {code}: {last_err}") def chunked(seq, size): for i in range(0, len(seq), size): yield seq[i:i + size] # --- Dedup --------------------------------------------------------------- def dedup_group_indices(candidates: list, threshold: float) -> list: """candidates: list of dicts with 'occasion' and 'score'. Returns the list of indices to KEEP -- one per near-duplicate cluster (the highest scorer), plus every non-duplicate candidate untouched.""" n = len(candidates) excluded = set() for i in range(n): if i in excluded: continue for j in range(i + 1, n): if j in excluded: continue ratio = difflib.SequenceMatcher( None, candidates[i]["occasion"].lower(), candidates[j]["occasion"].lower() ).ratio() if ratio >= threshold: # keep the higher scorer, exclude the other loser = j if candidates[i]["score"] >= candidates[j]["score"] else i excluded.add(loser) if loser == i: break # i is out; stop comparing it further return [i for i in range(n) if i not in excluded] # --- Selection ------------------------------------------------------------- def ref_to_verse_key(ref: str) -> str: return ref.strip() def select_top_n(candidates: list, n: int, verse_cap: int) -> list: """candidates: list of dicts with 'score' and 'ref', sorted by score descending on entry (caller's responsibility, but we sort here too to be safe). Greedily selects up to n, enforcing at most verse_cap picks per verse ref, relaxing the cap by 1 if quota can't be met.""" ordered = sorted(candidates, key=lambda c: c["score"], reverse=True) cap = verse_cap while True: counts = defaultdict(int) selected = [] for c in ordered: key = ref_to_verse_key(c["ref"]) if counts[key] < cap: selected.append(c) counts[key] += 1 if len(selected) >= n: break if len(selected) >= n or cap >= len(ordered): return selected[:n] cap += 1 # relax and try again # --- Main ------------------------------------------------------------------- def main(): parser = argparse.ArgumentParser( description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument("directory", help="Directory of classified per-category .txt files (e.g. txt)") parser.add_argument("--n-per-category", type=int, default=25, help="Shortlist size per category (default: 25)") parser.add_argument("--verse-cap", type=int, default=1, help="Max prayers per verse in the shortlist, before auto-relaxing (default: 1)") parser.add_argument("--dedup-threshold", type=float, default=0.82, help="Occasion-text similarity ratio to treat as a duplicate (default: 0.82)") parser.add_argument("--batch-size", type=int, default=20, help="Prayers per scoring API call (default: 20)") parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)") parser.add_argument("--dry-run", action="store_true", help="Report what would happen; delete/move/write nothing") args = parser.parse_args() directory = Path(args.directory) if not directory.is_dir(): sys.exit(f"Not a directory: {directory}") boneyard_dir = directory / "boneyard" files = sorted(directory.glob("*.txt")) if not files: sys.exit(f"No .txt files in {directory}") print(f"Found {len(files)} files in {directory}") if args.dry_run: print("(dry run -- nothing will be deleted, moved, or written)") print() by_category = defaultdict(list) # code -> list of candidate dicts (with 'path') deleted = [] ignored = [] for path in files: m = FILENAME_RE.match(path.name) if not m or m.group("code") not in VALID_CODES: ignored.append(path.name) # BONEYARD-*.txt or anything unrecognized continue code = m.group("code") text = path.read_text(encoding="utf-8") parsed = parse_file(text) if parsed is None: ignored.append(path.name) continue if not parsed["content"]: deleted.append(path) continue parsed["path"] = path by_category[code].append(parsed) print(f"Empty (no prayer written): {len(deleted)} -- will be deleted") print(f"Ignored (unrecognized name / boneyard holding-pen / unparseable): {len(ignored)}") for code in VALID_CODES: print(f" {code}: {len(by_category.get(code, []))} candidate(s)") print() if not args.dry_run: for path in deleted: path.unlink() boneyard_dir.mkdir(exist_ok=True) report_lines = [] total_selected = 0 total_excluded = 0 for code in VALID_CODES: candidates = by_category.get(code, []) name, _scope = CATEGORIES[code] print(f"=== {code} -- {name} ({len(candidates)} candidate(s)) ===") if not candidates: report_lines.append(f"\n## {code} -- {name}\n(no candidates)\n") continue # Score, in batches. for batch in chunked(candidates, args.batch_size): scored = score_batch(code, batch, args.model) for item, (score, note) in zip(batch, scored): item["score"] = score item["note"] = note # Dedup near-identical occasions. keep_idx = dedup_group_indices(candidates, args.dedup_threshold) keep_set = set(keep_idx) survivors = [candidates[i] for i in keep_idx] deduped_out = [candidates[i] for i in range(len(candidates)) if i not in keep_set] # Select top N with verse cap. selected = select_top_n(survivors, args.n_per_category, args.verse_cap) selected_paths = {c["path"] for c in selected} excluded = [c for c in survivors if c["path"] not in selected_paths] + deduped_out total_selected += len(selected) total_excluded += len(excluded) shortfall = args.n_per_category - len(selected) print(f" scored {len(candidates)}, deduped out {len(deduped_out)}, " f"selected {len(selected)}" + (f" ** SHORT by {shortfall} **" if shortfall > 0 else "")) # Move excluded files to boneyard/. if not args.dry_run: for c in excluded: dest = boneyard_dir / c["path"].name if dest.exists(): dest = boneyard_dir / f"{c['path'].stem}-dup{int(time.time()*1000)%100000}.txt" shutil.move(str(c["path"]), str(dest)) # Report. report_lines.append(f"\n## {code} -- {name} " f"({len(candidates)} scored, {len(selected)} selected" f"{', SHORT of quota' if shortfall > 0 else ''})\n") for rank, c in enumerate(sorted(selected, key=lambda x: x["score"], reverse=True), start=1): report_lines.append( f"{rank:2d}. [{c['score']:2d}] {c['ref']:>8} {c['path'].name}\n" f" {c['occasion']}\n" ) report_path = directory / "selection_report.txt" report_text = ( f"Selection report -- {args.n_per_category} per category, verse cap {args.verse_cap}\n" + "=" * 70 + "\n" + "".join(report_lines) ) if not args.dry_run: report_path.write_text(report_text, encoding="utf-8") print() print(f"Done. Selected: {total_selected}, moved to boneyard: {total_excluded}, " f"deleted (empty): {len(deleted)}") if not args.dry_run: print(f"Report written to {report_path}") else: print("(dry run -- report not written; rerun without --dry-run to apply)") if __name__ == "__main__": main()