Books/2026.08-PrayersForDailyLife/score_and_select.py
2026-09-09 19:59:36 -07:00

396 lines
15 KiB
Python

#!/usr/bin/env python3
"""
score_and_select.py
Run this after classify_occasions.py (and, ideally, revise_prayers.py)
have produced a large pool of finished, per-category prayer files in
./txt (named CODE-CCC.VV.txt / CODE-CCC.VV-b.txt per
rename_to_code_first.sh). This script narrows that pool down to a
manageable shortlist per category for final human selection.
Steps, in order:
1. DELETE any file with NO prayer content at all (bare occasion,
never written). These are backlog, not candidates.
2. SCORE every remaining candidate, batched per category, against a
fixed rubric (concreteness/imagery, single governing image,
absence of known stock-phrase problems, genuine category fit).
3. DE-DUPLICATE near-identical occasions within a category (e.g.
five different verses all drafted as "fear of the future") --
among a near-duplicate group, only the highest scorer is kept as
a candidate for selection; the rest are treated as excluded.
4. SELECT the top --n-per-category (default 25) per category, with a
verse cap (default 1 prayer per verse) so the shortlist doesn't
cluster on a few popular verses. The cap is relaxed automatically,
one step at a time, if a category can't otherwise reach quota.
5. MOVE every file that wasn't selected -- whether cut for score,
redundancy, or the verse cap -- into ./txt/boneyard/ (created if
needed). Nothing scored is ever deleted; only step 1's truly-empty
files are.
6. WRITE selection_report.txt: for each category, the selected
files ranked by score with their ref and occasion, plus a summary
of how many candidates existed and whether the category fell
short of quota (a signal to boost that category via
generate_petitions1.py later).
Files whose code prefix isn't one of the 7 valid categories (i.e. the
BONEYARD-CCC.VV.txt holding-pen files from classify_occasions.py) are
left alone entirely -- this script only touches classified files.
Usage:
pip install openai python-dotenv
python score_and_select.py txt
python score_and_select.py txt --dry-run
python score_and_select.py txt --n-per-category 25 --verse-cap 1
"""
import argparse
import difflib
import json
import os
import re
import shutil
import sys
import time
from collections import defaultdict
from pathlib import Path
from dotenv import load_dotenv
from categories import CATEGORIES, VALID_CODES, scope_block
load_dotenv()
API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY")
if not API_KEY:
sys.exit(
"No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) "
"to a .env file in the working directory, or export it in your shell."
)
try:
from openai import OpenAI
except ImportError:
sys.exit("Missing dependency. Run: pip install openai")
client = OpenAI(api_key=API_KEY)
# --- Parsing -----------------------------------------------------------
FULL_HEADER_RE = re.compile(
r"^(?P<ref>##\s*\S+)[ \t]*\n"
r">(?!>)[ \t]*(?P<title>[^\n]*)\n"
r">>[ \t]*(?P<quote>[^\n]*)\n"
)
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$")
FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<verse>\d{3}\.\d{2,3})(?:-[a-z])?\.txt$")
def parse_file(text: str):
"""Single-occasion new-format file -> {"ref","title","quote","occasion","content"}
or None. Multi-occasion (boneyard-style) or malformed files return None."""
stripped = text.lstrip("\n")
m = FULL_HEADER_RE.match(stripped)
if not m:
return None
ref = m.group("ref").lstrip("#").strip()
title = m.group("title").strip()
quote = m.group("quote").strip()
body = text[m.end():]
matches = list(OCCASION_LINE_RE.finditer(body))
if len(matches) != 1:
return None
mo = matches[0]
occasion = mo.group("text").strip()
content = body[mo.end():].strip()
return {"ref": ref, "title": title, "quote": quote, "occasion": occasion, "content": content}
# --- Scoring rubric & call ----------------------------------------------
RUBRIC = (
"Score each prayer 0-10 (integer) on overall quality as a candidate "
"for a finished devotional book, weighing:\n"
" - Concreteness: real images and specifics, not generic abstraction.\n"
" - A single governing image system, held all the way through, not "
"mixed metaphors.\n"
" - Freedom from tired stock phrasing (e.g. 'I know how fast/quickly...', "
"'the phone keeps ringing', vague summarizing lines like 'that thin life').\n"
" - Genuine fit to its stated category -- not just technically related, "
"but actually about that occasion.\n"
" - Craft: does the ending land, is the voice consistent, is it free of "
"typos/grammar errors.\n"
"A 9-10 is publication-ready as-is. A 5-6 has a real idea but needs "
"editing. Below 4 is generic filler or a poor category fit."
)
SYSTEM_PROMPT = (
"You are scoring a batch of devotional prayers, all written for the same "
"category, to help an author pick the strongest ~25 out of a much larger "
"pool for a finished book.\n\n"
f"Category: {{category_name}} -- {{category_scope}}\n\n"
f"{RUBRIC}\n\n"
"Echo the ref and occasion back exactly as given, for matching."
)
USER_PROMPT_TEMPLATE = """\
Score each of the following {count} prayers:
{items}
Respond ONLY with valid JSON in this exact shape, one entry per prayer \
above, in the same order:
{{
"scores": [
{{"ref": "...", "occasion": "...", "score": 7, "note": "one short phrase"}},
...
]
}}
"""
def build_batch_prompt(items: list) -> str:
blocks = []
for i, it in enumerate(items, start=1):
blocks.append(
f"{i}. Ref: {it['ref']}\n Occasion: {it['occasion']}\n "
f"Text:\n {it['content'].replace(chr(10), chr(10) + ' ')}"
)
return USER_PROMPT_TEMPLATE.format(count=len(items), items="\n\n".join(blocks))
def score_batch(code: str, items: list, model: str, retries: int = 3):
"""items: list of {ref, occasion, content, ...}. Returns list of
(score:int, note:str) aligned to items, in order."""
name, scope = CATEGORIES[code]
system = SYSTEM_PROMPT.format(category_name=name, category_scope=scope)
prompt = build_batch_prompt(items)
last_err = None
for attempt in range(1, retries + 1):
try:
resp = client.chat.completions.create(
model=model,
messages=[
{"role": "system", "content": system},
{"role": "user", "content": prompt},
],
temperature=0.2,
response_format={"type": "json_object"},
)
data = json.loads(resp.choices[0].message.content)
results = data["scores"]
if len(results) != len(items):
raise ValueError(f"Expected {len(items)} scores, got {len(results)}")
out = []
for expected, r in zip(items, results):
if r.get("ref", "").strip() != expected["ref"].strip():
raise ValueError(f"Ref mismatch: expected {expected['ref']!r}, got {r.get('ref')!r}")
score = r.get("score")
if not isinstance(score, (int, float)):
raise ValueError(f"Bad score for {expected['ref']}: {score!r}")
out.append((int(round(score)), r.get("note", "")))
return out
except Exception as e: # noqa: BLE001
last_err = e
print(f" batch attempt {attempt}/{retries} failed: {e}", file=sys.stderr)
time.sleep(1.5 * attempt)
raise RuntimeError(f"Giving up scoring a batch for {code}: {last_err}")
def chunked(seq, size):
for i in range(0, len(seq), size):
yield seq[i:i + size]
# --- Dedup ---------------------------------------------------------------
def dedup_group_indices(candidates: list, threshold: float) -> list:
"""candidates: list of dicts with 'occasion' and 'score'. Returns the
list of indices to KEEP -- one per near-duplicate cluster (the
highest scorer), plus every non-duplicate candidate untouched."""
n = len(candidates)
excluded = set()
for i in range(n):
if i in excluded:
continue
for j in range(i + 1, n):
if j in excluded:
continue
ratio = difflib.SequenceMatcher(
None, candidates[i]["occasion"].lower(), candidates[j]["occasion"].lower()
).ratio()
if ratio >= threshold:
# keep the higher scorer, exclude the other
loser = j if candidates[i]["score"] >= candidates[j]["score"] else i
excluded.add(loser)
if loser == i:
break # i is out; stop comparing it further
return [i for i in range(n) if i not in excluded]
# --- Selection -------------------------------------------------------------
def ref_to_verse_key(ref: str) -> str:
return ref.strip()
def select_top_n(candidates: list, n: int, verse_cap: int) -> list:
"""candidates: list of dicts with 'score' and 'ref', sorted by score
descending on entry (caller's responsibility, but we sort here too
to be safe). Greedily selects up to n, enforcing at most verse_cap
picks per verse ref, relaxing the cap by 1 if quota can't be met."""
ordered = sorted(candidates, key=lambda c: c["score"], reverse=True)
cap = verse_cap
while True:
counts = defaultdict(int)
selected = []
for c in ordered:
key = ref_to_verse_key(c["ref"])
if counts[key] < cap:
selected.append(c)
counts[key] += 1
if len(selected) >= n:
break
if len(selected) >= n or cap >= len(ordered):
return selected[:n]
cap += 1 # relax and try again
# --- Main -------------------------------------------------------------------
def main():
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument("directory", help="Directory of classified per-category .txt files (e.g. txt)")
parser.add_argument("--n-per-category", type=int, default=25, help="Shortlist size per category (default: 25)")
parser.add_argument("--verse-cap", type=int, default=1, help="Max prayers per verse in the shortlist, before auto-relaxing (default: 1)")
parser.add_argument("--dedup-threshold", type=float, default=0.82, help="Occasion-text similarity ratio to treat as a duplicate (default: 0.82)")
parser.add_argument("--batch-size", type=int, default=20, help="Prayers per scoring API call (default: 20)")
parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)")
parser.add_argument("--dry-run", action="store_true", help="Report what would happen; delete/move/write nothing")
args = parser.parse_args()
directory = Path(args.directory)
if not directory.is_dir():
sys.exit(f"Not a directory: {directory}")
boneyard_dir = directory / "boneyard"
files = sorted(directory.glob("*.txt"))
if not files:
sys.exit(f"No .txt files in {directory}")
print(f"Found {len(files)} files in {directory}")
if args.dry_run:
print("(dry run -- nothing will be deleted, moved, or written)")
print()
by_category = defaultdict(list) # code -> list of candidate dicts (with 'path')
deleted = []
ignored = []
for path in files:
m = FILENAME_RE.match(path.name)
if not m or m.group("code") not in VALID_CODES:
ignored.append(path.name) # BONEYARD-*.txt or anything unrecognized
continue
code = m.group("code")
text = path.read_text(encoding="utf-8")
parsed = parse_file(text)
if parsed is None:
ignored.append(path.name)
continue
if not parsed["content"]:
deleted.append(path)
continue
parsed["path"] = path
by_category[code].append(parsed)
print(f"Empty (no prayer written): {len(deleted)} -- will be deleted")
print(f"Ignored (unrecognized name / boneyard holding-pen / unparseable): {len(ignored)}")
for code in VALID_CODES:
print(f" {code}: {len(by_category.get(code, []))} candidate(s)")
print()
if not args.dry_run:
for path in deleted:
path.unlink()
boneyard_dir.mkdir(exist_ok=True)
report_lines = []
total_selected = 0
total_excluded = 0
for code in VALID_CODES:
candidates = by_category.get(code, [])
name, _scope = CATEGORIES[code]
print(f"=== {code} -- {name} ({len(candidates)} candidate(s)) ===")
if not candidates:
report_lines.append(f"\n## {code} -- {name}\n(no candidates)\n")
continue
# Score, in batches.
for batch in chunked(candidates, args.batch_size):
scored = score_batch(code, batch, args.model)
for item, (score, note) in zip(batch, scored):
item["score"] = score
item["note"] = note
# Dedup near-identical occasions.
keep_idx = dedup_group_indices(candidates, args.dedup_threshold)
keep_set = set(keep_idx)
survivors = [candidates[i] for i in keep_idx]
deduped_out = [candidates[i] for i in range(len(candidates)) if i not in keep_set]
# Select top N with verse cap.
selected = select_top_n(survivors, args.n_per_category, args.verse_cap)
selected_paths = {c["path"] for c in selected}
excluded = [c for c in survivors if c["path"] not in selected_paths] + deduped_out
total_selected += len(selected)
total_excluded += len(excluded)
shortfall = args.n_per_category - len(selected)
print(f" scored {len(candidates)}, deduped out {len(deduped_out)}, "
f"selected {len(selected)}" + (f" ** SHORT by {shortfall} **" if shortfall > 0 else ""))
# Move excluded files to boneyard/.
if not args.dry_run:
for c in excluded:
dest = boneyard_dir / c["path"].name
if dest.exists():
dest = boneyard_dir / f"{c['path'].stem}-dup{int(time.time()*1000)%100000}.txt"
shutil.move(str(c["path"]), str(dest))
# Report.
report_lines.append(f"\n## {code} -- {name} "
f"({len(candidates)} scored, {len(selected)} selected"
f"{', SHORT of quota' if shortfall > 0 else ''})\n")
for rank, c in enumerate(sorted(selected, key=lambda x: x["score"], reverse=True), start=1):
report_lines.append(
f"{rank:2d}. [{c['score']:2d}] {c['ref']:>8} {c['path'].name}\n"
f" {c['occasion']}\n"
)
report_path = directory / "selection_report.txt"
report_text = (
f"Selection report -- {args.n_per_category} per category, verse cap {args.verse_cap}\n"
+ "=" * 70 + "\n"
+ "".join(report_lines)
)
if not args.dry_run:
report_path.write_text(report_text, encoding="utf-8")
print()
print(f"Done. Selected: {total_selected}, moved to boneyard: {total_excluded}, "
f"deleted (empty): {len(deleted)}")
if not args.dry_run:
print(f"Report written to {report_path}")
else:
print("(dry run -- report not written; rerun without --dry-run to apply)")
if __name__ == "__main__":
main()