396 lines
15 KiB
Python
396 lines
15 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
score_and_select.py
|
|
|
|
Run this after classify_occasions.py (and, ideally, revise_prayers.py)
|
|
have produced a large pool of finished, per-category prayer files in
|
|
./txt (named CODE-CCC.VV.txt / CODE-CCC.VV-b.txt per
|
|
rename_to_code_first.sh). This script narrows that pool down to a
|
|
manageable shortlist per category for final human selection.
|
|
|
|
Steps, in order:
|
|
|
|
1. DELETE any file with NO prayer content at all (bare occasion,
|
|
never written). These are backlog, not candidates.
|
|
2. SCORE every remaining candidate, batched per category, against a
|
|
fixed rubric (concreteness/imagery, single governing image,
|
|
absence of known stock-phrase problems, genuine category fit).
|
|
3. DE-DUPLICATE near-identical occasions within a category (e.g.
|
|
five different verses all drafted as "fear of the future") --
|
|
among a near-duplicate group, only the highest scorer is kept as
|
|
a candidate for selection; the rest are treated as excluded.
|
|
4. SELECT the top --n-per-category (default 25) per category, with a
|
|
verse cap (default 1 prayer per verse) so the shortlist doesn't
|
|
cluster on a few popular verses. The cap is relaxed automatically,
|
|
one step at a time, if a category can't otherwise reach quota.
|
|
5. MOVE every file that wasn't selected -- whether cut for score,
|
|
redundancy, or the verse cap -- into ./txt/boneyard/ (created if
|
|
needed). Nothing scored is ever deleted; only step 1's truly-empty
|
|
files are.
|
|
6. WRITE selection_report.txt: for each category, the selected
|
|
files ranked by score with their ref and occasion, plus a summary
|
|
of how many candidates existed and whether the category fell
|
|
short of quota (a signal to boost that category via
|
|
generate_petitions1.py later).
|
|
|
|
Files whose code prefix isn't one of the 7 valid categories (i.e. the
|
|
BONEYARD-CCC.VV.txt holding-pen files from classify_occasions.py) are
|
|
left alone entirely -- this script only touches classified files.
|
|
|
|
Usage:
|
|
pip install openai python-dotenv
|
|
python score_and_select.py txt
|
|
python score_and_select.py txt --dry-run
|
|
python score_and_select.py txt --n-per-category 25 --verse-cap 1
|
|
"""
|
|
|
|
import argparse
|
|
import difflib
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import sys
|
|
import time
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
from categories import CATEGORIES, VALID_CODES, scope_block
|
|
|
|
load_dotenv()
|
|
|
|
API_KEY = os.environ.get("OPENAI_API_KEY") or os.environ.get("OPEN_API_KEY")
|
|
if not API_KEY:
|
|
sys.exit(
|
|
"No API key found. Add OPENAI_API_KEY=sk-... (or OPEN_API_KEY=sk-...) "
|
|
"to a .env file in the working directory, or export it in your shell."
|
|
)
|
|
|
|
try:
|
|
from openai import OpenAI
|
|
except ImportError:
|
|
sys.exit("Missing dependency. Run: pip install openai")
|
|
|
|
client = OpenAI(api_key=API_KEY)
|
|
|
|
# --- Parsing -----------------------------------------------------------
|
|
|
|
FULL_HEADER_RE = re.compile(
|
|
r"^(?P<ref>##\s*\S+)[ \t]*\n"
|
|
r">(?!>)[ \t]*(?P<title>[^\n]*)\n"
|
|
r">>[ \t]*(?P<quote>[^\n]*)\n"
|
|
)
|
|
OCCASION_LINE_RE = re.compile(r"(?m)^Occasion:[ \t]*(?P<text>.+)$")
|
|
|
|
FILENAME_RE = re.compile(r"^(?P<code>[A-Z]+)-(?P<verse>\d{3}\.\d{2,3})(?:-[a-z])?\.txt$")
|
|
|
|
|
|
def parse_file(text: str):
|
|
"""Single-occasion new-format file -> {"ref","title","quote","occasion","content"}
|
|
or None. Multi-occasion (boneyard-style) or malformed files return None."""
|
|
stripped = text.lstrip("\n")
|
|
m = FULL_HEADER_RE.match(stripped)
|
|
if not m:
|
|
return None
|
|
ref = m.group("ref").lstrip("#").strip()
|
|
title = m.group("title").strip()
|
|
quote = m.group("quote").strip()
|
|
body = text[m.end():]
|
|
matches = list(OCCASION_LINE_RE.finditer(body))
|
|
if len(matches) != 1:
|
|
return None
|
|
mo = matches[0]
|
|
occasion = mo.group("text").strip()
|
|
content = body[mo.end():].strip()
|
|
return {"ref": ref, "title": title, "quote": quote, "occasion": occasion, "content": content}
|
|
|
|
|
|
# --- Scoring rubric & call ----------------------------------------------
|
|
|
|
RUBRIC = (
|
|
"Score each prayer 0-10 (integer) on overall quality as a candidate "
|
|
"for a finished devotional book, weighing:\n"
|
|
" - Concreteness: real images and specifics, not generic abstraction.\n"
|
|
" - A single governing image system, held all the way through, not "
|
|
"mixed metaphors.\n"
|
|
" - Freedom from tired stock phrasing (e.g. 'I know how fast/quickly...', "
|
|
"'the phone keeps ringing', vague summarizing lines like 'that thin life').\n"
|
|
" - Genuine fit to its stated category -- not just technically related, "
|
|
"but actually about that occasion.\n"
|
|
" - Craft: does the ending land, is the voice consistent, is it free of "
|
|
"typos/grammar errors.\n"
|
|
"A 9-10 is publication-ready as-is. A 5-6 has a real idea but needs "
|
|
"editing. Below 4 is generic filler or a poor category fit."
|
|
)
|
|
|
|
SYSTEM_PROMPT = (
|
|
"You are scoring a batch of devotional prayers, all written for the same "
|
|
"category, to help an author pick the strongest ~25 out of a much larger "
|
|
"pool for a finished book.\n\n"
|
|
f"Category: {{category_name}} -- {{category_scope}}\n\n"
|
|
f"{RUBRIC}\n\n"
|
|
"Echo the ref and occasion back exactly as given, for matching."
|
|
)
|
|
|
|
USER_PROMPT_TEMPLATE = """\
|
|
Score each of the following {count} prayers:
|
|
|
|
{items}
|
|
|
|
Respond ONLY with valid JSON in this exact shape, one entry per prayer \
|
|
above, in the same order:
|
|
{{
|
|
"scores": [
|
|
{{"ref": "...", "occasion": "...", "score": 7, "note": "one short phrase"}},
|
|
...
|
|
]
|
|
}}
|
|
"""
|
|
|
|
|
|
def build_batch_prompt(items: list) -> str:
|
|
blocks = []
|
|
for i, it in enumerate(items, start=1):
|
|
blocks.append(
|
|
f"{i}. Ref: {it['ref']}\n Occasion: {it['occasion']}\n "
|
|
f"Text:\n {it['content'].replace(chr(10), chr(10) + ' ')}"
|
|
)
|
|
return USER_PROMPT_TEMPLATE.format(count=len(items), items="\n\n".join(blocks))
|
|
|
|
|
|
def score_batch(code: str, items: list, model: str, retries: int = 3):
|
|
"""items: list of {ref, occasion, content, ...}. Returns list of
|
|
(score:int, note:str) aligned to items, in order."""
|
|
name, scope = CATEGORIES[code]
|
|
system = SYSTEM_PROMPT.format(category_name=name, category_scope=scope)
|
|
prompt = build_batch_prompt(items)
|
|
last_err = None
|
|
for attempt in range(1, retries + 1):
|
|
try:
|
|
resp = client.chat.completions.create(
|
|
model=model,
|
|
messages=[
|
|
{"role": "system", "content": system},
|
|
{"role": "user", "content": prompt},
|
|
],
|
|
temperature=0.2,
|
|
response_format={"type": "json_object"},
|
|
)
|
|
data = json.loads(resp.choices[0].message.content)
|
|
results = data["scores"]
|
|
if len(results) != len(items):
|
|
raise ValueError(f"Expected {len(items)} scores, got {len(results)}")
|
|
out = []
|
|
for expected, r in zip(items, results):
|
|
if r.get("ref", "").strip() != expected["ref"].strip():
|
|
raise ValueError(f"Ref mismatch: expected {expected['ref']!r}, got {r.get('ref')!r}")
|
|
score = r.get("score")
|
|
if not isinstance(score, (int, float)):
|
|
raise ValueError(f"Bad score for {expected['ref']}: {score!r}")
|
|
out.append((int(round(score)), r.get("note", "")))
|
|
return out
|
|
except Exception as e: # noqa: BLE001
|
|
last_err = e
|
|
print(f" batch attempt {attempt}/{retries} failed: {e}", file=sys.stderr)
|
|
time.sleep(1.5 * attempt)
|
|
raise RuntimeError(f"Giving up scoring a batch for {code}: {last_err}")
|
|
|
|
|
|
def chunked(seq, size):
|
|
for i in range(0, len(seq), size):
|
|
yield seq[i:i + size]
|
|
|
|
|
|
# --- Dedup ---------------------------------------------------------------
|
|
|
|
def dedup_group_indices(candidates: list, threshold: float) -> list:
|
|
"""candidates: list of dicts with 'occasion' and 'score'. Returns the
|
|
list of indices to KEEP -- one per near-duplicate cluster (the
|
|
highest scorer), plus every non-duplicate candidate untouched."""
|
|
n = len(candidates)
|
|
excluded = set()
|
|
for i in range(n):
|
|
if i in excluded:
|
|
continue
|
|
for j in range(i + 1, n):
|
|
if j in excluded:
|
|
continue
|
|
ratio = difflib.SequenceMatcher(
|
|
None, candidates[i]["occasion"].lower(), candidates[j]["occasion"].lower()
|
|
).ratio()
|
|
if ratio >= threshold:
|
|
# keep the higher scorer, exclude the other
|
|
loser = j if candidates[i]["score"] >= candidates[j]["score"] else i
|
|
excluded.add(loser)
|
|
if loser == i:
|
|
break # i is out; stop comparing it further
|
|
return [i for i in range(n) if i not in excluded]
|
|
|
|
|
|
# --- Selection -------------------------------------------------------------
|
|
|
|
def ref_to_verse_key(ref: str) -> str:
|
|
return ref.strip()
|
|
|
|
|
|
def select_top_n(candidates: list, n: int, verse_cap: int) -> list:
|
|
"""candidates: list of dicts with 'score' and 'ref', sorted by score
|
|
descending on entry (caller's responsibility, but we sort here too
|
|
to be safe). Greedily selects up to n, enforcing at most verse_cap
|
|
picks per verse ref, relaxing the cap by 1 if quota can't be met."""
|
|
ordered = sorted(candidates, key=lambda c: c["score"], reverse=True)
|
|
cap = verse_cap
|
|
while True:
|
|
counts = defaultdict(int)
|
|
selected = []
|
|
for c in ordered:
|
|
key = ref_to_verse_key(c["ref"])
|
|
if counts[key] < cap:
|
|
selected.append(c)
|
|
counts[key] += 1
|
|
if len(selected) >= n:
|
|
break
|
|
if len(selected) >= n or cap >= len(ordered):
|
|
return selected[:n]
|
|
cap += 1 # relax and try again
|
|
|
|
|
|
# --- Main -------------------------------------------------------------------
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
|
)
|
|
parser.add_argument("directory", help="Directory of classified per-category .txt files (e.g. txt)")
|
|
parser.add_argument("--n-per-category", type=int, default=25, help="Shortlist size per category (default: 25)")
|
|
parser.add_argument("--verse-cap", type=int, default=1, help="Max prayers per verse in the shortlist, before auto-relaxing (default: 1)")
|
|
parser.add_argument("--dedup-threshold", type=float, default=0.82, help="Occasion-text similarity ratio to treat as a duplicate (default: 0.82)")
|
|
parser.add_argument("--batch-size", type=int, default=20, help="Prayers per scoring API call (default: 20)")
|
|
parser.add_argument("--model", default="gpt-5.4", help="OpenAI model to use (default: gpt-5.4)")
|
|
parser.add_argument("--dry-run", action="store_true", help="Report what would happen; delete/move/write nothing")
|
|
args = parser.parse_args()
|
|
|
|
directory = Path(args.directory)
|
|
if not directory.is_dir():
|
|
sys.exit(f"Not a directory: {directory}")
|
|
boneyard_dir = directory / "boneyard"
|
|
|
|
files = sorted(directory.glob("*.txt"))
|
|
if not files:
|
|
sys.exit(f"No .txt files in {directory}")
|
|
|
|
print(f"Found {len(files)} files in {directory}")
|
|
if args.dry_run:
|
|
print("(dry run -- nothing will be deleted, moved, or written)")
|
|
print()
|
|
|
|
by_category = defaultdict(list) # code -> list of candidate dicts (with 'path')
|
|
deleted = []
|
|
ignored = []
|
|
|
|
for path in files:
|
|
m = FILENAME_RE.match(path.name)
|
|
if not m or m.group("code") not in VALID_CODES:
|
|
ignored.append(path.name) # BONEYARD-*.txt or anything unrecognized
|
|
continue
|
|
code = m.group("code")
|
|
text = path.read_text(encoding="utf-8")
|
|
parsed = parse_file(text)
|
|
if parsed is None:
|
|
ignored.append(path.name)
|
|
continue
|
|
if not parsed["content"]:
|
|
deleted.append(path)
|
|
continue
|
|
parsed["path"] = path
|
|
by_category[code].append(parsed)
|
|
|
|
print(f"Empty (no prayer written): {len(deleted)} -- will be deleted")
|
|
print(f"Ignored (unrecognized name / boneyard holding-pen / unparseable): {len(ignored)}")
|
|
for code in VALID_CODES:
|
|
print(f" {code}: {len(by_category.get(code, []))} candidate(s)")
|
|
print()
|
|
|
|
if not args.dry_run:
|
|
for path in deleted:
|
|
path.unlink()
|
|
boneyard_dir.mkdir(exist_ok=True)
|
|
|
|
report_lines = []
|
|
total_selected = 0
|
|
total_excluded = 0
|
|
|
|
for code in VALID_CODES:
|
|
candidates = by_category.get(code, [])
|
|
name, _scope = CATEGORIES[code]
|
|
print(f"=== {code} -- {name} ({len(candidates)} candidate(s)) ===")
|
|
if not candidates:
|
|
report_lines.append(f"\n## {code} -- {name}\n(no candidates)\n")
|
|
continue
|
|
|
|
# Score, in batches.
|
|
for batch in chunked(candidates, args.batch_size):
|
|
scored = score_batch(code, batch, args.model)
|
|
for item, (score, note) in zip(batch, scored):
|
|
item["score"] = score
|
|
item["note"] = note
|
|
|
|
# Dedup near-identical occasions.
|
|
keep_idx = dedup_group_indices(candidates, args.dedup_threshold)
|
|
keep_set = set(keep_idx)
|
|
survivors = [candidates[i] for i in keep_idx]
|
|
deduped_out = [candidates[i] for i in range(len(candidates)) if i not in keep_set]
|
|
|
|
# Select top N with verse cap.
|
|
selected = select_top_n(survivors, args.n_per_category, args.verse_cap)
|
|
selected_paths = {c["path"] for c in selected}
|
|
excluded = [c for c in survivors if c["path"] not in selected_paths] + deduped_out
|
|
|
|
total_selected += len(selected)
|
|
total_excluded += len(excluded)
|
|
|
|
shortfall = args.n_per_category - len(selected)
|
|
print(f" scored {len(candidates)}, deduped out {len(deduped_out)}, "
|
|
f"selected {len(selected)}" + (f" ** SHORT by {shortfall} **" if shortfall > 0 else ""))
|
|
|
|
# Move excluded files to boneyard/.
|
|
if not args.dry_run:
|
|
for c in excluded:
|
|
dest = boneyard_dir / c["path"].name
|
|
if dest.exists():
|
|
dest = boneyard_dir / f"{c['path'].stem}-dup{int(time.time()*1000)%100000}.txt"
|
|
shutil.move(str(c["path"]), str(dest))
|
|
|
|
# Report.
|
|
report_lines.append(f"\n## {code} -- {name} "
|
|
f"({len(candidates)} scored, {len(selected)} selected"
|
|
f"{', SHORT of quota' if shortfall > 0 else ''})\n")
|
|
for rank, c in enumerate(sorted(selected, key=lambda x: x["score"], reverse=True), start=1):
|
|
report_lines.append(
|
|
f"{rank:2d}. [{c['score']:2d}] {c['ref']:>8} {c['path'].name}\n"
|
|
f" {c['occasion']}\n"
|
|
)
|
|
|
|
report_path = directory / "selection_report.txt"
|
|
report_text = (
|
|
f"Selection report -- {args.n_per_category} per category, verse cap {args.verse_cap}\n"
|
|
+ "=" * 70 + "\n"
|
|
+ "".join(report_lines)
|
|
)
|
|
if not args.dry_run:
|
|
report_path.write_text(report_text, encoding="utf-8")
|
|
|
|
print()
|
|
print(f"Done. Selected: {total_selected}, moved to boneyard: {total_excluded}, "
|
|
f"deleted (empty): {len(deleted)}")
|
|
if not args.dry_run:
|
|
print(f"Report written to {report_path}")
|
|
else:
|
|
print("(dry run -- report not written; rerun without --dry-run to apply)")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|