score-paragraphs: port GOES paragraph scorer as general-purpose script

Required --style, --json machine-readable output, mandatory --dryrun.
Engine unchanged; known limitation documented: already-scored paragraphs
are never re-examined (present in the original).

Claude-Session: https://claude.ai/code/session_01YQDoWNM7XPPii28khFWoMc
This commit is contained in:
Paul O'Reilly
2026-08-02 21:18:02 +12:00
parent 52eb484ef7
commit f78d292f05
3 changed files with 1409 additions and 0 deletions

733
scripts/score-paragraphs Executable file
View File

@@ -0,0 +1,733 @@
#!/usr/bin/env python3
"""Score markdown paragraphs against a style guide via `claude -p`.
Reads one or more markdown files, identifies paragraphs that need scoring
(new or changed), scores them in parallel via `claude -p` subprocesses, and
writes *-scored.md files with compact score blocks inserted after each
paragraph. Optionally also emits the results as machine-readable JSON.
Usage:
score-paragraphs --style style/voice.md book/chapter-01.md [chapter-02.md ...]
score-paragraphs --style style/voice.md book/chapter-01.md -o book/chapter-01-scored.md
score-paragraphs --style style/voice.md book/chapter-01.md --json /tmp/scores.json
score-paragraphs --style style/voice.md book/chapter-01.md --dryrun
Scoring dimensions (each 03):
E Evidence — specific claims backed by named sources / research links
J Judgment — direct language; no hedging where evidence supports a claim
V Voice — free of LLM tells, buzzwords, corporate softening
R Rhythm — sentence and paragraph length variety; not uniform blocks
G Register — appropriate to the target genre (not flat framework, not academic)
Score block format (inserted after each scored paragraph):
> `◈` E:3 · J:2 · V:3 · R:2 · G:3 = **13/15** `¶a3f5b2`
> ⚑ *"potentially" (s2) — evidence supports direct claim*
The `¶xxxxxxx` suffix is the first 7 chars of the SHA-256 of the paragraph text.
On re-runs, paragraphs whose hash matches their existing score block are skipped
automatically. Changed paragraphs are re-scored and their stale block replaced.
Use --force to re-score everything regardless of hash.
This is a generalised port of the GOES-repo score-paragraphs.py: --style
replaces a repo-relative default style guide, --book (GOES chapter/appendix
discovery) is gone, and --dryrun / --json are new. See
specs/score-paragraphs.spec.md for the full spec.
Requires: `claude` CLI on PATH (Claude Code) for any non-dryrun run.
"""
import argparse
import hashlib
import json
import re
import shutil
import subprocess
import sys
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Optional
# Paragraphs shorter than this (word count) are not scored
MIN_WORDS = 25
DIMENSIONS = ("evidence", "judgment", "voice", "rhythm", "register")
# ── Score block parsing / detection ──────────────────────────────────
SCORE_MARKER = "`◈`"
HASH_RE = re.compile(r"`¶([0-9a-f]{7})`")
SCORE_LINE_RE = re.compile(
r"E:(\d+)\s*·\s*J:(\d+)\s*·\s*V:(\d+)\s*·\s*R:(\d+)\s*·\s*G:(\d+)"
)
FLAG_ITEM_RE = re.compile(r"^⚑ \*(.*)\*$")
NOTE_ITEM_RE = re.compile(r"^_(.*)_$")
def para_hash(text: str) -> str:
"""Return first 7 hex chars of SHA-256 of the paragraph text."""
return hashlib.sha256(text.strip().encode()).hexdigest()[:7]
def is_score_block(line: str) -> bool:
return line.startswith("> ") and SCORE_MARKER in line
def extract_hash_from_score_block(block_text: str) -> Optional[str]:
"""Extract the embedded paragraph hash from a score block, or None."""
m = HASH_RE.search(block_text)
return m.group(1) if m else None
def parse_score_block(block_text: str) -> dict:
"""Reconstruct a scores dict from an already-written score block.
Used to recover the scores for a paragraph that was skipped this run
because its hash matched (--json needs its score too, but no `claude`
call is made to get it — it's read back out of the block text itself).
"""
lines = block_text.splitlines()
m = SCORE_LINE_RE.search(lines[0]) if lines else None
if not m:
return _error_score("unparseable score block")
e, j, v, r, g = (int(x) for x in m.groups())
flags = []
note = ""
if len(lines) > 1:
detail = lines[1]
if detail.startswith("> "):
detail = detail[2:]
for part in detail.split(" · "):
part = part.strip()
fm = FLAG_ITEM_RE.match(part)
if fm:
flags.append(fm.group(1))
continue
nm = NOTE_ITEM_RE.match(part)
if nm:
note = nm.group(1)
return {"evidence": e, "judgment": j, "voice": v, "rhythm": r,
"register": g, "flags": flags, "note": note}
# ── Markdown paragraph splitting ─────────────────────────────────────
def split_into_chunks(text: str) -> list:
"""Split markdown into chunks tagged as 'content' or 'skip'.
'skip': YAML frontmatter, fenced code blocks, headings, tables,
horizontal rules, empty lines, existing score blocks.
'content': substantive prose paragraphs worth scoring.
"""
lines = text.splitlines(keepends=True)
chunks = []
in_frontmatter = lines and lines[0].strip() == "---"
current_lines = []
def flush(kind):
if current_lines:
chunks.append({"kind": kind, "text": "".join(current_lines)})
current_lines.clear()
i = 0
while i < len(lines):
line = lines[i]
stripped = line.strip()
# YAML frontmatter
if in_frontmatter:
current_lines.append(line)
if stripped == "---" and i > 0:
in_frontmatter = False
flush("skip")
i += 1
continue
# Fenced code blocks
if stripped.startswith("```") or stripped.startswith("~~~"):
flush("skip")
current_lines.append(line)
fence = stripped[:3]
i += 1
while i < len(lines):
current_lines.append(lines[i])
if lines[i].strip().startswith(fence):
i += 1
break
i += 1
flush("skip")
continue
# Existing score blocks
#
# NOTE (verified during the port, not fixed — see "Known limitation"
# in specs/score-paragraphs.spec.md): this flush("skip") absorbs
# whatever is *currently accumulating* in current_lines, not just the
# score-block line itself. Since format_score_block() is written with
# zero blank-line separation from its paragraph, a paragraph that was
# already scored is still accumulating here when this branch fires —
# so it gets swallowed into the same "skip" chunk as its own score
# block, on every subsequent parse. That paragraph is no longer
# `kind == "content"` at all, so the hash-comparison branch below in
# `_jobs_for_file` (and even --force) never gets a chance to run
# against it. This makes that hash-comparison branch structurally
# unreachable for any file this script itself wrote — confirmed
# against real GOES output. The logic is kept as-is (unchanged
# engine, per the port's scope) rather than reworked here.
if is_score_block(line):
flush("skip")
current_lines.append(line)
i += 1
while i < len(lines) and lines[i].startswith("> "):
current_lines.append(lines[i])
i += 1
flush("skip")
continue
# Empty line — paragraph boundary
if not stripped:
flush("content")
current_lines.append(line)
flush("skip")
i += 1
continue
# Headings
if stripped.startswith("#"):
flush("content")
current_lines.append(line)
flush("skip")
i += 1
continue
# Tables
if stripped.startswith("|"):
flush("content")
current_lines.append(line)
i += 1
while i < len(lines) and lines[i].strip().startswith("|"):
current_lines.append(lines[i])
i += 1
flush("skip")
continue
# Horizontal rules
if re.match(r"^[-*_]{3,}$", stripped):
flush("content")
current_lines.append(line)
flush("skip")
i += 1
continue
current_lines.append(line)
i += 1
flush("content")
return chunks
def word_count(text: str) -> int:
return len(text.split())
# ── Claude scoring via `claude -p` ───────────────────────────────────
PROMPT_TEMPLATE = """\
You are a voice-quality reviewer. Score the paragraph at the end of this prompt \
against the style guide below.
{style_guide}
---
## Scoring rubric
Score the paragraph on five dimensions (each 03).
**E — Evidence (03)**
- 3: Specific claims backed by named research links or named adopters
- 2: Most claims grounded; one or two assertions lack explicit sourcing
- 1: Claims present but mostly asserted without named backing
- 0: Generic assertions with no evidence anchoring
**J — Judgment (03)**
- 3: Direct, committed language; no hedging where evidence supports a claim
- 2: Mostly direct; one unnecessary hedge (*could*, *might*, *may*, *potentially*)
- 1: Noticeable hedging; evidence would support stronger claims
- 0: Pervasive hedging throughout
**V — Voice (03)**
- 3: Practitioner register; free of LLM tells and burn-list buzzwords
- 2: Mostly clean; one slip into corporate softening or LLM phrasing
- 1: Several buzzwords (*leverage*, *empower*, *unlock*, *robust*, *transformative*, etc.)
- 0: Reads like an LLM or marketing copy
**R — Rhythm (03)**
- 3: Deliberate sentence and paragraph length variety
- 2: Mostly varied; slight tendency toward uniform medium-length sentences
- 1: Noticeably uniform sentence length throughout
- 0: Wall-of-text or robotic uniformity
**G — Register (03)**
- 3: Correctly matches the expected genre register (authored-prose, framework, blog, etc.)
- 2: Mostly correct; slight drift
- 1: Wrong register for the genre
- 0: Completely mismatched
## Smell-test flags
Report any present:
- Burn-list buzzword (*delve*, *journey*, *leverage*, *robust*, *foster*, *streamline*, \
*transformative*, *unlock*, *harness*, *pivotal*, *multifaceted*, *empower*, *synergize*)
- "It's important to note that" / "It's worth noting that"
- "In conclusion" / "Ultimately" / "In essence" as opener or closer
- Rule-of-three triplets in consecutive sentences
- Hedging where evidence supports a direct claim
- Throat-clearing opener ("In today's...", "In the ever-evolving...")
- Specific statistic without a research link
- Signposted conclusion that recaps and motivates
## Output format
Respond with ONLY a JSON object — no markdown fences, no preamble:
{{"evidence": <0-3>, "judgment": <0-3>, "voice": <0-3>, "rhythm": <0-3>, \
"register": <0-3>, "flags": ["<flag>", ...], "note": "<one actionable sentence or empty>"}}
If the text is too short or not scoreable prose, return all zeros and an empty note.
---
## Paragraph to score:
{paragraph}"""
def build_prompt(style_guide: str, paragraph: str) -> str:
return PROMPT_TEMPLATE.format(
style_guide=style_guide,
paragraph=paragraph.strip(),
)
def score_paragraph(style_guide: str, paragraph: str) -> dict:
"""Call `claude -p` to score one paragraph. Returns a score dict."""
prompt = build_prompt(style_guide, paragraph)
try:
result = subprocess.run(
["claude", "-p", prompt],
capture_output=True,
text=True,
timeout=120,
)
raw = result.stdout.strip()
# Strip markdown code fences if Claude wraps the JSON
raw = re.sub(r"^```(?:json)?\s*", "", raw)
raw = re.sub(r"\s*```\s*$", "", raw)
return json.loads(raw)
except subprocess.TimeoutExpired:
return _error_score("timeout")
except json.JSONDecodeError as e:
return _error_score(f"json parse error: {e}")
except Exception as e:
return _error_score(str(e))
def _error_score(reason: str) -> dict:
return {"evidence": 0, "judgment": 0, "voice": 0, "rhythm": 0,
"register": 0, "flags": [f"scoring error: {reason}"], "note": ""}
def format_score_block(scores: dict, phash: str) -> str:
"""Format a score dict as one or two blockquote lines, embedding the paragraph hash."""
e = scores.get("evidence", 0)
j = scores.get("judgment", 0)
v = scores.get("voice", 0)
r = scores.get("rhythm", 0)
g = scores.get("register", 0)
total = e + j + v + r + g
flags = scores.get("flags", [])
note = (scores.get("note") or "").strip()
line1 = f"> `◈` E:{e} · J:{j} · V:{v} · R:{r} · G:{g} = **{total}/15** `¶{phash}`"
lines = [line1]
parts = [f"⚑ *{f}*" for f in flags]
if note and note not in " ".join(flags):
parts.append(f"_{note}_")
if parts:
lines.append("> " + " · ".join(parts))
return "\n".join(lines)
def _json_paragraph_entry(index: int, phash: str, action: str, scores: dict) -> dict:
e = scores.get("evidence", 0)
j = scores.get("judgment", 0)
v = scores.get("voice", 0)
r = scores.get("rhythm", 0)
g = scores.get("register", 0)
return {
"index": index,
"hash": phash,
"action": action,
"scores": {"evidence": e, "judgment": j, "voice": v, "rhythm": r, "register": g},
"total": e + j + v + r + g,
"flags": scores.get("flags", []),
"note": (scores.get("note") or "").strip(),
}
def _summarize(paragraphs: list, paragraphs_skipped: int) -> dict:
totals = [p["total"] for p in paragraphs if p.get("total") is not None]
return {
"mean": round(sum(totals) / len(totals), 2) if totals else None,
"min": min(totals) if totals else None,
"paragraphs_scored": len(paragraphs),
"paragraphs_skipped": paragraphs_skipped,
}
# ── File processing ───────────────────────────────────────────────────
def _jobs_for_file(chunks: list, force: bool) -> list:
"""Return list of (chunk_idx, para_text, phash, stale_block_follows) for paragraphs
that need scoring."""
jobs = []
for idx, chunk in enumerate(chunks):
if chunk["kind"] != "content":
continue
para_text = chunk["text"]
if word_count(para_text) < MIN_WORDS:
continue
phash = para_hash(para_text)
stale_follows = False
if not force:
next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None
if next_chunk:
first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else ""
if next_chunk["kind"] == "skip" and is_score_block(first_line):
existing_hash = extract_hash_from_score_block(next_chunk["text"])
if existing_hash == phash:
continue # Unchanged — skip
stale_follows = True # Changed — stale block needs dropping
jobs.append((idx, para_text, phash, stale_follows))
return jobs
def dryrun_file(chunks: list, jobs: list, log) -> dict:
"""Report what would happen for one file, without calling claude or writing anything.
Returns a dict shaped like process_file's return value so the caller can
print the same summary lines and (optionally) assemble a preview JSON.
"""
job_idxs = {idx for idx, *_ in jobs}
total_eligible = sum(
1 for c in chunks
if c["kind"] == "content" and word_count(c["text"]) >= MIN_WORDS
)
total_short = sum(
1 for c in chunks
if c["kind"] == "content" and word_count(c["text"]) < MIN_WORDS
)
needs_scoring = len(jobs)
already_ok = total_eligible - needs_scoring
log(f" {needs_scoring} would be scored, {already_ok} unchanged "
f"(hash match, would skip), {total_short} too short (skipped)")
paragraphs = []
ordinal = 0
for idx, chunk in enumerate(chunks):
if chunk["kind"] != "content":
continue
wc = word_count(chunk["text"])
if wc < MIN_WORDS:
log(f" [skip too-short] paragraph ~{wc} words")
continue
ordinal += 1
phash = para_hash(chunk["text"])
if idx in job_idxs:
log(f" [would score] ¶{phash} (#{ordinal}, {wc} words) — "
f"would run: claude -p <scoring prompt>")
# Score is unknown pre-run — no claude call made in dryrun.
paragraphs.append({
"index": ordinal,
"hash": phash,
"action": "would_score",
"scores": None,
"total": None,
"flags": [],
"note": "",
})
else:
next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None
reused_scores = None
if next_chunk and next_chunk["kind"] == "skip":
first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else ""
if is_score_block(first_line):
reused_scores = parse_score_block(next_chunk["text"])
if reused_scores is None:
reused_scores = _error_score("could not reuse existing score block")
log(f" [skip unchanged] ¶{phash} (#{ordinal}) — "
f"hash matches existing score block")
paragraphs.append(_json_paragraph_entry(ordinal, phash, "reused", reused_scores))
return {
"scored": needs_scoring,
"skipped": already_ok,
"paragraphs": paragraphs,
"summary": _summarize(paragraphs, total_short),
}
def process_file(
input_path: Path,
output_path: Path,
style_guide: str,
max_workers: int,
force: bool,
log,
) -> dict:
"""Score paragraphs in input_path in parallel, write to output_path.
Returns a dict: {"scored", "skipped", "paragraphs", "summary"}.
"""
text = input_path.read_text(encoding="utf-8")
chunks = split_into_chunks(text)
jobs = _jobs_for_file(chunks, force)
total_eligible = sum(
1 for c in chunks
if c["kind"] == "content" and word_count(c["text"]) >= MIN_WORDS
)
total_short = sum(
1 for c in chunks
if c["kind"] == "content" and word_count(c["text"]) < MIN_WORDS
)
needs_scoring = len(jobs)
already_ok = total_eligible - needs_scoring
log(f" {input_path.name}: {needs_scoring} to score, "
f"{already_ok} unchanged (skipping)")
# Score all jobs in parallel
results = {} # chunk_idx -> (scores, phash)
stale = set() # chunk indices whose following score block should be dropped
if jobs:
completed = 0
with ThreadPoolExecutor(max_workers=max_workers) as executor:
future_map = {
executor.submit(score_paragraph, style_guide, para): (idx, phash, stale_follows)
for idx, para, phash, stale_follows in jobs
}
for future in as_completed(future_map):
idx, phash, stale_follows = future_map[future]
scores = future.result()
results[idx] = (scores, phash)
if stale_follows:
stale.add(idx)
completed += 1
total = sum(scores.get(k, 0) for k in DIMENSIONS)
log(f" [{completed}/{needs_scoring}] ¶{phash} {total}/15")
# Assemble output in chunk order, and the JSON paragraph list alongside it
output_parts = []
drop_next_score_block = False
json_paragraphs = []
ordinal = 0
for idx, chunk in enumerate(chunks):
if chunk["kind"] != "content":
first_line = chunk["text"].splitlines()[0] if chunk["text"] else ""
if drop_next_score_block and is_score_block(first_line):
drop_next_score_block = False
continue # Drop stale score block
drop_next_score_block = False
output_parts.append(chunk["text"])
continue
output_parts.append(chunk["text"])
wc = word_count(chunk["text"])
if wc < MIN_WORDS:
continue # Too short — never scored, no block, not in JSON
ordinal += 1
if idx in results:
scores, phash = results[idx]
output_parts.append(format_score_block(scores, phash) + "\n\n")
if idx in stale:
drop_next_score_block = True # Remove the now-replaced stale block
json_paragraphs.append(_json_paragraph_entry(ordinal, phash, "scored", scores))
else:
# Unchanged — hash matched, no claude call made. Recover its score
# from the existing block (already present in output_parts via
# the chunk text loop below, since we haven't touched that chunk).
phash = para_hash(chunk["text"])
next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None
reused_scores = None
if next_chunk and next_chunk["kind"] == "skip":
first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else ""
if is_score_block(first_line):
reused_scores = parse_score_block(next_chunk["text"])
if reused_scores is None:
reused_scores = _error_score("could not reuse existing score block")
json_paragraphs.append(_json_paragraph_entry(ordinal, phash, "reused", reused_scores))
output_path.write_text("".join(output_parts), encoding="utf-8")
return {
"scored": needs_scoring,
"skipped": already_ok,
"paragraphs": json_paragraphs,
"summary": _summarize(json_paragraphs, total_short),
}
# ── CLI ───────────────────────────────────────────────────────────────
def build_arg_parser() -> argparse.ArgumentParser:
ap = argparse.ArgumentParser(
prog="score-paragraphs",
description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter,
)
ap.add_argument("files", nargs="+", help="Markdown files to score")
ap.add_argument("--style", required=True, metavar="FILE",
help="Style guide markdown file to score paragraphs against")
ap.add_argument("-o", "--output",
help="Output file (only valid with a single input file)")
ap.add_argument("--json", metavar="FILE",
help="Also emit machine-readable JSON results to FILE, or - for stdout")
ap.add_argument("--parallel", type=int, default=4, metavar="N",
help="Number of parallel claude -p calls (default: 4)")
ap.add_argument("--force", action="store_true",
help="Re-score all paragraphs, ignoring existing hash matches")
ap.add_argument("--dryrun", "-n", action="store_true",
help="Preview what would be scored/skipped; makes no claude calls, "
"writes no files")
return ap
def main(argv=None) -> int:
ap = build_arg_parser()
args = ap.parse_args(argv if argv is not None else sys.argv[1:])
json_to_stdout = args.json == "-"
def log(msg: str = "") -> None:
print(msg, file=sys.stderr if json_to_stdout else sys.stdout, flush=True)
# Check claude is available (skip in dryrun — no calls are made)
if not args.dryrun and not shutil.which("claude"):
print("Error: 'claude' not found on PATH. Install Claude Code to continue.",
file=sys.stderr)
return 1
# Resolve input files
input_paths = []
for f in args.files:
p = Path(f)
if not p.is_absolute():
p = Path.cwd() / p
input_paths.append(p)
if args.output and len(input_paths) > 1:
print("Error: --output can only be used with a single input file", file=sys.stderr)
return 1
# Resolve output paths
io_pairs = []
for inp in input_paths:
if args.output:
out = Path(args.output)
else:
stem = inp.stem
if not stem.endswith("-scored"):
stem += "-scored"
out = inp.parent / (stem + inp.suffix)
io_pairs.append((inp, out))
for inp, _ in io_pairs:
if not inp.exists():
print(f"Error: file not found: {inp}", file=sys.stderr)
return 1
# Load style guide
style_path = Path(args.style)
if not style_path.is_absolute():
style_path = Path.cwd() / style_path
if not style_path.exists():
print(f"Error: style guide not found: {style_path}", file=sys.stderr)
return 1
style_guide = style_path.read_text(encoding="utf-8")
if args.dryrun:
log("[dryrun] score-paragraphs — no claude calls will be made, no files will be written")
log(f"Style: {style_path}")
log(f"Parallel: {args.parallel} workers")
total_scored = total_skipped = 0
json_files = []
for inp, out in io_pairs:
log(f"\nScoring: {inp} → {out}")
if args.dryrun:
chunks = split_into_chunks(inp.read_text(encoding="utf-8"))
jobs = _jobs_for_file(chunks, args.force)
result = dryrun_file(chunks, jobs, log)
else:
result = process_file(inp, out, style_guide, args.parallel, args.force, log)
total_scored += result["scored"]
total_skipped += result["skipped"]
if args.dryrun:
log(f" Would score: {result['scored']}, unchanged: {result['skipped']}")
else:
log(f" Done: {result['scored']} scored, {result['skipped']} unchanged")
json_files.append({
"input": str(inp),
"output": str(out),
"paragraphs": result["paragraphs"],
"summary": result["summary"],
})
if args.dryrun:
log(f"\nTotal: {total_scored} would be scored, {total_skipped} unchanged.")
else:
log(f"\nTotal: {total_scored} paragraphs scored, {total_skipped} unchanged.")
# --- JSON output -------------------------------------------------------
if args.json:
payload = {
"style": str(style_path),
"dryrun": args.dryrun,
"files": json_files,
}
json_text = json.dumps(payload, indent=2)
if args.json == "-":
# Stdout is not a filesystem change — honoured even under --dryrun.
print(json_text)
elif args.dryrun:
log(f"[dryrun] would write JSON results to {args.json}")
else:
Path(args.json).write_text(json_text + "\n", encoding="utf-8")
log(f"JSON results written to {args.json}")
return 0
if __name__ == "__main__":
sys.exit(main())