#!/usr/bin/env python3 """Score markdown paragraphs against a style guide via `claude -p`. Reads one or more markdown files, identifies paragraphs that need scoring (new or changed), scores them in parallel via `claude -p` subprocesses, and writes *-scored.md files with compact score blocks inserted after each paragraph. Optionally also emits the results as machine-readable JSON. Usage: score-paragraphs --style style/voice.md book/chapter-01.md [chapter-02.md ...] score-paragraphs --style style/voice.md book/chapter-01.md -o book/chapter-01-scored.md score-paragraphs --style style/voice.md book/chapter-01.md --json /tmp/scores.json score-paragraphs --style style/voice.md book/chapter-01.md --dryrun Scoring dimensions (each 0–3): E Evidence — specific claims backed by named sources / research links J Judgment — direct language; no hedging where evidence supports a claim V Voice — free of LLM tells, buzzwords, corporate softening R Rhythm — sentence and paragraph length variety; not uniform blocks G Register — appropriate to the target genre (not flat framework, not academic) Score block format (inserted after each scored paragraph): > `◈` E:3 · J:2 · V:3 · R:2 · G:3 = **13/15** `¶a3f5b2` > ⚑ *"potentially" (s2) — evidence supports direct claim* The `¶xxxxxxx` suffix is the first 7 chars of the SHA-256 of the paragraph text. On re-runs, paragraphs whose hash matches their existing score block are skipped automatically. Changed paragraphs are re-scored and their stale block replaced. Use --force to re-score everything regardless of hash. This is a generalised port of the GOES-repo score-paragraphs.py: --style replaces a repo-relative default style guide, --book (GOES chapter/appendix discovery) is gone, and --dryrun / --json are new. See specs/score-paragraphs.spec.md for the full spec. Requires: `claude` CLI on PATH (Claude Code) for any non-dryrun run. """ import argparse import hashlib import json import re import shutil import subprocess import sys from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path from typing import Optional # Paragraphs shorter than this (word count) are not scored MIN_WORDS = 25 DIMENSIONS = ("evidence", "judgment", "voice", "rhythm", "register") # ── Score block parsing / detection ────────────────────────────────── SCORE_MARKER = "`◈`" HASH_RE = re.compile(r"`¶([0-9a-f]{7})`") SCORE_LINE_RE = re.compile( r"E:(\d+)\s*·\s*J:(\d+)\s*·\s*V:(\d+)\s*·\s*R:(\d+)\s*·\s*G:(\d+)" ) FLAG_ITEM_RE = re.compile(r"^⚑ \*(.*)\*$") NOTE_ITEM_RE = re.compile(r"^_(.*)_$") def para_hash(text: str) -> str: """Return first 7 hex chars of SHA-256 of the paragraph text.""" return hashlib.sha256(text.strip().encode()).hexdigest()[:7] def is_score_block(line: str) -> bool: return line.startswith("> ") and SCORE_MARKER in line def extract_hash_from_score_block(block_text: str) -> Optional[str]: """Extract the embedded paragraph hash from a score block, or None.""" m = HASH_RE.search(block_text) return m.group(1) if m else None def parse_score_block(block_text: str) -> dict: """Reconstruct a scores dict from an already-written score block. Used to recover the scores for a paragraph that was skipped this run because its hash matched (--json needs its score too, but no `claude` call is made to get it — it's read back out of the block text itself). """ lines = block_text.splitlines() m = SCORE_LINE_RE.search(lines[0]) if lines else None if not m: return _error_score("unparseable score block") e, j, v, r, g = (int(x) for x in m.groups()) flags = [] note = "" if len(lines) > 1: detail = lines[1] if detail.startswith("> "): detail = detail[2:] for part in detail.split(" · "): part = part.strip() fm = FLAG_ITEM_RE.match(part) if fm: flags.append(fm.group(1)) continue nm = NOTE_ITEM_RE.match(part) if nm: note = nm.group(1) return {"evidence": e, "judgment": j, "voice": v, "rhythm": r, "register": g, "flags": flags, "note": note} # ── Markdown paragraph splitting ───────────────────────────────────── def split_into_chunks(text: str) -> list: """Split markdown into chunks tagged as 'content' or 'skip'. 'skip': YAML frontmatter, fenced code blocks, headings, tables, horizontal rules, empty lines, existing score blocks. 'content': substantive prose paragraphs worth scoring. """ lines = text.splitlines(keepends=True) chunks = [] in_frontmatter = lines and lines[0].strip() == "---" current_lines = [] def flush(kind): if current_lines: chunks.append({"kind": kind, "text": "".join(current_lines)}) current_lines.clear() i = 0 while i < len(lines): line = lines[i] stripped = line.strip() # YAML frontmatter if in_frontmatter: current_lines.append(line) if stripped == "---" and i > 0: in_frontmatter = False flush("skip") i += 1 continue # Fenced code blocks if stripped.startswith("```") or stripped.startswith("~~~"): flush("skip") current_lines.append(line) fence = stripped[:3] i += 1 while i < len(lines): current_lines.append(lines[i]) if lines[i].strip().startswith(fence): i += 1 break i += 1 flush("skip") continue # Existing score blocks # # NOTE (verified during the port, not fixed — see "Known limitation" # in specs/score-paragraphs.spec.md): this flush("skip") absorbs # whatever is *currently accumulating* in current_lines, not just the # score-block line itself. Since format_score_block() is written with # zero blank-line separation from its paragraph, a paragraph that was # already scored is still accumulating here when this branch fires — # so it gets swallowed into the same "skip" chunk as its own score # block, on every subsequent parse. That paragraph is no longer # `kind == "content"` at all, so the hash-comparison branch below in # `_jobs_for_file` (and even --force) never gets a chance to run # against it. This makes that hash-comparison branch structurally # unreachable for any file this script itself wrote — confirmed # against real GOES output. The logic is kept as-is (unchanged # engine, per the port's scope) rather than reworked here. if is_score_block(line): flush("skip") current_lines.append(line) i += 1 while i < len(lines) and lines[i].startswith("> "): current_lines.append(lines[i]) i += 1 flush("skip") continue # Empty line — paragraph boundary if not stripped: flush("content") current_lines.append(line) flush("skip") i += 1 continue # Headings if stripped.startswith("#"): flush("content") current_lines.append(line) flush("skip") i += 1 continue # Tables if stripped.startswith("|"): flush("content") current_lines.append(line) i += 1 while i < len(lines) and lines[i].strip().startswith("|"): current_lines.append(lines[i]) i += 1 flush("skip") continue # Horizontal rules if re.match(r"^[-*_]{3,}$", stripped): flush("content") current_lines.append(line) flush("skip") i += 1 continue current_lines.append(line) i += 1 flush("content") return chunks def word_count(text: str) -> int: return len(text.split()) # ── Claude scoring via `claude -p` ─────────────────────────────────── PROMPT_TEMPLATE = """\ You are a voice-quality reviewer. Score the paragraph at the end of this prompt \ against the style guide below. {style_guide} --- ## Scoring rubric Score the paragraph on five dimensions (each 0–3). **E — Evidence (0–3)** - 3: Specific claims backed by named research links or named adopters - 2: Most claims grounded; one or two assertions lack explicit sourcing - 1: Claims present but mostly asserted without named backing - 0: Generic assertions with no evidence anchoring **J — Judgment (0–3)** - 3: Direct, committed language; no hedging where evidence supports a claim - 2: Mostly direct; one unnecessary hedge (*could*, *might*, *may*, *potentially*) - 1: Noticeable hedging; evidence would support stronger claims - 0: Pervasive hedging throughout **V — Voice (0–3)** - 3: Practitioner register; free of LLM tells and burn-list buzzwords - 2: Mostly clean; one slip into corporate softening or LLM phrasing - 1: Several buzzwords (*leverage*, *empower*, *unlock*, *robust*, *transformative*, etc.) - 0: Reads like an LLM or marketing copy **R — Rhythm (0–3)** - 3: Deliberate sentence and paragraph length variety - 2: Mostly varied; slight tendency toward uniform medium-length sentences - 1: Noticeably uniform sentence length throughout - 0: Wall-of-text or robotic uniformity **G — Register (0–3)** - 3: Correctly matches the expected genre register (authored-prose, framework, blog, etc.) - 2: Mostly correct; slight drift - 1: Wrong register for the genre - 0: Completely mismatched ## Smell-test flags Report any present: - Burn-list buzzword (*delve*, *journey*, *leverage*, *robust*, *foster*, *streamline*, \ *transformative*, *unlock*, *harness*, *pivotal*, *multifaceted*, *empower*, *synergize*) - "It's important to note that" / "It's worth noting that" - "In conclusion" / "Ultimately" / "In essence" as opener or closer - Rule-of-three triplets in consecutive sentences - Hedging where evidence supports a direct claim - Throat-clearing opener ("In today's...", "In the ever-evolving...") - Specific statistic without a research link - Signposted conclusion that recaps and motivates ## Output format Respond with ONLY a JSON object — no markdown fences, no preamble: {{"evidence": <0-3>, "judgment": <0-3>, "voice": <0-3>, "rhythm": <0-3>, \ "register": <0-3>, "flags": ["", ...], "note": ""}} If the text is too short or not scoreable prose, return all zeros and an empty note. --- ## Paragraph to score: {paragraph}""" def build_prompt(style_guide: str, paragraph: str) -> str: return PROMPT_TEMPLATE.format( style_guide=style_guide, paragraph=paragraph.strip(), ) def score_paragraph(style_guide: str, paragraph: str) -> dict: """Call `claude -p` to score one paragraph. Returns a score dict.""" prompt = build_prompt(style_guide, paragraph) try: result = subprocess.run( ["claude", "-p", prompt], capture_output=True, text=True, timeout=120, ) raw = result.stdout.strip() # Strip markdown code fences if Claude wraps the JSON raw = re.sub(r"^```(?:json)?\s*", "", raw) raw = re.sub(r"\s*```\s*$", "", raw) return json.loads(raw) except subprocess.TimeoutExpired: return _error_score("timeout") except json.JSONDecodeError as e: return _error_score(f"json parse error: {e}") except Exception as e: return _error_score(str(e)) def _error_score(reason: str) -> dict: return {"evidence": 0, "judgment": 0, "voice": 0, "rhythm": 0, "register": 0, "flags": [f"scoring error: {reason}"], "note": ""} def format_score_block(scores: dict, phash: str) -> str: """Format a score dict as one or two blockquote lines, embedding the paragraph hash.""" e = scores.get("evidence", 0) j = scores.get("judgment", 0) v = scores.get("voice", 0) r = scores.get("rhythm", 0) g = scores.get("register", 0) total = e + j + v + r + g flags = scores.get("flags", []) note = (scores.get("note") or "").strip() line1 = f"> `◈` E:{e} · J:{j} · V:{v} · R:{r} · G:{g} = **{total}/15** `¶{phash}`" lines = [line1] parts = [f"⚑ *{f}*" for f in flags] if note and note not in " ".join(flags): parts.append(f"_{note}_") if parts: lines.append("> " + " · ".join(parts)) return "\n".join(lines) def _json_paragraph_entry(index: int, phash: str, action: str, scores: dict) -> dict: e = scores.get("evidence", 0) j = scores.get("judgment", 0) v = scores.get("voice", 0) r = scores.get("rhythm", 0) g = scores.get("register", 0) return { "index": index, "hash": phash, "action": action, "scores": {"evidence": e, "judgment": j, "voice": v, "rhythm": r, "register": g}, "total": e + j + v + r + g, "flags": scores.get("flags", []), "note": (scores.get("note") or "").strip(), } def _summarize(paragraphs: list, paragraphs_skipped: int) -> dict: totals = [p["total"] for p in paragraphs if p.get("total") is not None] return { "mean": round(sum(totals) / len(totals), 2) if totals else None, "min": min(totals) if totals else None, "paragraphs_scored": len(paragraphs), "paragraphs_skipped": paragraphs_skipped, } # ── File processing ─────────────────────────────────────────────────── def _jobs_for_file(chunks: list, force: bool) -> list: """Return list of (chunk_idx, para_text, phash, stale_block_follows) for paragraphs that need scoring.""" jobs = [] for idx, chunk in enumerate(chunks): if chunk["kind"] != "content": continue para_text = chunk["text"] if word_count(para_text) < MIN_WORDS: continue phash = para_hash(para_text) stale_follows = False if not force: next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None if next_chunk: first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else "" if next_chunk["kind"] == "skip" and is_score_block(first_line): existing_hash = extract_hash_from_score_block(next_chunk["text"]) if existing_hash == phash: continue # Unchanged — skip stale_follows = True # Changed — stale block needs dropping jobs.append((idx, para_text, phash, stale_follows)) return jobs def dryrun_file(chunks: list, jobs: list, log) -> dict: """Report what would happen for one file, without calling claude or writing anything. Returns a dict shaped like process_file's return value so the caller can print the same summary lines and (optionally) assemble a preview JSON. """ job_idxs = {idx for idx, *_ in jobs} total_eligible = sum( 1 for c in chunks if c["kind"] == "content" and word_count(c["text"]) >= MIN_WORDS ) total_short = sum( 1 for c in chunks if c["kind"] == "content" and word_count(c["text"]) < MIN_WORDS ) needs_scoring = len(jobs) already_ok = total_eligible - needs_scoring log(f" {needs_scoring} would be scored, {already_ok} unchanged " f"(hash match, would skip), {total_short} too short (skipped)") paragraphs = [] ordinal = 0 for idx, chunk in enumerate(chunks): if chunk["kind"] != "content": continue wc = word_count(chunk["text"]) if wc < MIN_WORDS: log(f" [skip too-short] paragraph ~{wc} words") continue ordinal += 1 phash = para_hash(chunk["text"]) if idx in job_idxs: log(f" [would score] ¶{phash} (#{ordinal}, {wc} words) — " f"would run: claude -p ") # Score is unknown pre-run — no claude call made in dryrun. paragraphs.append({ "index": ordinal, "hash": phash, "action": "would_score", "scores": None, "total": None, "flags": [], "note": "", }) else: next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None reused_scores = None if next_chunk and next_chunk["kind"] == "skip": first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else "" if is_score_block(first_line): reused_scores = parse_score_block(next_chunk["text"]) if reused_scores is None: reused_scores = _error_score("could not reuse existing score block") log(f" [skip unchanged] ¶{phash} (#{ordinal}) — " f"hash matches existing score block") paragraphs.append(_json_paragraph_entry(ordinal, phash, "reused", reused_scores)) return { "scored": needs_scoring, "skipped": already_ok, "paragraphs": paragraphs, "summary": _summarize(paragraphs, total_short), } def process_file( input_path: Path, output_path: Path, style_guide: str, max_workers: int, force: bool, log, ) -> dict: """Score paragraphs in input_path in parallel, write to output_path. Returns a dict: {"scored", "skipped", "paragraphs", "summary"}. """ text = input_path.read_text(encoding="utf-8") chunks = split_into_chunks(text) jobs = _jobs_for_file(chunks, force) total_eligible = sum( 1 for c in chunks if c["kind"] == "content" and word_count(c["text"]) >= MIN_WORDS ) total_short = sum( 1 for c in chunks if c["kind"] == "content" and word_count(c["text"]) < MIN_WORDS ) needs_scoring = len(jobs) already_ok = total_eligible - needs_scoring log(f" {input_path.name}: {needs_scoring} to score, " f"{already_ok} unchanged (skipping)") # Score all jobs in parallel results = {} # chunk_idx -> (scores, phash) stale = set() # chunk indices whose following score block should be dropped if jobs: completed = 0 with ThreadPoolExecutor(max_workers=max_workers) as executor: future_map = { executor.submit(score_paragraph, style_guide, para): (idx, phash, stale_follows) for idx, para, phash, stale_follows in jobs } for future in as_completed(future_map): idx, phash, stale_follows = future_map[future] scores = future.result() results[idx] = (scores, phash) if stale_follows: stale.add(idx) completed += 1 total = sum(scores.get(k, 0) for k in DIMENSIONS) log(f" [{completed}/{needs_scoring}] ¶{phash} {total}/15") # Assemble output in chunk order, and the JSON paragraph list alongside it output_parts = [] drop_next_score_block = False json_paragraphs = [] ordinal = 0 for idx, chunk in enumerate(chunks): if chunk["kind"] != "content": first_line = chunk["text"].splitlines()[0] if chunk["text"] else "" if drop_next_score_block and is_score_block(first_line): drop_next_score_block = False continue # Drop stale score block drop_next_score_block = False output_parts.append(chunk["text"]) continue output_parts.append(chunk["text"]) wc = word_count(chunk["text"]) if wc < MIN_WORDS: continue # Too short — never scored, no block, not in JSON ordinal += 1 if idx in results: scores, phash = results[idx] output_parts.append(format_score_block(scores, phash) + "\n\n") if idx in stale: drop_next_score_block = True # Remove the now-replaced stale block json_paragraphs.append(_json_paragraph_entry(ordinal, phash, "scored", scores)) else: # Unchanged — hash matched, no claude call made. Recover its score # from the existing block (already present in output_parts via # the chunk text loop below, since we haven't touched that chunk). phash = para_hash(chunk["text"]) next_chunk = chunks[idx + 1] if idx + 1 < len(chunks) else None reused_scores = None if next_chunk and next_chunk["kind"] == "skip": first_line = next_chunk["text"].splitlines()[0] if next_chunk["text"] else "" if is_score_block(first_line): reused_scores = parse_score_block(next_chunk["text"]) if reused_scores is None: reused_scores = _error_score("could not reuse existing score block") json_paragraphs.append(_json_paragraph_entry(ordinal, phash, "reused", reused_scores)) output_path.write_text("".join(output_parts), encoding="utf-8") return { "scored": needs_scoring, "skipped": already_ok, "paragraphs": json_paragraphs, "summary": _summarize(json_paragraphs, total_short), } # ── CLI ─────────────────────────────────────────────────────────────── def build_arg_parser() -> argparse.ArgumentParser: ap = argparse.ArgumentParser( prog="score-paragraphs", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter, ) ap.add_argument("files", nargs="+", help="Markdown files to score") ap.add_argument("--style", required=True, metavar="FILE", help="Style guide markdown file to score paragraphs against") ap.add_argument("-o", "--output", help="Output file (only valid with a single input file)") ap.add_argument("--json", metavar="FILE", help="Also emit machine-readable JSON results to FILE, or - for stdout") ap.add_argument("--parallel", type=int, default=4, metavar="N", help="Number of parallel claude -p calls (default: 4)") ap.add_argument("--force", action="store_true", help="Re-score all paragraphs, ignoring existing hash matches") ap.add_argument("--dryrun", "-n", action="store_true", help="Preview what would be scored/skipped; makes no claude calls, " "writes no files") return ap def main(argv=None) -> int: ap = build_arg_parser() args = ap.parse_args(argv if argv is not None else sys.argv[1:]) json_to_stdout = args.json == "-" def log(msg: str = "") -> None: print(msg, file=sys.stderr if json_to_stdout else sys.stdout, flush=True) # Check claude is available (skip in dryrun — no calls are made) if not args.dryrun and not shutil.which("claude"): print("Error: 'claude' not found on PATH. Install Claude Code to continue.", file=sys.stderr) return 1 # Resolve input files input_paths = [] for f in args.files: p = Path(f) if not p.is_absolute(): p = Path.cwd() / p input_paths.append(p) if args.output and len(input_paths) > 1: print("Error: --output can only be used with a single input file", file=sys.stderr) return 1 # Resolve output paths io_pairs = [] for inp in input_paths: if args.output: out = Path(args.output) else: stem = inp.stem if not stem.endswith("-scored"): stem += "-scored" out = inp.parent / (stem + inp.suffix) io_pairs.append((inp, out)) for inp, _ in io_pairs: if not inp.exists(): print(f"Error: file not found: {inp}", file=sys.stderr) return 1 # Load style guide style_path = Path(args.style) if not style_path.is_absolute(): style_path = Path.cwd() / style_path if not style_path.exists(): print(f"Error: style guide not found: {style_path}", file=sys.stderr) return 1 style_guide = style_path.read_text(encoding="utf-8") if args.dryrun: log("[dryrun] score-paragraphs — no claude calls will be made, no files will be written") log(f"Style: {style_path}") log(f"Parallel: {args.parallel} workers") total_scored = total_skipped = 0 json_files = [] for inp, out in io_pairs: log(f"\nScoring: {inp} → {out}") if args.dryrun: chunks = split_into_chunks(inp.read_text(encoding="utf-8")) jobs = _jobs_for_file(chunks, args.force) result = dryrun_file(chunks, jobs, log) else: result = process_file(inp, out, style_guide, args.parallel, args.force, log) total_scored += result["scored"] total_skipped += result["skipped"] if args.dryrun: log(f" Would score: {result['scored']}, unchanged: {result['skipped']}") else: log(f" Done: {result['scored']} scored, {result['skipped']} unchanged") json_files.append({ "input": str(inp), "output": str(out), "paragraphs": result["paragraphs"], "summary": result["summary"], }) if args.dryrun: log(f"\nTotal: {total_scored} would be scored, {total_skipped} unchanged.") else: log(f"\nTotal: {total_scored} paragraphs scored, {total_skipped} unchanged.") # --- JSON output ------------------------------------------------------- if args.json: payload = { "style": str(style_path), "dryrun": args.dryrun, "files": json_files, } json_text = json.dumps(payload, indent=2) if args.json == "-": # Stdout is not a filesystem change — honoured even under --dryrun. print(json_text) elif args.dryrun: log(f"[dryrun] would write JSON results to {args.json}") else: Path(args.json).write_text(json_text + "\n", encoding="utf-8") log(f"JSON results written to {args.json}") return 0 if __name__ == "__main__": sys.exit(main())