diff --git a/scripts/md-to-docx b/scripts/md-to-docx new file mode 100755 index 0000000..8fda3f7 --- /dev/null +++ b/scripts/md-to-docx @@ -0,0 +1,263 @@ +#!/usr/bin/env python3 +"""Convert a Markdown file to a professionally styled Word document (.docx).""" + +import argparse +import os +import re +import shutil +import subprocess +import sys +import tempfile + + +# --- Colour palette --- +DARK = (0x1A, 0x1A, 0x2E) # Near-black navy (H1, title) +ACCENT = (0x1B, 0x4D, 0x89) # Professional blue (H2, H3, links) +BODY = (0x33, 0x33, 0x33) # Dark grey (body text) +HEADER_BG = "1B4D89" +LIGHT_BG = "F2F6FA" + + +def check_dependencies(): + """Check that pandoc and python-docx are available. Returns (pandoc_version, docx_version) or exits.""" + pandoc_path = shutil.which("pandoc") + if not pandoc_path: + print("Error: pandoc not found. Install with: sudo apt install pandoc", file=sys.stderr) + sys.exit(1) + + result = subprocess.run([pandoc_path, "--version"], capture_output=True, text=True) + pandoc_version = result.stdout.split("\n")[0].replace("pandoc ", "") + + try: + import docx + docx_version = docx.__version__ + except ImportError: + print("Error: python-docx not installed. Install with: pip install python-docx", file=sys.stderr) + sys.exit(1) + + return pandoc_version, docx_version + + +def extract_title(md_path): + """Extract the first # heading from a markdown file.""" + with open(md_path, "r") as f: + for line in f: + match = re.match(r"^#\s+(.+)$", line.strip()) + if match: + return match.group(1) + return os.path.splitext(os.path.basename(md_path))[0] + + +def create_styled_reference(output_path): + """Generate a pandoc reference.docx with professional styling applied.""" + from docx import Document + from docx.shared import Pt, Cm, RGBColor + from docx.enum.text import WD_ALIGN_PARAGRAPH + + # Extract pandoc's default reference doc + subprocess.run( + ["pandoc", "-o", output_path, "--print-default-data-file", "reference.docx"], + capture_output=True, + check=True, + ) + + doc = Document(output_path) + + for section in doc.sections: + section.top_margin = Cm(2.5) + section.bottom_margin = Cm(2.5) + section.left_margin = Cm(2.5) + section.right_margin = Cm(2.5) + + style_defs = { + "Title": {"size": Pt(28), "bold": True, "color": DARK, "space_after": Pt(6), "alignment": WD_ALIGN_PARAGRAPH.LEFT}, + "Subtitle": {"size": Pt(14), "bold": False, "color": ACCENT, "space_after": Pt(24)}, + "Heading 1": {"size": Pt(22), "bold": True, "color": DARK, "space_before": Pt(36), "space_after": Pt(12)}, + "Heading 2": {"size": Pt(16), "bold": True, "color": ACCENT, "space_before": Pt(24), "space_after": Pt(8)}, + "Heading 3": {"size": Pt(13), "bold": True, "color": ACCENT, "space_before": Pt(18), "space_after": Pt(6)}, + "Normal": {"size": Pt(11), "bold": False, "color": BODY, "space_after": Pt(8), "line_spacing": Pt(16)}, + "Body Text": {"size": Pt(11), "bold": False, "color": BODY, "space_after": Pt(8), "line_spacing": Pt(16)}, + "First Paragraph": {"size": Pt(11), "bold": False, "color": BODY, "space_after": Pt(8), "line_spacing": Pt(16)}, + "Block Text": {"size": Pt(10), "bold": False, "color": BODY}, + "List Paragraph": {"size": Pt(11), "color": BODY}, + "Compact": {"size": Pt(11), "color": BODY}, + } + + for style_name, props in style_defs.items(): + try: + style = doc.styles[style_name] + except KeyError: + continue + + style.font.name = "Calibri" + if "size" in props: + style.font.size = props["size"] + if "bold" in props: + style.font.bold = props["bold"] + if "color" in props: + style.font.color.rgb = RGBColor(*props["color"]) + + pf = style.paragraph_format + for attr in ("space_before", "space_after", "line_spacing"): + if attr in props: + setattr(pf, attr, props[attr]) + if "alignment" in props: + pf.alignment = props["alignment"] + + try: + hl = doc.styles["Hyperlink"] + hl.font.color.rgb = RGBColor(*ACCENT) + hl.font.underline = True + except KeyError: + pass + + doc.save(output_path) + + +def postprocess_docx(docx_path): + """Apply table styling, page breaks, and font normalization.""" + from docx import Document + from docx.shared import Pt, RGBColor + from docx.enum.table import WD_TABLE_ALIGNMENT + from docx.oxml.ns import nsdecls, qn + from docx.oxml import parse_xml + + doc = Document(docx_path) + + # Style tables + for table in doc.tables: + table.alignment = WD_TABLE_ALIGNMENT.CENTER + + if not table.rows: + continue + + # Header row + for cell in table.rows[0].cells: + shading = parse_xml(f'') + cell._tc.get_or_add_tcPr().append(shading) + for paragraph in cell.paragraphs: + for run in paragraph.runs: + run.font.color.rgb = RGBColor(0xFF, 0xFF, 0xFF) + run.font.bold = True + run.font.size = Pt(10) + run.font.name = "Calibri" + + # Data rows with alternating shading + for i, row in enumerate(table.rows[1:], 1): + for cell in row.cells: + if i % 2 == 0: + shading = parse_xml(f'') + cell._tc.get_or_add_tcPr().append(shading) + for paragraph in cell.paragraphs: + for run in paragraph.runs: + run.font.size = Pt(10) + run.font.name = "Calibri" + + # Table borders + tbl_pr = table._tbl.tblPr + if tbl_pr is None: + tbl_pr = parse_xml(f'') + table._tbl.insert(0, tbl_pr) + + existing = tbl_pr.find(qn("w:tblBorders")) + if existing is not None: + tbl_pr.remove(existing) + + borders = parse_xml( + f'' + ' ' + ' ' + ' ' + ' ' + ' ' + ' ' + "" + ) + tbl_pr.append(borders) + + # Page break before "References" heading + for para in doc.paragraphs: + if para.text.strip() == "References" and para.style.name.startswith("Heading"): + pPr = para._p.get_or_add_pPr() + pPr.append(parse_xml(f'')) + break + + # Ensure all runs have Calibri + for para in doc.paragraphs: + for run in para.runs: + if run.font.name is None: + run.font.name = "Calibri" + + doc.save(docx_path) + + +def main(): + parser = argparse.ArgumentParser( + description="Convert a Markdown file to a professionally styled Word document (.docx).", + prog="md-to-docx", + ) + parser.add_argument("input", help="Path to the Markdown file to convert") + parser.add_argument("-o", "--output", help="Output file path (default: .docx)") + parser.add_argument("-t", "--title", help="Override document title") + parser.add_argument("--toc", action="store_true", help="Include a table of contents") + parser.add_argument("--toc-depth", type=int, default=2, choices=[1, 2, 3], help="TOC depth (default: 2)") + parser.add_argument("-n", "--dryrun", action="store_true", help="Preview without creating files") + args = parser.parse_args() + + # Validate input + if not os.path.isfile(args.input): + print(f"Error: file not found: {args.input}", file=sys.stderr) + sys.exit(1) + + input_path = os.path.abspath(args.input) + if args.output: + output_path = os.path.abspath(args.output) + else: + output_path = os.path.splitext(input_path)[0] + ".docx" + + title = args.title or extract_title(input_path) + pandoc_version, docx_version = check_dependencies() + + if args.dryrun: + print(f"[dryrun] Input: {input_path}") + print(f"[dryrun] Output: {output_path}") + print(f"[dryrun] Title: {title}") + toc_str = f"yes (depth: {args.toc_depth})" if args.toc else "no" + print(f"[dryrun] TOC: {toc_str}") + print(f"[dryrun] Dependencies OK: pandoc {pandoc_version}, python-docx {docx_version}") + print("[dryrun] Would generate styled .docx file") + return + + # Create styled reference doc in a temp directory + tmpdir = tempfile.mkdtemp(prefix="md-to-docx-") + ref_path = os.path.join(tmpdir, "reference.docx") + + try: + create_styled_reference(ref_path) + + # Build pandoc command + cmd = [ + "pandoc", input_path, + "-o", output_path, + f"--reference-doc={ref_path}", + "--metadata", f"title={title}", + "-f", "markdown+pipe_tables+inline_notes", + "--columns=72", + ] + if args.toc: + cmd.extend(["--toc", f"--toc-depth={args.toc_depth}"]) + + result = subprocess.run(cmd, capture_output=True, text=True) + if result.returncode != 0: + print(f"Error: pandoc failed:\n{result.stderr}", file=sys.stderr) + sys.exit(1) + + postprocess_docx(output_path) + print(output_path) + + finally: + shutil.rmtree(tmpdir, ignore_errors=True) + + +if __name__ == "__main__": + main() diff --git a/specs/md-to-docx.spec.md b/specs/md-to-docx.spec.md new file mode 100644 index 0000000..ffb95f3 --- /dev/null +++ b/specs/md-to-docx.spec.md @@ -0,0 +1,98 @@ +# md-to-docx + +## Purpose + +Convert a Markdown file to a professionally styled Word document (.docx) suitable for sharing with non-technical stakeholders (managers, C-levels, partners). + +## Dependencies + +- `pandoc` (3.x+) — Markdown to docx conversion +- `python3` with `python-docx` package — post-processing and styling + +## Usage + +``` +md-to-docx [OPTIONS] +``` + +### Arguments + +| Argument | Description | +|---|---| +| `` | Path to the Markdown file to convert (required) | + +### Options + +| Flag | Description | +|---|---| +| `--output`, `-o` | Output file path (default: same name as input with `.docx` extension) | +| `--title`, `-t` | Override document title (default: first `# heading` in the markdown) | +| `--toc` | Include a table of contents (default: off) | +| `--toc-depth` | TOC depth level, 1-3 (default: 2, only applies when `--toc` is set) | +| `--dryrun`, `-n` | Preview what would happen without creating files | +| `--help`, `-h` | Show usage information | + +## Behaviour + +1. **Validate inputs.** Check that `` exists and is readable. Check that `pandoc` is on PATH. Check that `python-docx` is importable. Exit 1 with a clear message if any check fails. + +2. **Extract title.** If `--title` is not provided, scan the markdown for the first `# heading` and use its text as the document title. If no heading found, use the filename (without extension). + +3. **Create styled reference doc.** Generate a temporary pandoc reference.docx and apply professional styling: + - Font: Calibri throughout + - Headings: navy (#1A1A2E) for H1, professional blue (#1B4D89) for H2/H3 + - Body: dark grey (#333333), 11pt, 16pt line spacing + - Comfortable margins (2.5cm all sides) + - Hyperlinks: blue (#1B4D89), underlined + +4. **Run pandoc.** Convert the markdown to docx using the styled reference doc. Include `--toc` if requested. + +5. **Post-process the docx.** Using python-docx: + - Style tables: blue header row with white text, alternating row shading, light grey borders + - Add page breaks before the "References" heading (if present) + - Ensure all runs have Calibri font set + +6. **Clean up.** Remove temporary reference doc. + +7. **Report.** Print the output path on success. + +## Dryrun Behaviour + +When `--dryrun` / `-n` is passed: + +``` +[dryrun] Input: /path/to/PROPOSAL.md +[dryrun] Output: /path/to/PROPOSAL.docx +[dryrun] Title: Building an Agentic Development Platform... +[dryrun] TOC: yes (depth: 2) +[dryrun] Dependencies OK: pandoc 3.1.3, python-docx 1.2.0 +[dryrun] Would generate styled .docx file +``` + +No files are created or modified. + +## Edge Cases + +- **Input file not found:** Exit 1 with `Error: file not found: ` +- **pandoc not installed:** Exit 1 with `Error: pandoc not found. Install with: sudo apt install pandoc` +- **python-docx not installed:** Exit 1 with `Error: python-docx not installed. Install with: pip install python-docx` +- **Output file already exists:** Overwrite without prompting (standard pipeline behaviour) +- **Markdown has no headings:** Use filename as title, skip TOC even if requested +- **Markdown has no tables:** Table styling step is a no-op +- **Markdown has no "References" section:** Page break insertion is a no-op + +## Examples + +```bash +# Basic conversion +md-to-docx PROPOSAL.md + +# Custom output path and title +md-to-docx -o ~/Documents/proposal-v2.docx -t "Agent Platform Proposal" PROPOSAL.md + +# With table of contents +md-to-docx --toc --toc-depth 3 SPEC.md + +# Preview without creating files +md-to-docx --dryrun PROPOSAL.md +``` diff --git a/tests/test-md-to-docx.sh b/tests/test-md-to-docx.sh new file mode 100755 index 0000000..5815ab3 --- /dev/null +++ b/tests/test-md-to-docx.sh @@ -0,0 +1,140 @@ +#!/usr/bin/env bash +# Test md-to-docx script via --dryrun mode +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")/.." && pwd)" +SCRIPT="$SCRIPT_DIR/scripts/md-to-docx" +PASS=0 +FAIL=0 +TMPDIR=$(mktemp -d) + +cleanup() { rm -rf "$TMPDIR"; } +trap cleanup EXIT + +green() { printf "\033[32m%s\033[0m\n" "$1"; } +red() { printf "\033[31m%s\033[0m\n" "$1"; } + +assert_contains() { + local label="$1" output="$2" expected="$3" + if echo "$output" | grep -qF "$expected"; then + green " PASS: $label" + PASS=$((PASS + 1)) + else + red " FAIL: $label" + red " Expected to contain: $expected" + red " Got: $output" + FAIL=$((FAIL + 1)) + fi +} + +assert_exit_code() { + local label="$1" actual="$2" expected="$3" + if [ "$actual" -eq "$expected" ]; then + green " PASS: $label" + PASS=$((PASS + 1)) + else + red " FAIL: $label (expected exit $expected, got $actual)" + FAIL=$((FAIL + 1)) + fi +} + +# --- Setup test fixtures --- +cat > "$TMPDIR/test.md" << 'EOF' +# Test Document Title + +## Introduction + +This is a test document with a table. + +| Column A | Column B | +|---|---| +| Value 1 | Value 2 | + +## References + +Some references here. +EOF + +cat > "$TMPDIR/no-heading.md" << 'EOF' +This file has no headings at all. +Just plain text. +EOF + +# ============================================================ +echo "=== Test: dryrun with defaults ===" +output=$("$SCRIPT" --dryrun "$TMPDIR/test.md" 2>&1) || true +assert_contains "shows input path" "$output" "[dryrun] Input:" +assert_contains "shows output path" "$output" "[dryrun] Output:" +assert_contains "extracts title from heading" "$output" "Test Document Title" +assert_contains "shows TOC off" "$output" "TOC: no" +assert_contains "shows dependencies" "$output" "Dependencies OK: pandoc" +assert_contains "shows would generate" "$output" "Would generate styled .docx file" + +# ============================================================ +echo "=== Test: dryrun with --toc ===" +output=$("$SCRIPT" --dryrun --toc --toc-depth 3 "$TMPDIR/test.md" 2>&1) || true +assert_contains "shows TOC on with depth" "$output" "TOC: yes (depth: 3)" + +# ============================================================ +echo "=== Test: dryrun with --title override ===" +output=$("$SCRIPT" --dryrun -t "Custom Title" "$TMPDIR/test.md" 2>&1) || true +assert_contains "uses custom title" "$output" "Custom Title" + +# ============================================================ +echo "=== Test: dryrun with --output override ===" +output=$("$SCRIPT" --dryrun -o "$TMPDIR/custom.docx" "$TMPDIR/test.md" 2>&1) || true +assert_contains "uses custom output path" "$output" "custom.docx" + +# ============================================================ +echo "=== Test: dryrun with no-heading file ===" +output=$("$SCRIPT" --dryrun "$TMPDIR/no-heading.md" 2>&1) || true +assert_contains "falls back to filename" "$output" "no-heading" + +# ============================================================ +echo "=== Test: missing input file ===" +output=$("$SCRIPT" --dryrun "$TMPDIR/nonexistent.md" 2>&1) || rc=$? +# Capture exit code properly +set +e +"$SCRIPT" --dryrun "$TMPDIR/nonexistent.md" > /dev/null 2>&1 +rc=$? +set -e +assert_exit_code "exits non-zero for missing file" "$rc" 1 + +output=$("$SCRIPT" --dryrun "$TMPDIR/nonexistent.md" 2>&1) || true +assert_contains "shows file not found error" "$output" "Error: file not found" + +# ============================================================ +echo "=== Test: --help flag ===" +output=$("$SCRIPT" --help 2>&1) || true +assert_contains "shows usage info" "$output" "Markdown" + +# ============================================================ +echo "=== Test: actual conversion (creates file) ===" +"$SCRIPT" "$TMPDIR/test.md" > /dev/null 2>&1 || true +if [ -f "$TMPDIR/test.docx" ]; then + green " PASS: docx file created" + PASS=$((PASS + 1)) + + # Check it's a valid zip (docx is a zip) + if file "$TMPDIR/test.docx" | grep -q "Zip\|Microsoft"; then + green " PASS: output is valid docx/zip format" + PASS=$((PASS + 1)) + else + red " FAIL: output is not a valid docx file" + FAIL=$((FAIL + 1)) + fi +else + red " FAIL: docx file not created" + FAIL=$((FAIL + 1)) + FAIL=$((FAIL + 1)) # count the format check as failed too +fi + +# ============================================================ +echo "" +echo "===============================" +if [ "$FAIL" -eq 0 ]; then + green "All $PASS tests passed." +else + red "$FAIL tests failed, $PASS passed." + exit 1 +fi