distill: best practices from 2026-04-19 cross-project run
Adds 3 new topic files (ai-parallel-agents, api-integration, python-patterns) and extends 21 existing topic files with new gotchas and patterns surfaced from memory across tracked projects. Index updated accordingly. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -1,41 +1,101 @@
|
||||
{
|
||||
"version": 1,
|
||||
"last_run": "2026-04-05T01:13:48Z",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z",
|
||||
"projects": {
|
||||
"agent-runtimes": {
|
||||
"path": "/home/paul/dev/claude/projects/agent-runtimes",
|
||||
"last_sha": "9cbef57e08c28b1ca68097178e02093c0835c7d7",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"last_sha": "f0a5c22477fe3ed9d97b95f9fecb02ac6671d170",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"claude-foundations": {
|
||||
"path": "/home/paul/dev/claude/projects/claude-foundations",
|
||||
"last_sha": "52b0f2b5c63954363a27dbb21edc3890e35cc43e",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"last_sha": "6dfa20c47ccd293d6c5de99524ed474f8f3ceef8",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"cluster-apps/octopus-deploy": {
|
||||
"path": "/home/paul/dev/claude/projects/cluster-apps/octopus-deploy",
|
||||
"last_sha": "3f23b8a2c3d4a120599f5a38a654e4af2419672f",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"cluster-bootstrap": {
|
||||
"path": "/home/paul/dev/claude/projects/cluster-bootstrap",
|
||||
"last_sha": "3623b88175e60908f81c099ddb03974436af4b72",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"last_sha": "9536d0880c308310b71d5e6b41c00bd0a3ef73c3",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"custom-claude-skills": {
|
||||
"path": "/home/paul/dev/claude/projects/custom-claude-skills",
|
||||
"last_sha": "3b954ff02afe8231ee43be92119a56d5763c90e0",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"last_sha": "7d7856a07971958c3e0b98720edeb685bac68678",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"hugo-accelerator": {
|
||||
"path": "/home/paul/dev/claude/projects/hugo-accelerator",
|
||||
"last_sha": "e56094d99b2a446d09911e32b381f9e411510fdf",
|
||||
"last_run": "2026-03-27T04:18:44Z"
|
||||
"last_sha": "206ab9bc11f60305beede2a2522dbcf794b6306b",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"small-scripts": {
|
||||
"path": "/home/paul/dev/claude/small-scripts",
|
||||
"last_sha": "529e49fe9803622f734974d46fec71ccb4de4d55",
|
||||
"last_run": "2026-04-05T01:13:48Z"
|
||||
"path": "/home/paul/dev/claude/projects/small-scripts",
|
||||
"last_sha": "5a1b4b14bc13fbc0a626f3450b913b3f788e4de4",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"ai-image-gen": {
|
||||
"path": "/home/paul/dev/claude/projects/ai-image-gen",
|
||||
"last_sha": "89c8605470e0188a5ad3765e6ce3b055adefba67",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"brainiac-app": {
|
||||
"path": "/home/paul/dev/claude/projects/brainiac-app",
|
||||
"last_sha": "7ff7a4ce605275fc7bf8b9fc664604af3e55e381",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"contracts": {
|
||||
"path": "/home/paul/dev/claude/projects/contracts",
|
||||
"last_sha": "d56c64f6ae95224c5ae68284ff7d3c21ede7a0c2",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"crud-accelerator": {
|
||||
"path": "/home/paul/dev/claude/projects/crud-accelerator",
|
||||
"last_sha": "b61b64b723ace6f9c7e6f4be94766955625cf90c",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"dns-manager": {
|
||||
"path": "/home/paul/dev/claude/projects/dns-manager",
|
||||
"last_sha": "f4ce7850a338e9ea35aebf64a1fd6461f955f011",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/ai-assisted-migration": {
|
||||
"path": "/home/paul/dev/claude/octopus/ai-assisted-migration",
|
||||
"last_sha": "761b7cb744a6eba575437a08d19844a7ceb0e5ab",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/customer-issue-sync": {
|
||||
"path": "/home/paul/dev/claude/octopus/customer-issue-sync",
|
||||
"last_sha": "ffeb6ea2409b35bbfcdf71db2710181ea44012a8",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/customers/duel-image": {
|
||||
"path": "/home/paul/dev/claude/octopus/customers/duel-image",
|
||||
"last_sha": "176c129af07ffaae4392ca9b82fa6c28c5be0909",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/customers/slb": {
|
||||
"path": "/home/paul/dev/claude/octopus/customers/slb",
|
||||
"last_sha": "bf4bf160a492ce6ea8880f9564a3b4d71e2fe2f5",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/goes": {
|
||||
"path": "/home/paul/dev/claude/octopus/goes",
|
||||
"last_sha": "b39e9c77f9e214bdb71f47543ad5eb821ae31c32",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/PlatformHub-Demo": {
|
||||
"path": "/home/paul/dev/claude/octopus/PlatformHub-Demo",
|
||||
"last_sha": "d8d039180b424ed0b0e56bc400228d2a51899804",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
},
|
||||
"octopus/the-case-for-agentic": {
|
||||
"path": "/home/paul/dev/claude/octopus/the-case-for-agentic",
|
||||
"last_sha": "69f3ca10de9994ed5386bd76f44fad7b1c23e8c4",
|
||||
"last_run": "2026-04-19T11:13:09.870821Z"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -4,27 +4,30 @@ Generalised best practices extracted from real project work via the `/distill-be
|
||||
|
||||
## Topics
|
||||
|
||||
- [Validation & Deployment](validation.md) — Validate locally, deploy once; full-chain testing; pre-flight checks; DB migration patterns; K8s constraint planning; deployment checklists
|
||||
- [Validation & Deployment](validation.md) — Validate locally, deploy once; full-chain testing; pre-flight checks; DB migration patterns; K8s constraint planning; deployment checklists; inert-by-default feature flags; integration failure categorisation; safe persistence pattern
|
||||
- [Security Architecture](security-architecture.md) — Server boundary rule: no credential crosses to the client; proxy + identity mapping pattern; defense in depth; anti-patterns
|
||||
- [Secrets Management](secrets-management.md) — SOPS + age, credential handling, file naming, encryption gotchas, .env source injection, URL-safe passwords
|
||||
- [Secrets Management](secrets-management.md) — SOPS + age, credential handling, file naming, encryption gotchas, .env source injection, URL-safe passwords, per-workload secret scoping
|
||||
- [Git & Source Control](git-source-control.md) — Commit practices, GitOps workflows, remote conventions
|
||||
- [Kubernetes Patterns](kubernetes.md) — Volume mounts, deployment strategies, naming, bootstrap ordering, ArgoCD SSA quirks, etcd tuning, Secret volume gotchas
|
||||
- [Kubernetes Patterns](kubernetes.md) — Volume mounts, deployment strategies, naming, bootstrap ordering, ArgoCD SSA quirks, etcd tuning, Secret volume gotchas, probe timeouts, Kustomize overlay image overrides, PodSecurity for monitoring, Cilium entity identities
|
||||
- [Helm Charts](helm.md) — Schema validation, version verification, values structure
|
||||
- [Ansible](ansible.md) — Inventory, templates, idempotency, credential safety
|
||||
- [Scripting](scripting.md) — Shell conventions, verification scripts, idempotency, colour output
|
||||
- [Documentation Standards](documentation.md) — CLAUDE.md, MEMORY.md, FUTURE.md, README.md structure and tiered memory
|
||||
- [Milestones & Reflections](milestones.md) — Milestone workflow, verification, reflection process
|
||||
- [Debugging Methodology](debugging.md) — Systematic diagnosis, full-chain testing, common pitfalls
|
||||
- [Debugging Methodology](debugging.md) — Systematic diagnosis, full-chain testing, common pitfalls, DB schema verification after deploy
|
||||
- [Claude Code Skills](skills-development.md) — Skill authoring, context injection, tool restrictions, read-only review skills, formatter/hook separation
|
||||
- [Linting & Formatting](linting.md) — Tool choices per language, PostToolUse hook, pre-commit integration, formatter contract
|
||||
- [Spec-Driven Development](spec-driven-development.md) — Spec structure, requirement numbering, test-first workflow, multi-model review, plan-first approach, agent prompt conventions
|
||||
- [Spec-Driven Development](spec-driven-development.md) — Spec structure, requirement numbering, test-first workflow, multi-model review, plan-first approach, agent prompt conventions, wave-based TDD dispatch
|
||||
- [Test-Driven Development](test-driven-development.md) — Edge case discovery, property-based testing, mutation testing, AI agent testing patterns, test architecture
|
||||
- [Networking & Infrastructure](networking.md) — nftables safety, systemd socket activation, Docker forwarding, TLS SNI vs Host header, wildcard certs
|
||||
- [Docker UID Matching](docker-uid-matching.md) — UID wrapper entrypoint for mounted volumes, gosu pattern, when to use vs K8s securityContext
|
||||
- [Database Selection](database-selection.md) — SQLite is not a production database; always use PostgreSQL for services with FQDNs, multiple consumers, or concurrent access
|
||||
- [Docker](docker.md) — gosu PID 1, GIT_SSH_COMMAND scope, slim image health checks, buildx local images, Compose networking/restart gotchas, volume paths, override merge behaviour
|
||||
- [Docker](docker.md) — gosu PID 1, GIT_SSH_COMMAND scope, slim image health checks, buildx local images, Compose networking/restart gotchas, volume paths, override merge behaviour, init script privilege order, payload size limits, bind-mount rm gotcha
|
||||
- [API Design](api-design.md) — Transport security, auth (OAuth2/JWT/mTLS), versioning, pagination, error handling, idempotency, rate limiting, input validation, zero-trust patterns
|
||||
- [API Integration](api-integration.md) — Client-side third-party API integration: capability verification, app-layer compensation, git+SOPS polling sync, bidirectional SoR
|
||||
- [Octopus Process Templates](octopus-process-templates.md) — OCL syntax, step template references, channel scoping, parameters, versioning, Platform Hub patterns
|
||||
- [LLM Code Security](llm-code-security.md) — Security vulnerabilities in AI-generated code: injection flaws, hardcoded secrets, hallucinated packages, over-permissive defaults, IaC risks, crypto mistakes, review checklist
|
||||
- [CI Container Builds](ci-container-builds.md) — Registry cache with inline metadata, buildx in DinD, layer ordering, pip caching, path filter gotchas, SHA tagging strategy
|
||||
- [Agent Repos & Container Agents](agent-repos.md) — Task submission, harnesses, monitoring, multi-model workflows, agent repo forks, workspace layout, artifact passing via git branches
|
||||
- [LLM Code Security](llm-code-security.md) — Security vulnerabilities in AI-generated code: injection flaws, hardcoded secrets, hallucinated packages, over-permissive defaults, IaC risks, crypto mistakes, operational vulnerabilities (idempotency, CI/CD integrity, supply chain provenance, concurrent access), review checklists
|
||||
- [CI Container Builds](ci-container-builds.md) — Registry cache with inline metadata, buildx in DinD, layer ordering, pip caching, path filter gotchas, SHA tagging strategy, runtime-mounted directory triggers
|
||||
- [Agent Repos & Container Agents](agent-repos.md) — Task submission, harnesses, monitoring, multi-model workflows, agent repo forks, workspace layout, artifact passing via git branches, read-only test protection, infrastructure failure modes, cost-effective model scope boundaries
|
||||
- [AI Parallel Agents](ai-parallel-agents.md) — Parallel agent orchestration: multi-facet research dispatch, file contention, WebFetch limits, narrow reads, dataset-wide audits
|
||||
- [Python Patterns](python-patterns.md) — Non-reentrant Lock deadlocks, Pydantic v2 extra='ignore' silent drops, subprocess routing callables for mocking, model_validator for cross-field validation
|
||||
|
||||
@@ -413,3 +413,63 @@ curl -s -X DELETE http://localhost:8100/tasks/{id}
|
||||
## Artifact Passing via Git Branches Instead of Env Vars
|
||||
|
||||
For multi-stage workflows where downstream tasks need upstream outputs, push artifacts to branches in an agent repo rather than embedding in prompts or env vars. This avoids K8s env var size limits (~228KB), survives pod restarts, provides a git audit trail, and scales to any artifact size. The downstream task clones the branch as a reference directory.
|
||||
|
||||
## Protect Test Files from Agent Modification via Root-Owned Read-Only Clone
|
||||
|
||||
When agents run tests, they may "fix" failing tests by weakening assertions rather than fixing the underlying code. Prevent this by cloning the test suite into `/workspace/reference/tests/` as a root-owned directory (the agent gets a permission error if it tries to write). The agent's working directory gets a symlink or copy of the tests at startup, but the authoritative copy is immutable. This is the same pattern as `/workspace/reference/main/` for source code — root ownership makes modification a hard error, not a policy.
|
||||
|
||||
## Infrastructure Failures Dominate Agent Failure Modes
|
||||
|
||||
In measured agent runs, the majority of task failures are infrastructure failures, not agent reasoning failures: network timeouts, SSH key not loaded, missing package in the base image, environment variable not propagated. Before debugging agent behaviour, check whether the failure is environmental — a task that consistently fails at "git clone" is an infrastructure problem, not an agent problem.
|
||||
|
||||
**Concrete ratio:** in one measured 12-task batch, 6/12 tasks failed and all 6 were infrastructure-class (wrong harness, missing payload fields, stale images) — zero model failures. Task templates that validate payload structure and harness compatibility before dispatch eliminate the entire dominant failure mode.
|
||||
|
||||
**Pre-dispatch validation checklist:**
|
||||
- SSH key reachable from the agent harness (test clone before dispatching)
|
||||
- Required env vars present (model API keys, registry credentials)
|
||||
- Harness image has all required tools (`uv`, `ruff`, `pytest`, etc.)
|
||||
- Target repo and branch exist
|
||||
- Network egress allows required domains
|
||||
- **Task templates** — long-term remediation. Validate payload structure and harness compatibility at template-render time, not via ad-hoc per-dispatch checks.
|
||||
|
||||
Invest in pre-dispatch validation scripts that catch the top-N infrastructure failures before the first agent container starts.
|
||||
|
||||
## Cross-Model Reviews Catch ~38% More Issues Than a Single Model
|
||||
|
||||
Running the same security or spec review with two different models (e.g., Opus + MiniMax) and comparing outputs catches ~38% more issues than either alone — in measured reviews, only 62% of findings overlap. Models converge on obvious issues but diverge on edge cases and design concerns. Worth the extra cost for security-critical specs and architecture reviews; overkill for routine code review.
|
||||
|
||||
**Pattern:** dispatch parallel review tasks to different models with identical prompts, union the findings, deduplicate against a shared issue key (file + line + category). Present the merged list to the human reviewer along with per-model attribution so reviewers can see where models agreed vs. diverged.
|
||||
|
||||
## Agent Worktree Branches Contain Files, Not Commits — Copy, Don't Merge
|
||||
|
||||
**Symptom:** Orchestrator merges an agent's branch and sees "Already up to date" because the agent wrote files to its worktree but never ran `git add` / `git commit`. Downstream tasks that depend on the upstream artifact then fail or silently use stale data.
|
||||
|
||||
**Fix:** Orchestration must explicitly copy files from dependency worktrees (driven by a `writes` field in the task manifest) into the consuming worktree. Git merge is insufficient when agent output is untracked.
|
||||
|
||||
**Alternative:** Require agents to commit before exit (finalize phase auto-commits everything under `/workspace/working/`), which unlocks git-branch artifact passing. The finalize-phase auto-commit described above is the canonical implementation — enforce it for any agent whose output other tasks depend on.
|
||||
|
||||
## Include Exact Dataclass/Context Schemas in Template-Writing Agent Prompts
|
||||
|
||||
**Symptom:** Agents writing templates invent their own mock context objects (e.g., dict-style `model["fields"]`) while real code provides a different shape (e.g., dataclass `model.fields`). Templates render against the mocks but produce attribute errors against real objects during integration.
|
||||
|
||||
**Fix:** Always include exact dataclass/type definitions of the render context in the agent prompt. During review, compare each agent's mock objects against the real normalized types before merging. Treat "wrote its own mock shape" as a review-blocking issue — the mock shape is a contract the agent must follow, not invent.
|
||||
|
||||
## Never Assume Web Search in Container Agents; Validate Version-Specific Claims Separately
|
||||
|
||||
**Symptom:** Agents in Docker/K8s containers have no `WebSearch` or `WebFetch` capability even with `--dangerously-skip-permissions`. Version numbers, release dates, and "actively maintained" claims come from training data and are often wrong (~30% inaccuracy on package versions in measured runs).
|
||||
|
||||
**Fix:**
|
||||
- Never tell a container agent to "use web search" — it has none and will silently fabricate from training data.
|
||||
- Run a separate web-validation pass (Opus or similar with network access) after research agents complete.
|
||||
- Treat all version claims as hypotheses until validated.
|
||||
- Budget web validation as a distinct pipeline stage, not an afterthought.
|
||||
|
||||
## Cost-Effective Models Need Explicit Scope Boundaries
|
||||
|
||||
Smaller/cheaper models (e.g., MiniMax, Haiku) need tighter scope constraints than capable flagship models. Without explicit boundaries, they drift into scope creep, run the full test suite when asked to write tests, or attempt broad refactors. For cost-effective model tasks:
|
||||
- **No full test suite runs** — specify which test file or test ID to run
|
||||
- **Concrete patterns, not open-ended** — "Write a test matching `test_cp_*.py` naming" not "Write tests for the control plane"
|
||||
- **Longer timeouts** — cheaper models are often slower per token; set `runtime.timeout` to 2-3× what flagship models need
|
||||
- **Explicit output location** — "Write to `results/output.md`" not "Write your findings"
|
||||
|
||||
Treat scope boundaries as a harness concern, not an agent concern — encode them in the prompt template or harness context, not in ad-hoc task prompts.
|
||||
|
||||
111
ai-parallel-agents.md
Normal file
111
ai-parallel-agents.md
Normal file
@@ -0,0 +1,111 @@
|
||||
# AI Parallel Agents
|
||||
|
||||
Patterns for orchestrating parallel AI agents inside a Claude Code (or similar) workflow — research fan-out, file-contention avoidance, tool-capability limits, context-budget discipline, and dataset-wide audits via background agents.
|
||||
|
||||
Cross-references: [Agent Repos & Container Agent Operations](agent-repos.md) covers the container-agent execution environment (harnesses, agent repos, CP dispatch). This file covers the orchestration patterns a main-thread Claude Code session uses when spawning and coordinating subagents.
|
||||
|
||||
---
|
||||
|
||||
## 1. Dispatch Parallel Agents for Multi-Facet Research
|
||||
|
||||
**Principle:** When researching a topic with multiple independent facets, dispatch parallel research agents rather than running sequential queries from the main thread.
|
||||
|
||||
**Why it matters:** Sequential investigation delays cross-facet contradiction discovery to the point where it is expensive to address (often after a plan has been written). Scoping each agent narrowly to one facet — per API, per jurisdiction, per sub-topic, per source-type filter — both parallelises the work and improves signal-to-noise because each agent's context is tightly focused on one domain.
|
||||
|
||||
**How to implement:**
|
||||
- Enumerate the facets of the research topic before dispatching. For an integration project, facets are usually "per external API". For a compliance question, facets are usually "per jurisdiction". For a market survey, facets are usually "per source category" (vendor docs, academic papers, blog posts).
|
||||
- Spawn one agent per facet, each with a narrow prompt: "Research X specifically in the context of Y. Do not cover adjacent topics."
|
||||
- For integration work that touches 2+ external APIs, spawn one research agent per API, plus a parallel plan-reviewer agent that checks cross-API compatibility as soon as the per-API agents report back.
|
||||
- Merge results in the main thread — deduplicate, flag contradictions, and surface cross-facet constraints.
|
||||
|
||||
**Anti-patterns:**
|
||||
- Running a single agent with "research topics X, Y, and Z" — context dilution, and any blocker in X delays Y and Z.
|
||||
- Sequential WebFetch calls from the main thread when the facets are independent.
|
||||
- Dispatching parallel agents with overlapping scope — they return the same information and the dedup cost eats the parallelism savings.
|
||||
- Waiting for research agent N to finish before dispatching reviewer agents — run them concurrently when they have no data dependency.
|
||||
|
||||
---
|
||||
|
||||
## 2. File Contention: Agents Return Text, Main Thread Writes
|
||||
|
||||
**Principle:** Never have two parallel agents edit the same file. Have agents RETURN prepared text in their final message and apply edits sequentially from the main thread.
|
||||
|
||||
**Why it matters:** Subagent writes to a shared file silently collide — one edit wins, the other is lost, and there is no error or warning. Verified in practice during a 195-claim verification across 16 files: 4 parallel agents, each handling 3-4 files, returning their prepared edits as text, applied by the main thread — zero conflicts, one round.
|
||||
|
||||
**How to implement:**
|
||||
- Partition the file set so each agent owns a non-overlapping subset. Never assign the same file to two agents.
|
||||
- Instruct agents explicitly: "Return your proposed edits as text in your final message. Do not write to disk." Include an example of the expected return format (e.g., file path + old/new block per edit).
|
||||
- When overlap is unavoidable (e.g., a cross-cutting change to every file), have agents RETURN their changes; apply edits from the main thread in a deterministic order.
|
||||
- For large batches, consider a two-phase pattern: parallel agents produce proposed edits as text; main thread reviews and applies.
|
||||
|
||||
**Anti-patterns:**
|
||||
- "Each agent edits the files relevant to its findings" — guaranteed to produce silent write collisions when scopes overlap.
|
||||
- Letting agents write to a shared directory without partitioning — last write wins, earlier writes vanish.
|
||||
- Assuming filesystem-level locking will save you — Claude Code agents do not hold locks across calls.
|
||||
- Relying on git to detect the collision — an agent that reads the pre-collision state and writes after the other agent has the correct "new" content, and git sees only the last write.
|
||||
|
||||
---
|
||||
|
||||
## 3. Subagents Cannot WebFetch — Perform Fetches in Main Thread
|
||||
|
||||
**Principle:** Subagents cannot use `WebFetch` or `WebSearch` (permission denied by the harness). Perform web fetches in the main conversation and delegate file processing — reading, verification, extraction, summarisation — to agents.
|
||||
|
||||
**Why it matters:** Telling a subagent to "go fetch X from the web" fails silently from the orchestrator's perspective — the subagent returns what it "knows" (training data) rather than what's on the web today. For container agents the restriction is stricter still: no network tools at all, and training-data version claims have been measured at ~30% inaccuracy.
|
||||
|
||||
**How to implement:**
|
||||
- Use `WebFetch` / `WebSearch` in the main thread to pull live content into files or into the context.
|
||||
- Pass the fetched content to subagents as text (in their prompt) or as file paths they can Read.
|
||||
- For research agents that need multi-source fetches, fetch everything in the main thread first, then dispatch agents to process/summarise the local content.
|
||||
- For container agents, treat all external network access as unavailable — run a separate web-validation pass from the main conversation after the container agent completes.
|
||||
|
||||
**Anti-patterns:**
|
||||
- "Dispatch a subagent to research X on the web" — subagent has no web tools, will fabricate from training data.
|
||||
- Telling container agents to "use web search" — they have none and will not tell you so.
|
||||
- Trusting version numbers, release dates, or "actively maintained" claims from a subagent that had no network access — validate separately.
|
||||
|
||||
---
|
||||
|
||||
## 4. Narrow Read Instructions to Prevent Context Blowup
|
||||
|
||||
**Principle:** Give agents narrow read instructions so each agent's context stays small. Name specific sections, grep patterns, or line ranges rather than "read the file."
|
||||
|
||||
**Why it matters:** Parallel agents that each read every full file consume tokens without improving output quality. A 16-file, 195-claim verification run blows up if every agent reads every file; it stays cheap if each agent reads only the section relevant to its claim. Context is the scarce resource in multi-agent workflows.
|
||||
|
||||
**How to implement:**
|
||||
- In the agent prompt, specify the exact section or line range: "Read only the `## Key Data Points` section of each file" or "Read lines 120-180 of spec/ingestion.md."
|
||||
- When the agent needs to find the relevant section itself, instruct it to grep first and Read only matching line windows — not the whole file.
|
||||
- For large source trees, supply a curated file list; do not let the agent glob an entire repo.
|
||||
- Review agent prompts for accidental "read everything" phrasing — "look through the spec" is unbounded, "read Section 3.2 of the spec" is bounded.
|
||||
|
||||
**Anti-patterns:**
|
||||
- "Read all the files in the spec directory and tell me X" — N agents × M files = huge context burn.
|
||||
- Open-ended research prompts with no read boundaries — agents default to reading everything.
|
||||
- Long agent prompts that themselves include full file contents when a section would suffice.
|
||||
- Failing to cross-reference against an existing index file (CLAUDE.md, MEMORY.md) — forcing every agent to rediscover the repo structure.
|
||||
|
||||
---
|
||||
|
||||
## 5. Background Agents for Dataset-Wide Audits
|
||||
|
||||
**Principle:** When a question requires surveying every record in a large dataset — field gap analysis, distribution across files, unknown-unknowns discovery — dispatch a background (container) agent with a single well-scoped sweep prompt and a research-document output. One complete pass produces better results than piecemeal interactive exploration.
|
||||
|
||||
**Why it matters:** Interactive main-thread exploration of large datasets is slow, context-expensive, and prone to anchoring on whatever the operator looked at first. A background agent with a single sweep prompt can process the full dataset in one pass, produce a structured research artifact, and surface patterns the interactive session would miss.
|
||||
|
||||
**How to implement:**
|
||||
- Frame the audit as a complete question: "For every ticket in the dataset, classify by [dimensions]. Produce a research document at `research/<audit-name>.md` with sections [X, Y, Z] and a summary table."
|
||||
- Give the agent the full dataset location (repo path, data export, dataset manifest) and explicit output instructions.
|
||||
- Prefer one deeper pass over many shallow passes — the marginal cost of one big research run is usually lower than the cumulative cost of back-and-forth.
|
||||
- Store research outputs in a dedicated folder (e.g., `research/`) separate from code and docs. Humans review, distill the useful findings, and archive or discard the raw research.
|
||||
- For recurring audits (weekly, per-milestone), template the prompt so it can be re-run with a date/scope parameter.
|
||||
|
||||
**Anti-patterns:**
|
||||
- Running the sweep interactively in the main thread — slow and context-expensive.
|
||||
- Multiple overlapping sweeps with no consolidation — hard to reason about what's covered.
|
||||
- No dedicated output folder — research artifacts get lost alongside code.
|
||||
- Never reading the research output after dispatch — a "fire and forget" audit with no human review has no value.
|
||||
|
||||
---
|
||||
|
||||
## Summary
|
||||
|
||||
Parallel agent orchestration multiplies throughput when agents have narrowly scoped work, disjoint file sets, and the right tool capabilities. The main thread holds responsibilities subagents cannot do: web fetches, file writes on contested paths, and final integration of returned text. For large datasets, a single background sweep with a research-document output usually beats interactive back-and-forth.
|
||||
134
api-integration.md
Normal file
134
api-integration.md
Normal file
@@ -0,0 +1,134 @@
|
||||
# API Integration
|
||||
|
||||
Client-side patterns for integrating with third-party REST/HTTP APIs. Focused on the realities of consuming APIs you do not control — capability gaps, missing filters, absent webhooks, inconsistent REST semantics, and the bidirectional sync problems that arise when two external systems must stay aligned.
|
||||
|
||||
Cross-references: [API Design](api-design.md) covers server-side API design (the APIs you build). [Secrets Management](secrets-management.md) covers credential storage for API tokens. [Validation & Deployment](validation.md) covers full-chain integration testing.
|
||||
|
||||
---
|
||||
|
||||
## 1. Verify API Capabilities Before Locking Architecture
|
||||
|
||||
**Principle:** Probe critical assumptions against live docs or test API calls before committing to an architecture. Assume standard REST conventions do NOT hold on any API you have not personally verified.
|
||||
|
||||
**Why it matters:** Mid-planning discovery of a missing capability can force a full redesign. Real examples: an API that has no `modified_at` filter (so incremental sync is impossible), no issue-linking primitive (so cross-system references cannot be stored natively), no webhooks (so push-based sync is off the table), or PATCH semantics that replace rather than merge nested objects. Each of these changes the architecture in a way that is expensive to discover after code is written.
|
||||
|
||||
**How to implement:**
|
||||
- Before sketching an integration, enumerate the capabilities you are relying on: filter/modification support, PATCH vs PUT semantics, webhook availability, custom field types, linking/relationship primitives, pagination cursor stability, rate limits.
|
||||
- For each capability, find a live docs reference or run a test call against a sandbox. Do not infer from "it's REST, so it should have X."
|
||||
- Document the actual capability matrix in the plan (or spec) — explicitly list what the API does NOT support, not just what it does.
|
||||
- When an integration targets multiple APIs, run this probe per API before deciding on a shared abstraction.
|
||||
|
||||
**Anti-patterns:**
|
||||
- Assuming every REST API supports an `updated_since` or `modified_at` filter.
|
||||
- Assuming PATCH merges; assuming PUT is idempotent on nested structures.
|
||||
- Assuming webhooks exist because "it's a modern SaaS product."
|
||||
- Assuming custom fields support reference/relationship types when the docs only mention strings and numbers.
|
||||
- Deferring capability verification to "we'll find out when we build it."
|
||||
|
||||
---
|
||||
|
||||
## 2. Compensate in the Application Layer for Missing API Features
|
||||
|
||||
**Principle:** When an external API lacks a feature you need, design the application layer to compensate rather than forcing the API to behave like you want.
|
||||
|
||||
**Why it matters:** You cannot change a third-party API. The cost of hoping for a capability is architecture that only works in a future that may never come. Compensation patterns are well-understood and usually add acceptable overhead — but only if you plan for them up front.
|
||||
|
||||
**How to implement:**
|
||||
- **No `modified_at` filter?** Hash-compare every record against a stored hash to detect changes. On each run, fetch all records, compute a content hash, diff against the previous hashes, apply changes for records whose hash changed.
|
||||
- **No reference/relationship type for custom fields?** Store foreign-key IDs in text fields with a documented format (e.g., `ticket:12345`). Parse on read.
|
||||
- **No webhooks?** Poll on a cursor. Persist the cursor in state so each run resumes from the last processed position.
|
||||
- **No persistent state store in the execution environment?** Use a git-backed flat-file store (see section 3).
|
||||
- **No idempotency key support on writes?** Implement dedupe at the application layer using a `(external_id, operation, hash)` tuple stored in your own state.
|
||||
- **No batch endpoints?** Implement client-side batching with concurrency caps and retry/backoff.
|
||||
|
||||
**Anti-patterns:**
|
||||
- "Let's ask the vendor to add X" as the primary plan — maybe they will, maybe they won't, probably not on your timeline.
|
||||
- Polling every record on every run with no hash/cursor — works at 10 records, implodes at 10,000.
|
||||
- Storing relationship data in a second external system with no back-reference — creates orphaned state.
|
||||
- Hand-rolling a webhook receiver when the source doesn't push — build a poller with a cursor instead.
|
||||
|
||||
---
|
||||
|
||||
## 3. Polling-Based Sync With Git + SOPS as State Store
|
||||
|
||||
**Principle:** For scheduled or runbook-style syncs where no webhooks are available and no persistent compute exists, use a dedicated git repo with SOPS-encrypted JSON flat files as the state store.
|
||||
|
||||
**Why it matters:** Ephemeral execution contexts (Octopus runbooks, cron jobs, scheduled GitHub Actions, short-lived containers) have no local disk, no database, and no shared cache. Spinning up a database for a low-throughput sync is overkill; using an HTTP KV service adds another dependency with its own credentials and failure mode. A git repo is free infrastructure that every execution context already knows how to use.
|
||||
|
||||
**How to implement:**
|
||||
- Create a dedicated repo (e.g., `<project>-sync-state`) separate from the application code.
|
||||
- Store state as JSON files, one per logical entity (e.g., `records.json`, `cursor.json`, `hashes.json`). Keep files human-readable.
|
||||
- Encrypt with SOPS + age so secrets in state (external IDs, customer references, email addresses) are never exposed in plaintext.
|
||||
- Each run: `git clone --depth 1` → SOPS decrypt → read/write → SOPS encrypt → `git add && git commit && git push`.
|
||||
- Make operations **idempotent** so a retry after a push conflict is safe. On conflict: re-pull, re-apply, re-push. A second writer racing for the same state is rare for scheduled runbooks, and idempotency handles it without coordination.
|
||||
- Write a meaningful commit message per run (`sync 2026-04-19T08:00Z: 3 created, 1 updated, 0 deleted`) — this IS your audit trail.
|
||||
|
||||
**Benefits:**
|
||||
- No storage infrastructure to provision or secure.
|
||||
- Free audit trail via git log — every state change is attributable to a run.
|
||||
- Human-readable state files for debugging.
|
||||
- Encrypted at rest via SOPS; decrypted only in the execution context.
|
||||
- Works in any ephemeral runner with git + age installed.
|
||||
- Rollback is `git revert`.
|
||||
|
||||
**Anti-patterns:**
|
||||
- Storing secrets in state files without SOPS ("it's a private repo").
|
||||
- Non-idempotent operations against the external API — turns push conflicts into data loss.
|
||||
- Treating state files as a database with complex queries; if you need queries, use a real database.
|
||||
- Shared state repo across unrelated integrations — blast radius and credential coupling get worse over time.
|
||||
- No commit message content — you lose the audit trail benefit.
|
||||
|
||||
---
|
||||
|
||||
## 4. Bidirectional Sync: Declare a System of Record
|
||||
|
||||
**Principle:** Before implementing any bidirectional sync, pick ONE system as the system of record (SoR) and declare conflicts resolve in its favour. Document the direction explicitly.
|
||||
|
||||
**Why it matters:** Without a canonical side, conflict resolution becomes ad hoc, races between writers produce unpredictable state, and users lose trust in both systems (because they cannot predict which change will win). "Last write wins" is not a conflict resolution strategy — it is a coin flip that sometimes destroys the wrong side's work.
|
||||
|
||||
**How to implement:**
|
||||
- Pick the SoR based on where humans actually work — the system with the richer UI, the auditable trail, or the regulatory obligation usually wins.
|
||||
- Document the direction in the integration spec: "SoR is Zendesk. PlanHat fields X, Y, Z are mirrored from Zendesk on every sync. PlanHat-side edits to X, Y, Z are overwritten."
|
||||
- Implement the sync loop as one-way reads from the SoR and one-way writes to the other system(s). Any "reverse" flow is a separate, explicit sync with its own direction and conflict rule.
|
||||
- Surface the direction in the user-facing UI when possible: show "read-only, synced from Zendesk" on mirrored fields so users don't waste effort editing them.
|
||||
|
||||
**Corollaries:**
|
||||
- **Only auto-create low-stakes artifacts.** Creating a tracking stub in a downstream system is fine. Creating a ticket or case that obligates a human response is not — leave high-stakes creation to humans in the SoR.
|
||||
- **Never mirror fields that could leak cross-tenant or cross-context data.** Internal notes, private comments, and any field that assumes a specific audience must not be synced to a system with different access controls. Review every field for leakage before adding it to the sync set.
|
||||
- **Separate read-fields and write-fields in config.** The set of fields the sync writes to system B should be a strict subset of the fields it reads from system A, and that mapping should be explicit code/config, not an implicit "copy everything."
|
||||
|
||||
**Anti-patterns:**
|
||||
- Bidirectional sync with no declared SoR — "both systems edit freely, we'll resolve conflicts later."
|
||||
- Last-write-wins across two independent writers — destroys work non-deterministically.
|
||||
- Auto-creating tickets, cases, or escalations in a downstream system without a human in the loop.
|
||||
- Mirroring entire records rather than an explicit allowlist of fields.
|
||||
- Documenting the SoR in a diagram only, without enforcing it in code.
|
||||
|
||||
---
|
||||
|
||||
## 5. Poll Cadence, Cursor Design, and Idempotency
|
||||
|
||||
**Principle:** Polling-based syncs need a deliberate cadence, a persistent cursor, and idempotent writes so missed runs and retries do not corrupt state.
|
||||
|
||||
**Why it matters:** Ephemeral runners fail. Network requests time out. Push conflicts happen when two runs overlap. The only way these are survivable is if every operation is safe to repeat.
|
||||
|
||||
**How to implement:**
|
||||
- **Cadence:** match the business tolerance for staleness, not the API's rate limit ceiling. A 15-minute poll is plenty for most customer-data syncs; sub-minute polls rarely justify their cost. Consider the rate limit as a constraint, not a target.
|
||||
- **Cursor persistence:** store the cursor in the same state store as the rest of the sync (section 3). On run start, read the cursor; on run end, advance it only if all writes succeeded. A half-failed run must NOT advance the cursor, or records will be silently skipped.
|
||||
- **Cursor type:** prefer a high-water-mark timestamp or an opaque server-provided cursor over page offsets. Offsets break under concurrent writes on the source.
|
||||
- **Idempotent writes:** every write to the downstream system must be safe to repeat. Use external IDs, idempotency keys, or upsert semantics. A retry after a transient failure must produce the same end state as a single successful write.
|
||||
- **Rate limit handling:** honour `Retry-After` headers. Implement exponential backoff with jitter. Do not retry in a tight loop — this turns a soft rate limit into a hard ban.
|
||||
- **Observability:** log per-run counts (read / created / updated / deleted / skipped / errored). Alert on error counts above a threshold or on consecutive empty runs (which may indicate the cursor has wedged).
|
||||
|
||||
**Anti-patterns:**
|
||||
- Advancing the cursor before writes complete — skips records on partial failure.
|
||||
- Polling faster than the business needs — wastes rate limit budget for no benefit.
|
||||
- Non-idempotent writes paired with at-least-once delivery — creates duplicates on retry.
|
||||
- Ignoring `Retry-After` — gets you rate-banned.
|
||||
- No per-run observability — silent failures accumulate until someone notices data is missing.
|
||||
|
||||
---
|
||||
|
||||
## Summary
|
||||
|
||||
Integrating with a third-party API is an exercise in compensating for its limitations. Verify capabilities before designing, build application-layer compensation for missing features, use git + SOPS as a state store when you have no persistent compute, and always declare a system of record for bidirectional syncs. Idempotency and observability are non-negotiable.
|
||||
@@ -152,10 +152,20 @@ DinD sidecars use ephemeral storage. Docker's local layer cache is lost when the
|
||||
|
||||
CI workflows with path filters (e.g., `paths: ["src/**", "Dockerfile"]`) won't trigger when only the workflow file itself changes. This means cache configuration changes require a matching source change to trigger a build. Push a trivial change to a matched path to test.
|
||||
|
||||
### CI Path Filters Must Include All COPY'd Directories
|
||||
### CI Path Filters Must Include All COPY'd Directories and Runtime-Mounted Paths
|
||||
|
||||
When a Dockerfile COPYs from a directory (e.g., `harnesses/`, `models/`), that directory must be in the CI workflow's `paths:` trigger filter. Otherwise, changes to those directories won't trigger image rebuilds, leaving deployed images stale. Always cross-check CI path triggers against Dockerfile COPY sources.
|
||||
|
||||
This also applies to directories that are **mounted at runtime** (not COPY'd) but whose contents affect the container's behaviour — e.g., a `config/` directory bind-mounted into a container via Compose or K8s volume mount. A change to mounted config doesn't change the image, but it may require a rolling restart or cache invalidation step that the CI workflow should trigger. Include these paths in the workflow trigger and add a separate step (e.g., `kubectl rollout restart`) rather than assuming a build is required.
|
||||
|
||||
### CI Image Tagging Strategy: Short SHA + Full SHA + Latest
|
||||
|
||||
Tag container images with three tags: `sha-<7char>` (human-readable in kubectl output), `<full-sha>` (exact traceability), and `latest` (local dev convenience). The `sha-` prefix distinguishes commit tags from version tags. Pin deploy manifests to commit SHAs via Kustomize `images:` blocks — `git blame` on the kustomization shows exactly when each version was deployed.
|
||||
|
||||
### Multi-Registry / Tier-Separated Image Publishing
|
||||
|
||||
When separating image tiers by registry (e.g., `org-nonprod/app` on push-to-main, `org-prod/app` on `v*` tag):
|
||||
|
||||
- Give CI a **service account that is a member of both registry orgs**, and store one credential per registry (do not share a single token across tiers).
|
||||
- Keep tag formats **distinct per tier** (e.g., `sha-<8>` for non-prod, `prod-sha-<8>` for prod) so downstream systems — Octopus channels, Kustomize overlays, audit tooling — can reason about provenance from the tag alone.
|
||||
- Gate the prod publish workflow on `v*` tags or an explicit release event, never on main-branch pushes.
|
||||
|
||||
@@ -98,3 +98,46 @@ Before deploying any new service, check:
|
||||
3. Switch to PostgreSQL **before the first deployment**, not after problems appear
|
||||
4. Add the database password to SOPS-encrypted secrets
|
||||
5. Verify the database connection works before adding consumers
|
||||
|
||||
## `ON CONFLICT DO NOTHING` Requires a Real Unique Constraint
|
||||
|
||||
SQLAlchemy `on_conflict_do_nothing()` and raw `ON CONFLICT DO NOTHING` only work when there is a matching unique constraint or unique index. Without one, the statement either silently does nothing or inserts a duplicate, depending on the exact phrasing.
|
||||
|
||||
**Trap:** "URL" looks like a natural unique key for a crawl/ingest table, so the LLM or developer adds `UNIQUE(url)` to make the upsert work. But URL content changes over time — two rows for the same URL at different timestamps are semantically distinct. The fake unique constraint then corrupts the model (or blocks legitimate re-ingestion).
|
||||
|
||||
**Rule:**
|
||||
|
||||
- If the natural key is truly unique (user ID, slug, message hash), add the unique constraint and use `on_conflict_do_nothing()`
|
||||
- If the natural key is *not* unique over time (URL, title, filename), use **query-before-insert** in a transaction, not a fake unique constraint
|
||||
- Don't invent unique constraints to make `ON CONFLICT` work — you're encoding a false invariant into the schema
|
||||
|
||||
## Alembic Multi-Schema Migrations
|
||||
|
||||
To run Alembic against multiple Postgres schemas in one database:
|
||||
|
||||
1. Set `include_schemas=True` in `env.py` so autogenerate sees non-default schemas
|
||||
2. Add an `include_name` filter so autogenerate only tracks the schemas you manage (otherwise it tries to "fix" `information_schema`, `pg_catalog`, etc.)
|
||||
3. **Issue `CREATE SCHEMA IF NOT EXISTS <name>`** before `run_migrations()` — otherwise the first migration fails on a missing schema
|
||||
|
||||
```python
|
||||
def include_name(name, type_, parent_names):
|
||||
if type_ == "schema":
|
||||
return name in {"app", "audit", "reporting"}
|
||||
return True
|
||||
|
||||
def run_migrations_online():
|
||||
connectable = engine_from_config(...)
|
||||
with connectable.connect() as connection:
|
||||
for schema in ("app", "audit", "reporting"):
|
||||
connection.execute(text(f'CREATE SCHEMA IF NOT EXISTS "{schema}"'))
|
||||
context.configure(
|
||||
connection=connection,
|
||||
include_schemas=True,
|
||||
include_name=include_name,
|
||||
...
|
||||
)
|
||||
with context.begin_transaction():
|
||||
context.run_migrations()
|
||||
```
|
||||
|
||||
Also set `version_table_schema` on `context.configure` if the Alembic version table should live in a specific schema rather than `public`.
|
||||
|
||||
31
debugging.md
31
debugging.md
@@ -22,6 +22,15 @@ After wiring up any new service:
|
||||
|
||||
Use `curl --resolve` to test specific paths without depending on DNS propagation.
|
||||
|
||||
## Split-Horizon DNS Can Hide Bugs from Local Testing
|
||||
|
||||
When `/etc/hosts` or internal DNS points a public hostname at an internal IP, local `curl` bypasses the external path (VPS, CDN, cloud LB) and masks bugs that are only visible to external users. Always verify production behaviour through the actual public path:
|
||||
|
||||
- `curl --resolve domain:443:<public-ip> https://domain/...` to force the real external IP
|
||||
- Or test from an external machine (phone on cellular, a cloud VM, etc.)
|
||||
|
||||
Applies to reverse-proxy routing bugs, HTTP/2 SAN mismatches, and TLS configuration that differs between internal and external ingress.
|
||||
|
||||
## When Something Doesn't Sync/Apply
|
||||
|
||||
- Check resource exclusions in the GitOps controller immediately
|
||||
@@ -79,6 +88,20 @@ Before building a new service, component, or script, read existing patterns in t
|
||||
|
||||
Known issues documented in CLAUDE.md or MEMORY.md but not applied to new scripts/configs waste debugging time. Search your own documentation before writing automation that touches areas with known gotchas.
|
||||
|
||||
## Read the Spec Before Proposing a Workaround
|
||||
|
||||
When a mid-implementation design question arises in a subsystem that already has a written spec, **read the spec before proposing a bridge hack**. The correct design is often already documented. One session burned hours considering Phase 1 bearer-token bypasses before realising the spec already defined the Phase 2 design (bootstrap tokens + mTLS).
|
||||
|
||||
Rule: specs exist to prevent this — grep `spec/` or re-read the relevant spec file before inventing a workaround.
|
||||
|
||||
## Budget Infrastructure-Recovery Time After Disruptions
|
||||
|
||||
After any disruption (power cut, network outage, cluster reboot, registry migration), start the next session with an **infrastructure health check before planning feature work**. Snap confinement quirks, stuck `Terminating` pods, read-only filesystems, and unreachable Git remotes each consume meaningful time to diagnose. Budget recovery as an explicit first phase rather than discovering it mid-task.
|
||||
|
||||
## Parallel Research Agents for Broad Topic Coverage
|
||||
|
||||
When researching a topic with multiple independent facets, dispatch **parallel research agents** (e.g., one per sub-topic or source type) rather than sequential queries. Scope each agent narrowly (e.g., jurisdiction, domain filter, doc set) to reduce noise and improve signal. Cheap when facets are independent; poor fit when later queries depend on earlier results.
|
||||
|
||||
## API Token Scope Errors
|
||||
|
||||
When an API endpoint returns a permission/scope error, read the error response body before guessing. Many APIs (Gitea, GitHub, GitLab) explicitly state the required scope in the error message (e.g., `required=[write:admin]`). This is faster and more reliable than consulting documentation or iterating one scope at a time.
|
||||
@@ -97,3 +120,11 @@ When a service appears down, check its logs for successful requests from other c
|
||||
## Structured JSON Logging from Application Entry Points
|
||||
|
||||
Web frameworks like uvicorn don't configure application-level loggers — only access logs appear by default. Named loggers have no handler unless `logging.basicConfig()` is called explicitly. This makes application logs invisible in production (K8s, Docker) with no error — just silence. Always call `logging.basicConfig()` with a structured format (JSON) early in application startup, before any `getLogger()` calls.
|
||||
|
||||
## Verify DB Schema Matches Application Models After Every Deployment
|
||||
|
||||
After deploying a new version of an application that uses an ORM or schema migration tool, verify that the live database schema matches what the application expects. Common failure mode: a migration ran in dev/staging but not in production, or a new field was added to a model without a corresponding migration.
|
||||
|
||||
**Quick check:** run the application's schema validation command, or compare `alembic current` vs `alembic head`, or run `SELECT column_name FROM information_schema.columns WHERE table_name='<table>'` and diff against the model definition.
|
||||
|
||||
**When to check:** after every deployment that touches models or migrations — not just on explicit migration commits. An ORM auto-create (e.g., SQLAlchemy `create_all`) can silently succeed while leaving optional columns missing, causing subtle bugs rather than hard crashes.
|
||||
|
||||
69
docker.md
69
docker.md
@@ -73,3 +73,72 @@ Docker Compose V2 merges list fields (ports, volumes, environment) by appending,
|
||||
## Volume Source Paths Must Be Absolute
|
||||
|
||||
Docker interprets relative paths in volume mount source fields as named volumes, not bind mounts. Use `os.path.abspath()` or equivalent when constructing volume source paths programmatically. The error message ("includes invalid characters for a local volume name") is misleading — the real issue is that the path is relative.
|
||||
|
||||
## Init Scripts Needing Root Must Run Before gosu/exec Privilege Drop
|
||||
|
||||
In Docker entrypoints that drop privileges via `gosu <user> "$@"` or `exec gosu <user> command`, any setup that requires root (creating directories, setting ownership, writing to system paths) must happen before the `gosu` call. Once `exec gosu` runs, the process is replaced with the unprivileged user — subsequent commands in the same shell context run as that user. Structure entrypoints as: (1) root-level setup, (2) `exec gosu <user> "$@"`.
|
||||
|
||||
## Payload Size Limits at Multiple Layers
|
||||
|
||||
Large payloads embedded in environment variables or command-line arguments hit hard limits at multiple layers:
|
||||
- **OS `ARG_MAX`** (~128KB on Linux): the kernel limit on aggregate environment size. Exceeded values cause `Argument list too long` errors.
|
||||
- **K8s env var limit** (~228KB base64 per variable): containers crash with exit 255 and zero logs.
|
||||
- **CLI argv**: even when payload delivery fits via env var or file mount, many CLIs (`claude --print <prompt>`, shell wrappers) pass the prompt on argv, which hits `ARG_MAX` independently. When a prompt exceeds ~64KB, write it to a temp file and pipe via stdin instead of passing it positionally.
|
||||
|
||||
Use mounted files (ConfigMaps, Secrets, host bind mounts) for any payload that approaches these limits. For inter-task artifact passing, use git branches or mounted volumes — not env var payloads. See also the [Kubernetes Patterns](kubernetes.md) entry on env var size limits.
|
||||
|
||||
**Meta-rule:** after fixing an exec-arg limit at one layer (env, argv, file mount), immediately check adjacent layers for the same pattern before declaring it done. The same payload often flows through multiple chokepoints.
|
||||
|
||||
## `rm -f` on Bind-Mounted Files Fails Under `set -e`
|
||||
|
||||
Attempting `rm -f /path/to/bind-mounted-file` when the file is a bind mount (e.g., a host file mounted read-only into a container) fails with "Device or resource busy" even with the `-f` flag. Under `set -e` this exits the script immediately. Use `rm -f path 2>/dev/null || true` to suppress the error and continue, or check whether the path is a bind mount before attempting deletion.
|
||||
|
||||
## Compose `profiles:` Blocks On-Demand Lifecycle Managers
|
||||
|
||||
Services gated by `profiles:` in `docker-compose.yml` are not created until the profile is activated. On-demand lifecycle managers — Sablier, autoheal-style wake-up tools, CI runners that start/stop existing containers on request — cannot manage what does not exist. The container has no Docker record for them to act on.
|
||||
|
||||
**Fix:** drop `profiles:` for services managed by an external lifecycle tool, and use `docker compose create` (not `up`) to materialise the containers in a stopped state. The lifecycle manager can then start them on demand.
|
||||
|
||||
## GPU-Agnostic Base Compose with Provider Override Files
|
||||
|
||||
Keep the base `docker-compose.yml` free of hardware-specific runtime config. Put NVIDIA/AMD/Apple GPU or accelerator runtime settings in separate override files composed in with `-f`:
|
||||
|
||||
```
|
||||
docker-compose.yml # GPU-agnostic base
|
||||
docker-compose.nvidia.yml # NVIDIA runtime, device reservations
|
||||
docker-compose.apple.yml # Apple Silicon / Metal settings
|
||||
docker-compose.amd.yml # ROCm overrides
|
||||
```
|
||||
|
||||
Deploy with `docker compose -f docker-compose.yml -f docker-compose.nvidia.yml up -d`. The same stack template deploys unchanged across heterogeneous hosts; adding a new accelerator type is an override file, not a fork of the base compose.
|
||||
|
||||
## Snap-Packaged Docker Breaks After Unclean Shutdown
|
||||
|
||||
**Symptom:** containers fail with "read-only file system" on `/opt/` or other snap-confined paths after a power cut or forced reboot, even though the disk is healthy.
|
||||
|
||||
**Cause:** snap auto-refresh or AppArmor profile corruption during unclean shutdown leaves Docker's snap confinement in an inconsistent state.
|
||||
|
||||
**Fix:** `snap restart docker` is usually enough; if not, `snap revert docker`. On production hosts, prefer distro-packaged or upstream Docker (`docker-ce` from Docker's apt repo) to avoid snap confinement entirely.
|
||||
|
||||
## Set PYTHONPATH in Dockerfiles for `src/` Layout Projects
|
||||
|
||||
Python projects using a `src/` layout must set `ENV PYTHONPATH=/app/src` in every Dockerfile that does `COPY src/ ./src/`. Local development hides the problem because `pip install -e .` wires up imports via the editable install — the container has no such install, so imports fail only at runtime inside the container.
|
||||
|
||||
Set `PYTHONPATH` when scaffolding the Dockerfile, not after the first failed container run. Same rule applies to any multi-stage build where the final stage copies `src/` without re-running `pip install`.
|
||||
|
||||
## Write CLI Entrypoints Alongside the Module They Run
|
||||
|
||||
When a module will run inside a container, create its `_cli.py` entry point and verify the Dockerfile `ENTRYPOINT` in the same phase/commit as the module itself. Deferring CLI shims to "a later phase" repeatedly leads to containers that build cleanly but have no runnable entry point — the build passes, the deployment looks healthy, and the first invocation fails with `No module named ...` or an `ENTRYPOINT` that points at nothing.
|
||||
|
||||
Treat "module + CLI shim + ENTRYPOINT verification" as one atomic unit of work.
|
||||
|
||||
## PostgreSQL Alpine Image Runs as UID 70, Not 999
|
||||
|
||||
The `postgres:*-alpine` images run postgres as **UID 70**, not the 999 used by Debian-based `postgres:*` images. Setting data-directory ownership to 999 (or 1000) via host `chown` or Ansible `file` modules silently breaks access — `pg_isready` may still pass while internal operations fail with "Permission denied" on WAL writes, replication slots, or extension installs.
|
||||
|
||||
**Always verify before setting ownership:**
|
||||
```
|
||||
docker exec --user postgres <container> id
|
||||
```
|
||||
|
||||
Match the automation to the actual UID. If a project mixes Alpine and Debian Postgres images across environments, treat the UID as per-environment config, not a hardcoded constant.
|
||||
|
||||
@@ -63,3 +63,50 @@ Tells Claude (and independent agents) what the project is currently working on.
|
||||
- Keep it current — remove entries when work is complete
|
||||
- Orient agents — this is the primary mechanism for pointing independent agents at the right work
|
||||
- Complement, don't duplicate — CLAUDE.md has conventions, MEMORY.md has learnings, CONTEXT.md has the current focus
|
||||
|
||||
## Research-Backed Document Generation
|
||||
|
||||
When generating structured documents (contracts, specs, reports) from research, require every assertion or clause to link to a sourced finding file. Use a pipeline: `references/` → `research/[topic]/findings/` → `templates/`, with markdown links from clauses back to findings. A clause or claim without a traceable source is flagged as a gap. This turns document generation into an auditable process — reviewers can trace any statement back to its evidence, and missing support becomes visible rather than invisible.
|
||||
|
||||
## Structured Q&A to Fill a CONTEXT Template
|
||||
|
||||
For drafting tasks with many unknowns, start from a full section template with every field marked TBD, then resolve iteratively via Q&A with the user. Track a confidence level per section (e.g., High/Medium/Low/TBD) so remaining gaps stay visible. Prevents the common failure of premature drafting on incomplete context — the template surfaces what you don't know before the first sentence is written.
|
||||
|
||||
## Sourced-Claim Writing: Match Claims to Sources Exactly
|
||||
|
||||
When writing claims backed by sources, the claim must match the source exactly in scope, magnitude, population, units, qualifiers, and attribution. Common failure modes:
|
||||
- **Population drift** — "82% of organisations" vs source "82% of container users"
|
||||
- **Unit mismatch** — dollars vs percentage, per-year vs per-month
|
||||
- **Dropped qualifiers** — "not from X" vs source "not just from X"
|
||||
- **Year mislabelling** — citing a 2024 figure as 2025
|
||||
- **Attribution drift** — attributing a blog post to a famous report it merely references
|
||||
|
||||
Fix: read the source before writing the sentence, never drop qualifiers when paraphrasing, and verify the attribution label against the finding's Source section. Paraphrasing from memory is the main cause of drift.
|
||||
|
||||
## Three-Layer Verification for Sourced Documents
|
||||
|
||||
For evidence-backed documents, run a three-layer check before shipping:
|
||||
|
||||
1. **Story vs finding file** — every claim in the narrative matches the finding it cites
|
||||
2. **Finding vs reference file** — every finding accurately represents the underlying source
|
||||
3. **Structural checks** — every number has a link, every finding has a URL, no unsupported qualifiers
|
||||
|
||||
Track verified files by checksum in a `VERIFICATION.md`, enforce via a pre-commit hook.
|
||||
|
||||
**Hard rule:** never update a checksum without re-running verification. Structural edits don't change claims but do change the checksum — bumping it without re-verifying silently defeats the control.
|
||||
|
||||
Second verification passes routinely catch 10-20+ new issues. Treat verification as iterative, not one-shot.
|
||||
|
||||
## Research Workflow: REFERENCES_CONSIDERED Per Topic
|
||||
|
||||
When running multi-topic research against a shared reference pool, create a `REFERENCES_CONSIDERED.md` per topic listing every reference and whether it was reviewed through that lens (YES / NO / N-A with a short reason). New topics introduced mid-project miss all previously-processed references unless the gap is made explicit. This file turns "have we checked everything?" into a zero-ambiguity work queue.
|
||||
|
||||
Create the file immediately when a topic is created, not retroactively — retroactive creation loses the signal of which references were genuinely reviewed vs. assumed-reviewed.
|
||||
|
||||
## Executive Document Structure: Conclusion-First, Glossary Pattern
|
||||
|
||||
For documents aimed at senior or executive audiences:
|
||||
|
||||
- **Conclusion-first** — put "Bottom Line" and "What We Should Do" ABOVE all evidence. Executive readers are top-down and may not reach the end; evidence becomes supporting detail rather than the main narrative.
|
||||
- **Glossary pattern** — replace inline concept explanations with a glossary. Each entry has a brief definition, a "Why leadership should care" hook, and a link to the full detail doc. Keeps the main flow readable while preserving depth for those who want it.
|
||||
- **Separate executive brief for long documents** — for compiled docs over ~60 pages, generate a separate 5-10 page brief containing just Bottom Line + Ask + key evidence summaries. The full document remains for reviewers who need detail; the brief is what actually gets read.
|
||||
|
||||
@@ -5,6 +5,11 @@
|
||||
- Use meaningful commit messages; prefer small, focused commits over large batches
|
||||
- Never commit secrets in plaintext — use SOPS + age or equivalent encryption
|
||||
- Enable pre-commit hooks where appropriate (secret detection, linting, formatting)
|
||||
- **Commit each implementation phase separately.** For multi-phase milestones, commit at each phase boundary with tests passing. Each commit should be self-contained and independently describable. Phase-by-phase commits surface issues early, keep history bisectable, and make later review and reflection far easier than one large end-of-milestone commit.
|
||||
|
||||
## Repo Initialization
|
||||
|
||||
- **Rename default branch immediately after `git init`.** `git init` still creates `master` on many systems despite `main` being the modern default. Run `git branch -m master main` (or configure `init.defaultBranch = main` globally) before first push — otherwise `git push -u origin main` fails with `src refspec main does not match any`. Bake this into any project-bootstrap script.
|
||||
|
||||
## Pre-Commit Hooks
|
||||
|
||||
@@ -12,6 +17,8 @@
|
||||
- Auto-encrypt files matching `.sops.yaml` rules that aren't yet encrypted
|
||||
- Enable with `git config core.hooksPath .githooks`
|
||||
- Consider secret detection, linting, and formatting hooks
|
||||
- **Copy hook scripts into `.git/hooks/`; don't symlink.** A symlinked hook resolves to the working-tree file, which changes on every branch switch — hooks then run stale (or wrong-branch) logic. Use an `install-hooks.sh` that copies the source script into `.git/hooks/`, and re-run it whenever the source changes.
|
||||
- **Document the sync-then-commit sequence when hooks require an up-to-date main.** If a pre-commit hook requires local `main` to match `origin/main`, the commit silently blocks when local `main` is even one commit behind. Standard sequence: `git checkout main && git pull --ff-only && git checkout <branch> && git rebase main`.
|
||||
|
||||
## GitOps Workflow
|
||||
|
||||
@@ -19,18 +26,43 @@
|
||||
- No manual changes without corresponding GitOps manifests — anything applied manually (e.g., `kubectl apply`, `helm install`) should immediately get a corresponding tracked manifest
|
||||
- For ArgoCD-managed clusters: edit in Git, push, sync — never edit live resources directly
|
||||
|
||||
## Separate Data Repos from Code Repos for GitOps Controllers
|
||||
|
||||
When a GitOps controller watches a git repo for declarative state (zone files, policy documents, config blobs), keep that data in a **repo separate from the controller's app code**. Mixing the two means every data change triggers CI builds (container rebuilds, test runs) for no reason, and pollutes the code repo's history with non-code churn.
|
||||
|
||||
Separation yields:
|
||||
- Pure data commits with no CI noise
|
||||
- Cleaner webhook targeting per-repo
|
||||
- Independent access control for data editors vs code maintainers
|
||||
- Simpler rollback semantics — revert a data change without touching the controller image
|
||||
|
||||
## Remote Conventions
|
||||
|
||||
- SSH workflows preferred over HTTPS for Git remotes
|
||||
- Use SSH config host aliases for multi-user setups (e.g., `gitea.example.com-<user>`)
|
||||
- Remote URL format: `git@<host-alias>:<org>/<repo>.git`
|
||||
- Optionally push-mirror to GitHub for public visibility
|
||||
- **Prefer internal hostnames for self-hosted Git remotes on the local network.** SSH push to a self-hosted Git server through a public/VPS hostname can return `kex_exchange_identification: read: Connection reset by peer`, especially during heavy agent activity — VPS routes often have connection limits or rate limiting that fail silently under load. On the local network, use the internal hostname in SSH aliases for heavy push workflows; reserve the public hostname for external access or HTTPS. Test with `ssh -T <alias>` immediately after configuring.
|
||||
|
||||
## Cross-User Pushes
|
||||
|
||||
- **Prefer SSH aliases over temp-URL swaps.** When a repo is owned by user A but your default SSH key authenticates as user B, configure a dedicated SSH host alias (`Host gitea.example.com-userA` with matching private key) and use `git@gitea.example.com-userA:org/repo.git`. Avoid temp-URL-swap (embedding an HTTPS token in the remote URL, pushing, then resetting) — it's clunky, prone to shell-quoting errors, and leaks tokens into reflog/history.
|
||||
- **API-created repos need the SSH user as a collaborator.** If a repo is created via API token (user A) but pushes use an SSH alias authenticating as user B, user B has no access by default. Add the SSH-authenticating user as admin collaborator via API before the first push. Applies to Gitea, GitHub, and any platform where API auth and SSH auth use different identities.
|
||||
|
||||
## Access and Clone Gotchas
|
||||
|
||||
- **Org repos require explicit collaborator grants.** Don't assume organizational membership implies write access — verify permissions before setting up automation or CI/CD.
|
||||
- **Shallow clones break push operations.** `git clone --depth 1` is fine for read-only CI jobs, but pipelines that push artifacts, tags, or mirror to other remotes need full clones.
|
||||
|
||||
## Cross-Remote Hygiene for Multi-Remote Projects
|
||||
|
||||
When a project has divergent remotes (e.g., local Gitea with granular commits + GitHub with squash-merged PRs), histories diverge and naive merges explode.
|
||||
|
||||
- **Never `git merge <remote>/<branch> --allow-unrelated-histories`** — it produces 50+ mass conflicts across unrelated files.
|
||||
- **Import specific files instead:** `git checkout <remote>/<branch> -- <specific-files>` brings those paths in as a normal local change.
|
||||
- **Always diff after a bulk checkout.** `git checkout` from a remote silently overwrites locally-modified files with older remote versions with zero warning. Run `git status` and review each touched file before staging; restore with `git checkout HEAD -- <file>` if an unwanted overwrite happened.
|
||||
- **After an upstream PR squash-merge, reset — don't rebase — local branches to main.** Rebase against a squashed history leaves phantom commits; a clean `git reset --hard origin/main` on the local branch is correct.
|
||||
|
||||
## Version Management
|
||||
|
||||
- Use the latest stable version of dependencies unless pinned for a reason
|
||||
@@ -55,7 +87,3 @@ When running parallel agents or tasks that modify the same repo:
|
||||
- Tasks with multiple dependencies get an octopus merge base branch
|
||||
- Worktrees share the `.git` object store — fast creation, minimal disk usage
|
||||
- Keep containers detached (`docker run -d`, not `--rm`) so logs survive for inspection after exit
|
||||
|
||||
## API-Created Repos Need SSH User as Collaborator
|
||||
|
||||
If a repo is created via API token (user A) but pushes use an SSH alias authenticating as user B, user B has no access by default. Add the SSH-authenticating user as admin collaborator via API before the first push. This applies to Gitea, GitHub, and any platform where API auth and SSH auth use different identities.
|
||||
|
||||
@@ -34,6 +34,7 @@ Manual bootstrap secrets (encryption keys, OIDC client secrets) must be document
|
||||
- **Liveness vs readiness probes serve different purposes.** TCP checks confirm the process is listening (liveness). Exec/command checks confirm the application is ready to serve (readiness). Don't conflate them.
|
||||
- **Probes must match application host validation.** Applications that validate Host headers (e.g., Next.js `ALLOWED_HOSTS`) will reject probes sent to the pod IP. Set `httpGet.httpHeaders` with the expected Host value.
|
||||
- **Don't load credentials into liveness probes.** If readiness requires an authenticated check (e.g., `sqlcmd`), use a simple TCP check for liveness and reserve the authenticated check for readiness only.
|
||||
- **`timeoutSeconds: 1` is too tight for services with DB connections or async startup.** The default probe timeout is 1 second, which causes spurious failures when a service is initialising a connection pool or running async startup tasks. Use 3–5 seconds as a minimum for any service that touches a database or has an async lifespan handler.
|
||||
|
||||
## Init Container Patterns
|
||||
|
||||
@@ -56,6 +57,23 @@ Manual bootstrap secrets (encryption keys, OIDC client secrets) must be document
|
||||
|
||||
- **Namespace PodSecurity labels must match container security contexts.** DinD, CSI drivers, and other privileged workloads need `pod-security.kubernetes.io/enforce: privileged` on their namespace. A `baseline` or `restricted` namespace silently blocks privileged pods.
|
||||
- **Document privileged namespace requirements.** When a workload needs elevated privileges, document the specific requirement (e.g., "Docker-in-Docker for CI builds") alongside the namespace label.
|
||||
- **Monitoring namespace requires privileged PodSecurity for node-exporter.** kube-prometheus-stack's node-exporter DaemonSet mounts host paths and uses `hostPID: true`. The monitoring namespace must be labelled `pod-security.kubernetes.io/enforce: privileged` or node-exporter pods will be silently blocked. Set this via GitOps namespace metadata — don't apply it manually or it will be reverted by the GitOps controller.
|
||||
|
||||
## Cilium Entity Identities for Monitoring Scraping
|
||||
|
||||
When writing Cilium network policies to allow Prometheus to scrape targets, the correct entity identity depends on the node role:
|
||||
|
||||
| Target | Cilium entity |
|
||||
|---|---|
|
||||
| kube-apiserver (port 6443) | `kube-apiserver` |
|
||||
| Worker node kubelet / node-exporter | `remote-node` |
|
||||
| Same-node kubelet (DaemonSet on same node) | `host` |
|
||||
|
||||
Using the wrong entity results in silent policy drops. Test with `cilium monitor --type drop` to identify mismatches.
|
||||
|
||||
## Kustomize Overlay `images:` Blocks Silently Override Base Tags
|
||||
|
||||
Kustomize `images:` blocks in an overlay apply to the entire rendered manifest, including any images defined in `base/`. If the base defines `image: my-app:v1.0.0` and the overlay has an `images:` block targeting `my-app`, the overlay's `newTag` silently wins — even if you intended the base tag to remain. When deploying a new image version via Kustomize, always update the `images:` block in the overlay, not just the base manifest. If the overlay doesn't have an `images:` block, add one rather than editing the base tag directly.
|
||||
|
||||
## ArgoCD Source Type Detection
|
||||
|
||||
@@ -64,7 +82,7 @@ Manual bootstrap secrets (encryption keys, OIDC client secrets) must be document
|
||||
|
||||
## Miscellaneous
|
||||
|
||||
- `enableServiceLinks: false` may be needed when K8s-injected service env vars conflict with app config (e.g., Authelia interprets `AUTHELIA_*` service vars as configuration).
|
||||
- `enableServiceLinks: false` may be needed when K8s-injected service env vars conflict with app config (e.g., Authelia interprets `AUTHELIA_*` service vars as configuration). A related symptom: pytest test collection fails with errors like `PORT=tcp://10.96.0.1:443` — K8s injects `<SERVICE>_PORT` as a full TCP URI, which many frameworks try to parse as an integer and crash. Setting `enableServiceLinks: false` on the pod removes all injected service env vars and resolves this class of error.
|
||||
- Proxmox VM names must match K8s node hostnames for cloud controller manager integration.
|
||||
- Metrics-server on Talos needs `--kubelet-insecure-tls` (self-signed kubelet certs).
|
||||
|
||||
@@ -103,3 +121,55 @@ Kubernetes has a hard limit on environment variable sizes (~228KB base64). Large
|
||||
## Non-Blocking Registration in FastAPI Lifespan Handlers
|
||||
|
||||
Blocking operations (external API calls, service registration) in application lifespan handlers prevent the HTTP server from starting. K8s liveness probes fail and the pod enters CrashLoopBackOff before the operation completes. Use background tasks (e.g., `asyncio.create_task`) for registration so health endpoints respond immediately while registration happens asynchronously. This applies to any K8s-deployed app framework with startup hooks (FastAPI, Flask, etc.).
|
||||
|
||||
## Verify Live Cluster State vs Deploy Repo Before Planning Changes
|
||||
|
||||
GitOps controllers (ArgoCD, Flux) preserve fields added by manual `kubectl patch`/`apply` when those fields aren't in the deploy repo — unknown fields are not removed unless the controller sees a conflicting managed field. Symptom: live ConfigMap/IngressRoute has values not in Git, causing "it works differently than the manifests say" debugging. Before changing GitOps-managed resources, diff `kubectl get -o yaml` against the deploy repo and add explicit values to Git so subsequent syncs reset any manual drift.
|
||||
|
||||
## Force-Delete Pods Stuck Terminating After Node Disruption
|
||||
|
||||
After power cut, kernel panic, or abrupt node loss, pods can be stuck in Terminating indefinitely (observed 22h+). The owning controller (Deployment, StatefulSet, ArgoCD application-controller) is blocked from creating a replacement, so upstream symptoms look like "GitOps stuck on old commit" or "service unreachable". Fix: `kubectl delete pod <name> --grace-period=0 --force`. The controller recreates immediately and reconciliation resumes.
|
||||
|
||||
## CNI L2 LoadBalancer Announcements: Use externalTrafficPolicy Cluster
|
||||
|
||||
With L2-announced LoadBalancer IPs (Cilium, MetalLB), `externalTrafficPolicy: Local` silently drops packets whenever the node winning the ARP lease doesn't run a backend pod — only that node holds a BPF/iptables LB entry. Use `externalTrafficPolicy: Cluster` and recover source IP at L7 (X-Forwarded-For, Proxy Protocol) instead.
|
||||
|
||||
## Restart CNI Agents After Agent-Affecting Config Changes
|
||||
|
||||
CNI Helm values that land in the agent ConfigMap (L2 announcements, Hubble, envoy features) do not take effect until agent pods restart — the agent logs a "config drift" warning but keeps running the old config. After changing agent-affecting values, `kubectl rollout restart daemonset/<cni-agent>` then `kubectl rollout restart deployment/<cni-operator>`.
|
||||
|
||||
## Upgrade Storage-Consuming Nodes Sequentially, Not Concurrently
|
||||
|
||||
Rolling upgrades that reboot multiple nodes concurrently can race external CSI controllers (Proxmox, vSphere, any hypervisor plugin doing hotplug). Parallel `ControllerPublishVolume`/`Unpublish` calls leave VolumeAttachments attached to the wrong VM or in a state where `attached: true` but the device is absent. Wait for each node Ready and CSI pods stable before upgrading the next; verify with `kubectl get volumeattachment`.
|
||||
|
||||
## Split Multi-Host IngressRoutes With Separate TLS Secrets
|
||||
|
||||
A Traefik IngressRoute using `Host(a.example) || Host(b.example)` can only reference one `tls.secretName`; the second domain silently falls back to Traefik's self-signed default cert. Split into one IngressRoute per Host match with its own `tls.secretName`. Generalises to any ingress controller pairing a single TLS secret per ingress object.
|
||||
|
||||
## Delete-and-Recreate, Don't Patch, When Adopting Manually-Applied Resources into GitOps
|
||||
|
||||
When a resource was first created with `kubectl apply` and then placed under GitOps ServerSideApply management, stale field-manager metadata causes perpetual OutOfSync that patching cannot resolve. Fix: `kubectl delete` the resource and let the GitOps controller recreate it with clean field ownership. Applies to any SSA-managed CRD adoption.
|
||||
|
||||
## etcd extraArgs Changes Require a Node Reboot on Immutable-OS Distros
|
||||
|
||||
On immutable-OS distros (Talos, Bottlerocket, Flatcar), etcd runs as a system service. Patching machine config with `cluster.etcd.extraArgs` reports "Applied without reboot" but etcd keeps old args until the process restarts — which only happens on a full node reboot. After etcd flag changes, roll the control plane (non-leader first, leader last) and verify with `etcd status`/logs.
|
||||
|
||||
## etcd Defrag Reclaims Space from Deleted Keys
|
||||
|
||||
etcd does not auto-reclaim space from deleted keys; the DB grows over time and hurts fsync latency on slow disks. Periodic `etcdctl defrag` (or Talos `etcd defrag`) reclaims 30-50% on typical clusters. Run on non-leader members first, leader last. Especially important on HDD or contended SSD.
|
||||
|
||||
## Hard-Reset Immutable-OS Nodes Stuck in Kernel-Level Boot
|
||||
|
||||
Immutable-OS API reboots (e.g., `talosctl reboot`) require the OS API running in userspace. When a node is stuck pre-userspace — XFS quotacheck after unclean shutdown, fsck, long kernel init — the API is unreachable. Use hypervisor-level hard reset (`qm reset`, `virsh reset`, cloud provider stop/start) to force a clean boot; kernel-level recovery usually completes in seconds.
|
||||
|
||||
## Reconciliation Controller Pattern: Pure Diff, I/O Reconciler
|
||||
|
||||
When writing a GitOps reconciliation controller, split into a pure function (desired vs actual → DiffResult, no I/O) and a separate reconciler class that handles all side effects (API calls, safety limits, logging). The pure diff is testable with zero mocks; the reconciler is mocked at its I/O boundary. Clean architecture for any controller reconciling declarative state against an external API.
|
||||
|
||||
## Controller Safety: Manage Only Declared Resources by Default
|
||||
|
||||
A reconciliation controller should touch only resources explicitly declared in its source-of-truth config; unmanaged resources should be logged but never deleted. Avoid "bulk replace" APIs that atomically overwrite everything — prefer per-record create/update/delete so incomplete declarations can't wipe records (NS, SOA, operator-managed). Design opt-in flags (`managed: all`, `conflict: alert|automatic`) from day one.
|
||||
|
||||
## Webhook-Triggered Reconciliation with Token Auth
|
||||
|
||||
Pair periodic reconciliation with an authenticated POST `/reconcile` endpoint so push events can trigger immediate sync. Use a 32+ char Bearer token with constant-time comparison, fail-closed (return 501) if the token is not configured. Avoids worst-case polling latency when a human just committed.
|
||||
|
||||
@@ -777,6 +777,116 @@ Use this checklist when reviewing LLM-generated code:
|
||||
|
||||
---
|
||||
|
||||
## 11. Gated Model Downloads Require Out-of-Band License Acceptance
|
||||
|
||||
### What goes wrong
|
||||
|
||||
HuggingFace (and similar model hubs) return **403 Forbidden** for "gated" models even when the HTTP request carries a valid user token. The hub enforces that the token's user has manually accepted the license agreement on the web UI for that *specific* model. Automation cannot bypass this — there is no API to accept the license.
|
||||
|
||||
This breaks reproducible-build scripts, container bake pipelines, and agent workflows that pull third-party ML models: the first run on a fresh account/token fails with an opaque 403 and no hint that human action is required.
|
||||
|
||||
### Pattern
|
||||
|
||||
Bake a pre-flight check into every model-download script:
|
||||
|
||||
```python
|
||||
def preflight_gated_model(model_id: str, token: str) -> None:
|
||||
r = requests.head(f"https://huggingface.co/{model_id}/resolve/main/config.json",
|
||||
headers={"Authorization": f"Bearer {token}"},
|
||||
allow_redirects=True)
|
||||
if r.status_code == 403:
|
||||
url = f"https://huggingface.co/{model_id}"
|
||||
raise SystemExit(
|
||||
f"Model {model_id} is gated. Open {url} in a browser, sign in as "
|
||||
f"the token owner, accept the license, then rerun this script."
|
||||
)
|
||||
r.raise_for_status()
|
||||
```
|
||||
|
||||
### Rules
|
||||
|
||||
- **Surface a human-readable error on 403** — do not retry, do not fall back to a different model silently
|
||||
- **Include the exact URL** to visit and the exact action required ("accept the license")
|
||||
- **Check every gated model** at pipeline start, not lazily at download time, so the human step is front-loaded
|
||||
- **Document which models are gated** in the project's README — license acceptance is per-user, so every new operator needs to do it once
|
||||
|
||||
Applies to any project pulling third-party ML models from HuggingFace, Meta's Llama portal, Stability AI's hub, or similar.
|
||||
|
||||
---
|
||||
|
||||
## 12. Operational Vulnerabilities in AI-Generated Code (Beyond Traditional SAST)
|
||||
|
||||
Traditional SAST tools (Bandit, Semgrep, SonarQube, Snyk) focus on known vulnerability patterns — injection, XSS, hardcoded secrets. But analysis of AI-generated code at scale (538,860 findings across 3,518 scans, SentinaLayer 2026) reveals that the **top vulnerability categories are structural and operational**, not traditional:
|
||||
|
||||
| Category | % of P0-P2 Findings | What SAST Misses |
|
||||
|---|---|---|
|
||||
| CI/CD Integrity Gaps | 31% | Gate ordering, missing dependency chains, workflow step sequencing |
|
||||
| Backend Reliability | 27% | Missing idempotency keys, premature health checks, retry logic gaps |
|
||||
| Security Overlay | 24% | Supply chain trust gaps, unverified checksums, unsigned artifacts |
|
||||
| Supply Chain Provenance | 11% | Missing signature verification, unpinned base images in multi-stage builds |
|
||||
| Data Layer Integrity | 7% | Unsafe write paths, missing concurrent-access guards, race conditions |
|
||||
|
||||
**The single most common P0-P2 finding: missing idempotency keys in webhook handlers (18% of all critical/high/medium findings).** This is not a vulnerability that any SAST tool checks for.
|
||||
|
||||
### Why AI agents produce these
|
||||
|
||||
AI coding agents optimise for **functional correctness** — does the code produce the right output for the happy path? They consistently miss:
|
||||
|
||||
1. **Retry safety.** Webhook handlers, API endpoints, and event processors that work correctly on first invocation but corrupt data on retry. Agents don't model what happens when the same request arrives twice.
|
||||
|
||||
2. **Ordering dependencies.** CI/CD pipelines where steps depend on prior steps' artifacts but the dependency isn't explicit. Works when steps happen to run in order, breaks under parallelism or partial failure.
|
||||
|
||||
3. **Health check timing.** Services that report healthy before their dependencies are ready. The agent sees "return 200 from /health" and implements it, without considering that the database connection pool hasn't warmed yet.
|
||||
|
||||
4. **Checksum and signature verification.** Agents download artifacts, pull images, and install packages without verifying integrity. The code works, but the supply chain is unverified.
|
||||
|
||||
5. **Concurrent write safety.** File operations, database writes, and cache updates that work under single-threaded testing but corrupt under concurrent access. Agents don't think about lock ordering or write-after-read races.
|
||||
|
||||
### Deterministic checks you can add to CI
|
||||
|
||||
These can be implemented as fast pre-scan rules (regex + AST) that run before expensive test suites:
|
||||
|
||||
**Idempotency:**
|
||||
- Webhook handlers without idempotency key extraction/dedup
|
||||
- POST endpoints that create resources without checking for existing duplicates
|
||||
- Event processors without at-least-once safety (no dedup by event ID)
|
||||
|
||||
**CI/CD Integrity:**
|
||||
- GitHub/Gitea Actions steps that reference artifacts from prior steps without explicit `needs:`
|
||||
- Dockerfile `COPY --from=` referencing stages without explicit ordering
|
||||
- Helm hooks without `hook-weight` when ordering matters
|
||||
|
||||
**Supply Chain:**
|
||||
- `curl | bash` or `wget -O- | sh` without checksum verification
|
||||
- Container images pulled by tag without digest pinning
|
||||
- Package installation without lockfile or hash verification
|
||||
- `go install` / `pip install` from URLs without integrity checks
|
||||
|
||||
**Health Checks:**
|
||||
- HTTP health endpoints that return 200 unconditionally (no dependency readiness check)
|
||||
- Liveness probes identical to readiness probes (should be different — liveness checks "am I stuck?", readiness checks "am I ready to serve?")
|
||||
- `initialDelaySeconds: 0` on readiness probes for services with startup dependencies
|
||||
|
||||
**Concurrent Access:**
|
||||
- File writes without advisory locking (`fcntl.flock` / `flock`)
|
||||
- Database read-then-write sequences without transactions or SELECT FOR UPDATE
|
||||
- Cache operations without atomic compare-and-swap
|
||||
|
||||
### Review checklist (operational)
|
||||
|
||||
- [ ] Every webhook handler extracts and deduplicates by an idempotency key
|
||||
- [ ] Every POST endpoint that creates resources checks for pre-existing duplicates
|
||||
- [ ] CI/CD steps declare explicit dependencies on prior steps' outputs
|
||||
- [ ] Downloaded artifacts are verified by checksum or signature before use
|
||||
- [ ] Container base images are pinned by digest, not just tag
|
||||
- [ ] Health check endpoints verify dependency readiness, not just "process is alive"
|
||||
- [ ] Readiness and liveness probes serve different purposes
|
||||
- [ ] File and database writes under concurrent access use appropriate locking
|
||||
- [ ] Event processors handle at-least-once delivery (idempotent or deduplicating)
|
||||
- [ ] Multi-stage Docker builds have explicit stage ordering and artifact dependencies
|
||||
|
||||
---
|
||||
|
||||
## Sources
|
||||
|
||||
- [Security Weaknesses of Copilot-Generated Code in GitHub Projects (ACM TOSEM)](https://dl.acm.org/doi/10.1145/3716848)
|
||||
|
||||
@@ -41,6 +41,20 @@ Good reflections capture:
|
||||
- Most avoidable waste and what would have prevented it
|
||||
- Concrete checklist items for future similar work
|
||||
|
||||
## Always Push Partial Work, Even on Test Failure
|
||||
|
||||
When dispatching agent tasks that commit work on exit, push the branch even when tests fail. Example failure mode: a task with 88/88 of its own tests passing but an unrelated dependency failure in the full suite had its work discarded because `finalize` was gated on "tests passed".
|
||||
|
||||
Partial work is almost always more valuable than nothing — the exit code and metadata still signal failure, and downstream users can cherry-pick or inspect the branch.
|
||||
|
||||
**Rule:** finalize/commit/push actions should not be gated on success. Only higher-level decisions (branch labels, PR creation, auto-merge eligibility) should key off test outcomes.
|
||||
|
||||
## Write Milestone Verify Scripts Manually, Not as Agent Deliverables
|
||||
|
||||
When agents complete tasks in a decomposed milestone, each agent naturally writes a verify script that covers only its own slice. If one of those slice-scoped scripts is labelled the milestone verify script, "milestone verified" really means "one slice verified" — the cross-slice integration is unchecked.
|
||||
|
||||
**Rule:** always write the milestone-level verify script manually, or dispatch it as a separate task whose input is the full milestone scope. Never fold milestone verification into one of the feature-implementation tasks.
|
||||
|
||||
## Evaluate Content Placement Before Building
|
||||
|
||||
Before creating a new document, system, or catalog, discuss where it belongs conceptually. Different content types have different lifecycles:
|
||||
|
||||
@@ -32,6 +32,16 @@ Auto-renewing proxies (Caddy, Traefik with Let's Encrypt, etc.) that also suppor
|
||||
|
||||
**Rule:** Use automatic certificate management for all sites. Don't mix file-loaded and automatic certs unless you understand the matching priority.
|
||||
|
||||
## Reverse Proxies Ignore Labels on Stopped Containers
|
||||
|
||||
Docker-label-based routing (Traefik, Caddy-docker-proxy, nginx-proxy) silently drops routes whose target containers are not running. Flags like `allowEmptyServices` do not help — the router only sees the labels of *running* containers.
|
||||
|
||||
This breaks on-demand and "scale-to-zero" backends: the proxy has no route to the stopped container, so the wake-up request never reaches whatever is meant to start it. Requests 404 (or worse, go to the wrong backend) until the container happens to be up.
|
||||
|
||||
**Rule:** For any backend that may not always be running, declare the route in **dynamic file config**, not container labels. File-config routes exist regardless of container state — the router can then proxy to a "wake" handler, return a holding page, or queue the request.
|
||||
|
||||
**Validation:** always test routing with the backend container **stopped**, not just running. If the route disappears when the container stops, the config is wrong for on-demand use.
|
||||
|
||||
## Cilium DNAT Resolves LB VIP Before NetworkPolicy Evaluation
|
||||
|
||||
Cilium performs DNAT on LoadBalancer VIP traffic before evaluating NetworkPolicy. Traffic to a VIP is rewritten to a backend pod IP before the policy check. For egress to LoadBalancer services in CiliumNetworkPolicy, use `toEndpoints` targeting the backend pods (by namespace/label), not `toCIDR` targeting the VIP.
|
||||
|
||||
@@ -190,13 +190,59 @@ packages "MyPackage" {
|
||||
- Keep major bumps for breaking changes (parameter renames, step removals)
|
||||
- Minor/patch for new optional parameters, script improvements, bug fixes
|
||||
|
||||
## Dual-Image / Multi-Registry Channel Pattern
|
||||
|
||||
When a project builds distinct images to separate registries per environment tier (e.g., non-prod vs prod), model each registry as its own **Feed** and create **one channel-scoped deployment step per feed**. Two channel-scoped steps are cleaner than a single step with version rules, because differing `PackageId` values (`org-nonprod/app` vs `org-prod/app`) cannot both be satisfied by a single step's package reference.
|
||||
|
||||
```hcl
|
||||
step "deploy-nonprod" {
|
||||
action {
|
||||
channels = ["non-prod"]
|
||||
packages "app" {
|
||||
feed = "registry-nonprod"
|
||||
package_id = "org-nonprod/app"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
step "deploy-prod" {
|
||||
action {
|
||||
channels = ["prod"]
|
||||
packages "app" {
|
||||
feed = "registry-prod"
|
||||
package_id = "org-prod/app"
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Release Creation: CI-Driven vs Feed Triggers
|
||||
|
||||
Octopus feed triggers silently ignore non-SemVer Docker tags (e.g., `sha-abc12345`) and poll on a ~3-minute interval. For SHA-tagged workflows or any non-SemVer scheme:
|
||||
|
||||
- **Preferred:** have CI `POST /api/releases` directly after pushing the image. This is instant, avoids simultaneous-trigger race conditions on `UQ_ReleaseVersionUnique`, and works with any tag format.
|
||||
- **Alternative:** tag with SemVer plus a timestamp build number (`1.2.3+YYYYMMDDHHMM`) for guaranteed uniqueness without needing external state.
|
||||
|
||||
Feed triggers remain fine for pure-SemVer tag schemes.
|
||||
|
||||
## Channel Version Rules and Enforcement
|
||||
|
||||
Channel version rules require **either a version range or a pre-release tag** — they cannot be empty. For non-SemVer tagging schemes, remove all rules (`Rules: []`) and rely on step channel-scoping (`action.channels`) for enforcement instead.
|
||||
|
||||
The **default channel** on any project cannot be deleted; either leave it unused or promote another channel to default before removing it.
|
||||
|
||||
## Auto-Promoting Lifecycles for Hands-Free Demos / PoCs
|
||||
|
||||
To demonstrate a full CI-to-deployed flow, set lifecycle phases to auto-deploy by moving environments from `OptionalDeploymentTargets` to `AutomaticDeploymentTargets`. Pair a non-prod lifecycle (Dev → Staging) with a prod lifecycle (Staging → Production) so each channel's releases promote automatically once created.
|
||||
|
||||
## Gotchas
|
||||
|
||||
1. **Step templates are space-scoped.** Process templates in Platform Hub cannot reference `ActionTemplates-*` IDs from other spaces. If you need reusable steps, use inline `action_type = "Octopus.Script"` in the process template OCL. Step templates are useful within a single space's projects, but not for cross-space process templates.
|
||||
2. **Process template names** cannot contain parentheses, slashes, or ampersands — only letters, numbers, periods, commas, dashes, underscores, and hashes.
|
||||
2. **Process template names** cannot contain parentheses, slashes, or ampersands — only letters, numbers, periods (`.`), commas (`,`), dashes (`-`), underscores (`_`), and hashes (`#`).
|
||||
3. **Heredoc for multi-line scripts** — use `<<-EOT` / `EOT` for PowerShell scripts that contain double quotes. The `-` prefix allows indented closing tags.
|
||||
4. **Every step needs a worker pool.** Process templates must have a `worker_pool` parameter (type `WorkerPool`), and every action must set `worker_pool_variable = "worker_pool"` referencing it. Without this, the template will fail to parse with "A step must specify a worker pool parameter".
|
||||
5. **Publishing and sharing is UI-only.** Process template sharing (which spaces can see/use a template) is stored in the Octopus database, not in Git/OCL. You must publish and share each template through the UI. There is no API or CLI for this currently.
|
||||
6. **Space creation via API requires a manager.** `POST /api/spaces` rejects an empty `SpaceManagersTeamMembers` with "select either teams and/or users as managers". Always include at least one user ID.
|
||||
|
||||
## Best Practices
|
||||
|
||||
|
||||
287
python-patterns.md
Normal file
287
python-patterns.md
Normal file
@@ -0,0 +1,287 @@
|
||||
# Python Patterns & Gotchas
|
||||
|
||||
Patterns, anti-patterns, and gotchas encountered in real Python projects. Covers concurrency, Pydantic, testing, and cross-field validation.
|
||||
|
||||
## Non-Reentrant `threading.Lock` Causes Deadlocks
|
||||
|
||||
Python's `threading.Lock` is **non-reentrant**: if the same thread tries to acquire a lock it already holds, it blocks forever. This is a common source of deadlocks in services where a method holding a lock calls another method that also acquires the same lock.
|
||||
|
||||
**Symptom:** Service hangs indefinitely with no error, CPU at 0%, no log output after the hang point.
|
||||
|
||||
**Fix:** Use `threading.RLock` (reentrant lock) for locks that may be acquired by the same thread multiple times — e.g., a `_persist()` method called both directly and from within a method that already holds the lock.
|
||||
|
||||
```python
|
||||
# Bad — deadlocks if update() calls persist() while holding _lock
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def update(self, key, value):
|
||||
with self._lock:
|
||||
self._data[key] = value
|
||||
self._persist() # acquires _lock again → deadlock
|
||||
|
||||
# Good — RLock allows re-acquisition by the same thread
|
||||
self._lock = threading.RLock()
|
||||
```
|
||||
|
||||
Use `Lock` only when you're certain a lock will never be re-acquired by the same thread. When in doubt, use `RLock`.
|
||||
|
||||
## Pydantic v2 `extra='ignore'` Silently Drops Unknown Fields
|
||||
|
||||
Pydantic v2 models with `model_config = ConfigDict(extra='ignore')` (or the class-level `class Config: extra = 'ignore'`) silently discard any fields not declared in the model. This is often the desired behaviour for API consumers that receive payloads with forward-compatible fields — but it becomes a bug when a required field is misspelled or renamed.
|
||||
|
||||
**Symptom:** An expected field is `None` or missing despite being present in the input dict. No validation error is raised.
|
||||
|
||||
**Debug pattern:** Temporarily switch to `extra='forbid'` to surface unexpected field names, which often reveals the misspelling or rename.
|
||||
|
||||
```python
|
||||
class TaskPayload(BaseModel):
|
||||
model_config = ConfigDict(extra='ignore')
|
||||
task_id: str
|
||||
prompt: str
|
||||
|
||||
# Silently drops 'task_id' if the input has 'taskId' (camelCase)
|
||||
payload = TaskPayload(**{"taskId": "abc", "prompt": "..."})
|
||||
print(payload.task_id) # None — no error raised
|
||||
```
|
||||
|
||||
**When to use `extra='ignore'`:** For external API payloads where forward-compatibility matters and the caller may add fields you don't care about. Always document that the model uses `extra='ignore'` so maintainers know unknown fields are dropped.
|
||||
|
||||
**Protocol / duplicate-store drift:** The same silent-drop failure mode appears at a different layer when multiple implementations share a `Protocol` or interface (e.g., `InMemoryStore` and `PostgresStore` both implementing a task store). Adding a field to the model isn't enough — each store's ORM-style `_to_row` / `_row_to_task` mappings (and any serialization helpers) must also learn the new field, or the field round-trips as its default/`None` in whichever store was missed. Symptom: value is present in-memory during tests but arrives as default/`None` in production where the other store is used.
|
||||
|
||||
**Rule:** After adding a field to a shared model, grep for every implementation of the Protocol and every `_to_row` / `_row_to_task` (or equivalent mapping) pair, and add a round-trip test per store that fails if the field is dropped.
|
||||
|
||||
## Pydantic v2: `BaseSettings` Moved to `pydantic-settings`
|
||||
|
||||
In Pydantic v2, `BaseSettings` was removed from the `pydantic` package and now lives in a separate `pydantic-settings` package. Code that does `from pydantic import BaseSettings` fails with `ImportError` on fresh installs.
|
||||
|
||||
**Fix:** Add `pydantic-settings>=2` as an explicit dependency in `pyproject.toml` whenever using `BaseSettings`. Do not rely on transitive installation via `pydantic` — it is not transitive.
|
||||
|
||||
```python
|
||||
# Pydantic v1
|
||||
from pydantic import BaseSettings
|
||||
|
||||
# Pydantic v2
|
||||
from pydantic_settings import BaseSettings
|
||||
```
|
||||
|
||||
## Use Routing Callables to Mock subprocess Without Patching Path Strings
|
||||
|
||||
Mocking `subprocess.run` (or `subprocess.Popen`) by patching the module path is brittle — the patch target must match exactly how the code imports it, and it breaks when code is refactored. A routing callable pattern is more robust:
|
||||
|
||||
```python
|
||||
# test helper — routes subprocess calls to per-command handlers
|
||||
def make_subprocess_router(routes: dict):
|
||||
"""
|
||||
routes: {command_prefix: mock_result_or_callable}
|
||||
e.g. {"git clone": CompletedProcess(...), "ssh": lambda cmd, **kw: ...}
|
||||
"""
|
||||
def router(cmd, **kwargs):
|
||||
for prefix, handler in routes.items():
|
||||
if isinstance(cmd, list) and " ".join(cmd[:len(prefix.split())]) == prefix:
|
||||
return handler(cmd, **kwargs) if callable(handler) else handler
|
||||
raise ValueError(f"Unrouted subprocess call: {cmd}")
|
||||
return router
|
||||
|
||||
# Usage in tests
|
||||
mock_runner = make_subprocess_router({
|
||||
"git clone": CompletedProcess([], 0, stdout="", stderr=""),
|
||||
"git push": CompletedProcess([], 0, stdout="", stderr=""),
|
||||
})
|
||||
|
||||
with patch.object(module_under_test, "subprocess_runner", mock_runner):
|
||||
result = module_under_test.run_task(payload)
|
||||
```
|
||||
|
||||
This avoids hard-coded patch target strings, handles multiple commands cleanly, and makes test setup readable.
|
||||
|
||||
## `model_validator` for Cross-Field Validation in Pydantic v2
|
||||
|
||||
Use `@model_validator(mode='after')` for validation that depends on multiple fields. Field-level validators (`@field_validator`) only see the single field being validated. Cross-field logic in field validators requires workaround hacks.
|
||||
|
||||
```python
|
||||
from pydantic import BaseModel, model_validator
|
||||
from typing import Optional
|
||||
|
||||
class TaskRuntime(BaseModel):
|
||||
cli: str = "claude"
|
||||
model: Optional[str] = None
|
||||
timeout: int = 1800
|
||||
|
||||
@model_validator(mode='after')
|
||||
def validate_model_for_cli(self) -> 'TaskRuntime':
|
||||
if self.cli == "openai_compat" and self.model is None:
|
||||
raise ValueError("model is required when cli='openai_compat'")
|
||||
return self
|
||||
```
|
||||
|
||||
**`mode='after'` vs `mode='before'`:**
|
||||
- `mode='after'` — runs after all field validators. `self` is the fully-constructed model instance. Use for cross-field checks.
|
||||
- `mode='before'` — runs on the raw input dict before field parsing. Use for input normalization (e.g., converting camelCase keys to snake_case).
|
||||
|
||||
## TLS 1.3 Post-Handshake Client Auth Requires Explicit `SSLContext`
|
||||
|
||||
Python mTLS clients using the convenience form `httpx.AsyncClient(cert=(crt, key))` fail against Go servers (Traefik, gRPC-Go, anything using the Go stdlib `crypto/tls`) because Go sends `CertificateRequest` as a TLS 1.3 **post-handshake** message. Python's `ssl` module ignores post-handshake client auth unless `SSLContext.post_handshake_auth=True` is set explicitly — and the convenience `cert=` argument does not set it.
|
||||
|
||||
**Symptom:** Server returns `401` "no client certificate" (or equivalent) even though the cert and key are configured correctly. Capping the negotiation to TLS 1.2 (`ssl.TLSVersion.TLSv1_2`) makes auth work — that's the diagnostic fingerprint.
|
||||
|
||||
**Fix:** Build an `SSLContext` explicitly and pass it as `verify=ctx`.
|
||||
|
||||
```python
|
||||
import ssl
|
||||
import httpx
|
||||
|
||||
ctx = ssl.create_default_context()
|
||||
ctx.post_handshake_auth = True
|
||||
ctx.load_verify_locations(cafile=ca_path)
|
||||
ctx.load_cert_chain(certfile=crt_path, keyfile=key_path)
|
||||
|
||||
client = httpx.AsyncClient(verify=ctx)
|
||||
```
|
||||
|
||||
**Diagnostic:** If forcing `minimum_version = maximum_version = ssl.TLSVersion.TLSv1_2` makes mTLS authenticate successfully while TLS 1.3 fails, you've hit the post-handshake gap. Don't ship the TLS 1.2 workaround — set `post_handshake_auth=True` instead.
|
||||
|
||||
## Recreate `venv` After Renaming or Moving a Project Directory
|
||||
|
||||
Python venvs embed absolute paths in pip shims (the shebang line of `.venv/bin/pip`, `.venv/bin/python` symlinks) and in `.pth` files for editable installs. After renaming or moving a project directory the venv looks intact — `python` runs, imports mostly work — but pip and editable installs break in subtle ways (`ModuleNotFoundError` for the local package, pip resolving against the wrong site-packages, etc.).
|
||||
|
||||
**Fix:** Recreate the venv. Don't try to patch paths in place.
|
||||
|
||||
```bash
|
||||
rm -rf .venv
|
||||
python3 -m venv .venv
|
||||
source .venv/bin/activate
|
||||
pip install -e ".[dev]"
|
||||
```
|
||||
|
||||
## Jinja2: Prefer `Environment` Whitespace Controls Over `{%- %}` in Code Templates
|
||||
|
||||
Using the hyphen-trim form `{%- ... -%}` inside Python (or other code-generating) templates collapses newlines between statements, producing syntactically invalid output — class bodies run together, method defs glue to the previous `return`, etc.
|
||||
|
||||
**Fix:** Configure whitespace once on the `Environment` and use plain `{% ... %}` tags inside code templates.
|
||||
|
||||
```python
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
|
||||
env = Environment(
|
||||
loader=FileSystemLoader("templates"),
|
||||
trim_blocks=True,
|
||||
lstrip_blocks=True,
|
||||
extensions=["jinja2.ext.do"], # if templates use {% do list.append(...) %}
|
||||
)
|
||||
```
|
||||
|
||||
- `trim_blocks=True` — strip the newline *after* a block tag.
|
||||
- `lstrip_blocks=True` — strip leading whitespace *before* a block tag on the same line.
|
||||
- `jinja2.ext.do` — enables `{% do %}` for list/dict mutation. Without it, templates using `{% do %}` fail at render time with a `TemplateSyntaxError`.
|
||||
|
||||
Inside code-generating templates, never use `{%-` or `-%}` — rely on the Environment settings.
|
||||
|
||||
## Packaging with `pyproject.toml` on Modern Linux
|
||||
|
||||
Three common setup failures, each with a one-line fix:
|
||||
|
||||
**(a) PEP 668 blocks system pip (Ubuntu 23.04+, Debian 12+, etc.).** Running `pip install` against the system Python raises `error: externally-managed-environment`. Always create a venv as step 1:
|
||||
|
||||
```bash
|
||||
python3 -m venv .venv
|
||||
source .venv/bin/activate
|
||||
pip install -e ".[dev]"
|
||||
```
|
||||
|
||||
**(b) setuptools flat-layout auto-discovery fails with multiple top-level dirs.** If your repo has more than one top-level directory (e.g., `mypkg/`, `tests/`, `scripts/`), setuptools' auto-discovery raises `Multiple top-level packages discovered` and refuses to build. Declare the package explicitly:
|
||||
|
||||
```toml
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["mypkg*"]
|
||||
```
|
||||
|
||||
**(c) `build-backend` typo.** The correct value is `"setuptools.build_meta"` (underscore, not plural). `"setuptools.backends"` / `"setuptools.build_metas"` yield `ModuleNotFoundError: No module named 'setuptools.backends'` at build time.
|
||||
|
||||
```toml
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
```
|
||||
|
||||
## Bridging `threading.Event` to `asyncio` — Use a Polling Bridge
|
||||
|
||||
When an HTTP server or other handler running in a **background thread** needs to signal an `asyncio` controller loop (e.g., webhook handler waking a reconciler), a plain `threading.Event.set()` cannot directly wake `asyncio.wait_for` or `asyncio.sleep` in the event-loop thread. The event loop only wakes for tasks it scheduled.
|
||||
|
||||
**Simplest bridge — async loop polls the event during its sleep interval:**
|
||||
|
||||
```python
|
||||
async def reconcile_loop(wake_event: threading.Event, interval: float = 30.0):
|
||||
while True:
|
||||
await do_reconcile()
|
||||
# Poll event every 1s instead of sleeping for the full interval
|
||||
for _ in range(int(interval)):
|
||||
if wake_event.is_set():
|
||||
wake_event.clear()
|
||||
break
|
||||
await asyncio.sleep(1)
|
||||
```
|
||||
|
||||
This accepts ~1s of bounded latency in exchange for a trivially correct bridge. Reach for `asyncio.run_coroutine_threadsafe(queue.put(...), loop)` only when your latency budget demands it — it's correct but adds complexity (you must capture the loop reference, handle loop shutdown, and deal with the queue from both sides).
|
||||
|
||||
## `dict.get(key, default)` Returns `None` When the Value Is Explicitly `null`
|
||||
|
||||
`raw.get("ttl", 300)` returns `None` — **not** `300` — when the key is present but its value is `null` / `None`. The default only applies when the key is *missing*. External APIs commonly return explicit `null` for optional fields (TTL, priority, timestamps, nullable FKs), which then crashes Pydantic validators that expect a non-null type, or propagates `None` through code that assumed the default kicked in.
|
||||
|
||||
**Fix — two options:**
|
||||
|
||||
```python
|
||||
# Option 1: falsy-coalesce at the call site (simple, but conflates 0/"" with null)
|
||||
ttl = raw.get("ttl") or 300
|
||||
|
||||
# Option 2: Pydantic before-validator mapping None → default (preferred for models)
|
||||
from pydantic import BaseModel, field_validator
|
||||
|
||||
class Record(BaseModel):
|
||||
ttl: int = 300
|
||||
|
||||
@field_validator("ttl", mode="before")
|
||||
@classmethod
|
||||
def _default_null(cls, v):
|
||||
return 300 if v is None else v
|
||||
```
|
||||
|
||||
Use the `or` form for throwaway scripts where `0` and `""` are not valid values. Use the Pydantic validator for models where you want to preserve legitimate falsy values while still mapping `null` to the default.
|
||||
|
||||
## Python Stdlib Sharp Edges: Helper Names, `re.sub`, Extensionless Imports
|
||||
|
||||
Three unrelated stdlib traps that all fail silently — no exception, wrong output:
|
||||
|
||||
**(1) Never use single-letter function names like `v`, `d`, `f`.** They silently shadow when callers rebind the same name (`v = dict(...)` in a caller's scope, list comprehensions using `v` as the loop variable, etc.). The failure mode is invisible: empty output, zero errors, no traceback. Use `fv()`, `format_value()`, `fmt_dict()` even for one-line helpers.
|
||||
|
||||
**(2) `re.sub()` interprets `$` and `\` in replacement strings as backreferences.** `\g<name>`, `\1`, `\\`, and even bare `\` in the replacement are all special — so any replacement containing literal dollar signs, backslashes, or backref-looking sequences is corrupted, and an equality check against the "expected" string fails without explaining why.
|
||||
|
||||
```python
|
||||
# Bad — '$' and '\' in replacement are reinterpreted
|
||||
re.sub(pattern, replacement_text, source)
|
||||
|
||||
# Good — escape replacement string
|
||||
re.sub(pattern, re.escape(replacement_text), source) # escapes backrefs but not $
|
||||
|
||||
# Better for arbitrary text — use search + slicing
|
||||
m = re.search(pattern, source)
|
||||
if m:
|
||||
source = source[:m.start()] + replacement_text + source[m.end():]
|
||||
```
|
||||
|
||||
When the replacement is arbitrary user-supplied or data-derived text, prefer `re.search()` + string slicing over `re.sub()`.
|
||||
|
||||
**(3) `importlib.util.spec_from_file_location` returns `None` for extensionless files.** Loading a script like `bin/my-tool` (no `.py`) via `spec_from_file_location` silently yields `None`, and subsequent `module_from_spec(None)` raises a confusing `AttributeError`.
|
||||
|
||||
```python
|
||||
import importlib.util
|
||||
import sys
|
||||
from importlib.machinery import SourceFileLoader
|
||||
|
||||
loader = SourceFileLoader("my_tool", "/path/to/bin/my-tool")
|
||||
spec = importlib.util.spec_from_loader("my_tool", loader)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules["my_tool"] = module # register BEFORE exec_module
|
||||
spec.loader.exec_module(module)
|
||||
```
|
||||
|
||||
The `sys.modules` registration must happen before `exec_module` — otherwise any `import my_tool` inside the loaded module creates a second, distinct module object.
|
||||
23
scripting.md
23
scripting.md
@@ -50,7 +50,30 @@ Every script that modifies state should support `--dryrun` / `-n`:
|
||||
- `grep` returns exit code 1 when no lines match — under `set -e`, this kills the script even when zero matches is a valid outcome. Append `|| true` to `grep` commands in pipelines where empty results are expected.
|
||||
- `while read` in a pipeline creates a subshell — variables modified inside the loop (counters, accumulators) are lost after the loop ends. Use process substitution (`while read line; do ...; done < <(command)`) or here-strings to keep the loop in the current shell.
|
||||
- **Order matters in sed/regex transformation pipelines.** Process more specific patterns before general ones. For example, if both `![[image.png]]` and `[[page]]` are valid patterns, process the image embed first — otherwise the general wikilink regex matches the inner `[[image.png]]` and the `!` prefix is left orphaned.
|
||||
- **Bare `except: pass` swallows `SystemExit` in Python.** `sys.exit(0)` inside a bare `except: pass` block is captured as a `SystemExit` exception and silently swallowed — the script continues instead of exiting. Use specific exception types in except clauses, or use `break`/`return` for loop early-exit, or re-raise after checking `isinstance(e, SystemExit)`. Applies to any script with loops that short-circuit on a condition inside exception handling.
|
||||
- **Never use `GROUPS` (or other reserved names) as a bash variable.** `GROUPS` is pre-set by bash completion and session initialisation with numeric group IDs — assignment appears to succeed but the pre-existing value often persists in sourcing contexts, producing bizarre "array iterates over 1000, 24, 27..." bugs. Other reserved/built-in names to avoid: `UID`, `EUID`, `PWD`, `OLDPWD`, `SHLVL`, `RANDOM`, `SECONDS`, `LINENO`, `PIPESTATUS`, `IFS`. Prefix project variables (`PROJECT_GROUPS`, `TEMPLATE_SLUGS`).
|
||||
|
||||
## JSON Construction in Scripts
|
||||
|
||||
Use Python (not shell) for constructing JSON payloads. Multi-line prompts with quotes, backticks, and special characters break shell-based JSON construction (printf/sed/heredocs). Python's `json.dump` handles escaping correctly every time. For scripts that need to construct and submit JSON payloads, write the construction logic in Python even if the rest of the script is bash. For long agent prompts, `--prompt-file` with temp files is cleaner than heredocs — writing prompts to `/tmp/*.md` files avoids shell escaping issues and enables review before submission.
|
||||
|
||||
**Build payloads in a variable first — never nest `$(...)` with mixed quoting.** Patterns like `response=$(octopus_post "/endpoint" "$(python3 -c "...$var")")` silently mangle variable values inside the inner substitution. Always assign the JSON body to a variable, then pass the variable:
|
||||
|
||||
```bash
|
||||
body=$(python3 -c "import json; print(json.dumps({'name': '$name'}))")
|
||||
response=$(octopus_post "/endpoint" "$body")
|
||||
```
|
||||
|
||||
**Helper functions must send status messages to stderr.** If a helper both returns a value on stdout and prints progress (`echo`, `ok`, `info`), the captured stdout will contain the status text mixed with the return value. Send status to stderr: `ok "msg" >&2`, so callers capturing stdout get only the return value.
|
||||
|
||||
## Markdown-to-PDF Pipeline (pandoc / ODT / LibreOffice)
|
||||
|
||||
Key gotchas when building branded PDFs from markdown via pandoc + ODT + LibreOffice:
|
||||
|
||||
1. **Pandoc treats standalone `---` lines as YAML frontmatter delimiters.** Replace horizontal rules with `* * *` after the frontmatter block so pandoc doesn't misparse the document.
|
||||
2. **Avoid Unicode in YAML frontmatter string values.** Em-dashes and other non-ASCII characters in frontmatter cause pandoc parse failures — use plain ASCII or escape.
|
||||
3. **Never use shell `sed` on ODF XML.** ODF uses namespaces (`fo:`, `style:`) that sed can't target reliably. Use Python `xml.etree.ElementTree` with namespace registration.
|
||||
4. **`--toc` leaves the TOC body empty in headless builds.** Pandoc creates the TOC element but leaves `<text:index-body>` empty, and LibreOffice headless doesn't populate it. Build TOC entries directly in XML.
|
||||
5. **Prefer direct XML manipulation over python-uno.** python-uno socket servers are unreliable in headless builds.
|
||||
6. **Cover-page page-number suppression requires a master-page chain.** Use a `CoverPage` master with `next-style-name = "Standard"`, not a single page style.
|
||||
7. **Prefer Python over shell `sed` for markdown compilation with footnotes/references.** Escaping and multi-line handling is far more reliable.
|
||||
|
||||
@@ -26,12 +26,22 @@ Each Ansible project needs `vars_plugins_enabled = host_group_vars,community.sop
|
||||
|
||||
## Credential Handling
|
||||
|
||||
- **Never pass secrets via command-line arguments** — visible in `ps` output
|
||||
- **Never pass secrets via command-line arguments** — visible in `ps` output to any process in the PID namespace, including other containers sharing the namespace
|
||||
- Use `@file` references, environment variables sourced at runtime, or stdin
|
||||
- For Ansible, use temp files with `trap rm` cleanup: `-e "@${tmpfile}"`
|
||||
- Read secrets at execution time and use them ephemerally — never cache or persist values
|
||||
- Reference the **existence** of a secret file in docs, never its contents
|
||||
|
||||
### Container Entrypoints
|
||||
|
||||
The `ps`-visibility problem is especially easy to hit in container entrypoints that wrap a CLI tool. Passing credentials as argv is visible to every process in the PID namespace. Pattern:
|
||||
|
||||
1. Write the credential to a temp file inside the container
|
||||
2. Use the tool's file-based import flag (e.g. `workspace import <file>`, `--credentials-file`, `@file`)
|
||||
3. `rm -f` the temp file before exec-ing the main process
|
||||
|
||||
Prefer stdin, env vars, or `@file` references over argv in every entrypoint script.
|
||||
|
||||
## Bootstrap Secrets
|
||||
|
||||
Some secrets are chicken-and-egg (e.g., the age decryption key for ArgoCD's KSOPS). These must be created manually as a bootstrap step and documented clearly.
|
||||
@@ -114,3 +124,27 @@ Shell `source` on .env files executes arbitrary commands — a crafted file with
|
||||
## URL-Safe Password Generation
|
||||
|
||||
Generated passwords that appear in connection strings (DATABASE_URL, AMQP URLs, etc.) must use URL-safe characters only: `A-Za-z0-9._-`. Characters like `^`, `@`, `:`, `/`, `+` break URL parsing in libraries like SQLAlchemy. Prevention via charset restriction is simpler and more reliable than URL-encoding passwords after generation.
|
||||
|
||||
## Scope Secret Delivery Per-Workload
|
||||
|
||||
When a harness or container environment makes secrets available (SSH keys, API tokens, credentials mounts), scope each secret to the specific workloads that need it. Global forwarding — mounting all credentials into every container or injecting all secrets into a shared env — leaks credentials to workloads that shouldn't have them.
|
||||
|
||||
**Anti-pattern:** Inject the Gitea admin token, Anthropic API key, and SSH private key into every agent container regardless of task.
|
||||
|
||||
**Pattern:** Use capability harness layers that compose per-task. A spec-planning task gets the planning context + SSH key. A code-review task gets the code-review context + API token. A read-only analysis task gets no write credentials at all.
|
||||
|
||||
This also limits blast radius when an agent is compromised or misbehaves — it can only escalate within the credentials it was explicitly given.
|
||||
|
||||
## Admin vs User API Tokens — Verify `is_admin` Before Assuming Scope
|
||||
|
||||
Not every token labelled "admin" has the platform's `is_admin` flag. Gitea's `cluster-administrator` user is a cluster admin but not a site admin — its token returns 403 on admin-API endpoints.
|
||||
|
||||
**When a token fails with 403 on admin endpoints:**
|
||||
|
||||
1. Don't assume the token is wrong or expired
|
||||
2. Check `GET /users/<owner>` — look for `"is_admin": true` on the owning user
|
||||
3. If `is_admin: false`, the token's owner lacks the platform privilege; a different user's token is required
|
||||
|
||||
**Gitea `Sudo` header quirk:** creating tokens on behalf of other users via the `Sudo` header returns 401 with a token. Use basic auth for that specific operation.
|
||||
|
||||
The general lesson: "admin" is overloaded (org admin vs site admin vs cluster admin vs repo admin). Always confirm which scope a token actually carries before blaming the token.
|
||||
|
||||
@@ -77,3 +77,37 @@ Some scenarios genuinely require client-side credentials (e.g., direct S3 upload
|
||||
**Wrong:** CMS authenticates directly with Gitea via OAuth popup. Gitea token lands in browser localStorage. CMS makes API calls directly to Gitea with the token.
|
||||
|
||||
**Right:** CMS sits behind Authelia (MFA). A proxy service intercepts the OAuth flow, issues a proxy session token (containing only the user's identity), and forwards all API calls to Gitea using a per-user server-side Gitea token. The Gitea token never leaves the server.
|
||||
|
||||
## Multi-Tenant Isolation Guard on Outbound Writes
|
||||
|
||||
When a service writes into per-tenant destinations — customer Slack channels, per-customer boards, per-org webhooks, tenant-scoped buckets — route every outbound write through a single **IsolationGuard** module.
|
||||
|
||||
### What the guard validates
|
||||
|
||||
For every outbound write, the guard checks:
|
||||
|
||||
1. **Destination ownership** — the target channel/board/webhook belongs to the tenant the write is scoped to
|
||||
2. **Payload cross-references** — the payload does not reference other tenants by name or ID (string match against the known tenant list)
|
||||
3. **Cross-tenant field stripping** — fields known to carry cross-tenant context (internal descriptions, linked-issue titles, audit trails) are removed or redacted before leaving the service
|
||||
|
||||
### Why centralise it
|
||||
|
||||
Per-tenant isolation is only testable if there is one place to exercise. If each call site inlines its own "scope this write" logic, cross-tenant leak tests must cover every call site and every future one. A single guard module:
|
||||
|
||||
- Gives tests one surface to fuzz with adversarial payloads
|
||||
- Makes it impossible to ship a new outbound path that forgets the check (the guard is the only API)
|
||||
- Centralises logging for any blocked write — leaks become observable, not silent
|
||||
|
||||
### Pattern
|
||||
|
||||
```
|
||||
CallSite ──→ IsolationGuard.send(tenant_id, destination, payload)
|
||||
│
|
||||
├── Validate destination ∈ tenant_id's destinations
|
||||
├── Scan payload for other tenants' names/IDs
|
||||
├── Strip cross-tenant fields per schema
|
||||
├── Log (tenant_id, destination, redacted fields)
|
||||
└── Dispatch to underlying transport
|
||||
```
|
||||
|
||||
No call site should import the underlying transport directly. Lint or grep for direct imports as a CI check.
|
||||
|
||||
@@ -65,3 +65,42 @@ Skills that review artifacts (plans, specs, designs) should be read-only — res
|
||||
## Separate Formatter Exit Codes from Hook Exit Codes
|
||||
|
||||
When integrating formatters with Claude Code hooks, keep formatter scripts and hook dispatch logic separate. Formatter scripts exit 0 (clean) or 1 (lint errors). The dispatch hook decides the final exit code semantically (e.g., exit 2 for PostToolUse feedback). This separation means the same formatter scripts work for both PostToolUse hooks and pre-commit hooks without modification.
|
||||
|
||||
## Pre-fetching Bounded Metadata vs. Unbounded Content
|
||||
|
||||
When a skill needs to examine large external artifacts (transcripts, logs, dumps), have a helper script pre-gather **metadata only** (a small JSON list: paths, sizes, headers) at skill-load time via `!`command``. Delegate the actual content extraction to a subagent invoked from the skill's instructions.
|
||||
|
||||
This keeps the main conversation's context small — the skill sees a compact index rather than megabytes of raw content — and lets each step pick the cheapest/strongest model for the job (metadata triage with Haiku, deep extraction with Sonnet). Never pre-fetch unbounded content via `!`command``; it bloats the system prompt and often blows past context limits.
|
||||
|
||||
## Shell Wrappers to Work Around `!`command`` Restrictions
|
||||
|
||||
When a skill needs dynamic values like `$(pwd)` in its pre-fetch command, write a thin shell wrapper script that does the substitution internally and expose the wrapper in `!`command``. The Claude Code permission checker rejects `$()` inside bang commands (see `!`command`` Gotchas above), but a wrapper invoked as a plain binary is fine.
|
||||
|
||||
Document the wrapper's reason-for-existence in a comment at the top of the script so it can be removed if the `$()` restriction ever lifts. Keep wrappers minimal — one responsibility each — so the indirection doesn't obscure what the skill is actually doing.
|
||||
|
||||
## Model Selection for Subagents by Task Type
|
||||
|
||||
When a skill spawns a subagent for a subtask, pick the model by cognitive demand:
|
||||
|
||||
| Task type | Model |
|
||||
|---|---|
|
||||
| Deterministic extraction, reformatting, metadata triage | Haiku |
|
||||
| Judgment required — detecting backtracking, gotchas, intent | Sonnet |
|
||||
| Planning, architecture, cross-file synthesis | Opus |
|
||||
|
||||
Document the rationale in the skill (a comment near the subagent invocation is enough) so future edits don't silently downgrade quality by picking a cheaper model without revisiting whether the task actually fits it.
|
||||
|
||||
## Subagent Path Discipline
|
||||
|
||||
Subagents inherit no cwd context from the parent skill — they start in whatever working directory the harness gives them, which is rarely where the parent was invoked. When a subagent writes output files, **pass absolute paths in its prompt** and never rely on relative paths resolving the way the parent expects.
|
||||
|
||||
Verify on the first real run that files land where intended; a subagent writing to a surprise cwd often fails silently (the parent looks for the file, doesn't find it, falls through to a default). `CLAUDE_PROJECT_ROOT` (see Portable Path Resolution above) is the canonical anchor — compute absolute output paths from it before handing them to the subagent.
|
||||
|
||||
## Separate Skill-Helper Scripts from General Scripts
|
||||
|
||||
Not every script in a repo is a skill helper. Maintain an explicit allowlist (e.g., a `SKILL_HELPERS` array in the installer) of scripts that get symlinked into the profile's skills-reachable dir (`$CLAUDE_CONFIG_DIR/scripts/`). Other scripts stay callable only by absolute repo path.
|
||||
|
||||
**Key points:**
|
||||
- **Symlinks, not copies.** Edits to the repo go live immediately without reinstalling.
|
||||
- **Hooks follow the same installer pattern** via their own allowlist loop — don't conflate hooks with general skill helpers.
|
||||
- **Explicit allowlist beats glob.** Auto-symlinking every script means new unrelated scripts silently become skill-reachable, which is a permission-scope surprise.
|
||||
|
||||
@@ -80,6 +80,16 @@ Guidelines:
|
||||
- Name the scenario descriptively — it becomes the test function's docstring
|
||||
- Include enough setup detail that an agent can write the test without guessing
|
||||
|
||||
## Plan Mode Produces Architecture, Not Contracts
|
||||
|
||||
Plans and specs answer different questions:
|
||||
- **Plans** answer "what we'll build" — milestones, tech choices, deployment shape, high-level architecture
|
||||
- **Specs** answer "how it must behave" — numbered requirements, scenarios, data models, interfaces
|
||||
|
||||
Skipping specs and jumping straight from plan to code forces retroactive spec writing once behaviour questions surface — and then tests written against the implementation have to be rewritten against the spec. This has been measured at ~30% of a session in one case.
|
||||
|
||||
**Rule:** for any multi-milestone coding project, enforce Plan → Spec → Test → Code. The plan is not a substitute for the spec.
|
||||
|
||||
## The Spec → Test → Code Workflow
|
||||
|
||||
This is the core development loop. Tests are written from the spec before code exists.
|
||||
@@ -229,6 +239,16 @@ Integration failures fall into a taxonomy of root causes. Categorize failures be
|
||||
|
||||
When splitting work into parallel agent tasks, ensure each task writes to distinct files. When two agents must modify the same file, make the shared changes small and predictable — identify the conflict point upfront so the merge is trivial. File-boundary decomposition produces zero-conflict assemblies.
|
||||
|
||||
### Parallel Agent Patterns: File Contention, WebFetch Limits, Narrow Reads
|
||||
|
||||
When using parallel subagents (Task tool with multiple concurrent invocations):
|
||||
|
||||
1. **Never have two agents edit the same file concurrently.** Subagent writes silently collide — one agent's edit wins and the other is lost with no error. Instead, have each agent RETURN prepared text in its response and apply the edits sequentially from the main thread. The main thread owns writes; agents produce content.
|
||||
2. **Subagents cannot use WebFetch** (permission denied in the subagent sandbox). Perform web fetches in the main conversation and pass the retrieved content to agents as input. Delegate file processing and analysis to agents, not network IO.
|
||||
3. **Give agents narrow read instructions** (e.g., "read only the Key Data Points section of finding X") to prevent expensive full-file reads that blow their context budget.
|
||||
|
||||
Verified pattern: 4 parallel agents splitting files 3-4 each verified 195 claims across 16 files in a single round, returning prepared findings to the main thread for sequential application.
|
||||
|
||||
### Choose Manual Implementation for Tightly-Coupled Cross-Component Work
|
||||
|
||||
When changes are small per file (5-15 lines) but tightly coupled across many files (each change depends on the previous), skip agent orchestration and implement manually. The assembly overhead exceeds the implementation time. Agent orchestration excels when tasks are independent and substantial; manual implementation excels when work is sequential and interconnected.
|
||||
@@ -278,3 +298,13 @@ Agents working in isolated worktrees or containers cannot discover mock targets
|
||||
## Plan-First Approach Eliminates Fix Cycles for Cross-Cutting Changes
|
||||
|
||||
For changes touching 5+ files across multiple subsystems, invest 30-45 minutes in exploration and planning before writing code. Measured results: sessions with plan-first had 0 fix commits; sessions with code-first had 7:1 fix:forward ratios. Use parallel exploration agents to cover different dimensions of the problem space.
|
||||
|
||||
## Wave-Based TDD Dispatch
|
||||
|
||||
For large milestones with many subsystems, dispatch test-writing and implementation as two explicit waves rather than interleaving them per agent:
|
||||
|
||||
**Wave 1 — Test agents:** Each agent receives the spec for one subsystem and writes all tests. No implementation code yet. Tests must all fail (or be skipped) at the end of Wave 1. Commit the test files to the agents branch.
|
||||
|
||||
**Wave 2 — Implementation agents:** Each agent receives the spec + the failing tests written by Wave 1. The agent's success criterion is "make your tests pass without modifying the test file." This hard separation prevents the common failure mode where an agent makes a test pass by weakening it.
|
||||
|
||||
**Human review gate between waves:** Before starting Wave 2, review the Wave 1 test files for coverage gaps and assert quality. It's cheaper to fix tests before implementation than after. Check that tests are genuinely failing (not just skipped), that assertions are specific, and that edge cases from the spec scenarios are covered.
|
||||
|
||||
@@ -371,6 +371,16 @@ When an implementation agent is gated by tests, **place tests in a read-only ref
|
||||
|
||||
Use filesystem enforcement, not prompt instructions. In a 3-way model comparison: Sonnet respected "DO NOT MODIFY" instructions; MiniMax edited tests 7 times; Haiku rewrote the entire test file. The filesystem makes modification impossible regardless of model.
|
||||
|
||||
### Test Infrastructure Files Must Be Protected from Agent Modification
|
||||
|
||||
Read-only protection must extend beyond test files to include **test infrastructure**: `conftest.py`, `pyproject.toml`, `pytest.ini`, `setup.cfg`, `tox.ini`. An agent can satisfy tests by adding pytest hooks in a writable `conftest.py` — for example, a `pytest_collection_modifyitems` hook that skips failures, or an autouse fixture that monkeypatches the system under test. Security reviews of agent gate implementations have repeatedly found `conftest.py` as a CRITICAL bypass vector.
|
||||
|
||||
**Rule:** the read-only reference directory must contain all test-discovery and test-configuration files, not just `test_*.py`. At gate-enforcement time, run pytest with `--rootdir` / `--confcutdir` pointed at the read-only tree so writable copies of these files in the working directory cannot override the authoritative configuration.
|
||||
|
||||
### Stand Up Real Test Infrastructure Early
|
||||
|
||||
Deferring a real test database/service (Docker Compose, testcontainers, ephemeral Postgres, etc.) pushes integration tests into a "deselected" bucket that nobody runs. Set up the test backing service in the first phase that touches it, so integration tests execute from day one instead of accumulating as tech debt. The cost of standing up ephemeral infrastructure is almost always lower than the cost of letting integration coverage rot.
|
||||
|
||||
### Model Selection for Implementation Agents
|
||||
|
||||
**Sonnet is the minimum viable model for constrained implementation tasks** (spec + test gate). Smaller and cheaper models modify test files or ignore constraints:
|
||||
@@ -482,6 +492,12 @@ When test files import optional packages (e.g., `sqlalchemy`, `psycopg`) at modu
|
||||
|
||||
Patching an entire module (e.g., `patch("mod.kubernetes.config")`) replaces exception classes with MagicMock objects. `except SomeException` then catches `MagicMock` instead of the real exception, causing tests to pass the wrong code path. Patch individual functions (`load_incluster_config`, `load_kube_config`) and leave exception classes intact so `except` clauses work correctly.
|
||||
|
||||
### MagicMock Returns Truthy in Controller Loops — Always Set Boolean Defaults
|
||||
|
||||
`MagicMock()` return values are truthy by default. If a controller loop calls `mock.should_stop()` or `mock.reconcile_triggered()` and the mock has no explicit `return_value`, the loop never exits — or never sleeps — because every call returns a truthy MagicMock. In one real incident this pattern consumed 40GB RAM before OOM.
|
||||
|
||||
**Always set `mock.method.return_value = False`** for boolean-returning methods used in loop predicates. Combine with `pytest-timeout` (e.g., `timeout = 10` in `pyproject.toml`) on any project with async loops, signal handlers, or sleep patterns so runaway tests die fast instead of starving the test host.
|
||||
|
||||
### Async Migration Requires Full Test Conversion
|
||||
|
||||
When migrating a codebase from sync to async, helper functions get converted but test functions are often left as sync `def`. Every test that calls an async function needs `async def` + `@pytest.mark.asyncio` + `await`. After any async migration, run tests and grep for `RuntimeWarning: coroutine '...' was never awaited` to find remaining sync-to-async gaps.
|
||||
|
||||
@@ -22,6 +22,8 @@ After wiring up any new service or endpoint, test end-to-end from the user's per
|
||||
- Test from the actual consumer (not same-namespace test pods for network policies)
|
||||
- Test DNS resolution after deploying FQDN-based policies
|
||||
|
||||
**Healthy != reachable.** A container showing `healthy` (pg_isready, HTTP 200, internal probe) can still have stale or broken external listeners — e.g., a TCP port bound to a namespace that was recreated, or an auth-backed endpoint accepting default credentials. Always probe the actual path users traverse: `nc -zw2 <ip> <port>` for TCP, `curl` the public URL, test auth APIs with known-bad and known-good creds. Probing the real path saves diagnosis time when the container's internal health signal diverges from external reachability.
|
||||
|
||||
## Pre-Flight Checks
|
||||
|
||||
Before starting a deploy or automation phase:
|
||||
@@ -112,3 +114,53 @@ When confirming whether a fix is deployed, `kubectl exec deploy/<name> -- ls <pa
|
||||
## Think Through K8s Constraints Before Coding Docker-First Solutions
|
||||
|
||||
Before implementing a feature that works in Docker, enumerate the K8s differences: read-only Secret volumes, root-owned files, no host-path mounts, separate pod filesystem, env var size limits. Design for both backends upfront. Planning for both environments from the start eliminates costly iteration cycles.
|
||||
|
||||
## Inert-by-Default Strategy for Risky Feature Flags
|
||||
|
||||
When shipping a significant behaviour change (new persistence layer, new auth path, changed state machine), ship the code in an inert state first — hidden behind a feature flag that defaults to off. Let the code deploy and stabilise in production without activating the behaviour. Then flip the flag in a separate commit or config change. This decouples "did the deploy succeed?" from "did the feature work?" and gives a clean rollback path (revert the flag) without reverting code.
|
||||
|
||||
**Pattern:** `ENABLE_NEW_FEATURE=false` ships with the code. A follow-up change flips it to `true`. If the feature has issues, revert only the flag change.
|
||||
|
||||
## Categorise Integration Failures by Root Cause Before Fixing
|
||||
|
||||
When an integration test run produces multiple failures, resist the urge to fix them one by one. Categorise first:
|
||||
|
||||
1. **Infrastructure** — missing env vars, wrong import paths, network timeouts
|
||||
2. **Interface drift** — a shared data model changed and callers weren't updated
|
||||
3. **Test rot** — test assumptions diverged from the current implementation
|
||||
4. **Logic bugs** — actual code defects
|
||||
|
||||
Fixing by category is more efficient than fixing in order: all infrastructure failures have the same root fix, all interface drift failures share a data model change, etc. Categorising upfront also reveals the true scope — "3 logic bugs" is manageable; "14 failures across 4 categories" requires a different approach.
|
||||
|
||||
## Use `additionalProperties: false` Everywhere in JSON Schemas
|
||||
|
||||
For schema-validated YAML/JSON configs, apply `additionalProperties: false` to every `type: object` — except intentionally open maps (explicitly documented). Typos in config keys (`maxlength` vs `max_length`) are the #1 config error source; without strict rejection, they silently pass validation and produce incorrect behaviour instead of a clear error. Helm covers this for charts; generalise the pattern to all schema-validated configs.
|
||||
|
||||
## JSON Schema Draft Choice Affects Tuple Validation Syntax
|
||||
|
||||
Draft 2020-12 uses `prefixItems` for positional tuple validation and `items: false` to disallow extras. The `items`-as-array syntax (`items: [{type: string}, {type: number}]`) is Draft-04/07 only — using it under 2020-12 raises `jsonschema.SchemaError`. Always declare `$schema` explicitly in the schema root so readers and validators agree on the dialect.
|
||||
|
||||
## Quote YAML Reserved-Word Keys
|
||||
|
||||
In YAML, unquoted reserved words as keys parse to their typed values: `null: true` becomes `{None: True}` in Python, not `{"null": True}`. The same applies to `true`, `false`, `yes`, `no`, `on`, `off`. Always quote reserved-word keys: `"null": true`. Add a schema-level regex or loader-level check to catch unquoted reserved-word keys before they silently break consumers — in one project this hit 45 occurrences across 4 manifests.
|
||||
|
||||
## Coverage Thresholds vs Marker-Filtered Test Runs
|
||||
|
||||
`pytest -m integration` (or any marker-filtered subset) fails `--cov-fail-under` because most tests are deselected, dropping total coverage below the gate even when the filtered tests pass. Similarly, stub modules with `raise NotImplementedError` drop coverage — def/import lines still count. Use `--no-cov` for marker-filtered/smoke CI steps and run full-coverage as a separate step. When scaffolding stubs during milestones, temporarily lower the threshold and restore it after implementation.
|
||||
|
||||
## Research Target Format and Feature Availability Before Building Artifacts
|
||||
|
||||
When integrating with an external platform (Platform Hub, Octopus, any product with its own schema/scope rules), spend the first hour reading the target format documentation (OCL syntax, API schema, scope rules) and verifying feature availability before committing to an approach. Two failure modes this prevents:
|
||||
|
||||
1. **Wrong-format artifacts** — e.g., 12 step templates created via API before discovering the target's process templates can't reference step templates cross-space. All the work was wasted.
|
||||
2. **Unavailable features** — ask "what features of this product are actually available to me right now?" before planning. Assuming an unreleased feature (e.g., Project Templates) is available caused a plan to evolve three times mid-implementation.
|
||||
|
||||
## Safe Persistence Pattern: Persist-on-Change, Rehydrate-at-Startup, Clean-Slate Fallback
|
||||
|
||||
For services that maintain in-memory state that should survive restarts (task queues, session state, model scores):
|
||||
|
||||
1. **Persist on change** — write state to disk immediately when it changes, not on a timer
|
||||
2. **Rehydrate at startup** — read the persisted state file during service initialisation
|
||||
3. **Clean-slate fallback** — if the state file is missing, corrupt, or incompatible, start with empty state and log a warning rather than crashing
|
||||
|
||||
This pattern handles pod restarts, rolling deploys, and upgrade scenarios without requiring an external database. The clean-slate fallback is critical — a startup crash due to a corrupt state file is worse than losing the state.
|
||||
|
||||
Reference in New Issue
Block a user