diff --git a/plugins/claude-config/.claude-plugin/plugin.json b/plugins/claude-config/.claude-plugin/plugin.json index b2a68f0e52..9c09178a1c 100644 --- a/plugins/claude-config/.claude-plugin/plugin.json +++ b/plugins/claude-config/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "claude-config", - "version": "0.45.0", + "version": "0.46.0", "description": "Nine configuration-health skills (plus setup) for a repo's Claude Code configuration: audit (settings.json / .mcp.json / hooks / plugins / permissions drift), audit-automation-gaps (evidence-gated verdicts on automation gaps), audit-permission-grants (allow-rule / allowed-tools grants for auto-mode durability and portability), audit-permission-state (the permission rules actually in effect: every settings scope merged with per-rule provenance, what auto mode drops on entry, config written where nothing reads it, and which managed intents are enforced versus loosenable), draft-auto-mode-rules (interview and draft a paste-ready autoMode classifier block; prints only, never writes), audit-instructions (locally-owned instruction surfaces vs current model capability, proposing removals/rewrites of instructions the model no longer needs, and detecting cross-surface instruction conflicts), audit-prompting-postures (the additive lane: posture guidance the prompting guide says a component's purpose needs but the component does not carry), audit-pass (one coordinated, ordered, resumable pass over a named target: three-scope inventory, run-time-derived exclusion set, stable finding identity, suppression memory, resume, one human gate, delegating every check to the plugin that owns it), and unhobble (the empirical bare-baseline experiment: reversibly strip a repo's standing instructions, log real stumbles against the current model, re-add only what evidence earns).", "author": { "name": "Melodic Software", diff --git a/plugins/claude-config/CHANGELOG.md b/plugins/claude-config/CHANGELOG.md index 1f9fdab586..a102789ac1 100644 --- a/plugins/claude-config/CHANGELOG.md +++ b/plugins/claude-config/CHANGELOG.md @@ -3,6 +3,50 @@ All notable changes to the `claude-config` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [0.46.0] + +### Added + +- **`audit-automation-gaps`**: `scripts/findings-state.sh` persists a run's verdict table, evidence + and implementation plans, so `--implement` has something to read in a later session. Keyed per + project through the shared `lib/state-key.sh`, so one checkout never reads another's verdicts, + and `--plugin-data` is required because `${CLAUDE_PLUGIN_DATA}` is absent from the Bash tool + environment. +- **`audit-automation-gaps`**: a Gotchas surface built from observed failures, a `## Next` section, + and presence-gated routing to `overengineering:audit` in both directions. +- **`audit-automation-gaps`**: eval cases covering the persisted artifact and the revised + `Already enforced` gate. + +### Changed + +- **`audit-automation-gaps`**: `scripts/inventory.sh` counts all seven documented hook locations + instead of the project `.claude` tree alone, which had it reporting 3 hook scripts and 0 skills + for a repository carrying 93 wired handlers and 271 skills. An unreadable scope now reports + `unreadable` rather than zero, conditional scopes are split from the standing set, and enablement + inputs are emitted with a pointer instead of a computed verdict. +- **`audit-automation-gaps`**: a `git log --grep` count is now treated as a ceiling rather than a + frequency in both gates that consumed it. A frequency claim requires a sample, reported with its + denominator and sample size; a ceiling already under 5 percent still settles YAGNI without one. +- **`audit-automation-gaps`**: the `Already enforced` gate takes a shift-left carve-out with three + falsifiable conjuncts, keeping the hook budget a hard gate, so a consumer documenting a budget + with no headroom left keeps the REJECT. +- **`audit-automation-gaps`**: one batched docs fetch is mandatory for any candidate whose mechanism + is a Claude Code surface; conditionality survives only for non-harness facts, and a skip records + how the fact was settled. +- **`audit-automation-gaps`**: the refusal gate and the implementation review both dispatch a + fresh-context judge, degrading to an inline review that records itself as same-context. +- **`audit-automation-gaps`**: the findings store reports state it cannot trust separately from + state that is absent. A history file, pointer or payload that exists but is unreadable, an + incomplete publish, or a payload that does not parse, all exit 5; nothing stored still exits 4. + A missing `jq` is the prerequisite it is, exit 2, rather than a corruption claim about the + operator's artifact. +- **`audit-automation-gaps`**: the audit writes twice, once when verdicts are presented and once + after the human selects, so `--implement` acts on what was chosen rather than on every candidate + the skill happened to pass. +- **`audit-automation-gaps`**: `scripts/inventory.sh` exits 2 on a usage error or a project root it + cannot enter, where it previously always exited 0. Its only consumer already falls back to + `inventory unavailable`, so a nonzero exit degrades rather than injecting wrong numbers. + ## [0.45.0] ### Added diff --git a/plugins/claude-config/skills/audit-automation-gaps/SKILL.md b/plugins/claude-config/skills/audit-automation-gaps/SKILL.md index 3e6fd07522..0bee880fee 100644 --- a/plugins/claude-config/skills/audit-automation-gaps/SKILL.md +++ b/plugins/claude-config/skills/audit-automation-gaps/SKILL.md @@ -13,6 +13,12 @@ metadata: Automation inventory: !`bash "${CLAUDE_PLUGIN_ROOT}/skills/audit-automation-gaps/scripts/inventory.sh" 2>/dev/null || echo "inventory unavailable"` +That block counts files present separately from handlers actually registered, and marks a scope it +could not read as `unreadable` rather than as zero. Treat it as a starting point, never as a +finding. Re-derive any count a verdict leans on, and where your own count disagrees with the block, +report both numbers and say which one the verdict used. A partial inventory that reads as complete +is this skill's own recorded failure: see [context/gotchas.md](context/gotchas.md). + ## Purpose Audit the repo's automation landscape and identify genuine gaps, not surface-level "you don't have X" @@ -28,6 +34,21 @@ architecture tests → unit/integration tests → git hooks → Claude Code hook documentation/behavioral rules. A consuming repo that documents its own hierarchy in `CLAUDE.md` or rules files overrides this default ordering, so read and use theirs. +## Boundary: no machine-readable hook enumerator + +This skill ships no hook-enumeration mode, and the absence is a decision rather than a gap. The +native `/hooks` browser already reports the harness's own resolved set: it lists every event with a +count of the hooks configured on it, and selecting one "shows its details: the event, matcher, type, +source file, and command". A script here would restate that set from the settings files instead of +from the resolution the harness actually performs, which is the weaker source. Revisiting the +decision requires first recording a native-surface verdict for `/hooks` in whatever registry the +repository keeps for native-overlap decisions, because whether such a reference should exist at all +is a human's call. This paragraph is a boundary, not a routing verdict. + +| Claim | Basis | As of | Recheck trigger | +|---|---|---|---| +| `/hooks` opens a read-only browser listing every hook event with its configured count, and selecting a hook "shows its details: the event, matcher, type, source file, and command" | [Automate actions with hooks](https://code.claude.com/docs/en/hooks-guide), the hooks-browser step | 2026-09-13 | A re-fetch finds the browser no longer read-only, or showing a different detail set | + ## Adapting to your environment (graceful degrade) This skill is self-contained. Where a phase names an adjacent capability, such as a research skill, an @@ -36,6 +57,13 @@ your setup provides an equivalent, use it; otherwise follow the inline guidance, own. Adjacent skills cover neighboring questions: the sibling `audit` skill (are the config FILES correct?) and the `audit` skill in the `claude-memory` plugin (is the instruction layer healthy?). +This skill proposes NEW automation. The inverse question, whether the automation already in place +still earns its carry cost, is owned by `/overengineering:audit`, whose evidence-earned-keep walk +treats every incumbent as a retirement candidate until evidence keeps it. Send a retirement question +there when the `overengineering` plugin is installed; when it is not, the **Already enforced** gate +still names the mechanism covering a concern, and any retirement observation is recorded in the +verdict for the human rather than acted on here. + ## Arguments Parse `$ARGUMENTS` for: @@ -44,6 +72,14 @@ Parse `$ARGUMENTS` for: - **`--implement`**: After presenting validated recommendations, implement user-chosen items sequentially - **Category filter** (`hooks`, `mcp`, `skills`, `subagents`, `scheduled`): Limit to a specific automation type. Default: `all` +There is deliberately no `ci` category. Two skills own that ground. The `ci-lanes` layer of +`/overengineering:audit` walks the lanes already running and judges whether each still earns its +carry cost, and `/review:audit-enforceability` maps one review finding through its enforcement-rung +crosswalk to the cheapest rung that could catch it, a CI lane included. Route a CI concern to +whichever fits when the `overengineering` or `review` plugin is installed; when neither is, record +the concern in the verdict as out of scope and name its owner, rather than widening this audit's +categories. + ## Track progress For any deep-dive run (Phases 1-4), keep a phase checklist and tick each phase + every anti-noise @@ -57,16 +93,22 @@ Analyze the current automation landscape directly, with no external recommender ### 1.1 Inventory Current State -Run `bash "${CLAUDE_PLUGIN_ROOT}/skills/audit-automation-gaps/scripts/inventory.sh"`, then read: +The pre-computed block already ran +`bash "${CLAUDE_PLUGIN_ROOT}/skills/audit-automation-gaps/scripts/inventory.sh"`; re-run it directly +if that block is missing. It settles the per-scope hook-location counts, the component counts, and +the enablement inputs. It deliberately leaves three things to you: effective plugin enablement, +because `enabledPlugins` follows precedence rather than the merge that hook entries follow; the +contents of any frontmatter `hooks:` block, which it counts but does not parse; and plugins +installed on the machine from outside this repository. Then read: | What | Source | |------|--------| -| Hooks | Every hook location: user + project + local `settings.json` → `hooks`, managed policy settings when readable, each enabled plugin's `hooks/hooks.json`, and skill or agent frontmatter `hooks:` | +| Hooks | Every hook location: user + project + local `settings.json` → `hooks`, managed policy settings when readable, each enabled plugin's `hooks/hooks.json`, and skill or agent frontmatter `hooks:`. Also record which hook EVENTS you considered: the documented event set is far wider than `PreToolUse` and `PostToolUse`, so a no-gaps hooks verdict has to name the events it examined, because an event nobody looked at cannot surface a candidate | | MCP servers | `.mcp.json` + `settings.json` → `disabledMcpjsonServers` | | Permissions | `settings.json` → `permissions` (deny/ask/allow) | | Enforcement | Ecosystem-specific build config (``, e.g. `Directory.Build.props` for .NET, `pyproject.toml` for Python, `package.json` for TS/JS), `.editorconfig` (top 20 lines), any enforcement-hierarchy section in the repo's own instruction files | | Codebase shape | File counts per language: `find . -name "*.cs" \| wc -l`, same for `.py`, `.ts`, `.sh`, `.ps1` | -| Incident history | `git log --oneline \| wc -l` for baseline, `git log --oneline -50` for recent activity | +| Incident history | `git log --oneline \| wc -l` for the denominator, then a `--grep` count for the concern. That count is a CEILING, never a frequency: read a sample before citing one, and carry the match count, the denominator, the sample size and what the sampled commits actually were into the verdict | ### 1.2 Gap Analysis @@ -94,7 +136,10 @@ For each candidate: - **Read the relevant config/code**: the specific files that would be affected - **Check the enforcement hierarchy**: walk up from the weakest level (docs) to the strongest (compiler). Stop when you find coverage. Document which level covers it -- **Check incident history**: `git log --grep` for related problems. Count incidents vs total commits for frequency +- **Check incident history**: `git log --grep` for related problems, then read a sample of the hits. + The count is a **ceiling**, the number of commits whose message mentions a word, not a count of + incidents and not a frequency. Report the match count, the total commit count, the sample size + read, and what the sampled commits actually were - **Measure performance**: if the candidate involves a CLI tool as a hook, time it: ```bash @@ -108,10 +153,30 @@ fixed number this skill supplies is a house rule and is labelled as one in the v Budget-resolution ladder, per-event costs, the documented levers, and the dated upstream-fact records: read [context/hook-timing.md](context/hook-timing.md). -### 2.2 Research (Targeted, Parallel) +### 2.2 Research (One Mandatory Batch, Then Conditional) + +Research splits by who owns the fact, and the two halves are not optional in the same way. + +**Mandatory: one batched docs fetch for harness mechanisms.** Any candidate whose mechanism is a +Claude Code surface gets its harness facts fetched before its verdict, however settled the local +evidence already looks. That covers hook event semantics and what each event can and cannot block, +`async` and `timeout` behavior, MCP server scope and configuration, and skill, agent or plugin +frontmatter fields. Collect every such question across every surviving candidate into ONE fetch +pass rather than one pass per candidate. The batch is mandatory because a docs pass routinely +changes the shape of a candidate that local evidence had already settled: a gate written as `exit 2` +on `PermissionRequest` reads as settled locally and is inert upstream, which no amount of local +reading surfaces. + +**Conditional: facts the harness does not own.** Third-party tool performance, an ecosystem's +standard formatter, server stability and auth, known issues in a tool the candidate would run. Fetch +these only where the verdict turns on them. -Launch **parallel research agents** for candidates that survive initial Explore (weren't immediately -disqualified by performance or existing enforcement). Group related queries to minimize agent count. +**When a conditional fetch is skipped, record how the fact was settled** in that candidate's +verdict: the measurement taken, the file read, or the existing dated record relied on. "Not needed" +and "not done" must stay distinguishable in the artifact. + +Group related queries to minimize agent count, and run the mandatory batch as one dispatch for all +candidates rather than one per candidate. **Per-candidate research questions by type:** @@ -126,24 +191,84 @@ disqualified by performance or existing enforcement). Group related queries to m Verify Claude Code mechanisms against official docs ([code.claude.com/docs/en/hooks](https://code.claude.com/docs/en/hooks), [code.claude.com/docs/en/mcp](https://code.claude.com/docs/en/mcp)); use web search for tool -performance and community practices. Delegate the fetches and keep the verdicts: agents gather -evidence, and the gate decisions in 2.3 stay in the main context. +performance and community practices. Research agents gather evidence and state no verdict; the +gates in 2.3 read that evidence and decide. ### 2.3 Vet / Validate (Quality Gates) -A candidate **fails** if ANY gate triggers: +A candidate **fails** if ANY gate triggers. + +**Run this gate in a fresh context.** Phase 1 generated these candidates in the context now being +asked to refuse them, and a context that produced work is a biased judge of it. Dispatch a +fresh-context subagent, handing it the candidate list, the Phase 2.1 and 2.2 evidence, and this +table, and keep the gate results it returns. Where the environment exposes no subagent surface, run +the gates inline and say in the report that the refusal gate ran in the same context that generated +the candidates, so the bias is visible rather than assumed away. | Gate | Condition | Evidence Required | |------|-----------|-------------------| -| **Already enforced** | A higher enforcement hierarchy level covers it | Name the level and mechanism | +| **Already enforced** | A higher enforcement hierarchy level covers it, and the shift-left carve-out below does not apply | Name the level and mechanism; where the candidate claims the carve-out, the result of each of its three conjuncts | | **Too slow** | Measured cost exceeds the resolved budget ([context/hook-timing.md](context/hook-timing.md)) for the event's actual cost: a blocked call on a blocking event, turn latency on `PostToolUse`, which never blocks | Timing measurement + the budget it was judged against and that budget's source | | **Not scriptable** | The mechanism can't be automated with available tools | Specific limitation cited | -| **Zero incidents** | Git history shows the problem has never occurred | Incident count + total commit count | +| **Zero incidents** | A read sample of the matching commits shows the problem has never occurred. A `--grep` count alone never triggers this gate, in either direction | Keyword match count, total commit count, the sample size read, and what the sampled commits turned out to be | | **Already exists** | A skill, behavioral rule, or convention already handles it | File path and line | -| **YAGNI** | Low frequency (<5% of commits) AND low severity | Frequency calculation | +| **YAGNI** | Low frequency (<5% of commits) AND low severity, where the frequency is sampled rather than counted | The match count over total commits as the ceiling; plus a sampled frequency whenever that ceiling is 5% or higher, and none when it is already below | | **Platform mismatch** | Requires infrastructure the user doesn't have | Platform requirement cited | | **Premature** | Depends on unfinished work (planned database, future CI) | What's missing cited | +#### Incident counts are ceilings + +**Zero incidents** and **YAGNI** both read a `git log --grep` count, and that count is an upper +bound on the concern, not a measure of it: it counts commits whose message mentions a word. On this +marketplace at merged main `49912c63`, `secret|gitleaks` matched 209 of 2202 commits while the first +15 read were dependency bumps, component syncs and CI pins, with no leaked secret among them, and +`markdownlint|lint` matched 1128, about 51%. Cite a frequency only after reading a sample and saying +what the sampled commits were. + +The discipline is asymmetric, which keeps it cheap. A ceiling already below the 5% **YAGNI** +threshold settles that gate on its own, because the true frequency cannot exceed the ceiling, so no +sample is needed. Only a PASS, or an affirmative claim that the concern is frequent, has to pay for +one. + +#### Shift-left carve-out to Already enforced + +A candidate that deliberately duplicates a check a higher rung already runs, at edit time, is not +rejected by **Already enforced** alone. Duplication earns a hearing only when all three conjuncts +below hold, each stated with its evidence. Any conjunct that cannot produce its evidence fails, and +the candidate returns to REJECT. The carve-out changes which gate decides, never the REJECT-by- +default posture: a candidate that clears all three is handed to **Too slow** and the remaining +gates, not passed. + +1. **Sampled frequency, never incident history.** The concern clears **Zero incidents** and + **YAGNI** on a count sampled from the current tree, not from `git log`. A check the higher rung + already runs suppresses its own incidents, so zero incidents in history is evidence about the + incumbent check, not about the concern. Sample instead: run the check over a stated N of files + or of recent edits and report hits over that N. Falsified when the sampled hit rate is under + 5 percent, which is the **YAGNI** threshold read against the sampled denominator. State N and + the hit count, so the arithmetic is checkable rather than asserted. +2. **Advisory and non-blocking.** The hook exits `0` on every path, surfaces findings as context + rather than rejecting the edit, and its own documentation names a commit hook or a CI lane as + the hard gate. Falsified by any exit path that blocks the tool call, or by documentation + presenting the hook itself as the gate. +3. **Fits the remaining budget, as a hard gate.** Resolve the budget through the two-rung ladder in + [context/hook-timing.md](context/hook-timing.md), then report four things: the measurement, the + budget row judged against, which rung supplied it, and the headroom arithmetic. Falsified when + the measured cost exceeds the stated headroom. Where the consuming repository documents a budget + and its own accounting records no headroom on that row, the verdict is REJECT at **Too slow** + whatever conjuncts 1 and 2 returned; the carve-out does not reopen a budget the consumer says is + already overspent. + +Why the carve-out is scoped this narrowly: the hooks documentation states that hooks give +"deterministic control: certain actions always happen rather than relying on the LLM to choose to +run them", which supports duplicating a check so it runs on every edit without exception. It states +nothing about where a check belongs, in CI or in a hook, so no latency or placement argument may be +offered as an upstream one. The record for both facts: + +| Claim | Basis | As of | Recheck trigger | +|---|---|---|---| +| Hooks exist to give deterministic control, so certain actions always happen rather than relying on the LLM to choose to run them | [Automate actions with hooks](https://code.claude.com/docs/en/hooks-guide), opening paragraph | 2026-09-13 | A re-fetch finds the opening statement of purpose no longer matching this row | +| No official page addresses whether a check belongs in CI or in a hook | Absence record in [context/hook-timing.md](context/hook-timing.md), upstream-fact records | 2026-09-12 | That record's own trigger fires | + These gates answer *should* we mechanize. Whether a candidate *can* be, and how far up the hierarchy it climbs, is a separate enforceability question: the **Not scriptable** gate maps to concerns no hook type reaches, **Already enforced** to a deterministic finding already escalated. `prompt` and @@ -152,7 +277,7 @@ hook type reaches, **Already enforced** to a deterministic finding already escal Produce a verdict per candidate: - **PASS**: clears all gates, provides genuine value -- **CONDITIONAL PASS**: value exists but only under specific conditions (document the condition) +- **CONDITIONAL**: value exists but only under specific conditions (document the condition) - **REJECT**: fails one or more gates (cite which gates and evidence) ## Phase 3: Present Results @@ -160,19 +285,25 @@ Produce a verdict per candidate: ### Summary Table ```markdown -| # | Candidate | Category | Verdict | Key Evidence | -|---|-----------|----------|---------|--------------| -| 1 | ... | Hook | REJECT | [gate]: [evidence] | -| 2 | ... | Skill | PASS | No existing coverage, [frequency]% of commits | +| # | ID | Candidate | Category | Verdict | Key Evidence | +|---|----|-----------|----------|---------|--------------| +| 1 | h1 | ... | hooks | REJECT | [gate]: [evidence] | +| 2 | s1 | ... | skills | PASS | No existing coverage, [n] of [total] commits, [k] sampled | ``` +`ID` is a short stable token for the candidate, and `Category` is one of `hooks`, `mcp`, `skills`, +`subagents`, `scheduled`. Both are what the persisted artifact is keyed and filtered by, so a row +without them cannot be written back or read by `--implement`. + ### Detailed Verdicts For each candidate, provide: - **Verdict** with gate results - **Evidence** (timing data, git history counts, enforcement mechanisms found) -- **For PASS/CONDITIONAL PASS**: implementation plan (files to change, effort estimate, test strategy) +- **For PASS/CONDITIONAL**: implementation plan (files to change, effort estimate, test strategy). + A `PASS` or `CONDITIONAL` without a plan cannot be persisted, because the plan is precisely what + `--implement` reads back in a later session **Order by value/impact**, highest value first. Value = frequency × severity × ease of implementation. @@ -186,16 +317,70 @@ If items pass: "Which items would you like to implement? I'll work through them If no items pass: State that clearly. A clean bill of health is a valid outcome. +### Persist the findings + +Write twice, because the selection does not exist yet when the verdicts do. + +Write once as soon as the verdicts are presented, so an audit is never lost to a session that ends +at the question above. Every candidate carries `approved: false` at this point, which is true: the +human has not chosen yet. + +Write again once the human answers, with `approved: true` on the candidates they picked. Run ids +are suffixed rather than overwritten and `read` serves the newest run, so the second write is what +a later `--implement` sees, and the first survives as the record of what the audit actually found. +Skip the second write only when nothing passed, since there is then nothing to select. + +```bash +bash "${CLAUDE_PLUGIN_ROOT}/skills/audit-automation-gaps/scripts/findings-state.sh" write \ + --plugin-data "${CLAUDE_PLUGIN_DATA}" --findings +``` + +`${CLAUDE_PLUGIN_DATA}` substitutes here in the skill text but is absent from the Bash tool's own +environment, so it has to travel as that argument; the script exits 2 rather than guessing when it +is missing. The payload is exactly one JSON object with a `candidates` array, each entry carrying +`id`, `candidate`, `category`, `verdict`, an `evidence` array, and a `plan` object on any `PASS` or +`CONDITIONAL`. Each entry also carries `approved`, the boolean that records the human's selection. +A malformed payload, or a stream carrying more than one document, exits 3 and writes nothing, +listing every problem at once. + +Writes are all or nothing. A run id already taken is reserved under the next free suffix rather than +overwritten, so two runs racing for one id both survive and each is told the id it actually got. + +Findings are keyed per project, so one checkout never reads another's verdicts. This tree is the +only durable copy: uninstalling the plugin without `--keep-data` deletes it. + ## Phase 4: Implement (if `--implement` or user requests) +Recover the prior run's verdicts first, rather than re-deriving them: + +```bash +bash "${CLAUDE_PLUGIN_ROOT}/skills/audit-automation-gaps/scripts/findings-state.sh" read \ + --plugin-data "${CLAUDE_PLUGIN_DATA}" +``` + +That serves this project's newest run. Implement the candidates whose `approved` is `true`. Where +no candidate carries `approved: true`, the stored run predates the human's selection, so ask which +items to implement rather than treating every passing candidate as chosen: a `PASS` verdict is this +skill's judgment, not the operator's consent. + +Exit 4 means this project has no stored run, which is the cue to audit first rather than to +implement from memory; the script never falls back to another project's artifact. Exit 5 is the different answer: stored state exists but cannot be trusted, +because a file is unreadable or a previous write half landed. Treat 5 as a state to repair or +re-audit, never as an absence, and say which of the two you got. `list` shows the earlier runs when +a specific one is wanted. + For each user-selected item: 1. **Explore**: re-read current state (may have changed since evaluation) -2. **Research + Validate**: verify the implementation approach against current docs +2. **Research + Validate**: verify the implementation approach against current docs. The audit + already ran one batched fetch for Claude Code surfaces in Phase 2.2, so re-fetch only what the + stored evidence does not already settle, or what has plausibly moved since the run was written 3. **Plan**: detailed implementation steps 4. **Implement**: execute with incremental validation and commit checkpoints 5. **Test**: verify the automation works (run hooks, test skills, etc.) -6. **Review**: self-review for consistency with existing patterns +6. **Review**: dispatch a fresh-context reviewer to check the change against existing patterns, + since the context that wrote it is a biased judge of it. Where no subagent surface exists, + review inline and record that the review was same-context 7. **Verify**: build/test the repo if code changed, confirm no regressions Route steps 2-7 through the consuming environment's own workflow skills when it has them; the inline @@ -207,8 +392,23 @@ Principles that govern every verdict: 1. **The enforcement hierarchy is your first check.** Most "gaps" are covered by the compiler-through-git-hooks levels 2. **Measure, don't assume.** Time every tool before recommending it as a hook. A formatter looks perfect until you measure 15+ seconds per file -3. **Check incident history.** Zero incidents across 100+ commits is strong evidence the problem doesn't exist in practice +3. **Check incident history, then read it.** Zero incidents across 100+ commits is strong evidence the problem doesn't exist in practice, but a `git log --grep` count is a ceiling on the incidents, not a count of them, and a check a higher rung already runs suppresses its own incidents 4. **Behavioral rules are valid enforcement.** A rule in CLAUDE.md is legitimate coverage; not everything needs a hook or script 5. **YAGNI is a quality gate, not laziness.** Automation for a task at 4% frequency is premature 6. **Premature is worse than missing.** Adding a database MCP before a database exists, or scheduling before CI exists, creates maintenance burden for zero value 7. **A clean bill of health is a valid outcome.** Not finding gaps means the automation is mature; don't manufacture recommendations to justify the audit + +## Next + +- Candidates passed and are ready to build: `/planning:plan`. +- Candidates passed and need slicing into grabbable tickets first: `/work-items:decompose`. +- The run raised a question about automation already in place: `/overengineering:audit`. +- A verdict is worth revisiting later rather than acting on now: `/work-items:track add`. + +## Gotchas + +Five failure modes this skill has actually produced, each with the evidence and the dated upstream +records behind it: a partial inventory anchoring every later verdict, keyword counts read as +frequencies, `PostToolUse` candidates phrased as gates, an `exit 2` gate on `PermissionRequest` that +is silently inert, and a candidate generator that reaches three of the 33 documented hook events. +Read [context/gotchas.md](context/gotchas.md) before any run whose verdicts a human will act on. diff --git a/plugins/claude-config/skills/audit-automation-gaps/context/gotchas.md b/plugins/claude-config/skills/audit-automation-gaps/context/gotchas.md new file mode 100644 index 0000000000..a32d8c8cb3 --- /dev/null +++ b/plugins/claude-config/skills/audit-automation-gaps/context/gotchas.md @@ -0,0 +1,90 @@ +# audit-automation-gaps: observed failure modes + +Every entry below is a failure this skill actually produced when it was run against a real +repository, not a hazard imagined for the list. Read before Phase 1 when the run will end in +verdicts a human acts on. + +## Contents + +- [A partial inventory anchors every later verdict](#a-partial-inventory-anchors-every-later-verdict) +- [A keyword count is a ceiling, never a frequency](#a-keyword-count-is-a-ceiling-never-a-frequency) +- [`PostToolUse` cannot block, so no `PostToolUse` candidate is a gate](#posttooluse-cannot-block-so-no-posttooluse-candidate-is-a-gate) +- [An `exit 2` gate on `PermissionRequest` is silently inert](#an-exit-2-gate-on-permissionrequest-is-silently-inert) +- [The candidate generator reaches a handful of the documented events](#the-candidate-generator-reaches-a-handful-of-the-documented-events) +- [Upstream records](#upstream-records) + +## A partial inventory anchors every later verdict + +Observed on 2026-09-13: the pre-computed inventory reported 3 hook scripts for a repository that +carried 143 of them, and every candidate generated afterwards was shaped by the small number. The +model did not re-derive the count; it reasoned from the figure it had been handed, and produced +"this repo has almost no hook automation" candidates against a repository that is dense with it. + +The failure class is the pre-computed number, not the particular bug. A count injected before +Phase 1 is an anchor, and a wrong anchor is not corrected by later reading because nothing in the +procedure asks for the count again. So: before generating candidates, re-derive any count the +verdicts will lean on with your own command, and where the re-derived number disagrees with the +injected one, say both in the report and use yours. A count and a wired mechanism are also +different facts. A hook script present on disk that no settings file or manifest registers is not +an active hook, and counting the two together overstates coverage in the opposite direction. + +## A keyword count is a ceiling, never a frequency + +Observed on 2026-09-13 against this marketplace at merged main `49912c63`: +`git log --oneline -E --grep='markdownlint|lint'` returns 1128 of 2202 commits, about 51 percent, +and `git log --oneline -E --grep='secret|gitleaks'` returns 209, about 9 percent. Reading the first +15 of those 209 shows dependency bumps, component syncs and CI pins. Not one is a leaked secret. + +Two gates read those numbers directly, so the error is not cosmetic. **Zero incidents** treats a +count as incidents that happened, and **YAGNI** treats a count as a frequency. A `--grep` count is +neither. It is the number of commits whose message mentions a word, which is an upper bound on the +incidents and usually a loose one. Cite a frequency only after reading a sample of the matching +commits and saying how many of them were the concern. + +The asymmetry is what makes this cheap rather than onerous. A ceiling that already sits below the +**YAGNI** threshold settles the gate on its own: if at most 4 percent of commits could be the +concern, the frequency is under 4 percent whatever the sample says, and no sample is needed. Only a +PASS, or an affirmative claim that the concern is frequent, has to pay for a sample. + +## `PostToolUse` cannot block, so no `PostToolUse` candidate is a gate + +A candidate phrased as "a `PostToolUse` hook that rejects the edit when the formatter finds a +problem" cannot exist. The tool has already run by the time the event fires, and exit code 2 there +shows the hook's stderr to Claude rather than preventing anything. Such a candidate is not a REJECT +at **Too slow** or **Already enforced**; it is malformed, and the right response is to re-state it +on a blocking event or to accept it as advisory before running it past any gate. + +## An `exit 2` gate on `PermissionRequest` is silently inert + +This is the same shape one rung worse, because it fails without any signal. `PermissionRequest` +does not honor exit code 2 at all: the permission flow proceeds unchanged and denial goes through a +JSON `decision` object instead. A hook written as `exit 2` there is not a weak gate or a slow gate. +It is no gate, it reports nothing, and the repository keeps a row in its enforcement inventory for a +mechanism that never fires. Any candidate whose mechanism is a per-event exit code gets its event's +exit-code semantics confirmed against the hooks reference in the Phase 2.2 mandatory batch, before +the verdict, not after. + +## The candidate generator reaches a handful of the documented events + +The hooks documentation names 33 events. This skill's own candidate guidance names three of them: +`context/gap-analysis.md` asks only about a `PostToolUse` formatter, and `context/hook-timing.md` +prices `PreToolUse`, `PostToolUse` and `PermissionRequest`. Everything else, session lifecycle, +compaction, subagent and task boundaries, file and config change, model switch, elicitation, is +absent from the questions the skill asks, so a genuine gap on one of those events cannot surface as +a candidate at all. That is a recall ceiling in the generator, and it is invisible in the output: a +clean bill of health is reported the same way whether the events were considered and dismissed or +never considered. When a run reports no gaps in the hooks category, say which events were actually +examined. + +## Upstream records + +Each row restates a volatile upstream specific, so it carries the four parts the upstream-drift +convention requires. Re-fetch the basis before acting on any row; the date is a ceiling on how +current the claim is, not authority. + +| Claim | Basis | As of | Recheck trigger | +|---|---|---|---| +| `PostToolUse` cannot block: the tool has already run, and exit code 2 shows the hook's stderr to Claude as a system reminder | [Claude Code hooks reference](https://code.claude.com/docs/en/hooks), exit-code-2 behavior per event | 2026-09-13 | A re-fetch finds the per-event exit-code table no longer matching this row | +| `PermissionRequest` does not honor exit code 2: "the permission flow proceeds unchanged. Deny through the `decision` object instead" | [Claude Code hooks reference](https://code.claude.com/docs/en/hooks), `PermissionRequest` exit-code behavior | 2026-09-13 | A re-fetch finds `PermissionRequest` honoring exit code 2, or the `decision` object renamed or removed | +| The hooks reference documents 33 hook events | [Claude Code hooks reference](https://code.claude.com/docs/en/hooks), the hook-events section headings, counted 2026-09-13 | 2026-09-13 | A re-fetch returns a different number of documented events | +| `/hooks` opens a read-only browser, and selecting a hook "shows its details: the event, matcher, type, source file, and command" | [Automate actions with hooks](https://code.claude.com/docs/en/hooks-guide), the hooks-browser step | 2026-09-13 | A re-fetch finds the browser no longer read-only, or showing a different detail set | diff --git a/plugins/claude-config/skills/audit-automation-gaps/evals/evals.json b/plugins/claude-config/skills/audit-automation-gaps/evals/evals.json index 22c0251024..80f8824b96 100644 --- a/plugins/claude-config/skills/audit-automation-gaps/evals/evals.json +++ b/plugins/claude-config/skills/audit-automation-gaps/evals/evals.json @@ -18,6 +18,18 @@ "prompt": "Our repo has grown to 50+ .NET projects and builds are getting slow. Someone suggested a build cache hook, a parallel test runner subagent, and a dependency graph skill. Deep-dive these — which add real value?", "expected_output": "Should research each recommendation against current .NET capabilities (incremental build, static-graph restore where already enabled). Should time actual build performance. Should check if dotnet test already runs in parallel. Should reject recommendations covered by existing MSBuild features and only pass genuinely novel automation.", "files": [] + }, + { + "id": 4, + "prompt": "Last week you audited our automation and I picked two items to build. Run --implement and pick up where we left off.", + "expected_output": "Should read the stored findings back via findings-state.sh read, passing the substituted ${CLAUDE_PLUGIN_DATA} as --plugin-data rather than assuming the variable is present in the shell, and should implement from the recovered plan rather than re-deriving verdicts from memory. Should implement only the candidates whose approved field is true; where no candidate is approved the stored run predates the human's selection, so it should ask which items to implement rather than treating every PASS as chosen, since a PASS is the skill's judgment and not the operator's consent. If the script exits 4, that means this project has no stored run, and the correct response is to say so and offer to audit first, never to implement from recollection or to read another project's artifact. Exit 5 is different from exit 4 and means stored state exists but cannot be trusted, which is a state to repair or re-audit rather than an absence.", + "files": [] + }, + { + "id": 5, + "prompt": "We already lint markdown in CI, but I want a PostToolUse hook that runs markdownlint on every edit too so I get told immediately. Is that worth adding?", + "expected_output": "Should NOT reject this on the Already enforced gate merely because CI covers it, and should NOT accept it merely because the hook is faster than CI, since no official documentation addresses CI-versus-hook placement. Should work the shift-left carve-out's three conjuncts explicitly: whether the concern clears the incident and YAGNI gates on a SAMPLED count rather than a raw git log --grep count, which is a ceiling and not a frequency; whether the proposed hook is advisory and non-blocking; and whether its measured cost fits the consuming repo's remaining hook-budget headroom. Should state that the budget is a hard gate, so a repo that documents a budget with no headroom left keeps the REJECT even when the first two conjuncts hold. Should name which budget source applied.", + "files": [] } ] } diff --git a/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.sh b/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.sh new file mode 100755 index 0000000000..e96fdeabb6 --- /dev/null +++ b/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.sh @@ -0,0 +1,829 @@ +#!/usr/bin/env bash +# findings-state.sh: persistence for audit-automation-gaps verdicts, so +# `--implement` in a LATER session can read back what a run decided. +# +# WHY THIS EXISTS. The skill takes an `--implement` flag and its Phase 4 acts on +# "user-selected items", but nothing persisted the verdict table, the evidence +# behind each verdict, or the implementation plans a PASS carries. An audit run +# and the implement run that acts on it are two sessions, so without a durable +# artifact `--implement` has nothing to read and the operator re-runs the whole +# audit to recover what was already decided. +# +# NAMING. `findings-state.sh`, not `findings.sh`: this is the state store for one +# component, the same role `audit-pass`'s `scripts/run-state.sh` fills for that +# skill, and it is the file this one is modelled on. `emit-findings.sh` in +# `audit-instructions` is a different job (composing a report body from scanner +# output) and deliberately resolves no home of its own. +# +# KEYING, WHICH IS NOT OPTIONAL. Everything is written under +# +# /audit-automation-gaps// +# +# where `` comes from `lib/state-key.sh` and nothing else. That is +# rule 1 of docs/conventions/plugin-data-report-keying/README.md, and the reason +# is the whole point of this file: `${CLAUDE_PLUGIN_DATA}` is keyed to the plugin +# identifier alone, so an unkeyed findings file is one file per MACHINE and a +# later `--implement` would be served another repository's approved items and its +# evidence. The key is derived by RUNNING the shared library (rule 1a), never by +# composing a path here and never by a second derivation of its own (rule 1's +# "do not mint a second scheme"). +# +# RETENTION. One file per run plus an appended history line, which is rule 2's +# third row: a same-day rerun must not erase the earlier verdict set, because the +# operator may have approved items from the earlier one. `latest` is a POINTER to +# the newest run id, not the artifact, so `read` with no `--run-id` still serves +# a real per-run file. A findings file is never overwritten: a colliding run id +# gets a `-2`, `-3`, ... suffix and the id actually written is reported back, the +# same non-overwrite naming `audit-instructions/scripts/emit-findings.sh` uses +# for its `--out`. +# +# HOW THAT PROMISE IS KEPT UNDER CONCURRENCY, which a check-then-write cannot. +# Testing `[[ -e "$target" ]]` and publishing afterwards leaves a window in which +# two writers pick the same name and the second `mv` destroys the first verdict +# set while both report success. So the suffix search IS the publish reservation: +# each candidate name is claimed with `set -C` (noclobber) plus a `>` redirect, +# which opens with O_EXCL, so exactly one writer can win a given name and the +# loser moves to the next suffix. The envelope is composed only after a name is +# won and lands on the reserved inode with `mv -f`. No lock file is taken: a +# stale lock would wedge every later run, whereas the history append is one short +# line through `>>` (a single atomic write below PIPE_BUF) and `latest` is +# last-writer-wins by definition. RESIDUAL: O_EXCL is atomic on local POSIX +# filesystems and on NFSv3+, not on NFSv2; a plugin data root on NFSv2 is outside +# what this guards, and is recorded rather than implied. +# +# A PUBLISH IS ALL OR NOTHING. The pointers a publish also owns (the history line +# and `latest`) are pre-flighted before anything is published, so the common +# damaged state refuses up front instead of half-landing. If the history append +# still fails, the findings file is rolled back rather than left as an orphan +# nothing points at. The one residue that cannot be rolled back, a `latest` that +# will not replace after the history line is committed, is REPORTED as an +# incomplete publish naming the run id, and `read`/`list` report orphan findings +# files as an incomplete publish rather than as an absence. +# +# RULE 3, NEVER SERVE WHAT YOU CANNOT ATTRIBUTE. `read` against a key with no +# artifact exits 4 saying so. It does not look in an unkeyed location, it does +# not adopt a legacy file, and it computes nothing from one. +# +# UNINSTALL FRAGILITY. This tree is the only durable copy of a run's verdicts. +# Uninstalling the plugin from its last scope deletes `${CLAUDE_PLUGIN_DATA}` +# unless `--keep-data` is passed, and that takes every project's findings with +# it. Say so to the operator before treating this as an archive. +# +# `--plugin-data` IS REQUIRED, AND THAT IS THE DESIGN. `${CLAUDE_PLUGIN_DATA}` is +# NOT in the Bash tool's environment: the placeholder substitutes in skill +# CONTENT, and it is exported to hook processes and MCP/LSP subprocesses, none of +# which the Bash tool is. Verified on this host: `printenv CLAUDE_PLUGIN_DATA` +# exits 1 with no output. So the skill passes the already-resolved path it can +# see, and an exported value is honored only where one genuinely exists. Absent +# both, every subcommand exits 2 naming the remedy rather than inventing a +# directory and writing a project's verdicts somewhere nobody will look. +# +# PORTABILITY. coreutils plus `jq` plus `git` and `sha256sum`/`shasum` through +# `lib/state-key.sh`. No GNU-only flags: no `readlink -f`, no in-place `sed`, no +# `date -d`, no `stat` format flags. `jq` is a hard requirement rather than a +# soft one, matching every other jq-using script in this plugin +# (`audit/scripts/*.sh`, `audit-permission-state/scripts/permission-state.sh`): a +# findings payload is a JSON document, and a half-checked document in an artifact +# `--implement` is the only reader of is worse than a refusal naming the tool. +# The requirement is checked by the two subcommands that touch a payload, `write` +# and `read`, and by neither `paths` nor `list`, which compute a path and serve +# the history file with `cat`. Checking it is not optional in `read`: it parses +# the stored envelope before serving it, so an absent `jq` is a failed command +# that a caller cannot tell from a document that did not parse, and reporting +# that as exit 5 tells an operator their verdicts were corrupted when the +# artifact is intact and the machine is short a tool. +# +# Usage: +# findings-state.sh paths --plugin-data [--root ] +# findings-state.sh write --plugin-data --findings [--run-id ] +# [--root ] +# findings-state.sh read --plugin-data [--run-id ] [--root ] +# findings-state.sh list --plugin-data [--root ] +# +# Exit codes: +# 0 the operation succeeded +# 2 usage error, rejected argument, or a missing prerequisite (no +# --plugin-data, no jq, no lib/state-key.sh) +# 3 `write` only: the findings payload is not well-formed JSON, is more than +# one JSON document, or does not carry the shape `--implement` reads back. +# Every problem is listed on stderr and NOTHING is written. +# 4 `read` only: no findings are persisted at this project's derived key. +# This is an answer, not a failure, and it is deliberately distinct from 2 +# so a caller can tell "nothing audited yet" from "you called me wrong". +# 5 the persisted state is not in a usable condition: a history file, latest +# pointer or findings file that exists but is not a readable regular file, +# an incomplete publish (findings files with no pointer, or a `latest` that +# could not be replaced), or a run id whose 99 suffixed names are all taken. +# Distinct from 4 because UNREADABLE IS NOT ABSENT: reporting a state this +# script cannot read as "nothing is persisted" is how an operator loses a +# verdict set they already approved items from. Distinct from 2 because the +# caller did nothing wrong. +set -uo pipefail + +PROG="findings-state.sh" + +# The component segment of rule 1's `//`. It is +# the skill's own directory name, so a second component writing under the same +# plugin data root cannot collide with this one. +COMPONENT="audit-automation-gaps" + +# Bumped when the on-disk envelope changes shape. `read` hands the whole envelope +# back with this field intact, so a later `--implement` can refuse an artifact it +# does not understand instead of misreading one. +SCHEMA_VERSION=1 + +# The vocabularies the payload is checked against. Categories are the skill's own +# `argument-hint` filter tokens; verdicts are its three Phase 2 outcomes. +CATEGORIES='["hooks","mcp","skills","subagents","scheduled"]' +VERDICTS='["PASS","CONDITIONAL","REJECT"]' + +EXIT_PAYLOAD=3 +EXIT_NO_ARTIFACT=4 +EXIT_STATE=5 + +# Staged files the cleanup trap removes. Globals rather than locals because the +# trap body runs after the function that created them has returned. +TMP_PAYLOAD="" +TMP_ENVELOPE="" +TMP_LATEST="" +# The findings name this run claimed with O_EXCL. Held for the same reason: the +# trap has to be able to release a name whose envelope never landed, or a failed +# run would burn a suffix and leave a zero-byte file behind. +RESERVED_TARGET="" + +# Set by parse_common_args and the resolvers below. +ARG_PLUGIN_DATA="" +ARG_ROOT="" +ARG_RUN_ID="" +ARG_FINDINGS="" +# "Was the flag given?" is tracked separately from its value. `--run-id ""` must +# be a rejected argument, not a silently omitted one: treating it as omitted +# would quietly serve the newest run to a caller that asked for a specific id. +ARG_RUN_ID_GIVEN=0 +ARG_FINDINGS_GIVEN=0 +PLUGIN_DATA="" +STATE_KEY="" +BASE_DIR="" +HISTORY_FILE="" +LATEST_FILE="" + +cleanup_tmp() { + if [[ -n "$TMP_PAYLOAD" ]]; then + rm -f "$TMP_PAYLOAD" + fi + if [[ -n "$TMP_ENVELOPE" ]]; then + rm -f "$TMP_ENVELOPE" + fi + if [[ -n "$TMP_LATEST" ]]; then + rm -f "$TMP_LATEST" + fi + # Only ever an EMPTY reservation: once the envelope has landed on it the name + # is cleared, so a published findings file is never removed from here. + if [[ -n "$RESERVED_TARGET" ]] && [[ ! -s "$RESERVED_TARGET" ]]; then + rm -f "$RESERVED_TARGET" + fi + return 0 +} +trap cleanup_tmp EXIT + +die() { + printf '%s: %s\n' "$PROG" "$1" >&2 + exit 2 +} + +# Exit 5: the persisted state is not usable. Separate from `die` because the +# caller did nothing wrong and, for `read` and `list`, separate from exit 4 +# because state this script cannot read is not state that is not there. +die_state() { + printf '%s: %s\n' "$PROG" "$1" >&2 + exit "$EXIT_STATE" +} + +# Refuse a damaged pointer BEFORE anything is published. `write` owns three +# artifacts (the findings file, the history line, the pointer) and a failure +# discovered after the first one has landed is the orphan this checks away: a +# findings file on disk that `read` and `list` then report as absent. +preflight_state_file() { + local path="$1" label="$2" + if [[ -e "$path" ]] && [[ ! -f "$path" ]]; then + die_state "$label exists but is not a regular file, so this run cannot be recorded and nothing was written: $path" + fi + if [[ -f "$path" ]] && [[ ! -w "$path" ]]; then + die_state "$label exists but is not writable, so this run cannot be recorded and nothing was written: $path" + fi +} + +# A publish that cannot be completed releases its reserved name rather than +# leaving a findings file nothing points at. +rollback_publish() { + if [[ -n "$RESERVED_TARGET" ]]; then + rm -f "$RESERVED_TARGET" + RESERVED_TARGET="" + fi + die_state "$1" +} + +usage() { + cat <<'EOF' +findings-state.sh: persist and read back audit-automation-gaps verdicts. + + findings-state.sh paths --plugin-data [--root ] + findings-state.sh write --plugin-data --findings [--run-id ] + [--root ] + findings-state.sh read --plugin-data [--run-id ] [--root ] + findings-state.sh list --plugin-data [--root ] + +--plugin-data is the resolved ${CLAUDE_PLUGIN_DATA} path. It is required because +that placeholder is not exported to the Bash tool; the skill substitutes it in +its own content and passes the result here. + +--root derives the state key for that directory instead of the current one. +--run-id defaults, on write, to a UTC timestamp; on read, to the newest run. +--findings is the payload document, or `-` to read it from stdin. + +paths prints plugin_data=, state_key=, dir=, history=, latest=. +write prints run_id=, state_key=, path=. Never overwrites a findings file. +read prints one run's envelope JSON. Exit 4 when this project has none. +list prints one JSON line per persisted run, oldest first, and nothing at all + when this project has none. + +Exit: 0 success; 2 usage or prerequisite; 3 rejected payload (write: not JSON, +more than one JSON document, or the wrong shape); 4 no artifact at this +project's key (read); 5 persisted state that exists but is not usable (a +history file, latest pointer or findings file that is not a readable regular +file, an incomplete publish, or 99 taken suffixes for one run id). 5 is not 4: +state this script cannot read is not state that is not there. +EOF +} + +now_iso() { date -u +%Y-%m-%dT%H:%M:%SZ; } +now_stamp() { date -u +%Y%m%dT%H%M%SZ; } + +require_jq() { + if ! command -v jq >/dev/null 2>&1; then + die "jq is required: the findings payload is a JSON document and this script will not half-check one" + fi +} + +# A run id becomes a FILENAME under the plugin's own tree. Accept only a plain +# segment: no separators, no leading dot. `lib/state-key.sh` validates its half +# of the same door (a remote URL that would become traversing directory +# components); the regex below is the other half, and it is the regex, not the +# `..` arm after it, that stops traversal: a segment with no `/` cannot climb. +# +# WHAT THE `..` ARM ACTUALLY GUARDS, since it is not traversal. A dot run is +# refused so the id stays a plain readable segment on disk and in the envelope, +# and so the check survives as the second half of the door if the regex is ever +# loosened to admit a separator. It is deliberately stricter than the +# `--plugin-data` rule below, which refuses only a real `..` PATH COMPONENT: a +# directory may legitimately be named `a..b`, a run id may not. +validate_run_id() { + local id="$1" + if [[ -z "$id" ]]; then + die "--run-id must not be empty" + fi + if [[ ! "$id" =~ ^[A-Za-z0-9][A-Za-z0-9_.-]*$ ]]; then + die "--run-id must be a plain path segment matching [A-Za-z0-9][A-Za-z0-9_.-]*: $id" + fi + case "$id" in + *..*) die "--run-id must not contain '..': $id" ;; + *) : ;; + esac +} + +require_absolute_path() { + local name="$1" value="$2" + case "$value" in + /* | ?:[\\/]*) : ;; + *) die "$name must be an absolute path: $value" ;; + esac +} + +# Sets PLUGIN_DATA. A GLOBAL RATHER THAN A PRINTED VALUE, deliberately: `die` +# must terminate the script, and a `$(...)` capture would confine that exit to a +# subshell while the caller carried on with an empty path. The same reasoning +# governs derive_state_key and resolve_locations below. +# +# An exported value is honored only where one genuinely exists (a hook context). +# The refusal names WHY the placeholder is unavailable, because "CLAUDE_PLUGIN_DATA +# is unset" sends a reader looking for an environment bug that is not there. +resolve_plugin_data() { + local value="$1" + if [[ -z "$value" ]]; then + value="${CLAUDE_PLUGIN_DATA:-}" + fi + if [[ -z "$value" ]]; then + die "--plugin-data is required: \${CLAUDE_PLUGIN_DATA} is not exported to the Bash tool, so pass the path substituted into the skill text" + fi + require_absolute_path "--plugin-data" "$value" + # A `..` PATH COMPONENT is the traversal, and it is the only thing refused + # here. A directory named `a..b` is a legal name and is accepted; the older + # substring test refused one, and the message it printed claimed a traversal + # the path did not contain. + case "$value" in + .. | ../* | ..[\\]* | *[/\\].. | *[/\\]..[/\\]*) + die "--plugin-data must not contain a '..' path component: $value" + ;; + *) : ;; + esac + PLUGIN_DATA="$value" +} + +# Sets STATE_KEY by RUNNING the shared library (rule 1a). Nothing here +# reimplements the scheme, and nothing composes a path from a placeholder. +derive_state_key() { + local root="$1" plugin_root state_key_lib + plugin_root="${CLAUDE_PLUGIN_ROOT:-$(cd "${BASH_SOURCE[0]%/*}/../../.." && pwd)}" + state_key_lib="$plugin_root/lib/state-key.sh" + if [[ ! -f "$state_key_lib" ]]; then + die "cannot find lib/state-key.sh at: $state_key_lib" + fi + if [[ -n "$root" ]]; then + if [[ ! -d "$root" ]]; then + die "--root is not a directory: $root" + fi + STATE_KEY=$(bash "$state_key_lib" --root "$root") + else + STATE_KEY=$(bash "$state_key_lib") + fi + if [[ -z "$STATE_KEY" ]]; then + die "lib/state-key.sh produced no state key" + fi +} + +# The physical repo root for the derived key, recorded in the envelope as an +# operator-facing breadcrumb. It is NOT the key, and nothing reads it back to +# decide attribution; the key does that. +derive_repo_root() { + local here="${1:-}" + if [[ -z "$here" ]]; then + here="." + fi + (cd "$here" 2>/dev/null && git rev-parse --show-toplevel 2>/dev/null | tr -d '\r') || true +} + +# Sets BASE_DIR, HISTORY_FILE, LATEST_FILE, STATE_KEY, PLUGIN_DATA. +resolve_locations() { + resolve_plugin_data "$1" + derive_state_key "$2" + BASE_DIR="$PLUGIN_DATA/$COMPONENT/$STATE_KEY" + HISTORY_FILE="$BASE_DIR/history.jsonl" + LATEST_FILE="$BASE_DIR/latest" +} + +# The jq program that judges a payload. It returns one line per problem and +# nothing at all for a conforming document, so the caller needs no exit-code +# contract from jq beyond "it parsed". +# +# WHAT IT REQUIRES, and why each field is here rather than left to the caller: +# `candidates` is the verdict table; `evidence` is what backed each verdict and +# is the half an operator cannot reconstruct from the table alone; `plan` is +# required on PASS and CONDITIONAL because that is precisely what `--implement` +# reads back. An empty `candidates` array is legal: the skill's own quality +# principles make a clean bill of health a valid outcome, and persisting that is +# a result, not an absence. +payload_problems_program() { + cat </dev/null 2>&1; then + printf '%s: the findings payload is not well-formed JSON\n' "$PROG" >&2 + jq . "$file" 2>&1 >/dev/null | head -5 >&2 + exit "$EXIT_PAYLOAD" + fi + # EXACTLY ONE DOCUMENT, because publication can only carry one. `jq` accepts a + # stream of concatenated documents and the check below would walk all of them, + # but the envelope is composed from the FIRST. A two-document stream whose + # second half holds the real verdicts would otherwise validate in full, report + # success, and persist only the empty first half: a confident success that + # stored none of the verdicts. Refusing is the honest half of the choice + # because merging would have to invent a rule for conflicting keys. + documents=$(jq -s 'length' "$file" 2>/dev/null) + if [[ "$documents" != "1" ]]; then + printf '%s: the findings payload must be exactly one JSON document, and this stream carries %s.\n' \ + "$PROG" "${documents:-an unreadable number of}" >&2 + printf '%s: only the first document would be persisted, so nothing is written. Merge them before writing.\n' \ + "$PROG" >&2 + exit "$EXIT_PAYLOAD" + fi + problems=$(jq -r "$(payload_problems_program)" "$file") + if [[ -n "$problems" ]]; then + printf '%s: the findings payload does not carry the shape --implement reads back:\n' "$PROG" >&2 + printf '%s\n' "$problems" >&2 + exit "$EXIT_PAYLOAD" + fi +} + +parse_common_args() { + local context="$1" + shift + ARG_PLUGIN_DATA="" + ARG_ROOT="" + ARG_RUN_ID="" + ARG_FINDINGS="" + ARG_RUN_ID_GIVEN=0 + ARG_FINDINGS_GIVEN=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --plugin-data) + [[ $# -ge 2 ]] || die "--plugin-data needs a path" + ARG_PLUGIN_DATA="$2" + shift 2 + ;; + --root) + [[ $# -ge 2 ]] || die "--root needs a path" + ARG_ROOT="$2" + shift 2 + ;; + --run-id) + [[ $# -ge 2 ]] || die "--run-id needs a value" + ARG_RUN_ID="$2" + ARG_RUN_ID_GIVEN=1 + shift 2 + ;; + --findings) + [[ $# -ge 2 ]] || die "--findings needs a path or -" + ARG_FINDINGS="$2" + ARG_FINDINGS_GIVEN=1 + shift 2 + ;; + *) die "unknown argument to $context: $1" ;; + esac + done +} + +cmd_paths() { + parse_common_args "paths" "$@" + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]] || [[ "$ARG_FINDINGS_GIVEN" -eq 1 ]]; then + die "paths takes only --plugin-data and --root" + fi + resolve_locations "$ARG_PLUGIN_DATA" "$ARG_ROOT" + printf 'plugin_data=%s\n' "$PLUGIN_DATA" + printf 'state_key=%s\n' "$STATE_KEY" + printf 'dir=%s\n' "$BASE_DIR" + printf 'history=%s\n' "$HISTORY_FILE" + printf 'latest=%s\n' "$LATEST_FILE" +} + +cmd_write() { + parse_common_args "write" "$@" + require_jq + if [[ "$ARG_FINDINGS_GIVEN" -eq 0 ]]; then + die "--findings is required: pass the payload document, or - to read it from stdin" + fi + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]]; then + validate_run_id "$ARG_RUN_ID" + fi + resolve_locations "$ARG_PLUGIN_DATA" "$ARG_ROOT" + + # WRITE IS THE ONLY SUBCOMMAND THAT CREATES ANYTHING, so containment is + # established here. Every path segment below the plugin data root is already + # constrained: COMPONENT is a literal, the state key is validated by + # lib/state-key.sh (which hashes any identity that is not a plain lowercase + # segment path), and the run id is validated above. So the directory is + # derived, never accepted from a caller, and there is no caller-supplied run + # directory to contain. + # + # DISCLOSED RESIDUAL: a symlink planted INSIDE the plugin's own data directory + # could still redirect the write, and this does not resolve physical paths to + # catch that. An attacker with write access there can write the findings file + # directly, so the guard would be moot; recorded rather than implied, because + # an unstated limit reads as no limit. + mkdir -p "$BASE_DIR" || die "cannot create the findings directory: $BASE_DIR" + + TMP_PAYLOAD="$BASE_DIR/.payload.$$" + TMP_ENVELOPE="$BASE_DIR/.envelope.$$" + if [[ "$ARG_FINDINGS" == "-" ]]; then + cat >"$TMP_PAYLOAD" || die "cannot stage the payload read from stdin" + else + if [[ ! -f "$ARG_FINDINGS" ]]; then + die "--findings is not a file: $ARG_FINDINGS" + fi + cp "$ARG_FINDINGS" "$TMP_PAYLOAD" || die "cannot stage the payload: $ARG_FINDINGS" + fi + + validate_payload "$TMP_PAYLOAD" + + local run_id + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]]; then + run_id="$ARG_RUN_ID" + else + run_id=$(now_stamp) + fi + + preflight_state_file "$HISTORY_FILE" "the history file" + preflight_state_file "$LATEST_FILE" "the latest pointer" + + # NEVER OVERWRITE, WHICH MEANS NEVER RACE. Two runs in the same second, or a + # caller reusing an id, get a suffix instead of silently replacing a verdict + # set the operator may have already approved items from. The id actually + # written is reported back so the caller records the one on disk rather than + # the one it asked for. + # + # The search and the publish are ONE atomic reservation per run: each candidate + # name is claimed by creating it under `set -C`, which is an O_EXCL open, so + # two concurrent writers cannot both believe they won the same name. A + # check-then-`mv` would leave a window between them in which the second writer + # destroys the first verdict set and both report success. + local target="" candidate="" suffix=1 + while :; do + if [[ "$suffix" -eq 1 ]]; then + candidate="$run_id" + else + candidate="$run_id-$suffix" + fi + target="$BASE_DIR/findings-$candidate.json" + if (set -C && : >"$target") 2>/dev/null; then + break + fi + # The claim can fail for a reason that is not "somebody holds this name", and + # walking 99 suffixes to report a cap that is not the problem would bury it. + if [[ ! -e "$target" ]]; then + die "cannot reserve the findings file: $target" + fi + suffix=$((suffix + 1)) + if [[ "$suffix" -gt 99 ]]; then + die_state "all 99 suffixed findings names are taken for run id: $run_id" + fi + done + run_id="$candidate" + RESERVED_TARGET="$target" + + local written_at repo_root + written_at=$(now_iso) + repo_root=$(derive_repo_root "$ARG_ROOT") + + jq -n \ + --slurpfile payload "$TMP_PAYLOAD" \ + --argjson schema "$SCHEMA_VERSION" \ + --argjson verds "$VERDICTS" \ + --arg run_id "$run_id" \ + --arg written_at "$written_at" \ + --arg state_key "$STATE_KEY" \ + --arg repo_root "$repo_root" \ + --arg file "findings-$run_id.json" ' + ($payload[0]) as $p + | ($p.candidates) as $c + | { + schema_version: $schema, + run_id: $run_id, + written_at: $written_at, + state_key: $state_key, + repo_root: $repo_root, + file: $file, + counts: ( {total: ($c | length)} + + ( $verds + | map(. as $v + | {key: ($v | ascii_downcase), + value: ([$c[] | select(.verdict == $v)] | length)}) + | from_entries ) ), + findings: $p + }' >"$TMP_ENVELOPE" || die "cannot compose the findings envelope" + + # Atomic publish onto the name this run reserved: a reader never observes a + # half-written findings file, and a crash mid-write leaves the reservation + # rather than a truncated JSON document `--implement` would fail to parse. + mv -f "$TMP_ENVELOPE" "$target" || rollback_publish "cannot publish the findings file: $target" + TMP_ENVELOPE="" + + # The history line is derived FROM the published envelope, so the two can never + # disagree about counts or timestamps. A failure here rolls the findings file + # back out rather than leaving one the history does not mention. + local line + line=$(jq -c '{schema_version, run_id, written_at, state_key, file, counts}' "$target") || + rollback_publish "cannot compose the history line, so the findings file was rolled back: $target" + printf '%s\n' "$line" >>"$HISTORY_FILE" || + rollback_publish "cannot append to the history file, so the findings file was rolled back: $HISTORY_FILE" + # Committed from here: the run is on disk and in the history, so a later + # failure is REPORTED rather than rolled back over a record that now exists. + RESERVED_TARGET="" + + # `latest` is a pointer, not the artifact. Written after the findings file, so + # a failure never leaves it naming a run that does not exist, and REPLACED + # rather than copied into: `mv` swaps the name atomically, where `cp` would + # write through whatever the name already is and let a reader see a half + # pointer. + TMP_LATEST="$LATEST_FILE.$$" + printf '%s\n' "$run_id" >"$TMP_LATEST" || + die_state "run $run_id is published and recorded, but the latest pointer could not be staged. Read it with --run-id $run_id: $LATEST_FILE" + mv -f "$TMP_LATEST" "$LATEST_FILE" || + die_state "run $run_id is published and recorded, but the latest pointer could not be replaced. Read it with --run-id $run_id: $LATEST_FILE" + TMP_LATEST="" + + printf 'run_id=%s\n' "$run_id" + printf 'state_key=%s\n' "$STATE_KEY" + printf 'path=%s\n' "$target" +} + +no_artifact() { + printf '%s: no findings are persisted for this project (state key %s) under %s\n' \ + "$PROG" "$STATE_KEY" "$BASE_DIR" >&2 + printf '%s: run the audit first. Nothing outside this key is read: an artifact with no project segment cannot be attributed to this one.\n' \ + "$PROG" >&2 + exit "$EXIT_NO_ARTIFACT" +} + +# Findings files with no pointer at them are an INCOMPLETE PUBLISH, and saying +# "nothing is persisted" over the top of them is the same lie in a different +# costume: the verdicts are right there on disk. Called only on the paths that +# were about to report absence, so a healthy tree never pays for it. +refuse_orphans() { + local missing="$1" file count=0 + for file in "$BASE_DIR"/findings-*.json; do + if [[ -e "$file" ]]; then + count=$((count + 1)) + fi + done + if [[ "$count" -eq 0 ]]; then + return 0 + fi + printf '%s: %s is missing, but %d findings file(s) exist under %s. This is an INCOMPLETE PUBLISH, not an absence:\n' \ + "$PROG" "$missing" "$count" "$BASE_DIR" >&2 + for file in "$BASE_DIR"/findings-*.json; do + if [[ -e "$file" ]]; then + printf '%s: %s\n' "$PROG" "${file##*/}" >&2 + fi + done + printf '%s: read one directly with --run-id , taking the id from the file name.\n' "$PROG" >&2 + exit "$EXIT_STATE" +} + +# `-f` alone answers "can this be served?" with a yes for a readable regular +# file and a no for everything else, and the two nos are not the same answer. +require_readable_file() { + local path="$1" label="$2" + if [[ -e "$path" ]] && [[ ! -f "$path" ]]; then + die_state "$label exists but is not a regular file, so it cannot be read (that is not the same as having none): $path" + fi + if [[ -f "$path" ]] && [[ ! -r "$path" ]]; then + die_state "$label exists but cannot be read (that is not the same as having none): $path" + fi +} + +cmd_read() { + parse_common_args "read" "$@" + # THE SAME PREREQUISITE CHECK `write` USES, and for a sharper reason here. + # `read` parses the stored envelope with `jq` before serving it, and a machine + # without `jq` fails that command with exit 127. Without this gate the failure + # is indistinguishable from a document that did not parse, so a perfectly + # valid persisted run is reported as damaged state under exit 5 and the + # operator is told an external writer corrupted their verdicts. A missing tool + # is a missing prerequisite: it exits 2 and names `jq`. + require_jq + if [[ "$ARG_FINDINGS_GIVEN" -eq 1 ]]; then + die "read does not take --findings" + fi + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]]; then + validate_run_id "$ARG_RUN_ID" + fi + resolve_locations "$ARG_PLUGIN_DATA" "$ARG_ROOT" + + if [[ -e "$BASE_DIR" ]] && [[ ! -d "$BASE_DIR" ]]; then + die_state "the findings directory exists but is not a directory, so nothing here can be read: $BASE_DIR" + fi + if [[ ! -d "$BASE_DIR" ]]; then + no_artifact + fi + + local run_id="" + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]]; then + run_id="$ARG_RUN_ID" + fi + if [[ -z "$run_id" ]]; then + require_readable_file "$LATEST_FILE" "the latest pointer" + if [[ ! -f "$LATEST_FILE" ]]; then + refuse_orphans "the latest pointer" + no_artifact + fi + run_id=$(tr -d '\r' <"$LATEST_FILE" | head -1) + if [[ -z "$run_id" ]]; then + die_state "the latest pointer is empty, so no run can be served by default: $LATEST_FILE" + fi + # VALIDATED EVEN THOUGH THIS SCRIPT WROTE IT. The pointer is a file on disk + # under a directory anything with write access can reach, so a poisoned + # `latest` is a caller-supplied run id by another route, and the name below + # is built from it. + validate_run_id "$run_id" + fi + + local target="$BASE_DIR/findings-$run_id.json" + require_readable_file "$target" "the findings file for run id $run_id" + if [[ ! -f "$target" ]]; then + printf '%s: no findings file for run id %s at: %s\n' "$PROG" "$run_id" "$target" >&2 + exit "$EXIT_NO_ARTIFACT" + fi + if [[ ! -s "$target" ]]; then + die_state "the findings file for run id $run_id is empty, so there is nothing to serve (a write may be in flight, or it was truncated): $target" + fi + # PARSED BEFORE IT IS SERVED, even though this script wrote it. Publication is + # staged and renamed, so this script cannot leave a corrupt file behind, but + # the tree is an ordinary directory that a foreign writer or a stray editor + # can reach. Serving bytes that do not parse would hand the caller something + # unusable under exit 0, which is the opposite of the promise exit 5 makes: + # state that exists but cannot be trusted is reported, never returned. + if ! jq -e 'type == "object"' "$target" >/dev/null 2>&1; then + die_state "the findings file for run id $run_id is not a JSON object, so it cannot be trusted (something outside this script wrote it): $target" + fi + cat "$target" +} + +cmd_list() { + parse_common_args "list" "$@" + if [[ "$ARG_RUN_ID_GIVEN" -eq 1 ]] || [[ "$ARG_FINDINGS_GIVEN" -eq 1 ]]; then + die "list takes only --plugin-data and --root" + fi + resolve_locations "$ARG_PLUGIN_DATA" "$ARG_ROOT" + + if [[ -e "$BASE_DIR" ]] && [[ ! -d "$BASE_DIR" ]]; then + die_state "the findings directory exists but is not a directory, so nothing here can be listed: $BASE_DIR" + fi + # A history file that exists but cannot be read is reported as itself, not as + # an empty listing: "no persisted runs" would send a caller off to re-run the + # whole audit over a verdict set that is sitting right there. + require_readable_file "$HISTORY_FILE" "the history file" + + # An empty listing is an ANSWER, not a failure: "this project has no persisted + # runs" is exactly what a caller deciding whether to audit or to implement + # needs, so stdout is empty and the exit code stays 0. Only `read`, which was + # asked for a specific artifact, treats absence as exit 4. + if [[ ! -f "$HISTORY_FILE" ]]; then + refuse_orphans "the history file" + printf '%s: no persisted runs for this project (state key %s)\n' "$PROG" "$STATE_KEY" >&2 + return 0 + fi + cat "$HISTORY_FILE" +} + +main() { + if [[ $# -lt 1 ]]; then + usage >&2 + exit 2 + fi + local command="$1" + shift + case "$command" in + -h | --help) + usage + exit 0 + ;; + paths) cmd_paths "$@" ;; + write) cmd_write "$@" ;; + read) cmd_read "$@" ;; + list) cmd_list "$@" ;; + *) + printf '%s: unknown command: %s\n' "$PROG" "$command" >&2 + usage >&2 + exit 2 + ;; + esac +} + +main "$@" diff --git a/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.test.sh b/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.test.sh new file mode 100755 index 0000000000..cc67cd4061 --- /dev/null +++ b/plugins/claude-config/skills/audit-automation-gaps/scripts/findings-state.test.sh @@ -0,0 +1,864 @@ +#!/usr/bin/env bash +# Regression tests for findings-state.sh (self-contained, ships with the plugin). +# +# THE CASE THIS SUITE EXISTS FOR is case 8: two repositories writing the same run +# id into one plugin data root must not collide. That is rule 1 of +# docs/conventions/plugin-data-report-keying/README.md, and an unkeyed +# implementation would serve one project's approved items and evidence to +# another. It is demonstrated with two real git fixtures rather than asserted. +# +# Three cases are NEGATIVE in the sense this repo means it: they mutate a copy of +# the script to delete exactly one check and assert the mutated copy reaches the +# outcome the real one refuses. A test that would still pass with the check +# deleted proves nothing. +# +# File-scoped: `read` is one of findings-state.sh's SUBCOMMAND NAMES, so every +# `run read ...` here is an argument word, never the bash builtin. This suite +# calls the builtin nowhere. +# shellcheck disable=SC2162 +set -uo pipefail + +# Fixture git isolation: an inherited GIT_DIR/GIT_WORK_TREE/GIT_CONFIG would +# redirect `git init` / `git config` into the caller's repository. +unset GIT_DIR GIT_WORK_TREE GIT_CONFIG + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +SCRIPT="$SCRIPT_DIR/findings-state.sh" +PLUGIN_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)" + +TEST_TMPDIR="$(mktemp -d)" +trap 'rm -rf "$TEST_TMPDIR"' EXIT + +FAILED=0 +CASE_NUM=0 +SKIPPED=0 + +pass() { + CASE_NUM=$((CASE_NUM + 1)) + printf 'PASS: %s\n' "$1" +} +# A host-capability skip is reported as itself and never through pass(), so a +# proof this host could not run can never be read off the summary as one that +# did. +skip() { + SKIPPED=$((SKIPPED + 1)) + printf 'SKIP (host: %s): %s\n' "$2" "$1" +} +# Under MSYS without winsymlinks, `ln -s` COPIES the target instead of linking +# it. Two cases below need a real symlink: one plants a dangling link to make an +# append fail, the other plants a link to prove the pointer is replaced rather +# than written through. Probe the round trip rather than the OS name. +host_makes_symlinks() { + local d rc=1 + d="$(mktemp -d)" + printf 'x\n' >"$d/target" + if ln -s target "$d/link" 2>/dev/null && + [[ -L "$d/link" ]] && [[ "$(readlink "$d/link" 2>/dev/null)" == "target" ]]; then + rc=0 + fi + rm -rf "$d" + return "$rc" +} +fail() { + CASE_NUM=$((CASE_NUM + 1)) + FAILED=$((FAILED + 1)) + printf 'FAIL: %s\n detail: %s\n' "$1" "$2" >&2 +} +assert_eq() { + if [[ "$2" == "$3" ]]; then pass "$1"; else fail "$1" "expected: $2, actual: $3"; fi +} +assert_exit() { + if [[ "$2" == "$3" ]]; then pass "$1"; else fail "$1" "expected exit $2, got $3"; fi +} +assert_contains() { + case "$2" in + *"$3"*) pass "$1" ;; + *) fail "$1" "expected to contain: $3" ;; + esac +} +assert_not_contains() { + case "$2" in + *"$3"*) fail "$1" "unexpected substring: $3" ;; + *) pass "$1" ;; + esac +} +assert_file() { + if [[ -f "$2" ]]; then pass "$1"; else fail "$1" "no such file: $2"; fi +} + +if ! command -v jq >/dev/null 2>&1; then + echo "SKIP: jq not installed" >&2 + exit 0 +fi + +# Every invocation runs with CLAUDE_PLUGIN_ROOT pinned, so a copy of the script +# placed outside the plugin tree still resolves lib/state-key.sh. +run() { + CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$SCRIPT" "$@" +} +run_copy() { + local copy="$1" + shift + CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$copy" "$@" +} + +DATA="$TEST_TMPDIR/plugin-data" +mkdir -p "$DATA" + +REPO_A="$TEST_TMPDIR/alpha" +REPO_B="$TEST_TMPDIR/beta" +mkdir -p "$REPO_A" "$REPO_B" +git -C "$REPO_A" init --quiet >/dev/null 2>&1 +git -C "$REPO_A" remote add origin https://github.com/example/alpha.git >/dev/null 2>&1 +git -C "$REPO_B" init --quiet >/dev/null 2>&1 +git -C "$REPO_B" remote add origin https://github.com/example/beta.git >/dev/null 2>&1 + +GOOD="$TEST_TMPDIR/good.json" +cat >"$GOOD" <<'EOF' +{ + "candidates": [ + { + "id": "1", + "candidate": "pre-commit shfmt hook", + "category": "hooks", + "verdict": "PASS", + "evidence": ["shfmt measured at 0.11s", "15% of commits touch shell"], + "plan": {"files": [".claude/hooks/shfmt.sh"], "effort": "S"}, + "approved": true + }, + { + "id": "2", + "candidate": "database MCP server", + "category": "mcp", + "verdict": "REJECT", + "evidence": ["Premature: no database exists"] + } + ], + "maturity": "One genuine gap." +} +EOF + +EMPTY="$TEST_TMPDIR/empty.json" +cat >"$EMPTY" <<'EOF' +{"candidates": [], "maturity": "A clean bill of health."} +EOF + +# --- Case 1: --help --------------------------------------------------------- + +rc=0 +OUT=$(run --help 2>&1) || rc=$? +assert_exit "--help exits 0" 0 "$rc" +assert_contains "--help names every subcommand" "$OUT" "findings-state.sh list" + +# --- Case 2: paths, and the keyed shape it reports -------------------------- + +rc=0 +OUT=$(run paths --plugin-data "$DATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "paths exits 0 on a git target" 0 "$rc" +assert_contains "paths echoes the plugin data dir it was given" "$OUT" "plugin_data=$DATA" +assert_contains "paths derives a state key through lib/state-key.sh" "$OUT" \ + "state_key=github.com/example/alpha/" +assert_contains "the tree is //" "$OUT" \ + "dir=$DATA/audit-automation-gaps/github.com/example/alpha/" + +rc=0 +OUT=$(run paths --plugin-data "$DATA" --root "$TEST_TMPDIR" 2>&1) || rc=$? +assert_exit "a non-repository directory still keys" 0 "$rc" +assert_contains "the non-repository rung is used" "$OUT" "state_key=nonrepo/" + +# --- Case 3: the missing-plugin-data failure must be LOUD ------------------- +# +# ${CLAUDE_PLUGIN_DATA} is not exported to the Bash tool, so the resolved path +# arrives as an argument. Guessing a directory here would write one project's +# verdicts somewhere nothing reads them back. + +rc=0 +OUT=$(env -u CLAUDE_PLUGIN_DATA CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" \ + bash "$SCRIPT" paths --root "$REPO_A" 2>&1) || rc=$? +assert_exit "paths exits 2 with no --plugin-data and no exported placeholder" 2 "$rc" +assert_contains "the refusal names why the placeholder is unavailable in Bash" "$OUT" \ + "not exported to the Bash tool" +assert_contains "the refusal names the remedy" "$OUT" "pass the path substituted into the skill text" + +rc=0 +OUT=$(env -u CLAUDE_PLUGIN_DATA CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" \ + bash "$SCRIPT" write --root "$REPO_A" --findings "$GOOD" 2>&1) || rc=$? +assert_exit "write exits 2 with no --plugin-data" 2 "$rc" +assert_contains "write's refusal is the same loud one" "$OUT" "not exported to the Bash tool" + +rc=0 +OUT=$(env -u CLAUDE_PLUGIN_DATA CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" \ + bash "$SCRIPT" read --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 2 with no --plugin-data" 2 "$rc" + +rc=0 +OUT=$(env -u CLAUDE_PLUGIN_DATA CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" \ + bash "$SCRIPT" list --root "$REPO_A" 2>&1) || rc=$? +assert_exit "list exits 2 with no --plugin-data" 2 "$rc" + +# An exported value is still honored where one genuinely exists (a hook context). +rc=0 +OUT=$(CLAUDE_PLUGIN_DATA="$DATA" CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" \ + bash "$SCRIPT" paths --root "$REPO_A" 2>&1) || rc=$? +assert_exit "an exported CLAUDE_PLUGIN_DATA is honored where it exists" 0 "$rc" +assert_contains "the exported value is the one used" "$OUT" "plugin_data=$DATA" + +rc=0 +OUT=$(run paths --plugin-data "relative/dir" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "a relative --plugin-data is refused" 2 "$rc" + +rc=0 +OUT=$(run paths --plugin-data "/tmp/../etc" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "a --plugin-data containing '..' is refused" 2 "$rc" + +# --- Case 4: run-id validation, the path-safety control --------------------- +# +# A run id becomes a filename under the plugin's own tree. lib/state-key.sh +# documents the traversal it defends its half against; this is the other half. + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id "a..b" 2>&1) || rc=$? +assert_exit "a run id containing '..' is refused" 2 "$rc" +assert_contains "the refusal names the traversal" "$OUT" "must not contain '..'" + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id "/etc/passwd" 2>&1) || rc=$? +assert_exit "a run id that is not a plain segment is refused" 2 "$rc" + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id "" 2>&1) || rc=$? +assert_exit "an empty run id is refused" 2 "$rc" + +# A DIRECTORY may legitimately be named `a..b`; a run id may not. The two rules +# differ on purpose, and each message says which rule it is enforcing. +mkdir -p "$TEST_TMPDIR/a..b/data" +rc=0 +OUT=$(run paths --plugin-data "$TEST_TMPDIR/a..b/data" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "a real directory named a..b keys like any other" 0 "$rc" +assert_contains "the a..b directory is the one used" "$OUT" "plugin_data=$TEST_TMPDIR/a..b/data" + +rc=0 +OUT=$(run paths --plugin-data "$TEST_TMPDIR/../etc" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "a real '..' path component is still refused" 2 "$rc" +assert_contains "the refusal names the component, not a substring" "$OUT" \ + "must not contain a '..' path component" + +# 4a: NEGATIVE. Delete the '..' arm and the dot-run id must reach the filename it +# would become. Asserting only that a message disappears would also pass if the +# id were refused a line later for some unrelated reason; the constructed path in +# the output is what shows the id actually got through to path construction. +DOTS_DATA="$TEST_TMPDIR/dots-data" +mkdir -p "$DOTS_DATA" +DOTS_KEY=$(run paths --plugin-data "$DOTS_DATA" --root "$REPO_A" | sed -n 's/^state_key=//p') +mkdir -p "$DOTS_DATA/audit-automation-gaps/$DOTS_KEY" +BROKEN_DOTS="$TEST_TMPDIR/broken-dots.sh" +sed "/must not contain '\.\.': \$id/s/.*/ *) : ;;/" "$SCRIPT" >"$BROKEN_DOTS" +if grep -qF "must not contain '..': \$id" "$BROKEN_DOTS"; then + fail "negative traversal case could not be constructed" \ + "the sed target no longer matches findings-state.sh, so the '..' rejection is UNVERIFIED by this run" +else + rc=0 + broken_out=$(run_copy "$BROKEN_DOTS" read --plugin-data "$DOTS_DATA" --root "$REPO_A" \ + --run-id "a..b" 2>&1) || rc=$? + assert_not_contains "with the '..' arm deleted the dot run is no longer refused" \ + "$broken_out" "must not contain" + assert_exit "with the arm deleted the id reaches the artifact lookup instead" 4 "$rc" + assert_contains "and it reaches it as a FILENAME built from the dot run" \ + "$broken_out" "findings-a..b.json" +fi + +# --- Case 5: read and list before anything is written ----------------------- +# +# Rule 3: an artifact that cannot be attributed to this project is never served, +# and absence is reported rather than papered over with an unkeyed fallback. + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 4 when this project has no artifact" 4 "$rc" +assert_contains "the empty read names the state key it looked under" "$OUT" \ + "github.com/example/alpha/" +assert_contains "the empty read states that nothing outside the key is read" "$OUT" \ + "Nothing outside this key is read" + +rc=0 +OUT=$(run list --plugin-data "$DATA" --root "$REPO_A" 2>/dev/null) || rc=$? +assert_exit "list exits 0 when this project has no artifact" 0 "$rc" +assert_eq "list prints nothing on stdout when there is nothing" "" "$OUT" + +# --- Case 6: write, then read it back --------------------------------------- + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --run-id run-0001 --findings "$GOOD" 2>&1) || rc=$? +assert_exit "write exits 0 on a conforming payload" 0 "$rc" +assert_contains "write reports the run id it wrote" "$OUT" "run_id=run-0001" +assert_contains "write reports the state key" "$OUT" "state_key=github.com/example/alpha/" +A_KEY=$(run paths --plugin-data "$DATA" --root "$REPO_A" | sed -n 's/^state_key=//p') +A_DIR="$DATA/audit-automation-gaps/$A_KEY" +assert_file "the findings file lands under the keyed tree" "$A_DIR/findings-run-0001.json" +assert_file "the history file is created" "$A_DIR/history.jsonl" +assert_file "the latest pointer is created" "$A_DIR/latest" + +rc=0 +ENVELOPE=$(run read --plugin-data "$DATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 0 once a run is persisted" 0 "$rc" +assert_eq "read serves the newest run without a --run-id" "run-0001" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.run_id')" +assert_eq "the envelope carries a schema version" "1" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.schema_version')" +assert_eq "the envelope records the state key it was written under" "$A_KEY" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.state_key')" +assert_eq "counts are derived from the verdict column" "2 1 0 1" \ + "$(printf '%s' "$ENVELOPE" | jq -r '[.counts.total, .counts.pass, .counts.conditional, .counts.reject] | join(" ")')" + +# The three payload parts this store exists to persist: the verdict table, the +# evidence behind each verdict, and the implementation plans. +assert_eq "the verdict table survives the round trip" "PASS REJECT" \ + "$(printf '%s' "$ENVELOPE" | jq -r '[.findings.candidates[].verdict] | join(" ")')" +assert_eq "the evidence survives the round trip" "2" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.findings.candidates[0].evidence | length')" +assert_eq "the implementation plan survives the round trip" ".claude/hooks/shfmt.sh" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.findings.candidates[0].plan.files[0]')" +assert_eq "an approval flag set by the operator survives the round trip" "true" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.findings.candidates[0].approved')" + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id run-0001 2>&1) || rc=$? +assert_exit "read accepts an explicit --run-id" 0 "$rc" + +rc=0 +OUT=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id no-such-run 2>&1) || rc=$? +assert_exit "read exits 4 for a run id with no file" 4 "$rc" + +# --- Case 7: stdin, the default run id, and non-overwrite ------------------- + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --findings - <"$EMPTY" 2>&1) || rc=$? +assert_exit "write accepts the payload on stdin" 0 "$rc" +assert_contains "the default run id is a UTC timestamp" "$OUT" "run_id=20" + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --run-id run-0001 --findings "$EMPTY" 2>&1) || rc=$? +assert_exit "a colliding run id still writes" 0 "$rc" +assert_contains "a colliding run id is suffixed rather than overwritten" "$OUT" "run_id=run-0001-2" +assert_eq "the first run's verdicts are untouched" "2" \ + "$(jq -r '.counts.total' "$A_DIR/findings-run-0001.json")" + +rc=0 +OUT=$(run list --plugin-data "$DATA" --root "$REPO_A" 2>/dev/null) || rc=$? +assert_exit "list exits 0" 0 "$rc" +assert_eq "list prints one JSON line per persisted run" "3" "$(printf '%s\n' "$OUT" | wc -l | tr -d ' ')" +assert_eq "every history line parses as JSON" "3" \ + "$(printf '%s\n' "$OUT" | jq -s 'length')" +assert_eq "the newest run is last" "run-0001-2" \ + "$(printf '%s\n' "$OUT" | jq -s -r '.[-1].run_id')" + +# --- Case 8: THE COLLISION PROOF -------------------------------------------- +# +# Two repositories, one plugin data root, the SAME run id, different payloads. +# Under an unkeyed layout the second write replaces the first and the first +# project is then served the second's approved items. + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_B" --run-id run-0001 --findings "$EMPTY" 2>&1) || rc=$? +assert_exit "repo B writes the same run id without error" 0 "$rc" +B_KEY=$(run paths --plugin-data "$DATA" --root "$REPO_B" | sed -n 's/^state_key=//p') +if [[ "$A_KEY" != "$B_KEY" ]]; then + pass "two repositories derive different state keys" +else + fail "two repositories derive different state keys" "both derived: $A_KEY" +fi +assert_file "repo B's findings file is a separate path" \ + "$DATA/audit-automation-gaps/$B_KEY/findings-run-0001.json" +assert_eq "repo A still serves ITS OWN verdict count" "2" \ + "$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id run-0001 | jq -r '.counts.total')" +assert_eq "repo B still serves ITS OWN verdict count" "0" \ + "$(run read --plugin-data "$DATA" --root "$REPO_B" --run-id run-0001 | jq -r '.counts.total')" +assert_eq "repo B's history holds only its own run" "1" \ + "$(run list --plugin-data "$DATA" --root "$REPO_B" 2>/dev/null | wc -l | tr -d ' ')" + +# 8a: NEGATIVE. Drop the state key out of the path and the two collide. +BROKEN_KEY="$TEST_TMPDIR/broken-key.sh" +# shellcheck disable=SC2016 # sed program text: the $ are literal characters in the script being mutated +sed 's|BASE_DIR="\$PLUGIN_DATA/\$COMPONENT/\$STATE_KEY"|BASE_DIR="$PLUGIN_DATA/$COMPONENT"|' \ + "$SCRIPT" >"$BROKEN_KEY" +# shellcheck disable=SC2016 # grep pattern text: same literal $ as the sed above +if grep -q 'BASE_DIR="\$PLUGIN_DATA/\$COMPONENT/\$STATE_KEY"' "$BROKEN_KEY"; then + fail "negative keying case could not be constructed" \ + "the sed target no longer matches findings-state.sh, so the keying is UNVERIFIED by this run" +else + UNKEYED="$TEST_TMPDIR/unkeyed-data" + mkdir -p "$UNKEYED" + run_copy "$BROKEN_KEY" write --plugin-data "$UNKEYED" --root "$REPO_A" \ + --run-id shared --findings "$GOOD" >/dev/null 2>&1 + run_copy "$BROKEN_KEY" write --plugin-data "$UNKEYED" --root "$REPO_B" \ + --run-id shared --findings "$EMPTY" >/dev/null 2>&1 + # Repo B's write landed beside repo A's under one unkeyed directory, so repo A + # is now served an artifact it did not produce: the `latest` pointer names + # repo B's run. + broken_latest=$(run_copy "$BROKEN_KEY" read --plugin-data "$UNKEYED" --root "$REPO_A" | + jq -r '.state_key') + assert_eq "with the state key deleted, repo A is served repo B's artifact" \ + "$B_KEY" "$broken_latest" +fi + +# --- Case 9: payload rejection ---------------------------------------------- + +BAD="$TEST_TMPDIR/bad.json" +cat >"$BAD" <<'EOF' +{ + "candidates": [ + {"id": "1", "candidate": "x", "category": "githooks", "verdict": "PASS", "evidence": ["e"]}, + {"id": "2", "category": "hooks", "verdict": "MAYBE", "evidence": "not an array"} + ] +} +EOF + +before=$(find "$A_DIR" -name 'findings-*.json' | wc -l | tr -d ' ') +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --run-id rejected --findings "$BAD" 2>&1) || rc=$? +assert_exit "a non-conforming payload exits 3" 3 "$rc" +assert_contains "an unknown category is named" "$OUT" '"category" must be one of' +assert_contains "an unknown verdict is named" "$OUT" '"verdict" must be one of' +assert_contains "a missing field is named" "$OUT" 'missing required field "candidate"' +assert_contains "evidence of the wrong type is named" "$OUT" '"evidence" must be an array of strings' +assert_contains "a PASS with no plan is named" "$OUT" 'must carry a "plan" object' +after=$(find "$A_DIR" -name 'findings-*.json' | wc -l | tr -d ' ') +assert_eq "a rejected payload writes no findings file" "$before" "$after" +assert_eq "a rejected payload leaves no staged temporary behind" "0" \ + "$(find "$A_DIR" -name '.payload.*' -o -name '.envelope.*' | wc -l | tr -d ' ')" + +rc=0 +OUT=$(printf 'not json at all' | run write --plugin-data "$DATA" --root "$REPO_A" --findings - 2>&1) || rc=$? +assert_exit "a payload that is not JSON exits 3" 3 "$rc" +assert_contains "the refusal says the payload is not well-formed JSON" "$OUT" "not well-formed JSON" + +rc=0 +OUT=$(printf '[1,2,3]' | run write --plugin-data "$DATA" --root "$REPO_A" --findings - 2>&1) || rc=$? +assert_exit "a JSON array payload exits 3" 3 "$rc" +assert_contains "the refusal says the payload must be an object" "$OUT" "must be a JSON object" + +rc=0 +OUT=$(printf '{"maturity":"none"}' | run write --plugin-data "$DATA" --root "$REPO_A" --findings - 2>&1) || rc=$? +assert_exit "a payload with no candidates key exits 3" 3 "$rc" +assert_contains "the refusal names the missing verdict table" "$OUT" 'must carry a "candidates" array' + +# An empty verdict table is a RESULT, not a rejection: the skill's own quality +# principles make a clean bill of health a valid outcome. +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_B" --run-id clean --findings "$EMPTY" 2>&1) || rc=$? +assert_exit "an empty candidates array is accepted" 0 "$rc" + +# 9a: the plan requirement, and its NEGATIVE. A PASS with no plan is the state +# that leaves --implement holding a verdict and nothing to act on. +NO_PLAN="$TEST_TMPDIR/no-plan.json" +cat >"$NO_PLAN" <<'EOF' +{"candidates": [{"id":"1","candidate":"x","category":"hooks","verdict":"PASS","evidence":["e"]}]} +EOF +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --run-id noplan --findings "$NO_PLAN" 2>&1) || rc=$? +assert_exit "a PASS with no plan is refused" 3 "$rc" + +BROKEN_PLAN="$TEST_TMPDIR/broken-plan.sh" +sed 's@c.plan | type) != "object"@c.plan | type) == "object"@' "$SCRIPT" >"$BROKEN_PLAN" +if grep -qF 'c.plan | type) != "object"' "$BROKEN_PLAN"; then + fail "negative plan-requirement case could not be constructed" \ + "the sed target no longer matches findings-state.sh, so the plan requirement is UNVERIFIED by this run" +else + rc=0 + run_copy "$BROKEN_PLAN" write --plugin-data "$DATA" --root "$REPO_A" \ + --run-id noplan-broken --findings "$NO_PLAN" >/dev/null 2>&1 || rc=$? + assert_exit "with the plan requirement inverted, a PASS with no plan is accepted" 0 "$rc" +fi + +# --- Case 10: argument surface ---------------------------------------------- + +rc=0 +OUT=$(run 2>&1) || rc=$? +assert_exit "no command exits 2" 2 "$rc" + +rc=0 +OUT=$(run bogus 2>&1) || rc=$? +assert_exit "an unknown command exits 2" 2 "$rc" +assert_contains "the refusal names the command" "$OUT" "unknown command: bogus" + +rc=0 +OUT=$(run paths --plugin-data "$DATA" --bogus x 2>&1) || rc=$? +assert_exit "an unknown argument exits 2" 2 "$rc" + +rc=0 +OUT=$(run paths --plugin-data "$DATA" --run-id x 2>&1) || rc=$? +assert_exit "paths refuses --run-id" 2 "$rc" + +rc=0 +OUT=$(run list --plugin-data "$DATA" --findings "$GOOD" 2>&1) || rc=$? +assert_exit "list refuses --findings" 2 "$rc" + +rc=0 +OUT=$(run read --plugin-data "$DATA" --findings "$GOOD" 2>&1) || rc=$? +assert_exit "read refuses --findings" 2 "$rc" + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "write with no --findings exits 2" 2 "$rc" + +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --findings "$TEST_TMPDIR/nope.json" 2>&1) || rc=$? +assert_exit "write with a missing payload file exits 2" 2 "$rc" + +rc=0 +OUT=$(run paths --plugin-data "$DATA" --root "$TEST_TMPDIR/not-there" 2>&1) || rc=$? +assert_exit "a --root that is not a directory exits 2" 2 "$rc" + +# --- Case 11: CONCURRENCY, the never-overwritten promise under a real race ---- +# +# Five writers, ONE run id, five DISTINCT payloads, all in flight at once. A +# check-then-publish implementation passes every earlier case in this file and +# still loses four verdict sets here while reporting success five times, which is +# the worst thing a persistence layer can do. The claim under test is not "it +# does not crash": it is that every payload that was reported written is on disk +# under the id that was reported back. + +CDATA="$TEST_TMPDIR/concurrent-data" +mkdir -p "$CDATA" +for i in 1 2 3 4 5; do + printf '{"candidates":[{"id":"%s","candidate":"c%s","category":"hooks","verdict":"REJECT","evidence":["e%s"]}]}\n' \ + "$i" "$i" "$i" >"$TEST_TMPDIR/conc-$i.json" +done +for i in 1 2 3 4 5; do + run write --plugin-data "$CDATA" --root "$REPO_A" --run-id shared \ + --findings "$TEST_TMPDIR/conc-$i.json" >"$TEST_TMPDIR/conc-out-$i" 2>&1 & +done +wait +CONC_DIR="$CDATA/audit-automation-gaps/$A_KEY" +assert_eq "five concurrent writes leave five findings files" "5" \ + "$(find "$CONC_DIR" -name 'findings-shared*.json' | wc -l | tr -d ' ')" +CONC_IDS=$(cat "$TEST_TMPDIR"/conc-out-* | sed -n 's/^run_id=//p' | sort) +assert_eq "each concurrent write reports a DISTINCT run id" "5" \ + "$(printf '%s\n' "$CONC_IDS" | sort -u | wc -l | tr -d ' ')" +assert_eq "the suffixes are the documented ones" "shared shared-2 shared-3 shared-4 shared-5" \ + "$(printf '%s\n' "$CONC_IDS" | tr '\n' ' ' | sed 's/ *$//')" +conc_missing="" +for id in $CONC_IDS; do + if [[ ! -f "$CONC_DIR/findings-$id.json" ]]; then + conc_missing="$conc_missing $id" + fi +done +assert_eq "every run id reported back names a file that exists" "" "$conc_missing" +assert_eq "every payload survives, none overwritten by another writer" "1 2 3 4 5" \ + "$(cat "$CONC_DIR"/findings-shared*.json | jq -r '.findings.candidates[0].id' | sort | tr '\n' ' ' | sed 's/ *$//')" +assert_eq "the history carries one row per surviving run, not five claiming one id" "5" \ + "$(jq -r '.run_id' <"$CONC_DIR/history.jsonl" | sort -u | wc -l | tr -d ' ')" +assert_eq "read serves a real per-run file after the race" "1" \ + "$(run read --plugin-data "$CDATA" --root "$REPO_A" | jq -r '.findings.candidates | length')" + +# --- Case 12: one payload document, because only one can be persisted --------- +# +# jq accepts a stream of concatenated documents, so a two-document payload used +# to validate in full and persist only the first. A stream whose second half +# holds the real verdicts then reported success with counts of zero. + +MULTI="$TEST_TMPDIR/multi.json" +cat >"$MULTI" <<'EOF' +{"candidates": [], "maturity": "a clean bill of health"} +{"candidates": [{"id":"9","candidate":"real gap","category":"hooks","verdict":"REJECT","evidence":["real evidence"]}]} +EOF +rc=0 +OUT=$(run write --plugin-data "$DATA" --root "$REPO_A" --run-id multi --findings "$MULTI" 2>&1) || rc=$? +assert_exit "a multi-document payload exits 3" 3 "$rc" +assert_contains "the refusal names the one-document rule" "$OUT" "exactly one JSON document" +assert_contains "the refusal says which half would have been kept" "$OUT" \ + "only the first document would be persisted" +assert_eq "a multi-document payload writes no findings file" "" \ + "$(find "$A_DIR" -name 'findings-multi*.json')" + +# --- Case 13: a publish is all or nothing, and an orphan is not an absence ---- + +ODATA="$TEST_TMPDIR/orphan-data" +ODIR="$ODATA/audit-automation-gaps/$A_KEY" +mkdir -p "$ODIR/history.jsonl" +rc=0 +OUT=$(run write --plugin-data "$ODATA" --root "$REPO_A" --run-id h1 --findings "$GOOD" 2>&1) || rc=$? +assert_exit "write refuses up front when the history file cannot be appended to" 5 "$rc" +assert_contains "the refusal says nothing was written" "$OUT" "nothing was written" +assert_eq "and nothing was: no orphan findings file is left on disk" "" \ + "$(find "$ODIR" -name 'findings-*.json')" + +# A findings file whose pointers are gone is an INCOMPLETE PUBLISH. Reporting it +# as "no findings are persisted" denies a verdict set that is sitting on disk. +PDATA="$TEST_TMPDIR/partial-data" +mkdir -p "$PDATA" +run write --plugin-data "$PDATA" --root "$REPO_A" --run-id partial --findings "$GOOD" >/dev/null 2>&1 +PDIR="$PDATA/audit-automation-gaps/$A_KEY" +rm -f "$PDIR/latest" "$PDIR/history.jsonl" +rc=0 +OUT=$(run read --plugin-data "$PDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read reports an orphaned findings file rather than absence" 5 "$rc" +assert_contains "the report names the incomplete publish" "$OUT" "INCOMPLETE PUBLISH" +assert_contains "the report names the file the operator can still read" "$OUT" "findings-partial.json" +assert_not_contains "and does not claim nothing is persisted" "$OUT" "no findings are persisted" +rc=0 +OUT=$(run list --plugin-data "$PDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "list reports the orphan rather than an empty listing" 5 "$rc" +assert_not_contains "list does not call an incomplete publish 'no persisted runs'" "$OUT" \ + "no persisted runs" +rc=0 +OUT=$(run read --plugin-data "$PDATA" --root "$REPO_A" --run-id partial 2>&1) || rc=$? +assert_exit "the orphaned run is still readable by its id" 0 "$rc" + +# The pre-flight cannot see every failure, so the publish is also rolled back +# when the append fails anyway. A DANGLING symlink is the reproduction: `-e` is +# false for one, so it passes the pre-flight, and the append into a directory +# that does not exist then fails with the findings file already published. +if host_makes_symlinks; then + RBDATA="$TEST_TMPDIR/rollback-data" + RBDIR="$RBDATA/audit-automation-gaps/$A_KEY" + mkdir -p "$RBDIR" + ln -s "$RBDIR/no-such-dir/history.jsonl" "$RBDIR/history.jsonl" + rc=0 + OUT=$(run write --plugin-data "$RBDATA" --root "$REPO_A" --run-id rb --findings "$GOOD" 2>&1) || rc=$? + assert_exit "a failed history append is reported as unusable state, not a usage error" 5 "$rc" + assert_contains "the report says the findings file was rolled back" "$OUT" "rolled back" + assert_eq "the published findings file is rolled back rather than orphaned" "" \ + "$(find "$RBDIR" -name 'findings-rb*.json')" + rc=0 + OUT=$(run read --plugin-data "$RBDATA" --root "$REPO_A" 2>&1) || rc=$? + assert_exit "and the rolled-back run leaves no orphan for read to report" 4 "$rc" +else + skip "a failed history append rolls the findings file back" "ln -s copies here" +fi + +# --- Case 14: UNREADABLE IS NOT ABSENT --------------------------------------- +# +# Directories stand in for the unreadable file here because the suite may run as +# root, where a chmod 000 file is still readable and the case would pass without +# testing anything. + +BDATA="$TEST_TMPDIR/broken-state-data" +BDIR="$BDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$BDIR/history.jsonl" "$BDIR/findings-d8.json" +printf 'd8\n' >"$BDIR/latest" +rc=0 +OUT=$(run list --plugin-data "$BDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "list exits 5 when the history file is not a regular file" 5 "$rc" +assert_contains "the report distinguishes unreadable from absent" "$OUT" \ + "that is not the same as having none" +assert_not_contains "list does not report unreadable state as no runs" "$OUT" "no persisted runs" + +rc=0 +OUT=$(run read --plugin-data "$BDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 5 when the findings file is not a regular file" 5 "$rc" +assert_contains "read names the findings file it could not read" "$OUT" "findings-d8.json" + +LDATA="$TEST_TMPDIR/broken-latest-data" +LDIR="$LDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$LDIR/latest" +rc=0 +OUT=$(run read --plugin-data "$LDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 5 when the latest pointer is not a regular file" 5 "$rc" +assert_contains "the report names the pointer" "$OUT" "the latest pointer exists" + +EDATA="$TEST_TMPDIR/empty-latest-data" +EDIR="$EDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$EDIR" +: >"$EDIR/latest" +rc=0 +OUT=$(run read --plugin-data "$EDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "an empty latest pointer is damaged state, not absence" 5 "$rc" + +FDATA="$TEST_TMPDIR/file-basedir-data" +mkdir -p "$FDATA/audit-automation-gaps/${A_KEY%/*}" +printf 'not a directory\n' >"$FDATA/audit-automation-gaps/$A_KEY" +rc=0 +OUT=$(run read --plugin-data "$FDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 5 when the keyed directory is a regular file" 5 "$rc" +rc=0 +OUT=$(run list --plugin-data "$FDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "list exits 5 when the keyed directory is a regular file" 5 "$rc" + +# A payload that exists, is readable, is not empty, and still does not parse. +# Publication stages and renames, so this script cannot leave one behind; a +# foreign writer or a stray editor can. Serving those bytes under exit 0 would +# hand the caller something unusable, which is what exit 5 exists to prevent. +CDATA="$TEST_TMPDIR/corrupt-payload-data" +CDIR="$CDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$CDIR" +printf 'corrupt-not-json\n' >"$CDIR/findings-c1.json" +printf 'c1\n' >"$CDIR/latest" +printf '{"run_id":"c1"}\n' >"$CDIR/history.jsonl" +rc=0 +OUT=$(run read --plugin-data "$CDATA" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "read exits 5 when the findings payload does not parse" 5 "$rc" +assert_contains "the report says the payload cannot be trusted" "$OUT" "cannot be trusted" +assert_not_contains "the corrupt bytes are never served" "$OUT" "corrupt-not-json" + +# --- Case 15: the suffix cap is state exhaustion, not a usage error ----------- + +CAPDATA="$TEST_TMPDIR/cap-data" +CAPDIR="$CAPDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$CAPDIR" +: >"$CAPDIR/findings-cap.json" +i=2 +while [[ "$i" -le 99 ]]; do + : >"$CAPDIR/findings-cap-$i.json" + i=$((i + 1)) +done +rc=0 +OUT=$(run write --plugin-data "$CAPDATA" --root "$REPO_A" --run-id cap --findings "$GOOD" 2>&1) || rc=$? +assert_exit "an exhausted suffix range exits 5, not the usage code" 5 "$rc" +assert_contains "the refusal names the run id whose names are taken" "$OUT" "run id: cap" + +# --- Case 16: what the envelope records, and how the pointer is replaced ------ +# +# repo_root and written_at are both derived per run. A constant in either place +# is invisible to a round-trip assertion that only checks the field is present, +# so both are compared against a value computed OUTSIDE the script. + +ENVELOPE=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id run-0001) +assert_eq "the envelope records the repo root it was derived from" \ + "$(git -C "$REPO_A" rev-parse --show-toplevel)" \ + "$(printf '%s' "$ENVELOPE" | jq -r '.repo_root')" +B_ENVELOPE=$(run read --plugin-data "$DATA" --root "$REPO_B" --run-id run-0001) +assert_eq "a different repository records a different repo root" \ + "$(git -C "$REPO_B" rev-parse --show-toplevel)" \ + "$(printf '%s' "$B_ENVELOPE" | jq -r '.repo_root')" + +stamp_before=$(date -u +%Y-%m-%d) +run write --plugin-data "$DATA" --root "$REPO_A" --run-id stamped --findings "$GOOD" >/dev/null 2>&1 +stamp_after=$(date -u +%Y-%m-%d) +stamped=$(run read --plugin-data "$DATA" --root "$REPO_A" --run-id stamped | jq -r '.written_at') +case "$stamped" in +"$stamp_before"T*Z | "$stamp_after"T*Z) + pass "written_at is stamped when the run is written" + ;; +*) + fail "written_at is stamped when the run is written" \ + "expected a UTC stamp dated $stamp_before or $stamp_after, got: $stamped" + ;; +esac + +# The `latest` pointer is a file under a directory anything with write access can +# reach, so the run id it yields is a caller-supplied id by another route. +POISON="$TEST_TMPDIR/poison-data" +PODIR="$POISON/audit-automation-gaps/$A_KEY" +mkdir -p "$PODIR" +run write --plugin-data "$POISON" --root "$REPO_A" --run-id clean --findings "$GOOD" >/dev/null 2>&1 +printf '../../etc/passwd\n' >"$PODIR/latest" +rc=0 +OUT=$(run read --plugin-data "$POISON" --root "$REPO_A" 2>&1) || rc=$? +assert_exit "a poisoned latest pointer is refused" 2 "$rc" +assert_contains "the refusal names the plain-segment rule" "$OUT" "plain path segment" + +# 16a: NEGATIVE. Delete the validation on the pointer path and the poisoned id +# reaches path construction. +BROKEN_PTR="$TEST_TMPDIR/broken-pointer.sh" +# shellcheck disable=SC2016 # sed program text: the $ is a literal character in the script being mutated +sed 's@validate_run_id "$run_id"@:@' "$SCRIPT" >"$BROKEN_PTR" +# shellcheck disable=SC2016 # grep pattern text: same literal $ as the sed above +if grep -qF 'validate_run_id "$run_id"' "$BROKEN_PTR"; then + fail "negative pointer-validation case could not be constructed" \ + "the sed target no longer matches findings-state.sh, so the pointer validation is UNVERIFIED by this run" +else + rc=0 + broken_out=$(run_copy "$BROKEN_PTR" read --plugin-data "$POISON" --root "$REPO_A" 2>&1) || rc=$? + assert_not_contains "with the pointer validation deleted the poisoned id is no longer refused" \ + "$broken_out" "plain path segment" + assert_contains "and it reaches the filesystem as a traversing path" \ + "$broken_out" "findings-../../etc/passwd.json" +fi + +# `latest` is REPLACED, never written through. A `cp` would follow whatever the +# name already is and clobber it, and would let a reader see a half pointer. +PTRDATA="$TEST_TMPDIR/pointer-data" +PTRDIR="$PTRDATA/audit-automation-gaps/$A_KEY" +mkdir -p "$PTRDIR" +if host_makes_symlinks; then + DECOY="$TEST_TMPDIR/decoy-pointer" + printf 'decoy\n' >"$DECOY" + ln -s "$DECOY" "$PTRDIR/latest" + run write --plugin-data "$PTRDATA" --root "$REPO_A" --run-id ptr --findings "$GOOD" >/dev/null 2>&1 + assert_eq "the write does not follow the old pointer and clobber its target" "decoy" "$(cat "$DECOY")" + if [[ -L "$PTRDIR/latest" ]]; then + fail "the latest pointer is replaced, not written through" "it is still a symlink to the decoy" + else + pass "the latest pointer is replaced, not written through" + fi + assert_eq "the replaced pointer names the run just written" "ptr" "$(cat "$PTRDIR/latest")" +else + skip "the latest pointer is replaced, not written through" "ln -s copies here" + run write --plugin-data "$PTRDATA" --root "$REPO_A" --run-id ptr --findings "$GOOD" >/dev/null 2>&1 +fi +assert_eq "no staged pointer temporary is left behind" "0" \ + "$(find "$PTRDIR" -name 'latest.*' | wc -l | tr -d ' ')" + +# --- Case 17: A MISSING TOOL IS NOT A DAMAGED ARTIFACT ------------------------ +# +# `read` parses the stored envelope with `jq` before serving it. On a host with +# no `jq` that command fails with exit 127, which is indistinguishable from a +# document that did not parse unless the prerequisite is checked first. Without +# the check a perfectly valid persisted run is reported under exit 5 as state an +# external writer corrupted, which is the script lying about the operator's +# data: the artifact is fine, the machine is missing a tool. The contract puts a +# missing prerequisite at exit 2, so that is what this asserts. +# +# `jq` is made unavailable by running the script under a PATH holding symlinks +# to the tools it needs and nothing else. Uninstalling `jq` is not an option and +# a wrapper that fails would still satisfy `command -v`. + +NOJQ_BIN="$TEST_TMPDIR/nojq-bin" +mkdir -p "$NOJQ_BIN" +for tool in bash sh git date sha256sum shasum tr sed cut head cat mv rm mkdir find wc ln chmod grep; do + tool_path="$(command -v "$tool" 2>/dev/null)" || continue + ln -s "$tool_path" "$NOJQ_BIN/$tool" 2>/dev/null || true +done +# Probe the shim rather than trust it: a host where `ln -s` copies, or where a +# needed tool is a shell function or builtin alias, would otherwise turn this +# case into a meaningless failure. +if PATH="$NOJQ_BIN" bash -c 'command -v git >/dev/null 2>&1 && ! command -v jq >/dev/null 2>&1' 2>/dev/null; then + NOJQ_DATA="$TEST_TMPDIR/nojq-data" + run write --plugin-data "$NOJQ_DATA" --root "$REPO_A" --run-id nojq --findings "$GOOD" >/dev/null 2>&1 + rc=0 + OUT=$(run read --plugin-data "$NOJQ_DATA" --root "$REPO_A" 2>&1) || rc=$? + assert_exit "the stored run this case reads back is valid while jq is present" 0 "$rc" + + rc=0 + OUT=$(PATH="$NOJQ_BIN" CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$SCRIPT" \ + read --plugin-data "$NOJQ_DATA" --root "$REPO_A" 2>&1) || rc=$? + assert_exit "read exits 2 with no jq, the prerequisite code and not the damaged-state one" 2 "$rc" + assert_contains "the refusal names jq as the missing prerequisite" "$OUT" "jq is required" + assert_not_contains "a valid stored run is never blamed for a missing jq" \ + "$OUT" "cannot be trusted" + assert_not_contains "and nothing claims an outside writer touched the artifact" \ + "$OUT" "something outside this script wrote it" + + # The gate belongs only where jq is actually used. `list` serves the history + # file with `cat` and `paths` computes a path, so requiring jq for either + # would break a host that can still answer both. + rc=0 + OUT=$(PATH="$NOJQ_BIN" CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$SCRIPT" \ + list --plugin-data "$NOJQ_DATA" --root "$REPO_A" 2>&1) || rc=$? + assert_exit "list still answers with no jq, because it uses none" 0 "$rc" + assert_contains "and it serves the recorded run" "$OUT" '"run_id":"nojq"' + + rc=0 + OUT=$(PATH="$NOJQ_BIN" CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$SCRIPT" \ + paths --plugin-data "$NOJQ_DATA" --root "$REPO_A" 2>&1) || rc=$? + assert_exit "paths still answers with no jq, because it uses none" 0 "$rc" + + # `write` composes the envelope with jq and already gated on it. Asserted here + # so the three subcommands are judged against one another in one place. + rc=0 + OUT=$(PATH="$NOJQ_BIN" CLAUDE_PLUGIN_ROOT="$PLUGIN_ROOT" bash "$SCRIPT" \ + write --plugin-data "$NOJQ_DATA" --root "$REPO_A" --run-id nojq-2 --findings "$GOOD" 2>&1) || rc=$? + assert_exit "write exits 2 with no jq" 2 "$rc" + assert_contains "write names jq too" "$OUT" "jq is required" +else + skip "read exits 2 with no jq, the prerequisite code and not the damaged-state one" \ + "a jq-free PATH could not be built here" +fi + +if [[ "$FAILED" -eq 0 ]]; then + printf '\nAll %d checks passed (%d skipped for host capability).\n' "$CASE_NUM" "$SKIPPED" + exit 0 +fi +printf '\n%d/%d checks failed (%d skipped for host capability).\n' "$FAILED" "$CASE_NUM" "$SKIPPED" >&2 +exit 1 diff --git a/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.sh b/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.sh index 3c35f97bc4..4165622430 100755 --- a/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.sh +++ b/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.sh @@ -1,57 +1,770 @@ #!/usr/bin/env bash # Automation landscape counts for the audit-automation-gaps skill. # -# Output: Hook scripts, Skills, Agents, MCP servers, Plugins enabled. -# Exit: always 0. +# The skill injects this script's stdout as pre-computed context, so the output is +# a COUNT TABLE and never a row dump. An all-scope enumerator that emitted one +# line per hook would spend more of the window than the audit it feeds. +# +# WHAT CHANGED AND WHY. The previous revision read only +# .claude/hooks, .claude/skills and .claude/agents, so on a repository carrying +# 77 plugins, 271 SKILL.md files and 137 hook files it reported 3 hook scripts, 0 +# skills, 0 agents and 1 plugin. Every number was a zero produced by not looking, +# and a zero produced by not looking reads exactly like a real absence. The skill +# body's own Phase 1.1 already instructs the model to read user, project and local +# settings, managed policy, every enabled plugin's hooks/hooks.json, and skill and +# subagent frontmatter; the script was behind its own skill's spec. +# +# THE SAME ZERO, THROUGH FIVE OTHER DOORS. A review of that rewrite found it +# reproducing the defect it existed to remove, so every walk and every count in +# this file now goes through one of four helpers, and none of them can answer 0 +# for a scope it did not actually read: +# find0 the only tree walk. Passes -H so a SYMLINKED .claude/skills, +# .claude/agents or .claude/hooks is descended instead of silently +# yielding nothing; prunes .git and node_modules so .git/hooks/* +# sample scripts cannot be counted as a plugin's hook scripts; and +# emits NUL-delimited paths so a newline inside a path cannot split +# one file into two. +# dir_status classifies a directory scope before it is walked. A path that +# exists but cannot be traversed, a dangling symlink included, is +# unreadable, never absent and never 0. All four directory scopes +# go through it, .claude/skills, .claude/agents, .claude/hooks and +# managed-settings.d, and each carries the status word the whole +# way out through count_dir or scope_count. Wiring one scope and +# not its siblings is how this class survived the first pass. +# jq_num FAILS instead of printing 0 when jq fails. A hooks or +# enabledPlugins key holding a string, a number or an array is +# reported invalid-json, because the wrong type is not "no hooks". +# count0 counts a NUL stream rather than lines. +# An unusable project root is likewise fatal: printing another directory's counts +# under the requested root's name was the worst shape of all, since the header +# named a directory the numbers did not come from. +# +# TWO MECHANICS THIS OUTPUT KEEPS APART. Hook entries MERGE across settings +# levels: user, project, local and managed each contribute, and none replaces +# another, so a per-scope additive count table is the honest shape for them. +# enabledPlugins is a PRECEDENCE key, not a merged one, and this script therefore +# reports the maps as read and computes no effective set. That is the same posture +# claude-ops:inventory and claude-ops:audit-native-overlap already take, and the +# Enablement section names where the verdict lives. +# +# STATUS, NEVER A SILENT ZERO. A location that could not be read reports +# unreadable, invalid-json, skipped or not-probed. A count is printed only for a +# location that was actually read. +# +# VOCABULARY is claude-ops:inventory's, not a second dialect for the same ideas: +# present / absent / unreadable / invalid-json / skipped / not-probed +# standing registered for the whole session, part of the always-on set +# conditional registered only when a skill is invoked, or only while a +# subagent runs, so it is not part of the standing set +# A hook script on disk is not a wired hook. Script counts live in Components, +# apart from the wired handler counts, and are never added to them. +# +# Claim: hooks are configured in exactly seven locations, user settings, project +# settings, project local settings, managed policy, a plugin's +# hooks/hooks.json, skill frontmatter and subagent frontmatter; hook entries +# merge across settings levels rather than replacing each other; skill hooks +# last for the rest of the session once the skill is invoked and subagent hooks +# only while that subagent runs. +# Basis: https://code.claude.com/docs/en/hooks, the "Hook locations" table. +# As of: 2026-09-13. +# Recheck trigger: that table gains, drops or renames a row, or the sentence +# stating that hook entries merge across settings levels changes. +# +# Claim: managed policy has a documented per-OS location, so this script probes it +# instead of reporting it not-probed. +# Basis: the paths are owned by ../../../lib/managed-scope.sh, which carries its +# own stamp against https://code.claude.com/docs/en/managed-settings; that page +# was re-read for this script and still lists +# /Library/Application Support/ClaudeCode/, /etc/claude-code/ and +# C:\Program Files\ClaudeCode\ plus an optional managed-settings.d directory. +# As of: 2026-09-13. +# Recheck trigger: managed-scope.sh's own recheck trigger fires, or this script +# starts reporting managed policy not-probed on a machine that has a policy. set -u usage() { cat <<'EOF' -inventory.sh — emit automation landscape counts (hooks, skills, MCP, plugins). +inventory.sh - per-scope automation counts for the audit-automation-gaps skill. Usage: inventory.sh [--help] -Exit: always 0. +It takes no other argument. An unrecognised argument is a usage error rather +than a silently ignored one, because a run that ignored its arguments would +report a full audit under a scope nobody asked for. + +Sections, in a stable order: + Hook locations one row per documented hook location, each with a status, a + standing-versus-conditional kind, and three counts: PROBED, + the files or paths this run examined; DECLARING, how many of + them register at least one hook; HANDLERS, how many hook + commands those registrations add. A frontmatter row carries no + HANDLERS figure because the YAML inside its hooks block is not + parsed here. + Components repository-tree counts for plugins, skills, agents, MCP + servers, and hook scripts present on disk + Enablement the enabledPlugins maps as read, per scope, with no verdict + Notes what a reader must know to not over-read the numbers + +A location this script could not read reports unreadable, invalid-json, skipped +or not-probed. It never reports that location as 0. A directory reached through +a symlink is walked, not skipped; a JSON file whose hooks or enabledPlugins key +holds the wrong type is invalid-json, not zero hooks; and .git and node_modules +are pruned from every walk so a sample hook or a vendored tree cannot inflate a +component count. + +Test seams, unset in normal use: + INVENTORY_PROJECT_DIR project root, instead of the git toplevel + INVENTORY_USER_SETTINGS user settings file, instead of the documented path + INVENTORY_MANAGED_PATH managed-settings.json, instead of the per-OS path + INVENTORY_JQ the jq command name, so a test can take jq away + +Exit: 0 once a table is printed; 2 for a usage error or a project root that +could not be entered. The output is a skill's pre-computed context, so a +partial read still exits 0 and says per row what it could not reach; only a run +that can produce no honest table at all exits nonzero. EOF } -case "${1:-}" in --h | --help) - usage - exit 0 +usage_error() { + printf 'inventory.sh: unrecognised argument: %s\n' "$1" >&2 + usage >&2 + exit 2 +} + +case "$#" in +0) ;; +1) + case "$1" in + -h | --help) + usage + exit 0 + ;; + *) usage_error "$1" ;; + esac + ;; +*) + case "$1" in + -h | --help) usage_error "$2" ;; + *) usage_error "$1" ;; + esac ;; -*) ;; esac -# Count files matching the given find args under , or 0 when is absent. -count_files() { - local dir="$1" - shift - [[ -d "$dir" ]] || { +jq_bin="${INVENTORY_JQ:-jq}" +have_jq=0 +command -v "$jq_bin" >/dev/null 2>&1 && have_jq=1 + +# Plugin root resolution mirrors permission-state.sh: parameter expansion plus +# builtins, so a missing external tool cannot silently turn managed policy into +# an absence. A reader that reports no policy because it could not load its own +# library is the exact failure this rewrite exists to remove. +plugin_root="${CLAUDE_PLUGIN_ROOT:-$(cd "${BASH_SOURCE[0]%/*}/../../.." && pwd)}" +managed_lib="$plugin_root/lib/managed-scope.sh" +have_managed_lib=0 +if [[ -r "$managed_lib" ]]; then + # shellcheck source=../../../lib/managed-scope.sh + source "$managed_lib" && have_managed_lib=1 +fi + +if [[ -n "${INVENTORY_PROJECT_DIR:-}" ]]; then + project_root="$INVENTORY_PROJECT_DIR" +else + project_root="$(git rev-parse --show-toplevel 2>/dev/null | tr -d '\r')" + [[ -n "$project_root" ]] || project_root="${CLAUDE_PROJECT_DIR:-$PWD}" +fi + +# A cd that fails is FATAL. The header prints the requested root, so a run that +# stayed in the invoking directory would attribute that directory's plugins, +# skills and handlers to a root they have nothing to do with. The `--` lets a +# relative root beginning with a dash be entered rather than parsed as an option. +if [[ ! -d "$project_root" ]] || ! cd -- "$project_root" 2>/dev/null; then + printf 'inventory.sh: project root %s is not a directory this run could enter; no counts were produced.\n' \ + "$project_root" >&2 + exit 2 +fi +project_root="$PWD" + +user_settings="${INVENTORY_USER_SETTINGS:-${CLAUDE_CONFIG_DIR:-${HOME:-}/.claude}/settings.json}" +project_settings=".claude/settings.json" +local_settings=".claude/settings.local.json" + +managed_file="" +managed_dropin="" +if [[ "$have_managed_lib" -eq 1 ]]; then + managed_file="$(mscope::base_file "${INVENTORY_MANAGED_PATH:-}")" + managed_dropin="$(mscope::dropin_dir "${INVENTORY_MANAGED_PATH:-}")" +fi + +notes=() +note() { notes+=("$1"); } + +# Classify one JSON surface. A path that exists but is not a regular file, or is +# not readable, is unreadable rather than absent: the caller must be able to tell +# "there is no policy here" from "I could not look". +json_status() { + local f="$1" + [[ -n "$f" ]] || { + printf 'not-probed' + return + } + [[ -e "$f" ]] || { + printf 'absent' + return + } + if [[ ! -f "$f" || ! -r "$f" ]]; then + printf 'unreadable' + return + fi + if [[ "$have_jq" -eq 0 ]]; then + printf 'skipped' + return + fi + if ! "$jq_bin" -e 'type == "object"' "$f" >/dev/null 2>&1; then + printf 'invalid-json' + return + fi + printf 'present' +} + +# Classify one directory scope before anything walks it. A symlink to a +# directory IS a directory here and is walked; a dangling symlink, a plain file +# in a directory's place, or a directory this uid cannot open is unreadable, so +# the count next to it is a status word instead of the 0 that would read as a +# real absence. +dir_status() { + local d="$1" + [[ -n "$d" ]] || { + printf 'not-probed' + return + } + if [[ -d "$d" ]]; then + if [[ -r "$d" && -x "$d" ]]; then + printf 'present' + else + printf 'unreadable' + fi + return + fi + if [[ -e "$d" || -L "$d" ]]; then + printf 'unreadable' + return + fi + printf 'absent' +} + +# The only tree walk in this script. Start points come first, then a literal --, +# then the find expression. +# +# -H follows a symlinked START POINT, so a .claude/skills that is a +# symlink to the real tree is descended. Without it find refuses +# to descend the start point and reports nothing, which is the +# "0 produced by not looking" this file exists to prevent. Only +# the start point is followed, so a symlink loop inside the tree +# still cannot trap the walk. +# prune .git carries a hooks/ directory whose *.sample files are not +# anybody's hook scripts, and a vendored node_modules tree can +# carry any layout at all. Both are pruned from every walk. +# -print0 a path may contain a newline. Line-delimited output would turn +# one such file into two records and lose the real one. +find0() { + local dirs=() + while [[ "$#" -gt 0 && "$1" != "--" ]]; do + dirs+=("$1") + shift + done + [[ "$#" -gt 0 ]] && shift + [[ "${#dirs[@]}" -gt 0 && "$#" -gt 0 ]] || return 0 + find -H "${dirs[@]}" \( -name .git -o -name node_modules \) -prune -o \( "$@" \) -print0 2>/dev/null +} + +# Counts a NUL-delimited stream on stdin. wc -l would count newlines, which are +# legal inside a path and would therefore over-count. +count0() { tr -cd '\000' | wc -c | tr -d ' '; } + +# A settings file names its hook block under the `hooks` key; a plugin manifest +# may wrap the same block or carry the events at its top level, so the two get +# their own programs rather than one guessing filter over both. +SETTINGS_EVENTS_JQ='[(.hooks // {}) | to_entries[] | select((.value | type) == "array" and (.value | length) > 0) | .key] | length' +SETTINGS_HANDLERS_JQ='[(.hooks // {}) | to_entries[] | .value[]? | (.hooks // []) | length] | add // 0' +PLUGIN_EVENTS_JQ='def hk: if has("hooks") then .hooks else . end; [hk | to_entries[] | select((.value | type) == "array" and (.value | length) > 0) | .key] | length' +PLUGIN_HANDLERS_JQ='def hk: if has("hooks") then .hooks else . end; [hk | to_entries[] | .value[]? | (.hooks // []) | length] | add // 0' + +# Runs one jq program over one file and prints its number. A jq that could not +# run, or a program whose result is not a plain non-negative integer, prints +# NOTHING and returns 1. Printing 0 there was the defect: a hooks key holding a +# string, a number, a boolean or an array made every query fail or return +# nothing, and the row then said "present, no hooks" about a file whose hook +# block is malformed. +jq_num() { + local program="$1" file="$2" out + out="$("$jq_bin" -r "$program" "$file" 2>/dev/null)" || return 1 + case "$out" in + '' | *[!0-9]*) return 1 ;; + *) printf '%s' "$out" ;; + esac +} + +# The jq type of one expression, or the empty string when jq could not run. An +# array answers most of the count programs without erroring, so the type has to +# be asked directly rather than inferred from a query that happened to succeed. +jq_type() { "$jq_bin" -r "$1 | type" "$2" 2>/dev/null; } + +# Discovered plugin roots. `.claude-plugin/plugin.json` is the documented marker, +# and the depth bound keeps the walk off vendored trees. +plugin_roots=() +while IFS= read -r -d '' manifest; do + [[ -n "$manifest" ]] || continue + plugin_roots+=("${manifest%/.claude-plugin/plugin.json}") +done < <(find -H . -maxdepth 4 \( -name .git -o -name node_modules \) -prune -o \ + -type f -path '*/.claude-plugin/plugin.json' -print0 2>/dev/null) + +# Walks every discovered plugin root, or emits nothing when this repository ships +# no plugins. +find0_plugin_roots() { + [[ "${#plugin_roots[@]}" -gt 0 ]] || return 0 + find0 "${plugin_roots[@]}" -- "$@" +} + +count_in_plugin_roots() { + [[ "${#plugin_roots[@]}" -gt 0 ]] || { + printf '0' + return + } + find0_plugin_roots "$@" | count0 +} + +# Counts files under one directory scope whose status was classified first. Only +# a scope that was actually walked gets a number; anything else gets its status +# word, so a count in this output is always a measurement. +count_dir() { + local status="$1" dir="$2" + shift 2 + case "$status" in + present) find0 "$dir" -- "$@" | count0 ;; + absent) printf '0' ;; + *) printf '%s' "$status" ;; + esac +} + +# Adds two figures either of which may be a status word rather than a number. +# A status word on either side makes the sum a floor, and it is printed as the +# two parts rather than collapsed into a number that was never measured. +add_counts() { + case "$1$2" in + *[!0-9]*) printf '%s+%s' "$1" "$2" ;; + *) printf '%s' "$(($1 + $2))" ;; + esac +} + +# One scope's contribution to a total: the number of files read from it, or its +# status word when the scope exists and could not be walked. Every scope that +# feeds a Components figure goes through this, so no scope can reach a total as +# a 0 it did not earn. count_dir applies the same rule to a scope it walks +# itself; this one is for a scope whose files were already gathered into a list. +scope_count() { + case "$1" in + present | absent) printf '%s' "$2" ;; + *) printf '%s' "$1" ;; + esac +} + +# Counts files whose YAML frontmatter opens a top-level `hooks:` block. The walk +# is frontmatter-only: a `hooks:` line in a skill body is prose about hooks, not +# a hook registration. awk prints only the tally and never a path, so a newline +# inside a path cannot be mistaken for a record separator on the way out either. +count_frontmatter_hooks() { + local list="$1" f + local files=() + while IFS= read -r -d '' f; do + [[ -n "$f" ]] || continue + files+=("$f") + done <"$list" + [[ "${#files[@]}" -gt 0 ]] || { printf '0' return } - find "$dir" "$@" 2>/dev/null | wc -l | tr -d ' ' + awk ' + FNR == 1 { in_fm = 0; opened = 0; seen = 0 } + FNR == 1 && $0 ~ /^---[[:space:]]*$/ { in_fm = 1; opened = 1; next } + opened && in_fm && $0 ~ /^---[[:space:]]*$/ { in_fm = 0; next } + opened && in_fm && seen == 0 && $0 ~ /^hooks:[[:space:]]*$/ { n += 1; seen = 1 } + END { printf "%d", n + 0 } + ' "${files[@]}" 2>/dev/null +} + +row() { + printf ' %-20s %-12s %-11s %7s %9s %8s %s\n' "$1" "$2" "$3" "$4" "$5" "$6" "$7" +} + +# --- Hook locations ----------------------------------------------------------- + +# Reads one settings-shaped file's hook block into three globals instead of +# printing them. note() appends to an array, and a command substitution would run +# it in a subshell whose appends die with it, so a file that could not be read +# would lose its explanation on the way back to the caller. shc_status is one +# vocabulary word; shc_events and shc_handlers are numbers only when it is +# present, and are zero otherwise so no caller can add a count nothing measured. +shc_status="" +shc_events=0 +shc_handlers=0 +settings_hook_counts() { + local file="$1" hooks_type + shc_events=0 + shc_handlers=0 + shc_status="$(json_status "$file")" + [[ "$shc_status" == "present" ]] || return 0 + hooks_type="$(jq_type '(.hooks // {})' "$file")" + if [[ "$hooks_type" != "object" ]]; then + note "$file is valid JSON, but its hooks key holds ${hooks_type:-a value jq could not type} rather than an object, so no hook entry could be read from it. A malformed hooks block is not an absence of hooks." + shc_status=invalid-json + return 0 + fi + if ! shc_events="$(jq_num "$SETTINGS_EVENTS_JQ" "$file")" || + ! shc_handlers="$(jq_num "$SETTINGS_HANDLERS_JQ" "$file")"; then + note "$file parses as JSON, but the hook query over it failed, so its hook entries are counted nowhere in this table." + shc_status=unreadable + shc_events=0 + shc_handlers=0 + return 0 + fi + return 0 +} + +settings_row() { + local label="$1" file="$2" declaring + settings_hook_counts "$file" + if [[ "$shc_status" != "present" ]]; then + row "$label" "$shc_status" standing 1 - - "$file" + return + fi + declaring=0 + [[ "$shc_events" -gt 0 ]] && declaring=1 + row "$label" "$shc_status" standing 1 "$declaring" "$shc_handlers" "$file" +} + +# The managed-policy row covers managed-settings.json TOGETHER WITH the readable +# managed-settings.d drop-ins, because those drop-ins merge on top of the base +# file rather than sitting beside it. Counting only the base file published an +# exact-looking handler figure for a policy whose standing hooks may live +# entirely in the drop-ins, which is the same confident wrong number this script +# exists to stop printing. A drop-in that could not be read contributes its +# status word instead of a count, so the figures render in the add_counts floor +# form rather than as a total nobody measured. +# +# Registrations, not the de-duplicated effective set: the merge concatenates and +# de-duplicates arrays, so a handler registered identically in the base file and +# in a drop-in is one entry in force and two here. The drop-in note says so. +managed_policy_row() { + local base="$1" dropin_dir="$2" dropin_status="$3" + local drops=() f base_status floor="" probed_floor="" + local probed=1 declaring=0 handlers=0 read_any=0 + local probed_out declaring_out handlers_out source_label + + settings_hook_counts "$base" + base_status="$shc_status" + case "$base_status" in + present) + read_any=1 + [[ "$shc_events" -gt 0 ]] && declaring=$((declaring + 1)) + handlers=$((handlers + shc_handlers)) + ;; + absent) ;; + *) floor="$base_status" ;; + esac + + case "$dropin_status" in + present) + while IFS= read -r -d '' f; do + [[ -n "$f" ]] || continue + drops+=("$f") + done < <(find0 "$dropin_dir" -- -maxdepth 1 -type f -name '*.json') + ;; + absent | not-probed) ;; + *) + probed_floor="$dropin_status" + [[ -n "$floor" ]] || floor="$dropin_status" + note "managed-settings.d exists at $dropin_dir but could not be traversed ($dropin_status), so how many drop-in files it holds is unknown to this run. They merge on top of managed-settings.json, so the managed policy in force may carry hooks nothing above accounts for, and the managed-policy row is a floor rather than a total." + ;; + esac + + for f in ${drops[@]+"${drops[@]}"}; do + probed=$((probed + 1)) + settings_hook_counts "$f" + if [[ "$shc_status" != "present" ]]; then + [[ -n "$floor" ]] || floor="$shc_status" + note "managed drop-in $f is $shc_status, so whatever it registers is missing from the managed-policy row, which is therefore a floor rather than a total." + continue + fi + read_any=1 + [[ "$shc_events" -gt 0 ]] && declaring=$((declaring + 1)) + handlers=$((handlers + shc_handlers)) + done + + if [[ "${#drops[@]}" -gt 0 ]]; then + note "managed-settings.d exists at $dropin_dir with ${#drops[@]} drop-in file(s); they merge on top of managed-settings.json and are counted in the managed-policy row above. That row counts registrations read, so a handler written identically in two of these files is one entry in force and two there." + fi + + # Nothing was read, so there is nothing to count and the row carries the base + # file's status word exactly as a single-file row would. + if [[ "$read_any" -eq 0 ]]; then + row managed-policy "$base_status" standing "$probed" - - "$base" + return + fi + + probed_out="$probed" + [[ -n "$probed_floor" ]] && probed_out="$(add_counts "$probed" "$probed_floor")" + declaring_out="$declaring" + handlers_out="$handlers" + if [[ -n "$floor" ]]; then + declaring_out="$(add_counts "$declaring" "$floor")" + handlers_out="$(add_counts "$handlers" "$floor")" + fi + source_label="$base" + [[ "${#drops[@]}" -gt 0 ]] && source_label="$base + ${#drops[@]} drop-in(s)" + row managed-policy present standing "$probed_out" "$declaring_out" "$handlers_out" "$source_label" } -repo_root="$(git rev-parse --show-toplevel 2>/dev/null | tr -d '\r')" -[[ -n "$repo_root" ]] && cd "$repo_root" || true +plugin_hook_row() { + local manifests=() m status total_events=0 total_handlers=0 declaring=0 scanned=0 + while IFS= read -r -d '' m; do + [[ -n "$m" ]] || continue + manifests+=("$m") + done < <(find0_plugin_roots -type f -path '*/hooks/hooks.json') + scanned="${#manifests[@]}" + if [[ "${#plugin_roots[@]}" -eq 0 ]]; then + row plugin-hooks-json absent standing 0 - - "no plugin root in this repository" + return + fi + if [[ "$have_jq" -eq 0 ]]; then + row plugin-hooks-json skipped standing "$scanned" - - "$scanned hooks.json, jq unavailable" + return + fi + for m in "${manifests[@]}"; do + status="$(json_status "$m")" + if [[ "$status" != "present" ]]; then + note "plugin hooks manifest $m is $status; its handlers are missing from the plugin-hooks-json row, which is therefore a floor rather than a total." + continue + fi + local e h + if ! e="$(jq_num "$PLUGIN_EVENTS_JQ" "$m")" || ! h="$(jq_num "$PLUGIN_HANDLERS_JQ" "$m")"; then + note "plugin hooks manifest $m parses as JSON, but its hooks block is the wrong shape to read; its handlers are missing from the plugin-hooks-json row, which is therefore a floor rather than a total." + continue + fi + total_events=$((total_events + e)) + total_handlers=$((total_handlers + h)) + [[ "$e" -gt 0 ]] && declaring=$((declaring + 1)) + done + row plugin-hooks-json present standing "$scanned" "$declaring" "$total_handlers" \ + "${#plugin_roots[@]} plugin roots, $total_events event slots" +} -hook_scripts="$(count_files .claude/hooks -maxdepth 1 -name '*.sh')" -skills="$(count_files .claude/skills -mindepth 1 -maxdepth 1 -type d)" -agents="$(count_files .claude/agents -maxdepth 1 -name '*.md')" -mcp=0 -plugins=0 -if [[ -f .mcp.json ]] && command -v jq >/dev/null 2>&1; then - mcp="$(jq '.mcpServers | length' .mcp.json 2>/dev/null || echo 0)" +# scope_status is the project-scope directory that feeds this list alongside the +# plugin roots. An empty list is only an ABSENCE when that scope was actually +# walked; when the scope exists and could not be walked, the row carries its +# status word and no counts at all, because 0 there would be the same claim the +# whole file exists to stop making. +frontmatter_row() { + local label="$1" listfile="$2" source_label="$3" scope_status="$4" scanned declaring + scanned="$(count0 <"$listfile")" + if [[ "$scanned" -eq 0 ]]; then + case "$scope_status" in + present | absent) row "$label" absent conditional 0 - - "$source_label" ;; + *) row "$label" "$scope_status" conditional - - - "$source_label" ;; + esac + return + fi + declaring="$(count_frontmatter_hooks "$listfile")" + row "$label" present conditional "$scanned" "$declaring" - "$source_label" +} + +tmpdir="$(mktemp -d 2>/dev/null)" +if [[ -z "$tmpdir" || ! -d "$tmpdir" ]]; then + echo "inventory.sh: could not create a work directory; no counts were produced." >&2 + exit 2 +fi +trap 'rm -rf "$tmpdir"' EXIT + +skills_dir_status="$(dir_status .claude/skills)" +agents_dir_status="$(dir_status .claude/agents)" +hooks_dir_status="$(dir_status .claude/hooks)" +for scope_pair in "skills:$skills_dir_status" "agents:$agents_dir_status" "hooks:$hooks_dir_status"; do + case "${scope_pair#*:}" in + present | absent) ;; + *) note ".claude/${scope_pair%%:*} exists but could not be traversed (${scope_pair#*:}); whatever it holds is missing from the counts below, which are floors rather than totals for that scope." ;; + esac +done + +# Each list is filled in two passes, plugin roots first and the project scope +# second, so the two contributions stay separable. The project scope's share is +# then rendered through scope_count, which substitutes its status word when the +# scope could not be walked. A list is therefore always what was READ, and the +# total beside it always says when it is only a floor. +skill_list="$tmpdir/skills" +agent_list="$tmpdir/agents" +: >"$skill_list" +: >"$agent_list" +find0_plugin_roots -type f -name 'SKILL.md' >>"$skill_list" +find0_plugin_roots -type f -path '*/agents/*.md' >>"$agent_list" +plugin_skills="$(count0 <"$skill_list")" +plugin_agents="$(count0 <"$agent_list")" +[[ "$skills_dir_status" == "present" ]] && + find0 .claude/skills -- -type f -name 'SKILL.md' >>"$skill_list" +[[ "$agents_dir_status" == "present" ]] && + find0 .claude/agents -- -type f -name '*.md' >>"$agent_list" + +skills_read="$(count0 <"$skill_list")" +agents_read="$(count0 <"$agent_list")" +skills_total="$(add_counts "$plugin_skills" \ + "$(scope_count "$skills_dir_status" "$((skills_read - plugin_skills))")")" +agents_total="$(add_counts "$plugin_agents" \ + "$(scope_count "$agents_dir_status" "$((agents_read - plugin_agents))")")" + +printf 'Claude Code automation inventory for %s\n' "$project_root" +printf '\nHook locations (7 documented; entries MERGE across settings levels)\n' +row LOCATION STATUS KIND PROBED DECLARING HANDLERS SOURCE +settings_row user-settings "$user_settings" +settings_row project-settings "$project_settings" +settings_row local-settings "$local_settings" +if [[ "$have_managed_lib" -eq 1 ]]; then + managed_policy_row "$managed_file" "$managed_dropin" "$(dir_status "$managed_dropin")" +else + row managed-policy not-probed standing - - - "managed-scope library unavailable" + note "Managed policy was NOT probed: $managed_lib could not be read, so no per-OS path was available. This is not evidence that no policy is deployed." fi -if [[ -f .claude/settings.json ]] && command -v jq >/dev/null 2>&1; then - plugins="$(jq '[.enabledPlugins // {} | to_entries[] | select(.value == true)] | length' .claude/settings.json 2>/dev/null || echo 0)" +plugin_hook_row +frontmatter_row skill-frontmatter "$skill_list" "$skills_total SKILL.md" "$skills_dir_status" +frontmatter_row subagent-frontmatter "$agent_list" "$agents_total agent definitions" "$agents_dir_status" + +# --- Components --------------------------------------------------------------- + +hook_scripts="$(count_in_plugin_roots -type f -path '*/hooks/*' ! -name '*.json' ! -name '*.test.sh')" +project_hook_scripts="$(count_dir "$hooks_dir_status" .claude/hooks -type f ! -name '*.json' ! -name '*.test.sh')" +hook_tests="$(add_counts \ + "$(count_in_plugin_roots -type f -path '*/hooks/*' -name '*.test.sh')" \ + "$(count_dir "$hooks_dir_status" .claude/hooks -type f -name '*.test.sh')")" +# Every .mcp.json this run will look at. The project-root file is emitted +# whenever anything is THERE, a directory or a dangling symlink included, so +# json_status gets to call it unreadable: the -f test that used to gate it +# dropped such a file before any status could be assigned, and the row then +# reported the absence of a configuration that exists. +mcp_candidates() { + [[ -e .mcp.json || -L .mcp.json ]] && printf '%s\0' .mcp.json + find0_plugin_roots -maxdepth 1 -type f -name '.mcp.json' +} + +# mcp_files counts the .mcp.json files this run EXAMINED, readable or not, and +# mcp_floor carries the status word of the first one it could not measure. A +# file that exists and could not be parsed therefore leaves the server figure in +# the add_counts floor form rather than at a numeric 0: "0 across 0 .mcp.json +# file(s)" for a repository that ships an MCP configuration was the same +# confident wrong number the directory scopes already stopped printing. +mcp_servers=0 +mcp_files=0 +mcp_floor="" +if [[ "$have_jq" -eq 1 ]]; then + while IFS= read -r -d '' f; do + [[ -n "$f" ]] || continue + mcp_files=$((mcp_files + 1)) + f_status="$(json_status "$f")" + if [[ "$f_status" != "present" ]]; then + [[ -n "$mcp_floor" ]] || mcp_floor="$f_status" + note "$f is $f_status, so the MCP servers it configures could not be counted; the mcp server figure is a floor." + continue + fi + if [[ "$(jq_type '(.mcpServers // {})' "$f")" != "object" ]]; then + [[ -n "$mcp_floor" ]] || mcp_floor=invalid-json + note "$f is valid JSON, but its mcpServers key is not an object, so its servers could not be counted; the mcp server figure is a floor." + continue + fi + if ! n="$(jq_num '(.mcpServers // {}) | length' "$f")"; then + [[ -n "$mcp_floor" ]] || mcp_floor=unreadable + note "$f parses as JSON, but its mcpServers count could not be read; the mcp server figure is a floor." + continue + fi + mcp_servers=$((mcp_servers + n)) + done < <(mcp_candidates) + [[ -n "$mcp_floor" ]] && mcp_servers="$(add_counts "$mcp_servers" "$mcp_floor")" +else + mcp_files="$(mcp_candidates | count0)" + mcp_servers=skipped + note "MCP server counts were skipped: jq is not on PATH. The count below is not a measurement. The .mcp.json files beside it were found and tallied, but nothing was read from them." fi -printf 'Hook scripts: %s\n' "$hook_scripts" -printf 'Skills: %s\n' "$skills" -printf 'Agents: %s\n' "$agents" -printf 'MCP servers: %s\n' "$mcp" -printf 'Plugins enabled: %s\n' "$plugins" +printf '\nComponents in this repository\n' +printf ' plugin roots %s\n' "${#plugin_roots[@]}" +printf ' skills %s SKILL.md\n' "$skills_total" +printf ' subagents %s definitions\n' "$agents_total" +printf ' mcp servers %s across %s .mcp.json file(s)\n' "$mcp_servers" "$mcp_files" +printf ' hook scripts on disk %s in plugin hooks/ dirs, %s in .claude/hooks (+%s test scripts)\n' \ + "$hook_scripts" "$project_hook_scripts" "$hook_tests" + +# --- Enablement inputs -------------------------------------------------------- + +ENABLED_ON_JQ='[(.enabledPlugins // {}) | to_entries[] | select(.value == true)] | length' +ENABLED_OFF_JQ='[(.enabledPlugins // {}) | to_entries[] | select(.value == false)] | length' + +enablement_row() { + local label="$1" file="$2" status on off ep_type + status="$(json_status "$file")" + if [[ "$status" != "present" ]]; then + printf ' %-10s %-12s %s\n' "$label" "$status" "$file" + return + fi + ep_type="$(jq_type '(.enabledPlugins // {})' "$file")" + if [[ "$ep_type" != "object" ]]; then + printf ' %-10s %-12s %s\n' "$label" invalid-json \ + "$file (enabledPlugins holds ${ep_type:-a value jq could not type}, not an object)" + return + fi + if ! on="$(jq_num "$ENABLED_ON_JQ" "$file")" || ! off="$(jq_num "$ENABLED_OFF_JQ" "$file")"; then + printf ' %-10s %-12s %s\n' "$label" unreadable "$file" + return + fi + printf ' %-10s %-12s %s true, %s false\n' "$label" "$status" "$on" "$off" +} + +printf '\nPlugin enablement inputs (enabledPlugins follows PRECEDENCE, not merge)\n' +enablement_row user "$user_settings" +enablement_row project "$project_settings" +enablement_row local "$local_settings" +if [[ "$have_managed_lib" -eq 1 ]]; then + enablement_row managed "$managed_file" +else + printf ' %-10s %-12s %s\n' managed not-probed "managed-scope library unavailable" +fi +if [[ -f .claude-plugin/marketplace.json && "$have_jq" -eq 1 ]]; then + # `length` answers on a string as well as on an array, so a plugins key of the + # wrong type would otherwise publish a character count as an entry count. + if [[ "$(jq_type '(.plugins // [])' .claude-plugin/marketplace.json)" != "array" ]]; then + printf ' %-10s %-12s %s\n' catalog invalid-json \ + ".claude-plugin/marketplace.json (plugins is not an array)" + elif cat_total="$(jq_num '(.plugins // []) | length' .claude-plugin/marketplace.json)" && + cat_off="$(jq_num '[(.plugins // [])[] | select(.defaultEnabled == false)] | length' .claude-plugin/marketplace.json)"; then + printf ' %-10s %-12s %s entries, %s with defaultEnabled false\n' catalog present "$cat_total" "$cat_off" + else + printf ' %-10s %-12s %s\n' catalog invalid-json .claude-plugin/marketplace.json + fi +fi +printf ' No effective enablement is computed here. Run /claude-ops:plugins audit for the verdict.\n' + +# --- Notes -------------------------------------------------------------------- + +note "A hook script on disk is not a wired hook: the Components counts are files present, and only the HANDLERS column counts entries a settings file or manifest actually registers." +note "skill-frontmatter and subagent-frontmatter rows are conditional, so they are not part of the standing set: a skill's hooks register only once that skill is invoked, and a subagent's only while that subagent runs. Do not fold them into the always-on set." +note "The frontmatter rows count files that open a hooks: block; the YAML inside those blocks is not parsed, so they carry no HANDLERS figure." +note "plugin-hooks-json covers plugin roots inside this repository. Plugins installed on this machine from elsewhere also contribute hooks; run /claude-ops:inventory for the machine-scope picture." +note "Every walk prunes .git and node_modules, so .git/hooks sample scripts and vendored trees are outside these counts." +if [[ "${CLAUDE_CODE_REMOTE:-}" == "true" ]]; then + note "CLAUDE_CODE_REMOTE=true: this is a cloud session, which does not read your own machine's ~/.claude/settings.json or .claude/settings.local.json, and reaches only server-managed settings. The user-settings row above is this container's file, so the effective set here differs from a desktop session's." +else + note "A cloud session reads a different scope set than this one: it does not read local user settings or .claude/settings.local.json, and only server-managed settings reach it. An effective set measured here is not the effective set there." +fi +note "Server-managed settings, delivered remotely at sign-in, have no local path and are invisible to any local reader, including this one." + +printf '\nNotes\n' +for n in "${notes[@]}"; do + printf ' - %s\n' "$n" +done + +exit 0 diff --git a/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.test.sh b/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.test.sh index 3f1de88f71..fced1f11a8 100755 --- a/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.test.sh +++ b/plugins/claude-config/skills/audit-automation-gaps/scripts/inventory.test.sh @@ -1,5 +1,20 @@ #!/usr/bin/env bash -# Tests for inventory.sh (self-contained — ships with the plugin). +# Tests for inventory.sh (self-contained, ships with the plugin). +# +# The behaviour under test is the one the previous revision got wrong: a location +# the script could not read must report WHY, never 0. A silent zero is +# indistinguishable from a real absence, which is what made the old output +# misleading rather than merely incomplete. +# +# WHY THIS SUITE PINS EXACT NUMBERS. An earlier revision of this file asserted +# only status words and banner text, so mutation testing walked straight through +# it: pinning jq_num to 7, forcing every counting helper to find nothing, leaving +# plugin_roots empty, setting hook_scripts to 999 and skills_total to 0 all left +# the suite green. A suite that cannot fail on a wrong number does not protect +# the one property this script exists for. Every count below is therefore +# asserted against a fixture built to a known size, as an exact value and never +# as "more than zero", and each hook-location row is pinned across all five of +# its fixed columns at once by row6. set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -20,21 +35,607 @@ fail() { assert_exit() { if [[ "$2" == "$3" ]]; then pass "$1"; else fail "$1" "expected exit $2, got $3"; fi } +assert_eq() { + if [[ "$2" == "$3" ]]; then pass "$1"; else fail "$1" "expected [$2], got [$3]"; fi +} assert_contains() { case "$2" in *"$3"*) pass "$1" ;; *) fail "$1" "expected to contain: $3" ;; esac } +assert_not_contains() { + case "$2" in + *"$3"*) fail "$1" "expected NOT to contain: $3" ;; + *) pass "$1" ;; + esac +} +# Asserts one line of output verbatim, which pins the numbers inside it. +assert_line() { + if printf '%s\n' "$2" | grep -Fxq -- "$3"; then + pass "$1" + else + fail "$1" "no line exactly equal to: $3" + fi +} +# Pulls the one hook-location row whose LOCATION column is