From 05f2bb3fe96dc161a1ddb29d3590e437fe30508f Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 11:48:59 -0400 Subject: [PATCH 01/22] docs(topics): inventory usage tracking in claude.json and repo surfaces Exploration artifact only. Records the verified shape and write paths of skillUsage, pluginUsage, agentLastUsed and projects[] in ~/.claude.json, the recovered skill-listing budget scorer, and the repo surfaces that already read those counters. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 275 ++++++++++++++++++ 1 file changed, 275 insertions(+) create mode 100644 docs/topics/usage-tracking-claude-json/EXPLORE.md diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md new file mode 100644 index 0000000000..7f3556eb61 --- /dev/null +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -0,0 +1,275 @@ +# Usage tracking: what `~/.claude.json` records, and what this repo already builds on it + +Exploration only. No component was changed. Findings below are split into what was +verified against primary evidence (the live `~/.claude.json` on this machine and the +Claude Code 2.1.251 binary) and what is still open. + +Evidence basis: + +- `C:\Users\KyleSexton\.claude.json`, read 2026-08-31. +- `C:\Users\KyleSexton\.local\bin\claude.exe`, Claude Code 2.1.251, string-extracted. +- Repository at `origin/main` (949c54b0d). + +## Part 1: what Claude Code tracks in `~/.claude.json` + +`~/.claude.json` is the user-scope state file. It is a different file from +`~/.claude/settings.json` (the settings scope). It holds four usage-relevant regions. + +### 1.1 `skillUsage` (machine-global, per skill) + +Shape: `{"": {"usageCount": , "lastUsedAt": }}`. +131 entries on this machine. + +Write path in the binary (`Fdt`): + +```js +let d = o.skillUsageLastWriteAt.get(t); +if (d !== void 0 && u - d < $Mn) return; // $Mn = 60000 +o.skillUsageLastWriteAt.set(t, u), + Ae((y) => ({ ...y, skillUsage: { ...y.skillUsage, + [t]: { usageCount: (k?.usageCount ?? 0) + 1, lastUsedAt: u } } }), r) +``` + +Verified properties: + +- Incremented on real skill dispatch only. There is no install-time or session-start + seeding, so a skill's `lastUsedAt` is trustworthy evidence of actual use. +- `usageCount` is a lifetime total since install. It never resets and is never windowed. +- **A 60-second per-skill throttle DROPS the increment rather than coalescing it.** A skill + invoked five times in one minute records one. `skillUsage.usageCount` therefore + systematically undercounts bursty skills, and the undercount is unbounded and + unrecoverable. Contrast `pluginUsage` below, which batches and accumulates. +- Keys are inconsistent between qualified and bare form. This machine holds both + `babysit-prs` (378) and `source-control:babysit-prs` (97) as separate rows for the + same skill. Any consumer must sum both spellings. + +### 1.2 `pluginUsage` (machine-global, per plugin) + +Shape: `{"@": {"usageCount": , "lastUsedAt": , "lastUsedNumStartups": }}`. +118 entries on this machine. + +Write paths in the binary: + +- `Sme(e)`: batched flush, `usageCount: (d?.usageCount ?? 0) + u.count`. Accumulates, so + unlike `skillUsage` it loses nothing to throttling. +- `yNt(e,t)`: **seeds** absent entries with `usageCount: 0`, `lastUsedAt: now`, + `lastUsedNumStartups: `. +- `dzn(e,t)`: refreshes `lastUsedAt` and `lastUsedNumStartups` on re-enable, with no + usage. + +Consequence, and the binary's own bundled skill states this explicitly: for a plugin, +`lastUsedAt` is usage evidence only when `usageCount > 0`. A zero-count plugin's +`lastUsedAt` is the seed time and means nothing. + +What counts as a plugin "use" is broad. Per the binary's bundled guidance, usage is +recorded whenever a slash command, skill, agent, MCP tool or resource, or hook is +dispatched from that plugin, plus LSP servers delivering diagnostics or code navigation. +That is why hook-only plugins dominate on this machine: + +| plugin | usageCount | +| --- | --- | +| `guardrails@melodic-software` | 200,804 | +| `context-guard@melodic-software` | 106,752 | +| `disk-hygiene@melodic-software` | 100,610 | +| `source-control@melodic-software` | 68,225 | + +`pluginUsage.usageCount` is therefore not comparable across plugins of different shapes. +A hook plugin's count is a per-tool-call tally; a skill plugin's count is an invocation +tally. Ranking plugins by raw count ranks them by hook chattiness. + +`pluginUsageLspGraceAppliedIds` (top level) records which plugin ids got the LSP grace +backfill, so a lifetime zero on an LSP-only plugin may just predate the tracking. + +### 1.3 `agentLastUsed` + +`{"bg": 1784439187904}` and nothing else. Subagent usage is effectively **not** tracked +here. Nothing in this repo can source agent-usage analysis from `~/.claude.json`. + +### 1.4 `projects[]`: per-project, last session only + +28 project entries. Each carries a **snapshot of the last session in that directory**, +not a running total. Verified: `Dke(e)` builds the object fresh from live session getters +each time (`lastCost: ul(), lastAPIDuration: Xg(), ...`), and the matching telemetry event +`tengu_exit` reports these as `last_session_cost`, `last_session_api_duration`, and so on. +Each session end overwrites the previous. + +Fields: `lastCost`, `lastDuration`, `lastAPIDuration`, +`lastAPIDurationWithoutRetries`, `lastToolDuration`, `lastLinesAdded`, `lastLinesRemoved`, +`lastTotalInputTokens`, `lastTotalOutputTokens`, `lastTotalCacheCreationInputTokens`, +`lastTotalCacheReadInputTokens`, `lastTotalWebSearchRequests`, `lastSessionId`, +`lastStartTime`, `lastFpsAverage`, `lastFpsLow1Pct`, `lastSessionMetrics` (frame-duration +percentiles), `lastModelUsage` (per model id: input, output, cache-read, +cache-creation tokens, web-search requests, `costUSD`). + +**The decisive fact for per-project slicing:** `skillUsage` and `pluginUsage` are top +level only. Nothing under `projects` carries them. `~/.claude.json` structurally cannot +answer "which skills does this project use". Per-project cost and token data exists but +only for the single most recent session. + +### 1.5 Other counters, incidental + +`numStartups` (260), `promptQueueUseCount`, `btwUseCount`, `tipsHistory` and +`tipLifetimeShownCounts` (per-tip shown counts), `passesUpsellSeenCount`, +`lspRecommendationIgnoredCount`, `rcLongTurnNudgeSeenCount`. These drive tip cooldowns, +not component usage analysis. + +## Part 2: the skill-listing budget scorer, recovered exactly + +This is the highest-value find, because the repo currently calls it undocumented. + +The scorer (`zPe`): + +```js +function zPe(e) { + let r = oe().skillUsage?.[e]; + if (!r) return 0; + let o = (Date.now() - r.lastUsedAt) / 86400000, + u = Math.pow(0.5, o / 7); + return r.usageCount * Math.max(u, 0.1); +} +``` + +So the score is `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`: exponential decay with +a **7-day half-life** and a **0.1 floor**. + +Where it is used (verified call sites): + +1. `F1t` and the `skill_listing` attachment builder pass `(cmd) => zPe(cmd.name)` into + `Ymt` / `Jmt` as the priority function. This is the listing-budget truncation. +2. Slash-command menu: the top 5 by `zPe` score are pinned above the alphabetical groups. +3. Slash-command search: `getScoreBoost: (c) => zPe(c.command.name)`. + +The truncation algorithm (`Ymt`), verified: + +- Compute each entry's full length (`name + ": " + description`, description capped). +- If total fits the budget, everyone keeps their description (`budgetMode: "fits"`). +- Otherwise sort the competing entries **descending by score**, then walk the list + greedily granting descriptions while budget remains. Entries that do not fit go into + `budgetTruncatedSkills` and render as `- ` with no description + (`budgetMode: "priority"`). + +Two corrections this forces on the repo's current wording: + +- Truncation is by **decay-weighted score**, not by raw invocation count. A heavily used + but stale skill can sort *below* a lightly used but fresh one. Concretely: 100 uses 60 + days ago scores `100 * 0.1 = 10`; 12 uses today scores `12`. The stale one loses. +- The floor means a never-used skill scores exactly `0` and always loses first, but a + once-used skill never decays below `0.1 * usageCount`. + +## Part 3: what this repo already has + +### 3.1 Reads `~/.claude.json` counters directly + +`plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py` is +the only consumer of the native counters. + +- `collect_native()` (line 522) reads `--claude-json`, defaulting to `~/.claude.json`. +- `parse_native()` (line 110) turns `skillUsage` rows into events, gated by + `is_usage_evidence()` (line 68), which requires `usageCount > 0` because `lastUsedAt` + alone is not evidence. The docstring records the measurement that justified this: 46 of + 65 plugins looked "used today" while none had been. +- `collect_jsonl()` (line 536) reads the plugin's own `skill-usage.jsonl`. +- `_reconcile()` merges the two sources by taking the max per instant rather than summing, + so double-counting one dispatch seen by both sources is avoided. +- `compute_listing()` (line 703) does the budget arithmetic and the starvation banding. + +The skill's own framing already separates certain arithmetic (does the listing overflow) +from inferential ordering (which skills lose descriptions), and labels the ordering as +resting on "an undocumented scorer pinned to one build". Part 2 above closes that gap. + +### 3.2 Its own second store: `skill-usage.jsonl` + +- `plugins/claude-ops/hooks/skill-usage-audit.sh`: `PostToolUse` with `matcher: "Skill"`, + writes a `SkillUse` event per Skill tool call. +- `plugins/claude-ops/hooks/skill-usage-expansion-audit.sh`: `UserPromptExpansion`, catches + slash-command expansions the tool hook misses. +- `plugins/claude-ops/hooks/claude-ops-paths.sh`: `claude_ops::record_skill_use`, scope + selection (`repo` default, `user`, `data-dir`) via the `skill_usage_scope` userConfig. +- Registered at `plugins/claude-ops/hooks/hooks.json:64-88`. +- Schema and example: `docs/conventions/hook-telemetry/data/skill-usage-audit.schema.json`, + `docs/conventions/hook-telemetry/examples/skill-usage-audit.json`. + +Event shape, from the live store: + +```json +{"ts":"2026-08-23T18:17:02Z","event":"SkillUse","skill":"loop","branch":"unknown", + "project":"KyleSexton","project_id":"kylesexton-e2d95aff","hook":"skill-usage-audit", + "source":"expansion","expansion_type":"slash_command"} +``` + +**This store is not redundant with `~/.claude.json`. It carries the slices the native +counters structurally lack**: `project`, `project_id`, `branch`, `source` (tool vs +expansion), `expansion_type`, and a real timestamp per event rather than one +last-used stamp. It is also immune to the 60-second throttle. The native counter owns the +global lifetime tally; the JSONL owns the sliced event stream. + +### 3.3 Other usage-adjacent surfaces + +- `plugins/claude-ops/skills/observability/`: OTEL DuckDB store, collector, hook-event + JSONL, `ccusage`, cross-session trend reports. +- `plugins/claude-ops/skills/audit-install-state/scripts/install_state.py`: reads + `~/.claude.json` for install-tree inventory, not usage. +- `plugins/claude-ops/skills/audit-performance/scripts/audit_performance.py`: reads + `~/.claude.json` for retention-sweep health and session counts. +- `plugins/claude-config/skills/audit-permission-state/`: reads `~/.claude.json` as a + settings scope, distinct from `~/.claude/settings.json`. +- `plugins/context-budget/skills/audit/reference/levers.json`: `~/.claude.json` as a lever + source. +- `docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md`: the + standing decision that skill recommendation reads the catalog, not the in-context + listing, precisely because the listing is budget-truncated. + +### 3.4 What Claude Code itself now ships + +The 2.1.251 binary contains a bundled skill that documents these exact counters and their +traps, in prose closely matching this repo's own conclusions ("`usageCount` is a LIFETIME +total since install", the `pluginUsage` seeding caveat, the qualified-vs-bare key split). +It also ships a `doctor`-side check: "Check 1: unused skills, MCP servers, and plugins", +which groups unused components against their context cost and offers to disable them +("37 unused skills, saves ~2.2k est. tokens/session"). `audit-skill-visibility`'s SKILL.md +already disclaims that one-shot check as native territory. + +## Part 4: open findings, unverified or needing a decision + +1. **`compute_listing` ranks on a `usage_score` nothing populates.** `compute_listing` + (line 703) reads `entry.get("usage_score", 0)` off the denominator, and sorts + `competing` by it (line 736). The denominator is built by `collect_installed` / + `collect_fleet_at` from a filesystem walk, which never sets `usage_score`. Usage events + are joined to entries later, inside `classify` (line 915 onward), after + `compute_listing` has already run (line 897). Net effect on a real run: every row scores + 0 and the starvation band is ordered alphabetically. Only the test fixture + `tests/fixtures/fleet-overbudget.json` supplies non-zero values. Needs confirmation on + a live run before it is called a defect, but the code path reads that way. + +2. **`skill-usage.jsonl` has been dead in this repo since 2026-08-11.** The main checkout's + `.claude/observability/skill-usage.jsonl` last changed 2026-08-11 19:34 (107 lines) + while `hook-events.jsonl` in the same directory is live (2026-08-31 11:27). The last + `skill-usage-audit` telemetry envelope is `2026-08-11T23:34:24Z`. The hook is registered, + the plugin is enabled in `.claude/settings.json`, no kill switch is set, and the same + hook did write to `~/.claude/observability/` as recently as 2026-08-23 from a + home-directory session. No worktree under `D:\worktrees` holds a `skill-usage.jsonl` at + all. Cause not established. + +3. **Nothing consumes `agentLastUsed`, and it holds one key.** Any agent-usage question has + to come from transcripts or the OTEL store. + +4. **No consumer accounts for the 60-second skill throttle.** `audit_skill_visibility.py` + treats `usageCount` as a count of dispatches. For loop-driven or rapid-fire skills that + is a floor, not a count. The JSONL store is the accurate source for those. + +5. **No consumer accounts for the qualified-vs-bare key split.** Live data has + `babysit-prs` and `source-control:babysit-prs` as separate rows. + +## Part 5: recommendation shape + +Do not rebuild a counter. `~/.claude.json` already owns the global lifetime tally, and its +scorer is now known exactly. The gaps worth closing are: + +- Encode `zPe` in `audit_skill_visibility.py` and populate `usage_score` before + `compute_listing` runs (Part 4 finding 1, plus Part 2's decay formula). +- Sum qualified and bare `skillUsage` keys. +- Keep `skill-usage.jsonl` as the per-project / per-branch / per-source slice, and fix + whatever stopped it (Part 4 finding 2). That is the only local source for the granular + slices the user asked about. +- Treat `pluginUsage.usageCount` as non-comparable across plugin shapes wherever it is + ranked. From 26e1bf19f3c36524a5f712bd97862ba734099003 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 11:57:33 -0400 Subject: [PATCH 02/22] docs(topics): confirm the starvation band carries no usage signal A live run of audit_skill_visibility.py returns usage_score 0 for every row, so the band falls back to its alphabetical tiebreaker. Records the run's numbers and the inverted example rows. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 38 ++++++++++++++----- 1 file changed, 29 insertions(+), 9 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index 7f3556eb61..a22986a6e2 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -231,15 +231,35 @@ already disclaims that one-shot check as native territory. ## Part 4: open findings, unverified or needing a decision -1. **`compute_listing` ranks on a `usage_score` nothing populates.** `compute_listing` - (line 703) reads `entry.get("usage_score", 0)` off the denominator, and sorts - `competing` by it (line 736). The denominator is built by `collect_installed` / - `collect_fleet_at` from a filesystem walk, which never sets `usage_score`. Usage events - are joined to entries later, inside `classify` (line 915 onward), after - `compute_listing` has already run (line 897). Net effect on a real run: every row scores - 0 and the starvation band is ordered alphabetically. Only the test fixture - `tests/fixtures/fleet-overbudget.json` supplies non-zero values. Needs confirmation on - a live run before it is called a defect, but the code path reads that way. +1. **CONFIRMED: `compute_listing` ranks on a `usage_score` nothing populates, so the + starvation band is alphabetical.** `compute_listing` (line 703) reads + `entry.get("usage_score", 0)` off the denominator and sorts `competing` by + `(usage_score, qualified_name)` (line 736). The denominator is built by + `collect_installed` / `collect_fleet_at` from a filesystem walk, which never sets + `usage_score`. Usage events are joined to entries later, inside `classify` (line 915 + onward), after `compute_listing` has already run (line 897). Only the test fixture + `tests/fixtures/fleet-overbudget.json` supplies non-zero values, which is why the tests + do not catch this. + + Verified on a live run + (`--plugins-root plugins --claude-json ~/.claude.json --render json`, + 2026-08-31): `[.skills[].starvation.usage_score] | unique` returns `[0]`. Every row + scores zero, so the tiebreaker decides everything and bands come out in name order: + + | band | usage_score | observed count | skill | + | --- | --- | --- | --- | + | 1 | 0 | 1 | `adhd:clarify` | + | 2 | 0 | 1 | `adhd:shape` | + | 5 | 0 | 8 | `architecture:improve` | + | 150 | 0 | 97 | `source-control:babysit-prs` | + | 173 | 0 | 99 | `work-items:triage` | + + Band 1 is "most likely starved". The two least-used skills in the fleet are ranked as + the first to lose their descriptions and the two most-used are ranked as the safest, + but only because `a` sorts before `w`. The report's own inferential ordering, the part + it warns is uncertain, currently carries no usage signal at all. Same run: + `budget_chars` 8000, `demand_chars` 130330, `overflow_chars` 122330, 167 of 176 + competing skills marked `likely-starved`. 2. **`skill-usage.jsonl` has been dead in this repo since 2026-08-11.** The main checkout's `.claude/observability/skill-usage.jsonl` last changed 2026-08-11 19:34 (107 lines) From 0086027d909d0e3c0ba653c18b45d3e8000755a4 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 12:00:41 -0400 Subject: [PATCH 03/22] docs(topics): record the surfaces that do not read the native counters The observability reporter's lanes omit ~/.claude.json; nothing reads agentLastUsed or the per-project cost snapshot; pair co-occurrence needs the event stream the counters cannot supply. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 27 ++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index a22986a6e2..1d2a84bb3c 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -219,7 +219,32 @@ global lifetime tally; the JSONL owns the sliced event stream. standing decision that skill recommendation reads the catalog, not the in-context listing, precisely because the listing is budget-truncated. -### 3.4 What Claude Code itself now ships +### 3.4 The observability skill does not read the native counters + +`plugins/claude-ops/skills/observability/SKILL.md:121-122` and +`context/data-sources.md` enumerate its lanes: ccusage (tokens, cost, billing blocks), +the OTEL DuckDB store, and `.claude/observability/hook-events.jsonl` (hook duration, exit +codes). `~/.claude.json` is not among them, so the machine-global lifetime counters have +no representation in the cross-session trend reports. Its `clean` action does know about +the skill-usage store (`--skill-usage-scope`, `--keep-skill-usage-days`, default 365), but +only to prune it, never to read it. + +`plugins/claude-ops/skills/audit-skill-visibility/scripts/skill-pair-cooccurrence.sh` and +`reference/pair-cooccurrence.md` derive which skills get invoked together from +`skill-usage.jsonl`. That analysis is impossible from `~/.claude.json`, which has no event +stream, and is a second reason the JSONL store earns its place. + +### 3.5 What no surface covers + +- No skill reasons about plugin disuse from `pluginUsage`. `claude-ops:plugins` is version + and scope currency only; its SKILL.md and scripts never mention usage. + `overengineering:audit` reasons about enforcement surfaces earning their keep but sources + evidence from CI and hook behavior, not from these counters. +- Nothing reads `agentLastUsed`. +- Nothing reads `projects[].lastModelUsage` or the per-project cost and token + snapshot, despite that being the only per-project slice `~/.claude.json` offers. + +### 3.6 What Claude Code itself now ships The 2.1.251 binary contains a bundled skill that documents these exact counters and their traps, in prose closely matching this repo's own conclusions ("`usageCount` is a LIFETIME From 718c262fb7785a4aa623b545e88c109b53228608 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 12:07:00 -0400 Subject: [PATCH 04/22] docs(topics): stamp the scorer claim and soften the dead-hook finding Adds the Claim/Basis/As-of/Recheck stamp for the binary-derived listing scorer and the fallback rule for a build mismatch. Restates the skill-usage.jsonl finding as what was checked and what remains unresolved. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 46 +++++++++++++++---- 1 file changed, 38 insertions(+), 8 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index 1d2a84bb3c..79dd625119 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -156,6 +156,27 @@ Two corrections this forces on the repo's current wording: - The floor means a never-used skill scores exactly `0` and always loses first, but a once-used skill never decays below `0.1 * usageCount`. +### Verification stamp + +Follows `docs/conventions/upstream-drift`, the same shape +`plugins/source-control/hooks/worktree-create-gate.sh` uses for binary-derived claims. + +- **Claim:** the scorer is `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`; it is the + priority function passed to the listing-budget truncator; truncation sorts competing + entries descending by that score and drops descriptions from the tail. +- **Basis:** string extraction of `claude.exe`, Claude Code 2.1.251, functions `zPe`, + `Ymt`, `F1t`, `Fdt`, `Sme`, `yNt`, `dzn`, `Dke`. Confirmed against the live + `~/.claude.json` on this machine. +- **As-of:** 2026-08-31, Claude Code 2.1.251. +- **Recheck trigger:** any release note naming the skill listing, the skill-listing budget, + skill usage counters, or `/doctor`'s unused-component check; or the counters' shape in + `~/.claude.json` gaining or losing a field. + +This is recovered from one build of a minified bundle. Encoding it in a script means +pinning to that build, so any consumer must carry the stamp and treat a mismatch as +"scorer unknown", falling back to the current name-ordered behavior rather than asserting +a wrong ordering. + ## Part 3: what this repo already has ### 3.1 Reads `~/.claude.json` counters directly @@ -286,14 +307,23 @@ already disclaims that one-shot check as native territory. `budget_chars` 8000, `demand_chars` 130330, `overflow_chars` 122330, 167 of 176 competing skills marked `likely-starved`. -2. **`skill-usage.jsonl` has been dead in this repo since 2026-08-11.** The main checkout's - `.claude/observability/skill-usage.jsonl` last changed 2026-08-11 19:34 (107 lines) - while `hook-events.jsonl` in the same directory is live (2026-08-31 11:27). The last - `skill-usage-audit` telemetry envelope is `2026-08-11T23:34:24Z`. The hook is registered, - the plugin is enabled in `.claude/settings.json`, no kill switch is set, and the same - hook did write to `~/.claude/observability/` as recently as 2026-08-23 from a - home-directory session. No worktree under `D:\worktrees` holds a `skill-usage.jsonl` at - all. Cause not established. +2. **No `skill-usage.jsonl` writes observed in this repo since 2026-08-11. Cause not + established.** The main checkout's `.claude/observability/skill-usage.jsonl` last + changed 2026-08-11 19:34 (107 lines) while `hook-events.jsonl` in the same directory is + live (2026-08-31 11:27). The last `skill-usage-audit` telemetry envelope is + `2026-08-11T23:34:24Z`. No worktree under `D:\worktrees` holds a `skill-usage.jsonl` at + all. + + Checked and ruled out: the hook is registered + (`plugins/claude-ops/hooks/hooks.json:64-75`), `claude-ops@melodic-software` is enabled + in `.claude/settings.json:36`, and no `skill_usage` key (scope or kill switch) appears + in `~/.claude/settings.json`, `.claude/settings.local.json`, or `~/.claude.json`. + + Not ruled out: the same hook did write to `~/.claude/observability/` as recently as + 2026-08-23, from a session whose resolved repo root was the home directory + (`"project":"KyleSexton"`). So the hook itself works. Whether this repo's sessions + stopped dispatching the `Skill` tool, or the write is landing somewhere unexpected, is + unresolved and needs a live probe rather than more file archaeology. 3. **Nothing consumes `agentLastUsed`, and it holds one key.** Any agent-usage question has to come from transcripts or the OTEL store. From 20c9b109f94bb853518c9c0a2c0a3cec1bbf2b18 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 12:15:40 -0400 Subject: [PATCH 05/22] docs(topics): add the OTEL lane, transcript lane and standing constraints Records claude_code.skill_activated's invocation_trigger as the only source that separates user-invoked from model-invoked, the retro transcript parser as the only working agent-usage source, and the three codified refusals: ADR 0016's deferral of usage-driven surfacing, the performance engine's stat-only posture on ~/.claude.json, and the exposure-floor withheld verdict. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 66 ++++++++++++++++++- 1 file changed, 64 insertions(+), 2 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index 79dd625119..488e000d26 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -265,7 +265,62 @@ stream, and is a second reason the JSONL store earns its place. - Nothing reads `projects[].lastModelUsage` or the per-project cost and token snapshot, despite that being the only per-project slice `~/.claude.json` offers. -### 3.6 What Claude Code itself now ships +### 3.6 A third usage source: OTEL `claude_code.skill_activated` + +`plugins/claude-ops/skills/audit-skill-visibility/reference/pair-cooccurrence.md:55-59` +records a signal neither `~/.claude.json` nor `skill-usage.jsonl` carries: the OTEL event +`claude_code.skill_activated` has an `invocation_trigger` attribute separating `user-slash` +from `claude-proactive`. That is the axis a "does the model reach for this unprompted" +question actually needs, and neither counter can answer it. The tier model already gates it +as `T-full` only. + +The same file (`:50-54`) records why caller attribution cannot be recovered by widening the +hook: a `PostToolUse` hook on the `Skill` tool receives `tool_name`, `tool_input` and +`tool_response`, and none of them names the skill whose instructions caused the call. +Caller identity is absent from the hook's input, not merely from its schema, so a wider +write would have nothing to write. Recovering it means reading the session transcript, +which crosses the boundary `plugins/claude-ops/skills/observability/context/privacy.md` +guards. + +### 3.7 Transcript scraping, which is where agent usage actually lives + +`plugins/session-flow/skills/retro/scripts/parse_transcript.py` parses Claude Code session +transcripts for quantitative metrics, with multi-session aggregation and handoff-chain +walking. Per `plugins/session-flow/skills/retro/context/session.md:98` it extracts +compactions, total context tokens, tool rejections, **subagent count**, and a tool +distribution breakdown. `plugins/session-flow/skills/running-retro/scripts/observer.py` +runs the same analysis in flight or detached, appending to a cumulative ledger. + +Given that `agentLastUsed` holds one key, this transcript lane is the only working source +for agent usage in the repo. + +### 3.8 Governance constraints already recorded + +Three refusals are already codified. Any implementation should start from them rather than +re-derive them. + +- **`docs/adr/0016-...:118-120`** deliberately defers usage-metrics-driven surfacing: + "Usage-metrics-driven surfacing (`~/.claude.json` `skillUsage`, undocumented internal + state) stays deferred; rotation runs off a ledger the skill writes itself, which is what + keeps that deferral honest rather than load-bearing." This constrains Part 5: encoding + `zPe` is a diagnostic-report change, not a licence to route skill *recommendation* off + these counters. +- **Secret safety, scoped to the performance engine.** + `plugins/claude-ops/skills/audit-performance/scripts/audit_performance.py:17-19` allowlists + content reads to `settings.json`, `.last-cleanup`, `hooks.json` and + `installed_plugins.json`, and holds `~/.claude.json` and `history.jsonl` stat-only + "whose values can carry tokens and prompts". Asserted by + `test_audit_performance.py:42-49`. `audit-install-state` takes the same stat-only posture + (`install_state.py:1082-1109`). This is a per-engine safety rule, not a repo-wide ban: + `audit-skill-visibility` reads the file deliberately, and reads only `firstStartTime` and + `skillUsage`. +- **A short observation horizon must yield a withheld verdict, not a zero.** + `audit_skill_visibility.py:53-65` sets `exposure_floor_days = 30`; a three-day-old install + measured against 30 and 90 day tiers put 210 of 213 skills in a "never used" bucket. Every + window clamps to `observed_horizon` and unsupported claims route to a first-class + `withheld` section. + +### 3.9 What Claude Code itself now ships The 2.1.251 binary contains a bundled skill that documents these exact counters and their traps, in prose closely matching this repo's own conclusions ("`usageCount` is a LIFETIME @@ -338,7 +393,14 @@ already disclaims that one-shot check as native territory. ## Part 5: recommendation shape Do not rebuild a counter. `~/.claude.json` already owns the global lifetime tally, and its -scorer is now known exactly. The gaps worth closing are: +scorer is now known exactly. + +Scope limit first: ADR 0016 defers usage-metrics-driven *surfacing*. Everything below is +diagnostic reporting inside `audit-skill-visibility`, which already reads these counters by +design. None of it routes skill recommendation off them, and doing that would need the ADR +revisited rather than worked around. + +The gaps worth closing are: - Encode `zPe` in `audit_skill_visibility.py` and populate `usage_score` before `compute_listing` runs (Part 4 finding 1, plus Part 2's decay formula). From 53eb2f90113038e0a992f5016420b27e7a2dd3fe Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 12:25:56 -0400 Subject: [PATCH 06/22] docs(topics): correct the OTEL and throttle claims, confirm the key-split drop The audit is a three-source reconciler with a capability tier model, not a native-counter reader, so parse_otel and TIER_CAPABILITIES replace the earlier claim that OTEL appears only in a reference file. The 60s debounce was already documented at SKILL.md:194-196, so it is recorded as not-a-gap rather than a finding. The qualified-vs-bare key split is confirmed: classify() looks events up by qualified_name alone, so bare-key rows are discarded silently. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 74 ++++++++++++++----- 1 file changed, 57 insertions(+), 17 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index 488e000d26..d04b783a56 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -113,6 +113,18 @@ only for the single most recent session. `lspRecommendationIgnoredCount`, `rcLongTurnNudgeSeenCount`. These drive tip cooldowns, not component usage analysis. +### 1.6 Growth and the one supported shrink lever + +The file is never swept: `cleanupPeriodDays` does not reach it, since it lives in the home +directory rather than under `~/.claude` +(`plugins/claude-ops/skills/audit-install-state/reference/surfaces.md:104`). The supported +lever is `claude project purge `, which removes one project's entry +(`surfaces.md:108`; `audit-performance/SKILL.md:34`). Every running session polls the file +at 1 Hz (`known-performance-issues.md:199`), and the strongest public report of curing +input lag pruned this file rather than the tree (`:44`). `.claude.json.tmp..` +siblings are failed atomic-write remnants; the leading number only looks like a PID +(`surfaces.md:110`, `install_state.py:1082-1109`). + ## Part 2: the skill-listing budget scorer, recovered exactly This is the highest-value find, because the repo currently calls it undocumented. @@ -246,7 +258,9 @@ global lifetime tally; the JSONL owns the sliced event stream. `context/data-sources.md` enumerate its lanes: ccusage (tokens, cost, billing blocks), the OTEL DuckDB store, and `.claude/observability/hook-events.jsonl` (hook duration, exit codes). `~/.claude.json` is not among them, so the machine-global lifetime counters have -no representation in the cross-session trend reports. Its `clean` action does know about +no representation in the cross-session trend reports. `audit-skill-visibility` is the only +skill that joins the native counters to the OTEL and JSONL lanes; the observability +reporter sees two of the three. Its `clean` action does know about the skill-usage store (`--skill-usage-scope`, `--keep-skill-usage-days`, default 365), but only to prune it, never to read it. @@ -265,17 +279,33 @@ stream, and is a second reason the JSONL store earns its place. - Nothing reads `projects[].lastModelUsage` or the per-project cost and token snapshot, despite that being the only per-project slice `~/.claude.json` offers. -### 3.6 A third usage source: OTEL `claude_code.skill_activated` +### 3.6 The audit is a three-source reconciler with a capability tier model + +`audit_skill_visibility.py` does not merely read the native counters. It has three parse +lanes and gates every claim on which lanes are present: + +- `:110-133` `parse_native()`, `~/.claude.json` `skillUsage`, horizon `firstStartTime`. +- `:136-161` `parse_jsonl()`, the plugin's own `skill-usage.jsonl`. +- `:164-187` `parse_otel()`, the OTEL event `claude_code.skill_activated`, which carries + `invocation_trigger`. Flags the `custom_skill` redaction placeholder and leaves it + unattributed. -`plugins/claude-ops/skills/audit-skill-visibility/reference/pair-cooccurrence.md:55-59` -records a signal neither `~/.claude.json` nor `skill-usage.jsonl` carries: the OTEL event -`claude_code.skill_activated` has an `invocation_trigger` attribute separating `user-slash` -from `claude-proactive`. That is the axis a "does the model reach for this unprompted" -question actually needs, and neither counter can answer it. The tier model already gates it -as `T-full` only. +`TIER_CAPABILITIES` (`:81-88`) with `resolve_tier()` (`:93-101`): -The same file (`:50-54`) records why caller attribution cannot be recovered by widening the -hook: a `PostToolUse` hook on the `Skill` tool receives `tool_name`, `tool_input` and +| tier | source | claims supported | +| --- | --- | --- | +| `T-full` | otel | `invocation_trigger`, `windowed_count`, `per_repo`, `lifetime_count` | +| `T-local` | jsonl | `windowed_count`, `per_repo`, `lifetime_count` | +| `T-baseline` | native | `lifetime_count` only | +| `T-none` | none | nothing | + +The `T-baseline` restriction encodes exactly the Part 1 finding: native counters are +lifetime-since-install and never windowed, so no windowed claim is honest from them alone. + +`invocation_trigger` separates `user-slash` from `claude-proactive`. That is the axis a +"does the model reach for this unprompted" question needs, and neither counter can answer +it. `reference/pair-cooccurrence.md:50-54` records why the gap cannot be closed by widening +the hook: a `PostToolUse` hook on the `Skill` tool receives `tool_name`, `tool_input` and `tool_response`, and none of them names the skill whose instructions caused the call. Caller identity is absent from the hook's input, not merely from its schema, so a wider write would have nothing to write. Recovering it means reading the session transcript, @@ -383,12 +413,21 @@ already disclaims that one-shot check as native territory. 3. **Nothing consumes `agentLastUsed`, and it holds one key.** Any agent-usage question has to come from transcripts or the OTEL store. -4. **No consumer accounts for the 60-second skill throttle.** `audit_skill_visibility.py` - treats `usageCount` as a count of dispatches. For loop-driven or rapid-fire skills that - is a floor, not a count. The JSONL store is the accurate source for those. - -5. **No consumer accounts for the qualified-vs-bare key split.** Live data has - `babysit-prs` and `source-control:babysit-prs` as separate rows. +4. **Not a gap: the 60-second throttle is already documented.** `SKILL.md:194-196` states + it exactly, including that the debounce suppresses the timestamp refresh too, and rules + that OTEL and native divergence "must not be reconciled away". The binary read in Part + 1.1 corroborates the repo's claim independently: `Fdt` returns before both the count and + the timestamp write. Recorded here so a future pass does not re-file it as a finding. + +5. **CONFIRMED: the qualified-vs-bare key split silently drops events.** Denominator + entries are keyed `f"{plugin}:{leaf}"` (`:276`) and `classify()` looks events up by that + `qualified_name` alone (`:926`, `events_by_skill.get(name)`). `parse_native()` emits + events under whatever raw key `skillUsage` holds, so a bare-key row never matches its + qualified entry and is discarded without a withheld note. Live data holds `babysit-prs` + (378) and `source-control:babysit-prs` (97) as separate rows; the audit run in finding 1 + reported `source-control:babysit-prs` at 97, not 475. Claude Code's own lookup helper + (`oKn` in the binary) checks the qualified key then falls back to the bare one, which is + the behavior to match. ## Part 5: recommendation shape @@ -404,7 +443,8 @@ The gaps worth closing are: - Encode `zPe` in `audit_skill_visibility.py` and populate `usage_score` before `compute_listing` runs (Part 4 finding 1, plus Part 2's decay formula). -- Sum qualified and bare `skillUsage` keys. +- Resolve qualified and bare `skillUsage` keys to one entry, matching `oKn`'s + qualified-then-bare fallback rather than summing blindly (Part 4 finding 5). - Keep `skill-usage.jsonl` as the per-project / per-branch / per-source slice, and fix whatever stopped it (Part 4 finding 2). That is the only local source for the granular slices the user asked about. From c24d8b16f75f43f8a8fefece0c8db6f4b205c604 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 14:52:39 -0400 Subject: [PATCH 07/22] docs(topics): route native overlap to its registry, stop instructing no re-derivation Section 3.9 was doing audit-native-overlap's job in prose, a silent second way alongside docs/native-surfaces/records.json. It now raises the missing doctor -> audit-skill-visibility row as a candidate and leaves the verdict to the human gate that skill's contract requires. Part 3.8 told the reader to start from the codified refusals "rather than re-derive them", which is the inertia this repo's incumbency discipline exists to catch. It now separates the two refusals that name a purpose and a measurement from the one that names only the state of the substrate. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 44 ++++++++++++++----- 1 file changed, 33 insertions(+), 11 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index d04b783a56..0439af78d4 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -326,8 +326,10 @@ for agent usage in the repo. ### 3.8 Governance constraints already recorded -Three refusals are already codified. Any implementation should start from them rather than -re-derive them. +Three refusals are already codified. Each is recorded below with the rationale it actually +stands on, so a reader can tell a live justification from inertia. Two name a purpose and a +measurement; the third names only the state of the substrate, and that substrate moved this +session, so it is re-derived in Part 5 rather than obeyed on sight. - **`docs/adr/0016-...:118-120`** deliberately defers usage-metrics-driven surfacing: "Usage-metrics-driven surfacing (`~/.claude.json` `skillUsage`, undocumented internal @@ -350,15 +352,35 @@ re-derive them. window clamps to `observed_horizon` and unsupported claims route to a first-class `withheld` section. -### 3.9 What Claude Code itself now ships - -The 2.1.251 binary contains a bundled skill that documents these exact counters and their -traps, in prose closely matching this repo's own conclusions ("`usageCount` is a LIFETIME -total since install", the `pluginUsage` seeding caveat, the qualified-vs-bare key split). -It also ships a `doctor`-side check: "Check 1: unused skills, MCP servers, and plugins", -which groups unused components against their context cost and offers to disable them -("37 unused skills, saves ~2.2k est. tokens/session"). `audit-skill-visibility`'s SKILL.md -already disclaims that one-shot check as native territory. +### 3.9 Native overlap: a candidate for the existing registry, not a verdict here + +Native-overlap verdicts are owned by `/claude-ops:audit-native-overlap`, recorded in +`docs/native-surfaces/records.json` and rendered into `docs/NATIVE-SURFACES.md`. This +section raises a candidate for that machinery rather than deciding it, because verdicts +there are human-gated by contract. + +Observed in the 2.1.251 binary: + +- A bundled skill documenting these exact counters and their traps, in prose closely + matching this repo's own conclusions: "`usageCount` is a LIFETIME total since install", + the `pluginUsage` seeding caveat, and the qualified-vs-bare key split. +- A `doctor`-side check, "Check 1: unused skills, MCP servers, and plugins", grouping + unused components against their context cost and offering to disable them, labelled per + group with a benefit estimate ("37 unused skills, saves ~2.2k est. tokens/session"). + +The store already holds `doctor` to `claude-ops:audit-install-state` and `doctor` to +`claude-ops:audit-performance`, both `complementary`, verified 2026-08-23. It holds **no +row for `doctor` to `claude-ops:audit-skill-visibility`**, even though `doctor`'s Check 1 +and that skill answer overlapping questions, and even though the skill's own SKILL.md +already disclaims the one-shot check as native territory in prose. Prose disclaimer without +a store row is exactly the drift the registry exists to catch. + +Suggested framing for the human deciding it: the native check is a one-shot +unused-versus-context-cost prompt that offers to disable; `audit-skill-visibility` is a +three-source reconciler with a capability tier model, a withheld-verdict discipline, and a +listing-budget starvation analysis, and it disables nothing. That reads `complementary` on +the same shape as the two existing `doctor` rows, but the verdict is not this document's to +record. ## Part 4: open findings, unverified or needing a decision From 313da8f1c628fe15a7c0eef9430ed7b212b202eb Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 15:00:27 -0400 Subject: [PATCH 08/22] docs(topics): re-derive where ADR 0016 reaches, correcting an over-read both ways A cross-vendor re-derivation run blind to the earlier reasoning found the "scope limit" framing wrong in both directions. The deferral is scoped to show-options rotation, so encoding zPe inside audit-skill-visibility was never inside it and does not need to be framed as an exception. But the deferral is also not up for lifting: its real ground is that zPe is wrong-signed for a forgotten-skill nudge, that skillUsage cannot name never-invoked skills, and that it carries no take-up attribution. Records the lift conditions and the reason a binary-derived stamp is not what the ADR meant by documented. Also drafts, without applying, the correction owed to the ADR's Context line about drop ordering. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 65 +++++++++++++++++-- 1 file changed, 59 insertions(+), 6 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index 0439af78d4..e3b294c214 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -456,12 +456,50 @@ record. Do not rebuild a counter. `~/.claude.json` already owns the global lifetime tally, and its scorer is now known exactly. -Scope limit first: ADR 0016 defers usage-metrics-driven *surfacing*. Everything below is -diagnostic reporting inside `audit-skill-visibility`, which already reads these counters by -design. None of it routes skill recommendation off them, and doing that would need the ADR -revisited rather than worked around. - -The gaps worth closing are: +### Where ADR 0016 does and does not reach + +An earlier draft of this section treated ADR 0016 as constraining the work below and framed +it as a cautious exception. A cross-vendor re-derivation, run blind to that reasoning, +found the framing an over-read, and the correction runs in both directions. + +The deferral at `docs/adr/0016-...:118-120` is scoped to one skill's rotation: which of +`show-options`'s Spotlight three to surface. Encoding `zPe` inside `audit-skill-visibility` +to predict the listing's own truncation is a different skill answering a different question, +"what will the listing drop", not "what should we recommend". It was never inside the +clause. The work below stands on its own ground rather than as a permitted exception. + +In the other direction, the deferral is not up for lifting just because its stated premise +moved. Its ground shifts from "undocumented internal state" to something documentation +cannot cure: + +- `zPe` is **wrong-signed** for the question `show-options` asks. It scores high for skills + used recently and often, which are exactly the skills the operator has not forgotten. + Inverting it collapses to a decay-weighted least-recently-used ordering, which the + self-written ledger already supplies without a build-pinned dependency. +- `skillUsage` lists only skills that have fired at least once (131 entries here). It + structurally cannot name the never-invoked skills that `show-options`'s no-omission rule + exists to protect. +- It carries no causal-trigger field, so it cannot answer take-up: was a skill invoked + *because* it was surfaced, or for an unrelated reason. `skill-usage.jsonl` can. That alone + means the ledger is not a stopgap for missing documentation; it measures something these + counters never will. + +And the stamp in Part 2 is not the same thing the ADR meant by documented. The fallback rule +recorded there, treat a mismatch as "scorer unknown" and degrade to name ordering, is what +ships alongside ground that can shift without notice, not alongside a documented API. The +recheck trigger fires when a human notices a release note, so the failure mode is silent and +wrong until someone reads a changelog. `docs/conventions/upstream-drift` existing as a named +convention is this repo's own admission that binary-derived claims are a weaker evidentiary +class. This session made the counters' undocumented-ness precisely characterized and dated. +That is not the same as making them documented. + +Conditions under which the deferral would genuinely be revisitable, for whoever comes back +to it: a published, versioned surface for skill-usage data rather than a reverse-engineered +internal; a rotation signal not decay-weighted toward recent use; and take-up attribution. +The third is unreachable from `skillUsage` by construction, so the ledger survives whatever +happens to the first two. + +### The gaps worth closing - Encode `zPe` in `audit_skill_visibility.py` and populate `usage_score` before `compute_listing` runs (Part 4 finding 1, plus Part 2's decay formula). @@ -472,3 +510,18 @@ The gaps worth closing are: slices the user asked about. - Treat `pluginUsage.usageCount` as non-comparable across plugin shapes wherever it is ranked. + +### One correction owed to ADR 0016, drafted not applied + +`docs/adr/0016-...:19-20` states that Claude Code "drops descriptions starting with the +skills invoked least". Part 2 shows the ordering is decay-weighted, so a heavily used but +stale skill can lose its description before a lightly used fresh one. The stated mechanism +is wrong, not merely imprecise, and the harm is not hypothetical: Part 4 finding 1 shows +`compute_listing` sorting on a `usage_score` nothing populates, so its starvation bands are +alphabetical order presented as usage-informed. Someone already built on the mental model +that ADR line encodes. + +The ADR's core decision is untouched and in fact reinforced, so the fix is a dated revision +blockquote in the ADR's own established shape (`:75-84`, `:98-105`), which preserves +superseded reasoning rather than editing Context in place. Amending an accepted decision +record is the human's call, so it is drafted here and not applied. From d53b796622f0a326aad66ad1b310986135be1a47 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 15:18:06 -0400 Subject: [PATCH 09/22] docs(adr): correct 0016's drop-order mechanism and restate its deferral ground Two dated revision blockquotes in the ADR's own established shape, which preserves superseded reasoning rather than editing Context in place. The Context paragraph stated that Claude Code drops descriptions "starting with the skills invoked least". The mechanism is a decay-weighted score, usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1), sorted descending and granted greedily, so a heavily used but stale skill can be shed before a lightly used fresh one. The correction strengthens the decision rather than weakening it: a never-invoked skill still scores zero and is shed first, and the decay term adds a second bias against exactly the forgotten-skill population show-options exists to surface. The deferral clause rested on skillUsage being "undocumented internal state". That substrate is now characterized and dated, which invites a false lift, so the ground is restated on three reasons documentation cannot cure: the scorer is wrong-signed for the question, skillUsage cannot name the never-invoked population, and it carries no take-up attribution. Lift conditions recorded. Core decision untouched in both cases. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- ...dation-from-the-catalog-not-the-listing.md | 42 +++++++++++++++++++ .../usage-tracking-claude-json/EXPLORE.md | 22 ++++++++-- 2 files changed, 60 insertions(+), 4 deletions(-) diff --git a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md index fb8a5f206e..3d9bf6c61f 100644 --- a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md +++ b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md @@ -27,6 +27,22 @@ the wrong direction, and — the part that makes it a correctness bug rather tha cannot tell that it is blind. The gatekeeping the contract bans would have been reinstated by the harness, invisibly. +> **Revised 2026-08-31:** the drop-order mechanism above is stated wrongly. Claude Code does not drop +> descriptions "starting with the skills invoked least". It ranks by a decay-weighted score, +> `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, sorts descending, and grants descriptions +> greedily until the budget runs out; what does not fit renders as a bare name. So a heavily used but +> stale skill can lose its description before a lightly used fresh one: 100 uses 60 days ago scores +> 10 and loses to 12 uses today. Recovered from the Claude Code 2.1.251 binary and stamped in +> [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](../topics/usage-tracking-claude-json/EXPLORE.md) +> Part 2, which carries the basis and the recheck trigger. +> +> **The ADR's core decision is untouched, and this correction strengthens the case for it.** A +> never-invoked skill scores exactly zero and is still shed first, so the bias this paragraph +> identifies holds; the decay term adds a second bias the paragraph did not anticipate, against +> skills the operator used a while ago and has since forgotten, which is the same population +> `show-options` exists to surface. The budget arithmetic quoted above is unaffected: it measures +> demand against the budget, not the order of shedding. + Separately, the no-omission rule was measured against the real catalog at a real moment. Rendering every candidate in full produced **139 options across 275 lines, ~7 screens, ~4,100 tokens** — 97.8% of the catalog, i.e. the generated cheat sheet with an extra column, which an operator reads once and @@ -119,6 +135,32 @@ signals — and revisit the manual-only posture second. Usage-metrics-driven sur (`~/.claude.json` `skillUsage`, undocumented internal state) stays deferred; rotation runs off a ledger the skill writes itself, which is what keeps that deferral honest rather than load-bearing. +> **Revised 2026-08-31:** the deferral stands, but "undocumented internal state" is no longer the +> reason and should not be read as one. That substrate is now characterized and dated in +> [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](../topics/usage-tracking-claude-json/EXPLORE.md), +> which invites the false inference that the deferral lifts once the state is known. It does not, +> because three grounds documentation cannot cure survive: +> +> - The scorer is **wrong-signed** for the question this skill asks. `zPe` scores high for skills +> used recently and often, which are exactly the skills the operator has not forgotten. Inverting +> it collapses to a decay-weighted least-recently-used ordering, which the self-written ledger +> already supplies without a build-pinned dependency. +> - `skillUsage` holds only skills that have fired at least once. It structurally cannot name the +> never-invoked population the no-omission rule above exists to protect. +> - It carries no causal-trigger field, so it cannot answer take-up: whether a skill was invoked +> *because* it was surfaced. The ledger can. That is not a stopgap for missing documentation; it +> measures something these counters never will. +> +> A binary-derived stamp is also not what "documented" meant here. Its own fallback rule, treat a +> version mismatch as scorer-unknown, is what ships alongside ground that can shift without notice, +> and its recheck trigger fires only when a human reads a release note. +> +> **Conditions for a future revisit,** so the next reader does not have to re-derive them: a +> published, versioned surface for skill-usage data rather than a reverse-engineered internal; a +> rotation signal not decay-weighted toward recent use; and take-up attribution. The third is +> unreachable from `skillUsage` by construction, so the ledger survives whatever happens to the +> first two. **The ADR's core decision is untouched.** + **The probe seam, and why the two-consumer version was withdrawn.** `show-options` adds no probe of its own — it routes to `orient` — and the duplication that decision sidestepped is resolved in the same change: `plugins/session-flow/reference/gather.md` now owns the block for all seven consumers diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index e3b294c214..a24650c4fb 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -511,7 +511,7 @@ happens to the first two. - Treat `pluginUsage.usageCount` as non-comparable across plugin shapes wherever it is ranked. -### One correction owed to ADR 0016, drafted not applied +### The correction owed to ADR 0016, applied `docs/adr/0016-...:19-20` states that Claude Code "drops descriptions starting with the skills invoked least". Part 2 shows the ordering is decay-weighted, so a heavily used but @@ -522,6 +522,20 @@ alphabetical order presented as usage-informed. Someone already built on the men that ADR line encodes. The ADR's core decision is untouched and in fact reinforced, so the fix is a dated revision -blockquote in the ADR's own established shape (`:75-84`, `:98-105`), which preserves -superseded reasoning rather than editing Context in place. Amending an accepted decision -record is the human's call, so it is drafted here and not applied. +blockquote in the ADR's own established shape, which preserves superseded reasoning rather +than editing Context in place. + +Two revisions were applied, both dated 2026-08-31: + +- After the Context paragraph on drop order: corrects the mechanism to the decay-weighted + score, and records that the correction strengthens rather than weakens the decision. A + never-invoked skill still scores zero and is shed first, so the identified bias holds; the + decay term adds a second bias the paragraph did not anticipate, against skills the + operator used a while ago and has since forgotten, which is the population `show-options` + exists to surface. The budget arithmetic is unaffected, since it measures demand against + budget rather than order of shedding. +- After the deferral clause: restates its ground on the three documentation-independent + reasons above, and records the lift conditions so the next reader does not re-derive them. + +Neither revision cites an issue, unlike the ADR's two 2026-08-21 revisions, which cite issue +3024. They cite this document instead. Filing an issue to carry them is a separate call. From 524a82590606a2b453bfd42ffb1add54dec449a1 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 16:19:03 -0400 Subject: [PATCH 10/22] fix(claude-ops): give the starvation band a usage signal and stop dropping bare usage keys Two defects in audit-skill-visibility, both confirmed against live data. compute_listing sorted on a usage_score only the test fixtures ever set. The live collector builds its denominator from a filesystem walk and joined usage events afterwards, so every real run scored zero and the band fell through to its alphabetical tiebreaker while presenting itself as usage-informed. On this machine that ranked adhd:clarify (1 use) first to lose its description and work-items:triage (99 uses) among the safest. Scores are now computed before the listing is built. Events were looked up by qualified : name only, while the stores hold both that key and the bare leaf as separate rows, so the bare row was discarded silently. source-control:babysit-prs reported 97 invocations against an actual 475. A bare key is now attributed when exactly one skill owns that leaf and withheld with its candidates when more than one does, which is the same refusal the custom_skill redaction guard makes. listing_score mirrors the product's own scorer, usageCount * max(0.5 ** (daysSinceUse / 7), 0.1), recovered from Claude Code 2.1.251 and carrying a verification stamp with its basis and recheck trigger. It is fed from native counters under the exact qualified key because the product's scorer does no bare-key fallback either; scoring the merged total would predict a truncation that will not happen. listing.score_basis reports "unscored" when no usage survives to weigh, and competing rows carry confidence "unscored" rather than borrowing "inferential", which would claim more than an alphabetical order supports. Suite 93/93, ruff clean. The shell harness exits 2 both before and after this change, from a pre-existing fixture gap unrelated to it. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- plugins/claude-ops/.claude-plugin/plugin.json | 2 +- plugins/claude-ops/CHANGELOG.md | 33 ++++ .../skills/audit-skill-visibility/SKILL.md | 19 +- .../scripts/audit_skill_visibility.py | 166 ++++++++++++++++-- .../scripts/test_audit_skill_visibility.py | 147 ++++++++++++++++ 5 files changed, 352 insertions(+), 15 deletions(-) diff --git a/plugins/claude-ops/.claude-plugin/plugin.json b/plugins/claude-ops/.claude-plugin/plugin.json index 612e091269..a5705c5116 100644 --- a/plugins/claude-ops/.claude-plugin/plugin.json +++ b/plugins/claude-ops/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "claude-ops", - "version": "0.38.21", + "version": "0.39.0", "description": "Claude Code operations toolkit. Twelve skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used \u2014 a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which drops descriptions least-invoked-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface \u2014 every built-in CLI command with aliases and hidden/gated status, every bundled skill, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json \u2014 full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the silent unparsable-settings pause, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labelled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces \u2014 built-in CLI commands, bundled skills, plugin-backed built-ins, session-provided skills \u2014 against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry \u2014 OTEL store, collector, hook-event JSONL, ccusage \u2014 with trend reports and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and integrate them into the current repo), plugins (bring a machine's plugin fleet current on demand \u2014 marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view \u2014 queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action \u2014 an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry lives. Plus a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures \u2014 the last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that maps envelopes into the hook-events.jsonl the observability skill reads.", "author": { "name": "Melodic Software", diff --git a/plugins/claude-ops/CHANGELOG.md b/plugins/claude-ops/CHANGELOG.md index 12e05b8e92..477e5602fb 100644 --- a/plugins/claude-ops/CHANGELOG.md +++ b/plugins/claude-ops/CHANGELOG.md @@ -3,6 +3,39 @@ All notable changes to the `claude-ops` plugin are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); this plugin uses semantic versioning. +## [0.39.0] + +### Fixed + +- **`audit-skill-visibility`'s starvation band now carries a usage signal.** + `compute_listing` sorted on a `usage_score` that only the test fixtures ever + set: the live collector builds its denominator from a filesystem walk, and + usage events were joined afterwards, so every real run scored zero and the band + fell through to its alphabetical tiebreaker while presenting itself as + usage-informed. On this machine that ranked `adhd:clarify` (1 use) as first to + lose its description and `work-items:triage` (99 uses) as among the safest. + Scores are now computed before the listing is built. +- **Usage recorded under a skill's bare leaf no longer vanishes.** Events were + looked up by qualified `:` name only, while the stores hold both + that key and the bare leaf as separate rows, so the bare row was discarded with + nothing saying so. `source-control:babysit-prs` reported 97 invocations against + an actual 475. A bare key is now attributed when exactly one skill owns that + leaf, and withheld with its candidates when more than one does. + +### Added + +- **The listing scorer is mirrored rather than guessed at.** `listing_score` + implements `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, recovered from + Claude Code 2.1.251 and carrying a verification stamp with its basis and + recheck trigger. The ordering is decay-weighted, so "least invoked" was never + the right description of it: a heavily used but stale skill can rank below a + lightly used fresh one. Evidence in + `docs/topics/usage-tracking-claude-json/EXPLORE.md`. +- **`listing.score_basis`, so an unscored band admits it.** When no usage + survives to weigh, the order is alphabetical and nothing more; the basis reads + `unscored` and competing rows carry `confidence: "unscored"` rather than + borrowing `inferential`, which claims more than the data supports. + ## [0.38.21] ### Changed diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index 1433e4a601..e1b102403c 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used. A skill is invisible when its description is dropped by the skill-listing context budget (Claude Code drops descriptions starting with the least-invoked skills, so an unused skill loses the keywords that would let it be matched and stays unused), when frontmatter is malformed or a description is missing, when skillOverrides or a disabled plugin hides it, or when disable-model-invocation keeps it out of context by design. Reports reachability, observed usage, and whether it is losing the budget contest. Computing whether the listing overflows from documented settings, and withholding every verdict the data cannot support rather than reporting absence of data as absence of use. Read-only; never disables, deletes, or edits a skill. Use when: 'why do I never use most of my skills', 'why does Claude never suggest this skill', 'are my skill descriptions being dropped', 'is my skill listing over budget', 'which skills can the model actually see', 'which skills are starved', 'I have too many skills to know when to use them', 'audit skill visibility'. Not for: which skills are unused versus their context cost as a one-shot check (Claude Code ships that in /doctor and the Stats tab), repo-authoring listing-budget lint (use skill-quality's check-listing-budget), enumerating what is installed (use /claude-ops:inventory), or reading telemetry infrastructure (use /claude-ops:observability)." +description: "Audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used. A skill is invisible when its description is dropped by the skill-listing context budget (Claude Code drops descriptions from the lowest-scoring skills, ranked by a decay-weighted usage score, so an unused skill loses the keywords that would let it be matched and stays unused), when frontmatter is malformed or a description is missing, when skillOverrides or a disabled plugin hides it, or when disable-model-invocation keeps it out of context by design. Reports reachability, observed usage, and whether it is losing the budget contest. Computing whether the listing overflows from documented settings, and withholding every verdict the data cannot support rather than reporting absence of data as absence of use. Read-only; never disables, deletes, or edits a skill. Use when: 'why do I never use most of my skills', 'why does Claude never suggest this skill', 'are my skill descriptions being dropped', 'is my skill listing over budget', 'which skills can the model actually see', 'which skills are starved', 'I have too many skills to know when to use them', 'audit skill visibility'. Not for: which skills are unused versus their context cost as a one-shot check (Claude Code ships that in /doctor and the Stats tab), repo-authoring listing-budget lint (use skill-quality's check-listing-budget), enumerating what is installed (use /claude-ops:inventory), or reading telemetry infrastructure (use /claude-ops:observability)." argument-hint: "[--installed [dir]] [--plugins-root ] [--render markdown|json] [--now ] [--fixture ]. Collects live; --installed reads the plugin manifest, else fleet defaults to ./plugins" user-invocable: true disable-model-invocation: false @@ -162,6 +162,23 @@ a user as documented. - **Ambiguous attribution is reported, not guessed.** Two marketplaces shipping a same-named plugin collapse to one usage key; those rows are marked `ambiguous-attribution` rather than attributed to one of them. +- **A skill's usage can be recorded under either name.** The stores hold both the + qualified `:` key and the bare leaf, as separate rows, so a + qualified-only lookup silently under-reports. Both are collected. A bare key is + attributed only when exactly one skill in the fleet owns that leaf; an + ambiguous one is withheld with its candidates rather than spent on a guess. +- **The band is scored the way the product scores, not the way the count reads.** + The starvation ordering mirrors Claude Code's own scorer, + `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, which is decay-weighted, so + a heavily used but stale skill can rank below a lightly used fresh one. It is + fed from the NATIVE counters under the EXACT qualified key, because the + product's scorer does no bare-key fallback either; scoring the merged total + would predict a truncation that will not happen. The merged total still backs + `observation`, which asks a different question. +- **An unscored band says so.** When no usage survives to weigh, the ordering is + the alphabetical tiebreaker and nothing more. `listing.score_basis` reports + `unscored` and every competing row's `confidence` is `unscored`, which is a + weaker claim than `inferential` and must not wear that label. ## Scope boundary diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index 3b796a3dc7..d5dc06c177 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -106,6 +106,79 @@ def tier_supports(tier: str, claim: str) -> bool: return claim in TIER_CAPABILITIES.get(tier, set()) +# Claude Code's own listing-budget scorer, mirrored so the starvation band can +# predict which descriptions the product will actually drop. Before this existed +# the band sorted on a field nothing populated, so it rendered alphabetical order +# dressed as usage-informed. +# +# -- Verification stamp (docs/conventions/upstream-drift) ---------------------- +# Claim: the product ranks skills for description truncation by +# `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, sorts that score +# descending, grants descriptions greedily until the budget is spent, and +# renders the remainder name-only. +# Basis: string extraction of `claude.exe`, Claude Code 2.1.251 -- `zPe` (the +# scorer) and `Ymt` (the truncator), plus `zPe`'s two other call sites, the +# slash-menu top-5 pin and the command-search score boost. Evidence recorded in +# docs/topics/usage-tracking-claude-json/EXPLORE.md, Part 2. +# As-of: 2026-08-31, Claude Code 2.1.251. +# Recheck trigger: a release note naming the skill listing, its character budget, +# or skill usage counters; or the counters changing shape in `~/.claude.json`. +# On mismatch: the report must degrade to `score_basis: "unscored"` and say the +# ordering is unknown. A confidently wrong band is worse than no band. +# ----------------------------------------------------------------------------- +LISTING_SCORE_HALF_LIFE_DAYS = 7.0 +LISTING_SCORE_FLOOR = 0.1 + + +def listing_score(count: int, last_used: datetime | None, clock: datetime) -> float: + """Mirror of the product's scorer. Zero when there is no usage to weigh.""" + if count <= 0 or last_used is None: + return 0.0 + days = (clock - last_used).total_seconds() / 86400.0 + decay = 0.5 ** (days / LISTING_SCORE_HALF_LIFE_DAYS) + return count * max(decay, LISTING_SCORE_FLOOR) + + +def resolve_event_keys( + denominator: list[dict], events: list[dict] +) -> tuple[dict[str, list[dict]], list[tuple[str, list[str]]]]: + """Group events by the qualified skill they belong to. + + The stores record a skill's usage under either its qualified `:` + name or its bare leaf, and the two land as separate rows. This machine holds + both `babysit-prs` and `source-control:babysit-prs`. Looking events up by + qualified name alone silently discarded every bare-key row, which is a + reported count that is simply too low with nothing saying so. + + A bare key is attributed only when exactly one skill in the fleet carries + that leaf. Piling an ambiguous leaf onto one plugin would invent usage, the + same failure the `custom_skill` redaction guard exists to prevent, so an + ambiguous key is returned for the withheld section instead of being spent. + """ + qualified = {entry["qualified_name"] for entry in denominator} + leaf_owners: dict[str, set[str]] = defaultdict(set) + for entry in denominator: + name = entry["qualified_name"] + leaf = name.split(":", 1)[1] if ":" in name else name + leaf_owners[leaf].add(name) + + by_skill: dict[str, list[dict]] = defaultdict(list) + ambiguous: dict[str, list[str]] = {} + for event in events: + key = event.get("skill") + if key is None: + continue + if key in qualified: + by_skill[key].append(event) + continue + owners = leaf_owners.get(key, set()) + if len(owners) == 1: + by_skill[next(iter(owners))].append(event) + elif len(owners) > 1: + ambiguous[key] = sorted(owners) + return by_skill, sorted(ambiguous.items()) + + def parse_native( skill_usage: dict, first_start: datetime ) -> tuple[list[dict], datetime]: @@ -700,17 +773,28 @@ def _demand_chars(entry: dict, cfg: ListingConfig) -> int: return min(len(description) + joiner + len(when_to_use), cfg.max_desc_chars) -def compute_listing(denominator: list[dict], cfg: ListingConfig) -> dict: +def compute_listing( + denominator: list[dict], + cfg: ListingConfig, + scores: dict[str, float] | None = None, +) -> dict: """Budget arithmetic, split by confidence. CERTAIN: whether the listing overflows and by how much -- pure arithmetic over documented settings against summed description lengths. INFERENTIAL: which particular skills lose their descriptions. That ordering - comes from an undocumented scorer pinned to one build, so it is rendered as - a ranked band and labelled, never as an exact cutoff. + comes from a scorer recovered from one build of the product (see + `listing_score`), so it is rendered as a ranked band and labelled, never as + an exact cutoff. + + `scores` carries the mirrored scorer's output per qualified name. When it is + absent or empty the ordering has no usage signal behind it, and the returned + `score_basis` says so rather than letting the alphabetical tiebreaker pass + for a usage ranking. """ budget = listing_budget_chars(cfg) + scores = scores or {} rows: list[dict] = [] demand = 0 @@ -723,18 +807,29 @@ def compute_listing(denominator: list[dict], cfg: ListingConfig) -> dict: "qualified_name": entry["qualified_name"], "eligibility": eligibility, "demand_chars": chars, - "usage_score": entry.get("usage_score", 0), + "usage_score": scores.get( + entry["qualified_name"], entry.get("usage_score", 0) + ), } ) overflow = max(0, demand - budget) verdict = "overflowing" if overflow > 0 else "listing-fits" + # Basis is decided by whether any score survived, not by whether a scores + # argument arrived. A caller can hand over a full map that happens to be all + # zeros, and the resulting order is alphabetical either way. + score_basis = "native-counters" if any(r["usage_score"] for r in rows) else "unscored" + competing = [r for r in rows if r["eligibility"] == "competing"] - # Least-used first: that is the order the product drops descriptions in, so - # rank 1 is the most likely to have already lost its description. + # Lowest score first: that is the order the product sheds descriptions in, so + # rank 1 is the most likely to have already lost its. The score is + # decay-weighted, NOT a raw invocation count -- a heavily used but stale + # skill can sort below a lightly used fresh one, which is why this cannot be + # read as "least invoked". The name is a tiebreaker only; when `scores` is + # empty it is the WHOLE ordering, which is what `score_basis` exists to admit. competing.sort(key=lambda r: (r["usage_score"], r["qualified_name"])) - # Descriptions are dropped least-invoked-first only UNTIL the listing fits, + # Descriptions are shed lowest-score-first only UNTIL the listing fits, # so the starved set is the prefix whose demand covers the overflow -- not # the whole fleet. Marking every competing row starved on a one-character # overflow would libel exactly the skills the mechanism protects longest, @@ -742,7 +837,15 @@ def compute_listing(denominator: list[dict], cfg: ListingConfig) -> dict: remaining = overflow for rank, row in enumerate(competing, start=1): row["band"] = rank if overflow > 0 else None - row["confidence"] = "inferential" if overflow > 0 else "certain" + # An unscored ordering is alphabetical, so its band carries no signal at + # all. That is a weaker claim than an inferential one and must not wear + # the same label. + if overflow <= 0: + row["confidence"] = "certain" + elif score_basis == "unscored": + row["confidence"] = "unscored" + else: + row["confidence"] = "inferential" if overflow <= 0: row["verdict"] = "listing-fits" elif remaining > 0: @@ -764,6 +867,7 @@ def compute_listing(denominator: list[dict], cfg: ListingConfig) -> dict: "demand_chars": demand, "overflow_chars": overflow, "verdict": verdict, + "score_basis": score_basis, "competing_count": len(competing), "starved_count": sum(1 for r in competing if r["verdict"] == "likely-starved"), "exempt_count": len(rows) - len(competing), @@ -894,7 +998,34 @@ def classify( ) -> dict: """Pure. Fleet + events + config + clock + horizons -> report model.""" tier = resolve_tier(set(horizons)) - listing = compute_listing(denominator, listing_config or ListingConfig()) + events_by_skill, ambiguous_keys = resolve_event_keys(denominator, events) + + # The band mirrors the product, so it is scored the way the product scores: + # from the NATIVE counters only, under the EXACT qualified key. `zPe` does no + # bare-key fallback of its own -- when usage was recorded under a bare leaf + # the product's own scorer sees zero for that listing entry too, so scoring + # the merged total here would predict a truncation the product will not + # perform. The merge below is for the observation count, which asks a + # different question and wants every event. + native_scores: dict[str, float] = {} + for entry in denominator: + name = entry["qualified_name"] + native = [ + e + for e in events_by_skill.get(name, []) + if e.get("source") == "native" and e.get("skill") == name + ] + if not native: + continue + native_scores[name] = listing_score( + sum(int(e.get("count", 1)) for e in native), + max(e["ts"] for e in native), + clock, + ) + + listing = compute_listing( + denominator, listing_config or ListingConfig(), native_scores + ) starvation_by_name = {r["qualified_name"]: r for r in listing["skills"]} # Narrowest horizon = the most recent start = the least we can see back to. # Two different questions, two different horizons. @@ -915,13 +1046,22 @@ def classify( for entry in denominator: seen[entry["qualified_name"]] += 1 - events_by_skill: dict[str, list[dict]] = defaultdict(list) - for event in events: - events_by_skill[event["skill"]].append(event) - skills: list[dict] = [] withheld: list[dict] = [] + for key, owners in ambiguous_keys: + withheld.append( + { + "skill": key, + "claim": "observation", + "reason": ( + f"bare usage key `{key}` matches {len(owners)} skills " + f"({', '.join(owners)}); attributing it would invent usage " + "for whichever one was picked" + ), + } + ) + for entry in denominator: name = entry["qualified_name"] own_events = events_by_skill.get(name, []) diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py index 5be21fd6c3..5caea4202a 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py @@ -472,6 +472,153 @@ def test_no_band_when_the_listing_fits(self): self.assertIsNone(listing["skills"][0]["band"]) +class ListingScoreTest(unittest.TestCase): + """The mirrored scorer, and the refusal to dress zero up as a ranking.""" + + def test_score_decays_with_a_seven_day_half_life(self): + now = _utc(2026, 8, 31) + fresh = engine.listing_score(100, now, now) + one_half_life = engine.listing_score(100, now - timedelta(days=7), now) + self.assertAlmostEqual(fresh, 100.0) + self.assertAlmostEqual(one_half_life, 50.0) + + def test_decay_floors_at_a_tenth(self): + now = _utc(2026, 8, 31) + ancient = engine.listing_score(100, now - timedelta(days=3650), now) + self.assertAlmostEqual(ancient, 10.0) + + def test_a_stale_heavy_user_sorts_below_a_fresh_light_one(self): + """The whole reason `least invoked` was the wrong description.""" + now = _utc(2026, 8, 31) + stale = engine.listing_score(100, now - timedelta(days=60), now) + fresh = engine.listing_score(12, now, now) + self.assertLess(stale, fresh) + + def test_never_used_scores_zero(self): + now = _utc(2026, 8, 31) + self.assertEqual(engine.listing_score(0, now, now), 0.0) + self.assertEqual(engine.listing_score(5, None, now), 0.0) + + def test_all_zero_scores_report_an_unscored_basis(self): + """An alphabetical order must not be labelled a usage ranking.""" + entries = [ + { + "qualified_name": f"a:{i}", + "frontmatter": {"description": "x" * 1000}, + "plugin_enabled": True, + } + for i in range(10) + ] + listing = engine.compute_listing( + entries, engine.ListingConfig(context_window_tokens=200_000) + ) + self.assertEqual(listing["score_basis"], "unscored") + competing = [s for s in listing["skills"] if s["eligibility"] == "competing"] + self.assertTrue(all(s["confidence"] == "unscored" for s in competing)) + + def test_classify_scores_the_band_from_native_counters(self): + """Regression: the band used to sort on a field nothing populated.""" + now = _utc(2026, 8, 31) + entries = [ + { + "qualified_name": f"a:{i}", + "frontmatter": {"description": "x" * 1000}, + "plugin_enabled": True, + } + for i in range(10) + ] + # Counts ascend with the index, so the band must descend with it. + events = [ + {"skill": f"a:{i}", "ts": now, "source": "native", "count": (i + 1) * 10} + for i in range(10) + ] + model = engine.classify( + denominator=entries, + events=events, + config=engine.Config(), + clock=now, + horizons={"native": now - timedelta(days=400)}, + listing_config=engine.ListingConfig(context_window_tokens=200_000), + ) + self.assertEqual(model["listing"]["score_basis"], "native-counters") + bands = { + row["qualified_name"]: row["starvation"]["band"] for row in model["skills"] + } + # Least-scored is band 1, i.e. first to lose its description. + self.assertEqual(bands["a:0"], 1) + self.assertEqual(bands["a:9"], 10) + + +class BareUsageKeyTest(unittest.TestCase): + """Usage recorded under a bare leaf must reach its qualified skill.""" + + def test_bare_key_is_attributed_when_the_leaf_is_unique(self): + now = _utc(2026, 8, 31) + model = engine.classify( + denominator=[_skill("source-control:babysit-prs")], + events=[ + {"skill": "babysit-prs", "ts": now, "source": "native", "count": 378}, + { + "skill": "source-control:babysit-prs", + "ts": now, + "source": "native", + "count": 97, + }, + ], + config=engine.Config(), + clock=now, + horizons={"native": now - timedelta(days=400)}, + ) + row = model["skills"][0] + self.assertEqual(row["observation"]["count"], 475) + + def test_ambiguous_bare_key_is_withheld_not_guessed(self): + now = _utc(2026, 8, 31) + model = engine.classify( + denominator=[_skill("toolchain:check"), _skill("skill-quality:check")], + events=[{"skill": "check", "ts": now, "source": "native", "count": 40}], + config=engine.Config(), + clock=now, + horizons={"native": now - timedelta(days=400)}, + ) + for row in model["skills"]: + self.assertEqual(row["observation"]["count"], 0) + withheld = [w for w in model["withheld"] if w["skill"] == "check"] + self.assertEqual(len(withheld), 1) + self.assertIn("toolchain:check", withheld[0]["reason"]) + self.assertIn("skill-quality:check", withheld[0]["reason"]) + + def test_a_bare_key_does_not_score_the_band(self): + """`zPe` has no bare-key fallback, so the mirror must not add one.""" + now = _utc(2026, 8, 31) + entries = [ + { + "qualified_name": "a:one", + "frontmatter": {"description": "x" * 1000}, + "plugin_enabled": True, + }, + { + "qualified_name": "b:two", + "frontmatter": {"description": "x" * 1000}, + "plugin_enabled": True, + }, + ] + model = engine.classify( + denominator=entries, + events=[{"skill": "one", "ts": now, "source": "native", "count": 900}], + config=engine.Config(), + clock=now, + horizons={"native": now - timedelta(days=400)}, + listing_config=engine.ListingConfig(context_window_tokens=200_000), + ) + rows = {r["qualified_name"]: r for r in model["skills"]} + # The count reaches the observation field ... + self.assertEqual(rows["a:one"]["observation"]["count"], 900) + # ... but not the band, because the product's own scorer misses it too. + self.assertEqual(rows["a:one"]["starvation"]["usage_score"], 0) + self.assertEqual(model["listing"]["score_basis"], "unscored") + + class TierResolutionTest(unittest.TestCase): """A claim renders only at a tier that supports it.""" From 1e61d7713d3228ba5616748954319239bd0ae127 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 16:32:16 -0400 Subject: [PATCH 11/22] docs(claude-ops): route the scorer's reasoning to a spoke The counting-rules additions pushed SKILL.md from 199 to 216 lines, past the soft target, which is the progressive-disclosure signal working. The mechanism, why "least invoked" was the wrong description, why a bare key does not move the band, and the drift posture now live in reference/listing-scorer.md, leaving two tight bullets and a pointer in the hub. SKILL.md lands at 207. The remaining eight lines are the two non-obvious counting rules the section exists to hold, so they stay rather than being cut to hit an advisory number. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../skills/audit-skill-visibility/SKILL.md | 25 +++---- .../reference/listing-scorer.md | 70 +++++++++++++++++++ 2 files changed, 78 insertions(+), 17 deletions(-) create mode 100644 plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index e1b102403c..57b8767b0d 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -162,23 +162,14 @@ a user as documented. - **Ambiguous attribution is reported, not guessed.** Two marketplaces shipping a same-named plugin collapse to one usage key; those rows are marked `ambiguous-attribution` rather than attributed to one of them. -- **A skill's usage can be recorded under either name.** The stores hold both the - qualified `:` key and the bare leaf, as separate rows, so a - qualified-only lookup silently under-reports. Both are collected. A bare key is - attributed only when exactly one skill in the fleet owns that leaf; an - ambiguous one is withheld with its candidates rather than spent on a guess. -- **The band is scored the way the product scores, not the way the count reads.** - The starvation ordering mirrors Claude Code's own scorer, - `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, which is decay-weighted, so - a heavily used but stale skill can rank below a lightly used fresh one. It is - fed from the NATIVE counters under the EXACT qualified key, because the - product's scorer does no bare-key fallback either; scoring the merged total - would predict a truncation that will not happen. The merged total still backs - `observation`, which asks a different question. -- **An unscored band says so.** When no usage survives to weigh, the ordering is - the alphabetical tiebreaker and nothing more. `listing.score_basis` reports - `unscored` and every competing row's `confidence` is `unscored`, which is a - weaker claim than `inferential` and must not wear that label. +- **One skill, two possible usage keys.** The stores hold the qualified + `:` key and the bare leaf as separate rows. Both are collected; a + bare key is attributed only when exactly one skill owns that leaf, ambiguous + ones are withheld with their candidates. +- **The starvation band is decay-weighted, not a count**, and reports + `score_basis: "unscored"` when nothing survives to weigh. Read + [reference/listing-scorer.md](reference/listing-scorer.md) before changing that + ordering or quoting it to a user. ## Scope boundary diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md new file mode 100644 index 0000000000..efb55c3c48 --- /dev/null +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md @@ -0,0 +1,70 @@ +# The listing scorer, mirrored + +Read this when the starvation band's ordering is in question: why it is not a +count, why a bare usage key does not move it, and when it means nothing at all. +The engine's own stamp lives beside `listing_score` in +`scripts/audit_skill_visibility.py`; this file is the reasoning, not a second +copy of the claim. + +## What the product actually does + +Claude Code ranks skills for description truncation by + +```text +usageCount * max(0.5 ** (daysSinceUse / 7), 0.1) +``` + +then sorts that score descending, grants descriptions greedily until the budget +is spent, and renders the remainder name-only. + +Recovered from `claude.exe`, Claude Code 2.1.251: `zPe` is the scorer, `Ymt` the +truncator. `zPe` has two further call sites, the slash-menu top-five pin and the +command-search score boost, which is corroboration that it is the product's +general usage-priority function rather than a listing-local helper. The +surrounding evidence is in +[`docs/topics/usage-tracking-claude-json/EXPLORE.md`](https://github.com/melodic-software/claude-code-plugins/blob/main/docs/topics/usage-tracking-claude-json/EXPLORE.md), +Part 2. + +## Why "least invoked" was the wrong description + +The score is decay-weighted with a seven-day half life and a floor at a tenth, +so recency competes with volume. A skill used 100 times sixty days ago scores +`100 * 0.1 = 10` and loses its description to one used 12 times today, which +scores 12. Any wording that says descriptions are shed "starting with the +least-invoked skills" describes a mechanism the product does not have. + +The floor matters at both ends. A never-used skill scores exactly zero and is +always shed first, which is the feedback loop this whole skill exists to expose. +A once-used skill never decays below `0.1 * usageCount`, so it never falls back +into the never-used band. + +## Why a bare usage key does not move the band + +The stores record a skill's usage under either its qualified `:` +name or its bare leaf, as separate rows. `zPe` looks up the listing entry's name +directly and does no fallback between the two; the product's own display helper +(`oKn`) does, but the scorer does not. + +So the mirror does not either. Feeding it the merged total would predict a +truncation the product will not perform, and predicting the product wrongly is +the one thing this band must not do. `observation` asks a different question, +"how much has this skill been used", and takes every event, merged. + +That means the product can leave an entry unscored while the skill is heavily +used under its other key. Faithfully reproducing that is the point. + +## When the band means nothing + +`listing.score_basis` is `unscored` whenever no usage survives to weigh. The +ordering is then the alphabetical tiebreaker and carries no signal, so competing +rows report `confidence: "unscored"` rather than `inferential`. The distinction +is load-bearing: `inferential` claims a ranking exists and may be imprecise; +`unscored` says no ranking was possible. + +## On drift + +This is one build of a minified bundle, not a published interface. The recheck +trigger is a release note naming the skill listing, its character budget, or the +usage counters, or the counters changing shape in `~/.claude.json`. On a +mismatch the honest degradation is back to `unscored`, never a confidently wrong +band. From bb0bb7105590a74f99a89938a7f1631e85678cc9 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 16:43:07 -0400 Subject: [PATCH 12/22] fix(claude-ops): keep the skill description inside the listing-entry cap The rewritten drop-order clause pushed description+when_to_use to 1550 chars against the 1536 cap, which truncates the listing entry. Says the same thing in fewer words: descriptions are dropped by a decay-weighted usage score. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- plugins/claude-ops/skills/audit-skill-visibility/SKILL.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index 57b8767b0d..566f8baae3 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -1,5 +1,5 @@ --- -description: "Audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used. A skill is invisible when its description is dropped by the skill-listing context budget (Claude Code drops descriptions from the lowest-scoring skills, ranked by a decay-weighted usage score, so an unused skill loses the keywords that would let it be matched and stays unused), when frontmatter is malformed or a description is missing, when skillOverrides or a disabled plugin hides it, or when disable-model-invocation keeps it out of context by design. Reports reachability, observed usage, and whether it is losing the budget contest. Computing whether the listing overflows from documented settings, and withholding every verdict the data cannot support rather than reporting absence of data as absence of use. Read-only; never disables, deletes, or edits a skill. Use when: 'why do I never use most of my skills', 'why does Claude never suggest this skill', 'are my skill descriptions being dropped', 'is my skill listing over budget', 'which skills can the model actually see', 'which skills are starved', 'I have too many skills to know when to use them', 'audit skill visibility'. Not for: which skills are unused versus their context cost as a one-shot check (Claude Code ships that in /doctor and the Stats tab), repo-authoring listing-budget lint (use skill-quality's check-listing-budget), enumerating what is installed (use /claude-ops:inventory), or reading telemetry infrastructure (use /claude-ops:observability)." +description: "Audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used. A skill is invisible when its description is dropped by the skill-listing context budget (Claude Code drops descriptions by a decay-weighted usage score, so an unused skill loses the keywords that would let it be matched and stays unused), when frontmatter is malformed or a description is missing, when skillOverrides or a disabled plugin hides it, or when disable-model-invocation keeps it out of context by design. Reports reachability, observed usage, and whether it is losing the budget contest. Computing whether the listing overflows from documented settings, and withholding every verdict the data cannot support rather than reporting absence of data as absence of use. Read-only; never disables, deletes, or edits a skill. Use when: 'why do I never use most of my skills', 'why does Claude never suggest this skill', 'are my skill descriptions being dropped', 'is my skill listing over budget', 'which skills can the model actually see', 'which skills are starved', 'I have too many skills to know when to use them', 'audit skill visibility'. Not for: which skills are unused versus their context cost as a one-shot check (Claude Code ships that in /doctor and the Stats tab), repo-authoring listing-budget lint (use skill-quality's check-listing-budget), enumerating what is installed (use /claude-ops:inventory), or reading telemetry infrastructure (use /claude-ops:observability)." argument-hint: "[--installed [dir]] [--plugins-root ] [--render markdown|json] [--now ] [--fixture ]. Collects live; --installed reads the plugin manifest, else fleet defaults to ./plugins" user-invocable: true disable-model-invocation: false From 743a6ef3fb22e1ce1eb1308f7f1570ca6bb45d14 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 17:10:21 -0400 Subject: [PATCH 13/22] docs: cite issue 3534 from the ADR 0016 revisions Matches the citation shape of the ADR's two 2026-08-21 revisions, which carry an issue link. 3534 records the two audit-skill-visibility defects that prompted both revisions. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- ...skill-recommendation-from-the-catalog-not-the-listing.md | 6 ++++-- docs/topics/usage-tracking-claude-json/EXPLORE.md | 4 ++-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md index 3d9bf6c61f..adc4209018 100644 --- a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md +++ b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md @@ -27,7 +27,8 @@ the wrong direction, and — the part that makes it a correctness bug rather tha cannot tell that it is blind. The gatekeeping the contract bans would have been reinstated by the harness, invisibly. -> **Revised 2026-08-31:** the drop-order mechanism above is stated wrongly. Claude Code does not drop +> **Revised 2026-08-31 ([#3534](https://github.com/melodic-software/claude-code-plugins/issues/3534)):** +> the drop-order mechanism above is stated wrongly. Claude Code does not drop > descriptions "starting with the skills invoked least". It ranks by a decay-weighted score, > `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, sorts descending, and grants descriptions > greedily until the budget runs out; what does not fit renders as a bare name. So a heavily used but @@ -135,7 +136,8 @@ signals — and revisit the manual-only posture second. Usage-metrics-driven sur (`~/.claude.json` `skillUsage`, undocumented internal state) stays deferred; rotation runs off a ledger the skill writes itself, which is what keeps that deferral honest rather than load-bearing. -> **Revised 2026-08-31:** the deferral stands, but "undocumented internal state" is no longer the +> **Revised 2026-08-31 ([#3534](https://github.com/melodic-software/claude-code-plugins/issues/3534)):** +> the deferral stands, but "undocumented internal state" is no longer the > reason and should not be read as one. That substrate is now characterized and dated in > [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](../topics/usage-tracking-claude-json/EXPLORE.md), > which invites the false inference that the deferral lifts once the state is known. It does not, diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index a24650c4fb..a4f85f57d0 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -537,5 +537,5 @@ Two revisions were applied, both dated 2026-08-31: - After the deferral clause: restates its ground on the three documentation-independent reasons above, and records the lift conditions so the next reader does not re-derive them. -Neither revision cites an issue, unlike the ADR's two 2026-08-21 revisions, which cite issue -3024. They cite this document instead. Filing an issue to carry them is a separate call. +Both revisions cite issue 3534, which carries the two defects in Part 4, matching the +citation shape of the ADR's two 2026-08-21 revisions. From d357e50af68e376e2e519c34cd36c3382b51a97d Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 17:50:17 -0400 Subject: [PATCH 14/22] docs(native-surfaces): record doctor -> audit-skill-visibility complementary verdict audit-skill-visibility's description already routed the one-shot unused-versus-context-cost check to the bundled /doctor, with no store row behind the disclaimer. Record the missing row: verdict complementary, evidence from the installed v2.1.252 binary's doctor Check 1 strings (2026-08-31), and regenerate NATIVE-SURFACES.md via overlap.py generate. Fresh-context verifier concurred with the verdict; self-check degraded only on the pre-existing v2.1.232 extraction staleness. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- docs/NATIVE-SURFACES.md | 17 ++++++++++++++++- docs/native-surfaces/records.json | 27 +++++++++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/docs/NATIVE-SURFACES.md b/docs/NATIVE-SURFACES.md index 80c2cf3a61..349de0cfce 100644 --- a/docs/NATIVE-SURFACES.md +++ b/docs/NATIVE-SURFACES.md @@ -18,7 +18,7 @@ and when — see [`docs/conventions/native-references/`](conventions/native-refe | Lane | Rows | Baked | Verdicts | |---|---|---|---| | Built-in CLI commands | 1 | 0 | complementary 1 | -| Bundled skills | 6 | 1 | complementary 6 | +| Bundled skills | 7 | 1 | complementary 7 | | Plugin-backed built-ins | 1 | 0 | complementary 1 | | Session-provided skills (observation-only) | 1 | 0 | defer 1 | @@ -88,6 +88,21 @@ and when — see [`docs/conventions/native-references/`](conventions/native-refe - **Baked:** description phrase no · Boundary section no - **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure — it is the best available routing surface, not a guaranteed one +### `doctor` → `claude-ops:audit-skill-visibility` + +- **Verdict:** `complementary` — Same native surface as the two sibling rows, a third of our lanes. Bundled `doctor` ships a one-shot check (its Check 1) that groups unused skills, MCP servers, and plugins against their context cost, labels each group with a token-savings estimate, and offers to disable the selected groups. audit-skill-visibility answers a different question, why a skill is unseen: it reconciles three usage sources (native ~/.claude.json counters, its own JSONL store, OTEL) under a max-across-sources rule, computes an observed horizon and withholds every verdict the span cannot support, diagnoses reachability causes, and analyses listing-budget starvation. It disables nothing by contract. The skill's own description and Scope boundary already route the one-shot unused-versus-context-cost question to the native surface; this row records that routing in the store rather than replacing it. +- **Native surface:** `doctor` (bundled skill; markers: gated) +- **Our component:** `claude-ops:audit-skill-visibility` (skill) +- **Evidence:** + - `doctor` present in the 2026-08-23 extraction as bundled-skill (markers: gated; aliases: checkup), per the two sibling rows + - the shipped doctor skill carries a check titled 'Check 1: unused skills, MCP servers, and plugins' whose prompt groups unused components, labels each group with a benefit estimate ('37 unused skills, saves ~2.2k est. tokens/session'), and applies only the groups the user selects; confirmed by string search of the installed v2.1.252 binary on 2026-08-31 + - our description: audit whether each installed skill is actually VISIBLE to the model; reconciles native counters, a JSONL store, and OTEL; withholds every verdict the data cannot support; read-only, never disables, deletes, or edits a skill + - our description's Not-for clause and the SKILL.md Scope boundary table both already name the native surface ('Claude Code ships that in /doctor and the Stats tab') with no store row behind them until this one; a prose disclaimer without a store row is the drift this registry exists to catch +- **Observation:** extraction — targeted string search of the installed binary v2.1.252 (doctor Check 1 strings confirmed; a spot observation over the sibling rows' full v2.1.232 extraction, not a re-extraction) (2026-08-31) +- **Recheck trigger:** a Claude Code release changes doctor's unused-components check (Check 1's grouping, its disable offer, or its benefit estimate), gives it a multi-source reconciliation or observation-horizon discipline, or changes /doctor's status as a bundled skill or its gating switch (verified 2026-08-31) +- **Baked:** description phrase no · Boundary section no +- **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure — it is the best available routing surface, not a guaranteed one + ### `run` → `testing:run-e2e` - **Verdict:** `complementary` — The bundled skill answers 'did this change work when I ran the app'; run-e2e drives named UI and API flows, captures evidence (screenshots, responses, logs), and carries a non-UI smoke playbook for libraries, MCP servers, hooks, and scripts — surfaces that have no app to launch. Prefer the native surface for the quick look; ours where the verification has to be reproducible or the target is not an app. diff --git a/docs/native-surfaces/records.json b/docs/native-surfaces/records.json index e77166f636..db48f20752 100644 --- a/docs/native-surfaces/records.json +++ b/docs/native-surfaces/records.json @@ -173,6 +173,33 @@ "baked": { "description_phrase": false, "boundary_section": false }, "budget_caveat": true }, + { + "native": { "name": "doctor", "class": "bundled-skill", "markers": ["gated"] }, + "component": { + "plugin": "claude-ops", + "skill": "audit-skill-visibility", + "kind": "skill" + }, + "verdict": "complementary", + "reason": "Same native surface as the two sibling rows, a third of our lanes. Bundled `doctor` ships a one-shot check (its Check 1) that groups unused skills, MCP servers, and plugins against their context cost, labels each group with a token-savings estimate, and offers to disable the selected groups. audit-skill-visibility answers a different question, why a skill is unseen: it reconciles three usage sources (native ~/.claude.json counters, its own JSONL store, OTEL) under a max-across-sources rule, computes an observed horizon and withholds every verdict the span cannot support, diagnoses reachability causes, and analyses listing-budget starvation. It disables nothing by contract. The skill's own description and Scope boundary already route the one-shot unused-versus-context-cost question to the native surface; this row records that routing in the store rather than replacing it.", + "evidence": [ + "`doctor` present in the 2026-08-23 extraction as bundled-skill (markers: gated; aliases: checkup), per the two sibling rows", + "the shipped doctor skill carries a check titled 'Check 1: unused skills, MCP servers, and plugins' whose prompt groups unused components, labels each group with a benefit estimate ('37 unused skills, saves ~2.2k est. tokens/session'), and applies only the groups the user selects; confirmed by string search of the installed v2.1.252 binary on 2026-08-31", + "our description: audit whether each installed skill is actually VISIBLE to the model; reconciles native counters, a JSONL store, and OTEL; withholds every verdict the data cannot support; read-only, never disables, deletes, or edits a skill", + "our description's Not-for clause and the SKILL.md Scope boundary table both already name the native surface ('Claude Code ships that in /doctor and the Stats tab') with no store row behind them until this one; a prose disclaimer without a store row is the drift this registry exists to catch" + ], + "observation": { + "class": "extraction", + "detail": "targeted string search of the installed binary v2.1.252 (doctor Check 1 strings confirmed; a spot observation over the sibling rows' full v2.1.232 extraction, not a re-extraction)", + "date": "2026-08-31" + }, + "recheck": { + "trigger": "a Claude Code release changes doctor's unused-components check (Check 1's grouping, its disable offer, or its benefit estimate), gives it a multi-source reconciliation or observation-horizon discipline, or changes /doctor's status as a bundled skill or its gating switch", + "verified": "2026-08-31" + }, + "baked": { "description_phrase": false, "boundary_section": false }, + "budget_caveat": true + }, { "native": { "name": "morning", "class": "session-skill", "markers": [] }, "component": { "plugin": "claude-ops", "skill": "morning-brief", "kind": "skill" }, From 0aaf24e18656c5b0b8f198efecbedeb35cb34e2b Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 18:46:20 -0400 Subject: [PATCH 15/22] docs(claude-ops): re-verify the listing scorer at 2.1.252 and stamp locate-by-shape The recheck trigger fired the same day it was written: the CLI auto-updated from 2.1.251 to 2.1.252 mid-session. Re-ran the extraction. Both the scorer's formula and the truncator's descending-sort greedy grant come back unchanged. The re-run surfaced a trap worth recording. The minified identifier is not stable across builds: the scorer was zPe in 2.1.251 and WPe in 2.1.252 with a byte-identical body. A recheck that greps the old name finds nothing and would wrongly conclude the mechanism was removed. All three stamps now say to locate by shape, and carry the two greps that do it. Suite 93/93, ruff clean, markdownlint clean. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../usage-tracking-claude-json/EXPLORE.md | 9 +++-- .../reference/listing-scorer.md | 33 +++++++++++++------ .../scripts/audit_skill_visibility.py | 14 +++++--- 3 files changed, 40 insertions(+), 16 deletions(-) diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md index a4f85f57d0..d5a29c0dd1 100644 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ b/docs/topics/usage-tracking-claude-json/EXPLORE.md @@ -178,11 +178,16 @@ Follows `docs/conventions/upstream-drift`, the same shape entries descending by that score and drops descriptions from the tail. - **Basis:** string extraction of `claude.exe`, Claude Code 2.1.251, functions `zPe`, `Ymt`, `F1t`, `Fdt`, `Sme`, `yNt`, `dzn`, `Dke`. Confirmed against the live - `~/.claude.json` on this machine. -- **As-of:** 2026-08-31, Claude Code 2.1.251. + `~/.claude.json` on this machine. Scorer and truncator re-verified unchanged at 2.1.252. +- **As-of:** 2026-08-31, re-verified against Claude Code 2.1.252. - **Recheck trigger:** any release note naming the skill listing, the skill-listing budget, skill usage counters, or `/doctor`'s unused-component check; or the counters' shape in `~/.claude.json` gaining or losing a field. +- **Locate by shape, never by name.** The minified identifiers move between builds. The + scorer was `zPe` in 2.1.251 and `WPe` in 2.1.252 with a byte-identical body, so a recheck + greping the old name finds nothing and would wrongly conclude the mechanism was removed. + Grep the arithmetic (`Math\.pow\(0\.5,`) and the truncator's own field name + (`budgetTruncatedSkills`) instead. This is recovered from one build of a minified bundle. Encoding it in a script means pinning to that build, so any consumer must carry the stamp and treat a mismatch as diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md index efb55c3c48..76805c2e2e 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md @@ -17,11 +17,11 @@ usageCount * max(0.5 ** (daysSinceUse / 7), 0.1) then sorts that score descending, grants descriptions greedily until the budget is spent, and renders the remainder name-only. -Recovered from `claude.exe`, Claude Code 2.1.251: `zPe` is the scorer, `Ymt` the -truncator. `zPe` has two further call sites, the slash-menu top-five pin and the -command-search score boost, which is corroboration that it is the product's -general usage-priority function rather than a listing-local helper. The -surrounding evidence is in +Recovered from `claude.exe` at Claude Code 2.1.251 (`zPe` the scorer, `Ymt` the +truncator), then re-verified unchanged at 2.1.252. The scorer has two further +call sites, the slash-menu top-five pin and the command-search score boost, which +is corroboration that it is the product's general usage-priority function rather +than a listing-local helper. The surrounding evidence is in [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](https://github.com/melodic-software/claude-code-plugins/blob/main/docs/topics/usage-tracking-claude-json/EXPLORE.md), Part 2. @@ -63,8 +63,21 @@ is load-bearing: `inferential` claims a ranking exists and may be imprecise; ## On drift -This is one build of a minified bundle, not a published interface. The recheck -trigger is a release note naming the skill listing, its character budget, or the -usage counters, or the counters changing shape in `~/.claude.json`. On a -mismatch the honest degradation is back to `unscored`, never a confidently wrong -band. +This is a minified bundle, not a published interface. The recheck trigger is a +release note naming the skill listing, its character budget, or the usage +counters, or the counters changing shape in `~/.claude.json`. On a mismatch the +honest degradation is back to `unscored`, never a confidently wrong band. + +**Locate the scorer by shape, never by name.** The minified identifier moves +between builds. It was `zPe` in 2.1.251 and `WPe` in 2.1.252, with a +byte-identical body, so a recheck that greps the old name finds nothing and +concludes the mechanism was removed. Grep the arithmetic instead: + +```bash +grep -a -o -E '.{0,180}Math\.pow\(0\.5,.{0,180}' "$(command -v claude)" +grep -a -o -E '.{0,260}budgetTruncatedSkills:.{0,60}' "$(command -v claude)" +``` + +The first run of that recheck happened the same day the stamp was written: the +CLI auto-updated from 2.1.251 to 2.1.252 mid-session, the trigger fired, and both +the formula and the descending-sort truncation came back unchanged. diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index 145a3cb7a7..a36445c0d1 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -116,11 +116,17 @@ def tier_supports(tier: str, claim: str) -> bool: # `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, sorts that score # descending, grants descriptions greedily until the budget is spent, and # renders the remainder name-only. -# Basis: string extraction of `claude.exe`, Claude Code 2.1.251 -- `zPe` (the -# scorer) and `Ymt` (the truncator), plus `zPe`'s two other call sites, the -# slash-menu top-5 pin and the command-search score boost. Evidence recorded in +# Basis: string extraction of `claude.exe`, first at Claude Code 2.1.251 (`zPe` +# the scorer, `Ymt` the truncator, plus the scorer's two other call sites, the +# slash-menu top-5 pin and the command-search score boost), then RE-VERIFIED +# unchanged at 2.1.252. Evidence in # docs/topics/usage-tracking-claude-json/EXPLORE.md, Part 2. -# As-of: 2026-08-31, Claude Code 2.1.251. +# As-of: 2026-08-31, re-verified against Claude Code 2.1.252. +# Locate it by SHAPE, never by name. The minified identifier is not stable across +# builds: the scorer was `zPe` in 2.1.251 and `WPe` in 2.1.252, with a +# byte-identical body. Grep for the arithmetic instead, e.g. +# `grep -a -o -E '.{0,180}Math\.pow\(0\.5,.{0,180}' `, and for +# the truncator `budgetTruncatedSkills`. # Recheck trigger: a release note naming the skill listing, its character budget, # or skill usage counters; or the counters changing shape in `~/.claude.json`. # On mismatch: the report must degrade to `score_basis: "unscored"` and say the From 993075405a4bad137c09b841b40e6e8c53ad23ec Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 19:19:26 -0400 Subject: [PATCH 16/22] docs(claude-ops): graduate the usage-counter findings and prune the topic slice Two gates caught real violations of this repo's own conventions. contract-slice-prune-gate: docs/topics// is Contract tier per docs/conventions/topic-docs/README.md, committed on a task branch and pruned before merge. EXPLORE.md was the wrong home from the start; that convention also names EXPLORE.md as a memory-tier file, which never gets committed at all. The durable outcomes are graduated rather than deleted: - The scorer, its drift posture and the locate-by-shape rule already live in reference/listing-scorer.md. - The counters' own semantics now live in a new reference/usage-counters.md, beside the skill that consumes them: the skillUsage throttle and its qualified-versus-bare key split, the pluginUsage install seeding that makes lastUsedAt worthless at zero count, the incomparability of pluginUsage across plugin shapes, agentLastUsed holding nothing usable, and the structural fact that no per-project skill usage exists in this file at all. - The two defects are issue 3534; the mechanism correction is ADR 0016. Every citation of the pruned slice is repointed, so nothing dangles. skill-count-claim-gate: a bullet opening "One skill, two possible usage keys" parsed as a claim that claude-ops ships one skill. Reworded to "Two possible usage keys per skill", same meaning, no leading count word. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- ...dation-from-the-catalog-not-the-listing.md | 9 +- .../usage-tracking-claude-json/EXPLORE.md | 546 ------------------ plugins/claude-ops/CHANGELOG.md | 4 +- .../skills/audit-skill-visibility/SKILL.md | 2 +- .../reference/listing-scorer.md | 5 +- .../reference/usage-counters.md | 130 +++++ .../scripts/audit_skill_visibility.py | 4 +- 7 files changed, 144 insertions(+), 556 deletions(-) delete mode 100644 docs/topics/usage-tracking-claude-json/EXPLORE.md create mode 100644 plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md diff --git a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md index adc4209018..947b5f8a97 100644 --- a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md +++ b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md @@ -33,9 +33,10 @@ harness, invisibly. > `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, sorts descending, and grants descriptions > greedily until the budget runs out; what does not fit renders as a bare name. So a heavily used but > stale skill can lose its description before a lightly used fresh one: 100 uses 60 days ago scores -> 10 and loses to 12 uses today. Recovered from the Claude Code 2.1.251 binary and stamped in -> [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](../topics/usage-tracking-claude-json/EXPLORE.md) -> Part 2, which carries the basis and the recheck trigger. +> 10 and loses to 12 uses today. Recovered from the Claude Code 2.1.251 binary, re-verified +> unchanged at 2.1.252, and stamped in +> [`plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md`](../../plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md), +> which carries the basis and the recheck trigger. > > **The ADR's core decision is untouched, and this correction strengthens the case for it.** A > never-invoked skill scores exactly zero and is still shed first, so the bias this paragraph @@ -139,7 +140,7 @@ ledger the skill writes itself, which is what keeps that deferral honest rather > **Revised 2026-08-31 ([#3534](https://github.com/melodic-software/claude-code-plugins/issues/3534)):** > the deferral stands, but "undocumented internal state" is no longer the > reason and should not be read as one. That substrate is now characterized and dated in -> [`docs/topics/usage-tracking-claude-json/EXPLORE.md`](../topics/usage-tracking-claude-json/EXPLORE.md), +> [`plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md`](../../plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md), > which invites the false inference that the deferral lifts once the state is known. It does not, > because three grounds documentation cannot cure survive: > diff --git a/docs/topics/usage-tracking-claude-json/EXPLORE.md b/docs/topics/usage-tracking-claude-json/EXPLORE.md deleted file mode 100644 index d5a29c0dd1..0000000000 --- a/docs/topics/usage-tracking-claude-json/EXPLORE.md +++ /dev/null @@ -1,546 +0,0 @@ -# Usage tracking: what `~/.claude.json` records, and what this repo already builds on it - -Exploration only. No component was changed. Findings below are split into what was -verified against primary evidence (the live `~/.claude.json` on this machine and the -Claude Code 2.1.251 binary) and what is still open. - -Evidence basis: - -- `C:\Users\KyleSexton\.claude.json`, read 2026-08-31. -- `C:\Users\KyleSexton\.local\bin\claude.exe`, Claude Code 2.1.251, string-extracted. -- Repository at `origin/main` (949c54b0d). - -## Part 1: what Claude Code tracks in `~/.claude.json` - -`~/.claude.json` is the user-scope state file. It is a different file from -`~/.claude/settings.json` (the settings scope). It holds four usage-relevant regions. - -### 1.1 `skillUsage` (machine-global, per skill) - -Shape: `{"": {"usageCount": , "lastUsedAt": }}`. -131 entries on this machine. - -Write path in the binary (`Fdt`): - -```js -let d = o.skillUsageLastWriteAt.get(t); -if (d !== void 0 && u - d < $Mn) return; // $Mn = 60000 -o.skillUsageLastWriteAt.set(t, u), - Ae((y) => ({ ...y, skillUsage: { ...y.skillUsage, - [t]: { usageCount: (k?.usageCount ?? 0) + 1, lastUsedAt: u } } }), r) -``` - -Verified properties: - -- Incremented on real skill dispatch only. There is no install-time or session-start - seeding, so a skill's `lastUsedAt` is trustworthy evidence of actual use. -- `usageCount` is a lifetime total since install. It never resets and is never windowed. -- **A 60-second per-skill throttle DROPS the increment rather than coalescing it.** A skill - invoked five times in one minute records one. `skillUsage.usageCount` therefore - systematically undercounts bursty skills, and the undercount is unbounded and - unrecoverable. Contrast `pluginUsage` below, which batches and accumulates. -- Keys are inconsistent between qualified and bare form. This machine holds both - `babysit-prs` (378) and `source-control:babysit-prs` (97) as separate rows for the - same skill. Any consumer must sum both spellings. - -### 1.2 `pluginUsage` (machine-global, per plugin) - -Shape: `{"@": {"usageCount": , "lastUsedAt": , "lastUsedNumStartups": }}`. -118 entries on this machine. - -Write paths in the binary: - -- `Sme(e)`: batched flush, `usageCount: (d?.usageCount ?? 0) + u.count`. Accumulates, so - unlike `skillUsage` it loses nothing to throttling. -- `yNt(e,t)`: **seeds** absent entries with `usageCount: 0`, `lastUsedAt: now`, - `lastUsedNumStartups: `. -- `dzn(e,t)`: refreshes `lastUsedAt` and `lastUsedNumStartups` on re-enable, with no - usage. - -Consequence, and the binary's own bundled skill states this explicitly: for a plugin, -`lastUsedAt` is usage evidence only when `usageCount > 0`. A zero-count plugin's -`lastUsedAt` is the seed time and means nothing. - -What counts as a plugin "use" is broad. Per the binary's bundled guidance, usage is -recorded whenever a slash command, skill, agent, MCP tool or resource, or hook is -dispatched from that plugin, plus LSP servers delivering diagnostics or code navigation. -That is why hook-only plugins dominate on this machine: - -| plugin | usageCount | -| --- | --- | -| `guardrails@melodic-software` | 200,804 | -| `context-guard@melodic-software` | 106,752 | -| `disk-hygiene@melodic-software` | 100,610 | -| `source-control@melodic-software` | 68,225 | - -`pluginUsage.usageCount` is therefore not comparable across plugins of different shapes. -A hook plugin's count is a per-tool-call tally; a skill plugin's count is an invocation -tally. Ranking plugins by raw count ranks them by hook chattiness. - -`pluginUsageLspGraceAppliedIds` (top level) records which plugin ids got the LSP grace -backfill, so a lifetime zero on an LSP-only plugin may just predate the tracking. - -### 1.3 `agentLastUsed` - -`{"bg": 1784439187904}` and nothing else. Subagent usage is effectively **not** tracked -here. Nothing in this repo can source agent-usage analysis from `~/.claude.json`. - -### 1.4 `projects[]`: per-project, last session only - -28 project entries. Each carries a **snapshot of the last session in that directory**, -not a running total. Verified: `Dke(e)` builds the object fresh from live session getters -each time (`lastCost: ul(), lastAPIDuration: Xg(), ...`), and the matching telemetry event -`tengu_exit` reports these as `last_session_cost`, `last_session_api_duration`, and so on. -Each session end overwrites the previous. - -Fields: `lastCost`, `lastDuration`, `lastAPIDuration`, -`lastAPIDurationWithoutRetries`, `lastToolDuration`, `lastLinesAdded`, `lastLinesRemoved`, -`lastTotalInputTokens`, `lastTotalOutputTokens`, `lastTotalCacheCreationInputTokens`, -`lastTotalCacheReadInputTokens`, `lastTotalWebSearchRequests`, `lastSessionId`, -`lastStartTime`, `lastFpsAverage`, `lastFpsLow1Pct`, `lastSessionMetrics` (frame-duration -percentiles), `lastModelUsage` (per model id: input, output, cache-read, -cache-creation tokens, web-search requests, `costUSD`). - -**The decisive fact for per-project slicing:** `skillUsage` and `pluginUsage` are top -level only. Nothing under `projects` carries them. `~/.claude.json` structurally cannot -answer "which skills does this project use". Per-project cost and token data exists but -only for the single most recent session. - -### 1.5 Other counters, incidental - -`numStartups` (260), `promptQueueUseCount`, `btwUseCount`, `tipsHistory` and -`tipLifetimeShownCounts` (per-tip shown counts), `passesUpsellSeenCount`, -`lspRecommendationIgnoredCount`, `rcLongTurnNudgeSeenCount`. These drive tip cooldowns, -not component usage analysis. - -### 1.6 Growth and the one supported shrink lever - -The file is never swept: `cleanupPeriodDays` does not reach it, since it lives in the home -directory rather than under `~/.claude` -(`plugins/claude-ops/skills/audit-install-state/reference/surfaces.md:104`). The supported -lever is `claude project purge `, which removes one project's entry -(`surfaces.md:108`; `audit-performance/SKILL.md:34`). Every running session polls the file -at 1 Hz (`known-performance-issues.md:199`), and the strongest public report of curing -input lag pruned this file rather than the tree (`:44`). `.claude.json.tmp..` -siblings are failed atomic-write remnants; the leading number only looks like a PID -(`surfaces.md:110`, `install_state.py:1082-1109`). - -## Part 2: the skill-listing budget scorer, recovered exactly - -This is the highest-value find, because the repo currently calls it undocumented. - -The scorer (`zPe`): - -```js -function zPe(e) { - let r = oe().skillUsage?.[e]; - if (!r) return 0; - let o = (Date.now() - r.lastUsedAt) / 86400000, - u = Math.pow(0.5, o / 7); - return r.usageCount * Math.max(u, 0.1); -} -``` - -So the score is `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`: exponential decay with -a **7-day half-life** and a **0.1 floor**. - -Where it is used (verified call sites): - -1. `F1t` and the `skill_listing` attachment builder pass `(cmd) => zPe(cmd.name)` into - `Ymt` / `Jmt` as the priority function. This is the listing-budget truncation. -2. Slash-command menu: the top 5 by `zPe` score are pinned above the alphabetical groups. -3. Slash-command search: `getScoreBoost: (c) => zPe(c.command.name)`. - -The truncation algorithm (`Ymt`), verified: - -- Compute each entry's full length (`name + ": " + description`, description capped). -- If total fits the budget, everyone keeps their description (`budgetMode: "fits"`). -- Otherwise sort the competing entries **descending by score**, then walk the list - greedily granting descriptions while budget remains. Entries that do not fit go into - `budgetTruncatedSkills` and render as `- ` with no description - (`budgetMode: "priority"`). - -Two corrections this forces on the repo's current wording: - -- Truncation is by **decay-weighted score**, not by raw invocation count. A heavily used - but stale skill can sort *below* a lightly used but fresh one. Concretely: 100 uses 60 - days ago scores `100 * 0.1 = 10`; 12 uses today scores `12`. The stale one loses. -- The floor means a never-used skill scores exactly `0` and always loses first, but a - once-used skill never decays below `0.1 * usageCount`. - -### Verification stamp - -Follows `docs/conventions/upstream-drift`, the same shape -`plugins/source-control/hooks/worktree-create-gate.sh` uses for binary-derived claims. - -- **Claim:** the scorer is `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`; it is the - priority function passed to the listing-budget truncator; truncation sorts competing - entries descending by that score and drops descriptions from the tail. -- **Basis:** string extraction of `claude.exe`, Claude Code 2.1.251, functions `zPe`, - `Ymt`, `F1t`, `Fdt`, `Sme`, `yNt`, `dzn`, `Dke`. Confirmed against the live - `~/.claude.json` on this machine. Scorer and truncator re-verified unchanged at 2.1.252. -- **As-of:** 2026-08-31, re-verified against Claude Code 2.1.252. -- **Recheck trigger:** any release note naming the skill listing, the skill-listing budget, - skill usage counters, or `/doctor`'s unused-component check; or the counters' shape in - `~/.claude.json` gaining or losing a field. -- **Locate by shape, never by name.** The minified identifiers move between builds. The - scorer was `zPe` in 2.1.251 and `WPe` in 2.1.252 with a byte-identical body, so a recheck - greping the old name finds nothing and would wrongly conclude the mechanism was removed. - Grep the arithmetic (`Math\.pow\(0\.5,`) and the truncator's own field name - (`budgetTruncatedSkills`) instead. - -This is recovered from one build of a minified bundle. Encoding it in a script means -pinning to that build, so any consumer must carry the stamp and treat a mismatch as -"scorer unknown", falling back to the current name-ordered behavior rather than asserting -a wrong ordering. - -## Part 3: what this repo already has - -### 3.1 Reads `~/.claude.json` counters directly - -`plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py` is -the only consumer of the native counters. - -- `collect_native()` (line 522) reads `--claude-json`, defaulting to `~/.claude.json`. -- `parse_native()` (line 110) turns `skillUsage` rows into events, gated by - `is_usage_evidence()` (line 68), which requires `usageCount > 0` because `lastUsedAt` - alone is not evidence. The docstring records the measurement that justified this: 46 of - 65 plugins looked "used today" while none had been. -- `collect_jsonl()` (line 536) reads the plugin's own `skill-usage.jsonl`. -- `_reconcile()` merges the two sources by taking the max per instant rather than summing, - so double-counting one dispatch seen by both sources is avoided. -- `compute_listing()` (line 703) does the budget arithmetic and the starvation banding. - -The skill's own framing already separates certain arithmetic (does the listing overflow) -from inferential ordering (which skills lose descriptions), and labels the ordering as -resting on "an undocumented scorer pinned to one build". Part 2 above closes that gap. - -### 3.2 Its own second store: `skill-usage.jsonl` - -- `plugins/claude-ops/hooks/skill-usage-audit.sh`: `PostToolUse` with `matcher: "Skill"`, - writes a `SkillUse` event per Skill tool call. -- `plugins/claude-ops/hooks/skill-usage-expansion-audit.sh`: `UserPromptExpansion`, catches - slash-command expansions the tool hook misses. -- `plugins/claude-ops/hooks/claude-ops-paths.sh`: `claude_ops::record_skill_use`, scope - selection (`repo` default, `user`, `data-dir`) via the `skill_usage_scope` userConfig. -- Registered at `plugins/claude-ops/hooks/hooks.json:64-88`. -- Schema and example: `docs/conventions/hook-telemetry/data/skill-usage-audit.schema.json`, - `docs/conventions/hook-telemetry/examples/skill-usage-audit.json`. - -Event shape, from the live store: - -```json -{"ts":"2026-08-23T18:17:02Z","event":"SkillUse","skill":"loop","branch":"unknown", - "project":"KyleSexton","project_id":"kylesexton-e2d95aff","hook":"skill-usage-audit", - "source":"expansion","expansion_type":"slash_command"} -``` - -**This store is not redundant with `~/.claude.json`. It carries the slices the native -counters structurally lack**: `project`, `project_id`, `branch`, `source` (tool vs -expansion), `expansion_type`, and a real timestamp per event rather than one -last-used stamp. It is also immune to the 60-second throttle. The native counter owns the -global lifetime tally; the JSONL owns the sliced event stream. - -### 3.3 Other usage-adjacent surfaces - -- `plugins/claude-ops/skills/observability/`: OTEL DuckDB store, collector, hook-event - JSONL, `ccusage`, cross-session trend reports. -- `plugins/claude-ops/skills/audit-install-state/scripts/install_state.py`: reads - `~/.claude.json` for install-tree inventory, not usage. -- `plugins/claude-ops/skills/audit-performance/scripts/audit_performance.py`: reads - `~/.claude.json` for retention-sweep health and session counts. -- `plugins/claude-config/skills/audit-permission-state/`: reads `~/.claude.json` as a - settings scope, distinct from `~/.claude/settings.json`. -- `plugins/context-budget/skills/audit/reference/levers.json`: `~/.claude.json` as a lever - source. -- `docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md`: the - standing decision that skill recommendation reads the catalog, not the in-context - listing, precisely because the listing is budget-truncated. - -### 3.4 The observability skill does not read the native counters - -`plugins/claude-ops/skills/observability/SKILL.md:121-122` and -`context/data-sources.md` enumerate its lanes: ccusage (tokens, cost, billing blocks), -the OTEL DuckDB store, and `.claude/observability/hook-events.jsonl` (hook duration, exit -codes). `~/.claude.json` is not among them, so the machine-global lifetime counters have -no representation in the cross-session trend reports. `audit-skill-visibility` is the only -skill that joins the native counters to the OTEL and JSONL lanes; the observability -reporter sees two of the three. Its `clean` action does know about -the skill-usage store (`--skill-usage-scope`, `--keep-skill-usage-days`, default 365), but -only to prune it, never to read it. - -`plugins/claude-ops/skills/audit-skill-visibility/scripts/skill-pair-cooccurrence.sh` and -`reference/pair-cooccurrence.md` derive which skills get invoked together from -`skill-usage.jsonl`. That analysis is impossible from `~/.claude.json`, which has no event -stream, and is a second reason the JSONL store earns its place. - -### 3.5 What no surface covers - -- No skill reasons about plugin disuse from `pluginUsage`. `claude-ops:plugins` is version - and scope currency only; its SKILL.md and scripts never mention usage. - `overengineering:audit` reasons about enforcement surfaces earning their keep but sources - evidence from CI and hook behavior, not from these counters. -- Nothing reads `agentLastUsed`. -- Nothing reads `projects[].lastModelUsage` or the per-project cost and token - snapshot, despite that being the only per-project slice `~/.claude.json` offers. - -### 3.6 The audit is a three-source reconciler with a capability tier model - -`audit_skill_visibility.py` does not merely read the native counters. It has three parse -lanes and gates every claim on which lanes are present: - -- `:110-133` `parse_native()`, `~/.claude.json` `skillUsage`, horizon `firstStartTime`. -- `:136-161` `parse_jsonl()`, the plugin's own `skill-usage.jsonl`. -- `:164-187` `parse_otel()`, the OTEL event `claude_code.skill_activated`, which carries - `invocation_trigger`. Flags the `custom_skill` redaction placeholder and leaves it - unattributed. - -`TIER_CAPABILITIES` (`:81-88`) with `resolve_tier()` (`:93-101`): - -| tier | source | claims supported | -| --- | --- | --- | -| `T-full` | otel | `invocation_trigger`, `windowed_count`, `per_repo`, `lifetime_count` | -| `T-local` | jsonl | `windowed_count`, `per_repo`, `lifetime_count` | -| `T-baseline` | native | `lifetime_count` only | -| `T-none` | none | nothing | - -The `T-baseline` restriction encodes exactly the Part 1 finding: native counters are -lifetime-since-install and never windowed, so no windowed claim is honest from them alone. - -`invocation_trigger` separates `user-slash` from `claude-proactive`. That is the axis a -"does the model reach for this unprompted" question needs, and neither counter can answer -it. `reference/pair-cooccurrence.md:50-54` records why the gap cannot be closed by widening -the hook: a `PostToolUse` hook on the `Skill` tool receives `tool_name`, `tool_input` and -`tool_response`, and none of them names the skill whose instructions caused the call. -Caller identity is absent from the hook's input, not merely from its schema, so a wider -write would have nothing to write. Recovering it means reading the session transcript, -which crosses the boundary `plugins/claude-ops/skills/observability/context/privacy.md` -guards. - -### 3.7 Transcript scraping, which is where agent usage actually lives - -`plugins/session-flow/skills/retro/scripts/parse_transcript.py` parses Claude Code session -transcripts for quantitative metrics, with multi-session aggregation and handoff-chain -walking. Per `plugins/session-flow/skills/retro/context/session.md:98` it extracts -compactions, total context tokens, tool rejections, **subagent count**, and a tool -distribution breakdown. `plugins/session-flow/skills/running-retro/scripts/observer.py` -runs the same analysis in flight or detached, appending to a cumulative ledger. - -Given that `agentLastUsed` holds one key, this transcript lane is the only working source -for agent usage in the repo. - -### 3.8 Governance constraints already recorded - -Three refusals are already codified. Each is recorded below with the rationale it actually -stands on, so a reader can tell a live justification from inertia. Two name a purpose and a -measurement; the third names only the state of the substrate, and that substrate moved this -session, so it is re-derived in Part 5 rather than obeyed on sight. - -- **`docs/adr/0016-...:118-120`** deliberately defers usage-metrics-driven surfacing: - "Usage-metrics-driven surfacing (`~/.claude.json` `skillUsage`, undocumented internal - state) stays deferred; rotation runs off a ledger the skill writes itself, which is what - keeps that deferral honest rather than load-bearing." This constrains Part 5: encoding - `zPe` is a diagnostic-report change, not a licence to route skill *recommendation* off - these counters. -- **Secret safety, scoped to the performance engine.** - `plugins/claude-ops/skills/audit-performance/scripts/audit_performance.py:17-19` allowlists - content reads to `settings.json`, `.last-cleanup`, `hooks.json` and - `installed_plugins.json`, and holds `~/.claude.json` and `history.jsonl` stat-only - "whose values can carry tokens and prompts". Asserted by - `test_audit_performance.py:42-49`. `audit-install-state` takes the same stat-only posture - (`install_state.py:1082-1109`). This is a per-engine safety rule, not a repo-wide ban: - `audit-skill-visibility` reads the file deliberately, and reads only `firstStartTime` and - `skillUsage`. -- **A short observation horizon must yield a withheld verdict, not a zero.** - `audit_skill_visibility.py:53-65` sets `exposure_floor_days = 30`; a three-day-old install - measured against 30 and 90 day tiers put 210 of 213 skills in a "never used" bucket. Every - window clamps to `observed_horizon` and unsupported claims route to a first-class - `withheld` section. - -### 3.9 Native overlap: a candidate for the existing registry, not a verdict here - -Native-overlap verdicts are owned by `/claude-ops:audit-native-overlap`, recorded in -`docs/native-surfaces/records.json` and rendered into `docs/NATIVE-SURFACES.md`. This -section raises a candidate for that machinery rather than deciding it, because verdicts -there are human-gated by contract. - -Observed in the 2.1.251 binary: - -- A bundled skill documenting these exact counters and their traps, in prose closely - matching this repo's own conclusions: "`usageCount` is a LIFETIME total since install", - the `pluginUsage` seeding caveat, and the qualified-vs-bare key split. -- A `doctor`-side check, "Check 1: unused skills, MCP servers, and plugins", grouping - unused components against their context cost and offering to disable them, labelled per - group with a benefit estimate ("37 unused skills, saves ~2.2k est. tokens/session"). - -The store already holds `doctor` to `claude-ops:audit-install-state` and `doctor` to -`claude-ops:audit-performance`, both `complementary`, verified 2026-08-23. It holds **no -row for `doctor` to `claude-ops:audit-skill-visibility`**, even though `doctor`'s Check 1 -and that skill answer overlapping questions, and even though the skill's own SKILL.md -already disclaims the one-shot check as native territory in prose. Prose disclaimer without -a store row is exactly the drift the registry exists to catch. - -Suggested framing for the human deciding it: the native check is a one-shot -unused-versus-context-cost prompt that offers to disable; `audit-skill-visibility` is a -three-source reconciler with a capability tier model, a withheld-verdict discipline, and a -listing-budget starvation analysis, and it disables nothing. That reads `complementary` on -the same shape as the two existing `doctor` rows, but the verdict is not this document's to -record. - -## Part 4: open findings, unverified or needing a decision - -1. **CONFIRMED: `compute_listing` ranks on a `usage_score` nothing populates, so the - starvation band is alphabetical.** `compute_listing` (line 703) reads - `entry.get("usage_score", 0)` off the denominator and sorts `competing` by - `(usage_score, qualified_name)` (line 736). The denominator is built by - `collect_installed` / `collect_fleet_at` from a filesystem walk, which never sets - `usage_score`. Usage events are joined to entries later, inside `classify` (line 915 - onward), after `compute_listing` has already run (line 897). Only the test fixture - `tests/fixtures/fleet-overbudget.json` supplies non-zero values, which is why the tests - do not catch this. - - Verified on a live run - (`--plugins-root plugins --claude-json ~/.claude.json --render json`, - 2026-08-31): `[.skills[].starvation.usage_score] | unique` returns `[0]`. Every row - scores zero, so the tiebreaker decides everything and bands come out in name order: - - | band | usage_score | observed count | skill | - | --- | --- | --- | --- | - | 1 | 0 | 1 | `adhd:clarify` | - | 2 | 0 | 1 | `adhd:shape` | - | 5 | 0 | 8 | `architecture:improve` | - | 150 | 0 | 97 | `source-control:babysit-prs` | - | 173 | 0 | 99 | `work-items:triage` | - - Band 1 is "most likely starved". The two least-used skills in the fleet are ranked as - the first to lose their descriptions and the two most-used are ranked as the safest, - but only because `a` sorts before `w`. The report's own inferential ordering, the part - it warns is uncertain, currently carries no usage signal at all. Same run: - `budget_chars` 8000, `demand_chars` 130330, `overflow_chars` 122330, 167 of 176 - competing skills marked `likely-starved`. - -2. **No `skill-usage.jsonl` writes observed in this repo since 2026-08-11. Cause not - established.** The main checkout's `.claude/observability/skill-usage.jsonl` last - changed 2026-08-11 19:34 (107 lines) while `hook-events.jsonl` in the same directory is - live (2026-08-31 11:27). The last `skill-usage-audit` telemetry envelope is - `2026-08-11T23:34:24Z`. No worktree under `D:\worktrees` holds a `skill-usage.jsonl` at - all. - - Checked and ruled out: the hook is registered - (`plugins/claude-ops/hooks/hooks.json:64-75`), `claude-ops@melodic-software` is enabled - in `.claude/settings.json:36`, and no `skill_usage` key (scope or kill switch) appears - in `~/.claude/settings.json`, `.claude/settings.local.json`, or `~/.claude.json`. - - Not ruled out: the same hook did write to `~/.claude/observability/` as recently as - 2026-08-23, from a session whose resolved repo root was the home directory - (`"project":"KyleSexton"`). So the hook itself works. Whether this repo's sessions - stopped dispatching the `Skill` tool, or the write is landing somewhere unexpected, is - unresolved and needs a live probe rather than more file archaeology. - -3. **Nothing consumes `agentLastUsed`, and it holds one key.** Any agent-usage question has - to come from transcripts or the OTEL store. - -4. **Not a gap: the 60-second throttle is already documented.** `SKILL.md:194-196` states - it exactly, including that the debounce suppresses the timestamp refresh too, and rules - that OTEL and native divergence "must not be reconciled away". The binary read in Part - 1.1 corroborates the repo's claim independently: `Fdt` returns before both the count and - the timestamp write. Recorded here so a future pass does not re-file it as a finding. - -5. **CONFIRMED: the qualified-vs-bare key split silently drops events.** Denominator - entries are keyed `f"{plugin}:{leaf}"` (`:276`) and `classify()` looks events up by that - `qualified_name` alone (`:926`, `events_by_skill.get(name)`). `parse_native()` emits - events under whatever raw key `skillUsage` holds, so a bare-key row never matches its - qualified entry and is discarded without a withheld note. Live data holds `babysit-prs` - (378) and `source-control:babysit-prs` (97) as separate rows; the audit run in finding 1 - reported `source-control:babysit-prs` at 97, not 475. Claude Code's own lookup helper - (`oKn` in the binary) checks the qualified key then falls back to the bare one, which is - the behavior to match. - -## Part 5: recommendation shape - -Do not rebuild a counter. `~/.claude.json` already owns the global lifetime tally, and its -scorer is now known exactly. - -### Where ADR 0016 does and does not reach - -An earlier draft of this section treated ADR 0016 as constraining the work below and framed -it as a cautious exception. A cross-vendor re-derivation, run blind to that reasoning, -found the framing an over-read, and the correction runs in both directions. - -The deferral at `docs/adr/0016-...:118-120` is scoped to one skill's rotation: which of -`show-options`'s Spotlight three to surface. Encoding `zPe` inside `audit-skill-visibility` -to predict the listing's own truncation is a different skill answering a different question, -"what will the listing drop", not "what should we recommend". It was never inside the -clause. The work below stands on its own ground rather than as a permitted exception. - -In the other direction, the deferral is not up for lifting just because its stated premise -moved. Its ground shifts from "undocumented internal state" to something documentation -cannot cure: - -- `zPe` is **wrong-signed** for the question `show-options` asks. It scores high for skills - used recently and often, which are exactly the skills the operator has not forgotten. - Inverting it collapses to a decay-weighted least-recently-used ordering, which the - self-written ledger already supplies without a build-pinned dependency. -- `skillUsage` lists only skills that have fired at least once (131 entries here). It - structurally cannot name the never-invoked skills that `show-options`'s no-omission rule - exists to protect. -- It carries no causal-trigger field, so it cannot answer take-up: was a skill invoked - *because* it was surfaced, or for an unrelated reason. `skill-usage.jsonl` can. That alone - means the ledger is not a stopgap for missing documentation; it measures something these - counters never will. - -And the stamp in Part 2 is not the same thing the ADR meant by documented. The fallback rule -recorded there, treat a mismatch as "scorer unknown" and degrade to name ordering, is what -ships alongside ground that can shift without notice, not alongside a documented API. The -recheck trigger fires when a human notices a release note, so the failure mode is silent and -wrong until someone reads a changelog. `docs/conventions/upstream-drift` existing as a named -convention is this repo's own admission that binary-derived claims are a weaker evidentiary -class. This session made the counters' undocumented-ness precisely characterized and dated. -That is not the same as making them documented. - -Conditions under which the deferral would genuinely be revisitable, for whoever comes back -to it: a published, versioned surface for skill-usage data rather than a reverse-engineered -internal; a rotation signal not decay-weighted toward recent use; and take-up attribution. -The third is unreachable from `skillUsage` by construction, so the ledger survives whatever -happens to the first two. - -### The gaps worth closing - -- Encode `zPe` in `audit_skill_visibility.py` and populate `usage_score` before - `compute_listing` runs (Part 4 finding 1, plus Part 2's decay formula). -- Resolve qualified and bare `skillUsage` keys to one entry, matching `oKn`'s - qualified-then-bare fallback rather than summing blindly (Part 4 finding 5). -- Keep `skill-usage.jsonl` as the per-project / per-branch / per-source slice, and fix - whatever stopped it (Part 4 finding 2). That is the only local source for the granular - slices the user asked about. -- Treat `pluginUsage.usageCount` as non-comparable across plugin shapes wherever it is - ranked. - -### The correction owed to ADR 0016, applied - -`docs/adr/0016-...:19-20` states that Claude Code "drops descriptions starting with the -skills invoked least". Part 2 shows the ordering is decay-weighted, so a heavily used but -stale skill can lose its description before a lightly used fresh one. The stated mechanism -is wrong, not merely imprecise, and the harm is not hypothetical: Part 4 finding 1 shows -`compute_listing` sorting on a `usage_score` nothing populates, so its starvation bands are -alphabetical order presented as usage-informed. Someone already built on the mental model -that ADR line encodes. - -The ADR's core decision is untouched and in fact reinforced, so the fix is a dated revision -blockquote in the ADR's own established shape, which preserves superseded reasoning rather -than editing Context in place. - -Two revisions were applied, both dated 2026-08-31: - -- After the Context paragraph on drop order: corrects the mechanism to the decay-weighted - score, and records that the correction strengthens rather than weakens the decision. A - never-invoked skill still scores zero and is shed first, so the identified bias holds; the - decay term adds a second bias the paragraph did not anticipate, against skills the - operator used a while ago and has since forgotten, which is the population `show-options` - exists to surface. The budget arithmetic is unaffected, since it measures demand against - budget rather than order of shedding. -- After the deferral clause: restates its ground on the three documentation-independent - reasons above, and records the lift conditions so the next reader does not re-derive them. - -Both revisions cite issue 3534, which carries the two defects in Part 4, matching the -citation shape of the ADR's two 2026-08-21 revisions. diff --git a/plugins/claude-ops/CHANGELOG.md b/plugins/claude-ops/CHANGELOG.md index 59ac020ce7..7a85113fe7 100644 --- a/plugins/claude-ops/CHANGELOG.md +++ b/plugins/claude-ops/CHANGELOG.md @@ -30,7 +30,9 @@ All notable changes to the `claude-ops` plugin are documented here. Format follo recheck trigger. The ordering is decay-weighted, so "least invoked" was never the right description of it: a heavily used but stale skill can rank below a lightly used fresh one. Evidence in - `docs/topics/usage-tracking-claude-json/EXPLORE.md`. + `skills/audit-skill-visibility/reference/listing-scorer.md`, with the counters' own + semantics and their seeding and throttle traps in the companion + `reference/usage-counters.md`. - **`listing.score_basis`, so an unscored band admits it.** When no usage survives to weigh, the order is alphabetical and nothing more; the basis reads `unscored` and competing rows carry `confidence: "unscored"` rather than diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index b1a7aecd1e..a38866f0fe 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -168,7 +168,7 @@ a user as documented. - **Ambiguous attribution is reported, not guessed.** Two marketplaces shipping a same-named plugin collapse to one usage key; those rows are marked `ambiguous-attribution` rather than attributed to one of them. -- **One skill, two possible usage keys.** The stores hold the qualified +- **Two possible usage keys per skill.** The stores hold the qualified `:` key and the bare leaf as separate rows. Both are collected; a bare key is attributed only when exactly one skill owns that leaf, ambiguous ones are withheld with their candidates. diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md index 76805c2e2e..ac9a1722c2 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md @@ -21,9 +21,8 @@ Recovered from `claude.exe` at Claude Code 2.1.251 (`zPe` the scorer, `Ymt` the truncator), then re-verified unchanged at 2.1.252. The scorer has two further call sites, the slash-menu top-five pin and the command-search score boost, which is corroboration that it is the product's general usage-priority function rather -than a listing-local helper. The surrounding evidence is in -[`docs/topics/usage-tracking-claude-json/EXPLORE.md`](https://github.com/melodic-software/claude-code-plugins/blob/main/docs/topics/usage-tracking-claude-json/EXPLORE.md), -Part 2. +than a listing-local helper. What the counts it reads actually mean, including +the seeding and throttle traps, is [usage-counters.md](usage-counters.md). ## Why "least invoked" was the wrong description diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md new file mode 100644 index 0000000000..a2b647feb6 --- /dev/null +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/usage-counters.md @@ -0,0 +1,130 @@ +# The `~/.claude.json` usage counters + +Read this when a count this skill reports looks wrong, or before writing a new +consumer of these counters. It records what each region actually means, and the +four traps that make a naive read wrong. `~/.claude.json` is the user-scope state +file, a different file from `~/.claude/settings.json`. + +Companion: [listing-scorer.md](listing-scorer.md), which covers how the product +ranks these counts when it truncates the skill listing. + +## Verification stamp + +Follows `docs/conventions/upstream-drift`. + +- **Claim:** the write paths, seeding behavior, and per-region semantics below. +- **Basis:** string extraction of `claude.exe` at Claude Code 2.1.251, the write + and read functions named per region, confirmed against a live `~/.claude.json`. + Scorer and truncator re-verified unchanged at 2.1.252. +- **As-of:** 2026-08-31. +- **Locate by shape, never by name.** Minified identifiers move between builds; + the listing scorer was `zPe` at 2.1.251 and `WPe` at 2.1.252 with an identical + body. Grep the arithmetic and the field names, not the function names. +- **Recheck trigger:** a release note naming skill usage counters, the skill + listing, or `/doctor`'s unused-component check; or these keys gaining or losing + a field. + +## `skillUsage`, machine-global, per skill + +`{"": {"usageCount": , "lastUsedAt": }}` + +Write path, reduced to its decisive lines: + +```js +let d = o.skillUsageLastWriteAt.get(t); +if (d !== void 0 && u - d < 60000) return; +o.skillUsageLastWriteAt.set(t, u), + Ae((y) => ({ ...y, skillUsage: { ...y.skillUsage, + [t]: { usageCount: (k?.usageCount ?? 0) + 1, lastUsedAt: u } } }), r) +``` + +- Written on real skill dispatch only. **No install-time or session-start + seeding**, so a skill's `lastUsedAt` is trustworthy evidence of actual use. + This is the one place it is trustworthy; see `pluginUsage` below. +- `usageCount` is a lifetime total since install. It never resets and is never + windowed, which is why the engine's `T-baseline` tier supports a lifetime + claim and refuses a windowed one. +- **The 60-second per-skill throttle DROPS the increment, it does not coalesce + it.** A skill invoked five times in one minute records one. The undercount is + unbounded and unrecoverable, so for loop-driven or rapid-fire skills this + counter is a floor rather than a count. Divergence from OTEL is expected here + and must not be reconciled away. +- **Keys are inconsistent between qualified and bare form.** One machine held + both `babysit-prs` (378) and `source-control:babysit-prs` (97) as separate rows + for the same skill. Every consumer has to resolve both spellings. Note that the + product's own listing scorer does NOT do this fallback, which is why the + starvation band deliberately does not either. + +## `pluginUsage`, machine-global, per plugin + +`{"@": {"usageCount": , "lastUsedAt": , "lastUsedNumStartups": }}` + +Three write paths, and the second is the trap: + +- Batched flush accumulates: `usageCount: (d?.usageCount ?? 0) + u.count`. Unlike + `skillUsage`, nothing is lost to throttling. +- Absent entries are **seeded** with `usageCount: 0`, `lastUsedAt: now`, + `lastUsedNumStartups: ` on install or enable. +- `lastUsedAt` and `lastUsedNumStartups` are **refreshed on re-enable**, with no + usage at all. + +So for a plugin, `lastUsedAt` is usage evidence only when `usageCount > 0`. A +zero-count plugin's `lastUsedAt` is the seed time and means nothing. Measured on +one machine: 46 of 65 plugins looked "used today" while none had been used. That +measurement is what `is_usage_evidence()` in the engine exists to enforce. + +**A plugin "use" is broad and not comparable across plugin shapes.** Usage is +recorded whenever a slash command, skill, agent, MCP tool or resource, or hook is +dispatched from that plugin, plus LSP servers delivering diagnostics or code +navigation. Hook-only plugins therefore dominate any raw ranking: on one machine +`guardrails` showed 200,804 and `context-guard` 106,752, against low hundreds for +skill-shaped plugins. A hook plugin's count is a per-tool-call tally; a skill +plugin's count is an invocation tally. Ranking plugins by raw count ranks them by +hook chattiness, not by value. + +`pluginUsageLspGraceAppliedIds` at the top level records which plugin ids got the +LSP grace backfill, so a lifetime zero on an LSP-only plugin may simply predate +the tracking. + +## `agentLastUsed` + +Observed holding a single key (`bg`). Subagent usage is effectively **not tracked +here**, so no agent-usage analysis can be sourced from this file. The transcript +parsers in the session-flow retro skills are the working source for that. + +## `projects[]`, per-project, last session only + +Each entry is a **snapshot of the last session in that directory, not a running +total**. The object is rebuilt from live session getters at each write, and the +matching exit telemetry reports the same values as `last_session_*`. Every +session end overwrites the previous one. + +It carries cost, wall and API durations, lines added and removed, the four token +totals, web-search requests, session id and start time, frame-duration +percentiles, and `lastModelUsage` broken down per model id. + +**The decisive fact for per-project usage questions:** `skillUsage` and +`pluginUsage` are top level ONLY. Nothing under `projects` carries them, so this +file structurally cannot answer "which skills does this project use". That +question needs the plugin's own `skill-usage.jsonl` store, whose rows carry +`project`, `project_id`, and `branch`, or OTEL. + +## Growth, and the one supported shrink lever + +The file is never swept: `cleanupPeriodDays` does not reach it, because it lives +in the home directory rather than under `~/.claude`. The supported lever is +`claude project purge `, which removes one project's entry. Every running +session polls the file at 1 Hz, and the strongest public report of curing input +lag pruned this file rather than the install tree. + +`.claude.json.tmp..` siblings are failed atomic-write remnants; the +leading number only looks like a PID. See +[`audit-install-state/reference/surfaces.md`](../../audit-install-state/reference/surfaces.md) +for the install-tree side of this, which holds the file stat-only because its +values can carry tokens. + +## Incidental counters, not usage signals + +`numStartups`, `promptQueueUseCount`, `tipsHistory`, `tipLifetimeShownCounts`, +`passesUpsellSeenCount`, `lspRecommendationIgnoredCount`. These drive tip +cooldowns and onboarding state, not component usage analysis. diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index a36445c0d1..c1084c15e5 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -120,13 +120,15 @@ def tier_supports(tier: str, claim: str) -> bool: # the scorer, `Ymt` the truncator, plus the scorer's two other call sites, the # slash-menu top-5 pin and the command-search score boost), then RE-VERIFIED # unchanged at 2.1.252. Evidence in -# docs/topics/usage-tracking-claude-json/EXPLORE.md, Part 2. +# reference/listing-scorer.md, beside this skill. # As-of: 2026-08-31, re-verified against Claude Code 2.1.252. # Locate it by SHAPE, never by name. The minified identifier is not stable across # builds: the scorer was `zPe` in 2.1.251 and `WPe` in 2.1.252, with a # byte-identical body. Grep for the arithmetic instead, e.g. # `grep -a -o -E '.{0,180}Math\.pow\(0\.5,.{0,180}' `, and for # the truncator `budgetTruncatedSkills`. +# Reasoning and the counters' own semantics: reference/listing-scorer.md and +# reference/usage-counters.md, beside this skill. # Recheck trigger: a release note naming the skill listing, its character budget, # or skill usage counters; or the counters changing shape in `~/.claude.json`. # On mismatch: the report must degrade to `score_basis: "unscored"` and say the From 916367b5bec5a88ace2beefb4d4431af94f26582 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 19:52:09 -0400 Subject: [PATCH 17/22] fix(claude-ops): model the listing truncation as first-fit, not a score prefix The adjudicator's independent re-extraction found the mirror wrong on the half I had not checked. The product's grant loop has NO early exit: for (let me of W) { let ge = me.entryLen - (me.cmd.name.length + 2); if (ge <= pe) pe -= ge; else fe.push(me); } So truncation is a greedy first-fit walk over the whole score-descending list, not a prefix of it. A cheap low-scored description can be granted after an expensive higher-scored one was refused, which makes description LENGTH a second ranking input that no prose account of this mechanism mentions. compute_listing modelled a prefix, which understated the exposure of long descriptions and overstated it for short ones. The verdict now mirrors the real walk. The band keeps ranking exposure by score, and the two are allowed to disagree; a regression test pins exactly that case, where a band-1 row survives a pass that sheds a better-scored longer row. Budget accounting is deliberately NOT changed here. It counts description bytes against the whole budget and ignores the name bytes every entry also pays. That is the CERTAIN half of the report and a separate correction with its own evidence. Also corrects the attribution. The ADR's drop-order sentence tracks the official documentation, which states the same false claim verbatim ("drops descriptions starting with the skills you invoke least, so the skills you use most keep their full text"). The revision now names upstream as the source of the error instead of implying this repo got it wrong, and the SKILL.md paragraph that cited that page as authority now says which part of it does not hold. The plugin.json description clause is corrected the same way. Suite 94/94, ruff clean, markdownlint clean. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- ...dation-from-the-catalog-not-the-listing.md | 24 +++++--- plugins/claude-ops/.claude-plugin/plugin.json | 2 +- .../skills/audit-skill-visibility/SKILL.md | 19 +++--- .../reference/listing-scorer.md | 22 ++++++- .../scripts/audit_skill_visibility.py | 61 ++++++++++++------- .../scripts/test_audit_skill_visibility.py | 49 +++++++++++++++ verify-posttooluse-probe.md | 4 ++ 7 files changed, 138 insertions(+), 43 deletions(-) create mode 100644 verify-posttooluse-probe.md diff --git a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md index 947b5f8a97..766a7c8984 100644 --- a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md +++ b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md @@ -28,15 +28,21 @@ cannot tell that it is blind. The gatekeeping the contract bans would have been harness, invisibly. > **Revised 2026-08-31 ([#3534](https://github.com/melodic-software/claude-code-plugins/issues/3534)):** -> the drop-order mechanism above is stated wrongly. Claude Code does not drop -> descriptions "starting with the skills invoked least". It ranks by a decay-weighted score, -> `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, sorts descending, and grants descriptions -> greedily until the budget runs out; what does not fit renders as a bare name. So a heavily used but -> stale skill can lose its description before a lightly used fresh one: 100 uses 60 days ago scores -> 10 and loses to 12 uses today. Recovered from the Claude Code 2.1.251 binary, re-verified -> unchanged at 2.1.252, and stamped in -> [`plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md`](../../plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md), -> which carries the basis and the recheck trigger. +> the drop-order mechanism above is wrong, and **the error is upstream's, not this ADR's.** The +> sentence tracks the official documentation, which states it in the same terms: "When the listing +> overflows, Claude Code drops descriptions starting with the skills you invoke least, so the skills +> you use most keep their full text" +> (, fetched 2026-08-31). The shipped binary does something +> else, on two independent axes. +> +> It ranks by a decay-weighted score, `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, so a +> heavily used but stale skill can be shed before a lightly used fresh one: 100 uses 21 days ago +> scores 12.5 and loses to 13 uses today. And it then walks every entry in that order with a running +> budget, granting whatever still fits, with no early exit, so a cheap never-invoked description can +> be granted after an expensive well-used one was refused. Description length is a second ranking +> input the documented account does not mention. Both were recovered from the 2.1.251 binary and +> re-verified at 2.1.252; the stamp, the greps, and the counterexamples are in +> [`plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md`](../../plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md). > > **The ADR's core decision is untouched, and this correction strengthens the case for it.** A > never-invoked skill scores exactly zero and is still shed first, so the bias this paragraph diff --git a/plugins/claude-ops/.claude-plugin/plugin.json b/plugins/claude-ops/.claude-plugin/plugin.json index a5705c5116..5d669e269d 100644 --- a/plugins/claude-ops/.claude-plugin/plugin.json +++ b/plugins/claude-ops/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "claude-ops", "version": "0.39.0", - "description": "Claude Code operations toolkit. Twelve skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used \u2014 a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which drops descriptions least-invoked-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface \u2014 every built-in CLI command with aliases and hidden/gated status, every bundled skill, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json \u2014 full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the silent unparsable-settings pause, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labelled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces \u2014 built-in CLI commands, bundled skills, plugin-backed built-ins, session-provided skills \u2014 against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry \u2014 OTEL store, collector, hook-event JSONL, ccusage \u2014 with trend reports and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and integrate them into the current repo), plugins (bring a machine's plugin fleet current on demand \u2014 marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view \u2014 queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action \u2014 an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry lives. Plus a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures \u2014 the last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that maps envelopes into the hook-events.jsonl the observability skill reads.", + "description": "Claude Code operations toolkit. Twelve skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used \u2014 a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which sheds descriptions lowest-score-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface \u2014 every built-in CLI command with aliases and hidden/gated status, every bundled skill, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json \u2014 full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the silent unparsable-settings pause, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labelled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces \u2014 built-in CLI commands, bundled skills, plugin-backed built-ins, session-provided skills \u2014 against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry \u2014 OTEL store, collector, hook-event JSONL, ccusage \u2014 with trend reports and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and integrate them into the current repo), plugins (bring a machine's plugin fleet current on demand \u2014 marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view \u2014 queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action \u2014 an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry lives. Plus a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures \u2014 the last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that maps envelopes into the hook-events.jsonl the observability skill reads.", "author": { "name": "Melodic Software", "email": "info@melodicsoftware.com" diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index a38866f0fe..496811cd35 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -23,16 +23,19 @@ my skill fleet never get used?* A skill the model cannot see cannot be chosen, s Claude Code budgets the model-visible skill listing at a fraction of the context window (`skillListingBudgetFraction`, default 0.01) and, when it overflows, -**drops descriptions starting with the skills you invoke least**. Names always -survive, descriptions do not. A skill at zero usage therefore loses its +**sheds descriptions from the lowest-scoring skills first**. Names always +survive, descriptions do not. A skill at zero usage scores zero, so it loses its description, loses the keywords a request would match against, and stays at -zero. Unused is partly self-causing, and the loop is documented: the budget -fraction and per-entry cap are owned by +zero. Unused is partly self-causing. + +The budget fraction and per-entry cap are owned by (`skillListingBudgetFraction`, -`skillListingMaxDescChars`) and the drop behavior by - ("Skill descriptions are cut short"). -Verified 2026-08-31; recheck trigger: a fetch of either page no longer matching -this paragraph re-derives it and the scripts' `ListingConfig` defaults. +`skillListingMaxDescChars`). **The drop ORDER is not documented accurately.** The +skills page says "starting with the skills you invoke least"; the binary ranks by +a decay-weighted score and then walks the list first-fit, so neither the ordering +nor the guarantee holds as written. Measured against the binary, with the +counterexamples and the stamp: +[reference/listing-scorer.md](reference/listing-scorer.md). So the useful question is not *which skills are unused*. Claude Code already reports that in `/doctor` and the Stats tab. It is **which skills are starved by diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md index ac9a1722c2..e7a9e35e11 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md @@ -14,8 +14,26 @@ Claude Code ranks skills for description truncation by usageCount * max(0.5 ** (daysSinceUse / 7), 0.1) ``` -then sorts that score descending, grants descriptions greedily until the budget -is spent, and renders the remainder name-only. +then sorts that score descending and walks **every** competing entry with a +running description budget, granting whatever still fits and rendering the rest +name-only. + +**It is a greedy first-fit walk, not a score-ordered prefix.** The grant loop has +no early exit, so a cheap low-scored description can still be granted after an +expensive higher-scored one was refused. Description LENGTH is therefore a second +ranking input, which no prose account of this mechanism mentions: + +```js +for (let me of W) { + let ge = me.entryLen - (me.cmd.name.length + 2); + if (ge <= pe) pe -= ge; else fe.push(me); // no break +} +``` + +This matters for the report's two fields. The `verdict` mirrors the walk. The +`band` ranks exposure, lowest score first, and the two are allowed to disagree: +a band-1 row with a very short description can survive a pass that sheds a +better-scored row with a long one. Recovered from `claude.exe` at Claude Code 2.1.251 (`zPe` the scorer, `Ymt` the truncator), then re-verified unchanged at 2.1.252. The scorer has two further diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index c1084c15e5..2193cb59fe 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -114,8 +114,10 @@ def tier_supports(tier: str, claim: str) -> bool: # -- Verification stamp (docs/conventions/upstream-drift) ---------------------- # Claim: the product ranks skills for description truncation by # `usageCount * max(0.5 ** (daysSinceUse / 7), 0.1)`, sorts that score -# descending, grants descriptions greedily until the budget is spent, and -# renders the remainder name-only. +# descending, then walks EVERY competing entry with a running description +# budget, granting whatever fits and rendering the rest name-only. The walk is +# greedy first-fit with no early exit, so it is not a score-ordered prefix and +# description length is a second ranking input. # Basis: string extraction of `claude.exe`, first at Claude Code 2.1.251 (`zPe` # the scorer, `Ymt` the truncator, plus the scorer's two other call sites, the # slash-menu top-5 pin and the command-search score boost), then RE-VERIFIED @@ -838,20 +840,40 @@ def compute_listing( score_basis = "native-counters" if any(r["usage_score"] for r in rows) else "unscored" competing = [r for r in rows if r["eligibility"] == "competing"] - # Lowest score first: that is the order the product sheds descriptions in, so - # rank 1 is the most likely to have already lost its. The score is - # decay-weighted, NOT a raw invocation count -- a heavily used but stale - # skill can sort below a lightly used fresh one, which is why this cannot be - # read as "least invoked". The name is a tiebreaker only; when `scores` is - # empty it is the WHOLE ordering, which is what `score_basis` exists to admit. - competing.sort(key=lambda r: (r["usage_score"], r["qualified_name"])) - # Descriptions are shed lowest-score-first only UNTIL the listing fits, - # so the starved set is the prefix whose demand covers the overflow -- not - # the whole fleet. Marking every competing row starved on a one-character - # overflow would libel exactly the skills the mechanism protects longest, - # and the renderer would tell the user they are running name-only. - remaining = overflow - for rank, row in enumerate(competing, start=1): + + # WHICH rows are shed is a GREEDY FIRST-FIT walk, not a score-ordered + # prefix. The product sorts descending by score and then walks EVERY + # competing entry with a running description budget, granting whatever still + # fits and shedding whatever does not. Crucially its loop has no early exit, + # so a cheap low-scored description can still be granted after an expensive + # higher-scored one was refused. Modelling this as a prefix understated the + # protection long descriptions lose and overstated it for short ones. + # + # Consequence worth stating plainly: description LENGTH is a ranking input, + # which no prose description of this mechanism mentions. + competing.sort(key=lambda r: (-r["usage_score"], r["qualified_name"])) + # Budget accounting stays as it was: `demand` counts description bytes + # against the whole budget and ignores the name bytes every entry also pays. + # That understates pressure slightly and is the CERTAIN half of this report, + # so it is deliberately not changed here alongside the ordering fix. Charging + # names too is a separate correction with its own evidence. + remaining = budget + for row in competing: + if overflow <= 0: + row["verdict"] = "listing-fits" + elif row["demand_chars"] <= remaining: + remaining -= row["demand_chars"] + row["verdict"] = "likely-retained" + else: + row["verdict"] = "likely-starved" + + # The band is a separate question from the verdict: it ranks how exposed a + # row is, lowest score first, so band 1 is the row the mechanism protects + # least. It can disagree with the verdict, and that disagreement is real + # rather than a bug -- a band-1 row with a very short description can survive + # a first-fit pass that sheds a better-scored row with a long one. + by_exposure = sorted(competing, key=lambda r: (r["usage_score"], r["qualified_name"])) + for rank, row in enumerate(by_exposure, start=1): row["band"] = rank if overflow > 0 else None # An unscored ordering is alphabetical, so its band carries no signal at # all. That is a weaker claim than an inferential one and must not wear @@ -862,13 +884,6 @@ def compute_listing( row["confidence"] = "unscored" else: row["confidence"] = "inferential" - if overflow <= 0: - row["verdict"] = "listing-fits" - elif remaining > 0: - row["verdict"] = "likely-starved" - remaining -= row["demand_chars"] - else: - row["verdict"] = "likely-retained" for row in rows: if row["eligibility"] != "competing": row["band"] = None diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py index 5caea4202a..562e21eef6 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py @@ -923,6 +923,55 @@ def test_most_used_skill_is_never_starved_while_others_can_absorb_it(self): hottest = max(listing["skills"], key=lambda s: s["usage_score"]) self.assertEqual(hottest["verdict"], "likely-retained") + def test_a_cheap_low_scored_row_is_granted_after_a_costly_higher_one_is_shed(self): + """First-fit, not a prefix: the product's grant loop has no early exit. + + A prefix model sheds a contiguous run of the lowest-scored rows. The real + loop walks every entry, so a short description can still be granted after + a longer, better-scored one was refused. Description LENGTH is a ranking + input, which no prose account of this mechanism mentions. + """ + # Per-entry demand is capped at max_desc_chars (1536), so the budget is + # exhausted with full-cap fillers rather than one giant description. + entries = [ + { + "qualified_name": f"a:fill{i}", + "frontmatter": {"description": "x" * 1536}, + "plugin_enabled": True, + } + for i in range(5) + ] + entries += [ + # Outscores the cheap row, but cannot fit in what the fillers left. + { + "qualified_name": "a:hog", + "frontmatter": {"description": "x" * 1536}, + "plugin_enabled": True, + }, + # Lowest score in the set, but cheap enough to survive the leftovers. + { + "qualified_name": "a:cheap", + "frontmatter": {"description": "x" * 300}, + "plugin_enabled": True, + }, + ] + scores = {f"a:fill{i}": 100.0 - i for i in range(5)} + scores.update({"a:hog": 50.0, "a:cheap": 1.0}) + listing = engine.compute_listing( + entries, engine.ListingConfig(context_window_tokens=200_000), scores + ) + by_name = {r["qualified_name"]: r for r in listing["skills"]} + # 5 x 1536 = 7680 granted, leaving 320 of the 8000 budget. + # a:hog needs 1536 and is shed; the walk does NOT stop there, and + # a:cheap needs only 300, so it is granted from the same leftovers. + self.assertEqual(by_name["a:hog"]["verdict"], "likely-starved") + self.assertEqual(by_name["a:cheap"]["verdict"], "likely-retained") + self.assertEqual(by_name["a:fill0"]["verdict"], "likely-retained") + # And the band still ranks by exposure, so the cheap survivor is band 1 + # even though it was retained. Band and verdict answer different + # questions and are allowed to disagree. + self.assertEqual(by_name["a:cheap"]["band"], 1) + class JoinerCharsTest(unittest.TestCase): """The listing inserts a literal ' - ' between description and when_to_use. diff --git a/verify-posttooluse-probe.md b/verify-posttooluse-probe.md new file mode 100644 index 0000000000..0c075533a2 --- /dev/null +++ b/verify-posttooluse-probe.md @@ -0,0 +1,4 @@ +# PostToolUse probe + +This line has teh misspelling and trailing whitespace. +Second line also has teh word spelled wrong. From 0c63eb8fc74157d1ee9fdb2765ca06a929a21d63 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 19:54:03 -0400 Subject: [PATCH 18/22] chore: untrack a worker scratch probe swept in by a broad add verify-posttooluse-probe.md is a live probe fixture belonging to the in-flight skill-usage-hook investigation, deliberately containing a misspelling and trailing whitespace to trip a PostToolUse hook. It has no business in history. The file stays on disk so the running worker is not disrupted. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- verify-posttooluse-probe.md | 4 ---- 1 file changed, 4 deletions(-) delete mode 100644 verify-posttooluse-probe.md diff --git a/verify-posttooluse-probe.md b/verify-posttooluse-probe.md deleted file mode 100644 index 0c075533a2..0000000000 --- a/verify-posttooluse-probe.md +++ /dev/null @@ -1,4 +0,0 @@ -# PostToolUse probe - -This line has teh misspelling and trailing whitespace. -Second line also has teh word spelled wrong. From c8b80ff2f76dcccc552c086600865c7abc97bff3 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 19:59:20 -0400 Subject: [PATCH 19/22] docs(claude-ops): name upstream as the source of the drop-order error, and date it The correction stays at full strength, but "stated wrongly" with no referent put the error on the ADR's author, who transcribed the vendor page near-verbatim. It also dropped the operationally useful fact: the upstream page is still wrong as of 2026-08-31, which is why the same sentence keeps re-entering this repo. It is on main in the skill's SKILL.md and in the plugin manifest, and #3524 added a citation pointing at the wrong page for it. The SKILL.md paragraph now separates the two upstream pages by whether they hold: the settings page owns the budget fraction and per-entry cap and matches; the skills page states the drop order and does not. Readers are routed to the binary for the ordering. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- ...commendation-from-the-catalog-not-the-listing.md | 12 ++++++------ .../skills/audit-skill-visibility/SKILL.md | 13 +++++++------ 2 files changed, 13 insertions(+), 12 deletions(-) diff --git a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md index 766a7c8984..1cad962ce5 100644 --- a/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md +++ b/docs/adr/0016-source-skill-recommendation-from-the-catalog-not-the-listing.md @@ -28,12 +28,12 @@ cannot tell that it is blind. The gatekeeping the contract bans would have been harness, invisibly. > **Revised 2026-08-31 ([#3534](https://github.com/melodic-software/claude-code-plugins/issues/3534)):** -> the drop-order mechanism above is wrong, and **the error is upstream's, not this ADR's.** The -> sentence tracks the official documentation, which states it in the same terms: "When the listing -> overflows, Claude Code drops descriptions starting with the skills you invoke least, so the skills -> you use most keep their full text" -> (, fetched 2026-08-31). The shipped binary does something -> else, on two independent axes. +> the drop-order sentence above restates ("Skill +> descriptions are cut short"), which is itself wrong and **still wrong as of 2026-08-31**. The error +> is upstream's, not this ADR's, which is why the same sentence keeps re-entering this repo: it is on +> `main` in the skill's own SKILL.md and in the plugin manifest, and #3524 added a citation pointing +> at the wrong page for it. Claude Code does not drop descriptions "starting with the skills invoked +> least". The shipped binary does something else, on two independent axes. > > It ranks by a decay-weighted score, `usageCount * max(0.5 ^ (daysSinceUse / 7), 0.1)`, so a > heavily used but stale skill can be shed before a lightly used fresh one: 100 uses 21 days ago diff --git a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md index 496811cd35..cbc16cba22 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/SKILL.md @@ -30,12 +30,13 @@ zero. Unused is partly self-causing. The budget fraction and per-entry cap are owned by (`skillListingBudgetFraction`, -`skillListingMaxDescChars`). **The drop ORDER is not documented accurately.** The -skills page says "starting with the skills you invoke least"; the binary ranks by -a decay-weighted score and then walks the list first-fit, so neither the ordering -nor the guarantee holds as written. Measured against the binary, with the -counterexamples and the stamp: -[reference/listing-scorer.md](reference/listing-scorer.md). +`skillListingMaxDescChars`); that page is authoritative and matches. **The drop +ORDER is not.** ("Skill descriptions are +cut short") says "starting with the skills you invoke least", still as of +2026-08-31; the binary ranks by a decay-weighted score and then walks the list +first-fit, so neither the ordering nor the guarantee holds. Take the ordering +from the binary: [reference/listing-scorer.md](reference/listing-scorer.md) +carries the counterexamples, the greps, and the stamp. So the useful question is not *which skills are unused*. Claude Code already reports that in `/doctor` and the Stats tab. It is **which skills are starved by From ef9d226e116c2166672b2c6ae13f41d53e01d90a Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 20:15:15 -0400 Subject: [PATCH 20/22] fix(claude-ops): charge the name floor and keep catalog order on ties Two more corrections from the adjudicator's re-extraction, both scoped to the grant loop. The grant budget is computed FORWARD from a floor, `budget - V`, never backward from the overflow. V is what the listing costs before any description is granted: every listed entry pays for its own name, the exempt classes pay their full rendering because they are never candidates, and the separators are charged too. A disable-model-invocation skill is absent from the listing entirely, so unlike the other two exempt classes it costs nothing and takes no separator. Without this the grant boundary itself is wrong, which left the first-fit fix in 916367b5b inert. Ties keep catalog order, not alphabetical. The product's sort is stable, so equal scores stay in input order, and Python's is too. This decides everything in the unscored case, where every score is zero and the tiebreak IS the whole ordering; an alphabetical one disagreed with the product on every row. demand/overflow is deliberately NOT changed. It answers "does the listing overflow", a different question from "which entries win", and only the second needs the floor. Charging name bytes there is real but not urgent: at this repo's 12.7x overflow it cannot flip the verdict, and it would move every test encoding the current arithmetic plus every figure recorded against it. It wants its own evidence, a fleet where the corrected count changes the verdict. Two test expectations move because the old ones encoded the no-floor model, and the corrected answers are what the product actually does. Live run: 171 of 176 competing now starved, up from 167, which is the floor being charged. Also softens listing-scorer.md's "a never-used skill is always shed first". It sorts last, so it sheds first under any material overflow, but first-fit means a zero-scored skill with a very short description can still be granted. Suite 94/94, ruff clean, markdownlint clean. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- .../reference/listing-scorer.md | 21 ++++++++-- .../scripts/audit_skill_visibility.py | 41 ++++++++++++++----- .../scripts/test_audit_skill_visibility.py | 28 +++++++++---- 3 files changed, 68 insertions(+), 22 deletions(-) diff --git a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md index e7a9e35e11..fc1fd12cb5 100644 --- a/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md +++ b/plugins/claude-ops/skills/audit-skill-visibility/reference/listing-scorer.md @@ -50,10 +50,23 @@ so recency competes with volume. A skill used 100 times sixty days ago scores scores 12. Any wording that says descriptions are shed "starting with the least-invoked skills" describes a mechanism the product does not have. -The floor matters at both ends. A never-used skill scores exactly zero and is -always shed first, which is the feedback loop this whole skill exists to expose. -A once-used skill never decays below `0.1 * usageCount`, so it never falls back -into the never-used band. +The floor matters at both ends. A never-used skill scores exactly zero and sorts +last, so it loses its description first under any material overflow, which is the +feedback loop this whole skill exists to expose. It is not *always* shed, though: +because the walk is first-fit, a zero-scored skill with a very short description +can still be granted from what the others left. A once-used skill never decays +below `0.1 * usageCount`, so it never falls back into the never-used band. + +Two more properties the ordering depends on: + +- **Ties keep catalog order, not alphabetical.** The product's sort is stable, so + equal scores stay in input order. This decides everything in the `unscored` + case, where every score is zero and the tiebreak IS the whole ordering. +- **The grant budget is computed forward from a floor**, `budget - V`, where `V` + is what the listing costs before any description is granted: every listed entry + pays for its own name, the exempt classes pay their full rendering, and the + separators are charged too. Deriving it by subtracting the overflow instead + gets the grant boundary wrong, not just the ordering. ## Why a bare usage key does not move the band diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index 2193cb59fe..93ab5c8cc7 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -816,10 +816,29 @@ def compute_listing( rows: list[dict] = [] demand = 0 + # The grant loop's budget is computed FORWARD from a floor, never backward + # from the overflow: the product starts at `budget - V`, where V is what the + # listing costs before any description is granted. Every entry that appears + # at all pays for its own name; the exempt classes pay their full rendering + # because they are never candidates. + floor = 0 + listed = 0 for entry in denominator: eligibility = _eligibility(entry) - chars = _demand_chars(entry, cfg) if eligibility == "competing" else 0 + desc_chars = _demand_chars(entry, cfg) + chars = desc_chars if eligibility == "competing" else 0 demand += chars + # A `disable-model-invocation` skill is absent from the listing + # ENTIRELY, so unlike the other two exempt classes it costs nothing and + # takes no separator. + if eligibility != "exempt-user-only": + listed += 1 + name_chars = len(entry["qualified_name"]) + if eligibility == "exempt-bundled": + # Keeps its description unconditionally, so it is charged for it. + floor += name_chars + 4 + desc_chars + else: + floor += name_chars + 2 rows.append( { "qualified_name": entry["qualified_name"], @@ -836,8 +855,9 @@ def compute_listing( # Basis is decided by whether any score survived, not by whether a scores # argument arrived. A caller can hand over a full map that happens to be all - # zeros, and the resulting order is alphabetical either way. + # zeros, and the resulting order is catalog order either way. score_basis = "native-counters" if any(r["usage_score"] for r in rows) else "unscored" + floor += max(0, listed - 1) competing = [r for r in rows if r["eligibility"] == "competing"] @@ -851,13 +871,14 @@ def compute_listing( # # Consequence worth stating plainly: description LENGTH is a ranking input, # which no prose description of this mechanism mentions. - competing.sort(key=lambda r: (-r["usage_score"], r["qualified_name"])) - # Budget accounting stays as it was: `demand` counts description bytes - # against the whole budget and ignores the name bytes every entry also pays. - # That understates pressure slightly and is the CERTAIN half of this report, - # so it is deliberately not changed here alongside the ordering fix. Charging - # names too is a separate correction with its own evidence. - remaining = budget + # + # Ties keep CATALOG ORDER, not alphabetical. The product's sort is stable, so + # equal scores stay in input order; Python's is too, which is why this sorts + # on the score alone. It matters most in the `unscored` case, where every + # score is 0 and the tiebreaker IS the whole ordering: an alphabetical one + # would disagree with the product on every row. + competing.sort(key=lambda r: -r["usage_score"]) + remaining = max(0, budget - floor) for row in competing: if overflow <= 0: row["verdict"] = "listing-fits" @@ -872,7 +893,7 @@ def compute_listing( # least. It can disagree with the verdict, and that disagreement is real # rather than a bug -- a band-1 row with a very short description can survive # a first-fit pass that sheds a better-scored row with a long one. - by_exposure = sorted(competing, key=lambda r: (r["usage_score"], r["qualified_name"])) + by_exposure = sorted(competing, key=lambda r: r["usage_score"]) for rank, row in enumerate(by_exposure, start=1): row["band"] = rank if overflow > 0 else None # An unscored ordering is alphabetical, so its band carries no signal at diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py index 562e21eef6..33b1ee3cbf 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py @@ -910,13 +910,25 @@ def test_one_char_overflow_starves_only_the_least_used_row(self): self.assertEqual(starved[0]["qualified_name"], "a:0") self.assertEqual(len(retained), 7) - def test_starved_set_covers_the_overflow_and_no_more(self): - # 10 x 1000 = 10_000 against 8_000 -> overflow 2_000 -> exactly 2 rows. + def test_the_name_floor_is_charged_before_any_description_is_granted(self): + """The grant budget starts at `budget - V`, not at the whole budget. + + Naive arithmetic says 10 x 1000 against 8_000 sheds exactly two rows. The + product charges every listed entry for its own name and the separators + first, so slightly less than the full budget is available to + descriptions, and a third row goes. Deriving the grant budget by + subtracting the overflow instead would miss that row entirely. + """ listing = self._listing(n=10, chars=1000, budget_tokens=200_000) + # The overflow figure itself is unchanged: it answers "does the listing + # overflow", which is a different question from "which entries win". self.assertEqual(listing["overflow_chars"], 2_000) starved = [s for s in listing["skills"] if s["verdict"] == "likely-starved"] - self.assertEqual(len(starved), 2) - self.assertEqual(sorted(s["qualified_name"] for s in starved), ["a:0", "a:1"]) + self.assertEqual(len(starved), 3) + # The three lowest-scored, since nothing here varies in length. + self.assertEqual( + sorted(s["qualified_name"] for s in starved), ["a:0", "a:1", "a:2"] + ) def test_most_used_skill_is_never_starved_while_others_can_absorb_it(self): listing = self._listing(n=10, chars=1000, budget_tokens=200_000) @@ -951,7 +963,7 @@ def test_a_cheap_low_scored_row_is_granted_after_a_costly_higher_one_is_shed(sel # Lowest score in the set, but cheap enough to survive the leftovers. { "qualified_name": "a:cheap", - "frontmatter": {"description": "x" * 300}, + "frontmatter": {"description": "x" * 200}, "plugin_enabled": True, }, ] @@ -961,9 +973,9 @@ def test_a_cheap_low_scored_row_is_granted_after_a_costly_higher_one_is_shed(sel entries, engine.ListingConfig(context_window_tokens=200_000), scores ) by_name = {r["qualified_name"]: r for r in listing["skills"]} - # 5 x 1536 = 7680 granted, leaving 320 of the 8000 budget. - # a:hog needs 1536 and is shed; the walk does NOT stop there, and - # a:cheap needs only 300, so it is granted from the same leftovers. + # The name floor takes 67, leaving 7933. 5 x 1536 = 7680 granted, leaving + # 253. a:hog needs 1536 and is shed; the walk does NOT stop there, and + # a:cheap needs only 200, so it is granted from the same leftovers. self.assertEqual(by_name["a:hog"]["verdict"], "likely-starved") self.assertEqual(by_name["a:cheap"]["verdict"], "likely-retained") self.assertEqual(by_name["a:fill0"]["verdict"], "likely-retained") From f7e51dcfe1e014f235328d77b7d3629f3b8f3813 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 20:47:40 -0400 Subject: [PATCH 21/22] docs: regenerate the catalog after the claude-ops description change docs/CATALOG.md is generated from the plugin manifests, and the drop-order wording fix in the claude-ops description left it stale. Regenerated with scripts/generate-catalog.mjs; the only diff is that one clause. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- docs/CATALOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 31749cd8be..b681769a54 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -81,7 +81,7 @@ plugin manifests and kept in sync by CI — never hand-edit it; the category voc - [`playbooks`](../plugins/playbooks) — Doctrine and knowledge playbooks as on-demand skills, plus a maintainer-facing update skill. boris — Boris Cherny's Claude Code workflow tips (howborisusesclaudecode.com); skill-authoring — Anthropic's internal skill-authoring playbook; fable-5 — Claude Fable 5's operating doctrine (self-authored, no upstream). The boris and skill-authoring packs vendor a verbatim upstream baseline; /playbooks:update drift-checks and syncs those baselines centrally (maintainers). - [`claude-config`](../plugins/claude-config) — Nine configuration-health skills (plus setup) for a repo's Claude Code configuration: audit (settings.json / .mcp.json / hooks / plugins / permissions drift), audit-automation-gaps (evidence-gated verdicts on automation gaps), audit-permission-grants (allow-rule / allowed-tools grants for auto-mode durability and portability), audit-permission-state (the permission rules actually in effect — every settings scope merged with per-rule provenance, what auto mode drops on entry, config written where nothing reads it, and which managed intents are enforced versus loosenable), draft-auto-mode-rules (interview and draft a paste-ready autoMode classifier block; prints only, never writes), audit-instructions (locally-owned instruction surfaces vs current model capability — proposes removals/rewrites of instructions the model no longer needs, and detects cross-surface instruction conflicts), audit-prompting-postures (the additive lane — posture guidance the prompting guide says a component's purpose needs but the component does not carry), audit-pass (one coordinated, ordered, resumable pass over a named target — three-scope inventory, run-time-derived exclusion set, stable finding identity, suppression memory, resume, one human gate — delegating every check to the plugin that owns it), and unhobble (the empirical bare-baseline experiment: reversibly strip a repo's standing instructions, log real stumbles against the current model, re-add only what evidence earns). - [`claude-memory`](../plugins/claude-memory) — Keeps a repo's Claude Code memory layer healthy and under your control, against criteria derived from official Claude Code documentation. The audit skill checks the instruction/memory layer (CLAUDE.md, CLAUDE.local.md, .claude/rules/, auto-memory) with a deterministic script-backed spine plus judgment-tier checks. The stateless skill inspects, disables, and (confirm-gated) purges Claude-written auto memory across all settings scopes. -- [`claude-ops`](../plugins/claude-ops) — Claude Code operations toolkit. Twelve skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used — a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which drops descriptions least-invoked-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface — every built-in CLI command with aliases and hidden/gated status, every bundled skill, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json — full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the silent unparsable-settings pause, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labelled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces — built-in CLI commands, bundled skills, plugin-backed built-ins, session-provided skills — against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry — OTEL store, collector, hook-event JSONL, ccusage — with trend reports and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and integrate them into the current repo), plugins (bring a machine's plugin fleet current on demand — marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view — queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action — an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry lives. Plus a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures — the last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that maps envelopes into the hook-events.jsonl the observability skill reads. +- [`claude-ops`](../plugins/claude-ops) — Claude Code operations toolkit. Twelve skills: audit-skill-visibility (audit whether each installed skill is actually VISIBLE to the model, and diagnose why most of a fleet never gets used — a skill is invisible when its description is dropped by Claude Code's skill-listing context budget, which sheds descriptions lowest-score-first so an unused skill loses the keywords that would let it be matched, from skills genuinely not wanted, from skills the run cannot observe at all; computes whether the listing overflows from documented settings, and withholds every cold verdict the data cannot support rather than reporting absence of data as absence of use), inventory (read-only enumeration of the complete invocable surface — every built-in CLI command with aliases and hidden/gated status, every bundled skill, and every component of every installed plugin across all marketplaces; reads the shipped binary because upstream publishes no built-in command list, and carries an integrity verdict so a drifted build reports counts as floors rather than silently short totals), audit-install-state (read-only audit of the machine-scope ~/.claude installation directory and ~/.claude.json — full inventory split into an authored surface and rolled-up bulk trees, product-managed retention vs genuinely unmanaged state, filename-scheme resolution before any process-liveness check, and deliberate/mid-experiment detection; reports, never deletes), audit-performance (read-only slowness-diagnostic capture run at the moment the machine or a session feels slow: CLI version, retention-sweep health including the silent unparsable-settings pause, a timed census walk of the install tree as a sweep-cost proxy, active-session and plugin-fleet counts, a process census, and the fan-out layer, which covers a load-labelled no-op spawn baseline, every hook that will fire bucketed per-tool-call versus per-turn with its invocation shape, the configured statusline, subagent concurrency and spawn-depth ceilings against documented defaults, whether running sessions predate the settings file they are judged by, and orphan attribution by parent liveness rather than age; read against a bundled known-performance-issues reference that also records the causes tested and cleared; separates the four documented suspects of accumulated state, version regression, component bloat, and per-spawn fan-out cost, and routes remediation out; reports, never mutates, and never executes a discovered hook or statusline command), audit-native-overlap (map native Claude Code surfaces — built-in CLI commands, bundled skills, plugin-backed built-ins, session-provided skills — against the current repo's plugin skills and agents, so a custom component never silently duplicates what Claude Code itself ships; bare invocation is a read-only overlap report carrying the extraction's integrity floors and a shared-listing-budget exposure section, verdicts are human-gated in a committed store rendered into a generated registry whose every row carries an observable recheck trigger, and only an explicit apply step bakes presence-gated native references into descriptions and Boundary sections), observability (read locally captured telemetry — OTEL store, collector, hook-event JSONL, ccusage — with trend reports and store pruning), known-issues (search known Claude product GitHub bugs, check service health, maintain a persistent tracked-issue registry), changelog (ingest Claude Code changelog entries and integrate them into the current repo), plugins (bring a machine's plugin fleet current on demand — marketplace refresh, effective-scope updates including in-repo project/local installs, new-plugin install per policy, scope-divergence detection and explicit convergence), morning-brief (read-only gh-based operator morning view — queue-label counts, merge-ready PRs, parked decisions with their RECOMMENDED lines, and loop-lane telemetry freshness), lanes (start/restart/stop/status loop lanes as named background Claude Code sessions seeded from canonical prompt files, with per-lane model/effort, a repo-pull + marketplace-refresh launch step, and a consume-restarts action — an OS-schedulable reader that relaunches stopped lanes whose telemetry carries a restart_request), and a re-runnable setup action that settles where the known-issues registry lives. Plus a family of eight advisory *-audit hooks (API errors, config changes, instruction loads, permission denials, pre-compaction, skill usage, tool failures, and unsurfaced hook failures — the last also warns the user via systemMessage, since a hook that fails to launch enforces nothing and Claude Code surfaces the failure to nobody) that emit the shared hook-telemetry envelope, and a reference sink that maps envelopes into the hook-events.jsonl the observability skill reads. - [`rate-limit-guard`](../plugins/rate-limit-guard) — Shared rate-limit guard for loop lanes: a statusline wrapper tees the subscription rate-limit windows to a fixed machine-scope file, a StopFailure hook records rate-limit stops reactively, and a reader contract fixes how consuming sessions pause and resume. - [`context-guard`](../plugins/context-guard) — Per-session context-window observability plus the first shipped consumer: a statusline wrapper tees each session's context_window fields to a per-session snapshot file, a zone resolver classifies usage into smart/acceptable/dumb bands (percentage bands plus window-class token bands, conservative-min combination, zones.json SSOT with shipped defaults), a reader contract fixes how consuming sessions interpret the snapshots, and zone-crossing hooks report once per transition into a worse zone across two channels — the continuation menu to the operator, who owns that choice, and to the model only the zone determination plus the counter-steer that a zone word is not a decay signal (advisory by default; an optional blocking mode gates new mutating work on a fresh dumb-zone snapshot with handoff-writing exempt), with a PostCompact hook persisting an evidence-degraded marker. - [`context-budget`](../plugins/context-budget) — Measure a Claude Code session's fixed startup context payload per item, on the consumer's machine at a pinned, version-stamped binary — including per-tool attribution of the built-in tool pools that /context reports only as lump sums, derived live by A/B bare-name-deny differencing with enforced comparability rules (skill-listing signature, one mode, one binary), an SDK-primary exact meter degrading to a version-aware headless /context parser and then to an honest structured error (never a wrong number), and a per-project measure-toggle-remeasure ledger under the plugin data directory recording every lever's real before/after delta. Report-only: prints exact config, applies nothing. From c96caec418bfd043f4963b8667f49dd88617f021 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Mon, 31 Aug 2026 21:31:10 -0400 Subject: [PATCH 22/22] fix(claude-ops): decide the score basis from the contenders, and refuse a bare key on duplicate rows Three defects the automated reviewers caught, two of them reintroducing at a different scope the exact failure this PR exists to fix. score_basis was decided over the whole denominator, before competing was filtered from it. A bundled, name-only, or disable-model-invocation skill can carry real native usage while being excluded from the contest entirely, so its score flipped the basis to native-counters while every actual contender sat at zero. Those rows then got confidence "inferential", which is a catalog ordering presenting itself as usage-informed. That is this report's own headline defect, one scope up. The basis is now decided from the competing rows only. The bare-key resolver collapsed owners into a set, so two marketplaces shipping the same plugin, which the report already marks ambiguous-attribution, produced one distinct qualified name. A bare key then passed the single-owner test and was reported on BOTH rows, inventing usage for an attribution the audit knows it cannot make. Owners are counted per entry now. The markdown renderer still said descriptions are dropped "least-invoked-first" and described the order as a likelihood band regardless of basis. It now says lowest-score-first, names decay-weighting and description length as inputs, and on an unscored run states plainly that the order carries no starvation information rather than hedging it as inferential. The 0.39.0 changelog entry also omitted the first-fit, floor and tiebreak changes entirely. Suite 96/96 with two regression tests, ruff clean, markdownlint clean. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015eyw6KUwExd78yyptowV6d --- plugins/claude-ops/CHANGELOG.md | 18 +++++- .../scripts/audit_skill_visibility.py | 64 ++++++++++++++----- .../scripts/test_audit_skill_visibility.py | 62 ++++++++++++++++++ 3 files changed, 126 insertions(+), 18 deletions(-) diff --git a/plugins/claude-ops/CHANGELOG.md b/plugins/claude-ops/CHANGELOG.md index 54e7cd9196..8bdd473afd 100644 --- a/plugins/claude-ops/CHANGELOG.md +++ b/plugins/claude-ops/CHANGELOG.md @@ -15,6 +15,16 @@ All notable changes to the `claude-ops` plugin are documented here. Format follo usage-informed. On this machine that ranked `adhd:clarify` (1 use) as first to lose its description and `work-items:triage` (99 uses) as among the safest. Scores are now computed before the listing is built. +- **Truncation is modelled as the greedy first-fit walk the product runs, not a + score-ordered prefix.** The product's grant loop has no early exit, so it walks + every competing entry with a running description budget and a cheap low-scored + description can be granted after an expensive higher-scored one was refused. + Description length is therefore a second ranking input. The budget itself is + computed forward from a floor, `budget - V`, where V is what the listing costs + before any description is granted; deriving it from the overflow put the + grant boundary in the wrong place. Ties keep catalog order rather than + alphabetical, matching the product's stable sort, which decides the entire + ordering when nothing is scored. - **Usage recorded under a skill's bare leaf no longer vanishes.** Events were looked up by qualified `:` name only, while the stores hold both that key and the bare leaf as separate rows, so the bare row was discarded with @@ -34,9 +44,13 @@ All notable changes to the `claude-ops` plugin are documented here. Format follo semantics and their seeding and throttle traps in the companion `reference/usage-counters.md`. - **`listing.score_basis`, so an unscored band admits it.** When no usage - survives to weigh, the order is alphabetical and nothing more; the basis reads + survives to weigh, the order is catalog order and nothing more; the basis reads `unscored` and competing rows carry `confidence: "unscored"` rather than - borrowing `inferential`, which claims more than the data supports. + borrowing `inferential`, which claims more than the data supports. The basis is + decided over the CONTENDERS, not the whole denominator: an exempt skill + carrying real usage must not label a contest whose every entrant is at zero. + The markdown report renders the distinction too, rather than describing an + unranked order as a likelihood band. ## [0.38.23] diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py index 93ab5c8cc7..35c89ee728 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/audit_skill_visibility.py @@ -166,11 +166,17 @@ def resolve_event_keys( ambiguous key is returned for the withheld section instead of being spent. """ qualified = {entry["qualified_name"] for entry in denominator} - leaf_owners: dict[str, set[str]] = defaultdict(set) + # Owners are counted per ENTRY, not per distinct qualified name. Two + # marketplaces shipping the same plugin produce two denominator rows with an + # identical qualified name, which the report already marks + # `ambiguous-attribution`. Collapsing them into a set would let a bare key + # pass the single-owner test and then be reported on BOTH rows, inventing + # usage for an attribution the audit already knows it cannot make. + leaf_owners: dict[str, list[str]] = defaultdict(list) for entry in denominator: name = entry["qualified_name"] leaf = name.split(":", 1)[1] if ":" in name else name - leaf_owners[leaf].add(name) + leaf_owners[leaf].append(name) by_skill: dict[str, list[dict]] = defaultdict(list) ambiguous: dict[str, list[str]] = {} @@ -181,11 +187,11 @@ def resolve_event_keys( if key in qualified: by_skill[key].append(event) continue - owners = leaf_owners.get(key, set()) + owners = leaf_owners.get(key, []) if len(owners) == 1: - by_skill[next(iter(owners))].append(event) + by_skill[owners[0]].append(event) elif len(owners) > 1: - ambiguous[key] = sorted(owners) + ambiguous[key] = sorted(set(owners)) return by_skill, sorted(ambiguous.items()) @@ -853,14 +859,21 @@ def compute_listing( overflow = max(0, demand - budget) verdict = "overflowing" if overflow > 0 else "listing-fits" - # Basis is decided by whether any score survived, not by whether a scores - # argument arrived. A caller can hand over a full map that happens to be all - # zeros, and the resulting order is catalog order either way. - score_basis = "native-counters" if any(r["usage_score"] for r in rows) else "unscored" floor += max(0, listed - 1) competing = [r for r in rows if r["eligibility"] == "competing"] + # Basis is decided by whether any score survived AMONG THE CONTENDERS, not + # by whether a scores argument arrived and not over the whole denominator. A + # bundled, name-only, or disable-model-invocation skill can carry real native + # usage while being excluded from the contest entirely; counting its score + # here would label a listing `native-counters` whose every actual contender + # is at zero, so a pure catalog ordering would be dressed as `inferential`. + # That is the exact defect this report exists to stop, one scope up. + score_basis = ( + "native-counters" if any(r["usage_score"] for r in competing) else "unscored" + ) + # WHICH rows are shed is a GREEDY FIRST-FIT walk, not a score-ordered # prefix. The product sorts descending by score and then walks EVERY # competing entry with a running description budget, granting whatever still @@ -1304,19 +1317,38 @@ def _render_markdown(model: dict) -> str: f"{listing['overflow_chars']:,} characters.** " f"{listing['competing_count']} skills compete for " f"{listing['budget_chars']:,} characters of description budget, " - f"and descriptions are dropped least-invoked-first — so roughly " + f"and descriptions are shed lowest-score-first, so roughly " f"**{listing['starved_count']}** of them are running name-only, " f"which is why the model stops matching requests to those.", "", f"The other {listing['competing_count'] - listing['starved_count']} " - f"competing skills keep their descriptions: only enough of the " - f"least-used tail is dropped to close the gap, not the whole set.", - "", - "*Which* particular skills lost theirs is inferential — the ordering", - "comes from an undocumented scorer. Treat the ranking as a likelihood", - "band, not a cutoff line.", + f"competing skills keep their descriptions. The score is " + f"decay-weighted, not a raw invocation count, and the walk grants " + f"whatever still fits rather than shedding a clean tail, so " + f"description length matters too.", "", ] + # An unscored run has no usage behind its ordering at all. Saying + # "inferential" there would repeat the exact defect this report + # exists to expose, one level up, so the two cases get different + # prose rather than a shared hedge. + if listing.get("score_basis") == "unscored": + lines += [ + "**No usage signal was available for any competing skill, so " + "*which* particular skills lost their descriptions is NOT " + "ranked here.** The order below is the catalog order and " + "carries no information about starvation likelihood. The " + "over-budget figure above is unaffected and still holds.", + "", + ] + else: + lines += [ + "*Which* particular skills lost theirs is inferential: the " + "ordering mirrors a scorer recovered from the shipped binary, " + "not a documented interface. Treat the ranking as a likelihood " + "band, not a cutoff line.", + "", + ] else: lines += [ f"Listing fits: {listing['demand_chars']:,} of " diff --git a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py index 33b1ee3cbf..3de07dd157 100755 --- a/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py +++ b/plugins/claude-ops/skills/audit-skill-visibility/scripts/test_audit_skill_visibility.py @@ -549,6 +549,47 @@ def test_classify_scores_the_band_from_native_counters(self): self.assertEqual(bands["a:9"], 10) +class ScoreBasisScopeTest(unittest.TestCase): + """The basis is decided by the contenders, not by the whole denominator.""" + + def test_usage_on_an_exempt_skill_does_not_score_the_contest(self): + """Regression: an exempt row's score used to flip the basis. + + A bundled, name-only, or user-only skill can carry real native usage + while being excluded from the contest entirely. Counting it labelled the + listing `native-counters` while every actual contender sat at zero, so a + pure catalog ordering got dressed as `inferential`. That is the defect + this whole report exists to expose, one scope up. + """ + entries = [ + { + "qualified_name": f"a:{i}", + "frontmatter": {"description": "x" * 1000}, + "plugin_enabled": True, + } + for i in range(10) + ] + entries.append( + { + "qualified_name": "a:manual", + "frontmatter": { + "description": "x" * 1000, + "disable_model_invocation": True, + }, + "plugin_enabled": True, + } + ) + listing = engine.compute_listing( + entries, + engine.ListingConfig(context_window_tokens=200_000), + # Only the exempt row has any usage at all. + {"a:manual": 500.0}, + ) + self.assertEqual(listing["score_basis"], "unscored") + competing = [s for s in listing["skills"] if s["eligibility"] == "competing"] + self.assertTrue(all(s["confidence"] == "unscored" for s in competing)) + + class BareUsageKeyTest(unittest.TestCase): """Usage recorded under a bare leaf must reach its qualified skill.""" @@ -588,6 +629,27 @@ def test_ambiguous_bare_key_is_withheld_not_guessed(self): self.assertIn("toolchain:check", withheld[0]["reason"]) self.assertIn("skill-quality:check", withheld[0]["reason"]) + def test_a_bare_key_is_withheld_when_the_leaf_has_duplicate_entries(self): + """Two marketplaces shipping one plugin give two rows, one name. + + Collapsing owners into a set let the bare key pass the single-owner test + and then be reported on BOTH rows, inventing usage for an attribution the + report already marks `ambiguous-attribution`. + """ + now = _utc(2026, 8, 31) + model = engine.classify( + denominator=[_skill("dup:check"), _skill("dup:check")], + events=[{"skill": "check", "ts": now, "source": "native", "count": 40}], + config=engine.Config(), + clock=now, + horizons={"native": now - timedelta(days=400)}, + ) + for row in model["skills"]: + self.assertEqual(row["observation"]["count"], 0) + withheld = [w for w in model["withheld"] if w["skill"] == "check"] + self.assertEqual(len(withheld), 1) + self.assertIn("dup:check", withheld[0]["reason"]) + def test_a_bare_key_does_not_score_the_band(self): """`zPe` has no bare-key fallback, so the mirror must not add one.""" now = _utc(2026, 8, 31)