From 5b2e4ce09272af44ea0395a2f51363a907601cac Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:53:59 -0400 Subject: [PATCH 01/22] fix(claude-ops): repair inventory extraction for Claude Code 2.1.284 The brace reader treated template-literal ${...} substitutions as text, so quotes inside a regex in a substitution desynchronized it and one brace pair swallowed 21 MB of the bundle; 15 of 152 commands resolved. Substitutions are now tokenized as code. Also resolves registerSlidesSkill, literal-table skill rosters, and constant-named commands; tightens registrar and registration-token matching instead of widening thresholds. Co-Authored-By: Claude Opus 5.5 --- .../skills/inventory/reference/extraction.md | 36 +- .../skills/inventory/scripts/inventory.py | 361 ++++++++++++++---- .../inventory/scripts/test_inventory.py | 100 ++++- 3 files changed, 414 insertions(+), 83 deletions(-) diff --git a/plugins/claude-ops/skills/inventory/reference/extraction.md b/plugins/claude-ops/skills/inventory/reference/extraction.md index 94c6cd88a4..ed4f54943f 100644 --- a/plugins/claude-ops/skills/inventory/reference/extraction.md +++ b/plugins/claude-ops/skills/inventory/reference/extraction.md @@ -32,6 +32,7 @@ appended. Two layouts have been observed: |---|---|---| | Single run | 2.1.228 (PE32+, Windows) | one ~25 MB printable run under a `// @bun @bytecode @bun-cjs` header | | Fragmented bytecode | 2.1.263 (ELF, Linux) | ~3,200 printable runs from 256 bytes to several MB, scattered through ~214 MB after the first `// @bun` marker | +| Fragmented bytecode | 2.1.284 (ELF, Linux) | ~4,200 runs joining to ~45 MB, ~243 MB file; registrar `ps` by ESM export | In the fragmented layout the registrations sit in small runs (the `doctor` registration in about 1.3 KB), the hoisted name constants sit megabytes ahead of the calls that use them, and the export @@ -51,8 +52,10 @@ command lane looked healthy. The rule now, stated exactly: from the first `BUNDLE_MARKERS` occurrence to end of file, every printable run of at least `MIN_RUN_BYTES` (256) bytes, found with one `RUN_RE` pass, joined with newlines. `sources.binary` records `runs`, `joined_bytes`, `region_rule`, `runs_below_floor`, and -`elapsed_seconds`. `runs_below_floor` counts registration tokens (`({name:`) that sit in runs -shorter than the floor: a registration below the floor is counted, never silently lost. The floor +`elapsed_seconds`. `runs_below_floor` counts registration tokens (`({name:` followed directly by a +quote, backtick, or identifier) that sit in runs shorter than the floor: a registration below the +floor is counted, never silently lost. `({name: "` with a space is message prose from the bytecode +string table (2.1.284 carries two) and is not counted. The floor is what keeps the pass cheap; measured on 2.1.263, 256 bytes recovers every named surface in about 4 seconds, while a 64 KiB floor recovers 13 of 33 names and a 1-byte floor takes a minute. @@ -91,6 +94,25 @@ everywhere, so it is trusted only when that nearest binding lies within `eo({name:r,...})`) resolves, while a loop variable whose only binding is megabytes away does not. No preceding binding is unresolved, never guessed. +Two further shapes, both first seen in 2.1.284: + +- **Descriptor member.** `let t=c;ps({name:t.name,description:t.description,...})` with + `c={name:o,...}` and `o="slides"` (`registerSlidesSkill`). The identifier is followed through + `a=b` aliases to its object literal by the same nearest-preceding rule; that object's `name` and + description stand in for the registration's member reads. +- **Loop over a literal table.** ``for(let{kind:e,description:n}of Qi)ps({name:`artifact-${e}`,...})`` + with `Qi=[{kind:"report",...},...]`. When the table binding is an array of object literals, each + row is expanded into its own skill and `bundled_skill_notes.rosters_resolved` records the table + and row count. A loop over anything else stays a `dynamic_roster`. + +`resolved` counts registration calls, not rows, so an expanded roster never masks an unresolved +call. A registrar call must stand alone: `productionRemoteToolsAnnounceDeps({` ends in `ps` and is +not a call to `ps`. + +Built-in commands use the same constant rule for `name:EMr` (`commit-push-pr`, `limit-reset`, +`low-priority`, `claim-credit` in 2.1.284). A single-character `name:e` in a command literal is a +factory parameter and is never resolved. + Bundle markers are not module boundaries in the fragmented layout: the real bindings sit about 170 marker occurrences ahead of their registrations, so a marker-scoped rule finds nothing. Locality is measured in bytes. @@ -110,6 +132,13 @@ command's and each registration's fields are then read from its own literal. A r this bundle goes wrong in one of two ways: it misses `/artifacts` entirely, or it invents `/alias` and `/todos` as commands. +A template substitution `${...}` is code, not text: it can hold regex literals, nested templates, +and object literals, so the main tokenizer walks it and the `}` that returns to the substitution's +depth resumes the template text. In 2.1.284 `` `prints ${to(fn.replace(/^(["'])(.*)\1$/,"$2"))}` `` +put quote characters inside a regex inside a substitution; skipping the substitution as quoted +text desynced the reader, swallowed 21 MB into one brace pair, and left 15 of 152 command literals +resolved with every canary missing. + A call to the registrar identifier whose object carries no `name:` is another module's function sharing the minified name, not a registration; it is counted in `same_identifier_calls_skipped` and never inflates the resolved-versus-seen gap. @@ -179,7 +208,8 @@ check failed, and each maps to one edit: | broken: joined region under 1 MB | Packer layout changed | Add the new marker to `BUNDLE_MARKERS`, or lower the floor after measuring it | | `bundled_skills` degraded: unrecognized registrar export | A new registration path may exist | Inspect it; add to `KNOWN_REGISTRAR_EXPORTS` if it funnels into the known registrar, otherwise extract it | | `bundled_skills` degraded: computed names unresolved | A binding shape the locality rule does not see | Report as a floor; extend `_KEBAB_BINDING_RE` only if the count grows | -| `bundled_skills` degraded: dynamic roster | A family registered in a loop or template | Acceptable; the names are enumerable only by running the binary | +| `bundled_skills` degraded: dynamic roster | A family registered in a loop or template over something other than a literal table | Acceptable; the names are enumerable only by running the binary | +| `builtin_commands` broken: yield collapses and one brace pair spans megabytes | The tokenizer desynced on a new syntax shape | Find the largest pairs in `build_brace_map`, read the text at the open brace, fix the tokenizer state that misread it | After revalidating, bump `VALIDATED_AGAINST`. Leaving it stale is not a bug: every report then says its counts are believed rather than verified, which is the honest state until someone checks. diff --git a/plugins/claude-ops/skills/inventory/scripts/inventory.py b/plugins/claude-ops/skills/inventory/scripts/inventory.py index 613b2999f7..1e48047022 100755 --- a/plugins/claude-ops/skills/inventory/scripts/inventory.py +++ b/plugins/claude-ops/skills/inventory/scripts/inventory.py @@ -47,7 +47,7 @@ # the skill's evals. Drift from it is not an error - the extraction is designed # to survive ordinary releases - but it downgrades every count from "verified" # to "believed", which the report has to say out loud. -VALIDATED_AGAINST = "2.1.263" +VALIDATED_AGAINST = "2.1.284" # Commands that have shipped in every build observed. Their absence means the # extraction broke, not that Anthropic deleted /help. This is the cheapest @@ -71,6 +71,7 @@ "registerScheduleRemoteAgentsSkill", "registerAgentProxyEnvFn", "registerDesignCanvasSkill", + "registerSlidesSkill", "registerWorkflowAuthoringSkill", } ) @@ -96,10 +97,12 @@ # below it the regex costs minutes and above it registrations go missing. MIN_RUN_BYTES = 256 RUN_RE = re.compile(rb"[\t\n\r\x20-\x7e]{%d,}" % MIN_RUN_BYTES) -# Every registration literal, of any registrar, opens with this token. Counting +# Every registration literal, of any registrar, opens with `({name:`. Counting # it in the raw region against the joined source is how a registration that -# sits in a run below the floor is counted rather than silently lost. -REGISTRATION_TOKEN = b"({name:" +# sits in a run below the floor is counted rather than silently lost. Minified +# source puts the value flush against the colon; `({name: "` with a space is +# prose in a message string (the bytecode string table holds several), not code. +REGISTRATION_TOKEN_RE = re.compile(rb"\(\{name:(?=[\"`A-Za-z_$])") # A single-character identifier is a function-local minifier name reused # everywhere, so its nearest preceding string binding is only trusted when it @@ -279,9 +282,9 @@ def _select_region( ) meta["runs"] = len(runs) meta["joined_bytes"] = len(joined) - meta["runs_below_floor"] = max( - 0, data.count(REGISTRATION_TOKEN, first) - joined.count(REGISTRATION_TOKEN) - ) + in_region = sum(1 for _ in REGISTRATION_TOKEN_RE.finditer(data, first)) + in_joined = sum(1 for _ in REGISTRATION_TOKEN_RE.finditer(joined)) + meta["runs_below_floor"] = max(0, in_region - in_joined) meta["elapsed_seconds"] = round(time.perf_counter() - started, 3) if len(joined) < 1_000_000: meta["error"] = ( @@ -413,6 +416,11 @@ def build_brace_map(s: str) -> BraceMap: """ stack: list[int] = [] pairs: dict[int, int] = {} + # Brace depth at which each open template `${` substitution began. The + # substitution is ordinary code (it can hold regex literals, nested + # templates, and object literals), so the main loop tokenizes it; the `}` + # that returns to that depth resumes the template's literal text. + templates: list[int] = [] i, n = 0, len(s) prev_sig = "\n" prev_word = "" @@ -444,8 +452,21 @@ def build_brace_map(s: str) -> BraceMap: prev_sig, prev_word = q, "" continue if c == "`": - i = _skip_template(s, i, n) - prev_sig, prev_word = "`", "" + i, opened = _scan_template_text(s, i + 1, n) + if opened: + templates.append(len(stack)) + prev_sig, prev_word = "{", "" + else: + prev_sig, prev_word = "`", "" + continue + if c == "}" and templates and templates[-1] == len(stack): + templates.pop() + i, opened = _scan_template_text(s, i + 1, n) + if opened: + templates.append(len(stack)) + prev_sig, prev_word = "{", "" + else: + prev_sig, prev_word = "`", "" continue if c == "/": if prev_word in _REGEX_KEYWORDS or prev_sig in _REGEX_PRECEDERS: @@ -473,38 +494,22 @@ def build_brace_map(s: str) -> BraceMap: return BraceMap(pairs=pairs, opens=sorted(pairs)) -def _skip_template(s: str, i: int, n: int) -> int: - i += 1 +def _scan_template_text(s: str, i: int, n: int) -> tuple[int, bool]: + """Skip a template literal's text from `i`; True when it stopped at a `${`. + + Returns the index just past the closing backtick, or just past the `${` + that opens a substitution the caller must tokenize as code. + """ while i < n: if s[i] == "\\": i += 2 continue if s[i] == "`": - return i + 1 + return i + 1, False if s[i] == "$" and i + 1 < n and s[i + 1] == "{": - depth = 1 - i += 2 - while i < n and depth: - if s[i] in "\"'`": - q = s[i] - i += 1 - while i < n: - if s[i] == "\\": - i += 2 - continue - if s[i] == q: - i += 1 - break - i += 1 - continue - if s[i] == "{": - depth += 1 - elif s[i] == "}": - depth -= 1 - i += 1 - continue + return i + 2, True i += 1 - return i + return i, False def _skip_regex(s: str, i: int, n: int) -> int: @@ -545,6 +550,7 @@ def _unescape(raw: str) -> str: _TYPE_RE = re.compile(r'type:"(local|local-jsx|prompt)"') _NAME_RE = re.compile(r"(?:^|[,{])name:" + _STR) +_NAME_IDENT_RE = re.compile(r"(?:^\{|,)name:([A-Za-z_$][A-Za-z0-9_$]*)(?=[,}])") _UFN_RE = re.compile(r"userFacingName\(\)\{return" + _STR) _DESC_RE = re.compile(r"(?:^|[,{])description:" + _STR) _MENUDESC_RE = re.compile(r"(?:menuDescription|description):" + _STR) @@ -568,7 +574,7 @@ def extract_builtin_commands(src: str, braces: BraceMap) -> dict[str, dict[str, enclosing object is resolved by brace depth, then its own fields are read, so an adjacent command's description cannot bleed in. """ - out: dict[str, dict[str, Any]] = {} + literals: list[tuple[re.Match[str], int, str]] = [] for m in _TYPE_RE.finditer(src): enc = braces.enclosing(m.start()) if not enc: @@ -577,12 +583,33 @@ def extract_builtin_commands(src: str, braces: BraceMap) -> dict[str, dict[str, if close_i - open_i > 4000: # Too large to be a command literal - this is some enclosing scope. continue - body = src[open_i : close_i + 1] + literals.append((m, open_i, src[open_i : close_i + 1])) + + # A name held in a hoisted constant (`name:EMr` where `EMr="commit-push-pr"`) + # resolves by the same nearest-preceding rule as a bundled skill. A + # single-character identifier here is a factory's parameter + # (`function t1t(e,...){return{type:...,name:e}}`), not one command, so it + # is never resolved. + idents = { + c.group(1) + for _, _, body in literals + if not _NAME_RE.search(body) + for c in [_NAME_IDENT_RE.search(body)] + if c and len(c.group(1)) > 1 + } + index = build_const_index(src, idents) + + out: dict[str, dict[str, Any]] = {} + for m, open_i, body in literals: nm = _NAME_RE.search(body) or _UFN_RE.search(body) - if not nm: - continue - name = _unescape(nm.group(1)) - if not _NAME_OK.fullmatch(name or ""): + if nm: + name = _unescape(nm.group(1)) + else: + c = _NAME_IDENT_RE.search(body) + if not c or c.group(1) not in idents: + continue + name = resolve_name_ident(c.group(1), open_i, index) + if not name or not _NAME_OK.fullmatch(name): continue desc = _DESC_RE.search(body) aliases = _read_aliases(body) @@ -715,6 +742,117 @@ def resolve_name_ident( return value +def _nearest_binding(src: str, ident: str, at: int) -> int | None: + """Offset of the value in the nearest `ident=` binding before `at`. + + The same locality rule as `resolve_name_ident`: nearest preceding wins, and + a single-character identifier is only looked for within + `SHORT_IDENT_LOCALITY_BYTES`. + """ + lo = max(0, at - SHORT_IDENT_LOCALITY_BYTES) if len(ident) == 1 else 0 + pattern = re.compile(r"(?])\s*") + last = None + for m in pattern.finditer(src, lo, at): + last = m + return last.end() if last else None + + +def _resolve_object( + src: str, braces: BraceMap, ident: str, at: int, hops: int = 4 +) -> tuple[int, str] | None: + """The object literal an identifier is bound to, following `a=b` aliases. + + Returns the literal's open-brace offset and its text, or None when the + nearest binding is anything else. + """ + for _ in range(hops): + v = _nearest_binding(src, ident, at) + if v is None: + return None + if src.startswith("{", v): + close = braces.pairs.get(v) + return None if close is None else (v, src[v : close + 1]) + alias = re.match(_IDENT + r"(?=[;,)\s])", src[v : v + 64]) + if not alias: + return None + ident, at = alias.group(0), v + return None + + +def _resolve_descriptor_name( + src: str, braces: BraceMap, ident: str, at: int +) -> tuple[str, str] | None: + """Resolve `name:x.name`: the registration reads its fields from a descriptor. + + `let t=c;registrar({name:t.name,description:t.description,...})` where + `c={name:o,...}` and `o="slides"`. Returns the name and the descriptor's + text, whose fields stand in for the registration's member reads. + """ + obj = _resolve_object(src, braces, ident, at) + if obj is None: + return None + open_i, body = obj + m = re.search(r"(?:^\{|,)name:(?:" + _STR + r"|(" + _IDENT + r")\b)", body) + if not m: + return None + if m.group(1) is not None: + return _unescape(m.group(1)), body + v = _nearest_binding(src, m.group(2), open_i) + lit = re.match(_STR, src[v : v + 256]) if v is not None else None + return (_unescape(lit.group(1)), body) if lit else None + + +_ROSTER_HEAD_RE = re.compile( + r"for\(\s*(?:let|const|var)\s*\{(?P[^{}]*)\}\s*of\s*(?P" + + _IDENT + + r")\s*\)\s*\{?\s*$" +) + + +def _resolve_roster( + src: str, braces: BraceMap, call_start: int +) -> tuple[str, dict[str, str], list[str]] | None: + """A registration looping over a literal table: the table and its rows. + + `for(let{kind:e,description:n}of Qi)registrar({name:\\`artifact-${e}\\`,...})` + where `Qi=[{kind:"table",description:"..."},...]` is enumerable without + running the binary. Returns the table identifier, the destructuring map + (local variable to row key), and each row's text. None when the loop, the + table binding, or any row is not a plain literal. + """ + pre = src[max(0, call_start - 300) : call_start] + head = _ROSTER_HEAD_RE.search(pre) + if not head: + return None + var_to_key: dict[str, str] = {} + for part in head.group("pattern").split(","): + key, _, var = part.strip().partition(":") + if not re.fullmatch(_IDENT, key) or (var and not re.fullmatch(_IDENT, var)): + return None + var_to_key[var or key] = key + table = head.group("table") + v = _nearest_binding(src, table, call_start - len(pre) + head.start()) + if v is None or not src.startswith("[", v): + return None + rows: list[str] = [] + i = v + 1 + while True: + while i < len(src) and src[i] in " \t\r\n,": + i += 1 + if src.startswith("]", i): + return table, var_to_key, rows + close = braces.pairs.get(i) if src.startswith("{", i) else None + if close is None: + return None + rows.append(src[i : close + 1]) + i = close + 1 + + +def _row_field(row: str, key: str) -> str | None: + m = re.search(r"(?:^\{|,)" + re.escape(key) + ":" + _STR, row) + return _unescape(m.group(1)) if m else None + + _FOR_HEAD_RE = re.compile(r"for\((?P[^()]*)\)\s*\{?\s*$") _INVOCATION_FIELDS: tuple[tuple[str, str], ...] = ( ("user_invocable", "userInvocable"), @@ -797,8 +935,12 @@ def extract_bundled_skills( calls: list[tuple[int, str, re.Match[str]]] = [] unbounded = 0 same_ident_calls = 0 - name_re = re.compile(r"\bname:(?:" + _STR + r"|(" + _IDENT + r")|(`[^`]*`))") - for m in re.finditer(re.escape(fn) + r"\(\{", src): + name_re = re.compile( + r"\bname:(?:" + _STR + r"|(" + _IDENT + r")(\.name\b)?|(`[^`]*`))" + ) + # A call is the registrar identifier standing alone: `xps({` is another + # function whose name merely ends in the registrar's. + for m in re.finditer(r"(? None: + name = rec["name"] prev = out.get(name) if prev is None: out[name] = rec - continue + return existing = registrations_of(prev) if any(_same_registration(e, rec) for e in existing): - continue + return # A genuine collision: keep every registration, keyed by name. for e in existing: e["collision"] = True @@ -868,8 +983,51 @@ def extract_bundled_skills( if name not in collisions: collisions.append(name) + for call_start, body, nm in calls: + seen += 1 + descriptor = "" + if nm.group(1) is not None: + name = _unescape(nm.group(1)) + elif nm.group(4) is not None or _is_loop_registration( + src, call_start, nm.group(2) + ): + # A template literal or a loop variable: one call registering a + # family, enumerable statically only when the loop walks a + # literal table. + roster = _resolve_roster(src, braces, call_start) + rows = _roster_records(body, nm, roster) if roster else None + if roster is None or rows is None: + dynamic_rosters += 1 + dynamic_patterns.append( + nm.group(4) or f"for(... of ...) over {nm.group(2)}" + ) + continue + resolved_calls += 1 + rosters[roster[0]] = len(rows) + for rec in rows: + add(rec) + continue + elif nm.group(3) is not None: + found = _resolve_descriptor_name(src, braces, nm.group(2), call_start) + if found is None: + unresolved.append(f"{nm.group(2)}.name") + continue + name, descriptor = found + else: + resolved = resolve_name_ident(nm.group(2), call_start, index) + if resolved is None: + unresolved.append(nm.group(2)) + continue + name = resolved + resolved_calls += 1 + add(_skill_record(name, body, descriptor)) + + # Counted per call, not per row: a roster call is one registration however + # many rows it expands to, so it can never mask an unresolved one. notes["registrations_seen"] = seen - notes["resolved"] = sum(len(registrations_of(e)) for e in out.values()) + notes["resolved"] = resolved_calls + if rosters: + notes["rosters_resolved"] = rosters if unresolved: notes["unresolved_dynamic_names"] = sorted(set(unresolved)) if dynamic_rosters: @@ -884,6 +1042,55 @@ def extract_bundled_skills( return out, notes +def _skill_record(name: str, body: str, descriptor: str = "") -> dict[str, Any]: + """One bundled-skill row from its registration literal. + + `descriptor` is the object a `name:x.name` registration reads its fields + from; a description the literal does not carry as a string is read there. + """ + desc = _MENUDESC_RE.search(body) or _MENUDESC_RE.search(descriptor) + rec: dict[str, Any] = { + "name": name, + "source": "bundled-skill", + "description": _unescape(desc.group(1)) if desc else "", + "aliases": _read_aliases(body), + "gated": "isEnabled" in body, + "hidden": "isHidden" in body, + } + rec.update(read_invocation_fields(body)) + return rec + + +def _roster_records( + body: str, + nm: re.Match[str], + roster: tuple[str, dict[str, str], list[str]], +) -> list[dict[str, Any]] | None: + """Expand a looped registration over its table's rows; None if any row fails.""" + _, var_to_key, rows = roster + template = nm.group(4) + field = re.search(r"(?:menuDescription|description):(" + _IDENT + ")", body) + out: list[dict[str, Any]] = [] + for row in rows: + values = {var: _row_field(row, key) for var, key in var_to_key.items()} + if template: + parts = re.split(r"\$\{(" + _IDENT + r")\}", template[1:-1]) + if any(values.get(v) is None for v in parts[1::2]): + return None + name = "".join( + p if k % 2 == 0 else str(values[p]) for k, p in enumerate(parts) + ) + else: + name = values.get(nm.group(2)) or "" + if not _NAME_OK.fullmatch(name): + return None + rec = _skill_record(name, body) + if not rec["description"] and field: + rec["description"] = values.get(field.group(1)) or "" + out.append(rec) + return out + + def _same_registration(a: dict[str, Any], b: dict[str, Any]) -> bool: # `flag_driven` is part of the identity: a constant-true invocation field # and a function-valued one read as the same boolean, and the difference diff --git a/plugins/claude-ops/skills/inventory/scripts/test_inventory.py b/plugins/claude-ops/skills/inventory/scripts/test_inventory.py index b310534ece..6077b58d10 100755 --- a/plugins/claude-ops/skills/inventory/scripts/test_inventory.py +++ b/plugins/claude-ops/skills/inventory/scripts/test_inventory.py @@ -37,6 +37,24 @@ def test_ignores_braces_inside_template(self) -> None: bm = inv.build_brace_map("x={a:`v${b}`}") self.assertEqual(len(bm.pairs), 1) + def test_regex_inside_a_template_substitution_does_not_desync(self) -> None: + # 2.1.284: `\`prints ${to(fn.replace(/^(["'])(.*)\1$/,"$2"))}\`` - a + # substitution holding a regex with quote characters. Skipping the + # substitution as quoted text swallowed 21 MB into one brace pair and + # left /clear, /help, and most commands unresolved. + src = ( + 'if(a){x=`p ${f(/^(["\'])(.*)\\1$/,"$2")}`}' + 'var c={type:"local",name:"clear"};' + ) + bm = inv.build_brace_map(src) + enc = bm.enclosing(src.index('name:"clear"')) + assert enc is not None + self.assertEqual(src[enc[0] : enc[1] + 1], '{type:"local",name:"clear"}') + + def test_object_literal_inside_a_template_substitution_is_paired(self) -> None: + bm = inv.build_brace_map("x=`a ${f({k:`b ${c}`})} d`;y={e:1}") + self.assertEqual(len(bm.pairs), 2) + def test_division_is_not_a_regex(self) -> None: # `a/b` followed by `{` must not swallow the object as a regex body. bm = inv.build_brace_map("y=a/b;x={c:1}") @@ -104,6 +122,22 @@ def test_userfacingname_is_a_fallback(self) -> None: ) self.assertIn("autofix-pr", self._extract(src)) + def test_hoisted_constant_name_resolves(self) -> None: + src = ( + 'var EMr="commit-push-pr";' + 'var pGo={type:"prompt",name:EMr,description:"Commit, push, and open a PR"};' + ) + self.assertEqual( + self._extract(src)["commit-push-pr"]["description"], + "Commit, push, and open a PR", + ) + + def test_factory_parameter_name_is_not_resolved(self) -> None: + # `name:e` in a factory is whatever the caller passes, never the + # nearest `e="..."` in scope. + src = 'var e="wrong";function t1t(e,n){return{type:"local-jsx",name:e,description:n}}' + self.assertEqual(self._extract(src), {}) + def test_internal_names_are_marked(self) -> None: src = 'x={type:"prompt",name:"mcp__",description:"d"};' self.assertTrue(self._extract(src)["mcp__"]["internal"]) @@ -410,6 +444,21 @@ def test_read_bundle_counts_a_registration_below_the_floor(self) -> None: self.assertNotIn('name:"tiny"', src) self.assertEqual(meta["runs_below_floor"], 1) + def test_read_bundle_does_not_count_prose_in_the_string_table(self) -> None: + # The bytecode string table holds message text such as + # `or Workflow({name: "` - a space after the colon, so not code. + layout = ( + self.MARKER + + self.BIG + + self.GAP + + b' or Workflow({name: "' + + self.GAP + + self.DOCTOR + + b"b" * 300 + ) + _, meta = inv.read_bundle(self._write(layout)) + self.assertEqual(meta["runs_below_floor"], 0) + def test_read_bundle_rejects_a_region_under_one_megabyte(self) -> None: src, meta = inv.read_bundle(self._write(self.MARKER + self.DOCTOR + b"b" * 300)) self.assertIsNone(src) @@ -485,7 +534,7 @@ def test_a_lone_far_binding_of_a_short_identifier_is_not_resolved(self) -> None: self.HEAD + 'var e="linux";' + "z" * (inv.SHORT_IDENT_LOCALITY_BYTES + 10) - + 'eo({name:e,menuDescription:"D"});' + + ';eo({name:e,menuDescription:"D"});' ) skills, notes = self._skills(src) self.assertEqual(skills, {}) @@ -498,7 +547,7 @@ def test_a_short_identifier_bound_nearby_resolves(self) -> None: self.HEAD + 'var r="design";' + "z" * 4000 - + 'eo({name:r,menuDescription:"Draft"});' + + ';eo({name:r,menuDescription:"Draft"});' ) skills, _ = self._skills(src) self.assertIn("design", skills) @@ -508,7 +557,7 @@ def test_a_long_identifier_bound_far_ahead_resolves(self) -> None: self.HEAD + 'var kYe="simplify";' + "z" * (inv.SHORT_IDENT_LOCALITY_BYTES * 2) - + 'eo({name:kYe,menuDescription:"Clean up"});' + + ';eo({name:kYe,menuDescription:"Clean up"});' ) skills, _ = self._skills(src) self.assertIn("simplify", skills) @@ -540,6 +589,51 @@ def test_a_same_identifier_call_without_a_name_is_not_a_registration(self) -> No self.assertEqual(notes["registrations_seen"], 1) self.assertEqual(notes["same_identifier_calls_skipped"], 1) + def test_a_descriptor_member_name_resolves(self) -> None: + # registerSlidesSkill in 2.1.284: the registration reads its fields + # from a descriptor object whose name is a hoisted constant. + src = ( + self.HEAD + 'var o="slides";var c={name:o,intent:"slides",' + 'description:"Make a new Slides deck artifact from a brief"};' + "function Wvn(){let t=c;eo({name:t.name,description:t.description," + "isEnabled:()=>l(t),userInvocable:!0,disableModelInvocation:!0})}" + ) + skills, notes = self._skills(src) + self.assertEqual( + skills["slides"]["description"], + "Make a new Slides deck artifact from a brief", + ) + self.assertTrue(skills["slides"]["disable_model_invocation"]) + self.assertNotIn("unresolved_dynamic_names", notes) + + def test_a_loop_over_a_literal_table_is_enumerated(self) -> None: + src = ( + self.HEAD + + 'var Qi=[{kind:"report",description:"R"},{kind:"explainer",description:"E"}];' + "function Zt(){for(let{kind:e,description:n}of Qi)" + "eo({name:`artifact-${e}`,description:n,userInvocable:!0})}" + 'var za=[{kind:"doc",description:"D"}];' + "function Cn(){for(let{kind:e,description:s}of za)" + "eo({name:e,description:s,userInvocable:!0})}" + ) + skills, notes = self._skills(src) + self.assertEqual( + sorted(skills), ["artifact-explainer", "artifact-report", "doc"] + ) + self.assertEqual(skills["artifact-report"]["description"], "R") + self.assertEqual(skills["doc"]["description"], "D") + self.assertEqual(notes["rosters_resolved"], {"Qi": 2, "za": 1}) + self.assertNotIn("dynamic_roster", notes) + self.assertEqual(notes["registrations_seen"], notes["resolved"]) + + def test_a_function_whose_name_ends_in_the_registrar_is_not_a_call(self) -> None: + # `productionRemoteToolsAnnounceDeps({...name:be.name...})` ends in + # the 2.1.284 registrar `ps`; it is not a registration. + src = self.HEAD + 'xeo({bridge:()=>({name:be.name})});eo({name:"run"});' + skills, notes = self._skills(src) + self.assertEqual(list(skills), ["run"]) + self.assertEqual(notes["registrations_seen"], 1) + def test_a_loop_registration_is_a_dynamic_roster(self) -> None: src = ( self.HEAD From 5f15d4782aa3046548ed31f4a989703cd2ef00e9 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:04:14 -0400 Subject: [PATCH 02/22] feat(claude-ops): discover native overlap candidates dynamically detect now scores every native surface (builtin commands, bundled skills, plugin-backed built-ins, and bundled workflows when the inventory carries that lane) against every repo skill and agent from name and description tokens, and emits pairs over a threshold, top-k per surface, as origin "discovered" beside the seeded pairs. Pairs already in the store are listed as existing with their verdict; seeds absorb their discovered twin. Each candidate carries invocable_by from model_invocable/user_invocable (older inventories degrade to unknown) and a recommended_integration label; model_invocable false sets the model-invocation-disabled marker the store's suggest-only rule reads. bundled-workflow joins the provenance classes. Co-Authored-By: Claude Opus 5.5 --- .../native-references/CHANGELOG.md | 9 + docs/conventions/native-references/README.md | 1 + docs/native-surfaces.md | 5 + .../audit-native-overlap/scripts/discover.py | 299 +++++++++++++++ .../audit-native-overlap/scripts/overlap.py | 358 ++++++++++++++---- .../scripts/test_overlap.py | 283 +++++++++++++- 6 files changed, 876 insertions(+), 79 deletions(-) create mode 100644 plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py diff --git a/docs/conventions/native-references/CHANGELOG.md b/docs/conventions/native-references/CHANGELOG.md index 4e57acd3de..62269856ba 100644 --- a/docs/conventions/native-references/CHANGELOG.md +++ b/docs/conventions/native-references/CHANGELOG.md @@ -6,6 +6,15 @@ major change; additive guidance is minor; clarification is a patch. The doc ship unnumbered, which this file reads as **1.0**; the entry below is the first recorded change and lands the changelog the README said would arrive with it. +## [3.1.0] - 2026-09-29 + +Minor: additive guidance. + +- **`bundled-workflow` is a provenance class.** Claude Code bundles workflows (`deep-research`) + beside its skills, and the overlap store can now record a row against one. Its runtime + relationship is `route` or `suggest`, never `wrap`, because the Native step invokes through the + Skill tool and a workflow is not a skill. + ## [3.0.0] - 2026-09-28 Major: the canonical route-gate token changes, which the Versioning section names as a major change. diff --git a/docs/conventions/native-references/README.md b/docs/conventions/native-references/README.md index d56b9fc32d..2a45bb5025 100644 --- a/docs/conventions/native-references/README.md +++ b/docs/conventions/native-references/README.md @@ -236,6 +236,7 @@ at runtime. Every store row carries one of `route`, `wrap`, or `suggest`. | `builtin-command` | `route` or `suggest`. Never `wrap`: a built-in command is not invoked by the Skill tool | | `bundled-skill` | `route` or `wrap` | | `bundled-skill` carrying `model-invocation-disabled` | `suggest` only. The model never lists the surface, so a route phrase is dead text. The marker is set from the registration the row's evidence names, never from the bare name | +| `bundled-workflow` | `route` or `suggest`. Never `wrap`: the Native step invokes through the Skill tool, and a workflow is not a skill | | `plugin-backed-builtin` | `route` or `wrap` | | `marketplace-plugin` | `route` or `wrap`. The wrap grammar for this class is seam-phrasing's, not the Native step below | | `session-skill` | `route` only | diff --git a/docs/native-surfaces.md b/docs/native-surfaces.md index f493419f33..0868c7a027 100644 --- a/docs/native-surfaces.md +++ b/docs/native-surfaces.md @@ -19,6 +19,7 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen |---|---|---|---|---| | Built-in CLI commands | 5 | 5 | route 1, suggest 4 | complementary 5 | | Bundled skills | 13 | 12 | route 8, suggest 3, wrap 2 | complementary 12, defer 1 | +| Bundled workflows | 0 | 0 | none | none | | Plugin-backed built-ins | 1 | 1 | route 1 | complementary 1 | | Session-provided skills (observation-only) | 1 | 0 | route 1 | defer 1 | | First-party marketplace plugins | 2 | 2 | route 2 | complementary 2 | @@ -341,6 +342,10 @@ and when. See [`docs/conventions/native-references/`](conventions/native-referen - **Baked:** description phrase yes · Boundary section yes · Native step no · suggest sentence no - **Budget caveat:** the baked phrase may be dropped from the skill listing under budget pressure. It is the best available routing surface, not a guaranteed one +## Bundled workflows + +No rows recorded in this lane. + ## Plugin-backed built-ins ### `security-review` → `review:security-review` diff --git a/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py b/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py new file mode 100644 index 0000000000..51d6b358b8 --- /dev/null +++ b/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py @@ -0,0 +1,299 @@ +"""Dynamic overlap discovery: score every native surface against every component. + +Seeded pairs only find what someone already thought of. Discovery scores each +(native surface, repo component) pair from name and description tokens, so a +surface a new Claude Code release adds is compared against the whole fleet +without anyone hand-seeding it. It proposes candidates; it never decides one. + +Scoring, standard library only: + + score = NAME_WEIGHT * name_score + (1 - NAME_WEIGHT) * text_score, + times SINGLE_TOKEN_FACTOR when a described surface shares one token + + name_score IDF-weighted share of the native name's tokens (or one alias's, + whichever covers best) found in the component's skill or agent + name; a token found only in its plugin name earns + PLUGIN_NAME_CREDIT of its weight. + text_score geometric mean of the TF-IDF cosine between the two weighted + bags and the share of the native bag the component covers + (native name x3, aliases x2, description and argument hint x1; + component name x3, plugin name x1, description x1). Coverage + keeps a long component description from diluting a one-line + native one; cosine keeps a long description from matching + everything. + +Tokens are lower-cased, stoplisted, lightly stemmed, split off an `auto`/`sub` +prefix, and a few abbreviations are expanded (`pr` -> pull request), so +`verify` meets `verification` and `subagent` meets `agent`. +""" + +from __future__ import annotations + +import math +import re +from collections import Counter +from dataclasses import dataclass, field +from typing import Any, Iterable + +DEFAULT_THRESHOLD = 0.30 +DEFAULT_TOP_K = 3 +NAME_WEIGHT = 0.5 +PLUGIN_NAME_CREDIT = 0.5 +SINGLE_TOKEN_FACTOR = 0.7 + +STOPWORDS = frozenset( + """ + a about after all also an and any are as at be been before but by can + could do does doing each either every for from get gets give given go + has have how i if in into is it its it's just let like make makes may me + more most my need no none not now of off on one only or other our out + over own per please same see set should show so some such than that the + their them then there these they this those through to too under up us + use used uses using very via want was we what when where whether which + while who why will with without would yet you your + claude code skill skills plugin plugins built builtin bundled native + command commands invoke invoked invocation user users model setup + """.split() +) +ABBREVIATIONS: dict[str, tuple[str, ...]] = { + "pr": ("pull", "request"), + "prs": ("pull", "request"), + "repo": ("repository",), + "repos": ("repository",), +} +MORPH_PREFIXES = ("auto", "sub") +SUFFIXES = ( + "ications", + "ication", + "ations", + "ation", + "ings", + "ing", + "ies", + "ied", + "ers", + "er", + "es", + "ed", + "s", + "y", + "e", +) +TOKEN_RE = re.compile(r"[a-z0-9]+") + + +def stem(word: str) -> str: + """Strip one common suffix, keeping a stem of at least three letters. + + A bare trailing `e` needs a five-letter stem and `es` a sibilant before it, + so `states` meets `state` and never `stats`. + """ + for suffix in SUFFIXES: + if suffix == "es" and not word[:-2].endswith(("s", "x", "z", "ch", "sh")): + continue + floor = 5 if suffix == "e" else 3 + if word.endswith(suffix) and len(word) - len(suffix) >= floor: + return word[: -len(suffix)] + return word + + +def tokenize(text: str) -> list[str]: + tokens: list[str] = [] + for raw in TOKEN_RE.findall((text or "").lower()): + if len(raw) < 2 or raw.isdigit() or raw in STOPWORDS: + continue + words = [raw, *ABBREVIATIONS.get(raw, ())] + for prefix in MORPH_PREFIXES: + if raw.startswith(prefix) and len(raw) - len(prefix) >= 4: + words.append(raw[len(prefix) :]) + tokens.extend(stem(w) for w in words if w not in STOPWORDS) + return tokens + + +def _bag(*parts: tuple[str, float]) -> Counter: + bag: Counter = Counter() + for text, weight in parts: + for token in tokenize(text): + bag[token] += weight + return bag + + +@dataclass +class Surface: + """One native surface: its identity, its text, and its name token sets.""" + + name: str + klass: str + lane: str + registrations: list[dict[str, Any]] + bag: Counter = field(default_factory=Counter) + name_sets: list[set[str]] = field(default_factory=list) + described: bool = False + + @classmethod + def build( + cls, name: str, klass: str, lane: str, registrations: list[dict[str, Any]] + ) -> "Surface": + aliases = sorted( + { + a + for r in registrations + for a in (r.get("aliases") or []) + if isinstance(a, str) + } + ) + text = " ".join( + str(r.get(key) or "") + for r in registrations + for key in ("description", "argument_hint") + ) + bag = _bag((name, 3.0), (" ".join(aliases), 2.0), (text, 1.0)) + name_sets = [s for s in (set(tokenize(n)) for n in [name, *aliases]) if s] + return cls( + name, klass, lane, registrations, bag, name_sets, bool(tokenize(text)) + ) + + +@dataclass +class Component: + plugin: str + name: str + kind: str + bag: Counter + name_tokens: set[str] + plugin_tokens: set[str] + + @classmethod + def build(cls, plugin: str, name: str, kind: str, description: str) -> "Component": + bag = _bag((name, 3.0), (plugin, 1.0), (description, 1.0)) + return cls(plugin, name, kind, bag, set(tokenize(name)), set(tokenize(plugin))) + + +def idf_table(bags: Iterable[Counter]) -> dict[str, float]: + bags = list(bags) + df: Counter = Counter() + for bag in bags: + df.update(bag.keys()) + total = len(bags) + return {t: math.log((1 + total) / (1 + n)) + 1.0 for t, n in df.items()} + + +def _vector(bag: Counter, idf: dict[str, float]) -> dict[str, float]: + return {t: (1 + math.log(w)) * idf.get(t, 1.0) for t, w in bag.items() if w > 0} + + +def _norm(vec: dict[str, float]) -> float: + return math.sqrt(sum(v * v for v in vec.values())) or 1.0 + + +def score_pair( + surface: Surface, + component: Component, + idf: dict[str, float], + vectors: dict[int, tuple[dict[str, float], float]], +) -> tuple[float, list[str]]: + """(score, matched tokens ordered by contribution) for one pair.""" + sv, sn = vectors[id(surface)] + cv, cn = vectors[id(component)] + shared = set(sv) & set(cv) + cosine = sum(sv[t] * cv[t] for t in shared) / (sn * cn) + coverage = sum(sv[t] ** 2 for t in shared) / (sn * sn) + text_score = math.sqrt(cosine * coverage) + name_score = 0.0 + for names in surface.name_sets: + total = sum(idf.get(t, 1.0) for t in names) + hit = sum( + idf.get(t, 1.0) + * (1.0 if t in component.name_tokens else PLUGIN_NAME_CREDIT) + for t in names & (component.name_tokens | component.plugin_tokens) + ) + name_score = max(name_score, hit / total if total else 0.0) + score = NAME_WEIGHT * name_score + (1 - NAME_WEIGHT) * text_score + if len(shared) == 1 and surface.described: + score *= SINGLE_TOKEN_FACTOR + matched = sorted(shared, key=lambda t: (-sv[t] * cv[t], t)) + return round(score, 4), matched + + +def discover( + surfaces: list[Surface], + components: list[Component], + *, + threshold: float = DEFAULT_THRESHOLD, + top_k: int = DEFAULT_TOP_K, +) -> list[tuple[Surface, Component, float, list[str]]]: + """Every pair at or over the threshold, at most top_k per native surface.""" + idf = idf_table([s.bag for s in surfaces] + [c.bag for c in components]) + vectors: dict[int, tuple[dict[str, float], float]] = {} + for item in [*surfaces, *components]: + vec = _vector(item.bag, idf) + vectors[id(item)] = (vec, _norm(vec)) + found: list[tuple[Surface, Component, float, list[str]]] = [] + for surface in surfaces: + if not surface.bag: + continue + scored = [] + for component in components: + score, matched = score_pair(surface, component, idf, vectors) + if score >= threshold and matched: + scored.append((surface, component, score, matched)) + scored.sort(key=lambda item: (-item[2], item[1].plugin, item[1].name)) + found.extend(scored[:top_k]) + return found + + +def invocability(registrations: list[dict[str, Any]]) -> dict[str, Any]: + """Who can invoke a surface, from the fields its registrations carry. + + `model_invocable` wins; an older extraction's `disable_model_invocation` + stands in for it; anything else is unknown (None). Registrations that + disagree (a name collision) are unknown too: the name alone cannot say + which registration a caller would reach. + """ + + def agreed(values: list[Any]) -> Any: + return values[0] if values and all(v == values[0] for v in values) else None + + models: list[Any] = [] + users: list[Any] = [] + hints: list[Any] = [] + for entry in registrations: + if isinstance(entry.get("model_invocable"), bool): + models.append(entry["model_invocable"]) + elif isinstance(entry.get("disable_model_invocation"), bool): + models.append(not entry["disable_model_invocation"]) + else: + models.append(None) + users.append( + entry.get("user_invocable") + if isinstance(entry.get("user_invocable"), bool) + else None + ) + hint = entry.get("argument_hint") + hints.append(hint if isinstance(hint, str) and hint else None) + model, user = agreed(models), agreed(users) + invocable_by = { + (True, True): "model+user", + (False, True): "user-only", + (True, False): "model-only", + }.get((model, user), "unknown") + return { + "model_invocable": model, + "user_invocable": user, + "argument_hint": agreed(hints), + "invocable_by": invocable_by, + } + + +def recommended_integration(klass: str, invocable_by: str) -> str | None: + """A recommendation label for the human, never a verdict or a store value. + + A user-only surface can only be suggested (ours tells the model to suggest + the user type it); a model-invocable one can be routed to or wrapped, + except a built-in command, which the store never lets take `wrap`. + """ + if invocable_by == "user-only": + return "suggest" + if invocable_by in ("model+user", "model-only"): + return "route" if klass == "builtin-command" else "route-or-wrap" + return None diff --git a/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py b/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py index d86244809c..d1273536f6 100755 --- a/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py +++ b/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py @@ -55,6 +55,8 @@ # shape, so a consumer can never read it by a rule of its own. from registrations import registrations_of # noqa: E402 (path set above; plugin-bundled module) +import discover # noqa: E402 (sibling module; the script's own directory is on sys.path) + MIN_PYTHON = (3, 11) STORE_SCHEMA = 1 @@ -71,6 +73,7 @@ NATIVE_CLASSES = ( "builtin-command", "bundled-skill", + "bundled-workflow", "plugin-backed-builtin", "session-skill", "marketplace-plugin", @@ -115,12 +118,21 @@ # Extraction lanes the sibling extractor reports integrity for, and the # native class each one carries. Session-provided and marketplace classes have # no lane: nothing about them is derivable from the binary. -LANE_ORDER = ("builtin_commands", "bundled_skills", "plugin_backed") +# `bundled_workflows` is optional: an extraction that predates the lane lacks +# the key, and the lane is then not reported rather than read as broken. +LANE_ORDER = ( + "builtin_commands", + "bundled_skills", + "bundled_workflows", + "plugin_backed", +) LANE_OF_CLASS = { "builtin-command": "builtin_commands", "bundled-skill": "bundled_skills", + "bundled-workflow": "bundled_workflows", "plugin-backed-builtin": "plugin_backed", } +CLASS_OF_LANE = {lane: klass for klass, lane in LANE_OF_CLASS.items()} # Lane order in the generated view: (class, section heading, singular noun used @@ -129,6 +141,7 @@ LANES: tuple[tuple[str, str, str], ...] = ( ("builtin-command", "Built-in CLI commands", "built-in command"), ("bundled-skill", "Bundled skills", "bundled skill"), + ("bundled-workflow", "Bundled workflows", "bundled workflow"), ("plugin-backed-builtin", "Plugin-backed built-ins", "plugin-backed built-in"), ( "session-skill", @@ -357,6 +370,96 @@ def scan_components(repo: Path) -> dict[str, list[str]]: return found +def load_components(repo: Path) -> list[discover.Component]: + """The target side with its routing text, for discovery scoring.""" + found = scan_components(repo) + corpus: list[discover.Component] = [] + for kind, ids in (("skill", found["skills"]), ("agent", found["agents"])): + for ident in ids: + plugin, name = ident.split(":", 1) + path = component_path(repo, plugin, name, kind) + try: + frontmatter, _body = split_frontmatter(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError): + frontmatter = "" + corpus.append( + discover.Component.build( + plugin, name, kind, frontmatter_description(frontmatter) + ) + ) + return corpus + + +def native_surfaces(lane_payloads: dict[str, Any]) -> list[discover.Surface]: + """Every scorable native surface. An `internal` registration is skipped: + it is plumbing the product never offers anyone to type or call.""" + surfaces: list[discover.Surface] = [] + for lane, payload in lane_payloads.items(): + for name, entry in payload.items(): + if lane == "plugin_backed": + registrations = [{"name": name, "plugin_name": entry}] + else: + registrations = registrations_of(entry) + if not registrations or any(r.get("internal") for r in registrations): + continue + surfaces.append( + discover.Surface.build(name, CLASS_OF_LANE[lane], lane, registrations) + ) + return surfaces + + +def registration_evidence(registrations: list[dict[str, Any]]) -> list[str]: + """Per-registration evidence lines: markers, aliases, description, invocability.""" + evidence: list[str] = [] + if len(registrations) > 1: + evidence.append( + f"name collision: {len(registrations)} distinct registrations share " + "this name in the extraction; a per-registration property (model " + "invocability, gating) is read from the registration the row's " + "evidence names, never from the bare name" + ) + for position, entry in enumerate(registrations, start=1): + tag = f"[{position}] " if len(registrations) > 1 else "" + markers = [m for m in NATIVE_MARKERS if entry.get(m)] + if markers: + evidence.append(f"{tag}markers: {', '.join(markers)}") + if entry.get("aliases"): + evidence.append(f"{tag}aliases: {', '.join(entry['aliases'])}") + if entry.get("description"): + evidence.append(f"{tag}native description: {entry['description']}") + model = discover.invocability([entry])["model_invocable"] + if model is not None: + flagged = {"disable_model_invocation", "model_invocable"} & set( + entry.get("flag_driven") or [] + ) + evidence.append( + f"{tag}model invocation: {'enabled' if model else 'disabled'}" + + (" (flag-driven at runtime)" if flagged else "") + ) + return evidence + + +def _store_verdicts(store_path: Path) -> dict[tuple[str, str, str, str], Any]: + """Store identity -> verdict. An absent or unreadable store yields none: + detect then reports every pair as NEW, and self-check owns store health.""" + store, _error = load_json(store_path) + rows = store.get("rows") if isinstance(store, dict) else None + verdicts: dict[tuple[str, str, str, str], Any] = {} + for row in rows if isinstance(rows, list) else []: + try: + component = row["component"] + key = ( + str(row["native"]["name"]), + str(component["plugin"]), + str(component["skill"]), + str(component.get("kind", "skill")), + ) + except (KeyError, TypeError): + continue + verdicts[key] = row.get("verdict") + return verdicts + + def current_cli_version() -> tuple[str | None, str]: """Best-effort current CLI version from a cheap `claude --version` call. @@ -523,10 +626,7 @@ def validate_row(row: Any, index: int) -> list[str]: if not isinstance(baked, dict) or not all( isinstance(baked.get(key), bool) for key in BAKED_FLAGS ): - problems.append( - f"{label}: `baked` must carry boolean " - f"{', '.join(BAKED_FLAGS)}" - ) + problems.append(f"{label}: `baked` must carry boolean {', '.join(BAKED_FLAGS)}") if not isinstance(row.get("budget_caveat"), bool): problems.append(f"{label}: `budget_caveat` must be a boolean") @@ -555,9 +655,7 @@ def _integration_problems(row: dict[str, Any], label: str) -> list[str]: """ integration = row.get("integration") if integration not in INTEGRATIONS: - return [ - f"{label}: `integration` must be one of {', '.join(INTEGRATIONS)}" - ] + return [f"{label}: `integration` must be one of {', '.join(INTEGRATIONS)}"] problems: list[str] = [] native = row.get("native") if isinstance(row.get("native"), dict) else {} klass = native.get("class") @@ -575,15 +673,12 @@ def _integration_problems(row: dict[str, Any], label: str) -> list[str]: problems.append( f"{label}: a `session-skill` row takes `integration` `route`" ) - elif klass == "builtin-command": + elif klass in ("builtin-command", "bundled-workflow"): if integration not in ("route", "suggest"): problems.append( - f"{label}: a `builtin-command` row takes `route` or `suggest`, " - "never `wrap`" + f"{label}: a `{klass}` row takes `route` or `suggest`, never `wrap`" ) - elif ( - klass == "bundled-skill" and "model-invocation-disabled" in markers - ): + elif klass == "bundled-skill" and "model-invocation-disabled" in markers: if integration != "suggest": problems.append( f"{label}: a bundled skill marked `model-invocation-disabled` " @@ -592,9 +687,7 @@ def _integration_problems(row: dict[str, Any], label: str) -> list[str]: ) elif klass in ROUTE_OR_WRAP_CLASSES: if integration not in ("route", "wrap"): - problems.append( - f"{label}: a `{klass}` row takes `route` or `wrap`" - ) + problems.append(f"{label}: a `{klass}` row takes `route` or `wrap`") if integration in ("wrap", "suggest"): evidence = row.get("evidence") if isinstance(row.get("evidence"), list) else [] if not any( @@ -987,10 +1080,14 @@ def _presence_descriptions(plugins_dir: Path) -> list[tuple[str, str]]: continue description = frontmatter_description(frontmatter) if description: - found.append((f"{skill_md.parents[2].name}:{skill_md.parent.name}", description)) + found.append( + (f"{skill_md.parents[2].name}:{skill_md.parent.name}", description) + ) for manifest in sorted(plugins_dir.glob("*/.claude-plugin/plugin.json")): try: - description = json.loads(manifest.read_text(encoding="utf-8")).get("description") + description = json.loads(manifest.read_text(encoding="utf-8")).get( + "description" + ) except (OSError, ValueError, AttributeError): continue if isinstance(description, str) and description: @@ -1096,17 +1193,89 @@ def lane_state(lane: str | None) -> dict[str, Any] | None: known_skills = set(components["skills"]) known_agents = set(components["agents"]) + # The workflow lane is optional: an extraction that predates it simply + # has no such lane, which is not a missing consumed key. + lane_payloads = { + lane: inventory.get(lane) + for lane in LANE_ORDER + if isinstance(inventory.get(lane), dict) + } native_index: dict[str, dict[str, Any]] = {} - for name, entry in (inventory.get("bundled_skills") or {}).items(): - native_index[name] = {"class": "bundled-skill", "entry": entry} - for name, entry in (inventory.get("builtin_commands") or {}).items(): - native_index.setdefault(name, {"class": "builtin-command", "entry": entry}) + for lane in ("bundled_workflows", "bundled_skills", "builtin_commands"): + for name, entry in (lane_payloads.get(lane) or {}).items(): + native_index.setdefault( + name, {"class": CLASS_OF_LANE[lane], "entry": entry} + ) for name, plugin in (inventory.get("plugin_backed") or {}).items(): native_index[name] = { "class": "plugin-backed-builtin", "entry": {"name": name, "plugin_name": plugin}, } + store_verdicts = _store_verdicts(Path(args.store)) + + def broken_lane_evidence(lanes_in_play: set[str]) -> tuple[bool | None, list[str]]: + if not lanes_in_play: + return None, [] # session-provided and marketplace rows have no lane + notes: list[str] = [] + broken = [ + lane + for lane in sorted(lanes_in_play) + if (lane_state(lane) or {}).get("status") == "broken" + ] + for lane in broken: + problems = (lane_state(lane) or {}).get("problems") or [] + notes.append( + f"the `{lane}` lane of this extraction is broken" + + (f" ({'; '.join(problems)})" if problems else "") + + " - presence or absence in that lane is not re-derivable from " + "this run" + ) + return not broken, notes + + def native_block( + name: str, klass: str, seen: dict[str, Any] | None + ) -> dict[str, Any]: + registrations = registrations_of(seen["entry"]) if seen else [] + who = discover.invocability(registrations) + markers = [m for m in NATIVE_MARKERS if any(r.get(m) for r in registrations)] + if who["model_invocable"] is False: + markers.append("model-invocation-disabled") + return { + "name": name, + "class": klass, + "observed": seen is not None, + "markers": sorted(set(markers), key=NATIVE_MARKERS.index), + "model_invocable": who["model_invocable"], + "user_invocable": who["user_invocable"], + "argument_hint": who["argument_hint"], + "invocable_by": who["invocable_by"], + } + + def key_of(name: Any, component: dict[str, Any]) -> tuple[str, str, str, str]: + return ( + str(name), + str(component.get("plugin")), + str(component.get("skill")), + str(component.get("kind", "skill")), + ) + + corpus = load_components(repo) + surfaces = native_surfaces(lane_payloads) + scored = ( + discover.discover(surfaces, corpus, threshold=args.threshold, top_k=args.top_k) + if args.top_k > 0 + else [] + ) + scores = { + key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}): ( + s, + score, + matched, + ) + for s, c, score, matched in scored + } + candidates: list[dict[str, Any]] = [] for pair in pairs_data["pairs"]: # shape guaranteed by validate_pairs above native = pair.get("native", {}) @@ -1123,59 +1292,17 @@ def lane_state(lane: str | None) -> dict[str, Any] | None: evidence.append( f"`{native.get('name')}` present in the extraction as {seen['class']}" ) - registrations = registrations_of(seen["entry"]) - if len(registrations) > 1: - evidence.append( - f"name collision: {len(registrations)} distinct registrations share " - "this name in the extraction; a per-registration property (model " - "invocability, gating) is read from the registration the row's " - "evidence names, never from the bare name" - ) - for position, entry in enumerate(registrations, start=1): - tag = f"[{position}] " if len(registrations) > 1 else "" - markers = [m for m in NATIVE_MARKERS if entry.get(m)] - if markers: - evidence.append(f"{tag}markers: {', '.join(markers)}") - if entry.get("aliases"): - evidence.append(f"{tag}aliases: {', '.join(entry['aliases'])}") - if entry.get("description"): - evidence.append(f"{tag}native description: {entry['description']}") - if "disable_model_invocation" in entry: - mode = ( - "disabled" if entry["disable_model_invocation"] else "enabled" - ) - flagged = "disable_model_invocation" in ( - entry.get("flag_driven") or [] - ) - evidence.append( - f"{tag}model invocation: {mode}" - + (" (flag-driven at runtime)" if flagged else "") - ) + evidence.extend(registration_evidence(registrations_of(seen["entry"]))) # The lane a candidate's re-derivability depends on: the seeded class # when the name is absent from the extraction, the observed class when # present, and both when the two disagree (a class collision), so a # broken lane on either side marks the candidate. seeded_lane = LANE_OF_CLASS.get(native.get("class")) observed_lane = LANE_OF_CLASS.get(seen["class"]) if seen else None - relevant_lanes = {lane for lane in (seeded_lane, observed_lane) if lane} - re_derivable: bool | None - if not relevant_lanes: - re_derivable = None # session-provided and marketplace rows have no lane - else: - broken = [ - lane - for lane in sorted(relevant_lanes) - if (lane_state(lane) or {}).get("status") == "broken" - ] - re_derivable = not broken - for lane in broken: - problems = (lane_state(lane) or {}).get("problems") or [] - evidence.append( - f"the `{lane}` lane of this extraction is broken" - + (f" ({'; '.join(problems)})" if problems else "") - + " - presence or absence in that lane is not re-derivable from " - "this run" - ) + re_derivable, notes = broken_lane_evidence( + {lane for lane in (seeded_lane, observed_lane) if lane} + ) + evidence.extend(notes) kind = component.get("kind", "skill") pool = known_agents if kind == "agent" else known_skills target_present = target in pool @@ -1183,17 +1310,68 @@ def lane_state(lane: str | None) -> dict[str, Any] | None: evidence.append(f"target `{target}` not found in the repo tree at {repo}") if pair.get("why"): evidence.append(f"seeded rationale: {pair['why']}") + key = key_of(native.get("name"), component) + _surface, score, matched = scores.pop(key, (None, None, None)) + klass = (seen or {}).get("class", native.get("class")) + block = native_block(native.get("name"), klass, seen) + block["seeded_class"] = native.get("class") candidates.append( { - "native": { - "name": native.get("name"), - "class": (seen or {}).get("class", native.get("class")), - "seeded_class": native.get("class"), - "observed": seen is not None, - }, + "origin": "seeded", + "native": block, "component": component, "component_present": target_present, "re_derivable": re_derivable, + "score": score, + "matched_tokens": matched, + "recommended_integration": discover.recommended_integration( + klass, block["invocable_by"] + ), + "store_verdict": store_verdicts.get(key), + "verdict": None, + "evidence": evidence, + } + ) + + existing: list[dict[str, Any]] = [] + for key, (surface, score, matched) in sorted( + scores.items(), key=lambda item: (item[0][0], -item[1][1], item[0][1:]) + ): + name, plugin, skill, kind = key + component = {"plugin": plugin, "skill": skill, "kind": kind} + if key in store_verdicts: + existing.append( + { + "native": name, + "class": surface.klass, + "component": component, + "score": score, + "store_verdict": store_verdicts[key], + } + ) + continue + seen = {"class": surface.klass, "entry": surface.registrations} + block = native_block(name, surface.klass, seen) + re_derivable, notes = broken_lane_evidence({surface.lane}) + evidence = [ + f"`{name}` present in the extraction as {surface.klass}", + *registration_evidence(surface.registrations), + f"discovered: score {score} from shared tokens {', '.join(matched[:8])}", + *notes, + ] + candidates.append( + { + "origin": "discovered", + "native": block, + "component": component, + "component_present": True, + "re_derivable": re_derivable, + "score": score, + "matched_tokens": matched, + "recommended_integration": discover.recommended_integration( + surface.klass, block["invocable_by"] + ), + "store_verdict": None, "verdict": None, "evidence": evidence, } @@ -1242,6 +1420,16 @@ def lane_state(lane: str | None) -> dict[str, Any] | None: "skills": len(components["skills"]), "agents": len(components["agents"]), }, + "discovery": { + "threshold": args.threshold, + "top_k": args.top_k, + "lanes_scored": sorted(lane_payloads), + "surfaces_scored": len(surfaces), + "components_scored": len(corpus), + "seeded": sum(1 for c in candidates if c["origin"] == "seeded"), + "discovered": sum(1 for c in candidates if c["origin"] == "discovered"), + "existing": existing, + }, "candidates": candidates, "note": ( "Candidates only. No verdict is assigned here: every verdict is a human's, " @@ -1521,6 +1709,24 @@ def add_paths(sub: argparse.ArgumentParser) -> None: detect.add_argument( "--out", help="write the candidate report here instead of stdout" ) + detect.add_argument( + "--threshold", + type=float, + default=discover.DEFAULT_THRESHOLD, + help=( + "lowest similarity score a discovered pair needs, 0-1 " + f"(default: {discover.DEFAULT_THRESHOLD})" + ), + ) + detect.add_argument( + "--top-k", + type=int, + default=discover.DEFAULT_TOP_K, + help=( + "most discovered components kept per native surface; 0 turns " + f"discovery off (default: {discover.DEFAULT_TOP_K})" + ), + ) add_paths(detect) generate = subparsers.add_parser( diff --git a/plugins/claude-ops/skills/audit-native-overlap/scripts/test_overlap.py b/plugins/claude-ops/skills/audit-native-overlap/scripts/test_overlap.py index 748b1a4c5d..cfcf4c247a 100755 --- a/plugins/claude-ops/skills/audit-native-overlap/scripts/test_overlap.py +++ b/plugins/claude-ops/skills/audit-native-overlap/scripts/test_overlap.py @@ -20,6 +20,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) +import discover # noqa: E402 - path shim above must run first import overlap # noqa: E402 - path shim above must run first FIXTURE_CLI_VERSION = "2.1.232" @@ -617,6 +618,18 @@ def test_builtin_command_rejects_wrap(self): problems = overlap.validate_row(row, 0) self.assertTrue(any("never `wrap`" in problem for problem in problems)) + def test_bundled_workflow_rejects_wrap(self): + row = deep_copy(BASE_ROW) + row["native"] = { + "name": "deep-research", + "class": "bundled-workflow", + "markers": [], + } + row["integration"] = "wrap" + row["baked"]["boundary_section"] = False + problems = overlap.validate_row(row, 0) + self.assertTrue(any("never `wrap`" in problem for problem in problems)) + def test_builtin_command_allows_suggest_with_invocation_evidence(self): row = deep_copy(BASE_ROW) row["native"] = {"name": "export", "class": "builtin-command", "markers": []} @@ -631,7 +644,9 @@ def test_session_skill_rejects_suggest(self): row["integration"] = "suggest" row["observation"]["class"] = "live-roster" row["baked"]["boundary_section"] = False - row["evidence"] = ["invocation mode: session roster, not a bundled registration"] + row["evidence"] = [ + "invocation mode: session roster, not a bundled registration" + ] problems = overlap.validate_row(row, 0) self.assertTrue(any("session-skill" in problem for problem in problems)) @@ -1302,7 +1317,9 @@ def test_a_lane_attributed_advisory_degrades_only_its_lane(self): report["integrity"]["lanes"]["builtin_commands"]["counts_are"], "totals" ) - def test_an_inventory_without_lanes_keeps_the_old_behaviour(self): # identifier, not prose # spellchecker:disable-line + def test_an_inventory_without_lanes_keeps_the_old_behaviour( + self, + ): # identifier, not prose # spellchecker:disable-line self.write_inventory() out = self.repo.root / "candidates.json" self.assertEqual(self.detect(out), 0) @@ -1556,7 +1573,9 @@ def test_bare_available_phrase_is_not_a_suggest_orphan(self): def test_native_step_forward_parity_wants_the_heading(self): row = deep_copy(BASE_ROW) row["integration"] = "wrap" - row["evidence"] = ["invocation mode: model-invocable, no disableModelInvocation"] + row["evidence"] = [ + "invocation mode: model-invocable, no disableModelInvocation" + ] row["baked"]["native_step"] = True self.repo.write_store(make_store([row])) self.repo.generate() @@ -1597,5 +1616,263 @@ def test_scan_of_a_tree_without_plugins_is_empty(self): self.assertEqual(found, {"skills": [], "agents": []}) +def surface(name, klass="bundled-skill", **fields): + lane = overlap.LANE_OF_CLASS[klass] + return discover.Surface.build(name, klass, lane, [{"name": name, **fields}]) + + +class DiscoveryScoringTests(unittest.TestCase): + def test_tokenize_stems_expands_and_stoplists(self): + self.assertEqual(discover.tokenize("verify"), discover.tokenize("verification")) + self.assertEqual(discover.tokenize("pr"), ["pr", "pull", "request"]) + self.assertIn("agent", discover.tokenize("subagents")) + self.assertEqual(discover.tokenize("the skill for Claude Code"), []) + self.assertNotEqual(discover.tokenize("states"), discover.tokenize("stats")) + + def test_a_name_match_outscores_an_unrelated_component(self): + native = surface("commit", description="Create a git commit") + ours = discover.Component.build( + "source-control", "commit", "skill", "Create a git commit with a trailer." + ) + other = discover.Component.build( + "songwriting", "rhyme", "skill", "Find rhymes for a lyric line." + ) + found = discover.discover([native], [ours, other], threshold=0.0, top_k=5) + ranked = [(c.plugin, c.name) for _s, c, _score, _m in found] + self.assertEqual(ranked[0], ("source-control", "commit")) + self.assertNotIn(("songwriting", "rhyme"), ranked) # no shared token at all + self.assertIn("commit", found[0][3]) + + def test_threshold_and_top_k_bound_the_result(self): + native = surface("pr", description="Create a pull request") + corpus = [ + discover.Component.build( + "vcs", f"pull-request-{n}", "skill", "Open a pull request." + ) + for n in range(5) + ] + self.assertEqual(len(discover.discover([native], corpus, top_k=2)), 2) + self.assertEqual( + discover.discover([native], corpus, threshold=1.01, top_k=5), [] + ) + + def test_invocability_maps_every_combination(self): + cases = [ + ({"model_invocable": True, "user_invocable": True}, "model+user"), + ({"model_invocable": False, "user_invocable": True}, "user-only"), + ({"model_invocable": True, "user_invocable": False}, "model-only"), + ({"user_invocable": True}, "unknown"), + ({}, "unknown"), + ({"disable_model_invocation": True, "user_invocable": True}, "user-only"), + ] + for fields, expected in cases: + with self.subTest(fields=fields): + self.assertEqual( + discover.invocability([fields])["invocable_by"], expected + ) + + def test_disagreeing_registrations_are_unknown(self): + who = discover.invocability( + [ + {"model_invocable": True, "user_invocable": True}, + {"model_invocable": False, "user_invocable": True}, + ] + ) + self.assertIsNone(who["model_invocable"]) + self.assertEqual(who["invocable_by"], "unknown") + + def test_recommended_integration_is_a_label_per_invocability(self): + rec = discover.recommended_integration + self.assertEqual(rec("bundled-skill", "user-only"), "suggest") + self.assertEqual(rec("bundled-skill", "model+user"), "route-or-wrap") + self.assertEqual(rec("builtin-command", "model+user"), "route") + self.assertIsNone(rec("bundled-skill", "unknown")) + + +class DiscoveryDetectTests(unittest.TestCase): + def setUp(self): + self.repo = TempRepo() + self.addCleanup(self.repo.cleanup) + self.repo.write_skill( + "source-control", + "commit", + description="Create a git commit with a trailer.", + ) + self.repo.write_skill( + "songwriting", "rhyme", description="Find rhymes for a lyric." + ) + self.pairs_path = self.repo.root / "pairs.json" + self.pairs_path.write_text( + json.dumps({"schema": 1, "pairs": []}), encoding="utf-8" + ) + self.inventory_path = self.repo.root / "inventory.json" + self.out = self.repo.root / "candidates.json" + + def write_inventory(self, **overrides): + payload = { + "schema": 1, + "builtin_commands": {}, + "bundled_skills": { + "commit": {"name": "commit", "description": "Create a git commit"} + }, + "plugin_backed": {}, + "integrity": {"status": "ok"}, + } + payload.update(overrides) + self.inventory_path.write_text(json.dumps(payload), encoding="utf-8") + + def detect(self, *extra): + code = overlap.main( + [ + "detect", + "--repo", + str(self.repo.root), + "--inventory", + str(self.inventory_path), + "--pairs", + str(self.pairs_path), + "--out", + str(self.out), + *extra, + ] + ) + return code, json.loads(self.out.read_text(encoding="utf-8")) + + def discovered(self, report): + return [c for c in report["candidates"] if c["origin"] == "discovered"] + + def test_a_discovered_candidate_carries_score_tokens_and_no_verdict(self): + self.write_inventory() + code, report = self.detect() + self.assertEqual(code, 0) + [candidate] = self.discovered(report) + self.assertEqual(candidate["component"]["skill"], "commit") + self.assertGreater(candidate["score"], 0.3) + self.assertIn("commit", candidate["matched_tokens"]) + self.assertIsNone(candidate["verdict"]) + self.assertIsNone(candidate["store_verdict"]) + self.assertEqual(report["discovery"]["discovered"], 1) + + def test_a_high_threshold_or_zero_top_k_discovers_nothing(self): + self.write_inventory() + self.assertEqual(self.discovered(self.detect("--threshold", "1.01")[1]), []) + self.assertEqual(self.discovered(self.detect("--top-k", "0")[1]), []) + + def test_a_seeded_pair_is_not_repeated_as_discovered(self): + self.write_inventory() + pair = { + "native": {"name": "commit", "class": "bundled-skill"}, + "component": { + "plugin": "source-control", + "skill": "commit", + "kind": "skill", + }, + } + self.pairs_path.write_text( + json.dumps({"schema": 1, "pairs": [pair]}), encoding="utf-8" + ) + _code, report = self.detect() + self.assertEqual(self.discovered(report), []) + [seeded] = report["candidates"] + self.assertEqual(seeded["origin"], "seeded") + self.assertIsNotNone(seeded["score"]) + + def test_a_pair_already_in_the_store_is_reported_as_existing(self): + row = deep_copy(BASE_ROW) + row["native"] = {"name": "commit", "class": "bundled-skill", "markers": []} + row["component"] = { + "plugin": "source-control", + "skill": "commit", + "kind": "skill", + } + self.repo.write_store(make_store([row])) + self.write_inventory() + _code, report = self.detect() + self.assertEqual(self.discovered(report), []) + [existing] = report["discovery"]["existing"] + self.assertEqual(existing["store_verdict"], "complementary") + + def test_internal_commands_are_never_scored(self): + self.write_inventory( + bundled_skills={}, + builtin_commands={ + "commit": {"name": "commit", "description": "Commit", "internal": True} + }, + ) + _code, report = self.detect() + self.assertEqual(report["discovery"]["surfaces_scored"], 0) + self.assertEqual(self.discovered(report), []) + + def test_the_workflow_lane_is_optional(self): + self.write_inventory() + code, report = self.detect() + self.assertEqual(code, 0) + self.assertNotIn("bundled_workflows", report["discovery"]["lanes_scored"]) + self.repo.write_skill( + "discovery", "research-deep", description="Dispatch deep external research." + ) + self.write_inventory( + bundled_workflows={ + "deep-research": { + "name": "deep-research", + "description": "Deep research", + } + } + ) + code, report = self.detect() + self.assertEqual(code, 0) + self.assertIn("bundled_workflows", report["discovery"]["lanes_scored"]) + classes = { + c["native"]["name"]: c["native"]["class"] for c in self.discovered(report) + } + self.assertEqual(classes.get("deep-research"), "bundled-workflow") + + def test_missing_invocability_fields_degrade_to_unknown(self): + self.write_inventory() + _code, report = self.detect() + [candidate] = self.discovered(report) + self.assertEqual(candidate["native"]["invocable_by"], "unknown") + self.assertIsNone(candidate["native"]["model_invocable"]) + self.assertIsNone(candidate["native"]["argument_hint"]) + self.assertIsNone(candidate["recommended_integration"]) + + def test_a_user_only_surface_recommends_suggest_and_carries_the_marker(self): + self.write_inventory( + bundled_skills={ + "commit": { + "name": "commit", + "description": "Create a git commit", + "model_invocable": False, + "user_invocable": True, + "argument_hint": "[message]", + } + } + ) + _code, report = self.detect() + [candidate] = self.discovered(report) + self.assertEqual(candidate["native"]["invocable_by"], "user-only") + self.assertEqual(candidate["native"]["argument_hint"], "[message]") + self.assertIn("model-invocation-disabled", candidate["native"]["markers"]) + self.assertEqual(candidate["recommended_integration"], "suggest") + self.assertIn("model invocation: disabled", candidate["evidence"]) + + def test_a_model_invocable_skill_recommends_route_or_wrap(self): + self.write_inventory( + bundled_skills={ + "commit": { + "name": "commit", + "description": "Create a git commit", + "model_invocable": True, + "user_invocable": True, + } + } + ) + _code, report = self.detect() + [candidate] = self.discovered(report) + self.assertEqual(candidate["native"]["invocable_by"], "model+user") + self.assertEqual(candidate["native"]["markers"], []) + self.assertEqual(candidate["recommended_integration"], "route-or-wrap") + + if __name__ == "__main__": unittest.main() From 1ad73a5369446365e1b94c720667d5db5cb91983 Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:05:22 -0400 Subject: [PATCH 03/22] feat(claude-ops): seed conceptual native overlaps and score every seed Seven pairs whose overlap is conceptual rather than lexical (recap, fork, subtask, batch, explain-usage x2, fewer-permission-prompts) score below the discovery cut against Claude Code 2.1.284, so they join the seeded pairs. Seeded candidates now report their lexical score even below the cut. Co-Authored-By: Claude Opus 5.5 --- .../reference/canonical-pairs.json | 35 +++++++++++++++ .../audit-native-overlap/scripts/discover.py | 43 +++++++++++++------ .../audit-native-overlap/scripts/overlap.py | 32 ++++++++------ 3 files changed, 84 insertions(+), 26 deletions(-) diff --git a/plugins/claude-ops/skills/audit-native-overlap/reference/canonical-pairs.json b/plugins/claude-ops/skills/audit-native-overlap/reference/canonical-pairs.json index 82e2b6c358..df31b44820 100644 --- a/plugins/claude-ops/skills/audit-native-overlap/reference/canonical-pairs.json +++ b/plugins/claude-ops/skills/audit-native-overlap/reference/canonical-pairs.json @@ -77,6 +77,41 @@ "component": { "plugin": "prototype", "skill": "explore-directions", "kind": "skill" }, "why": "explore-directions offers the design-canvas Artifact as one of its mockup surfaces; same live integration, sibling component." }, + { + "native": { "name": "recap", "class": "builtin-command" }, + "component": { "plugin": "session-flow", "skill": "orient", "kind": "skill" }, + "why": "Both answer 'where does this session stand'; recap summarizes the conversation, orient reads durable state the conversation never saw. The two share almost no vocabulary, so discovery scores the pair low." + }, + { + "native": { "name": "fork", "class": "builtin-command" }, + "component": { "plugin": "session-flow", "skill": "continue-in-background", "kind": "skill" }, + "why": "Both hand the current work to a background agent that carries the conversation forward." + }, + { + "native": { "name": "subtask", "class": "builtin-command" }, + "component": { "plugin": "session-flow", "skill": "continue-in-background", "kind": "skill" }, + "why": "Both send an agent off with the session's context; subtask returns its result here, continue-in-background detaches it." + }, + { + "native": { "name": "batch", "class": "bundled-skill" }, + "component": { "plugin": "implementation", "skill": "implement-dispatch", "kind": "skill" }, + "why": "Both plan a large change and fan it out to worker agents; the name tokens never meet, so discovery misses the pair." + }, + { + "native": { "name": "explain-usage", "class": "bundled-skill" }, + "component": { "plugin": "claude-ops", "skill": "observability", "kind": "skill" }, + "why": "Both explain where a session's tokens and cost went." + }, + { + "native": { "name": "explain-usage", "class": "bundled-skill" }, + "component": { "plugin": "context-budget", "skill": "audit", "kind": "skill" }, + "why": "Both attribute context consumption to what caused it; the audit measures the startup payload, explain-usage the whole session." + }, + { + "native": { "name": "fewer-permission-prompts", "class": "bundled-skill" }, + "component": { "plugin": "claude-config", "skill": "audit-permission-state", "kind": "skill" }, + "why": "Both work from the permission rules in effect; the native skill writes an allowlist from observed usage, the audit reports what is in effect and what auto mode drops." + }, { "native": { "name": "design-sync", "class": "bundled-skill" }, "component": { "plugin": "visualization", "skill": "visualize", "kind": "skill" }, diff --git a/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py b/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py index 51d6b358b8..1f705b95da 100644 --- a/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py +++ b/plugins/claude-ops/skills/audit-native-overlap/scripts/discover.py @@ -215,33 +215,50 @@ def score_pair( return round(score, 4), matched -def discover( - surfaces: list[Surface], - components: list[Component], - *, - threshold: float = DEFAULT_THRESHOLD, - top_k: int = DEFAULT_TOP_K, -) -> list[tuple[Surface, Component, float, list[str]]]: - """Every pair at or over the threshold, at most top_k per native surface.""" +Scored = tuple[Surface, Component, float, list[str]] + + +def score_all(surfaces: list[Surface], components: list[Component]) -> list[Scored]: + """Every pair sharing at least one token, best first per native surface.""" idf = idf_table([s.bag for s in surfaces] + [c.bag for c in components]) vectors: dict[int, tuple[dict[str, float], float]] = {} for item in [*surfaces, *components]: vec = _vector(item.bag, idf) vectors[id(item)] = (vec, _norm(vec)) - found: list[tuple[Surface, Component, float, list[str]]] = [] + found: list[Scored] = [] for surface in surfaces: - if not surface.bag: - continue scored = [] for component in components: score, matched = score_pair(surface, component, idf, vectors) - if score >= threshold and matched: + if matched: scored.append((surface, component, score, matched)) scored.sort(key=lambda item: (-item[2], item[1].plugin, item[1].name)) - found.extend(scored[:top_k]) + found.extend(scored) return found +def select(scored: list[Scored], *, threshold: float, top_k: int) -> list[Scored]: + """The pairs at or over the threshold, at most top_k per native surface.""" + kept: list[Scored] = [] + per_surface: Counter = Counter() + for item in scored: + if item[2] >= threshold and per_surface[id(item[0])] < top_k: + per_surface[id(item[0])] += 1 + kept.append(item) + return kept + + +def discover( + surfaces: list[Surface], + components: list[Component], + *, + threshold: float = DEFAULT_THRESHOLD, + top_k: int = DEFAULT_TOP_K, +) -> list[Scored]: + """Every pair at or over the threshold, at most top_k per native surface.""" + return select(score_all(surfaces, components), threshold=threshold, top_k=top_k) + + def invocability(registrations: list[dict[str, Any]]) -> dict[str, Any]: """Who can invoke a surface, from the fields its registrations carry. diff --git a/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py b/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py index d1273536f6..d7befd2cca 100755 --- a/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py +++ b/plugins/claude-ops/skills/audit-native-overlap/scripts/overlap.py @@ -1262,19 +1262,24 @@ def key_of(name: Any, component: dict[str, Any]) -> tuple[str, str, str, str]: corpus = load_components(repo) surfaces = native_surfaces(lane_payloads) - scored = ( - discover.discover(surfaces, corpus, threshold=args.threshold, top_k=args.top_k) - if args.top_k > 0 - else [] + + def keyed(scored: list[discover.Scored]) -> dict[tuple[str, str, str, str], Any]: + return { + key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}): ( + s, + score, + matched, + ) + for s, c, score, matched in scored + } + + # Seeds read their score from every scored pair, so a seed below the + # discovery cut still shows how far lexical evidence alone would carry it. + all_scored = discover.score_all(surfaces, corpus) + seed_scores = keyed(all_scored) + scores = keyed( + discover.select(all_scored, threshold=args.threshold, top_k=args.top_k) ) - scores = { - key_of(s.name, {"plugin": c.plugin, "skill": c.name, "kind": c.kind}): ( - s, - score, - matched, - ) - for s, c, score, matched in scored - } candidates: list[dict[str, Any]] = [] for pair in pairs_data["pairs"]: # shape guaranteed by validate_pairs above @@ -1311,7 +1316,8 @@ def key_of(name: Any, component: dict[str, Any]) -> tuple[str, str, str, str]: if pair.get("why"): evidence.append(f"seeded rationale: {pair['why']}") key = key_of(native.get("name"), component) - _surface, score, matched = scores.pop(key, (None, None, None)) + scores.pop(key, None) + _surface, score, matched = seed_scores.get(key, (None, None, None)) klass = (seen or {}).get("class", native.get("class")) block = native_block(native.get("name"), klass, seen) block["seeded_class"] = native.get("class") From 194457d5434d3e559d4d4e1d7fda3882ebbdc03f Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:06:43 -0400 Subject: [PATCH 04/22] docs(claude-ops): describe seeded and discovered overlap candidates The report structure now shows each candidate's origin, score, invocable_by and recommended integration, and the detection posture states how discovery scores and where seeds still earn their place. The plugin_backed lane and code-review alias gotchas are re-verified against Claude Code 2.1.284. Co-Authored-By: Claude Opus 5.5 --- .../skills/audit-native-overlap/SKILL.md | 57 ++++++++++++++----- 1 file changed, 44 insertions(+), 13 deletions(-) diff --git a/plugins/claude-ops/skills/audit-native-overlap/SKILL.md b/plugins/claude-ops/skills/audit-native-overlap/SKILL.md index b41bd5e4c8..fabdd301fb 100644 --- a/plugins/claude-ops/skills/audit-native-overlap/SKILL.md +++ b/plugins/claude-ops/skills/audit-native-overlap/SKILL.md @@ -84,7 +84,10 @@ python3 "${CLAUDE_PLUGIN_ROOT}/skills/audit-native-overlap/scripts/overlap.py" s ``` `--repo`, `--store`, `--view`, and `--pairs` are flags with repo-relative defaults, so a consumer -repository with a different layout points them wherever its files live. +repository with a different layout points them wherever its files live. `detect` also takes +`--threshold` (lowest discovery score kept, default `0.30`) and `--top-k` (most components kept per +native surface, default `3`; `0` turns discovery off). Lower the threshold to `0.25` for a +recall-first sweep; below that, most added pairs share one incidental word. Every subcommand exits `0` ok, `1` broken, `3` degraded, the sibling extractor's contract, not the shell gates' `0/1/2`, so one lane can carry both; `2` stays argparse's usage error. A degraded exit @@ -94,10 +97,35 @@ for the repo's test discovery. ## Detection posture. Floor-honest -Under-recall stated honestly beats confident completeness. Three rules: +Under-recall stated honestly beats confident completeness. Candidates come from two origins: + +- **Seeded**: every pair in `reference/canonical-pairs.json`, emitted whether or not the extraction + shows the native side. +- **Discovered**: every native surface in the extraction, internal ones excepted, scored against + every skill and agent in the repo from name, alias, and description tokens + (`scripts/discover.py` states the formula). A pair at or over `--threshold`, among the `--top-k` + best for its surface, is emitted with its score and the shared tokens as evidence. A pair the + store already records is listed under `discovery.existing` with its verdict instead; a pair that + is also seeded stays one seeded candidate carrying the score. + +Discovery is lexical. A one-line native description rarely shares words with the component it +duplicates in concept (`recap` against `session-flow:orient`), so such a pair belongs in the seeds +once a human confirms it; a pair discovery already finds needs no seed. + +Every candidate carries `native.invocable_by` (`model+user`, `user-only`, `model-only`, or +`unknown`), read from the registration's `model_invocable` and `user_invocable` fields, or from +`disable_model_invocation` in an older extraction; a field the extraction lacks makes it `unknown`. +`model_invocable: false` also sets the `model-invocation-disabled` marker the store's suggest-only +rule reads. From it comes `recommended_integration`: `suggest` for a user-only surface, `route` or +`route-or-wrap` for a model-invocable one (`route` for a built-in command, which never takes +`wrap`), and nothing when unknown. It is a label for the human writing the row, never a store value. + +Three rules: - **Carry the integrity floor through, per lane.** The inventory reports integrity per lane - (`builtin_commands`, `bundled_skills`, `plugin_backed`). A `degraded` lane makes every count from + (`builtin_commands`, `bundled_skills`, `plugin_backed`, and `bundled_workflows` when the + extraction has that lane; an extraction without it is not an error, the lane is simply not + reported). A `degraded` lane makes every count from that lane a floor, and the report says so in the same sentence as the number. A `broken` lane's counts are omitted, the report names the lane and its cause, and every candidate whose lane is broken is marked `re_derivable: false` (its presence or absence in that lane proves nothing @@ -118,9 +146,11 @@ Inventory status per lane (ok | degraded | broken), cli_version vs validated_aga that means for every count below; a broken lane is named with its cause. ## Overlap candidates -One row per (native surface, our component): native name + provenance class + hidden/gated -markers, our component, the evidence, and the store's current verdict — or NEW where the -store has no row yet. +One row per (native surface, our component): origin (seeded | discovered), native name + +provenance class + hidden/gated markers, invocable_by, our component, score and shared tokens, +recommended integration (a label, not a verdict), the evidence, and the store's current verdict, +or NEW where the store has no row yet. Discovered pairs the store already records follow as one +line each with their verdict. ## Registry state Rows whose recheck trigger has fired, rows missing a baked line, rows baked but unverified. @@ -133,8 +163,8 @@ to name-only degradation. Which substrate produced which section, and anything the run could not resolve. ``` -Provenance classes are never merged into one list. A bundled skill, a built-in command, a -plugin-backed built-in, and a session-provided skill have different disable switches and different +Provenance classes are never merged into one list. A bundled skill, a bundled workflow, a built-in +command, a plugin-backed built-in, and a session-provided skill have different disable switches and different rosters per host; a merged list cannot be acted on. ## Budget exposure, a presence-gated seam @@ -308,14 +338,15 @@ Two upstream facts this skill depends on, each with the trigger that obliges re- - **A plugin skill never shadows a native one.** Ours are namespaced, so both resolve and the model chooses. That is why the routing lives in descriptions rather than in a name. - **`plugin_backed` is its own lane.** `security-review` is reported there, not under - `builtin_commands`. Read the wrong key and the row looks absent. Verified 2026-09-06 against - Claude Code 2.1.263, by running `inventory.py --binary-only` on this machine and reading the - `plugin_backed` key, which holds `security-review` and nothing else. Recheck when the extractor's - provenance lanes change or a release note moves a bundled surface between them. + `builtin_commands`. Read the wrong key and the row looks absent. Verified 2026-09-29 against + Claude Code 2.1.284, by reading the `plugin_backed` key of an `inventory.py --binary-only` + extraction on this machine, which holds `security-review` and nothing else. Recheck when the + extractor's provenance lanes change or a release note moves a bundled surface between them. - **A bundled skill can carry aliases.** `code-review` answers to `review`; treating an alias as a separate surface produces a duplicate row for one capability. Basis: gives `/code-review` the line "Alias: `/review`". - Verified 2026-09-06 against Claude Code 2.1.263 and that page as fetched that day. Recheck when + Verified 2026-09-29 against Claude Code 2.1.284 (the extraction lists `review` as the alias) and + that page as fetched that day. Recheck when the commands page drops the alias line or a release note renames a bundled skill. - **Absent from the binary is not absent from the product.** Session-provided skills exist only in a live roster. "Not in the extraction" is a statement about the extraction. From 4ffcf8c7003c624ec3319e6609e09393f70e7aae Mon Sep 17 00:00:00 2001 From: Kyle Sexton <153232337+kyle-sexton@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:17:06 -0400 Subject: [PATCH 05/22] feat(claude-ops): resolve built-in hints, invocability, workflows, docs cross-check The inventory now emits argument_hint and description resolved from getters, constants, function references, and concatenations; user_invocable and model_invocable on every command and bundled skill, null when the bundle decides at runtime; a bundled_workflows lane with a deep-research canary; and a --docs mode that classifies each name against the commands page and attaches changelog history as a labeled heuristic. Co-Authored-By: Claude Opus 5.5 --- .../inventory/scripts/docs_crosscheck.py | 405 +++++++ .../skills/inventory/scripts/inventory.py | 1006 ++++++++++++++++- .../inventory/scripts/test_inventory.py | 364 ++++++ 3 files changed, 1741 insertions(+), 34 deletions(-) create mode 100644 plugins/claude-ops/skills/inventory/scripts/docs_crosscheck.py diff --git a/plugins/claude-ops/skills/inventory/scripts/docs_crosscheck.py b/plugins/claude-ops/skills/inventory/scripts/docs_crosscheck.py new file mode 100644 index 0000000000..41bfd46ed1 --- /dev/null +++ b/plugins/claude-ops/skills/inventory/scripts/docs_crosscheck.py @@ -0,0 +1,405 @@ +"""Classify the extracted built-in surface against the official command docs. + +A separate lane from the binary extraction: the binary says what exists, the +docs say what is documented, and this module only compares the two. It never +adds a name to or removes one from a binary lane, and a fetch failure degrades +only its own block. + +Python 3.11+, standard library only. +""" + +from __future__ import annotations + +import re +import sys +import urllib.error +import urllib.request +from pathlib import Path +from typing import Any + +_LIB_DIR = Path(__file__).resolve().parents[3] / "lib" +if str(_LIB_DIR) not in sys.path: + sys.path.insert(0, str(_LIB_DIR)) + +from registrations import registrations_of # noqa: E402 (path set above) + +COMMANDS_URL = "https://code.claude.com/docs/en/commands.md" +CHANGELOG_URL = ( + "https://raw.githubusercontent.com/anthropics/claude-code/main/CHANGELOG.md" +) + +STATUSES = ( + "documented", + "undocumented", + "alias", + "docs_alias_but_registered", + "removed_in_docs", + "removed_in_docs_but_registered", + "docs_only", +) + +_SECTION = "## All commands" +_ROW_RE = re.compile( + r"^\|\s*`/(?P[a-z0-9][a-z0-9:_-]*)(?P[^`]*)`\s*\|\s*(?P.*?)\s*\|\s*$" +) +_KIND_RE = re.compile(r"^\*\*\[?(Skill|Workflow)\]?(?:\([^)]*\))?\.?\*\*\.?\s*") +_ALIAS_OF_RE = re.compile(r"^Alias (?:for|of) \[?`/([a-z0-9:_-]+)") +_TOKENS = r"((?:`/[a-z0-9:_-]+`(?:,\s*|\s+and\s+|,\s*and\s+)?)+)" +_ALIAS_LIST_RE = re.compile(r"\bAlias(?:es)?:\s*" + _TOKENS) +_ARE_ALIASES_RE = re.compile(_TOKENS + r"\s+are aliases\b") +_IS_ALIAS_RE = re.compile(r"`/([a-z0-9:_-]+)` is an alias\b") +_REMOVED_RE = re.compile(r"^Removed(?: in v?(\d+\.\d+\.\d+))?\b") +_LINK_RE = re.compile(r"\[([^\]]*)\]\([^)]*\)") + +_VERSION_RE = re.compile(r"^##\s+\[?v?(\d+\.\d+\.\d+)") +_EVENT_WORDS = ( + ("added", re.compile(r"\b(?:added|adds|introduc(?:ed|es))\b", re.I)), + ("renamed", re.compile(r"\brenam(?:ed|es)\b", re.I)), + ("removed", re.compile(r"\bremov(?:ed|es)\b", re.I)), + ("deprecated", re.compile(r"\bdeprecat(?:ed|es|ion)\b", re.I)), + ("alias", re.compile(r"\balias(?:es|ed)?\b", re.I)), +) +_EVENT_TEXT_MAX = 240 + + +def fetch_text(url: str, timeout: float = 20.0) -> tuple[str | None, str | None]: + """The body at `url`, or None and the reason. Never raises.""" + req = urllib.request.Request(url, headers={"User-Agent": "claude-ops-inventory"}) + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + return resp.read().decode("utf-8", "replace"), None + except (urllib.error.URLError, TimeoutError, OSError, ValueError) as exc: + return None, f"{type(exc).__name__}: {exc}" + + +def _slashes(tokens: str) -> list[str]: + return re.findall(r"`/([a-z0-9:_-]+)`", tokens) + + +def parse_commands_table(text: str) -> dict[str, dict[str, Any]]: + """Rows of the commands page's "All commands" table, keyed by name. + + Each row records its argument synopsis, kind (command, skill, workflow), + whether the row itself is an alias (and of what), the aliases it declares, + and whether it is marked removed. "Removed" counts only at the start of + the row text: a row that says skills were "added or removed" is not. + """ + start = text.find(_SECTION) + if start < 0: + return {} + end = text.find("\n## ", start + len(_SECTION)) + rows: dict[str, dict[str, Any]] = {} + for line in text[start : end if end > 0 else len(text)].splitlines(): + m = _ROW_RE.match(line) + if not m: + continue + name = m.group("name") + body = m.group("text").replace("\\|", "|") + kind_m = _KIND_RE.match(body) + kind = kind_m.group(1).lower() if kind_m else "command" + body = body[kind_m.end() :] if kind_m else body + row: dict[str, Any] = { + "args": m.group("args").replace("\\|", "|").strip() or None, + "kind": kind, + "alias_of": None, + "is_alias": False, + "aliases": [], + "removed": False, + "removed_version": None, + "summary": _LINK_RE.sub(r"\1", body).split(". ")[0].strip()[:300], + } + alias_of = _ALIAS_OF_RE.match(body) + if alias_of: + row["is_alias"], row["alias_of"] = True, alias_of.group(1) + removed = _REMOVED_RE.match(body) + if removed: + row["removed"], row["removed_version"] = True, removed.group(1) + declared: list[str] = [] + for pattern in (_ALIAS_LIST_RE, _ARE_ALIASES_RE): + for hit in pattern.finditer(body): + declared += _slashes(hit.group(1)) + for hit in _IS_ALIAS_RE.finditer(body): + if hit.group(1) == name: + row["is_alias"] = True + else: + declared.append(hit.group(1)) + row["aliases"] = sorted(set(declared) - {name}) + rows[name] = row + return rows + + +def _vkey(version: str) -> tuple[int, ...]: + return tuple(int(p) for p in version.split(".")) + + +def parse_changelog(text: str, names: set[str]) -> dict[str, dict[str, Any]]: + """Per name: the earliest version whose entry mentions `/name`, and the + lines that also use an add, rename, remove, deprecate, or alias word. + + A heuristic: a mention is not proof of introduction, and a word match is + not proof the line is about that change. + """ + if not names: + return {} + token = re.compile( + r"(? tuple[dict[str, dict[str, Any]], dict[str, str]]: + """Registered names with their lane facts, and each binary alias's owner.""" + names: dict[str, dict[str, Any]] = {} + alias_owner: dict[str, str] = {} + lanes = ( + ("builtin_commands", "command"), + ("bundled_skills", "skill"), + ("bundled_workflows", "workflow"), + ) + for lane, kind in lanes: + for name, entry in (report.get(lane) or {}).items(): + regs = registrations_of(entry) + if not regs: + continue + aliases = sorted({a for r in regs for a in r.get("aliases") or []}) + names[name] = { + "kind": kind, + "aliases": aliases, + "internal": all(r.get("internal") for r in regs), + "hidden": any(r.get("hidden") for r in regs), + "user_invocable": regs[0].get("user_invocable"), + "model_invocable": regs[0].get("model_invocable"), + } + for a in aliases: + alias_owner.setdefault(a, name) + for name in report.get("plugin_backed") or {}: + names.setdefault( + name, + { + "kind": "command", + "aliases": [], + "internal": False, + "hidden": False, + "plugin_backed": True, + }, + ) + return names, alias_owner + + +def classify( + report: dict[str, Any], + rows: dict[str, dict[str, Any]], + changelog: dict[str, dict[str, Any]] | None = None, +) -> dict[str, dict[str, Any]]: + """One entry per name the binary registers or the docs list. + + An internal registration (never user-typed) counts as registered when the + docs name it, and is otherwise left out rather than listed undocumented. + """ + binary, alias_owner = _binary_index(report) + docs_aliases: dict[str, str] = {} + for name, row in rows.items(): + for a in row["aliases"]: + docs_aliases.setdefault(a, name) + if row["alias_of"]: + docs_aliases.setdefault(name, row["alias_of"]) + + out: dict[str, dict[str, Any]] = {} + for name in sorted(set(binary) | set(rows) | set(docs_aliases)): + row = rows.get(name) + reg = binary.get(name) + if reg and reg["internal"] and row is None and name not in docs_aliases: + continue + bin_alias_of = alias_owner.get(name) + docs_alias_of = ( + docs_aliases.get(name) if not (row and not row["is_alias"]) else None + ) + if row and row["removed"]: + registered = reg is not None or bin_alias_of is not None + status = ( + "removed_in_docs_but_registered" if registered else "removed_in_docs" + ) + elif (row and row["is_alias"]) or (row is None and name in docs_aliases): + if reg is not None: + status = "docs_alias_but_registered" + elif bin_alias_of is not None: + status = "alias" + else: + status = "docs_only" + elif row is not None: + status = "documented" if reg is not None or bin_alias_of else "docs_only" + else: + status = "undocumented" + entry: dict[str, Any] = { + "status": status, + "binary_kind": reg["kind"] if reg else ("alias" if bin_alias_of else None), + "binary_alias_of": bin_alias_of, + "docs_kind": ( + "alias" + if (row and row["is_alias"]) or (row is None and name in docs_aliases) + else (row["kind"] if row else None) + ), + "docs_args": row["args"] if row else None, + "docs_alias_of": docs_alias_of, + "docs_summary": row["summary"] if row else None, + } + if row and row["removed_version"]: + entry["docs_removed_version"] = row["removed_version"] + if reg and reg.get("plugin_backed"): + entry["plugin_backed"] = True + entry["kind_mismatch"] = bool( + row and reg and not row["is_alias"] and row["kind"] != reg["kind"] + ) + if reg: + documented_aliases = set(row["aliases"]) if row else set() + documented_aliases |= { + a for a, owner in docs_aliases.items() if owner == name + } + only_docs = sorted(documented_aliases - set(reg["aliases"])) + only_binary = sorted(set(reg["aliases"]) - documented_aliases) + if row and (only_docs or only_binary): + entry["alias_disagreement"] = { + "docs_only": only_docs, + "binary_only": only_binary, + } + if docs_alias_of and bin_alias_of and docs_alias_of != bin_alias_of: + entry["alias_target_disagreement"] = { + "docs": docs_alias_of, + "binary": bin_alias_of, + } + entry["changelog"] = changelog.get(name) if changelog is not None else None + out[name] = entry + return out + + +def build_crosscheck( + report: dict[str, Any], + docs_file: str | None = None, + changelog_file: str | None = None, +) -> dict[str, Any]: + """The `docs_crosscheck` block: statuses per name plus its own health.""" + block: dict[str, Any] = { + "status": "ok", + "problems": [], + "advisories": [], + "method": { + "source_of_truth": "the binary lanes decide what exists; the docs only classify it", + "statuses": list(STATUSES), + "changelog": "heuristic: earliest version whose entry mentions /name, and " + "lines using an add, rename, remove, deprecate, or alias word; a mention " + "is not proof of introduction", + }, + "sources": {}, + } + if not report.get("sources", {}).get("binary", {}).get("available"): + block["status"] = "unavailable" + block["problems"].append( + "the binary was not read; there is nothing to cross-check" + ) + return block + + if docs_file: + try: + docs_text, err = Path(docs_file).read_text(encoding="utf-8"), None + except OSError as exc: + docs_text, err = None, f"{type(exc).__name__}: {exc}" + block["sources"]["commands"] = {"file": docs_file} + else: + docs_text, err = fetch_text(COMMANDS_URL) + block["sources"]["commands"] = {"url": COMMANDS_URL} + if docs_text is None: + block["status"] = "unavailable" + block["sources"]["commands"]["error"] = err + block["problems"].append(f"commands page unavailable: {err}") + return block + + rows = parse_commands_table(docs_text) + block["sources"]["commands"]["rows"] = len(rows) + if not rows: + block["status"] = "broken" + block["problems"].append( + f"no rows parsed under '{_SECTION}' - the commands page layout changed" + ) + return block + + changelog: dict[str, dict[str, Any]] | None = None + if changelog_file: + try: + cl_text, cl_err = Path(changelog_file).read_text(encoding="utf-8"), None + except OSError as exc: + cl_text, cl_err = None, f"{type(exc).__name__}: {exc}" + block["sources"]["changelog"] = {"file": changelog_file} + else: + cl_text, cl_err = fetch_text(CHANGELOG_URL) + block["sources"]["changelog"] = {"url": CHANGELOG_URL} + if cl_text is None: + block["sources"]["changelog"]["error"] = cl_err + block["advisories"].append( + f"changelog unavailable, no version history attached: {cl_err}" + ) + else: + binary, _ = _binary_index(report) + names = ( + set(binary) | set(rows) | {a for r in rows.values() for a in r["aliases"]} + ) + changelog = parse_changelog(cl_text, names) + + names_block = classify(report, rows, changelog) + lanes = (report.get("integrity") or {}).get("lanes") or {} + unhealthy = sorted(lane for lane, e in lanes.items() if e.get("status") != "ok") + if unhealthy: + block["advisories"].append( + "binary lane(s) not ok: " + + ", ".join(unhealthy) + + " - an undocumented or docs_only status may reflect extraction, not the product" + ) + counts = {s: 0 for s in STATUSES} + for entry in names_block.values(): + counts[entry["status"]] += 1 + block["counts"] = counts + block["kind_mismatches"] = sorted( + n for n, e in names_block.items() if e["kind_mismatch"] + ) + block["alias_disagreements"] = sorted( + n + for n, e in names_block.items() + if "alias_disagreement" in e or "alias_target_disagreement" in e + ) + block["names"] = names_block + if block["advisories"]: + block["status"] = "degraded" + return block diff --git a/plugins/claude-ops/skills/inventory/scripts/inventory.py b/plugins/claude-ops/skills/inventory/scripts/inventory.py index 1e48047022..997f1eafff 100755 --- a/plugins/claude-ops/skills/inventory/scripts/inventory.py +++ b/plugins/claude-ops/skills/inventory/scripts/inventory.py @@ -41,6 +41,10 @@ # extractor writes and the consumer reads has exactly one home. from registrations import registrations_of # noqa: E402 (path set above; plugin-bundled module) +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from docs_crosscheck import build_crosscheck # noqa: E402 (sibling module) + MIN_PYTHON = (3, 11) # The CLI release this extractor was last verified against by a human running @@ -115,6 +119,18 @@ # Extraction lanes, in the order the self-check prints them. Each lane carries # its own status so one broken lane never silently voids the others' counts. LANES = ("builtin_commands", "bundled_skills", "plugin_backed") +# Evaluated whenever the binary is read; optional in `check_integrity` so a +# caller that extracts no workflows is not reported as a broken lane. +WORKFLOW_LANE = "bundled_workflows" + +# A bundled workflow that has shipped in every build since workflows gained a +# bundled roster. Absence means the workflow scan broke. +WORKFLOW_CANARY = ("deep-research",) + +# A description or argument hint held in a single-character identifier is +# trusted only when its binding lies this close to the registration: such names +# are function-local, and a farther binding belongs to another function. +SHORT_VALUE_LOCALITY_BYTES = 4_096 # Component types a plugin may ship, from the plugin manifest schema and the # standard plugin layout. Directory is the default location; the manifest may @@ -544,6 +560,573 @@ def _unescape(raw: str) -> str: return raw +# -------------------------------------------------------------------------- +# Static values: descriptions and argument hints +# -------------------------------------------------------------------------- + +_QUOTES = "\"'`" +_ELLIPSIS = "\u2026" +_ID_START = set("abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ_$") +_NON_STRING_WORDS = frozenset( + {"void", "null", "undefined", "true", "false", "typeof", "new", "await", "this"} +) +_JS_ESCAPES = {"n": "\n", "t": "\t", "r": "\r", "b": "\b", "f": "\f", "v": "\v"} +_ESCAPE_RE = re.compile(r"\\(u\{[0-9a-fA-F]+\}|u[0-9a-fA-F]{4}|x[0-9a-fA-F]{2}|[\s\S])") +_MAX_HOPS = 4 +_NONSTRING = object() + + +def _js_unescape(raw: str) -> str: + def sub(m: re.Match[str]) -> str: + e = m.group(1) + if e.startswith("u{"): + return chr(int(e[2:-1], 16)) + if e[0] in "ux" and len(e) > 1: + return chr(int(e[1:], 16)) + if e == "\n": + return "" + return _JS_ESCAPES.get(e, e) + + text = _ESCAPE_RE.sub(sub, raw) + return text.encode("utf-16", "surrogatepass").decode("utf-16", "replace") + + +def _read_literal(src: str, i: int, n: int) -> tuple[str, int, bool]: + """A string or template literal at `i`: (text, index past it, has substitution). + + A template substitution renders as an ellipsis: its value exists only at + runtime. Raises ValueError on an unterminated literal. + """ + q = src[i] + j = i + 1 + if q != "`": + while j < n: + ch = src[j] + if ch == "\\": + j += 2 + continue + if ch == q: + return _js_unescape(src[i + 1 : j]), j + 1, False + if ch == "\n": + break + j += 1 + raise ValueError("unterminated string") + parts: list[str] = [] + seg, subst = j, False + while j < n: + ch = src[j] + if ch == "\\": + j += 2 + continue + if ch == "`": + parts.append(_js_unescape(src[seg:j])) + return "".join(parts), j + 1, subst + if ch == "$" and src.startswith("{", j + 1): + parts.append(_js_unescape(src[seg:j]) + _ELLIPSIS) + subst = True + j = _skip_substitution(src, j + 2, n) + seg = j + continue + j += 1 + raise ValueError("unterminated template") + + +def _skip_substitution(src: str, i: int, n: int) -> int: + depth = 1 + while i < n: + ch = src[i] + if ch in _QUOTES: + i = _read_literal(src, i, n)[1] + continue + if ch == "{": + depth += 1 + elif ch == "}": + depth -= 1 + if depth == 0: + return i + 1 + i += 1 + raise ValueError("unterminated substitution") + + +def _skip_ws(src: str, i: int, n: int) -> int: + while i < n and src[i] in " \t\r\n": + i += 1 + return i + + +def _ident_end(src: str, i: int) -> int: + n = len(src) + while i < n and src[i] in _ID_CHARS: + i += 1 + return i + + +def _match_close(src: str, braces: BraceMap, i: int, n: int) -> int: + """Index past the `(` or `[` group opening at `i`.""" + depth = 0 + while i < n: + ch = src[i] + if ch in _QUOTES: + i = _read_literal(src, i, n)[1] + continue + if ch == "{": + close = braces.pairs.get(i) + if close is None: + raise ValueError("unmatched brace") + i = close + 1 + continue + if ch in "([": + depth += 1 + elif ch in ")]": + depth -= 1 + if depth == 0: + return i + 1 + i += 1 + raise ValueError("unterminated group") + + +@dataclass +class _Values: + """String values an expression can take, collected in source order.""" + + variants: list[str] = field(default_factory=list) + unresolved: int = 0 + via: set[str] = field(default_factory=set) + + def add(self, value: str) -> None: + if value in self.variants: + self.variants.remove(value) + self.variants.append(value) + + +def _object_fields( + src: str, braces: BraceMap, open_i: int +) -> dict[str, tuple[str, int]]: + """Top-level fields of the object literal at `open_i`. + + Maps each key to ("value", offset of its value) or ("getter", offset of + the getter body's `{`). Nested objects, strings, and groups are skipped, + so a nested object's field never answers for the outer one. + """ + close = braces.pairs.get(open_i) + if close is None: + return {} + fields: dict[str, tuple[str, int]] = {} + head = re.compile(r"(?:(get|set|async)\s+)?(" + _IDENT + r")\s*(:|\()") + i, expect_key = open_i + 1, True + while i < close: + ch = src[i] + if ch in " \t\r\n": + i += 1 + continue + if ch == ",": + expect_key = True + i += 1 + continue + if expect_key and ch in _ID_START: + m = head.match(src, i, close) + if m: + key = m.group(2) + if m.group(3) == ":" and not m.group(1): + fields.setdefault(key, ("value", m.end())) + i, expect_key = m.end(), False + continue + j = _skip_ws(src, _match_close(src, braces, m.end() - 1, close), close) + if m.group(1) == "get" and src.startswith("{", j): + fields.setdefault(key, ("getter", j)) + i, expect_key = j, False + continue + expect_key = False + if ch in _QUOTES: + i = _read_literal(src, i, close)[1] + elif ch == "{": + end = braces.pairs.get(i) + if end is None: + break + i = end + 1 + elif ch in "([": + i = _match_close(src, braces, i, close) + else: + i += 1 + return fields + + +def _scan( + src: str, + braces: BraceMap, + i: int, + end: int, + acc: _Values, + *, + block: bool, + hops: int, + anchor: int | None, +) -> int: + """Collect the string values an expression (or a function body) yields. + + Expression mode reads one expression from `i`, stopping at a top-level + `,`, `;`, or closing bracket. Block mode reads a function body and + collects what each `return` yields. Within a yielded expression, the + operands in value position (the start, and after a ternary `?` or `:`) + are the candidates; an operand followed by `?` is a condition, not a + value. + """ + active = at_value = not block + depth = 0 + prev, prev_word = "", "" + n = min(end, len(src)) + while i < n: + c = src[i] + if c in " \t\r\n": + i += 1 + continue + if ( + active + and at_value + and depth == 0 + and (c in _QUOTES or c in _ID_START or c == "(") + ): + i = _operand(src, braces, i, n, acc, hops=hops, anchor=anchor) + at_value, prev, prev_word = False, "x", "" + continue + if c in _QUOTES: + i = _read_literal(src, i, n)[1] + at_value, prev = False, c + continue + if c == "{": + close = braces.pairs.get(i) + if close is None: + raise ValueError("unmatched brace") + if ( + block + and depth == 0 + and (prev == ")" or prev_word in ("else", "try", "finally")) + ): + _scan( + src, braces, i + 1, close, acc, block=True, hops=hops, anchor=anchor + ) + i, at_value, prev, prev_word = close + 1, False, "}", "" + continue + if c == "}": + break + if c in "([": + depth += 1 + elif c in ")]": + if depth == 0: + break + depth -= 1 + elif depth == 0 and c in ",;": + if not block: + break + if c == ";": + active = False + elif depth == 0 and c == "?": + if src.startswith(("??", "?."), i): + i += 2 + at_value, prev = False, "?" + continue + at_value = active + i, prev = i + 1, c + continue + elif depth == 0 and c == ":": + at_value = active + i, prev = i + 1, c + continue + elif c in _ID_START: + j = _ident_end(src, i) + word = src[i:j] + if block and depth == 0 and word == "return": + active = at_value = True + else: + at_value = False + i, prev, prev_word = j, "x", word + continue + at_value, prev, prev_word = False, c, "" + i += 1 + return i + + +def _operand( + src: str, + braces: BraceMap, + i: int, + n: int, + acc: _Values, + *, + hops: int, + anchor: int | None, +) -> int: + """Read one operand in value position; record it when it is a string value.""" + if src[i] == "(": + close = _match_close(src, braces, i, n) + k = _skip_ws(src, close, n) + if not src.startswith("=>", k): + return close + return _arrow_body(src, braces, k + 2, n, acc, hops=hops, anchor=anchor) + parts: list[Any] = [] + while True: + i = _skip_ws(src, i, n) + if i >= n: + break + c = src[i] + if c in _QUOTES: + text, i, subst = _read_literal(src, i, n) + parts.append(text) + acc.via.add("template" if subst else "literal") + elif c in _ID_START: + j = _ident_end(src, i) + word = src[i:j] + k = _skip_ws(src, j, n) + if src.startswith("=>", k): + return _arrow_body(src, braces, k + 2, n, acc, hops=hops, anchor=anchor) + if word in _NON_STRING_WORDS: + i = _skip_ws(src, j, n) + if ( + word in ("void", "typeof", "new", "await") + and i < n + and src[i] in _ID_CHARS + ): + i = _ident_end(src, i) + parts.append(_NONSTRING) + else: + chain, i = _read_chain(src, braces, i, n) + parts.append( + _resolve_chain(src, braces, chain, i, acc, hops=hops, anchor=anchor) + ) + elif c == "!" or c.isdigit(): + i = _ident_end(src, i + 1) + parts.append(_NONSTRING) + else: + break + k = _skip_ws(src, i, n) + if src.startswith("+", k) and not src.startswith(("++", "+="), k): + i = k + 1 + continue + i = k + break + t = src[i] if i < n else "" + if t == "?" and not src.startswith(("??", "?."), i): + return i + if t and t not in ":,;})]": + return i + if len(parts) == 1: + p = parts[0] + if p is None: + acc.unresolved += 1 + elif isinstance(p, list): + for v in p: + acc.add(v) + elif isinstance(p, str): + acc.add(p) + elif len(parts) > 1 and any(isinstance(p, str) for p in parts): + acc.add( + "".join( + p + if isinstance(p, str) + else (p[-1] if isinstance(p, list) and p else _ELLIPSIS) + for p in parts + ) + ) + return i + + +def _arrow_body( + src: str, + braces: BraceMap, + k: int, + n: int, + acc: _Values, + *, + hops: int, + anchor: int | None, +) -> int: + acc.via.add("arrow") + k = _skip_ws(src, k, n) + if src.startswith("{", k): + close = braces.pairs.get(k) + if close is None: + raise ValueError("unmatched brace") + _scan(src, braces, k + 1, close, acc, block=True, hops=hops, anchor=anchor) + return close + 1 + return _scan(src, braces, k, n, acc, block=False, hops=hops, anchor=anchor) + + +def _read_chain( + src: str, braces: BraceMap, i: int, n: int +) -> tuple[list[tuple[str, str]], int]: + """An identifier with its member reads and calls: `a`, `f()`, `a.b`, `a.b()`.""" + j = _ident_end(src, i) + chain = [("id", src[i:j])] + i = j + while i < n: + if src.startswith("?.", i): + i += 1 + if src.startswith(".", i) and i + 1 < n and src[i + 1] in _ID_START: + j = _ident_end(src, i + 1) + chain.append(("prop", src[i + 1 : j])) + i = j + elif src.startswith("(", i): + j = _match_close(src, braces, i, n) + chain.append(("call", src[i + 1 : j - 1].strip())) + i = j + elif src.startswith("[", i): + j = _match_close(src, braces, i, n) + chain.append(("index", "")) + i = j + else: + break + return chain, i + + +def _binding_value(src: str, ident: str, at: int) -> int | None: + """Offset of the nearest `ident=` value before `at`, under the locality rule.""" + v = _nearest_binding(src, ident, at) + if v is None or (len(ident) == 1 and at - v > SHORT_VALUE_LOCALITY_BYTES): + return None + return v + + +def _function_body(src: str, braces: BraceMap, ident: str, at: int) -> int | None: + """The `{` of `function ident(){...}`: nearest before `at`, else first after.""" + if len(ident) == 1: + return None + pattern = re.compile(r"function\s+" + re.escape(ident) + r"\s*\(\s*\)\s*\{") + last = None + for m in pattern.finditer(src, 0, at): + last = m + if last is None: + last = pattern.search(src, at) + return None if last is None else last.end() - 1 + + +def _resolve_chain( + src: str, + braces: BraceMap, + chain: list[tuple[str, str]], + pos: int, + acc: _Values, + *, + hops: int, + anchor: int | None, +) -> list[str] | None: + """The string values a constant, a no-argument call, or a member read yields.""" + if hops <= 0: + return None + at = anchor if anchor is not None else pos + ident = chain[0][1] + sub = _Values() + # A bare identifier naming a function declaration is a function-valued + # field, which the registrars read through a getter: resolve it as a call + # when the declaration is nearer than any `ident=` binding. + v = _binding_value(src, ident, at) if len(chain) == 1 else None + fn_body = _function_body(src, braces, ident, at) if len(chain) == 1 else None + if fn_body is not None and v is not None and (fn_body > at or fn_body < v): + fn_body = None + if len(chain) == 1 and fn_body is None: + if v is None: + return None + _scan(src, braces, v, len(src), sub, block=False, hops=hops - 1, anchor=None) + acc.via.add("constant") + elif chain[1:] == [("call", "")] or fn_body is not None: + body = ( + fn_body if fn_body is not None else _function_body(src, braces, ident, at) + ) + close = None if body is None else braces.pairs.get(body) + if close is None: + return None + _scan(src, braces, body + 1, close, sub, block=True, hops=hops - 1, anchor=None) + acc.via.add("call") + elif len(chain) == 2 and chain[1][0] == "prop": + if len(ident) == 1: + return None + obj = _resolve_object(src, braces, ident, at) + if obj is None: + return None + found = _eval_field( + src, braces, obj[0], chain[1][1], sub, hops=hops - 1, anchor=None + ) + if found is None: + return None + acc.via.add("constant") + else: + return None + acc.via |= sub.via + return sub.variants or None + + +def _eval_field( + src: str, + braces: BraceMap, + open_i: int, + key: str, + acc: _Values, + *, + hops: int = _MAX_HOPS, + anchor: int | None = None, +) -> str | None: + """Evaluate one top-level field into `acc`; returns its form, or None when absent.""" + entry = _object_fields(src, braces, open_i).get(key) + if entry is None: + return None + kind, pos = entry + if kind == "getter": + close = braces.pairs.get(pos) + if close is None: + return "getter" + _scan(src, braces, pos + 1, close, acc, block=True, hops=hops, anchor=anchor) + return "getter" + close = braces.pairs.get(open_i, len(src)) + _scan(src, braces, pos, close, acc, block=False, hops=hops, anchor=anchor) + return "value" + + +def resolve_field( + src: str, braces: BraceMap, open_i: int, key: str, anchor: int | None = None +) -> dict[str, Any] | None: + """Statically resolve one string field of an object literal. + + Returns None when the object has no such field. Otherwise `value` is the + fallthrough variant (the else branch of a ternary, the last `return` of a + getter), which is what a default session shows; `variants` lists every + alternative when there is more than one; `source` names the form: + literal, template (a runtime substitution rendered as an ellipsis), + constant, call, getter, arrow, or unresolved. + """ + acc = _Values() + try: + form = _eval_field(src, braces, open_i, key, acc, anchor=anchor) + except (ValueError, IndexError, RecursionError): + form, acc = "value", _Values() + if form is None: + return None + if not acc.variants: + source = "unresolved" + elif form == "getter": + source = "getter" + else: + source = next( + (s for s in ("arrow", "call", "constant", "template") if s in acc.via), + "literal", + ) + out: dict[str, Any] = { + "value": acc.variants[-1] if acc.variants else None, + "source": source, + } + if len(acc.variants) > 1: + out["variants"] = acc.variants + return out + + +def _apply_field( + rec: dict[str, Any], key: str, resolved: dict[str, Any] | None +) -> None: + """Write a resolved field as ``, `_source`, and `_variants`.""" + rec[key] = resolved["value"] if resolved else None + rec[f"{key}_source"] = resolved["source"] if resolved else "absent" + if resolved and "variants" in resolved: + rec[f"{key}_variants"] = resolved["variants"] + + # -------------------------------------------------------------------------- # Extracting the registries # -------------------------------------------------------------------------- @@ -552,8 +1135,6 @@ def _unescape(raw: str) -> str: _NAME_RE = re.compile(r"(?:^|[,{])name:" + _STR) _NAME_IDENT_RE = re.compile(r"(?:^\{|,)name:([A-Za-z_$][A-Za-z0-9_$]*)(?=[,}])") _UFN_RE = re.compile(r"userFacingName\(\)\{return" + _STR) -_DESC_RE = re.compile(r"(?:^|[,{])description:" + _STR) -_MENUDESC_RE = re.compile(r"(?:menuDescription|description):" + _STR) _ALIAS_RE = re.compile(r"aliases:\[([^\]]*)\]") _NAME_OK = re.compile(r"[a-zA-Z0-9][a-zA-Z0-9:_-]{0,40}") @@ -611,18 +1192,26 @@ def extract_builtin_commands(src: str, braces: BraceMap) -> dict[str, dict[str, name = resolve_name_ident(c.group(1), open_i, index) if not name or not _NAME_OK.fullmatch(name): continue - desc = _DESC_RE.search(body) aliases = _read_aliases(body) rec = { "name": name, "source": "builtin", "type": m.group(1), - "description": _unescape(desc.group(1)) if desc else "", "aliases": aliases, "hidden": "isHidden" in body, "gated": "isEnabled" in body, "internal": name in INTERNAL_NAMES, } + _apply_field( + rec, "description", resolve_field(src, braces, open_i, "description") + ) + rec["description"] = rec["description"] or "" + _apply_field( + rec, "argument_hint", resolve_field(src, braces, open_i, "argumentHint") + ) + rec.update(read_invocation_fields(body)) + origin = resolve_field(src, braces, open_i, "source") + rec.update(command_invocability(rec, origin["value"] if origin else None)) prev = out.get(name) if prev is None or (not prev["description"] and rec["description"]): if prev is not None: @@ -633,6 +1222,47 @@ def extract_builtin_commands(src: str, braces: BraceMap) -> dict[str, dict[str, return out +def command_invocability(rec: dict[str, Any], origin: str | None) -> dict[str, Any]: + """Who can invoke a built-in command, as far as the bundle decides it. + + The Skill tool admits a command only when it is `type:"prompt"`, not + `disableModelInvocation`, and (for a built-in) declares `source:"builtin"`; + `local` and `local-jsx` commands never reach it. A function-valued field + is decided at runtime, so it reads null rather than a guess. A per-machine + skill override in settings can still turn a command off; that is config, + not the bundle, and is not read here. + """ + flag_driven = set(rec.get("flag_driven") or []) + if "user_invocable" in flag_driven: + user: bool | None = None + else: + user = rec.get("user_invocable", True) + if rec["type"] != "prompt": + model: bool | None = False + elif "disable_model_invocation" in flag_driven: + model = None + elif rec.get("disable_model_invocation"): + model = False + elif origin == "builtin": + model = True + else: + model = None + return {"user_invocable": user, "model_invocable": model} + + +def skill_invocability(rec: dict[str, Any]) -> dict[str, Any]: + """Who can invoke a bundled skill: `userInvocable` defaults true and + `disableModelInvocation` false in the bundled registrar; a function-valued + field becomes a runtime getter and reads null.""" + flag_driven = set(rec.get("flag_driven") or []) + user = None if "user_invocable" in flag_driven else rec.get("user_invocable", True) + if "disable_model_invocation" in flag_driven: + model = None + else: + model = not rec.get("disable_model_invocation", False) + return {"user_invocable": user, "model_invocable": model} + + def _read_aliases(body: str) -> list[str]: m = _ALIAS_RE.search(body) if not m: @@ -781,12 +1411,12 @@ def _resolve_object( def _resolve_descriptor_name( src: str, braces: BraceMap, ident: str, at: int -) -> tuple[str, str] | None: +) -> tuple[str, int] | None: """Resolve `name:x.name`: the registration reads its fields from a descriptor. `let t=c;registrar({name:t.name,description:t.description,...})` where `c={name:o,...}` and `o="slides"`. Returns the name and the descriptor's - text, whose fields stand in for the registration's member reads. + offset, whose fields stand in for the registration's member reads. """ obj = _resolve_object(src, braces, ident, at) if obj is None: @@ -796,10 +1426,10 @@ def _resolve_descriptor_name( if not m: return None if m.group(1) is not None: - return _unescape(m.group(1)), body + return _unescape(m.group(1)), open_i v = _nearest_binding(src, m.group(2), open_i) lit = re.match(_STR, src[v : v + 256]) if v is not None else None - return (_unescape(lit.group(1)), body) if lit else None + return (_unescape(lit.group(1)), open_i) if lit else None _ROSTER_HEAD_RE = re.compile( @@ -932,7 +1562,7 @@ def extract_bundled_skills( # object carries no `name:` is another module's function that happens to # share the minified identifier, not a registration, so it is counted # apart and never inflates the resolved-versus-seen gap. - calls: list[tuple[int, str, re.Match[str]]] = [] + calls: list[tuple[int, int, str, re.Match[str]]] = [] unbounded = 0 same_ident_calls = 0 name_re = re.compile( @@ -953,9 +1583,9 @@ def extract_bundled_skills( if not nm: same_ident_calls += 1 continue - calls.append((m.start(), body, nm)) + calls.append((m.start(), open_i, body, nm)) - idents = {nm.group(2) for _, _, nm in calls if nm.group(2) and not nm.group(3)} + idents = {nm.group(2) for *_, nm in calls if nm.group(2) and not nm.group(3)} index = build_const_index(src, idents) out: dict[str, Any] = {} unresolved: list[str] = [] @@ -983,9 +1613,9 @@ def add(rec: dict[str, Any]) -> None: if name not in collisions: collisions.append(name) - for call_start, body, nm in calls: + for call_start, open_i, body, nm in calls: seen += 1 - descriptor = "" + descriptor: int | None = None if nm.group(1) is not None: name = _unescape(nm.group(1)) elif nm.group(4) is not None or _is_loop_registration( @@ -995,7 +1625,7 @@ def add(rec: dict[str, Any]) -> None: # family, enumerable statically only when the loop walks a # literal table. roster = _resolve_roster(src, braces, call_start) - rows = _roster_records(body, nm, roster) if roster else None + rows = _roster_records(src, braces, open_i, nm, roster) if roster else None if roster is None or rows is None: dynamic_rosters += 1 dynamic_patterns.append( @@ -1020,7 +1650,7 @@ def add(rec: dict[str, Any]) -> None: continue name = resolved resolved_calls += 1 - add(_skill_record(name, body, descriptor)) + add(_skill_record(src, braces, name, open_i, descriptor)) # Counted per call, not per row: a roster call is one registration however # many rows it expands to, so it can never mask an unresolved one. @@ -1042,34 +1672,66 @@ def add(rec: dict[str, Any]) -> None: return out, notes -def _skill_record(name: str, body: str, descriptor: str = "") -> dict[str, Any]: - """One bundled-skill row from its registration literal. +def _skill_record( + src: str, + braces: BraceMap, + name: str, + open_i: int, + descriptor: int | None = None, + row: dict[str, str | None] | None = None, +) -> dict[str, Any]: + """One bundled-skill row from its registration literal at `open_i`. - `descriptor` is the object a `name:x.name` registration reads its fields - from; a description the literal does not carry as a string is read there. + `descriptor` is the offset of the object a `name:x.name` registration + reads its fields from; a field the literal does not resolve is read + there. `row` maps a looped registration's variables to one table row's + values, which answer for a field that reads a loop variable. """ - desc = _MENUDESC_RE.search(body) or _MENUDESC_RE.search(descriptor) + close = braces.pairs[open_i] + body = src[open_i : close + 1] rec: dict[str, Any] = { "name": name, "source": "bundled-skill", - "description": _unescape(desc.group(1)) if desc else "", "aliases": _read_aliases(body), "gated": "isEnabled" in body, "hidden": "isHidden" in body, } + + def field_of(key: str) -> dict[str, Any] | None: + if row: + m = re.search(r"(?:^\{|,)" + key + r":(" + _IDENT + r")(?=[,}])", body) + if m and m.group(1) in row: + value = row[m.group(1)] + return {"value": value, "source": "roster"} if value else None + found = resolve_field(src, braces, open_i, key) + if (found is None or found["value"] is None) and descriptor is not None: + found = resolve_field(src, braces, descriptor, key) or found + return found + + desc = field_of("description") + menu = field_of("menuDescription") + if (desc is None or desc["value"] is None) and menu and menu["value"]: + desc = menu + _apply_field(rec, "description", desc) + rec["description"] = rec["description"] or "" + if menu and menu["value"]: + rec["menu_description"] = menu["value"] + _apply_field(rec, "argument_hint", field_of("argumentHint")) rec.update(read_invocation_fields(body)) + rec.update(skill_invocability(rec)) return rec def _roster_records( - body: str, + src: str, + braces: BraceMap, + open_i: int, nm: re.Match[str], roster: tuple[str, dict[str, str], list[str]], ) -> list[dict[str, Any]] | None: """Expand a looped registration over its table's rows; None if any row fails.""" _, var_to_key, rows = roster template = nm.group(4) - field = re.search(r"(?:menuDescription|description):(" + _IDENT + ")", body) out: list[dict[str, Any]] = [] for row in rows: values = {var: _row_field(row, key) for var, key in var_to_key.items()} @@ -1084,10 +1746,7 @@ def _roster_records( name = values.get(nm.group(2)) or "" if not _NAME_OK.fullmatch(name): return None - rec = _skill_record(name, body) - if not rec["description"] and field: - rec["description"] = values.get(field.group(1)) or "" - out.append(rec) + out.append(_skill_record(src, braces, name, open_i, row=values)) return out @@ -1096,9 +1755,14 @@ def _same_registration(a: dict[str, Any], b: dict[str, Any]) -> bool: # and a function-valued one read as the same boolean, and the difference # (decided at runtime versus fixed) is exactly the evidence a collision # exists to preserve. - keys = ("description", "aliases", "gated", "hidden", "flag_driven") + tuple( - k for k, _ in _INVOCATION_FIELDS - ) + keys = ( + "description", + "argument_hint", + "aliases", + "gated", + "hidden", + "flag_driven", + ) + tuple(k for k, _ in _INVOCATION_FIELDS) return all(a.get(k) == b.get(k) for k in keys) @@ -1115,6 +1779,198 @@ def extract_plugin_backed(src: str) -> dict[str, str]: return out +_WORKFLOW_PUSH = ".bundledWorkflows.push(" +_FUNC_HEAD_RE = re.compile(r"function\s+(" + _IDENT + r")\s*\(([^()]*)\)\s*\{") + + +def _split_args(src: str, braces: BraceMap, i: int) -> list[int]: + """Start offsets of each top-level argument of the call whose `(` ends at `i`.""" + n = len(src) + starts = [_skip_ws(src, i, n)] + depth = 0 + while i < n: + ch = src[i] + if ch in _QUOTES: + i = _read_literal(src, i, n)[1] + continue + if ch == "{": + close = braces.pairs.get(i) + if close is None: + raise ValueError("unmatched brace") + i = close + 1 + continue + if ch in "([": + depth += 1 + elif ch in ")]": + if depth == 0: + return starts + depth -= 1 + elif ch == "," and depth == 0: + starts.append(_skip_ws(src, i + 1, n)) + i += 1 + raise ValueError("unterminated call") + + +def _workflow_registrars(src: str) -> dict[str, tuple[int | None, int | None]]: + """Each function that pushes onto `bundledWorkflows`, with its argument layout. + + The registrar has no readable export name, so it is found by what it + does: `function f(script,meta,opts){...bundledWorkflows.push({...meta, + script, disableModelInvocation: opts?.disableModelInvocation})}`. Returns + the index of the spread (meta) argument and of the options argument. + """ + out: dict[str, tuple[int | None, int | None]] = {} + for m in re.finditer(re.escape(_WORKFLOW_PUSH), src): + heads = list(_FUNC_HEAD_RE.finditer(src, max(0, m.start() - 400), m.start())) + if not heads: + continue + head = heads[-1] + params = [p.strip() for p in head.group(2).split(",")] + tail = src[m.end() : m.end() + 400] + spread = re.search(r"\.\.\.(" + _IDENT + r")", tail) + opts = re.search(r"(" + _IDENT + r")\?\.disableModelInvocation", tail) + out[head.group(1)] = ( + params.index(spread.group(1)) + if spread and spread.group(1) in params + else None, + params.index(opts.group(1)) if opts and opts.group(1) in params else None, + ) + return out + + +def _array_titles(src: str, braces: BraceMap, ident: str, at: int) -> list[str] | None: + """`title` of each row of the array literal an identifier is bound to.""" + v = _binding_value(src, ident, at) + if v is None or not src.startswith("[", v): + return None + titles: list[str] = [] + i = v + 1 + while True: + i = _skip_ws(src, i, len(src)) + if src.startswith(",", i): + i += 1 + continue + if src.startswith("]", i): + return titles + close = braces.pairs.get(i) if src.startswith("{", i) else None + if close is None: + return None + title = resolve_field(src, braces, i, "title") + if title and title["value"]: + titles.append(title["value"]) + i = close + 1 + + +def extract_bundled_workflows( + src: str, braces: BraceMap +) -> tuple[dict[str, dict[str, Any]], dict[str, Any]]: + """Bundled workflows (`/deep-research`), keyed by name, plus resolution notes. + + A registration is `registrar(\\`