feat(gate): P17 keeps the engine tool surface out of skill content
The first-party agent-tool layer is Experimental and its surface moves between builds. A recipe that names a toolset does not fail loudly on the next build -- the tool is simply absent, the search finds nothing, and "nothing found" reads as "no problem here". That is the exact confusion this bundle documents, so the tokens are barred rather than discouraged. Grounded in measurement against a live editor, not in reading: - the aggregator plugin lists 21 dependencies; the server reported 53 registered toolsets from 22 plugins, one dependency contributing none; - the visible tool count flips between 3 meta-tools and every tool registered natively, on one project setting. - gate.py: ENGINE_TOOL_TOKENS and check_engine_tool_surface, registered in CHECKS; rule floor raised to 17. - test_gate.py: one poison naming a toolset, a meta-tool and a host:port. - ADR-0004 records the decision, its confirmation criteria and what is deliberately deferred. - README and CONTRIBUTING state the rule where a contributor meets it. Verified: gate.py 0 violations over 17 rules; test_gate.py 17/17 redden on their fixtures, baseline clean, surface coverage intact. The token list matches nothing under skills/ today, checked before the rule landed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -49,6 +49,22 @@ HARNESS_TOKENS = (
|
||||
r'(?<![A-Za-z0-9_./-])\.claude/',
|
||||
)
|
||||
|
||||
# Engine tool surface. The first-party agent-tool layer shipped with the editor
|
||||
# is Experimental and says so: "APIs and data formats are subject to change at
|
||||
# any time". A recipe that names a toolset, a meta-tool or an endpoint dates
|
||||
# itself to one engine build and fails silently on the next -- the tool is
|
||||
# simply absent, which reads as "nothing found here" rather than as an error.
|
||||
# Measured surface drift is why this is a rule and not advice: one editor
|
||||
# session exposed 53 toolsets from 21 aggregated plugins, and the visible tool
|
||||
# count flips between 3 and several hundred on a single project setting.
|
||||
ENGINE_TOOL_TOKENS = (
|
||||
r'\b(?:list_toolsets|describe_toolset|call_tool)\b',
|
||||
r'\b[A-Za-z][A-Za-z0-9_]*Toolsets?\b',
|
||||
r'\bAgentSkill[A-Za-z]*\b',
|
||||
r'\bModelContextProtocol[A-Za-z]*\b',
|
||||
r'\b(?:127\.0\.0\.1|localhost)\s*:\s*\d+',
|
||||
)
|
||||
|
||||
TEXT_SUFFIXES = {'.md', '.json', '.txt', '.yaml', '.yml'}
|
||||
JUNK_SUFFIXES = {'.pyc', '.pyo', '.html', '.htm', '.zip', '.exe', '.dll',
|
||||
'.uasset', '.umap', '.pdb', '.log'}
|
||||
@@ -115,12 +131,14 @@ RULES = [
|
||||
Rule('P15', 'failure-mode entry ids are not sequential in document order'),
|
||||
Rule('P16', 'two skills share an entry-id prefix, so the archive resolver '
|
||||
'silently collapses them'),
|
||||
Rule('P17', 'skill content names an engine tool surface (toolset, meta-tool '
|
||||
'or endpoint) that is Experimental and version-bound'),
|
||||
]
|
||||
|
||||
# A rule that never fires on its fixture does not exist. The fixture corpus in
|
||||
# _gate/test_gate.py asserts one poison per rule; this guards against a rule
|
||||
# being silently dropped from the list itself.
|
||||
assert len(RULES) >= 16, 'rule set shrank; a rule was lost'
|
||||
assert len(RULES) >= 17, 'rule set shrank; a rule was lost'
|
||||
assert len({r.id for r in RULES}) == len(RULES), 'duplicate rule id'
|
||||
|
||||
|
||||
@@ -300,6 +318,40 @@ def check_harness_neutral(root: Path) -> list[Violation]:
|
||||
return out
|
||||
|
||||
|
||||
def check_engine_tool_surface(root: Path) -> list[Violation]:
|
||||
"""P17: nothing under skills/ may name the engine's own agent-tool surface.
|
||||
|
||||
P13 keeps the content free of one *harness*; this keeps it free of one
|
||||
*engine build*. The distinction matters because the failure looks different:
|
||||
a harness token tells the reader to run something they do not have, while a
|
||||
toolset name tells them to run something that used to exist. The second is
|
||||
quieter -- the tool is missing, the search returns nothing, and "nothing
|
||||
found" is indistinguishable from "no problem here", which is the exact
|
||||
failure this bundle exists to document.
|
||||
|
||||
Detect recipes may still say what to look for. They may not say which
|
||||
first-party tool to look with.
|
||||
"""
|
||||
out: list[Violation] = []
|
||||
base = root / 'skills'
|
||||
if not base.is_dir():
|
||||
return out
|
||||
patterns = [re.compile(p) for p in ENGINE_TOOL_TOKENS]
|
||||
for path in sorted(base.rglob('*')):
|
||||
if not path.is_file() or path.suffix.lower() not in TEXT_SUFFIXES:
|
||||
continue
|
||||
rel = path.relative_to(root).as_posix()
|
||||
for n, line in enumerate(path.read_text(
|
||||
encoding='utf-8', errors='replace').splitlines(), 1):
|
||||
for pat in patterns:
|
||||
m = pat.search(line)
|
||||
if m:
|
||||
out.append(Violation('P17', rel, n,
|
||||
f'engine tool surface: {m.group(0)}'))
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def check_catalog(root: Path) -> list[Violation]:
|
||||
"""P14: catalog.json describes every skill without vendor vocabulary.
|
||||
|
||||
@@ -555,7 +607,8 @@ def check_manifest(root: Path) -> list[Violation]:
|
||||
|
||||
|
||||
CHECKS = (check_text_rules, check_top_level, check_junk, check_skills,
|
||||
check_manifest, check_harness_neutral, check_catalog,
|
||||
check_manifest, check_harness_neutral, check_engine_tool_surface,
|
||||
check_catalog,
|
||||
check_id_prefixes)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user