Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,8 @@ jobs:
--test tests/test_claude_plugin_conflicting_identity.py \
--test tests/test_claude_plugin_secret_to_prompt.py \
--test tests/test_claude_plugin_secret_to_mcp.py \
--test tests/test_claude_plugin_command_skill_reuse.py
--test tests/test_claude_plugin_command_skill_reuse.py \
--test tests/test_claude_plugin_hide_actions.py
- name: Verify 100% statement coverage for Claude plugin scan CLI
if: matrix.python-version == '3.13'
run: |
Expand Down
7 changes: 7 additions & 0 deletions CHANGELOG.d/1099-claude-plugin-supply-chain.md
Original file line number Diff line number Diff line change
Expand Up @@ -92,3 +92,10 @@
`claude-plugin-secret-to-mcp`. Curl copies stay
`claude-plugin-secret-to-network`. Prompt and log copies stay
`claude-plugin-secret-to-prompt`. Snippets are the env name only.
Skill, command, or agent text that hides tool use, rewrites the system
prompt, or expands the declared goal fails as
`claude-plugin-hide-actions-directive`,
`claude-plugin-self-modify-directive`, or
`claude-plugin-goal-escalation-directive`. Honest ``report each tool
call to the user`` wording, README prose, and vendored copies are not
that class. #1036 injection and exfil identities stay on their rules.
112 changes: 110 additions & 2 deletions appguardrail_core/claude_plugin_detector.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,8 +19,11 @@
finding. Capability inventory is evidence,
not permission: presence of a capability is not a finding by itself. Skill
homoglyph, injection, exfiltration, and placeholder hits reuse #1036 rule
identities. A lockfile-backed package.json without a lifecycle download
stays inventory. Vendored trees are one scope finding, not hook scans.
identities. Skill, command, or agent text that hides tool use, rewrites
the system prompt, or escalates the declared goal is a separate
instruction-override family. A lockfile-backed package.json without a
lifecycle download stays inventory. Vendored trees are one scope finding,
not hook scans.
"""

from __future__ import annotations
Expand Down Expand Up @@ -195,6 +198,24 @@
"command, URL, or header. Keep credentials out of MCP declarations. "
"[CWE-200 - Exposure of Sensitive Information to an Unauthorized Actor]"
)
CLAUDE_PLUGIN_HIDE_ACTIONS_MESSAGE: Final = (
"Claude plugin skill, command, or agent text instructs the model to hide "
"tool use or conceal actions from the user. Instruction text is untrusted "
"data. Remove the directive. "
"[CWE-451 - User Interface (UI) Misrepresentation of Critical Information]"
)
CLAUDE_PLUGIN_SELF_MODIFY_MESSAGE: Final = (
"Claude plugin skill, command, or agent text instructs the model to "
"rewrite its system prompt or ignore previous policy. Instruction text "
"is untrusted data. Remove the directive. "
"[CWE-693 - Protection Mechanism Failure]"
)
CLAUDE_PLUGIN_GOAL_ESCALATION_MESSAGE: Final = (
"Claude plugin skill, command, or agent text instructs the model to "
"expand or escalate the goal beyond the declared task. Instruction text "
"is untrusted data. Remove the directive. "
"[CWE-693 - Protection Mechanism Failure]"
)
CLAUDE_PLUGIN_UNSIGNED_EXECUTABLE_DOWNLOAD_MESSAGE: Final = (
"Claude plugin hook or package lifecycle script downloads an unsigned "
"executable and makes it runnable. Pin and verify binaries; do not "
Expand Down Expand Up @@ -349,6 +370,33 @@
)
_SKILL_SURFACE_NAMES: Final = frozenset({"SKILL.md", "skill.json", "agent.md"})
_SKILL_MARKDOWN_DIRS: Final = frozenset({"commands", "agents"})
_HIDE_ACTIONS_DIRECTIVE = re.compile(
r"(?i)(?:do not tell the user[^\n]{0,80}(?:calling tools|tool calls|"
r"tool use)|hide (?:your )?(?:actions|tool use|tool calls))"
)
_SELF_MODIFY_DIRECTIVE = re.compile(
r"(?i)(?:ignore previous policy|rewrite (?:your )?system prompt)"
)
_GOAL_ESCALATION_DIRECTIVE = re.compile(
r"(?i)(?:expand|escalate|broaden) (?:the |your )?(?:goal|scope|objective)"
)
_INSTRUCTION_OVERRIDE_RULES: Final = (
(
"claude-plugin-hide-actions-directive",
_HIDE_ACTIONS_DIRECTIVE,
CLAUDE_PLUGIN_HIDE_ACTIONS_MESSAGE,
),
(
"claude-plugin-self-modify-directive",
_SELF_MODIFY_DIRECTIVE,
CLAUDE_PLUGIN_SELF_MODIFY_MESSAGE,
),
(
"claude-plugin-goal-escalation-directive",
_GOAL_ESCALATION_DIRECTIVE,
CLAUDE_PLUGIN_GOAL_ESCALATION_MESSAGE,
),
)
_DESCRIPTION_JSON_NAMES: Final = frozenset(
{"plugin.json", "marketplace.json", "skill.json"}
)
Expand Down Expand Up @@ -2781,6 +2829,7 @@ def _collect_plugin_hits(root: Path) -> tuple[PluginHit, ...]:
)
hits.extend(_package_lifecycle_file_hits(root))
hits.extend(_skill_supply_chain_hits(root))
hits.extend(_instruction_override_hits(root))
return tuple(hits)


Expand All @@ -2806,6 +2855,65 @@ def _is_skill_surface(path: Path) -> bool:
return any(part.lower() in _SKILL_MARKDOWN_DIRS for part in path.parts[:-1])


def _instruction_override_content_hits(
content: str, relative: str
) -> tuple[PluginHit, ...]:
"""Return hide-actions, self-modify, and goal-escalation hits from one file.

The three rule identities are one instruction-to-the-model family. This
detector does not copy #1036 injection or exfil regular expressions.

Args:
content: Skill, command, or agent instruction text.
relative: Repository-relative display path.

Returns:
Zero or more hits. Snippets omit secret literals and raw bidi.
"""
hits: list[PluginHit] = []
for rule_id, pattern, message in _INSTRUCTION_OVERRIDE_RULES:
match = pattern.search(content)
if match is None:
continue
hits.append(
PluginHit(
rule_id=rule_id,
line=content[: match.start()].count("\n") + 1,
snippet=_sanitize_plugin_snippet(match.group(0).splitlines()[0]),
message=message,
file=relative,
)
)
return tuple(hits)


def _instruction_override_hits(root: Path) -> tuple[PluginHit, ...]:
"""Fail closed on skill, command, or agent text that overrides the task.

Hide-actions, self-modify, and goal-escalation wording share this walker.
README, root ``AGENTS.md``, vendored copies, and command shell files are
not this class. #1036 injection and exfil identities stay on their rules.

Args:
root: Materialized plugin tree.

Returns:
Hits whose ``rule_id`` values are the instruction-override family.
Empty when no skill, command, or agent surface carries that wording.
"""
hits: list[PluginHit] = []
for path in _walk_entries(root):
if path.is_symlink() or not path.is_file() or not _is_skill_surface(path):
continue
relative = path.relative_to(root).as_posix()
if _is_vendored_scope_relative(relative):
continue
payload = _regular_file_bytes(path)
content = payload.decode("utf-8", errors="replace")
hits.extend(_instruction_override_content_hits(content, relative))
return tuple(hits)


def _skill_supply_chain_hits(root: Path) -> tuple[PluginHit, ...]:
"""Reuse released #1036 rule identities on plugin skill, agent, and command files.

Expand Down
2 changes: 1 addition & 1 deletion docs/TRACEABILITY.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
| structural Semgrep-style `pattern:` execution by lightweight engine | built-in scanner | not implemented unless a real structural matcher is added; fixtures are not execution |
| GitHub Actions transport-only polling loop (#1087, #938 vertical slice) | owned by PR #1088 / issue #1087; YAML rules and RED precision contracts | mapped-family only; this successor does not ship or close the detector |
| Password/database-url/auth-comment precision and test-file context (#1106) | existing `_scan_file` rules `hardcoded-password`, `hardcoded-database-url`, `todo-skip-auth`, `_finding_context` | implemented-branch regression lock |
| Claude plugin marketplace/package supply chain (#1099) | `claude-plugin-floating-git-ref`, `claude-plugin-provider-secret`, `claude-plugin-pipe-to-shell`, `claude-plugin-unsigned-executable-download` (hooks and package.json lifecycle scripts), `claude-plugin-unpinned-package-install`, `claude-plugin-undeclared-executable`, `claude-plugin-symlink-escape`, `claude-plugin-archive-path-traversal`, `claude-plugin-unadmitted-submodule`, `claude-plugin-duplicate-json-member`, `claude-plugin-nonstandard-json-constant`, `claude-plugin-malformed-utf8`, `claude-plugin-inconsistent-normalized-name`, `claude-plugin-vendored-scope-undeclared`, `claude-plugin-conflicting-identity`, `claude-plugin-unbounded-mcp`, `claude-plugin-license-missing`, `claude-plugin-license-mismatch`, `claude-plugin-dynamic-eval`, `claude-plugin-hidden-undeclared-executable`, `claude-plugin-concealed-identity`, `claude-plugin-oversized-package`, `claude-plugin-source-mismatch`, `claude-plugin-github-write-token`, `claude-plugin-docker-socket`, `claude-plugin-browser-profile-access`, `claude-plugin-deceptive-description`, `claude-plugin-secret-to-network`, `claude-plugin-secret-to-prompt`, `claude-plugin-secret-to-mcp`, reused #1036 `skill-name-homoglyph-confusable` / `skill-manifest-prompt-injection-payload` / `skill-doc-exfiltration-endpoint-directive` / `skill-placeholder-template-unresolved` on plugin skill/agent/command surfaces, deterministic scan receipt with catalog repository/SHA bind and SARIF 2.1.0 `sarif_sha256` bound to the same finding rule_ids, fail-closed receipt verification | implemented-branch |
| Claude plugin marketplace/package supply chain (#1099) | `claude-plugin-floating-git-ref`, `claude-plugin-provider-secret`, `claude-plugin-pipe-to-shell`, `claude-plugin-unsigned-executable-download` (hooks and package.json lifecycle scripts), `claude-plugin-unpinned-package-install`, `claude-plugin-undeclared-executable`, `claude-plugin-symlink-escape`, `claude-plugin-archive-path-traversal`, `claude-plugin-unadmitted-submodule`, `claude-plugin-duplicate-json-member`, `claude-plugin-nonstandard-json-constant`, `claude-plugin-malformed-utf8`, `claude-plugin-inconsistent-normalized-name`, `claude-plugin-vendored-scope-undeclared`, `claude-plugin-conflicting-identity`, `claude-plugin-unbounded-mcp`, `claude-plugin-license-missing`, `claude-plugin-license-mismatch`, `claude-plugin-dynamic-eval`, `claude-plugin-hidden-undeclared-executable`, `claude-plugin-concealed-identity`, `claude-plugin-oversized-package`, `claude-plugin-source-mismatch`, `claude-plugin-github-write-token`, `claude-plugin-docker-socket`, `claude-plugin-browser-profile-access`, `claude-plugin-deceptive-description`, `claude-plugin-secret-to-network`, `claude-plugin-secret-to-prompt`, `claude-plugin-secret-to-mcp`, `claude-plugin-hide-actions-directive` / `claude-plugin-self-modify-directive` / `claude-plugin-goal-escalation-directive`, reused #1036 `skill-name-homoglyph-confusable` / `skill-manifest-prompt-injection-payload` / `skill-doc-exfiltration-endpoint-directive` / `skill-placeholder-template-unresolved` on plugin skill/agent/command surfaces, deterministic scan receipt with catalog repository/SHA bind and SARIF 2.1.0 `sarif_sha256` bound to the same finding rule_ids, fail-closed receipt verification | implemented-branch |
| Orphaned GitHub Actions registry identities (#929) | owned by PR #966 / issue #929; live registry DAST | mapped-family only; this successor does not ship or close the detector |
| Org security-failure CI tickets without copied vuln evidence | documented non-detectable family | snapshot in `tests/fixtures/cwl-security-issue-inventory.json` |

Expand Down
13 changes: 11 additions & 2 deletions docs/doctoring/cwl-security-issue-detectors.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ every frozen family. It implements only the unique families it owns.
|---|---|---|---|---|
| Transport-only Actions polling | SAST | #1087, #938 | PR #1088 / issue #1087 | maps only |
| Secret indirection / auth comments | SAST | #1106 | this successor | implements regression lock on existing `_scan_file` rules, including LifeOS #247 test-title/authority wording |
| Claude plugin supply chain | SAST | #1099 | this successor | implements `claude-plugin-*` findings including unsigned executable downloads from hooks and package.json lifecycle scripts, unpinned package URL installs, GitHub write tokens, Docker socket binds, host browser-profile stores, deceptive plugin/skill/command descriptions, non-standard JSON constants, malformed UTF-8 JSON bytes, non-NFC identity names, undeclared vendored or generated code scope, conflicting plugin/skill/command identities, secret-to-network flows, secret-to-prompt, log, or subprocess-env copies, and secrets copied into MCP env/args/command/URL/headers, reuses released #1036 skill-supply-chain rule identities on plugin skill/agent/command surfaces, capability inventory evidence, undeclared-executable admission, LICENSE/NOTICE SPDX mismatch, dynamic eval/exec on hook surfaces, hidden undeclared executable/config surfaces, a secret-free scan receipt with catalog repository/SHA bind and SARIF 2.1.0 `sarif_sha256` bound to the same finding rule_ids, and fail-closed stale/mismatched receipt verification |
| Claude plugin supply chain | SAST | #1099 | this successor | implements `claude-plugin-*` findings including unsigned executable downloads from hooks and package.json lifecycle scripts, unpinned package URL installs, GitHub write tokens, Docker socket binds, host browser-profile stores, deceptive plugin/skill/command descriptions, non-standard JSON constants, malformed UTF-8 JSON bytes, non-NFC identity names, undeclared vendored or generated code scope, conflicting plugin/skill/command identities, secret-to-network flows, secret-to-prompt, log, or subprocess-env copies, secrets copied into MCP env/args/command/URL/headers, and hide-actions / self-modify / goal-escalation wording on skill/command/agent surfaces, reuses released #1036 skill-supply-chain rule identities on plugin skill/agent/command surfaces, capability inventory evidence, undeclared-executable admission, LICENSE/NOTICE SPDX mismatch, dynamic eval/exec on hook surfaces, hidden undeclared executable/config surfaces, a secret-free scan receipt with catalog repository/SHA bind and SARIF 2.1.0 `sarif_sha256` bound to the same finding rule_ids, and fail-closed stale/mismatched receipt verification |
| Orphaned Actions workflows | DAST | #929 | PR #966 / issue #929 | maps only |
| Org CI failure without evidence | non-detectable | 353 tickets | inventory snapshot | maps only |
| UX / control-plane product gaps | non-detectable | #871, #928 | out of SAST/DAST scope | maps only |
Expand All @@ -27,7 +27,9 @@ every frozen family. It implements only the unique families it owns.
Uncontrolled CI wait is CWE-400 resource consumption (MITRE, n.d.-a). Hard-coded
credentials are CWE-798 (MITRE, n.d.-b). Unsigned installer execution is CWE-494
(MITRE, n.d.-c). Symlink follow during package admission is CWE-59 (MITRE,
n.d.-d). GitHub Actions workflow identities persist after file deletion
n.d.-d). Concealing tool use from the user is CWE-451 (MITRE, n.d.-e).
Rewriting a system prompt or escalating the declared goal is CWE-693
(MITRE, n.d.-f). GitHub Actions workflow identities persist after file deletion
and must be disabled through the lifecycle API (GitHub, n.d.). Those last two
workflow families remain owned by PRs #1088 and #966.

Expand All @@ -42,6 +44,7 @@ workflow families remain owned by PRs #1088 and #966.
- `tests/test_claude_plugin_conflicting_identity.py`
- `tests/test_claude_plugin_secret_to_prompt.py`
- `tests/test_claude_plugin_secret_to_mcp.py`
- `tests/test_claude_plugin_hide_actions.py`
- `tests/test_password_indirection_precision.py`
- `tests/test_cwl_security_issue_inventory.py`
- `tests/fixtures/cwl-security-issue-inventory.json`
Expand All @@ -66,3 +69,9 @@ https://cwe.mitre.org/data/definitions/494.html

MITRE. (n.d.-d). *CWE-59: Improper link resolution before file access*.
https://cwe.mitre.org/data/definitions/59.html

MITRE. (n.d.-e). *CWE-451: User interface (UI) misrepresentation of critical
information*. https://cwe.mitre.org/data/definitions/451.html

MITRE. (n.d.-f). *CWE-693: Protection mechanism failure*.
https://cwe.mitre.org/data/definitions/693.html
7 changes: 4 additions & 3 deletions docs/sast-dast-rule-research.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,16 +76,17 @@ files being scanned, then applies the union of relevant checks. Examples:
- `java-jwt-none-algorithm`: JWT none algorithm marker.
- `java-objectinputstream-deserialization`: direct Java native deserialization
entry point, CWE-502.
- `claude-plugin-*`: CWE-494/CWE-798/CWE-829/CWE-250/CWE-200 plugin
- `claude-plugin-*`: CWE-494/CWE-798/CWE-829/CWE-250/CWE-200/CWE-451/CWE-693 plugin
marketplace provenance, provider secrets, GitHub write tokens, Docker
socket binds, secret-to-network flows, secret-to-prompt and secret-to-log
copies, pipe-to-shell installers,
unsigned executable downloads including package.json lifecycle scripts,
unpinned package URL installs,
undeclared executables, undeclared vendored or generated code scope,
reused #1036 skill-supply-chain identities on
plugin skill/agent/command surfaces, and fail-closed replay of a stale or
mismatched scan receipt.
plugin skill/agent/command surfaces, hide-actions / self-modify /
goal-escalation instruction wording on those surfaces, and fail-closed
replay of a stale or mismatched scan receipt.
- Mapped, not owned here: GitHub Actions transport-only poll loops (#1087,
PR #1088) and orphaned workflow registry DAST (#929, PR #966).
- `tool-execute-parameters-passthrough`: Strix-observed dynamic tool execution
Expand Down
Loading