diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2d6298d8..f587c138 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -72,6 +72,27 @@ jobs: - name: Lint if: ${{ needs.classify-changes.outputs.docs_only != 'true' }} run: uv run ruff check . + # The Harbor verifier runs these files with the task image's python3, which + # can be older than the Python SkillEvaluator itself needs. Compiling catches + # syntax; the smoke script also runs code that needs newer runtime features. + - name: Compile Harbor verifier files on Python 3.9 and 3.11 + if: ${{ needs.classify-changes.outputs.docs_only != 'true' }} + shell: bash + run: | + set -euo pipefail + files=( + src/skillevaluator/tier3/harbor/templates/*.py + src/skillevaluator/tier3/eval_core/log_converters.py + src/skillevaluator/tier3/eval_core/codex_tool_call_normalizer.py + src/skillevaluator/evidence.py + ) + for version in 3.9 3.11; do + uv run --no-project --python "$version" -- python -I -c \ + "import sys; [compile(open(p, encoding='utf-8').read(), p, 'exec') for p in sys.argv[1:]]" \ + "${files[@]}" + uv run --no-project --python "$version" --with "idna>=3.10,<4" -- python -I \ + scripts/ci/smoke_harbor_verifier.py + done - name: Run tests with coverage if: ${{ needs.classify-changes.outputs.docs_only != 'true' }} run: >- diff --git a/.gitleaks.toml b/.gitleaks.toml index c8652ed9..9b8781de 100644 --- a/.gitleaks.toml +++ b/.gitleaks.toml @@ -10,6 +10,20 @@ targetRules = ["generic-api-key"] regexes = ['''^sk-AbCdEf1234567890$'''] paths = ['''^tests/test_tier3_(progress|result_display)\.py$'''] +[[allowlists]] +description = "Synthetic API key used by plugin hook inline-secret tests" +condition = "AND" +targetRules = ["generic-api-key"] +regexes = ['''^abcd1234secret$'''] +paths = ['''^tests/validators/test_plugin_component_risk\.py$'''] + +[[allowlists]] +description = "Synthetic Docker registry login (base64 of a fake user:password) used by the scanner-credential tests" +condition = "AND" +targetRules = ["generic-api-key"] +regexes = ['''^dXNlcjpodW50ZXIy$'''] +paths = ['''^tests/validators/test_audits_endpoints_review_fixes\.py$'''] + [[allowlists]] description = "Synthetic basic-auth, cloud-key and URL-token fixtures in the first revisions of the security-evidence and MCP-hardening tests; both files now assemble them from split literals" condition = "AND" diff --git a/CHANGELOG.md b/CHANGELOG.md index a925cd72..ea1dc658 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -38,6 +38,41 @@ All notable changes to SkillEvaluator are documented in this file. Effectiveness and Integration results, and the behavior the run excluded. - Added a Plugin Evaluation docs page covering all three tiers, the plugin dataset fields, verdicts, and current limitations. +- Review plugin subagent and command privileges in Tier 1: unrestricted `Bash` and `bypassPermissions` are HIGH; wildcard tools, `acceptEdits`, and subagents that inherit every tool next to write-capable MCP servers are MEDIUM. +- Add a static hook risk model for plugins: per-handler event, matcher, handler type, and risk flags, with findings for broad auto-approval, remote code execution, HTTP endpoints (`hooks.allowed_urls` policy allowlist, compared as parsed URLs), context injection, and files outside the plugin root. Matchers are classified without a backtracking regex engine, flat hooks (Cursor and Agent Plugins client extensions) get the same checks, hook scripts are read from their first 64 KiB even when larger or not UTF-8, and a script that cannot be analyzed (`plugin_hook_script_unanalyzed`) or a hooks source past the scan limits (`plugin_hook_scan_truncated`) is a HIGH finding. +- Extend the opt-in `dependency` check for plugins to npm lockfiles and `package.json` and to container images (MCP `docker`/`podman` commands from every manifest the plugin ships, and Dockerfiles), with OSV-Scanner, `npm audit --package-lock-only`, Grype, or Trivy. The plugin audit never passes silently: no scanner, a failed scanner run (pip-audit included), or a manifest, lockfile, or Dockerfile that cannot be read within its bounds makes the check `INCOMPLETE`. +- Add opt-in `validate --resolve-endpoints` (policy `endpoints.resolve`) DNS and single-`HEAD` redirect checks for MCP (including servers that only an additional manifest declares) and HTTP hook URLs that reach private, link-local, or metadata addresses. Hosts with an unexpanded `${VAR}` are skipped. +- Add the opt-in `claude-validate` Tier 1 check, which compares SkillEvaluator's plugin verdict with `claude plugin validate --strict` when the Claude Code CLI is installed. +- Render plugin privileges, hook risk, the CVE audit summary, validator parity, and endpoint checks in the JSON, Markdown, HTML, and CLI reports. +- Plugin evaluation supports Codex (`.codex-plugin/plugin.json`), Cursor (`.cursor-plugin/plugin.json`), + and Agent Plugins v1 (root `plugin.json`) manifests in every tier, with deterministic manifest precedence, + per-format required-field checks, component mapping (Codex apps and Agent Plugins extensions are + inventoried as unsupported), and MEDIUM `plugin_manifest_conflict` findings when a plugin's manifests + disagree on name or version. Codex `http_headers` are checked together with `headers`, a root + `plugin.json` that is not UTF-8 or exceeds the 1 MiB manifest limit no longer fails discovery, and an + additional manifest with such content is a MEDIUM `plugin_manifest_additional_invalid`. +- Add `--plugin-load {wrapper,native,auto}` to `tier3 evaluate-plugin` and `validate --tier3` (default `wrapper`); `native` stages the with-plugin arm the way Claude Code, Codex, OpenCode, or Hermes load plugins, while the member-skills and no-plugin arms stay unchanged. +- Record `plugin_load` (requested mode and per-agent mode, reason, adapter, and component modes) in plugin provenance, and render a Plugin Loading section in the HTML, Markdown, and CLI reports. +- Wrap natively staged plugin hooks with the hook census logger (`/skilleval/hook_census.sh`) using the Tier 1 `hook_risk` ids. +- Write a per-trial load census (`/logs/agent/skilleval-load-census.json`) under native loading, collect it per agent, and mark verified components `loaded` in component coverage. +- Support `hermes` as an experimental Tier 3 agent in Docker and cloud environments with the Anthropic provider. The OpenAI and OpenAI-compatible providers are refused for Hermes, because Hermes routes `openai/MODEL` to OpenRouter and would send the key there. +- Plugin Tier 3 collects a hook execution census (`hook_census.sh`) into + `plugin_signals.hook_census` and marks natively staged hooks that ran as + `exercised` in component coverage. +- Plugin subagents and commands observed at runtime count toward activation + coverage and are marked `exercised` in component coverage when available. +- Plugin Tier 3 plants a random canary credential in every arm's workspace and + scores its exfiltration as a critical `canary_exfiltration` security finding, + with per-arm canary results. Shell commands are checked per statement, + including `sh -c`/`bash -lc`/`eval` payloads, heredocs, and argv-list + commands, with taint tracking across statements; URLs count only for + network, MCP, and command tools. +- `tier3 evaluate-plugin` and `validate` accept `--probe-mcp` to probe + author-supplied URL MCP servers from the host (advisory `mcp_proof`). The + probe sends only literal declared headers; `--probe-mcp-env NAME` opts one + host variable in to `${NAME}` header expansion. +- Plugin reports render the hook census, canary results, and MCP proof. +- `harbor.plugin_canary: false` in `evals/config.yml` turns off the canary decoy for a plugin run, and the run config records the choice. - Configurable evidence bundle budgets (`SKILL_EVAL_ACCURACY_BUDGET`, `SKILL_EVAL_GOAL_ACCURACY_BUDGET`, `SKILL_EVAL_BEHAVIOR_CHECK_BUDGET`) and final response limit (`SKILL_EVAL_BEHAVIOR_FINAL_RESPONSE_LIMIT`). ### Changed @@ -46,6 +81,21 @@ All notable changes to SkillEvaluator are documented in this file. applicable (N/A) instead of a fabricated 1.0 when an eval case has no `ground_truth` or `expected_behavior`; N/A metrics are left out of overall scores, averages, and lift, and render as N/A. +- `plugin_agent_unrestricted_bash` is now LOW (advisory) and also covers subagents + that omit `tools`: a subagent's tools list limits it but never pre-approves Bash. +- Tier 3 judges now see what the agent wrote. Write calls (write and edit tools, + `apply_patch` with OpenCode's `patchText` or Codex's `input`, Hermes `patch`, + and shell writes, including Hermes `terminal` and `execute_code`) keep their + body in judge evidence instead of the first 200 characters. Each write body + gets an even share of the room left, at least 1,800 characters, and any cut + keeps the start and end with a marker that says how much was cut. File + changes leave room for the newest tool results and for a short tool history, + which lists each write as one line, so skill calls and test runs stay in + view. An over-budget history drops its middle first, and its copy of the + final answer, which FINAL RESPONSE already shows, shrinks before any tool + call drops. Judge evidence is secret-redacted; placeholder keys such as + `sk-your-key-here` stay as written. Scores for cases that write files can + change. ### Fixed @@ -91,6 +141,128 @@ All notable changes to SkillEvaluator are documented in this file. - Re-rendered plugin Tier 3 reports (`view`, standalone renders, and the report delivered after a plugin run) keep plugin provenance and INCOMPLETE status by reading the bounded, no-follow `plugin_provenance.json` sidecar. +- Tier 3 JWT log redaction, on the host and in the Harbor verifier, keeps memory flat on long token runs instead of using about 75 bytes per character. +- The Harbor verifier files parse again on task images whose `python3` is older than 3.12, and CI now compiles them on Python 3.9 and 3.11. +- The Tier 1 PII scan stays linear on very long lines: the email, JWT, database URL, and connection-string patterns no longer take minutes on one crafted line. An email local part over 64 characters is reported by its last 64. +- Local mode restores SIGPIPE before it runs a command, so a pipeline such as `yes | head` ends instead of hanging until the timeout. +- `--checks claude-validate` compares only Claude Code (`.claude-plugin`) plugins and reports other formats as not applicable. It parses the messages of Claude Code's text report, leaves output with no verdict INCOMPLETE, and names files relative to the plugin root. +- The Tier 1 PII scan still reports a database URL whose password holds a raw `/`, and no longer reports a host and port followed by a path or text (such as `redis://cache:6379/0?owner=ops@corp.io`) as a credential. +- `--checks claude-validate` still finds a `claude` installed through Volta, asdf, or nvm under the throwaway HOME, does not count Claude Code's notes as warnings in the text report, and leaves a JSON report that names no manifest and no files INCOMPLETE. +- Native plugin loading rewrites only Harbor's real launch command (its launcher line, at a shell-command boundary), so task or plugin MCP text with the launch words no longer gets the setup or `--plugin-dir`, and a run whose launch is missing or repeated fails instead of running without the plugin. +- Codex native MCP config and the wrapper `plugin_mcp_servers.toml` are written as TOML strings (an emoji or DEL no longer breaks Codex's `config.toml`), the Codex TOML is parsed at staging time, and a broken `plugin_mcp_servers.toml` fails the run instead of silently dropping every plugin MCP server. +- OpenCode native loading stages a plugin agent named like an OpenCode built-in (`build`, `plan`, `general`, ...) as `-` instead of replacing the built-in, and turns subagent `tools`/`disallowedTools` into OpenCode `permission` rules instead of granting every tool. +- Hermes native loading labels its components `wrapper`, and `--plugin-load auto` uses the wrapper for Hermes, because its with-plugin task is the wrapper task plus the load census. +- Codex and OpenCode native rules drop Cursor rule frontmatter and stage only always-on rules; glob-scoped, agent-requested, and manual Cursor rules are reported as not loaded. +- Tier 3 treats an MCP server that launches through the manifest format's own plugin-root placeholder (`${PLUGIN_ROOT}`, `${CURSOR_PLUGIN_ROOT}`, braced or bare) or a relative `cwd` as a plugin-file launch: it is no longer staged as runnable for wrapper, Codex, OpenCode, or Hermes arms, and the run is reported INCOMPLETE. +- `--plugin-load native|auto` stages `${CLAUDE_PLUGIN_ROOT}` MCP servers in the native Claude Code `.mcp.json` (the plugin tree is copied there), and a plugin whose only component is such a server is evaluated instead of skipped. +- `--plugin-load` resolves the per-agent plan before staging: the skip decision and the INCOMPLETE rule follow what each with-plugin arm stages, agent- or command-only plugins run where an adapter loads them, and the plugin is snapshotted only when some agent loads it natively. +- A permission-bypass flag in a plugin hook, subagent, settings file, or LSP server now blocks only the adapters that would stage it: `auto` falls back to the wrapper for that agent with the reason recorded, and `native` fails only for those agents. +- Native Claude Code staging translates Cursor hook events to Claude Code events and roots relative hook commands at `${CLAUDE_PLUGIN_ROOT}`; Cursor events with no equivalent get a `not_loaded` census row. +- Native Claude Code staging keeps `userConfig`, the plugin `settings.json` keys Claude Code applies (`agent`, `subagentStatusLine`), LSP servers, and MCP `env`/`headers` (refusing literal secrets), and lists `bin/` and Codex apps in the load census. +- Native Claude Code staging copies member skills that Claude Code would not read into `skills/` (failing on a name clash), and the load census and routing aliases now cover every skill the staged plugin loads, including declared skill directories. +- Tier 3 reports a plugin run INCOMPLETE when an MCP server uses a `${user_config.*}` value that a with-plugin arm cannot fill in: the wrapper never does, and native Claude Code applies only `userConfig` defaults. +- Native Claude Code user rules keep their full file names, so `style.md` and `style.mdc` no longer overwrite each other. +- `--plugin-load native` with an environment that cannot load natively (such as `--env-mode local`) now fails with that reason before the environment preflight instead of a missing local CLI error, and `auto` without a native snapshot falls back to the wrapper. +- `--plugin-load` plans with the `harbor.task_source` pinned in the evals `config.yml`, as the run does, so a plugin pinned to native Harbor tasks is no longer planned (and reported complete) as a native Claude Code run. +- Native Claude Code staging rewrites a Codex or Agent Plugins `${PLUGIN_ROOT}` in hook commands to `${CLAUDE_PLUGIN_ROOT}`, so those hook scripts run. +- Native Claude Code staging roots any relative word in a Cursor hook command that names a plugin file (`sh scripts/x.sh`), and translates more Cursor hook events (`afterShellExecution`, `afterMCPExecution`, `beforeReadFile`, `sessionEnd`, `subagentStart`, `subagentStop`, `preCompact`). +- Native Claude Code staging fails when a copied member skill has the same name as a skill in a declared skill directory, since Claude Code would load both under one name. +- The coverage reason of a `${CLAUDE_PLUGIN_ROOT}` MCP server now names its own gap, such as an unfilled `${user_config.*}` key, instead of always blaming the other with-plugin arms. +- The plugin hook census counts exit 2 as a blocked run (the hook's deny + decision), not a failure, and exit 126/127 as not started; reports show both. +- Native plugin loading marks a component `loaded` only when the harness reports + it (Claude Code's startup event; a failed or pending MCP server is not + loaded). A file listing is now `listed` and keeps the row `staged`, the Plugin + Loading table no longer calls it verified, census notes are appended to the + row reason, and a native run with no load census in any trial is INCOMPLETE. +- The load census ignores entries for components that were never staged and + only promotes types the agent loads natively; the OpenCode checks require its + launch config variables, and the Hermes MCP check needs an exact + `mcp_servers` key. +- A plugin hook counts as exercised only from runs of the exact handler ids that + were natively staged for that agent, and not when every run failed to start; + the evidence is labeled self-reported. +- A plugin run with failed trials or a failed arm keeps its with-plugin + evidence: the load and hook census read every trial, `plugin_provenance.json` + is written before the error, and Tier 3 reports an INCOMPLETE result instead + of a skip; a crash in the report-only sum-of-parts arm no longer ends the run. +- The Tier 3 verifier retries a judge call that timed out, lost its connection, + or got HTTP 408/429/5xx, with a short backoff bounded inside the verifier + timeout; other errors are not retried. +- Plugin signals and in-agent MCP proof map native MCP tool names (Claude Code `mcp__plugin____`, OpenCode `_`, Hermes `mcp__`, Codex bare names via its session log) to the declared server, and never credit one server's calls to a similarly named one. +- Plugin signals count Claude Code `Skill(:)` calls as command activations, credit every SKILL.md in a chained shell read, and no longer mark a skill read as failed because the skill text says something is "not available". +- The Harbor verifier matches Claude Code native `:` names exactly and keeps `:` calls out of skill activation and routing grades. +- Plugin signals still mark a shell read of a SKILL.md as failed when the error line comes after Codex status lines or after other output. +- `--probe-mcp` counts the DNS lookup against its 20-second deadline and stops at once, without client tracebacks, on an oversized response or an off-origin SSE endpoint. +- The canary no longer scores common benign commands as a critical leak: a whole environment handed to a child process, loopback-only calls, `if`/`while` tests, `echo` text, commit messages, search and exclude patterns, bare variable names, symlinks, and the literal `$NAME` in tool arguments. +- The canary summary counts the decoys the verifier found (`planted_file`, `decoy_missing`), compares arms by leak rate, names sum-of-parts leaks, and says "Canary not confirmed" instead of a green pass when the decoy was missing. +- Tier 1 hook risk now covers skill and command frontmatter hooks, plugin + monitors, and Cursor and GitHub Copilot hooks (their approval events, + `permission: allow` output, and `bash`/`powershell` commands), and flags plugin + skill `allowed-tools` like a command's (`plugin_skill_unrestricted_bash`). +- Tier 1 remote-code checks no longer flag fetch-then-parse hooks + (`curl … | python3 -m json.tool`, `jq . out.json`), and now catch `| $SHELL`, + `| busybox sh`, `source /dev/stdin`, evaluated download variables, + `exec(urlopen(…))`, git or URL package runners, unpacked archives, and a file + one hook downloads and another runs. +- Tier 1 scans hook scripts whole (up to 1 MiB) and one script level deeper (also + after `cd ${CLAUDE_PLUGIN_ROOT}`), reads hard-linked scripts instead of failing + them, and reports an approval hook whose plugin script cannot be read. +- Tier 1 adds MEDIUM checks for auto-approving write, fetch, or MCP tools, unpinned + hook package runners, hooks that run `${CLAUDE_PLUGIN_DATA}` code, the `auto` and + `acceptEdits` permission modes, and LSP server command forms. +- Tier 1 classifies Bash grants the way Claude Code does (`Bash()`, `Bash(**)`, + `Bash(python3:*)`, `Bash(sh -c *)`) for commands, skills, and shipped settings, + and flags Codex bypass flags (`--dangerously-bypass-hook-trust`, `-a never`, + `-c approval_policy=never`). +- Tier 1 again flags a download piped into `python3 -c`, `node -e`, `perl -e`, + `ruby -e`, or `sh -c` when that program evaluates its input + (`exec(sys.stdin.read())`, `eval "$(cat)"`, `bash -c "$(cat)"`). +- Tier 1 follows a script named inside a hook script only where it is run, not + where it is echoed, tested with `[ -f … ]`, or mentioned in a comment. +- Tier 1 matches hook downloads and runs in linear time, so a 1 MiB hook script + or many hooks sharing scripts no longer stall it; a hook that runs more than + 2048 distinct files while the plugin downloads something is reported unanalyzed. +- Tier 1 expands simple script variables in download and run paths + (`D="$CLAUDE_PLUGIN_DATA/bin"; curl -o "$D/tool"`) and counts a relative + command such as `d/run` as running an unpacked download. +- Tier 1 treats a server-scoped MCP matcher (`mcp__github__.*`) as covering MCP + tools for auto-approve, and counts Codex's `-a never` and `-c approval_policy=never` + only after a `codex` command (`grep -a never file` is no longer HIGH). +- Plugin dependency audits no longer pass on missing evidence: `npm audit` runs with `--no-offline`, and an npm manifest with more than 5,000 packages or a missing pip-audit makes a plugin audit INCOMPLETE. +- `--resolve-endpoints` resolves every name before any request, classifies every DNS answer (up to 64), gives each `HEAD` a wall-clock deadline, no longer delays exit on a hanging DNS lookup, and is INCOMPLETE when the endpoint cap, the time budget, or DNS leaves an endpoint unchecked. +- The plugin dependency audit also CVE-audits the packages that `npx`, `bunx`, `pnpm dlx`, `uvx`, and `pipx run` MCP servers install. +- Declared skill folders (for example Cursor `"skills": "./my-skills/"`) now get skill schema checks; skills only in `skills/` are advisory when the format's declared `skills` replace it. +- Codex skills are now inventoried at any depth below `skills/` or a declared skills folder, as Codex discovers them. +- Agent Plugins 1.1.0 is now a MEDIUM unrecognized version and a `$schema` with surrounding whitespace is HIGH, because Codex and Hermes accept only the exact 1.0.0 identifiers. +- A `.codex-plugin/plugin.json` overlay beside a root Agent Plugins manifest is validated as the documented overlay instead of being reported invalid. +- Missing Codex `version`, `description`, or `author` is now MEDIUM, because the Codex runtime loads the plugin without them. +- Markdown and terminal reports show N/A instead of crashing when a baseline score is missing, the lift stays unknown instead of +0.00, and one failing report format no longer stops the others from being written. +- Plugin reports no longer say hooks, subagents and commands cannot be evaluated: they say Tier 3 does not stage them in wrapper mode and Tier 1 checks them statically, and `BENCHMARK.md` drops a type from the excluded list when Tier 3 staged, loaded or exercised it. +- The Markdown report says how many endpoints it left out past 200 and how many `claude plugin validate` errors and warnings it left out past 20, and the docs no longer say the canary token reaches only the verifier. +- HTML reports pass skill and agent names to click handlers through `data-*` attributes, so a crafted skill directory name can no longer run script, and the report sets a Content Security Policy that blocks outside scripts (except the pinned Chart.js file) and outbound requests. +- A lone surrogate in plugin text (for example `"\ud83d"` in a subagent name) no longer crashes the terminal output or the Markdown, HTML and BENCHMARK reports, and terminal output shows bidi and zero-width characters as visible `\uXXXX` escapes. +- Markdown and `BENCHMARK.md` reports keep plugin text inert: terminal escape sequences are removed, link and emphasis syntax is escaped, and plugin values sit in `` elements instead of backtick code spans that showed escapes literally. +- SARIF gives bundled-skill findings repository-relative URIs with `uriBaseId` instead of an encoded `[skill] /absolute/path`, marks a failed Tier 3 run `executionSuccessful: false` with a notification, and reports a plugin-attributable canary leak as an error result; `BENCHMARK.md` shows canary results too. +- The terminal report prints the Tier 3 plugin block (verdict, coverage, loading, canary, census, Integration) for every plugin run, not only when AGENT_EVAL fails or has findings. +- `tier3 evaluate-plugin` no longer says Integration recorded no sum-of-parts comparison next to a measured Integration lift: its summary builds the Integration block the same way the reports do. +- The Markdown Integration line prints the lift once, followed by its interval, instead of repeating the estimate. +- Plugin cost tables show "not priced" with a short note when a run recorded tokens but no USD cost (common for gateway model ids), and the docs explain where USD comes from. +- `validate` and the single-check commands exit 1 when a requested report could not be written; the other reports and `BENCHMARK.md` are still written, and the quiet footer no longer links the missing file. +- SARIF no longer marks an advisory Tier 3 run with no task source as a failed run, and it adds a warning notification when a Tier 3 run was skipped (for example an engine crash or timeout) or is INCOMPLETE. +- A lone surrogate in the plugin provenance sidecar (for example in the plugin name) no longer crashes Tier 3 result building, `view` or re-rendered reports. +- The Markdown report escapes the reason shown for a skipped Tier 3 run and puts the requested plugin load mode in a `` element, and the JSON report counts a plugin-attributable canary leak as critical in `severity_counts`. +- Report-only activation coverage credits an OpenCode plugin agent staged under a new name (`-build` for an agent named `build`) to the declared agent, instead of leaving it never exercised. +- Skill-scoped checks (version, quality, lint, and folder schema walks) no longer skip every skill of a plugin or folder that lives under a directory named `results`, `versions`, or `evals`, and they treat a first-level `skills/evals/` or `skills/versions/` folder as a skill, like plugin skill discovery. +- Native Claude Code staging no longer loads a glob-scoped Cursor rule on every task: `globs` (or `paths`) become the user rule's `paths` frontmatter, always-on rules lose their Cursor frontmatter, and agent-requested and manual rules are reported as not loaded. +- The Harbor verifier, custom grader runner, and dataset metric also run on Python 3.9 task images, not only parse there: the canary check no longer calls `zip(strict=)`, the judge retry and reward helpers use tuples instead of `A | B` in `isinstance`, and a 3.9 read timeout (`socket.timeout`) is retried. CI now runs a smoke script on Python 3.9 and 3.11 after compiling. +- Runs without a sum-of-parts arm (the default effectiveness lift, or `--skip-baseline`) no longer print "Integration completeness issues" in the terminal and HTML reports: such runs record `complete: null`, and reports re-rendered from older runs that saved `complete: false` with a skipped sum-of-parts arm stay quiet too. +- The HTML Plugin Signals "Activation coverage" row shows the share of declared components exercised (for example 67% for 2 of 3) instead of "n/a exercised", because it no longer reads the per-component rate map as one number. +- Tier 3 coverage reasons follow the resolved plugin-load plan: a native arm's rule row says "staged as a native rule for " instead of the wrapper `SKILL.md`, hooks, subagents, and commands a native arm stages are "staged natively for " before any census arrives, and other rows say why each arm does not stage them (the generated wrapper, or "unsupported by the native adapter"). +- The `tier3 evaluate-plugin` run summary now prints the Component coverage block, like `validate`: it is printed after the plugin provenance is built instead of from the raw engine result. +- A plugin run that is INCOMPLETE only because it did not complete (for example every trial failed on a bad key) or because a native load was never confirmed now says so in the report conclusion ("Evaluation INCOMPLETE - the run did not complete") and in the Dependency Completeness hint, instead of blaming unresolved dependencies when nothing was deferred. +- The Codex, OpenCode, and Hermes load censuses list each MCP server that launches from plugin files as not loaded ("launches from plugin files; not staged by the native adapter"), so their Plugin Loading rows no longer under-count what did not load. +- Load-census evidence longer than the census text limit is cut in the middle, so a long container or temp path keeps its file name (`.../skills//SKILL.md`). ### Security @@ -111,6 +283,34 @@ All notable changes to SkillEvaluator are documented in this file. a plugin that bundles skills now also cover the plugin's root content (`scripts/`, `hooks/`, `.mcp.json`, ...), scanning each file once; a linked, hard-linked, or special entry anywhere in the plugin tree fails closed. +- `--checks claude-validate` runs `claude plugin validate` with an allowlisted environment (no credentials; proxy passwords removed) and a throwaway HOME and Claude config directory. +- Fixed an exponential-time Hermes launch pattern (CodeQL py/redos) that a plugin MCP URL could reach to hang a native Hermes run. +- Hermes is refused with the OpenAI and OpenAI-compatible providers, because Harbor's Hermes agent lets Hermes route `openai/MODEL` to OpenRouter and would send the evaluator's OpenAI key there. +- `--probe-mcp-env` accepts `NAME=HOST` and `NAME@SERVER` to send a variable to one host or server, prints which variables may go to which host before probing, and records what was sent; probed tool names and details drop terminal control characters. +- The canary and the other verifier security checks now cover Hermes tools, OpenCode MCP tools and `filePath` writes, Codex `workdir`, Claude Code subagent tool calls, whole-workspace uploads, DNS, cloud, forge, and container clients, `/dev/tcp` descriptors, scripts piped into a shell, and files tainted in an earlier tool call. +- Canary evidence now gets the shared credential redaction, and the outside-file read-back is deduplicated, reads likely files first, and records when it hits its cap. +- The canary treats a call as loopback-only only when every target is a literal loopback address with no proxy, and it now catches sending a symbolic link to the decoy, `awk`/`jq`/`yq` reads of the canary variable, `echo` output captured as a path, escaped `printf`/`echo -e` writes, and package uploads (`npm`/`yarn`/`pnpm`/`cargo publish`, `gem push`). +- Hook and MCP URLs are read the way WHATWG clients (Node, the MCP SDKs) read them, so a backslash or a missing `//` no longer slips past `hooks.allowed_urls` or the metadata checks; such URLs are HIGH findings, and every HTTP hook header is checked for inline secrets. +- The endpoint policy treats documentation, reserved, broadcast, and multicast addresses as non-public and adds the Oracle, ECS task, and EKS Pod Identity credential addresses to the metadata set; `--resolve-endpoints` classifies a redirect `Location` where WHATWG clients follow it. +- Container image audits check each registry host against the endpoint policy (metadata addresses are always refused, private registries need `mcp.allowed_private_hosts`) and run Grype and Trivy with an empty Docker config. +- `--resolve-endpoints` classifies a redirect `Location` that starts with three or more slashes or backslashes by the host a client follows, and decodes a percent-encoded host name before the DNS lookup. +- An HTTP hook URL with a password before a backslash is an inline secret again, an MCP URL such as `https:user:password@host` is too, and MCP URL findings no longer print that password. +- Grype and Trivy scans of a plugin-chosen registry also drop the scanners' registry login variables, `DOCKER_AUTH_CONFIG`, and Podman's auth file. +- Plugin checks now cover what other clients load from the same folder: Claude Code `--plugin-dir` defaults in a folder without `.claude-plugin/plugin.json`, Codex reading a Claude Code or Cursor manifest with its own rules, and hooks and apps in an Agent Plugins `extensions["com.openai"]` object. +- An additional client manifest that is over 1 MiB or not UTF-8 is now HIGH `plugin_manifest_additional_unreadable` and is read leniently so its hooks and MCP servers are still checked. +- A root Agent Plugins `plugin.json` over 1 MiB is now parsed whole (up to 8 MiB; a larger one always counts), so padding before `$schema` or escaped slashes no longer let the folder pass as a skill or a decoy manifest win. +- A Codex manifest path that Codex drops (no `./`, `./`, `..`, or a root variable) no longer turns off the checks on the default `hooks/hooks.json`, `.mcp.json`, `skills/`, `commands/`, or `.app.json`. +- Root variables such as `${CLAUDE_PLUGIN_ROOT}` in Claude Code and Codex manifest component paths are now HIGH `plugin_component_path_invalid`. +- Case variants of client manifests, such as `.Codex-Plugin/plugin.json`, are now found and checked, with a HIGH `manifest_case_variant`. +- Skills in `skills/evals/`, `skills/results/`, and `skills/versions/` are now discovered and scanned; a `SKILL.md` deeper in such a folder, or a manifest path into one, is HIGH. +- A skill in an `evals/`, `results/`, or `versions/` folder inside a declared skills folder (for example `my-skills/evals/`) is now inventoried, schema-checked, and HIGH `plugin_skill_in_unscanned_folder`, because clients load it but Tier 1 scans skip it. +- The Tier 1 MCP inline-secret check stays linear on a long JWT-like value (`eyJeyJ...`), which took about a second per 64 KB value and added up across many MCP args, env values, and headers. +- MCP and HTTP hook URLs with an invisible format character (such as a zero-width space in the host) or Unicode whitespace at either end are now HIGH, like other whitespace and control characters; only leading or trailing ASCII whitespace is still allowed. +- Tier 3 coverage and the dependency (CVE) audit now read an additional client manifest that is over 1 MiB or not UTF-8 leniently, like Tier 1, so its hooks and MCP servers no longer drop out of coverage or the audit; a component only another client loads names that client in its Tier 3 coverage reason. +- Tier 1 whole-tree scans now scan a skill that a declared skills folder loads from an `evals/`, `results/`, or `versions/` folder (such as `my-skills/evals/`, or a declared `./evals/`) as its own skill unit, after a no-follow check, instead of only failing it HIGH; a `SKILL.md` deep inside a skill's own artifact folder stays HIGH `plugin_skill_in_unscanned_folder`. +- A hook that pipes a download into a shell or interpreter is CRITICAL remote code again when a comment or a redirection follows the shell (`curl … | sh # install`, `| sh > /dev/null 2>&1`, `| python3 # c`); only a real script path makes the download plain data. +- Decoding `\uXXXX` and `\xXX` escapes in hook scripts is linear: a hook script that is a long run of backslashes no longer stalls every Tier 1 check that builds the plugin inventory (a 1 MiB script took hours). +- The canary check stays linear on code padded with blank lines: a plugin could steer the agent into one large code or shell call that made the verifier spend minutes on the environment-copy pattern and hit its timeout, losing the trial's canary result. ## 0.4.0 - 2026-09-30 diff --git a/docs/cli-reference.mdx b/docs/cli-reference.mdx index 230ba4fe..6f120d62 100644 --- a/docs/cli-reference.mdx +++ b/docs/cli-reference.mdx @@ -202,7 +202,7 @@ Applies to the whole run: target typing, policy profile, reports, tier selection | `--policy FILE` | none | Custom policy YAML overlaid on top of `--profile`. | | `--repo-root DIRECTORY` | git top-level of the plugin | Plugin only: repository root used to resolve same-repository skill and rule references in Tier 1 and Tier 3. The missing-dependency gate also requires this root to have a git `origin` remote; otherwise references stay `unresolved` (advisory). | -Auto-detection recognizes: a `SKILL.md` for skills, `.mdc` files for rules, `workflow-rules.mdc` for workflows, and — for plugins — a bundle-reference `agent_plugin.yaml`/`.yml` manifest or a contained `.claude-plugin/plugin.json` manifest. Plugins are validated against their public contract; quality, lint, and version checks run on each skill bundled under the plugin's `skills/` directory. Plugins run Tiers 1 and 2 by default; request plugin Tier 3 with `--tier3`. See [Plugin Evaluation](plugin-evaluation.mdx). +Auto-detection recognizes: a `SKILL.md` for skills, `.mdc` files for rules, `workflow-rules.mdc` for workflows, and — for plugins — a bundle-reference `agent_plugin.yaml`/`.yml` manifest, or a contained `.claude-plugin/`, `.codex-plugin/`, or `.cursor-plugin/` `plugin.json` or Agent Plugins root `plugin.json` manifest ([precedence](plugin-evaluation.mdx#manifest-detection)). Plugins are validated against their public contract; quality, lint, and version checks run on each skill bundled under the plugin's `skills/` directory. Plugins run Tiers 1 and 2 by default; request plugin Tier 3 with `--tier3`. See [Plugin Evaluation](plugin-evaluation.mdx). `validate` also takes the standard `-r/--report` and `-o/--output-dir` options described under [Global conventions](#global-conventions). @@ -212,7 +212,8 @@ Static checks; LLM-free by default. Tier 1 gates the exit code and always runs. | Flag | Default | Effect | | --- | --- | --- | -| `--checks, --tier1-checks TEXT` | all applicable | Comma-separated subset of Tier 1 checks. Default choices: `schema`, `version`, `security`, `pii`, `license`, `code-integrity`, `unicode`, `quality`, `lint`; opt-in (not run by default): `dependency`. `quality`/`lint`/`version` are skill-only and skipped for rules and workflows. | +| `--checks, --tier1-checks TEXT` | all applicable | Comma-separated subset of Tier 1 checks. Default choices: `schema`, `version`, `security`, `pii`, `license`, `code-integrity`, `unicode`, `quality`, `lint`; opt-in (not run by default): `dependency`, and, for plugins, `claude-validate` (parity with `claude plugin validate --strict`). `quality`/`lint`/`version` are skill-only and skipped for rules and workflows. | +| `--resolve-endpoints` | off | Plugin only, opt-in network check: resolve MCP and HTTP hook URL hosts and send one credential-free `HEAD` (no redirects followed) to flag names or redirects that reach private, link-local, or cloud-metadata addresses. Also enabled by `endpoints.resolve: true` in the policy. See [Plugin Evaluation](plugin-evaluation.mdx#opt-in-endpoint-dns-and-redirect-checks). | | `--fail-fast` | off | Stop on the first failing check instead of collecting all issues. | | `-c, --continue-on-failure` | off | Run the full pipeline without stopping early; record all issues in the reports. Overrides `--fail-fast`, and for folder validation keeps scanning every skill past a CRITICAL finding. | | `--llm, --tier1-llm / --no-llm, --no-tier1-llm` | `no-llm` | Enable LLM-backed security analysis (requires a configured public provider — see [Providers & Credentials](configuration.mdx)). | @@ -242,6 +243,9 @@ The following flags are forwarded to the live-eval engine when Tier 3 is enabled | `-a, --agents TEXT` | provider-native | Comma-separated Harbor agents to evaluate. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. | | `--env-mode` | `docker` | Harbor environment backend (full list under [tier3 evaluate](#tier3-evaluate)). | | `--lift-mode [effectiveness\|integration\|both]` | `effectiveness` | Plugin only: compare the coordinated plugin with no plugin, its member skills staged individually, or both. Integration requires explicit cross-component dataset evidence. | +| `--plugin-load [wrapper\|native\|auto]` | `wrapper` | Plugin only: how the with-plugin arm loads the plugin: the generated wrapper skill, the harness's own plugin layout, or native where supported. See [Native loading](plugin-evaluation.mdx#native-loading). | +| `--probe-mcp` | off | Plugin only: before Tier 3, probe each author-supplied URL MCP server from the host (`initialize` + `tools/list`, bounded, no redirects) under the endpoint policy and `mcp.allowed_private_hosts` from `--policy`. Advisory; recorded as `mcp_proof`. See [Public MCP proof](plugin-evaluation.mdx#public-mcp-proof). | +| `--probe-mcp-env NAME` | none | Plugin only, with `--probe-mcp`: a host environment variable the probe may expand into a server's declared `${NAME}` headers (repeatable). Without it, the probe sends only literal header values; a header that references any other variable is not sent. | | `--skip-baseline` | off | Skip the without-skill baseline (no lift analysis, faster). | | `--n-concurrent INTEGER` | unset | Concurrent eval cases per agent. | | `--max-agents INTEGER` | unset | Maximum agents to run in parallel. | @@ -394,7 +398,7 @@ Use `skillevaluator tier3 PATH` for the complete workflow with automatic missing | Flag | Default | Effect | | --- | --- | --- | -| `-a, --agents TEXT` | provider-native | Comma-separated Harbor agents. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. Supported: `claude-code`, `codex`, `opencode`; the alias `claude` is accepted for `claude-code`. See [Agents & Sandboxes](agents-and-sandboxes.mdx). | +| `-a, --agents TEXT` | provider-native | Comma-separated Harbor agents. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. Supported: `claude-code`, `codex`, `opencode`; experimental: `hermes` (container environments only, Anthropic provider); the alias `claude` is accepted for `claude-code`. See [Agents & Sandboxes](agents-and-sandboxes.mdx). | | `--env-mode` | `docker` | Where trials run. All 16 values: `docker`, `daytona`, `e2b`, `modal`, `runloop`, `langsmith`, `gke`, `novita`, `apple-container`, `singularity`, `islo`, `tensorlake`, `cwsandbox`, `wandb`, `use-computer`, `local`. Cloud modes are provider-managed Harbor backends enabled by the matching Harbor extra; `local` runs on your host. See [Agents & Sandboxes](agents-and-sandboxes.mdx). | | `--autopilot` | off | Create one eval case when no dataset/task source exists, then evaluate. The case is LLM-generated with the configured provider, with a deterministic keyless template fallback; an existing source is never overwritten. | | `--skip-baseline` | off | Skip the without-skill baseline (no lift analysis, faster). | @@ -433,10 +437,13 @@ Flags marked "unset" fall back to their matching keys in `evals/config.yml` wher ## tier3 evaluate-plugin Run Tier 3 against a public plugin without fetching remote components. The -command accepts a bundle-reference `agent_plugin.yaml`/`.yml` target or a -contained `.claude-plugin/plugin.json` target, stages the locally evaluable -skills, rules, and MCP declarations into a temporary wrapper, and records any -unresolved remote references in plugin provenance. +command accepts a plugin root or its manifest in any supported format: a +bundle-reference `agent_plugin.yaml`/`.yml`, a contained `.claude-plugin/`, +`.codex-plugin/`, or `.cursor-plugin/` `plugin.json`, or an Agent Plugins root +`plugin.json` ([precedence](plugin-evaluation.mdx#manifest-detection)). It +stages the locally evaluable skills, rules, and MCP declarations into a +temporary wrapper (or the harness's own layout with `--plugin-load`), and +records any unresolved remote references in plugin provenance. ```bash title="Evaluate plugin effectiveness and Integration" skillevaluator tier3 evaluate-plugin ./my-plugin --lift-mode both \ @@ -450,6 +457,9 @@ skillevaluator tier3 evaluate-plugin ./my-plugin --lift-mode both \ | `--env-mode` | `docker` | Harbor environment backend; accepts the same values as [tier3 evaluate](#tier3-evaluate). | | `--skip-baseline` | off | Skip the no-plugin baseline. Invalid with `integration` or `both`, which require a baseline. | | `--lift-mode [effectiveness\|integration\|both]` | `effectiveness` | Compare the coordinated plugin with no plugin, its member skills staged individually, or both. `both` falls back to effectiveness when composition evidence is missing. | +| `--plugin-load [wrapper\|native\|auto]` | `wrapper` | How the with-plugin arm loads the plugin. `native` fails for agents or environments without a native adapter; `auto` falls back to the wrapper. Baseline arms are unchanged. See [Native loading](plugin-evaluation.mdx#native-loading). | +| `--probe-mcp` | off | Before the run, probe each author-supplied URL MCP server from the host (`initialize` + `tools/list`, bounded, no redirects) under the endpoint policy. This command takes no policy file and uses a bundled profile (`SKILLEVALUATOR_PROFILE` or the default), none of which allowlists private hosts, so private hosts are not probed; a custom `mcp.allowed_private_hosts` allowlist applies to `--probe-mcp` only through `validate --tier3 --policy `. Advisory; recorded as `mcp_proof` in plugin provenance. See [Public MCP proof](plugin-evaluation.mdx#public-mcp-proof). | +| `--probe-mcp-env NAME` | none | With `--probe-mcp`: a host environment variable the probe may expand into a server's declared `${NAME}` headers (repeatable). Without it, the probe sends only literal header values; a header that references any other variable is not sent. | | `--n-attempts INTEGER` | unset | Attempts per eval case (pass@k). | | `--pass-threshold FLOAT` | unset | Score threshold (0.0–1.0) for a case to count as passed. | | `--stop-on-pass / --no-stop-on-pass` | unset | Stop a case's remaining attempts once one passes. | diff --git a/docs/eval-datasets.mdx b/docs/eval-datasets.mdx index 117fb5a0..20cbe814 100644 --- a/docs/eval-datasets.mdx +++ b/docs/eval-datasets.mdx @@ -283,6 +283,7 @@ grading: | `harbor.max_agents` | integer ≥ 1 | Cap on agents evaluated in one run. | | `harbor.timeout_multiplier` | number > 0 | Scales task timeouts. | | `harbor.agent_runtime_preflight` | boolean | Opt-in extra agent-only execution of the first staged task before the measured evaluation. Default: `false`; the CLI flag overrides this value. | +| `harbor.plugin_canary` | boolean | Plant the canary decoy credential in every arm of a plugin run. Default: `true`. Set `false` to run a plugin evaluation without it. | | `harbor.agent_workdir` | string | Working directory for the agent inside the container. | | `harbor.resources` | `cpus`, `memory_mb`, `storage_mb` | Per-container resource requests. | | `harbor.runtime_env` | list or mapping | Non-credential task values passed into the container. Prefer a list of plain names — each expands to `${NAME}` from your shell; a mapping sets explicit templates. Entries that name or reference operator-owned credentials (`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `NVIDIA_API_KEY`, base-URL variables, AWS credential variables) fail with a hard error. Alias: `passthrough_env`. | diff --git a/docs/plugin-evaluation.mdx b/docs/plugin-evaluation.mdx index 765da879..9b29a932 100644 --- a/docs/plugin-evaluation.mdx +++ b/docs/plugin-evaluation.mdx @@ -30,7 +30,7 @@ Each tier answers a different question about the plugin: ```mermaid flowchart LR - P(["Plugin root
agent_plugin.yaml or
.claude-plugin/plugin.json"]) + P(["Plugin root
agent_plugin.yaml, a Claude, Codex,
or Cursor plugin.json, or an
Agent Plugins plugin.json"]) subgraph T1["Tier 1: static, offline, gating"] direction TB @@ -93,8 +93,10 @@ and its count is part of `not_evaluated` (the number of components not staged). Reports make the gaps explicit instead of letting them read as a pass: - **Tier 1** inventories every component with a `support` level of - `evaluated`, `static_only`, or `unsupported`, and lists the component types it - cannot evaluate in `unsupported_types_present`. + `evaluated`, `static_only`, or `unsupported`, and lists the component types + that Tier 3 does not stage in wrapper mode in `unsupported_types_present`. + Tier 1 still checks hooks, subagents and commands statically (Hook risk and + Subagent and command privileges). - **Tier 3** records a coverage state and a reason for every component. When a declared skill, rule, or MCP server that the plugin needs could not be evaluated, the run is **INCOMPLETE**, never a full pass. See @@ -112,18 +114,103 @@ Reports make the gaps explicit instead of letting them read as a pass: ### Manifest detection A plugin root is a directory that contains one of these manifests. When more -than one exists, the first one in this order wins: +than one exists, the first one in this order is the **selected** manifest: -1. `agent_plugin.yaml` -2. `agent_plugin.yml` -3. `.claude-plugin/plugin.json` +| Order | Manifest | `manifest_type` | Format | +| --- | --- | --- | --- | +| 1 | `agent_plugin.yaml` | `agent_bundle_yaml` | [Bundle reference](#bundle-reference-manifest-agent_pluginyaml) | +| 2 | `agent_plugin.yml` | `agent_bundle_yaml` | Bundle reference | +| 3 | `.claude-plugin/plugin.json` | `claude_plugin_json` | [Claude Code](#contained-manifest-claude-pluginpluginjson) | +| 4 | `plugin.json` at the root | `agent_plugins_v1` | [Agent Plugins v1](#agent-plugins-v1-manifest-pluginjson) | +| 5 | `.codex-plugin/plugin.json` | `codex_plugin_json` | [Codex](#codex-manifest-codex-pluginpluginjson) | +| 6 | `.cursor-plugin/plugin.json` | `cursor_plugin_json` | [Cursor](#cursor-manifest-cursor-pluginpluginjson) | + +A root `plugin.json` is an Agent Plugins manifest only when it declares an +`https://agent-plugins.org/schemas//plugin.schema.json` `$schema`, the +rule that VS Code and Codex apply. Any other root `plugin.json`, such as the +legacy Copilot format, is not a supported manifest and is ignored. Passing that +file directly does not change this: `validate` resolves a manifest path to its +plugin root and runs the same discovery, so without another manifest +`validate --type plugin` fails with HIGH `manifest_missing`, whose message +explains the `$schema` rule. Add the Agent Plugins `$schema` to validate the +file as an Agent Plugins manifest. + +A root `plugin.json` that is not UTF-8 still counts when its bytes name the +schema host (for example, a UTF-16 file), so its encoding error is reported. +Clients read a manifest of any size, so a root `plugin.json` larger than the +1 MiB manifest limit is still parsed whole, up to 8 MiB. Whitespace before +`$schema` or escaped slashes (`https:\/\/agent-plugins.org\/...`) do not hide +the opt-in. A file that does not parse counts when it names the schema host, +and a file over 8 MiB always counts. Such a file is the Agent Plugins manifest, +and as the selected manifest it fails HIGH `manifest_unsafe` because it cannot +be read. Auto-detection uses the same rule, so such a folder stays a plugin +instead of becoming a skill. Only a linked, hard-linked, special, or changed +file fails discovery. + +Clients open fixed manifest paths. On a case-insensitive filesystem, such as +the macOS default, a client that opens `.codex-plugin/plugin.json` reads +`.Codex-Plugin/plugin.json`. So the client manifests (every JSON one) are +matched without regard to case. The exact spelling wins when both exist. A +case variant is used as the manifest it stands for and gets a HIGH +`manifest_case_variant`, because the plugin then differs by platform. + +When both `agent_plugin.yaml` and `agent_plugin.yml` exist, `agent_plugin.yml` +is read as an additional bundle manifest: skill and rule refs and `mcp` entries +that the selected manifest does not declare are inventoried with `declared_by`. You can pass the plugin root or the manifest file itself. Auto-detection recognizes plugins; add `--type plugin` to be explicit. Every manifest variant -must be one regular, single-link file inside the plugin root. A linked, +must be one regular, single-link file inside the plugin root. Discovery is one +bounded walk, two levels deep, that never follows links. A linked, hard-linked, or special manifest fails as `manifest_outside_root` or `unsafe_plugin_filesystem` even when another variant would otherwise win. +The order is SkillEvaluator's own choice, so results stay deterministic. It +keeps an existing `agent_plugin.yaml` or Claude Code evaluation unchanged when +a plugin adds another client's manifest. Clients use their own order: Codex, +for example, prefers an Agent Plugins `plugin.json`, then +`.codex-plugin/plugin.json`. + +#### Multiple manifests + +The selected manifest drives the evaluation: its name, its component fields, +and what Tier 3 stages. Every other manifest found is an **additional +declaration**: + +- It is recorded under `manifest_declarations`, with its `manifest_type`, + `name`, `version`, `status` (`parsed`, `invalid`, `unreadable`, or + `unsafe`), and the Agent Plugins `spec_version` when it has one. The CLI prints a `plugin_manifests` + line, and the Markdown and HTML reports add a **Plugin manifests** table. +- A `name` or `version` that differs from the selected manifest is a MEDIUM + `plugin_manifest_conflict`, because clients that load different manifests see + different plugins. +- An additional manifest that cannot be parsed or fails its format's required + fields is a MEDIUM `plugin_manifest_additional_invalid` (status `invalid`). +- An additional client manifest that is not UTF-8 or is larger than the 1 MiB + manifest limit is a HIGH `plugin_manifest_additional_unreadable` (status + `unreadable`). Clients do not share those limits: Codex reads a manifest of + any size and prefers `.codex-plugin` over `.claude-plugin`, and Claude Code + reads a Latin-1 manifest. So it is read again leniently (up to 8 MiB, with + replacement characters), and what it declares is checked like any other + additional manifest. Tier 3 coverage and the dependency audit read it the + same way, so its hooks and MCP servers are listed and CVE-audited. Only a + linked, hard-linked, special, or changed one is `unsafe` and fails as a + security check. +- A `.codex-plugin/plugin.json` beside a root Agent Plugins manifest is the + [Codex overlay](https://developers.openai.com/plugins/build/plugins) that + OpenAI documents. It holds only OpenAI settings (`apps`, `hooks`, and + `interface`), so only their types are checked, and its row has + `"overlay": true`. Its hooks are analyzed like any other hooks. +- Its components are inventoried with the same rules as the selected manifest, + and their static checks (paths, MCP policy, hook bypass flags, hook risk, and + subagent and command privileges) run too, so an MCP server, hook, subagent, + or command that only another client's manifest loads is still checked (for + clients that read this folder through another manifest, see + [Other clients that load the same folder](#other-clients-that-load-the-same-folder)). A component that + the selected manifest does not also declare carries `declared_by`, is never + `evaluated` (an `evaluated` type becomes `static_only`), and is not counted in + the context-cost estimate. Tier 3 records it as `not_staged` or `unsupported`. + ### Bundle-reference manifest: `agent_plugin.yaml` A bundle-reference plugin references skills and rules that live elsewhere in @@ -188,7 +275,9 @@ plugin schema is not validated. The component fields (`skills`, `commands`, the plugin root: an absolute, home-relative, drive-letter, or `..` path gets a HIGH `plugin_component_path_escape` finding. Claude Code expects `./`-relative paths, so a path without `./` gets a MEDIUM `plugin_component_path_style` -finding. A `${CLAUDE_PLUGIN_ROOT}/` prefix is treated as the root. +finding. Claude Code expands `${CLAUDE_PLUGIN_ROOT}` only in hook commands and +MCP server fields, and rejects a manifest whose component path starts with +it, so such a path is a HIGH `plugin_component_path_invalid`. - my-plugin/ @@ -214,6 +303,225 @@ finding. A `${CLAUDE_PLUGIN_ROOT}/` prefix is treated as the root. - evals.json +### Agent Plugins v1 manifest: `plugin.json` + +[Agent Plugins](https://agent-plugins.org) is an open, cross-client format. A +plugin has a root `plugin.json`, skills in `skills//SKILL.md`, and MCP +servers in a root `mcp.json`. The manifest cannot declare or relocate +components. Client-specific data lives in the manifest's `extensions` object, +keyed by a reverse-domain namespace, and client-specific files live in a +top-level directory of the same name. + +```json title="plugin.json" +{ + "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", + "name": "release-helper", + "version": "1.0.0", + "description": "Triage tickets and draft release notes.", + "extensions": { "com.example.client": { "setting": true } } +} +``` + +SkillEvaluator implements the 1.0.0 specification (`spec/1.0.0.md` and +`schemas/1.0.0/*.schema.json` in `agentplugins/agent-plugins-spec` at +`ff8ab5e`). The 1.1.0 working draft has the same rules, but Codex and Hermes +accept only the exact 1.0.0 schema identifiers, so 1.1.0 is reported like any +other unrecognized 1.x version. + +| Rule | Severity | +| --- | --- | +| `$schema` is required and must name an Agent Plugins plugin schema (`schema:$schema:missing` or `:invalid`) | HIGH | +| `$schema` has leading or trailing whitespace; Codex and Hermes compare it exactly (`schema:$schema:invalid`) | HIGH | +| A major version other than 1 (`schema:$schema:unsupported_version`) | HIGH | +| A 1.x version other than 1.0.0, including the 1.1.0 draft, is validated with the 1.0.0 rules; Codex and Hermes do not load it (`schema:$schema:unrecognized_version`) | MEDIUM | +| `name` is required: 1 to 64 characters of `a-z`, `0-9`, `-`, and `.`; it starts and ends with a letter or digit, with no `--` or `..` (`schema:name:pattern`) | HIGH | +| `version`, `description`, `homepage`, `repository`, and `license` are strings, `keywords` is a string array, and `author` allows only string `name`, `email`, and `url` | HIGH | +| Any other top-level field, such as `skills` or `mcpServers` (`plugin_manifest_unknown_field`); clients report and ignore it | MEDIUM | +| `extensions` is not an object; clients ignore it | MEDIUM | + +The `mcp.json` file must declare the Agent Plugins MCP `$schema` of the same +version as `plugin.json`. Otherwise clients disable the plugin's MCP servers, +and SkillEvaluator reports a HIGH `mcp_config_schema_mismatch`. Each server +must declare `type`: `stdio`, `streamable-http`, or `sse`. A missing `type` is a +HIGH `mcp_transport_missing`. `streamable-http` is checked and staged as +`http`. Only immediate children of `skills/` are skills. + +OpenAI-specific settings live in `extensions["com.openai"]`. Its `hooks` (a +path, an array of paths, an inline hooks object, or an array of them) get the +same bypass and risk checks as any other hooks, and its `apps` path is +inventoried as `app` components. Both use the Codex path rules below. + +### Codex manifest: `.codex-plugin/plugin.json` + +The Codex format is the one described by +[developers.openai.com/plugins](https://developers.openai.com/plugins/build/plugins) +and validated by the `plugin-eval` plugin in `openai/plugins` at `610a632`. +Codex also reads an Agent Plugins `plugin.json`, and calls this format its +compatibility format. + +```json title=".codex-plugin/plugin.json" +{ + "name": "release-helper", + "version": "1.0.0", + "description": "Triage tickets and draft release notes.", + "author": { "name": "Example Maintainers" }, + "skills": "./skills/", + "mcpServers": "./.mcp.json", + "apps": "./.app.json", + "interface": { "displayName": "Release Helper", "shortDescription": "Triage and release notes" } +} +``` + +| Rule | Severity | +| --- | --- | +| `name` is required (`schema:name:missing`) | HIGH | +| `version` or `description` is not a string; the Codex runtime does not load the plugin | HIGH | +| `version`, `description`, or `author` is missing; Codex plugin packaging expects them, but the runtime loads the plugin without them (`schema::missing`) | MEDIUM | +| `name` is at most 64 characters, starts with a letter or digit, and uses only letters, digits, `_`, and `-` | HIGH | +| `name` is not lowercase kebab-case | MEDIUM | +| `version` is not a semantic version, or `author.name` is missing | MEDIUM | +| `interface` is missing, or lacks any of the ten presentation fields `plugin-eval` requires | MEDIUM | +| A component path does not start with `./`; the Codex loader ignores such a field (`plugin_component_path_style`) | MEDIUM | +| A component path starts with a root variable such as `${PLUGIN_ROOT}`, which Codex does not expand there (`plugin_component_path_invalid`) | HIGH | + +Codex paths are plugin-root relative. Codex keeps a path only when it starts +with `./`, is not `./`, and has no `..` segment; it drops any other path and +loads the default location instead. So a declared field replaces its default +location only when Codex keeps it: `skills/`, `.mcp.json`, `.app.json`, +`hooks/hooks.json`, and `commands/`. A dropped path still gets its finding, and +the default location is inventoried and checked. Skills are found at any depth +below `skills/` or a declared skills folder, as Codex searches them +recursively. MCP entries use Codex's server fields: + +- `type` is optional; `streamable_http` and `streamable-http` mean `http`. +- `http_headers` are merged into `headers` (for a header in both, the + `http_headers` value is used unless only the `headers` value is a literal + credential), so a literal credential in either is a CRITICAL + `mcp_inline_secret`, in Tier 1 and when Tier 3 stages the server. +- An inline `bearer_token`, which Codex rejects in plugins, is a HIGH + `mcp_inline_bearer_token`. +- `env_vars`, `env_http_headers`, `bearer_token_env_var`, + `http_headers_helper`, and `oauth` configure auth that the evaluation runtime + does not apply. A staged server that declares them makes the Tier 3 run + `INCOMPLETE`, like `env` and `headers`. + +### Cursor manifest: `.cursor-plugin/plugin.json` + +The Cursor format is described by the +[Cursor plugins reference](https://cursor.com/docs/reference/plugins) and the +published schema (`schemas/plugin.schema.json` in `cursor/plugins` at +`4b4d98e`). + +```json title=".cursor-plugin/plugin.json" +{ + "name": "release-helper", + "version": "1.0.0", + "description": "Triage tickets and draft release notes.", + "author": { "name": "Example Maintainers" }, + "skills": "./skills/", + "rules": "./rules/", + "mcpServers": "./mcp.json" +} +``` + +| Rule | Severity | +| --- | --- | +| `name` is required and lowercase kebab-case: `a-z`, `0-9`, `-`, and `.`, starting and ending with a letter or digit | HIGH | +| Declared fields have the schema's types; `author` requires `name` and allows only `name` and `email` | HIGH | +| A top-level field outside the published schema (`plugin_manifest_unknown_field`) | MEDIUM | +| `version` is not a semantic version; the docs ask for one but the schema does not enforce it | LOW | + +Paths are plugin-root relative; `./` is conventional but not required. A +declared field replaces default-folder discovery for its type. The defaults are +`skills/`, `rules/` (`.md`, `.mdc`, and `.markdown` files), `agents/`, +`commands/` (which also accepts `.txt`), `hooks/hooks.json`, and a root +`mcp.json`. `${CURSOR_PLUGIN_ROOT}` and `${CLAUDE_PLUGIN_ROOT}` name the root. A +root `SKILL.md` makes a single-skill plugin when no skills are declared and +there is no `skills/` directory; it is inventoried but not staged. Cursor hooks list +handlers directly under each event; the hook risk model reads each one as a +one-handler matcher group, so the same checks apply. + +### Field support by format + +Each format's component fields map onto the same inventory types. Fields that +only carry metadata (`homepage`, `keywords`, `interface`, `logo`, `variables`, +and so on) are type-checked where the table above says so, and otherwise +ignored. + +| Component | Claude Code | Agent Plugins v1 | Codex | Cursor | Support | +| --- | --- | --- | --- | --- | --- | +| Identity | `name` | `$schema`, `name`, `version` | `name`; `version`, `description`, `author` advisory | `name` | Validated | +| `skill` | `skills` + `skills/` | `skills/` (fixed, immediate children) | `skills` or `skills/` (any depth) | `skills` or `skills/`; root `SKILL.md` | `evaluated` | +| `rule` | `rules` + `rules/` | Extension `rules/` only | None | `rules` or `rules/` | `evaluated`; extension rules `unsupported` | +| `mcp` | `mcpServers` + `.mcp.json` | `mcp.json` (fixed) | `mcpServers` or `.mcp.json` | `mcpServers` or `mcp.json` | See [Component inventory](#component-inventory) | +| `hook` | `hooks` + `hooks/hooks.json` | `extensions["com.openai"].hooks`; extension `hooks/hooks.json` | `hooks` or `hooks/hooks.json` | `hooks` or `hooks/hooks.json` | `unsupported` | +| `agent` | `agents` or `agents/` | Extension `agents/` | None | `agents` or `agents/` | `unsupported` | +| `command` | `commands` or `commands/` | Extension `commands/` | `commands` or `commands/` | `commands` or `commands/` | `unsupported` | +| `app` | None | `extensions["com.openai"].apps` | `apps` or `.app.json` | None | `unsupported` | +| `extension` | None | `extensions` keys and top-level reverse-domain directories | None | None | `unsupported` | +| `lsp`, `output_style`, `monitor`, `settings` | Yes | None | None | None | `unsupported` | + +"A or B" means a declared field replaces the default location; "A + B" means +both are read. + +Every declared skill folder gets the same skill schema checks as the skills in +`skills/`. When a format's declared `skills` replace `skills/` (Cursor, and +Codex when it keeps the path), the skills found only in `skills/` are not part +of that client's plugin, so their skill schema findings are advisory (MEDIUM +at most, marked `advisory`). + +### Other clients that load the same folder + +Each manifest is inventoried with its own client's rules. Two clients also +load a plugin folder through a manifest that is not their own, and their view +is checked too: + +- **Claude Code** loads any folder with `--plugin-dir`. Without a + `.claude-plugin/plugin.json`, it reads its default locations: `skills/`, + `agents/`, `commands/`, `hooks/hooks.json`, `.mcp.json`, `.lsp.json`, + `output-styles/`, `monitors/monitors.json`, and `settings.json`. +- **Codex** takes the first of a root Agent Plugins `plugin.json`, + `.codex-plugin/plugin.json`, `.claude-plugin/plugin.json`, and + `.cursor-plugin/plugin.json`. When that is a Claude Code or Cursor manifest, + it reads it with the Codex path rules and defaults (`.mcp.json`, + `hooks/hooks.json`, `commands/`, and `skills/` at any depth). + +So the `.mcp.json` server or `hooks/hooks.json` hook of a Cursor or Agent +Plugins plugin, or the subagents in its `agents/`, get the MCP policy, hook +risk, and privilege checks. A component that only such a view loads is listed +with `declared_by` (the manifest that client reads, or the default location +itself) and `loaded_by` (the client), is never `evaluated`, and is not staged +by Tier 3; its Tier 3 coverage reason names the client that loads it, and its +MCP servers are CVE-audited. A file that the plugin's own manifests already +load is checked once, with their rules. + +### Skill folders named evals, results, or versions + +Tier 1 file walks skip folders named `evals`, `results`, and `versions` (and +their dotted forms), because they hold evaluation output and snapshots. Inside +`skills/`, that is true only inside a skill: a folder with one of those names +directly under `skills/`, such as `skills/evals/` or `skills/versions/v2/`, is +a skill folder that clients load, so it is discovered and scanned like any +other skill. Two cases stay unscanned and are reported instead: + +- A `SKILL.md` deeper inside such a folder, for example in a skill's own + `evals/results/`, is a HIGH `plugin_skill_in_unscanned_folder`, because Codex + searches `skills/` recursively and loads it. Keep Tier 3 results out of the + plugin with `--results-dir` or `SKILLEVALUATOR_RESULTS_DIR`. +- A manifest path into such a folder for any component other than skills, for + example `"agents": "./evals/agents/"`, is a HIGH + `plugin_component_path_unscanned`. + +A declared skills folder, such as `"skills": "./my-skills/"` or even +`"skills": "./evals/"`, is searched the same way, so `my-skills/evals/SKILL.md` +is listed and gets the skill schema checks. Clients load such a skill, so the +whole-tree scans (security, PII, secrets, code integrity, license, dependency, +and Unicode) scan its folder as its own skill unit, after the same no-follow +check as the rest of the tree, even though they prune those folder names +elsewhere. A `SKILL.md` deeper inside a skill's own `evals/`, `results/`, or +`versions/` folder is still a HIGH `plugin_skill_in_unscanned_folder`, as it is +under `skills/`. + ### Reference sources A skill or rule reference names a public source-control system and a @@ -272,11 +580,20 @@ In a bundle-reference `agent_plugin.yaml`, the `mcp` list holds provider-only `.mcp.json` in a bundle-reference plugin is inventoried and statically checked, but Tier 3 does not stage it. +The other contained formats use the same forms with their own default file: +`.mcp.json` for Codex, and `mcp.json` (no leading dot) for Cursor and Agent +Plugins. In Codex and Cursor, a declared `mcpServers` replaces the default +file; in Agent Plugins the location is fixed. Codex and Agent Plugins entries +are normalized before the static policy runs, as described in their manifest +sections, so every format's servers get the same checks. + ### Component inventory Tier 1 and Tier 3 build the same static inventory from the declared fields and -from the Claude Code default locations. The `support` column says what -SkillEvaluator can do with each type today: +from the default locations of the selected manifest's format. This table uses +the Claude Code field names; see [Field support by format](#field-support-by-format) +for the other formats. The `support` column says what SkillEvaluator can do +with each type today: | Type | Declared by | Default location | Support | | --- | --- | --- | --- | @@ -290,9 +607,12 @@ SkillEvaluator can do with each type today: | `output_style` | `outputStyles` | `output-styles/*.md` | `unsupported` | | `monitor` | `experimental.monitors` (or `monitors`) | `monitors/monitors.json` | `unsupported` | | `settings` | `settings` in `plugin.json` | `settings.json`, `.claude/settings.json`, `.claude/settings.local.json` | `unsupported` | +| `app` | Codex `apps` | `.app.json` (Codex) | `unsupported`. Each app alias is listed | +| `extension` | Agent Plugins `extensions` keys | Top-level reverse-domain directories such as `com.github.copilot/` | `unsupported`. The `agents/`, `commands/`, `rules/`, and `hooks/hooks.json` inside a namespace directory are listed as their own types, all `unsupported` | Each inventoried component has an `origin`: `declared`, `packaged`, or -`declared+packaged`. It also has a root-relative `path` and a `findings` +`declared+packaged`. A component that only an additional manifest declares +also has `declared_by`, the path of that manifest. It also has a root-relative `path` and a `findings` count, which attributes Tier 1 findings from every validator to the component that owns the file. Inventory reads are bounded and never follow links. Hooks, LSP, monitors, settings, and MCP config files are read from their own budget, so @@ -305,7 +625,9 @@ entries past the cap would go unchecked. `unsupported` means "inventoried, and statically checked where a check exists, but not evaluated." For example, hooks, LSP servers, monitors, and settings are scanned for permission-bypass flags and dangerous environment overrides, but -they are never run. +they are never run. Hooks also get a per-handler +[risk model](#hook-risk-model), and subagents and commands a +[privilege review](#subagent-command-and-skill-privileges). ## Tier 1: static validation @@ -380,15 +702,172 @@ checks are skipped when a plugin bundles no skills. | `plugin_component_unreadable` | HIGH | A hooks, LSP, monitors, or settings JSON file is unreadable, invalid, or larger than 256 KiB. Its bypass and environment checks cannot run, so it blocks | | `plugin_component_list_truncated` | HIGH | A declared component list, an LSP server map, or a `commands` map has more than 256 entries | | `plugin_settings_bypass_permissions` | HIGH | Shipped settings set `permissions.defaultMode` to `bypassPermissions` | -| `plugin_settings_broad_allow` | HIGH | Shipped settings pre-approve unrestricted tools: `Bash`, `Bash(*)`, `Bash(:*)`, `Bash(*:*)`, or `*` | +| `plugin_settings_broad_allow` | HIGH | Shipped settings pre-approve `*` or a `Bash` rule that runs any command, classified like the command `allowed-tools` check below (`Bash`, `Bash(*)`, `Bash(**)`, `Bash(python3:*)`, ...) | +| `plugin_settings_permission_mode` | MEDIUM | Shipped settings set `permissions.defaultMode` to `acceptEdits` or `auto` | | `plugin_settings_auto_approve` | MEDIUM | Shipped settings set `enableAllProjectMcpServers` | -| `plugin_permission_bypass_flag` | HIGH | A hook, LSP server, monitor, or settings file contains an agent permission-bypass flag or option, the same forms as `mcp_permission_bypass_flag` | +| `plugin_permission_bypass_flag` | HIGH | A hook, LSP server, monitor, or settings file contains an agent permission-bypass flag or option, the same forms as `mcp_permission_bypass_flag`. That includes Codex's `--dangerously-bypass-hook-trust`, `--ask-for-approval never` (`-a never`), and `-c approval_policy=never`. The short `-a never` and the `-c` / `--config` overrides count only after a `codex` command in the same string or argv, so `grep -a never file` is not flagged | +| `plugin_permission_mode_flag` | MEDIUM | A hook, LSP server, monitor, or settings file passes `--permission-mode auto` or `--permission-mode acceptEdits` to an agent CLI | +| `plugin_lsp_command_*`, `plugin_lsp_unpinned_package` | as for MCP | An `.lsp.json` server's `command` and `args` get the MCP stdio command checks: a shell `-c` program (`plugin_lsp_command_dangerous_form`, CRITICAL), shell metacharacters, inline credentials, floating versions, and unpinned package runners | | `plugin_permission_bypass_scan_truncated` | HIGH | A hook, LSP server, monitor, or settings config is too large to scan completely for bypass flags | | `plugin_env_code_injection` | HIGH | Settings or LSP `env` preloads code (`LD_PRELOAD`, `LD_AUDIT`, `DYLD_INSERT_LIBRARIES`, or `NODE_OPTIONS` with `--require`, `-r`, `--import`, `--loader`, or `--experimental-loader`) | | `plugin_env_traffic_redirect` | MEDIUM | Settings or LSP `env` overrides a model API base URL, a proxy, or trusted CA certificates | | `plugin_env_file_shipped` | MEDIUM | The plugin ships `.env` or `.env.*`. Template names such as `.env.example` and `.env.sample` are allowed. The file is never opened; secret scanning covers the values | | `plugin_env_scan_incomplete` | LOW | The `.env` scan stopped early: the plugin has more than 4,096 entries, or a directory could not be opened or listed | +### Subagent, command, and skill privileges + +Tier 1 reads the frontmatter of every subagent (`agents/*.md`), command +(`commands/*.md`, or a `commands` map entry in `plugin.json`), and skill +(`SKILL.md`) with a bounded, no-follow read. Nothing is loaded or run. It +records these fields per component under `plugin.privileges`: + +- subagents: `tools`, `disallowedTools`, `model`, `permissionMode`, and whether + the subagent inherits every tool because it omits `tools`; +- commands and skills: `allowed-tools` (or the map entry's `allowedTools`), + `model`, and whether the model may invoke it (`disable-model-invocation` is + not `true`). Claude Code pre-approves a skill's `allowed-tools` while the + skill is active, the same as a command's. + +Hooks in a skill's or command's frontmatter (`hooks:`) are registered by +Claude Code while the skill or command is active, so they get the hook risk +model below, with the source `#hooks`. + +`tools` and `allowed-tools` accept a YAML list or a comma- or space-separated +string; separators inside parentheses, as in `Bash(git add *)`, do not split +an entry. + +| Check | Severity | Flags | +| --- | --- | --- | +| `plugin_command_unrestricted_bash`, `plugin_skill_unrestricted_bash` | HIGH | `allowed-tools` pre-approves a `Bash` grant that runs any command, so any shell command runs without a prompt while the command or skill is active. Grants are classified the way Claude Code classifies its own allow rules: no content, empty content, or only `*` (`Bash`, `Bash()`, `Bash(*)`, `Bash(**)`, plus `Bash(:*)` and `Bash(*:*)`), or an interpreter or wrapper prefix (`python`, `python3`, `node`, `deno`, `tsx`, `ruby`, `perl`, `php`, `lua`, `npx`, `bunx`, `npm run`, `yarn run`, `pnpm run`, `bun run`, `bash`, `sh`, `zsh`, `fish`, `ssh`, `eval`, `exec`, `env`, `xargs`, `sudo`) as ``, `:*`, ` *`, `*`, or ` -